@plurnk/plurnk-providers 1.7.0 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +25 -13
- package/README.md +8 -1
- package/SPEC.md +93 -15
- package/dist/AiSdkProvider.d.ts +6 -3
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +108 -51
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +1 -0
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +2 -0
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -0
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +3 -0
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +11 -10
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +7 -6
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +2 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +29 -6
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +94 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +2 -0
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +5 -4
- package/dist/cost.js.map +1 -1
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +13 -2
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +3 -2
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +11 -4
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +9 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -3
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +1 -1
- package/dist/notices.d.ts.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +163 -19
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +8 -2
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +10 -1
- package/dist/types.js.map +1 -1
- package/package.json +9 -9
- package/src/AiSdkProvider.test.ts +206 -32
- package/src/AiSdkProvider.ts +140 -54
- package/src/Mock.ts +2 -0
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +5 -0
- package/src/ProviderRegistry.test.ts +27 -14
- package/src/ProviderRegistry.ts +19 -10
- package/src/accounting.test.ts +6 -2
- package/src/accounting.ts +7 -6
- package/src/aiSdkTransport.ts +32 -7
- package/src/catalogProvider.test.ts +151 -19
- package/src/catalogProvider.ts +125 -3
- package/src/compatibleProvider.test.ts +13 -10
- package/src/compatibleProvider.ts +2 -0
- package/src/cost.ts +5 -4
- package/src/discover.test.ts +27 -0
- package/src/discover.ts +20 -3
- package/src/env.test.ts +23 -8
- package/src/env.ts +17 -8
- package/src/index.ts +16 -8
- package/src/notices.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/providerDefaults.test.ts +50 -0
- package/src/sdkModels.test.ts +142 -8
- package/src/sdkModels.ts +201 -19
- package/src/types.ts +16 -0
package/src/AiSdkProvider.ts
CHANGED
|
@@ -20,11 +20,13 @@ import type {
|
|
|
20
20
|
ProviderRequestSettlement,
|
|
21
21
|
ProviderResponse,
|
|
22
22
|
ProviderUsage,
|
|
23
|
+
ReasoningPolicy,
|
|
23
24
|
} from "./types.ts";
|
|
24
25
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
25
|
-
import
|
|
26
|
-
import {
|
|
27
|
-
import type
|
|
26
|
+
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
27
|
+
import type { CallWarning, JSONValue } from "ai";
|
|
28
|
+
import { MAX_PROVIDER_TIMEOUT_MS, type Reasoning, type ReasoningResponseStyle } from "./env.ts";
|
|
29
|
+
import { UnsupportedReasoningPolicyError } from "./types.ts";
|
|
28
30
|
import {
|
|
29
31
|
executeAiSdkModel,
|
|
30
32
|
executeOpenAICompatible,
|
|
@@ -59,6 +61,25 @@ export type CacheAffinity =
|
|
|
59
61
|
|
|
60
62
|
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
61
63
|
|
|
64
|
+
const isJsonObject = (value: JSONValue | undefined): value is Record<string, JSONValue> =>
|
|
65
|
+
typeof value === "object" && value !== null && !Array.isArray(value);
|
|
66
|
+
|
|
67
|
+
const mergeJsonObjects = (
|
|
68
|
+
left: Record<string, JSONValue | undefined>,
|
|
69
|
+
right: Record<string, JSONValue | undefined>,
|
|
70
|
+
): Record<string, JSONValue | undefined> => Object.fromEntries(
|
|
71
|
+
[...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
|
|
72
|
+
const leftValue = left[key];
|
|
73
|
+
const rightValue = right[key];
|
|
74
|
+
return [
|
|
75
|
+
key,
|
|
76
|
+
isJsonObject(leftValue) && isJsonObject(rightValue)
|
|
77
|
+
? mergeJsonObjects(leftValue, rightValue)
|
|
78
|
+
: rightValue ?? leftValue,
|
|
79
|
+
];
|
|
80
|
+
}),
|
|
81
|
+
);
|
|
82
|
+
|
|
62
83
|
export type AiSdkProviderConfig = {
|
|
63
84
|
model: string;
|
|
64
85
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
@@ -75,6 +96,11 @@ export type AiSdkProviderConfig = {
|
|
|
75
96
|
maxOutputTokens?: number | null;
|
|
76
97
|
outputBudget?: number | null;
|
|
77
98
|
reasoningBudget?: number | null;
|
|
99
|
+
supportedReasoningPolicies?: readonly ReasoningPolicy[];
|
|
100
|
+
// Native AI SDK projection for adaptive. `high` is the graded fallback;
|
|
101
|
+
// provider-default is reserved for a documented native option/default.
|
|
102
|
+
adaptiveReasoning?: "high" | "provider-default";
|
|
103
|
+
adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
|
|
78
104
|
// Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
|
|
79
105
|
// visible output and add an explicit provider reasoning budget. This marker lets the
|
|
80
106
|
// adapter subtract that subset so the resulting wire cap remains PLURNK's
|
|
@@ -125,7 +151,7 @@ export type AiSdkProviderConfig = {
|
|
|
125
151
|
requiresOutputBudget?: boolean;
|
|
126
152
|
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
127
153
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
128
|
-
// { mode: off|adaptive|
|
|
154
|
+
// { mode: off|adaptive|low|medium|high, budget: independent optional cap }.
|
|
129
155
|
// to the backend's mechanism via reasoningStyle; budget is only ever an
|
|
130
156
|
// explicit magnitude, never a hidden activation flag.
|
|
131
157
|
reasoning: Reasoning;
|
|
@@ -265,19 +291,32 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
|
265
291
|
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
266
292
|
};
|
|
267
293
|
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
if (budget <= 4000) return "medium";
|
|
272
|
-
return "high";
|
|
294
|
+
const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" => {
|
|
295
|
+
if (mode === "low" || mode === "medium" || mode === "high") return mode;
|
|
296
|
+
throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
|
|
273
297
|
};
|
|
274
298
|
|
|
275
|
-
//
|
|
276
|
-
//
|
|
277
|
-
//
|
|
278
|
-
//
|
|
279
|
-
const
|
|
280
|
-
|
|
299
|
+
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
300
|
+
// PLURNK's durable contract names an effort. These fractions match the native
|
|
301
|
+
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
302
|
+
// model's physical maximum. The minimum is imposed by the provider protocol.
|
|
303
|
+
const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
304
|
+
adaptive: 0.6,
|
|
305
|
+
low: 0.1,
|
|
306
|
+
medium: 0.3,
|
|
307
|
+
high: 0.6,
|
|
308
|
+
} satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
|
|
309
|
+
const MANUAL_REASONING_MINIMUM = 1024;
|
|
310
|
+
|
|
311
|
+
const providerWarningMessage = (warning: CallWarning): string => {
|
|
312
|
+
switch (warning.type) {
|
|
313
|
+
case "unsupported":
|
|
314
|
+
case "compatibility":
|
|
315
|
+
return `${warning.type} ${warning.feature}${warning.details === undefined ? "" : `: ${warning.details}`}`;
|
|
316
|
+
case "deprecated": return `deprecated ${warning.setting}: ${warning.message}`;
|
|
317
|
+
case "other": return warning.message;
|
|
318
|
+
}
|
|
319
|
+
};
|
|
281
320
|
|
|
282
321
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
283
322
|
// these. Two families:
|
|
@@ -321,6 +360,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
321
360
|
#reasoningBudget: number | null;
|
|
322
361
|
#additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
323
362
|
#reasoning: Reasoning;
|
|
363
|
+
#supportedReasoningPolicies: readonly ReasoningPolicy[];
|
|
364
|
+
#adaptiveReasoning: "high" | "provider-default";
|
|
365
|
+
#adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
324
366
|
#temperature: number;
|
|
325
367
|
#repeatPenalty: number;
|
|
326
368
|
#frequencyPenalty: number;
|
|
@@ -390,6 +432,18 @@ export default class AiSdkProvider implements Provider {
|
|
|
390
432
|
this.#reasoningBudget = config.reasoningBudget ?? null;
|
|
391
433
|
this.#additiveReasoningProvider = config.additiveReasoningProvider;
|
|
392
434
|
this.#reasoning = config.reasoning;
|
|
435
|
+
this.#supportedReasoningPolicies = Object.freeze([
|
|
436
|
+
...new Set(config.supportedReasoningPolicies ?? REASONING_POLICIES),
|
|
437
|
+
]);
|
|
438
|
+
this.#adaptiveReasoning = config.adaptiveReasoning ?? "high";
|
|
439
|
+
this.#adaptiveReasoningProviderOptions = config.adaptiveReasoningProviderOptions;
|
|
440
|
+
if (!this.#supportedReasoningPolicies.includes(this.#reasoning.mode)) {
|
|
441
|
+
throw new UnsupportedReasoningPolicyError(
|
|
442
|
+
config.source ?? "provider",
|
|
443
|
+
this.#reasoning.mode,
|
|
444
|
+
this.#supportedReasoningPolicies,
|
|
445
|
+
);
|
|
446
|
+
}
|
|
393
447
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
394
448
|
// required tuning fields must fail at construction, not silently send
|
|
395
449
|
// undefined sampling on every grammar request.
|
|
@@ -435,6 +489,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
435
489
|
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
436
490
|
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
437
491
|
}
|
|
492
|
+
if (this.#languageModel === undefined && this.#additiveReasoningProvider !== undefined) {
|
|
493
|
+
throw new Error(`${this.#source}: additive reasoning projection requires an AI SDK model`);
|
|
494
|
+
}
|
|
438
495
|
if (this.#cacheAffinity?.target === "provider-option"
|
|
439
496
|
&& Object.hasOwn(
|
|
440
497
|
this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
|
|
@@ -476,16 +533,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
476
533
|
throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
|
|
477
534
|
}
|
|
478
535
|
if (this.#reasoningStyle === "anthropic"
|
|
479
|
-
&& this.#reasoning.mode
|
|
536
|
+
&& this.#reasoning.mode !== "off"
|
|
537
|
+
&& this.#reasoning.mode !== "adaptive"
|
|
480
538
|
&& this.#reasoning.budget === null
|
|
481
539
|
&& this.#reasoningBudget === null) {
|
|
482
540
|
throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
483
541
|
}
|
|
484
|
-
if (this.#additiveReasoningProvider !== undefined
|
|
485
|
-
&& this.#reasoning.mode === "on"
|
|
486
|
-
&& this.#reasoningBudget === null) {
|
|
487
|
-
throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
|
|
488
|
-
}
|
|
489
542
|
if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
|
|
490
543
|
throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
|
|
491
544
|
}
|
|
@@ -515,6 +568,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
515
568
|
get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
|
|
516
569
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
517
570
|
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
571
|
+
get supportedReasoningPolicies(): readonly ReasoningPolicy[] { return this.#supportedReasoningPolicies; }
|
|
518
572
|
get inputCapacity(): number | null {
|
|
519
573
|
return effectiveInputCapacity({
|
|
520
574
|
contextWindow: this.#contextWindow,
|
|
@@ -628,7 +682,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
628
682
|
case "template": {
|
|
629
683
|
const allowance = mode === "off"
|
|
630
684
|
? 0
|
|
631
|
-
:
|
|
685
|
+
: budget;
|
|
632
686
|
return {
|
|
633
687
|
chat_template_kwargs: { enable_thinking: on },
|
|
634
688
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -637,9 +691,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
637
691
|
}
|
|
638
692
|
case "think": return on ? { think: true } : {};
|
|
639
693
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
694
|
+
case "effort": return mode === "off"
|
|
695
|
+
? {}
|
|
696
|
+
: { reasoning_effort: mode === "adaptive" ? "high" : fixedEffort(mode) };
|
|
643
697
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
644
698
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
645
699
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -649,25 +703,23 @@ export default class AiSdkProvider implements Provider {
|
|
|
649
703
|
// efforts 400.
|
|
650
704
|
case "effort_explicit": return mode === "off"
|
|
651
705
|
? { reasoning_effort: "none" }
|
|
652
|
-
: mode === "
|
|
706
|
+
: mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
653
707
|
// {§deepseek-reasoning-request}
|
|
654
708
|
case "thinking_effort": return mode === "off"
|
|
655
709
|
? { thinking: { type: "disabled" } }
|
|
656
|
-
: mode === "
|
|
710
|
+
: mode === "adaptive" ? { thinking: { type: "enabled" } } : {
|
|
657
711
|
thinking: { type: "enabled" },
|
|
658
|
-
|
|
659
|
-
}
|
|
660
|
-
// Anthropic
|
|
661
|
-
// enabled with the explicit reasoning subset; adaptive →
|
|
662
|
-
// omit (the API default).
|
|
712
|
+
reasoning_effort: fixedEffort(mode),
|
|
713
|
+
};
|
|
714
|
+
// Anthropic-compatible native dynamic or manual budget mode.
|
|
663
715
|
case "anthropic": return mode === "off"
|
|
664
716
|
? { thinking: { type: "disabled" } }
|
|
665
|
-
: mode === "
|
|
717
|
+
: mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
|
|
666
718
|
thinking: {
|
|
667
719
|
type: "enabled",
|
|
668
720
|
budget_tokens: budget!,
|
|
669
721
|
},
|
|
670
|
-
}
|
|
722
|
+
};
|
|
671
723
|
case "none": return {};
|
|
672
724
|
}
|
|
673
725
|
}
|
|
@@ -809,22 +861,26 @@ export default class AiSdkProvider implements Provider {
|
|
|
809
861
|
|
|
810
862
|
#requestProviderOptions(
|
|
811
863
|
workerId: string,
|
|
812
|
-
|
|
864
|
+
nativeReasoningBudget: number | null,
|
|
813
865
|
): AiSdkProviderOptions | undefined {
|
|
814
866
|
const responseOptions = this.#reasoning.mode === "off"
|
|
815
867
|
? undefined
|
|
816
868
|
: this.#reasoningResponseProviderOptions;
|
|
817
|
-
const
|
|
869
|
+
const adaptiveOptions = this.#reasoning.mode === "adaptive"
|
|
870
|
+
&& nativeReasoningBudget === null
|
|
871
|
+
? this.#adaptiveReasoningProviderOptions
|
|
872
|
+
: undefined;
|
|
873
|
+
const nativeReasoning = nativeReasoningBudget !== null
|
|
818
874
|
? this.#additiveReasoningProvider === "anthropic"
|
|
819
|
-
? { anthropic: { thinking: { type: "enabled", budgetTokens:
|
|
875
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
820
876
|
: this.#additiveReasoningProvider === "bedrock"
|
|
821
|
-
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens:
|
|
877
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
822
878
|
: undefined
|
|
823
879
|
: undefined;
|
|
824
880
|
const options: AiSdkProviderOptions = {};
|
|
825
|
-
for (const part of [responseOptions, nativeReasoning]) {
|
|
881
|
+
for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
|
|
826
882
|
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
827
|
-
options[provider] =
|
|
883
|
+
options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
|
|
828
884
|
}
|
|
829
885
|
}
|
|
830
886
|
if (this.#cacheAffinity?.target === "provider-option") {
|
|
@@ -836,16 +892,38 @@ export default class AiSdkProvider implements Provider {
|
|
|
836
892
|
|
|
837
893
|
#nativeMaxOutputTokens(
|
|
838
894
|
outputBudget: number | null,
|
|
839
|
-
|
|
895
|
+
nativeReasoningBudget: number | null,
|
|
840
896
|
): number | undefined {
|
|
841
897
|
if (outputBudget === null) return undefined;
|
|
842
|
-
return
|
|
843
|
-
|
|
844
|
-
&& reasoningBudget !== null
|
|
845
|
-
? outputBudget - reasoningBudget
|
|
898
|
+
return nativeReasoningBudget !== null
|
|
899
|
+
? outputBudget - nativeReasoningBudget
|
|
846
900
|
: outputBudget;
|
|
847
901
|
}
|
|
848
902
|
|
|
903
|
+
#nativeReasoningBudget(
|
|
904
|
+
outputBudget: number | null,
|
|
905
|
+
configuredReasoningBudget: number | null,
|
|
906
|
+
): number | null {
|
|
907
|
+
if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off") return null;
|
|
908
|
+
if (configuredReasoningBudget !== null) return configuredReasoningBudget;
|
|
909
|
+
if (this.#adaptiveReasoningProviderOptions !== undefined) return null;
|
|
910
|
+
if (outputBudget === null) {
|
|
911
|
+
throw new TypeError(
|
|
912
|
+
`${this.#source}: manual provider reasoning requires a resolved total output budget`,
|
|
913
|
+
);
|
|
914
|
+
}
|
|
915
|
+
if (outputBudget <= MANUAL_REASONING_MINIMUM) {
|
|
916
|
+
throw new TypeError(
|
|
917
|
+
`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`,
|
|
918
|
+
);
|
|
919
|
+
}
|
|
920
|
+
const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
|
|
921
|
+
return Math.min(
|
|
922
|
+
outputBudget - 1,
|
|
923
|
+
Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)),
|
|
924
|
+
);
|
|
925
|
+
}
|
|
926
|
+
|
|
849
927
|
#accounting(
|
|
850
928
|
outcome: ProviderRequestAccounting["outcome"],
|
|
851
929
|
usage: ProviderUsage | undefined,
|
|
@@ -895,6 +973,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
895
973
|
);
|
|
896
974
|
}
|
|
897
975
|
const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
|
|
976
|
+
const nativeReasoningBudget = this.#nativeReasoningBudget(
|
|
977
|
+
capacity.outputBudget,
|
|
978
|
+
capacity.reasoningBudget,
|
|
979
|
+
);
|
|
898
980
|
|
|
899
981
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
900
982
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
@@ -1008,7 +1090,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1008
1090
|
: await executeAiSdkModel({
|
|
1009
1091
|
languageModel: this.#languageModel,
|
|
1010
1092
|
headers: requestHeaders,
|
|
1011
|
-
providerOptions: this.#requestProviderOptions(workerId,
|
|
1093
|
+
providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
|
|
1012
1094
|
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
1013
1095
|
messages,
|
|
1014
1096
|
signal: operationSignal,
|
|
@@ -1032,17 +1114,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
1032
1114
|
? sampling.stop
|
|
1033
1115
|
: undefined,
|
|
1034
1116
|
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
1035
|
-
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget,
|
|
1117
|
+
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
1036
1118
|
reasoning: this.#reasoning.mode === "off"
|
|
1037
1119
|
? "none"
|
|
1038
1120
|
: this.#reasoning.mode === "adaptive"
|
|
1039
|
-
?
|
|
1040
|
-
: this.#
|
|
1041
|
-
? "provider-default"
|
|
1042
|
-
: effortFromReasoning({
|
|
1043
|
-
mode: this.#reasoning.mode,
|
|
1044
|
-
budget: capacity.reasoningBudget,
|
|
1045
|
-
}),
|
|
1121
|
+
? this.#adaptiveReasoning
|
|
1122
|
+
: fixedEffort(this.#reasoning.mode),
|
|
1046
1123
|
});
|
|
1047
1124
|
} catch (error) {
|
|
1048
1125
|
const failure = transportFailureEvidence(error);
|
|
@@ -1152,6 +1229,15 @@ export default class AiSdkProvider implements Provider {
|
|
|
1152
1229
|
}
|
|
1153
1230
|
|
|
1154
1231
|
let notices: ProviderNotice[] | undefined;
|
|
1232
|
+
for (const warning of raw.warnings) {
|
|
1233
|
+
(notices ??= []).push({
|
|
1234
|
+
source: this.#source,
|
|
1235
|
+
kind: "provider_warning",
|
|
1236
|
+
level: "warn",
|
|
1237
|
+
message: providerWarningMessage(warning),
|
|
1238
|
+
position: null,
|
|
1239
|
+
});
|
|
1240
|
+
}
|
|
1155
1241
|
const usage = raw.usage;
|
|
1156
1242
|
if (sendGrammar !== undefined
|
|
1157
1243
|
&& this.tokenize !== undefined
|
package/src/Mock.ts
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
9
|
import { resolveGenerationEnvelopeFromEnv } from "./env.ts";
|
|
10
|
+
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
10
11
|
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
11
12
|
import { ProviderError } from "./errors.ts";
|
|
12
13
|
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
@@ -69,6 +70,7 @@ export default class Mock implements Provider {
|
|
|
69
70
|
get maxOutputTokens(): number | null { return null; }
|
|
70
71
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
71
72
|
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
73
|
+
get supportedReasoningPolicies() { return REASONING_POLICIES; }
|
|
72
74
|
get inputCapacity(): number | null {
|
|
73
75
|
return effectiveInputCapacity({
|
|
74
76
|
contextWindow: this.#contextWindow,
|
package/src/Pool.test.ts
CHANGED
|
@@ -53,6 +53,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
53
53
|
maxOutputTokens: opts.maxOutputTokens ?? null,
|
|
54
54
|
outputBudget: opts.outputBudget ?? null,
|
|
55
55
|
reasoningBudget: opts.reasoningBudget ?? null,
|
|
56
|
+
supportedReasoningPolicies: ["off", "adaptive", "low", "medium", "high"],
|
|
56
57
|
inputCapacity: effectiveInputCapacity({
|
|
57
58
|
contextWindow: opts.window === undefined ? 48_000 : opts.window,
|
|
58
59
|
maxInputTokens: opts.maxInputTokens ?? null,
|
package/src/Pool.ts
CHANGED
|
@@ -7,6 +7,7 @@ import Meta, {
|
|
|
7
7
|
type PluginAttributionContext,
|
|
8
8
|
} from "@plurnk/plurnk-meta";
|
|
9
9
|
import { effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, requestCapacityDecision } from "./capacity.ts";
|
|
10
|
+
import type { ReasoningPolicy } from "./types.ts";
|
|
10
11
|
|
|
11
12
|
// A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
|
|
12
13
|
// transient retries before throwing one of these, so re-hitting the same
|
|
@@ -89,6 +90,10 @@ export default class Pool implements Provider {
|
|
|
89
90
|
get maxOutputTokens(): number | null { return this.#minimumKnown((provider) => provider.maxOutputTokens); }
|
|
90
91
|
get outputBudget(): number | null { return this.#minimumKnown((provider) => provider.outputBudget); }
|
|
91
92
|
get reasoningBudget(): number | null { return this.#minimumKnown((provider) => provider.reasoningBudget); }
|
|
93
|
+
get supportedReasoningPolicies(): readonly ReasoningPolicy[] {
|
|
94
|
+
return this.#backends[0].supportedReasoningPolicies.filter((policy) =>
|
|
95
|
+
this.#backends.every((provider) => provider.supportedReasoningPolicies.includes(policy)));
|
|
96
|
+
}
|
|
92
97
|
get inputCapacity(): number | null { return this.#minimumKnown((provider) => provider.inputCapacity); }
|
|
93
98
|
|
|
94
99
|
// Served id / capabilities aggregate CONSERVATIVELY: a worker could land on any
|
|
@@ -4,7 +4,7 @@ import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./
|
|
|
4
4
|
import type { PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
5
5
|
|
|
6
6
|
const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
|
|
7
|
-
async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
|
|
7
|
+
async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>(), grammarStyles: new Map() });
|
|
8
8
|
|
|
9
9
|
// Alias parsing is tested in @plurnk/plurnk-aliases (its owner). Here we
|
|
10
10
|
// exercise the resolution + two-tier instantiation this module owns; the active
|
|
@@ -31,7 +31,7 @@ test("instantiateProvider: cataloged name resolves in-framework, no scan, no imp
|
|
|
31
31
|
let scanned = false;
|
|
32
32
|
const p = await instantiateProvider("openai", { ...fullEnv }, "m",
|
|
33
33
|
async (s) => { imports.push(s); return {}; },
|
|
34
|
-
async () => { scanned = true; return { registry: new Map(), skipped: new Map(), attributions: new Map() }; });
|
|
34
|
+
async () => { scanned = true; return { registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }; });
|
|
35
35
|
assert.equal(p.model, "m");
|
|
36
36
|
assert.deepEqual(imports, []); // tier 1 never touches the importer…
|
|
37
37
|
assert.equal(scanned, false); // …nor the scan
|
|
@@ -77,7 +77,7 @@ test("instantiateProvider: a selected plugin composes its static and runtime att
|
|
|
77
77
|
registry: new Map([["acme", "@acme/ai-provider"]]),
|
|
78
78
|
skipped: new Map(),
|
|
79
79
|
attributions: new Map([["acme", "static:provider"]]),
|
|
80
|
-
packageAttributions: new Map([["@acme/ai-provider", ["static:provider"]]]),
|
|
80
|
+
packageAttributions: new Map([["@acme/ai-provider", ["static:provider"]]]), grammarStyles: new Map(),
|
|
81
81
|
}),
|
|
82
82
|
);
|
|
83
83
|
|
|
@@ -107,7 +107,7 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
|
|
|
107
107
|
return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
108
108
|
});
|
|
109
109
|
await instantiateProvider("openai", { ...fullEnv }, "m", // fullEnv.OPENAI_BASE_URL is http://x — the override must win
|
|
110
|
-
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
|
|
110
|
+
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
|
|
111
111
|
"http://hazel2:8080/v1");
|
|
112
112
|
assert.ok(probed.some((u) => u === "http://hazel2:8080/v1/models"), `probe hit the override host; saw ${probed.join(", ")}`);
|
|
113
113
|
assert.equal(probed.some((u) => u.startsWith("http://x")), false); // never the per-name OPENAI_BASE_URL
|
|
@@ -193,13 +193,13 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
|
|
|
193
193
|
});
|
|
194
194
|
const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW_turbo: "12345", PLURNK_PROVIDERS_LLAMA_SERVER_turbo: "1" };
|
|
195
195
|
const p = await instantiateProvider("openai", env, "m",
|
|
196
|
-
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
|
|
196
|
+
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
|
|
197
197
|
undefined, "turbo");
|
|
198
198
|
assert.equal(p.contextWindow, 12345); // _turbo CONTEXT_WINDOW reached the provider
|
|
199
199
|
assert.equal(p.constrainsOutput, true); // _turbo LLAMA_SERVER pin reached it too
|
|
200
200
|
// same env, DIFFERENT alias: neither override applies
|
|
201
201
|
const q = await instantiateProvider("openai", env, "m",
|
|
202
|
-
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
|
|
202
|
+
async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
|
|
203
203
|
undefined, "plain");
|
|
204
204
|
assert.equal(q.contextWindow, null);
|
|
205
205
|
assert.equal(q.constrainsOutput, false);
|
|
@@ -310,7 +310,7 @@ test("a catalog provider with unknown model metadata never falls through to plug
|
|
|
310
310
|
"cloudflare",
|
|
311
311
|
{
|
|
312
312
|
CLOUDFLARE_ACCOUNT_ID: "account",
|
|
313
|
-
|
|
313
|
+
CLOUDFLARE_API_KEY: "token",
|
|
314
314
|
},
|
|
315
315
|
"vendor/model-outside-snapshot",
|
|
316
316
|
async () => { throw new Error("plugin import must not run"); },
|
|
@@ -319,7 +319,7 @@ test("a catalog provider with unknown model metadata never falls through to plug
|
|
|
319
319
|
return {
|
|
320
320
|
registry: new Map([["cloudflare", "@plurnk/plurnk-providers-cloudflare"]]),
|
|
321
321
|
skipped: new Map(),
|
|
322
|
-
attributions: new Map(),
|
|
322
|
+
attributions: new Map(), grammarStyles: new Map(),
|
|
323
323
|
};
|
|
324
324
|
},
|
|
325
325
|
),
|
|
@@ -334,7 +334,7 @@ test("explicit metadata constructs an out-of-snapshot Cloudflare model in the co
|
|
|
334
334
|
"cloudflare",
|
|
335
335
|
{
|
|
336
336
|
CLOUDFLARE_ACCOUNT_ID: "account",
|
|
337
|
-
|
|
337
|
+
CLOUDFLARE_API_KEY: "token",
|
|
338
338
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "128000",
|
|
339
339
|
},
|
|
340
340
|
"vendor/model-outside-snapshot",
|
|
@@ -344,7 +344,7 @@ test("explicit metadata constructs an out-of-snapshot Cloudflare model in the co
|
|
|
344
344
|
return {
|
|
345
345
|
registry: new Map([["cloudflare", "@plurnk/plurnk-providers-cloudflare"]]),
|
|
346
346
|
skipped: new Map(),
|
|
347
|
-
attributions: new Map(),
|
|
347
|
+
attributions: new Map(), grammarStyles: new Map(),
|
|
348
348
|
};
|
|
349
349
|
},
|
|
350
350
|
);
|
|
@@ -382,7 +382,7 @@ test("{§provider-tagged-reasoning} a Cloudflare model alias carries its explici
|
|
|
382
382
|
{
|
|
383
383
|
...fullEnv,
|
|
384
384
|
CLOUDFLARE_ACCOUNT_ID: "account",
|
|
385
|
-
|
|
385
|
+
CLOUDFLARE_API_KEY: "token",
|
|
386
386
|
PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE: "verbatim",
|
|
387
387
|
PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_cfds1: "think-tags",
|
|
388
388
|
},
|
|
@@ -406,7 +406,17 @@ test("two Fireworks aliases independently select default and priority service ti
|
|
|
406
406
|
const bodies: Record<string, unknown>[] = [];
|
|
407
407
|
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
408
408
|
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
409
|
-
|
|
409
|
+
const chunk = {
|
|
410
|
+
id: "fireworks-test",
|
|
411
|
+
object: "chat.completion.chunk",
|
|
412
|
+
created: 1,
|
|
413
|
+
model: "fireworks-test",
|
|
414
|
+
choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
|
|
415
|
+
};
|
|
416
|
+
return new Response(`data: ${JSON.stringify(chunk)}\n\ndata: [DONE]\n\n`, {
|
|
417
|
+
status: 200,
|
|
418
|
+
headers: { "content-type": "text/event-stream" },
|
|
419
|
+
});
|
|
410
420
|
});
|
|
411
421
|
const env = {
|
|
412
422
|
...fullEnv,
|
|
@@ -419,7 +429,7 @@ test("two Fireworks aliases independently select default and priority service ti
|
|
|
419
429
|
PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
|
|
420
430
|
};
|
|
421
431
|
const imports = async () => ({});
|
|
422
|
-
const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
|
|
432
|
+
const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() });
|
|
423
433
|
const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
|
|
424
434
|
const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
|
|
425
435
|
await fast.generate({ workerId: "fast-worker", messages: [] });
|
|
@@ -445,5 +455,8 @@ test("loadActiveProvider: resolves the alias cascade to an installed AI SDK prov
|
|
|
445
455
|
});
|
|
446
456
|
|
|
447
457
|
test("loadActiveProvider: throws a named error when no alias is active", async () => {
|
|
448
|
-
await assert.rejects(
|
|
458
|
+
await assert.rejects(
|
|
459
|
+
() => loadActiveProvider({ ...fullEnv }),
|
|
460
|
+
/set PLURNK_MODEL to a declared alias or provider\/model route/,
|
|
461
|
+
);
|
|
449
462
|
});
|
package/src/ProviderRegistry.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
|
-
// Provider instantiation + active-
|
|
1
|
+
// Provider instantiation + active model-route resolution. Alias parsing (the
|
|
2
2
|
// PLURNK_MODEL_<alias>=<provider>/<model> cascade + PLURNK_BASEURL_<alias>
|
|
3
|
-
// overrides) lives in @plurnk/plurnk-aliases
|
|
4
|
-
//
|
|
3
|
+
// overrides) lives in @plurnk/plurnk-aliases; this module resolves the active
|
|
4
|
+
// alias-or-route selector to a Provider.
|
|
5
5
|
//
|
|
6
6
|
// {§provider-resolution} Models.dev catalog → PLURNK provider declaration
|
|
7
7
|
// → local protocol adapter → scope-agnostic AI SDK plugin discovery. Generic
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
import type { AiSdkProviderPlugin, Provider } from "./types.ts";
|
|
12
12
|
import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
|
|
13
13
|
import { discover, type DiscoverOptions, type Discovery } from "./discover.ts";
|
|
14
|
-
import {
|
|
14
|
+
import { resolveActiveRoute } from "@plurnk/plurnk-aliases";
|
|
15
15
|
import { scopeEnvToAlias } from "./env.ts";
|
|
16
16
|
import { ollamaProviderFromEnv } from "./ollama.ts";
|
|
17
17
|
import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
|
|
@@ -61,7 +61,7 @@ export const instantiateProvider = async (
|
|
|
61
61
|
if (catalog !== null) return catalog;
|
|
62
62
|
if (name === "ollama") return ollamaProviderFromEnv(env, model, baseUrl === undefined ? undefined : { baseUrl });
|
|
63
63
|
if (name === "openai" || name === "plurnk") return compatibleProviderFromEnv(name, env, model, baseUrl);
|
|
64
|
-
const { registry, skipped, packageAttributions = new Map() } = await providerPackages(discoverFn, env);
|
|
64
|
+
const { registry, skipped, packageAttributions = new Map(), grammarStyles = new Map() } = await providerPackages(discoverFn, env);
|
|
65
65
|
const specifier = registry.get(name);
|
|
66
66
|
if (specifier === undefined) {
|
|
67
67
|
const declined = skipped.get(name);
|
|
@@ -99,21 +99,30 @@ export const instantiateProvider = async (
|
|
|
99
99
|
languageModel: sdkProvider.languageModel(model),
|
|
100
100
|
contextWindow,
|
|
101
101
|
attributions,
|
|
102
|
+
...(grammarStyles.get(name) === undefined ? {} : { grammarStyle: grammarStyles.get(name) }),
|
|
102
103
|
});
|
|
103
104
|
};
|
|
104
105
|
|
|
105
106
|
// Test-only: drop the memoized discovery so a fresh scan/injection runs next.
|
|
106
107
|
export const resetDiscoveryCache = (): void => { discoveredCache = null; };
|
|
107
108
|
|
|
108
|
-
// Boot convenience: resolve the active
|
|
109
|
+
// Boot convenience: resolve the active selector and instantiate its exact route.
|
|
109
110
|
export const loadActiveProvider = async (
|
|
110
111
|
env: NodeJS.ProcessEnv = process.env,
|
|
111
112
|
importImpl: ImportModule = importModule,
|
|
112
113
|
discoverFn: DiscoverFn = discover,
|
|
113
114
|
): Promise<Provider> => {
|
|
114
|
-
const
|
|
115
|
-
if (
|
|
116
|
-
throw new Error("no active provider: set PLURNK_MODEL to
|
|
115
|
+
const route = resolveActiveRoute(env);
|
|
116
|
+
if (route === null) {
|
|
117
|
+
throw new Error("no active provider: set PLURNK_MODEL to a declared alias or provider/model route");
|
|
117
118
|
}
|
|
118
|
-
return instantiateProvider(
|
|
119
|
+
return instantiateProvider(
|
|
120
|
+
route.provider,
|
|
121
|
+
env,
|
|
122
|
+
route.model,
|
|
123
|
+
importImpl,
|
|
124
|
+
discoverFn,
|
|
125
|
+
route.baseUrl,
|
|
126
|
+
route.alias,
|
|
127
|
+
);
|
|
119
128
|
};
|
package/src/accounting.test.ts
CHANGED
|
@@ -89,6 +89,10 @@ test("aggregateProviderAccounting preserves request order and only sums known fi
|
|
|
89
89
|
},
|
|
90
90
|
]);
|
|
91
91
|
assert.deepEqual(accounting.requests.map(({ provider }) => provider), ["provider:a", "provider:b"]);
|
|
92
|
-
assert.
|
|
93
|
-
|
|
92
|
+
assert.deepEqual(accounting.usage, {
|
|
93
|
+
inputTokens: 2,
|
|
94
|
+
outputTokens: 3,
|
|
95
|
+
totalTokens: 5,
|
|
96
|
+
}, "a response-less failure is skipped, never allowed to erase reported usage");
|
|
97
|
+
assert.equal(accounting.costUsd, "0.25", "a response-less failure is skipped; the expressible cost survives");
|
|
94
98
|
});
|
package/src/accounting.ts
CHANGED
|
@@ -126,12 +126,13 @@ const sumKnown = (
|
|
|
126
126
|
requests: readonly ProviderRequestAccounting[],
|
|
127
127
|
read: (usage: ProviderUsage) => number | undefined,
|
|
128
128
|
): number | undefined => {
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
? undefined
|
|
134
|
-
|
|
129
|
+
// {§tokenomics-provider-usage} — aggregate usage sums every reported
|
|
130
|
+
// quantity; an unreported one (a response-less failure) is skipped, never
|
|
131
|
+
// invented as zero and never allowed to erase the reported evidence.
|
|
132
|
+
const known = requests
|
|
133
|
+
.map((request) => request.usage === undefined ? undefined : read(request.usage))
|
|
134
|
+
.filter((value): value is number => value !== undefined);
|
|
135
|
+
return known.length === 0 ? undefined : known.reduce((sum, value) => sum + value, 0);
|
|
135
136
|
};
|
|
136
137
|
|
|
137
138
|
export const aggregateProviderAccounting = (
|