@plurnk/plurnk-providers 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +25 -13
- package/README.md +8 -1
- package/SPEC.md +103 -15
- package/dist/AiSdkProvider.d.ts +7 -4
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +150 -53
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +2 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +6 -1
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -0
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +3 -0
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +11 -10
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +16 -8
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +33 -6
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +94 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +2 -0
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +5 -4
- package/dist/cost.js.map +1 -1
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +13 -2
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +3 -2
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +11 -4
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +10 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -3
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +1 -1
- package/dist/notices.d.ts.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +163 -19
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +10 -2
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +10 -1
- package/dist/types.js.map +1 -1
- package/package.json +9 -9
- package/src/AiSdkProvider.test.ts +214 -34
- package/src/AiSdkProvider.ts +181 -56
- package/src/Mock.test.ts +6 -1
- package/src/Mock.ts +6 -1
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +5 -0
- package/src/ProviderRegistry.test.ts +27 -14
- package/src/ProviderRegistry.ts +19 -10
- package/src/accounting.test.ts +30 -2
- package/src/accounting.ts +16 -8
- package/src/aiSdkTransport.test.ts +3 -0
- package/src/aiSdkTransport.ts +38 -8
- package/src/catalogProvider.test.ts +151 -19
- package/src/catalogProvider.ts +125 -3
- package/src/compatibleProvider.test.ts +13 -10
- package/src/compatibleProvider.ts +2 -0
- package/src/cost.ts +5 -4
- package/src/discover.test.ts +27 -0
- package/src/discover.ts +20 -3
- package/src/env.test.ts +23 -8
- package/src/env.ts +18 -9
- package/src/errors.test.ts +2 -2
- package/src/index.ts +17 -8
- package/src/notices.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/providerDefaults.test.ts +50 -0
- package/src/sdkModels.test.ts +142 -8
- package/src/sdkModels.ts +201 -19
- package/src/types.ts +22 -0
package/src/AiSdkProvider.ts
CHANGED
|
@@ -20,11 +20,13 @@ import type {
|
|
|
20
20
|
ProviderRequestSettlement,
|
|
21
21
|
ProviderResponse,
|
|
22
22
|
ProviderUsage,
|
|
23
|
+
ReasoningPolicy,
|
|
23
24
|
} from "./types.ts";
|
|
24
25
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
25
|
-
import
|
|
26
|
-
import {
|
|
27
|
-
import type
|
|
26
|
+
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
27
|
+
import type { CallWarning, JSONValue } from "ai";
|
|
28
|
+
import { MAX_PROVIDER_TIMEOUT_MS, type Reasoning, type ReasoningResponseStyle } from "./env.ts";
|
|
29
|
+
import { UnsupportedReasoningPolicyError } from "./types.ts";
|
|
28
30
|
import {
|
|
29
31
|
executeAiSdkModel,
|
|
30
32
|
executeOpenAICompatible,
|
|
@@ -59,6 +61,25 @@ export type CacheAffinity =
|
|
|
59
61
|
|
|
60
62
|
export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
|
|
61
63
|
|
|
64
|
+
const isJsonObject = (value: JSONValue | undefined): value is Record<string, JSONValue> =>
|
|
65
|
+
typeof value === "object" && value !== null && !Array.isArray(value);
|
|
66
|
+
|
|
67
|
+
const mergeJsonObjects = (
|
|
68
|
+
left: Record<string, JSONValue | undefined>,
|
|
69
|
+
right: Record<string, JSONValue | undefined>,
|
|
70
|
+
): Record<string, JSONValue | undefined> => Object.fromEntries(
|
|
71
|
+
[...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
|
|
72
|
+
const leftValue = left[key];
|
|
73
|
+
const rightValue = right[key];
|
|
74
|
+
return [
|
|
75
|
+
key,
|
|
76
|
+
isJsonObject(leftValue) && isJsonObject(rightValue)
|
|
77
|
+
? mergeJsonObjects(leftValue, rightValue)
|
|
78
|
+
: rightValue ?? leftValue,
|
|
79
|
+
];
|
|
80
|
+
}),
|
|
81
|
+
);
|
|
82
|
+
|
|
62
83
|
export type AiSdkProviderConfig = {
|
|
63
84
|
model: string;
|
|
64
85
|
url?: string; // OpenAI-compatible chat-completions URL
|
|
@@ -75,6 +96,11 @@ export type AiSdkProviderConfig = {
|
|
|
75
96
|
maxOutputTokens?: number | null;
|
|
76
97
|
outputBudget?: number | null;
|
|
77
98
|
reasoningBudget?: number | null;
|
|
99
|
+
supportedReasoningPolicies?: readonly ReasoningPolicy[];
|
|
100
|
+
// Native AI SDK projection for adaptive. `high` is the graded fallback;
|
|
101
|
+
// provider-default is reserved for a documented native option/default.
|
|
102
|
+
adaptiveReasoning?: "high" | "provider-default";
|
|
103
|
+
adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
|
|
78
104
|
// Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
|
|
79
105
|
// visible output and add an explicit provider reasoning budget. This marker lets the
|
|
80
106
|
// adapter subtract that subset so the resulting wire cap remains PLURNK's
|
|
@@ -125,7 +151,7 @@ export type AiSdkProviderConfig = {
|
|
|
125
151
|
requiresOutputBudget?: boolean;
|
|
126
152
|
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
127
153
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
128
|
-
// { mode: off|adaptive|
|
|
154
|
+
// { mode: off|adaptive|low|medium|high, budget: independent optional cap }.
|
|
129
155
|
// to the backend's mechanism via reasoningStyle; budget is only ever an
|
|
130
156
|
// explicit magnitude, never a hidden activation flag.
|
|
131
157
|
reasoning: Reasoning;
|
|
@@ -194,6 +220,13 @@ class ProviderRequestAccountingError extends Error {
|
|
|
194
220
|
}
|
|
195
221
|
}
|
|
196
222
|
|
|
223
|
+
class ProviderReasoningObserverError extends Error {
|
|
224
|
+
constructor(cause: unknown) {
|
|
225
|
+
super("provider reasoning could not be observed", { cause });
|
|
226
|
+
this.name = "ProviderReasoningObserverError";
|
|
227
|
+
}
|
|
228
|
+
}
|
|
229
|
+
|
|
197
230
|
// Drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
198
231
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
199
232
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -265,19 +298,32 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
|
265
298
|
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
266
299
|
};
|
|
267
300
|
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
if (budget <= 4000) return "medium";
|
|
272
|
-
return "high";
|
|
301
|
+
const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" => {
|
|
302
|
+
if (mode === "low" || mode === "medium" || mode === "high") return mode;
|
|
303
|
+
throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
|
|
273
304
|
};
|
|
274
305
|
|
|
275
|
-
//
|
|
276
|
-
//
|
|
277
|
-
//
|
|
278
|
-
//
|
|
279
|
-
const
|
|
280
|
-
|
|
306
|
+
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
307
|
+
// PLURNK's durable contract names an effort. These fractions match the native
|
|
308
|
+
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
309
|
+
// model's physical maximum. The minimum is imposed by the provider protocol.
|
|
310
|
+
const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
311
|
+
adaptive: 0.6,
|
|
312
|
+
low: 0.1,
|
|
313
|
+
medium: 0.3,
|
|
314
|
+
high: 0.6,
|
|
315
|
+
} satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
|
|
316
|
+
const MANUAL_REASONING_MINIMUM = 1024;
|
|
317
|
+
|
|
318
|
+
const providerWarningMessage = (warning: CallWarning): string => {
|
|
319
|
+
switch (warning.type) {
|
|
320
|
+
case "unsupported":
|
|
321
|
+
case "compatibility":
|
|
322
|
+
return `${warning.type} ${warning.feature}${warning.details === undefined ? "" : `: ${warning.details}`}`;
|
|
323
|
+
case "deprecated": return `deprecated ${warning.setting}: ${warning.message}`;
|
|
324
|
+
case "other": return warning.message;
|
|
325
|
+
}
|
|
326
|
+
};
|
|
281
327
|
|
|
282
328
|
// Body keys the provider owns — a caller's `sampling` passthrough may not set
|
|
283
329
|
// these. Two families:
|
|
@@ -321,6 +367,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
321
367
|
#reasoningBudget: number | null;
|
|
322
368
|
#additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
323
369
|
#reasoning: Reasoning;
|
|
370
|
+
#supportedReasoningPolicies: readonly ReasoningPolicy[];
|
|
371
|
+
#adaptiveReasoning: "high" | "provider-default";
|
|
372
|
+
#adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
324
373
|
#temperature: number;
|
|
325
374
|
#repeatPenalty: number;
|
|
326
375
|
#frequencyPenalty: number;
|
|
@@ -390,6 +439,18 @@ export default class AiSdkProvider implements Provider {
|
|
|
390
439
|
this.#reasoningBudget = config.reasoningBudget ?? null;
|
|
391
440
|
this.#additiveReasoningProvider = config.additiveReasoningProvider;
|
|
392
441
|
this.#reasoning = config.reasoning;
|
|
442
|
+
this.#supportedReasoningPolicies = Object.freeze([
|
|
443
|
+
...new Set(config.supportedReasoningPolicies ?? REASONING_POLICIES),
|
|
444
|
+
]);
|
|
445
|
+
this.#adaptiveReasoning = config.adaptiveReasoning ?? "high";
|
|
446
|
+
this.#adaptiveReasoningProviderOptions = config.adaptiveReasoningProviderOptions;
|
|
447
|
+
if (!this.#supportedReasoningPolicies.includes(this.#reasoning.mode)) {
|
|
448
|
+
throw new UnsupportedReasoningPolicyError(
|
|
449
|
+
config.source ?? "provider",
|
|
450
|
+
this.#reasoning.mode,
|
|
451
|
+
this.#supportedReasoningPolicies,
|
|
452
|
+
);
|
|
453
|
+
}
|
|
393
454
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
394
455
|
// required tuning fields must fail at construction, not silently send
|
|
395
456
|
// undefined sampling on every grammar request.
|
|
@@ -435,6 +496,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
435
496
|
if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
|
|
436
497
|
throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
|
|
437
498
|
}
|
|
499
|
+
if (this.#languageModel === undefined && this.#additiveReasoningProvider !== undefined) {
|
|
500
|
+
throw new Error(`${this.#source}: additive reasoning projection requires an AI SDK model`);
|
|
501
|
+
}
|
|
438
502
|
if (this.#cacheAffinity?.target === "provider-option"
|
|
439
503
|
&& Object.hasOwn(
|
|
440
504
|
this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
|
|
@@ -476,16 +540,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
476
540
|
throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
|
|
477
541
|
}
|
|
478
542
|
if (this.#reasoningStyle === "anthropic"
|
|
479
|
-
&& this.#reasoning.mode
|
|
543
|
+
&& this.#reasoning.mode !== "off"
|
|
544
|
+
&& this.#reasoning.mode !== "adaptive"
|
|
480
545
|
&& this.#reasoning.budget === null
|
|
481
546
|
&& this.#reasoningBudget === null) {
|
|
482
547
|
throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
483
548
|
}
|
|
484
|
-
if (this.#additiveReasoningProvider !== undefined
|
|
485
|
-
&& this.#reasoning.mode === "on"
|
|
486
|
-
&& this.#reasoningBudget === null) {
|
|
487
|
-
throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
|
|
488
|
-
}
|
|
489
549
|
if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
|
|
490
550
|
throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
|
|
491
551
|
}
|
|
@@ -515,6 +575,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
515
575
|
get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
|
|
516
576
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
517
577
|
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
578
|
+
get supportedReasoningPolicies(): readonly ReasoningPolicy[] { return this.#supportedReasoningPolicies; }
|
|
518
579
|
get inputCapacity(): number | null {
|
|
519
580
|
return effectiveInputCapacity({
|
|
520
581
|
contextWindow: this.#contextWindow,
|
|
@@ -628,7 +689,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
628
689
|
case "template": {
|
|
629
690
|
const allowance = mode === "off"
|
|
630
691
|
? 0
|
|
631
|
-
:
|
|
692
|
+
: budget;
|
|
632
693
|
return {
|
|
633
694
|
chat_template_kwargs: { enable_thinking: on },
|
|
634
695
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -637,9 +698,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
637
698
|
}
|
|
638
699
|
case "think": return on ? { think: true } : {};
|
|
639
700
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
701
|
+
case "effort": return mode === "off"
|
|
702
|
+
? {}
|
|
703
|
+
: { reasoning_effort: mode === "adaptive" ? "high" : fixedEffort(mode) };
|
|
643
704
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
644
705
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
645
706
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -649,25 +710,23 @@ export default class AiSdkProvider implements Provider {
|
|
|
649
710
|
// efforts 400.
|
|
650
711
|
case "effort_explicit": return mode === "off"
|
|
651
712
|
? { reasoning_effort: "none" }
|
|
652
|
-
: mode === "
|
|
713
|
+
: mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
653
714
|
// {§deepseek-reasoning-request}
|
|
654
715
|
case "thinking_effort": return mode === "off"
|
|
655
716
|
? { thinking: { type: "disabled" } }
|
|
656
|
-
: mode === "
|
|
717
|
+
: mode === "adaptive" ? { thinking: { type: "enabled" } } : {
|
|
657
718
|
thinking: { type: "enabled" },
|
|
658
|
-
|
|
659
|
-
}
|
|
660
|
-
// Anthropic
|
|
661
|
-
// enabled with the explicit reasoning subset; adaptive →
|
|
662
|
-
// omit (the API default).
|
|
719
|
+
reasoning_effort: fixedEffort(mode),
|
|
720
|
+
};
|
|
721
|
+
// Anthropic-compatible native dynamic or manual budget mode.
|
|
663
722
|
case "anthropic": return mode === "off"
|
|
664
723
|
? { thinking: { type: "disabled" } }
|
|
665
|
-
: mode === "
|
|
724
|
+
: mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
|
|
666
725
|
thinking: {
|
|
667
726
|
type: "enabled",
|
|
668
727
|
budget_tokens: budget!,
|
|
669
728
|
},
|
|
670
|
-
}
|
|
729
|
+
};
|
|
671
730
|
case "none": return {};
|
|
672
731
|
}
|
|
673
732
|
}
|
|
@@ -809,22 +868,26 @@ export default class AiSdkProvider implements Provider {
|
|
|
809
868
|
|
|
810
869
|
#requestProviderOptions(
|
|
811
870
|
workerId: string,
|
|
812
|
-
|
|
871
|
+
nativeReasoningBudget: number | null,
|
|
813
872
|
): AiSdkProviderOptions | undefined {
|
|
814
873
|
const responseOptions = this.#reasoning.mode === "off"
|
|
815
874
|
? undefined
|
|
816
875
|
: this.#reasoningResponseProviderOptions;
|
|
817
|
-
const
|
|
876
|
+
const adaptiveOptions = this.#reasoning.mode === "adaptive"
|
|
877
|
+
&& nativeReasoningBudget === null
|
|
878
|
+
? this.#adaptiveReasoningProviderOptions
|
|
879
|
+
: undefined;
|
|
880
|
+
const nativeReasoning = nativeReasoningBudget !== null
|
|
818
881
|
? this.#additiveReasoningProvider === "anthropic"
|
|
819
|
-
? { anthropic: { thinking: { type: "enabled", budgetTokens:
|
|
882
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
820
883
|
: this.#additiveReasoningProvider === "bedrock"
|
|
821
|
-
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens:
|
|
884
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
|
|
822
885
|
: undefined
|
|
823
886
|
: undefined;
|
|
824
887
|
const options: AiSdkProviderOptions = {};
|
|
825
|
-
for (const part of [responseOptions, nativeReasoning]) {
|
|
888
|
+
for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
|
|
826
889
|
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
827
|
-
options[provider] =
|
|
890
|
+
options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
|
|
828
891
|
}
|
|
829
892
|
}
|
|
830
893
|
if (this.#cacheAffinity?.target === "provider-option") {
|
|
@@ -836,16 +899,38 @@ export default class AiSdkProvider implements Provider {
|
|
|
836
899
|
|
|
837
900
|
#nativeMaxOutputTokens(
|
|
838
901
|
outputBudget: number | null,
|
|
839
|
-
|
|
902
|
+
nativeReasoningBudget: number | null,
|
|
840
903
|
): number | undefined {
|
|
841
904
|
if (outputBudget === null) return undefined;
|
|
842
|
-
return
|
|
843
|
-
|
|
844
|
-
&& reasoningBudget !== null
|
|
845
|
-
? outputBudget - reasoningBudget
|
|
905
|
+
return nativeReasoningBudget !== null
|
|
906
|
+
? outputBudget - nativeReasoningBudget
|
|
846
907
|
: outputBudget;
|
|
847
908
|
}
|
|
848
909
|
|
|
910
|
+
#nativeReasoningBudget(
|
|
911
|
+
outputBudget: number | null,
|
|
912
|
+
configuredReasoningBudget: number | null,
|
|
913
|
+
): number | null {
|
|
914
|
+
if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off") return null;
|
|
915
|
+
if (configuredReasoningBudget !== null) return configuredReasoningBudget;
|
|
916
|
+
if (this.#adaptiveReasoningProviderOptions !== undefined) return null;
|
|
917
|
+
if (outputBudget === null) {
|
|
918
|
+
throw new TypeError(
|
|
919
|
+
`${this.#source}: manual provider reasoning requires a resolved total output budget`,
|
|
920
|
+
);
|
|
921
|
+
}
|
|
922
|
+
if (outputBudget <= MANUAL_REASONING_MINIMUM) {
|
|
923
|
+
throw new TypeError(
|
|
924
|
+
`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`,
|
|
925
|
+
);
|
|
926
|
+
}
|
|
927
|
+
const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
|
|
928
|
+
return Math.min(
|
|
929
|
+
outputBudget - 1,
|
|
930
|
+
Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)),
|
|
931
|
+
);
|
|
932
|
+
}
|
|
933
|
+
|
|
849
934
|
#accounting(
|
|
850
935
|
outcome: ProviderRequestAccounting["outcome"],
|
|
851
936
|
usage: ProviderUsage | undefined,
|
|
@@ -864,7 +949,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
864
949
|
});
|
|
865
950
|
}
|
|
866
951
|
|
|
867
|
-
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
|
|
952
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, observeReasoning, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
|
|
868
953
|
// {§provider-interface} The worker identity is required.
|
|
869
954
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
870
955
|
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
@@ -895,6 +980,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
895
980
|
);
|
|
896
981
|
}
|
|
897
982
|
const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
|
|
983
|
+
const nativeReasoningBudget = this.#nativeReasoningBudget(
|
|
984
|
+
capacity.outputBudget,
|
|
985
|
+
capacity.reasoningBudget,
|
|
986
|
+
);
|
|
898
987
|
|
|
899
988
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
900
989
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
@@ -937,7 +1026,25 @@ export default class AiSdkProvider implements Provider {
|
|
|
937
1026
|
: operationTimeout === undefined
|
|
938
1027
|
? signal
|
|
939
1028
|
: AbortSignal.any([signal, operationTimeout]);
|
|
1029
|
+
const emitReasoning = observeReasoning === undefined
|
|
1030
|
+
? undefined
|
|
1031
|
+
: (delta: string): void => {
|
|
1032
|
+
if (delta.length === 0) return;
|
|
1033
|
+
try {
|
|
1034
|
+
observeReasoning(delta);
|
|
1035
|
+
} catch (cause) {
|
|
1036
|
+
throw new ProviderReasoningObserverError(cause);
|
|
1037
|
+
}
|
|
1038
|
+
};
|
|
1039
|
+
let successfulReasoningStream = "";
|
|
940
1040
|
const executeRequest = async () => {
|
|
1041
|
+
let requestReasoningStream = "";
|
|
1042
|
+
const observeRequestReasoning = emitReasoning === undefined
|
|
1043
|
+
? undefined
|
|
1044
|
+
: (delta: string): void => {
|
|
1045
|
+
requestReasoningStream += delta;
|
|
1046
|
+
emitReasoning(delta);
|
|
1047
|
+
};
|
|
941
1048
|
let settle: ProviderRequestSettlement | undefined;
|
|
942
1049
|
try {
|
|
943
1050
|
settle = await observeRequest?.({
|
|
@@ -1004,11 +1111,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
1004
1111
|
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
1005
1112
|
streaming: this.#streaming,
|
|
1006
1113
|
captureRawBody: this.#rawBody,
|
|
1114
|
+
...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
|
|
1007
1115
|
})
|
|
1008
1116
|
: await executeAiSdkModel({
|
|
1009
1117
|
languageModel: this.#languageModel,
|
|
1010
1118
|
headers: requestHeaders,
|
|
1011
|
-
providerOptions: this.#requestProviderOptions(workerId,
|
|
1119
|
+
providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
|
|
1012
1120
|
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
1013
1121
|
messages,
|
|
1014
1122
|
signal: operationSignal,
|
|
@@ -1017,6 +1125,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1017
1125
|
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
1018
1126
|
streaming: this.#streaming,
|
|
1019
1127
|
captureRawBody: this.#rawBody,
|
|
1128
|
+
...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
|
|
1020
1129
|
temperature: this.#tuningFloors
|
|
1021
1130
|
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
1022
1131
|
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
@@ -1032,17 +1141,12 @@ export default class AiSdkProvider implements Provider {
|
|
|
1032
1141
|
? sampling.stop
|
|
1033
1142
|
: undefined,
|
|
1034
1143
|
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
1035
|
-
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget,
|
|
1144
|
+
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
|
|
1036
1145
|
reasoning: this.#reasoning.mode === "off"
|
|
1037
1146
|
? "none"
|
|
1038
1147
|
: this.#reasoning.mode === "adaptive"
|
|
1039
|
-
?
|
|
1040
|
-
: this.#
|
|
1041
|
-
? "provider-default"
|
|
1042
|
-
: effortFromReasoning({
|
|
1043
|
-
mode: this.#reasoning.mode,
|
|
1044
|
-
budget: capacity.reasoningBudget,
|
|
1045
|
-
}),
|
|
1148
|
+
? this.#adaptiveReasoning
|
|
1149
|
+
: fixedEffort(this.#reasoning.mode),
|
|
1046
1150
|
});
|
|
1047
1151
|
} catch (error) {
|
|
1048
1152
|
const failure = transportFailureEvidence(error);
|
|
@@ -1054,6 +1158,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1054
1158
|
);
|
|
1055
1159
|
throw error;
|
|
1056
1160
|
}
|
|
1161
|
+
successfulReasoningStream = requestReasoningStream;
|
|
1057
1162
|
await settleAccounting(
|
|
1058
1163
|
"response",
|
|
1059
1164
|
response.usage,
|
|
@@ -1071,7 +1176,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
1071
1176
|
raw = await retry(executeRequest);
|
|
1072
1177
|
} catch (err) {
|
|
1073
1178
|
if (err instanceof ProviderRequestObserverError
|
|
1074
|
-
|| err instanceof ProviderRequestAccountingError
|
|
1179
|
+
|| err instanceof ProviderRequestAccountingError
|
|
1180
|
+
|| err instanceof ProviderReasoningObserverError) throw err.cause;
|
|
1075
1181
|
if (signal?.aborted) throw err;
|
|
1076
1182
|
if (operationTimeout?.aborted) {
|
|
1077
1183
|
const timeout = new ProviderTimeoutError("operation", this.#operationTimeoutMs, err);
|
|
@@ -1150,8 +1256,27 @@ export default class AiSdkProvider implements Provider {
|
|
|
1150
1256
|
raw.content = projectedReasoning.content;
|
|
1151
1257
|
raw.reasoning = projectedReasoning.reasoning;
|
|
1152
1258
|
}
|
|
1259
|
+
if (emitReasoning !== undefined
|
|
1260
|
+
&& raw.reasoning.length > 0
|
|
1261
|
+
&& successfulReasoningStream !== raw.reasoning) {
|
|
1262
|
+
const missing = raw.reasoning.startsWith(successfulReasoningStream)
|
|
1263
|
+
? raw.reasoning.slice(successfulReasoningStream.length)
|
|
1264
|
+
: successfulReasoningStream.length === 0
|
|
1265
|
+
? raw.reasoning
|
|
1266
|
+
: `\n${raw.reasoning}`;
|
|
1267
|
+
emitReasoning(missing);
|
|
1268
|
+
}
|
|
1153
1269
|
|
|
1154
1270
|
let notices: ProviderNotice[] | undefined;
|
|
1271
|
+
for (const warning of raw.warnings) {
|
|
1272
|
+
(notices ??= []).push({
|
|
1273
|
+
source: this.#source,
|
|
1274
|
+
kind: "provider_warning",
|
|
1275
|
+
level: "warn",
|
|
1276
|
+
message: providerWarningMessage(warning),
|
|
1277
|
+
position: null,
|
|
1278
|
+
});
|
|
1279
|
+
}
|
|
1155
1280
|
const usage = raw.usage;
|
|
1156
1281
|
if (sendGrammar !== undefined
|
|
1157
1282
|
&& this.tokenize !== undefined
|
package/src/Mock.test.ts
CHANGED
|
@@ -40,7 +40,11 @@ test("Mock: prompt counting is exact for its declared mock vocabulary", async ()
|
|
|
40
40
|
|
|
41
41
|
test("Mock: generate resolves a valid ProviderResponse shape", async () => {
|
|
42
42
|
const m = build([{ assistant: { content: "hello", reasoning: "cot" } }]);
|
|
43
|
-
const
|
|
43
|
+
const reasoning: string[] = [];
|
|
44
|
+
const { assistant, assistantRaw, accounting } = await m.generate({
|
|
45
|
+
messages: [],
|
|
46
|
+
observeReasoning: (delta) => reasoning.push(delta),
|
|
47
|
+
});
|
|
44
48
|
assert.equal(assistant.content, "hello");
|
|
45
49
|
assert.equal(assistant.reasoning, "cot");
|
|
46
50
|
assert.deepEqual(accounting[0]?.usage, {
|
|
@@ -56,6 +60,7 @@ test("Mock: generate resolves a valid ProviderResponse shape", async () => {
|
|
|
56
60
|
assert.equal(assistant.finishReason, "stop");
|
|
57
61
|
assert.equal(assistant.model, "mock");
|
|
58
62
|
assert.equal(assistantRaw, null); // present, defaulted
|
|
63
|
+
assert.deepEqual(reasoning, ["cot"]);
|
|
59
64
|
});
|
|
60
65
|
|
|
61
66
|
test("Mock: generate applies caller-supplied overrides", async () => {
|
package/src/Mock.ts
CHANGED
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
9
|
import { resolveGenerationEnvelopeFromEnv } from "./env.ts";
|
|
10
|
+
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
10
11
|
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
11
12
|
import { ProviderError } from "./errors.ts";
|
|
12
13
|
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
@@ -69,6 +70,7 @@ export default class Mock implements Provider {
|
|
|
69
70
|
get maxOutputTokens(): number | null { return null; }
|
|
70
71
|
get outputBudget(): number | null { return this.#outputBudget; }
|
|
71
72
|
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
73
|
+
get supportedReasoningPolicies() { return REASONING_POLICIES; }
|
|
72
74
|
get inputCapacity(): number | null {
|
|
73
75
|
return effectiveInputCapacity({
|
|
74
76
|
contextWindow: this.#contextWindow,
|
|
@@ -113,7 +115,7 @@ export default class Mock implements Provider {
|
|
|
113
115
|
});
|
|
114
116
|
}
|
|
115
117
|
|
|
116
|
-
async generate({ messages, maxOutputTokens, signal, grammar, observeRequest }: MockGenerateArgs): Promise<MockReturnedResponse> {
|
|
118
|
+
async generate({ messages, maxOutputTokens, signal, grammar, observeRequest, observeReasoning }: MockGenerateArgs): Promise<MockReturnedResponse> {
|
|
117
119
|
// Honor abort before consuming the queue — an aborted call makes no
|
|
118
120
|
// "wire call" and must not exhaust a queued response
|
|
119
121
|
// ({§provider-failure-normalization}).
|
|
@@ -167,6 +169,9 @@ export default class Mock implements Provider {
|
|
|
167
169
|
model: a.model ?? "mock",
|
|
168
170
|
...(a.ops !== undefined ? { ops: a.ops } : {}),
|
|
169
171
|
};
|
|
172
|
+
if (assistant.reasoning !== null && assistant.reasoning.length > 0) {
|
|
173
|
+
observeReasoning?.(assistant.reasoning);
|
|
174
|
+
}
|
|
170
175
|
const grammarEvidence = next.grammarEvidence
|
|
171
176
|
?? (grammar === undefined
|
|
172
177
|
? undefined
|
package/src/Pool.test.ts
CHANGED
|
@@ -53,6 +53,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
53
53
|
maxOutputTokens: opts.maxOutputTokens ?? null,
|
|
54
54
|
outputBudget: opts.outputBudget ?? null,
|
|
55
55
|
reasoningBudget: opts.reasoningBudget ?? null,
|
|
56
|
+
supportedReasoningPolicies: ["off", "adaptive", "low", "medium", "high"],
|
|
56
57
|
inputCapacity: effectiveInputCapacity({
|
|
57
58
|
contextWindow: opts.window === undefined ? 48_000 : opts.window,
|
|
58
59
|
maxInputTokens: opts.maxInputTokens ?? null,
|
package/src/Pool.ts
CHANGED
|
@@ -7,6 +7,7 @@ import Meta, {
|
|
|
7
7
|
type PluginAttributionContext,
|
|
8
8
|
} from "@plurnk/plurnk-meta";
|
|
9
9
|
import { effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, requestCapacityDecision } from "./capacity.ts";
|
|
10
|
+
import type { ReasoningPolicy } from "./types.ts";
|
|
10
11
|
|
|
11
12
|
// A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
|
|
12
13
|
// transient retries before throwing one of these, so re-hitting the same
|
|
@@ -89,6 +90,10 @@ export default class Pool implements Provider {
|
|
|
89
90
|
get maxOutputTokens(): number | null { return this.#minimumKnown((provider) => provider.maxOutputTokens); }
|
|
90
91
|
get outputBudget(): number | null { return this.#minimumKnown((provider) => provider.outputBudget); }
|
|
91
92
|
get reasoningBudget(): number | null { return this.#minimumKnown((provider) => provider.reasoningBudget); }
|
|
93
|
+
get supportedReasoningPolicies(): readonly ReasoningPolicy[] {
|
|
94
|
+
return this.#backends[0].supportedReasoningPolicies.filter((policy) =>
|
|
95
|
+
this.#backends.every((provider) => provider.supportedReasoningPolicies.includes(policy)));
|
|
96
|
+
}
|
|
92
97
|
get inputCapacity(): number | null { return this.#minimumKnown((provider) => provider.inputCapacity); }
|
|
93
98
|
|
|
94
99
|
// Served id / capabilities aggregate CONSERVATIVELY: a worker could land on any
|