@plurnk/plurnk-providers 1.11.0 → 1.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +5 -1
- package/SPEC.md +57 -31
- package/dist/AiSdkProvider.d.ts +5 -1
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +29 -7
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +13 -3
- package/dist/accounting.js.map +1 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +91 -34
- package/dist/catalogProvider.js.map +1 -1
- package/dist/index.d.ts +2 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -1
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +103 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +2 -8
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.ts +42 -11
- package/src/accounting.test.ts +39 -0
- package/src/accounting.ts +13 -4
- package/src/catalogProvider.test.ts +80 -13
- package/src/catalogProvider.ts +127 -38
- package/src/index.ts +2 -1
- package/src/sdkModels.test.ts +42 -1
- package/src/sdkModels.ts +144 -7
- package/src/types.ts +4 -15
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@plurnk/plurnk-providers",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.12.0",
|
|
4
4
|
"description": "PLURNK's stable model-provider contract and AI SDK adapter.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"plurnk",
|
|
@@ -71,6 +71,7 @@
|
|
|
71
71
|
"prepack": "npm run build"
|
|
72
72
|
},
|
|
73
73
|
"dependencies": {
|
|
74
|
+
"@ai-sdk/provider": "^4.0.8",
|
|
74
75
|
"@ai-sdk/amazon-bedrock": "^5.0.0",
|
|
75
76
|
"@ai-sdk/anthropic": "^4.0.0",
|
|
76
77
|
"@ai-sdk/cerebras": "^3.0.0",
|
|
@@ -83,11 +84,11 @@
|
|
|
83
84
|
"@ai-sdk/togetherai": "^3.0.0",
|
|
84
85
|
"@ai-sdk/xai": "^4.0.0",
|
|
85
86
|
"@openrouter/ai-sdk-provider": "^3.0.0",
|
|
86
|
-
"@plurnk/gbnf": "1.
|
|
87
|
-
"@plurnk/plurnk-aliases": "1.
|
|
88
|
-
"@plurnk/plurnk-contracts": "1.
|
|
89
|
-
"@plurnk/plurnk-meta": "1.
|
|
90
|
-
"@plurnk/plurnk-models": "1.
|
|
87
|
+
"@plurnk/gbnf": "1.12.0",
|
|
88
|
+
"@plurnk/plurnk-aliases": "1.12.0",
|
|
89
|
+
"@plurnk/plurnk-contracts": "1.12.0",
|
|
90
|
+
"@plurnk/plurnk-meta": "1.12.0",
|
|
91
|
+
"@plurnk/plurnk-models": "1.12.0",
|
|
91
92
|
"ai": "^7.0.37",
|
|
92
93
|
"zod": "^4.4.3"
|
|
93
94
|
},
|
package/src/AiSdkProvider.ts
CHANGED
|
@@ -52,6 +52,9 @@ export type ProviderFetch = typeof globalThis.fetch;
|
|
|
52
52
|
// mapping retains any backend-specific omission/explicit-disable constraint.
|
|
53
53
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "effort_required" | "thinking_effort" | "template" | "anthropic";
|
|
54
54
|
|
|
55
|
+
export type NativeReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
|
|
56
|
+
export type CompatibleReasoningEffort = NativeReasoningEffort | "max";
|
|
57
|
+
|
|
55
58
|
// GBNF transport is a local llama-server capability. "none" means no
|
|
56
59
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
57
60
|
export type GrammarStyle = "none" | "llamacpp";
|
|
@@ -98,9 +101,15 @@ export type AiSdkProviderConfig = {
|
|
|
98
101
|
outputBudget?: number | null;
|
|
99
102
|
reasoningBudget?: number | null;
|
|
100
103
|
supportedReasoningPolicies?: readonly ReasoningPolicy[];
|
|
101
|
-
// Native AI SDK projection for adaptive.
|
|
102
|
-
// provider-default is reserved for a documented native
|
|
103
|
-
|
|
104
|
+
// Native AI SDK projection for adaptive. Models.dev supplies the route's
|
|
105
|
+
// admissible values; provider-default is reserved for a documented native
|
|
106
|
+
// dynamic option/default or a route with no caller-selectable effort.
|
|
107
|
+
adaptiveReasoning?: NativeReasoningEffort | "provider-default";
|
|
108
|
+
// OpenAI-compatible effort transports accept route-native values beyond
|
|
109
|
+
// the AI SDK's generic vocabulary. These are still catalog facts, not
|
|
110
|
+
// provider-name branches.
|
|
111
|
+
compatibleAdaptiveReasoning?: CompatibleReasoningEffort | "provider-default";
|
|
112
|
+
compatibleOffReasoning?: "none";
|
|
104
113
|
adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
|
|
105
114
|
// Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
|
|
106
115
|
// visible output and add an explicit provider reasoning budget. This marker lets the
|
|
@@ -369,7 +378,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
369
378
|
#additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
370
379
|
#reasoning: Reasoning;
|
|
371
380
|
#supportedReasoningPolicies: readonly ReasoningPolicy[];
|
|
372
|
-
#adaptiveReasoning:
|
|
381
|
+
#adaptiveReasoning: NativeReasoningEffort | "provider-default";
|
|
382
|
+
#compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
|
|
383
|
+
#compatibleOffReasoning: "none" | undefined;
|
|
373
384
|
#adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
374
385
|
#temperature: number;
|
|
375
386
|
#repeatPenalty: number;
|
|
@@ -444,6 +455,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
444
455
|
...new Set(config.supportedReasoningPolicies ?? REASONING_POLICIES),
|
|
445
456
|
]);
|
|
446
457
|
this.#adaptiveReasoning = config.adaptiveReasoning ?? "high";
|
|
458
|
+
this.#compatibleAdaptiveReasoning = config.compatibleAdaptiveReasoning ?? "high";
|
|
459
|
+
this.#compatibleOffReasoning = config.compatibleOffReasoning;
|
|
447
460
|
this.#adaptiveReasoningProviderOptions = config.adaptiveReasoningProviderOptions;
|
|
448
461
|
if (!this.#supportedReasoningPolicies.includes(this.#reasoning.mode)) {
|
|
449
462
|
throw new UnsupportedReasoningPolicyError(
|
|
@@ -700,13 +713,31 @@ export default class AiSdkProvider implements Provider {
|
|
|
700
713
|
case "think": return on ? { think: true } : {};
|
|
701
714
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
702
715
|
case "effort": return mode === "off"
|
|
703
|
-
?
|
|
704
|
-
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
716
|
+
? this.#compatibleOffReasoning === undefined
|
|
717
|
+
? {}
|
|
718
|
+
: { reasoning_effort: this.#compatibleOffReasoning }
|
|
719
|
+
: mode === "adaptive"
|
|
720
|
+
? this.#compatibleAdaptiveReasoning === "provider-default"
|
|
721
|
+
? {}
|
|
722
|
+
: { reasoning_effort: this.#compatibleAdaptiveReasoning }
|
|
723
|
+
: { reasoning_effort: fixedEffort(mode) };
|
|
724
|
+
// Graded reasoning is mandatory when the route advertises an effort
|
|
725
|
+
// value. Cataloged routes supply the exact strongest legal value;
|
|
726
|
+
// construction rejects an unsupported off or fixed policy.
|
|
727
|
+
case "effort_required": {
|
|
728
|
+
if (mode === "off") {
|
|
729
|
+
if (this.#compatibleOffReasoning === undefined) {
|
|
730
|
+
throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
|
|
731
|
+
}
|
|
732
|
+
return { reasoning_effort: this.#compatibleOffReasoning };
|
|
733
|
+
}
|
|
734
|
+
if (mode === "adaptive") {
|
|
735
|
+
return this.#compatibleAdaptiveReasoning === "provider-default"
|
|
736
|
+
? {}
|
|
737
|
+
: { reasoning_effort: this.#compatibleAdaptiveReasoning };
|
|
738
|
+
}
|
|
739
|
+
return { reasoning_effort: fixedEffort(mode) };
|
|
740
|
+
}
|
|
710
741
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
711
742
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
712
743
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
package/src/accounting.test.ts
CHANGED
|
@@ -120,3 +120,42 @@ test("aggregateProviderAccounting omits unknown nested usage fields from its JSO
|
|
|
120
120
|
outputTokenDetails: { reasoningTokens: 2 },
|
|
121
121
|
});
|
|
122
122
|
});
|
|
123
|
+
|
|
124
|
+
test("aggregateProviderAccounting does not invent complete detail partitions across heterogeneous requests", () => {
|
|
125
|
+
const accounting = aggregateProviderAccounting([
|
|
126
|
+
{
|
|
127
|
+
provider: "provider:detailed",
|
|
128
|
+
model: "m",
|
|
129
|
+
outcome: "response",
|
|
130
|
+
usage: {
|
|
131
|
+
inputTokens: 2,
|
|
132
|
+
outputTokens: 0,
|
|
133
|
+
totalTokens: 2,
|
|
134
|
+
inputTokenDetails: {
|
|
135
|
+
noCacheTokens: 2,
|
|
136
|
+
cacheReadTokens: 0,
|
|
137
|
+
cacheWriteTokens: 0,
|
|
138
|
+
},
|
|
139
|
+
},
|
|
140
|
+
cost: { kind: "unknown", reason: "no direct cost" },
|
|
141
|
+
},
|
|
142
|
+
{
|
|
143
|
+
provider: "provider:totals-only",
|
|
144
|
+
model: "m",
|
|
145
|
+
outcome: "response",
|
|
146
|
+
usage: { inputTokens: 4, outputTokens: 0, totalTokens: 4 },
|
|
147
|
+
cost: { kind: "unknown", reason: "no direct cost" },
|
|
148
|
+
},
|
|
149
|
+
]);
|
|
150
|
+
|
|
151
|
+
assert.deepEqual(accounting.usage, {
|
|
152
|
+
inputTokens: 6,
|
|
153
|
+
outputTokens: 0,
|
|
154
|
+
totalTokens: 6,
|
|
155
|
+
inputTokenDetails: {
|
|
156
|
+
noCacheTokens: 2,
|
|
157
|
+
cacheReadTokens: 0,
|
|
158
|
+
cacheWriteTokens: 0,
|
|
159
|
+
},
|
|
160
|
+
});
|
|
161
|
+
});
|
package/src/accounting.ts
CHANGED
|
@@ -132,7 +132,12 @@ const sumKnown = (
|
|
|
132
132
|
const known = requests
|
|
133
133
|
.map((request) => request.usage === undefined ? undefined : read(request.usage))
|
|
134
134
|
.filter((value): value is number => value !== undefined);
|
|
135
|
-
|
|
135
|
+
if (known.length === 0) return undefined;
|
|
136
|
+
const sum = known.reduce((total, value) => total + value, 0);
|
|
137
|
+
if (!Number.isSafeInteger(sum)) {
|
|
138
|
+
throw new TypeError("aggregate provider usage exceeds the safe-integer range");
|
|
139
|
+
}
|
|
140
|
+
return sum;
|
|
136
141
|
};
|
|
137
142
|
|
|
138
143
|
export const aggregateProviderAccounting = (
|
|
@@ -179,16 +184,20 @@ export const aggregateProviderAccounting = (
|
|
|
179
184
|
...(textTokens === undefined ? {} : { textTokens }),
|
|
180
185
|
...(reasoningTokens === undefined ? {} : { reasoningTokens }),
|
|
181
186
|
};
|
|
182
|
-
|
|
187
|
+
// Each physical request was validated above. Aggregate fields deliberately
|
|
188
|
+
// sum their own known evidence independently: heterogeneous providers may
|
|
189
|
+
// report different detail subsets, so the projection must not reinterpret
|
|
190
|
+
// their union as one complete per-request partition.
|
|
191
|
+
const usage: ProviderUsage | null = inputTokens === undefined && outputTokens === undefined && totalTokens === undefined
|
|
183
192
|
&& inputTokenDetails === undefined && outputTokenDetails === undefined
|
|
184
193
|
? null
|
|
185
|
-
:
|
|
194
|
+
: {
|
|
186
195
|
...(inputTokens === undefined ? {} : { inputTokens }),
|
|
187
196
|
...(outputTokens === undefined ? {} : { outputTokens }),
|
|
188
197
|
...(totalTokens === undefined ? {} : { totalTokens }),
|
|
189
198
|
...(inputTokenDetails === undefined ? {} : { inputTokenDetails }),
|
|
190
199
|
...(outputTokenDetails === undefined ? {} : { outputTokenDetails }),
|
|
191
|
-
}
|
|
200
|
+
};
|
|
192
201
|
return {
|
|
193
202
|
requests: [...requests],
|
|
194
203
|
usage,
|
|
@@ -48,7 +48,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
|
|
|
48
48
|
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
49
49
|
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
50
50
|
}, "deepseek-v4-flash");
|
|
51
|
-
assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "high"]);
|
|
51
|
+
assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high"]);
|
|
52
52
|
|
|
53
53
|
assert.throws(
|
|
54
54
|
() => catalogProviderFromEnv("deepseek", {
|
|
@@ -57,7 +57,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
|
|
|
57
57
|
PLURNK_PROVIDERS_REASONING: "medium",
|
|
58
58
|
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
59
59
|
}, "deepseek-v4-flash"),
|
|
60
|
-
/reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
|
|
60
|
+
/reasoning policy 'medium' is unsupported; supported policies: off, adaptive, low, high/,
|
|
61
61
|
);
|
|
62
62
|
|
|
63
63
|
const mistral = catalogProviderFromEnv("mistral", {
|
|
@@ -82,7 +82,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
|
|
|
82
82
|
assert.deepEqual(gemini?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Gemini 3's mandatory minimum is not advertised as off");
|
|
83
83
|
});
|
|
84
84
|
|
|
85
|
-
test("
|
|
85
|
+
test("Models.dev controls Cloudflare's exact effort vocabulary", async () => {
|
|
86
86
|
const bodies: Record<string, unknown>[] = [];
|
|
87
87
|
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
88
88
|
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
@@ -91,7 +91,7 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
|
|
|
91
91
|
id: "cloudflare-effort",
|
|
92
92
|
object: "chat.completion.chunk",
|
|
93
93
|
created: 1,
|
|
94
|
-
model: "@cf/
|
|
94
|
+
model: "@cf/qwen/qwen3.8-27b",
|
|
95
95
|
choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
|
|
96
96
|
})}`,
|
|
97
97
|
"data: [DONE]",
|
|
@@ -106,14 +106,14 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
|
|
|
106
106
|
const low = catalogProviderFromEnv("cloudflare", {
|
|
107
107
|
...cloudflareEnv,
|
|
108
108
|
PLURNK_PROVIDERS_REASONING: "low",
|
|
109
|
-
}, "@cf/
|
|
110
|
-
assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium"
|
|
109
|
+
}, "@cf/qwen/qwen3.8-27b");
|
|
110
|
+
assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium"]);
|
|
111
111
|
await low?.generate({ workerId: "cloudflare-low", messages: [{ role: "user", content: "hello" }] });
|
|
112
112
|
|
|
113
113
|
const adaptive = catalogProviderFromEnv("cloudflare", {
|
|
114
114
|
...cloudflareEnv,
|
|
115
115
|
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
116
|
-
}, "@cf/
|
|
116
|
+
}, "@cf/qwen/qwen3.8-27b");
|
|
117
117
|
await adaptive?.generate({ workerId: "cloudflare-adaptive", messages: [{ role: "user", content: "hello" }] });
|
|
118
118
|
|
|
119
119
|
const nonReasoning = catalogProviderFromEnv("cloudflare", {
|
|
@@ -123,10 +123,71 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
|
|
|
123
123
|
assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
124
124
|
await nonReasoning?.generate({ workerId: "cloudflare-granite", messages: [{ role: "user", content: "hello" }] });
|
|
125
125
|
|
|
126
|
-
|
|
126
|
+
const ungradedReasoner = catalogProviderFromEnv("cloudflare", {
|
|
127
|
+
...cloudflareEnv,
|
|
128
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
129
|
+
}, "@cf/zai-org/glm-5.3-flash");
|
|
130
|
+
assert.deepEqual(ungradedReasoner?.supportedReasoningPolicies, ["adaptive"]);
|
|
131
|
+
await ungradedReasoner?.generate({ workerId: "cloudflare-ungraded", messages: [{ role: "user", content: "hello" }] });
|
|
132
|
+
|
|
133
|
+
assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["low", "xhigh", undefined, undefined]);
|
|
134
|
+
assert.throws(
|
|
135
|
+
() => catalogProviderFromEnv("cloudflare", cloudflareEnv, "@cf/qwen/qwen3.8-27b"),
|
|
136
|
+
/reasoning policy 'off' is unsupported; supported policies: adaptive, low, medium/,
|
|
137
|
+
);
|
|
127
138
|
assert.throws(
|
|
128
|
-
() => catalogProviderFromEnv("cloudflare",
|
|
129
|
-
|
|
139
|
+
() => catalogProviderFromEnv("cloudflare", {
|
|
140
|
+
...cloudflareEnv,
|
|
141
|
+
PLURNK_PROVIDERS_REASONING: "high",
|
|
142
|
+
}, "@cf/qwen/qwen3.8-27b"),
|
|
143
|
+
/reasoning policy 'high' is unsupported; supported policies: adaptive, low, medium/,
|
|
144
|
+
);
|
|
145
|
+
});
|
|
146
|
+
|
|
147
|
+
test("an operator-declared effort vocabulary extends Models.dev's for a provider's reasoning routes (#439)", async () => {
|
|
148
|
+
const bodies: Record<string, unknown>[] = [];
|
|
149
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
150
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
151
|
+
return new Response([
|
|
152
|
+
`data: ${JSON.stringify({
|
|
153
|
+
id: "cloudflare-declared",
|
|
154
|
+
object: "chat.completion.chunk",
|
|
155
|
+
created: 1,
|
|
156
|
+
model: "@cf/zai-org/glm-5.3-flash",
|
|
157
|
+
choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
|
|
158
|
+
})}`,
|
|
159
|
+
"data: [DONE]",
|
|
160
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
161
|
+
});
|
|
162
|
+
const declaredEnv = {
|
|
163
|
+
...env,
|
|
164
|
+
CLOUDFLARE_ACCOUNT_ID: "account",
|
|
165
|
+
CLOUDFLARE_API_KEY: "token",
|
|
166
|
+
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_STYLE: "effort_required",
|
|
167
|
+
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "low, medium,high",
|
|
168
|
+
};
|
|
169
|
+
// Models.dev lists this route as reasoning with no effort vocabulary; the declaration supplies one.
|
|
170
|
+
const low = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "low" }, "@cf/zai-org/glm-5.3-flash");
|
|
171
|
+
assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"]);
|
|
172
|
+
await low?.generate({ workerId: "declared-low", messages: [{ role: "user", content: "hello" }] });
|
|
173
|
+
const adaptive = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/zai-org/glm-5.3-flash");
|
|
174
|
+
await adaptive?.generate({ workerId: "declared-adaptive", messages: [{ role: "user", content: "hello" }] });
|
|
175
|
+
assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["low", "high"], "a fixed level keeps its name; adaptive takes the strongest declared effort");
|
|
176
|
+
// A declared `none` admits `off` on an effort transport; the catalog's own vocabulary stays in the union.
|
|
177
|
+
const withOff = catalogProviderFromEnv("cloudflare", {
|
|
178
|
+
...declaredEnv,
|
|
179
|
+
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "none,high",
|
|
180
|
+
}, "@cf/qwen/qwen3.8-27b");
|
|
181
|
+
assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high"]);
|
|
182
|
+
// The declaration never turns a non-reasoning route into a reasoning one.
|
|
183
|
+
const nonReasoning = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/ibm-granite/granite-4.0-h-micro");
|
|
184
|
+
assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
185
|
+
assert.throws(
|
|
186
|
+
() => catalogProviderFromEnv("cloudflare", {
|
|
187
|
+
...declaredEnv,
|
|
188
|
+
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "low,turbo",
|
|
189
|
+
}, "@cf/zai-org/glm-5.3-flash"),
|
|
190
|
+
/PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS has invalid value "turbo"; declarable efforts: none, minimal, low, medium, high, xhigh, max/,
|
|
130
191
|
);
|
|
131
192
|
});
|
|
132
193
|
|
|
@@ -201,7 +262,7 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
201
262
|
assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
|
|
202
263
|
});
|
|
203
264
|
|
|
204
|
-
test("xAI's native chat contract
|
|
265
|
+
test("xAI's native chat contract requests its strongest cataloged effort and caps the complete reasoning response", async () => {
|
|
205
266
|
let call: { headers: Headers; body: Record<string, unknown> } | undefined;
|
|
206
267
|
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
207
268
|
call = {
|
|
@@ -254,7 +315,7 @@ test("xAI's native chat contract affirmatively requests adaptive high and caps t
|
|
|
254
315
|
});
|
|
255
316
|
|
|
256
317
|
assert.equal(call?.body.max_completion_tokens, 16);
|
|
257
|
-
assert.equal(call?.body.reasoning_effort, "
|
|
318
|
+
assert.equal(call?.body.reasoning_effort, "xhigh", "adaptive uses the strongest route-advertised generic effort");
|
|
258
319
|
assert.equal("max_tokens" in (call?.body ?? {}), false);
|
|
259
320
|
assert.equal(call?.headers.get("x-grok-conv-id"), "xai-worker");
|
|
260
321
|
assert.equal(result?.assistant.reasoning, "consider");
|
|
@@ -365,7 +426,7 @@ test("Meta Muse adaptive reasoning is requested even when the endpoint returns n
|
|
|
365
426
|
messages: [{ role: "user", content: "hello" }],
|
|
366
427
|
});
|
|
367
428
|
|
|
368
|
-
assert.equal(body?.reasoning_effort, "
|
|
429
|
+
assert.equal(body?.reasoning_effort, "xhigh");
|
|
369
430
|
assert.equal(result?.assistant.reasoning, null);
|
|
370
431
|
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 37);
|
|
371
432
|
});
|
|
@@ -504,6 +565,7 @@ test("native Anthropic adaptive policy uses adaptive thinking rather than a fixe
|
|
|
504
565
|
contextWindow: 16_384,
|
|
505
566
|
maxOutputTokens: 8_192,
|
|
506
567
|
reasoning: true,
|
|
568
|
+
reasoningOptions: [{ type: "toggle" }],
|
|
507
569
|
attachment: true,
|
|
508
570
|
toolCall: true,
|
|
509
571
|
modalities: { input: ["text", "image"], output: ["text"] },
|
|
@@ -635,8 +697,10 @@ test("cataloged unknown model fails unless its context is explicit", () => {
|
|
|
635
697
|
...env,
|
|
636
698
|
XAI_API_KEY: "test-key",
|
|
637
699
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
700
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
638
701
|
}, "not-in-the-catalog");
|
|
639
702
|
assert.equal(provider?.contextWindow, 8192);
|
|
703
|
+
assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
640
704
|
});
|
|
641
705
|
|
|
642
706
|
test("Models.dev is the only fallback rate table", async () => {
|
|
@@ -662,6 +726,8 @@ test("Models.dev is the only fallback rate table", async () => {
|
|
|
662
726
|
const cataloged = catalogProviderFromEnv("deepseek", {
|
|
663
727
|
...env,
|
|
664
728
|
DEEPSEEK_API_KEY: "test-key",
|
|
729
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
730
|
+
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
665
731
|
}, "deepseek-v4-flash");
|
|
666
732
|
assert.notEqual(cataloged, null);
|
|
667
733
|
const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
|
|
@@ -675,6 +741,7 @@ test("Models.dev is the only fallback rate table", async () => {
|
|
|
675
741
|
...env,
|
|
676
742
|
XAI_API_KEY: "test-key",
|
|
677
743
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
744
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
678
745
|
}, "not-in-the-catalog");
|
|
679
746
|
const uncatalogedResponse = await uncataloged!.generate({ workerId: "uncataloged", messages: [] });
|
|
680
747
|
assert.deepEqual(uncatalogedResponse.accounting[0]?.cost, {
|
package/src/catalogProvider.ts
CHANGED
|
@@ -1,4 +1,9 @@
|
|
|
1
|
-
import {
|
|
1
|
+
import {
|
|
2
|
+
lookupProvider,
|
|
3
|
+
resolveModel,
|
|
4
|
+
type ModelInfo,
|
|
5
|
+
type ModelReasoningEffort,
|
|
6
|
+
} from "@plurnk/plurnk-models";
|
|
2
7
|
import {
|
|
3
8
|
contextWindowFromEnv,
|
|
4
9
|
effectiveContextWindow,
|
|
@@ -12,7 +17,13 @@ import {
|
|
|
12
17
|
reasoningFromEnv,
|
|
13
18
|
reasoningResponseStyleFromEnv,
|
|
14
19
|
} from "./env.ts";
|
|
15
|
-
import AiSdkProvider, {
|
|
20
|
+
import AiSdkProvider, {
|
|
21
|
+
type AiSdkProviderConfig,
|
|
22
|
+
type CompatibleReasoningEffort,
|
|
23
|
+
type GrammarStyle,
|
|
24
|
+
type NativeReasoningEffort,
|
|
25
|
+
type ReasoningStyle,
|
|
26
|
+
} from "./AiSdkProvider.ts";
|
|
16
27
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
17
28
|
import { providerSource } from "./notices.ts";
|
|
18
29
|
import type { Provider, ProviderCostNormalizer } from "./types.ts";
|
|
@@ -40,23 +51,92 @@ const reasoningStyleFromEnv = (
|
|
|
40
51
|
return value as ReasoningStyle;
|
|
41
52
|
};
|
|
42
53
|
|
|
54
|
+
// {§provider-reasoning-policy} — the operator's affirmative declaration of efforts a provider's
|
|
55
|
+
// reasoning routes accept beyond Models.dev; the daemon adds none on its own (#439).
|
|
56
|
+
const DECLARABLE_EFFORTS: ReadonlySet<ModelReasoningEffort> = new Set([
|
|
57
|
+
"none", "minimal", "low", "medium", "high", "xhigh", "max",
|
|
58
|
+
]);
|
|
59
|
+
const declaredEffortsFromEnv = (
|
|
60
|
+
env: NodeJS.ProcessEnv,
|
|
61
|
+
name: string,
|
|
62
|
+
): readonly ModelReasoningEffort[] => {
|
|
63
|
+
const prefix = name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase();
|
|
64
|
+
const key = `PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_EFFORTS`;
|
|
65
|
+
const value = env[key];
|
|
66
|
+
if (value === undefined || value.trim().length === 0) return [];
|
|
67
|
+
const efforts = value.split(",").map((part) => part.trim()).filter((part) => part.length > 0);
|
|
68
|
+
const invalid = efforts.find((effort) => !DECLARABLE_EFFORTS.has(effort as ModelReasoningEffort));
|
|
69
|
+
if (invalid !== undefined) {
|
|
70
|
+
throw new Error(`${name} provider: ${key} has invalid value "${invalid}"; declarable efforts: ${[...DECLARABLE_EFFORTS].join(", ")}`);
|
|
71
|
+
}
|
|
72
|
+
return [...new Set(efforts as ModelReasoningEffort[])];
|
|
73
|
+
};
|
|
43
74
|
const activationPolicies = Object.freeze(["off", "adaptive"] as const);
|
|
44
75
|
const deepSeekPolicies = Object.freeze(["off", "adaptive", "high"] as const);
|
|
45
|
-
const adaptiveOnly = Object.freeze(["adaptive"] as const);
|
|
46
76
|
const reasoningWithoutOff = Object.freeze(["adaptive", "low", "medium", "high"] as const);
|
|
47
77
|
|
|
48
|
-
const
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
78
|
+
const reasoningEffortOrder = Object.freeze([
|
|
79
|
+
"minimal", "low", "medium", "high", "xhigh", "max",
|
|
80
|
+
] as const);
|
|
81
|
+
const nativeReasoningEfforts = new Set<NativeReasoningEffort>([
|
|
82
|
+
"minimal", "low", "medium", "high", "xhigh",
|
|
83
|
+
]);
|
|
84
|
+
const compatibleReasoningEfforts = new Set<CompatibleReasoningEffort>(reasoningEffortOrder);
|
|
85
|
+
const compatibleEffortStyles = new Set<ReasoningStyle>([
|
|
86
|
+
"effort", "effort_explicit", "effort_required", "thinking_effort",
|
|
87
|
+
]);
|
|
88
|
+
const compatibleToggleStyles = new Set<ReasoningStyle>([
|
|
89
|
+
"think", "include_reasoning", "effort_explicit", "thinking_effort", "template", "anthropic",
|
|
90
|
+
]);
|
|
91
|
+
|
|
92
|
+
const catalogEfforts = (
|
|
93
|
+
info: ModelInfo,
|
|
94
|
+
declared: readonly ModelReasoningEffort[] = [],
|
|
95
|
+
): readonly ModelReasoningEffort[] => [
|
|
96
|
+
...new Set([
|
|
97
|
+
...(info.reasoningOptions
|
|
98
|
+
?.filter((option) => option.type === "effort")
|
|
99
|
+
.flatMap((option) => option.values) ?? []),
|
|
100
|
+
...declared,
|
|
101
|
+
]),
|
|
102
|
+
];
|
|
53
103
|
|
|
54
|
-
const
|
|
55
|
-
|
|
104
|
+
const strongestCatalogEffort = <T extends NativeReasoningEffort | CompatibleReasoningEffort>(
|
|
105
|
+
info: ModelInfo,
|
|
106
|
+
supported: ReadonlySet<T>,
|
|
107
|
+
declared: readonly ModelReasoningEffort[] = [],
|
|
108
|
+
): T | undefined => {
|
|
109
|
+
const values = new Set(catalogEfforts(info, declared));
|
|
110
|
+
return reasoningEffortOrder.findLast((effort) => supported.has(effort as T) && values.has(effort)) as T | undefined;
|
|
111
|
+
};
|
|
112
|
+
|
|
113
|
+
const catalogSupportsToggle = (info: ModelInfo): boolean =>
|
|
114
|
+
info.reasoningOptions?.some((option) => option.type === "toggle") === true;
|
|
56
115
|
|
|
57
|
-
const
|
|
58
|
-
|
|
59
|
-
|
|
116
|
+
const catalogSupportedReasoningPolicies = ({
|
|
117
|
+
info,
|
|
118
|
+
native,
|
|
119
|
+
style,
|
|
120
|
+
declared,
|
|
121
|
+
}: {
|
|
122
|
+
info: ModelInfo;
|
|
123
|
+
native: boolean;
|
|
124
|
+
style: ReasoningStyle;
|
|
125
|
+
declared: readonly ModelReasoningEffort[];
|
|
126
|
+
}): readonly ReasoningPolicy[] => {
|
|
127
|
+
if (!info.reasoning) return activationPolicies;
|
|
128
|
+
const efforts = new Set(catalogEfforts(info, declared));
|
|
129
|
+
const effortTransport = native || compatibleEffortStyles.has(style);
|
|
130
|
+
const off = native
|
|
131
|
+
? efforts.has("none") || catalogSupportsToggle(info)
|
|
132
|
+
: (efforts.has("none") && compatibleEffortStyles.has(style))
|
|
133
|
+
|| (catalogSupportsToggle(info) && compatibleToggleStyles.has(style));
|
|
134
|
+
return REASONING_POLICIES.filter((policy) => policy === "adaptive"
|
|
135
|
+
|| policy === "off" && off
|
|
136
|
+
|| (policy === "low" || policy === "medium" || policy === "high")
|
|
137
|
+
&& effortTransport
|
|
138
|
+
&& efforts.has(policy));
|
|
139
|
+
};
|
|
60
140
|
|
|
61
141
|
const anthropicSupportsAdaptiveThinking = (model: string): boolean =>
|
|
62
142
|
/claude-(?:opus-(?:4-[678]|5)|sonnet-(?:4-6|5)|fable-5)/.test(model);
|
|
@@ -65,29 +145,18 @@ const supportedReasoningPolicies = ({
|
|
|
65
145
|
info,
|
|
66
146
|
native,
|
|
67
147
|
style,
|
|
68
|
-
|
|
69
|
-
model,
|
|
148
|
+
declared,
|
|
70
149
|
}: {
|
|
71
150
|
info?: ModelInfo;
|
|
72
151
|
native: boolean;
|
|
73
152
|
style: ReasoningStyle;
|
|
74
|
-
|
|
75
|
-
model: string;
|
|
153
|
+
declared: readonly ModelReasoningEffort[];
|
|
76
154
|
}): readonly ReasoningPolicy[] => {
|
|
77
155
|
if (info !== undefined && info.reasoning !== true) return activationPolicies;
|
|
78
|
-
if (
|
|
79
|
-
|
|
80
|
-
return mistralSupportsEffort(model) ? deepSeekPolicies : adaptiveOnly;
|
|
81
|
-
}
|
|
82
|
-
if (sdkPackage === "@ai-sdk/xai") {
|
|
83
|
-
if (xaiReasoningIsModelFixed(model)) return adaptiveOnly;
|
|
84
|
-
if (model === "grok-4.6") return reasoningWithoutOff;
|
|
85
|
-
}
|
|
86
|
-
if (sdkPackage === "@ai-sdk/google" && googleReasoningCannotBeOff(model)) {
|
|
87
|
-
return reasoningWithoutOff;
|
|
88
|
-
}
|
|
89
|
-
return REASONING_POLICIES;
|
|
156
|
+
if (info?.reasoningOptions !== undefined) {
|
|
157
|
+
return catalogSupportedReasoningPolicies({ info, native, style, declared });
|
|
90
158
|
}
|
|
159
|
+
if (native) return activationPolicies;
|
|
91
160
|
if (style === "effort" || style === "effort_explicit") return REASONING_POLICIES;
|
|
92
161
|
if (style === "effort_required") return reasoningWithoutOff;
|
|
93
162
|
if (style === "thinking_effort") return deepSeekPolicies;
|
|
@@ -98,10 +167,14 @@ const adaptiveReasoningProjection = ({
|
|
|
98
167
|
sdkPackage,
|
|
99
168
|
model,
|
|
100
169
|
reasoningCapable,
|
|
170
|
+
info,
|
|
171
|
+
declared,
|
|
101
172
|
}: {
|
|
102
173
|
sdkPackage?: string;
|
|
103
174
|
model: string;
|
|
104
175
|
reasoningCapable: boolean;
|
|
176
|
+
info?: ModelInfo;
|
|
177
|
+
declared: readonly ModelReasoningEffort[];
|
|
105
178
|
}): Pick<AiSdkProviderConfig, "adaptiveReasoning" | "adaptiveReasoningProviderOptions"> => {
|
|
106
179
|
if (!reasoningCapable) return { adaptiveReasoning: "provider-default" };
|
|
107
180
|
if (sdkPackage === "@ai-sdk/google" && /^gemini-2\.5(?:-|$)/i.test(model)) {
|
|
@@ -128,13 +201,12 @@ const adaptiveReasoningProjection = ({
|
|
|
128
201
|
},
|
|
129
202
|
};
|
|
130
203
|
}
|
|
131
|
-
if (
|
|
132
|
-
return {
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
return { adaptiveReasoning: "provider-default" };
|
|
204
|
+
if (info?.reasoningOptions !== undefined) {
|
|
205
|
+
return {
|
|
206
|
+
adaptiveReasoning: strongestCatalogEffort(info, nativeReasoningEfforts, declared) ?? "provider-default",
|
|
207
|
+
};
|
|
136
208
|
}
|
|
137
|
-
return { adaptiveReasoning: "
|
|
209
|
+
return { adaptiveReasoning: "provider-default" };
|
|
138
210
|
};
|
|
139
211
|
|
|
140
212
|
export const providerFromSdkModel = ({
|
|
@@ -191,14 +263,30 @@ export const providerFromSdkModel = ({
|
|
|
191
263
|
);
|
|
192
264
|
const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
|
|
193
265
|
const reasoningCapable = info?.reasoning === true;
|
|
194
|
-
const
|
|
266
|
+
const declaredEfforts = declaredEffortsFromEnv(env, name);
|
|
267
|
+
const declaredReasoningStyle = info?.reasoning === false
|
|
195
268
|
? "none"
|
|
196
269
|
: reasoningStyleFromEnv(env, name) ?? "none";
|
|
270
|
+
const reasoningStyle = info?.reasoningOptions !== undefined
|
|
271
|
+
&& (declaredReasoningStyle === "effort" || declaredReasoningStyle === "effort_required")
|
|
272
|
+
&& catalogEfforts(info, declaredEfforts).length === 0
|
|
273
|
+
? "none"
|
|
274
|
+
: declaredReasoningStyle;
|
|
197
275
|
const adaptiveReasoning = adaptiveReasoningProjection({
|
|
198
276
|
sdkPackage,
|
|
199
277
|
model,
|
|
200
278
|
reasoningCapable,
|
|
279
|
+
info,
|
|
280
|
+
declared: declaredEfforts,
|
|
201
281
|
});
|
|
282
|
+
const compatibleAdaptiveReasoning = languageModel !== undefined || info?.reasoningOptions === undefined
|
|
283
|
+
? undefined
|
|
284
|
+
: strongestCatalogEffort(info, compatibleReasoningEfforts, declaredEfforts) ?? "provider-default";
|
|
285
|
+
const compatibleOffReasoning = languageModel !== undefined
|
|
286
|
+
|| info === undefined
|
|
287
|
+
|| !catalogEfforts(info, declaredEfforts).includes("none")
|
|
288
|
+
? undefined
|
|
289
|
+
: "none" as const;
|
|
202
290
|
|
|
203
291
|
const catalogCost = info?.cost;
|
|
204
292
|
const rates = catalogCost === undefined ? null : {
|
|
@@ -235,10 +323,11 @@ export const providerFromSdkModel = ({
|
|
|
235
323
|
info,
|
|
236
324
|
native: languageModel !== undefined,
|
|
237
325
|
style: reasoningStyle,
|
|
238
|
-
|
|
239
|
-
model,
|
|
326
|
+
declared: declaredEfforts,
|
|
240
327
|
}),
|
|
241
328
|
...adaptiveReasoning,
|
|
329
|
+
...(compatibleAdaptiveReasoning === undefined ? {} : { compatibleAdaptiveReasoning }),
|
|
330
|
+
...(compatibleOffReasoning === undefined ? {} : { compatibleOffReasoning }),
|
|
242
331
|
...(additiveReasoningProvider === undefined || !reasoningCapable
|
|
243
332
|
? {}
|
|
244
333
|
: { additiveReasoningProvider }),
|
package/src/index.ts
CHANGED
|
@@ -45,7 +45,8 @@ export {
|
|
|
45
45
|
loadActiveProvider,
|
|
46
46
|
resetDiscoveryCache,
|
|
47
47
|
} from "./ProviderRegistry.ts";
|
|
48
|
-
export { providerReadiness } from "./sdkModels.ts";
|
|
48
|
+
export { createEmbeddingModel, providerReadiness } from "./sdkModels.ts";
|
|
49
|
+
export type { EmbeddingModelResolution } from "./sdkModels.ts";
|
|
49
50
|
|
|
50
51
|
// Scope-agnostic plugin discovery ({§plugin-family-kind}).
|
|
51
52
|
export { discover } from "./discover.ts";
|