@plurnk/plurnk-providers 1.11.0 → 1.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@plurnk/plurnk-providers",
3
- "version": "1.11.0",
3
+ "version": "1.13.0",
4
4
  "description": "PLURNK's stable model-provider contract and AI SDK adapter.",
5
5
  "keywords": [
6
6
  "plurnk",
@@ -71,6 +71,7 @@
71
71
  "prepack": "npm run build"
72
72
  },
73
73
  "dependencies": {
74
+ "@ai-sdk/provider": "^4.0.8",
74
75
  "@ai-sdk/amazon-bedrock": "^5.0.0",
75
76
  "@ai-sdk/anthropic": "^4.0.0",
76
77
  "@ai-sdk/cerebras": "^3.0.0",
@@ -83,11 +84,11 @@
83
84
  "@ai-sdk/togetherai": "^3.0.0",
84
85
  "@ai-sdk/xai": "^4.0.0",
85
86
  "@openrouter/ai-sdk-provider": "^3.0.0",
86
- "@plurnk/gbnf": "1.11.0",
87
- "@plurnk/plurnk-aliases": "1.11.0",
88
- "@plurnk/plurnk-contracts": "1.11.0",
89
- "@plurnk/plurnk-meta": "1.11.0",
90
- "@plurnk/plurnk-models": "1.11.0",
87
+ "@plurnk/gbnf": "1.13.0",
88
+ "@plurnk/plurnk-aliases": "1.13.0",
89
+ "@plurnk/plurnk-contracts": "1.13.0",
90
+ "@plurnk/plurnk-meta": "1.13.0",
91
+ "@plurnk/plurnk-models": "1.13.0",
91
92
  "ai": "^7.0.37",
92
93
  "zod": "^4.4.3"
93
94
  },
@@ -52,6 +52,9 @@ export type ProviderFetch = typeof globalThis.fetch;
52
52
  // mapping retains any backend-specific omission/explicit-disable constraint.
53
53
  export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "effort_required" | "thinking_effort" | "template" | "anthropic";
54
54
 
55
+ export type NativeReasoningEffort = "minimal" | "low" | "medium" | "high" | "xhigh";
56
+ export type CompatibleReasoningEffort = NativeReasoningEffort | "max";
57
+
55
58
  // GBNF transport is a local llama-server capability. "none" means no
56
59
  // service-managed constrained sampling; endpoint-owned settings are not inferred.
57
60
  export type GrammarStyle = "none" | "llamacpp";
@@ -98,9 +101,15 @@ export type AiSdkProviderConfig = {
98
101
  outputBudget?: number | null;
99
102
  reasoningBudget?: number | null;
100
103
  supportedReasoningPolicies?: readonly ReasoningPolicy[];
101
- // Native AI SDK projection for adaptive. `high` is the graded fallback;
102
- // provider-default is reserved for a documented native option/default.
103
- adaptiveReasoning?: "high" | "provider-default";
104
+ // Native AI SDK projection for adaptive. Models.dev supplies the route's
105
+ // admissible values; provider-default is reserved for a documented native
106
+ // dynamic option/default or a route with no caller-selectable effort.
107
+ adaptiveReasoning?: NativeReasoningEffort | "provider-default";
108
+ // OpenAI-compatible effort transports accept route-native values beyond
109
+ // the AI SDK's generic vocabulary. These are still catalog facts, not
110
+ // provider-name branches.
111
+ compatibleAdaptiveReasoning?: CompatibleReasoningEffort | "provider-default";
112
+ compatibleOffReasoning?: "none";
104
113
  adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
105
114
  // Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
106
115
  // visible output and add an explicit provider reasoning budget. This marker lets the
@@ -369,7 +378,9 @@ export default class AiSdkProvider implements Provider {
369
378
  #additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
370
379
  #reasoning: Reasoning;
371
380
  #supportedReasoningPolicies: readonly ReasoningPolicy[];
372
- #adaptiveReasoning: "high" | "provider-default";
381
+ #adaptiveReasoning: NativeReasoningEffort | "provider-default";
382
+ #compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
383
+ #compatibleOffReasoning: "none" | undefined;
373
384
  #adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
374
385
  #temperature: number;
375
386
  #repeatPenalty: number;
@@ -444,6 +455,8 @@ export default class AiSdkProvider implements Provider {
444
455
  ...new Set(config.supportedReasoningPolicies ?? REASONING_POLICIES),
445
456
  ]);
446
457
  this.#adaptiveReasoning = config.adaptiveReasoning ?? "high";
458
+ this.#compatibleAdaptiveReasoning = config.compatibleAdaptiveReasoning ?? "high";
459
+ this.#compatibleOffReasoning = config.compatibleOffReasoning;
447
460
  this.#adaptiveReasoningProviderOptions = config.adaptiveReasoningProviderOptions;
448
461
  if (!this.#supportedReasoningPolicies.includes(this.#reasoning.mode)) {
449
462
  throw new UnsupportedReasoningPolicyError(
@@ -700,13 +713,31 @@ export default class AiSdkProvider implements Provider {
700
713
  case "think": return on ? { think: true } : {};
701
714
  case "include_reasoning": return on ? { include_reasoning: true } : {};
702
715
  case "effort": return mode === "off"
703
- ? {}
704
- : { reasoning_effort: mode === "adaptive" ? "high" : fixedEffort(mode) };
705
- // Graded reasoning is mandatory: adaptive uses PLURNK's high fallback,
706
- // fixed levels preserve their intent, and construction rejects off.
707
- case "effort_required": return {
708
- reasoning_effort: mode === "adaptive" ? "high" : fixedEffort(mode),
709
- };
716
+ ? this.#compatibleOffReasoning === undefined
717
+ ? {}
718
+ : { reasoning_effort: this.#compatibleOffReasoning }
719
+ : mode === "adaptive"
720
+ ? this.#compatibleAdaptiveReasoning === "provider-default"
721
+ ? {}
722
+ : { reasoning_effort: this.#compatibleAdaptiveReasoning }
723
+ : { reasoning_effort: fixedEffort(mode) };
724
+ // Graded reasoning is mandatory when the route advertises an effort
725
+ // value. Cataloged routes supply the exact strongest legal value;
726
+ // construction rejects an unsupported off or fixed policy.
727
+ case "effort_required": {
728
+ if (mode === "off") {
729
+ if (this.#compatibleOffReasoning === undefined) {
730
+ throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
731
+ }
732
+ return { reasoning_effort: this.#compatibleOffReasoning };
733
+ }
734
+ if (mode === "adaptive") {
735
+ return this.#compatibleAdaptiveReasoning === "provider-default"
736
+ ? {}
737
+ : { reasoning_effort: this.#compatibleAdaptiveReasoning };
738
+ }
739
+ return { reasoning_effort: fixedEffort(mode) };
740
+ }
710
741
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
711
742
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
712
743
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -120,3 +120,42 @@ test("aggregateProviderAccounting omits unknown nested usage fields from its JSO
120
120
  outputTokenDetails: { reasoningTokens: 2 },
121
121
  });
122
122
  });
123
+
124
+ test("aggregateProviderAccounting does not invent complete detail partitions across heterogeneous requests", () => {
125
+ const accounting = aggregateProviderAccounting([
126
+ {
127
+ provider: "provider:detailed",
128
+ model: "m",
129
+ outcome: "response",
130
+ usage: {
131
+ inputTokens: 2,
132
+ outputTokens: 0,
133
+ totalTokens: 2,
134
+ inputTokenDetails: {
135
+ noCacheTokens: 2,
136
+ cacheReadTokens: 0,
137
+ cacheWriteTokens: 0,
138
+ },
139
+ },
140
+ cost: { kind: "unknown", reason: "no direct cost" },
141
+ },
142
+ {
143
+ provider: "provider:totals-only",
144
+ model: "m",
145
+ outcome: "response",
146
+ usage: { inputTokens: 4, outputTokens: 0, totalTokens: 4 },
147
+ cost: { kind: "unknown", reason: "no direct cost" },
148
+ },
149
+ ]);
150
+
151
+ assert.deepEqual(accounting.usage, {
152
+ inputTokens: 6,
153
+ outputTokens: 0,
154
+ totalTokens: 6,
155
+ inputTokenDetails: {
156
+ noCacheTokens: 2,
157
+ cacheReadTokens: 0,
158
+ cacheWriteTokens: 0,
159
+ },
160
+ });
161
+ });
package/src/accounting.ts CHANGED
@@ -132,7 +132,12 @@ const sumKnown = (
132
132
  const known = requests
133
133
  .map((request) => request.usage === undefined ? undefined : read(request.usage))
134
134
  .filter((value): value is number => value !== undefined);
135
- return known.length === 0 ? undefined : known.reduce((sum, value) => sum + value, 0);
135
+ if (known.length === 0) return undefined;
136
+ const sum = known.reduce((total, value) => total + value, 0);
137
+ if (!Number.isSafeInteger(sum)) {
138
+ throw new TypeError("aggregate provider usage exceeds the safe-integer range");
139
+ }
140
+ return sum;
136
141
  };
137
142
 
138
143
  export const aggregateProviderAccounting = (
@@ -179,16 +184,20 @@ export const aggregateProviderAccounting = (
179
184
  ...(textTokens === undefined ? {} : { textTokens }),
180
185
  ...(reasoningTokens === undefined ? {} : { reasoningTokens }),
181
186
  };
182
- const usage = inputTokens === undefined && outputTokens === undefined && totalTokens === undefined
187
+ // Each physical request was validated above. Aggregate fields deliberately
188
+ // sum their own known evidence independently: heterogeneous providers may
189
+ // report different detail subsets, so the projection must not reinterpret
190
+ // their union as one complete per-request partition.
191
+ const usage: ProviderUsage | null = inputTokens === undefined && outputTokens === undefined && totalTokens === undefined
183
192
  && inputTokenDetails === undefined && outputTokenDetails === undefined
184
193
  ? null
185
- : validateProviderUsage({
194
+ : {
186
195
  ...(inputTokens === undefined ? {} : { inputTokens }),
187
196
  ...(outputTokens === undefined ? {} : { outputTokens }),
188
197
  ...(totalTokens === undefined ? {} : { totalTokens }),
189
198
  ...(inputTokenDetails === undefined ? {} : { inputTokenDetails }),
190
199
  ...(outputTokenDetails === undefined ? {} : { outputTokenDetails }),
191
- });
200
+ };
192
201
  return {
193
202
  requests: [...requests],
194
203
  usage,
@@ -48,7 +48,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
48
48
  PLURNK_PROVIDERS_REASONING: "adaptive",
49
49
  PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
50
50
  }, "deepseek-v4-flash");
51
- assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "high"]);
51
+ assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high"]);
52
52
 
53
53
  assert.throws(
54
54
  () => catalogProviderFromEnv("deepseek", {
@@ -57,7 +57,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
57
57
  PLURNK_PROVIDERS_REASONING: "medium",
58
58
  PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
59
59
  }, "deepseek-v4-flash"),
60
- /reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
60
+ /reasoning policy 'medium' is unsupported; supported policies: off, adaptive, low, high/,
61
61
  );
62
62
 
63
63
  const mistral = catalogProviderFromEnv("mistral", {
@@ -82,7 +82,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
82
82
  assert.deepEqual(gemini?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Gemini 3's mandatory minimum is not advertised as off");
83
83
  });
84
84
 
85
- test("Cloudflare graded reasoning sends exact effort and never advertises unsupported off", async () => {
85
+ test("Models.dev controls Cloudflare's exact effort vocabulary", async () => {
86
86
  const bodies: Record<string, unknown>[] = [];
87
87
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
88
88
  bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
@@ -91,7 +91,7 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
91
91
  id: "cloudflare-effort",
92
92
  object: "chat.completion.chunk",
93
93
  created: 1,
94
- model: "@cf/zai-org/glm-5.3-flash",
94
+ model: "@cf/qwen/qwen3.8-27b",
95
95
  choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
96
96
  })}`,
97
97
  "data: [DONE]",
@@ -106,14 +106,14 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
106
106
  const low = catalogProviderFromEnv("cloudflare", {
107
107
  ...cloudflareEnv,
108
108
  PLURNK_PROVIDERS_REASONING: "low",
109
- }, "@cf/zai-org/glm-5.3-flash");
110
- assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"]);
109
+ }, "@cf/qwen/qwen3.8-27b");
110
+ assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium"]);
111
111
  await low?.generate({ workerId: "cloudflare-low", messages: [{ role: "user", content: "hello" }] });
112
112
 
113
113
  const adaptive = catalogProviderFromEnv("cloudflare", {
114
114
  ...cloudflareEnv,
115
115
  PLURNK_PROVIDERS_REASONING: "adaptive",
116
- }, "@cf/zai-org/glm-5.3-flash");
116
+ }, "@cf/qwen/qwen3.8-27b");
117
117
  await adaptive?.generate({ workerId: "cloudflare-adaptive", messages: [{ role: "user", content: "hello" }] });
118
118
 
119
119
  const nonReasoning = catalogProviderFromEnv("cloudflare", {
@@ -123,10 +123,71 @@ test("Cloudflare graded reasoning sends exact effort and never advertises unsupp
123
123
  assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
124
124
  await nonReasoning?.generate({ workerId: "cloudflare-granite", messages: [{ role: "user", content: "hello" }] });
125
125
 
126
- assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["low", "high", undefined]);
126
+ const ungradedReasoner = catalogProviderFromEnv("cloudflare", {
127
+ ...cloudflareEnv,
128
+ PLURNK_PROVIDERS_REASONING: "adaptive",
129
+ }, "@cf/zai-org/glm-5.3-flash");
130
+ assert.deepEqual(ungradedReasoner?.supportedReasoningPolicies, ["adaptive"]);
131
+ await ungradedReasoner?.generate({ workerId: "cloudflare-ungraded", messages: [{ role: "user", content: "hello" }] });
132
+
133
+ assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["low", "xhigh", undefined, undefined]);
134
+ assert.throws(
135
+ () => catalogProviderFromEnv("cloudflare", cloudflareEnv, "@cf/qwen/qwen3.8-27b"),
136
+ /reasoning policy 'off' is unsupported; supported policies: adaptive, low, medium/,
137
+ );
127
138
  assert.throws(
128
- () => catalogProviderFromEnv("cloudflare", cloudflareEnv, "@cf/zai-org/glm-5.3-flash"),
129
- /reasoning policy 'off' is unsupported; supported policies: adaptive, low, medium, high/,
139
+ () => catalogProviderFromEnv("cloudflare", {
140
+ ...cloudflareEnv,
141
+ PLURNK_PROVIDERS_REASONING: "high",
142
+ }, "@cf/qwen/qwen3.8-27b"),
143
+ /reasoning policy 'high' is unsupported; supported policies: adaptive, low, medium/,
144
+ );
145
+ });
146
+
147
+ test("an operator-declared effort vocabulary extends Models.dev's for a provider's reasoning routes (#439)", async () => {
148
+ const bodies: Record<string, unknown>[] = [];
149
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
150
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
151
+ return new Response([
152
+ `data: ${JSON.stringify({
153
+ id: "cloudflare-declared",
154
+ object: "chat.completion.chunk",
155
+ created: 1,
156
+ model: "@cf/zai-org/glm-5.3-flash",
157
+ choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
158
+ })}`,
159
+ "data: [DONE]",
160
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
161
+ });
162
+ const declaredEnv = {
163
+ ...env,
164
+ CLOUDFLARE_ACCOUNT_ID: "account",
165
+ CLOUDFLARE_API_KEY: "token",
166
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_STYLE: "effort_required",
167
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "low, medium,high",
168
+ };
169
+ // Models.dev lists this route as reasoning with no effort vocabulary; the declaration supplies one.
170
+ const low = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "low" }, "@cf/zai-org/glm-5.3-flash");
171
+ assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"]);
172
+ await low?.generate({ workerId: "declared-low", messages: [{ role: "user", content: "hello" }] });
173
+ const adaptive = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/zai-org/glm-5.3-flash");
174
+ await adaptive?.generate({ workerId: "declared-adaptive", messages: [{ role: "user", content: "hello" }] });
175
+ assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["low", "high"], "a fixed level keeps its name; adaptive takes the strongest declared effort");
176
+ // A declared `none` admits `off` on an effort transport; the catalog's own vocabulary stays in the union.
177
+ const withOff = catalogProviderFromEnv("cloudflare", {
178
+ ...declaredEnv,
179
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "none,high",
180
+ }, "@cf/qwen/qwen3.8-27b");
181
+ assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high"]);
182
+ // The declaration never turns a non-reasoning route into a reasoning one.
183
+ const nonReasoning = catalogProviderFromEnv("cloudflare", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/ibm-granite/granite-4.0-h-micro");
184
+ assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
185
+ assert.throws(
186
+ () => catalogProviderFromEnv("cloudflare", {
187
+ ...declaredEnv,
188
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS: "low,turbo",
189
+ }, "@cf/zai-org/glm-5.3-flash"),
190
+ /PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_REASONING_EFFORTS has invalid value "turbo"; declarable efforts: none, minimal, low, medium, high, xhigh, max/,
130
191
  );
131
192
  });
132
193
 
@@ -201,7 +262,7 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
201
262
  assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
202
263
  });
203
264
 
204
- test("xAI's native chat contract affirmatively requests adaptive high and caps the complete reasoning response", async () => {
265
+ test("xAI's native chat contract requests its strongest cataloged effort and caps the complete reasoning response", async () => {
205
266
  let call: { headers: Headers; body: Record<string, unknown> } | undefined;
206
267
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
207
268
  call = {
@@ -254,7 +315,7 @@ test("xAI's native chat contract affirmatively requests adaptive high and caps t
254
315
  });
255
316
 
256
317
  assert.equal(call?.body.max_completion_tokens, 16);
257
- assert.equal(call?.body.reasoning_effort, "high", "adaptive is affirmative on xAI's graded route");
318
+ assert.equal(call?.body.reasoning_effort, "xhigh", "adaptive uses the strongest route-advertised generic effort");
258
319
  assert.equal("max_tokens" in (call?.body ?? {}), false);
259
320
  assert.equal(call?.headers.get("x-grok-conv-id"), "xai-worker");
260
321
  assert.equal(result?.assistant.reasoning, "consider");
@@ -365,7 +426,7 @@ test("Meta Muse adaptive reasoning is requested even when the endpoint returns n
365
426
  messages: [{ role: "user", content: "hello" }],
366
427
  });
367
428
 
368
- assert.equal(body?.reasoning_effort, "high");
429
+ assert.equal(body?.reasoning_effort, "xhigh");
369
430
  assert.equal(result?.assistant.reasoning, null);
370
431
  assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 37);
371
432
  });
@@ -504,6 +565,7 @@ test("native Anthropic adaptive policy uses adaptive thinking rather than a fixe
504
565
  contextWindow: 16_384,
505
566
  maxOutputTokens: 8_192,
506
567
  reasoning: true,
568
+ reasoningOptions: [{ type: "toggle" }],
507
569
  attachment: true,
508
570
  toolCall: true,
509
571
  modalities: { input: ["text", "image"], output: ["text"] },
@@ -635,8 +697,10 @@ test("cataloged unknown model fails unless its context is explicit", () => {
635
697
  ...env,
636
698
  XAI_API_KEY: "test-key",
637
699
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
700
+ PLURNK_PROVIDERS_REASONING: "adaptive",
638
701
  }, "not-in-the-catalog");
639
702
  assert.equal(provider?.contextWindow, 8192);
703
+ assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
640
704
  });
641
705
 
642
706
  test("Models.dev is the only fallback rate table", async () => {
@@ -662,6 +726,8 @@ test("Models.dev is the only fallback rate table", async () => {
662
726
  const cataloged = catalogProviderFromEnv("deepseek", {
663
727
  ...env,
664
728
  DEEPSEEK_API_KEY: "test-key",
729
+ PLURNK_PROVIDERS_REASONING: "adaptive",
730
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
665
731
  }, "deepseek-v4-flash");
666
732
  assert.notEqual(cataloged, null);
667
733
  const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
@@ -675,6 +741,7 @@ test("Models.dev is the only fallback rate table", async () => {
675
741
  ...env,
676
742
  XAI_API_KEY: "test-key",
677
743
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
744
+ PLURNK_PROVIDERS_REASONING: "adaptive",
678
745
  }, "not-in-the-catalog");
679
746
  const uncatalogedResponse = await uncataloged!.generate({ workerId: "uncataloged", messages: [] });
680
747
  assert.deepEqual(uncatalogedResponse.accounting[0]?.cost, {
@@ -1,4 +1,9 @@
1
- import { lookupProvider, resolveModel, type ModelInfo } from "@plurnk/plurnk-models";
1
+ import {
2
+ lookupProvider,
3
+ resolveModel,
4
+ type ModelInfo,
5
+ type ModelReasoningEffort,
6
+ } from "@plurnk/plurnk-models";
2
7
  import {
3
8
  contextWindowFromEnv,
4
9
  effectiveContextWindow,
@@ -12,7 +17,13 @@ import {
12
17
  reasoningFromEnv,
13
18
  reasoningResponseStyleFromEnv,
14
19
  } from "./env.ts";
15
- import AiSdkProvider, { type AiSdkProviderConfig, type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
20
+ import AiSdkProvider, {
21
+ type AiSdkProviderConfig,
22
+ type CompatibleReasoningEffort,
23
+ type GrammarStyle,
24
+ type NativeReasoningEffort,
25
+ type ReasoningStyle,
26
+ } from "./AiSdkProvider.ts";
16
27
  import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
17
28
  import { providerSource } from "./notices.ts";
18
29
  import type { Provider, ProviderCostNormalizer } from "./types.ts";
@@ -40,23 +51,92 @@ const reasoningStyleFromEnv = (
40
51
  return value as ReasoningStyle;
41
52
  };
42
53
 
54
+ // {§provider-reasoning-policy} — the operator's affirmative declaration of efforts a provider's
55
+ // reasoning routes accept beyond Models.dev; the daemon adds none on its own (#439).
56
+ const DECLARABLE_EFFORTS: ReadonlySet<ModelReasoningEffort> = new Set([
57
+ "none", "minimal", "low", "medium", "high", "xhigh", "max",
58
+ ]);
59
+ const declaredEffortsFromEnv = (
60
+ env: NodeJS.ProcessEnv,
61
+ name: string,
62
+ ): readonly ModelReasoningEffort[] => {
63
+ const prefix = name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase();
64
+ const key = `PLURNK_PROVIDERS_PROVIDER_${prefix}_REASONING_EFFORTS`;
65
+ const value = env[key];
66
+ if (value === undefined || value.trim().length === 0) return [];
67
+ const efforts = value.split(",").map((part) => part.trim()).filter((part) => part.length > 0);
68
+ const invalid = efforts.find((effort) => !DECLARABLE_EFFORTS.has(effort as ModelReasoningEffort));
69
+ if (invalid !== undefined) {
70
+ throw new Error(`${name} provider: ${key} has invalid value "${invalid}"; declarable efforts: ${[...DECLARABLE_EFFORTS].join(", ")}`);
71
+ }
72
+ return [...new Set(efforts as ModelReasoningEffort[])];
73
+ };
43
74
  const activationPolicies = Object.freeze(["off", "adaptive"] as const);
44
75
  const deepSeekPolicies = Object.freeze(["off", "adaptive", "high"] as const);
45
- const adaptiveOnly = Object.freeze(["adaptive"] as const);
46
76
  const reasoningWithoutOff = Object.freeze(["adaptive", "low", "medium", "high"] as const);
47
77
 
48
- const mistralSupportsEffort = (model: string): boolean =>
49
- model === "mistral-small-latest"
50
- || model === "mistral-small-2603"
51
- || model === "mistral-medium-3"
52
- || model === "mistral-medium-3.5";
78
+ const reasoningEffortOrder = Object.freeze([
79
+ "minimal", "low", "medium", "high", "xhigh", "max",
80
+ ] as const);
81
+ const nativeReasoningEfforts = new Set<NativeReasoningEffort>([
82
+ "minimal", "low", "medium", "high", "xhigh",
83
+ ]);
84
+ const compatibleReasoningEfforts = new Set<CompatibleReasoningEffort>(reasoningEffortOrder);
85
+ const compatibleEffortStyles = new Set<ReasoningStyle>([
86
+ "effort", "effort_explicit", "effort_required", "thinking_effort",
87
+ ]);
88
+ const compatibleToggleStyles = new Set<ReasoningStyle>([
89
+ "think", "include_reasoning", "effort_explicit", "thinking_effort", "template", "anthropic",
90
+ ]);
91
+
92
+ const catalogEfforts = (
93
+ info: ModelInfo,
94
+ declared: readonly ModelReasoningEffort[] = [],
95
+ ): readonly ModelReasoningEffort[] => [
96
+ ...new Set([
97
+ ...(info.reasoningOptions
98
+ ?.filter((option) => option.type === "effort")
99
+ .flatMap((option) => option.values) ?? []),
100
+ ...declared,
101
+ ]),
102
+ ];
53
103
 
54
- const xaiReasoningIsModelFixed = (model: string): boolean =>
55
- /^grok-4\.20(?:-\d{4})?-(?:non-)?reasoning$/.test(model);
104
+ const strongestCatalogEffort = <T extends NativeReasoningEffort | CompatibleReasoningEffort>(
105
+ info: ModelInfo,
106
+ supported: ReadonlySet<T>,
107
+ declared: readonly ModelReasoningEffort[] = [],
108
+ ): T | undefined => {
109
+ const values = new Set(catalogEfforts(info, declared));
110
+ return reasoningEffortOrder.findLast((effort) => supported.has(effort as T) && values.has(effort)) as T | undefined;
111
+ };
112
+
113
+ const catalogSupportsToggle = (info: ModelInfo): boolean =>
114
+ info.reasoningOptions?.some((option) => option.type === "toggle") === true;
56
115
 
57
- const googleReasoningCannotBeOff = (model: string): boolean =>
58
- /^gemini-2\.5-pro(?:-|$)/i.test(model)
59
- || /^gemini-(?:[3-9]|\d{2})[.-]/i.test(model);
116
+ const catalogSupportedReasoningPolicies = ({
117
+ info,
118
+ native,
119
+ style,
120
+ declared,
121
+ }: {
122
+ info: ModelInfo;
123
+ native: boolean;
124
+ style: ReasoningStyle;
125
+ declared: readonly ModelReasoningEffort[];
126
+ }): readonly ReasoningPolicy[] => {
127
+ if (!info.reasoning) return activationPolicies;
128
+ const efforts = new Set(catalogEfforts(info, declared));
129
+ const effortTransport = native || compatibleEffortStyles.has(style);
130
+ const off = native
131
+ ? efforts.has("none") || catalogSupportsToggle(info)
132
+ : (efforts.has("none") && compatibleEffortStyles.has(style))
133
+ || (catalogSupportsToggle(info) && compatibleToggleStyles.has(style));
134
+ return REASONING_POLICIES.filter((policy) => policy === "adaptive"
135
+ || policy === "off" && off
136
+ || (policy === "low" || policy === "medium" || policy === "high")
137
+ && effortTransport
138
+ && efforts.has(policy));
139
+ };
60
140
 
61
141
  const anthropicSupportsAdaptiveThinking = (model: string): boolean =>
62
142
  /claude-(?:opus-(?:4-[678]|5)|sonnet-(?:4-6|5)|fable-5)/.test(model);
@@ -65,29 +145,18 @@ const supportedReasoningPolicies = ({
65
145
  info,
66
146
  native,
67
147
  style,
68
- sdkPackage,
69
- model,
148
+ declared,
70
149
  }: {
71
150
  info?: ModelInfo;
72
151
  native: boolean;
73
152
  style: ReasoningStyle;
74
- sdkPackage?: string;
75
- model: string;
153
+ declared: readonly ModelReasoningEffort[];
76
154
  }): readonly ReasoningPolicy[] => {
77
155
  if (info !== undefined && info.reasoning !== true) return activationPolicies;
78
- if (native) {
79
- if (sdkPackage === "@ai-sdk/mistral") {
80
- return mistralSupportsEffort(model) ? deepSeekPolicies : adaptiveOnly;
81
- }
82
- if (sdkPackage === "@ai-sdk/xai") {
83
- if (xaiReasoningIsModelFixed(model)) return adaptiveOnly;
84
- if (model === "grok-4.6") return reasoningWithoutOff;
85
- }
86
- if (sdkPackage === "@ai-sdk/google" && googleReasoningCannotBeOff(model)) {
87
- return reasoningWithoutOff;
88
- }
89
- return REASONING_POLICIES;
156
+ if (info?.reasoningOptions !== undefined) {
157
+ return catalogSupportedReasoningPolicies({ info, native, style, declared });
90
158
  }
159
+ if (native) return activationPolicies;
91
160
  if (style === "effort" || style === "effort_explicit") return REASONING_POLICIES;
92
161
  if (style === "effort_required") return reasoningWithoutOff;
93
162
  if (style === "thinking_effort") return deepSeekPolicies;
@@ -98,10 +167,14 @@ const adaptiveReasoningProjection = ({
98
167
  sdkPackage,
99
168
  model,
100
169
  reasoningCapable,
170
+ info,
171
+ declared,
101
172
  }: {
102
173
  sdkPackage?: string;
103
174
  model: string;
104
175
  reasoningCapable: boolean;
176
+ info?: ModelInfo;
177
+ declared: readonly ModelReasoningEffort[];
105
178
  }): Pick<AiSdkProviderConfig, "adaptiveReasoning" | "adaptiveReasoningProviderOptions"> => {
106
179
  if (!reasoningCapable) return { adaptiveReasoning: "provider-default" };
107
180
  if (sdkPackage === "@ai-sdk/google" && /^gemini-2\.5(?:-|$)/i.test(model)) {
@@ -128,13 +201,12 @@ const adaptiveReasoningProjection = ({
128
201
  },
129
202
  };
130
203
  }
131
- if (sdkPackage === "@ai-sdk/mistral" && !mistralSupportsEffort(model)) {
132
- return { adaptiveReasoning: "provider-default" };
133
- }
134
- if (sdkPackage === "@ai-sdk/xai" && xaiReasoningIsModelFixed(model)) {
135
- return { adaptiveReasoning: "provider-default" };
204
+ if (info?.reasoningOptions !== undefined) {
205
+ return {
206
+ adaptiveReasoning: strongestCatalogEffort(info, nativeReasoningEfforts, declared) ?? "provider-default",
207
+ };
136
208
  }
137
- return { adaptiveReasoning: "high" };
209
+ return { adaptiveReasoning: "provider-default" };
138
210
  };
139
211
 
140
212
  export const providerFromSdkModel = ({
@@ -191,14 +263,30 @@ export const providerFromSdkModel = ({
191
263
  );
192
264
  const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
193
265
  const reasoningCapable = info?.reasoning === true;
194
- const reasoningStyle = info?.reasoning === false
266
+ const declaredEfforts = declaredEffortsFromEnv(env, name);
267
+ const declaredReasoningStyle = info?.reasoning === false
195
268
  ? "none"
196
269
  : reasoningStyleFromEnv(env, name) ?? "none";
270
+ const reasoningStyle = info?.reasoningOptions !== undefined
271
+ && (declaredReasoningStyle === "effort" || declaredReasoningStyle === "effort_required")
272
+ && catalogEfforts(info, declaredEfforts).length === 0
273
+ ? "none"
274
+ : declaredReasoningStyle;
197
275
  const adaptiveReasoning = adaptiveReasoningProjection({
198
276
  sdkPackage,
199
277
  model,
200
278
  reasoningCapable,
279
+ info,
280
+ declared: declaredEfforts,
201
281
  });
282
+ const compatibleAdaptiveReasoning = languageModel !== undefined || info?.reasoningOptions === undefined
283
+ ? undefined
284
+ : strongestCatalogEffort(info, compatibleReasoningEfforts, declaredEfforts) ?? "provider-default";
285
+ const compatibleOffReasoning = languageModel !== undefined
286
+ || info === undefined
287
+ || !catalogEfforts(info, declaredEfforts).includes("none")
288
+ ? undefined
289
+ : "none" as const;
202
290
 
203
291
  const catalogCost = info?.cost;
204
292
  const rates = catalogCost === undefined ? null : {
@@ -235,10 +323,11 @@ export const providerFromSdkModel = ({
235
323
  info,
236
324
  native: languageModel !== undefined,
237
325
  style: reasoningStyle,
238
- sdkPackage,
239
- model,
326
+ declared: declaredEfforts,
240
327
  }),
241
328
  ...adaptiveReasoning,
329
+ ...(compatibleAdaptiveReasoning === undefined ? {} : { compatibleAdaptiveReasoning }),
330
+ ...(compatibleOffReasoning === undefined ? {} : { compatibleOffReasoning }),
242
331
  ...(additiveReasoningProvider === undefined || !reasoningCapable
243
332
  ? {}
244
333
  : { additiveReasoningProvider }),
package/src/index.ts CHANGED
@@ -45,7 +45,8 @@ export {
45
45
  loadActiveProvider,
46
46
  resetDiscoveryCache,
47
47
  } from "./ProviderRegistry.ts";
48
- export { providerReadiness } from "./sdkModels.ts";
48
+ export { createEmbeddingModel, providerReadiness } from "./sdkModels.ts";
49
+ export type { EmbeddingModelResolution } from "./sdkModels.ts";
49
50
 
50
51
  // Scope-agnostic plugin discovery ({§plugin-family-kind}).
51
52
  export { discover } from "./discover.ts";