@plurnk/plurnk-providers 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +25 -13
- package/README.md +8 -1
- package/SPEC.md +103 -15
- package/dist/AiSdkProvider.d.ts +7 -4
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +150 -53
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +2 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +6 -1
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -0
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +3 -0
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +11 -10
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +16 -8
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +33 -6
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +94 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +2 -0
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +5 -4
- package/dist/cost.js.map +1 -1
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +13 -2
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +3 -2
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +11 -4
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +10 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -3
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +1 -1
- package/dist/notices.d.ts.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +163 -19
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +10 -2
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +10 -1
- package/dist/types.js.map +1 -1
- package/package.json +9 -9
- package/src/AiSdkProvider.test.ts +214 -34
- package/src/AiSdkProvider.ts +181 -56
- package/src/Mock.test.ts +6 -1
- package/src/Mock.ts +6 -1
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +5 -0
- package/src/ProviderRegistry.test.ts +27 -14
- package/src/ProviderRegistry.ts +19 -10
- package/src/accounting.test.ts +30 -2
- package/src/accounting.ts +16 -8
- package/src/aiSdkTransport.test.ts +3 -0
- package/src/aiSdkTransport.ts +38 -8
- package/src/catalogProvider.test.ts +151 -19
- package/src/catalogProvider.ts +125 -3
- package/src/compatibleProvider.test.ts +13 -10
- package/src/compatibleProvider.ts +2 -0
- package/src/cost.ts +5 -4
- package/src/discover.test.ts +27 -0
- package/src/discover.ts +20 -3
- package/src/env.test.ts +23 -8
- package/src/env.ts +18 -9
- package/src/errors.test.ts +2 -2
- package/src/index.ts +17 -8
- package/src/notices.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/providerDefaults.test.ts +50 -0
- package/src/sdkModels.test.ts +142 -8
- package/src/sdkModels.ts +201 -19
- package/src/types.ts +22 -0
|
@@ -2,7 +2,8 @@ import test, { mock } from "node:test";
|
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { once } from "node:events";
|
|
4
4
|
import { createServer } from "node:http";
|
|
5
|
-
import { catalogProviderFromEnv } from "./catalogProvider.ts";
|
|
5
|
+
import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
|
|
6
|
+
import type { LanguageModel } from "ai";
|
|
6
7
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
7
8
|
|
|
8
9
|
const env = {
|
|
@@ -37,6 +38,48 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
37
38
|
assert.equal(provider?.maxOutputTokens, 32_768);
|
|
38
39
|
assert.equal(provider?.outputBudget, 32_768);
|
|
39
40
|
assert.equal(provider?.reasoningBudget, null);
|
|
41
|
+
assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
test("provider adapters advertise only reasoning policies they can preserve", () => {
|
|
45
|
+
const deepseek = catalogProviderFromEnv("deepseek", {
|
|
46
|
+
...env,
|
|
47
|
+
DEEPSEEK_API_KEY: "test-key",
|
|
48
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
49
|
+
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
50
|
+
}, "deepseek-v4-flash");
|
|
51
|
+
assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "high"]);
|
|
52
|
+
|
|
53
|
+
assert.throws(
|
|
54
|
+
() => catalogProviderFromEnv("deepseek", {
|
|
55
|
+
...env,
|
|
56
|
+
DEEPSEEK_API_KEY: "test-key",
|
|
57
|
+
PLURNK_PROVIDERS_REASONING: "medium",
|
|
58
|
+
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
59
|
+
}, "deepseek-v4-flash"),
|
|
60
|
+
/reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
|
|
61
|
+
);
|
|
62
|
+
|
|
63
|
+
const mistral = catalogProviderFromEnv("mistral", {
|
|
64
|
+
...env,
|
|
65
|
+
MISTRAL_API_KEY: "test-key",
|
|
66
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
67
|
+
}, "mistral-small-latest");
|
|
68
|
+
assert.deepEqual(mistral?.supportedReasoningPolicies, ["off", "adaptive", "high"], "Mistral's low/medium coercion is not advertised as exact support");
|
|
69
|
+
|
|
70
|
+
const grok = catalogProviderFromEnv("xai", {
|
|
71
|
+
...env,
|
|
72
|
+
XAI_API_KEY: "test-key",
|
|
73
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
74
|
+
}, "grok-4.6");
|
|
75
|
+
assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Grok 4.6 cannot disable reasoning");
|
|
76
|
+
|
|
77
|
+
const gemini = catalogProviderFromEnv("google", {
|
|
78
|
+
...env,
|
|
79
|
+
GEMINI_API_KEY: "test-key",
|
|
80
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
81
|
+
}, "gemini-3.7-flash");
|
|
82
|
+
assert.deepEqual(gemini?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Gemini 3's mandatory minimum is not advertised as off");
|
|
40
83
|
});
|
|
41
84
|
|
|
42
85
|
test("an operator context window caps catalog physics and percentage output policy", () => {
|
|
@@ -110,7 +153,7 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
|
|
|
110
153
|
assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
|
|
111
154
|
});
|
|
112
155
|
|
|
113
|
-
test("xAI's native chat contract caps the complete reasoning response", async () => {
|
|
156
|
+
test("xAI's native chat contract affirmatively requests adaptive high and caps the complete reasoning response", async () => {
|
|
114
157
|
let call: { headers: Headers; body: Record<string, unknown> } | undefined;
|
|
115
158
|
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
116
159
|
call = {
|
|
@@ -122,21 +165,21 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
|
|
|
122
165
|
id: "response-xai",
|
|
123
166
|
object: "chat.completion.chunk",
|
|
124
167
|
created: 1,
|
|
125
|
-
model: "grok-
|
|
168
|
+
model: "grok-4.6",
|
|
126
169
|
choices: [{ index: 0, delta: { reasoning_content: "consider" }, finish_reason: null }],
|
|
127
170
|
})}`,
|
|
128
171
|
`data: ${JSON.stringify({
|
|
129
172
|
id: "response-xai",
|
|
130
173
|
object: "chat.completion.chunk",
|
|
131
174
|
created: 2,
|
|
132
|
-
model: "grok-
|
|
175
|
+
model: "grok-4.6",
|
|
133
176
|
choices: [{ index: 0, delta: { content: "OK" }, finish_reason: "stop" }],
|
|
134
177
|
})}`,
|
|
135
178
|
`data: ${JSON.stringify({
|
|
136
179
|
id: "response-xai",
|
|
137
180
|
object: "chat.completion.chunk",
|
|
138
181
|
created: 3,
|
|
139
|
-
model: "grok-
|
|
182
|
+
model: "grok-4.6",
|
|
140
183
|
choices: [],
|
|
141
184
|
usage: {
|
|
142
185
|
prompt_tokens: 5,
|
|
@@ -155,7 +198,7 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
|
|
|
155
198
|
...env,
|
|
156
199
|
XAI_API_KEY: "test-key",
|
|
157
200
|
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
158
|
-
}, "grok-
|
|
201
|
+
}, "grok-4.6");
|
|
159
202
|
const result = await provider?.generate({
|
|
160
203
|
workerId: "xai-worker",
|
|
161
204
|
messages: [{ role: "user", content: "hello" }],
|
|
@@ -163,6 +206,7 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
|
|
|
163
206
|
});
|
|
164
207
|
|
|
165
208
|
assert.equal(call?.body.max_completion_tokens, 16);
|
|
209
|
+
assert.equal(call?.body.reasoning_effort, "high", "adaptive is affirmative on xAI's graded route");
|
|
166
210
|
assert.equal("max_tokens" in (call?.body ?? {}), false);
|
|
167
211
|
assert.equal(call?.headers.get("x-grok-conv-id"), "xai-worker");
|
|
168
212
|
assert.equal(result?.assistant.reasoning, "consider");
|
|
@@ -221,14 +265,14 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
|
|
|
221
265
|
const provider = catalogProviderFromEnv("cerebras", {
|
|
222
266
|
...env,
|
|
223
267
|
CEREBRAS_API_KEY: "test-key",
|
|
224
|
-
PLURNK_PROVIDERS_REASONING: "
|
|
268
|
+
PLURNK_PROVIDERS_REASONING: "high",
|
|
225
269
|
}, "gemma-4-31b");
|
|
226
270
|
const result = await provider?.generate({
|
|
227
271
|
workerId: "worker",
|
|
228
272
|
messages: [{ role: "user", content: "hello" }],
|
|
229
273
|
});
|
|
230
274
|
|
|
231
|
-
assert.equal(body?.reasoning_effort, "
|
|
275
|
+
assert.equal(body?.reasoning_effort, "high", "the native SDK preserves the explicit durable effort");
|
|
232
276
|
assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
|
|
233
277
|
assert.equal(result?.assistant.reasoning, "consider");
|
|
234
278
|
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
@@ -236,7 +280,7 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
|
|
|
236
280
|
|
|
237
281
|
test("Google adaptive reasoning requests and preserves readable thought summaries", async () => {
|
|
238
282
|
const bodies: Array<{
|
|
239
|
-
generationConfig?: { thinkingConfig?: { includeThoughts?: boolean } };
|
|
283
|
+
generationConfig?: { thinkingConfig?: { includeThoughts?: boolean; thinkingLevel?: string; thinkingBudget?: number } };
|
|
240
284
|
}> = [];
|
|
241
285
|
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
242
286
|
bodies.push(JSON.parse(String(init?.body)) as typeof bodies[number]);
|
|
@@ -275,25 +319,109 @@ test("Google adaptive reasoning requests and preserves readable thought summarie
|
|
|
275
319
|
|
|
276
320
|
assert.deepEqual(bodies[0]?.generationConfig?.thinkingConfig, {
|
|
277
321
|
includeThoughts: true,
|
|
278
|
-
|
|
322
|
+
thinkingLevel: "high",
|
|
323
|
+
}, "Gemini 3 adaptive selects its documented high/dynamic posture and readable summary");
|
|
279
324
|
assert.equal(result?.assistant.reasoning, "consider");
|
|
280
325
|
assert.equal(result?.assistant.content, "done");
|
|
281
326
|
assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
|
|
282
327
|
|
|
283
|
-
|
|
328
|
+
assert.throws(
|
|
329
|
+
() => catalogProviderFromEnv("google", {
|
|
330
|
+
...env,
|
|
331
|
+
GEMINI_API_KEY: "test-key",
|
|
332
|
+
PLURNK_PROVIDERS_REASONING: "off",
|
|
333
|
+
}, "gemini-3.7-flash"),
|
|
334
|
+
/reasoning policy 'off' is unsupported/,
|
|
335
|
+
"Gemini 3's mandatory minimum thinking is not mislabeled as off",
|
|
336
|
+
);
|
|
337
|
+
|
|
338
|
+
const dynamic25 = catalogProviderFromEnv("google", {
|
|
284
339
|
...env,
|
|
285
340
|
GEMINI_API_KEY: "test-key",
|
|
286
|
-
PLURNK_PROVIDERS_REASONING: "
|
|
287
|
-
}, "gemini-
|
|
288
|
-
await
|
|
341
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
342
|
+
}, "gemini-2.5-flash");
|
|
343
|
+
await dynamic25?.generate({
|
|
289
344
|
workerId: "worker",
|
|
290
345
|
messages: [{ role: "user", content: "hello" }],
|
|
291
346
|
});
|
|
292
|
-
assert.
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
347
|
+
assert.deepEqual(bodies[1]?.generationConfig?.thinkingConfig, {
|
|
348
|
+
includeThoughts: true,
|
|
349
|
+
thinkingBudget: -1,
|
|
350
|
+
}, "Gemini 2.5 adaptive uses the provider's native dynamic budget sentinel");
|
|
351
|
+
});
|
|
352
|
+
|
|
353
|
+
test("native Anthropic adaptive policy uses adaptive thinking rather than a fixed high effort", async () => {
|
|
354
|
+
let request: Record<string, unknown> | undefined;
|
|
355
|
+
const languageModel = {
|
|
356
|
+
specificationVersion: "v4",
|
|
357
|
+
provider: "anthropic.messages",
|
|
358
|
+
modelId: "claude-sonnet-4-6",
|
|
359
|
+
supportedUrls: {},
|
|
360
|
+
doGenerate: async (options: Record<string, unknown>) => {
|
|
361
|
+
request = options;
|
|
362
|
+
return {
|
|
363
|
+
content: [{ type: "text", text: "ok" }],
|
|
364
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
365
|
+
usage: {
|
|
366
|
+
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
367
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
368
|
+
},
|
|
369
|
+
response: { id: "response", modelId: "claude-sonnet-4-6" },
|
|
370
|
+
warnings: [],
|
|
371
|
+
};
|
|
372
|
+
},
|
|
373
|
+
doStream: async (options: Record<string, unknown>) => {
|
|
374
|
+
request = options;
|
|
375
|
+
return {
|
|
376
|
+
stream: new ReadableStream({
|
|
377
|
+
start(controller) {
|
|
378
|
+
controller.enqueue({ type: "stream-start", warnings: [] });
|
|
379
|
+
controller.enqueue({ type: "response-metadata", id: "response", modelId: "claude-sonnet-4-6" });
|
|
380
|
+
controller.enqueue({ type: "text-start", id: "text-1" });
|
|
381
|
+
controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
|
|
382
|
+
controller.enqueue({ type: "text-end", id: "text-1" });
|
|
383
|
+
controller.enqueue({
|
|
384
|
+
type: "finish",
|
|
385
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
386
|
+
usage: {
|
|
387
|
+
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
388
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
389
|
+
},
|
|
390
|
+
});
|
|
391
|
+
controller.close();
|
|
392
|
+
},
|
|
393
|
+
}),
|
|
394
|
+
response: {},
|
|
395
|
+
};
|
|
396
|
+
},
|
|
397
|
+
} as unknown as LanguageModel;
|
|
398
|
+
const provider = providerFromSdkModel({
|
|
399
|
+
name: "anthropic",
|
|
400
|
+
env: {
|
|
401
|
+
...env,
|
|
402
|
+
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
403
|
+
PLURNK_PROVIDERS_STREAMING: "0",
|
|
404
|
+
},
|
|
405
|
+
model: "claude-sonnet-4-6",
|
|
406
|
+
languageModel,
|
|
407
|
+
sdkPackage: "@ai-sdk/anthropic",
|
|
408
|
+
additiveReasoningProvider: "anthropic",
|
|
409
|
+
contextWindow: 16_384,
|
|
410
|
+
info: {
|
|
411
|
+
name: "Claude Sonnet 4.6",
|
|
412
|
+
contextWindow: 16_384,
|
|
413
|
+
maxOutputTokens: 8_192,
|
|
414
|
+
reasoning: true,
|
|
415
|
+
attachment: true,
|
|
416
|
+
toolCall: true,
|
|
417
|
+
modalities: { input: ["text", "image"], output: ["text"] },
|
|
418
|
+
},
|
|
419
|
+
});
|
|
420
|
+
await provider.generate({ workerId: "worker", messages: [{ role: "user", content: "hello" }] });
|
|
421
|
+
assert.equal(request?.reasoning, "provider-default");
|
|
422
|
+
assert.deepEqual(request?.providerOptions, {
|
|
423
|
+
anthropic: { thinking: { type: "adaptive", display: "summarized" } },
|
|
424
|
+
});
|
|
297
425
|
});
|
|
298
426
|
|
|
299
427
|
test("native provider routes project their documented cache controls through the actual SDK request", async (t) => {
|
|
@@ -378,6 +506,8 @@ test("native provider routes project their documented cache controls through the
|
|
|
378
506
|
const provider = catalogProviderFromEnv("openrouter", {
|
|
379
507
|
...env,
|
|
380
508
|
OPENROUTER_API_KEY: "test-key",
|
|
509
|
+
OPENROUTER_HTTP_REFERER: "https://github.com/plurnk/plurnk-service",
|
|
510
|
+
OPENROUTER_APP_TITLE: "Plurnk",
|
|
381
511
|
}, "anthropic/claude-sonnet-4.6", `http://127.0.0.1:${address.port}/api/v1`);
|
|
382
512
|
await provider?.generate({
|
|
383
513
|
workerId: "openrouter-worker",
|
|
@@ -387,6 +517,8 @@ test("native provider routes project their documented cache controls through the
|
|
|
387
517
|
],
|
|
388
518
|
});
|
|
389
519
|
assert.equal(call?.headers.get("x-session-id"), "openrouter-worker");
|
|
520
|
+
assert.equal(call?.headers.get("http-referer"), "https://github.com/plurnk/plurnk-service");
|
|
521
|
+
assert.equal(call?.headers.get("x-openrouter-title"), "Plurnk");
|
|
390
522
|
assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[0], {
|
|
391
523
|
role: "system",
|
|
392
524
|
content: [{
|
package/src/catalogProvider.ts
CHANGED
|
@@ -12,10 +12,11 @@ import {
|
|
|
12
12
|
reasoningFromEnv,
|
|
13
13
|
reasoningResponseStyleFromEnv,
|
|
14
14
|
} from "./env.ts";
|
|
15
|
-
import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
15
|
+
import AiSdkProvider, { type AiSdkProviderConfig, type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
16
16
|
import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
|
|
17
17
|
import { providerSource } from "./notices.ts";
|
|
18
18
|
import type { Provider, ProviderCostNormalizer } from "./types.ts";
|
|
19
|
+
import { REASONING_POLICIES, type ReasoningPolicy } from "@plurnk/plurnk-contracts";
|
|
19
20
|
import { estimateProviderCost } from "./cost.ts";
|
|
20
21
|
import { emitWarningOnce } from "./warnings.ts";
|
|
21
22
|
import type { LanguageModel } from "ai";
|
|
@@ -39,6 +40,102 @@ const reasoningStyleFromEnv = (
|
|
|
39
40
|
return value as ReasoningStyle;
|
|
40
41
|
};
|
|
41
42
|
|
|
43
|
+
const activationPolicies = Object.freeze(["off", "adaptive"] as const);
|
|
44
|
+
const deepSeekPolicies = Object.freeze(["off", "adaptive", "high"] as const);
|
|
45
|
+
const adaptiveOnly = Object.freeze(["adaptive"] as const);
|
|
46
|
+
const reasoningWithoutOff = Object.freeze(["adaptive", "low", "medium", "high"] as const);
|
|
47
|
+
|
|
48
|
+
const mistralSupportsEffort = (model: string): boolean =>
|
|
49
|
+
model === "mistral-small-latest"
|
|
50
|
+
|| model === "mistral-small-2603"
|
|
51
|
+
|| model === "mistral-medium-3"
|
|
52
|
+
|| model === "mistral-medium-3.5";
|
|
53
|
+
|
|
54
|
+
const xaiReasoningIsModelFixed = (model: string): boolean =>
|
|
55
|
+
/^grok-4\.20(?:-\d{4})?-(?:non-)?reasoning$/.test(model);
|
|
56
|
+
|
|
57
|
+
const googleReasoningCannotBeOff = (model: string): boolean =>
|
|
58
|
+
/^gemini-2\.5-pro(?:-|$)/i.test(model)
|
|
59
|
+
|| /^gemini-(?:[3-9]|\d{2})[.-]/i.test(model);
|
|
60
|
+
|
|
61
|
+
const anthropicSupportsAdaptiveThinking = (model: string): boolean =>
|
|
62
|
+
/claude-(?:opus-(?:4-[678]|5)|sonnet-(?:4-6|5)|fable-5)/.test(model);
|
|
63
|
+
|
|
64
|
+
const supportedReasoningPolicies = ({
|
|
65
|
+
info,
|
|
66
|
+
native,
|
|
67
|
+
style,
|
|
68
|
+
sdkPackage,
|
|
69
|
+
model,
|
|
70
|
+
}: {
|
|
71
|
+
info?: ModelInfo;
|
|
72
|
+
native: boolean;
|
|
73
|
+
style: ReasoningStyle;
|
|
74
|
+
sdkPackage?: string;
|
|
75
|
+
model: string;
|
|
76
|
+
}): readonly ReasoningPolicy[] => {
|
|
77
|
+
if (info !== undefined && info.reasoning !== true) return activationPolicies;
|
|
78
|
+
if (native) {
|
|
79
|
+
if (sdkPackage === "@ai-sdk/mistral") {
|
|
80
|
+
return mistralSupportsEffort(model) ? deepSeekPolicies : adaptiveOnly;
|
|
81
|
+
}
|
|
82
|
+
if (sdkPackage === "@ai-sdk/xai") {
|
|
83
|
+
if (xaiReasoningIsModelFixed(model)) return adaptiveOnly;
|
|
84
|
+
if (model === "grok-4.6") return reasoningWithoutOff;
|
|
85
|
+
}
|
|
86
|
+
if (sdkPackage === "@ai-sdk/google" && googleReasoningCannotBeOff(model)) {
|
|
87
|
+
return reasoningWithoutOff;
|
|
88
|
+
}
|
|
89
|
+
return REASONING_POLICIES;
|
|
90
|
+
}
|
|
91
|
+
if (style === "effort" || style === "effort_explicit") return REASONING_POLICIES;
|
|
92
|
+
if (style === "thinking_effort") return deepSeekPolicies;
|
|
93
|
+
return activationPolicies;
|
|
94
|
+
};
|
|
95
|
+
|
|
96
|
+
const adaptiveReasoningProjection = ({
|
|
97
|
+
sdkPackage,
|
|
98
|
+
model,
|
|
99
|
+
reasoningCapable,
|
|
100
|
+
}: {
|
|
101
|
+
sdkPackage?: string;
|
|
102
|
+
model: string;
|
|
103
|
+
reasoningCapable: boolean;
|
|
104
|
+
}): Pick<AiSdkProviderConfig, "adaptiveReasoning" | "adaptiveReasoningProviderOptions"> => {
|
|
105
|
+
if (!reasoningCapable) return { adaptiveReasoning: "provider-default" };
|
|
106
|
+
if (sdkPackage === "@ai-sdk/google" && /^gemini-2\.5(?:-|$)/i.test(model)) {
|
|
107
|
+
return {
|
|
108
|
+
adaptiveReasoning: "provider-default",
|
|
109
|
+
adaptiveReasoningProviderOptions: {
|
|
110
|
+
google: { thinkingConfig: { thinkingBudget: -1 } },
|
|
111
|
+
},
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
if (sdkPackage === "@ai-sdk/anthropic" && anthropicSupportsAdaptiveThinking(model)) {
|
|
115
|
+
return {
|
|
116
|
+
adaptiveReasoning: "provider-default",
|
|
117
|
+
adaptiveReasoningProviderOptions: {
|
|
118
|
+
anthropic: { thinking: { type: "adaptive", display: "summarized" } },
|
|
119
|
+
},
|
|
120
|
+
};
|
|
121
|
+
}
|
|
122
|
+
if (sdkPackage === "@ai-sdk/amazon-bedrock" && anthropicSupportsAdaptiveThinking(model)) {
|
|
123
|
+
return {
|
|
124
|
+
adaptiveReasoning: "provider-default",
|
|
125
|
+
adaptiveReasoningProviderOptions: {
|
|
126
|
+
bedrock: { reasoningConfig: { type: "adaptive" } },
|
|
127
|
+
},
|
|
128
|
+
};
|
|
129
|
+
}
|
|
130
|
+
if (sdkPackage === "@ai-sdk/mistral" && !mistralSupportsEffort(model)) {
|
|
131
|
+
return { adaptiveReasoning: "provider-default" };
|
|
132
|
+
}
|
|
133
|
+
if (sdkPackage === "@ai-sdk/xai" && xaiReasoningIsModelFixed(model)) {
|
|
134
|
+
return { adaptiveReasoning: "provider-default" };
|
|
135
|
+
}
|
|
136
|
+
return { adaptiveReasoning: "high" };
|
|
137
|
+
};
|
|
138
|
+
|
|
42
139
|
export const providerFromSdkModel = ({
|
|
43
140
|
name,
|
|
44
141
|
env,
|
|
@@ -54,6 +151,8 @@ export const providerFromSdkModel = ({
|
|
|
54
151
|
systemCacheProviderOptions,
|
|
55
152
|
reasoningResponseProviderOptions,
|
|
56
153
|
additiveReasoningProvider,
|
|
154
|
+
sdkPackage,
|
|
155
|
+
grammarStyle,
|
|
57
156
|
}: {
|
|
58
157
|
name: string;
|
|
59
158
|
env: NodeJS.ProcessEnv;
|
|
@@ -65,10 +164,14 @@ export const providerFromSdkModel = ({
|
|
|
65
164
|
contextWindow: number;
|
|
66
165
|
info?: ModelInfo;
|
|
67
166
|
attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
167
|
+
// {§provider-grammar-transport} — plugin-declared constrained-decoding
|
|
168
|
+
// capability; "none" keeps the grammar off the wire.
|
|
169
|
+
grammarStyle?: GrammarStyle;
|
|
68
170
|
cacheAffinity?: CacheAffinity;
|
|
69
171
|
systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
70
172
|
reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
71
173
|
additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
174
|
+
sdkPackage?: string;
|
|
72
175
|
}): Provider => {
|
|
73
176
|
emitWarningOnce(
|
|
74
177
|
`${name} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
|
|
@@ -86,6 +189,13 @@ export const providerFromSdkModel = ({
|
|
|
86
189
|
maxOutputTokens,
|
|
87
190
|
);
|
|
88
191
|
const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
|
|
192
|
+
const reasoningStyle = reasoningStyleFromEnv(env, name) ?? "none";
|
|
193
|
+
const reasoningCapable = info?.reasoning === true;
|
|
194
|
+
const adaptiveReasoning = adaptiveReasoningProjection({
|
|
195
|
+
sdkPackage,
|
|
196
|
+
model,
|
|
197
|
+
reasoningCapable,
|
|
198
|
+
});
|
|
89
199
|
|
|
90
200
|
const catalogCost = info?.cost;
|
|
91
201
|
const rates = catalogCost === undefined ? null : {
|
|
@@ -118,7 +228,17 @@ export const providerFromSdkModel = ({
|
|
|
118
228
|
maxOutputTokens,
|
|
119
229
|
outputBudget: envelope.outputBudget,
|
|
120
230
|
reasoningBudget: reasoning.budget,
|
|
121
|
-
|
|
231
|
+
supportedReasoningPolicies: supportedReasoningPolicies({
|
|
232
|
+
info,
|
|
233
|
+
native: languageModel !== undefined,
|
|
234
|
+
style: reasoningStyle,
|
|
235
|
+
sdkPackage,
|
|
236
|
+
model,
|
|
237
|
+
}),
|
|
238
|
+
...adaptiveReasoning,
|
|
239
|
+
...(additiveReasoningProvider === undefined || !reasoningCapable
|
|
240
|
+
? {}
|
|
241
|
+
: { additiveReasoningProvider }),
|
|
122
242
|
fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
|
|
123
243
|
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
|
|
124
244
|
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
|
|
@@ -130,7 +250,7 @@ export const providerFromSdkModel = ({
|
|
|
130
250
|
frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
|
|
131
251
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
132
252
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
133
|
-
reasoningStyle
|
|
253
|
+
reasoningStyle,
|
|
134
254
|
...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
|
|
135
255
|
...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
|
|
136
256
|
? { systemCacheProviderOptions }
|
|
@@ -141,6 +261,7 @@ export const providerFromSdkModel = ({
|
|
|
141
261
|
serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
|
|
142
262
|
estimateCost,
|
|
143
263
|
source: providerSource(name),
|
|
264
|
+
...(grammarStyle === undefined ? {} : { grammarStyle }),
|
|
144
265
|
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
|
|
145
266
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
|
|
146
267
|
&& env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
|
|
@@ -186,6 +307,7 @@ export const catalogProviderFromEnv = (
|
|
|
186
307
|
systemCacheProviderOptions: sdk.systemCacheProviderOptions,
|
|
187
308
|
reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
|
|
188
309
|
additiveReasoningProvider: sdk.additiveReasoningProvider,
|
|
310
|
+
sdkPackage: sdk.catalog?.npm,
|
|
189
311
|
contextWindow,
|
|
190
312
|
info,
|
|
191
313
|
});
|
|
@@ -21,6 +21,17 @@ const env = {
|
|
|
21
21
|
PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
22
22
|
};
|
|
23
23
|
|
|
24
|
+
const streamedChatResponse = (content: string) => new Response([
|
|
25
|
+
`data: ${JSON.stringify({
|
|
26
|
+
id: "test-completion",
|
|
27
|
+
object: "chat.completion.chunk",
|
|
28
|
+
created: 1,
|
|
29
|
+
model: "local",
|
|
30
|
+
choices: [{ index: 0, delta: { content }, finish_reason: "stop" }],
|
|
31
|
+
})}`,
|
|
32
|
+
"data: [DONE]",
|
|
33
|
+
].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
|
|
34
|
+
|
|
24
35
|
test.afterEach(() => mock.restoreAll());
|
|
25
36
|
|
|
26
37
|
test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
|
|
@@ -30,11 +41,7 @@ test("an undifferentiated compatible endpoint receives no guessed prompt-cache f
|
|
|
30
41
|
return new Response(JSON.stringify({ data: [{ id: "local", n_ctx: 8192 }] }));
|
|
31
42
|
}
|
|
32
43
|
body = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
33
|
-
return
|
|
34
|
-
model: "local",
|
|
35
|
-
choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
|
|
36
|
-
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
|
|
37
|
-
}), { headers: { "content-type": "application/json" } });
|
|
44
|
+
return streamedChatResponse("ok");
|
|
38
45
|
});
|
|
39
46
|
|
|
40
47
|
const provider = await compatibleProviderFromEnv("openai", env, "local");
|
|
@@ -55,11 +62,7 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
|
55
62
|
}));
|
|
56
63
|
}
|
|
57
64
|
body = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
58
|
-
return
|
|
59
|
-
model: "local",
|
|
60
|
-
choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
|
|
61
|
-
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
|
|
62
|
-
}), { headers: { "content-type": "application/json" } });
|
|
65
|
+
return streamedChatResponse("ok");
|
|
63
66
|
});
|
|
64
67
|
|
|
65
68
|
const provider = await compatibleProviderFromEnv("openai", {
|
|
@@ -182,6 +182,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
182
182
|
}
|
|
183
183
|
const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
|
|
184
184
|
const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
|
|
185
|
+
const supportedReasoningPolicies = ["off", "adaptive"] as const;
|
|
185
186
|
return new AiSdkProvider({
|
|
186
187
|
model,
|
|
187
188
|
url,
|
|
@@ -191,6 +192,7 @@ export const compatibleProviderFromEnv = async (
|
|
|
191
192
|
maxOutputTokens: null,
|
|
192
193
|
outputBudget: envelope.outputBudget,
|
|
193
194
|
reasoningBudget: reasoning.budget,
|
|
195
|
+
supportedReasoningPolicies,
|
|
194
196
|
fetchTimeoutMs: timeout,
|
|
195
197
|
operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
|
|
196
198
|
firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
|
package/src/cost.ts
CHANGED
|
@@ -132,8 +132,9 @@ export const addDecimals = (values: readonly string[]): string => {
|
|
|
132
132
|
};
|
|
133
133
|
|
|
134
134
|
export const sumProviderCostsUsd = (costs: readonly ProviderCost[]): string | null => {
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
135
|
+
// {§tokenomics-provider-usage} — a request without USD-expressible cost
|
|
136
|
+
// (an uncataloged model, or a response-less failure) is skipped; it never
|
|
137
|
+
// erases the expressible evidence. Null only when nothing is expressible.
|
|
138
|
+
const values = costs.map(providerCostUsd).filter((value): value is string => value !== null);
|
|
139
|
+
return values.length === 0 ? null : addDecimals(values);
|
|
139
140
|
};
|
package/src/discover.test.ts
CHANGED
|
@@ -47,6 +47,33 @@ test("discover: a provider package missing plurnk.name is ignored, not crashed",
|
|
|
47
47
|
assert.deepEqual([...registry.keys()], ["named"]);
|
|
48
48
|
});
|
|
49
49
|
|
|
50
|
+
test("{§provider-grammar-transport} discover: the manifest grammarStyle declaration is recorded and validated", async (t) => {
|
|
51
|
+
const root = await buildModules(t, {
|
|
52
|
+
"@acme/llamacpp-rail": {
|
|
53
|
+
name: "@acme/llamacpp-rail",
|
|
54
|
+
plurnk: { kind: "provider", name: "rail", grammarStyle: "llamacpp" },
|
|
55
|
+
},
|
|
56
|
+
"@acme/plain": {
|
|
57
|
+
name: "@acme/plain",
|
|
58
|
+
plurnk: { kind: "provider", name: "plain" },
|
|
59
|
+
},
|
|
60
|
+
});
|
|
61
|
+
const { grammarStyles } = await discover({ cwd: root });
|
|
62
|
+
assert.equal(grammarStyles.get("rail"), "llamacpp");
|
|
63
|
+
assert.equal(grammarStyles.get("plain"), "none");
|
|
64
|
+
assert.equal(grammarStyles.get("absent"), undefined);
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test("{§provider-grammar-transport} discover: an invalid grammarStyle fails loudly, never guessing", async (t) => {
|
|
68
|
+
const root = await buildModules(t, {
|
|
69
|
+
"@acme/bad-grammar": {
|
|
70
|
+
name: "@acme/bad-grammar",
|
|
71
|
+
plurnk: { kind: "provider", name: "bad", grammarStyle: "guff" },
|
|
72
|
+
},
|
|
73
|
+
});
|
|
74
|
+
await assert.rejects(discover({ cwd: root }), /grammarStyle must be "none" or "llamacpp"/);
|
|
75
|
+
});
|
|
76
|
+
|
|
50
77
|
test("discover: an array kind claims no provider family", async (t) => {
|
|
51
78
|
const root = await buildModules(t, {
|
|
52
79
|
"@acme/dual": {
|
package/src/discover.ts
CHANGED
|
@@ -6,6 +6,7 @@ import type {
|
|
|
6
6
|
PluginAttribution,
|
|
7
7
|
PluginAttributionDeclaration,
|
|
8
8
|
} from "@plurnk/plurnk-meta";
|
|
9
|
+
import type { GrammarStyle } from "./AiSdkProvider.ts";
|
|
9
10
|
|
|
10
11
|
// Scope-agnostic discovery of installed AI SDK provider packages
|
|
11
12
|
// ({§plugin-family-kind}).
|
|
@@ -40,6 +41,9 @@ export type Discovery = {
|
|
|
40
41
|
// Published name-keyed projection retained for 1.x consumers.
|
|
41
42
|
attributions: Map<string, string | string[]>;
|
|
42
43
|
packageAttributions: PackageAttributions;
|
|
44
|
+
// {§provider-grammar-transport} — plugin-declared constrained-decoding
|
|
45
|
+
// capability per provider name; "none" unless the manifest declares one.
|
|
46
|
+
grammarStyles: Map<string, GrammarStyle>;
|
|
43
47
|
};
|
|
44
48
|
|
|
45
49
|
|
|
@@ -51,6 +55,7 @@ export const discover = async (options: DiscoverOptions = {}): Promise<Discovery
|
|
|
51
55
|
const skipped = new Map<string, string>();
|
|
52
56
|
const attributions = new Map<string, PluginAttributionDeclaration>();
|
|
53
57
|
const packageAttributions = new Map<string, PluginAttribution>();
|
|
58
|
+
const grammarStyles = new Map<string, GrammarStyle>();
|
|
54
59
|
for (const dir of dirs) {
|
|
55
60
|
const info = await readProviderInfo(dir);
|
|
56
61
|
if (info === null) continue;
|
|
@@ -67,11 +72,12 @@ export const discover = async (options: DiscoverOptions = {}): Promise<Discovery
|
|
|
67
72
|
}
|
|
68
73
|
const tags = Meta.normalizeAttribution(info.attribution, info.packageName);
|
|
69
74
|
registry.set(info.name, info.packageName);
|
|
75
|
+
grammarStyles.set(info.name, info.grammarStyle);
|
|
70
76
|
const attribution = attributionProjection(info.attribution, tags);
|
|
71
77
|
if (attribution !== undefined) attributions.set(info.name, attribution);
|
|
72
78
|
if (tags.length > 0) packageAttributions.set(info.packageName, tags);
|
|
73
79
|
}
|
|
74
|
-
return { registry, skipped, attributions, packageAttributions };
|
|
80
|
+
return { registry, skipped, attributions, packageAttributions, grammarStyles };
|
|
75
81
|
};
|
|
76
82
|
|
|
77
83
|
// Enumerate every installed package directory — scoped and unscoped — under
|
|
@@ -84,7 +90,7 @@ const defaultPackageDirs = async (cwd: string): Promise<string[]> => {
|
|
|
84
90
|
// One inert manifest record for a provider package, or null for anything that
|
|
85
91
|
// isn't one. Attribution remains unknown until trust admission, then the shared
|
|
86
92
|
// {§plugin-attribution} boundary validates it.
|
|
87
|
-
type ProviderInfo = { name: string; packageName: string; attribution: unknown };
|
|
93
|
+
type ProviderInfo = { name: string; packageName: string; attribution: unknown; grammarStyle: GrammarStyle };
|
|
88
94
|
|
|
89
95
|
const readProviderInfo = async (dir: string): Promise<ProviderInfo | null> => {
|
|
90
96
|
let raw: string;
|
|
@@ -107,7 +113,18 @@ const readProviderInfo = async (dir: string): Promise<ProviderInfo | null> => {
|
|
|
107
113
|
if (!Meta.declaresKind(plurnkRec, "provider")) return null;
|
|
108
114
|
if (typeof plurnkRec.name !== "string" || plurnkRec.name === "") return null;
|
|
109
115
|
if (typeof record.name !== "string" || record.name === "") return null;
|
|
110
|
-
|
|
116
|
+
const grammarStyle = plurnkRec.grammarStyle;
|
|
117
|
+
if (grammarStyle !== undefined && grammarStyle !== "none" && grammarStyle !== "llamacpp") {
|
|
118
|
+
throw new Error(
|
|
119
|
+
`${record.name}: plurnk.grammarStyle must be "none" or "llamacpp", got ${JSON.stringify(grammarStyle)}.`,
|
|
120
|
+
);
|
|
121
|
+
}
|
|
122
|
+
return {
|
|
123
|
+
name: plurnkRec.name,
|
|
124
|
+
packageName: record.name,
|
|
125
|
+
attribution: plurnkRec.attribution,
|
|
126
|
+
grammarStyle: grammarStyle === undefined ? "none" : grammarStyle,
|
|
127
|
+
};
|
|
111
128
|
};
|
|
112
129
|
|
|
113
130
|
const attributionProjection = (
|