@plurnk/plurnk-providers 1.7.0 → 1.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +25 -13
- package/README.md +8 -1
- package/SPEC.md +93 -15
- package/dist/AiSdkProvider.d.ts +6 -3
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +108 -51
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +1 -0
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +2 -0
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -0
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +3 -0
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +11 -10
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +7 -6
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +2 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +29 -6
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +94 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +2 -0
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +5 -4
- package/dist/cost.js.map +1 -1
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +13 -2
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +3 -2
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +11 -4
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +9 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -3
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +1 -1
- package/dist/notices.d.ts.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +163 -19
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +8 -2
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +10 -1
- package/dist/types.js.map +1 -1
- package/package.json +9 -9
- package/src/AiSdkProvider.test.ts +206 -32
- package/src/AiSdkProvider.ts +140 -54
- package/src/Mock.ts +2 -0
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +5 -0
- package/src/ProviderRegistry.test.ts +27 -14
- package/src/ProviderRegistry.ts +19 -10
- package/src/accounting.test.ts +6 -2
- package/src/accounting.ts +7 -6
- package/src/aiSdkTransport.ts +32 -7
- package/src/catalogProvider.test.ts +151 -19
- package/src/catalogProvider.ts +125 -3
- package/src/compatibleProvider.test.ts +13 -10
- package/src/compatibleProvider.ts +2 -0
- package/src/cost.ts +5 -4
- package/src/discover.test.ts +27 -0
- package/src/discover.ts +20 -3
- package/src/env.test.ts +23 -8
- package/src/env.ts +17 -8
- package/src/index.ts +16 -8
- package/src/notices.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/providerDefaults.test.ts +50 -0
- package/src/sdkModels.test.ts +142 -8
- package/src/sdkModels.ts +201 -19
- package/src/types.ts +16 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import AiSdkProvider, {
|
|
3
|
+
import AiSdkProvider, { type AiSdkProviderConfig } from "./AiSdkProvider.ts";
|
|
4
4
|
import { ProviderError } from "./errors.ts";
|
|
5
5
|
import { providerCostNormalizer } from "./accounting.ts";
|
|
6
6
|
import type { LanguageModel } from "ai";
|
|
@@ -60,9 +60,18 @@ const sseStream = (chunks: unknown[]) => {
|
|
|
60
60
|
|
|
61
61
|
const installFetch = (chunks: unknown[]) => {
|
|
62
62
|
const calls: { url: string; init: RequestInit }[] = [];
|
|
63
|
+
const completeChunks = chunks.some((value) => {
|
|
64
|
+
const choices = (value as { choices?: Array<{ finish_reason?: unknown }> }).choices;
|
|
65
|
+
return choices?.some(({ finish_reason }) => finish_reason != null) ?? false;
|
|
66
|
+
})
|
|
67
|
+
? chunks
|
|
68
|
+
: [...chunks, { choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }];
|
|
63
69
|
mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
|
|
64
70
|
calls.push({ url, init });
|
|
65
|
-
return new Response(sseStream(
|
|
71
|
+
return new Response(sseStream(completeChunks), {
|
|
72
|
+
status: 200,
|
|
73
|
+
headers: { "content-type": "text/event-stream" },
|
|
74
|
+
});
|
|
66
75
|
});
|
|
67
76
|
return calls;
|
|
68
77
|
};
|
|
@@ -296,14 +305,6 @@ const flush = () => new Promise<void>((r) => setImmediate(r));
|
|
|
296
305
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
297
306
|
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
298
307
|
|
|
299
|
-
test("effortFromBudget: maps budget to tiers", () => {
|
|
300
|
-
assert.equal(effortFromBudget(1), "low");
|
|
301
|
-
assert.equal(effortFromBudget(1000), "low");
|
|
302
|
-
assert.equal(effortFromBudget(1001), "medium");
|
|
303
|
-
assert.equal(effortFromBudget(4000), "medium");
|
|
304
|
-
assert.equal(effortFromBudget(4001), "high");
|
|
305
|
-
});
|
|
306
|
-
|
|
307
308
|
test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
308
309
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
309
310
|
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
@@ -602,6 +603,58 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
602
603
|
assert.notEqual(assistantRaw, undefined);
|
|
603
604
|
});
|
|
604
605
|
|
|
606
|
+
test("an unsupported fixed reasoning policy fails before provider I/O", () => {
|
|
607
|
+
assert.throws(
|
|
608
|
+
() => testProvider({
|
|
609
|
+
...injectedBase,
|
|
610
|
+
reasoning: { mode: "medium", budget: null },
|
|
611
|
+
supportedReasoningPolicies: ["off", "adaptive", "high"],
|
|
612
|
+
}),
|
|
613
|
+
/reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
|
|
614
|
+
);
|
|
615
|
+
});
|
|
616
|
+
|
|
617
|
+
test("native SDK warnings survive as source-attributed provider Notices", async () => {
|
|
618
|
+
const usage = {
|
|
619
|
+
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
620
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
621
|
+
};
|
|
622
|
+
const languageModel = {
|
|
623
|
+
specificationVersion: "v4",
|
|
624
|
+
provider: "native.test",
|
|
625
|
+
modelId: "warning-test",
|
|
626
|
+
supportedUrls: {},
|
|
627
|
+
doGenerate: async () => ({
|
|
628
|
+
content: [{ type: "text", text: "ok" }],
|
|
629
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
630
|
+
usage,
|
|
631
|
+
response: { id: "response-warning", modelId: "warning-test" },
|
|
632
|
+
warnings: [{
|
|
633
|
+
type: "compatibility",
|
|
634
|
+
feature: "reasoning",
|
|
635
|
+
details: "reasoning low was mapped to high",
|
|
636
|
+
}],
|
|
637
|
+
}),
|
|
638
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
639
|
+
} as unknown as LanguageModel;
|
|
640
|
+
const response = await testProvider({
|
|
641
|
+
...injectedBase,
|
|
642
|
+
url: undefined,
|
|
643
|
+
model: "warning-test",
|
|
644
|
+
languageModel,
|
|
645
|
+
source: "provider:warning-test",
|
|
646
|
+
streaming: false,
|
|
647
|
+
}).generate({ workerId: "warnings", messages: [] });
|
|
648
|
+
|
|
649
|
+
assert.deepEqual(response.notices, [{
|
|
650
|
+
source: "provider:warning-test",
|
|
651
|
+
kind: "provider_warning",
|
|
652
|
+
level: "warn",
|
|
653
|
+
message: "compatibility reasoning: reasoning low was mapped to high",
|
|
654
|
+
position: null,
|
|
655
|
+
}]);
|
|
656
|
+
});
|
|
657
|
+
|
|
605
658
|
test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
|
|
606
659
|
const usage = {
|
|
607
660
|
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
@@ -1157,21 +1210,21 @@ test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the
|
|
|
1157
1210
|
assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
|
|
1158
1211
|
});
|
|
1159
1212
|
|
|
1160
|
-
test("reasoningStyle 'effort'
|
|
1161
|
-
for (const
|
|
1162
|
-
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions",
|
|
1213
|
+
test("reasoningStyle 'effort' preserves each fixed portable effort", async () => {
|
|
1214
|
+
for (const mode of ["low", "medium", "high"] as const) {
|
|
1215
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode, budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
1163
1216
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1164
1217
|
await p.generate({ workerId: "r", messages: [] });
|
|
1165
|
-
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort,
|
|
1218
|
+
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, mode);
|
|
1166
1219
|
mock.restoreAll();
|
|
1167
1220
|
}
|
|
1168
1221
|
});
|
|
1169
1222
|
|
|
1170
|
-
test("reasoningStyle 'effort_explicit': off
|
|
1223
|
+
test("reasoningStyle 'effort_explicit': off sends none, adaptive omits, fixed effort remains exact", async () => {
|
|
1171
1224
|
// expected === null → the field must be ABSENT from the wire body. Fireworks
|
|
1172
1225
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
1173
1226
|
// Adaptive = the backend's own default posture = omission.
|
|
1174
|
-
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "
|
|
1227
|
+
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "low", budget: null }, "low"], [{ mode: "high", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "low" | "high"; budget: number | null }, string | null]>) {
|
|
1175
1228
|
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", ...(reasoning.budget === null ? {} : { outputBudget: reasoning.budget + 1, reasoningBudget: reasoning.budget }), fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
1176
1229
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1177
1230
|
await p.generate({ workerId: "r", messages: [] });
|
|
@@ -1185,9 +1238,9 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends
|
|
|
1185
1238
|
test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
|
|
1186
1239
|
const cases = [
|
|
1187
1240
|
[{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
|
|
1188
|
-
[{ mode: "adaptive", budget: null }, {}],
|
|
1189
|
-
[{ mode: "
|
|
1190
|
-
[{ mode: "
|
|
1241
|
+
[{ mode: "adaptive", budget: null }, { thinking: { type: "enabled" } }],
|
|
1242
|
+
[{ mode: "high", budget: null }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
1243
|
+
[{ mode: "high", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
1191
1244
|
] as const;
|
|
1192
1245
|
for (const [reasoning, expected] of cases) {
|
|
1193
1246
|
const p = testProvider({
|
|
@@ -1478,14 +1531,14 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
|
|
|
1478
1531
|
|
|
1479
1532
|
test("reasoningStyle 'template' carries the explicit reasoning subset and rejects an additive envelope", async () => {
|
|
1480
1533
|
const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, outputBudget: 224, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
|
|
1481
|
-
const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "
|
|
1534
|
+
const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "high", budget: 32 } });
|
|
1482
1535
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1483
1536
|
await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
|
|
1484
1537
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1485
1538
|
assert.equal(body.thinking_budget_tokens, 32);
|
|
1486
1539
|
assert.equal(body.reasoning_format, "auto");
|
|
1487
1540
|
assert.throws(
|
|
1488
|
-
() => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "
|
|
1541
|
+
() => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "high", budget: 224 } }),
|
|
1489
1542
|
/reasoningBudget must be smaller than the total outputBudget/,
|
|
1490
1543
|
);
|
|
1491
1544
|
});
|
|
@@ -1500,7 +1553,7 @@ test("reasoningStyle 'template' explicit activation uses the configured reasonin
|
|
|
1500
1553
|
fetchTimeoutMs: 5000,
|
|
1501
1554
|
temperature: 0.2,
|
|
1502
1555
|
repeatPenalty: 1.15,
|
|
1503
|
-
reasoning: { mode: "
|
|
1556
|
+
reasoning: { mode: "high", budget: 64 },
|
|
1504
1557
|
retryAttempts: 0,
|
|
1505
1558
|
reasoningStyle: "template",
|
|
1506
1559
|
});
|
|
@@ -1911,7 +1964,7 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
1911
1964
|
{ status: 409, retryAfter: 0 },
|
|
1912
1965
|
{ status: 429, retryAfter: 0 },
|
|
1913
1966
|
{ status: 503, retryAfter: 0 },
|
|
1914
|
-
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
|
|
1967
|
+
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
|
|
1915
1968
|
]);
|
|
1916
1969
|
const p = testProvider({ ...retryCfg, retryAttempts: 4 });
|
|
1917
1970
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -1950,6 +2003,43 @@ test("streamed-body silence retries and returns the retry's complete output", as
|
|
|
1950
2003
|
mock.restoreAll();
|
|
1951
2004
|
});
|
|
1952
2005
|
|
|
2006
|
+
test("an Undici stream termination retries and returns the retry's complete output", async () => {
|
|
2007
|
+
let calls = 0;
|
|
2008
|
+
mock.method(globalThis, "fetch", async () => {
|
|
2009
|
+
calls++;
|
|
2010
|
+
if (calls === 1) {
|
|
2011
|
+
return new Response(new ReadableStream({
|
|
2012
|
+
start(controller) {
|
|
2013
|
+
controller.enqueue(new TextEncoder().encode(
|
|
2014
|
+
'data: {"id":"terminated","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
2015
|
+
));
|
|
2016
|
+
const socket = Object.assign(new Error("other side closed"), { code: "UND_ERR_SOCKET" });
|
|
2017
|
+
controller.error(new TypeError("terminated", { cause: socket }));
|
|
2018
|
+
},
|
|
2019
|
+
}), { status: 200 });
|
|
2020
|
+
}
|
|
2021
|
+
return new Response(sseStream([
|
|
2022
|
+
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
2023
|
+
]), { status: 200 });
|
|
2024
|
+
});
|
|
2025
|
+
const p = testProvider({
|
|
2026
|
+
model: "m",
|
|
2027
|
+
url: "http://x/v1/chat/completions",
|
|
2028
|
+
fetchTimeoutMs: 5000,
|
|
2029
|
+
streamIdleTimeoutMs: 0,
|
|
2030
|
+
temperature: 0.2,
|
|
2031
|
+
repeatPenalty: 1.15,
|
|
2032
|
+
reasoning: { mode: "off", budget: null },
|
|
2033
|
+
retryAttempts: 1,
|
|
2034
|
+
source: "provider:test",
|
|
2035
|
+
});
|
|
2036
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
2037
|
+
assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the terminated partial");
|
|
2038
|
+
assert.equal(calls, 2, "the terminated stream retried once and the retry succeeded");
|
|
2039
|
+
assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
|
|
2040
|
+
mock.restoreAll();
|
|
2041
|
+
});
|
|
2042
|
+
|
|
1953
2043
|
test("streamed-body silence does not replay when retries are disabled", async () => {
|
|
1954
2044
|
let calls = 0;
|
|
1955
2045
|
mock.method(globalThis, "fetch", async () => {
|
|
@@ -2230,13 +2320,13 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
2230
2320
|
|
|
2231
2321
|
test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thinking parameter", async () => {
|
|
2232
2322
|
// N>0 → enabled with budget_tokens
|
|
2233
|
-
const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "
|
|
2323
|
+
const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 4096 }, reasoningStyle: "anthropic" });
|
|
2234
2324
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2235
2325
|
await capped.generate({ workerId: "r", messages: [] });
|
|
2236
2326
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
|
|
2237
2327
|
|
|
2238
2328
|
mock.restoreAll();
|
|
2239
|
-
const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "
|
|
2329
|
+
const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 2048 }, reasoningStyle: "anthropic" });
|
|
2240
2330
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2241
2331
|
await unbudgeted.generate({ workerId: "r", messages: [] });
|
|
2242
2332
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
|
|
@@ -2249,11 +2339,11 @@ test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thi
|
|
|
2249
2339
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
|
|
2250
2340
|
|
|
2251
2341
|
mock.restoreAll();
|
|
2252
|
-
//
|
|
2342
|
+
// Adaptive is explicit, so the provider—not omission—owns the posture.
|
|
2253
2343
|
const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
|
|
2254
2344
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2255
2345
|
await adaptive.generate({ workerId: "r", messages: [] });
|
|
2256
|
-
assert.
|
|
2346
|
+
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "adaptive" });
|
|
2257
2347
|
});
|
|
2258
2348
|
|
|
2259
2349
|
// — non-streaming transport (streaming:false) —
|
|
@@ -2476,6 +2566,10 @@ test("native request projections compose reasoning visibility, affinity, and sys
|
|
|
2476
2566
|
reasoningResponseProviderOptions: {
|
|
2477
2567
|
google: { thinkingConfig: { includeThoughts: true } },
|
|
2478
2568
|
},
|
|
2569
|
+
adaptiveReasoning: "provider-default",
|
|
2570
|
+
adaptiveReasoningProviderOptions: {
|
|
2571
|
+
google: { thinkingConfig: { thinkingBudget: -1 } },
|
|
2572
|
+
},
|
|
2479
2573
|
systemCacheProviderOptions: {
|
|
2480
2574
|
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
2481
2575
|
},
|
|
@@ -2490,7 +2584,7 @@ test("native request projections compose reasoning visibility, affinity, and sys
|
|
|
2490
2584
|
});
|
|
2491
2585
|
|
|
2492
2586
|
assert.deepEqual(request?.providerOptions, {
|
|
2493
|
-
google: { thinkingConfig: { includeThoughts: true } },
|
|
2587
|
+
google: { thinkingConfig: { includeThoughts: true, thinkingBudget: -1 } },
|
|
2494
2588
|
openai: { promptCacheKey: "worker-native" },
|
|
2495
2589
|
});
|
|
2496
2590
|
assert.deepEqual(request?.prompt, [
|
|
@@ -2532,13 +2626,13 @@ test("native AI SDK reasoning turns on without an operator token budget", async
|
|
|
2532
2626
|
fetchTimeoutMs: 5000,
|
|
2533
2627
|
temperature: 0.2,
|
|
2534
2628
|
repeatPenalty: 1.15,
|
|
2535
|
-
reasoning: { mode: "
|
|
2629
|
+
reasoning: { mode: "high", budget: null },
|
|
2536
2630
|
retryAttempts: 0,
|
|
2537
2631
|
streaming: false,
|
|
2538
2632
|
});
|
|
2539
2633
|
const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
|
|
2540
2634
|
|
|
2541
|
-
assert.equal(request?.reasoning, "
|
|
2635
|
+
assert.equal(request?.reasoning, "high");
|
|
2542
2636
|
assert.equal(response.assistant.reasoning, "consider");
|
|
2543
2637
|
});
|
|
2544
2638
|
|
|
@@ -2577,13 +2671,93 @@ test("native additive-reasoning adapters preserve one total output budget", asyn
|
|
|
2577
2671
|
fetchTimeoutMs: 5000,
|
|
2578
2672
|
temperature: 0.2,
|
|
2579
2673
|
repeatPenalty: 1.15,
|
|
2580
|
-
reasoning: { mode: "
|
|
2674
|
+
reasoning: { mode: "high", budget: 2048 },
|
|
2581
2675
|
retryAttempts: 0,
|
|
2582
2676
|
streaming: false,
|
|
2583
2677
|
});
|
|
2584
2678
|
await p.generate({ workerId: `worker-${provider}`, messages: [], maxOutputTokens: 1500 });
|
|
2585
2679
|
assert.equal(request?.maxOutputTokens, 1);
|
|
2586
|
-
assert.equal(request?.reasoning, "
|
|
2680
|
+
assert.equal(request?.reasoning, "high", "the fixed policy remains distinct from its independent token budget");
|
|
2587
2681
|
assert.deepEqual(request?.providerOptions, expected);
|
|
2588
2682
|
}
|
|
2589
2683
|
});
|
|
2684
|
+
|
|
2685
|
+
test("native additive-reasoning adapters derive a bounded manual allowance when the model has no adaptive mode", async () => {
|
|
2686
|
+
for (const [provider, expected] of [
|
|
2687
|
+
["anthropic", { anthropic: { thinking: { type: "enabled", budgetTokens: 1024 } } }],
|
|
2688
|
+
["bedrock", { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: 1024 } } }],
|
|
2689
|
+
] as const) {
|
|
2690
|
+
let request: Record<string, unknown> | undefined;
|
|
2691
|
+
const languageModel = {
|
|
2692
|
+
specificationVersion: "v4",
|
|
2693
|
+
provider: `native.${provider}`,
|
|
2694
|
+
modelId: "native-additive",
|
|
2695
|
+
supportedUrls: {},
|
|
2696
|
+
doGenerate: async (options: Record<string, unknown>) => {
|
|
2697
|
+
request = options;
|
|
2698
|
+
return {
|
|
2699
|
+
content: [{ type: "text", text: "ok" }],
|
|
2700
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
2701
|
+
usage: {
|
|
2702
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
2703
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
2704
|
+
},
|
|
2705
|
+
response: { id: "response", modelId: "native-additive" },
|
|
2706
|
+
warnings: [],
|
|
2707
|
+
};
|
|
2708
|
+
},
|
|
2709
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2710
|
+
} as unknown as LanguageModel;
|
|
2711
|
+
const p = testProvider({
|
|
2712
|
+
model: "native-additive",
|
|
2713
|
+
languageModel,
|
|
2714
|
+
additiveReasoningProvider: provider,
|
|
2715
|
+
outputBudget: 1500,
|
|
2716
|
+
reasoningBudget: null,
|
|
2717
|
+
fetchTimeoutMs: 5000,
|
|
2718
|
+
temperature: 0.2,
|
|
2719
|
+
repeatPenalty: 1.15,
|
|
2720
|
+
reasoning: { mode: "adaptive", budget: null },
|
|
2721
|
+
retryAttempts: 0,
|
|
2722
|
+
streaming: false,
|
|
2723
|
+
});
|
|
2724
|
+
await p.generate({ workerId: `worker-${provider}`, messages: [] });
|
|
2725
|
+
assert.equal(request?.maxOutputTokens, 476);
|
|
2726
|
+
assert.equal(request?.reasoning, "high", "adaptive retains its documented high fallback");
|
|
2727
|
+
assert.deepEqual(request?.providerOptions, expected);
|
|
2728
|
+
}
|
|
2729
|
+
});
|
|
2730
|
+
|
|
2731
|
+
test("a manual-reasoning model rejects an envelope below its provider minimum before I/O", async () => {
|
|
2732
|
+
let calls = 0;
|
|
2733
|
+
const languageModel = {
|
|
2734
|
+
specificationVersion: "v4",
|
|
2735
|
+
provider: "native.anthropic",
|
|
2736
|
+
modelId: "manual-reasoning",
|
|
2737
|
+
supportedUrls: {},
|
|
2738
|
+
doGenerate: async () => {
|
|
2739
|
+
calls++;
|
|
2740
|
+
throw new Error("provider I/O must not begin");
|
|
2741
|
+
},
|
|
2742
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2743
|
+
} as unknown as LanguageModel;
|
|
2744
|
+
const p = testProvider({
|
|
2745
|
+
model: "manual-reasoning",
|
|
2746
|
+
languageModel,
|
|
2747
|
+
additiveReasoningProvider: "anthropic",
|
|
2748
|
+
outputBudget: 1024,
|
|
2749
|
+
reasoningBudget: null,
|
|
2750
|
+
fetchTimeoutMs: 5000,
|
|
2751
|
+
temperature: 0.2,
|
|
2752
|
+
repeatPenalty: 1.15,
|
|
2753
|
+
reasoning: { mode: "adaptive", budget: null },
|
|
2754
|
+
retryAttempts: 0,
|
|
2755
|
+
streaming: false,
|
|
2756
|
+
});
|
|
2757
|
+
|
|
2758
|
+
await assert.rejects(
|
|
2759
|
+
p.generate({ workerId: "worker", messages: [] }),
|
|
2760
|
+
/total output budget must exceed the provider's 1024-token minimum reasoning allowance/,
|
|
2761
|
+
);
|
|
2762
|
+
assert.equal(calls, 0);
|
|
2763
|
+
});
|