@plurnk/plurnk-providers 1.7.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +25 -13
- package/README.md +8 -1
- package/SPEC.md +103 -15
- package/dist/AiSdkProvider.d.ts +7 -4
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +150 -53
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +2 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +6 -1
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -0
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +3 -0
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +11 -10
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +16 -8
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +33 -6
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +4 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +94 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +2 -0
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +5 -4
- package/dist/cost.js.map +1 -1
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +13 -2
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +3 -2
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +11 -4
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +10 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +6 -3
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +1 -1
- package/dist/notices.d.ts.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +163 -19
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +10 -2
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +10 -1
- package/dist/types.js.map +1 -1
- package/package.json +9 -9
- package/src/AiSdkProvider.test.ts +214 -34
- package/src/AiSdkProvider.ts +181 -56
- package/src/Mock.test.ts +6 -1
- package/src/Mock.ts +6 -1
- package/src/Pool.test.ts +1 -0
- package/src/Pool.ts +5 -0
- package/src/ProviderRegistry.test.ts +27 -14
- package/src/ProviderRegistry.ts +19 -10
- package/src/accounting.test.ts +30 -2
- package/src/accounting.ts +16 -8
- package/src/aiSdkTransport.test.ts +3 -0
- package/src/aiSdkTransport.ts +38 -8
- package/src/catalogProvider.test.ts +151 -19
- package/src/catalogProvider.ts +125 -3
- package/src/compatibleProvider.test.ts +13 -10
- package/src/compatibleProvider.ts +2 -0
- package/src/cost.ts +5 -4
- package/src/discover.test.ts +27 -0
- package/src/discover.ts +20 -3
- package/src/env.test.ts +23 -8
- package/src/env.ts +18 -9
- package/src/errors.test.ts +2 -2
- package/src/index.ts +17 -8
- package/src/notices.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/providerDefaults.test.ts +50 -0
- package/src/sdkModels.test.ts +142 -8
- package/src/sdkModels.ts +201 -19
- package/src/types.ts +22 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import AiSdkProvider, {
|
|
3
|
+
import AiSdkProvider, { type AiSdkProviderConfig } from "./AiSdkProvider.ts";
|
|
4
4
|
import { ProviderError } from "./errors.ts";
|
|
5
5
|
import { providerCostNormalizer } from "./accounting.ts";
|
|
6
6
|
import type { LanguageModel } from "ai";
|
|
@@ -60,9 +60,18 @@ const sseStream = (chunks: unknown[]) => {
|
|
|
60
60
|
|
|
61
61
|
const installFetch = (chunks: unknown[]) => {
|
|
62
62
|
const calls: { url: string; init: RequestInit }[] = [];
|
|
63
|
+
const completeChunks = chunks.some((value) => {
|
|
64
|
+
const choices = (value as { choices?: Array<{ finish_reason?: unknown }> }).choices;
|
|
65
|
+
return choices?.some(({ finish_reason }) => finish_reason != null) ?? false;
|
|
66
|
+
})
|
|
67
|
+
? chunks
|
|
68
|
+
: [...chunks, { choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }];
|
|
63
69
|
mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
|
|
64
70
|
calls.push({ url, init });
|
|
65
|
-
return new Response(sseStream(
|
|
71
|
+
return new Response(sseStream(completeChunks), {
|
|
72
|
+
status: 200,
|
|
73
|
+
headers: { "content-type": "text/event-stream" },
|
|
74
|
+
});
|
|
66
75
|
});
|
|
67
76
|
return calls;
|
|
68
77
|
};
|
|
@@ -296,14 +305,6 @@ const flush = () => new Promise<void>((r) => setImmediate(r));
|
|
|
296
305
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
297
306
|
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
298
307
|
|
|
299
|
-
test("effortFromBudget: maps budget to tiers", () => {
|
|
300
|
-
assert.equal(effortFromBudget(1), "low");
|
|
301
|
-
assert.equal(effortFromBudget(1000), "low");
|
|
302
|
-
assert.equal(effortFromBudget(1001), "medium");
|
|
303
|
-
assert.equal(effortFromBudget(4000), "medium");
|
|
304
|
-
assert.equal(effortFromBudget(4001), "high");
|
|
305
|
-
});
|
|
306
|
-
|
|
307
308
|
test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
308
309
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
309
310
|
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
@@ -602,6 +603,58 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
602
603
|
assert.notEqual(assistantRaw, undefined);
|
|
603
604
|
});
|
|
604
605
|
|
|
606
|
+
test("an unsupported fixed reasoning policy fails before provider I/O", () => {
|
|
607
|
+
assert.throws(
|
|
608
|
+
() => testProvider({
|
|
609
|
+
...injectedBase,
|
|
610
|
+
reasoning: { mode: "medium", budget: null },
|
|
611
|
+
supportedReasoningPolicies: ["off", "adaptive", "high"],
|
|
612
|
+
}),
|
|
613
|
+
/reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
|
|
614
|
+
);
|
|
615
|
+
});
|
|
616
|
+
|
|
617
|
+
test("native SDK warnings survive as source-attributed provider Notices", async () => {
|
|
618
|
+
const usage = {
|
|
619
|
+
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
620
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
621
|
+
};
|
|
622
|
+
const languageModel = {
|
|
623
|
+
specificationVersion: "v4",
|
|
624
|
+
provider: "native.test",
|
|
625
|
+
modelId: "warning-test",
|
|
626
|
+
supportedUrls: {},
|
|
627
|
+
doGenerate: async () => ({
|
|
628
|
+
content: [{ type: "text", text: "ok" }],
|
|
629
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
630
|
+
usage,
|
|
631
|
+
response: { id: "response-warning", modelId: "warning-test" },
|
|
632
|
+
warnings: [{
|
|
633
|
+
type: "compatibility",
|
|
634
|
+
feature: "reasoning",
|
|
635
|
+
details: "reasoning low was mapped to high",
|
|
636
|
+
}],
|
|
637
|
+
}),
|
|
638
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
639
|
+
} as unknown as LanguageModel;
|
|
640
|
+
const response = await testProvider({
|
|
641
|
+
...injectedBase,
|
|
642
|
+
url: undefined,
|
|
643
|
+
model: "warning-test",
|
|
644
|
+
languageModel,
|
|
645
|
+
source: "provider:warning-test",
|
|
646
|
+
streaming: false,
|
|
647
|
+
}).generate({ workerId: "warnings", messages: [] });
|
|
648
|
+
|
|
649
|
+
assert.deepEqual(response.notices, [{
|
|
650
|
+
source: "provider:warning-test",
|
|
651
|
+
kind: "provider_warning",
|
|
652
|
+
level: "warn",
|
|
653
|
+
message: "compatibility reasoning: reasoning low was mapped to high",
|
|
654
|
+
position: null,
|
|
655
|
+
}]);
|
|
656
|
+
});
|
|
657
|
+
|
|
605
658
|
test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
|
|
606
659
|
const usage = {
|
|
607
660
|
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
@@ -965,12 +1018,18 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered le
|
|
|
965
1018
|
usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
|
|
966
1019
|
});
|
|
967
1020
|
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
968
|
-
const
|
|
1021
|
+
const reasoning: string[] = [];
|
|
1022
|
+
const response = await testProvider(config).generate({
|
|
1023
|
+
workerId: "tagged-buffer",
|
|
1024
|
+
messages: [],
|
|
1025
|
+
observeReasoning: (delta) => reasoning.push(delta),
|
|
1026
|
+
});
|
|
969
1027
|
|
|
970
1028
|
assert.equal(response.assistant.reasoning, "12345");
|
|
971
1029
|
assert.equal(response.assistant.content, "abcde");
|
|
972
1030
|
assert.equal(response.accounting[0]?.usage?.outputTokens, 10);
|
|
973
1031
|
assert.equal(response.accounting[0]?.usage?.outputTokenDetails, undefined);
|
|
1032
|
+
assert.deepEqual(reasoning, ["12345"], "buffered normalization converges on the same observer");
|
|
974
1033
|
});
|
|
975
1034
|
|
|
976
1035
|
test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
|
|
@@ -1157,21 +1216,21 @@ test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the
|
|
|
1157
1216
|
assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
|
|
1158
1217
|
});
|
|
1159
1218
|
|
|
1160
|
-
test("reasoningStyle 'effort'
|
|
1161
|
-
for (const
|
|
1162
|
-
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions",
|
|
1219
|
+
test("reasoningStyle 'effort' preserves each fixed portable effort", async () => {
|
|
1220
|
+
for (const mode of ["low", "medium", "high"] as const) {
|
|
1221
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode, budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
1163
1222
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1164
1223
|
await p.generate({ workerId: "r", messages: [] });
|
|
1165
|
-
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort,
|
|
1224
|
+
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, mode);
|
|
1166
1225
|
mock.restoreAll();
|
|
1167
1226
|
}
|
|
1168
1227
|
});
|
|
1169
1228
|
|
|
1170
|
-
test("reasoningStyle 'effort_explicit': off
|
|
1229
|
+
test("reasoningStyle 'effort_explicit': off sends none, adaptive omits, fixed effort remains exact", async () => {
|
|
1171
1230
|
// expected === null → the field must be ABSENT from the wire body. Fireworks
|
|
1172
1231
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
1173
1232
|
// Adaptive = the backend's own default posture = omission.
|
|
1174
|
-
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "
|
|
1233
|
+
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "low", budget: null }, "low"], [{ mode: "high", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "low" | "high"; budget: number | null }, string | null]>) {
|
|
1175
1234
|
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", ...(reasoning.budget === null ? {} : { outputBudget: reasoning.budget + 1, reasoningBudget: reasoning.budget }), fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
1176
1235
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1177
1236
|
await p.generate({ workerId: "r", messages: [] });
|
|
@@ -1185,9 +1244,9 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends
|
|
|
1185
1244
|
test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
|
|
1186
1245
|
const cases = [
|
|
1187
1246
|
[{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
|
|
1188
|
-
[{ mode: "adaptive", budget: null }, {}],
|
|
1189
|
-
[{ mode: "
|
|
1190
|
-
[{ mode: "
|
|
1247
|
+
[{ mode: "adaptive", budget: null }, { thinking: { type: "enabled" } }],
|
|
1248
|
+
[{ mode: "high", budget: null }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
1249
|
+
[{ mode: "high", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
1191
1250
|
] as const;
|
|
1192
1251
|
for (const [reasoning, expected] of cases) {
|
|
1193
1252
|
const p = testProvider({
|
|
@@ -1478,14 +1537,14 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
|
|
|
1478
1537
|
|
|
1479
1538
|
test("reasoningStyle 'template' carries the explicit reasoning subset and rejects an additive envelope", async () => {
|
|
1480
1539
|
const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, outputBudget: 224, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
|
|
1481
|
-
const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "
|
|
1540
|
+
const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "high", budget: 32 } });
|
|
1482
1541
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1483
1542
|
await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
|
|
1484
1543
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1485
1544
|
assert.equal(body.thinking_budget_tokens, 32);
|
|
1486
1545
|
assert.equal(body.reasoning_format, "auto");
|
|
1487
1546
|
assert.throws(
|
|
1488
|
-
() => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "
|
|
1547
|
+
() => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "high", budget: 224 } }),
|
|
1489
1548
|
/reasoningBudget must be smaller than the total outputBudget/,
|
|
1490
1549
|
);
|
|
1491
1550
|
});
|
|
@@ -1500,7 +1559,7 @@ test("reasoningStyle 'template' explicit activation uses the configured reasonin
|
|
|
1500
1559
|
fetchTimeoutMs: 5000,
|
|
1501
1560
|
temperature: 0.2,
|
|
1502
1561
|
repeatPenalty: 1.15,
|
|
1503
|
-
reasoning: { mode: "
|
|
1562
|
+
reasoning: { mode: "high", budget: 64 },
|
|
1504
1563
|
retryAttempts: 0,
|
|
1505
1564
|
reasoningStyle: "template",
|
|
1506
1565
|
});
|
|
@@ -1867,7 +1926,7 @@ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details
|
|
|
1867
1926
|
assert.equal(err.status, 429);
|
|
1868
1927
|
assert.equal(err.problem.status, 429);
|
|
1869
1928
|
assert.equal(err.problem.detail, err.message);
|
|
1870
|
-
assert.equal(err.problem.type, "https://problems.plurnk.
|
|
1929
|
+
assert.equal(err.problem.type, "https://problems.plurnk.xyz/provider/test/rate-limit");
|
|
1871
1930
|
return true;
|
|
1872
1931
|
});
|
|
1873
1932
|
});
|
|
@@ -1911,7 +1970,7 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
1911
1970
|
{ status: 409, retryAfter: 0 },
|
|
1912
1971
|
{ status: 429, retryAfter: 0 },
|
|
1913
1972
|
{ status: 503, retryAfter: 0 },
|
|
1914
|
-
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
|
|
1973
|
+
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
|
|
1915
1974
|
]);
|
|
1916
1975
|
const p = testProvider({ ...retryCfg, retryAttempts: 4 });
|
|
1917
1976
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -1950,6 +2009,43 @@ test("streamed-body silence retries and returns the retry's complete output", as
|
|
|
1950
2009
|
mock.restoreAll();
|
|
1951
2010
|
});
|
|
1952
2011
|
|
|
2012
|
+
test("an Undici stream termination retries and returns the retry's complete output", async () => {
|
|
2013
|
+
let calls = 0;
|
|
2014
|
+
mock.method(globalThis, "fetch", async () => {
|
|
2015
|
+
calls++;
|
|
2016
|
+
if (calls === 1) {
|
|
2017
|
+
return new Response(new ReadableStream({
|
|
2018
|
+
start(controller) {
|
|
2019
|
+
controller.enqueue(new TextEncoder().encode(
|
|
2020
|
+
'data: {"id":"terminated","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
2021
|
+
));
|
|
2022
|
+
const socket = Object.assign(new Error("other side closed"), { code: "UND_ERR_SOCKET" });
|
|
2023
|
+
controller.error(new TypeError("terminated", { cause: socket }));
|
|
2024
|
+
},
|
|
2025
|
+
}), { status: 200 });
|
|
2026
|
+
}
|
|
2027
|
+
return new Response(sseStream([
|
|
2028
|
+
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
2029
|
+
]), { status: 200 });
|
|
2030
|
+
});
|
|
2031
|
+
const p = testProvider({
|
|
2032
|
+
model: "m",
|
|
2033
|
+
url: "http://x/v1/chat/completions",
|
|
2034
|
+
fetchTimeoutMs: 5000,
|
|
2035
|
+
streamIdleTimeoutMs: 0,
|
|
2036
|
+
temperature: 0.2,
|
|
2037
|
+
repeatPenalty: 1.15,
|
|
2038
|
+
reasoning: { mode: "off", budget: null },
|
|
2039
|
+
retryAttempts: 1,
|
|
2040
|
+
source: "provider:test",
|
|
2041
|
+
});
|
|
2042
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
2043
|
+
assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the terminated partial");
|
|
2044
|
+
assert.equal(calls, 2, "the terminated stream retried once and the retry succeeded");
|
|
2045
|
+
assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
|
|
2046
|
+
mock.restoreAll();
|
|
2047
|
+
});
|
|
2048
|
+
|
|
1953
2049
|
test("streamed-body silence does not replay when retries are disabled", async () => {
|
|
1954
2050
|
let calls = 0;
|
|
1955
2051
|
mock.method(globalThis, "fetch", async () => {
|
|
@@ -2230,13 +2326,13 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
2230
2326
|
|
|
2231
2327
|
test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thinking parameter", async () => {
|
|
2232
2328
|
// N>0 → enabled with budget_tokens
|
|
2233
|
-
const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "
|
|
2329
|
+
const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 4096 }, reasoningStyle: "anthropic" });
|
|
2234
2330
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2235
2331
|
await capped.generate({ workerId: "r", messages: [] });
|
|
2236
2332
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
|
|
2237
2333
|
|
|
2238
2334
|
mock.restoreAll();
|
|
2239
|
-
const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "
|
|
2335
|
+
const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 2048 }, reasoningStyle: "anthropic" });
|
|
2240
2336
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2241
2337
|
await unbudgeted.generate({ workerId: "r", messages: [] });
|
|
2242
2338
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
|
|
@@ -2249,11 +2345,11 @@ test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thi
|
|
|
2249
2345
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
|
|
2250
2346
|
|
|
2251
2347
|
mock.restoreAll();
|
|
2252
|
-
//
|
|
2348
|
+
// Adaptive is explicit, so the provider—not omission—owns the posture.
|
|
2253
2349
|
const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
|
|
2254
2350
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2255
2351
|
await adaptive.generate({ workerId: "r", messages: [] });
|
|
2256
|
-
assert.
|
|
2352
|
+
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "adaptive" });
|
|
2257
2353
|
});
|
|
2258
2354
|
|
|
2259
2355
|
// — non-streaming transport (streaming:false) —
|
|
@@ -2476,6 +2572,10 @@ test("native request projections compose reasoning visibility, affinity, and sys
|
|
|
2476
2572
|
reasoningResponseProviderOptions: {
|
|
2477
2573
|
google: { thinkingConfig: { includeThoughts: true } },
|
|
2478
2574
|
},
|
|
2575
|
+
adaptiveReasoning: "provider-default",
|
|
2576
|
+
adaptiveReasoningProviderOptions: {
|
|
2577
|
+
google: { thinkingConfig: { thinkingBudget: -1 } },
|
|
2578
|
+
},
|
|
2479
2579
|
systemCacheProviderOptions: {
|
|
2480
2580
|
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
2481
2581
|
},
|
|
@@ -2490,7 +2590,7 @@ test("native request projections compose reasoning visibility, affinity, and sys
|
|
|
2490
2590
|
});
|
|
2491
2591
|
|
|
2492
2592
|
assert.deepEqual(request?.providerOptions, {
|
|
2493
|
-
google: { thinkingConfig: { includeThoughts: true } },
|
|
2593
|
+
google: { thinkingConfig: { includeThoughts: true, thinkingBudget: -1 } },
|
|
2494
2594
|
openai: { promptCacheKey: "worker-native" },
|
|
2495
2595
|
});
|
|
2496
2596
|
assert.deepEqual(request?.prompt, [
|
|
@@ -2532,13 +2632,13 @@ test("native AI SDK reasoning turns on without an operator token budget", async
|
|
|
2532
2632
|
fetchTimeoutMs: 5000,
|
|
2533
2633
|
temperature: 0.2,
|
|
2534
2634
|
repeatPenalty: 1.15,
|
|
2535
|
-
reasoning: { mode: "
|
|
2635
|
+
reasoning: { mode: "high", budget: null },
|
|
2536
2636
|
retryAttempts: 0,
|
|
2537
2637
|
streaming: false,
|
|
2538
2638
|
});
|
|
2539
2639
|
const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
|
|
2540
2640
|
|
|
2541
|
-
assert.equal(request?.reasoning, "
|
|
2641
|
+
assert.equal(request?.reasoning, "high");
|
|
2542
2642
|
assert.equal(response.assistant.reasoning, "consider");
|
|
2543
2643
|
});
|
|
2544
2644
|
|
|
@@ -2577,13 +2677,93 @@ test("native additive-reasoning adapters preserve one total output budget", asyn
|
|
|
2577
2677
|
fetchTimeoutMs: 5000,
|
|
2578
2678
|
temperature: 0.2,
|
|
2579
2679
|
repeatPenalty: 1.15,
|
|
2580
|
-
reasoning: { mode: "
|
|
2680
|
+
reasoning: { mode: "high", budget: 2048 },
|
|
2581
2681
|
retryAttempts: 0,
|
|
2582
2682
|
streaming: false,
|
|
2583
2683
|
});
|
|
2584
2684
|
await p.generate({ workerId: `worker-${provider}`, messages: [], maxOutputTokens: 1500 });
|
|
2585
2685
|
assert.equal(request?.maxOutputTokens, 1);
|
|
2586
|
-
assert.equal(request?.reasoning, "
|
|
2686
|
+
assert.equal(request?.reasoning, "high", "the fixed policy remains distinct from its independent token budget");
|
|
2587
2687
|
assert.deepEqual(request?.providerOptions, expected);
|
|
2588
2688
|
}
|
|
2589
2689
|
});
|
|
2690
|
+
|
|
2691
|
+
test("native additive-reasoning adapters derive a bounded manual allowance when the model has no adaptive mode", async () => {
|
|
2692
|
+
for (const [provider, expected] of [
|
|
2693
|
+
["anthropic", { anthropic: { thinking: { type: "enabled", budgetTokens: 1024 } } }],
|
|
2694
|
+
["bedrock", { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: 1024 } } }],
|
|
2695
|
+
] as const) {
|
|
2696
|
+
let request: Record<string, unknown> | undefined;
|
|
2697
|
+
const languageModel = {
|
|
2698
|
+
specificationVersion: "v4",
|
|
2699
|
+
provider: `native.${provider}`,
|
|
2700
|
+
modelId: "native-additive",
|
|
2701
|
+
supportedUrls: {},
|
|
2702
|
+
doGenerate: async (options: Record<string, unknown>) => {
|
|
2703
|
+
request = options;
|
|
2704
|
+
return {
|
|
2705
|
+
content: [{ type: "text", text: "ok" }],
|
|
2706
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
2707
|
+
usage: {
|
|
2708
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
2709
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
2710
|
+
},
|
|
2711
|
+
response: { id: "response", modelId: "native-additive" },
|
|
2712
|
+
warnings: [],
|
|
2713
|
+
};
|
|
2714
|
+
},
|
|
2715
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2716
|
+
} as unknown as LanguageModel;
|
|
2717
|
+
const p = testProvider({
|
|
2718
|
+
model: "native-additive",
|
|
2719
|
+
languageModel,
|
|
2720
|
+
additiveReasoningProvider: provider,
|
|
2721
|
+
outputBudget: 1500,
|
|
2722
|
+
reasoningBudget: null,
|
|
2723
|
+
fetchTimeoutMs: 5000,
|
|
2724
|
+
temperature: 0.2,
|
|
2725
|
+
repeatPenalty: 1.15,
|
|
2726
|
+
reasoning: { mode: "adaptive", budget: null },
|
|
2727
|
+
retryAttempts: 0,
|
|
2728
|
+
streaming: false,
|
|
2729
|
+
});
|
|
2730
|
+
await p.generate({ workerId: `worker-${provider}`, messages: [] });
|
|
2731
|
+
assert.equal(request?.maxOutputTokens, 476);
|
|
2732
|
+
assert.equal(request?.reasoning, "high", "adaptive retains its documented high fallback");
|
|
2733
|
+
assert.deepEqual(request?.providerOptions, expected);
|
|
2734
|
+
}
|
|
2735
|
+
});
|
|
2736
|
+
|
|
2737
|
+
test("a manual-reasoning model rejects an envelope below its provider minimum before I/O", async () => {
|
|
2738
|
+
let calls = 0;
|
|
2739
|
+
const languageModel = {
|
|
2740
|
+
specificationVersion: "v4",
|
|
2741
|
+
provider: "native.anthropic",
|
|
2742
|
+
modelId: "manual-reasoning",
|
|
2743
|
+
supportedUrls: {},
|
|
2744
|
+
doGenerate: async () => {
|
|
2745
|
+
calls++;
|
|
2746
|
+
throw new Error("provider I/O must not begin");
|
|
2747
|
+
},
|
|
2748
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2749
|
+
} as unknown as LanguageModel;
|
|
2750
|
+
const p = testProvider({
|
|
2751
|
+
model: "manual-reasoning",
|
|
2752
|
+
languageModel,
|
|
2753
|
+
additiveReasoningProvider: "anthropic",
|
|
2754
|
+
outputBudget: 1024,
|
|
2755
|
+
reasoningBudget: null,
|
|
2756
|
+
fetchTimeoutMs: 5000,
|
|
2757
|
+
temperature: 0.2,
|
|
2758
|
+
repeatPenalty: 1.15,
|
|
2759
|
+
reasoning: { mode: "adaptive", budget: null },
|
|
2760
|
+
retryAttempts: 0,
|
|
2761
|
+
streaming: false,
|
|
2762
|
+
});
|
|
2763
|
+
|
|
2764
|
+
await assert.rejects(
|
|
2765
|
+
p.generate({ workerId: "worker", messages: [] }),
|
|
2766
|
+
/total output budget must exceed the provider's 1024-token minimum reasoning allowance/,
|
|
2767
|
+
);
|
|
2768
|
+
assert.equal(calls, 0);
|
|
2769
|
+
});
|