@plurnk/plurnk-providers 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/.env.defaults +25 -13
  2. package/README.md +8 -1
  3. package/SPEC.md +93 -15
  4. package/dist/AiSdkProvider.d.ts +6 -3
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +108 -51
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +1 -0
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +2 -0
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -0
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +3 -0
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts.map +1 -1
  17. package/dist/ProviderRegistry.js +11 -10
  18. package/dist/ProviderRegistry.js.map +1 -1
  19. package/dist/accounting.d.ts.map +1 -1
  20. package/dist/accounting.js +7 -6
  21. package/dist/accounting.js.map +1 -1
  22. package/dist/aiSdkTransport.d.ts +2 -1
  23. package/dist/aiSdkTransport.d.ts.map +1 -1
  24. package/dist/aiSdkTransport.js +29 -6
  25. package/dist/aiSdkTransport.js.map +1 -1
  26. package/dist/catalogProvider.d.ts +4 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +94 -3
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +2 -0
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/cost.d.ts.map +1 -1
  34. package/dist/cost.js +5 -4
  35. package/dist/cost.js.map +1 -1
  36. package/dist/discover.d.ts +2 -0
  37. package/dist/discover.d.ts.map +1 -1
  38. package/dist/discover.js +13 -2
  39. package/dist/discover.js.map +1 -1
  40. package/dist/env.d.ts +3 -2
  41. package/dist/env.d.ts.map +1 -1
  42. package/dist/env.js +11 -4
  43. package/dist/env.js.map +1 -1
  44. package/dist/index.d.ts +9 -5
  45. package/dist/index.d.ts.map +1 -1
  46. package/dist/index.js +6 -3
  47. package/dist/index.js.map +1 -1
  48. package/dist/notices.d.ts +1 -1
  49. package/dist/notices.d.ts.map +1 -1
  50. package/dist/openai.d.ts +1 -1
  51. package/dist/openai.d.ts.map +1 -1
  52. package/dist/openai.js +1 -1
  53. package/dist/openai.js.map +1 -1
  54. package/dist/sdkModels.d.ts +2 -0
  55. package/dist/sdkModels.d.ts.map +1 -1
  56. package/dist/sdkModels.js +163 -19
  57. package/dist/sdkModels.js.map +1 -1
  58. package/dist/types.d.ts +8 -2
  59. package/dist/types.d.ts.map +1 -1
  60. package/dist/types.js +10 -1
  61. package/dist/types.js.map +1 -1
  62. package/package.json +9 -9
  63. package/src/AiSdkProvider.test.ts +206 -32
  64. package/src/AiSdkProvider.ts +140 -54
  65. package/src/Mock.ts +2 -0
  66. package/src/Pool.test.ts +1 -0
  67. package/src/Pool.ts +5 -0
  68. package/src/ProviderRegistry.test.ts +27 -14
  69. package/src/ProviderRegistry.ts +19 -10
  70. package/src/accounting.test.ts +6 -2
  71. package/src/accounting.ts +7 -6
  72. package/src/aiSdkTransport.ts +32 -7
  73. package/src/catalogProvider.test.ts +151 -19
  74. package/src/catalogProvider.ts +125 -3
  75. package/src/compatibleProvider.test.ts +13 -10
  76. package/src/compatibleProvider.ts +2 -0
  77. package/src/cost.ts +5 -4
  78. package/src/discover.test.ts +27 -0
  79. package/src/discover.ts +20 -3
  80. package/src/env.test.ts +23 -8
  81. package/src/env.ts +17 -8
  82. package/src/index.ts +16 -8
  83. package/src/notices.ts +1 -1
  84. package/src/openai.ts +1 -1
  85. package/src/providerDefaults.test.ts +50 -0
  86. package/src/sdkModels.test.ts +142 -8
  87. package/src/sdkModels.ts +201 -19
  88. package/src/types.ts +16 -0
@@ -1,6 +1,6 @@
1
1
  import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import AiSdkProvider, { effortFromBudget, type AiSdkProviderConfig } from "./AiSdkProvider.ts";
3
+ import AiSdkProvider, { type AiSdkProviderConfig } from "./AiSdkProvider.ts";
4
4
  import { ProviderError } from "./errors.ts";
5
5
  import { providerCostNormalizer } from "./accounting.ts";
6
6
  import type { LanguageModel } from "ai";
@@ -60,9 +60,18 @@ const sseStream = (chunks: unknown[]) => {
60
60
 
61
61
  const installFetch = (chunks: unknown[]) => {
62
62
  const calls: { url: string; init: RequestInit }[] = [];
63
+ const completeChunks = chunks.some((value) => {
64
+ const choices = (value as { choices?: Array<{ finish_reason?: unknown }> }).choices;
65
+ return choices?.some(({ finish_reason }) => finish_reason != null) ?? false;
66
+ })
67
+ ? chunks
68
+ : [...chunks, { choices: [{ index: 0, delta: {}, finish_reason: "stop" }] }];
63
69
  mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
64
70
  calls.push({ url, init });
65
- return new Response(sseStream(chunks), { status: 200 });
71
+ return new Response(sseStream(completeChunks), {
72
+ status: 200,
73
+ headers: { "content-type": "text/event-stream" },
74
+ });
66
75
  });
67
76
  return calls;
68
77
  };
@@ -296,14 +305,6 @@ const flush = () => new Promise<void>((r) => setImmediate(r));
296
305
  import { resetEmittedWarnings } from "./warnings.ts";
297
306
  test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
298
307
 
299
- test("effortFromBudget: maps budget to tiers", () => {
300
- assert.equal(effortFromBudget(1), "low");
301
- assert.equal(effortFromBudget(1000), "low");
302
- assert.equal(effortFromBudget(1001), "medium");
303
- assert.equal(effortFromBudget(4000), "medium");
304
- assert.equal(effortFromBudget(4001), "high");
305
- });
306
-
307
308
  test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
308
309
  const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
309
310
  const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
@@ -602,6 +603,58 @@ test("generate maps a streamed response into ProviderResponse", async () => {
602
603
  assert.notEqual(assistantRaw, undefined);
603
604
  });
604
605
 
606
+ test("an unsupported fixed reasoning policy fails before provider I/O", () => {
607
+ assert.throws(
608
+ () => testProvider({
609
+ ...injectedBase,
610
+ reasoning: { mode: "medium", budget: null },
611
+ supportedReasoningPolicies: ["off", "adaptive", "high"],
612
+ }),
613
+ /reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
614
+ );
615
+ });
616
+
617
+ test("native SDK warnings survive as source-attributed provider Notices", async () => {
618
+ const usage = {
619
+ inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
620
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
621
+ };
622
+ const languageModel = {
623
+ specificationVersion: "v4",
624
+ provider: "native.test",
625
+ modelId: "warning-test",
626
+ supportedUrls: {},
627
+ doGenerate: async () => ({
628
+ content: [{ type: "text", text: "ok" }],
629
+ finishReason: { unified: "stop", raw: "stop" },
630
+ usage,
631
+ response: { id: "response-warning", modelId: "warning-test" },
632
+ warnings: [{
633
+ type: "compatibility",
634
+ feature: "reasoning",
635
+ details: "reasoning low was mapped to high",
636
+ }],
637
+ }),
638
+ doStream: async () => { throw new Error("streaming is not under test"); },
639
+ } as unknown as LanguageModel;
640
+ const response = await testProvider({
641
+ ...injectedBase,
642
+ url: undefined,
643
+ model: "warning-test",
644
+ languageModel,
645
+ source: "provider:warning-test",
646
+ streaming: false,
647
+ }).generate({ workerId: "warnings", messages: [] });
648
+
649
+ assert.deepEqual(response.notices, [{
650
+ source: "provider:warning-test",
651
+ kind: "provider_warning",
652
+ level: "warn",
653
+ message: "compatibility reasoning: reasoning low was mapped to high",
654
+ position: null,
655
+ }]);
656
+ });
657
+
605
658
  test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
606
659
  const usage = {
607
660
  inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
@@ -1157,21 +1210,21 @@ test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the
1157
1210
  assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
1158
1211
  });
1159
1212
 
1160
- test("reasoningStyle 'effort' enables at the portable default without inventing a budget", async () => {
1161
- for (const [budget, expected] of [[null, "medium"], [5000, "high"]] as const) {
1162
- const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", ...(budget === null ? {} : { outputBudget: budget + 1, reasoningBudget: budget }), fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget }, retryAttempts: 0, reasoningStyle: "effort" });
1213
+ test("reasoningStyle 'effort' preserves each fixed portable effort", async () => {
1214
+ for (const mode of ["low", "medium", "high"] as const) {
1215
+ const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode, budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
1163
1216
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1164
1217
  await p.generate({ workerId: "r", messages: [] });
1165
- assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, expected);
1218
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, mode);
1166
1219
  mock.restoreAll();
1167
1220
  }
1168
1221
  });
1169
1222
 
1170
- test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
1223
+ test("reasoningStyle 'effort_explicit': off sends none, adaptive omits, fixed effort remains exact", async () => {
1171
1224
  // expected === null → the field must be ABSENT from the wire body. Fireworks
1172
1225
  // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
1173
1226
  // Adaptive = the backend's own default posture = omission.
1174
- for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: null }, "medium"], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
1227
+ for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "low", budget: null }, "low"], [{ mode: "high", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "low" | "high"; budget: number | null }, string | null]>) {
1175
1228
  const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", ...(reasoning.budget === null ? {} : { outputBudget: reasoning.budget + 1, reasoningBudget: reasoning.budget }), fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
1176
1229
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1177
1230
  await p.generate({ workerId: "r", messages: [] });
@@ -1185,9 +1238,9 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends
1185
1238
  test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
1186
1239
  const cases = [
1187
1240
  [{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
1188
- [{ mode: "adaptive", budget: null }, {}],
1189
- [{ mode: "on", budget: null }, { thinking: { type: "enabled" } }],
1190
- [{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
1241
+ [{ mode: "adaptive", budget: null }, { thinking: { type: "enabled" } }],
1242
+ [{ mode: "high", budget: null }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
1243
+ [{ mode: "high", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
1191
1244
  ] as const;
1192
1245
  for (const [reasoning, expected] of cases) {
1193
1246
  const p = testProvider({
@@ -1478,14 +1531,14 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
1478
1531
 
1479
1532
  test("reasoningStyle 'template' carries the explicit reasoning subset and rejects an additive envelope", async () => {
1480
1533
  const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, outputBudget: 224, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
1481
- const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "on", budget: 32 } });
1534
+ const p = testProvider({ ...base, reasoningBudget: 32, reasoning: { mode: "high", budget: 32 } });
1482
1535
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1483
1536
  await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
1484
1537
  const body = JSON.parse(calls[0].init.body as string);
1485
1538
  assert.equal(body.thinking_budget_tokens, 32);
1486
1539
  assert.equal(body.reasoning_format, "auto");
1487
1540
  assert.throws(
1488
- () => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "on", budget: 224 } }),
1541
+ () => testProvider({ ...base, reasoningBudget: 224, reasoning: { mode: "high", budget: 224 } }),
1489
1542
  /reasoningBudget must be smaller than the total outputBudget/,
1490
1543
  );
1491
1544
  });
@@ -1500,7 +1553,7 @@ test("reasoningStyle 'template' explicit activation uses the configured reasonin
1500
1553
  fetchTimeoutMs: 5000,
1501
1554
  temperature: 0.2,
1502
1555
  repeatPenalty: 1.15,
1503
- reasoning: { mode: "on", budget: 64 },
1556
+ reasoning: { mode: "high", budget: 64 },
1504
1557
  retryAttempts: 0,
1505
1558
  reasoningStyle: "template",
1506
1559
  });
@@ -1911,7 +1964,7 @@ test("retry: a transient failure retries and a later success resolves", async ()
1911
1964
  { status: 409, retryAfter: 0 },
1912
1965
  { status: 429, retryAfter: 0 },
1913
1966
  { status: 503, retryAfter: 0 },
1914
- { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
1967
+ { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
1915
1968
  ]);
1916
1969
  const p = testProvider({ ...retryCfg, retryAttempts: 4 });
1917
1970
  const res = await p.generate({ workerId: "r", messages: [] });
@@ -1950,6 +2003,43 @@ test("streamed-body silence retries and returns the retry's complete output", as
1950
2003
  mock.restoreAll();
1951
2004
  });
1952
2005
 
2006
+ test("an Undici stream termination retries and returns the retry's complete output", async () => {
2007
+ let calls = 0;
2008
+ mock.method(globalThis, "fetch", async () => {
2009
+ calls++;
2010
+ if (calls === 1) {
2011
+ return new Response(new ReadableStream({
2012
+ start(controller) {
2013
+ controller.enqueue(new TextEncoder().encode(
2014
+ 'data: {"id":"terminated","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
2015
+ ));
2016
+ const socket = Object.assign(new Error("other side closed"), { code: "UND_ERR_SOCKET" });
2017
+ controller.error(new TypeError("terminated", { cause: socket }));
2018
+ },
2019
+ }), { status: 200 });
2020
+ }
2021
+ return new Response(sseStream([
2022
+ { choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
2023
+ ]), { status: 200 });
2024
+ });
2025
+ const p = testProvider({
2026
+ model: "m",
2027
+ url: "http://x/v1/chat/completions",
2028
+ fetchTimeoutMs: 5000,
2029
+ streamIdleTimeoutMs: 0,
2030
+ temperature: 0.2,
2031
+ repeatPenalty: 1.15,
2032
+ reasoning: { mode: "off", budget: null },
2033
+ retryAttempts: 1,
2034
+ source: "provider:test",
2035
+ });
2036
+ const result = await p.generate({ workerId: "r", messages: [] });
2037
+ assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the terminated partial");
2038
+ assert.equal(calls, 2, "the terminated stream retried once and the retry succeeded");
2039
+ assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
2040
+ mock.restoreAll();
2041
+ });
2042
+
1953
2043
  test("streamed-body silence does not replay when retries are disabled", async () => {
1954
2044
  let calls = 0;
1955
2045
  mock.method(globalThis, "fetch", async () => {
@@ -2230,13 +2320,13 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
2230
2320
 
2231
2321
  test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thinking parameter", async () => {
2232
2322
  // N>0 → enabled with budget_tokens
2233
- const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
2323
+ const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", outputBudget: 8192, reasoningBudget: 4096, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 4096 }, reasoningStyle: "anthropic" });
2234
2324
  let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
2235
2325
  await capped.generate({ workerId: "r", messages: [] });
2236
2326
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
2237
2327
 
2238
2328
  mock.restoreAll();
2239
- const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 2048 }, reasoningStyle: "anthropic" });
2329
+ const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, outputBudget: 4096, reasoningBudget: 2048, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "high", budget: 2048 }, reasoningStyle: "anthropic" });
2240
2330
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
2241
2331
  await unbudgeted.generate({ workerId: "r", messages: [] });
2242
2332
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
@@ -2249,11 +2339,11 @@ test("reasoningStyle 'anthropic' maps the configured reasoning subset to the thi
2249
2339
  assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
2250
2340
 
2251
2341
  mock.restoreAll();
2252
- // -1 adaptive → omit (API default depth)
2342
+ // Adaptive is explicit, so the provider—not omission—owns the posture.
2253
2343
  const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
2254
2344
  calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
2255
2345
  await adaptive.generate({ workerId: "r", messages: [] });
2256
- assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
2346
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "adaptive" });
2257
2347
  });
2258
2348
 
2259
2349
  // — non-streaming transport (streaming:false) —
@@ -2476,6 +2566,10 @@ test("native request projections compose reasoning visibility, affinity, and sys
2476
2566
  reasoningResponseProviderOptions: {
2477
2567
  google: { thinkingConfig: { includeThoughts: true } },
2478
2568
  },
2569
+ adaptiveReasoning: "provider-default",
2570
+ adaptiveReasoningProviderOptions: {
2571
+ google: { thinkingConfig: { thinkingBudget: -1 } },
2572
+ },
2479
2573
  systemCacheProviderOptions: {
2480
2574
  anthropic: { cacheControl: { type: "ephemeral" } },
2481
2575
  },
@@ -2490,7 +2584,7 @@ test("native request projections compose reasoning visibility, affinity, and sys
2490
2584
  });
2491
2585
 
2492
2586
  assert.deepEqual(request?.providerOptions, {
2493
- google: { thinkingConfig: { includeThoughts: true } },
2587
+ google: { thinkingConfig: { includeThoughts: true, thinkingBudget: -1 } },
2494
2588
  openai: { promptCacheKey: "worker-native" },
2495
2589
  });
2496
2590
  assert.deepEqual(request?.prompt, [
@@ -2532,13 +2626,13 @@ test("native AI SDK reasoning turns on without an operator token budget", async
2532
2626
  fetchTimeoutMs: 5000,
2533
2627
  temperature: 0.2,
2534
2628
  repeatPenalty: 1.15,
2535
- reasoning: { mode: "on", budget: null },
2629
+ reasoning: { mode: "high", budget: null },
2536
2630
  retryAttempts: 0,
2537
2631
  streaming: false,
2538
2632
  });
2539
2633
  const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
2540
2634
 
2541
- assert.equal(request?.reasoning, "medium");
2635
+ assert.equal(request?.reasoning, "high");
2542
2636
  assert.equal(response.assistant.reasoning, "consider");
2543
2637
  });
2544
2638
 
@@ -2577,13 +2671,93 @@ test("native additive-reasoning adapters preserve one total output budget", asyn
2577
2671
  fetchTimeoutMs: 5000,
2578
2672
  temperature: 0.2,
2579
2673
  repeatPenalty: 1.15,
2580
- reasoning: { mode: "on", budget: 2048 },
2674
+ reasoning: { mode: "high", budget: 2048 },
2581
2675
  retryAttempts: 0,
2582
2676
  streaming: false,
2583
2677
  });
2584
2678
  await p.generate({ workerId: `worker-${provider}`, messages: [], maxOutputTokens: 1500 });
2585
2679
  assert.equal(request?.maxOutputTokens, 1);
2586
- assert.equal(request?.reasoning, "provider-default");
2680
+ assert.equal(request?.reasoning, "high", "the fixed policy remains distinct from its independent token budget");
2587
2681
  assert.deepEqual(request?.providerOptions, expected);
2588
2682
  }
2589
2683
  });
2684
+
2685
+ test("native additive-reasoning adapters derive a bounded manual allowance when the model has no adaptive mode", async () => {
2686
+ for (const [provider, expected] of [
2687
+ ["anthropic", { anthropic: { thinking: { type: "enabled", budgetTokens: 1024 } } }],
2688
+ ["bedrock", { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: 1024 } } }],
2689
+ ] as const) {
2690
+ let request: Record<string, unknown> | undefined;
2691
+ const languageModel = {
2692
+ specificationVersion: "v4",
2693
+ provider: `native.${provider}`,
2694
+ modelId: "native-additive",
2695
+ supportedUrls: {},
2696
+ doGenerate: async (options: Record<string, unknown>) => {
2697
+ request = options;
2698
+ return {
2699
+ content: [{ type: "text", text: "ok" }],
2700
+ finishReason: { unified: "stop", raw: "stop" },
2701
+ usage: {
2702
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
2703
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
2704
+ },
2705
+ response: { id: "response", modelId: "native-additive" },
2706
+ warnings: [],
2707
+ };
2708
+ },
2709
+ doStream: async () => { throw new Error("streaming is not under test"); },
2710
+ } as unknown as LanguageModel;
2711
+ const p = testProvider({
2712
+ model: "native-additive",
2713
+ languageModel,
2714
+ additiveReasoningProvider: provider,
2715
+ outputBudget: 1500,
2716
+ reasoningBudget: null,
2717
+ fetchTimeoutMs: 5000,
2718
+ temperature: 0.2,
2719
+ repeatPenalty: 1.15,
2720
+ reasoning: { mode: "adaptive", budget: null },
2721
+ retryAttempts: 0,
2722
+ streaming: false,
2723
+ });
2724
+ await p.generate({ workerId: `worker-${provider}`, messages: [] });
2725
+ assert.equal(request?.maxOutputTokens, 476);
2726
+ assert.equal(request?.reasoning, "high", "adaptive retains its documented high fallback");
2727
+ assert.deepEqual(request?.providerOptions, expected);
2728
+ }
2729
+ });
2730
+
2731
+ test("a manual-reasoning model rejects an envelope below its provider minimum before I/O", async () => {
2732
+ let calls = 0;
2733
+ const languageModel = {
2734
+ specificationVersion: "v4",
2735
+ provider: "native.anthropic",
2736
+ modelId: "manual-reasoning",
2737
+ supportedUrls: {},
2738
+ doGenerate: async () => {
2739
+ calls++;
2740
+ throw new Error("provider I/O must not begin");
2741
+ },
2742
+ doStream: async () => { throw new Error("streaming is not under test"); },
2743
+ } as unknown as LanguageModel;
2744
+ const p = testProvider({
2745
+ model: "manual-reasoning",
2746
+ languageModel,
2747
+ additiveReasoningProvider: "anthropic",
2748
+ outputBudget: 1024,
2749
+ reasoningBudget: null,
2750
+ fetchTimeoutMs: 5000,
2751
+ temperature: 0.2,
2752
+ repeatPenalty: 1.15,
2753
+ reasoning: { mode: "adaptive", budget: null },
2754
+ retryAttempts: 0,
2755
+ streaming: false,
2756
+ });
2757
+
2758
+ await assert.rejects(
2759
+ p.generate({ workerId: "worker", messages: [] }),
2760
+ /total output budget must exceed the provider's 1024-token minimum reasoning allowance/,
2761
+ );
2762
+ assert.equal(calls, 0);
2763
+ });