@plurnk/plurnk-providers 1.14.1 → 1.14.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/.env.defaults +21 -9
  2. package/SPEC.md +54 -12
  3. package/dist/AiSdkProvider.d.ts +2 -2
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +24 -13
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Pool.d.ts.map +1 -1
  8. package/dist/Pool.js +2 -0
  9. package/dist/Pool.js.map +1 -1
  10. package/dist/aiSdkTransport.d.ts.map +1 -1
  11. package/dist/aiSdkTransport.js +29 -32
  12. package/dist/aiSdkTransport.js.map +1 -1
  13. package/dist/capacity.d.ts +8 -0
  14. package/dist/capacity.d.ts.map +1 -1
  15. package/dist/capacity.js +21 -0
  16. package/dist/capacity.js.map +1 -1
  17. package/dist/catalogProvider.d.ts.map +1 -1
  18. package/dist/catalogProvider.js +8 -3
  19. package/dist/catalogProvider.js.map +1 -1
  20. package/dist/compatibleProvider.d.ts.map +1 -1
  21. package/dist/compatibleProvider.js +6 -3
  22. package/dist/compatibleProvider.js.map +1 -1
  23. package/dist/errors.d.ts.map +1 -1
  24. package/dist/errors.js +23 -2
  25. package/dist/errors.js.map +1 -1
  26. package/dist/types.d.ts +1 -0
  27. package/dist/types.d.ts.map +1 -1
  28. package/dist/types.js +9 -1
  29. package/dist/types.js.map +1 -1
  30. package/package.json +6 -6
  31. package/src/AiSdkProvider.test.ts +138 -112
  32. package/src/AiSdkProvider.ts +28 -17
  33. package/src/Pool.test.ts +2 -0
  34. package/src/Pool.ts +2 -0
  35. package/src/aiSdkTransport.test.ts +31 -16
  36. package/src/aiSdkTransport.ts +32 -34
  37. package/src/capacity.test.ts +33 -1
  38. package/src/capacity.ts +34 -0
  39. package/src/catalogProvider.test.ts +22 -0
  40. package/src/catalogProvider.ts +7 -2
  41. package/src/compatibleProvider.test.ts +12 -0
  42. package/src/compatibleProvider.ts +6 -3
  43. package/src/errors.test.ts +1 -0
  44. package/src/errors.ts +22 -1
  45. package/src/types.ts +12 -1
@@ -289,19 +289,19 @@ test("the adapter preserves nonstandard reasoning accounting after SDK parsing",
289
289
  });
290
290
  });
291
291
 
292
- test("normalizeRetryAttemptError — attempt, first-content, and stream-idle deadlines are retryable transients ({§provider-connectivity})", () => {
292
+ test("normalizeRetryAttemptError — deadlines surface at once, never transport-retried ({§provider-connectivity}, #479)", () => {
293
293
  const first = normalizeRetryAttemptError(new ProviderTimeoutError("first_content", 180000));
294
294
  assert.equal(APICallError.isInstance(first), true);
295
- assert.equal((first as APICallError).isRetryable, true);
295
+ assert.equal((first as APICallError).isRetryable, false);
296
296
  const attempt = normalizeRetryAttemptError(new ProviderTimeoutError("attempt", 60000));
297
- assert.equal((attempt as APICallError).isRetryable, true);
297
+ assert.equal((attempt as APICallError).isRetryable, false);
298
298
  const idle = normalizeRetryAttemptError(new ProviderTimeoutError("stream_idle", 120000));
299
- assert.equal((idle as APICallError).isRetryable, true);
299
+ assert.equal((idle as APICallError).isRetryable, false);
300
300
  const operation = new ProviderTimeoutError("operation", 2700000);
301
301
  assert.equal(normalizeRetryAttemptError(operation), operation);
302
302
  });
303
303
 
304
- test("normalizeRetryAttemptError — a 2xx APICallError without a directive becomes retryable; a directive outranks; non-2xx untouched (#446)", () => {
304
+ test("normalizeRetryAttemptError — only provider-directed waits retry: 429, Retry-After, or a directive (#479 supersedes #446)", () => {
305
305
  const garbage = new APICallError({
306
306
  message: "Failed to process successful response",
307
307
  url: "https://api.example/v1/chat/completions",
@@ -310,28 +310,43 @@ test("normalizeRetryAttemptError — a 2xx APICallError without a directive beco
310
310
  responseBody: "not json",
311
311
  isRetryable: false,
312
312
  });
313
- const normalized = normalizeRetryAttemptError(garbage) as APICallError;
314
- assert.ok(APICallError.isInstance(normalized));
315
- assert.equal(normalized.isRetryable, true, "2xx invalid-response consumes the retry budget");
316
- assert.equal(normalized.cause, garbage, "the original failure rides as the cause");
317
- assert.equal(normalized.statusCode, 200);
313
+ assert.equal(normalizeRetryAttemptError(garbage), garbage, "2xx invalid-response surfaces at once — no promoted budget (#446 superseded)");
318
314
 
319
315
  const directed = new APICallError({
320
316
  message: "Failed to process successful response",
321
317
  url: "https://api.example/v1/chat/completions",
322
318
  requestBodyValues: {},
323
319
  statusCode: 200,
324
- responseHeaders: { "x-should-retry": "false" },
320
+ responseHeaders: { "x-should-retry": "true" },
325
321
  isRetryable: false,
326
322
  });
327
- assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, false, "an explicit directive outranks the 2xx default");
323
+ assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, true, "an explicit directive still outranks");
328
324
 
329
- const clientError = new APICallError({
330
- message: "bad request",
325
+ const rateLimited = new APICallError({
326
+ message: "slow down",
331
327
  url: "https://api.example/v1/chat/completions",
332
328
  requestBodyValues: {},
333
- statusCode: 400,
329
+ statusCode: 429,
334
330
  isRetryable: false,
335
331
  });
336
- assert.equal(normalizeRetryAttemptError(clientError), clientError, "non-2xx stays untouched");
332
+ assert.equal((normalizeRetryAttemptError(rateLimited) as APICallError).isRetryable, true, "a 429 is the provider-directed wait");
333
+
334
+ const directedWait = new APICallError({
335
+ message: "maintenance",
336
+ url: "https://api.example/v1/chat/completions",
337
+ requestBodyValues: {},
338
+ statusCode: 503,
339
+ responseHeaders: { "retry-after": "1" },
340
+ isRetryable: true,
341
+ });
342
+ assert.equal((normalizeRetryAttemptError(directedWait) as APICallError).isRetryable, true, "Retry-After on any status is a directed wait");
343
+
344
+ const bareServerError = new APICallError({
345
+ message: "internal error",
346
+ url: "https://api.example/v1/chat/completions",
347
+ requestBodyValues: {},
348
+ statusCode: 503,
349
+ isRetryable: true,
350
+ });
351
+ assert.equal((normalizeRetryAttemptError(bareServerError) as APICallError).isRetryable, false, "a bare 5xx surfaces at once for the engine's recovery");
337
352
  });
@@ -29,16 +29,26 @@ const retryDirective = (
29
29
  return null;
30
30
  };
31
31
 
32
+ // {§provider-connectivity} (#479): a Retry-After header on any status is the
33
+ // provider directing a wait (RFC 9110 defines it on 503 exactly for this);
34
+ // its presence, like a bare 429, earns the bounded transport retry.
35
+ const retryAfterPresent = (
36
+ headers: Headers | Readonly<Record<string, string>>,
37
+ ): boolean => {
38
+ const raw = headers instanceof Headers
39
+ ? headers.get("retry-after")
40
+ : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.[1];
41
+ return raw !== undefined && raw !== null && raw.trim() !== "";
42
+ };
43
+
32
44
  const errorStructure: ProviderErrorStructure<z.infer<typeof errorSchema>> = {
33
45
  errorSchema,
34
46
  errorToMessage: ({ error }) => error.message,
35
47
  isRetryable(response) {
36
- return retryDirective(response.status, response.headers) ?? (
37
- response.status === 408
38
- || response.status === 409
39
- || response.status === 429
40
- || response.status >= 500
41
- );
48
+ // {§provider-connectivity} (#479): only a provider-directed wait — 429,
49
+ // a Retry-After, or an explicit X-Should-Retry — earns a transport retry.
50
+ return retryDirective(response.status, response.headers)
51
+ ?? (response.status === 429 || retryAfterPresent(response.headers));
42
52
  },
43
53
  };
44
54
 
@@ -254,16 +264,16 @@ export const transportFailureOutputObserved = (error: unknown): boolean => {
254
264
 
255
265
  export const normalizeRetryAttemptError = (error: unknown): unknown => {
256
266
  if (!APICallError.isInstance(error)) {
257
- // Attempt, first-content, and stream-idle deadlines are retryable network
258
- // failures that consume the ordinary retry budget within the operation
259
- // deadline ({§provider-connectivity}); the stall is reported, never swallowed.
267
+ // Attempt, first-content, and stream-idle deadlines surface on the first
268
+ // failure ({§provider-connectivity}, #479): the engine's {§provider-recovery}
269
+ // owns re-issue with backoff and park; the stall is reported, never swallowed.
260
270
  if (error instanceof ProviderTimeoutError && error.phase !== "operation") {
261
271
  return retainStreamFailureValues(error, new APICallError({
262
272
  message: error.message,
263
273
  url: "model:generation",
264
274
  requestBodyValues: {},
265
275
  cause: error,
266
- isRetryable: true,
276
+ isRetryable: false,
267
277
  }));
268
278
  }
269
279
  // Node's Undici stream reader reports a peer-aborted HTTP/2 body as this
@@ -276,33 +286,21 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
276
286
  url: "model:generation",
277
287
  requestBodyValues: {},
278
288
  cause: error,
279
- isRetryable: true,
289
+ isRetryable: false,
280
290
  }));
281
291
  }
282
292
  return error;
283
293
  }
284
294
  const directed = retryDirective(error.statusCode, error.responseHeaders ?? {});
285
- // A 2xx APICallError with no explicit retry directive is a provider
286
- // invalid-response: the exchange succeeded and the body was unusable (a
287
- // serializer hiccup, a truncated frame). That is the same transient class as
288
- // a transport failure and consumes the same bounded retry budget; an explicit
289
- // `x-should-retry` directive outranks this default, and the terminal
290
- // classification after the budget stays the non-retryable 502 (#446).
291
- if (directed === null
292
- && typeof error.statusCode === "number" && error.statusCode >= 200 && error.statusCode < 300
293
- && error.isRetryable !== true) {
294
- return retainStreamFailureValues(error, new APICallError({
295
- message: error.message,
296
- url: error.url,
297
- requestBodyValues: error.requestBodyValues,
298
- statusCode: error.statusCode,
299
- responseHeaders: error.responseHeaders,
300
- responseBody: error.responseBody,
301
- cause: error,
302
- isRetryable: true,
303
- }));
304
- }
305
- if (directed === null || directed === error.isRetryable) return error;
295
+ // {§provider-connectivity} (#479): without an explicit directive the only
296
+ // transport-retryable failures are the provider-directed waits — a 429, or
297
+ // any status carrying Retry-After; those live in headers the engine never
298
+ // sees. Every other failure — a bare 408/409/5xx, a network error, and the
299
+ // 2xx invalid-response #446 once promoted — surfaces at once;
300
+ // {§provider-recovery} owns re-issue.
301
+ const policy = directed
302
+ ?? (error.statusCode === 429 || retryAfterPresent(error.responseHeaders ?? {}));
303
+ if (policy === error.isRetryable) return error;
306
304
  return retainStreamFailureValues(error, new APICallError({
307
305
  message: error.message,
308
306
  url: error.url,
@@ -311,7 +309,7 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
311
309
  responseHeaders: error.responseHeaders,
312
310
  responseBody: error.responseBody,
313
311
  cause: error,
314
- isRetryable: directed,
312
+ isRetryable: policy,
315
313
  data: error.data,
316
314
  }));
317
315
  };
@@ -330,7 +328,7 @@ const executeModel = async (
330
328
  url: "model:generation",
331
329
  requestBodyValues: {},
332
330
  cause: timeout,
333
- isRetryable: true,
331
+ isRetryable: false,
334
332
  });
335
333
  }
336
334
  };
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import assert from "node:assert/strict";
3
- import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
3
+ import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, flexedResponseMax } from "./capacity.ts";
4
4
 
5
5
  test("effective output budget is caller-tightenable and physically capped", () => {
6
6
  assert.equal(effectiveOutputBudget({
@@ -90,3 +90,35 @@ test("only exact overflow rejects before provider I/O", () => {
90
90
  measurement: { kind: "upper_bound", tokens: 60, source: "bound" },
91
91
  }).decision, "admit");
92
92
  });
93
+
94
+
95
+ test("(#482) flexedResponseMax harvests exact slack above the floor", () => {
96
+ assert.equal(
97
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
98
+ 47_644,
99
+ "small prompt: the window remainder minus margin",
100
+ );
101
+ assert.equal(
102
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 40_000, margin: 256 }),
103
+ 8_000,
104
+ "a full packet keeps the guaranteed floor even when margin eats the slack",
105
+ );
106
+ assert.equal(
107
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: 16_000, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
108
+ 16_000,
109
+ "the model's own output cap bounds the harvest",
110
+ );
111
+ assert.equal(
112
+ flexedResponseMax({ contextWindow: null, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
113
+ 8_000,
114
+ "no window, no flex",
115
+ );
116
+ });
117
+
118
+ test("(#482) assessRequestCapacity flexes only exact measurements", () => {
119
+ const base = { contextWindow: 48_000, maxInputTokens: null, maxOutputTokens: null, outputBudget: 8_000, reasoningBudget: null };
120
+ const exact = assessRequestCapacity({ ...base, measurement: { kind: "exact", tokens: 1_000, source: "t" } });
121
+ assert.equal(exact.responseMax, 48_000 - 1_000 - 256, "exact prompts harvest the slack");
122
+ const estimate = assessRequestCapacity({ ...base, measurement: { kind: "estimate", tokens: 1_000, source: "t", detail: "chars/2 test estimate" } });
123
+ assert.equal(estimate.responseMax, 8_000, "estimates keep the floor — they prove nothing about the remainder");
124
+ });
package/src/capacity.ts CHANGED
@@ -81,6 +81,34 @@ export const effectiveInputCapacity = ({
81
81
  return capacities.length === 0 ? null : Math.min(...capacities);
82
82
  };
83
83
 
84
+ // {§provider-flexed-allowance} (#482): the configured output budget is the floor
85
+ // curation packed the input against; window room the actual prompt left
86
+ // unclaimed is guaranteed free and becomes response runway. Only an exact
87
+ // prompt measurement may claim slack — an estimate proves nothing about the
88
+ // true remainder — and the model's own maxOutputTokens still caps the grant.
89
+ export const WIRE_FLEX_MARGIN = 256;
90
+
91
+ export const flexedResponseMax = ({
92
+ contextWindow,
93
+ maxOutputTokens,
94
+ outputBudget,
95
+ promptTokens,
96
+ margin,
97
+ }: {
98
+ contextWindow: number | null;
99
+ maxOutputTokens: number | null;
100
+ outputBudget: number | null;
101
+ promptTokens: number;
102
+ margin: number;
103
+ }): number | null => {
104
+ if (outputBudget === null || contextWindow === null) return outputBudget;
105
+ if (!Number.isSafeInteger(promptTokens) || promptTokens < 0) {
106
+ throw new TypeError("promptTokens must be a non-negative safe integer");
107
+ }
108
+ const flexed = Math.max(outputBudget, contextWindow - promptTokens - margin);
109
+ return maxOutputTokens === null ? flexed : Math.min(flexed, Math.max(outputBudget, maxOutputTokens));
110
+ };
111
+
84
112
  export const requestCapacityDecision = (
85
113
  inputCapacity: number | null,
86
114
  measurement: PromptTokenMeasurement,
@@ -126,6 +154,11 @@ export const assessRequestCapacity = ({
126
154
  }
127
155
  const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
128
156
  const inputCapacity = effectiveInputCapacity({ contextWindow, maxInputTokens, outputBudget });
157
+ // {§provider-flexed-allowance}: exact measurements harvest the slack; every
158
+ // other measurement kind keeps the floor.
159
+ const responseMax = prompt.kind === "exact"
160
+ ? flexedResponseMax({ contextWindow, maxOutputTokens, outputBudget, promptTokens: prompt.tokens, margin: WIRE_FLEX_MARGIN })
161
+ : outputBudget;
129
162
 
130
163
  return {
131
164
  decision: requestCapacityDecision(inputCapacity, prompt),
@@ -135,6 +168,7 @@ export const assessRequestCapacity = ({
135
168
  outputBudget,
136
169
  reasoningBudget,
137
170
  inputCapacity,
171
+ responseMax,
138
172
  prompt,
139
173
  };
140
174
  };
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
3
3
  import { once } from "node:events";
4
4
  import { createServer } from "node:http";
5
5
  import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
6
+ import { withProviderDefaults } from "./defaults.ts";
6
7
  import type { LanguageModel } from "ai";
7
8
  import { resetEmittedWarnings } from "./warnings.ts";
8
9
 
@@ -41,6 +42,27 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
41
42
  assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
42
43
  });
43
44
 
45
+ test("(#472) an effort refusal names the operator's declaration lever with the exact key", () => {
46
+ assert.throws(
47
+ () => catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
48
+ ...env,
49
+ FIREWORKS_API_KEY: "test-key",
50
+ PLURNK_PROVIDERS_REASONING: "low",
51
+ }), "accounts/fireworks/models/glm-5p3-flash"),
52
+ /declare it: PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS=low/,
53
+ "the refusal is actionable: it names the exact env declaration",
54
+ );
55
+ // And the lever works: the same route with the declaration constructs.
56
+ const declared = catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
57
+ ...env,
58
+ FIREWORKS_API_KEY: "test-key",
59
+ PLURNK_PROVIDERS_REASONING: "low",
60
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
61
+ }), "accounts/fireworks/models/glm-5p3-flash");
62
+ assert.ok(declared, "the declared effort admits the route");
63
+ assert.ok(declared.supportedReasoningPolicies.includes("low"), "low is admitted through the operator's declaration");
64
+ });
65
+
44
66
  test("provider adapters advertise only reasoning policies they can preserve", () => {
45
67
  const deepseek = catalogProviderFromEnv("deepseek", {
46
68
  ...env,
@@ -10,6 +10,7 @@ import {
10
10
  effectiveContextWindow,
11
11
  dataCaptureFromEnv,
12
12
  generationEnvelopeFromEnv,
13
+ parseOptionalFloat,
13
14
  parseRequiredFloat,
14
15
  parseRequiredInt,
15
16
  parseTimeoutMs,
@@ -154,6 +155,10 @@ const supportedReasoningPolicies = ({
154
155
  style: ReasoningStyle;
155
156
  declared: readonly ModelReasoningEffort[];
156
157
  }): readonly ReasoningPolicy[] => {
158
+ // The template style is the operator's declaration that the rail's own chat
159
+ // template governs reasoning: a fixed effort rides in verbatim, native SDK or
160
+ // not, and a word the template does not know fails loudly on the first request.
161
+ if (style === "template") return REASONING_POLICIES;
157
162
  if (info !== undefined && info.reasoning !== true) return activationPolicies;
158
163
  if (info?.reasoningOptions !== undefined) {
159
164
  return catalogSupportedReasoningPolicies({ info, native, style, declared });
@@ -356,8 +361,8 @@ export const providerFromSdkModel = ({
356
361
  streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
357
362
  reasoning,
358
363
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
359
- temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
360
- repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
364
+ temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
365
+ repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
361
366
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
362
367
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
363
368
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
@@ -83,6 +83,18 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
83
83
  assert.equal("dry_allowed_length" in (body ?? {}), false);
84
84
  });
85
85
 
86
+ test("(#483) a detected llama-server rail admits the operator's stated effort", async () => {
87
+ mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
88
+ const url = String(input);
89
+ if (url.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }] }));
90
+ if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
91
+ throw new Error(`unexpected request ${url}`);
92
+ });
93
+ const provider = await compatibleProviderFromEnv("openai", { ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
94
+ assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
95
+ assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
96
+ });
97
+
86
98
  test("detected llama-server measures the complete chat request through input_tokens", async () => {
87
99
  let countUrl: string | undefined;
88
100
  let countBody: Record<string, unknown> | undefined;
@@ -1,3 +1,4 @@
1
+ import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
1
2
  import AiSdkProvider, { type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
2
3
  import {
3
4
  contextWindowFromEnv,
@@ -182,7 +183,9 @@ export const compatibleProviderFromEnv = async (
182
183
  }
183
184
  const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
184
185
  const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
185
- const supportedReasoningPolicies = ["off", "adaptive"] as const;
186
+ // A detected llama-server rail runs the template style: the chat template governs
187
+ // reasoning, so the operator's stated effort is admitted and forwarded verbatim (#483).
188
+ const supportedReasoningPolicies = reasoningStyle === "template" ? REASONING_POLICIES : (["off", "adaptive"] as const);
186
189
  return new AiSdkProvider({
187
190
  model,
188
191
  url,
@@ -200,8 +203,8 @@ export const compatibleProviderFromEnv = async (
200
203
  reasoning,
201
204
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
202
205
  reasoningStyle,
203
- temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
204
- repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", provider, 0),
206
+ temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
207
+ repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", provider, 0),
205
208
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", provider, 0),
206
209
  dryMultiplier: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_MULTIPLIER, "PLURNK_PROVIDERS_DRY_MULTIPLIER", provider, 0) ?? undefined,
207
210
  dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
@@ -126,6 +126,7 @@ test("#161: ProviderError carries resource-interrupted attempt evidence outside
126
126
  maxInputTokens: null,
127
127
  maxOutputTokens: null,
128
128
  outputBudget: null,
129
+ responseMax: null,
129
130
  reasoningBudget: null,
130
131
  inputCapacity: null,
131
132
  prompt: {
package/src/errors.ts CHANGED
@@ -46,6 +46,17 @@ export const providerTimeoutOf = (error: unknown): ProviderTimeoutError | null =
46
46
  return null;
47
47
  };
48
48
 
49
+ const peerTerminated = (err: unknown): boolean => {
50
+ const seen = new Set<unknown>();
51
+ let cur: unknown = err;
52
+ while (typeof cur === "object" && cur !== null && !seen.has(cur)) {
53
+ seen.add(cur);
54
+ if (cur instanceof TypeError && cur.message.trim().toLowerCase() === "terminated") return true;
55
+ cur = (cur as { cause?: unknown }).cause;
56
+ }
57
+ return false;
58
+ };
59
+
49
60
  const wireError = (body: string): { type: string | null; code: string | null; message: string | null } => {
50
61
  try {
51
62
  const { error } = JSON.parse(body) as { error?: { type?: unknown } };
@@ -100,13 +111,23 @@ export const classifyProviderError = (
100
111
  : "The provider request failed without a diagnostic message.";
101
112
  const body = err.responseBody ?? "";
102
113
  const wire = wireError(body);
114
+ // A peer-terminated body inside a 2xx exchange is a network truth, not
115
+ // an invalid response — the SDK wraps Undici's TypeError("terminated")
116
+ // as a processing failure, but the KIND must stay engine-recoverable
117
+ // (#479). Walk the cause chain for the termination signature.
118
+ if (peerTerminated(err)) {
119
+ return { kind: "network_failure", message, retryable: err.isRetryable };
120
+ }
103
121
  if (status === 401 || status === 403) return { kind: "unauthorized", message };
104
122
  if (status === 402) return { kind: "quota_exceeded", message };
105
123
  if (status === 429) return { kind: "rate_limit", message, retryable: err.isRetryable };
106
124
  if (status === 408 || status === 409) {
107
125
  return { kind: "network_failure", message, retryable: err.isRetryable };
108
126
  }
109
- if (status === 0 && err.isRetryable) return { kind: "network_failure", message };
127
+ // Status 0 is a transport-level failure (no HTTP exchange settled); its
128
+ // KIND stays network_failure regardless of the transport's retry policy
129
+ // (#479) — the engine's recovery keys on kind, never the retryable flag.
130
+ if (status === 0) return { kind: "network_failure", message, retryable: err.isRetryable };
110
131
  if (status >= 500) return { kind: "network_failure", message, retryable: err.isRetryable };
111
132
  if (status === 413 || (
112
133
  (status === 400 || status === 422)
package/src/types.ts CHANGED
@@ -32,7 +32,15 @@ export class UnsupportedReasoningPolicyError extends Error {
32
32
  readonly supported: readonly ReasoningPolicy[];
33
33
 
34
34
  constructor(source: string, policy: ReasoningPolicy, supported: readonly ReasoningPolicy[]) {
35
- super(`${source}: reasoning policy '${policy}' is unsupported; supported policies: ${supported.join(", ")}`);
35
+ // A graded-effort refusal names the operator's declaration lever (#472):
36
+ // the catalog is authoritative per model, and where a provider's API
37
+ // accepts more than its catalog entry lists, the remedy is the
38
+ // operator's, never a vendored fact in shipped defaults.
39
+ const providerName = /^provider:(.+)$/.exec(source)?.[1];
40
+ const lever = policy === "off" || policy === "adaptive"
41
+ ? ""
42
+ : ` If the provider's API documents this effort beyond its catalog entry, declare it: PLURNK_PROVIDERS_PROVIDER_${(providerName ?? "<NAME>").replaceAll(/[^a-zA-Z0-9<>]/g, "_").toUpperCase()}_REASONING_EFFORTS=${policy} (comma-separated).`;
43
+ super(`${source}: reasoning policy '${policy}' is unsupported; supported policies: ${supported.join(", ")}.${lever}`);
36
44
  this.name = "UnsupportedReasoningPolicyError";
37
45
  this.policy = policy;
38
46
  this.supported = supported;
@@ -82,6 +90,9 @@ export interface ProviderRequestCapacity {
82
90
  readonly outputBudget: number | null;
83
91
  readonly reasoningBudget: number | null;
84
92
  readonly inputCapacity: number | null;
93
+ // {§provider-flexed-allowance} (#482): the response allowance actually
94
+ // granted this request — the floor, or the exactly-measured slack above it.
95
+ readonly responseMax: number | null;
85
96
  readonly prompt: PromptTokenMeasurement;
86
97
  }
87
98