@plurnk/plurnk-providers 1.14.0 → 1.14.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.env.defaults +21 -9
  2. package/SPEC.md +54 -12
  3. package/dist/AiSdkProvider.d.ts +2 -2
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +36 -15
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/Pool.d.ts.map +1 -1
  8. package/dist/Pool.js +2 -0
  9. package/dist/Pool.js.map +1 -1
  10. package/dist/aiSdkTransport.d.ts.map +1 -1
  11. package/dist/aiSdkTransport.js +29 -32
  12. package/dist/aiSdkTransport.js.map +1 -1
  13. package/dist/capacity.d.ts +8 -0
  14. package/dist/capacity.d.ts.map +1 -1
  15. package/dist/capacity.js +21 -0
  16. package/dist/capacity.js.map +1 -1
  17. package/dist/catalogProvider.d.ts.map +1 -1
  18. package/dist/catalogProvider.js +10 -4
  19. package/dist/catalogProvider.js.map +1 -1
  20. package/dist/compatibleProvider.d.ts.map +1 -1
  21. package/dist/compatibleProvider.js +6 -3
  22. package/dist/compatibleProvider.js.map +1 -1
  23. package/dist/errors.d.ts.map +1 -1
  24. package/dist/errors.js +23 -2
  25. package/dist/errors.js.map +1 -1
  26. package/dist/sdkModels.d.ts.map +1 -1
  27. package/dist/sdkModels.js +6 -0
  28. package/dist/sdkModels.js.map +1 -1
  29. package/dist/types.d.ts +1 -0
  30. package/dist/types.d.ts.map +1 -1
  31. package/dist/types.js +9 -1
  32. package/dist/types.js.map +1 -1
  33. package/package.json +6 -6
  34. package/src/AiSdkProvider.test.ts +138 -112
  35. package/src/AiSdkProvider.ts +41 -20
  36. package/src/Pool.test.ts +2 -0
  37. package/src/Pool.ts +2 -0
  38. package/src/aiSdkTransport.test.ts +31 -16
  39. package/src/aiSdkTransport.ts +32 -34
  40. package/src/capacity.test.ts +33 -1
  41. package/src/capacity.ts +34 -0
  42. package/src/catalogProvider.test.ts +28 -6
  43. package/src/catalogProvider.ts +9 -3
  44. package/src/compatibleProvider.test.ts +12 -0
  45. package/src/compatibleProvider.ts +6 -3
  46. package/src/env.test.ts +4 -4
  47. package/src/errors.test.ts +1 -0
  48. package/src/errors.ts +22 -1
  49. package/src/sdkModels.ts +6 -0
  50. package/src/types.ts +12 -1
@@ -173,8 +173,8 @@ export type AiSdkProviderConfig = {
173
173
  // `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
174
174
  // (greedy-under-mask loops without it) — the VALUE is operator config;
175
175
  // WHERE it applies stays mechanism.
176
- temperature: number;
177
- repeatPenalty: number;
176
+ temperature: number | null;
177
+ repeatPenalty: number | null;
178
178
  // Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
179
179
  // repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
180
180
  // Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
@@ -310,11 +310,19 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
310
310
  return { content, reasoning: "", projected: false, contentStart: 0 };
311
311
  };
312
312
 
313
- const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" => {
314
- if (mode === "low" || mode === "medium" || mode === "high") return mode;
313
+ const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" | "max" => {
314
+ if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max") return mode;
315
315
  throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
316
316
  };
317
317
 
318
+ // The native SDK effort surface tops at xhigh; admission never grants a native
319
+ // route "max", so reaching it here is a contract violation, not a fallback site.
320
+ const nativeFixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" => {
321
+ const effort = fixedEffort(mode);
322
+ if (effort === "max") throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
323
+ return effort;
324
+ };
325
+
318
326
  // Anthropic's older manual-reasoning protocol needs an absolute allowance while
319
327
  // PLURNK's durable contract names an effort. These fractions match the native
320
328
  // SDK's policy projection, but apply to PLURNK's total envelope rather than the
@@ -324,6 +332,8 @@ const MANUAL_REASONING_FRACTIONS = Object.freeze({
324
332
  low: 0.1,
325
333
  medium: 0.3,
326
334
  high: 0.6,
335
+ xhigh: 0.75,
336
+ max: 0.85,
327
337
  } satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
328
338
  const MANUAL_REASONING_MINIMUM = 1024;
329
339
 
@@ -385,8 +395,8 @@ export default class AiSdkProvider implements Provider {
385
395
  #compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
386
396
  #compatibleOffReasoning: "none" | undefined;
387
397
  #adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
388
- #temperature: number;
389
- #repeatPenalty: number;
398
+ #temperature: number | null;
399
+ #repeatPenalty: number | null;
390
400
  #frequencyPenalty: number;
391
401
  #dryMultiplier: number | undefined;
392
402
  #dryBase: number | undefined;
@@ -472,8 +482,8 @@ export default class AiSdkProvider implements Provider {
472
482
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
473
483
  // required tuning fields must fail at construction, not silently send
474
484
  // undefined sampling on every grammar request.
475
- if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
476
- throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
485
+ if (config.temperature === undefined || config.repeatPenalty === undefined) {
486
+ throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty declared (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY; null = provider default)`);
477
487
  }
478
488
  this.#temperature = config.temperature;
479
489
  this.#repeatPenalty = config.repeatPenalty;
@@ -708,8 +718,11 @@ export default class AiSdkProvider implements Provider {
708
718
  const allowance = mode === "off"
709
719
  ? 0
710
720
  : budget;
721
+ // A fixed effort rides into the template as its own variable; adaptive
722
+ // and off send none and leave the template's default in force.
723
+ const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
711
724
  return {
712
- chat_template_kwargs: { enable_thinking: on },
725
+ chat_template_kwargs: { enable_thinking: on, ...templateEffort },
713
726
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
714
727
  ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
715
728
  };
@@ -803,9 +816,9 @@ export default class AiSdkProvider implements Provider {
803
816
  #grammarBody(grammar: string | undefined): Record<string, unknown> {
804
817
  if (grammar === undefined) return {};
805
818
  switch (this.#grammarStyle) {
806
- // Greedy decoding under hard constraint loops without a repeat-penalty
807
- // floor — llama.cpp spells it `repeat_penalty`.
808
- case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
819
+ // Grammar-constrained decoding can loop under the mask; a configured
820
+ // per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
821
+ case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
809
822
  case "none": return {};
810
823
  }
811
824
  }
@@ -825,7 +838,7 @@ export default class AiSdkProvider implements Provider {
825
838
  // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
826
839
  // Each rides only when its operator knob is set; absent = the box's default.
827
840
  case "llamacpp": return {
828
- repeat_penalty: this.#repeatPenalty,
841
+ ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
829
842
  ...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
830
843
  ...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
831
844
  dry_multiplier: this.#dryMultiplier,
@@ -1023,7 +1036,9 @@ export default class AiSdkProvider implements Provider {
1023
1036
  { capacity, extensions: { capacityStage: "preflight", capacity } },
1024
1037
  );
1025
1038
  }
1026
- const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
1039
+ // {§provider-flexed-allowance} (#482): the wire grants the flexed
1040
+ // allowance — the floor, or the exactly-measured slack above it.
1041
+ const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
1027
1042
  const nativeReasoningBudget = this.#nativeReasoningBudget(
1028
1043
  capacity.outputBudget,
1029
1044
  capacity.reasoningBudget,
@@ -1036,7 +1051,7 @@ export default class AiSdkProvider implements Provider {
1036
1051
  const body: Record<string, unknown> = {
1037
1052
  // Floors are suppressed on router-owned-tuning providers (plurnk) —
1038
1053
  // the router's per-model tuning must not be overridden by client floors.
1039
- ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
1054
+ ...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
1040
1055
  ...this.#samplingBody(sampling),
1041
1056
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
1042
1057
  model: this.#model,
@@ -1172,7 +1187,7 @@ export default class AiSdkProvider implements Provider {
1172
1187
  captureRawBody: this.#rawBody,
1173
1188
  ...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
1174
1189
  temperature: this.#tuningFloors
1175
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
1190
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature ?? undefined)
1176
1191
  : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
1177
1192
  topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
1178
1193
  topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
@@ -1191,7 +1206,7 @@ export default class AiSdkProvider implements Provider {
1191
1206
  ? "none"
1192
1207
  : this.#reasoning.mode === "adaptive"
1193
1208
  ? this.#adaptiveReasoning
1194
- : fixedEffort(this.#reasoning.mode),
1209
+ : nativeFixedEffort(this.#reasoning.mode),
1195
1210
  });
1196
1211
  } catch (error) {
1197
1212
  if (transportFailureOutputObserved(error)) recoveredAfterOutput = true;
@@ -1387,9 +1402,14 @@ export default class AiSdkProvider implements Provider {
1387
1402
  ...(meta !== undefined ? { meta } : {}),
1388
1403
  ...(notices !== undefined ? { notices } : {}),
1389
1404
  };
1390
- if (capacity.outputBudget !== null
1405
+ // {§provider-flexed-allowance} (#482): conformance judges the GRANT the
1406
+ // wire actually sent, not the configured floor — output between the two
1407
+ // is overflow tolerance working, never a provider fault. Run7 loop-death
1408
+ // was this guard still holding the floor after the flex landed.
1409
+ const grantedOutput = capacity.responseMax ?? capacity.outputBudget;
1410
+ if (grantedOutput !== null
1391
1411
  && usage?.outputTokens !== undefined
1392
- && usage.outputTokens > capacity.outputBudget) {
1412
+ && usage.outputTokens > grantedOutput) {
1393
1413
  const attempt: ProviderAttempt = {
1394
1414
  assistant: { ...assistant, finishReason: raw.finishReason },
1395
1415
  ...evidence,
@@ -1397,13 +1417,14 @@ export default class AiSdkProvider implements Provider {
1397
1417
  throw new ProviderError(
1398
1418
  this.#source,
1399
1419
  "invalid_response",
1400
- `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`,
1420
+ `The provider reported ${usage.outputTokens} output tokens after receiving a granted output allowance of ${grantedOutput}.`,
1401
1421
  {
1402
1422
  attempt,
1403
1423
  accounting,
1404
1424
  extensions: {
1405
1425
  stage: "provider-response",
1406
1426
  outputBudget: capacity.outputBudget,
1427
+ grantedOutput,
1407
1428
  reportedOutputTokens: usage.outputTokens,
1408
1429
  },
1409
1430
  },
package/src/Pool.test.ts CHANGED
@@ -31,6 +31,7 @@ const RESP: ProviderResponse = {
31
31
  outputBudget: 12_000,
32
32
  reasoningBudget: null,
33
33
  inputCapacity: 36_000,
34
+ responseMax: 12_000,
34
35
  prompt: { kind: "exact", tokens: 0, source: "test:exact" },
35
36
  },
36
37
  };
@@ -76,6 +77,7 @@ const backend = (opts: FakeOpts = {}) => {
76
77
  outputBudget: maxOutputTokens ?? opts.outputBudget ?? null,
77
78
  reasoningBudget: opts.reasoningBudget ?? null,
78
79
  inputCapacity: null,
80
+ responseMax: maxOutputTokens ?? opts.outputBudget ?? null,
79
81
  prompt: opts.promptMeasurement ?? {
80
82
  kind: "exact",
81
83
  tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
package/src/Pool.ts CHANGED
@@ -190,6 +190,8 @@ export default class Pool implements Provider {
190
190
  outputBudget: minimum(envelopes.map((envelope) => envelope.outputBudget)),
191
191
  reasoningBudget: minimum(envelopes.map((envelope) => envelope.reasoningBudget)),
192
192
  inputCapacity,
193
+ // A pool spans members whose windows differ; it never flexes ({§provider-flexed-allowance}).
194
+ responseMax: minimum(envelopes.map((envelope) => envelope.outputBudget)),
193
195
  prompt: measurement,
194
196
  };
195
197
  }
@@ -289,19 +289,19 @@ test("the adapter preserves nonstandard reasoning accounting after SDK parsing",
289
289
  });
290
290
  });
291
291
 
292
- test("normalizeRetryAttemptError — attempt, first-content, and stream-idle deadlines are retryable transients ({§provider-connectivity})", () => {
292
+ test("normalizeRetryAttemptError — deadlines surface at once, never transport-retried ({§provider-connectivity}, #479)", () => {
293
293
  const first = normalizeRetryAttemptError(new ProviderTimeoutError("first_content", 180000));
294
294
  assert.equal(APICallError.isInstance(first), true);
295
- assert.equal((first as APICallError).isRetryable, true);
295
+ assert.equal((first as APICallError).isRetryable, false);
296
296
  const attempt = normalizeRetryAttemptError(new ProviderTimeoutError("attempt", 60000));
297
- assert.equal((attempt as APICallError).isRetryable, true);
297
+ assert.equal((attempt as APICallError).isRetryable, false);
298
298
  const idle = normalizeRetryAttemptError(new ProviderTimeoutError("stream_idle", 120000));
299
- assert.equal((idle as APICallError).isRetryable, true);
299
+ assert.equal((idle as APICallError).isRetryable, false);
300
300
  const operation = new ProviderTimeoutError("operation", 2700000);
301
301
  assert.equal(normalizeRetryAttemptError(operation), operation);
302
302
  });
303
303
 
304
- test("normalizeRetryAttemptError — a 2xx APICallError without a directive becomes retryable; a directive outranks; non-2xx untouched (#446)", () => {
304
+ test("normalizeRetryAttemptError — only provider-directed waits retry: 429, Retry-After, or a directive (#479 supersedes #446)", () => {
305
305
  const garbage = new APICallError({
306
306
  message: "Failed to process successful response",
307
307
  url: "https://api.example/v1/chat/completions",
@@ -310,28 +310,43 @@ test("normalizeRetryAttemptError — a 2xx APICallError without a directive beco
310
310
  responseBody: "not json",
311
311
  isRetryable: false,
312
312
  });
313
- const normalized = normalizeRetryAttemptError(garbage) as APICallError;
314
- assert.ok(APICallError.isInstance(normalized));
315
- assert.equal(normalized.isRetryable, true, "2xx invalid-response consumes the retry budget");
316
- assert.equal(normalized.cause, garbage, "the original failure rides as the cause");
317
- assert.equal(normalized.statusCode, 200);
313
+ assert.equal(normalizeRetryAttemptError(garbage), garbage, "2xx invalid-response surfaces at once — no promoted budget (#446 superseded)");
318
314
 
319
315
  const directed = new APICallError({
320
316
  message: "Failed to process successful response",
321
317
  url: "https://api.example/v1/chat/completions",
322
318
  requestBodyValues: {},
323
319
  statusCode: 200,
324
- responseHeaders: { "x-should-retry": "false" },
320
+ responseHeaders: { "x-should-retry": "true" },
325
321
  isRetryable: false,
326
322
  });
327
- assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, false, "an explicit directive outranks the 2xx default");
323
+ assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, true, "an explicit directive still outranks");
328
324
 
329
- const clientError = new APICallError({
330
- message: "bad request",
325
+ const rateLimited = new APICallError({
326
+ message: "slow down",
331
327
  url: "https://api.example/v1/chat/completions",
332
328
  requestBodyValues: {},
333
- statusCode: 400,
329
+ statusCode: 429,
334
330
  isRetryable: false,
335
331
  });
336
- assert.equal(normalizeRetryAttemptError(clientError), clientError, "non-2xx stays untouched");
332
+ assert.equal((normalizeRetryAttemptError(rateLimited) as APICallError).isRetryable, true, "a 429 is the provider-directed wait");
333
+
334
+ const directedWait = new APICallError({
335
+ message: "maintenance",
336
+ url: "https://api.example/v1/chat/completions",
337
+ requestBodyValues: {},
338
+ statusCode: 503,
339
+ responseHeaders: { "retry-after": "1" },
340
+ isRetryable: true,
341
+ });
342
+ assert.equal((normalizeRetryAttemptError(directedWait) as APICallError).isRetryable, true, "Retry-After on any status is a directed wait");
343
+
344
+ const bareServerError = new APICallError({
345
+ message: "internal error",
346
+ url: "https://api.example/v1/chat/completions",
347
+ requestBodyValues: {},
348
+ statusCode: 503,
349
+ isRetryable: true,
350
+ });
351
+ assert.equal((normalizeRetryAttemptError(bareServerError) as APICallError).isRetryable, false, "a bare 5xx surfaces at once for the engine's recovery");
337
352
  });
@@ -29,16 +29,26 @@ const retryDirective = (
29
29
  return null;
30
30
  };
31
31
 
32
+ // {§provider-connectivity} (#479): a Retry-After header on any status is the
33
+ // provider directing a wait (RFC 9110 defines it on 503 exactly for this);
34
+ // its presence, like a bare 429, earns the bounded transport retry.
35
+ const retryAfterPresent = (
36
+ headers: Headers | Readonly<Record<string, string>>,
37
+ ): boolean => {
38
+ const raw = headers instanceof Headers
39
+ ? headers.get("retry-after")
40
+ : Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.[1];
41
+ return raw !== undefined && raw !== null && raw.trim() !== "";
42
+ };
43
+
32
44
  const errorStructure: ProviderErrorStructure<z.infer<typeof errorSchema>> = {
33
45
  errorSchema,
34
46
  errorToMessage: ({ error }) => error.message,
35
47
  isRetryable(response) {
36
- return retryDirective(response.status, response.headers) ?? (
37
- response.status === 408
38
- || response.status === 409
39
- || response.status === 429
40
- || response.status >= 500
41
- );
48
+ // {§provider-connectivity} (#479): only a provider-directed wait — 429,
49
+ // a Retry-After, or an explicit X-Should-Retry — earns a transport retry.
50
+ return retryDirective(response.status, response.headers)
51
+ ?? (response.status === 429 || retryAfterPresent(response.headers));
42
52
  },
43
53
  };
44
54
 
@@ -254,16 +264,16 @@ export const transportFailureOutputObserved = (error: unknown): boolean => {
254
264
 
255
265
  export const normalizeRetryAttemptError = (error: unknown): unknown => {
256
266
  if (!APICallError.isInstance(error)) {
257
- // Attempt, first-content, and stream-idle deadlines are retryable network
258
- // failures that consume the ordinary retry budget within the operation
259
- // deadline ({§provider-connectivity}); the stall is reported, never swallowed.
267
+ // Attempt, first-content, and stream-idle deadlines surface on the first
268
+ // failure ({§provider-connectivity}, #479): the engine's {§provider-recovery}
269
+ // owns re-issue with backoff and park; the stall is reported, never swallowed.
260
270
  if (error instanceof ProviderTimeoutError && error.phase !== "operation") {
261
271
  return retainStreamFailureValues(error, new APICallError({
262
272
  message: error.message,
263
273
  url: "model:generation",
264
274
  requestBodyValues: {},
265
275
  cause: error,
266
- isRetryable: true,
276
+ isRetryable: false,
267
277
  }));
268
278
  }
269
279
  // Node's Undici stream reader reports a peer-aborted HTTP/2 body as this
@@ -276,33 +286,21 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
276
286
  url: "model:generation",
277
287
  requestBodyValues: {},
278
288
  cause: error,
279
- isRetryable: true,
289
+ isRetryable: false,
280
290
  }));
281
291
  }
282
292
  return error;
283
293
  }
284
294
  const directed = retryDirective(error.statusCode, error.responseHeaders ?? {});
285
- // A 2xx APICallError with no explicit retry directive is a provider
286
- // invalid-response: the exchange succeeded and the body was unusable (a
287
- // serializer hiccup, a truncated frame). That is the same transient class as
288
- // a transport failure and consumes the same bounded retry budget; an explicit
289
- // `x-should-retry` directive outranks this default, and the terminal
290
- // classification after the budget stays the non-retryable 502 (#446).
291
- if (directed === null
292
- && typeof error.statusCode === "number" && error.statusCode >= 200 && error.statusCode < 300
293
- && error.isRetryable !== true) {
294
- return retainStreamFailureValues(error, new APICallError({
295
- message: error.message,
296
- url: error.url,
297
- requestBodyValues: error.requestBodyValues,
298
- statusCode: error.statusCode,
299
- responseHeaders: error.responseHeaders,
300
- responseBody: error.responseBody,
301
- cause: error,
302
- isRetryable: true,
303
- }));
304
- }
305
- if (directed === null || directed === error.isRetryable) return error;
295
+ // {§provider-connectivity} (#479): without an explicit directive the only
296
+ // transport-retryable failures are the provider-directed waits — a 429, or
297
+ // any status carrying Retry-After; those live in headers the engine never
298
+ // sees. Every other failure — a bare 408/409/5xx, a network error, and the
299
+ // 2xx invalid-response #446 once promoted — surfaces at once;
300
+ // {§provider-recovery} owns re-issue.
301
+ const policy = directed
302
+ ?? (error.statusCode === 429 || retryAfterPresent(error.responseHeaders ?? {}));
303
+ if (policy === error.isRetryable) return error;
306
304
  return retainStreamFailureValues(error, new APICallError({
307
305
  message: error.message,
308
306
  url: error.url,
@@ -311,7 +309,7 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
311
309
  responseHeaders: error.responseHeaders,
312
310
  responseBody: error.responseBody,
313
311
  cause: error,
314
- isRetryable: directed,
312
+ isRetryable: policy,
315
313
  data: error.data,
316
314
  }));
317
315
  };
@@ -330,7 +328,7 @@ const executeModel = async (
330
328
  url: "model:generation",
331
329
  requestBodyValues: {},
332
330
  cause: timeout,
333
- isRetryable: true,
331
+ isRetryable: false,
334
332
  });
335
333
  }
336
334
  };
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import assert from "node:assert/strict";
3
- import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
3
+ import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, flexedResponseMax } from "./capacity.ts";
4
4
 
5
5
  test("effective output budget is caller-tightenable and physically capped", () => {
6
6
  assert.equal(effectiveOutputBudget({
@@ -90,3 +90,35 @@ test("only exact overflow rejects before provider I/O", () => {
90
90
  measurement: { kind: "upper_bound", tokens: 60, source: "bound" },
91
91
  }).decision, "admit");
92
92
  });
93
+
94
+
95
+ test("(#482) flexedResponseMax harvests exact slack above the floor", () => {
96
+ assert.equal(
97
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
98
+ 47_644,
99
+ "small prompt: the window remainder minus margin",
100
+ );
101
+ assert.equal(
102
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 40_000, margin: 256 }),
103
+ 8_000,
104
+ "a full packet keeps the guaranteed floor even when margin eats the slack",
105
+ );
106
+ assert.equal(
107
+ flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: 16_000, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
108
+ 16_000,
109
+ "the model's own output cap bounds the harvest",
110
+ );
111
+ assert.equal(
112
+ flexedResponseMax({ contextWindow: null, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
113
+ 8_000,
114
+ "no window, no flex",
115
+ );
116
+ });
117
+
118
+ test("(#482) assessRequestCapacity flexes only exact measurements", () => {
119
+ const base = { contextWindow: 48_000, maxInputTokens: null, maxOutputTokens: null, outputBudget: 8_000, reasoningBudget: null };
120
+ const exact = assessRequestCapacity({ ...base, measurement: { kind: "exact", tokens: 1_000, source: "t" } });
121
+ assert.equal(exact.responseMax, 48_000 - 1_000 - 256, "exact prompts harvest the slack");
122
+ const estimate = assessRequestCapacity({ ...base, measurement: { kind: "estimate", tokens: 1_000, source: "t", detail: "chars/2 test estimate" } });
123
+ assert.equal(estimate.responseMax, 8_000, "estimates keep the floor — they prove nothing about the remainder");
124
+ });
package/src/capacity.ts CHANGED
@@ -81,6 +81,34 @@ export const effectiveInputCapacity = ({
81
81
  return capacities.length === 0 ? null : Math.min(...capacities);
82
82
  };
83
83
 
84
+ // {§provider-flexed-allowance} (#482): the configured output budget is the floor
85
+ // curation packed the input against; window room the actual prompt left
86
+ // unclaimed is guaranteed free and becomes response runway. Only an exact
87
+ // prompt measurement may claim slack — an estimate proves nothing about the
88
+ // true remainder — and the model's own maxOutputTokens still caps the grant.
89
+ export const WIRE_FLEX_MARGIN = 256;
90
+
91
+ export const flexedResponseMax = ({
92
+ contextWindow,
93
+ maxOutputTokens,
94
+ outputBudget,
95
+ promptTokens,
96
+ margin,
97
+ }: {
98
+ contextWindow: number | null;
99
+ maxOutputTokens: number | null;
100
+ outputBudget: number | null;
101
+ promptTokens: number;
102
+ margin: number;
103
+ }): number | null => {
104
+ if (outputBudget === null || contextWindow === null) return outputBudget;
105
+ if (!Number.isSafeInteger(promptTokens) || promptTokens < 0) {
106
+ throw new TypeError("promptTokens must be a non-negative safe integer");
107
+ }
108
+ const flexed = Math.max(outputBudget, contextWindow - promptTokens - margin);
109
+ return maxOutputTokens === null ? flexed : Math.min(flexed, Math.max(outputBudget, maxOutputTokens));
110
+ };
111
+
84
112
  export const requestCapacityDecision = (
85
113
  inputCapacity: number | null,
86
114
  measurement: PromptTokenMeasurement,
@@ -126,6 +154,11 @@ export const assessRequestCapacity = ({
126
154
  }
127
155
  const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
128
156
  const inputCapacity = effectiveInputCapacity({ contextWindow, maxInputTokens, outputBudget });
157
+ // {§provider-flexed-allowance}: exact measurements harvest the slack; every
158
+ // other measurement kind keeps the floor.
159
+ const responseMax = prompt.kind === "exact"
160
+ ? flexedResponseMax({ contextWindow, maxOutputTokens, outputBudget, promptTokens: prompt.tokens, margin: WIRE_FLEX_MARGIN })
161
+ : outputBudget;
129
162
 
130
163
  return {
131
164
  decision: requestCapacityDecision(inputCapacity, prompt),
@@ -135,6 +168,7 @@ export const assessRequestCapacity = ({
135
168
  outputBudget,
136
169
  reasoningBudget,
137
170
  inputCapacity,
171
+ responseMax,
138
172
  prompt,
139
173
  };
140
174
  };
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
3
3
  import { once } from "node:events";
4
4
  import { createServer } from "node:http";
5
5
  import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
6
+ import { withProviderDefaults } from "./defaults.ts";
6
7
  import type { LanguageModel } from "ai";
7
8
  import { resetEmittedWarnings } from "./warnings.ts";
8
9
 
@@ -41,6 +42,27 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
41
42
  assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
42
43
  });
43
44
 
45
+ test("(#472) an effort refusal names the operator's declaration lever with the exact key", () => {
46
+ assert.throws(
47
+ () => catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
48
+ ...env,
49
+ FIREWORKS_API_KEY: "test-key",
50
+ PLURNK_PROVIDERS_REASONING: "low",
51
+ }), "accounts/fireworks/models/glm-5p3-flash"),
52
+ /declare it: PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS=low/,
53
+ "the refusal is actionable: it names the exact env declaration",
54
+ );
55
+ // And the lever works: the same route with the declaration constructs.
56
+ const declared = catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
57
+ ...env,
58
+ FIREWORKS_API_KEY: "test-key",
59
+ PLURNK_PROVIDERS_REASONING: "low",
60
+ PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
61
+ }), "accounts/fireworks/models/glm-5p3-flash");
62
+ assert.ok(declared, "the declared effort admits the route");
63
+ assert.ok(declared.supportedReasoningPolicies.includes("low"), "low is admitted through the operator's declaration");
64
+ });
65
+
44
66
  test("provider adapters advertise only reasoning policies they can preserve", () => {
45
67
  const deepseek = catalogProviderFromEnv("deepseek", {
46
68
  ...env,
@@ -48,7 +70,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
48
70
  PLURNK_PROVIDERS_REASONING: "adaptive",
49
71
  PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
50
72
  }, "deepseek-v4-flash");
51
- assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high"]);
73
+ assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high", "max"]);
52
74
 
53
75
  assert.throws(
54
76
  () => catalogProviderFromEnv("deepseek", {
@@ -72,7 +94,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
72
94
  XAI_API_KEY: "test-key",
73
95
  PLURNK_PROVIDERS_REASONING: "adaptive",
74
96
  }, "grok-4.6");
75
- assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Grok 4.6 cannot disable reasoning");
97
+ assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high", "xhigh"], "Grok 4.6 cannot disable reasoning");
76
98
 
77
99
  const gemini = catalogProviderFromEnv("google", {
78
100
  ...env,
@@ -107,7 +129,7 @@ test("Models.dev controls Cloudflare's exact effort vocabulary", async () => {
107
129
  ...cloudflareEnv,
108
130
  PLURNK_PROVIDERS_REASONING: "low",
109
131
  }, "@cf/qwen/qwen3.8-27b");
110
- assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium"]);
132
+ assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium", "xhigh"]);
111
133
  await low?.generate({ workerId: "cloudflare-low", messages: [{ role: "user", content: "hello" }] });
112
134
 
113
135
  const adaptive = catalogProviderFromEnv("cloudflare-workers-ai", {
@@ -178,7 +200,7 @@ test("an operator-declared effort vocabulary extends Models.dev's for a provider
178
200
  ...declaredEnv,
179
201
  PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_EFFORTS: "none,high",
180
202
  }, "@cf/qwen/qwen3.8-27b");
181
- assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high"]);
203
+ assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high", "xhigh"]);
182
204
  // The declaration never turns a non-reasoning route into a reasoning one.
183
205
  const nonReasoning = catalogProviderFromEnv("cloudflare-workers-ai", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/ibm-granite/granite-4.0-h-micro");
184
206
  assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
@@ -848,6 +870,6 @@ test("(#458) declared efforts union into the supported set under the models.dev-
848
870
  PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_STYLE: "effort_explicit",
849
871
  PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
850
872
  }, "accounts/fireworks/models/glm-5p3-flash");
851
- // "max" stays outside the portable wire vocabulary; "off" requires a declared "none".
852
- assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high"]);
873
+ // (#474) "max" joined the portable vocabulary; "off" still requires a declared "none".
874
+ assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high", "max"]);
853
875
  });
@@ -10,6 +10,7 @@ import {
10
10
  effectiveContextWindow,
11
11
  dataCaptureFromEnv,
12
12
  generationEnvelopeFromEnv,
13
+ parseOptionalFloat,
13
14
  parseRequiredFloat,
14
15
  parseRequiredInt,
15
16
  parseTimeoutMs,
@@ -134,7 +135,8 @@ const catalogSupportedReasoningPolicies = ({
134
135
  || (catalogSupportsToggle(info) && compatibleToggleStyles.has(style));
135
136
  return REASONING_POLICIES.filter((policy) => policy === "adaptive"
136
137
  || policy === "off" && off
137
- || (policy === "low" || policy === "medium" || policy === "high")
138
+ || (policy === "low" || policy === "medium" || policy === "high"
139
+ || policy === "xhigh" || policy === "max")
138
140
  && effortTransport
139
141
  && efforts.has(policy));
140
142
  };
@@ -153,6 +155,10 @@ const supportedReasoningPolicies = ({
153
155
  style: ReasoningStyle;
154
156
  declared: readonly ModelReasoningEffort[];
155
157
  }): readonly ReasoningPolicy[] => {
158
+ // The template style is the operator's declaration that the rail's own chat
159
+ // template governs reasoning: a fixed effort rides in verbatim, native SDK or
160
+ // not, and a word the template does not know fails loudly on the first request.
161
+ if (style === "template") return REASONING_POLICIES;
156
162
  if (info !== undefined && info.reasoning !== true) return activationPolicies;
157
163
  if (info?.reasoningOptions !== undefined) {
158
164
  return catalogSupportedReasoningPolicies({ info, native, style, declared });
@@ -355,8 +361,8 @@ export const providerFromSdkModel = ({
355
361
  streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
356
362
  reasoning,
357
363
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
358
- temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
359
- repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
364
+ temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
365
+ repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
360
366
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
361
367
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
362
368
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
@@ -83,6 +83,18 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
83
83
  assert.equal("dry_allowed_length" in (body ?? {}), false);
84
84
  });
85
85
 
86
+ test("(#483) a detected llama-server rail admits the operator's stated effort", async () => {
87
+ mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
88
+ const url = String(input);
89
+ if (url.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }] }));
90
+ if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
91
+ throw new Error(`unexpected request ${url}`);
92
+ });
93
+ const provider = await compatibleProviderFromEnv("openai", { ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
94
+ assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
95
+ assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
96
+ });
97
+
86
98
  test("detected llama-server measures the complete chat request through input_tokens", async () => {
87
99
  let countUrl: string | undefined;
88
100
  let countBody: Record<string, unknown> | undefined;