@plurnk/plurnk-providers 1.14.0 → 1.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +21 -9
- package/SPEC.md +54 -12
- package/dist/AiSdkProvider.d.ts +2 -2
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +36 -15
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +2 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +29 -32
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +8 -0
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +21 -0
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +10 -4
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +6 -3
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +23 -2
- package/dist/errors.js.map +1 -1
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +6 -0
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +1 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +9 -1
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +138 -112
- package/src/AiSdkProvider.ts +41 -20
- package/src/Pool.test.ts +2 -0
- package/src/Pool.ts +2 -0
- package/src/aiSdkTransport.test.ts +31 -16
- package/src/aiSdkTransport.ts +32 -34
- package/src/capacity.test.ts +33 -1
- package/src/capacity.ts +34 -0
- package/src/catalogProvider.test.ts +28 -6
- package/src/catalogProvider.ts +9 -3
- package/src/compatibleProvider.test.ts +12 -0
- package/src/compatibleProvider.ts +6 -3
- package/src/env.test.ts +4 -4
- package/src/errors.test.ts +1 -0
- package/src/errors.ts +22 -1
- package/src/sdkModels.ts +6 -0
- package/src/types.ts +12 -1
package/src/AiSdkProvider.ts
CHANGED
|
@@ -173,8 +173,8 @@ export type AiSdkProviderConfig = {
|
|
|
173
173
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
174
174
|
// (greedy-under-mask loops without it) — the VALUE is operator config;
|
|
175
175
|
// WHERE it applies stays mechanism.
|
|
176
|
-
temperature: number;
|
|
177
|
-
repeatPenalty: number;
|
|
176
|
+
temperature: number | null;
|
|
177
|
+
repeatPenalty: number | null;
|
|
178
178
|
// Anti-degeneration guard on the cloud path (grammarStyle "none"), where the
|
|
179
179
|
// repeat_penalty multiplier isn't available - the OpenAI-standard frequency_penalty.
|
|
180
180
|
// Optional (default 0 = off) so an out-of-date plugin that omits it just runs unguarded
|
|
@@ -310,11 +310,19 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
|
310
310
|
return { content, reasoning: "", projected: false, contentStart: 0 };
|
|
311
311
|
};
|
|
312
312
|
|
|
313
|
-
const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" => {
|
|
314
|
-
if (mode === "low" || mode === "medium" || mode === "high") return mode;
|
|
313
|
+
const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" | "max" => {
|
|
314
|
+
if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max") return mode;
|
|
315
315
|
throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
|
|
316
316
|
};
|
|
317
317
|
|
|
318
|
+
// The native SDK effort surface tops at xhigh; admission never grants a native
|
|
319
|
+
// route "max", so reaching it here is a contract violation, not a fallback site.
|
|
320
|
+
const nativeFixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" | "xhigh" => {
|
|
321
|
+
const effort = fixedEffort(mode);
|
|
322
|
+
if (effort === "max") throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
|
|
323
|
+
return effort;
|
|
324
|
+
};
|
|
325
|
+
|
|
318
326
|
// Anthropic's older manual-reasoning protocol needs an absolute allowance while
|
|
319
327
|
// PLURNK's durable contract names an effort. These fractions match the native
|
|
320
328
|
// SDK's policy projection, but apply to PLURNK's total envelope rather than the
|
|
@@ -324,6 +332,8 @@ const MANUAL_REASONING_FRACTIONS = Object.freeze({
|
|
|
324
332
|
low: 0.1,
|
|
325
333
|
medium: 0.3,
|
|
326
334
|
high: 0.6,
|
|
335
|
+
xhigh: 0.75,
|
|
336
|
+
max: 0.85,
|
|
327
337
|
} satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
|
|
328
338
|
const MANUAL_REASONING_MINIMUM = 1024;
|
|
329
339
|
|
|
@@ -385,8 +395,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
385
395
|
#compatibleAdaptiveReasoning: CompatibleReasoningEffort | "provider-default";
|
|
386
396
|
#compatibleOffReasoning: "none" | undefined;
|
|
387
397
|
#adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
|
|
388
|
-
#temperature: number;
|
|
389
|
-
#repeatPenalty: number;
|
|
398
|
+
#temperature: number | null;
|
|
399
|
+
#repeatPenalty: number | null;
|
|
390
400
|
#frequencyPenalty: number;
|
|
391
401
|
#dryMultiplier: number | undefined;
|
|
392
402
|
#dryBase: number | undefined;
|
|
@@ -472,8 +482,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
472
482
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
473
483
|
// required tuning fields must fail at construction, not silently send
|
|
474
484
|
// undefined sampling on every grammar request.
|
|
475
|
-
if (
|
|
476
|
-
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
|
|
485
|
+
if (config.temperature === undefined || config.repeatPenalty === undefined) {
|
|
486
|
+
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty declared (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY; null = provider default)`);
|
|
477
487
|
}
|
|
478
488
|
this.#temperature = config.temperature;
|
|
479
489
|
this.#repeatPenalty = config.repeatPenalty;
|
|
@@ -708,8 +718,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
708
718
|
const allowance = mode === "off"
|
|
709
719
|
? 0
|
|
710
720
|
: budget;
|
|
721
|
+
// A fixed effort rides into the template as its own variable; adaptive
|
|
722
|
+
// and off send none and leave the template's default in force.
|
|
723
|
+
const templateEffort = mode === "off" || mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
|
|
711
724
|
return {
|
|
712
|
-
chat_template_kwargs: { enable_thinking: on },
|
|
725
|
+
chat_template_kwargs: { enable_thinking: on, ...templateEffort },
|
|
713
726
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
714
727
|
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
715
728
|
};
|
|
@@ -803,9 +816,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
803
816
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
804
817
|
if (grammar === undefined) return {};
|
|
805
818
|
switch (this.#grammarStyle) {
|
|
806
|
-
//
|
|
807
|
-
//
|
|
808
|
-
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
819
|
+
// Grammar-constrained decoding can loop under the mask; a configured
|
|
820
|
+
// per-alias repeat_penalty is the measured remedy ({§provider-sampling-passthrough}).
|
|
821
|
+
case "llamacpp": return { grammar, ...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}) };
|
|
809
822
|
case "none": return {};
|
|
810
823
|
}
|
|
811
824
|
}
|
|
@@ -825,7 +838,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
825
838
|
// repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
|
|
826
839
|
// Each rides only when its operator knob is set; absent = the box's default.
|
|
827
840
|
case "llamacpp": return {
|
|
828
|
-
repeat_penalty: this.#repeatPenalty,
|
|
841
|
+
...(this.#repeatPenalty !== null ? { repeat_penalty: this.#repeatPenalty } : {}),
|
|
829
842
|
...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
|
|
830
843
|
...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
|
|
831
844
|
dry_multiplier: this.#dryMultiplier,
|
|
@@ -1023,7 +1036,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
1023
1036
|
{ capacity, extensions: { capacityStage: "preflight", capacity } },
|
|
1024
1037
|
);
|
|
1025
1038
|
}
|
|
1026
|
-
|
|
1039
|
+
// {§provider-flexed-allowance} (#482): the wire grants the flexed
|
|
1040
|
+
// allowance — the floor, or the exactly-measured slack above it.
|
|
1041
|
+
const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
|
|
1027
1042
|
const nativeReasoningBudget = this.#nativeReasoningBudget(
|
|
1028
1043
|
capacity.outputBudget,
|
|
1029
1044
|
capacity.reasoningBudget,
|
|
@@ -1036,7 +1051,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1036
1051
|
const body: Record<string, unknown> = {
|
|
1037
1052
|
// Floors are suppressed on router-owned-tuning providers (plurnk) —
|
|
1038
1053
|
// the router's per-model tuning must not be overridden by client floors.
|
|
1039
|
-
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
1054
|
+
...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#repetitionPenaltyBody() } : {}),
|
|
1040
1055
|
...this.#samplingBody(sampling),
|
|
1041
1056
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
1042
1057
|
model: this.#model,
|
|
@@ -1172,7 +1187,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1172
1187
|
captureRawBody: this.#rawBody,
|
|
1173
1188
|
...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
|
|
1174
1189
|
temperature: this.#tuningFloors
|
|
1175
|
-
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
1190
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature ?? undefined)
|
|
1176
1191
|
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
1177
1192
|
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
1178
1193
|
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
@@ -1191,7 +1206,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
1191
1206
|
? "none"
|
|
1192
1207
|
: this.#reasoning.mode === "adaptive"
|
|
1193
1208
|
? this.#adaptiveReasoning
|
|
1194
|
-
:
|
|
1209
|
+
: nativeFixedEffort(this.#reasoning.mode),
|
|
1195
1210
|
});
|
|
1196
1211
|
} catch (error) {
|
|
1197
1212
|
if (transportFailureOutputObserved(error)) recoveredAfterOutput = true;
|
|
@@ -1387,9 +1402,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
1387
1402
|
...(meta !== undefined ? { meta } : {}),
|
|
1388
1403
|
...(notices !== undefined ? { notices } : {}),
|
|
1389
1404
|
};
|
|
1390
|
-
|
|
1405
|
+
// {§provider-flexed-allowance} (#482): conformance judges the GRANT the
|
|
1406
|
+
// wire actually sent, not the configured floor — output between the two
|
|
1407
|
+
// is overflow tolerance working, never a provider fault. Run7 loop-death
|
|
1408
|
+
// was this guard still holding the floor after the flex landed.
|
|
1409
|
+
const grantedOutput = capacity.responseMax ?? capacity.outputBudget;
|
|
1410
|
+
if (grantedOutput !== null
|
|
1391
1411
|
&& usage?.outputTokens !== undefined
|
|
1392
|
-
&& usage.outputTokens >
|
|
1412
|
+
&& usage.outputTokens > grantedOutput) {
|
|
1393
1413
|
const attempt: ProviderAttempt = {
|
|
1394
1414
|
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
1395
1415
|
...evidence,
|
|
@@ -1397,13 +1417,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
1397
1417
|
throw new ProviderError(
|
|
1398
1418
|
this.#source,
|
|
1399
1419
|
"invalid_response",
|
|
1400
|
-
`The provider reported ${usage.outputTokens} output tokens after receiving a
|
|
1420
|
+
`The provider reported ${usage.outputTokens} output tokens after receiving a granted output allowance of ${grantedOutput}.`,
|
|
1401
1421
|
{
|
|
1402
1422
|
attempt,
|
|
1403
1423
|
accounting,
|
|
1404
1424
|
extensions: {
|
|
1405
1425
|
stage: "provider-response",
|
|
1406
1426
|
outputBudget: capacity.outputBudget,
|
|
1427
|
+
grantedOutput,
|
|
1407
1428
|
reportedOutputTokens: usage.outputTokens,
|
|
1408
1429
|
},
|
|
1409
1430
|
},
|
package/src/Pool.test.ts
CHANGED
|
@@ -31,6 +31,7 @@ const RESP: ProviderResponse = {
|
|
|
31
31
|
outputBudget: 12_000,
|
|
32
32
|
reasoningBudget: null,
|
|
33
33
|
inputCapacity: 36_000,
|
|
34
|
+
responseMax: 12_000,
|
|
34
35
|
prompt: { kind: "exact", tokens: 0, source: "test:exact" },
|
|
35
36
|
},
|
|
36
37
|
};
|
|
@@ -76,6 +77,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
76
77
|
outputBudget: maxOutputTokens ?? opts.outputBudget ?? null,
|
|
77
78
|
reasoningBudget: opts.reasoningBudget ?? null,
|
|
78
79
|
inputCapacity: null,
|
|
80
|
+
responseMax: maxOutputTokens ?? opts.outputBudget ?? null,
|
|
79
81
|
prompt: opts.promptMeasurement ?? {
|
|
80
82
|
kind: "exact",
|
|
81
83
|
tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
|
package/src/Pool.ts
CHANGED
|
@@ -190,6 +190,8 @@ export default class Pool implements Provider {
|
|
|
190
190
|
outputBudget: minimum(envelopes.map((envelope) => envelope.outputBudget)),
|
|
191
191
|
reasoningBudget: minimum(envelopes.map((envelope) => envelope.reasoningBudget)),
|
|
192
192
|
inputCapacity,
|
|
193
|
+
// A pool spans members whose windows differ; it never flexes ({§provider-flexed-allowance}).
|
|
194
|
+
responseMax: minimum(envelopes.map((envelope) => envelope.outputBudget)),
|
|
193
195
|
prompt: measurement,
|
|
194
196
|
};
|
|
195
197
|
}
|
|
@@ -289,19 +289,19 @@ test("the adapter preserves nonstandard reasoning accounting after SDK parsing",
|
|
|
289
289
|
});
|
|
290
290
|
});
|
|
291
291
|
|
|
292
|
-
test("normalizeRetryAttemptError —
|
|
292
|
+
test("normalizeRetryAttemptError — deadlines surface at once, never transport-retried ({§provider-connectivity}, #479)", () => {
|
|
293
293
|
const first = normalizeRetryAttemptError(new ProviderTimeoutError("first_content", 180000));
|
|
294
294
|
assert.equal(APICallError.isInstance(first), true);
|
|
295
|
-
assert.equal((first as APICallError).isRetryable,
|
|
295
|
+
assert.equal((first as APICallError).isRetryable, false);
|
|
296
296
|
const attempt = normalizeRetryAttemptError(new ProviderTimeoutError("attempt", 60000));
|
|
297
|
-
assert.equal((attempt as APICallError).isRetryable,
|
|
297
|
+
assert.equal((attempt as APICallError).isRetryable, false);
|
|
298
298
|
const idle = normalizeRetryAttemptError(new ProviderTimeoutError("stream_idle", 120000));
|
|
299
|
-
assert.equal((idle as APICallError).isRetryable,
|
|
299
|
+
assert.equal((idle as APICallError).isRetryable, false);
|
|
300
300
|
const operation = new ProviderTimeoutError("operation", 2700000);
|
|
301
301
|
assert.equal(normalizeRetryAttemptError(operation), operation);
|
|
302
302
|
});
|
|
303
303
|
|
|
304
|
-
test("normalizeRetryAttemptError —
|
|
304
|
+
test("normalizeRetryAttemptError — only provider-directed waits retry: 429, Retry-After, or a directive (#479 supersedes #446)", () => {
|
|
305
305
|
const garbage = new APICallError({
|
|
306
306
|
message: "Failed to process successful response",
|
|
307
307
|
url: "https://api.example/v1/chat/completions",
|
|
@@ -310,28 +310,43 @@ test("normalizeRetryAttemptError — a 2xx APICallError without a directive beco
|
|
|
310
310
|
responseBody: "not json",
|
|
311
311
|
isRetryable: false,
|
|
312
312
|
});
|
|
313
|
-
|
|
314
|
-
assert.ok(APICallError.isInstance(normalized));
|
|
315
|
-
assert.equal(normalized.isRetryable, true, "2xx invalid-response consumes the retry budget");
|
|
316
|
-
assert.equal(normalized.cause, garbage, "the original failure rides as the cause");
|
|
317
|
-
assert.equal(normalized.statusCode, 200);
|
|
313
|
+
assert.equal(normalizeRetryAttemptError(garbage), garbage, "2xx invalid-response surfaces at once — no promoted budget (#446 superseded)");
|
|
318
314
|
|
|
319
315
|
const directed = new APICallError({
|
|
320
316
|
message: "Failed to process successful response",
|
|
321
317
|
url: "https://api.example/v1/chat/completions",
|
|
322
318
|
requestBodyValues: {},
|
|
323
319
|
statusCode: 200,
|
|
324
|
-
responseHeaders: { "x-should-retry": "
|
|
320
|
+
responseHeaders: { "x-should-retry": "true" },
|
|
325
321
|
isRetryable: false,
|
|
326
322
|
});
|
|
327
|
-
assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable,
|
|
323
|
+
assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, true, "an explicit directive still outranks");
|
|
328
324
|
|
|
329
|
-
const
|
|
330
|
-
message: "
|
|
325
|
+
const rateLimited = new APICallError({
|
|
326
|
+
message: "slow down",
|
|
331
327
|
url: "https://api.example/v1/chat/completions",
|
|
332
328
|
requestBodyValues: {},
|
|
333
|
-
statusCode:
|
|
329
|
+
statusCode: 429,
|
|
334
330
|
isRetryable: false,
|
|
335
331
|
});
|
|
336
|
-
assert.equal(normalizeRetryAttemptError(
|
|
332
|
+
assert.equal((normalizeRetryAttemptError(rateLimited) as APICallError).isRetryable, true, "a 429 is the provider-directed wait");
|
|
333
|
+
|
|
334
|
+
const directedWait = new APICallError({
|
|
335
|
+
message: "maintenance",
|
|
336
|
+
url: "https://api.example/v1/chat/completions",
|
|
337
|
+
requestBodyValues: {},
|
|
338
|
+
statusCode: 503,
|
|
339
|
+
responseHeaders: { "retry-after": "1" },
|
|
340
|
+
isRetryable: true,
|
|
341
|
+
});
|
|
342
|
+
assert.equal((normalizeRetryAttemptError(directedWait) as APICallError).isRetryable, true, "Retry-After on any status is a directed wait");
|
|
343
|
+
|
|
344
|
+
const bareServerError = new APICallError({
|
|
345
|
+
message: "internal error",
|
|
346
|
+
url: "https://api.example/v1/chat/completions",
|
|
347
|
+
requestBodyValues: {},
|
|
348
|
+
statusCode: 503,
|
|
349
|
+
isRetryable: true,
|
|
350
|
+
});
|
|
351
|
+
assert.equal((normalizeRetryAttemptError(bareServerError) as APICallError).isRetryable, false, "a bare 5xx surfaces at once for the engine's recovery");
|
|
337
352
|
});
|
package/src/aiSdkTransport.ts
CHANGED
|
@@ -29,16 +29,26 @@ const retryDirective = (
|
|
|
29
29
|
return null;
|
|
30
30
|
};
|
|
31
31
|
|
|
32
|
+
// {§provider-connectivity} (#479): a Retry-After header on any status is the
|
|
33
|
+
// provider directing a wait (RFC 9110 defines it on 503 exactly for this);
|
|
34
|
+
// its presence, like a bare 429, earns the bounded transport retry.
|
|
35
|
+
const retryAfterPresent = (
|
|
36
|
+
headers: Headers | Readonly<Record<string, string>>,
|
|
37
|
+
): boolean => {
|
|
38
|
+
const raw = headers instanceof Headers
|
|
39
|
+
? headers.get("retry-after")
|
|
40
|
+
: Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.[1];
|
|
41
|
+
return raw !== undefined && raw !== null && raw.trim() !== "";
|
|
42
|
+
};
|
|
43
|
+
|
|
32
44
|
const errorStructure: ProviderErrorStructure<z.infer<typeof errorSchema>> = {
|
|
33
45
|
errorSchema,
|
|
34
46
|
errorToMessage: ({ error }) => error.message,
|
|
35
47
|
isRetryable(response) {
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|| response.status >= 500
|
|
41
|
-
);
|
|
48
|
+
// {§provider-connectivity} (#479): only a provider-directed wait — 429,
|
|
49
|
+
// a Retry-After, or an explicit X-Should-Retry — earns a transport retry.
|
|
50
|
+
return retryDirective(response.status, response.headers)
|
|
51
|
+
?? (response.status === 429 || retryAfterPresent(response.headers));
|
|
42
52
|
},
|
|
43
53
|
};
|
|
44
54
|
|
|
@@ -254,16 +264,16 @@ export const transportFailureOutputObserved = (error: unknown): boolean => {
|
|
|
254
264
|
|
|
255
265
|
export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
256
266
|
if (!APICallError.isInstance(error)) {
|
|
257
|
-
// Attempt, first-content, and stream-idle deadlines
|
|
258
|
-
//
|
|
259
|
-
//
|
|
267
|
+
// Attempt, first-content, and stream-idle deadlines surface on the first
|
|
268
|
+
// failure ({§provider-connectivity}, #479): the engine's {§provider-recovery}
|
|
269
|
+
// owns re-issue with backoff and park; the stall is reported, never swallowed.
|
|
260
270
|
if (error instanceof ProviderTimeoutError && error.phase !== "operation") {
|
|
261
271
|
return retainStreamFailureValues(error, new APICallError({
|
|
262
272
|
message: error.message,
|
|
263
273
|
url: "model:generation",
|
|
264
274
|
requestBodyValues: {},
|
|
265
275
|
cause: error,
|
|
266
|
-
isRetryable:
|
|
276
|
+
isRetryable: false,
|
|
267
277
|
}));
|
|
268
278
|
}
|
|
269
279
|
// Node's Undici stream reader reports a peer-aborted HTTP/2 body as this
|
|
@@ -276,33 +286,21 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
|
276
286
|
url: "model:generation",
|
|
277
287
|
requestBodyValues: {},
|
|
278
288
|
cause: error,
|
|
279
|
-
isRetryable:
|
|
289
|
+
isRetryable: false,
|
|
280
290
|
}));
|
|
281
291
|
}
|
|
282
292
|
return error;
|
|
283
293
|
}
|
|
284
294
|
const directed = retryDirective(error.statusCode, error.responseHeaders ?? {});
|
|
285
|
-
//
|
|
286
|
-
//
|
|
287
|
-
//
|
|
288
|
-
//
|
|
289
|
-
//
|
|
290
|
-
//
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
return retainStreamFailureValues(error, new APICallError({
|
|
295
|
-
message: error.message,
|
|
296
|
-
url: error.url,
|
|
297
|
-
requestBodyValues: error.requestBodyValues,
|
|
298
|
-
statusCode: error.statusCode,
|
|
299
|
-
responseHeaders: error.responseHeaders,
|
|
300
|
-
responseBody: error.responseBody,
|
|
301
|
-
cause: error,
|
|
302
|
-
isRetryable: true,
|
|
303
|
-
}));
|
|
304
|
-
}
|
|
305
|
-
if (directed === null || directed === error.isRetryable) return error;
|
|
295
|
+
// {§provider-connectivity} (#479): without an explicit directive the only
|
|
296
|
+
// transport-retryable failures are the provider-directed waits — a 429, or
|
|
297
|
+
// any status carrying Retry-After; those live in headers the engine never
|
|
298
|
+
// sees. Every other failure — a bare 408/409/5xx, a network error, and the
|
|
299
|
+
// 2xx invalid-response #446 once promoted — surfaces at once;
|
|
300
|
+
// {§provider-recovery} owns re-issue.
|
|
301
|
+
const policy = directed
|
|
302
|
+
?? (error.statusCode === 429 || retryAfterPresent(error.responseHeaders ?? {}));
|
|
303
|
+
if (policy === error.isRetryable) return error;
|
|
306
304
|
return retainStreamFailureValues(error, new APICallError({
|
|
307
305
|
message: error.message,
|
|
308
306
|
url: error.url,
|
|
@@ -311,7 +309,7 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
|
311
309
|
responseHeaders: error.responseHeaders,
|
|
312
310
|
responseBody: error.responseBody,
|
|
313
311
|
cause: error,
|
|
314
|
-
isRetryable:
|
|
312
|
+
isRetryable: policy,
|
|
315
313
|
data: error.data,
|
|
316
314
|
}));
|
|
317
315
|
};
|
|
@@ -330,7 +328,7 @@ const executeModel = async (
|
|
|
330
328
|
url: "model:generation",
|
|
331
329
|
requestBodyValues: {},
|
|
332
330
|
cause: timeout,
|
|
333
|
-
isRetryable:
|
|
331
|
+
isRetryable: false,
|
|
334
332
|
});
|
|
335
333
|
}
|
|
336
334
|
};
|
package/src/capacity.test.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import assert from "node:assert/strict";
|
|
3
|
-
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
3
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, flexedResponseMax } from "./capacity.ts";
|
|
4
4
|
|
|
5
5
|
test("effective output budget is caller-tightenable and physically capped", () => {
|
|
6
6
|
assert.equal(effectiveOutputBudget({
|
|
@@ -90,3 +90,35 @@ test("only exact overflow rejects before provider I/O", () => {
|
|
|
90
90
|
measurement: { kind: "upper_bound", tokens: 60, source: "bound" },
|
|
91
91
|
}).decision, "admit");
|
|
92
92
|
});
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
test("(#482) flexedResponseMax harvests exact slack above the floor", () => {
|
|
96
|
+
assert.equal(
|
|
97
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
98
|
+
47_644,
|
|
99
|
+
"small prompt: the window remainder minus margin",
|
|
100
|
+
);
|
|
101
|
+
assert.equal(
|
|
102
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 40_000, margin: 256 }),
|
|
103
|
+
8_000,
|
|
104
|
+
"a full packet keeps the guaranteed floor even when margin eats the slack",
|
|
105
|
+
);
|
|
106
|
+
assert.equal(
|
|
107
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: 16_000, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
108
|
+
16_000,
|
|
109
|
+
"the model's own output cap bounds the harvest",
|
|
110
|
+
);
|
|
111
|
+
assert.equal(
|
|
112
|
+
flexedResponseMax({ contextWindow: null, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
113
|
+
8_000,
|
|
114
|
+
"no window, no flex",
|
|
115
|
+
);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
test("(#482) assessRequestCapacity flexes only exact measurements", () => {
|
|
119
|
+
const base = { contextWindow: 48_000, maxInputTokens: null, maxOutputTokens: null, outputBudget: 8_000, reasoningBudget: null };
|
|
120
|
+
const exact = assessRequestCapacity({ ...base, measurement: { kind: "exact", tokens: 1_000, source: "t" } });
|
|
121
|
+
assert.equal(exact.responseMax, 48_000 - 1_000 - 256, "exact prompts harvest the slack");
|
|
122
|
+
const estimate = assessRequestCapacity({ ...base, measurement: { kind: "estimate", tokens: 1_000, source: "t", detail: "chars/2 test estimate" } });
|
|
123
|
+
assert.equal(estimate.responseMax, 8_000, "estimates keep the floor — they prove nothing about the remainder");
|
|
124
|
+
});
|
package/src/capacity.ts
CHANGED
|
@@ -81,6 +81,34 @@ export const effectiveInputCapacity = ({
|
|
|
81
81
|
return capacities.length === 0 ? null : Math.min(...capacities);
|
|
82
82
|
};
|
|
83
83
|
|
|
84
|
+
// {§provider-flexed-allowance} (#482): the configured output budget is the floor
|
|
85
|
+
// curation packed the input against; window room the actual prompt left
|
|
86
|
+
// unclaimed is guaranteed free and becomes response runway. Only an exact
|
|
87
|
+
// prompt measurement may claim slack — an estimate proves nothing about the
|
|
88
|
+
// true remainder — and the model's own maxOutputTokens still caps the grant.
|
|
89
|
+
export const WIRE_FLEX_MARGIN = 256;
|
|
90
|
+
|
|
91
|
+
export const flexedResponseMax = ({
|
|
92
|
+
contextWindow,
|
|
93
|
+
maxOutputTokens,
|
|
94
|
+
outputBudget,
|
|
95
|
+
promptTokens,
|
|
96
|
+
margin,
|
|
97
|
+
}: {
|
|
98
|
+
contextWindow: number | null;
|
|
99
|
+
maxOutputTokens: number | null;
|
|
100
|
+
outputBudget: number | null;
|
|
101
|
+
promptTokens: number;
|
|
102
|
+
margin: number;
|
|
103
|
+
}): number | null => {
|
|
104
|
+
if (outputBudget === null || contextWindow === null) return outputBudget;
|
|
105
|
+
if (!Number.isSafeInteger(promptTokens) || promptTokens < 0) {
|
|
106
|
+
throw new TypeError("promptTokens must be a non-negative safe integer");
|
|
107
|
+
}
|
|
108
|
+
const flexed = Math.max(outputBudget, contextWindow - promptTokens - margin);
|
|
109
|
+
return maxOutputTokens === null ? flexed : Math.min(flexed, Math.max(outputBudget, maxOutputTokens));
|
|
110
|
+
};
|
|
111
|
+
|
|
84
112
|
export const requestCapacityDecision = (
|
|
85
113
|
inputCapacity: number | null,
|
|
86
114
|
measurement: PromptTokenMeasurement,
|
|
@@ -126,6 +154,11 @@ export const assessRequestCapacity = ({
|
|
|
126
154
|
}
|
|
127
155
|
const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
|
|
128
156
|
const inputCapacity = effectiveInputCapacity({ contextWindow, maxInputTokens, outputBudget });
|
|
157
|
+
// {§provider-flexed-allowance}: exact measurements harvest the slack; every
|
|
158
|
+
// other measurement kind keeps the floor.
|
|
159
|
+
const responseMax = prompt.kind === "exact"
|
|
160
|
+
? flexedResponseMax({ contextWindow, maxOutputTokens, outputBudget, promptTokens: prompt.tokens, margin: WIRE_FLEX_MARGIN })
|
|
161
|
+
: outputBudget;
|
|
129
162
|
|
|
130
163
|
return {
|
|
131
164
|
decision: requestCapacityDecision(inputCapacity, prompt),
|
|
@@ -135,6 +168,7 @@ export const assessRequestCapacity = ({
|
|
|
135
168
|
outputBudget,
|
|
136
169
|
reasoningBudget,
|
|
137
170
|
inputCapacity,
|
|
171
|
+
responseMax,
|
|
138
172
|
prompt,
|
|
139
173
|
};
|
|
140
174
|
};
|
|
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
|
|
|
3
3
|
import { once } from "node:events";
|
|
4
4
|
import { createServer } from "node:http";
|
|
5
5
|
import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
|
|
6
|
+
import { withProviderDefaults } from "./defaults.ts";
|
|
6
7
|
import type { LanguageModel } from "ai";
|
|
7
8
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
8
9
|
|
|
@@ -41,6 +42,27 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
41
42
|
assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
42
43
|
});
|
|
43
44
|
|
|
45
|
+
test("(#472) an effort refusal names the operator's declaration lever with the exact key", () => {
|
|
46
|
+
assert.throws(
|
|
47
|
+
() => catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
|
|
48
|
+
...env,
|
|
49
|
+
FIREWORKS_API_KEY: "test-key",
|
|
50
|
+
PLURNK_PROVIDERS_REASONING: "low",
|
|
51
|
+
}), "accounts/fireworks/models/glm-5p3-flash"),
|
|
52
|
+
/declare it: PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS=low/,
|
|
53
|
+
"the refusal is actionable: it names the exact env declaration",
|
|
54
|
+
);
|
|
55
|
+
// And the lever works: the same route with the declaration constructs.
|
|
56
|
+
const declared = catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
|
|
57
|
+
...env,
|
|
58
|
+
FIREWORKS_API_KEY: "test-key",
|
|
59
|
+
PLURNK_PROVIDERS_REASONING: "low",
|
|
60
|
+
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
|
|
61
|
+
}), "accounts/fireworks/models/glm-5p3-flash");
|
|
62
|
+
assert.ok(declared, "the declared effort admits the route");
|
|
63
|
+
assert.ok(declared.supportedReasoningPolicies.includes("low"), "low is admitted through the operator's declaration");
|
|
64
|
+
});
|
|
65
|
+
|
|
44
66
|
test("provider adapters advertise only reasoning policies they can preserve", () => {
|
|
45
67
|
const deepseek = catalogProviderFromEnv("deepseek", {
|
|
46
68
|
...env,
|
|
@@ -48,7 +70,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
|
|
|
48
70
|
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
49
71
|
PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
|
|
50
72
|
}, "deepseek-v4-flash");
|
|
51
|
-
assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high"]);
|
|
73
|
+
assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "low", "high", "max"]);
|
|
52
74
|
|
|
53
75
|
assert.throws(
|
|
54
76
|
() => catalogProviderFromEnv("deepseek", {
|
|
@@ -72,7 +94,7 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
|
|
|
72
94
|
XAI_API_KEY: "test-key",
|
|
73
95
|
PLURNK_PROVIDERS_REASONING: "adaptive",
|
|
74
96
|
}, "grok-4.6");
|
|
75
|
-
assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Grok 4.6 cannot disable reasoning");
|
|
97
|
+
assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high", "xhigh"], "Grok 4.6 cannot disable reasoning");
|
|
76
98
|
|
|
77
99
|
const gemini = catalogProviderFromEnv("google", {
|
|
78
100
|
...env,
|
|
@@ -107,7 +129,7 @@ test("Models.dev controls Cloudflare's exact effort vocabulary", async () => {
|
|
|
107
129
|
...cloudflareEnv,
|
|
108
130
|
PLURNK_PROVIDERS_REASONING: "low",
|
|
109
131
|
}, "@cf/qwen/qwen3.8-27b");
|
|
110
|
-
assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium"]);
|
|
132
|
+
assert.deepEqual(low?.supportedReasoningPolicies, ["adaptive", "low", "medium", "xhigh"]);
|
|
111
133
|
await low?.generate({ workerId: "cloudflare-low", messages: [{ role: "user", content: "hello" }] });
|
|
112
134
|
|
|
113
135
|
const adaptive = catalogProviderFromEnv("cloudflare-workers-ai", {
|
|
@@ -178,7 +200,7 @@ test("an operator-declared effort vocabulary extends Models.dev's for a provider
|
|
|
178
200
|
...declaredEnv,
|
|
179
201
|
PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_EFFORTS: "none,high",
|
|
180
202
|
}, "@cf/qwen/qwen3.8-27b");
|
|
181
|
-
assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high"]);
|
|
203
|
+
assert.deepEqual(withOff?.supportedReasoningPolicies, ["off", "adaptive", "low", "medium", "high", "xhigh"]);
|
|
182
204
|
// The declaration never turns a non-reasoning route into a reasoning one.
|
|
183
205
|
const nonReasoning = catalogProviderFromEnv("cloudflare-workers-ai", { ...declaredEnv, PLURNK_PROVIDERS_REASONING: "adaptive" }, "@cf/ibm-granite/granite-4.0-h-micro");
|
|
184
206
|
assert.deepEqual(nonReasoning?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
@@ -848,6 +870,6 @@ test("(#458) declared efforts union into the supported set under the models.dev-
|
|
|
848
870
|
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_STYLE: "effort_explicit",
|
|
849
871
|
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
|
|
850
872
|
}, "accounts/fireworks/models/glm-5p3-flash");
|
|
851
|
-
// "max"
|
|
852
|
-
assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high"]);
|
|
873
|
+
// (#474) "max" joined the portable vocabulary; "off" still requires a declared "none".
|
|
874
|
+
assert.deepEqual(provider?.supportedReasoningPolicies, ["adaptive", "low", "high", "max"]);
|
|
853
875
|
});
|
package/src/catalogProvider.ts
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
effectiveContextWindow,
|
|
11
11
|
dataCaptureFromEnv,
|
|
12
12
|
generationEnvelopeFromEnv,
|
|
13
|
+
parseOptionalFloat,
|
|
13
14
|
parseRequiredFloat,
|
|
14
15
|
parseRequiredInt,
|
|
15
16
|
parseTimeoutMs,
|
|
@@ -134,7 +135,8 @@ const catalogSupportedReasoningPolicies = ({
|
|
|
134
135
|
|| (catalogSupportsToggle(info) && compatibleToggleStyles.has(style));
|
|
135
136
|
return REASONING_POLICIES.filter((policy) => policy === "adaptive"
|
|
136
137
|
|| policy === "off" && off
|
|
137
|
-
|| (policy === "low" || policy === "medium" || policy === "high"
|
|
138
|
+
|| (policy === "low" || policy === "medium" || policy === "high"
|
|
139
|
+
|| policy === "xhigh" || policy === "max")
|
|
138
140
|
&& effortTransport
|
|
139
141
|
&& efforts.has(policy));
|
|
140
142
|
};
|
|
@@ -153,6 +155,10 @@ const supportedReasoningPolicies = ({
|
|
|
153
155
|
style: ReasoningStyle;
|
|
154
156
|
declared: readonly ModelReasoningEffort[];
|
|
155
157
|
}): readonly ReasoningPolicy[] => {
|
|
158
|
+
// The template style is the operator's declaration that the rail's own chat
|
|
159
|
+
// template governs reasoning: a fixed effort rides in verbatim, native SDK or
|
|
160
|
+
// not, and a word the template does not know fails loudly on the first request.
|
|
161
|
+
if (style === "template") return REASONING_POLICIES;
|
|
156
162
|
if (info !== undefined && info.reasoning !== true) return activationPolicies;
|
|
157
163
|
if (info?.reasoningOptions !== undefined) {
|
|
158
164
|
return catalogSupportedReasoningPolicies({ info, native, style, declared });
|
|
@@ -355,8 +361,8 @@ export const providerFromSdkModel = ({
|
|
|
355
361
|
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
|
|
356
362
|
reasoning,
|
|
357
363
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
|
|
358
|
-
temperature:
|
|
359
|
-
repeatPenalty:
|
|
364
|
+
temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
365
|
+
repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
360
366
|
frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
|
|
361
367
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
362
368
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
@@ -83,6 +83,18 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
|
83
83
|
assert.equal("dry_allowed_length" in (body ?? {}), false);
|
|
84
84
|
});
|
|
85
85
|
|
|
86
|
+
test("(#483) a detected llama-server rail admits the operator's stated effort", async () => {
|
|
87
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
|
|
88
|
+
const url = String(input);
|
|
89
|
+
if (url.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }] }));
|
|
90
|
+
if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
|
|
91
|
+
throw new Error(`unexpected request ${url}`);
|
|
92
|
+
});
|
|
93
|
+
const provider = await compatibleProviderFromEnv("openai", { ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
|
|
94
|
+
assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
|
|
95
|
+
assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
|
|
96
|
+
});
|
|
97
|
+
|
|
86
98
|
test("detected llama-server measures the complete chat request through input_tokens", async () => {
|
|
87
99
|
let countUrl: string | undefined;
|
|
88
100
|
let countBody: Record<string, unknown> | undefined;
|