@plurnk/plurnk-providers 1.14.1 → 1.14.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +21 -9
- package/SPEC.md +54 -12
- package/dist/AiSdkProvider.d.ts +2 -2
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +24 -13
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +2 -0
- package/dist/Pool.js.map +1 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +29 -32
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +8 -0
- package/dist/capacity.d.ts.map +1 -1
- package/dist/capacity.js +21 -0
- package/dist/capacity.js.map +1 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +8 -3
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +6 -3
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +23 -2
- package/dist/errors.js.map +1 -1
- package/dist/types.d.ts +1 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +9 -1
- package/dist/types.js.map +1 -1
- package/package.json +6 -6
- package/src/AiSdkProvider.test.ts +138 -112
- package/src/AiSdkProvider.ts +28 -17
- package/src/Pool.test.ts +2 -0
- package/src/Pool.ts +2 -0
- package/src/aiSdkTransport.test.ts +31 -16
- package/src/aiSdkTransport.ts +32 -34
- package/src/capacity.test.ts +33 -1
- package/src/capacity.ts +34 -0
- package/src/catalogProvider.test.ts +22 -0
- package/src/catalogProvider.ts +7 -2
- package/src/compatibleProvider.test.ts +12 -0
- package/src/compatibleProvider.ts +6 -3
- package/src/errors.test.ts +1 -0
- package/src/errors.ts +22 -1
- package/src/types.ts +12 -1
|
@@ -289,19 +289,19 @@ test("the adapter preserves nonstandard reasoning accounting after SDK parsing",
|
|
|
289
289
|
});
|
|
290
290
|
});
|
|
291
291
|
|
|
292
|
-
test("normalizeRetryAttemptError —
|
|
292
|
+
test("normalizeRetryAttemptError — deadlines surface at once, never transport-retried ({§provider-connectivity}, #479)", () => {
|
|
293
293
|
const first = normalizeRetryAttemptError(new ProviderTimeoutError("first_content", 180000));
|
|
294
294
|
assert.equal(APICallError.isInstance(first), true);
|
|
295
|
-
assert.equal((first as APICallError).isRetryable,
|
|
295
|
+
assert.equal((first as APICallError).isRetryable, false);
|
|
296
296
|
const attempt = normalizeRetryAttemptError(new ProviderTimeoutError("attempt", 60000));
|
|
297
|
-
assert.equal((attempt as APICallError).isRetryable,
|
|
297
|
+
assert.equal((attempt as APICallError).isRetryable, false);
|
|
298
298
|
const idle = normalizeRetryAttemptError(new ProviderTimeoutError("stream_idle", 120000));
|
|
299
|
-
assert.equal((idle as APICallError).isRetryable,
|
|
299
|
+
assert.equal((idle as APICallError).isRetryable, false);
|
|
300
300
|
const operation = new ProviderTimeoutError("operation", 2700000);
|
|
301
301
|
assert.equal(normalizeRetryAttemptError(operation), operation);
|
|
302
302
|
});
|
|
303
303
|
|
|
304
|
-
test("normalizeRetryAttemptError —
|
|
304
|
+
test("normalizeRetryAttemptError — only provider-directed waits retry: 429, Retry-After, or a directive (#479 supersedes #446)", () => {
|
|
305
305
|
const garbage = new APICallError({
|
|
306
306
|
message: "Failed to process successful response",
|
|
307
307
|
url: "https://api.example/v1/chat/completions",
|
|
@@ -310,28 +310,43 @@ test("normalizeRetryAttemptError — a 2xx APICallError without a directive beco
|
|
|
310
310
|
responseBody: "not json",
|
|
311
311
|
isRetryable: false,
|
|
312
312
|
});
|
|
313
|
-
|
|
314
|
-
assert.ok(APICallError.isInstance(normalized));
|
|
315
|
-
assert.equal(normalized.isRetryable, true, "2xx invalid-response consumes the retry budget");
|
|
316
|
-
assert.equal(normalized.cause, garbage, "the original failure rides as the cause");
|
|
317
|
-
assert.equal(normalized.statusCode, 200);
|
|
313
|
+
assert.equal(normalizeRetryAttemptError(garbage), garbage, "2xx invalid-response surfaces at once — no promoted budget (#446 superseded)");
|
|
318
314
|
|
|
319
315
|
const directed = new APICallError({
|
|
320
316
|
message: "Failed to process successful response",
|
|
321
317
|
url: "https://api.example/v1/chat/completions",
|
|
322
318
|
requestBodyValues: {},
|
|
323
319
|
statusCode: 200,
|
|
324
|
-
responseHeaders: { "x-should-retry": "
|
|
320
|
+
responseHeaders: { "x-should-retry": "true" },
|
|
325
321
|
isRetryable: false,
|
|
326
322
|
});
|
|
327
|
-
assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable,
|
|
323
|
+
assert.equal((normalizeRetryAttemptError(directed) as APICallError).isRetryable, true, "an explicit directive still outranks");
|
|
328
324
|
|
|
329
|
-
const
|
|
330
|
-
message: "
|
|
325
|
+
const rateLimited = new APICallError({
|
|
326
|
+
message: "slow down",
|
|
331
327
|
url: "https://api.example/v1/chat/completions",
|
|
332
328
|
requestBodyValues: {},
|
|
333
|
-
statusCode:
|
|
329
|
+
statusCode: 429,
|
|
334
330
|
isRetryable: false,
|
|
335
331
|
});
|
|
336
|
-
assert.equal(normalizeRetryAttemptError(
|
|
332
|
+
assert.equal((normalizeRetryAttemptError(rateLimited) as APICallError).isRetryable, true, "a 429 is the provider-directed wait");
|
|
333
|
+
|
|
334
|
+
const directedWait = new APICallError({
|
|
335
|
+
message: "maintenance",
|
|
336
|
+
url: "https://api.example/v1/chat/completions",
|
|
337
|
+
requestBodyValues: {},
|
|
338
|
+
statusCode: 503,
|
|
339
|
+
responseHeaders: { "retry-after": "1" },
|
|
340
|
+
isRetryable: true,
|
|
341
|
+
});
|
|
342
|
+
assert.equal((normalizeRetryAttemptError(directedWait) as APICallError).isRetryable, true, "Retry-After on any status is a directed wait");
|
|
343
|
+
|
|
344
|
+
const bareServerError = new APICallError({
|
|
345
|
+
message: "internal error",
|
|
346
|
+
url: "https://api.example/v1/chat/completions",
|
|
347
|
+
requestBodyValues: {},
|
|
348
|
+
statusCode: 503,
|
|
349
|
+
isRetryable: true,
|
|
350
|
+
});
|
|
351
|
+
assert.equal((normalizeRetryAttemptError(bareServerError) as APICallError).isRetryable, false, "a bare 5xx surfaces at once for the engine's recovery");
|
|
337
352
|
});
|
package/src/aiSdkTransport.ts
CHANGED
|
@@ -29,16 +29,26 @@ const retryDirective = (
|
|
|
29
29
|
return null;
|
|
30
30
|
};
|
|
31
31
|
|
|
32
|
+
// {§provider-connectivity} (#479): a Retry-After header on any status is the
|
|
33
|
+
// provider directing a wait (RFC 9110 defines it on 503 exactly for this);
|
|
34
|
+
// its presence, like a bare 429, earns the bounded transport retry.
|
|
35
|
+
const retryAfterPresent = (
|
|
36
|
+
headers: Headers | Readonly<Record<string, string>>,
|
|
37
|
+
): boolean => {
|
|
38
|
+
const raw = headers instanceof Headers
|
|
39
|
+
? headers.get("retry-after")
|
|
40
|
+
: Object.entries(headers).find(([name]) => name.toLowerCase() === "retry-after")?.[1];
|
|
41
|
+
return raw !== undefined && raw !== null && raw.trim() !== "";
|
|
42
|
+
};
|
|
43
|
+
|
|
32
44
|
const errorStructure: ProviderErrorStructure<z.infer<typeof errorSchema>> = {
|
|
33
45
|
errorSchema,
|
|
34
46
|
errorToMessage: ({ error }) => error.message,
|
|
35
47
|
isRetryable(response) {
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|| response.status >= 500
|
|
41
|
-
);
|
|
48
|
+
// {§provider-connectivity} (#479): only a provider-directed wait — 429,
|
|
49
|
+
// a Retry-After, or an explicit X-Should-Retry — earns a transport retry.
|
|
50
|
+
return retryDirective(response.status, response.headers)
|
|
51
|
+
?? (response.status === 429 || retryAfterPresent(response.headers));
|
|
42
52
|
},
|
|
43
53
|
};
|
|
44
54
|
|
|
@@ -254,16 +264,16 @@ export const transportFailureOutputObserved = (error: unknown): boolean => {
|
|
|
254
264
|
|
|
255
265
|
export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
256
266
|
if (!APICallError.isInstance(error)) {
|
|
257
|
-
// Attempt, first-content, and stream-idle deadlines
|
|
258
|
-
//
|
|
259
|
-
//
|
|
267
|
+
// Attempt, first-content, and stream-idle deadlines surface on the first
|
|
268
|
+
// failure ({§provider-connectivity}, #479): the engine's {§provider-recovery}
|
|
269
|
+
// owns re-issue with backoff and park; the stall is reported, never swallowed.
|
|
260
270
|
if (error instanceof ProviderTimeoutError && error.phase !== "operation") {
|
|
261
271
|
return retainStreamFailureValues(error, new APICallError({
|
|
262
272
|
message: error.message,
|
|
263
273
|
url: "model:generation",
|
|
264
274
|
requestBodyValues: {},
|
|
265
275
|
cause: error,
|
|
266
|
-
isRetryable:
|
|
276
|
+
isRetryable: false,
|
|
267
277
|
}));
|
|
268
278
|
}
|
|
269
279
|
// Node's Undici stream reader reports a peer-aborted HTTP/2 body as this
|
|
@@ -276,33 +286,21 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
|
276
286
|
url: "model:generation",
|
|
277
287
|
requestBodyValues: {},
|
|
278
288
|
cause: error,
|
|
279
|
-
isRetryable:
|
|
289
|
+
isRetryable: false,
|
|
280
290
|
}));
|
|
281
291
|
}
|
|
282
292
|
return error;
|
|
283
293
|
}
|
|
284
294
|
const directed = retryDirective(error.statusCode, error.responseHeaders ?? {});
|
|
285
|
-
//
|
|
286
|
-
//
|
|
287
|
-
//
|
|
288
|
-
//
|
|
289
|
-
//
|
|
290
|
-
//
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
return retainStreamFailureValues(error, new APICallError({
|
|
295
|
-
message: error.message,
|
|
296
|
-
url: error.url,
|
|
297
|
-
requestBodyValues: error.requestBodyValues,
|
|
298
|
-
statusCode: error.statusCode,
|
|
299
|
-
responseHeaders: error.responseHeaders,
|
|
300
|
-
responseBody: error.responseBody,
|
|
301
|
-
cause: error,
|
|
302
|
-
isRetryable: true,
|
|
303
|
-
}));
|
|
304
|
-
}
|
|
305
|
-
if (directed === null || directed === error.isRetryable) return error;
|
|
295
|
+
// {§provider-connectivity} (#479): without an explicit directive the only
|
|
296
|
+
// transport-retryable failures are the provider-directed waits — a 429, or
|
|
297
|
+
// any status carrying Retry-After; those live in headers the engine never
|
|
298
|
+
// sees. Every other failure — a bare 408/409/5xx, a network error, and the
|
|
299
|
+
// 2xx invalid-response #446 once promoted — surfaces at once;
|
|
300
|
+
// {§provider-recovery} owns re-issue.
|
|
301
|
+
const policy = directed
|
|
302
|
+
?? (error.statusCode === 429 || retryAfterPresent(error.responseHeaders ?? {}));
|
|
303
|
+
if (policy === error.isRetryable) return error;
|
|
306
304
|
return retainStreamFailureValues(error, new APICallError({
|
|
307
305
|
message: error.message,
|
|
308
306
|
url: error.url,
|
|
@@ -311,7 +309,7 @@ export const normalizeRetryAttemptError = (error: unknown): unknown => {
|
|
|
311
309
|
responseHeaders: error.responseHeaders,
|
|
312
310
|
responseBody: error.responseBody,
|
|
313
311
|
cause: error,
|
|
314
|
-
isRetryable:
|
|
312
|
+
isRetryable: policy,
|
|
315
313
|
data: error.data,
|
|
316
314
|
}));
|
|
317
315
|
};
|
|
@@ -330,7 +328,7 @@ const executeModel = async (
|
|
|
330
328
|
url: "model:generation",
|
|
331
329
|
requestBodyValues: {},
|
|
332
330
|
cause: timeout,
|
|
333
|
-
isRetryable:
|
|
331
|
+
isRetryable: false,
|
|
334
332
|
});
|
|
335
333
|
}
|
|
336
334
|
};
|
package/src/capacity.test.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import assert from "node:assert/strict";
|
|
3
|
-
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
3
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, flexedResponseMax } from "./capacity.ts";
|
|
4
4
|
|
|
5
5
|
test("effective output budget is caller-tightenable and physically capped", () => {
|
|
6
6
|
assert.equal(effectiveOutputBudget({
|
|
@@ -90,3 +90,35 @@ test("only exact overflow rejects before provider I/O", () => {
|
|
|
90
90
|
measurement: { kind: "upper_bound", tokens: 60, source: "bound" },
|
|
91
91
|
}).decision, "admit");
|
|
92
92
|
});
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
test("(#482) flexedResponseMax harvests exact slack above the floor", () => {
|
|
96
|
+
assert.equal(
|
|
97
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
98
|
+
47_644,
|
|
99
|
+
"small prompt: the window remainder minus margin",
|
|
100
|
+
);
|
|
101
|
+
assert.equal(
|
|
102
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 40_000, margin: 256 }),
|
|
103
|
+
8_000,
|
|
104
|
+
"a full packet keeps the guaranteed floor even when margin eats the slack",
|
|
105
|
+
);
|
|
106
|
+
assert.equal(
|
|
107
|
+
flexedResponseMax({ contextWindow: 48_000, maxOutputTokens: 16_000, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
108
|
+
16_000,
|
|
109
|
+
"the model's own output cap bounds the harvest",
|
|
110
|
+
);
|
|
111
|
+
assert.equal(
|
|
112
|
+
flexedResponseMax({ contextWindow: null, maxOutputTokens: null, outputBudget: 8_000, promptTokens: 100, margin: 256 }),
|
|
113
|
+
8_000,
|
|
114
|
+
"no window, no flex",
|
|
115
|
+
);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
test("(#482) assessRequestCapacity flexes only exact measurements", () => {
|
|
119
|
+
const base = { contextWindow: 48_000, maxInputTokens: null, maxOutputTokens: null, outputBudget: 8_000, reasoningBudget: null };
|
|
120
|
+
const exact = assessRequestCapacity({ ...base, measurement: { kind: "exact", tokens: 1_000, source: "t" } });
|
|
121
|
+
assert.equal(exact.responseMax, 48_000 - 1_000 - 256, "exact prompts harvest the slack");
|
|
122
|
+
const estimate = assessRequestCapacity({ ...base, measurement: { kind: "estimate", tokens: 1_000, source: "t", detail: "chars/2 test estimate" } });
|
|
123
|
+
assert.equal(estimate.responseMax, 8_000, "estimates keep the floor — they prove nothing about the remainder");
|
|
124
|
+
});
|
package/src/capacity.ts
CHANGED
|
@@ -81,6 +81,34 @@ export const effectiveInputCapacity = ({
|
|
|
81
81
|
return capacities.length === 0 ? null : Math.min(...capacities);
|
|
82
82
|
};
|
|
83
83
|
|
|
84
|
+
// {§provider-flexed-allowance} (#482): the configured output budget is the floor
|
|
85
|
+
// curation packed the input against; window room the actual prompt left
|
|
86
|
+
// unclaimed is guaranteed free and becomes response runway. Only an exact
|
|
87
|
+
// prompt measurement may claim slack — an estimate proves nothing about the
|
|
88
|
+
// true remainder — and the model's own maxOutputTokens still caps the grant.
|
|
89
|
+
export const WIRE_FLEX_MARGIN = 256;
|
|
90
|
+
|
|
91
|
+
export const flexedResponseMax = ({
|
|
92
|
+
contextWindow,
|
|
93
|
+
maxOutputTokens,
|
|
94
|
+
outputBudget,
|
|
95
|
+
promptTokens,
|
|
96
|
+
margin,
|
|
97
|
+
}: {
|
|
98
|
+
contextWindow: number | null;
|
|
99
|
+
maxOutputTokens: number | null;
|
|
100
|
+
outputBudget: number | null;
|
|
101
|
+
promptTokens: number;
|
|
102
|
+
margin: number;
|
|
103
|
+
}): number | null => {
|
|
104
|
+
if (outputBudget === null || contextWindow === null) return outputBudget;
|
|
105
|
+
if (!Number.isSafeInteger(promptTokens) || promptTokens < 0) {
|
|
106
|
+
throw new TypeError("promptTokens must be a non-negative safe integer");
|
|
107
|
+
}
|
|
108
|
+
const flexed = Math.max(outputBudget, contextWindow - promptTokens - margin);
|
|
109
|
+
return maxOutputTokens === null ? flexed : Math.min(flexed, Math.max(outputBudget, maxOutputTokens));
|
|
110
|
+
};
|
|
111
|
+
|
|
84
112
|
export const requestCapacityDecision = (
|
|
85
113
|
inputCapacity: number | null,
|
|
86
114
|
measurement: PromptTokenMeasurement,
|
|
@@ -126,6 +154,11 @@ export const assessRequestCapacity = ({
|
|
|
126
154
|
}
|
|
127
155
|
const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
|
|
128
156
|
const inputCapacity = effectiveInputCapacity({ contextWindow, maxInputTokens, outputBudget });
|
|
157
|
+
// {§provider-flexed-allowance}: exact measurements harvest the slack; every
|
|
158
|
+
// other measurement kind keeps the floor.
|
|
159
|
+
const responseMax = prompt.kind === "exact"
|
|
160
|
+
? flexedResponseMax({ contextWindow, maxOutputTokens, outputBudget, promptTokens: prompt.tokens, margin: WIRE_FLEX_MARGIN })
|
|
161
|
+
: outputBudget;
|
|
129
162
|
|
|
130
163
|
return {
|
|
131
164
|
decision: requestCapacityDecision(inputCapacity, prompt),
|
|
@@ -135,6 +168,7 @@ export const assessRequestCapacity = ({
|
|
|
135
168
|
outputBudget,
|
|
136
169
|
reasoningBudget,
|
|
137
170
|
inputCapacity,
|
|
171
|
+
responseMax,
|
|
138
172
|
prompt,
|
|
139
173
|
};
|
|
140
174
|
};
|
|
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
|
|
|
3
3
|
import { once } from "node:events";
|
|
4
4
|
import { createServer } from "node:http";
|
|
5
5
|
import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
|
|
6
|
+
import { withProviderDefaults } from "./defaults.ts";
|
|
6
7
|
import type { LanguageModel } from "ai";
|
|
7
8
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
8
9
|
|
|
@@ -41,6 +42,27 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
|
|
|
41
42
|
assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
|
|
42
43
|
});
|
|
43
44
|
|
|
45
|
+
test("(#472) an effort refusal names the operator's declaration lever with the exact key", () => {
|
|
46
|
+
assert.throws(
|
|
47
|
+
() => catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
|
|
48
|
+
...env,
|
|
49
|
+
FIREWORKS_API_KEY: "test-key",
|
|
50
|
+
PLURNK_PROVIDERS_REASONING: "low",
|
|
51
|
+
}), "accounts/fireworks/models/glm-5p3-flash"),
|
|
52
|
+
/declare it: PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS=low/,
|
|
53
|
+
"the refusal is actionable: it names the exact env declaration",
|
|
54
|
+
);
|
|
55
|
+
// And the lever works: the same route with the declaration constructs.
|
|
56
|
+
const declared = catalogProviderFromEnv("fireworks-ai", withProviderDefaults({
|
|
57
|
+
...env,
|
|
58
|
+
FIREWORKS_API_KEY: "test-key",
|
|
59
|
+
PLURNK_PROVIDERS_REASONING: "low",
|
|
60
|
+
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_AI_REASONING_EFFORTS: "low,high,max",
|
|
61
|
+
}), "accounts/fireworks/models/glm-5p3-flash");
|
|
62
|
+
assert.ok(declared, "the declared effort admits the route");
|
|
63
|
+
assert.ok(declared.supportedReasoningPolicies.includes("low"), "low is admitted through the operator's declaration");
|
|
64
|
+
});
|
|
65
|
+
|
|
44
66
|
test("provider adapters advertise only reasoning policies they can preserve", () => {
|
|
45
67
|
const deepseek = catalogProviderFromEnv("deepseek", {
|
|
46
68
|
...env,
|
package/src/catalogProvider.ts
CHANGED
|
@@ -10,6 +10,7 @@ import {
|
|
|
10
10
|
effectiveContextWindow,
|
|
11
11
|
dataCaptureFromEnv,
|
|
12
12
|
generationEnvelopeFromEnv,
|
|
13
|
+
parseOptionalFloat,
|
|
13
14
|
parseRequiredFloat,
|
|
14
15
|
parseRequiredInt,
|
|
15
16
|
parseTimeoutMs,
|
|
@@ -154,6 +155,10 @@ const supportedReasoningPolicies = ({
|
|
|
154
155
|
style: ReasoningStyle;
|
|
155
156
|
declared: readonly ModelReasoningEffort[];
|
|
156
157
|
}): readonly ReasoningPolicy[] => {
|
|
158
|
+
// The template style is the operator's declaration that the rail's own chat
|
|
159
|
+
// template governs reasoning: a fixed effort rides in verbatim, native SDK or
|
|
160
|
+
// not, and a word the template does not know fails loudly on the first request.
|
|
161
|
+
if (style === "template") return REASONING_POLICIES;
|
|
157
162
|
if (info !== undefined && info.reasoning !== true) return activationPolicies;
|
|
158
163
|
if (info?.reasoningOptions !== undefined) {
|
|
159
164
|
return catalogSupportedReasoningPolicies({ info, native, style, declared });
|
|
@@ -356,8 +361,8 @@ export const providerFromSdkModel = ({
|
|
|
356
361
|
streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
|
|
357
362
|
reasoning,
|
|
358
363
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
|
|
359
|
-
temperature:
|
|
360
|
-
repeatPenalty:
|
|
364
|
+
temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
365
|
+
repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
361
366
|
frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
|
|
362
367
|
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
363
368
|
errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
|
|
@@ -83,6 +83,18 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
|
|
|
83
83
|
assert.equal("dry_allowed_length" in (body ?? {}), false);
|
|
84
84
|
});
|
|
85
85
|
|
|
86
|
+
test("(#483) a detected llama-server rail admits the operator's stated effort", async () => {
|
|
87
|
+
mock.method(globalThis, "fetch", async (input: string | URL | Request) => {
|
|
88
|
+
const url = String(input);
|
|
89
|
+
if (url.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "served.gguf", meta: { n_ctx: 8192 } }] }));
|
|
90
|
+
if (url.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }));
|
|
91
|
+
throw new Error(`unexpected request ${url}`);
|
|
92
|
+
});
|
|
93
|
+
const provider = await compatibleProviderFromEnv("openai", { ...env, PLURNK_PROVIDERS_REASONING: "medium" }, "local");
|
|
94
|
+
assert.ok(provider.supportedReasoningPolicies.includes("medium"), "the template governs: medium is admitted on a llama-server rail");
|
|
95
|
+
assert.ok(provider.supportedReasoningPolicies.includes("low") && provider.supportedReasoningPolicies.includes("high"), "the whole policy vocabulary rides; the template refuses unknown words itself");
|
|
96
|
+
});
|
|
97
|
+
|
|
86
98
|
test("detected llama-server measures the complete chat request through input_tokens", async () => {
|
|
87
99
|
let countUrl: string | undefined;
|
|
88
100
|
let countBody: Record<string, unknown> | undefined;
|
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
|
|
1
2
|
import AiSdkProvider, { type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
|
|
2
3
|
import {
|
|
3
4
|
contextWindowFromEnv,
|
|
@@ -182,7 +183,9 @@ export const compatibleProviderFromEnv = async (
|
|
|
182
183
|
}
|
|
183
184
|
const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
|
|
184
185
|
const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
|
|
185
|
-
|
|
186
|
+
// A detected llama-server rail runs the template style: the chat template governs
|
|
187
|
+
// reasoning, so the operator's stated effort is admitted and forwarded verbatim (#483).
|
|
188
|
+
const supportedReasoningPolicies = reasoningStyle === "template" ? REASONING_POLICIES : (["off", "adaptive"] as const);
|
|
186
189
|
return new AiSdkProvider({
|
|
187
190
|
model,
|
|
188
191
|
url,
|
|
@@ -200,8 +203,8 @@ export const compatibleProviderFromEnv = async (
|
|
|
200
203
|
reasoning,
|
|
201
204
|
reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
|
|
202
205
|
reasoningStyle,
|
|
203
|
-
temperature:
|
|
204
|
-
repeatPenalty:
|
|
206
|
+
temperature: parseOptionalFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
|
|
207
|
+
repeatPenalty: parseOptionalFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", provider, 0),
|
|
205
208
|
frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", provider, 0),
|
|
206
209
|
dryMultiplier: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_MULTIPLIER, "PLURNK_PROVIDERS_DRY_MULTIPLIER", provider, 0) ?? undefined,
|
|
207
210
|
dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
|
package/src/errors.test.ts
CHANGED
package/src/errors.ts
CHANGED
|
@@ -46,6 +46,17 @@ export const providerTimeoutOf = (error: unknown): ProviderTimeoutError | null =
|
|
|
46
46
|
return null;
|
|
47
47
|
};
|
|
48
48
|
|
|
49
|
+
const peerTerminated = (err: unknown): boolean => {
|
|
50
|
+
const seen = new Set<unknown>();
|
|
51
|
+
let cur: unknown = err;
|
|
52
|
+
while (typeof cur === "object" && cur !== null && !seen.has(cur)) {
|
|
53
|
+
seen.add(cur);
|
|
54
|
+
if (cur instanceof TypeError && cur.message.trim().toLowerCase() === "terminated") return true;
|
|
55
|
+
cur = (cur as { cause?: unknown }).cause;
|
|
56
|
+
}
|
|
57
|
+
return false;
|
|
58
|
+
};
|
|
59
|
+
|
|
49
60
|
const wireError = (body: string): { type: string | null; code: string | null; message: string | null } => {
|
|
50
61
|
try {
|
|
51
62
|
const { error } = JSON.parse(body) as { error?: { type?: unknown } };
|
|
@@ -100,13 +111,23 @@ export const classifyProviderError = (
|
|
|
100
111
|
: "The provider request failed without a diagnostic message.";
|
|
101
112
|
const body = err.responseBody ?? "";
|
|
102
113
|
const wire = wireError(body);
|
|
114
|
+
// A peer-terminated body inside a 2xx exchange is a network truth, not
|
|
115
|
+
// an invalid response — the SDK wraps Undici's TypeError("terminated")
|
|
116
|
+
// as a processing failure, but the KIND must stay engine-recoverable
|
|
117
|
+
// (#479). Walk the cause chain for the termination signature.
|
|
118
|
+
if (peerTerminated(err)) {
|
|
119
|
+
return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
120
|
+
}
|
|
103
121
|
if (status === 401 || status === 403) return { kind: "unauthorized", message };
|
|
104
122
|
if (status === 402) return { kind: "quota_exceeded", message };
|
|
105
123
|
if (status === 429) return { kind: "rate_limit", message, retryable: err.isRetryable };
|
|
106
124
|
if (status === 408 || status === 409) {
|
|
107
125
|
return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
108
126
|
}
|
|
109
|
-
|
|
127
|
+
// Status 0 is a transport-level failure (no HTTP exchange settled); its
|
|
128
|
+
// KIND stays network_failure regardless of the transport's retry policy
|
|
129
|
+
// (#479) — the engine's recovery keys on kind, never the retryable flag.
|
|
130
|
+
if (status === 0) return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
110
131
|
if (status >= 500) return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
111
132
|
if (status === 413 || (
|
|
112
133
|
(status === 400 || status === 422)
|
package/src/types.ts
CHANGED
|
@@ -32,7 +32,15 @@ export class UnsupportedReasoningPolicyError extends Error {
|
|
|
32
32
|
readonly supported: readonly ReasoningPolicy[];
|
|
33
33
|
|
|
34
34
|
constructor(source: string, policy: ReasoningPolicy, supported: readonly ReasoningPolicy[]) {
|
|
35
|
-
|
|
35
|
+
// A graded-effort refusal names the operator's declaration lever (#472):
|
|
36
|
+
// the catalog is authoritative per model, and where a provider's API
|
|
37
|
+
// accepts more than its catalog entry lists, the remedy is the
|
|
38
|
+
// operator's, never a vendored fact in shipped defaults.
|
|
39
|
+
const providerName = /^provider:(.+)$/.exec(source)?.[1];
|
|
40
|
+
const lever = policy === "off" || policy === "adaptive"
|
|
41
|
+
? ""
|
|
42
|
+
: ` If the provider's API documents this effort beyond its catalog entry, declare it: PLURNK_PROVIDERS_PROVIDER_${(providerName ?? "<NAME>").replaceAll(/[^a-zA-Z0-9<>]/g, "_").toUpperCase()}_REASONING_EFFORTS=${policy} (comma-separated).`;
|
|
43
|
+
super(`${source}: reasoning policy '${policy}' is unsupported; supported policies: ${supported.join(", ")}.${lever}`);
|
|
36
44
|
this.name = "UnsupportedReasoningPolicyError";
|
|
37
45
|
this.policy = policy;
|
|
38
46
|
this.supported = supported;
|
|
@@ -82,6 +90,9 @@ export interface ProviderRequestCapacity {
|
|
|
82
90
|
readonly outputBudget: number | null;
|
|
83
91
|
readonly reasoningBudget: number | null;
|
|
84
92
|
readonly inputCapacity: number | null;
|
|
93
|
+
// {§provider-flexed-allowance} (#482): the response allowance actually
|
|
94
|
+
// granted this request — the floor, or the exactly-measured slack above it.
|
|
95
|
+
readonly responseMax: number | null;
|
|
85
96
|
readonly prompt: PromptTokenMeasurement;
|
|
86
97
|
}
|
|
87
98
|
|