@plurnk/plurnk-providers 1.5.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -34
- package/README.md +15 -0
- package/SPEC.md +242 -89
- package/dist/AiSdkProvider.d.ts +33 -33
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +442 -133
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +10 -11
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +87 -25
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +9 -24
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +86 -25
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +8 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +45 -41
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +26 -12
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +13 -11
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +83 -46
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +17 -3
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +91 -8
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +7 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -3
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/sdkModels.d.ts +7 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +43 -13
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +55 -33
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +22 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +169 -83
- package/dist/usage.js.map +1 -1
- package/package.json +18 -7
- package/src/AiSdkProvider.test.ts +964 -206
- package/src/AiSdkProvider.ts +545 -155
- package/src/Mock.test.ts +69 -30
- package/src/Mock.ts +99 -29
- package/src/Pool.test.ts +90 -19
- package/src/Pool.ts +96 -27
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +119 -18
- package/src/accountingPublic.ts +9 -0
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +2 -0
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +339 -30
- package/src/catalogProvider.ts +65 -47
- package/src/compatibleProvider.test.ts +7 -5
- package/src/compatibleProvider.ts +29 -13
- package/src/cost.test.ts +86 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +103 -25
- package/src/env.ts +153 -65
- package/src/errors.test.ts +80 -2
- package/src/errors.ts +107 -8
- package/src/index.ts +26 -7
- package/src/ollama.test.ts +5 -3
- package/src/ollama.ts +3 -3
- package/src/promptTokens.ts +8 -5
- package/src/sdkModels.test.ts +77 -8
- package/src/sdkModels.ts +51 -15
- package/src/types.ts +112 -51
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +214 -93
package/src/env.ts
CHANGED
|
@@ -8,6 +8,16 @@ export const parseRequiredInt = (raw: string | undefined, name: string, label: s
|
|
|
8
8
|
return n;
|
|
9
9
|
};
|
|
10
10
|
|
|
11
|
+
export const MAX_PROVIDER_TIMEOUT_MS = 2_147_483_647;
|
|
12
|
+
|
|
13
|
+
export const parseTimeoutMs = (raw: string | undefined, name: string, label: string): number => {
|
|
14
|
+
const timeoutMs = parseRequiredInt(raw, name, label);
|
|
15
|
+
if (timeoutMs > MAX_PROVIDER_TIMEOUT_MS) {
|
|
16
|
+
throw new Error(`${label} provider: ${name} must be at most ${MAX_PROVIDER_TIMEOUT_MS} milliseconds (got "${raw}")`);
|
|
17
|
+
}
|
|
18
|
+
return timeoutMs;
|
|
19
|
+
};
|
|
20
|
+
|
|
11
21
|
export const parseOptionalInt = (raw: string | undefined, name: string, label: string): number | null => {
|
|
12
22
|
if (raw === undefined || raw.length === 0) return null;
|
|
13
23
|
const n = Number(raw);
|
|
@@ -34,15 +44,6 @@ export const requireEnv = (raw: string | undefined, name: string, label: string)
|
|
|
34
44
|
return raw;
|
|
35
45
|
};
|
|
36
46
|
|
|
37
|
-
export const promptCacheKeyFromEnv = (env: NodeJS.ProcessEnv, label: string): boolean => {
|
|
38
|
-
const name = "PLURNK_PROVIDERS_PROMPT_CACHE_KEY";
|
|
39
|
-
const value = env[name];
|
|
40
|
-
if (value !== "0" && value !== "1") {
|
|
41
|
-
throw new Error(`${label} provider: ${name} must be "0" or "1"`);
|
|
42
|
-
}
|
|
43
|
-
return value === "1";
|
|
44
|
-
};
|
|
45
|
-
|
|
46
47
|
// {§provider-configuration} A still-set retired knob fails
|
|
47
48
|
// hard pointing at its successor — never silently coexists with the new floor.
|
|
48
49
|
// The retired names appear ONLY as this function's ARGUMENTS at the call sites
|
|
@@ -52,6 +53,27 @@ const shedRenamed = (env: NodeJS.ProcessEnv, oldName: string, newName: string, l
|
|
|
52
53
|
if (stale !== undefined && stale.length > 0) throw new Error(`${label} provider: ${oldName} was renamed to ${newName} (${ref}); update the env`);
|
|
53
54
|
};
|
|
54
55
|
|
|
56
|
+
export type CacheWritePolicy = "off" | "stable-system";
|
|
57
|
+
|
|
58
|
+
export const cacheAffinityFromEnv = (env: NodeJS.ProcessEnv, label: string): boolean => {
|
|
59
|
+
const name = "PLURNK_PROVIDERS_CACHE_AFFINITY";
|
|
60
|
+
shedRenamed(env, "PLURNK_PROVIDERS_PROMPT_CACHE_KEY", name, label, "{§provider-cache-affinity}"); // lexicon-allow
|
|
61
|
+
const value = env[name];
|
|
62
|
+
if (value !== "0" && value !== "1") {
|
|
63
|
+
throw new Error(`${label} provider: ${name} must be "0" or "1"`);
|
|
64
|
+
}
|
|
65
|
+
return value === "1";
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
export const cacheWritePolicyFromEnv = (env: NodeJS.ProcessEnv, label: string): CacheWritePolicy => {
|
|
69
|
+
const name = "PLURNK_PROVIDERS_CACHE_WRITE_POLICY";
|
|
70
|
+
const value = env[name];
|
|
71
|
+
if (value !== "off" && value !== "stable-system") {
|
|
72
|
+
throw new Error(`${label} provider: ${name} must be "off" or "stable-system"`);
|
|
73
|
+
}
|
|
74
|
+
return value;
|
|
75
|
+
};
|
|
76
|
+
|
|
55
77
|
// {§provider-evidence} Data-capture knobs are read identically by every provider
|
|
56
78
|
// (standard AND
|
|
57
79
|
// plugin) so the opt-in surface is one source of truth. Both OFF by default —
|
|
@@ -79,8 +101,8 @@ export const contextWindowFromEnv = (env: NodeJS.ProcessEnv, label: string): num
|
|
|
79
101
|
return parseOptionalInt(env.PLURNK_PROVIDERS_CONTEXT_WINDOW, "PLURNK_PROVIDERS_CONTEXT_WINDOW", label);
|
|
80
102
|
};
|
|
81
103
|
|
|
82
|
-
// {§model-fact-resolution} — an operator value caps known model
|
|
83
|
-
// declares the
|
|
104
|
+
// {§model-fact-resolution} — an operator value caps known natural model
|
|
105
|
+
// capacity and declares the envelope only when no natural value is known.
|
|
84
106
|
export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number): number;
|
|
85
107
|
export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number | null): number | null;
|
|
86
108
|
export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number | null): number | null {
|
|
@@ -91,20 +113,14 @@ export function effectiveContextWindow(operatorCap: number | null, naturalWindow
|
|
|
91
113
|
: Math.min(operatorCap, naturalWindow);
|
|
92
114
|
}
|
|
93
115
|
|
|
94
|
-
// {§provider-generation-envelope}
|
|
95
|
-
//
|
|
96
|
-
//
|
|
97
|
-
//
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
// Per-alias suffixes override for measured envelopes; absolutes win over the
|
|
103
|
-
// window derivation entirely.
|
|
104
|
-
export type ReserveSpec = { percent: number } | { tokens: number };
|
|
105
|
-
|
|
106
|
-
const parseReserve = (raw: string | undefined, name: string, label: string): ReserveSpec => {
|
|
107
|
-
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the window like "10%", or an absolute token count)`);
|
|
116
|
+
// {§provider-generation-envelope} Generation has one total output budget. An
|
|
117
|
+
// optional reasoning budget is a subset, never an additive second reserve.
|
|
118
|
+
// Percentages are of the effective context window; absolutes remain useful for
|
|
119
|
+
// measured local deployments. Physical model limits always cap operator policy.
|
|
120
|
+
export type TokenBudgetSpec = { percent: number } | { tokens: number };
|
|
121
|
+
|
|
122
|
+
const parseTokenBudget = (raw: string | undefined, name: string, label: string): TokenBudgetSpec => {
|
|
123
|
+
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the context window like "35%", or an absolute token count)`);
|
|
108
124
|
const pct = /^([0-9]+(?:\.[0-9]+)?)%$/.exec(raw);
|
|
109
125
|
if (pct !== null) {
|
|
110
126
|
const p = Number(pct[1]);
|
|
@@ -116,37 +132,110 @@ const parseReserve = (raw: string | undefined, name: string, label: string): Res
|
|
|
116
132
|
return { tokens: n };
|
|
117
133
|
};
|
|
118
134
|
|
|
119
|
-
|
|
120
|
-
reasoningReserve: parseReserve(env.PLURNK_PROVIDERS_REASONING_RESERVE, "PLURNK_PROVIDERS_REASONING_RESERVE", label),
|
|
121
|
-
completionReserve: parseReserve(env.PLURNK_PROVIDERS_COMPLETION_RESERVE, "PLURNK_PROVIDERS_COMPLETION_RESERVE", label),
|
|
122
|
-
});
|
|
123
|
-
|
|
124
|
-
// Resolve a ReserveSpec against a known window: absolutes stand alone; a
|
|
135
|
+
// Resolve a token budget against a known window: absolutes stand alone; a
|
|
125
136
|
// percentage needs the window (null when unknown — the underivable/no-cap case).
|
|
126
|
-
export const
|
|
127
|
-
"tokens" in spec
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
137
|
+
export const resolveTokenBudget = (spec: TokenBudgetSpec, window: number | null): number | null =>
|
|
138
|
+
"tokens" in spec
|
|
139
|
+
? spec.tokens
|
|
140
|
+
: window === null
|
|
141
|
+
? null
|
|
142
|
+
: Math.max(1, Math.round(spec.percent * window));
|
|
143
|
+
|
|
144
|
+
export type GenerationEnvelope = {
|
|
145
|
+
readonly outputBudget: number | null;
|
|
146
|
+
readonly reasoningBudget: number | null;
|
|
147
|
+
};
|
|
148
|
+
|
|
149
|
+
const shedRetiredEnvelope = (env: NodeJS.ProcessEnv, label: string): void => {
|
|
150
|
+
for (const name of ["PLURNK_PROVIDERS_REASONING_RESERVE", "PLURNK_PROVIDERS_COMPLETION_RESERVE"] as const) {
|
|
151
|
+
if (env[name] !== undefined && env[name] !== "") {
|
|
152
|
+
throw new Error(
|
|
153
|
+
`${label} provider: ${name} is retired; reasoning is now a subset of the total generation envelope. Replace the old pair with PLURNK_PROVIDERS_OUTPUT_BUDGET and optional PLURNK_PROVIDERS_REASONING_BUDGET ({§provider-generation-envelope})`,
|
|
154
|
+
);
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
|
|
159
|
+
const optionalTokenBudget = (
|
|
160
|
+
raw: string | undefined,
|
|
161
|
+
name: string,
|
|
162
|
+
label: string,
|
|
163
|
+
): TokenBudgetSpec | null => raw === undefined || raw.length === 0
|
|
164
|
+
? null
|
|
165
|
+
: parseTokenBudget(raw, name, label);
|
|
166
|
+
|
|
167
|
+
const resolveGeneration = (
|
|
168
|
+
outputSpec: TokenBudgetSpec | null,
|
|
169
|
+
reasoningSpec: TokenBudgetSpec | null,
|
|
170
|
+
contextWindow: number | null,
|
|
171
|
+
maxOutputTokens: number | null,
|
|
172
|
+
label: string,
|
|
173
|
+
): GenerationEnvelope => {
|
|
174
|
+
const requestedOutput = outputSpec === null ? null : resolveTokenBudget(outputSpec, contextWindow);
|
|
175
|
+
const physicalCaps = [contextWindow, maxOutputTokens].filter((value): value is number => value !== null);
|
|
176
|
+
const outputBudget = requestedOutput === null
|
|
177
|
+
? null
|
|
178
|
+
: Math.min(requestedOutput, ...physicalCaps);
|
|
179
|
+
if (contextWindow !== null && outputBudget !== null && outputBudget >= contextWindow) {
|
|
180
|
+
throw new Error(
|
|
181
|
+
`${label} provider: PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}) must leave positive input capacity inside the context window (${contextWindow})`,
|
|
182
|
+
);
|
|
183
|
+
}
|
|
184
|
+
const reasoningBudget = reasoningSpec === null
|
|
185
|
+
? null
|
|
186
|
+
: resolveTokenBudget(reasoningSpec, contextWindow);
|
|
187
|
+
if (reasoningBudget !== null && outputBudget === null) {
|
|
188
|
+
throw new Error(
|
|
189
|
+
`${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET; reasoning is a subset of total output`,
|
|
190
|
+
);
|
|
191
|
+
}
|
|
192
|
+
if (reasoningBudget !== null && outputBudget !== null && reasoningBudget >= outputBudget) {
|
|
193
|
+
throw new Error(
|
|
194
|
+
`${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET (${reasoningBudget}) exceeds the effective PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}); reasoning is a subset of total output`,
|
|
195
|
+
);
|
|
196
|
+
}
|
|
197
|
+
return { outputBudget, reasoningBudget };
|
|
198
|
+
};
|
|
199
|
+
|
|
200
|
+
// Standard providers receive the shipped OUTPUT_BUDGET floor and fail hard if
|
|
201
|
+
// it is absent. Mock uses the tolerant sibling below so ordinary unit fixtures
|
|
202
|
+
// make no generation claim unless a test deliberately configures one.
|
|
203
|
+
export const generationEnvelopeFromEnv = (
|
|
204
|
+
env: NodeJS.ProcessEnv,
|
|
205
|
+
label: string,
|
|
206
|
+
contextWindow: number | null,
|
|
207
|
+
maxOutputTokens: number | null,
|
|
208
|
+
): GenerationEnvelope => {
|
|
209
|
+
shedRetiredEnvelope(env, label);
|
|
210
|
+
return resolveGeneration(
|
|
211
|
+
parseTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", label),
|
|
212
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", label),
|
|
213
|
+
contextWindow,
|
|
214
|
+
maxOutputTokens,
|
|
215
|
+
label,
|
|
216
|
+
);
|
|
217
|
+
};
|
|
218
|
+
|
|
219
|
+
export const resolveGenerationEnvelopeFromEnv = (
|
|
220
|
+
env: NodeJS.ProcessEnv,
|
|
221
|
+
contextWindow: number | null,
|
|
222
|
+
maxOutputTokens: number | null = null,
|
|
223
|
+
): GenerationEnvelope => {
|
|
224
|
+
shedRetiredEnvelope(env, "mock");
|
|
225
|
+
return resolveGeneration(
|
|
226
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", "mock"),
|
|
227
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", "mock"),
|
|
228
|
+
contextWindow,
|
|
229
|
+
maxOutputTokens,
|
|
230
|
+
"mock",
|
|
231
|
+
);
|
|
142
232
|
};
|
|
143
233
|
|
|
144
234
|
// {§provider-configuration} The side-channel reasoning knobs — activation and budget
|
|
145
235
|
// are separate vars, so a numeric budget can never silently flip wire flags:
|
|
146
236
|
// PLURNK_PROVIDERS_REASONING off | adaptive | on (REQUIRED, fail-hard)
|
|
147
|
-
// PLURNK_PROVIDERS_REASONING_BUDGET
|
|
148
|
-
//
|
|
149
|
-
// request-scoped allowance and cannot exceed the physical reasoning reserve.
|
|
237
|
+
// PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
|
|
238
|
+
// output budget, used for tier/budget mapping where the backend supports it.
|
|
150
239
|
// The provider maps intent to the backend's mechanism; the consumer states
|
|
151
240
|
// intent, never mechanism. PLAN is a separate public intended-goals record.
|
|
152
241
|
export type ReasoningMode = "off" | "adaptive" | "on";
|
|
@@ -167,20 +256,18 @@ export const reasoningResponseStyleFromEnv = (
|
|
|
167
256
|
return raw;
|
|
168
257
|
};
|
|
169
258
|
|
|
170
|
-
export const reasoningFromEnv = (
|
|
259
|
+
export const reasoningFromEnv = (
|
|
260
|
+
env: NodeJS.ProcessEnv,
|
|
261
|
+
label: string,
|
|
262
|
+
resolvedBudget: number | null = null,
|
|
263
|
+
): Reasoning => {
|
|
171
264
|
shedRenamed(env, "PLURNK_PROVIDERS_THINKING", "PLURNK_PROVIDERS_REASONING", label, "provider configuration contract"); // lexicon-allow
|
|
172
265
|
shedRenamed(env, "PLURNK_PROVIDERS_THINKING_CAPACITY", "PLURNK_PROVIDERS_REASONING_BUDGET", label, "provider configuration contract"); // lexicon-allow
|
|
173
266
|
const name = "PLURNK_PROVIDERS_REASONING";
|
|
174
267
|
const raw = env[name];
|
|
175
268
|
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (off | adaptive | on)`);
|
|
176
269
|
if (raw !== "off" && raw !== "adaptive" && raw !== "on") throw new Error(`${label} provider: ${name} must be one of "off", "adaptive", "on" (got "${raw}")`);
|
|
177
|
-
|
|
178
|
-
const capName = "PLURNK_PROVIDERS_REASONING_BUDGET";
|
|
179
|
-
const capRaw = env[capName];
|
|
180
|
-
if (capRaw === undefined || capRaw.length === 0) throw new Error(`${label} provider: ${capName} must be set when ${name}=on`);
|
|
181
|
-
const n = Number(capRaw);
|
|
182
|
-
if (!Number.isInteger(n) || n <= 0) throw new Error(`${label} provider: ${capName} must be a positive integer (got "${capRaw}")`);
|
|
183
|
-
return { mode: "on", budget: n };
|
|
270
|
+
return { mode: raw, budget: raw === "off" ? null : resolvedBudget };
|
|
184
271
|
};
|
|
185
272
|
|
|
186
273
|
// ── Per-alias knob scoping (per-alias scoping doctrine, user 2026-07-03): PLURNK_PROVIDERS_<KNOB>[_<alias>] ──
|
|
@@ -191,22 +278,24 @@ export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoni
|
|
|
191
278
|
// facts (API keys, canonical endpoints) remain vendor-named; the per-alias
|
|
192
279
|
// endpoint override stays PLURNK_BASEURL_<alias> (its existing precedent).
|
|
193
280
|
export const PROVIDERS_KNOBS = Object.freeze([
|
|
194
|
-
"
|
|
195
|
-
"PLURNK_PROVIDERS_COMPLETION_RESERVE",
|
|
281
|
+
"PLURNK_PROVIDERS_OUTPUT_BUDGET",
|
|
196
282
|
"PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
|
|
197
283
|
"PLURNK_PROVIDERS_REASONING_BUDGET",
|
|
198
284
|
"PLURNK_PROVIDERS_REASONING",
|
|
199
285
|
"PLURNK_PROVIDERS_CONTEXT_WINDOW",
|
|
200
286
|
"PLURNK_PROVIDERS_RETRY_ATTEMPTS",
|
|
201
287
|
"PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT",
|
|
288
|
+
"PLURNK_PROVIDERS_OPERATION_TIMEOUT",
|
|
202
289
|
"PLURNK_PROVIDERS_FETCH_TIMEOUT",
|
|
290
|
+
"PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT",
|
|
203
291
|
"PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
|
|
204
292
|
"PLURNK_PROVIDERS_LLAMA_SERVER",
|
|
205
293
|
"PLURNK_PROVIDERS_TEMPERATURE",
|
|
206
294
|
"PLURNK_PROVIDERS_REPEAT_PENALTY",
|
|
207
295
|
"PLURNK_PROVIDERS_FREQUENCY_PENALTY",
|
|
208
296
|
"PLURNK_PROVIDERS_SERVICE_TIER",
|
|
209
|
-
"
|
|
297
|
+
"PLURNK_PROVIDERS_CACHE_WRITE_POLICY",
|
|
298
|
+
"PLURNK_PROVIDERS_CACHE_AFFINITY",
|
|
210
299
|
"PLURNK_PROVIDERS_REPEAT_LAST_N",
|
|
211
300
|
"PLURNK_PROVIDERS_DRY_MULTIPLIER",
|
|
212
301
|
"PLURNK_PROVIDERS_DRY_BASE",
|
|
@@ -225,10 +314,9 @@ export const PROVIDERS_KNOBS = Object.freeze([
|
|
|
225
314
|
// single overlay.
|
|
226
315
|
//
|
|
227
316
|
// `knobs` (optional) lets a CONSUMER scope its OWN closed knob list with this
|
|
228
|
-
// same parser — e.g.
|
|
229
|
-
//
|
|
230
|
-
//
|
|
231
|
-
// rules. Default stays the providers-family list; my call sites pass nothing.
|
|
317
|
+
// same parser — e.g. service loop policy or prompt projection — without
|
|
318
|
+
// reimplementing the suffix/collision rules. Default stays the
|
|
319
|
+
// providers-family list; provider call sites pass nothing.
|
|
232
320
|
export const scopeEnvToAlias = (env: NodeJS.ProcessEnv, alias: string, knobs: readonly string[] = PROVIDERS_KNOBS): NodeJS.ProcessEnv => {
|
|
233
321
|
const folded = alias.toLowerCase();
|
|
234
322
|
const out: NodeJS.ProcessEnv = { ...env };
|
package/src/errors.test.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { APICallError, RetryError } from "ai";
|
|
4
|
-
import { ProviderError, classifyProviderError, toProviderError } from "./errors.ts";
|
|
4
|
+
import { ProviderError, ProviderTimeoutError, classifyProviderError, toProviderError } from "./errors.ts";
|
|
5
5
|
import type { ProviderAttempt } from "./types.ts";
|
|
6
6
|
import { providerSource } from "./notices.ts";
|
|
7
7
|
|
|
@@ -29,12 +29,46 @@ test("classifyProviderError maps HTTP status to kind", () => {
|
|
|
29
29
|
assert.equal(k(403), "unauthorized");
|
|
30
30
|
assert.equal(k(402), "quota_exceeded");
|
|
31
31
|
assert.equal(k(429), "rate_limit");
|
|
32
|
+
assert.equal(k(408), "network_failure");
|
|
33
|
+
assert.equal(k(409), "network_failure");
|
|
32
34
|
assert.equal(k(500), "network_failure");
|
|
33
35
|
assert.equal(k(503), "network_failure");
|
|
36
|
+
assert.equal(k(413), "capacity_exceeded");
|
|
34
37
|
assert.equal(k(400), "invalid_response");
|
|
35
38
|
assert.equal(k(404), "invalid_response");
|
|
36
39
|
});
|
|
37
40
|
|
|
41
|
+
test("capacity normalization prefers structured provider codes and keeps generic 400s distinct", () => {
|
|
42
|
+
const openai = apiError(400, JSON.stringify({
|
|
43
|
+
error: {
|
|
44
|
+
type: "invalid_request_error",
|
|
45
|
+
code: "context_length_exceeded",
|
|
46
|
+
message: "maximum context length exceeded",
|
|
47
|
+
},
|
|
48
|
+
}));
|
|
49
|
+
assert.equal(classifyProviderError(openai).kind, "capacity_exceeded");
|
|
50
|
+
const normalized = toProviderError(openai, "provider:openai");
|
|
51
|
+
assert.equal(normalized.status, 413);
|
|
52
|
+
assert.equal(normalized.problem.capacityStage, "upstream");
|
|
53
|
+
assert.equal(normalized.problem.providerStatus, 400);
|
|
54
|
+
assert.equal(classifyProviderError(apiError(400, JSON.stringify({
|
|
55
|
+
error: { type: "invalid_request_error", code: "bad_temperature", message: "bad temperature" },
|
|
56
|
+
}))).kind, "invalid_response");
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
test("provider retry directives survive HTTP failure normalization", () => {
|
|
60
|
+
const final = new APICallError({
|
|
61
|
+
message: "edge router says not to replay",
|
|
62
|
+
url: "https://example.test/v1/chat/completions",
|
|
63
|
+
requestBodyValues: {},
|
|
64
|
+
statusCode: 524,
|
|
65
|
+
isRetryable: false,
|
|
66
|
+
});
|
|
67
|
+
const error = toProviderError(final, "provider:test");
|
|
68
|
+
assert.equal(error.kind, "network_failure");
|
|
69
|
+
assert.equal(error.problem.retryable, false);
|
|
70
|
+
});
|
|
71
|
+
|
|
38
72
|
test("classifyProviderError: a 422 flagged grammar_invalid is distinct; other 422s are invalid responses", () => {
|
|
39
73
|
const rejected = apiError(422, JSON.stringify({ error: { type: "grammar_invalid", message: "non-conforming emission rejected: ..." } }));
|
|
40
74
|
assert.equal(classifyProviderError(rejected).kind, "grammar_invalid");
|
|
@@ -75,11 +109,31 @@ test("#161: ProviderError carries resource-interrupted attempt evidence outside
|
|
|
75
109
|
assistant: {
|
|
76
110
|
content: "partial",
|
|
77
111
|
reasoning: null,
|
|
78
|
-
usage: { prompt: 3, completion: 1, reasoning: 0, cached: 0, total: 4 },
|
|
79
112
|
finishReason: "resource_interrupted",
|
|
80
113
|
model: "served-model",
|
|
81
114
|
},
|
|
82
115
|
assistantRaw: { rawFinishReason: "insufficient_system_resource" },
|
|
116
|
+
accounting: [{
|
|
117
|
+
provider: "provider:deepseek",
|
|
118
|
+
model: "served-model",
|
|
119
|
+
outcome: "response",
|
|
120
|
+
usage: { inputTokens: 3, outputTokens: 1, totalTokens: 4 },
|
|
121
|
+
cost: { kind: "unknown", reason: "fixture has no monetary evidence" },
|
|
122
|
+
}],
|
|
123
|
+
capacity: {
|
|
124
|
+
decision: "defer",
|
|
125
|
+
contextWindow: null,
|
|
126
|
+
maxInputTokens: null,
|
|
127
|
+
maxOutputTokens: null,
|
|
128
|
+
outputBudget: null,
|
|
129
|
+
reasoningBudget: null,
|
|
130
|
+
inputCapacity: null,
|
|
131
|
+
prompt: {
|
|
132
|
+
kind: "unavailable",
|
|
133
|
+
source: "fixture",
|
|
134
|
+
detail: "the interrupted fixture has no preflight measurement",
|
|
135
|
+
},
|
|
136
|
+
},
|
|
83
137
|
} as ProviderAttempt;
|
|
84
138
|
const error = new ProviderError(
|
|
85
139
|
"provider:deepseek",
|
|
@@ -96,6 +150,7 @@ test("#161: ProviderError carries resource-interrupted attempt evidence outside
|
|
|
96
150
|
);
|
|
97
151
|
|
|
98
152
|
assert.equal(error.attempt, attempt);
|
|
153
|
+
assert.deepEqual(error.accounting, attempt.accounting);
|
|
99
154
|
assert.deepEqual(error.problem, {
|
|
100
155
|
type: "https://problems.plurnk.dev/provider/deepseek/resource-interrupted",
|
|
101
156
|
title: "Resource interrupted",
|
|
@@ -146,3 +201,26 @@ test("retry exhaustion is explicit and does not recommend another automatic repl
|
|
|
146
201
|
assert.equal(error.problem.retryExhausted, true);
|
|
147
202
|
assert.equal(error.cause, cause);
|
|
148
203
|
});
|
|
204
|
+
|
|
205
|
+
test("retry exhaustion retains the exact inner deadline phase", () => {
|
|
206
|
+
const failures = [1, 2].map(() => new APICallError({
|
|
207
|
+
message: "attempt timed out",
|
|
208
|
+
url: "https://example.test/v1/chat/completions",
|
|
209
|
+
requestBodyValues: {},
|
|
210
|
+
cause: new ProviderTimeoutError("attempt", 10),
|
|
211
|
+
isRetryable: true,
|
|
212
|
+
}));
|
|
213
|
+
const cause = new RetryError({
|
|
214
|
+
message: "Failed after 2 attempts.",
|
|
215
|
+
reason: "maxRetriesExceeded",
|
|
216
|
+
errors: failures,
|
|
217
|
+
});
|
|
218
|
+
const error = toProviderError(cause, "provider:test");
|
|
219
|
+
assert.equal(error.kind, "network_failure");
|
|
220
|
+
assert.equal(error.status, 503);
|
|
221
|
+
assert.equal(error.problem.retryable, false);
|
|
222
|
+
assert.equal(error.problem.attempts, 2);
|
|
223
|
+
assert.equal(error.problem.retryExhausted, true);
|
|
224
|
+
assert.equal(error.problem.timeoutPhase, "attempt");
|
|
225
|
+
assert.equal(error.problem.timeoutMs, 10);
|
|
226
|
+
});
|
package/src/errors.ts
CHANGED
|
@@ -1,16 +1,18 @@
|
|
|
1
1
|
import { Problems, type ProblemDetails } from "@plurnk/plurnk-contracts";
|
|
2
2
|
import { APICallError, RetryError } from "ai";
|
|
3
3
|
import { providerSource } from "./notices.ts";
|
|
4
|
-
import type { ProviderAttempt } from "./types.ts";
|
|
4
|
+
import type { ProviderAttempt, ProviderRequestAccounting, ProviderRequestCapacity } from "./types.ts";
|
|
5
5
|
|
|
6
6
|
export type ProviderErrorKind =
|
|
7
7
|
| "rate_limit"
|
|
8
8
|
| "network_failure"
|
|
9
|
+
| "deadline_exceeded"
|
|
9
10
|
| "model_refused"
|
|
10
11
|
| "invalid_response"
|
|
11
12
|
| "unauthorized"
|
|
12
13
|
| "quota_exceeded"
|
|
13
14
|
| "grammar_invalid"
|
|
15
|
+
| "capacity_exceeded"
|
|
14
16
|
| "resource_interrupted";
|
|
15
17
|
|
|
16
18
|
export interface ClassifiedProviderError {
|
|
@@ -19,16 +21,50 @@ export interface ClassifiedProviderError {
|
|
|
19
21
|
retryable?: boolean;
|
|
20
22
|
attempts?: number;
|
|
21
23
|
retryExhausted?: boolean;
|
|
24
|
+
extensions?: Readonly<Record<string, unknown>>;
|
|
22
25
|
}
|
|
23
26
|
|
|
27
|
+
export type ProviderTimeoutPhase = "attempt" | "first_content" | "stream_idle" | "operation";
|
|
28
|
+
|
|
29
|
+
export class ProviderTimeoutError extends Error {
|
|
30
|
+
readonly phase: ProviderTimeoutPhase;
|
|
31
|
+
readonly timeoutMs: number;
|
|
32
|
+
|
|
33
|
+
constructor(phase: ProviderTimeoutPhase, timeoutMs: number, cause?: unknown) {
|
|
34
|
+
const labels: Record<ProviderTimeoutPhase, string> = {
|
|
35
|
+
attempt: "Provider attempt",
|
|
36
|
+
first_content: "First provider content",
|
|
37
|
+
stream_idle: "Provider stream idle",
|
|
38
|
+
operation: "Provider operation",
|
|
39
|
+
};
|
|
40
|
+
super(`${labels[phase]} exceeded its ${timeoutMs} ms deadline.`, cause === undefined ? undefined : { cause });
|
|
41
|
+
this.name = "ProviderTimeoutError";
|
|
42
|
+
this.phase = phase;
|
|
43
|
+
this.timeoutMs = timeoutMs;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export const providerTimeoutOf = (error: unknown): ProviderTimeoutError | null => {
|
|
48
|
+
const seen = new Set<unknown>();
|
|
49
|
+
let current = error;
|
|
50
|
+
while (typeof current === "object" && current !== null && !seen.has(current)) {
|
|
51
|
+
if (current instanceof ProviderTimeoutError) return current;
|
|
52
|
+
seen.add(current);
|
|
53
|
+
current = (current as { cause?: unknown }).cause;
|
|
54
|
+
}
|
|
55
|
+
return null;
|
|
56
|
+
};
|
|
57
|
+
|
|
24
58
|
const defaultStatus = (kind: ProviderErrorKind): number => {
|
|
25
59
|
switch (kind) {
|
|
26
60
|
case "unauthorized": return 401;
|
|
27
61
|
case "quota_exceeded": return 402;
|
|
62
|
+
case "capacity_exceeded": return 413;
|
|
28
63
|
case "rate_limit": return 429;
|
|
29
64
|
case "model_refused":
|
|
30
65
|
case "grammar_invalid": return 422;
|
|
31
66
|
case "invalid_response": return 502;
|
|
67
|
+
case "deadline_exceeded": return 504;
|
|
32
68
|
case "network_failure":
|
|
33
69
|
case "resource_interrupted": return 503;
|
|
34
70
|
}
|
|
@@ -39,8 +75,10 @@ const retryable = (kind: ProviderErrorKind): boolean => {
|
|
|
39
75
|
case "rate_limit":
|
|
40
76
|
case "network_failure":
|
|
41
77
|
return true;
|
|
78
|
+
case "deadline_exceeded":
|
|
42
79
|
case "invalid_response":
|
|
43
80
|
case "grammar_invalid":
|
|
81
|
+
case "capacity_exceeded":
|
|
44
82
|
case "resource_interrupted":
|
|
45
83
|
case "model_refused":
|
|
46
84
|
case "unauthorized":
|
|
@@ -60,11 +98,13 @@ const buildProblem = (
|
|
|
60
98
|
const code: Record<ProviderErrorKind, string> = {
|
|
61
99
|
rate_limit: "rate-limit",
|
|
62
100
|
network_failure: "network-failure",
|
|
101
|
+
deadline_exceeded: "deadline-exceeded",
|
|
63
102
|
model_refused: "model-refused",
|
|
64
103
|
invalid_response: "invalid-response",
|
|
65
104
|
unauthorized: "unauthorized",
|
|
66
105
|
quota_exceeded: "quota-exceeded",
|
|
67
106
|
grammar_invalid: "grammar-invalid",
|
|
107
|
+
capacity_exceeded: "capacity-exceeded",
|
|
68
108
|
resource_interrupted: "resource-interrupted",
|
|
69
109
|
};
|
|
70
110
|
return Problems.create(source, code[kind], status, message, {
|
|
@@ -84,6 +124,8 @@ export class ProviderError extends Error {
|
|
|
84
124
|
readonly kind: ProviderErrorKind;
|
|
85
125
|
readonly problem: ProblemDetails;
|
|
86
126
|
readonly attempt?: ProviderAttempt;
|
|
127
|
+
readonly capacity?: ProviderRequestCapacity;
|
|
128
|
+
#accounting: ProviderRequestAccounting[];
|
|
87
129
|
|
|
88
130
|
constructor(
|
|
89
131
|
source: string,
|
|
@@ -95,6 +137,8 @@ export class ProviderError extends Error {
|
|
|
95
137
|
retryable?: boolean;
|
|
96
138
|
extensions?: Readonly<Record<string, unknown>>;
|
|
97
139
|
attempt?: ProviderAttempt;
|
|
140
|
+
accounting?: readonly ProviderRequestAccounting[];
|
|
141
|
+
capacity?: ProviderRequestCapacity;
|
|
98
142
|
} = {},
|
|
99
143
|
) {
|
|
100
144
|
super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
|
|
@@ -102,6 +146,8 @@ export class ProviderError extends Error {
|
|
|
102
146
|
this.source = providerSource(source);
|
|
103
147
|
this.kind = kind;
|
|
104
148
|
this.attempt = options.attempt;
|
|
149
|
+
this.capacity = options.capacity ?? options.attempt?.capacity;
|
|
150
|
+
this.#accounting = [...(options.accounting ?? options.attempt?.accounting ?? [])];
|
|
105
151
|
const status = options.status !== null && options.status !== undefined
|
|
106
152
|
&& Number.isInteger(options.status) && options.status >= 400 && options.status <= 599
|
|
107
153
|
? options.status
|
|
@@ -119,17 +165,35 @@ export class ProviderError extends Error {
|
|
|
119
165
|
get status(): number {
|
|
120
166
|
return this.problem.status;
|
|
121
167
|
}
|
|
168
|
+
|
|
169
|
+
get accounting(): readonly ProviderRequestAccounting[] {
|
|
170
|
+
return this.#accounting;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
// A capacity pool adds the already-settled requests from prior backends as
|
|
174
|
+
// the same failure crosses that orchestration boundary.
|
|
175
|
+
prependAccounting(accounting: readonly ProviderRequestAccounting[]): void {
|
|
176
|
+
if (accounting.length > 0) this.#accounting = [...accounting, ...this.#accounting];
|
|
177
|
+
}
|
|
122
178
|
}
|
|
123
179
|
|
|
124
|
-
const
|
|
180
|
+
const wireError = (body: string): { type: string | null; code: string | null; message: string | null } => {
|
|
125
181
|
try {
|
|
126
182
|
const { error } = JSON.parse(body) as { error?: { type?: unknown } };
|
|
127
|
-
|
|
183
|
+
const record = error as { type?: unknown; code?: unknown; message?: unknown } | undefined;
|
|
184
|
+
return {
|
|
185
|
+
type: typeof record?.type === "string" ? record.type : null,
|
|
186
|
+
code: typeof record?.code === "string" || typeof record?.code === "number" ? String(record.code) : null,
|
|
187
|
+
message: typeof record?.message === "string" ? record.message : null,
|
|
188
|
+
};
|
|
128
189
|
} catch {
|
|
129
|
-
return null;
|
|
190
|
+
return { type: null, code: null, message: null };
|
|
130
191
|
}
|
|
131
192
|
};
|
|
132
193
|
|
|
194
|
+
const CAPACITY_CODE = /^(?:context_length_exceeded|context_window_exceeded|input_too_long|prompt_too_long|request_too_large|token_limit_exceeded|max_tokens_exceeded)$/i;
|
|
195
|
+
const CAPACITY_MESSAGE = /(?:maximum context length|context (?:length|window).*(?:exceed|too (?:large|long)|maximum)|(?:input|prompt|request).*(?:token|length|size).*(?:exceed|too (?:large|long)|maximum))/i;
|
|
196
|
+
|
|
133
197
|
const preview = (value: unknown, limit: number | undefined): string => {
|
|
134
198
|
const text = value instanceof Error ? value.message : String(value);
|
|
135
199
|
return limit !== undefined && text.length > limit
|
|
@@ -150,17 +214,46 @@ export const classifyProviderError = (
|
|
|
150
214
|
};
|
|
151
215
|
}
|
|
152
216
|
if (APICallError.isInstance(err)) {
|
|
217
|
+
const timeout = providerTimeoutOf(err);
|
|
218
|
+
if (timeout !== null) {
|
|
219
|
+
return {
|
|
220
|
+
kind: "network_failure",
|
|
221
|
+
message: timeout.message,
|
|
222
|
+
extensions: {
|
|
223
|
+
timeoutPhase: timeout.phase,
|
|
224
|
+
timeoutMs: timeout.timeoutMs,
|
|
225
|
+
},
|
|
226
|
+
};
|
|
227
|
+
}
|
|
153
228
|
const status = err.statusCode ?? 0;
|
|
154
229
|
const message = err.message.trim().length > 0
|
|
155
230
|
? preview(err.message, detailLimit)
|
|
156
231
|
: "The provider request failed without a diagnostic message.";
|
|
157
232
|
const body = err.responseBody ?? "";
|
|
233
|
+
const wire = wireError(body);
|
|
158
234
|
if (status === 401 || status === 403) return { kind: "unauthorized", message };
|
|
159
235
|
if (status === 402) return { kind: "quota_exceeded", message };
|
|
160
|
-
if (status === 429) return { kind: "rate_limit", message };
|
|
236
|
+
if (status === 429) return { kind: "rate_limit", message, retryable: err.isRetryable };
|
|
237
|
+
if (status === 408 || status === 409) {
|
|
238
|
+
return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
239
|
+
}
|
|
161
240
|
if (status === 0 && err.isRetryable) return { kind: "network_failure", message };
|
|
162
|
-
if (status >= 500) return { kind: "network_failure", message };
|
|
163
|
-
if (status ===
|
|
241
|
+
if (status >= 500) return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
242
|
+
if (status === 413 || (
|
|
243
|
+
(status === 400 || status === 422)
|
|
244
|
+
&& (
|
|
245
|
+
(wire.code !== null && CAPACITY_CODE.test(wire.code))
|
|
246
|
+
|| (wire.type !== null && CAPACITY_CODE.test(wire.type))
|
|
247
|
+
|| CAPACITY_MESSAGE.test(wire.message ?? message)
|
|
248
|
+
)
|
|
249
|
+
)) {
|
|
250
|
+
return {
|
|
251
|
+
kind: "capacity_exceeded",
|
|
252
|
+
message,
|
|
253
|
+
extensions: status === 413 ? undefined : { providerStatus: status },
|
|
254
|
+
};
|
|
255
|
+
}
|
|
256
|
+
if (status === 422 && wire.type === "grammar_invalid") {
|
|
164
257
|
return { kind: "grammar_invalid", message };
|
|
165
258
|
}
|
|
166
259
|
return { kind: "invalid_response", message };
|
|
@@ -188,21 +281,27 @@ export const toProviderError = (
|
|
|
188
281
|
err: unknown,
|
|
189
282
|
source: string,
|
|
190
283
|
detailLimit?: number,
|
|
284
|
+
capacity?: ProviderRequestCapacity,
|
|
191
285
|
): ProviderError => {
|
|
192
286
|
if (err instanceof ProviderError) return err;
|
|
193
287
|
const underlying = RetryError.isInstance(err) ? err.lastError : err;
|
|
194
288
|
const classified = classifyProviderError(err, detailLimit);
|
|
195
289
|
const { kind, message } = classified;
|
|
196
|
-
const
|
|
290
|
+
const upstreamStatus = APICallError.isInstance(underlying) ? underlying.statusCode ?? null : null;
|
|
291
|
+
const status = kind === "capacity_exceeded" ? 413 : upstreamStatus;
|
|
197
292
|
return new ProviderError(source, kind, message, {
|
|
198
293
|
status,
|
|
199
294
|
cause: err,
|
|
200
295
|
retryable: classified.retryable,
|
|
201
296
|
extensions: {
|
|
297
|
+
...(classified.extensions ?? {}),
|
|
298
|
+
...(kind === "capacity_exceeded" ? { capacityStage: "upstream" } : {}),
|
|
299
|
+
...(capacity === undefined ? {} : { capacity }),
|
|
202
300
|
...(classified.attempts === undefined ? {} : { attempts: classified.attempts }),
|
|
203
301
|
...(classified.retryExhausted === undefined
|
|
204
302
|
? {}
|
|
205
303
|
: { retryExhausted: classified.retryExhausted }),
|
|
206
304
|
},
|
|
305
|
+
capacity,
|
|
207
306
|
});
|
|
208
307
|
};
|