@plurnk/plurnk-providers 1.6.0 → 1.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +9 -16
- package/README.md +15 -0
- package/SPEC.md +124 -45
- package/dist/AiSdkProvider.d.ts +16 -9
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +151 -54
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +8 -4
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +53 -18
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -4
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +68 -12
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +18 -20
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +10 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/env.d.ts +8 -10
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +54 -37
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +5 -3
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +33 -6
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/sdkModels.d.ts +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +5 -8
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +24 -4
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +7 -2
- package/dist/usage.js.map +1 -1
- package/package.json +17 -7
- package/src/AiSdkProvider.test.ts +198 -37
- package/src/AiSdkProvider.ts +192 -59
- package/src/Mock.test.ts +32 -18
- package/src/Mock.ts +58 -19
- package/src/Pool.test.ts +71 -13
- package/src/Pool.ts +78 -13
- package/src/ProviderRegistry.test.ts +1 -1
- package/src/accounting.ts +0 -1
- package/src/accountingPublic.ts +9 -0
- package/src/boundaries.test.ts +1 -0
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +82 -9
- package/src/catalogProvider.ts +24 -21
- package/src/compatibleProvider.test.ts +1 -2
- package/src/compatibleProvider.ts +10 -7
- package/src/cost.test.ts +31 -0
- package/src/env.test.ts +49 -20
- package/src/env.ts +114 -51
- package/src/errors.test.ts +33 -0
- package/src/errors.ts +41 -6
- package/src/index.ts +5 -2
- package/src/ollama.test.ts +1 -2
- package/src/promptTokens.ts +8 -5
- package/src/sdkModels.test.ts +2 -5
- package/src/sdkModels.ts +6 -8
- package/src/types.ts +41 -19
- package/src/usage.ts +7 -2
package/src/env.ts
CHANGED
|
@@ -113,20 +113,14 @@ export function effectiveContextWindow(operatorCap: number | null, naturalWindow
|
|
|
113
113
|
: Math.min(operatorCap, naturalWindow);
|
|
114
114
|
}
|
|
115
115
|
|
|
116
|
-
// {§provider-generation-envelope}
|
|
117
|
-
//
|
|
118
|
-
//
|
|
119
|
-
//
|
|
120
|
-
|
|
121
|
-
// ("10%") or an absolute token count ("4096"); the floor ships percentages so
|
|
122
|
-
// every window-advertising endpoint (llama-server n_ctx, the plurnk.ai router,
|
|
123
|
-
// a cataloged cloud model) arrives at sane defaults with ZERO operator tuning.
|
|
124
|
-
// Per-alias suffixes override for measured envelopes; absolutes win over the
|
|
125
|
-
// window derivation entirely.
|
|
126
|
-
export type ReserveSpec = { percent: number } | { tokens: number };
|
|
116
|
+
// {§provider-generation-envelope} Generation has one total output budget. An
|
|
117
|
+
// optional reasoning budget is a subset, never an additive second reserve.
|
|
118
|
+
// Percentages are of the effective context window; absolutes remain useful for
|
|
119
|
+
// measured local deployments. Physical model limits always cap operator policy.
|
|
120
|
+
export type TokenBudgetSpec = { percent: number } | { tokens: number };
|
|
127
121
|
|
|
128
|
-
const
|
|
129
|
-
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the window like "
|
|
122
|
+
const parseTokenBudget = (raw: string | undefined, name: string, label: string): TokenBudgetSpec => {
|
|
123
|
+
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the context window like "35%", or an absolute token count)`);
|
|
130
124
|
const pct = /^([0-9]+(?:\.[0-9]+)?)%$/.exec(raw);
|
|
131
125
|
if (pct !== null) {
|
|
132
126
|
const p = Number(pct[1]);
|
|
@@ -138,37 +132,110 @@ const parseReserve = (raw: string | undefined, name: string, label: string): Res
|
|
|
138
132
|
return { tokens: n };
|
|
139
133
|
};
|
|
140
134
|
|
|
141
|
-
|
|
142
|
-
reasoningReserve: parseReserve(env.PLURNK_PROVIDERS_REASONING_RESERVE, "PLURNK_PROVIDERS_REASONING_RESERVE", label),
|
|
143
|
-
completionReserve: parseReserve(env.PLURNK_PROVIDERS_COMPLETION_RESERVE, "PLURNK_PROVIDERS_COMPLETION_RESERVE", label),
|
|
144
|
-
});
|
|
145
|
-
|
|
146
|
-
// Resolve a ReserveSpec against a known window: absolutes stand alone; a
|
|
135
|
+
// Resolve a token budget against a known window: absolutes stand alone; a
|
|
147
136
|
// percentage needs the window (null when unknown — the underivable/no-cap case).
|
|
148
|
-
export const
|
|
149
|
-
"tokens" in spec
|
|
137
|
+
export const resolveTokenBudget = (spec: TokenBudgetSpec, window: number | null): number | null =>
|
|
138
|
+
"tokens" in spec
|
|
139
|
+
? spec.tokens
|
|
140
|
+
: window === null
|
|
141
|
+
? null
|
|
142
|
+
: Math.max(1, Math.round(spec.percent * window));
|
|
150
143
|
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
}
|
|
144
|
+
export type GenerationEnvelope = {
|
|
145
|
+
readonly outputBudget: number | null;
|
|
146
|
+
readonly reasoningBudget: number | null;
|
|
147
|
+
};
|
|
148
|
+
|
|
149
|
+
const shedRetiredEnvelope = (env: NodeJS.ProcessEnv, label: string): void => {
|
|
150
|
+
for (const name of ["PLURNK_PROVIDERS_REASONING_RESERVE", "PLURNK_PROVIDERS_COMPLETION_RESERVE"] as const) {
|
|
151
|
+
if (env[name] !== undefined && env[name] !== "") {
|
|
152
|
+
throw new Error(
|
|
153
|
+
`${label} provider: ${name} is retired; reasoning is now a subset of the total generation envelope. Replace the old pair with PLURNK_PROVIDERS_OUTPUT_BUDGET and optional PLURNK_PROVIDERS_REASONING_BUDGET ({§provider-generation-envelope})`,
|
|
154
|
+
);
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
};
|
|
158
|
+
|
|
159
|
+
const optionalTokenBudget = (
|
|
160
|
+
raw: string | undefined,
|
|
161
|
+
name: string,
|
|
162
|
+
label: string,
|
|
163
|
+
): TokenBudgetSpec | null => raw === undefined || raw.length === 0
|
|
164
|
+
? null
|
|
165
|
+
: parseTokenBudget(raw, name, label);
|
|
166
|
+
|
|
167
|
+
const resolveGeneration = (
|
|
168
|
+
outputSpec: TokenBudgetSpec | null,
|
|
169
|
+
reasoningSpec: TokenBudgetSpec | null,
|
|
170
|
+
contextWindow: number | null,
|
|
171
|
+
maxOutputTokens: number | null,
|
|
172
|
+
label: string,
|
|
173
|
+
): GenerationEnvelope => {
|
|
174
|
+
const requestedOutput = outputSpec === null ? null : resolveTokenBudget(outputSpec, contextWindow);
|
|
175
|
+
const physicalCaps = [contextWindow, maxOutputTokens].filter((value): value is number => value !== null);
|
|
176
|
+
const outputBudget = requestedOutput === null
|
|
177
|
+
? null
|
|
178
|
+
: Math.min(requestedOutput, ...physicalCaps);
|
|
179
|
+
if (contextWindow !== null && outputBudget !== null && outputBudget >= contextWindow) {
|
|
180
|
+
throw new Error(
|
|
181
|
+
`${label} provider: PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}) must leave positive input capacity inside the context window (${contextWindow})`,
|
|
182
|
+
);
|
|
183
|
+
}
|
|
184
|
+
const reasoningBudget = reasoningSpec === null
|
|
185
|
+
? null
|
|
186
|
+
: resolveTokenBudget(reasoningSpec, contextWindow);
|
|
187
|
+
if (reasoningBudget !== null && outputBudget === null) {
|
|
188
|
+
throw new Error(
|
|
189
|
+
`${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET; reasoning is a subset of total output`,
|
|
190
|
+
);
|
|
191
|
+
}
|
|
192
|
+
if (reasoningBudget !== null && outputBudget !== null && reasoningBudget >= outputBudget) {
|
|
193
|
+
throw new Error(
|
|
194
|
+
`${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET (${reasoningBudget}) exceeds the effective PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}); reasoning is a subset of total output`,
|
|
195
|
+
);
|
|
196
|
+
}
|
|
197
|
+
return { outputBudget, reasoningBudget };
|
|
198
|
+
};
|
|
199
|
+
|
|
200
|
+
// Standard providers receive the shipped OUTPUT_BUDGET floor and fail hard if
|
|
201
|
+
// it is absent. Mock uses the tolerant sibling below so ordinary unit fixtures
|
|
202
|
+
// make no generation claim unless a test deliberately configures one.
|
|
203
|
+
export const generationEnvelopeFromEnv = (
|
|
204
|
+
env: NodeJS.ProcessEnv,
|
|
205
|
+
label: string,
|
|
206
|
+
contextWindow: number | null,
|
|
207
|
+
maxOutputTokens: number | null,
|
|
208
|
+
): GenerationEnvelope => {
|
|
209
|
+
shedRetiredEnvelope(env, label);
|
|
210
|
+
return resolveGeneration(
|
|
211
|
+
parseTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", label),
|
|
212
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", label),
|
|
213
|
+
contextWindow,
|
|
214
|
+
maxOutputTokens,
|
|
215
|
+
label,
|
|
216
|
+
);
|
|
217
|
+
};
|
|
218
|
+
|
|
219
|
+
export const resolveGenerationEnvelopeFromEnv = (
|
|
220
|
+
env: NodeJS.ProcessEnv,
|
|
221
|
+
contextWindow: number | null,
|
|
222
|
+
maxOutputTokens: number | null = null,
|
|
223
|
+
): GenerationEnvelope => {
|
|
224
|
+
shedRetiredEnvelope(env, "mock");
|
|
225
|
+
return resolveGeneration(
|
|
226
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", "mock"),
|
|
227
|
+
optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", "mock"),
|
|
228
|
+
contextWindow,
|
|
229
|
+
maxOutputTokens,
|
|
230
|
+
"mock",
|
|
231
|
+
);
|
|
164
232
|
};
|
|
165
233
|
|
|
166
234
|
// {§provider-configuration} The side-channel reasoning knobs — activation and budget
|
|
167
235
|
// are separate vars, so a numeric budget can never silently flip wire flags:
|
|
168
236
|
// PLURNK_PROVIDERS_REASONING off | adaptive | on (REQUIRED, fail-hard)
|
|
169
|
-
// PLURNK_PROVIDERS_REASONING_BUDGET optional
|
|
170
|
-
//
|
|
171
|
-
// request-scoped allowance and cannot exceed the physical reasoning reserve.
|
|
237
|
+
// PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
|
|
238
|
+
// output budget, used for tier/budget mapping where the backend supports it.
|
|
172
239
|
// The provider maps intent to the backend's mechanism; the consumer states
|
|
173
240
|
// intent, never mechanism. PLAN is a separate public intended-goals record.
|
|
174
241
|
export type ReasoningMode = "off" | "adaptive" | "on";
|
|
@@ -189,20 +256,18 @@ export const reasoningResponseStyleFromEnv = (
|
|
|
189
256
|
return raw;
|
|
190
257
|
};
|
|
191
258
|
|
|
192
|
-
export const reasoningFromEnv = (
|
|
259
|
+
export const reasoningFromEnv = (
|
|
260
|
+
env: NodeJS.ProcessEnv,
|
|
261
|
+
label: string,
|
|
262
|
+
resolvedBudget: number | null = null,
|
|
263
|
+
): Reasoning => {
|
|
193
264
|
shedRenamed(env, "PLURNK_PROVIDERS_THINKING", "PLURNK_PROVIDERS_REASONING", label, "provider configuration contract"); // lexicon-allow
|
|
194
265
|
shedRenamed(env, "PLURNK_PROVIDERS_THINKING_CAPACITY", "PLURNK_PROVIDERS_REASONING_BUDGET", label, "provider configuration contract"); // lexicon-allow
|
|
195
266
|
const name = "PLURNK_PROVIDERS_REASONING";
|
|
196
267
|
const raw = env[name];
|
|
197
268
|
if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (off | adaptive | on)`);
|
|
198
269
|
if (raw !== "off" && raw !== "adaptive" && raw !== "on") throw new Error(`${label} provider: ${name} must be one of "off", "adaptive", "on" (got "${raw}")`);
|
|
199
|
-
|
|
200
|
-
const capName = "PLURNK_PROVIDERS_REASONING_BUDGET";
|
|
201
|
-
const capRaw = env[capName];
|
|
202
|
-
if (capRaw === undefined || capRaw.length === 0) return { mode: "on", budget: null };
|
|
203
|
-
const n = Number(capRaw);
|
|
204
|
-
if (!Number.isInteger(n) || n <= 0) throw new Error(`${label} provider: ${capName} must be a positive integer (got "${capRaw}")`);
|
|
205
|
-
return { mode: "on", budget: n };
|
|
270
|
+
return { mode: raw, budget: raw === "off" ? null : resolvedBudget };
|
|
206
271
|
};
|
|
207
272
|
|
|
208
273
|
// ── Per-alias knob scoping (per-alias scoping doctrine, user 2026-07-03): PLURNK_PROVIDERS_<KNOB>[_<alias>] ──
|
|
@@ -213,8 +278,7 @@ export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoni
|
|
|
213
278
|
// facts (API keys, canonical endpoints) remain vendor-named; the per-alias
|
|
214
279
|
// endpoint override stays PLURNK_BASEURL_<alias> (its existing precedent).
|
|
215
280
|
export const PROVIDERS_KNOBS = Object.freeze([
|
|
216
|
-
"
|
|
217
|
-
"PLURNK_PROVIDERS_COMPLETION_RESERVE",
|
|
281
|
+
"PLURNK_PROVIDERS_OUTPUT_BUDGET",
|
|
218
282
|
"PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
|
|
219
283
|
"PLURNK_PROVIDERS_REASONING_BUDGET",
|
|
220
284
|
"PLURNK_PROVIDERS_REASONING",
|
|
@@ -250,10 +314,9 @@ export const PROVIDERS_KNOBS = Object.freeze([
|
|
|
250
314
|
// single overlay.
|
|
251
315
|
//
|
|
252
316
|
// `knobs` (optional) lets a CONSUMER scope its OWN closed knob list with this
|
|
253
|
-
// same parser — e.g.
|
|
254
|
-
//
|
|
255
|
-
//
|
|
256
|
-
// rules. Default stays the providers-family list; my call sites pass nothing.
|
|
317
|
+
// same parser — e.g. service loop policy or prompt projection — without
|
|
318
|
+
// reimplementing the suffix/collision rules. Default stays the
|
|
319
|
+
// providers-family list; provider call sites pass nothing.
|
|
257
320
|
export const scopeEnvToAlias = (env: NodeJS.ProcessEnv, alias: string, knobs: readonly string[] = PROVIDERS_KNOBS): NodeJS.ProcessEnv => {
|
|
258
321
|
const folded = alias.toLowerCase();
|
|
259
322
|
const out: NodeJS.ProcessEnv = { ...env };
|
package/src/errors.test.ts
CHANGED
|
@@ -33,10 +33,29 @@ test("classifyProviderError maps HTTP status to kind", () => {
|
|
|
33
33
|
assert.equal(k(409), "network_failure");
|
|
34
34
|
assert.equal(k(500), "network_failure");
|
|
35
35
|
assert.equal(k(503), "network_failure");
|
|
36
|
+
assert.equal(k(413), "capacity_exceeded");
|
|
36
37
|
assert.equal(k(400), "invalid_response");
|
|
37
38
|
assert.equal(k(404), "invalid_response");
|
|
38
39
|
});
|
|
39
40
|
|
|
41
|
+
test("capacity normalization prefers structured provider codes and keeps generic 400s distinct", () => {
|
|
42
|
+
const openai = apiError(400, JSON.stringify({
|
|
43
|
+
error: {
|
|
44
|
+
type: "invalid_request_error",
|
|
45
|
+
code: "context_length_exceeded",
|
|
46
|
+
message: "maximum context length exceeded",
|
|
47
|
+
},
|
|
48
|
+
}));
|
|
49
|
+
assert.equal(classifyProviderError(openai).kind, "capacity_exceeded");
|
|
50
|
+
const normalized = toProviderError(openai, "provider:openai");
|
|
51
|
+
assert.equal(normalized.status, 413);
|
|
52
|
+
assert.equal(normalized.problem.capacityStage, "upstream");
|
|
53
|
+
assert.equal(normalized.problem.providerStatus, 400);
|
|
54
|
+
assert.equal(classifyProviderError(apiError(400, JSON.stringify({
|
|
55
|
+
error: { type: "invalid_request_error", code: "bad_temperature", message: "bad temperature" },
|
|
56
|
+
}))).kind, "invalid_response");
|
|
57
|
+
});
|
|
58
|
+
|
|
40
59
|
test("provider retry directives survive HTTP failure normalization", () => {
|
|
41
60
|
const final = new APICallError({
|
|
42
61
|
message: "edge router says not to replay",
|
|
@@ -101,6 +120,20 @@ test("#161: ProviderError carries resource-interrupted attempt evidence outside
|
|
|
101
120
|
usage: { inputTokens: 3, outputTokens: 1, totalTokens: 4 },
|
|
102
121
|
cost: { kind: "unknown", reason: "fixture has no monetary evidence" },
|
|
103
122
|
}],
|
|
123
|
+
capacity: {
|
|
124
|
+
decision: "defer",
|
|
125
|
+
contextWindow: null,
|
|
126
|
+
maxInputTokens: null,
|
|
127
|
+
maxOutputTokens: null,
|
|
128
|
+
outputBudget: null,
|
|
129
|
+
reasoningBudget: null,
|
|
130
|
+
inputCapacity: null,
|
|
131
|
+
prompt: {
|
|
132
|
+
kind: "unavailable",
|
|
133
|
+
source: "fixture",
|
|
134
|
+
detail: "the interrupted fixture has no preflight measurement",
|
|
135
|
+
},
|
|
136
|
+
},
|
|
104
137
|
} as ProviderAttempt;
|
|
105
138
|
const error = new ProviderError(
|
|
106
139
|
"provider:deepseek",
|
package/src/errors.ts
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import { Problems, type ProblemDetails } from "@plurnk/plurnk-contracts";
|
|
2
2
|
import { APICallError, RetryError } from "ai";
|
|
3
3
|
import { providerSource } from "./notices.ts";
|
|
4
|
-
import type { ProviderAttempt, ProviderRequestAccounting } from "./types.ts";
|
|
4
|
+
import type { ProviderAttempt, ProviderRequestAccounting, ProviderRequestCapacity } from "./types.ts";
|
|
5
5
|
|
|
6
6
|
export type ProviderErrorKind =
|
|
7
7
|
| "rate_limit"
|
|
@@ -12,6 +12,7 @@ export type ProviderErrorKind =
|
|
|
12
12
|
| "unauthorized"
|
|
13
13
|
| "quota_exceeded"
|
|
14
14
|
| "grammar_invalid"
|
|
15
|
+
| "capacity_exceeded"
|
|
15
16
|
| "resource_interrupted";
|
|
16
17
|
|
|
17
18
|
export interface ClassifiedProviderError {
|
|
@@ -58,6 +59,7 @@ const defaultStatus = (kind: ProviderErrorKind): number => {
|
|
|
58
59
|
switch (kind) {
|
|
59
60
|
case "unauthorized": return 401;
|
|
60
61
|
case "quota_exceeded": return 402;
|
|
62
|
+
case "capacity_exceeded": return 413;
|
|
61
63
|
case "rate_limit": return 429;
|
|
62
64
|
case "model_refused":
|
|
63
65
|
case "grammar_invalid": return 422;
|
|
@@ -76,6 +78,7 @@ const retryable = (kind: ProviderErrorKind): boolean => {
|
|
|
76
78
|
case "deadline_exceeded":
|
|
77
79
|
case "invalid_response":
|
|
78
80
|
case "grammar_invalid":
|
|
81
|
+
case "capacity_exceeded":
|
|
79
82
|
case "resource_interrupted":
|
|
80
83
|
case "model_refused":
|
|
81
84
|
case "unauthorized":
|
|
@@ -101,6 +104,7 @@ const buildProblem = (
|
|
|
101
104
|
unauthorized: "unauthorized",
|
|
102
105
|
quota_exceeded: "quota-exceeded",
|
|
103
106
|
grammar_invalid: "grammar-invalid",
|
|
107
|
+
capacity_exceeded: "capacity-exceeded",
|
|
104
108
|
resource_interrupted: "resource-interrupted",
|
|
105
109
|
};
|
|
106
110
|
return Problems.create(source, code[kind], status, message, {
|
|
@@ -120,6 +124,7 @@ export class ProviderError extends Error {
|
|
|
120
124
|
readonly kind: ProviderErrorKind;
|
|
121
125
|
readonly problem: ProblemDetails;
|
|
122
126
|
readonly attempt?: ProviderAttempt;
|
|
127
|
+
readonly capacity?: ProviderRequestCapacity;
|
|
123
128
|
#accounting: ProviderRequestAccounting[];
|
|
124
129
|
|
|
125
130
|
constructor(
|
|
@@ -133,6 +138,7 @@ export class ProviderError extends Error {
|
|
|
133
138
|
extensions?: Readonly<Record<string, unknown>>;
|
|
134
139
|
attempt?: ProviderAttempt;
|
|
135
140
|
accounting?: readonly ProviderRequestAccounting[];
|
|
141
|
+
capacity?: ProviderRequestCapacity;
|
|
136
142
|
} = {},
|
|
137
143
|
) {
|
|
138
144
|
super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
|
|
@@ -140,6 +146,7 @@ export class ProviderError extends Error {
|
|
|
140
146
|
this.source = providerSource(source);
|
|
141
147
|
this.kind = kind;
|
|
142
148
|
this.attempt = options.attempt;
|
|
149
|
+
this.capacity = options.capacity ?? options.attempt?.capacity;
|
|
143
150
|
this.#accounting = [...(options.accounting ?? options.attempt?.accounting ?? [])];
|
|
144
151
|
const status = options.status !== null && options.status !== undefined
|
|
145
152
|
&& Number.isInteger(options.status) && options.status >= 400 && options.status <= 599
|
|
@@ -170,15 +177,23 @@ export class ProviderError extends Error {
|
|
|
170
177
|
}
|
|
171
178
|
}
|
|
172
179
|
|
|
173
|
-
const
|
|
180
|
+
const wireError = (body: string): { type: string | null; code: string | null; message: string | null } => {
|
|
174
181
|
try {
|
|
175
182
|
const { error } = JSON.parse(body) as { error?: { type?: unknown } };
|
|
176
|
-
|
|
183
|
+
const record = error as { type?: unknown; code?: unknown; message?: unknown } | undefined;
|
|
184
|
+
return {
|
|
185
|
+
type: typeof record?.type === "string" ? record.type : null,
|
|
186
|
+
code: typeof record?.code === "string" || typeof record?.code === "number" ? String(record.code) : null,
|
|
187
|
+
message: typeof record?.message === "string" ? record.message : null,
|
|
188
|
+
};
|
|
177
189
|
} catch {
|
|
178
|
-
return null;
|
|
190
|
+
return { type: null, code: null, message: null };
|
|
179
191
|
}
|
|
180
192
|
};
|
|
181
193
|
|
|
194
|
+
const CAPACITY_CODE = /^(?:context_length_exceeded|context_window_exceeded|input_too_long|prompt_too_long|request_too_large|token_limit_exceeded|max_tokens_exceeded)$/i;
|
|
195
|
+
const CAPACITY_MESSAGE = /(?:maximum context length|context (?:length|window).*(?:exceed|too (?:large|long)|maximum)|(?:input|prompt|request).*(?:token|length|size).*(?:exceed|too (?:large|long)|maximum))/i;
|
|
196
|
+
|
|
182
197
|
const preview = (value: unknown, limit: number | undefined): string => {
|
|
183
198
|
const text = value instanceof Error ? value.message : String(value);
|
|
184
199
|
return limit !== undefined && text.length > limit
|
|
@@ -215,6 +230,7 @@ export const classifyProviderError = (
|
|
|
215
230
|
? preview(err.message, detailLimit)
|
|
216
231
|
: "The provider request failed without a diagnostic message.";
|
|
217
232
|
const body = err.responseBody ?? "";
|
|
233
|
+
const wire = wireError(body);
|
|
218
234
|
if (status === 401 || status === 403) return { kind: "unauthorized", message };
|
|
219
235
|
if (status === 402) return { kind: "quota_exceeded", message };
|
|
220
236
|
if (status === 429) return { kind: "rate_limit", message, retryable: err.isRetryable };
|
|
@@ -223,7 +239,21 @@ export const classifyProviderError = (
|
|
|
223
239
|
}
|
|
224
240
|
if (status === 0 && err.isRetryable) return { kind: "network_failure", message };
|
|
225
241
|
if (status >= 500) return { kind: "network_failure", message, retryable: err.isRetryable };
|
|
226
|
-
if (status ===
|
|
242
|
+
if (status === 413 || (
|
|
243
|
+
(status === 400 || status === 422)
|
|
244
|
+
&& (
|
|
245
|
+
(wire.code !== null && CAPACITY_CODE.test(wire.code))
|
|
246
|
+
|| (wire.type !== null && CAPACITY_CODE.test(wire.type))
|
|
247
|
+
|| CAPACITY_MESSAGE.test(wire.message ?? message)
|
|
248
|
+
)
|
|
249
|
+
)) {
|
|
250
|
+
return {
|
|
251
|
+
kind: "capacity_exceeded",
|
|
252
|
+
message,
|
|
253
|
+
extensions: status === 413 ? undefined : { providerStatus: status },
|
|
254
|
+
};
|
|
255
|
+
}
|
|
256
|
+
if (status === 422 && wire.type === "grammar_invalid") {
|
|
227
257
|
return { kind: "grammar_invalid", message };
|
|
228
258
|
}
|
|
229
259
|
return { kind: "invalid_response", message };
|
|
@@ -251,22 +281,27 @@ export const toProviderError = (
|
|
|
251
281
|
err: unknown,
|
|
252
282
|
source: string,
|
|
253
283
|
detailLimit?: number,
|
|
284
|
+
capacity?: ProviderRequestCapacity,
|
|
254
285
|
): ProviderError => {
|
|
255
286
|
if (err instanceof ProviderError) return err;
|
|
256
287
|
const underlying = RetryError.isInstance(err) ? err.lastError : err;
|
|
257
288
|
const classified = classifyProviderError(err, detailLimit);
|
|
258
289
|
const { kind, message } = classified;
|
|
259
|
-
const
|
|
290
|
+
const upstreamStatus = APICallError.isInstance(underlying) ? underlying.statusCode ?? null : null;
|
|
291
|
+
const status = kind === "capacity_exceeded" ? 413 : upstreamStatus;
|
|
260
292
|
return new ProviderError(source, kind, message, {
|
|
261
293
|
status,
|
|
262
294
|
cause: err,
|
|
263
295
|
retryable: classified.retryable,
|
|
264
296
|
extensions: {
|
|
265
297
|
...(classified.extensions ?? {}),
|
|
298
|
+
...(kind === "capacity_exceeded" ? { capacityStage: "upstream" } : {}),
|
|
299
|
+
...(capacity === undefined ? {} : { capacity }),
|
|
266
300
|
...(classified.attempts === undefined ? {} : { attempts: classified.attempts }),
|
|
267
301
|
...(classified.retryExhausted === undefined
|
|
268
302
|
? {}
|
|
269
303
|
: { retryExhausted: classified.retryExhausted }),
|
|
270
304
|
},
|
|
305
|
+
capacity,
|
|
271
306
|
});
|
|
272
307
|
};
|
package/src/index.ts
CHANGED
|
@@ -16,6 +16,8 @@ export type {
|
|
|
16
16
|
ProviderCallKind,
|
|
17
17
|
ProviderGenerateArgs,
|
|
18
18
|
ProviderRequestAccounting,
|
|
19
|
+
ProviderRequestCapacity,
|
|
20
|
+
ProviderRequestCapacityDecision,
|
|
19
21
|
ProviderRequestIdentity,
|
|
20
22
|
ProviderRequestObserver,
|
|
21
23
|
ProviderRequestSettlement,
|
|
@@ -25,6 +27,7 @@ export type {
|
|
|
25
27
|
TokenAlternative,
|
|
26
28
|
} from "./types.ts";
|
|
27
29
|
export { assertPromptTokenMeasurement } from "./promptTokens.ts";
|
|
30
|
+
export { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, requestCapacityDecision } from "./capacity.ts";
|
|
28
31
|
|
|
29
32
|
// Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases, so
|
|
30
33
|
// the "." surface is unchanged for existing importers and there's one source of
|
|
@@ -50,8 +53,8 @@ export type { AiSdkProviderConfig, ReasoningStyle, GrammarStyle } from "./AiSdkP
|
|
|
50
53
|
// DECISION stays the consumer's, by choosing which pool to call.
|
|
51
54
|
export { default as Pool } from "./Pool.ts";
|
|
52
55
|
export type { ProviderFetch } from "./AiSdkProvider.ts";
|
|
53
|
-
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow,
|
|
54
|
-
export type { Reasoning, ReasoningMode, ReasoningResponseStyle,
|
|
56
|
+
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow, generationEnvelopeFromEnv, resolveGenerationEnvelopeFromEnv, resolveTokenBudget, PROVIDERS_KNOBS } from "./env.ts";
|
|
57
|
+
export type { GenerationEnvelope, Reasoning, ReasoningMode, ReasoningResponseStyle, TokenBudgetSpec } from "./env.ts";
|
|
55
58
|
export { normalizeUsage, calculateCostUsdDecimal, validateProviderUsage } from "./usage.ts";
|
|
56
59
|
export {
|
|
57
60
|
addDecimals,
|
package/src/ollama.test.ts
CHANGED
|
@@ -11,8 +11,7 @@ const env = Object.freeze({
|
|
|
11
11
|
PLURNK_PROVIDERS_TEMPERATURE: "0.2",
|
|
12
12
|
PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
|
|
13
13
|
PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
|
|
14
|
-
|
|
15
|
-
PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
14
|
+
PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
|
|
16
15
|
PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
17
16
|
PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
|
|
18
17
|
PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
|
package/src/promptTokens.ts
CHANGED
|
@@ -4,6 +4,7 @@ const KINDS = new Set<PromptTokenMeasurement["kind"]>([
|
|
|
4
4
|
"exact",
|
|
5
5
|
"upper_bound",
|
|
6
6
|
"estimate",
|
|
7
|
+
"unavailable",
|
|
7
8
|
]);
|
|
8
9
|
|
|
9
10
|
export const assertPromptTokenMeasurement = (
|
|
@@ -13,19 +14,21 @@ export const assertPromptTokenMeasurement = (
|
|
|
13
14
|
if (typeof value !== "object" || value === null) {
|
|
14
15
|
throw new TypeError(`${owner}: prompt token measurement must be an object`);
|
|
15
16
|
}
|
|
16
|
-
const candidate = value as
|
|
17
|
-
|
|
17
|
+
const candidate = value as Record<string, unknown>;
|
|
18
|
+
const kind = candidate.kind as PromptTokenMeasurement["kind"];
|
|
19
|
+
if (!KINDS.has(kind)) {
|
|
18
20
|
throw new TypeError(`${owner}: prompt token measurement has invalid kind ${JSON.stringify(candidate.kind)}`);
|
|
19
21
|
}
|
|
20
|
-
if (
|
|
22
|
+
if (kind !== "unavailable"
|
|
23
|
+
&& (!Number.isInteger(candidate.tokens) || (candidate.tokens as number) < 0)) {
|
|
21
24
|
throw new TypeError(`${owner}: prompt token measurement tokens must be a non-negative integer`);
|
|
22
25
|
}
|
|
23
26
|
if (typeof candidate.source !== "string" || candidate.source.length === 0) {
|
|
24
27
|
throw new TypeError(`${owner}: prompt token measurement source must be a non-empty string`);
|
|
25
28
|
}
|
|
26
|
-
if (
|
|
29
|
+
if ((kind === "estimate" || kind === "unavailable")
|
|
27
30
|
&& (typeof candidate.detail !== "string" || candidate.detail.length === 0)) {
|
|
28
|
-
throw new TypeError(`${owner}:
|
|
31
|
+
throw new TypeError(`${owner}: ${kind} prompt token measurement requires detail`);
|
|
29
32
|
}
|
|
30
33
|
return value as PromptTokenMeasurement;
|
|
31
34
|
};
|
package/src/sdkModels.test.ts
CHANGED
|
@@ -19,11 +19,8 @@ test("createSdkModel uses Models.dev provider facts and operator credentials", (
|
|
|
19
19
|
const sdk = createSdkModel("xai", "grok-build-0.1", { XAI_API_KEY: "test-key" });
|
|
20
20
|
assert.notEqual(sdk, null);
|
|
21
21
|
assert.equal(sdk?.catalog?.npm, "@ai-sdk/xai");
|
|
22
|
-
assert.
|
|
23
|
-
assert.
|
|
24
|
-
url: "https://api.x.ai/v1/chat/completions",
|
|
25
|
-
headers: { Authorization: "Bearer test-key" },
|
|
26
|
-
});
|
|
22
|
+
assert.notEqual(sdk?.languageModel, undefined);
|
|
23
|
+
assert.equal(sdk?.compatible, undefined);
|
|
27
24
|
assert.deepEqual(sdk?.cacheAffinity, { target: "header", name: "x-grok-conv-id" });
|
|
28
25
|
assert.notEqual(sdk?.normalizeCost, undefined);
|
|
29
26
|
});
|
package/src/sdkModels.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { createGroq } from "@ai-sdk/groq";
|
|
|
7
7
|
import { createMistral } from "@ai-sdk/mistral";
|
|
8
8
|
import { createOpenAI } from "@ai-sdk/openai";
|
|
9
9
|
import { createTogetherAI } from "@ai-sdk/togetherai";
|
|
10
|
+
import { createXai } from "@ai-sdk/xai";
|
|
10
11
|
import { createOpenRouter } from "@openrouter/ai-sdk-provider";
|
|
11
12
|
import { lookupProvider, type ProviderInfo } from "@plurnk/plurnk-models";
|
|
12
13
|
import type { LanguageModel } from "ai";
|
|
@@ -24,6 +25,7 @@ export type SdkModel = {
|
|
|
24
25
|
readonly cacheAffinity?: CacheAffinity;
|
|
25
26
|
readonly systemCacheProviderOptions?: AiSdkProviderOptions;
|
|
26
27
|
readonly reasoningResponseProviderOptions?: AiSdkProviderOptions;
|
|
28
|
+
readonly additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
27
29
|
readonly catalog: ProviderInfo | null;
|
|
28
30
|
};
|
|
29
31
|
|
|
@@ -162,24 +164,19 @@ export const createSdkModel = (
|
|
|
162
164
|
},
|
|
163
165
|
catalog,
|
|
164
166
|
};
|
|
165
|
-
case "@ai-sdk/xai":
|
|
166
|
-
const key = requireApiKey(provider, env, catalog);
|
|
167
|
-
const compatibleBase = url ?? "https://api.x.ai/v1";
|
|
167
|
+
case "@ai-sdk/xai":
|
|
168
168
|
return {
|
|
169
|
-
|
|
170
|
-
url: `${compatibleBase}/chat/completions`,
|
|
171
|
-
headers: { Authorization: `Bearer ${key}` },
|
|
172
|
-
},
|
|
169
|
+
languageModel: createXai({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).chat(model),
|
|
173
170
|
...(catalog.id === "xai"
|
|
174
171
|
? { cacheAffinity: { target: "header" as const, name: "x-grok-conv-id" } }
|
|
175
172
|
: {}),
|
|
176
173
|
...(normalizeCost === undefined ? {} : { normalizeCost }),
|
|
177
174
|
catalog,
|
|
178
175
|
};
|
|
179
|
-
}
|
|
180
176
|
case "@ai-sdk/anthropic":
|
|
181
177
|
return {
|
|
182
178
|
languageModel: createAnthropic({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
|
|
179
|
+
additiveReasoningProvider: "anthropic",
|
|
183
180
|
...(catalog.id === "anthropic"
|
|
184
181
|
? { systemCacheProviderOptions: { anthropic: { cacheControl } } }
|
|
185
182
|
: {}),
|
|
@@ -195,6 +192,7 @@ export const createSdkModel = (
|
|
|
195
192
|
apiKey: env.AWS_BEARER_TOKEN_BEDROCK,
|
|
196
193
|
baseURL: url,
|
|
197
194
|
}).languageModel(model),
|
|
195
|
+
additiveReasoningProvider: "bedrock",
|
|
198
196
|
catalog,
|
|
199
197
|
};
|
|
200
198
|
case "@openrouter/ai-sdk-provider":
|