@plurnk/plurnk-providers 1.6.0 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/.env.defaults +9 -16
  2. package/README.md +17 -0
  3. package/SPEC.md +128 -45
  4. package/dist/AiSdkProvider.d.ts +16 -9
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +151 -54
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +8 -4
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +53 -18
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -4
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +68 -12
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js.map +1 -1
  18. package/dist/accountingPublic.d.ts +5 -0
  19. package/dist/accountingPublic.d.ts.map +1 -0
  20. package/dist/accountingPublic.js +3 -0
  21. package/dist/accountingPublic.js.map +1 -0
  22. package/dist/capacity.d.ts +26 -0
  23. package/dist/capacity.d.ts.map +1 -0
  24. package/dist/capacity.js +90 -0
  25. package/dist/capacity.js.map +1 -0
  26. package/dist/catalogProvider.d.ts +2 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +18 -20
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +10 -7
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/env.d.ts +8 -10
  34. package/dist/env.d.ts.map +1 -1
  35. package/dist/env.js +54 -37
  36. package/dist/env.js.map +1 -1
  37. package/dist/errors.d.ts +6 -22
  38. package/dist/errors.d.ts.map +1 -1
  39. package/dist/errors.js +30 -91
  40. package/dist/errors.js.map +1 -1
  41. package/dist/index.d.ts +4 -3
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +2 -1
  44. package/dist/index.js.map +1 -1
  45. package/dist/promptTokens.d.ts.map +1 -1
  46. package/dist/promptTokens.js +7 -4
  47. package/dist/promptTokens.js.map +1 -1
  48. package/dist/providerError.d.ts +25 -0
  49. package/dist/providerError.d.ts.map +1 -0
  50. package/dist/providerError.js +91 -0
  51. package/dist/providerError.js.map +1 -0
  52. package/dist/sdkModels.d.ts +1 -0
  53. package/dist/sdkModels.d.ts.map +1 -1
  54. package/dist/sdkModels.js +5 -8
  55. package/dist/sdkModels.js.map +1 -1
  56. package/dist/types.d.ts +24 -4
  57. package/dist/types.d.ts.map +1 -1
  58. package/dist/usage.d.ts +1 -0
  59. package/dist/usage.d.ts.map +1 -1
  60. package/dist/usage.js +7 -2
  61. package/dist/usage.js.map +1 -1
  62. package/package.json +22 -7
  63. package/src/AiSdkProvider.test.ts +198 -37
  64. package/src/AiSdkProvider.ts +192 -59
  65. package/src/Mock.test.ts +32 -18
  66. package/src/Mock.ts +58 -19
  67. package/src/Pool.test.ts +71 -13
  68. package/src/Pool.ts +78 -13
  69. package/src/ProviderRegistry.test.ts +1 -1
  70. package/src/accounting.ts +0 -1
  71. package/src/accountingPublic.ts +9 -0
  72. package/src/boundaries.test.ts +28 -15
  73. package/src/capacity.test.ts +92 -0
  74. package/src/capacity.ts +140 -0
  75. package/src/catalogProvider.test.ts +82 -9
  76. package/src/catalogProvider.ts +24 -21
  77. package/src/compatibleProvider.test.ts +1 -2
  78. package/src/compatibleProvider.ts +10 -7
  79. package/src/cost.test.ts +31 -0
  80. package/src/env.test.ts +49 -20
  81. package/src/env.ts +114 -51
  82. package/src/errors.test.ts +33 -0
  83. package/src/errors.ts +38 -134
  84. package/src/index.ts +5 -2
  85. package/src/ollama.test.ts +1 -2
  86. package/src/promptTokens.ts +8 -5
  87. package/src/providerError.ts +139 -0
  88. package/src/sdkModels.test.ts +2 -5
  89. package/src/sdkModels.ts +6 -8
  90. package/src/types.ts +41 -19
  91. package/src/usage.ts +7 -2
package/src/env.ts CHANGED
@@ -113,20 +113,14 @@ export function effectiveContextWindow(operatorCap: number | null, naturalWindow
113
113
  : Math.min(operatorCap, naturalWindow);
114
114
  }
115
115
 
116
- // {§provider-generation-envelope} How much of a
117
- // effective context window is reserved for reasoning and for completion — the
118
- // remainder (minus the consumer's own packing-safety margin) is the prompt
119
- // gauge. Provider-owned: the window is the effective hard envelope and these are
120
- // amounts OF it. Each knob accepts a percentage of the window
121
- // ("10%") or an absolute token count ("4096"); the floor ships percentages so
122
- // every window-advertising endpoint (llama-server n_ctx, the plurnk.ai router,
123
- // a cataloged cloud model) arrives at sane defaults with ZERO operator tuning.
124
- // Per-alias suffixes override for measured envelopes; absolutes win over the
125
- // window derivation entirely.
126
- export type ReserveSpec = { percent: number } | { tokens: number };
116
+ // {§provider-generation-envelope} Generation has one total output budget. An
117
+ // optional reasoning budget is a subset, never an additive second reserve.
118
+ // Percentages are of the effective context window; absolutes remain useful for
119
+ // measured local deployments. Physical model limits always cap operator policy.
120
+ export type TokenBudgetSpec = { percent: number } | { tokens: number };
127
121
 
128
- const parseReserve = (raw: string | undefined, name: string, label: string): ReserveSpec => {
129
- if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the window like "10%", or an absolute token count)`);
122
+ const parseTokenBudget = (raw: string | undefined, name: string, label: string): TokenBudgetSpec => {
123
+ if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (a percentage of the context window like "35%", or an absolute token count)`);
130
124
  const pct = /^([0-9]+(?:\.[0-9]+)?)%$/.exec(raw);
131
125
  if (pct !== null) {
132
126
  const p = Number(pct[1]);
@@ -138,37 +132,110 @@ const parseReserve = (raw: string | undefined, name: string, label: string): Res
138
132
  return { tokens: n };
139
133
  };
140
134
 
141
- export const envelopeFromEnv = (env: NodeJS.ProcessEnv, label: string): { reasoningReserve: ReserveSpec; completionReserve: ReserveSpec } => ({
142
- reasoningReserve: parseReserve(env.PLURNK_PROVIDERS_REASONING_RESERVE, "PLURNK_PROVIDERS_REASONING_RESERVE", label),
143
- completionReserve: parseReserve(env.PLURNK_PROVIDERS_COMPLETION_RESERVE, "PLURNK_PROVIDERS_COMPLETION_RESERVE", label),
144
- });
145
-
146
- // Resolve a ReserveSpec against a known window: absolutes stand alone; a
135
+ // Resolve a token budget against a known window: absolutes stand alone; a
147
136
  // percentage needs the window (null when unknown — the underivable/no-cap case).
148
- export const resolveReserve = (spec: ReserveSpec, window: number | null): number | null =>
149
- "tokens" in spec ? spec.tokens : window === null ? null : Math.round(spec.percent * window);
137
+ export const resolveTokenBudget = (spec: TokenBudgetSpec, window: number | null): number | null =>
138
+ "tokens" in spec
139
+ ? spec.tokens
140
+ : window === null
141
+ ? null
142
+ : Math.max(1, Math.round(spec.percent * window));
150
143
 
151
- // TOLERANT envelope read for fixtures + consumers that resolve reserves against
152
- // a known window and treat ABSENCE as "no claim" (null), never fail-hard —
153
- // unlike envelopeFromEnv (the REQUIRED provider path, fed by the shipped floor).
154
- // Mock uses this so the service's partition/budget suite drives the SAME
155
- // env→reserve resolution a real provider does, without the floor. An invalid
156
- // value still throws (a malformed reserve is an error, not an absence).
157
- export const resolveEnvelopeFromEnv = (env: NodeJS.ProcessEnv, window: number | null): { reasoningReserve: number | null; completionReserve: number | null } => {
158
- const one = (raw: string | undefined, name: string): number | null =>
159
- raw === undefined || raw.length === 0 ? null : resolveReserve(parseReserve(raw, name, "mock"), window);
160
- return {
161
- reasoningReserve: one(env.PLURNK_PROVIDERS_REASONING_RESERVE, "PLURNK_PROVIDERS_REASONING_RESERVE"),
162
- completionReserve: one(env.PLURNK_PROVIDERS_COMPLETION_RESERVE, "PLURNK_PROVIDERS_COMPLETION_RESERVE"),
163
- };
144
+ export type GenerationEnvelope = {
145
+ readonly outputBudget: number | null;
146
+ readonly reasoningBudget: number | null;
147
+ };
148
+
149
+ const shedRetiredEnvelope = (env: NodeJS.ProcessEnv, label: string): void => {
150
+ for (const name of ["PLURNK_PROVIDERS_REASONING_RESERVE", "PLURNK_PROVIDERS_COMPLETION_RESERVE"] as const) {
151
+ if (env[name] !== undefined && env[name] !== "") {
152
+ throw new Error(
153
+ `${label} provider: ${name} is retired; reasoning is now a subset of the total generation envelope. Replace the old pair with PLURNK_PROVIDERS_OUTPUT_BUDGET and optional PLURNK_PROVIDERS_REASONING_BUDGET ({§provider-generation-envelope})`,
154
+ );
155
+ }
156
+ }
157
+ };
158
+
159
+ const optionalTokenBudget = (
160
+ raw: string | undefined,
161
+ name: string,
162
+ label: string,
163
+ ): TokenBudgetSpec | null => raw === undefined || raw.length === 0
164
+ ? null
165
+ : parseTokenBudget(raw, name, label);
166
+
167
+ const resolveGeneration = (
168
+ outputSpec: TokenBudgetSpec | null,
169
+ reasoningSpec: TokenBudgetSpec | null,
170
+ contextWindow: number | null,
171
+ maxOutputTokens: number | null,
172
+ label: string,
173
+ ): GenerationEnvelope => {
174
+ const requestedOutput = outputSpec === null ? null : resolveTokenBudget(outputSpec, contextWindow);
175
+ const physicalCaps = [contextWindow, maxOutputTokens].filter((value): value is number => value !== null);
176
+ const outputBudget = requestedOutput === null
177
+ ? null
178
+ : Math.min(requestedOutput, ...physicalCaps);
179
+ if (contextWindow !== null && outputBudget !== null && outputBudget >= contextWindow) {
180
+ throw new Error(
181
+ `${label} provider: PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}) must leave positive input capacity inside the context window (${contextWindow})`,
182
+ );
183
+ }
184
+ const reasoningBudget = reasoningSpec === null
185
+ ? null
186
+ : resolveTokenBudget(reasoningSpec, contextWindow);
187
+ if (reasoningBudget !== null && outputBudget === null) {
188
+ throw new Error(
189
+ `${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET; reasoning is a subset of total output`,
190
+ );
191
+ }
192
+ if (reasoningBudget !== null && outputBudget !== null && reasoningBudget >= outputBudget) {
193
+ throw new Error(
194
+ `${label} provider: PLURNK_PROVIDERS_REASONING_BUDGET (${reasoningBudget}) exceeds the effective PLURNK_PROVIDERS_OUTPUT_BUDGET (${outputBudget}); reasoning is a subset of total output`,
195
+ );
196
+ }
197
+ return { outputBudget, reasoningBudget };
198
+ };
199
+
200
+ // Standard providers receive the shipped OUTPUT_BUDGET floor and fail hard if
201
+ // it is absent. Mock uses the tolerant sibling below so ordinary unit fixtures
202
+ // make no generation claim unless a test deliberately configures one.
203
+ export const generationEnvelopeFromEnv = (
204
+ env: NodeJS.ProcessEnv,
205
+ label: string,
206
+ contextWindow: number | null,
207
+ maxOutputTokens: number | null,
208
+ ): GenerationEnvelope => {
209
+ shedRetiredEnvelope(env, label);
210
+ return resolveGeneration(
211
+ parseTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", label),
212
+ optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", label),
213
+ contextWindow,
214
+ maxOutputTokens,
215
+ label,
216
+ );
217
+ };
218
+
219
+ export const resolveGenerationEnvelopeFromEnv = (
220
+ env: NodeJS.ProcessEnv,
221
+ contextWindow: number | null,
222
+ maxOutputTokens: number | null = null,
223
+ ): GenerationEnvelope => {
224
+ shedRetiredEnvelope(env, "mock");
225
+ return resolveGeneration(
226
+ optionalTokenBudget(env.PLURNK_PROVIDERS_OUTPUT_BUDGET, "PLURNK_PROVIDERS_OUTPUT_BUDGET", "mock"),
227
+ optionalTokenBudget(env.PLURNK_PROVIDERS_REASONING_BUDGET, "PLURNK_PROVIDERS_REASONING_BUDGET", "mock"),
228
+ contextWindow,
229
+ maxOutputTokens,
230
+ "mock",
231
+ );
164
232
  };
165
233
 
166
234
  // {§provider-configuration} The side-channel reasoning knobs — activation and budget
167
235
  // are separate vars, so a numeric budget can never silently flip wire flags:
168
236
  // PLURNK_PROVIDERS_REASONING off | adaptive | on (REQUIRED, fail-hard)
169
- // PLURNK_PROVIDERS_REASONING_BUDGET optional positive int when REASONING=on —
170
- // an explicit magnitude for tier/budget mapping. On llama-server it is the
171
- // request-scoped allowance and cannot exceed the physical reasoning reserve.
237
+ // PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
238
+ // output budget, used for tier/budget mapping where the backend supports it.
172
239
  // The provider maps intent to the backend's mechanism; the consumer states
173
240
  // intent, never mechanism. PLAN is a separate public intended-goals record.
174
241
  export type ReasoningMode = "off" | "adaptive" | "on";
@@ -189,20 +256,18 @@ export const reasoningResponseStyleFromEnv = (
189
256
  return raw;
190
257
  };
191
258
 
192
- export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoning => {
259
+ export const reasoningFromEnv = (
260
+ env: NodeJS.ProcessEnv,
261
+ label: string,
262
+ resolvedBudget: number | null = null,
263
+ ): Reasoning => {
193
264
  shedRenamed(env, "PLURNK_PROVIDERS_THINKING", "PLURNK_PROVIDERS_REASONING", label, "provider configuration contract"); // lexicon-allow
194
265
  shedRenamed(env, "PLURNK_PROVIDERS_THINKING_CAPACITY", "PLURNK_PROVIDERS_REASONING_BUDGET", label, "provider configuration contract"); // lexicon-allow
195
266
  const name = "PLURNK_PROVIDERS_REASONING";
196
267
  const raw = env[name];
197
268
  if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (off | adaptive | on)`);
198
269
  if (raw !== "off" && raw !== "adaptive" && raw !== "on") throw new Error(`${label} provider: ${name} must be one of "off", "adaptive", "on" (got "${raw}")`);
199
- if (raw !== "on") return { mode: raw, budget: null };
200
- const capName = "PLURNK_PROVIDERS_REASONING_BUDGET";
201
- const capRaw = env[capName];
202
- if (capRaw === undefined || capRaw.length === 0) return { mode: "on", budget: null };
203
- const n = Number(capRaw);
204
- if (!Number.isInteger(n) || n <= 0) throw new Error(`${label} provider: ${capName} must be a positive integer (got "${capRaw}")`);
205
- return { mode: "on", budget: n };
270
+ return { mode: raw, budget: raw === "off" ? null : resolvedBudget };
206
271
  };
207
272
 
208
273
  // ── Per-alias knob scoping (per-alias scoping doctrine, user 2026-07-03): PLURNK_PROVIDERS_<KNOB>[_<alias>] ──
@@ -213,8 +278,7 @@ export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoni
213
278
  // facts (API keys, canonical endpoints) remain vendor-named; the per-alias
214
279
  // endpoint override stays PLURNK_BASEURL_<alias> (its existing precedent).
215
280
  export const PROVIDERS_KNOBS = Object.freeze([
216
- "PLURNK_PROVIDERS_REASONING_RESERVE",
217
- "PLURNK_PROVIDERS_COMPLETION_RESERVE",
281
+ "PLURNK_PROVIDERS_OUTPUT_BUDGET",
218
282
  "PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
219
283
  "PLURNK_PROVIDERS_REASONING_BUDGET",
220
284
  "PLURNK_PROVIDERS_REASONING",
@@ -250,10 +314,9 @@ export const PROVIDERS_KNOBS = Object.freeze([
250
314
  // single overlay.
251
315
  //
252
316
  // `knobs` (optional) lets a CONSUMER scope its OWN closed knob list with this
253
- // same parser — e.g. the service's window-partition vars (PLURNK_SERVICE_CONTEXT_WINDOW/
254
- // MAX_TURNS/...), so a 64k cloud envelope and a 12k gemma envelope
255
- // coexist per-alias without the service reimplementing the suffix/collision
256
- // rules. Default stays the providers-family list; my call sites pass nothing.
317
+ // same parser — e.g. service loop policy or prompt projection — without
318
+ // reimplementing the suffix/collision rules. Default stays the
319
+ // providers-family list; provider call sites pass nothing.
257
320
  export const scopeEnvToAlias = (env: NodeJS.ProcessEnv, alias: string, knobs: readonly string[] = PROVIDERS_KNOBS): NodeJS.ProcessEnv => {
258
321
  const folded = alias.toLowerCase();
259
322
  const out: NodeJS.ProcessEnv = { ...env };
@@ -33,10 +33,29 @@ test("classifyProviderError maps HTTP status to kind", () => {
33
33
  assert.equal(k(409), "network_failure");
34
34
  assert.equal(k(500), "network_failure");
35
35
  assert.equal(k(503), "network_failure");
36
+ assert.equal(k(413), "capacity_exceeded");
36
37
  assert.equal(k(400), "invalid_response");
37
38
  assert.equal(k(404), "invalid_response");
38
39
  });
39
40
 
41
+ test("capacity normalization prefers structured provider codes and keeps generic 400s distinct", () => {
42
+ const openai = apiError(400, JSON.stringify({
43
+ error: {
44
+ type: "invalid_request_error",
45
+ code: "context_length_exceeded",
46
+ message: "maximum context length exceeded",
47
+ },
48
+ }));
49
+ assert.equal(classifyProviderError(openai).kind, "capacity_exceeded");
50
+ const normalized = toProviderError(openai, "provider:openai");
51
+ assert.equal(normalized.status, 413);
52
+ assert.equal(normalized.problem.capacityStage, "upstream");
53
+ assert.equal(normalized.problem.providerStatus, 400);
54
+ assert.equal(classifyProviderError(apiError(400, JSON.stringify({
55
+ error: { type: "invalid_request_error", code: "bad_temperature", message: "bad temperature" },
56
+ }))).kind, "invalid_response");
57
+ });
58
+
40
59
  test("provider retry directives survive HTTP failure normalization", () => {
41
60
  const final = new APICallError({
42
61
  message: "edge router says not to replay",
@@ -101,6 +120,20 @@ test("#161: ProviderError carries resource-interrupted attempt evidence outside
101
120
  usage: { inputTokens: 3, outputTokens: 1, totalTokens: 4 },
102
121
  cost: { kind: "unknown", reason: "fixture has no monetary evidence" },
103
122
  }],
123
+ capacity: {
124
+ decision: "defer",
125
+ contextWindow: null,
126
+ maxInputTokens: null,
127
+ maxOutputTokens: null,
128
+ outputBudget: null,
129
+ reasoningBudget: null,
130
+ inputCapacity: null,
131
+ prompt: {
132
+ kind: "unavailable",
133
+ source: "fixture",
134
+ detail: "the interrupted fixture has no preflight measurement",
135
+ },
136
+ },
104
137
  } as ProviderAttempt;
105
138
  const error = new ProviderError(
106
139
  "provider:deepseek",
package/src/errors.ts CHANGED
@@ -1,18 +1,10 @@
1
- import { Problems, type ProblemDetails } from "@plurnk/plurnk-contracts";
2
1
  import { APICallError, RetryError } from "ai";
3
- import { providerSource } from "./notices.ts";
4
- import type { ProviderAttempt, ProviderRequestAccounting } from "./types.ts";
2
+ import { ProviderError } from "./providerError.ts";
3
+ import type { ProviderErrorKind } from "./providerError.ts";
4
+ import type { ProviderRequestCapacity } from "./types.ts";
5
5
 
6
- export type ProviderErrorKind =
7
- | "rate_limit"
8
- | "network_failure"
9
- | "deadline_exceeded"
10
- | "model_refused"
11
- | "invalid_response"
12
- | "unauthorized"
13
- | "quota_exceeded"
14
- | "grammar_invalid"
15
- | "resource_interrupted";
6
+ export { ProviderError } from "./providerError.ts";
7
+ export type { ProviderErrorKind } from "./providerError.ts";
16
8
 
17
9
  export interface ClassifiedProviderError {
18
10
  kind: ProviderErrorKind;
@@ -54,131 +46,23 @@ export const providerTimeoutOf = (error: unknown): ProviderTimeoutError | null =
54
46
  return null;
55
47
  };
56
48
 
57
- const defaultStatus = (kind: ProviderErrorKind): number => {
58
- switch (kind) {
59
- case "unauthorized": return 401;
60
- case "quota_exceeded": return 402;
61
- case "rate_limit": return 429;
62
- case "model_refused":
63
- case "grammar_invalid": return 422;
64
- case "invalid_response": return 502;
65
- case "deadline_exceeded": return 504;
66
- case "network_failure":
67
- case "resource_interrupted": return 503;
68
- }
69
- };
70
-
71
- const retryable = (kind: ProviderErrorKind): boolean => {
72
- switch (kind) {
73
- case "rate_limit":
74
- case "network_failure":
75
- return true;
76
- case "deadline_exceeded":
77
- case "invalid_response":
78
- case "grammar_invalid":
79
- case "resource_interrupted":
80
- case "model_refused":
81
- case "unauthorized":
82
- case "quota_exceeded":
83
- return false;
84
- }
85
- };
86
-
87
- const buildProblem = (
88
- source: string,
89
- kind: ProviderErrorKind,
90
- message: string,
91
- status: number,
92
- extensions: Readonly<Record<string, unknown>>,
93
- retryableOverride: boolean | undefined,
94
- ): ProblemDetails => {
95
- const code: Record<ProviderErrorKind, string> = {
96
- rate_limit: "rate-limit",
97
- network_failure: "network-failure",
98
- deadline_exceeded: "deadline-exceeded",
99
- model_refused: "model-refused",
100
- invalid_response: "invalid-response",
101
- unauthorized: "unauthorized",
102
- quota_exceeded: "quota-exceeded",
103
- grammar_invalid: "grammar-invalid",
104
- resource_interrupted: "resource-interrupted",
105
- };
106
- return Problems.create(source, code[kind], status, message, {
107
- providerKind: kind,
108
- stage: "provider-request",
109
- retryable: retryableOverride ?? retryable(kind),
110
- ...extensions,
111
- });
112
- };
113
-
114
- // A provider operation failed before a completed exchange existed. An
115
- // interrupted response may still carry attempt evidence for its consumer.
116
- // The standardized Problem is the public failure contract; kind remains the
117
- // provider pool's routing discriminator and is repeated as a Problem extension.
118
- export class ProviderError extends Error {
119
- readonly source: string;
120
- readonly kind: ProviderErrorKind;
121
- readonly problem: ProblemDetails;
122
- readonly attempt?: ProviderAttempt;
123
- #accounting: ProviderRequestAccounting[];
124
-
125
- constructor(
126
- source: string,
127
- kind: ProviderErrorKind,
128
- message: string,
129
- options: {
130
- status?: number | null;
131
- cause?: unknown;
132
- retryable?: boolean;
133
- extensions?: Readonly<Record<string, unknown>>;
134
- attempt?: ProviderAttempt;
135
- accounting?: readonly ProviderRequestAccounting[];
136
- } = {},
137
- ) {
138
- super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
139
- this.name = "ProviderError";
140
- this.source = providerSource(source);
141
- this.kind = kind;
142
- this.attempt = options.attempt;
143
- this.#accounting = [...(options.accounting ?? options.attempt?.accounting ?? [])];
144
- const status = options.status !== null && options.status !== undefined
145
- && Number.isInteger(options.status) && options.status >= 400 && options.status <= 599
146
- ? options.status
147
- : defaultStatus(kind);
148
- this.problem = buildProblem(
149
- this.source,
150
- kind,
151
- message,
152
- status,
153
- options.extensions ?? {},
154
- options.retryable,
155
- );
156
- }
157
-
158
- get status(): number {
159
- return this.problem.status;
160
- }
161
-
162
- get accounting(): readonly ProviderRequestAccounting[] {
163
- return this.#accounting;
164
- }
165
-
166
- // A capacity pool adds the already-settled requests from prior backends as
167
- // the same failure crosses that orchestration boundary.
168
- prependAccounting(accounting: readonly ProviderRequestAccounting[]): void {
169
- if (accounting.length > 0) this.#accounting = [...accounting, ...this.#accounting];
170
- }
171
- }
172
-
173
- const wireErrorType = (body: string): string | null => {
49
+ const wireError = (body: string): { type: string | null; code: string | null; message: string | null } => {
174
50
  try {
175
51
  const { error } = JSON.parse(body) as { error?: { type?: unknown } };
176
- return typeof error?.type === "string" ? error.type : null;
52
+ const record = error as { type?: unknown; code?: unknown; message?: unknown } | undefined;
53
+ return {
54
+ type: typeof record?.type === "string" ? record.type : null,
55
+ code: typeof record?.code === "string" || typeof record?.code === "number" ? String(record.code) : null,
56
+ message: typeof record?.message === "string" ? record.message : null,
57
+ };
177
58
  } catch {
178
- return null;
59
+ return { type: null, code: null, message: null };
179
60
  }
180
61
  };
181
62
 
63
+ const CAPACITY_CODE = /^(?:context_length_exceeded|context_window_exceeded|input_too_long|prompt_too_long|request_too_large|token_limit_exceeded|max_tokens_exceeded)$/i;
64
+ const CAPACITY_MESSAGE = /(?:maximum context length|context (?:length|window).*(?:exceed|too (?:large|long)|maximum)|(?:input|prompt|request).*(?:token|length|size).*(?:exceed|too (?:large|long)|maximum))/i;
65
+
182
66
  const preview = (value: unknown, limit: number | undefined): string => {
183
67
  const text = value instanceof Error ? value.message : String(value);
184
68
  return limit !== undefined && text.length > limit
@@ -215,6 +99,7 @@ export const classifyProviderError = (
215
99
  ? preview(err.message, detailLimit)
216
100
  : "The provider request failed without a diagnostic message.";
217
101
  const body = err.responseBody ?? "";
102
+ const wire = wireError(body);
218
103
  if (status === 401 || status === 403) return { kind: "unauthorized", message };
219
104
  if (status === 402) return { kind: "quota_exceeded", message };
220
105
  if (status === 429) return { kind: "rate_limit", message, retryable: err.isRetryable };
@@ -223,7 +108,21 @@ export const classifyProviderError = (
223
108
  }
224
109
  if (status === 0 && err.isRetryable) return { kind: "network_failure", message };
225
110
  if (status >= 500) return { kind: "network_failure", message, retryable: err.isRetryable };
226
- if (status === 422 && wireErrorType(body) === "grammar_invalid") {
111
+ if (status === 413 || (
112
+ (status === 400 || status === 422)
113
+ && (
114
+ (wire.code !== null && CAPACITY_CODE.test(wire.code))
115
+ || (wire.type !== null && CAPACITY_CODE.test(wire.type))
116
+ || CAPACITY_MESSAGE.test(wire.message ?? message)
117
+ )
118
+ )) {
119
+ return {
120
+ kind: "capacity_exceeded",
121
+ message,
122
+ extensions: status === 413 ? undefined : { providerStatus: status },
123
+ };
124
+ }
125
+ if (status === 422 && wire.type === "grammar_invalid") {
227
126
  return { kind: "grammar_invalid", message };
228
127
  }
229
128
  return { kind: "invalid_response", message };
@@ -251,22 +150,27 @@ export const toProviderError = (
251
150
  err: unknown,
252
151
  source: string,
253
152
  detailLimit?: number,
153
+ capacity?: ProviderRequestCapacity,
254
154
  ): ProviderError => {
255
155
  if (err instanceof ProviderError) return err;
256
156
  const underlying = RetryError.isInstance(err) ? err.lastError : err;
257
157
  const classified = classifyProviderError(err, detailLimit);
258
158
  const { kind, message } = classified;
259
- const status = APICallError.isInstance(underlying) ? underlying.statusCode ?? null : null;
159
+ const upstreamStatus = APICallError.isInstance(underlying) ? underlying.statusCode ?? null : null;
160
+ const status = kind === "capacity_exceeded" ? 413 : upstreamStatus;
260
161
  return new ProviderError(source, kind, message, {
261
162
  status,
262
163
  cause: err,
263
164
  retryable: classified.retryable,
264
165
  extensions: {
265
166
  ...(classified.extensions ?? {}),
167
+ ...(kind === "capacity_exceeded" ? { capacityStage: "upstream" } : {}),
168
+ ...(capacity === undefined ? {} : { capacity }),
266
169
  ...(classified.attempts === undefined ? {} : { attempts: classified.attempts }),
267
170
  ...(classified.retryExhausted === undefined
268
171
  ? {}
269
172
  : { retryExhausted: classified.retryExhausted }),
270
173
  },
174
+ capacity,
271
175
  });
272
176
  };
package/src/index.ts CHANGED
@@ -16,6 +16,8 @@ export type {
16
16
  ProviderCallKind,
17
17
  ProviderGenerateArgs,
18
18
  ProviderRequestAccounting,
19
+ ProviderRequestCapacity,
20
+ ProviderRequestCapacityDecision,
19
21
  ProviderRequestIdentity,
20
22
  ProviderRequestObserver,
21
23
  ProviderRequestSettlement,
@@ -25,6 +27,7 @@ export type {
25
27
  TokenAlternative,
26
28
  } from "./types.ts";
27
29
  export { assertPromptTokenMeasurement } from "./promptTokens.ts";
30
+ export { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, requestCapacityDecision } from "./capacity.ts";
28
31
 
29
32
  // Alias cascade — re-exported from the zero-dep @plurnk/plurnk-aliases, so
30
33
  // the "." surface is unchanged for existing importers and there's one source of
@@ -50,8 +53,8 @@ export type { AiSdkProviderConfig, ReasoningStyle, GrammarStyle } from "./AiSdkP
50
53
  // DECISION stays the consumer's, by choosing which pool to call.
51
54
  export { default as Pool } from "./Pool.ts";
52
55
  export type { ProviderFetch } from "./AiSdkProvider.ts";
53
- export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow, envelopeFromEnv, resolveReserve, PROVIDERS_KNOBS } from "./env.ts";
54
- export type { Reasoning, ReasoningMode, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
56
+ export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, effectiveContextWindow, generationEnvelopeFromEnv, resolveGenerationEnvelopeFromEnv, resolveTokenBudget, PROVIDERS_KNOBS } from "./env.ts";
57
+ export type { GenerationEnvelope, Reasoning, ReasoningMode, ReasoningResponseStyle, TokenBudgetSpec } from "./env.ts";
55
58
  export { normalizeUsage, calculateCostUsdDecimal, validateProviderUsage } from "./usage.ts";
56
59
  export {
57
60
  addDecimals,
@@ -11,8 +11,7 @@ const env = Object.freeze({
11
11
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
12
12
  PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
13
13
  PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
14
- PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
15
- PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
14
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
16
15
  PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
17
16
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
18
17
  PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
@@ -4,6 +4,7 @@ const KINDS = new Set<PromptTokenMeasurement["kind"]>([
4
4
  "exact",
5
5
  "upper_bound",
6
6
  "estimate",
7
+ "unavailable",
7
8
  ]);
8
9
 
9
10
  export const assertPromptTokenMeasurement = (
@@ -13,19 +14,21 @@ export const assertPromptTokenMeasurement = (
13
14
  if (typeof value !== "object" || value === null) {
14
15
  throw new TypeError(`${owner}: prompt token measurement must be an object`);
15
16
  }
16
- const candidate = value as Partial<PromptTokenMeasurement>;
17
- if (!KINDS.has(candidate.kind as PromptTokenMeasurement["kind"])) {
17
+ const candidate = value as Record<string, unknown>;
18
+ const kind = candidate.kind as PromptTokenMeasurement["kind"];
19
+ if (!KINDS.has(kind)) {
18
20
  throw new TypeError(`${owner}: prompt token measurement has invalid kind ${JSON.stringify(candidate.kind)}`);
19
21
  }
20
- if (!Number.isInteger(candidate.tokens) || candidate.tokens! < 0) {
22
+ if (kind !== "unavailable"
23
+ && (!Number.isInteger(candidate.tokens) || (candidate.tokens as number) < 0)) {
21
24
  throw new TypeError(`${owner}: prompt token measurement tokens must be a non-negative integer`);
22
25
  }
23
26
  if (typeof candidate.source !== "string" || candidate.source.length === 0) {
24
27
  throw new TypeError(`${owner}: prompt token measurement source must be a non-empty string`);
25
28
  }
26
- if (candidate.kind === "estimate"
29
+ if ((kind === "estimate" || kind === "unavailable")
27
30
  && (typeof candidate.detail !== "string" || candidate.detail.length === 0)) {
28
- throw new TypeError(`${owner}: estimated prompt token measurement requires detail`);
31
+ throw new TypeError(`${owner}: ${kind} prompt token measurement requires detail`);
29
32
  }
30
33
  return value as PromptTokenMeasurement;
31
34
  };