@plurnk/plurnk-providers 1.6.0 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/.env.defaults +9 -16
  2. package/README.md +17 -0
  3. package/SPEC.md +128 -45
  4. package/dist/AiSdkProvider.d.ts +16 -9
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +151 -54
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +8 -4
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +53 -18
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -4
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +68 -12
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js.map +1 -1
  18. package/dist/accountingPublic.d.ts +5 -0
  19. package/dist/accountingPublic.d.ts.map +1 -0
  20. package/dist/accountingPublic.js +3 -0
  21. package/dist/accountingPublic.js.map +1 -0
  22. package/dist/capacity.d.ts +26 -0
  23. package/dist/capacity.d.ts.map +1 -0
  24. package/dist/capacity.js +90 -0
  25. package/dist/capacity.js.map +1 -0
  26. package/dist/catalogProvider.d.ts +2 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +18 -20
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +10 -7
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/env.d.ts +8 -10
  34. package/dist/env.d.ts.map +1 -1
  35. package/dist/env.js +54 -37
  36. package/dist/env.js.map +1 -1
  37. package/dist/errors.d.ts +6 -22
  38. package/dist/errors.d.ts.map +1 -1
  39. package/dist/errors.js +30 -91
  40. package/dist/errors.js.map +1 -1
  41. package/dist/index.d.ts +4 -3
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +2 -1
  44. package/dist/index.js.map +1 -1
  45. package/dist/promptTokens.d.ts.map +1 -1
  46. package/dist/promptTokens.js +7 -4
  47. package/dist/promptTokens.js.map +1 -1
  48. package/dist/providerError.d.ts +25 -0
  49. package/dist/providerError.d.ts.map +1 -0
  50. package/dist/providerError.js +91 -0
  51. package/dist/providerError.js.map +1 -0
  52. package/dist/sdkModels.d.ts +1 -0
  53. package/dist/sdkModels.d.ts.map +1 -1
  54. package/dist/sdkModels.js +5 -8
  55. package/dist/sdkModels.js.map +1 -1
  56. package/dist/types.d.ts +24 -4
  57. package/dist/types.d.ts.map +1 -1
  58. package/dist/usage.d.ts +1 -0
  59. package/dist/usage.d.ts.map +1 -1
  60. package/dist/usage.js +7 -2
  61. package/dist/usage.js.map +1 -1
  62. package/package.json +22 -7
  63. package/src/AiSdkProvider.test.ts +198 -37
  64. package/src/AiSdkProvider.ts +192 -59
  65. package/src/Mock.test.ts +32 -18
  66. package/src/Mock.ts +58 -19
  67. package/src/Pool.test.ts +71 -13
  68. package/src/Pool.ts +78 -13
  69. package/src/ProviderRegistry.test.ts +1 -1
  70. package/src/accounting.ts +0 -1
  71. package/src/accountingPublic.ts +9 -0
  72. package/src/boundaries.test.ts +28 -15
  73. package/src/capacity.test.ts +92 -0
  74. package/src/capacity.ts +140 -0
  75. package/src/catalogProvider.test.ts +82 -9
  76. package/src/catalogProvider.ts +24 -21
  77. package/src/compatibleProvider.test.ts +1 -2
  78. package/src/compatibleProvider.ts +10 -7
  79. package/src/cost.test.ts +31 -0
  80. package/src/env.test.ts +49 -20
  81. package/src/env.ts +114 -51
  82. package/src/errors.test.ts +33 -0
  83. package/src/errors.ts +38 -134
  84. package/src/index.ts +5 -2
  85. package/src/ollama.test.ts +1 -2
  86. package/src/promptTokens.ts +8 -5
  87. package/src/providerError.ts +139 -0
  88. package/src/sdkModels.test.ts +2 -5
  89. package/src/sdkModels.ts +6 -8
  90. package/src/types.ts +41 -19
  91. package/src/usage.ts +7 -2
@@ -0,0 +1,139 @@
1
+ import { Problems, type ProblemDetails } from "@plurnk/plurnk-contracts";
2
+ import { providerSource } from "./notices.ts";
3
+ import type { ProviderAttempt, ProviderRequestAccounting, ProviderRequestCapacity } from "./types.ts";
4
+
5
+ export type ProviderErrorKind =
6
+ | "rate_limit"
7
+ | "network_failure"
8
+ | "deadline_exceeded"
9
+ | "model_refused"
10
+ | "invalid_response"
11
+ | "unauthorized"
12
+ | "quota_exceeded"
13
+ | "grammar_invalid"
14
+ | "capacity_exceeded"
15
+ | "resource_interrupted";
16
+
17
+ const defaultStatus = (kind: ProviderErrorKind): number => {
18
+ switch (kind) {
19
+ case "unauthorized": return 401;
20
+ case "quota_exceeded": return 402;
21
+ case "capacity_exceeded": return 413;
22
+ case "rate_limit": return 429;
23
+ case "model_refused":
24
+ case "grammar_invalid": return 422;
25
+ case "invalid_response": return 502;
26
+ case "deadline_exceeded": return 504;
27
+ case "network_failure":
28
+ case "resource_interrupted": return 503;
29
+ }
30
+ };
31
+
32
+ const retryable = (kind: ProviderErrorKind): boolean => {
33
+ switch (kind) {
34
+ case "rate_limit":
35
+ case "network_failure":
36
+ return true;
37
+ case "deadline_exceeded":
38
+ case "invalid_response":
39
+ case "grammar_invalid":
40
+ case "capacity_exceeded":
41
+ case "resource_interrupted":
42
+ case "model_refused":
43
+ case "unauthorized":
44
+ case "quota_exceeded":
45
+ return false;
46
+ }
47
+ };
48
+
49
+ const buildProblem = (
50
+ source: string,
51
+ kind: ProviderErrorKind,
52
+ message: string,
53
+ status: number,
54
+ extensions: Readonly<Record<string, unknown>>,
55
+ retryableOverride: boolean | undefined,
56
+ ): ProblemDetails => {
57
+ const code: Record<ProviderErrorKind, string> = {
58
+ rate_limit: "rate-limit",
59
+ network_failure: "network-failure",
60
+ deadline_exceeded: "deadline-exceeded",
61
+ model_refused: "model-refused",
62
+ invalid_response: "invalid-response",
63
+ unauthorized: "unauthorized",
64
+ quota_exceeded: "quota-exceeded",
65
+ grammar_invalid: "grammar-invalid",
66
+ capacity_exceeded: "capacity-exceeded",
67
+ resource_interrupted: "resource-interrupted",
68
+ };
69
+ return Problems.create(source, code[kind], status, message, {
70
+ providerKind: kind,
71
+ stage: "provider-request",
72
+ retryable: retryableOverride ?? retryable(kind),
73
+ ...extensions,
74
+ });
75
+ };
76
+
77
+ // A provider operation failed before a completed exchange existed. An
78
+ // interrupted response may still carry attempt evidence for its consumer.
79
+ // The standardized Problem is the public failure contract; kind remains the
80
+ // provider pool's routing discriminator and is repeated as a Problem extension.
81
+ export class ProviderError extends Error {
82
+ readonly source: string;
83
+ readonly kind: ProviderErrorKind;
84
+ readonly problem: ProblemDetails;
85
+ readonly attempt?: ProviderAttempt;
86
+ readonly capacity?: ProviderRequestCapacity;
87
+ #accounting: ProviderRequestAccounting[];
88
+
89
+ constructor(
90
+ source: string,
91
+ kind: ProviderErrorKind,
92
+ message: string,
93
+ options: {
94
+ status?: number | null;
95
+ cause?: unknown;
96
+ retryable?: boolean;
97
+ extensions?: Readonly<Record<string, unknown>>;
98
+ attempt?: ProviderAttempt;
99
+ accounting?: readonly ProviderRequestAccounting[];
100
+ capacity?: ProviderRequestCapacity;
101
+ } = {},
102
+ ) {
103
+ super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
104
+ this.name = "ProviderError";
105
+ this.source = providerSource(source);
106
+ this.kind = kind;
107
+ this.attempt = options.attempt;
108
+ this.capacity = options.capacity ?? options.attempt?.capacity;
109
+ this.#accounting = [...(options.accounting ?? options.attempt?.accounting ?? [])];
110
+ const status = options.status !== null && options.status !== undefined
111
+ && Number.isInteger(options.status) && options.status >= 400 && options.status <= 599
112
+ ? options.status
113
+ : defaultStatus(kind);
114
+ this.problem = buildProblem(
115
+ this.source,
116
+ kind,
117
+ message,
118
+ status,
119
+ options.extensions ?? {},
120
+ options.retryable,
121
+ );
122
+ }
123
+
124
+ get status(): number {
125
+ return this.problem.status;
126
+ }
127
+
128
+ get accounting(): readonly ProviderRequestAccounting[] {
129
+ return this.#accounting;
130
+ }
131
+
132
+ // A capacity pool adds the already-settled requests from prior backends as
133
+ // the same failure crosses that orchestration boundary.
134
+ prependAccounting(accounting: readonly ProviderRequestAccounting[]): void {
135
+ if (accounting.length > 0) this.#accounting = [...accounting, ...this.#accounting];
136
+ }
137
+ }
138
+
139
+ export type { ProviderAttempt, ProviderRequestAccounting, ProviderRequestCapacity } from "./types.ts";
@@ -19,11 +19,8 @@ test("createSdkModel uses Models.dev provider facts and operator credentials", (
19
19
  const sdk = createSdkModel("xai", "grok-build-0.1", { XAI_API_KEY: "test-key" });
20
20
  assert.notEqual(sdk, null);
21
21
  assert.equal(sdk?.catalog?.npm, "@ai-sdk/xai");
22
- assert.equal(sdk?.languageModel, undefined);
23
- assert.deepEqual(sdk?.compatible, {
24
- url: "https://api.x.ai/v1/chat/completions",
25
- headers: { Authorization: "Bearer test-key" },
26
- });
22
+ assert.notEqual(sdk?.languageModel, undefined);
23
+ assert.equal(sdk?.compatible, undefined);
27
24
  assert.deepEqual(sdk?.cacheAffinity, { target: "header", name: "x-grok-conv-id" });
28
25
  assert.notEqual(sdk?.normalizeCost, undefined);
29
26
  });
package/src/sdkModels.ts CHANGED
@@ -7,6 +7,7 @@ import { createGroq } from "@ai-sdk/groq";
7
7
  import { createMistral } from "@ai-sdk/mistral";
8
8
  import { createOpenAI } from "@ai-sdk/openai";
9
9
  import { createTogetherAI } from "@ai-sdk/togetherai";
10
+ import { createXai } from "@ai-sdk/xai";
10
11
  import { createOpenRouter } from "@openrouter/ai-sdk-provider";
11
12
  import { lookupProvider, type ProviderInfo } from "@plurnk/plurnk-models";
12
13
  import type { LanguageModel } from "ai";
@@ -24,6 +25,7 @@ export type SdkModel = {
24
25
  readonly cacheAffinity?: CacheAffinity;
25
26
  readonly systemCacheProviderOptions?: AiSdkProviderOptions;
26
27
  readonly reasoningResponseProviderOptions?: AiSdkProviderOptions;
28
+ readonly additiveReasoningProvider?: "anthropic" | "bedrock";
27
29
  readonly catalog: ProviderInfo | null;
28
30
  };
29
31
 
@@ -162,24 +164,19 @@ export const createSdkModel = (
162
164
  },
163
165
  catalog,
164
166
  };
165
- case "@ai-sdk/xai": {
166
- const key = requireApiKey(provider, env, catalog);
167
- const compatibleBase = url ?? "https://api.x.ai/v1";
167
+ case "@ai-sdk/xai":
168
168
  return {
169
- compatible: {
170
- url: `${compatibleBase}/chat/completions`,
171
- headers: { Authorization: `Bearer ${key}` },
172
- },
169
+ languageModel: createXai({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).chat(model),
173
170
  ...(catalog.id === "xai"
174
171
  ? { cacheAffinity: { target: "header" as const, name: "x-grok-conv-id" } }
175
172
  : {}),
176
173
  ...(normalizeCost === undefined ? {} : { normalizeCost }),
177
174
  catalog,
178
175
  };
179
- }
180
176
  case "@ai-sdk/anthropic":
181
177
  return {
182
178
  languageModel: createAnthropic({ apiKey: requireApiKey(provider, env, catalog), baseURL: url }).languageModel(model),
179
+ additiveReasoningProvider: "anthropic",
183
180
  ...(catalog.id === "anthropic"
184
181
  ? { systemCacheProviderOptions: { anthropic: { cacheControl } } }
185
182
  : {}),
@@ -195,6 +192,7 @@ export const createSdkModel = (
195
192
  apiKey: env.AWS_BEARER_TOKEN_BEDROCK,
196
193
  baseURL: url,
197
194
  }).languageModel(model),
195
+ additiveReasoningProvider: "bedrock",
198
196
  catalog,
199
197
  };
200
198
  case "@openrouter/ai-sdk-provider":
package/src/types.ts CHANGED
@@ -10,10 +10,8 @@ import type {
10
10
  PluginAttributionSource,
11
11
  } from "@plurnk/plurnk-meta";
12
12
  import type {
13
- ProviderAccounting,
14
13
  ProviderCost,
15
14
  ProviderRequestAccounting,
16
- ProviderUsage,
17
15
  } from "@plurnk/plurnk-contracts";
18
16
 
19
17
  export type {
@@ -46,8 +44,29 @@ export type PromptTokenMeasurement =
46
44
  readonly tokens: number;
47
45
  readonly source: string;
48
46
  readonly detail: string;
47
+ }
48
+ | {
49
+ readonly kind: "unavailable";
50
+ readonly source: string;
51
+ readonly detail: string;
49
52
  };
50
53
 
54
+ export type ProviderRequestCapacityDecision = "admit" | "defer" | "reject";
55
+
56
+ // Complete pre-I/O evidence for one logical request. `defer` is intentional:
57
+ // an estimate or incomplete limit set cannot safely veto a request, so the
58
+ // upstream provider remains the capacity oracle.
59
+ export interface ProviderRequestCapacity {
60
+ readonly decision: ProviderRequestCapacityDecision;
61
+ readonly contextWindow: number | null;
62
+ readonly maxInputTokens: number | null;
63
+ readonly maxOutputTokens: number | null;
64
+ readonly outputBudget: number | null;
65
+ readonly reasoningBudget: number | null;
66
+ readonly inputCapacity: number | null;
67
+ readonly prompt: PromptTokenMeasurement;
68
+ }
69
+
51
70
  export type ChargedCost = Extract<ProviderCost, { kind: "charged" }>;
52
71
 
53
72
  // Evidence exposed by the transport to the provider adapter that owns its
@@ -149,6 +168,7 @@ export interface ProviderResponse<TFinish extends ProviderAttemptFinishReason =
149
168
  // Ordered physical request evidence, including automatic retries and pool
150
169
  // failover that preceded this response. {§provider-request-accounting}
151
170
  readonly accounting: readonly ProviderRequestAccounting[];
171
+ readonly capacity: ProviderRequestCapacity;
152
172
  // {§gbnf-response-observation} — evidence only; the consumer owns the verdict.
153
173
  readonly grammarEvidence?: GrammarEvidence;
154
174
  // Per-turn provider→client metadata bag: the backend's non-standard top-level
@@ -179,7 +199,7 @@ export interface ProviderGenerateArgs {
179
199
  readonly primaryWorkerId?: string;
180
200
  readonly signal?: AbortSignal;
181
201
  readonly grammar?: string;
182
- readonly maxTokens?: number;
202
+ readonly maxOutputTokens?: number;
183
203
  readonly attributions?: string[];
184
204
  readonly client?: string;
185
205
  readonly strikes?: number;
@@ -202,11 +222,9 @@ export interface Provider {
202
222
  // to constrain and which root variant to send is consumer policy
203
223
  // ({§gbnf-response-observation}).
204
224
  //
205
- // `maxTokens` is the consumer's per-call output ceiling (wire `max_tokens`).
206
- // Without it, most servers generate UNBOUNDED (llama-server n_predict -1) —
207
- // under a multi-op grammar that degenerates to the context wall,
208
- // so a constrained consumer is expected to pass it. Policy stays the
209
- // consumer's; the provider only transports.
225
+ // `maxOutputTokens` may tighten the provider's configured total output
226
+ // budget for this call. It includes visible output and hidden reasoning;
227
+ // the adapter owns projection into each backend's native wire semantics.
210
228
  //
211
229
  // `workerId` is the REQUIRED, opaque, stable identity of the consumer's work
212
230
  // stream (loop/run). Providers MAY key backend affinity on it — e.g.
@@ -255,6 +273,11 @@ export interface Provider {
255
273
  // including any stricter operator cap. `null` means unknown; under
256
274
  // llama-server parallelism the probed natural value is per slot.
257
275
  readonly contextWindow: number | null;
276
+ readonly maxInputTokens: number | null;
277
+ readonly maxOutputTokens: number | null;
278
+ readonly outputBudget: number | null;
279
+ readonly reasoningBudget: number | null;
280
+ readonly inputCapacity: number | null;
258
281
  readonly model: string;
259
282
  // Optional: the backend's self-reported served model id, from a
260
283
  // /v1/models-shaped probe (llama-server today; any such backend). For a local
@@ -274,17 +297,16 @@ export interface Provider {
274
297
  // clamp an over-ask (fireworks/xai, verified live) never set this; undefined
275
298
  // = no claim. Introspectable so a consumer can refuse AT BOOT a local alias
276
299
  // with no declared envelope, instead of dying mid-turn in partition math.
277
- readonly requiresMaxTokens?: boolean;
278
- // Optional generation-envelope reserves ({§provider-generation-envelope}) — the amounts of
279
- // the effective window reserved for reasoning and completion: floor
280
- // percentages of `contextWindow`, or absolute per-alias pins that win
281
- // outright. The consumer's prompt budget is `contextWindow - reasoningReserve
282
- // - completionReserve - <its own packing-safety margin>`; the generation cap
283
- // is the two pooled. `null` = underivable (window unknown, no absolute pin) →
284
- // the consumer's no-cap path. Absent = a bare sibling makes NO claim (treated
285
- // as null). All first-party providers claim, so null means genuinely-unknown.
286
- readonly reasoningReserve?: number | null;
287
- readonly completionReserve?: number | null;
300
+ readonly requiresOutputBudget?: boolean;
301
+ // The adapter owns request-specific physical admission. A proven fit may
302
+ // admit and an exact overflow may reject before I/O. Incomplete limits,
303
+ // estimates, bounds that do not prove fit, and unavailable measurements
304
+ // defer to the upstream provider as the capacity oracle.
305
+ assessRequestCapacity(
306
+ messages: readonly ChatMessage[],
307
+ maxOutputTokens?: number,
308
+ signal?: AbortSignal,
309
+ ): Promise<ProviderRequestCapacity>;
288
310
  // Provider-owned preflight measurement of the complete chat request,
289
311
  // including provider/template framing when the adapter can know it.
290
312
  // Estimates are explicit and MUST NOT authorize hard context-envelope admission.
package/src/usage.ts CHANGED
@@ -199,6 +199,7 @@ export const normalizeUsage = (raw: RawUsage | null | undefined): ProviderUsage
199
199
  export type TokenRates = {
200
200
  input: number;
201
201
  output: number;
202
+ reasoning?: number;
202
203
  cacheRead?: number;
203
204
  cacheWrite?: number;
204
205
  };
@@ -235,17 +236,21 @@ export const calculateCostUsdDecimal = (
235
236
 
236
237
  const cacheReadRate = rates.cacheRead ?? rates.input;
237
238
  const cacheWriteRate = rates.cacheWrite ?? rates.input;
239
+ const reasoningRate = rates.reasoning ?? rates.output;
238
240
  const cacheReadTokens = usage.inputTokenDetails?.cacheReadTokens;
239
241
  const cacheWriteTokens = usage.inputTokenDetails?.cacheWriteTokens;
242
+ const reasoningTokens = usage.outputTokenDetails?.reasoningTokens;
240
243
  if (cacheReadRate !== rates.input && cacheReadTokens === undefined) return null;
241
244
  if (cacheWriteRate !== rates.input && cacheWriteTokens === undefined) return null;
245
+ if (reasoningRate !== rates.output && reasoningTokens === undefined) return null;
242
246
 
243
- const parts = [rates.input, rates.output, cacheReadRate, cacheWriteRate].map(decimalParts);
247
+ const parts = [rates.input, rates.output, reasoningRate, cacheReadRate, cacheWriteRate].map(decimalParts);
244
248
  const rateScale = Math.max(...parts.map(({ scale }) => scale));
245
- const [input, output, cacheRead, cacheWrite] = parts.map(({ coefficient, scale }) =>
249
+ const [input, output, reasoning, cacheRead, cacheWrite] = parts.map(({ coefficient, scale }) =>
246
250
  coefficient * 10n ** BigInt(rateScale - scale));
247
251
  const coefficient = BigInt(usage.inputTokens) * input!
248
252
  + BigInt(usage.outputTokens) * output!
253
+ + BigInt(reasoningTokens ?? 0) * (reasoning! - output!)
249
254
  + BigInt(cacheReadTokens ?? 0) * (cacheRead! - input!)
250
255
  + BigInt(cacheWriteTokens ?? 0) * (cacheWrite! - input!);
251
256
  if (coefficient < 0n) throw new TypeError("provider token-rate details produced a negative estimate");