@plurnk/plurnk-providers 1.6.0 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/.env.defaults +9 -16
  2. package/README.md +17 -0
  3. package/SPEC.md +128 -45
  4. package/dist/AiSdkProvider.d.ts +16 -9
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +151 -54
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +8 -4
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +53 -18
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -4
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +68 -12
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js.map +1 -1
  18. package/dist/accountingPublic.d.ts +5 -0
  19. package/dist/accountingPublic.d.ts.map +1 -0
  20. package/dist/accountingPublic.js +3 -0
  21. package/dist/accountingPublic.js.map +1 -0
  22. package/dist/capacity.d.ts +26 -0
  23. package/dist/capacity.d.ts.map +1 -0
  24. package/dist/capacity.js +90 -0
  25. package/dist/capacity.js.map +1 -0
  26. package/dist/catalogProvider.d.ts +2 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +18 -20
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +10 -7
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/env.d.ts +8 -10
  34. package/dist/env.d.ts.map +1 -1
  35. package/dist/env.js +54 -37
  36. package/dist/env.js.map +1 -1
  37. package/dist/errors.d.ts +6 -22
  38. package/dist/errors.d.ts.map +1 -1
  39. package/dist/errors.js +30 -91
  40. package/dist/errors.js.map +1 -1
  41. package/dist/index.d.ts +4 -3
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +2 -1
  44. package/dist/index.js.map +1 -1
  45. package/dist/promptTokens.d.ts.map +1 -1
  46. package/dist/promptTokens.js +7 -4
  47. package/dist/promptTokens.js.map +1 -1
  48. package/dist/providerError.d.ts +25 -0
  49. package/dist/providerError.d.ts.map +1 -0
  50. package/dist/providerError.js +91 -0
  51. package/dist/providerError.js.map +1 -0
  52. package/dist/sdkModels.d.ts +1 -0
  53. package/dist/sdkModels.d.ts.map +1 -1
  54. package/dist/sdkModels.js +5 -8
  55. package/dist/sdkModels.js.map +1 -1
  56. package/dist/types.d.ts +24 -4
  57. package/dist/types.d.ts.map +1 -1
  58. package/dist/usage.d.ts +1 -0
  59. package/dist/usage.d.ts.map +1 -1
  60. package/dist/usage.js +7 -2
  61. package/dist/usage.js.map +1 -1
  62. package/package.json +22 -7
  63. package/src/AiSdkProvider.test.ts +198 -37
  64. package/src/AiSdkProvider.ts +192 -59
  65. package/src/Mock.test.ts +32 -18
  66. package/src/Mock.ts +58 -19
  67. package/src/Pool.test.ts +71 -13
  68. package/src/Pool.ts +78 -13
  69. package/src/ProviderRegistry.test.ts +1 -1
  70. package/src/accounting.ts +0 -1
  71. package/src/accountingPublic.ts +9 -0
  72. package/src/boundaries.test.ts +28 -15
  73. package/src/capacity.test.ts +92 -0
  74. package/src/capacity.ts +140 -0
  75. package/src/catalogProvider.test.ts +82 -9
  76. package/src/catalogProvider.ts +24 -21
  77. package/src/compatibleProvider.test.ts +1 -2
  78. package/src/compatibleProvider.ts +10 -7
  79. package/src/cost.test.ts +31 -0
  80. package/src/env.test.ts +49 -20
  81. package/src/env.ts +114 -51
  82. package/src/errors.test.ts +33 -0
  83. package/src/errors.ts +38 -134
  84. package/src/index.ts +5 -2
  85. package/src/ollama.test.ts +1 -2
  86. package/src/promptTokens.ts +8 -5
  87. package/src/providerError.ts +139 -0
  88. package/src/sdkModels.test.ts +2 -5
  89. package/src/sdkModels.ts +6 -8
  90. package/src/types.ts +41 -19
  91. package/src/usage.ts +7 -2
@@ -0,0 +1,140 @@
1
+ import type {
2
+ PromptTokenMeasurement,
3
+ ProviderRequestCapacity,
4
+ } from "./types.ts";
5
+ import { assertPromptTokenMeasurement } from "./promptTokens.ts";
6
+
7
+ const positiveOrNull = (value: number | null, name: string): number | null => {
8
+ if (value === null) return null;
9
+ if (!Number.isSafeInteger(value) || value <= 0) {
10
+ throw new TypeError(`${name} must be a positive safe integer or null`);
11
+ }
12
+ return value;
13
+ };
14
+
15
+ export const effectiveOutputBudget = ({
16
+ requested,
17
+ configured,
18
+ maxOutputTokens,
19
+ contextWindow,
20
+ }: {
21
+ requested?: number;
22
+ configured: number | null;
23
+ maxOutputTokens: number | null;
24
+ contextWindow: number | null;
25
+ }): number | null => {
26
+ if (requested !== undefined && (!Number.isSafeInteger(requested) || requested <= 0)) {
27
+ throw new TypeError("maxOutputTokens must be a positive safe integer");
28
+ }
29
+ const policies = [requested ?? null, configured].filter((value): value is number => value !== null);
30
+ if (policies.length === 0) return null;
31
+ const physical = [
32
+ positiveOrNull(maxOutputTokens, "maxOutputTokens"),
33
+ positiveOrNull(contextWindow, "contextWindow"),
34
+ ].filter((value): value is number => value !== null);
35
+ return Math.min(...policies, ...physical);
36
+ };
37
+
38
+ export const effectiveReasoningBudget = ({
39
+ configured,
40
+ outputBudget,
41
+ }: {
42
+ configured: number | null;
43
+ outputBudget: number | null;
44
+ }): number | null => {
45
+ positiveOrNull(configured, "reasoningBudget");
46
+ positiveOrNull(outputBudget, "outputBudget");
47
+ if (configured === null) return null;
48
+ if (outputBudget === null) {
49
+ throw new TypeError("a reasoning budget requires a resolved total output budget");
50
+ }
51
+ if (outputBudget < 2) {
52
+ throw new TypeError("maxOutputTokens must leave at least one token outside the reasoning budget");
53
+ }
54
+ return Math.min(configured, outputBudget - 1);
55
+ };
56
+
57
+ export const effectiveInputCapacity = ({
58
+ contextWindow,
59
+ maxInputTokens,
60
+ outputBudget,
61
+ }: {
62
+ contextWindow: number | null;
63
+ maxInputTokens: number | null;
64
+ outputBudget: number | null;
65
+ }): number | null => {
66
+ positiveOrNull(contextWindow, "contextWindow");
67
+ positiveOrNull(maxInputTokens, "maxInputTokens");
68
+ positiveOrNull(outputBudget, "outputBudget");
69
+ const combinedCapacity = contextWindow !== null && outputBudget !== null
70
+ ? contextWindow - outputBudget
71
+ : null;
72
+ if (combinedCapacity !== null && combinedCapacity <= 0) {
73
+ throw new TypeError(
74
+ `outputBudget (${outputBudget}) must leave positive input capacity inside contextWindow (${contextWindow})`,
75
+ );
76
+ }
77
+ const capacities = [
78
+ maxInputTokens,
79
+ combinedCapacity,
80
+ ].filter((value): value is number => value !== null);
81
+ return capacities.length === 0 ? null : Math.min(...capacities);
82
+ };
83
+
84
+ export const requestCapacityDecision = (
85
+ inputCapacity: number | null,
86
+ measurement: PromptTokenMeasurement,
87
+ ): ProviderRequestCapacity["decision"] => {
88
+ const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
89
+ // An upper bound can prove fit when it is below the limit, but exceeding
90
+ // the limit proves nothing about the unknown exact count. Estimates never
91
+ // authorize or reject; the provider remains the capacity oracle.
92
+ return inputCapacity === null
93
+ || prompt.kind === "estimate"
94
+ || prompt.kind === "unavailable"
95
+ ? "defer"
96
+ : prompt.tokens <= inputCapacity
97
+ ? "admit"
98
+ : prompt.kind === "exact"
99
+ ? "reject"
100
+ : "defer";
101
+ };
102
+
103
+ export const assessRequestCapacity = ({
104
+ contextWindow,
105
+ maxInputTokens,
106
+ maxOutputTokens,
107
+ outputBudget,
108
+ reasoningBudget,
109
+ measurement,
110
+ }: {
111
+ contextWindow: number | null;
112
+ maxInputTokens: number | null;
113
+ maxOutputTokens: number | null;
114
+ outputBudget: number | null;
115
+ reasoningBudget: number | null;
116
+ measurement: PromptTokenMeasurement;
117
+ }): ProviderRequestCapacity => {
118
+ positiveOrNull(contextWindow, "contextWindow");
119
+ positiveOrNull(maxInputTokens, "maxInputTokens");
120
+ positiveOrNull(maxOutputTokens, "maxOutputTokens");
121
+ positiveOrNull(outputBudget, "outputBudget");
122
+ positiveOrNull(reasoningBudget, "reasoningBudget");
123
+ if (reasoningBudget !== null
124
+ && (outputBudget === null || reasoningBudget >= outputBudget)) {
125
+ throw new TypeError("reasoningBudget must be a strict subset of outputBudget");
126
+ }
127
+ const prompt = assertPromptTokenMeasurement(measurement, "provider capacity");
128
+ const inputCapacity = effectiveInputCapacity({ contextWindow, maxInputTokens, outputBudget });
129
+
130
+ return {
131
+ decision: requestCapacityDecision(inputCapacity, prompt),
132
+ contextWindow,
133
+ maxInputTokens,
134
+ maxOutputTokens,
135
+ outputBudget,
136
+ reasoningBudget,
137
+ inputCapacity,
138
+ prompt,
139
+ };
140
+ };
@@ -16,8 +16,7 @@ const env = {
16
16
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
17
17
  PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
18
18
  PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
19
- PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
20
- PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
19
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
21
20
  PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
22
21
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
23
22
  PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
@@ -34,18 +33,20 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
34
33
  assert.notEqual(provider, null);
35
34
  assert.equal(provider?.model, "gpt-4.1-mini");
36
35
  assert.equal(provider?.contextWindow, 1_047_576);
37
- assert.equal(provider?.reasoningReserve, 16_384);
38
- assert.equal(provider?.completionReserve, 32_768);
36
+ assert.equal(provider?.maxInputTokens, null);
37
+ assert.equal(provider?.maxOutputTokens, 32_768);
38
+ assert.equal(provider?.outputBudget, 32_768);
39
+ assert.equal(provider?.reasoningBudget, null);
39
40
  });
40
41
 
41
- test("an operator context window caps catalog physics and percentage reserves derive from the cap", () => {
42
+ test("an operator context window caps catalog physics and percentage output policy", () => {
42
43
  const provider = catalogProviderFromEnv("openai", {
43
44
  ...env,
44
45
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "128000",
45
46
  }, "gpt-4.1-mini");
46
47
  assert.equal(provider?.contextWindow, 128_000);
47
- assert.equal(provider?.reasoningReserve, 16_000);
48
- assert.equal(provider?.completionReserve, 32_000);
48
+ assert.equal(provider?.outputBudget, 32_768, "the model output maximum remains the tighter cap");
49
+ assert.equal(provider?.reasoningBudget, null);
49
50
 
50
51
  const oversized = catalogProviderFromEnv("openai", {
51
52
  ...env,
@@ -89,7 +90,7 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
89
90
  const result = await provider?.generate({
90
91
  workerId: "worker",
91
92
  messages: [{ role: "user", content: "hello" }],
92
- maxTokens: 64,
93
+ maxOutputTokens: 64,
93
94
  sampling: { top_p: 0.8, seed: 7 },
94
95
  });
95
96
 
@@ -109,6 +110,78 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
109
110
  assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
110
111
  });
111
112
 
113
+ test("xAI's native chat contract caps the complete reasoning response", async () => {
114
+ let call: { headers: Headers; body: Record<string, unknown> } | undefined;
115
+ mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
116
+ call = {
117
+ headers: new Headers(init?.headers),
118
+ body: JSON.parse(String(init?.body)) as Record<string, unknown>,
119
+ };
120
+ return new Response([
121
+ `data: ${JSON.stringify({
122
+ id: "response-xai",
123
+ object: "chat.completion.chunk",
124
+ created: 1,
125
+ model: "grok-build-0.1",
126
+ choices: [{ index: 0, delta: { reasoning_content: "consider" }, finish_reason: null }],
127
+ })}`,
128
+ `data: ${JSON.stringify({
129
+ id: "response-xai",
130
+ object: "chat.completion.chunk",
131
+ created: 2,
132
+ model: "grok-build-0.1",
133
+ choices: [{ index: 0, delta: { content: "OK" }, finish_reason: "stop" }],
134
+ })}`,
135
+ `data: ${JSON.stringify({
136
+ id: "response-xai",
137
+ object: "chat.completion.chunk",
138
+ created: 3,
139
+ model: "grok-build-0.1",
140
+ choices: [],
141
+ usage: {
142
+ prompt_tokens: 5,
143
+ completion_tokens: 4,
144
+ total_tokens: 9,
145
+ prompt_tokens_details: { cached_tokens: 2 },
146
+ completion_tokens_details: { reasoning_tokens: 3 },
147
+ cost_in_usd_ticks: 1_230_000,
148
+ },
149
+ })}`,
150
+ "data: [DONE]",
151
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
152
+ });
153
+
154
+ const provider = catalogProviderFromEnv("xai", {
155
+ ...env,
156
+ XAI_API_KEY: "test-key",
157
+ PLURNK_PROVIDERS_REASONING: "adaptive",
158
+ }, "grok-build-0.1");
159
+ const result = await provider?.generate({
160
+ workerId: "xai-worker",
161
+ messages: [{ role: "user", content: "hello" }],
162
+ maxOutputTokens: 16,
163
+ });
164
+
165
+ assert.equal(call?.body.max_completion_tokens, 16);
166
+ assert.equal("max_tokens" in (call?.body ?? {}), false);
167
+ assert.equal(call?.headers.get("x-grok-conv-id"), "xai-worker");
168
+ assert.equal(result?.assistant.reasoning, "consider");
169
+ assert.equal(result?.assistant.content, "OK");
170
+ assert.deepEqual(result?.accounting[0]?.usage, {
171
+ inputTokens: 5,
172
+ outputTokens: 4,
173
+ totalTokens: 9,
174
+ inputTokenDetails: { cacheReadTokens: 2 },
175
+ outputTokenDetails: { textTokens: 1, reasoningTokens: 3 },
176
+ });
177
+ assert.deepEqual(result?.accounting[0]?.cost, {
178
+ kind: "charged",
179
+ amount: { amount: "1230000", currency: "USDTICK" },
180
+ usdEquivalent: "0.000123",
181
+ source: "xAI response usage.cost_in_usd_ticks",
182
+ });
183
+ });
184
+
112
185
  test("Cerebras explicit reasoning activation needs no operator effort or token budget", async () => {
113
186
  let body: Record<string, unknown> | undefined;
114
187
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
@@ -347,7 +420,7 @@ test("Models.dev is the only fallback rate table", async () => {
347
420
  `data: ${JSON.stringify({
348
421
  id: "response",
349
422
  model: "served",
350
- choices: [{ delta: { content: "ok" }, finish_reason: "stop" }],
423
+ choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
351
424
  })}`,
352
425
  `data: ${JSON.stringify({
353
426
  id: "response",
@@ -3,7 +3,7 @@ import {
3
3
  contextWindowFromEnv,
4
4
  effectiveContextWindow,
5
5
  dataCaptureFromEnv,
6
- envelopeFromEnv,
6
+ generationEnvelopeFromEnv,
7
7
  parseRequiredFloat,
8
8
  parseRequiredInt,
9
9
  parseTimeoutMs,
@@ -11,8 +11,6 @@ import {
11
11
  cacheWritePolicyFromEnv,
12
12
  reasoningFromEnv,
13
13
  reasoningResponseStyleFromEnv,
14
- resolveReserve,
15
- type ReserveSpec,
16
14
  } from "./env.ts";
17
15
  import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
18
16
  import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
@@ -55,6 +53,7 @@ export const providerFromSdkModel = ({
55
53
  cacheAffinity,
56
54
  systemCacheProviderOptions,
57
55
  reasoningResponseProviderOptions,
56
+ additiveReasoningProvider,
58
57
  }: {
59
58
  name: string;
60
59
  env: NodeJS.ProcessEnv;
@@ -69,32 +68,32 @@ export const providerFromSdkModel = ({
69
68
  cacheAffinity?: CacheAffinity;
70
69
  systemCacheProviderOptions?: AiSdkProviderOptions;
71
70
  reasoningResponseProviderOptions?: AiSdkProviderOptions;
71
+ additiveReasoningProvider?: "anthropic" | "bedrock";
72
72
  }): Provider => {
73
73
  emitWarningOnce(
74
- `${name} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
74
+ `${name} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
75
75
  "PLURNK_PROMPT_COUNT_ESTIMATE",
76
76
  );
77
77
 
78
- const reasoning = reasoningFromEnv(env, name);
79
- const { reasoningReserve: configuredReasoning, completionReserve: configuredCompletion } = envelopeFromEnv(env, name);
80
- const completionReserve: ReserveSpec = "tokens" in configuredCompletion
81
- ? configuredCompletion
82
- : info?.maxOutput === undefined
83
- ? configuredCompletion
84
- : { tokens: Math.min(info.maxOutput, Math.round(configuredCompletion.percent * contextWindow)) };
85
- const completionTokens = resolveReserve(completionReserve, contextWindow);
86
- const reasoningReserve: ReserveSpec = "tokens" in configuredReasoning
87
- ? configuredReasoning
88
- : reasoning.budget !== null
89
- ? { tokens: reasoning.budget }
90
- : completionTokens === null
91
- ? configuredReasoning
92
- : { tokens: Math.round(completionTokens / 2) };
78
+ const maxInputTokens = info?.maxInputTokens ?? null;
79
+ const maxOutputTokens = info?.maxOutputTokens === undefined
80
+ ? null
81
+ : Math.min(info.maxOutputTokens, contextWindow);
82
+ const envelope = generationEnvelopeFromEnv(
83
+ env,
84
+ name,
85
+ contextWindow,
86
+ maxOutputTokens,
87
+ );
88
+ const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
93
89
 
94
90
  const catalogCost = info?.cost;
95
91
  const rates = catalogCost === undefined ? null : {
96
92
  input: catalogCost.inputPer1M,
97
93
  output: catalogCost.outputPer1M,
94
+ ...(catalogCost.reasoningPer1M === undefined
95
+ ? {}
96
+ : { reasoning: catalogCost.reasoningPer1M }),
98
97
  ...(catalogCost.cacheReadPer1M === undefined
99
98
  ? {}
100
99
  : { cacheRead: catalogCost.cacheReadPer1M }),
@@ -115,6 +114,11 @@ export const providerFromSdkModel = ({
115
114
  ...(url === undefined ? {} : { url }),
116
115
  ...(headers === undefined ? {} : { headers: { ...headers } }),
117
116
  contextWindow,
117
+ maxInputTokens,
118
+ maxOutputTokens,
119
+ outputBudget: envelope.outputBudget,
120
+ reasoningBudget: reasoning.budget,
121
+ ...(additiveReasoningProvider === undefined ? {} : { additiveReasoningProvider }),
118
122
  fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
119
123
  operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
120
124
  firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
@@ -124,8 +128,6 @@ export const providerFromSdkModel = ({
124
128
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
125
129
  repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
126
130
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
127
- reasoningReserve,
128
- completionReserve,
129
131
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
130
132
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
131
133
  reasoningStyle: reasoningStyleFromEnv(env, name),
@@ -183,6 +185,7 @@ export const catalogProviderFromEnv = (
183
185
  cacheAffinity: sdk.cacheAffinity,
184
186
  systemCacheProviderOptions: sdk.systemCacheProviderOptions,
185
187
  reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
188
+ additiveReasoningProvider: sdk.additiveReasoningProvider,
186
189
  contextWindow,
187
190
  info,
188
191
  });
@@ -12,8 +12,7 @@ const env = {
12
12
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
13
13
  PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
14
14
  PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
15
- PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
16
- PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
15
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
17
16
  PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
18
17
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
19
18
  PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
@@ -3,7 +3,7 @@ import {
3
3
  contextWindowFromEnv,
4
4
  effectiveContextWindow,
5
5
  dataCaptureFromEnv,
6
- envelopeFromEnv,
6
+ generationEnvelopeFromEnv,
7
7
  parseOptionalFloat,
8
8
  parseOptionalInt,
9
9
  parseRequiredFloat,
@@ -176,21 +176,26 @@ export const compatibleProviderFromEnv = async (
176
176
 
177
177
  if (!llamaServer) {
178
178
  emitWarningOnce(
179
- `${provider} provider: request-level prompt counting is a chars/2 estimate; hard context-envelope admission fails closed without exact or bounded evidence`,
179
+ `${provider} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
180
180
  "PLURNK_PROMPT_COUNT_ESTIMATE",
181
181
  );
182
182
  }
183
- const { reasoningReserve, completionReserve } = envelopeFromEnv(env, provider);
183
+ const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
184
+ const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
184
185
  return new AiSdkProvider({
185
186
  model,
186
187
  url,
187
188
  headers,
188
189
  contextWindow,
190
+ maxInputTokens: null,
191
+ maxOutputTokens: null,
192
+ outputBudget: envelope.outputBudget,
193
+ reasoningBudget: reasoning.budget,
189
194
  fetchTimeoutMs: timeout,
190
195
  operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
191
196
  firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
192
197
  streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
193
- reasoning: reasoningFromEnv(env, provider),
198
+ reasoning,
194
199
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
195
200
  reasoningStyle,
196
201
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
@@ -200,8 +205,6 @@ export const compatibleProviderFromEnv = async (
200
205
  dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
201
206
  dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", provider) ?? undefined,
202
207
  repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", provider) ?? undefined,
203
- reasoningReserve,
204
- completionReserve,
205
208
  tuningFloors: provider !== "plurnk",
206
209
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
207
210
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
@@ -222,6 +225,6 @@ export const compatibleProviderFromEnv = async (
222
225
  tokenizeUrl,
223
226
  promptTokensUrl,
224
227
  servedModel: probe.servedModel ?? undefined,
225
- requiresMaxTokens: llamaServer || undefined,
228
+ requiresOutputBudget: llamaServer || undefined,
226
229
  });
227
230
  };
package/src/cost.test.ts CHANGED
@@ -54,6 +54,37 @@ test("a distinct cache rate requires the applicable token category", () => {
54
54
  );
55
55
  });
56
56
 
57
+ test("a distinct reasoning rate prices reasoning as a subset of output", () => {
58
+ const withReasoning: ProviderUsage = {
59
+ inputTokens: 100,
60
+ outputTokens: 50,
61
+ totalTokens: 150,
62
+ outputTokenDetails: { reasoningTokens: 20 },
63
+ };
64
+ assert.deepEqual(
65
+ estimateProviderCost(withReasoning, {
66
+ input: 1,
67
+ output: 2,
68
+ reasoning: 5,
69
+ }, "Models.dev"),
70
+ {
71
+ kind: "estimated",
72
+ amount: { amount: "0.00026", currency: "USD" },
73
+ source: "Models.dev",
74
+ },
75
+ );
76
+ });
77
+
78
+ test("a distinct reasoning rate requires reported reasoning usage", () => {
79
+ assert.deepEqual(
80
+ estimateProviderCost(usage, { input: 1, output: 2, reasoning: 5 }, "Models.dev"),
81
+ {
82
+ kind: "unknown",
83
+ reason: "the provider response omitted a token category with a distinct Models.dev rate",
84
+ },
85
+ );
86
+ });
87
+
57
88
  test("decimal aggregation is exact and becomes unknown if any request is unknown", () => {
58
89
  assert.equal(addDecimals(["0.1", "0.02", "3"]), "3.12");
59
90
  assert.equal(sumProviderCostsUsd([
package/src/env.test.ts CHANGED
@@ -3,6 +3,7 @@ import { strict as assert } from "node:assert";
3
3
  import {
4
4
  cacheAffinityFromEnv,
5
5
  cacheWritePolicyFromEnv,
6
+ generationEnvelopeFromEnv,
6
7
  parseRequiredInt,
7
8
  parseOptionalInt,
8
9
  parseTimeoutMs,
@@ -51,11 +52,10 @@ test("reasoningFromEnv: activation is independent from an optional explicit budg
51
52
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "off" }, "openai"), { mode: "off", budget: null });
52
53
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "adaptive" }, "openai"), { mode: "adaptive", budget: null });
53
54
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on" }, "openai"), { mode: "on", budget: null });
54
- assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "4096" }, "openai"), { mode: "on", budget: 4096 });
55
+ assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on" }, "openai", 4096), { mode: "on", budget: 4096 });
56
+ assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "adaptive" }, "openai", 4096), { mode: "adaptive", budget: 4096 });
55
57
  assert.throws(() => reasoningFromEnv({}, "openai"), /PLURNK_PROVIDERS_REASONING must be set/);
56
58
  assert.throws(() => reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "8192" }, "openai"), /must be one of "off", "adaptive", "on"/); // the old numeric habit fails loudly
57
- assert.throws(() => reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "0" }, "openai"), /positive integer/);
58
- assert.throws(() => reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "1.5" }, "openai"), /positive integer/);
59
59
  });
60
60
 
61
61
  test("{§provider-tagged-reasoning} response style is explicit and invalid values fail at the provider boundary", () => {
@@ -112,7 +112,7 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
112
112
  PLURNK_PROVIDERS_REASONING_BUDGET_TURBODERP: "4096", // case-folds like PLURNK_MODEL_ keys
113
113
  PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_TURBODERP: "think-tags",
114
114
  PLURNK_PROVIDERS_CONTEXT_WINDOW_turboderp: "8000",
115
- PLURNK_PROVIDERS_COMPLETION_RESERVE_turboderp: "4096",
115
+ PLURNK_PROVIDERS_OUTPUT_BUDGET_turboderp: "4096",
116
116
  PLURNK_PROVIDERS_CONTEXT_WINDOW_other: "1",
117
117
  } as NodeJS.ProcessEnv;
118
118
  const scoped = scopeEnvToAlias(env, "turboderp");
@@ -120,7 +120,7 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
120
120
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING_BUDGET, "4096");
121
121
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE, "think-tags");
122
122
  assert.equal(scoped.PLURNK_PROVIDERS_CONTEXT_WINDOW, "8000");
123
- assert.equal(scoped.PLURNK_PROVIDERS_COMPLETION_RESERVE, "4096");
123
+ assert.equal(scoped.PLURNK_PROVIDERS_OUTPUT_BUDGET, "4096");
124
124
  assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING, "off"); // fallback intact
125
125
  });
126
126
 
@@ -171,16 +171,16 @@ test("contextWindowFromEnv: reads the new name, sheds CONTEXT_SIZE hard, null wh
171
171
 
172
172
  test("scopeEnvToAlias: a caller-supplied knob list scopes consumer-owned vars", async () => {
173
173
  const { scopeEnvToAlias } = await import("./env.ts");
174
- const SERVICE_KNOBS = ["PLURNK_SERVICE_MAX_TURNS", "PLURNK_SERVICE_LOOP_TIMEOUT", "PLURNK_SERVICE_EXEC_HOLD_MS", "PLURNK_SERVICE_SAFETY"];
174
+ const SERVICE_KNOBS = ["PLURNK_SERVICE_MAX_TURNS", "PLURNK_SERVICE_LOOP_TIMEOUT", "PLURNK_SERVICE_EXEC_HOLD_MS", "PLURNK_SERVICE_PROMPT_PROJECTION"];
175
175
  const env = {
176
- PLURNK_SERVICE_MAX_TURNS: "163840", PLURNK_SERVICE_LOOP_TIMEOUT: "16384", PLURNK_SERVICE_EXEC_HOLD_MS: "49152", PLURNK_SERVICE_SAFETY: "1024",
176
+ PLURNK_SERVICE_MAX_TURNS: "163840", PLURNK_SERVICE_LOOP_TIMEOUT: "16384", PLURNK_SERVICE_EXEC_HOLD_MS: "49152", PLURNK_SERVICE_PROMPT_PROJECTION: "25%",
177
177
  PLURNK_SERVICE_MAX_TURNS_turboderp: "78848", PLURNK_SERVICE_LOOP_TIMEOUT_turboderp: "4096", PLURNK_SERVICE_EXEC_HOLD_MS_TURBODERP: "8192", // case-folds
178
178
  } as NodeJS.ProcessEnv;
179
179
  const gemma = scopeEnvToAlias(env, "turboderp", SERVICE_KNOBS);
180
180
  assert.equal(gemma.PLURNK_SERVICE_MAX_TURNS, "78848");
181
181
  assert.equal(gemma.PLURNK_SERVICE_LOOP_TIMEOUT, "4096");
182
182
  assert.equal(gemma.PLURNK_SERVICE_EXEC_HOLD_MS, "8192");
183
- assert.equal(gemma.PLURNK_SERVICE_SAFETY, "1024"); // bare fallback intact
183
+ assert.equal(gemma.PLURNK_SERVICE_PROMPT_PROJECTION, "25%"); // bare fallback intact
184
184
  const cloud = scopeEnvToAlias(env, "fireslow", SERVICE_KNOBS);
185
185
  assert.equal(cloud.PLURNK_SERVICE_LOOP_TIMEOUT, "16384"); // 64k envelope untouched by gemma overrides
186
186
  assert.equal(cloud.PLURNK_SERVICE_EXEC_HOLD_MS, "49152");
@@ -241,23 +241,52 @@ test("the shipped DRY floor is off and claims no universally safe shape", async
241
241
 
242
242
  // -- {§provider-generation-envelope} --
243
243
 
244
- test("envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
245
- const { envelopeFromEnv } = await import("./env.ts");
244
+ test("generationEnvelopeFromEnv: output is total and reasoning is an optional subset", () => {
246
245
  assert.deepEqual(
247
- envelopeFromEnv({ PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "4096" } as NodeJS.ProcessEnv, "x"),
248
- { reasoningReserve: { percent: 0.1 }, completionReserve: { tokens: 4096 } },
246
+ generationEnvelopeFromEnv({
247
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
248
+ PLURNK_PROVIDERS_REASONING_BUDGET: "4096",
249
+ } as NodeJS.ProcessEnv, "x", 100_000, 32_000),
250
+ { outputBudget: 32_000, reasoningBudget: 4096 },
251
+ );
252
+ assert.throws(() => generationEnvelopeFromEnv({}, "x", 100_000, null), /PLURNK_PROVIDERS_OUTPUT_BUDGET must be set/);
253
+ assert.throws(() => generationEnvelopeFromEnv({ PLURNK_PROVIDERS_OUTPUT_BUDGET: "150%" }, "x", 100_000, null), /percentage must be in \(0, 100\)/);
254
+ assert.throws(() => generationEnvelopeFromEnv({ PLURNK_PROVIDERS_OUTPUT_BUDGET: "-5" }, "x", 100_000, null), /positive integer token count/);
255
+ assert.throws(() => generationEnvelopeFromEnv({ PLURNK_PROVIDERS_OUTPUT_BUDGET: "100000" }, "x", 100_000, null), /must leave positive input capacity/);
256
+ assert.throws(() => generationEnvelopeFromEnv({
257
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "20%",
258
+ PLURNK_PROVIDERS_REASONING_BUDGET: "25%",
259
+ }, "x", 100_000, null), /reasoning is a subset of total output/);
260
+ assert.deepEqual(
261
+ generationEnvelopeFromEnv({ PLURNK_PROVIDERS_OUTPUT_BUDGET: "1%" }, "x", 2, null),
262
+ { outputBudget: 1, reasoningBudget: null },
263
+ "a valid percentage always resolves to at least one whole token",
264
+ );
265
+ assert.throws(
266
+ () => generationEnvelopeFromEnv({ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%" }, "x", 1, null),
267
+ /must leave positive input capacity/,
249
268
  );
250
- assert.throws(() => envelopeFromEnv({ PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%" } as NodeJS.ProcessEnv, "x"), /PLURNK_PROVIDERS_REASONING_RESERVE must be set/);
251
- assert.throws(() => envelopeFromEnv({ PLURNK_PROVIDERS_REASONING_RESERVE: "150%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%" } as NodeJS.ProcessEnv, "x"), /percentage must be in \(0, 100\)/);
252
- assert.throws(() => envelopeFromEnv({ PLURNK_PROVIDERS_REASONING_RESERVE: "-5", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%" } as NodeJS.ProcessEnv, "x"), /positive integer token count/);
253
269
  });
254
270
 
255
271
  test("envelope knobs are per-alias scopable (measured envelope per box)", async () => {
256
- const { scopeEnvToAlias, envelopeFromEnv } = await import("./env.ts");
272
+ const { scopeEnvToAlias } = await import("./env.ts");
257
273
  const env = {
258
- PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
259
- PLURNK_PROVIDERS_REASONING_RESERVE_turboderp: "4096", PLURNK_PROVIDERS_COMPLETION_RESERVE_turboderp: "8192",
274
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
275
+ PLURNK_PROVIDERS_REASONING_BUDGET: "10%",
276
+ PLURNK_PROVIDERS_OUTPUT_BUDGET_turboderp: "8192",
277
+ PLURNK_PROVIDERS_REASONING_BUDGET_turboderp: "4096",
260
278
  } as NodeJS.ProcessEnv;
261
- assert.deepEqual(envelopeFromEnv(scopeEnvToAlias(env, "turboderp"), "x"), { reasoningReserve: { tokens: 4096 }, completionReserve: { tokens: 8192 } });
262
- assert.deepEqual(envelopeFromEnv(scopeEnvToAlias(env, "jennifer"), "x"), { reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
279
+ assert.deepEqual(generationEnvelopeFromEnv(scopeEnvToAlias(env, "turboderp"), "x", 49_152, null), { outputBudget: 8192, reasoningBudget: 4096 });
280
+ assert.deepEqual(generationEnvelopeFromEnv(scopeEnvToAlias(env, "jennifer"), "x", 100_000, null), { outputBudget: 35_000, reasoningBudget: 10_000 });
281
+ });
282
+
283
+ test("retired additive reserve knobs fail rather than creating a dual contract", () => {
284
+ assert.throws(() => generationEnvelopeFromEnv({
285
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
286
+ PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
287
+ }, "x", 100_000, null), /PLURNK_PROVIDERS_REASONING_RESERVE is retired/);
288
+ assert.throws(() => generationEnvelopeFromEnv({
289
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
290
+ PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
291
+ }, "x", 100_000, null), /PLURNK_PROVIDERS_COMPLETION_RESERVE is retired/);
263
292
  });