@plurnk/plurnk-providers 1.5.0 → 1.6.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/.env.defaults +41 -34
  2. package/README.md +15 -0
  3. package/SPEC.md +242 -89
  4. package/dist/AiSdkProvider.d.ts +33 -33
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +442 -133
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +10 -11
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +87 -25
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +9 -24
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +86 -25
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts +5 -2
  17. package/dist/accounting.d.ts.map +1 -1
  18. package/dist/accounting.js +100 -16
  19. package/dist/accounting.js.map +1 -1
  20. package/dist/accountingPublic.d.ts +5 -0
  21. package/dist/accountingPublic.d.ts.map +1 -0
  22. package/dist/accountingPublic.js +3 -0
  23. package/dist/accountingPublic.js.map +1 -0
  24. package/dist/aiSdkTransport.d.ts +9 -2
  25. package/dist/aiSdkTransport.d.ts.map +1 -1
  26. package/dist/aiSdkTransport.js +160 -62
  27. package/dist/aiSdkTransport.js.map +1 -1
  28. package/dist/capacity.d.ts +26 -0
  29. package/dist/capacity.d.ts.map +1 -0
  30. package/dist/capacity.js +90 -0
  31. package/dist/capacity.js.map +1 -0
  32. package/dist/catalogProvider.d.ts +8 -3
  33. package/dist/catalogProvider.d.ts.map +1 -1
  34. package/dist/catalogProvider.js +45 -41
  35. package/dist/catalogProvider.js.map +1 -1
  36. package/dist/compatibleProvider.d.ts.map +1 -1
  37. package/dist/compatibleProvider.js +26 -12
  38. package/dist/compatibleProvider.js.map +1 -1
  39. package/dist/cost.d.ts +10 -10
  40. package/dist/cost.d.ts.map +1 -1
  41. package/dist/cost.js +90 -42
  42. package/dist/cost.js.map +1 -1
  43. package/dist/env.d.ts +13 -11
  44. package/dist/env.d.ts.map +1 -1
  45. package/dist/env.js +83 -46
  46. package/dist/env.js.map +1 -1
  47. package/dist/errors.d.ts +17 -3
  48. package/dist/errors.d.ts.map +1 -1
  49. package/dist/errors.js +91 -8
  50. package/dist/errors.js.map +1 -1
  51. package/dist/index.d.ts +7 -6
  52. package/dist/index.d.ts.map +1 -1
  53. package/dist/index.js +5 -3
  54. package/dist/index.js.map +1 -1
  55. package/dist/ollama.js +3 -3
  56. package/dist/ollama.js.map +1 -1
  57. package/dist/promptTokens.d.ts.map +1 -1
  58. package/dist/promptTokens.js +7 -4
  59. package/dist/promptTokens.js.map +1 -1
  60. package/dist/sdkModels.d.ts +7 -2
  61. package/dist/sdkModels.d.ts.map +1 -1
  62. package/dist/sdkModels.js +43 -13
  63. package/dist/sdkModels.js.map +1 -1
  64. package/dist/types.d.ts +55 -33
  65. package/dist/types.d.ts.map +1 -1
  66. package/dist/usage.d.ts +22 -5
  67. package/dist/usage.d.ts.map +1 -1
  68. package/dist/usage.js +169 -83
  69. package/dist/usage.js.map +1 -1
  70. package/package.json +18 -7
  71. package/src/AiSdkProvider.test.ts +964 -206
  72. package/src/AiSdkProvider.ts +545 -155
  73. package/src/Mock.test.ts +69 -30
  74. package/src/Mock.ts +99 -29
  75. package/src/Pool.test.ts +90 -19
  76. package/src/Pool.ts +96 -27
  77. package/src/ProviderRegistry.test.ts +16 -11
  78. package/src/accounting.test.ts +58 -22
  79. package/src/accounting.ts +119 -18
  80. package/src/accountingPublic.ts +9 -0
  81. package/src/aiSdkTransport.test.ts +42 -49
  82. package/src/aiSdkTransport.ts +174 -62
  83. package/src/boundaries.test.ts +2 -0
  84. package/src/capacity.test.ts +92 -0
  85. package/src/capacity.ts +140 -0
  86. package/src/catalogProvider.test.ts +339 -30
  87. package/src/catalogProvider.ts +65 -47
  88. package/src/compatibleProvider.test.ts +7 -5
  89. package/src/compatibleProvider.ts +29 -13
  90. package/src/cost.test.ts +86 -36
  91. package/src/cost.ts +111 -50
  92. package/src/defaults.test.ts +13 -3
  93. package/src/env.test.ts +103 -25
  94. package/src/env.ts +153 -65
  95. package/src/errors.test.ts +80 -2
  96. package/src/errors.ts +107 -8
  97. package/src/index.ts +26 -7
  98. package/src/ollama.test.ts +5 -3
  99. package/src/ollama.ts +3 -3
  100. package/src/promptTokens.ts +8 -5
  101. package/src/sdkModels.test.ts +77 -8
  102. package/src/sdkModels.ts +51 -15
  103. package/src/types.ts +112 -51
  104. package/src/usage.test.ts +112 -116
  105. package/src/usage.ts +214 -93
@@ -3,24 +3,24 @@ import {
3
3
  contextWindowFromEnv,
4
4
  effectiveContextWindow,
5
5
  dataCaptureFromEnv,
6
- envelopeFromEnv,
6
+ generationEnvelopeFromEnv,
7
7
  parseRequiredFloat,
8
8
  parseRequiredInt,
9
- promptCacheKeyFromEnv,
9
+ parseTimeoutMs,
10
+ cacheAffinityFromEnv,
11
+ cacheWritePolicyFromEnv,
10
12
  reasoningFromEnv,
11
13
  reasoningResponseStyleFromEnv,
12
- resolveReserve,
13
- type ReserveSpec,
14
14
  } from "./env.ts";
15
15
  import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
16
16
  import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
17
17
  import { providerSource } from "./notices.ts";
18
- import type { AuthoritativeChargeNormalizer, Provider, ProviderUsage } from "./types.ts";
19
- import { calculateCostUsd, calculateCostUsdDecimal } from "./usage.ts";
18
+ import type { Provider, ProviderCostNormalizer } from "./types.ts";
19
+ import { estimateProviderCost } from "./cost.ts";
20
20
  import { emitWarningOnce } from "./warnings.ts";
21
21
  import type { LanguageModel } from "ai";
22
+ import type { AiSdkProviderOptions, CacheAffinity } from "./AiSdkProvider.ts";
22
23
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
23
- import type { ProviderCost } from "@plurnk/plurnk-contracts";
24
24
 
25
25
  const reasoningStyleFromEnv = (
26
26
  env: NodeJS.ProcessEnv,
@@ -44,88 +44,102 @@ export const providerFromSdkModel = ({
44
44
  env,
45
45
  model,
46
46
  languageModel,
47
- normalizeCharge,
47
+ normalizeCost,
48
48
  url,
49
49
  headers,
50
50
  contextWindow,
51
51
  info,
52
52
  attributions,
53
+ cacheAffinity,
54
+ systemCacheProviderOptions,
55
+ reasoningResponseProviderOptions,
56
+ additiveReasoningProvider,
53
57
  }: {
54
58
  name: string;
55
59
  env: NodeJS.ProcessEnv;
56
60
  model: string;
57
61
  languageModel?: LanguageModel;
58
- normalizeCharge?: AuthoritativeChargeNormalizer;
62
+ normalizeCost?: ProviderCostNormalizer;
59
63
  url?: string;
60
64
  headers?: Readonly<Record<string, string>>;
61
65
  contextWindow: number;
62
66
  info?: ModelInfo;
63
67
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
68
+ cacheAffinity?: CacheAffinity;
69
+ systemCacheProviderOptions?: AiSdkProviderOptions;
70
+ reasoningResponseProviderOptions?: AiSdkProviderOptions;
71
+ additiveReasoningProvider?: "anthropic" | "bedrock";
64
72
  }): Provider => {
65
73
  emitWarningOnce(
66
- `${name} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
74
+ `${name} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
67
75
  "PLURNK_PROMPT_COUNT_ESTIMATE",
68
76
  );
69
77
 
70
- const reasoning = reasoningFromEnv(env, name);
71
- const { reasoningReserve: configuredReasoning, completionReserve: configuredCompletion } = envelopeFromEnv(env, name);
72
- const completionReserve: ReserveSpec = "tokens" in configuredCompletion
73
- ? configuredCompletion
74
- : info?.maxOutput === undefined
75
- ? configuredCompletion
76
- : { tokens: Math.min(info.maxOutput, Math.round(configuredCompletion.percent * contextWindow)) };
77
- const completionTokens = resolveReserve(completionReserve, contextWindow);
78
- const reasoningReserve: ReserveSpec = "tokens" in configuredReasoning
79
- ? configuredReasoning
80
- : reasoning.budget !== null
81
- ? { tokens: reasoning.budget }
82
- : completionTokens === null
83
- ? configuredReasoning
84
- : { tokens: Math.round(completionTokens / 2) };
78
+ const maxInputTokens = info?.maxInputTokens ?? null;
79
+ const maxOutputTokens = info?.maxOutputTokens === undefined
80
+ ? null
81
+ : Math.min(info.maxOutputTokens, contextWindow);
82
+ const envelope = generationEnvelopeFromEnv(
83
+ env,
84
+ name,
85
+ contextWindow,
86
+ maxOutputTokens,
87
+ );
88
+ const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
85
89
 
86
90
  const catalogCost = info?.cost;
87
91
  const rates = catalogCost === undefined ? null : {
88
92
  input: catalogCost.inputPer1M,
89
93
  output: catalogCost.outputPer1M,
90
- cached: catalogCost.cacheReadPer1M ?? catalogCost.inputPer1M,
94
+ ...(catalogCost.reasoningPer1M === undefined
95
+ ? {}
96
+ : { reasoning: catalogCost.reasoningPer1M }),
97
+ ...(catalogCost.cacheReadPer1M === undefined
98
+ ? {}
99
+ : { cacheRead: catalogCost.cacheReadPer1M }),
100
+ ...(catalogCost.cacheWritePer1M === undefined
101
+ ? {}
102
+ : { cacheWrite: catalogCost.cacheWritePer1M }),
91
103
  };
92
- const calculateCost = rates === null
93
- ? undefined
94
- : (usage: ProviderUsage): number => calculateCostUsd(usage, rates);
95
- const calculateCharge: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }> = rates === null
96
- ? () => ({ kind: "unknown", reason: "the response reported no cost and Models.dev has no rate for this model" })
97
- : rates.input === 0 && rates.output === 0 && rates.cached === 0
98
- ? () => ({ kind: "free", source: "Models.dev catalog rates" })
99
- : (usage: ProviderUsage) => ({
100
- kind: "estimated",
101
- usd: calculateCostUsdDecimal(usage, rates),
102
- source: "Models.dev catalog rates",
103
- });
104
+ const estimateCost = (usage: Parameters<typeof estimateProviderCost>[0]) =>
105
+ estimateProviderCost(usage, rates, "Models.dev catalog rates");
106
+ const affinityEnabled = cacheAffinityFromEnv(env, name);
107
+ const cacheWritePolicy = cacheWritePolicyFromEnv(env, name);
104
108
 
105
109
  return new AiSdkProvider({
106
110
  model,
107
111
  ...(attributions === undefined ? {} : { attributions }),
108
112
  ...(languageModel === undefined ? {} : { languageModel }),
109
- ...(normalizeCharge === undefined ? {} : { normalizeCharge }),
113
+ ...(normalizeCost === undefined ? {} : { normalizeCost }),
110
114
  ...(url === undefined ? {} : { url }),
111
115
  ...(headers === undefined ? {} : { headers: { ...headers } }),
112
116
  contextWindow,
113
- fetchTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
114
- streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
117
+ maxInputTokens,
118
+ maxOutputTokens,
119
+ outputBudget: envelope.outputBudget,
120
+ reasoningBudget: reasoning.budget,
121
+ ...(additiveReasoningProvider === undefined ? {} : { additiveReasoningProvider }),
122
+ fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
123
+ operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
124
+ firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
125
+ streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name),
115
126
  reasoning,
116
127
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, name),
117
128
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
118
129
  repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
119
130
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
120
- reasoningReserve,
121
- completionReserve,
122
131
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
123
132
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
124
133
  reasoningStyle: reasoningStyleFromEnv(env, name),
125
- promptCacheKey: url === undefined ? false : promptCacheKeyFromEnv(env, name),
134
+ ...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
135
+ ...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
136
+ ? { systemCacheProviderOptions }
137
+ : {}),
138
+ ...(reasoningResponseProviderOptions === undefined
139
+ ? {}
140
+ : { reasoningResponseProviderOptions }),
126
141
  serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
127
- calculateCost,
128
- calculateCharge,
142
+ estimateCost,
129
143
  source: providerSource(name),
130
144
  gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
131
145
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
@@ -165,9 +179,13 @@ export const catalogProviderFromEnv = (
165
179
  env,
166
180
  model: wireModel,
167
181
  languageModel: sdk.languageModel,
168
- normalizeCharge: sdk.normalizeCharge,
182
+ normalizeCost: sdk.normalizeCost,
169
183
  url: sdk.compatible?.url,
170
184
  headers: sdk.compatible?.headers,
185
+ cacheAffinity: sdk.cacheAffinity,
186
+ systemCacheProviderOptions: sdk.systemCacheProviderOptions,
187
+ reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
188
+ additiveReasoningProvider: sdk.additiveReasoningProvider,
171
189
  contextWindow,
172
190
  info,
173
191
  });
@@ -5,23 +5,25 @@ import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
5
5
  const env = {
6
6
  OPENAI_BASE_URL: "http://local.test/v1",
7
7
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "1000",
8
+ PLURNK_PROVIDERS_OPERATION_TIMEOUT: "3000",
9
+ PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "1000",
8
10
  PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
9
11
  PLURNK_PROVIDERS_REASONING: "off",
10
12
  PLURNK_PROVIDERS_TEMPERATURE: "0.2",
11
13
  PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15",
12
14
  PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0",
13
- PLURNK_PROVIDERS_REASONING_RESERVE: "10%",
14
- PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
15
+ PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%",
15
16
  PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
16
17
  PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512",
17
18
  PLURNK_PROVIDERS_PROBE_ATTEMPTS: "1",
18
19
  PLURNK_PROVIDERS_PROBE_DELAY: "0",
19
- PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
20
+ PLURNK_PROVIDERS_CACHE_AFFINITY: "1",
21
+ PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
20
22
  };
21
23
 
22
24
  test.afterEach(() => mock.restoreAll());
23
25
 
24
- test("compatible endpoints preserve configured prompt-cache affinity", async () => {
26
+ test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
25
27
  let body: Record<string, unknown> | undefined;
26
28
  mock.method(globalThis, "fetch", async (input: string | URL | Request, init?: RequestInit) => {
27
29
  if (String(input).endsWith("/models")) {
@@ -41,7 +43,7 @@ test("compatible endpoints preserve configured prompt-cache affinity", async ()
41
43
  messages: [{ role: "user", content: "hello" }],
42
44
  });
43
45
 
44
- assert.equal(body?.prompt_cache_key, "worker-affinity");
46
+ assert.equal("prompt_cache_key" in (body ?? {}), false);
45
47
  });
46
48
 
47
49
  test("the server-wide DRY-off floor emits no DRY request fields", async () => {
@@ -3,16 +3,19 @@ import {
3
3
  contextWindowFromEnv,
4
4
  effectiveContextWindow,
5
5
  dataCaptureFromEnv,
6
- envelopeFromEnv,
6
+ generationEnvelopeFromEnv,
7
7
  parseOptionalFloat,
8
8
  parseOptionalInt,
9
9
  parseRequiredFloat,
10
10
  parseRequiredInt,
11
- promptCacheKeyFromEnv,
11
+ parseTimeoutMs,
12
+ cacheAffinityFromEnv,
13
+ cacheWritePolicyFromEnv,
12
14
  reasoningFromEnv,
13
15
  reasoningResponseStyleFromEnv,
14
16
  } from "./env.ts";
15
17
  import { providerSource } from "./notices.ts";
18
+ import { plurnkCostNormalizer } from "./accounting.ts";
16
19
  import type { Provider } from "./types.ts";
17
20
  import { emitWarningOnce } from "./warnings.ts";
18
21
 
@@ -49,7 +52,10 @@ const probeModels = async (
49
52
  ): Promise<EndpointProbe> => {
50
53
  const modelsUrl = url.replace(/\/chat\/completions$/, "/models");
51
54
  try {
52
- const response = await fetch(modelsUrl, { headers, signal: AbortSignal.timeout(timeout) });
55
+ const response = await fetch(modelsUrl, {
56
+ headers,
57
+ ...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
58
+ });
53
59
  if (!response.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
54
60
  const data = await response.json() as {
55
61
  data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }>;
@@ -93,7 +99,7 @@ const probeProps = async (
93
99
  try {
94
100
  const response = await fetch(url.replace(/\/v1\/chat\/completions$/, "/props"), {
95
101
  headers,
96
- signal: AbortSignal.timeout(timeout),
102
+ ...(timeout > 0 ? { signal: AbortSignal.timeout(timeout) } : {}),
97
103
  });
98
104
  if (!response.ok) return { slotCount: null, eosText: null };
99
105
  const data = await response.json() as { total_slots?: number; eos_token?: string };
@@ -112,12 +118,17 @@ export const compatibleProviderFromEnv = async (
112
118
  model: string,
113
119
  baseUrlOverride?: string,
114
120
  ): Promise<Provider> => {
121
+ // The knobs remain universal and fail hard when malformed, but this local /
122
+ // first-party compatible route declares no vendor cache projection. llama-server
123
+ // already owns slot affinity and the first-party endpoint receives worker metadata.
124
+ cacheAffinityFromEnv(env, provider);
125
+ cacheWritePolicyFromEnv(env, provider);
115
126
  const url = chatUrl(provider, env, baseUrlOverride);
116
127
  const apiKey = provider === "openai" ? env.OPENAI_API_KEY : env.PLURNK_API_KEY;
117
128
  const headers: Record<string, string> = apiKey === undefined || apiKey.length === 0
118
129
  ? {}
119
130
  : { Authorization: `Bearer ${apiKey}` };
120
- const timeout = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
131
+ const timeout = parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", provider);
121
132
  const attempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", provider);
122
133
  const probe = await probeModelsRetrying(
123
134
  url,
@@ -165,19 +176,26 @@ export const compatibleProviderFromEnv = async (
165
176
 
166
177
  if (!llamaServer) {
167
178
  emitWarningOnce(
168
- `${provider} provider: physical prompt counting is a chars/2 estimate; over-policy recovery fails closed without exact or bounded request evidence`,
179
+ `${provider} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
169
180
  "PLURNK_PROMPT_COUNT_ESTIMATE",
170
181
  );
171
182
  }
172
- const { reasoningReserve, completionReserve } = envelopeFromEnv(env, provider);
183
+ const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
184
+ const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
173
185
  return new AiSdkProvider({
174
186
  model,
175
187
  url,
176
188
  headers,
177
189
  contextWindow,
190
+ maxInputTokens: null,
191
+ maxOutputTokens: null,
192
+ outputBudget: envelope.outputBudget,
193
+ reasoningBudget: reasoning.budget,
178
194
  fetchTimeoutMs: timeout,
179
- streamIdleTimeoutMs: parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
180
- reasoning: reasoningFromEnv(env, provider),
195
+ operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
196
+ firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
197
+ streamIdleTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", provider),
198
+ reasoning,
181
199
  reasoningResponseStyle: reasoningResponseStyleFromEnv(env, provider),
182
200
  reasoningStyle,
183
201
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", provider, 0),
@@ -187,12 +205,9 @@ export const compatibleProviderFromEnv = async (
187
205
  dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", provider, 0) ?? undefined,
188
206
  dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", provider) ?? undefined,
189
207
  repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", provider) ?? undefined,
190
- reasoningReserve,
191
- completionReserve,
192
208
  tuningFloors: provider !== "plurnk",
193
209
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", provider),
194
210
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
195
- promptCacheKey: promptCacheKeyFromEnv(env, provider),
196
211
  source: providerSource(provider),
197
212
  grammarStyle,
198
213
  gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
@@ -200,6 +215,7 @@ export const compatibleProviderFromEnv = async (
200
215
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
201
216
  ...dataCaptureFromEnv(env, provider),
202
217
  firstPartyMetadata: provider === "plurnk",
218
+ normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
203
219
  apiKeyRejectedMessage: provider === "plurnk"
204
220
  ? "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired)."
205
221
  : undefined,
@@ -209,6 +225,6 @@ export const compatibleProviderFromEnv = async (
209
225
  tokenizeUrl,
210
226
  promptTokensUrl,
211
227
  servedModel: probe.servedModel ?? undefined,
212
- requiresMaxTokens: llamaServer || undefined,
228
+ requiresOutputBudget: llamaServer || undefined,
213
229
  });
214
230
  };
package/src/cost.test.ts CHANGED
@@ -1,62 +1,112 @@
1
1
  import assert from "node:assert/strict";
2
2
  import test from "node:test";
3
3
  import {
4
- providerCostFor,
4
+ addDecimals,
5
+ estimateProviderCost,
5
6
  providerCostUsd,
6
- validateAuthoritativeCharge,
7
+ resolveProviderCost,
8
+ sumProviderCostsUsd,
9
+ validateChargedCost,
7
10
  } from "./cost.ts";
8
11
  import type { ProviderUsage } from "./types.ts";
9
12
 
10
13
  const usage: ProviderUsage = {
11
- prompt: 1,
12
- completion: 1,
13
- reasoning: 0,
14
- cached: 0,
15
- total: 2,
14
+ inputTokens: 1,
15
+ outputTokens: 1,
16
+ totalTokens: 2,
16
17
  };
17
18
 
18
- test("authoritative provider charge wins over a local estimate", () => {
19
- const provider = {
20
- calculateCost: () => 12,
21
- calculateCharge: () => ({ kind: "estimated", usd: "12", source: "catalog" } as const),
22
- };
23
- const charge = {
24
- kind: "authoritative",
19
+ test("direct charged evidence wins over a Models.dev estimate", () => {
20
+ const charged = {
21
+ kind: "charged",
25
22
  amount: { amount: "0.0000042", currency: "XMR" },
26
23
  usdEquivalent: "0.73",
27
- source: "settled upstream turn charge",
24
+ source: "settled upstream request charge",
28
25
  } as const;
29
- assert.deepEqual(providerCostFor(provider, usage, charge), charge);
30
- assert.equal(providerCostUsd(charge), 0.73);
26
+ const estimated = estimateProviderCost(usage, { input: 1, output: 1 }, "Models.dev");
27
+ assert.deepEqual(resolveProviderCost(charged, estimated), charged);
28
+ assert.equal(providerCostUsd(charged), "0.73");
31
29
  });
32
30
 
33
- test("explicit free remains distinguishable from unknown", () => {
34
- const free = providerCostFor({
35
- calculateCost: () => 0,
36
- calculateCharge: () => ({ kind: "free", source: "local model" }),
37
- }, usage);
38
- const unknown = providerCostFor({ calculateCost: () => 0 }, usage);
39
- assert.deepEqual(free, { kind: "free", source: "local model" });
31
+ test("an exact zero estimate remains distinguishable from unknown cost", () => {
32
+ const zero = estimateProviderCost(usage, { input: 0, output: 0 }, "Models.dev");
33
+ const unknown = estimateProviderCost(usage, null, "Models.dev");
34
+ assert.deepEqual(zero, {
35
+ kind: "estimated",
36
+ amount: { amount: "0", currency: "USD" },
37
+ source: "Models.dev",
38
+ });
40
39
  assert.deepEqual(unknown, {
41
40
  kind: "unknown",
42
- reason: "the response reported no cost and Models.dev has no rate for this model",
41
+ reason: "Models.dev has no complete rate for this model",
43
42
  });
44
- assert.equal(providerCostUsd(free), 0);
43
+ assert.equal(providerCostUsd(zero), "0");
45
44
  assert.equal(providerCostUsd(unknown), null);
46
45
  });
47
46
 
48
- test("legacy calculateCost is not a monetary-reporting authority", () => {
49
- const cost = providerCostFor({ calculateCost: () => 0.25 }, usage);
50
- assert.deepEqual(cost, {
51
- kind: "unknown",
52
- reason: "the response reported no cost and Models.dev has no rate for this model",
53
- });
54
- assert.equal(providerCostUsd(cost), null);
47
+ test("a distinct cache rate requires the applicable token category", () => {
48
+ assert.deepEqual(
49
+ estimateProviderCost(usage, { input: 1, output: 1, cacheRead: 0.1 }, "Models.dev"),
50
+ {
51
+ kind: "unknown",
52
+ reason: "the provider response omitted a token category with a distinct Models.dev rate",
53
+ },
54
+ );
55
+ });
56
+
57
+ test("a distinct reasoning rate prices reasoning as a subset of output", () => {
58
+ const withReasoning: ProviderUsage = {
59
+ inputTokens: 100,
60
+ outputTokens: 50,
61
+ totalTokens: 150,
62
+ outputTokenDetails: { reasoningTokens: 20 },
63
+ };
64
+ assert.deepEqual(
65
+ estimateProviderCost(withReasoning, {
66
+ input: 1,
67
+ output: 2,
68
+ reasoning: 5,
69
+ }, "Models.dev"),
70
+ {
71
+ kind: "estimated",
72
+ amount: { amount: "0.00026", currency: "USD" },
73
+ source: "Models.dev",
74
+ },
75
+ );
76
+ });
77
+
78
+ test("a distinct reasoning rate requires reported reasoning usage", () => {
79
+ assert.deepEqual(
80
+ estimateProviderCost(usage, { input: 1, output: 2, reasoning: 5 }, "Models.dev"),
81
+ {
82
+ kind: "unknown",
83
+ reason: "the provider response omitted a token category with a distinct Models.dev rate",
84
+ },
85
+ );
86
+ });
87
+
88
+ test("decimal aggregation is exact and becomes unknown if any request is unknown", () => {
89
+ assert.equal(addDecimals(["0.1", "0.02", "3"]), "3.12");
90
+ assert.equal(sumProviderCostsUsd([
91
+ {
92
+ kind: "charged",
93
+ amount: { amount: "0.1", currency: "USD" },
94
+ source: "provider",
95
+ },
96
+ {
97
+ kind: "estimated",
98
+ amount: { amount: "0.02", currency: "USD" },
99
+ source: "Models.dev",
100
+ },
101
+ ]), "0.12");
102
+ assert.equal(sumProviderCostsUsd([
103
+ { kind: "unknown", reason: "no evidence" },
104
+ ]), null);
55
105
  });
56
106
 
57
- test("rejects malformed money instead of mining or coercing it", () => {
58
- assert.throws(() => validateAuthoritativeCharge({
59
- kind: "authoritative",
107
+ test("malformed charged money is rejected instead of coerced", () => {
108
+ assert.throws(() => validateChargedCost({
109
+ kind: "charged",
60
110
  amount: { amount: "1e3", currency: "usd" },
61
111
  usdEquivalent: "1000",
62
112
  source: "wire",