@plurnk/plurnk-providers 1.3.12 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +39 -22
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
package/src/env.test.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { parseRequiredInt, parseOptionalInt, requireEnv, reasoningFromEnv, tokenRatesFromEnv } from "./env.ts";
3
+ import { parseRequiredInt, parseOptionalInt, requireEnv, reasoningFromEnv, reasoningResponseStyleFromEnv, tokenRatesFromEnv } from "./env.ts";
4
4
 
5
5
  test("parseRequiredInt: parses a non-negative integer", () => {
6
6
  assert.equal(parseRequiredInt("600000", "PLURNK_PROVIDERS_FETCH_TIMEOUT", "openai"), 600000);
@@ -29,7 +29,7 @@ test("parseOptionalInt: rejects fractional and negative values", () => {
29
29
  assert.throws(() => parseOptionalInt("-8", "PLURNK_PROVIDERS_CONTEXT_WINDOW", "openai"), /must be a non-negative integer/);
30
30
  });
31
31
 
32
- test("reasoningFromEnv: activation modes parse; budget required IFF on; fail-hard on everything else (#33)", () => {
32
+ test("reasoningFromEnv: activation modes parse; budget required IFF on; fail-hard on everything else", () => {
33
33
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "off" }, "openai"), { mode: "off", budget: null });
34
34
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "adaptive" }, "openai"), { mode: "adaptive", budget: null });
35
35
  assert.deepEqual(reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "4096" }, "openai"), { mode: "on", budget: 4096 });
@@ -40,6 +40,19 @@ test("reasoningFromEnv: activation modes parse; budget required IFF on; fail-har
40
40
  assert.throws(() => reasoningFromEnv({ PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "1.5" }, "openai"), /positive integer/);
41
41
  });
42
42
 
43
+ test("{§provider-tagged-reasoning} response style is explicit and invalid values fail at the provider boundary", () => {
44
+ assert.equal(reasoningResponseStyleFromEnv({}, "cloudflare"), "verbatim");
45
+ assert.equal(reasoningResponseStyleFromEnv({
46
+ PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE: "think-tags",
47
+ }, "cloudflare"), "think-tags");
48
+ assert.throws(
49
+ () => reasoningResponseStyleFromEnv({
50
+ PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE: "auto",
51
+ }, "cloudflare"),
52
+ /cloudflare provider: PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE must be "verbatim" or "think-tags" \(got "auto"\)/,
53
+ );
54
+ });
55
+
43
56
  test("requireEnv: returns the value or throws a named error", () => {
44
57
  assert.equal(requireEnv("sk-x", "OPENAI_API_KEY", "openai"), "sk-x");
45
58
  assert.throws(() => requireEnv(undefined, "GROQ_API_KEY", "groq"), /groq provider: GROQ_API_KEY must be set/);
@@ -79,15 +92,17 @@ test("scopeEnvToAlias: suffixed knob wins, bare is the fallback, other aliases i
79
92
  PLURNK_PROVIDERS_REASONING: "off",
80
93
  PLURNK_PROVIDERS_REASONING_turboderp: "on",
81
94
  PLURNK_PROVIDERS_REASONING_BUDGET_TURBODERP: "4096", // case-folds like PLURNK_MODEL_ keys
82
- PLURNK_PROVIDERS_CONTEXT_WINDOW_turboderp: "8000", // #525 gate knob — the client window cap
95
+ PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_TURBODERP: "think-tags",
96
+ PLURNK_PROVIDERS_CONTEXT_WINDOW_turboderp: "8000",
83
97
  PLURNK_PROVIDERS_COMPLETION_RESERVE_turboderp: "4096",
84
98
  PLURNK_PROVIDERS_CONTEXT_WINDOW_other: "1",
85
99
  } as NodeJS.ProcessEnv;
86
100
  const scoped = scopeEnvToAlias(env, "turboderp");
87
101
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING, "on");
88
102
  assert.equal(scoped.PLURNK_PROVIDERS_REASONING_BUDGET, "4096");
89
- assert.equal(scoped.PLURNK_PROVIDERS_CONTEXT_WINDOW, "8000"); // #525: the alias-scoped window cap promotes to bare (the release-gate knob)
90
- assert.equal(scoped.PLURNK_PROVIDERS_COMPLETION_RESERVE, "4096"); // the #507 reserves promote too
103
+ assert.equal(scoped.PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE, "think-tags");
104
+ assert.equal(scoped.PLURNK_PROVIDERS_CONTEXT_WINDOW, "8000");
105
+ assert.equal(scoped.PLURNK_PROVIDERS_COMPLETION_RESERVE, "4096");
91
106
  assert.equal(scopeEnvToAlias(env, "plain").PLURNK_PROVIDERS_REASONING, "off"); // fallback intact
92
107
  });
93
108
 
@@ -103,7 +118,7 @@ test("scopeEnvToAlias: aliases with underscores resolve; a bare knob is never mi
103
118
  assert.equal(scopeEnvToAlias(env, "budget").PLURNK_PROVIDERS_REASONING, "off"); // collision guard
104
119
  });
105
120
 
106
- test("#36 dataCaptureFromEnv: both knobs OFF by default, ON when set (TOP_LOGPROBS = the OpenAI top_logprobs count)", async () => {
121
+ test("dataCaptureFromEnv: both knobs OFF by default, ON when set (TOP_LOGPROBS = the OpenAI top_logprobs count)", async () => {
107
122
  const { dataCaptureFromEnv } = await import("./env.ts");
108
123
  assert.deepEqual(dataCaptureFromEnv({} as NodeJS.ProcessEnv, "x"), { topLogprobs: null, rawBody: false });
109
124
  assert.deepEqual(dataCaptureFromEnv({ PLURNK_PROVIDERS_RAWBODY: "0" } as NodeJS.ProcessEnv, "x"), { topLogprobs: null, rawBody: false });
@@ -120,7 +135,7 @@ test("OpenAI-lexicon shed: a still-set PLURNK_PROVIDERS_LOGPROB fails hard with
120
135
  );
121
136
  });
122
137
 
123
- test("#472 contextWindowFromEnv: reads the new name, sheds CONTEXT_SIZE hard, null when unset", async () => {
138
+ test("contextWindowFromEnv: reads the new name, sheds CONTEXT_SIZE hard, null when unset", async () => {
124
139
  const { contextWindowFromEnv } = await import("./env.ts");
125
140
  assert.equal(contextWindowFromEnv({ PLURNK_PROVIDERS_CONTEXT_WINDOW: "131072" } as NodeJS.ProcessEnv, "openai"), 131072);
126
141
  assert.equal(contextWindowFromEnv({} as NodeJS.ProcessEnv, "openai"), null);
@@ -130,7 +145,7 @@ test("#472 contextWindowFromEnv: reads the new name, sheds CONTEXT_SIZE hard, nu
130
145
  );
131
146
  });
132
147
 
133
- test("scopeEnvToAlias: a caller-supplied knob list scopes CONSUMER vars (service window partition)", async () => {
148
+ test("scopeEnvToAlias: a caller-supplied knob list scopes consumer-owned vars", async () => {
134
149
  const { scopeEnvToAlias } = await import("./env.ts");
135
150
  const SERVICE_KNOBS = ["PLURNK_SERVICE_MAX_TURNS", "PLURNK_SERVICE_LOOP_TIMEOUT", "PLURNK_SERVICE_EXEC_HOLD_MS", "PLURNK_SERVICE_SAFETY"];
136
151
  const env = {
@@ -150,7 +165,7 @@ test("scopeEnvToAlias: a caller-supplied knob list scopes CONSUMER vars (service
150
165
  assert.equal(mixed.PLURNK_PROVIDERS_REASONING, "off");
151
166
  });
152
167
 
153
- test("#36 capture knobs are per-alias scopable: enable on a scraping alias, serving alias stays clean", async () => {
168
+ test("capture knobs are per-alias scopable: enable on a scraping alias, serving alias stays clean", async () => {
154
169
  const { scopeEnvToAlias, dataCaptureFromEnv } = await import("./env.ts");
155
170
  const env = {
156
171
  PLURNK_PROVIDERS_TOP_LOGPROBS_fireslow: "3",
@@ -160,10 +175,10 @@ test("#36 capture knobs are per-alias scopable: enable on a scraping alias, serv
160
175
  assert.deepEqual(dataCaptureFromEnv(scopeEnvToAlias(env, "grokfast"), "x"), { topLogprobs: null, rawBody: false });
161
176
  });
162
177
 
163
- // Reader-declares (#44 ecosystem standard): every knob the code reads appears in
178
+ // Every knob the code reads appears in
164
179
  // the shipped .env.defaults — set (the floor) or commented (documented optional).
165
180
  // The file IS the operator documentation; this keeps it from drifting off the code.
166
- test("#44: every PROVIDERS_KNOBS entry appears in the shipped .env.defaults", async () => {
181
+ test("every PROVIDERS_KNOBS entry appears in the shipped .env.defaults", async () => {
167
182
  const { readFileSync } = await import("node:fs");
168
183
  const { PROVIDERS_KNOBS } = await import("./env.ts");
169
184
  const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
@@ -172,37 +187,37 @@ test("#44: every PROVIDERS_KNOBS entry appears in the shipped .env.defaults", as
172
187
  assert.ok(defaults.includes("PLURNK_PROVIDERS_GBNF="), "GBNF (service-read, providers-namespace) must be declared with its default");
173
188
  });
174
189
 
175
- // #399: the family word is REASONING (industry standard). Old names fail hard
190
+ // The family word is REASONING (industry standard). Old names fail hard
176
191
  // with the migration pointer — never silently coexist with the new floor.
177
- test("#399: still-set old THINKING names fail hard with the rename pointer", () => {
192
+ test("still-set old THINKING names fail hard with the rename pointer", () => {
178
193
  assert.throws(
179
194
  () => reasoningFromEnv({ PLURNK_PROVIDERS_THINKING: "on", PLURNK_PROVIDERS_REASONING: "adaptive" }, "openai"),
180
- /PLURNK_PROVIDERS_THINKING was renamed to PLURNK_PROVIDERS_REASONING \(#399\)/,
195
+ /PLURNK_PROVIDERS_THINKING was renamed to PLURNK_PROVIDERS_REASONING \(provider configuration contract\)/,
181
196
  );
182
197
  assert.throws(
183
198
  () => reasoningFromEnv({ PLURNK_PROVIDERS_THINKING_CAPACITY: "4096", PLURNK_PROVIDERS_REASONING: "adaptive" }, "openai"),
184
- /PLURNK_PROVIDERS_THINKING_CAPACITY was renamed to PLURNK_PROVIDERS_REASONING_BUDGET \(#399\)/,
199
+ /PLURNK_PROVIDERS_THINKING_CAPACITY was renamed to PLURNK_PROVIDERS_REASONING_BUDGET \(provider configuration contract\)/,
185
200
  );
186
201
  });
187
202
 
188
- test("#399: the shipped floor activates reasoning by default (adaptive — owner ruling)", async () => {
203
+ test("the shipped floor activates reasoning by default (adaptive)", async () => {
189
204
  const { readFileSync } = await import("node:fs");
190
205
  const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
191
206
  assert.ok(defaults.includes("PLURNK_PROVIDERS_REASONING=adaptive"), "floor must ship REASONING=adaptive");
192
207
  assert.ok(!defaults.match(/^PLURNK_PROVIDERS_REASONING_BUDGET=/m), "no shipped magnitude — budget is on-mode only");
193
208
  });
194
209
 
195
- test("#567: the shipped DRY floor stays off while retaining the measured alias-safe shape", async () => {
210
+ test("the shipped DRY floor is off and claims no universally safe shape", async () => {
196
211
  const { readFileSync } = await import("node:fs");
197
212
  const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
198
- assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "one-model tuning is not a universal sampler floor");
199
- assert.match(defaults, /^PLURNK_PROVIDERS_DRY_BASE=1\.75$/m);
200
- assert.match(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32$/m, "the measured identifier-safe shape remains available for alias opt-in");
213
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "a fidelity-corrupting sampler cannot be a portable floor");
214
+ assert.doesNotMatch(defaults, /^PLURNK_PROVIDERS_DRY_BASE=/m);
215
+ assert.doesNotMatch(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=/m);
201
216
  });
202
217
 
203
- // -- #507: envelope reserves (owner-ruled migration from PLURNK_SERVICE_*) --
218
+ // -- {§provider-generation-envelope} --
204
219
 
205
- test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
220
+ test("envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
206
221
  const { envelopeFromEnv } = await import("./env.ts");
207
222
  assert.deepEqual(
208
223
  envelopeFromEnv({ PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "4096" } as NodeJS.ProcessEnv, "x"),
@@ -213,7 +228,7 @@ test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fai
213
228
  assert.throws(() => envelopeFromEnv({ PLURNK_PROVIDERS_REASONING_RESERVE: "-5", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%" } as NodeJS.ProcessEnv, "x"), /positive integer token count/);
214
229
  });
215
230
 
216
- test("#507 envelope knobs are per-alias scopable (measured envelope per box)", async () => {
231
+ test("envelope knobs are per-alias scopable (measured envelope per box)", async () => {
217
232
  const { scopeEnvToAlias, envelopeFromEnv } = await import("./env.ts");
218
233
  const env = {
219
234
  PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
package/src/env.ts CHANGED
@@ -43,7 +43,7 @@ export const promptCacheKeyFromEnv = (env: NodeJS.ProcessEnv, label: string): bo
43
43
  return value === "1";
44
44
  };
45
45
 
46
- // Old-name shed (the #399/#472 lexicon renames): a still-set retired knob fails
46
+ // {§provider-configuration} A still-set retired knob fails
47
47
  // hard pointing at its successor — never silently coexists with the new floor.
48
48
  // The retired names appear ONLY as this function's ARGUMENTS at the call sites
49
49
  // (each lexicon-allow), never as a live identifier.
@@ -52,7 +52,8 @@ const shedRenamed = (env: NodeJS.ProcessEnv, oldName: string, newName: string, l
52
52
  if (stale !== undefined && stale.length > 0) throw new Error(`${label} provider: ${oldName} was renamed to ${newName} (${ref}); update the env`);
53
53
  };
54
54
 
55
- // Data-capture knobs (#36), read identically by every provider (standard AND
55
+ // {§provider-evidence} Data-capture knobs are read identically by every provider
56
+ // (standard AND
56
57
  // plugin) so the opt-in surface is one source of truth. Both OFF by default —
57
58
  // the flag is the isolation, so serving turns request and carry nothing.
58
59
  // PLURNK_PROVIDERS_TOP_LOGPROBS "off" or a non-negative int = the OpenAI
@@ -71,17 +72,29 @@ export const dataCaptureFromEnv = (env: NodeJS.ProcessEnv, label: string): { top
71
72
  };
72
73
  };
73
74
 
74
- // The context-window pin (SPEC §4) under its OpenAI-lexicon name (#472) — the
75
- // industry term is "context window" (OpenAI/Anthropic docs, models.dev
76
- // contextWindow); CONTEXT_SIZE was home-grown. One reader for base AND
77
- // plugins, so the shed fires everywhere the knob is honored.
75
+ // {§model-fact-resolution} — one context-window reader for every provider path;
76
+ // the retired CONTEXT_SIZE spelling fails visibly at the same boundary.
78
77
  export const contextWindowFromEnv = (env: NodeJS.ProcessEnv, label: string): number | null => {
79
- shedRenamed(env, "PLURNK_PROVIDERS_CONTEXT_SIZE", "PLURNK_PROVIDERS_CONTEXT_WINDOW", label, "the industry term, #472"); // lexicon-allow
78
+ shedRenamed(env, "PLURNK_PROVIDERS_CONTEXT_SIZE", "PLURNK_PROVIDERS_CONTEXT_WINDOW", label, "{§model-fact-resolution}"); // lexicon-allow
80
79
  return parseOptionalInt(env.PLURNK_PROVIDERS_CONTEXT_WINDOW, "PLURNK_PROVIDERS_CONTEXT_WINDOW", label);
81
80
  };
82
81
 
82
+ // {§model-fact-resolution} — an operator value caps known model physics and
83
+ // declares the window only when no natural value is known.
84
+ export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number): number;
85
+ export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number | null): number | null;
86
+ export function effectiveContextWindow(operatorCap: number | null, naturalWindow: number | null): number | null {
87
+ return operatorCap === null
88
+ ? naturalWindow
89
+ : naturalWindow === null
90
+ ? operatorCap
91
+ : Math.min(operatorCap, naturalWindow);
92
+ }
93
+
83
94
  export type ProviderTokenRates = { input: number; cached: number; output: number };
84
95
 
96
+ // {§model-fact-resolution} — any explicit rate opts into one complete operator
97
+ // estimate; cached input alone may default to the explicit input rate.
85
98
  export const tokenRatesFromEnv = (env: NodeJS.ProcessEnv, label: string): ProviderTokenRates | null => {
86
99
  const inputName = "PLURNK_PROVIDERS_INPUT_USD_PER_MILLION";
87
100
  const cachedName = "PLURNK_PROVIDERS_CACHE_READ_USD_PER_MILLION";
@@ -99,13 +112,11 @@ export const tokenRatesFromEnv = (env: NodeJS.ProcessEnv, label: string): Provid
99
112
  };
100
113
  };
101
114
 
102
- // The generation-envelope reserves (#507, owner-ruled migration): how much of a
115
+ // {§provider-generation-envelope} How much of a
103
116
  // DETECTED context window is reserved for reasoning and for completion — the
104
117
  // remainder (minus the consumer's own packing-safety margin) is the prompt
105
118
  // budget. Provider-owned: the window is a provider fact and these are amounts OF
106
- // it; the former PLURNK_SERVICE_{CONTEXT_WINDOW,REASONING,COMPLETION} knobs were
107
- // provider quantities wearing a service prefix (born as #352 packet parameters
108
- // before this surface existed). Each knob accepts a percentage of the window
119
+ // it. Each knob accepts a percentage of the window
109
120
  // ("10%") or an absolute token count ("4096"); the floor ships percentages so
110
121
  // every window-advertising endpoint (llama-server n_ctx, the plurnk.ai router,
111
122
  // a cataloged cloud model) arrives at sane defaults with ZERO operator tuning.
@@ -151,21 +162,35 @@ export const resolveEnvelopeFromEnv = (env: NodeJS.ProcessEnv, window: number |
151
162
  };
152
163
  };
153
164
 
154
- // The side-channel reasoning knobs (SPEC §4, #32/#33) — ACTIVATION and BUDGET
165
+ // {§provider-configuration} The side-channel reasoning knobs — activation and budget
155
166
  // are separate vars, so a numeric budget can never silently flip wire flags:
156
167
  // PLURNK_PROVIDERS_REASONING off | adaptive | on (REQUIRED, fail-hard)
157
168
  // PLURNK_PROVIDERS_REASONING_BUDGET positive int, REQUIRED iff REASONING=on —
158
- // the magnitude for tier/budget mapping. On llama.cpp the ENFORCEMENT is the
159
- // box's --reasoning-budget launch flag (per-request numerics are ignored):
160
- // env budget and launch flag are the same number, changed together.
169
+ // the magnitude for tier/budget mapping. On llama-server it is the explicit
170
+ // request-scoped allowance and cannot exceed the physical reasoning reserve.
161
171
  // The provider maps intent to the backend's mechanism; the consumer states
162
- // intent, never mechanism. (In-DSL PLAN reasoning is a grammar concern.)
172
+ // intent, never mechanism. PLAN is a separate public intended-goals record.
163
173
  export type ReasoningMode = "off" | "adaptive" | "on";
164
174
  export type Reasoning = { mode: ReasoningMode; budget: number | null };
165
175
 
176
+ export type ReasoningResponseStyle = "verbatim" | "think-tags";
177
+
178
+ export const reasoningResponseStyleFromEnv = (
179
+ env: NodeJS.ProcessEnv,
180
+ label: string,
181
+ ): ReasoningResponseStyle => {
182
+ const name = "PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE";
183
+ const raw = env[name];
184
+ if (raw === undefined || raw.length === 0) return "verbatim";
185
+ if (raw !== "verbatim" && raw !== "think-tags") {
186
+ throw new Error(`${label} provider: ${name} must be "verbatim" or "think-tags" (got "${raw}")`);
187
+ }
188
+ return raw;
189
+ };
190
+
166
191
  export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoning => {
167
- shedRenamed(env, "PLURNK_PROVIDERS_THINKING", "PLURNK_PROVIDERS_REASONING", label, "#399"); // lexicon-allow
168
- shedRenamed(env, "PLURNK_PROVIDERS_THINKING_CAPACITY", "PLURNK_PROVIDERS_REASONING_BUDGET", label, "#399"); // lexicon-allow
192
+ shedRenamed(env, "PLURNK_PROVIDERS_THINKING", "PLURNK_PROVIDERS_REASONING", label, "provider configuration contract"); // lexicon-allow
193
+ shedRenamed(env, "PLURNK_PROVIDERS_THINKING_CAPACITY", "PLURNK_PROVIDERS_REASONING_BUDGET", label, "provider configuration contract"); // lexicon-allow
169
194
  const name = "PLURNK_PROVIDERS_REASONING";
170
195
  const raw = env[name];
171
196
  if (raw === undefined || raw.length === 0) throw new Error(`${label} provider: ${name} must be set (off | adaptive | on)`);
@@ -189,6 +214,7 @@ export const reasoningFromEnv = (env: NodeJS.ProcessEnv, label: string): Reasoni
189
214
  export const PROVIDERS_KNOBS = Object.freeze([
190
215
  "PLURNK_PROVIDERS_REASONING_RESERVE",
191
216
  "PLURNK_PROVIDERS_COMPLETION_RESERVE",
217
+ "PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE",
192
218
  "PLURNK_PROVIDERS_REASONING_BUDGET",
193
219
  "PLURNK_PROVIDERS_REASONING",
194
220
  "PLURNK_PROVIDERS_CONTEXT_WINDOW",
@@ -196,6 +222,7 @@ export const PROVIDERS_KNOBS = Object.freeze([
196
222
  "PLURNK_PROVIDERS_CACHE_READ_USD_PER_MILLION",
197
223
  "PLURNK_PROVIDERS_OUTPUT_USD_PER_MILLION",
198
224
  "PLURNK_PROVIDERS_RETRY_ATTEMPTS",
225
+ "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT",
199
226
  "PLURNK_PROVIDERS_FETCH_TIMEOUT",
200
227
  "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
201
228
  "PLURNK_PROVIDERS_LLAMA_SERVER",
@@ -0,0 +1,148 @@
1
+ import test from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { APICallError, RetryError } from "ai";
4
+ import { ProviderError, classifyProviderError, toProviderError } from "./errors.ts";
5
+ import type { ProviderAttempt } from "./types.ts";
6
+ import { providerSource } from "./notices.ts";
7
+
8
+ const apiError = (statusCode: number, responseBody = "body") => new APICallError({
9
+ message: `request failed (${statusCode})`,
10
+ url: "https://example.test/v1/chat/completions",
11
+ requestBodyValues: {},
12
+ statusCode,
13
+ responseBody,
14
+ });
15
+
16
+ const SOURCE_PATTERN = /^[a-z]+(:[a-z][a-z0-9-]*)?$/;
17
+
18
+ test("providerSource produces a schema-valid colon-namespaced source", () => {
19
+ assert.equal(providerSource("openai"), "provider:openai");
20
+ assert.match(providerSource("openrouter"), SOURCE_PATTERN);
21
+ assert.equal(providerSource("@scope/custom_provider"), "provider:scope-custom-provider");
22
+ assert.equal(providerSource("2fast"), "provider:p-2fast");
23
+ assert.throws(() => providerSource(""), /must name a provider/);
24
+ });
25
+
26
+ test("classifyProviderError maps HTTP status to kind", () => {
27
+ const k = (status: number) => classifyProviderError(apiError(status)).kind;
28
+ assert.equal(k(401), "unauthorized");
29
+ assert.equal(k(403), "unauthorized");
30
+ assert.equal(k(402), "quota_exceeded");
31
+ assert.equal(k(429), "rate_limit");
32
+ assert.equal(k(500), "network_failure");
33
+ assert.equal(k(503), "network_failure");
34
+ assert.equal(k(400), "invalid_response");
35
+ assert.equal(k(404), "invalid_response");
36
+ });
37
+
38
+ test("classifyProviderError: a 422 flagged grammar_invalid is distinct; other 422s are invalid responses", () => {
39
+ const rejected = apiError(422, JSON.stringify({ error: { type: "grammar_invalid", message: "non-conforming emission rejected: ..." } }));
40
+ assert.equal(classifyProviderError(rejected).kind, "grammar_invalid");
41
+ assert.equal(classifyProviderError(apiError(422, JSON.stringify({ error: { type: "invalid_request_error" } }))).kind, "invalid_response");
42
+ assert.equal(classifyProviderError(apiError(422, "<html>Bad</html>")).kind, "invalid_response");
43
+ });
44
+
45
+ test("classifyProviderError treats non-HTTP errors as network_failure", () => {
46
+ assert.equal(classifyProviderError(new TypeError("fetch failed")).kind, "network_failure");
47
+ const timeout = Object.assign(new Error("timed out"), { name: "TimeoutError" });
48
+ assert.equal(classifyProviderError(timeout).kind, "network_failure");
49
+ });
50
+
51
+ test("ProviderError carries a validated RFC 9457 Problem Details object", () => {
52
+ const e = new ProviderError("provider:openai", "rate_limit", "OpenAI 429 - slow down", { status: 429 });
53
+ assert.deepEqual(e.problem, {
54
+ type: "https://problems.plurnk.dev/provider/openai/rate-limit",
55
+ title: "Rate limit",
56
+ status: 429,
57
+ detail: "OpenAI 429 - slow down",
58
+ providerKind: "rate_limit",
59
+ stage: "provider-request",
60
+ retryable: true,
61
+ });
62
+ assert.match(e.source, SOURCE_PATTERN);
63
+ assert.ok(e instanceof Error);
64
+ assert.equal(e.status, 429);
65
+ });
66
+
67
+ test("ProviderError marks failures that require changed external state as non-retryable", () => {
68
+ const unauthorized = new ProviderError("provider:openai", "unauthorized", "Invalid API key.");
69
+ assert.equal(unauthorized.problem.retryable, false);
70
+ assert.equal(unauthorized.problem.stage, "provider-request");
71
+ });
72
+
73
+ test("#161: ProviderError carries resource-interrupted attempt evidence outside Problem Details", () => {
74
+ const attempt = {
75
+ assistant: {
76
+ content: "partial",
77
+ reasoning: null,
78
+ usage: { prompt: 3, completion: 1, reasoning: 0, cached: 0, total: 4 },
79
+ finishReason: "resource_interrupted",
80
+ model: "served-model",
81
+ },
82
+ assistantRaw: { rawFinishReason: "insufficient_system_resource" },
83
+ } as ProviderAttempt;
84
+ const error = new ProviderError(
85
+ "provider:deepseek",
86
+ "resource_interrupted",
87
+ "The provider interrupted generation because inference resources were unavailable.",
88
+ {
89
+ attempt,
90
+ extensions: {
91
+ stage: "provider-response",
92
+ finishReason: "resource_interrupted",
93
+ rawFinishReason: "insufficient_system_resource",
94
+ },
95
+ },
96
+ );
97
+
98
+ assert.equal(error.attempt, attempt);
99
+ assert.deepEqual(error.problem, {
100
+ type: "https://problems.plurnk.dev/provider/deepseek/resource-interrupted",
101
+ title: "Resource interrupted",
102
+ status: 503,
103
+ detail: "The provider interrupted generation because inference resources were unavailable.",
104
+ providerKind: "resource_interrupted",
105
+ stage: "provider-response",
106
+ retryable: false,
107
+ finishReason: "resource_interrupted",
108
+ rawFinishReason: "insufficient_system_resource",
109
+ });
110
+ });
111
+
112
+ test("toProviderError classifies and tags an HTTP error with the source + status", () => {
113
+ const cause = apiError(401, "no key");
114
+ const pe = toProviderError(cause, "provider:groq");
115
+ assert.equal(pe.kind, "unauthorized");
116
+ assert.equal(pe.source, "provider:groq");
117
+ assert.equal(pe.status, 401);
118
+ assert.equal(pe.cause, cause);
119
+ });
120
+
121
+ test("toProviderError passes an existing ProviderError through unchanged", () => {
122
+ const original = new ProviderError("provider:xai", "rate_limit", "429");
123
+ assert.equal(toProviderError(original, "provider:other"), original);
124
+ });
125
+
126
+ test("provider diagnostics are bounded without losing structured failure facts", () => {
127
+ const cause = apiError(502);
128
+ Object.defineProperty(cause, "message", { value: "abcdefghij" });
129
+ const error = toProviderError(cause, "provider:test", 4);
130
+ assert.equal(error.problem.detail, "abcd...");
131
+ assert.equal(error.problem.status, 502);
132
+ assert.equal(error.problem.providerKind, "network_failure");
133
+ assert.equal(error.cause, cause);
134
+ });
135
+
136
+ test("retry exhaustion is explicit and does not recommend another automatic replay", () => {
137
+ const failures = [apiError(429), apiError(429), apiError(429)];
138
+ const cause = new RetryError({
139
+ message: "Failed after 3 attempts.",
140
+ reason: "maxRetriesExceeded",
141
+ errors: failures,
142
+ });
143
+ const error = toProviderError(cause, "provider:test");
144
+ assert.equal(error.problem.retryable, false);
145
+ assert.equal(error.problem.attempts, 3);
146
+ assert.equal(error.problem.retryExhausted, true);
147
+ assert.equal(error.cause, cause);
148
+ });
package/src/errors.ts ADDED
@@ -0,0 +1,207 @@
1
+ import { Problems, type ProblemDetails } from "@plurnk/plurnk-contracts";
2
+ import { APICallError, RetryError } from "ai";
3
+ import { providerSource } from "./notices.ts";
4
+ import type { ProviderAttempt } from "./types.ts";
5
+
6
+ export type ProviderErrorKind =
7
+ | "rate_limit"
8
+ | "network_failure"
9
+ | "model_refused"
10
+ | "invalid_response"
11
+ | "unauthorized"
12
+ | "quota_exceeded"
13
+ | "grammar_invalid"
14
+ | "resource_interrupted";
15
+
16
+ export interface ClassifiedProviderError {
17
+ kind: ProviderErrorKind;
18
+ message: string;
19
+ retryable?: boolean;
20
+ attempts?: number;
21
+ retryExhausted?: boolean;
22
+ }
23
+
24
+ const defaultStatus = (kind: ProviderErrorKind): number => {
25
+ switch (kind) {
26
+ case "unauthorized": return 401;
27
+ case "quota_exceeded": return 402;
28
+ case "rate_limit": return 429;
29
+ case "model_refused":
30
+ case "grammar_invalid": return 422;
31
+ case "invalid_response": return 502;
32
+ case "network_failure":
33
+ case "resource_interrupted": return 503;
34
+ }
35
+ };
36
+
37
+ const retryable = (kind: ProviderErrorKind): boolean => {
38
+ switch (kind) {
39
+ case "rate_limit":
40
+ case "network_failure":
41
+ return true;
42
+ case "invalid_response":
43
+ case "grammar_invalid":
44
+ case "resource_interrupted":
45
+ case "model_refused":
46
+ case "unauthorized":
47
+ case "quota_exceeded":
48
+ return false;
49
+ }
50
+ };
51
+
52
+ const buildProblem = (
53
+ source: string,
54
+ kind: ProviderErrorKind,
55
+ message: string,
56
+ status: number,
57
+ extensions: Readonly<Record<string, unknown>>,
58
+ retryableOverride: boolean | undefined,
59
+ ): ProblemDetails => {
60
+ const code: Record<ProviderErrorKind, string> = {
61
+ rate_limit: "rate-limit",
62
+ network_failure: "network-failure",
63
+ model_refused: "model-refused",
64
+ invalid_response: "invalid-response",
65
+ unauthorized: "unauthorized",
66
+ quota_exceeded: "quota-exceeded",
67
+ grammar_invalid: "grammar-invalid",
68
+ resource_interrupted: "resource-interrupted",
69
+ };
70
+ return Problems.create(source, code[kind], status, message, {
71
+ providerKind: kind,
72
+ stage: "provider-request",
73
+ retryable: retryableOverride ?? retryable(kind),
74
+ ...extensions,
75
+ });
76
+ };
77
+
78
+ // A provider operation failed before a completed exchange existed. An
79
+ // interrupted response may still carry attempt evidence for its consumer.
80
+ // The standardized Problem is the public failure contract; kind remains the
81
+ // provider pool's routing discriminator and is repeated as a Problem extension.
82
+ export class ProviderError extends Error {
83
+ readonly source: string;
84
+ readonly kind: ProviderErrorKind;
85
+ readonly problem: ProblemDetails;
86
+ readonly attempt?: ProviderAttempt;
87
+
88
+ constructor(
89
+ source: string,
90
+ kind: ProviderErrorKind,
91
+ message: string,
92
+ options: {
93
+ status?: number | null;
94
+ cause?: unknown;
95
+ retryable?: boolean;
96
+ extensions?: Readonly<Record<string, unknown>>;
97
+ attempt?: ProviderAttempt;
98
+ } = {},
99
+ ) {
100
+ super(message, options.cause !== undefined ? { cause: options.cause } : undefined);
101
+ this.name = "ProviderError";
102
+ this.source = providerSource(source);
103
+ this.kind = kind;
104
+ this.attempt = options.attempt;
105
+ const status = options.status !== null && options.status !== undefined
106
+ && Number.isInteger(options.status) && options.status >= 400 && options.status <= 599
107
+ ? options.status
108
+ : defaultStatus(kind);
109
+ this.problem = buildProblem(
110
+ this.source,
111
+ kind,
112
+ message,
113
+ status,
114
+ options.extensions ?? {},
115
+ options.retryable,
116
+ );
117
+ }
118
+
119
+ get status(): number {
120
+ return this.problem.status;
121
+ }
122
+ }
123
+
124
+ const wireErrorType = (body: string): string | null => {
125
+ try {
126
+ const { error } = JSON.parse(body) as { error?: { type?: unknown } };
127
+ return typeof error?.type === "string" ? error.type : null;
128
+ } catch {
129
+ return null;
130
+ }
131
+ };
132
+
133
+ const preview = (value: unknown, limit: number | undefined): string => {
134
+ const text = value instanceof Error ? value.message : String(value);
135
+ return limit !== undefined && text.length > limit
136
+ ? `${text.slice(0, limit)}...`
137
+ : text;
138
+ };
139
+
140
+ export const classifyProviderError = (
141
+ err: unknown,
142
+ detailLimit?: number,
143
+ ): ClassifiedProviderError => {
144
+ if (RetryError.isInstance(err)) {
145
+ return {
146
+ ...classifyProviderError(err.lastError, detailLimit),
147
+ retryable: false,
148
+ attempts: err.errors.length,
149
+ retryExhausted: err.reason === "maxRetriesExceeded",
150
+ };
151
+ }
152
+ if (APICallError.isInstance(err)) {
153
+ const status = err.statusCode ?? 0;
154
+ const message = err.message.trim().length > 0
155
+ ? preview(err.message, detailLimit)
156
+ : "The provider request failed without a diagnostic message.";
157
+ const body = err.responseBody ?? "";
158
+ if (status === 401 || status === 403) return { kind: "unauthorized", message };
159
+ if (status === 402) return { kind: "quota_exceeded", message };
160
+ if (status === 429) return { kind: "rate_limit", message };
161
+ if (status >= 500) return { kind: "network_failure", message };
162
+ if (status === 422 && wireErrorType(body) === "grammar_invalid") {
163
+ return { kind: "grammar_invalid", message };
164
+ }
165
+ return { kind: "invalid_response", message };
166
+ }
167
+ const wire = err as { message?: unknown; type?: unknown };
168
+ if (wire?.type === "grammar_invalid") {
169
+ return {
170
+ kind: "grammar_invalid",
171
+ message: typeof wire.message === "string"
172
+ ? preview(wire.message, detailLimit)
173
+ : "The provider rejected the response grammar.",
174
+ };
175
+ }
176
+ const error = err as { message?: string };
177
+ return {
178
+ kind: "network_failure",
179
+ message: preview(
180
+ (error?.message ?? String(err)) || "The provider request failed.",
181
+ detailLimit,
182
+ ),
183
+ };
184
+ };
185
+
186
+ export const toProviderError = (
187
+ err: unknown,
188
+ source: string,
189
+ detailLimit?: number,
190
+ ): ProviderError => {
191
+ if (err instanceof ProviderError) return err;
192
+ const underlying = RetryError.isInstance(err) ? err.lastError : err;
193
+ const classified = classifyProviderError(err, detailLimit);
194
+ const { kind, message } = classified;
195
+ const status = APICallError.isInstance(underlying) ? underlying.statusCode ?? null : null;
196
+ return new ProviderError(source, kind, message, {
197
+ status,
198
+ cause: err,
199
+ retryable: classified.retryable,
200
+ extensions: {
201
+ ...(classified.attempts === undefined ? {} : { attempts: classified.attempts }),
202
+ ...(classified.retryExhausted === undefined
203
+ ? {}
204
+ : { retryExhausted: classified.retryExhausted }),
205
+ },
206
+ });
207
+ };