@plurnk/plurnk-providers 1.16.4 → 1.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/.env.defaults +87 -200
  2. package/README.md +2 -1
  3. package/SPEC.md +73 -66
  4. package/dist/AiSdkProvider.d.ts +0 -1
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +25 -55
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/AiSdkRequestBody.d.ts +0 -1
  9. package/dist/AiSdkRequestBody.d.ts.map +1 -1
  10. package/dist/AiSdkRequestBody.js +1 -17
  11. package/dist/AiSdkRequestBody.js.map +1 -1
  12. package/dist/LeadingReasoning.d.ts +16 -0
  13. package/dist/LeadingReasoning.d.ts.map +1 -0
  14. package/dist/LeadingReasoning.js +81 -0
  15. package/dist/LeadingReasoning.js.map +1 -0
  16. package/dist/aiSdkTransport.d.ts +7 -0
  17. package/dist/aiSdkTransport.d.ts.map +1 -1
  18. package/dist/aiSdkTransport.js +19 -5
  19. package/dist/aiSdkTransport.js.map +1 -1
  20. package/dist/catalogProvider.d.ts +3 -1
  21. package/dist/catalogProvider.d.ts.map +1 -1
  22. package/dist/catalogProvider.js +24 -18
  23. package/dist/catalogProvider.js.map +1 -1
  24. package/dist/compatibleProvider.d.ts.map +1 -1
  25. package/dist/compatibleProvider.js +0 -3
  26. package/dist/compatibleProvider.js.map +1 -1
  27. package/dist/env.d.ts.map +1 -1
  28. package/dist/env.js +0 -1
  29. package/dist/env.js.map +1 -1
  30. package/dist/errors.d.ts +6 -0
  31. package/dist/errors.d.ts.map +1 -1
  32. package/dist/errors.js +46 -15
  33. package/dist/errors.js.map +1 -1
  34. package/dist/index.d.ts +2 -2
  35. package/dist/index.d.ts.map +1 -1
  36. package/dist/index.js +2 -1
  37. package/dist/index.js.map +1 -1
  38. package/dist/providerError.d.ts +1 -1
  39. package/dist/providerError.d.ts.map +1 -1
  40. package/dist/providerError.js +3 -0
  41. package/dist/providerError.js.map +1 -1
  42. package/dist/sdkModels.d.ts +1 -13
  43. package/dist/sdkModels.d.ts.map +1 -1
  44. package/dist/sdkModels.js +2 -94
  45. package/dist/sdkModels.js.map +1 -1
  46. package/dist/types.d.ts.map +1 -1
  47. package/docs/models.md +90 -0
  48. package/package.json +6 -6
  49. package/src/AiSdkProvider.test.ts +99 -38
  50. package/src/AiSdkProvider.ts +24 -78
  51. package/src/AiSdkRequestBody.ts +1 -16
  52. package/src/LeadingReasoning.test.ts +34 -0
  53. package/src/LeadingReasoning.ts +87 -0
  54. package/src/ProviderRegistry.test.ts +19 -1
  55. package/src/aiSdkTransport.test.ts +56 -0
  56. package/src/aiSdkTransport.ts +31 -5
  57. package/src/boundaries.test.ts +1 -0
  58. package/src/catalogProvider.test.ts +110 -20
  59. package/src/catalogProvider.ts +40 -18
  60. package/src/compatibleProvider.ts +0 -3
  61. package/src/env.test.ts +1 -0
  62. package/src/env.ts +1 -2
  63. package/src/errors.test.ts +58 -7
  64. package/src/errors.ts +55 -18
  65. package/src/index.ts +2 -2
  66. package/src/providerError.ts +4 -0
  67. package/src/sdkModels.test.ts +1 -42
  68. package/src/sdkModels.ts +1 -123
  69. package/src/types.ts +8 -10
@@ -89,6 +89,25 @@ const wireUsageOf = (
89
89
  return undefined;
90
90
  };
91
91
 
92
+ // {§provider-usage-refusal} — the provider's bookkeeping is not the exchange. When the reported
93
+ // counters cannot be normalized (a reasoning detail exceeding its output aggregate, a total that
94
+ // contradicts its parts), the response stands, usage is unknown (never invented, clamped, or zero),
95
+ // and the counters as reported ride beside the refusal so forensics can see what the wire said.
96
+ export type UsageRefusal = { readonly reason: string; readonly usage: unknown };
97
+
98
+ const settledUsage = (
99
+ values: readonly unknown[],
100
+ sdkUsage: LanguageModelUsage | undefined,
101
+ ): { usage?: ProviderUsage; usageRefusal?: UsageRefusal } => {
102
+ try {
103
+ const usage = wireUsageOf(values) ?? (sdkUsage === undefined ? undefined : usageOf(sdkUsage));
104
+ return usage === undefined ? {} : { usage };
105
+ } catch (cause) {
106
+ if (!(cause instanceof TypeError)) throw cause;
107
+ return { usageRefusal: { reason: cause.message, usage: wireUsageEvidenceOf(values) ?? sdkUsage } };
108
+ }
109
+ };
110
+
92
111
  const wireUsageEvidenceOf = (values: readonly unknown[]): unknown => {
93
112
  for (let index = values.length - 1; index >= 0; index -= 1) {
94
113
  const record = recordOf(values[index]);
@@ -169,6 +188,7 @@ export type AiSdkTransportRequest = {
169
188
  streaming: boolean;
170
189
  captureRawBody: boolean;
171
190
  observeReasoning?: ProviderReasoningObserver;
191
+ observeText?: (delta: string) => void;
172
192
  };
173
193
 
174
194
  export type AiSdkTransportResponse = {
@@ -179,6 +199,7 @@ export type AiSdkTransportResponse = {
179
199
  finishReason: ProviderAttemptFinishReason;
180
200
  rawFinishReason?: string;
181
201
  usage?: ProviderUsage;
202
+ usageRefusal?: UsageRefusal;
182
203
  metadata: Record<string, unknown>;
183
204
  reasoningEncrypted: Array<{
184
205
  id: string | null;
@@ -437,7 +458,7 @@ const executeModelOnce = async (
437
458
  reasoningProjected: evidence.reasoningProjected,
438
459
  finishReason: finishReasonOf(rawFinishReason),
439
460
  ...(rawFinishReason === undefined ? {} : { rawFinishReason }),
440
- usage: wireUsageOf(values) ?? usageOf(result.usage),
461
+ ...settledUsage(values, result.usage),
441
462
  metadata: metadataOf(values),
442
463
  reasoningEncrypted: evidence.reasoningEncrypted,
443
464
  logprobs: evidence.logprobs,
@@ -472,7 +493,10 @@ const executeModelOnce = async (
472
493
  try {
473
494
  for await (const part of result.fullStream) {
474
495
  if (part.type === "raw") rawChunks.push(part.rawValue);
475
- if (part.type === "text-delta" && part.text.length > 0) outputObserved = true;
496
+ if (part.type === "text-delta" && part.text.length > 0) {
497
+ outputObserved = true;
498
+ request.observeText?.(part.text);
499
+ }
476
500
  if (part.type === "reasoning-delta" && part.text.length > 0) {
477
501
  outputObserved = true;
478
502
  request.observeReasoning?.(part.text);
@@ -504,7 +528,7 @@ const executeModelOnce = async (
504
528
  reasoningProjected: evidence.reasoningProjected,
505
529
  finishReason: finishReasonOf(rawFinishReason),
506
530
  ...(rawFinishReason === undefined ? {} : { rawFinishReason }),
507
- usage: wireUsageOf(rawChunks) ?? usageOf(await result.usage),
531
+ ...settledUsage(rawChunks, await result.usage),
508
532
  metadata: metadataOf(rawChunks),
509
533
  reasoningEncrypted: evidence.reasoningEncrypted,
510
534
  logprobs: evidence.logprobs,
@@ -563,6 +587,7 @@ export const executeOpenAICompatible = async (
563
587
  streaming: request.streaming,
564
588
  captureRawBody: request.captureRawBody,
565
589
  ...(request.observeReasoning === undefined ? {} : { observeReasoning: request.observeReasoning }),
590
+ ...(request.observeText === undefined ? {} : { observeText: request.observeText }),
566
591
  });
567
592
  };
568
593
 
@@ -577,6 +602,7 @@ const responseBodyValues = (error: APICallError): readonly unknown[] => {
577
602
 
578
603
  export type AiSdkTransportFailureEvidence = {
579
604
  readonly usage?: ProviderUsage;
605
+ readonly usageRefusal?: UsageRefusal;
580
606
  readonly chargeEvidence: ProviderChargeEvidence;
581
607
  readonly status?: number;
582
608
  };
@@ -587,7 +613,7 @@ export const transportFailureEvidence = (
587
613
  const values = typeof error === "object" && error !== null
588
614
  ? streamFailureValues.get(error) ?? (APICallError.isInstance(error) ? responseBodyValues(error) : [])
589
615
  : [];
590
- const usage = wireUsageOf(values);
616
+ const settled = settledUsage(values, undefined);
591
617
  const usageEvidence = wireUsageEvidenceOf(values);
592
618
  const charge = wireChargeEvidenceOf(values);
593
619
  const wireStatus = values
@@ -600,7 +626,7 @@ export const transportFailureEvidence = (
600
626
  ? wireStatus as number
601
627
  : undefined;
602
628
  return {
603
- ...(usage === undefined ? {} : { usage }),
629
+ ...settled,
604
630
  chargeEvidence: {
605
631
  ...(charge === undefined ? {} : { charge }),
606
632
  ...(usageEvidence === undefined ? {} : { usage: usageEvidence }),
@@ -52,6 +52,7 @@ test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery",
52
52
  "cost.ts",
53
53
  "env.ts",
54
54
  "errors.ts",
55
+ "LeadingReasoning.ts",
55
56
  "notices.ts",
56
57
  "openai.ts",
57
58
  "promptTokens.ts",
@@ -2,10 +2,27 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { once } from "node:events";
4
4
  import { createServer } from "node:http";
5
- import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
5
+ import { catalogProviderFromEnv, catalogReasoningPolicies, providerFromSdkModel } from "./catalogProvider.ts";
6
+ import { lookupProvider, resolveModel } from "@plurnk/plurnk-models";
7
+ import { calculateCostUsdDecimal } from "./usage.ts";
8
+
9
+ // A rate literal breaks at every catalog refresh (the 1.17.0 stamp moved DeepSeek's rates); the
10
+ // claim under test is that the estimate is the catalog's rates applied to the reported usage.
11
+ const catalogRatesOf = (provider: string, model: string) => {
12
+ const cost = resolveModel(provider, model)?.info.cost;
13
+ if (cost === undefined) throw new Error(`${provider}/${model} has no catalog rates`);
14
+ return {
15
+ input: cost.inputPer1M,
16
+ output: cost.outputPer1M,
17
+ ...(cost.reasoningPer1M === undefined ? {} : { reasoning: cost.reasoningPer1M }),
18
+ ...(cost.cacheReadPer1M === undefined ? {} : { cacheRead: cost.cacheReadPer1M }),
19
+ ...(cost.cacheWritePer1M === undefined ? {} : { cacheWrite: cost.cacheWritePer1M }),
20
+ };
21
+ };
6
22
  import { withProviderDefaults } from "./defaults.ts";
7
23
  import type { LanguageModel } from "ai";
8
24
  import { resetEmittedWarnings } from "./warnings.ts";
25
+ import { UnsupportedReasoningPolicyError } from "./types.ts";
9
26
 
10
27
  const env = {
11
28
  OPENAI_API_KEY: "test-key",
@@ -104,6 +121,73 @@ test("provider adapters advertise only reasoning policies they can preserve", ()
104
121
  assert.deepEqual(gemini?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Gemini 3's mandatory minimum is not advertised as off");
105
122
  });
106
123
 
124
+ test("{§provider-reasoning-policy}: catalog discovery and construction agree on native and compatible routes", () => {
125
+ const fetch = mock.method(globalThis, "fetch", () => {
126
+ throw new Error("capability discovery and construction require no provider request");
127
+ });
128
+ const configured = withProviderDefaults({
129
+ ...env,
130
+ DEEPSEEK_API_KEY: "test-key",
131
+ GEMINI_API_KEY: "test-key",
132
+ MISTRAL_API_KEY: "test-key",
133
+ CLOUDFLARE_ACCOUNT_ID: "test-account",
134
+ CLOUDFLARE_API_KEY: "test-key",
135
+ PLURNK_PROVIDERS_REASONING: "adaptive",
136
+ PLURNK_PROVIDERS_PROVIDER_CLOUDFLARE_WORKERS_AI_REASONING_STYLE: "effort_required",
137
+ });
138
+ for (const [name, model] of [
139
+ ["openai", "gpt-4.1-mini"],
140
+ ["google", "gemini-3.7-flash"],
141
+ ["mistral", "mistral-small-latest"],
142
+ ["deepseek", "deepseek-v4-flash"],
143
+ ["cloudflare-workers-ai", "@cf/qwen/qwen3.8-27b"],
144
+ ["cloudflare-workers-ai", "@cf/ibm-granite/granite-4.0-h-micro"],
145
+ ["cloudflare-workers-ai", "@cf/zai-org/glm-5.3-flash"],
146
+ ]) {
147
+ const providerInfo = lookupProvider(name);
148
+ const info = resolveModel(name, model)?.info;
149
+ assert.ok(providerInfo);
150
+ assert.ok(info);
151
+ for (const declaration of ["", "medium,high,max"]) {
152
+ const key = `PLURNK_PROVIDERS_PROVIDER_${name.replaceAll(/[^a-zA-Z0-9]/g, "_").toUpperCase()}_REASONING_EFFORTS`;
153
+ const localEnv = { ...configured, [key]: declaration };
154
+ const provider = catalogProviderFromEnv(name, localEnv, model);
155
+ assert.ok(provider);
156
+ assert.deepEqual(
157
+ catalogReasoningPolicies(providerInfo, info, localEnv),
158
+ provider.supportedReasoningPolicies,
159
+ `${name}/${model} with declared efforts ${JSON.stringify(declaration)}`,
160
+ );
161
+ }
162
+ }
163
+ assert.equal(fetch.mock.callCount(), 0);
164
+ });
165
+
166
+ test("{§provider-reasoning-policy}: discovery never offers a fixed effort the installed transport rejects", () => {
167
+ const info = resolveModel("deepseek", "deepseek-v4-flash")?.info;
168
+ assert.ok(info);
169
+ const declared = withProviderDefaults({
170
+ ...env,
171
+ PLURNK_PROVIDERS_REASONING: "adaptive",
172
+ PLURNK_PROVIDERS_PROVIDER_OPENAI_REASONING_EFFORTS: "medium,xhigh",
173
+ PLURNK_PROVIDERS_PROVIDER_OPENROUTER_REASONING_EFFORTS: "medium,xhigh",
174
+ });
175
+ for (const [name, expected] of [
176
+ ["openai", ["off", "adaptive", "low", "medium", "high", "xhigh"]],
177
+ ["openrouter", ["off", "adaptive", "low", "medium", "high"]],
178
+ ] as const) {
179
+ const catalog = lookupProvider(name);
180
+ assert.ok(catalog);
181
+ assert.deepEqual(catalogReasoningPolicies(catalog, info, declared), expected);
182
+ assert.throws(() => providerFromSdkModel({
183
+ name, model: "fixture", contextWindow: 32_768, info,
184
+ env: { ...declared, PLURNK_PROVIDERS_REASONING: "max" },
185
+ languageModel: { specificationVersion: "v4" } as LanguageModel,
186
+ sdkPackage: catalog.npm,
187
+ }), UnsupportedReasoningPolicyError);
188
+ }
189
+ });
190
+
107
191
  test("Models.dev controls Cloudflare's exact effort vocabulary", async () => {
108
192
  const bodies: Record<string, unknown>[] = [];
109
193
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
@@ -357,7 +441,7 @@ test("xAI's native chat contract requests its strongest cataloged effort and cap
357
441
  });
358
442
  });
359
443
 
360
- test("Cerebras explicit reasoning activation needs no operator effort or token budget", async () => {
444
+ test("Cerebras explicit and adaptive reasoning do not invent a token budget", async () => {
361
445
  let body: Record<string, unknown> | undefined;
362
446
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
363
447
  body = JSON.parse(String(init?.body)) as Record<string, unknown>;
@@ -366,21 +450,21 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
366
450
  id: "chatcmpl-cerebras",
367
451
  object: "chat.completion.chunk",
368
452
  created: 1,
369
- model: "gemma-4-31b",
453
+ model: "qwen-3.8-27b",
370
454
  choices: [{ index: 0, delta: { reasoning: "consider" }, finish_reason: null }],
371
455
  })}`,
372
456
  `data: ${JSON.stringify({
373
457
  id: "chatcmpl-cerebras",
374
458
  object: "chat.completion.chunk",
375
459
  created: 2,
376
- model: "gemma-4-31b",
460
+ model: "qwen-3.8-27b",
377
461
  choices: [{ index: 0, delta: { content: "done" }, finish_reason: "stop" }],
378
462
  })}`,
379
463
  `data: ${JSON.stringify({
380
464
  id: "chatcmpl-cerebras",
381
465
  object: "chat.completion.chunk",
382
466
  created: 3,
383
- model: "gemma-4-31b",
467
+ model: "qwen-3.8-27b",
384
468
  choices: [],
385
469
  usage: {
386
470
  prompt_tokens: 2,
@@ -393,20 +477,22 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
393
477
  ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
394
478
  });
395
479
 
396
- const provider = catalogProviderFromEnv("cerebras", {
397
- ...env,
398
- CEREBRAS_API_KEY: "test-key",
399
- PLURNK_PROVIDERS_REASONING: "high",
400
- }, "gemma-4-31b");
401
- const result = await provider?.generate({
402
- workerId: "worker",
403
- messages: [{ role: "user", content: "hello" }],
404
- });
480
+ for (const policy of ["high", "adaptive"] as const) {
481
+ const provider = catalogProviderFromEnv("cerebras", {
482
+ ...env,
483
+ CEREBRAS_API_KEY: "test-key",
484
+ PLURNK_PROVIDERS_REASONING: policy,
485
+ }, "qwen-3.8-27b");
486
+ const result = await provider?.generate({
487
+ workerId: "worker",
488
+ messages: [{ role: "user", content: "hello" }],
489
+ });
405
490
 
406
- assert.equal(body?.reasoning_effort, "high", "the native SDK preserves the explicit durable effort");
407
- assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
408
- assert.equal(result?.assistant.reasoning, "consider");
409
- assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
491
+ assert.equal(body?.reasoning_effort, "high", `${policy} reaches the native SDK as high`);
492
+ assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
493
+ assert.equal(result?.assistant.reasoning, "consider");
494
+ assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
495
+ }
410
496
  });
411
497
 
412
498
  test("Meta Muse adaptive reasoning is requested even when the endpoint returns no readable trace", async () => {
@@ -807,9 +893,11 @@ test("Models.dev is the only fallback rate table", async () => {
807
893
  }, "deepseek-v4-flash");
808
894
  assert.notEqual(cataloged, null);
809
895
  const catalogedResponse = await cataloged!.generate({ workerId: "cataloged", messages: [] });
896
+ const catalogAmount = calculateCostUsdDecimal(catalogedResponse.accounting[0]!.usage!, catalogRatesOf("deepseek", "deepseek-v4-flash"));
897
+ assert.match(String(catalogAmount), /^0\.0*[1-9]/, "the catalog prices this usage");
810
898
  assert.deepEqual(catalogedResponse.accounting[0]?.cost, {
811
899
  kind: "estimated",
812
- amount: { amount: "0.00012712", currency: "USD" },
900
+ amount: { amount: catalogAmount, currency: "USD" },
813
901
  source: "Models.dev catalog rates",
814
902
  });
815
903
 
@@ -854,9 +942,11 @@ test("{§operator-cost-override} declared rates overlay the catalog and the sour
854
942
  PLURNK_PROVIDERS_COST: "input=0.22,output=0.66,cacheRead=0.007",
855
943
  }, "deepseek-v4-flash");
856
944
  const response = await overridden!.generate({ workerId: "overridden", messages: [] });
945
+ const overriddenAmount = calculateCostUsdDecimal(response.accounting[0]!.usage!, { ...catalogRatesOf("deepseek", "deepseek-v4-flash"), input: 0.22, output: 0.66, cacheRead: 0.007 });
946
+ assert.notEqual(overriddenAmount, calculateCostUsdDecimal(response.accounting[0]!.usage!, catalogRatesOf("deepseek", "deepseek-v4-flash")), "the override changes the price");
857
947
  assert.deepEqual(response.accounting[0]?.cost, {
858
948
  kind: "estimated",
859
- amount: { amount: "0.0002148", currency: "USD" },
949
+ amount: { amount: overriddenAmount, currency: "USD" },
860
950
  source: "operator PLURNK_PROVIDERS_COST override over Models.dev catalog rates",
861
951
  });
862
952
  mock.restoreAll();
@@ -3,6 +3,7 @@ import {
3
3
  resolveModel,
4
4
  type ModelInfo,
5
5
  type ModelReasoningEffort,
6
+ type ProviderInfo,
6
7
  } from "@plurnk/plurnk-models";
7
8
  import {
8
9
  costOverrideFromEnv,
@@ -176,6 +177,41 @@ const supportedReasoningPolicies = ({
176
177
  return activationPolicies;
177
178
  };
178
179
 
180
+ const reasoningCapabilities = (
181
+ name: string,
182
+ env: NodeJS.ProcessEnv,
183
+ info: ModelInfo | undefined,
184
+ native: boolean,
185
+ sdkPackage?: string,
186
+ ) => {
187
+ const declaredEfforts = declaredEffortsFromEnv(env, name);
188
+ const declaredStyle = info?.reasoning === false ? "none" : reasoningStyleFromEnv(env, name) ?? "none";
189
+ const style = info?.reasoningOptions !== undefined
190
+ && (declaredStyle === "effort" || declaredStyle === "effort_required")
191
+ && catalogEfforts(info, declaredEfforts).length === 0
192
+ ? "none"
193
+ : declaredStyle;
194
+ const [first, ...rest] = supportedReasoningPolicies({ info, native, style, declared: declaredEfforts })
195
+ .filter((policy) => (!native || policy !== "max")
196
+ && (sdkPackage !== "@openrouter/ai-sdk-provider" || policy !== "xhigh"));
197
+ if (first === undefined) throw new TypeError(`${name} provider: no portable reasoning policy is representable`);
198
+ return {
199
+ declaredEfforts,
200
+ style,
201
+ policies: [first, ...rest] satisfies [ReasoningPolicy, ...ReasoningPolicy[]],
202
+ };
203
+ };
204
+
205
+ // {§provider-reasoning-policy} Read-only discovery and construction share admission;
206
+ // the compatible SDK package is the only catalog transport without a native model.
207
+ export const catalogReasoningPolicies = (
208
+ provider: ProviderInfo,
209
+ info: ModelInfo,
210
+ env: NodeJS.ProcessEnv,
211
+ ): readonly [ReasoningPolicy, ...ReasoningPolicy[]] => reasoningCapabilities(
212
+ provider.id, env, info, provider.npm !== "@ai-sdk/openai-compatible", provider.npm,
213
+ ).policies;
214
+
179
215
  const adaptiveReasoningProjection = ({
180
216
  sdkPackage,
181
217
  model,
@@ -276,15 +312,9 @@ export const providerFromSdkModel = ({
276
312
  );
277
313
  const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
278
314
  const reasoningCapable = info?.reasoning === true;
279
- const declaredEfforts = declaredEffortsFromEnv(env, name);
280
- const declaredReasoningStyle = info?.reasoning === false
281
- ? "none"
282
- : reasoningStyleFromEnv(env, name) ?? "none";
283
- const reasoningStyle = info?.reasoningOptions !== undefined
284
- && (declaredReasoningStyle === "effort" || declaredReasoningStyle === "effort_required")
285
- && catalogEfforts(info, declaredEfforts).length === 0
286
- ? "none"
287
- : declaredReasoningStyle;
315
+ const { declaredEfforts, style: reasoningStyle, policies } = reasoningCapabilities(
316
+ name, env, info, languageModel !== undefined, sdkPackage,
317
+ );
288
318
  const adaptiveReasoning = adaptiveReasoningProjection({
289
319
  sdkPackage,
290
320
  model,
@@ -351,12 +381,7 @@ export const providerFromSdkModel = ({
351
381
  reasoningBudget: reasoning.budget,
352
382
  // #457 — a catalog-declared toggle control makes `adaptive` an explicit enable.
353
383
  reasoningToggle: info?.reasoningOptions?.some((option) => option.type === "toggle") === true,
354
- supportedReasoningPolicies: supportedReasoningPolicies({
355
- info,
356
- native: languageModel !== undefined,
357
- style: reasoningStyle,
358
- declared: declaredEfforts,
359
- }),
384
+ supportedReasoningPolicies: policies,
360
385
  ...adaptiveReasoning,
361
386
  ...(compatibleAdaptiveReasoning === undefined ? {} : { compatibleAdaptiveReasoning }),
362
387
  ...(compatibleOffReasoning === undefined ? {} : { compatibleOffReasoning }),
@@ -386,9 +411,6 @@ export const providerFromSdkModel = ({
386
411
  estimateCost,
387
412
  source: providerSource(name),
388
413
  ...(grammarStyle === undefined ? {} : { grammarStyle }),
389
- gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
390
- && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
391
- && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
392
414
  ...dataCaptureFromEnv(env, name),
393
415
  });
394
416
  };
@@ -215,9 +215,6 @@ export const compatibleProviderFromEnv = async (
215
215
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", provider),
216
216
  source: providerSource(provider),
217
217
  grammarStyle,
218
- gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
219
- && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
220
- && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
221
218
  ...dataCaptureFromEnv(env, provider),
222
219
  firstPartyMetadata: provider === "plurnk",
223
220
  normalizeCost: provider === "plurnk" ? plurnkCostNormalizer : undefined,
package/src/env.test.ts CHANGED
@@ -211,6 +211,7 @@ test("every PROVIDERS_KNOBS entry appears in the shipped .env.defaults", async (
211
211
  const missing = PROVIDERS_KNOBS.filter((k) => !defaults.includes(k));
212
212
  assert.deepEqual([...missing], [], "knobs read by code but undeclared in .env.defaults");
213
213
  assert.ok(defaults.includes("PLURNK_PROVIDERS_GBNF="), "GBNF (service-read, providers-namespace) must be declared with its default");
214
+ assert.equal(defaults.includes("PLURNK_PROVIDERS_GBNF_DEBUG"), false, "the debug knob is gone with the generator (#588)");
214
215
  });
215
216
 
216
217
  // The family word is REASONING (industry standard). Old names fail hard
package/src/env.ts CHANGED
@@ -243,7 +243,7 @@ export const resolveGenerationEnvelopeFromEnv = (
243
243
  // PLURNK_PROVIDERS_REASONING_BUDGET optional reasoning subset of the total
244
244
  // output budget, used for tier/budget mapping where the backend supports it.
245
245
  // The provider maps intent to the backend's mechanism; the consumer states
246
- // intent, never mechanism. PLAN is a separate public complete-Plan record.
246
+ // intent, never mechanism. TASK inventory is separate from provider reasoning.
247
247
  export type Reasoning = { mode: ReasoningPolicy; budget: number | null };
248
248
 
249
249
  export const parseReasoningPolicy = (value: unknown, label: string): ReasoningPolicy => {
@@ -316,7 +316,6 @@ export const PROVIDERS_KNOBS = Object.freeze([
316
316
  "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH",
317
317
  "PLURNK_PROVIDERS_PROBE_ATTEMPTS",
318
318
  "PLURNK_PROVIDERS_PROBE_DELAY",
319
- "PLURNK_PROVIDERS_GBNF_DEBUG",
320
319
  "PLURNK_PROVIDERS_TOP_LOGPROBS",
321
320
  "PLURNK_PROVIDERS_RAWBODY",
322
321
  ]);
@@ -1,6 +1,6 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
- import { APICallError, RetryError } from "ai";
3
+ import { APICallError, JSONParseError, RetryError, TypeValidationError } from "ai";
4
4
  import { ProviderError, ProviderTimeoutError, classifyProviderError, toProviderError } from "./errors.ts";
5
5
  import type { ProviderAttempt } from "./types.ts";
6
6
  import { providerSource } from "./notices.ts";
@@ -15,6 +15,22 @@ const apiError = (statusCode: number, responseBody = "body") => new APICallError
15
15
 
16
16
  const SOURCE_PATTERN = /^[a-z]+(:[a-z][a-z0-9-]*)?$/;
17
17
 
18
+ test("HTTP request rejection preserves its cause without becoming a response-contract strike", () => {
19
+ for (const status of [400, 404, 405, 422]) {
20
+ const message = "The requested model is not available on this endpoint.";
21
+ const cause = apiError(status, JSON.stringify({ error: { message } }));
22
+ Object.defineProperty(cause, "message", { value: message });
23
+ const failure = toProviderError(cause, "provider:example");
24
+ assert.equal(failure.kind, "request_rejected");
25
+ assert.equal(failure.status, status);
26
+ assert.equal(failure.message, message);
27
+ assert.equal(failure.problem.detail, message);
28
+ assert.equal(failure.problem.retryable, false);
29
+ assert.equal(failure.cause, cause);
30
+ }
31
+ assert.equal(classifyProviderError(apiError(200)).kind, "invalid_response", "a malformed successful response still violates the response contract");
32
+ });
33
+
18
34
  test("providerSource produces a schema-valid colon-namespaced source", () => {
19
35
  assert.equal(providerSource("openai"), "provider:openai");
20
36
  assert.match(providerSource("openrouter"), SOURCE_PATTERN);
@@ -34,8 +50,8 @@ test("classifyProviderError maps HTTP status to kind", () => {
34
50
  assert.equal(k(500), "network_failure");
35
51
  assert.equal(k(503), "network_failure");
36
52
  assert.equal(k(413), "capacity_exceeded");
37
- assert.equal(k(400), "invalid_response");
38
- assert.equal(k(404), "invalid_response");
53
+ assert.equal(k(400), "request_rejected");
54
+ assert.equal(k(404), "request_rejected");
39
55
  });
40
56
 
41
57
  test("capacity normalization prefers structured provider codes and keeps generic 400s distinct", () => {
@@ -53,7 +69,7 @@ test("capacity normalization prefers structured provider codes and keeps generic
53
69
  assert.equal(normalized.problem.providerStatus, 400);
54
70
  assert.equal(classifyProviderError(apiError(400, JSON.stringify({
55
71
  error: { type: "invalid_request_error", code: "bad_temperature", message: "bad temperature" },
56
- }))).kind, "invalid_response");
72
+ }))).kind, "request_rejected");
57
73
  });
58
74
 
59
75
  test("provider retry directives survive HTTP failure normalization", () => {
@@ -69,11 +85,11 @@ test("provider retry directives survive HTTP failure normalization", () => {
69
85
  assert.equal(error.problem.retryable, false);
70
86
  });
71
87
 
72
- test("classifyProviderError: a 422 flagged grammar_invalid is distinct; other 422s are invalid responses", () => {
88
+ test("classifyProviderError: a 422 flagged grammar_invalid is distinct from other request rejections", () => {
73
89
  const rejected = apiError(422, JSON.stringify({ error: { type: "grammar_invalid", message: "non-conforming emission rejected: ..." } }));
74
90
  assert.equal(classifyProviderError(rejected).kind, "grammar_invalid");
75
- assert.equal(classifyProviderError(apiError(422, JSON.stringify({ error: { type: "invalid_request_error" } }))).kind, "invalid_response");
76
- assert.equal(classifyProviderError(apiError(422, "<html>Bad</html>")).kind, "invalid_response");
91
+ assert.equal(classifyProviderError(apiError(422, JSON.stringify({ error: { type: "invalid_request_error" } }))).kind, "request_rejected");
92
+ assert.equal(classifyProviderError(apiError(422, "<html>Bad</html>")).kind, "request_rejected");
77
93
  });
78
94
 
79
95
  test("classifyProviderError treats non-HTTP errors as network_failure", () => {
@@ -225,3 +241,38 @@ test("retry exhaustion retains the exact inner deadline phase", () => {
225
241
  assert.equal(error.problem.timeoutPhase, "attempt");
226
242
  assert.equal(error.problem.timeoutMs, 10);
227
243
  });
244
+
245
+ // {§provider-failure-cause} — the SDK wraps every processing failure of a 2xx body in one message;
246
+ // the durable Problem keeps a bounded classification of what actually happened underneath.
247
+ const processingFailure = (cause: Error) => new APICallError({
248
+ message: "Failed to process successful response",
249
+ url: "https://example.test/v1/chat/completions",
250
+ requestBodyValues: {},
251
+ statusCode: 200,
252
+ responseHeaders: { authorization: "Bearer never-copied", "x-request-id": "req_1" },
253
+ cause,
254
+ });
255
+
256
+ test("#593: a terminated body, invalid JSON, a schema-invalid body, and an internal fault are told apart in durable evidence", () => {
257
+ const terminated = toProviderError(processingFailure(new TypeError("terminated")), "provider:test", 512);
258
+ assert.equal(terminated.problem.providerKind, "network_failure");
259
+ assert.deepEqual(terminated.problem.cause, { causeKind: "transport_terminated", causeName: "TypeError", causeMessage: "terminated" });
260
+
261
+ const parse = toProviderError(processingFailure(new JSONParseError({ text: "<html>502 Bad Gateway</html>".repeat(40), cause: new SyntaxError("Unexpected token <") })), "provider:test", 512);
262
+ assert.equal(parse.problem.providerKind, "invalid_response");
263
+ assert.deepEqual(parse.problem.cause, { causeKind: "invalid_json", causeName: "AI_JSONParseError", causeMessage: "JSON parsing failed (1120 characters)" });
264
+ assert.doesNotMatch(JSON.stringify(parse.problem), /Bad Gateway/, "the payload never enters the Problem");
265
+
266
+ const schema = toProviderError(processingFailure(new TypeValidationError({ value: { choices: "not-an-array", secret: "sk-live" }, cause: new Error("Expected array, received string at choices\nmore lines") })), "provider:test", 512);
267
+ assert.equal(schema.problem.providerKind, "invalid_response");
268
+ assert.deepEqual(schema.problem.cause, { causeKind: "schema_invalid", causeName: "AI_TypeValidationError", causeMessage: "Type validation failed: Expected array, received string at choices" });
269
+ assert.doesNotMatch(JSON.stringify(schema.problem), /sk-live|not-an-array/, "the value never enters the Problem");
270
+
271
+ const internal = toProviderError(processingFailure(new RangeError("Invalid array length\n at Array.push (stack frame)")), "provider:test", 512);
272
+ assert.deepEqual(internal.problem.cause, { causeKind: "internal", causeName: "RangeError", causeMessage: "Invalid array length" });
273
+ assert.doesNotMatch(JSON.stringify(internal.problem), /stack frame|authorization|Bearer|req_1/, "no stack, no headers");
274
+
275
+ const bounded = toProviderError(processingFailure(new Error("x".repeat(100))), "provider:test", 8);
276
+ assert.equal((bounded.problem.cause as { causeMessage: string }).causeMessage, "xxxxxxxx...", "the cause message honours the configured detail limit");
277
+ assert.equal(toProviderError(apiError(502), "provider:test", 512).problem.cause, undefined, "an error without a cause reports none");
278
+ });
package/src/errors.ts CHANGED
@@ -1,4 +1,4 @@
1
- import { APICallError, RetryError } from "ai";
1
+ import { APICallError, InvalidResponseDataError, JSONParseError, RetryError, TypeValidationError } from "ai";
2
2
  import { ProviderError } from "./providerError.ts";
3
3
  import type { ProviderErrorKind } from "./providerError.ts";
4
4
  import type { ProviderRequestCapacity } from "./types.ts";
@@ -81,6 +81,34 @@ const preview = (value: unknown, limit: number | undefined): string => {
81
81
  : text;
82
82
  };
83
83
 
84
+ export type ProviderFailureCauseKind = "transport_terminated" | "invalid_json" | "schema_invalid" | "invalid_response_data" | "internal";
85
+
86
+ export interface ProviderFailureCause {
87
+ readonly causeKind: ProviderFailureCauseKind;
88
+ readonly causeName: string;
89
+ readonly causeMessage: string;
90
+ }
91
+
92
+ const firstLine = (text: string): string => text.split("\n")[0]?.trim() ?? "";
93
+
94
+ // {§provider-failure-cause} — what the SDK wrapped. "Failed to process successful response" is
95
+ // the SDK's one message for a body that terminated, a body that was not JSON, a body that failed
96
+ // the response schema, and its own internal faults; the durable Problem keeps a bounded
97
+ // classification of the cause so forensics can tell them apart. The SDK's parse and validation
98
+ // messages embed the payload, so those two report shape, never text.
99
+ const failureCause = (err: APICallError, detailLimit: number | undefined): ProviderFailureCause | null => {
100
+ const cause = err.cause;
101
+ if (!(cause instanceof Error)) return null;
102
+ if (peerTerminated(cause)) return { causeKind: "transport_terminated", causeName: cause.name, causeMessage: preview(firstLine(cause.message), detailLimit) };
103
+ if (JSONParseError.isInstance(cause)) return { causeKind: "invalid_json", causeName: cause.name, causeMessage: `JSON parsing failed (${cause.text.length} characters)` };
104
+ if (TypeValidationError.isInstance(cause)) {
105
+ const inner = cause.cause instanceof Error ? firstLine(cause.cause.message) : "";
106
+ return { causeKind: "schema_invalid", causeName: cause.name, causeMessage: inner === "" ? "Type validation failed" : `Type validation failed: ${preview(inner, detailLimit)}` };
107
+ }
108
+ if (InvalidResponseDataError.isInstance(cause)) return { causeKind: "invalid_response_data", causeName: cause.name, causeMessage: preview(firstLine(cause.message), detailLimit) };
109
+ return { causeKind: "internal", causeName: cause.name, causeMessage: preview(firstLine(cause.message), detailLimit) };
110
+ };
111
+
84
112
  export const classifyProviderError = (
85
113
  err: unknown,
86
114
  detailLimit?: number,
@@ -94,6 +122,31 @@ export const classifyProviderError = (
94
122
  };
95
123
  }
96
124
  if (APICallError.isInstance(err)) {
125
+ const classified = classifyApiCallError(err, detailLimit);
126
+ const cause = failureCause(err, detailLimit);
127
+ return cause === null ? classified : { ...classified, extensions: { ...(classified.extensions ?? {}), cause } };
128
+ }
129
+ const wire = err as { message?: unknown; type?: unknown };
130
+ if (wire?.type === "grammar_invalid") {
131
+ return {
132
+ kind: "grammar_invalid",
133
+ message: typeof wire.message === "string"
134
+ ? preview(wire.message, detailLimit)
135
+ : "The provider rejected the response grammar.",
136
+ };
137
+ }
138
+ const error = err as { message?: string };
139
+ return {
140
+ kind: "network_failure",
141
+ message: preview(
142
+ (error?.message ?? String(err)) || "The provider request failed.",
143
+ detailLimit,
144
+ ),
145
+ };
146
+ };
147
+
148
+ const classifyApiCallError = (err: APICallError, detailLimit: number | undefined): ClassifiedProviderError => {
149
+ {
97
150
  const timeout = providerTimeoutOf(err);
98
151
  if (timeout !== null) {
99
152
  return {
@@ -146,25 +199,9 @@ export const classifyProviderError = (
146
199
  if (status === 422 && wire.type === "grammar_invalid") {
147
200
  return { kind: "grammar_invalid", message };
148
201
  }
202
+ if (status >= 400 && status < 500) return { kind: "request_rejected", message };
149
203
  return { kind: "invalid_response", message };
150
204
  }
151
- const wire = err as { message?: unknown; type?: unknown };
152
- if (wire?.type === "grammar_invalid") {
153
- return {
154
- kind: "grammar_invalid",
155
- message: typeof wire.message === "string"
156
- ? preview(wire.message, detailLimit)
157
- : "The provider rejected the response grammar.",
158
- };
159
- }
160
- const error = err as { message?: string };
161
- return {
162
- kind: "network_failure",
163
- message: preview(
164
- (error?.message ?? String(err)) || "The provider request failed.",
165
- detailLimit,
166
- ),
167
- };
168
205
  };
169
206
 
170
207
  export const toProviderError = (
package/src/index.ts CHANGED
@@ -45,8 +45,8 @@ export {
45
45
  loadActiveProvider,
46
46
  resetDiscoveryCache,
47
47
  } from "./ProviderRegistry.ts";
48
- export { createEmbeddingModel, providerReadiness } from "./sdkModels.ts";
49
- export type { EmbeddingModelResolution } from "./sdkModels.ts";
48
+ export { providerReadiness } from "./sdkModels.ts";
49
+ export { catalogReasoningPolicies } from "./catalogProvider.ts";
50
50
 
51
51
  // Scope-agnostic plugin discovery ({§plugin-family-kind}).
52
52
  export { discover } from "./discover.ts";