@plurnk/plurnk-providers 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/.env.defaults +25 -13
  2. package/README.md +8 -1
  3. package/SPEC.md +93 -15
  4. package/dist/AiSdkProvider.d.ts +6 -3
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +108 -51
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +1 -0
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +2 -0
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -0
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +3 -0
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts.map +1 -1
  17. package/dist/ProviderRegistry.js +11 -10
  18. package/dist/ProviderRegistry.js.map +1 -1
  19. package/dist/accounting.d.ts.map +1 -1
  20. package/dist/accounting.js +7 -6
  21. package/dist/accounting.js.map +1 -1
  22. package/dist/aiSdkTransport.d.ts +2 -1
  23. package/dist/aiSdkTransport.d.ts.map +1 -1
  24. package/dist/aiSdkTransport.js +29 -6
  25. package/dist/aiSdkTransport.js.map +1 -1
  26. package/dist/catalogProvider.d.ts +4 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +94 -3
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +2 -0
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/cost.d.ts.map +1 -1
  34. package/dist/cost.js +5 -4
  35. package/dist/cost.js.map +1 -1
  36. package/dist/discover.d.ts +2 -0
  37. package/dist/discover.d.ts.map +1 -1
  38. package/dist/discover.js +13 -2
  39. package/dist/discover.js.map +1 -1
  40. package/dist/env.d.ts +3 -2
  41. package/dist/env.d.ts.map +1 -1
  42. package/dist/env.js +11 -4
  43. package/dist/env.js.map +1 -1
  44. package/dist/index.d.ts +9 -5
  45. package/dist/index.d.ts.map +1 -1
  46. package/dist/index.js +6 -3
  47. package/dist/index.js.map +1 -1
  48. package/dist/notices.d.ts +1 -1
  49. package/dist/notices.d.ts.map +1 -1
  50. package/dist/openai.d.ts +1 -1
  51. package/dist/openai.d.ts.map +1 -1
  52. package/dist/openai.js +1 -1
  53. package/dist/openai.js.map +1 -1
  54. package/dist/sdkModels.d.ts +2 -0
  55. package/dist/sdkModels.d.ts.map +1 -1
  56. package/dist/sdkModels.js +163 -19
  57. package/dist/sdkModels.js.map +1 -1
  58. package/dist/types.d.ts +8 -2
  59. package/dist/types.d.ts.map +1 -1
  60. package/dist/types.js +10 -1
  61. package/dist/types.js.map +1 -1
  62. package/package.json +9 -9
  63. package/src/AiSdkProvider.test.ts +206 -32
  64. package/src/AiSdkProvider.ts +140 -54
  65. package/src/Mock.ts +2 -0
  66. package/src/Pool.test.ts +1 -0
  67. package/src/Pool.ts +5 -0
  68. package/src/ProviderRegistry.test.ts +27 -14
  69. package/src/ProviderRegistry.ts +19 -10
  70. package/src/accounting.test.ts +6 -2
  71. package/src/accounting.ts +7 -6
  72. package/src/aiSdkTransport.ts +32 -7
  73. package/src/catalogProvider.test.ts +151 -19
  74. package/src/catalogProvider.ts +125 -3
  75. package/src/compatibleProvider.test.ts +13 -10
  76. package/src/compatibleProvider.ts +2 -0
  77. package/src/cost.ts +5 -4
  78. package/src/discover.test.ts +27 -0
  79. package/src/discover.ts +20 -3
  80. package/src/env.test.ts +23 -8
  81. package/src/env.ts +17 -8
  82. package/src/index.ts +16 -8
  83. package/src/notices.ts +1 -1
  84. package/src/openai.ts +1 -1
  85. package/src/providerDefaults.test.ts +50 -0
  86. package/src/sdkModels.test.ts +142 -8
  87. package/src/sdkModels.ts +201 -19
  88. package/src/types.ts +16 -0
@@ -1,5 +1,5 @@
1
1
  import { createOpenAICompatible, type ProviderErrorStructure } from "@ai-sdk/openai-compatible";
2
- import { APICallError, generateText, streamText, type JSONValue, type LanguageModel, type LanguageModelUsage } from "ai";
2
+ import { APICallError, generateText, streamText, type CallWarning, type JSONValue, type LanguageModel, type LanguageModelUsage } from "ai";
3
3
  import { z } from "zod/v4";
4
4
  import type { ChatMessage, ProviderAttemptFinishReason, ProviderChargeEvidence, ProviderUsage, TokenLogprob } from "./types.ts";
5
5
  import { normalizeUsage, type RawUsage } from "./usage.ts";
@@ -176,6 +176,7 @@ export type AiSdkTransportResponse = {
176
176
  logprobs: TokenLogprob[];
177
177
  chargeEvidence: ProviderChargeEvidence;
178
178
  rawBody?: unknown;
179
+ warnings: readonly CallWarning[];
179
180
  };
180
181
 
181
182
  export type AiSdkModelRequest = Omit<AiSdkTransportRequest, "url" | "model" | "body" | "fetch"> & {
@@ -221,11 +222,32 @@ const transportTimeout = (
221
222
 
222
223
  const streamFailureValues = new WeakMap<object, readonly unknown[]>();
223
224
 
224
- const applyRetryDirective = (error: unknown): unknown => {
225
- if (!APICallError.isInstance(error)) return error;
225
+ const retainStreamFailureValues = <T extends object>(source: object, target: T): T => {
226
+ const values = streamFailureValues.get(source);
227
+ if (values !== undefined) streamFailureValues.set(target, values);
228
+ return target;
229
+ };
230
+
231
+ const normalizeRetryAttemptError = (error: unknown): unknown => {
232
+ if (!APICallError.isInstance(error)) {
233
+ // Node's Undici stream reader reports a peer-aborted HTTP/2 body as this
234
+ // raw TypeError after headers have arrived. Normalize it at the attempt
235
+ // boundary so the owned scheduler sees the same retryability that the
236
+ // public ProviderError contract would otherwise assign too late.
237
+ if (error instanceof TypeError && error.message.trim().toLowerCase() === "terminated") {
238
+ return retainStreamFailureValues(error, new APICallError({
239
+ message: error.message,
240
+ url: "model:generation",
241
+ requestBodyValues: {},
242
+ cause: error,
243
+ isRetryable: true,
244
+ }));
245
+ }
246
+ return error;
247
+ }
226
248
  const directed = retryDirective(error.statusCode, error.responseHeaders ?? {});
227
249
  if (directed === null || directed === error.isRetryable) return error;
228
- return new APICallError({
250
+ return retainStreamFailureValues(error, new APICallError({
229
251
  message: error.message,
230
252
  url: error.url,
231
253
  requestBodyValues: error.requestBodyValues,
@@ -235,7 +257,7 @@ const applyRetryDirective = (error: unknown): unknown => {
235
257
  cause: error,
236
258
  isRetryable: directed,
237
259
  data: error.data,
238
- });
260
+ }));
239
261
  };
240
262
 
241
263
  const executeModel = async (
@@ -246,7 +268,7 @@ const executeModel = async (
246
268
  } catch (cause) {
247
269
  if (request.signal?.aborted) throw request.signal.reason;
248
270
  const timeout = transportTimeout(cause, request);
249
- if (timeout === null) throw applyRetryDirective(cause);
271
+ if (timeout === null) throw normalizeRetryAttemptError(cause);
250
272
  throw new APICallError({
251
273
  message: timeout.message,
252
274
  url: "model:generation",
@@ -364,6 +386,7 @@ const executeModelOnce = async (
364
386
  },
365
387
  },
366
388
  ...(request.captureRawBody ? { rawBody } : {}),
389
+ warnings: result.warnings ?? [],
367
390
  };
368
391
  }
369
392
 
@@ -389,9 +412,10 @@ const executeModelOnce = async (
389
412
  const content = await result.text;
390
413
  const reasoningText = evidence.reasoning || (await result.reasoningText) || "";
391
414
  const rawFinishReason = await result.rawFinishReason;
392
- const [response, providerMetadata] = await Promise.all([
415
+ const [response, providerMetadata, warnings] = await Promise.all([
393
416
  result.response,
394
417
  result.providerMetadata,
418
+ result.warnings,
395
419
  ]);
396
420
  return {
397
421
  model: response.modelId,
@@ -416,6 +440,7 @@ const executeModelOnce = async (
416
440
  },
417
441
  },
418
442
  ...(request.captureRawBody ? { rawBody: rawChunks } : {}),
443
+ warnings: warnings ?? [],
419
444
  };
420
445
  };
421
446
 
@@ -2,7 +2,8 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { once } from "node:events";
4
4
  import { createServer } from "node:http";
5
- import { catalogProviderFromEnv } from "./catalogProvider.ts";
5
+ import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
6
+ import type { LanguageModel } from "ai";
6
7
  import { resetEmittedWarnings } from "./warnings.ts";
7
8
 
8
9
  const env = {
@@ -37,6 +38,48 @@ test("catalog provider resolves model physics and Models.dev USD rates", () => {
37
38
  assert.equal(provider?.maxOutputTokens, 32_768);
38
39
  assert.equal(provider?.outputBudget, 32_768);
39
40
  assert.equal(provider?.reasoningBudget, null);
41
+ assert.deepEqual(provider?.supportedReasoningPolicies, ["off", "adaptive"]);
42
+ });
43
+
44
+ test("provider adapters advertise only reasoning policies they can preserve", () => {
45
+ const deepseek = catalogProviderFromEnv("deepseek", {
46
+ ...env,
47
+ DEEPSEEK_API_KEY: "test-key",
48
+ PLURNK_PROVIDERS_REASONING: "adaptive",
49
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
50
+ }, "deepseek-v4-flash");
51
+ assert.deepEqual(deepseek?.supportedReasoningPolicies, ["off", "adaptive", "high"]);
52
+
53
+ assert.throws(
54
+ () => catalogProviderFromEnv("deepseek", {
55
+ ...env,
56
+ DEEPSEEK_API_KEY: "test-key",
57
+ PLURNK_PROVIDERS_REASONING: "medium",
58
+ PLURNK_PROVIDERS_PROVIDER_DEEPSEEK_REASONING_STYLE: "thinking_effort",
59
+ }, "deepseek-v4-flash"),
60
+ /reasoning policy 'medium' is unsupported; supported policies: off, adaptive, high/,
61
+ );
62
+
63
+ const mistral = catalogProviderFromEnv("mistral", {
64
+ ...env,
65
+ MISTRAL_API_KEY: "test-key",
66
+ PLURNK_PROVIDERS_REASONING: "adaptive",
67
+ }, "mistral-small-latest");
68
+ assert.deepEqual(mistral?.supportedReasoningPolicies, ["off", "adaptive", "high"], "Mistral's low/medium coercion is not advertised as exact support");
69
+
70
+ const grok = catalogProviderFromEnv("xai", {
71
+ ...env,
72
+ XAI_API_KEY: "test-key",
73
+ PLURNK_PROVIDERS_REASONING: "adaptive",
74
+ }, "grok-4.6");
75
+ assert.deepEqual(grok?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Grok 4.6 cannot disable reasoning");
76
+
77
+ const gemini = catalogProviderFromEnv("google", {
78
+ ...env,
79
+ GEMINI_API_KEY: "test-key",
80
+ PLURNK_PROVIDERS_REASONING: "adaptive",
81
+ }, "gemini-3.7-flash");
82
+ assert.deepEqual(gemini?.supportedReasoningPolicies, ["adaptive", "low", "medium", "high"], "Gemini 3's mandatory minimum is not advertised as off");
40
83
  });
41
84
 
42
85
  test("an operator context window caps catalog physics and percentage output policy", () => {
@@ -110,7 +153,7 @@ test("official AI SDK provider owns the native request while PLURNK owns call se
110
153
  assert.equal(calls[0]?.body.prompt_cache_key, "worker", "the official OpenAI SDK projects the documented affinity key");
111
154
  });
112
155
 
113
- test("xAI's native chat contract caps the complete reasoning response", async () => {
156
+ test("xAI's native chat contract affirmatively requests adaptive high and caps the complete reasoning response", async () => {
114
157
  let call: { headers: Headers; body: Record<string, unknown> } | undefined;
115
158
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
116
159
  call = {
@@ -122,21 +165,21 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
122
165
  id: "response-xai",
123
166
  object: "chat.completion.chunk",
124
167
  created: 1,
125
- model: "grok-build-0.1",
168
+ model: "grok-4.6",
126
169
  choices: [{ index: 0, delta: { reasoning_content: "consider" }, finish_reason: null }],
127
170
  })}`,
128
171
  `data: ${JSON.stringify({
129
172
  id: "response-xai",
130
173
  object: "chat.completion.chunk",
131
174
  created: 2,
132
- model: "grok-build-0.1",
175
+ model: "grok-4.6",
133
176
  choices: [{ index: 0, delta: { content: "OK" }, finish_reason: "stop" }],
134
177
  })}`,
135
178
  `data: ${JSON.stringify({
136
179
  id: "response-xai",
137
180
  object: "chat.completion.chunk",
138
181
  created: 3,
139
- model: "grok-build-0.1",
182
+ model: "grok-4.6",
140
183
  choices: [],
141
184
  usage: {
142
185
  prompt_tokens: 5,
@@ -155,7 +198,7 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
155
198
  ...env,
156
199
  XAI_API_KEY: "test-key",
157
200
  PLURNK_PROVIDERS_REASONING: "adaptive",
158
- }, "grok-build-0.1");
201
+ }, "grok-4.6");
159
202
  const result = await provider?.generate({
160
203
  workerId: "xai-worker",
161
204
  messages: [{ role: "user", content: "hello" }],
@@ -163,6 +206,7 @@ test("xAI's native chat contract caps the complete reasoning response", async ()
163
206
  });
164
207
 
165
208
  assert.equal(call?.body.max_completion_tokens, 16);
209
+ assert.equal(call?.body.reasoning_effort, "high", "adaptive is affirmative on xAI's graded route");
166
210
  assert.equal("max_tokens" in (call?.body ?? {}), false);
167
211
  assert.equal(call?.headers.get("x-grok-conv-id"), "xai-worker");
168
212
  assert.equal(result?.assistant.reasoning, "consider");
@@ -221,14 +265,14 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
221
265
  const provider = catalogProviderFromEnv("cerebras", {
222
266
  ...env,
223
267
  CEREBRAS_API_KEY: "test-key",
224
- PLURNK_PROVIDERS_REASONING: "on",
268
+ PLURNK_PROVIDERS_REASONING: "high",
225
269
  }, "gemma-4-31b");
226
270
  const result = await provider?.generate({
227
271
  workerId: "worker",
228
272
  messages: [{ role: "user", content: "hello" }],
229
273
  });
230
274
 
231
- assert.equal(body?.reasoning_effort, "medium", "the native SDK projects unqualified on to its enabled posture");
275
+ assert.equal(body?.reasoning_effort, "high", "the native SDK preserves the explicit durable effort");
232
276
  assert.equal("thinking_budget_tokens" in (body ?? {}), false, "activation does not invent a token budget");
233
277
  assert.equal(result?.assistant.reasoning, "consider");
234
278
  assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
@@ -236,7 +280,7 @@ test("Cerebras explicit reasoning activation needs no operator effort or token b
236
280
 
237
281
  test("Google adaptive reasoning requests and preserves readable thought summaries", async () => {
238
282
  const bodies: Array<{
239
- generationConfig?: { thinkingConfig?: { includeThoughts?: boolean } };
283
+ generationConfig?: { thinkingConfig?: { includeThoughts?: boolean; thinkingLevel?: string; thinkingBudget?: number } };
240
284
  }> = [];
241
285
  mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
242
286
  bodies.push(JSON.parse(String(init?.body)) as typeof bodies[number]);
@@ -275,25 +319,109 @@ test("Google adaptive reasoning requests and preserves readable thought summarie
275
319
 
276
320
  assert.deepEqual(bodies[0]?.generationConfig?.thinkingConfig, {
277
321
  includeThoughts: true,
278
- }, "adaptive leaves thinking depth to Google while requesting its readable summary");
322
+ thinkingLevel: "high",
323
+ }, "Gemini 3 adaptive selects its documented high/dynamic posture and readable summary");
279
324
  assert.equal(result?.assistant.reasoning, "consider");
280
325
  assert.equal(result?.assistant.content, "done");
281
326
  assert.equal(result?.accounting[0]?.usage?.outputTokenDetails?.reasoningTokens, 1);
282
327
 
283
- const disabled = catalogProviderFromEnv("google", {
328
+ assert.throws(
329
+ () => catalogProviderFromEnv("google", {
330
+ ...env,
331
+ GEMINI_API_KEY: "test-key",
332
+ PLURNK_PROVIDERS_REASONING: "off",
333
+ }, "gemini-3.7-flash"),
334
+ /reasoning policy 'off' is unsupported/,
335
+ "Gemini 3's mandatory minimum thinking is not mislabeled as off",
336
+ );
337
+
338
+ const dynamic25 = catalogProviderFromEnv("google", {
284
339
  ...env,
285
340
  GEMINI_API_KEY: "test-key",
286
- PLURNK_PROVIDERS_REASONING: "off",
287
- }, "gemini-3.7-flash");
288
- await disabled?.generate({
341
+ PLURNK_PROVIDERS_REASONING: "adaptive",
342
+ }, "gemini-2.5-flash");
343
+ await dynamic25?.generate({
289
344
  workerId: "worker",
290
345
  messages: [{ role: "user", content: "hello" }],
291
346
  });
292
- assert.equal(
293
- bodies[1]?.generationConfig?.thinkingConfig?.includeThoughts,
294
- undefined,
295
- "off does not request readable thoughts",
296
- );
347
+ assert.deepEqual(bodies[1]?.generationConfig?.thinkingConfig, {
348
+ includeThoughts: true,
349
+ thinkingBudget: -1,
350
+ }, "Gemini 2.5 adaptive uses the provider's native dynamic budget sentinel");
351
+ });
352
+
353
+ test("native Anthropic adaptive policy uses adaptive thinking rather than a fixed high effort", async () => {
354
+ let request: Record<string, unknown> | undefined;
355
+ const languageModel = {
356
+ specificationVersion: "v4",
357
+ provider: "anthropic.messages",
358
+ modelId: "claude-sonnet-4-6",
359
+ supportedUrls: {},
360
+ doGenerate: async (options: Record<string, unknown>) => {
361
+ request = options;
362
+ return {
363
+ content: [{ type: "text", text: "ok" }],
364
+ finishReason: { unified: "stop", raw: "stop" },
365
+ usage: {
366
+ inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
367
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
368
+ },
369
+ response: { id: "response", modelId: "claude-sonnet-4-6" },
370
+ warnings: [],
371
+ };
372
+ },
373
+ doStream: async (options: Record<string, unknown>) => {
374
+ request = options;
375
+ return {
376
+ stream: new ReadableStream({
377
+ start(controller) {
378
+ controller.enqueue({ type: "stream-start", warnings: [] });
379
+ controller.enqueue({ type: "response-metadata", id: "response", modelId: "claude-sonnet-4-6" });
380
+ controller.enqueue({ type: "text-start", id: "text-1" });
381
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
382
+ controller.enqueue({ type: "text-end", id: "text-1" });
383
+ controller.enqueue({
384
+ type: "finish",
385
+ finishReason: { unified: "stop", raw: "stop" },
386
+ usage: {
387
+ inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
388
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
389
+ },
390
+ });
391
+ controller.close();
392
+ },
393
+ }),
394
+ response: {},
395
+ };
396
+ },
397
+ } as unknown as LanguageModel;
398
+ const provider = providerFromSdkModel({
399
+ name: "anthropic",
400
+ env: {
401
+ ...env,
402
+ PLURNK_PROVIDERS_REASONING: "adaptive",
403
+ PLURNK_PROVIDERS_STREAMING: "0",
404
+ },
405
+ model: "claude-sonnet-4-6",
406
+ languageModel,
407
+ sdkPackage: "@ai-sdk/anthropic",
408
+ additiveReasoningProvider: "anthropic",
409
+ contextWindow: 16_384,
410
+ info: {
411
+ name: "Claude Sonnet 4.6",
412
+ contextWindow: 16_384,
413
+ maxOutputTokens: 8_192,
414
+ reasoning: true,
415
+ attachment: true,
416
+ toolCall: true,
417
+ modalities: { input: ["text", "image"], output: ["text"] },
418
+ },
419
+ });
420
+ await provider.generate({ workerId: "worker", messages: [{ role: "user", content: "hello" }] });
421
+ assert.equal(request?.reasoning, "provider-default");
422
+ assert.deepEqual(request?.providerOptions, {
423
+ anthropic: { thinking: { type: "adaptive", display: "summarized" } },
424
+ });
297
425
  });
298
426
 
299
427
  test("native provider routes project their documented cache controls through the actual SDK request", async (t) => {
@@ -378,6 +506,8 @@ test("native provider routes project their documented cache controls through the
378
506
  const provider = catalogProviderFromEnv("openrouter", {
379
507
  ...env,
380
508
  OPENROUTER_API_KEY: "test-key",
509
+ OPENROUTER_HTTP_REFERER: "https://github.com/plurnk/plurnk-service",
510
+ OPENROUTER_APP_TITLE: "Plurnk",
381
511
  }, "anthropic/claude-sonnet-4.6", `http://127.0.0.1:${address.port}/api/v1`);
382
512
  await provider?.generate({
383
513
  workerId: "openrouter-worker",
@@ -387,6 +517,8 @@ test("native provider routes project their documented cache controls through the
387
517
  ],
388
518
  });
389
519
  assert.equal(call?.headers.get("x-session-id"), "openrouter-worker");
520
+ assert.equal(call?.headers.get("http-referer"), "https://github.com/plurnk/plurnk-service");
521
+ assert.equal(call?.headers.get("x-openrouter-title"), "Plurnk");
390
522
  assert.deepEqual((call?.body.messages as unknown[] | undefined)?.[0], {
391
523
  role: "system",
392
524
  content: [{
@@ -12,10 +12,11 @@ import {
12
12
  reasoningFromEnv,
13
13
  reasoningResponseStyleFromEnv,
14
14
  } from "./env.ts";
15
- import AiSdkProvider, { type ReasoningStyle } from "./AiSdkProvider.ts";
15
+ import AiSdkProvider, { type AiSdkProviderConfig, type GrammarStyle, type ReasoningStyle } from "./AiSdkProvider.ts";
16
16
  import { configuredProviderInfo, createSdkModel } from "./sdkModels.ts";
17
17
  import { providerSource } from "./notices.ts";
18
18
  import type { Provider, ProviderCostNormalizer } from "./types.ts";
19
+ import { REASONING_POLICIES, type ReasoningPolicy } from "@plurnk/plurnk-contracts";
19
20
  import { estimateProviderCost } from "./cost.ts";
20
21
  import { emitWarningOnce } from "./warnings.ts";
21
22
  import type { LanguageModel } from "ai";
@@ -39,6 +40,102 @@ const reasoningStyleFromEnv = (
39
40
  return value as ReasoningStyle;
40
41
  };
41
42
 
43
+ const activationPolicies = Object.freeze(["off", "adaptive"] as const);
44
+ const deepSeekPolicies = Object.freeze(["off", "adaptive", "high"] as const);
45
+ const adaptiveOnly = Object.freeze(["adaptive"] as const);
46
+ const reasoningWithoutOff = Object.freeze(["adaptive", "low", "medium", "high"] as const);
47
+
48
+ const mistralSupportsEffort = (model: string): boolean =>
49
+ model === "mistral-small-latest"
50
+ || model === "mistral-small-2603"
51
+ || model === "mistral-medium-3"
52
+ || model === "mistral-medium-3.5";
53
+
54
+ const xaiReasoningIsModelFixed = (model: string): boolean =>
55
+ /^grok-4\.20(?:-\d{4})?-(?:non-)?reasoning$/.test(model);
56
+
57
+ const googleReasoningCannotBeOff = (model: string): boolean =>
58
+ /^gemini-2\.5-pro(?:-|$)/i.test(model)
59
+ || /^gemini-(?:[3-9]|\d{2})[.-]/i.test(model);
60
+
61
+ const anthropicSupportsAdaptiveThinking = (model: string): boolean =>
62
+ /claude-(?:opus-(?:4-[678]|5)|sonnet-(?:4-6|5)|fable-5)/.test(model);
63
+
64
+ const supportedReasoningPolicies = ({
65
+ info,
66
+ native,
67
+ style,
68
+ sdkPackage,
69
+ model,
70
+ }: {
71
+ info?: ModelInfo;
72
+ native: boolean;
73
+ style: ReasoningStyle;
74
+ sdkPackage?: string;
75
+ model: string;
76
+ }): readonly ReasoningPolicy[] => {
77
+ if (info !== undefined && info.reasoning !== true) return activationPolicies;
78
+ if (native) {
79
+ if (sdkPackage === "@ai-sdk/mistral") {
80
+ return mistralSupportsEffort(model) ? deepSeekPolicies : adaptiveOnly;
81
+ }
82
+ if (sdkPackage === "@ai-sdk/xai") {
83
+ if (xaiReasoningIsModelFixed(model)) return adaptiveOnly;
84
+ if (model === "grok-4.6") return reasoningWithoutOff;
85
+ }
86
+ if (sdkPackage === "@ai-sdk/google" && googleReasoningCannotBeOff(model)) {
87
+ return reasoningWithoutOff;
88
+ }
89
+ return REASONING_POLICIES;
90
+ }
91
+ if (style === "effort" || style === "effort_explicit") return REASONING_POLICIES;
92
+ if (style === "thinking_effort") return deepSeekPolicies;
93
+ return activationPolicies;
94
+ };
95
+
96
+ const adaptiveReasoningProjection = ({
97
+ sdkPackage,
98
+ model,
99
+ reasoningCapable,
100
+ }: {
101
+ sdkPackage?: string;
102
+ model: string;
103
+ reasoningCapable: boolean;
104
+ }): Pick<AiSdkProviderConfig, "adaptiveReasoning" | "adaptiveReasoningProviderOptions"> => {
105
+ if (!reasoningCapable) return { adaptiveReasoning: "provider-default" };
106
+ if (sdkPackage === "@ai-sdk/google" && /^gemini-2\.5(?:-|$)/i.test(model)) {
107
+ return {
108
+ adaptiveReasoning: "provider-default",
109
+ adaptiveReasoningProviderOptions: {
110
+ google: { thinkingConfig: { thinkingBudget: -1 } },
111
+ },
112
+ };
113
+ }
114
+ if (sdkPackage === "@ai-sdk/anthropic" && anthropicSupportsAdaptiveThinking(model)) {
115
+ return {
116
+ adaptiveReasoning: "provider-default",
117
+ adaptiveReasoningProviderOptions: {
118
+ anthropic: { thinking: { type: "adaptive", display: "summarized" } },
119
+ },
120
+ };
121
+ }
122
+ if (sdkPackage === "@ai-sdk/amazon-bedrock" && anthropicSupportsAdaptiveThinking(model)) {
123
+ return {
124
+ adaptiveReasoning: "provider-default",
125
+ adaptiveReasoningProviderOptions: {
126
+ bedrock: { reasoningConfig: { type: "adaptive" } },
127
+ },
128
+ };
129
+ }
130
+ if (sdkPackage === "@ai-sdk/mistral" && !mistralSupportsEffort(model)) {
131
+ return { adaptiveReasoning: "provider-default" };
132
+ }
133
+ if (sdkPackage === "@ai-sdk/xai" && xaiReasoningIsModelFixed(model)) {
134
+ return { adaptiveReasoning: "provider-default" };
135
+ }
136
+ return { adaptiveReasoning: "high" };
137
+ };
138
+
42
139
  export const providerFromSdkModel = ({
43
140
  name,
44
141
  env,
@@ -54,6 +151,8 @@ export const providerFromSdkModel = ({
54
151
  systemCacheProviderOptions,
55
152
  reasoningResponseProviderOptions,
56
153
  additiveReasoningProvider,
154
+ sdkPackage,
155
+ grammarStyle,
57
156
  }: {
58
157
  name: string;
59
158
  env: NodeJS.ProcessEnv;
@@ -65,10 +164,14 @@ export const providerFromSdkModel = ({
65
164
  contextWindow: number;
66
165
  info?: ModelInfo;
67
166
  attributions?: (context: PluginAttributionContext) => PluginAttribution;
167
+ // {§provider-grammar-transport} — plugin-declared constrained-decoding
168
+ // capability; "none" keeps the grammar off the wire.
169
+ grammarStyle?: GrammarStyle;
68
170
  cacheAffinity?: CacheAffinity;
69
171
  systemCacheProviderOptions?: AiSdkProviderOptions;
70
172
  reasoningResponseProviderOptions?: AiSdkProviderOptions;
71
173
  additiveReasoningProvider?: "anthropic" | "bedrock";
174
+ sdkPackage?: string;
72
175
  }): Provider => {
73
176
  emitWarningOnce(
74
177
  `${name} provider: request-level prompt counting is a chars/2 estimate; capacity is deferred to the provider`,
@@ -86,6 +189,13 @@ export const providerFromSdkModel = ({
86
189
  maxOutputTokens,
87
190
  );
88
191
  const reasoning = reasoningFromEnv(env, name, envelope.reasoningBudget);
192
+ const reasoningStyle = reasoningStyleFromEnv(env, name) ?? "none";
193
+ const reasoningCapable = info?.reasoning === true;
194
+ const adaptiveReasoning = adaptiveReasoningProjection({
195
+ sdkPackage,
196
+ model,
197
+ reasoningCapable,
198
+ });
89
199
 
90
200
  const catalogCost = info?.cost;
91
201
  const rates = catalogCost === undefined ? null : {
@@ -118,7 +228,17 @@ export const providerFromSdkModel = ({
118
228
  maxOutputTokens,
119
229
  outputBudget: envelope.outputBudget,
120
230
  reasoningBudget: reasoning.budget,
121
- ...(additiveReasoningProvider === undefined ? {} : { additiveReasoningProvider }),
231
+ supportedReasoningPolicies: supportedReasoningPolicies({
232
+ info,
233
+ native: languageModel !== undefined,
234
+ style: reasoningStyle,
235
+ sdkPackage,
236
+ model,
237
+ }),
238
+ ...adaptiveReasoning,
239
+ ...(additiveReasoningProvider === undefined || !reasoningCapable
240
+ ? {}
241
+ : { additiveReasoningProvider }),
122
242
  fetchTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name),
123
243
  operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", name),
124
244
  firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", name),
@@ -130,7 +250,7 @@ export const providerFromSdkModel = ({
130
250
  frequencyPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
131
251
  retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
132
252
  errorDetailLimit: parseRequiredInt(env.PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT, "PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT", name),
133
- reasoningStyle: reasoningStyleFromEnv(env, name),
253
+ reasoningStyle,
134
254
  ...(affinityEnabled && cacheAffinity !== undefined ? { cacheAffinity } : {}),
135
255
  ...(cacheWritePolicy === "stable-system" && systemCacheProviderOptions !== undefined
136
256
  ? { systemCacheProviderOptions }
@@ -141,6 +261,7 @@ export const providerFromSdkModel = ({
141
261
  serviceTier: env.PLURNK_PROVIDERS_SERVICE_TIER,
142
262
  estimateCost,
143
263
  source: providerSource(name),
264
+ ...(grammarStyle === undefined ? {} : { grammarStyle }),
144
265
  gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined
145
266
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== ""
146
267
  && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
@@ -186,6 +307,7 @@ export const catalogProviderFromEnv = (
186
307
  systemCacheProviderOptions: sdk.systemCacheProviderOptions,
187
308
  reasoningResponseProviderOptions: sdk.reasoningResponseProviderOptions,
188
309
  additiveReasoningProvider: sdk.additiveReasoningProvider,
310
+ sdkPackage: sdk.catalog?.npm,
189
311
  contextWindow,
190
312
  info,
191
313
  });
@@ -21,6 +21,17 @@ const env = {
21
21
  PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
22
22
  };
23
23
 
24
+ const streamedChatResponse = (content: string) => new Response([
25
+ `data: ${JSON.stringify({
26
+ id: "test-completion",
27
+ object: "chat.completion.chunk",
28
+ created: 1,
29
+ model: "local",
30
+ choices: [{ index: 0, delta: { content }, finish_reason: "stop" }],
31
+ })}`,
32
+ "data: [DONE]",
33
+ ].join("\n\n"), { headers: { "content-type": "text/event-stream" } });
34
+
24
35
  test.afterEach(() => mock.restoreAll());
25
36
 
26
37
  test("an undifferentiated compatible endpoint receives no guessed prompt-cache field", async () => {
@@ -30,11 +41,7 @@ test("an undifferentiated compatible endpoint receives no guessed prompt-cache f
30
41
  return new Response(JSON.stringify({ data: [{ id: "local", n_ctx: 8192 }] }));
31
42
  }
32
43
  body = JSON.parse(String(init?.body)) as Record<string, unknown>;
33
- return new Response(JSON.stringify({
34
- model: "local",
35
- choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
36
- usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
37
- }), { headers: { "content-type": "application/json" } });
44
+ return streamedChatResponse("ok");
38
45
  });
39
46
 
40
47
  const provider = await compatibleProviderFromEnv("openai", env, "local");
@@ -55,11 +62,7 @@ test("the server-wide DRY-off floor emits no DRY request fields", async () => {
55
62
  }));
56
63
  }
57
64
  body = JSON.parse(String(init?.body)) as Record<string, unknown>;
58
- return new Response(JSON.stringify({
59
- model: "local",
60
- choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
61
- usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
62
- }), { headers: { "content-type": "application/json" } });
65
+ return streamedChatResponse("ok");
63
66
  });
64
67
 
65
68
  const provider = await compatibleProviderFromEnv("openai", {
@@ -182,6 +182,7 @@ export const compatibleProviderFromEnv = async (
182
182
  }
183
183
  const envelope = generationEnvelopeFromEnv(env, provider, contextWindow, null);
184
184
  const reasoning = reasoningFromEnv(env, provider, envelope.reasoningBudget);
185
+ const supportedReasoningPolicies = ["off", "adaptive"] as const;
185
186
  return new AiSdkProvider({
186
187
  model,
187
188
  url,
@@ -191,6 +192,7 @@ export const compatibleProviderFromEnv = async (
191
192
  maxOutputTokens: null,
192
193
  outputBudget: envelope.outputBudget,
193
194
  reasoningBudget: reasoning.budget,
195
+ supportedReasoningPolicies,
194
196
  fetchTimeoutMs: timeout,
195
197
  operationTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_OPERATION_TIMEOUT, "PLURNK_PROVIDERS_OPERATION_TIMEOUT", provider),
196
198
  firstContentTimeoutMs: parseTimeoutMs(env.PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT, "PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT", provider),
package/src/cost.ts CHANGED
@@ -132,8 +132,9 @@ export const addDecimals = (values: readonly string[]): string => {
132
132
  };
133
133
 
134
134
  export const sumProviderCostsUsd = (costs: readonly ProviderCost[]): string | null => {
135
- const values = costs.map(providerCostUsd);
136
- return values.some((value) => value === null)
137
- ? null
138
- : addDecimals(values as string[]);
135
+ // {§tokenomics-provider-usage} — a request without USD-expressible cost
136
+ // (an uncataloged model, or a response-less failure) is skipped; it never
137
+ // erases the expressible evidence. Null only when nothing is expressible.
138
+ const values = costs.map(providerCostUsd).filter((value): value is string => value !== null);
139
+ return values.length === 0 ? null : addDecimals(values);
139
140
  };
@@ -47,6 +47,33 @@ test("discover: a provider package missing plurnk.name is ignored, not crashed",
47
47
  assert.deepEqual([...registry.keys()], ["named"]);
48
48
  });
49
49
 
50
+ test("{§provider-grammar-transport} discover: the manifest grammarStyle declaration is recorded and validated", async (t) => {
51
+ const root = await buildModules(t, {
52
+ "@acme/llamacpp-rail": {
53
+ name: "@acme/llamacpp-rail",
54
+ plurnk: { kind: "provider", name: "rail", grammarStyle: "llamacpp" },
55
+ },
56
+ "@acme/plain": {
57
+ name: "@acme/plain",
58
+ plurnk: { kind: "provider", name: "plain" },
59
+ },
60
+ });
61
+ const { grammarStyles } = await discover({ cwd: root });
62
+ assert.equal(grammarStyles.get("rail"), "llamacpp");
63
+ assert.equal(grammarStyles.get("plain"), "none");
64
+ assert.equal(grammarStyles.get("absent"), undefined);
65
+ });
66
+
67
+ test("{§provider-grammar-transport} discover: an invalid grammarStyle fails loudly, never guessing", async (t) => {
68
+ const root = await buildModules(t, {
69
+ "@acme/bad-grammar": {
70
+ name: "@acme/bad-grammar",
71
+ plurnk: { kind: "provider", name: "bad", grammarStyle: "guff" },
72
+ },
73
+ });
74
+ await assert.rejects(discover({ cwd: root }), /grammarStyle must be "none" or "llamacpp"/);
75
+ });
76
+
50
77
  test("discover: an array kind claims no provider family", async (t) => {
51
78
  const root = await buildModules(t, {
52
79
  "@acme/dual": {