@plurnk/plurnk-providers 1.7.0 → 1.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. package/.env.defaults +25 -13
  2. package/README.md +8 -1
  3. package/SPEC.md +93 -15
  4. package/dist/AiSdkProvider.d.ts +6 -3
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +108 -51
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +1 -0
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +2 -0
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +2 -0
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +3 -0
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts.map +1 -1
  17. package/dist/ProviderRegistry.js +11 -10
  18. package/dist/ProviderRegistry.js.map +1 -1
  19. package/dist/accounting.d.ts.map +1 -1
  20. package/dist/accounting.js +7 -6
  21. package/dist/accounting.js.map +1 -1
  22. package/dist/aiSdkTransport.d.ts +2 -1
  23. package/dist/aiSdkTransport.d.ts.map +1 -1
  24. package/dist/aiSdkTransport.js +29 -6
  25. package/dist/aiSdkTransport.js.map +1 -1
  26. package/dist/catalogProvider.d.ts +4 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +94 -3
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +2 -0
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/cost.d.ts.map +1 -1
  34. package/dist/cost.js +5 -4
  35. package/dist/cost.js.map +1 -1
  36. package/dist/discover.d.ts +2 -0
  37. package/dist/discover.d.ts.map +1 -1
  38. package/dist/discover.js +13 -2
  39. package/dist/discover.js.map +1 -1
  40. package/dist/env.d.ts +3 -2
  41. package/dist/env.d.ts.map +1 -1
  42. package/dist/env.js +11 -4
  43. package/dist/env.js.map +1 -1
  44. package/dist/index.d.ts +9 -5
  45. package/dist/index.d.ts.map +1 -1
  46. package/dist/index.js +6 -3
  47. package/dist/index.js.map +1 -1
  48. package/dist/notices.d.ts +1 -1
  49. package/dist/notices.d.ts.map +1 -1
  50. package/dist/openai.d.ts +1 -1
  51. package/dist/openai.d.ts.map +1 -1
  52. package/dist/openai.js +1 -1
  53. package/dist/openai.js.map +1 -1
  54. package/dist/sdkModels.d.ts +2 -0
  55. package/dist/sdkModels.d.ts.map +1 -1
  56. package/dist/sdkModels.js +163 -19
  57. package/dist/sdkModels.js.map +1 -1
  58. package/dist/types.d.ts +8 -2
  59. package/dist/types.d.ts.map +1 -1
  60. package/dist/types.js +10 -1
  61. package/dist/types.js.map +1 -1
  62. package/package.json +9 -9
  63. package/src/AiSdkProvider.test.ts +206 -32
  64. package/src/AiSdkProvider.ts +140 -54
  65. package/src/Mock.ts +2 -0
  66. package/src/Pool.test.ts +1 -0
  67. package/src/Pool.ts +5 -0
  68. package/src/ProviderRegistry.test.ts +27 -14
  69. package/src/ProviderRegistry.ts +19 -10
  70. package/src/accounting.test.ts +6 -2
  71. package/src/accounting.ts +7 -6
  72. package/src/aiSdkTransport.ts +32 -7
  73. package/src/catalogProvider.test.ts +151 -19
  74. package/src/catalogProvider.ts +125 -3
  75. package/src/compatibleProvider.test.ts +13 -10
  76. package/src/compatibleProvider.ts +2 -0
  77. package/src/cost.ts +5 -4
  78. package/src/discover.test.ts +27 -0
  79. package/src/discover.ts +20 -3
  80. package/src/env.test.ts +23 -8
  81. package/src/env.ts +17 -8
  82. package/src/index.ts +16 -8
  83. package/src/notices.ts +1 -1
  84. package/src/openai.ts +1 -1
  85. package/src/providerDefaults.test.ts +50 -0
  86. package/src/sdkModels.test.ts +142 -8
  87. package/src/sdkModels.ts +201 -19
  88. package/src/types.ts +16 -0
@@ -20,11 +20,13 @@ import type {
20
20
  ProviderRequestSettlement,
21
21
  ProviderResponse,
22
22
  ProviderUsage,
23
+ ReasoningPolicy,
23
24
  } from "./types.ts";
24
25
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
25
- import type { JSONValue } from "ai";
26
- import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
27
- import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
26
+ import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
27
+ import type { CallWarning, JSONValue } from "ai";
28
+ import { MAX_PROVIDER_TIMEOUT_MS, type Reasoning, type ReasoningResponseStyle } from "./env.ts";
29
+ import { UnsupportedReasoningPolicyError } from "./types.ts";
28
30
  import {
29
31
  executeAiSdkModel,
30
32
  executeOpenAICompatible,
@@ -59,6 +61,25 @@ export type CacheAffinity =
59
61
 
60
62
  export type AiSdkProviderOptions = Record<string, Record<string, JSONValue | undefined>>;
61
63
 
64
+ const isJsonObject = (value: JSONValue | undefined): value is Record<string, JSONValue> =>
65
+ typeof value === "object" && value !== null && !Array.isArray(value);
66
+
67
+ const mergeJsonObjects = (
68
+ left: Record<string, JSONValue | undefined>,
69
+ right: Record<string, JSONValue | undefined>,
70
+ ): Record<string, JSONValue | undefined> => Object.fromEntries(
71
+ [...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
72
+ const leftValue = left[key];
73
+ const rightValue = right[key];
74
+ return [
75
+ key,
76
+ isJsonObject(leftValue) && isJsonObject(rightValue)
77
+ ? mergeJsonObjects(leftValue, rightValue)
78
+ : rightValue ?? leftValue,
79
+ ];
80
+ }),
81
+ );
82
+
62
83
  export type AiSdkProviderConfig = {
63
84
  model: string;
64
85
  url?: string; // OpenAI-compatible chat-completions URL
@@ -75,6 +96,11 @@ export type AiSdkProviderConfig = {
75
96
  maxOutputTokens?: number | null;
76
97
  outputBudget?: number | null;
77
98
  reasoningBudget?: number | null;
99
+ supportedReasoningPolicies?: readonly ReasoningPolicy[];
100
+ // Native AI SDK projection for adaptive. `high` is the graded fallback;
101
+ // provider-default is reserved for a documented native option/default.
102
+ adaptiveReasoning?: "high" | "provider-default";
103
+ adaptiveReasoningProviderOptions?: AiSdkProviderOptions;
78
104
  // Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
79
105
  // visible output and add an explicit provider reasoning budget. This marker lets the
80
106
  // adapter subtract that subset so the resulting wire cap remains PLURNK's
@@ -125,7 +151,7 @@ export type AiSdkProviderConfig = {
125
151
  requiresOutputBudget?: boolean;
126
152
  // The side-channel reasoning intent — REQUIRED, no in-code default
127
153
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
128
- // { mode: off|adaptive|on, budget: optional when on }. The provider maps it
154
+ // { mode: off|adaptive|low|medium|high, budget: independent optional cap }.
129
155
  // to the backend's mechanism via reasoningStyle; budget is only ever an
130
156
  // explicit magnitude, never a hidden activation flag.
131
157
  reasoning: Reasoning;
@@ -265,19 +291,32 @@ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
265
291
  return { content, reasoning: "", projected: false, contentStart: 0 };
266
292
  };
267
293
 
268
- // Shared budget→effort breakpoints (xai and google had identical copies).
269
- export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
270
- if (budget <= 1000) return "low";
271
- if (budget <= 4000) return "medium";
272
- return "high";
294
+ const fixedEffort = (mode: ReasoningPolicy): "low" | "medium" | "high" => {
295
+ if (mode === "low" || mode === "medium" || mode === "high") return mode;
296
+ throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
273
297
  };
274
298
 
275
- // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
276
- // is the neutral activation projection for an explicit, unqualified `on`; it
277
- // changes no PLURNK output budget. An operator reasoning subset, when present, remains
278
- // the only input to the existing magnitude-to-tier projection.
279
- const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
280
- reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
299
+ // Anthropic's older manual-reasoning protocol needs an absolute allowance while
300
+ // PLURNK's durable contract names an effort. These fractions match the native
301
+ // SDK's policy projection, but apply to PLURNK's total envelope rather than the
302
+ // model's physical maximum. The minimum is imposed by the provider protocol.
303
+ const MANUAL_REASONING_FRACTIONS = Object.freeze({
304
+ adaptive: 0.6,
305
+ low: 0.1,
306
+ medium: 0.3,
307
+ high: 0.6,
308
+ } satisfies Record<Exclude<ReasoningPolicy, "off">, number>);
309
+ const MANUAL_REASONING_MINIMUM = 1024;
310
+
311
+ const providerWarningMessage = (warning: CallWarning): string => {
312
+ switch (warning.type) {
313
+ case "unsupported":
314
+ case "compatibility":
315
+ return `${warning.type} ${warning.feature}${warning.details === undefined ? "" : `: ${warning.details}`}`;
316
+ case "deprecated": return `deprecated ${warning.setting}: ${warning.message}`;
317
+ case "other": return warning.message;
318
+ }
319
+ };
281
320
 
282
321
  // Body keys the provider owns — a caller's `sampling` passthrough may not set
283
322
  // these. Two families:
@@ -321,6 +360,9 @@ export default class AiSdkProvider implements Provider {
321
360
  #reasoningBudget: number | null;
322
361
  #additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
323
362
  #reasoning: Reasoning;
363
+ #supportedReasoningPolicies: readonly ReasoningPolicy[];
364
+ #adaptiveReasoning: "high" | "provider-default";
365
+ #adaptiveReasoningProviderOptions: AiSdkProviderOptions | undefined;
324
366
  #temperature: number;
325
367
  #repeatPenalty: number;
326
368
  #frequencyPenalty: number;
@@ -390,6 +432,18 @@ export default class AiSdkProvider implements Provider {
390
432
  this.#reasoningBudget = config.reasoningBudget ?? null;
391
433
  this.#additiveReasoningProvider = config.additiveReasoningProvider;
392
434
  this.#reasoning = config.reasoning;
435
+ this.#supportedReasoningPolicies = Object.freeze([
436
+ ...new Set(config.supportedReasoningPolicies ?? REASONING_POLICIES),
437
+ ]);
438
+ this.#adaptiveReasoning = config.adaptiveReasoning ?? "high";
439
+ this.#adaptiveReasoningProviderOptions = config.adaptiveReasoningProviderOptions;
440
+ if (!this.#supportedReasoningPolicies.includes(this.#reasoning.mode)) {
441
+ throw new UnsupportedReasoningPolicyError(
442
+ config.source ?? "provider",
443
+ this.#reasoning.mode,
444
+ this.#supportedReasoningPolicies,
445
+ );
446
+ }
393
447
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
394
448
  // required tuning fields must fail at construction, not silently send
395
449
  // undefined sampling on every grammar request.
@@ -435,6 +489,9 @@ export default class AiSdkProvider implements Provider {
435
489
  if (this.#languageModel === undefined && this.#reasoningResponseProviderOptions !== undefined) {
436
490
  throw new Error(`${this.#source}: reasoning response provider options require an AI SDK model`);
437
491
  }
492
+ if (this.#languageModel === undefined && this.#additiveReasoningProvider !== undefined) {
493
+ throw new Error(`${this.#source}: additive reasoning projection requires an AI SDK model`);
494
+ }
438
495
  if (this.#cacheAffinity?.target === "provider-option"
439
496
  && Object.hasOwn(
440
497
  this.#reasoningResponseProviderOptions?.[this.#cacheAffinity.provider] ?? {},
@@ -476,16 +533,12 @@ export default class AiSdkProvider implements Provider {
476
533
  throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
477
534
  }
478
535
  if (this.#reasoningStyle === "anthropic"
479
- && this.#reasoning.mode === "on"
536
+ && this.#reasoning.mode !== "off"
537
+ && this.#reasoning.mode !== "adaptive"
480
538
  && this.#reasoning.budget === null
481
539
  && this.#reasoningBudget === null) {
482
540
  throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
483
541
  }
484
- if (this.#additiveReasoningProvider !== undefined
485
- && this.#reasoning.mode === "on"
486
- && this.#reasoningBudget === null) {
487
- throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
488
- }
489
542
  if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
490
543
  throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
491
544
  }
@@ -515,6 +568,7 @@ export default class AiSdkProvider implements Provider {
515
568
  get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
516
569
  get outputBudget(): number | null { return this.#outputBudget; }
517
570
  get reasoningBudget(): number | null { return this.#reasoningBudget; }
571
+ get supportedReasoningPolicies(): readonly ReasoningPolicy[] { return this.#supportedReasoningPolicies; }
518
572
  get inputCapacity(): number | null {
519
573
  return effectiveInputCapacity({
520
574
  contextWindow: this.#contextWindow,
@@ -628,7 +682,7 @@ export default class AiSdkProvider implements Provider {
628
682
  case "template": {
629
683
  const allowance = mode === "off"
630
684
  ? 0
631
- : mode === "on" && budget !== null ? budget : this.#reasoningBudget;
685
+ : budget;
632
686
  return {
633
687
  chat_template_kwargs: { enable_thinking: on },
634
688
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -637,9 +691,9 @@ export default class AiSdkProvider implements Provider {
637
691
  }
638
692
  case "think": return on ? { think: true } : {};
639
693
  case "include_reasoning": return on ? { include_reasoning: true } : {};
640
- // Explicit on uses the portable enabled posture or a tier derived
641
- // from an explicit budget; off/adaptive omit the field.
642
- case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
694
+ case "effort": return mode === "off"
695
+ ? {}
696
+ : { reasoning_effort: mode === "adaptive" ? "high" : fixedEffort(mode) };
643
697
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
644
698
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
645
699
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -649,25 +703,23 @@ export default class AiSdkProvider implements Provider {
649
703
  // efforts 400.
650
704
  case "effort_explicit": return mode === "off"
651
705
  ? { reasoning_effort: "none" }
652
- : mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
706
+ : mode === "adaptive" ? {} : { reasoning_effort: fixedEffort(mode) };
653
707
  // {§deepseek-reasoning-request}
654
708
  case "thinking_effort": return mode === "off"
655
709
  ? { thinking: { type: "disabled" } }
656
- : mode === "on" ? {
710
+ : mode === "adaptive" ? { thinking: { type: "enabled" } } : {
657
711
  thinking: { type: "enabled" },
658
- ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
659
- } : {};
660
- // Anthropic compat: explicit thinking object. off → disabled; on →
661
- // enabled with the explicit reasoning subset; adaptive →
662
- // omit (the API default).
712
+ reasoning_effort: fixedEffort(mode),
713
+ };
714
+ // Anthropic-compatible native dynamic or manual budget mode.
663
715
  case "anthropic": return mode === "off"
664
716
  ? { thinking: { type: "disabled" } }
665
- : mode === "on" ? {
717
+ : mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
666
718
  thinking: {
667
719
  type: "enabled",
668
720
  budget_tokens: budget!,
669
721
  },
670
- } : {};
722
+ };
671
723
  case "none": return {};
672
724
  }
673
725
  }
@@ -809,22 +861,26 @@ export default class AiSdkProvider implements Provider {
809
861
 
810
862
  #requestProviderOptions(
811
863
  workerId: string,
812
- reasoningBudget: number | null,
864
+ nativeReasoningBudget: number | null,
813
865
  ): AiSdkProviderOptions | undefined {
814
866
  const responseOptions = this.#reasoning.mode === "off"
815
867
  ? undefined
816
868
  : this.#reasoningResponseProviderOptions;
817
- const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
869
+ const adaptiveOptions = this.#reasoning.mode === "adaptive"
870
+ && nativeReasoningBudget === null
871
+ ? this.#adaptiveReasoningProviderOptions
872
+ : undefined;
873
+ const nativeReasoning = nativeReasoningBudget !== null
818
874
  ? this.#additiveReasoningProvider === "anthropic"
819
- ? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
875
+ ? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
820
876
  : this.#additiveReasoningProvider === "bedrock"
821
- ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
877
+ ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
822
878
  : undefined
823
879
  : undefined;
824
880
  const options: AiSdkProviderOptions = {};
825
- for (const part of [responseOptions, nativeReasoning]) {
881
+ for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
826
882
  for (const [provider, values] of Object.entries(part ?? {})) {
827
- options[provider] = { ...options[provider], ...values };
883
+ options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
828
884
  }
829
885
  }
830
886
  if (this.#cacheAffinity?.target === "provider-option") {
@@ -836,16 +892,38 @@ export default class AiSdkProvider implements Provider {
836
892
 
837
893
  #nativeMaxOutputTokens(
838
894
  outputBudget: number | null,
839
- reasoningBudget: number | null,
895
+ nativeReasoningBudget: number | null,
840
896
  ): number | undefined {
841
897
  if (outputBudget === null) return undefined;
842
- return this.#additiveReasoningProvider !== undefined
843
- && this.#reasoning.mode === "on"
844
- && reasoningBudget !== null
845
- ? outputBudget - reasoningBudget
898
+ return nativeReasoningBudget !== null
899
+ ? outputBudget - nativeReasoningBudget
846
900
  : outputBudget;
847
901
  }
848
902
 
903
+ #nativeReasoningBudget(
904
+ outputBudget: number | null,
905
+ configuredReasoningBudget: number | null,
906
+ ): number | null {
907
+ if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off") return null;
908
+ if (configuredReasoningBudget !== null) return configuredReasoningBudget;
909
+ if (this.#adaptiveReasoningProviderOptions !== undefined) return null;
910
+ if (outputBudget === null) {
911
+ throw new TypeError(
912
+ `${this.#source}: manual provider reasoning requires a resolved total output budget`,
913
+ );
914
+ }
915
+ if (outputBudget <= MANUAL_REASONING_MINIMUM) {
916
+ throw new TypeError(
917
+ `${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`,
918
+ );
919
+ }
920
+ const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
921
+ return Math.min(
922
+ outputBudget - 1,
923
+ Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)),
924
+ );
925
+ }
926
+
849
927
  #accounting(
850
928
  outcome: ProviderRequestAccounting["outcome"],
851
929
  usage: ProviderUsage | undefined,
@@ -895,6 +973,10 @@ export default class AiSdkProvider implements Provider {
895
973
  );
896
974
  }
897
975
  const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
976
+ const nativeReasoningBudget = this.#nativeReasoningBudget(
977
+ capacity.outputBudget,
978
+ capacity.reasoningBudget,
979
+ );
898
980
 
899
981
  // Assembly order = precedence: the family's sampling DEFAULTS
900
982
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
@@ -1008,7 +1090,7 @@ export default class AiSdkProvider implements Provider {
1008
1090
  : await executeAiSdkModel({
1009
1091
  languageModel: this.#languageModel,
1010
1092
  headers: requestHeaders,
1011
- providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
1093
+ providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
1012
1094
  systemProviderOptions: this.#systemCacheProviderOptions,
1013
1095
  messages,
1014
1096
  signal: operationSignal,
@@ -1032,17 +1114,12 @@ export default class AiSdkProvider implements Provider {
1032
1114
  ? sampling.stop
1033
1115
  : undefined,
1034
1116
  seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
1035
- maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
1117
+ maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
1036
1118
  reasoning: this.#reasoning.mode === "off"
1037
1119
  ? "none"
1038
1120
  : this.#reasoning.mode === "adaptive"
1039
- ? "provider-default"
1040
- : this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
1041
- ? "provider-default"
1042
- : effortFromReasoning({
1043
- mode: this.#reasoning.mode,
1044
- budget: capacity.reasoningBudget,
1045
- }),
1121
+ ? this.#adaptiveReasoning
1122
+ : fixedEffort(this.#reasoning.mode),
1046
1123
  });
1047
1124
  } catch (error) {
1048
1125
  const failure = transportFailureEvidence(error);
@@ -1152,6 +1229,15 @@ export default class AiSdkProvider implements Provider {
1152
1229
  }
1153
1230
 
1154
1231
  let notices: ProviderNotice[] | undefined;
1232
+ for (const warning of raw.warnings) {
1233
+ (notices ??= []).push({
1234
+ source: this.#source,
1235
+ kind: "provider_warning",
1236
+ level: "warn",
1237
+ message: providerWarningMessage(warning),
1238
+ position: null,
1239
+ });
1240
+ }
1155
1241
  const usage = raw.usage;
1156
1242
  if (sendGrammar !== undefined
1157
1243
  && this.tokenize !== undefined
package/src/Mock.ts CHANGED
@@ -7,6 +7,7 @@
7
7
 
8
8
  import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
9
9
  import { resolveGenerationEnvelopeFromEnv } from "./env.ts";
10
+ import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
10
11
  import { validateProviderRequestAccounting } from "./accounting.ts";
11
12
  import { ProviderError } from "./errors.ts";
12
13
  import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
@@ -69,6 +70,7 @@ export default class Mock implements Provider {
69
70
  get maxOutputTokens(): number | null { return null; }
70
71
  get outputBudget(): number | null { return this.#outputBudget; }
71
72
  get reasoningBudget(): number | null { return this.#reasoningBudget; }
73
+ get supportedReasoningPolicies() { return REASONING_POLICIES; }
72
74
  get inputCapacity(): number | null {
73
75
  return effectiveInputCapacity({
74
76
  contextWindow: this.#contextWindow,
package/src/Pool.test.ts CHANGED
@@ -53,6 +53,7 @@ const backend = (opts: FakeOpts = {}) => {
53
53
  maxOutputTokens: opts.maxOutputTokens ?? null,
54
54
  outputBudget: opts.outputBudget ?? null,
55
55
  reasoningBudget: opts.reasoningBudget ?? null,
56
+ supportedReasoningPolicies: ["off", "adaptive", "low", "medium", "high"],
56
57
  inputCapacity: effectiveInputCapacity({
57
58
  contextWindow: opts.window === undefined ? 48_000 : opts.window,
58
59
  maxInputTokens: opts.maxInputTokens ?? null,
package/src/Pool.ts CHANGED
@@ -7,6 +7,7 @@ import Meta, {
7
7
  type PluginAttributionContext,
8
8
  } from "@plurnk/plurnk-meta";
9
9
  import { effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, requestCapacityDecision } from "./capacity.ts";
10
+ import type { ReasoningPolicy } from "./types.ts";
10
11
 
11
12
  // A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
12
13
  // transient retries before throwing one of these, so re-hitting the same
@@ -89,6 +90,10 @@ export default class Pool implements Provider {
89
90
  get maxOutputTokens(): number | null { return this.#minimumKnown((provider) => provider.maxOutputTokens); }
90
91
  get outputBudget(): number | null { return this.#minimumKnown((provider) => provider.outputBudget); }
91
92
  get reasoningBudget(): number | null { return this.#minimumKnown((provider) => provider.reasoningBudget); }
93
+ get supportedReasoningPolicies(): readonly ReasoningPolicy[] {
94
+ return this.#backends[0].supportedReasoningPolicies.filter((policy) =>
95
+ this.#backends.every((provider) => provider.supportedReasoningPolicies.includes(policy)));
96
+ }
92
97
  get inputCapacity(): number | null { return this.#minimumKnown((provider) => provider.inputCapacity); }
93
98
 
94
99
  // Served id / capabilities aggregate CONSERVATIVELY: a worker could land on any
@@ -4,7 +4,7 @@ import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./
4
4
  import type { PluginAttributionContext } from "@plurnk/plurnk-meta";
5
5
 
6
6
  const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
7
- async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
7
+ async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>(), grammarStyles: new Map() });
8
8
 
9
9
  // Alias parsing is tested in @plurnk/plurnk-aliases (its owner). Here we
10
10
  // exercise the resolution + two-tier instantiation this module owns; the active
@@ -31,7 +31,7 @@ test("instantiateProvider: cataloged name resolves in-framework, no scan, no imp
31
31
  let scanned = false;
32
32
  const p = await instantiateProvider("openai", { ...fullEnv }, "m",
33
33
  async (s) => { imports.push(s); return {}; },
34
- async () => { scanned = true; return { registry: new Map(), skipped: new Map(), attributions: new Map() }; });
34
+ async () => { scanned = true; return { registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }; });
35
35
  assert.equal(p.model, "m");
36
36
  assert.deepEqual(imports, []); // tier 1 never touches the importer…
37
37
  assert.equal(scanned, false); // …nor the scan
@@ -77,7 +77,7 @@ test("instantiateProvider: a selected plugin composes its static and runtime att
77
77
  registry: new Map([["acme", "@acme/ai-provider"]]),
78
78
  skipped: new Map(),
79
79
  attributions: new Map([["acme", "static:provider"]]),
80
- packageAttributions: new Map([["@acme/ai-provider", ["static:provider"]]]),
80
+ packageAttributions: new Map([["@acme/ai-provider", ["static:provider"]]]), grammarStyles: new Map(),
81
81
  }),
82
82
  );
83
83
 
@@ -107,7 +107,7 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
107
107
  return new Response(JSON.stringify({ data: [] }), { status: 200 });
108
108
  });
109
109
  await instantiateProvider("openai", { ...fullEnv }, "m", // fullEnv.OPENAI_BASE_URL is http://x — the override must win
110
- async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
110
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
111
111
  "http://hazel2:8080/v1");
112
112
  assert.ok(probed.some((u) => u === "http://hazel2:8080/v1/models"), `probe hit the override host; saw ${probed.join(", ")}`);
113
113
  assert.equal(probed.some((u) => u.startsWith("http://x")), false); // never the per-name OPENAI_BASE_URL
@@ -193,13 +193,13 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
193
193
  });
194
194
  const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW_turbo: "12345", PLURNK_PROVIDERS_LLAMA_SERVER_turbo: "1" };
195
195
  const p = await instantiateProvider("openai", env, "m",
196
- async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
196
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
197
197
  undefined, "turbo");
198
198
  assert.equal(p.contextWindow, 12345); // _turbo CONTEXT_WINDOW reached the provider
199
199
  assert.equal(p.constrainsOutput, true); // _turbo LLAMA_SERVER pin reached it too
200
200
  // same env, DIFFERENT alias: neither override applies
201
201
  const q = await instantiateProvider("openai", env, "m",
202
- async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
202
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() }),
203
203
  undefined, "plain");
204
204
  assert.equal(q.contextWindow, null);
205
205
  assert.equal(q.constrainsOutput, false);
@@ -310,7 +310,7 @@ test("a catalog provider with unknown model metadata never falls through to plug
310
310
  "cloudflare",
311
311
  {
312
312
  CLOUDFLARE_ACCOUNT_ID: "account",
313
- CLOUDFLARE_API_TOKEN: "token",
313
+ CLOUDFLARE_API_KEY: "token",
314
314
  },
315
315
  "vendor/model-outside-snapshot",
316
316
  async () => { throw new Error("plugin import must not run"); },
@@ -319,7 +319,7 @@ test("a catalog provider with unknown model metadata never falls through to plug
319
319
  return {
320
320
  registry: new Map([["cloudflare", "@plurnk/plurnk-providers-cloudflare"]]),
321
321
  skipped: new Map(),
322
- attributions: new Map(),
322
+ attributions: new Map(), grammarStyles: new Map(),
323
323
  };
324
324
  },
325
325
  ),
@@ -334,7 +334,7 @@ test("explicit metadata constructs an out-of-snapshot Cloudflare model in the co
334
334
  "cloudflare",
335
335
  {
336
336
  CLOUDFLARE_ACCOUNT_ID: "account",
337
- CLOUDFLARE_API_TOKEN: "token",
337
+ CLOUDFLARE_API_KEY: "token",
338
338
  PLURNK_PROVIDERS_CONTEXT_WINDOW: "128000",
339
339
  },
340
340
  "vendor/model-outside-snapshot",
@@ -344,7 +344,7 @@ test("explicit metadata constructs an out-of-snapshot Cloudflare model in the co
344
344
  return {
345
345
  registry: new Map([["cloudflare", "@plurnk/plurnk-providers-cloudflare"]]),
346
346
  skipped: new Map(),
347
- attributions: new Map(),
347
+ attributions: new Map(), grammarStyles: new Map(),
348
348
  };
349
349
  },
350
350
  );
@@ -382,7 +382,7 @@ test("{§provider-tagged-reasoning} a Cloudflare model alias carries its explici
382
382
  {
383
383
  ...fullEnv,
384
384
  CLOUDFLARE_ACCOUNT_ID: "account",
385
- CLOUDFLARE_API_TOKEN: "token",
385
+ CLOUDFLARE_API_KEY: "token",
386
386
  PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE: "verbatim",
387
387
  PLURNK_PROVIDERS_REASONING_RESPONSE_STYLE_cfds1: "think-tags",
388
388
  },
@@ -406,7 +406,17 @@ test("two Fireworks aliases independently select default and priority service ti
406
406
  const bodies: Record<string, unknown>[] = [];
407
407
  mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
408
408
  bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
409
- return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
409
+ const chunk = {
410
+ id: "fireworks-test",
411
+ object: "chat.completion.chunk",
412
+ created: 1,
413
+ model: "fireworks-test",
414
+ choices: [{ index: 0, delta: { content: "ok" }, finish_reason: "stop" }],
415
+ };
416
+ return new Response(`data: ${JSON.stringify(chunk)}\n\ndata: [DONE]\n\n`, {
417
+ status: 200,
418
+ headers: { "content-type": "text/event-stream" },
419
+ });
410
420
  });
411
421
  const env = {
412
422
  ...fullEnv,
@@ -419,7 +429,7 @@ test("two Fireworks aliases independently select default and priority service ti
419
429
  PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
420
430
  };
421
431
  const imports = async () => ({});
422
- const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
432
+ const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map(), grammarStyles: new Map() });
423
433
  const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
424
434
  const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
425
435
  await fast.generate({ workerId: "fast-worker", messages: [] });
@@ -445,5 +455,8 @@ test("loadActiveProvider: resolves the alias cascade to an installed AI SDK prov
445
455
  });
446
456
 
447
457
  test("loadActiveProvider: throws a named error when no alias is active", async () => {
448
- await assert.rejects(() => loadActiveProvider({ ...fullEnv }), /set PLURNK_MODEL to an alias/);
458
+ await assert.rejects(
459
+ () => loadActiveProvider({ ...fullEnv }),
460
+ /set PLURNK_MODEL to a declared alias or provider\/model route/,
461
+ );
449
462
  });
@@ -1,7 +1,7 @@
1
- // Provider instantiation + active-alias resolution. Alias PARSING (the
1
+ // Provider instantiation + active model-route resolution. Alias parsing (the
2
2
  // PLURNK_MODEL_<alias>=<provider>/<model> cascade + PLURNK_BASEURL_<alias>
3
- // overrides) lives in @plurnk/plurnk-aliases — the zero-dep parser shared with
4
- // thin clients; this module resolves the active alias to a Provider.
3
+ // overrides) lives in @plurnk/plurnk-aliases; this module resolves the active
4
+ // alias-or-route selector to a Provider.
5
5
  //
6
6
  // {§provider-resolution} Models.dev catalog → PLURNK provider declaration
7
7
  // → local protocol adapter → scope-agnostic AI SDK plugin discovery. Generic
@@ -11,7 +11,7 @@
11
11
  import type { AiSdkProviderPlugin, Provider } from "./types.ts";
12
12
  import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
13
13
  import { discover, type DiscoverOptions, type Discovery } from "./discover.ts";
14
- import { resolveActiveAlias } from "@plurnk/plurnk-aliases";
14
+ import { resolveActiveRoute } from "@plurnk/plurnk-aliases";
15
15
  import { scopeEnvToAlias } from "./env.ts";
16
16
  import { ollamaProviderFromEnv } from "./ollama.ts";
17
17
  import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
@@ -61,7 +61,7 @@ export const instantiateProvider = async (
61
61
  if (catalog !== null) return catalog;
62
62
  if (name === "ollama") return ollamaProviderFromEnv(env, model, baseUrl === undefined ? undefined : { baseUrl });
63
63
  if (name === "openai" || name === "plurnk") return compatibleProviderFromEnv(name, env, model, baseUrl);
64
- const { registry, skipped, packageAttributions = new Map() } = await providerPackages(discoverFn, env);
64
+ const { registry, skipped, packageAttributions = new Map(), grammarStyles = new Map() } = await providerPackages(discoverFn, env);
65
65
  const specifier = registry.get(name);
66
66
  if (specifier === undefined) {
67
67
  const declined = skipped.get(name);
@@ -99,21 +99,30 @@ export const instantiateProvider = async (
99
99
  languageModel: sdkProvider.languageModel(model),
100
100
  contextWindow,
101
101
  attributions,
102
+ ...(grammarStyles.get(name) === undefined ? {} : { grammarStyle: grammarStyles.get(name) }),
102
103
  });
103
104
  };
104
105
 
105
106
  // Test-only: drop the memoized discovery so a fresh scan/injection runs next.
106
107
  export const resetDiscoveryCache = (): void => { discoveredCache = null; };
107
108
 
108
- // Boot convenience: resolve the active alias cascade and instantiate it.
109
+ // Boot convenience: resolve the active selector and instantiate its exact route.
109
110
  export const loadActiveProvider = async (
110
111
  env: NodeJS.ProcessEnv = process.env,
111
112
  importImpl: ImportModule = importModule,
112
113
  discoverFn: DiscoverFn = discover,
113
114
  ): Promise<Provider> => {
114
- const alias = resolveActiveAlias(env);
115
- if (alias === null) {
116
- throw new Error("no active provider: set PLURNK_MODEL to an alias declared via PLURNK_MODEL_<alias>=<provider>/<model>");
115
+ const route = resolveActiveRoute(env);
116
+ if (route === null) {
117
+ throw new Error("no active provider: set PLURNK_MODEL to a declared alias or provider/model route");
117
118
  }
118
- return instantiateProvider(alias.provider, env, alias.model, importImpl, discoverFn, alias.baseUrl, alias.alias);
119
+ return instantiateProvider(
120
+ route.provider,
121
+ env,
122
+ route.model,
123
+ importImpl,
124
+ discoverFn,
125
+ route.baseUrl,
126
+ route.alias,
127
+ );
119
128
  };
@@ -89,6 +89,10 @@ test("aggregateProviderAccounting preserves request order and only sums known fi
89
89
  },
90
90
  ]);
91
91
  assert.deepEqual(accounting.requests.map(({ provider }) => provider), ["provider:a", "provider:b"]);
92
- assert.equal(accounting.usage, null, "an unknown failed request prevents fabricated totals");
93
- assert.equal(accounting.costUsd, null);
92
+ assert.deepEqual(accounting.usage, {
93
+ inputTokens: 2,
94
+ outputTokens: 3,
95
+ totalTokens: 5,
96
+ }, "a response-less failure is skipped, never allowed to erase reported usage");
97
+ assert.equal(accounting.costUsd, "0.25", "a response-less failure is skipped; the expressible cost survives");
94
98
  });
package/src/accounting.ts CHANGED
@@ -126,12 +126,13 @@ const sumKnown = (
126
126
  requests: readonly ProviderRequestAccounting[],
127
127
  read: (usage: ProviderUsage) => number | undefined,
128
128
  ): number | undefined => {
129
- const values = requests.map((request) => request.usage === undefined
130
- ? undefined
131
- : read(request.usage));
132
- return values.some((value) => value === undefined)
133
- ? undefined
134
- : (values as number[]).reduce((sum, value) => sum + value, 0);
129
+ // {§tokenomics-provider-usage} — aggregate usage sums every reported
130
+ // quantity; an unreported one (a response-less failure) is skipped, never
131
+ // invented as zero and never allowed to erase the reported evidence.
132
+ const known = requests
133
+ .map((request) => request.usage === undefined ? undefined : read(request.usage))
134
+ .filter((value): value is number => value !== undefined);
135
+ return known.length === 0 ? undefined : known.reduce((sum, value) => sum + value, 0);
135
136
  };
136
137
 
137
138
  export const aggregateProviderAccounting = (