@plurnk/plurnk-providers 1.6.0 → 1.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/.env.defaults +9 -16
  2. package/README.md +17 -0
  3. package/SPEC.md +128 -45
  4. package/dist/AiSdkProvider.d.ts +16 -9
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +151 -54
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +8 -4
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +53 -18
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -4
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +68 -12
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/accounting.d.ts.map +1 -1
  17. package/dist/accounting.js.map +1 -1
  18. package/dist/accountingPublic.d.ts +5 -0
  19. package/dist/accountingPublic.d.ts.map +1 -0
  20. package/dist/accountingPublic.js +3 -0
  21. package/dist/accountingPublic.js.map +1 -0
  22. package/dist/capacity.d.ts +26 -0
  23. package/dist/capacity.d.ts.map +1 -0
  24. package/dist/capacity.js +90 -0
  25. package/dist/capacity.js.map +1 -0
  26. package/dist/catalogProvider.d.ts +2 -1
  27. package/dist/catalogProvider.d.ts.map +1 -1
  28. package/dist/catalogProvider.js +18 -20
  29. package/dist/catalogProvider.js.map +1 -1
  30. package/dist/compatibleProvider.d.ts.map +1 -1
  31. package/dist/compatibleProvider.js +10 -7
  32. package/dist/compatibleProvider.js.map +1 -1
  33. package/dist/env.d.ts +8 -10
  34. package/dist/env.d.ts.map +1 -1
  35. package/dist/env.js +54 -37
  36. package/dist/env.js.map +1 -1
  37. package/dist/errors.d.ts +6 -22
  38. package/dist/errors.d.ts.map +1 -1
  39. package/dist/errors.js +30 -91
  40. package/dist/errors.js.map +1 -1
  41. package/dist/index.d.ts +4 -3
  42. package/dist/index.d.ts.map +1 -1
  43. package/dist/index.js +2 -1
  44. package/dist/index.js.map +1 -1
  45. package/dist/promptTokens.d.ts.map +1 -1
  46. package/dist/promptTokens.js +7 -4
  47. package/dist/promptTokens.js.map +1 -1
  48. package/dist/providerError.d.ts +25 -0
  49. package/dist/providerError.d.ts.map +1 -0
  50. package/dist/providerError.js +91 -0
  51. package/dist/providerError.js.map +1 -0
  52. package/dist/sdkModels.d.ts +1 -0
  53. package/dist/sdkModels.d.ts.map +1 -1
  54. package/dist/sdkModels.js +5 -8
  55. package/dist/sdkModels.js.map +1 -1
  56. package/dist/types.d.ts +24 -4
  57. package/dist/types.d.ts.map +1 -1
  58. package/dist/usage.d.ts +1 -0
  59. package/dist/usage.d.ts.map +1 -1
  60. package/dist/usage.js +7 -2
  61. package/dist/usage.js.map +1 -1
  62. package/package.json +22 -7
  63. package/src/AiSdkProvider.test.ts +198 -37
  64. package/src/AiSdkProvider.ts +192 -59
  65. package/src/Mock.test.ts +32 -18
  66. package/src/Mock.ts +58 -19
  67. package/src/Pool.test.ts +71 -13
  68. package/src/Pool.ts +78 -13
  69. package/src/ProviderRegistry.test.ts +1 -1
  70. package/src/accounting.ts +0 -1
  71. package/src/accountingPublic.ts +9 -0
  72. package/src/boundaries.test.ts +28 -15
  73. package/src/capacity.test.ts +92 -0
  74. package/src/capacity.ts +140 -0
  75. package/src/catalogProvider.test.ts +82 -9
  76. package/src/catalogProvider.ts +24 -21
  77. package/src/compatibleProvider.test.ts +1 -2
  78. package/src/compatibleProvider.ts +10 -7
  79. package/src/cost.test.ts +31 -0
  80. package/src/env.test.ts +49 -20
  81. package/src/env.ts +114 -51
  82. package/src/errors.test.ts +33 -0
  83. package/src/errors.ts +38 -134
  84. package/src/index.ts +5 -2
  85. package/src/ollama.test.ts +1 -2
  86. package/src/promptTokens.ts +8 -5
  87. package/src/providerError.ts +139 -0
  88. package/src/sdkModels.test.ts +2 -5
  89. package/src/sdkModels.ts +6 -8
  90. package/src/types.ts +41 -19
  91. package/src/usage.ts +7 -2
@@ -11,11 +11,12 @@ import type {
11
11
  GrammarEvidence,
12
12
  PromptTokenMeasurement,
13
13
  Provider,
14
+ ProviderAttempt,
14
15
  ProviderCostNormalizer,
15
16
  ProviderCallKind,
16
17
  ProviderGenerateArgs,
17
18
  ProviderRequestAccounting,
18
- ProviderRequestObserver,
19
+ ProviderRequestCapacity,
19
20
  ProviderRequestSettlement,
20
21
  ProviderResponse,
21
22
  ProviderUsage,
@@ -23,7 +24,7 @@ import type {
23
24
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
24
25
  import type { JSONValue } from "ai";
25
26
  import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
26
- import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
27
+ import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
27
28
  import {
28
29
  executeAiSdkModel,
29
30
  executeOpenAICompatible,
@@ -40,6 +41,7 @@ import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk
40
41
  import { resolveProviderCost } from "./cost.ts";
41
42
  import { validateProviderRequestAccounting } from "./accounting.ts";
42
43
  import { validateProviderUsage } from "./usage.ts";
44
+ import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
43
45
 
44
46
  export type ProviderFetch = typeof globalThis.fetch;
45
47
 
@@ -69,6 +71,15 @@ export type AiSdkProviderConfig = {
69
71
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
70
72
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
71
73
  contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
74
+ maxInputTokens?: number | null;
75
+ maxOutputTokens?: number | null;
76
+ outputBudget?: number | null;
77
+ reasoningBudget?: number | null;
78
+ // Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
79
+ // visible output and add an explicit provider reasoning budget. This marker lets the
80
+ // adapter subtract that subset so the resulting wire cap remains PLURNK's
81
+ // one total output budget.
82
+ additiveReasoningProvider?: "anthropic" | "bedrock";
72
83
  reasoningStyle?: ReasoningStyle; // default "none"
73
84
  reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
74
85
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
@@ -109,9 +120,9 @@ export type AiSdkProviderConfig = {
109
120
  // when no probe ran or it read no row.
110
121
  servedModel?: string;
111
122
  // Backend decodes unbounded without a caller cap (llama-server n_predict
112
- // to the wall) — surfaced as Provider.requiresMaxTokens so consumers can
123
+ // to the wall) — surfaced as Provider.requiresOutputBudget so consumers can
113
124
  // boot-refuse an envelope-less local alias. Default unset (no claim).
114
- requiresMaxTokens?: boolean;
125
+ requiresOutputBudget?: boolean;
115
126
  // The side-channel reasoning intent — REQUIRED, no in-code default
116
127
  // (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
117
128
  // { mode: off|adaptive|on, budget: optional when on }. The provider maps it
@@ -157,14 +168,12 @@ export type AiSdkProviderConfig = {
157
168
  // gated per-alias.
158
169
  topLogprobs?: number | null;
159
170
  rawBody?: boolean;
160
- // {§provider-generation-envelope} The generation-envelope reserves, env-read via
161
- // envelopeFromEnv — a percentage of the DETECTED window or an absolute token
171
+ // {§provider-generation-envelope} The generation budgets, env-read via the
172
+ // common envelope parser — a percentage of the detected window or an absolute token
162
173
  // count. Optional so an out-of-date sibling keeps constructing (no claim);
163
174
  // the standard factory always supplies them. Resolved against contextWindow
164
175
  // at read time (getters), so a probe that lands after config assembly still
165
176
  // derives correctly.
166
- reasoningReserve?: ReserveSpec;
167
- completionReserve?: ReserveSpec;
168
177
  // The plurnk.ai router owns tuning — false suppresses the
169
178
  // client-side temperature/penalty FLOORS on this provider (caller `sampling`
170
179
  // still passes through verbatim). Default true (floors ride).
@@ -265,7 +274,7 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
265
274
 
266
275
  // AI SDK's portable reasoning control has no boolean-enabled value. `medium`
267
276
  // is the neutral activation projection for an explicit, unqualified `on`; it
268
- // changes no PLURNK token reserve. An operator budget, when present, remains
277
+ // changes no PLURNK output budget. An operator reasoning subset, when present, remains
269
278
  // the only input to the existing magnitude-to-tier projection.
270
279
  const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
271
280
  reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
@@ -278,7 +287,7 @@ const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
278
287
  // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
279
288
  // body doctrine, §2: native tool_calls return null content = a broken turn),
280
289
  // modalities/audio (text-only contract), prediction (decode semantics, not
281
- // sampling), and the token caps (the envelope is the managed maxTokens —
290
+ // sampling), and the token caps (the envelope is the managed maxOutputTokens —
282
291
  // sampling must not bypass the consumer's cap).
283
292
  // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
284
293
  // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
@@ -306,6 +315,11 @@ export default class AiSdkProvider implements Provider {
306
315
  #apiKeyRejectedMessage: string | undefined;
307
316
  #eosText: string | undefined;
308
317
  #contextWindow: number | null;
318
+ #maxInputTokens: number | null;
319
+ #maxOutputTokens: number | null;
320
+ #outputBudget: number | null;
321
+ #reasoningBudget: number | null;
322
+ #additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
309
323
  #reasoning: Reasoning;
310
324
  #temperature: number;
311
325
  #repeatPenalty: number;
@@ -334,12 +348,10 @@ export default class AiSdkProvider implements Provider {
334
348
  #retryAttempts: number;
335
349
  #errorDetailLimit: number | undefined;
336
350
  #topLogprobs: number | null;
337
- #reasoningReserve: ReserveSpec | undefined;
338
- #completionReserve: ReserveSpec | undefined;
339
351
  #tuningFloors: boolean;
340
352
  #rawBody: boolean;
341
353
  #servedModel: string | undefined;
342
- #requiresMaxTokens: boolean | undefined;
354
+ #requiresOutputBudget: boolean | undefined;
343
355
  readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
344
356
 
345
357
  // Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
@@ -372,6 +384,11 @@ export default class AiSdkProvider implements Provider {
372
384
  this.#headers = config.headers ?? {};
373
385
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
374
386
  this.#contextWindow = config.contextWindow ?? null;
387
+ this.#maxInputTokens = config.maxInputTokens ?? null;
388
+ this.#maxOutputTokens = config.maxOutputTokens ?? null;
389
+ this.#outputBudget = config.outputBudget ?? null;
390
+ this.#reasoningBudget = config.reasoningBudget ?? null;
391
+ this.#additiveReasoningProvider = config.additiveReasoningProvider;
375
392
  this.#reasoning = config.reasoning;
376
393
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
377
394
  // required tuning fields must fail at construction, not silently send
@@ -435,25 +452,42 @@ export default class AiSdkProvider implements Provider {
435
452
  this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
436
453
  this.#slotCount = config.slotCount ?? null;
437
454
  this.#topLogprobs = config.topLogprobs ?? null;
438
- this.#reasoningReserve = config.reasoningReserve;
439
- this.#completionReserve = config.completionReserve;
440
455
  this.#tuningFloors = config.tuningFloors ?? true;
441
456
  this.#rawBody = config.rawBody ?? false;
442
457
  this.#servedModel = config.servedModel;
443
- this.#requiresMaxTokens = config.requiresMaxTokens;
444
- const reasoningReserve = this.reasoningReserve;
445
- if (this.#reasoningStyle === "template"
446
- && this.#reasoning.mode === "on"
447
- && this.#reasoning.budget !== null
448
- && reasoningReserve !== null
449
- && this.#reasoning.budget > reasoningReserve) {
450
- throw new Error(`${this.#source}: PLURNK_PROVIDERS_REASONING_BUDGET (${this.#reasoning.budget}) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE (${reasoningReserve})`);
458
+ this.#requiresOutputBudget = config.requiresOutputBudget;
459
+ for (const [name, value] of [
460
+ ["contextWindow", this.#contextWindow],
461
+ ["maxInputTokens", this.#maxInputTokens],
462
+ ["maxOutputTokens", this.#maxOutputTokens],
463
+ ["outputBudget", this.#outputBudget],
464
+ ["reasoningBudget", this.#reasoningBudget],
465
+ ] as const) {
466
+ if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
467
+ throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
468
+ }
469
+ }
470
+ if (this.#reasoningBudget !== null
471
+ && this.#outputBudget !== null
472
+ && this.#reasoningBudget >= this.#outputBudget) {
473
+ throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
474
+ }
475
+ if (this.#reasoning.budget !== this.#reasoningBudget) {
476
+ throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
451
477
  }
452
478
  if (this.#reasoningStyle === "anthropic"
453
479
  && this.#reasoning.mode === "on"
454
480
  && this.#reasoning.budget === null
455
- && reasoningReserve === null) {
456
- throw new Error(`${this.#source}: explicit Anthropic reasoning requires a resolved reasoning reserve or PLURNK_PROVIDERS_REASONING_BUDGET`);
481
+ && this.#reasoningBudget === null) {
482
+ throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
483
+ }
484
+ if (this.#additiveReasoningProvider !== undefined
485
+ && this.#reasoning.mode === "on"
486
+ && this.#reasoningBudget === null) {
487
+ throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
488
+ }
489
+ if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
490
+ throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
457
491
  }
458
492
  const { tokenizeUrl } = config;
459
493
  if (tokenizeUrl !== undefined) {
@@ -477,20 +511,22 @@ export default class AiSdkProvider implements Provider {
477
511
  }
478
512
 
479
513
  get contextWindow(): number | null { return this.#contextWindow; }
480
- // {§provider-generation-envelope} Envelope reserves — absolute pins stand alone; percentages need the
481
- // detected window; null = underivable (no claim for core's no-cap path).
482
- #resolveReserve(spec: ReserveSpec | undefined): number | null {
483
- if (spec === undefined) return null;
484
- if ("tokens" in spec) return spec.tokens;
485
- return this.#contextWindow === null ? null : Math.round(spec.percent * this.#contextWindow);
514
+ get maxInputTokens(): number | null { return this.#maxInputTokens; }
515
+ get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
516
+ get outputBudget(): number | null { return this.#outputBudget; }
517
+ get reasoningBudget(): number | null { return this.#reasoningBudget; }
518
+ get inputCapacity(): number | null {
519
+ return effectiveInputCapacity({
520
+ contextWindow: this.#contextWindow,
521
+ maxInputTokens: this.#maxInputTokens,
522
+ outputBudget: this.#outputBudget,
523
+ });
486
524
  }
487
- get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
488
- get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
489
525
  get model(): string { return this.#model; }
490
526
  // Backend's self-reported served id; undefined when unprobed/unknown.
491
527
  get servedModel(): string | undefined { return this.#servedModel; }
492
528
  // Resolved "decodes unbounded without a cap" fact; undefined = no claim.
493
- get requiresMaxTokens(): boolean | undefined { return this.#requiresMaxTokens; }
529
+ get requiresOutputBudget(): boolean | undefined { return this.#requiresOutputBudget; }
494
530
  // Resolved capability: will a transported grammar actually constrain
495
531
  // this backend's decode? Introspectable so a consumer can verify the rails
496
532
  // are LIVE without spending a generation on a forcing-grammar probe.
@@ -553,17 +589,46 @@ export default class AiSdkProvider implements Provider {
553
589
  );
554
590
  }
555
591
  }
592
+
593
+ async assessRequestCapacity(
594
+ messages: readonly ChatMessage[],
595
+ maxOutputTokens?: number,
596
+ signal?: AbortSignal,
597
+ ): Promise<ProviderRequestCapacity> {
598
+ const outputBudget = effectiveOutputBudget({
599
+ requested: maxOutputTokens,
600
+ configured: this.#outputBudget,
601
+ maxOutputTokens: this.#maxOutputTokens,
602
+ contextWindow: this.#contextWindow,
603
+ });
604
+ const reasoningBudget = effectiveReasoningBudget({
605
+ configured: this.#reasoningBudget,
606
+ outputBudget,
607
+ });
608
+ return assessRequestCapacity({
609
+ contextWindow: this.#contextWindow,
610
+ maxInputTokens: this.#maxInputTokens,
611
+ maxOutputTokens: this.#maxOutputTokens,
612
+ outputBudget,
613
+ reasoningBudget,
614
+ measurement: await this.countPromptTokens(messages, signal),
615
+ });
616
+ }
556
617
  // Reasoning activation and allowance are independent of grammar transport;
557
618
  // only the response representation becomes lossless when evidence is needed.
558
619
  // The llama-server template mapping is owned by {§llama-reasoning-request}.
559
- #reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
560
- const { mode, budget } = this.#reasoning;
620
+ #reasoningBody(
621
+ preserveGrammarSentence = false,
622
+ reasoningBudget = this.#reasoningBudget,
623
+ ): Record<string, unknown> {
624
+ const { mode } = this.#reasoning;
625
+ const budget = reasoningBudget;
561
626
  const on = mode !== "off";
562
627
  switch (this.#reasoningStyle) {
563
628
  case "template": {
564
629
  const allowance = mode === "off"
565
630
  ? 0
566
- : mode === "on" && budget !== null ? budget : this.reasoningReserve;
631
+ : mode === "on" && budget !== null ? budget : this.#reasoningBudget;
567
632
  return {
568
633
  chat_template_kwargs: { enable_thinking: on },
569
634
  reasoning_format: preserveGrammarSentence ? "none" : "auto",
@@ -574,7 +639,7 @@ export default class AiSdkProvider implements Provider {
574
639
  case "include_reasoning": return on ? { include_reasoning: true } : {};
575
640
  // Explicit on uses the portable enabled posture or a tier derived
576
641
  // from an explicit budget; off/adaptive omit the field.
577
- case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
642
+ case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
578
643
  // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
579
644
  // reason-by-default model (DeepSeek V4: default 'high') reasoning.
580
645
  // ADAPTIVE omits the field: the backend's own default posture IS the
@@ -584,7 +649,7 @@ export default class AiSdkProvider implements Provider {
584
649
  // efforts 400.
585
650
  case "effort_explicit": return mode === "off"
586
651
  ? { reasoning_effort: "none" }
587
- : mode === "on" ? { reasoning_effort: effortFromReasoning(this.#reasoning) } : {};
652
+ : mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
588
653
  // {§deepseek-reasoning-request}
589
654
  case "thinking_effort": return mode === "off"
590
655
  ? { thinking: { type: "disabled" } }
@@ -593,14 +658,14 @@ export default class AiSdkProvider implements Provider {
593
658
  ...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
594
659
  } : {};
595
660
  // Anthropic compat: explicit thinking object. off → disabled; on →
596
- // enabled with the explicit budget or resolved reserve; adaptive →
661
+ // enabled with the explicit reasoning subset; adaptive →
597
662
  // omit (the API default).
598
663
  case "anthropic": return mode === "off"
599
664
  ? { thinking: { type: "disabled" } }
600
665
  : mode === "on" ? {
601
666
  thinking: {
602
667
  type: "enabled",
603
- budget_tokens: budget ?? this.reasoningReserve!,
668
+ budget_tokens: budget!,
604
669
  },
605
670
  } : {};
606
671
  case "none": return {};
@@ -742,19 +807,43 @@ export default class AiSdkProvider implements Provider {
742
807
  return out;
743
808
  }
744
809
 
745
- #requestProviderOptions(workerId: string): AiSdkProviderOptions | undefined {
746
- const reasoningOptions = this.#reasoning.mode === "off"
810
+ #requestProviderOptions(
811
+ workerId: string,
812
+ reasoningBudget: number | null,
813
+ ): AiSdkProviderOptions | undefined {
814
+ const responseOptions = this.#reasoning.mode === "off"
747
815
  ? undefined
748
816
  : this.#reasoningResponseProviderOptions;
749
- if (this.#cacheAffinity?.target !== "provider-option") return reasoningOptions;
750
- const { provider, name } = this.#cacheAffinity;
751
- return {
752
- ...reasoningOptions,
753
- [provider]: {
754
- ...reasoningOptions?.[provider],
755
- [name]: workerId,
756
- },
757
- };
817
+ const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
818
+ ? this.#additiveReasoningProvider === "anthropic"
819
+ ? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
820
+ : this.#additiveReasoningProvider === "bedrock"
821
+ ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
822
+ : undefined
823
+ : undefined;
824
+ const options: AiSdkProviderOptions = {};
825
+ for (const part of [responseOptions, nativeReasoning]) {
826
+ for (const [provider, values] of Object.entries(part ?? {})) {
827
+ options[provider] = { ...options[provider], ...values };
828
+ }
829
+ }
830
+ if (this.#cacheAffinity?.target === "provider-option") {
831
+ const { provider, name } = this.#cacheAffinity;
832
+ options[provider] = { ...options[provider], [name]: workerId };
833
+ }
834
+ return Object.keys(options).length === 0 ? undefined : options;
835
+ }
836
+
837
+ #nativeMaxOutputTokens(
838
+ outputBudget: number | null,
839
+ reasoningBudget: number | null,
840
+ ): number | undefined {
841
+ if (outputBudget === null) return undefined;
842
+ return this.#additiveReasoningProvider !== undefined
843
+ && this.#reasoning.mode === "on"
844
+ && reasoningBudget !== null
845
+ ? outputBudget - reasoningBudget
846
+ : outputBudget;
758
847
  }
759
848
 
760
849
  #accounting(
@@ -775,7 +864,7 @@ export default class AiSdkProvider implements Provider {
775
864
  });
776
865
  }
777
866
 
778
- async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
867
+ async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
779
868
  // {§provider-interface} The worker identity is required.
780
869
  if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
781
870
  if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
@@ -793,6 +882,20 @@ export default class AiSdkProvider implements Provider {
793
882
  const preserveGrammarSentence = wantGrammar
794
883
  && this.#reasoningStyle === "template";
795
884
 
885
+ const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
886
+ if (capacity.decision === "reject") {
887
+ if (capacity.prompt.kind !== "exact") {
888
+ throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
889
+ }
890
+ throw new ProviderError(
891
+ this.#source,
892
+ "capacity_exceeded",
893
+ `The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`,
894
+ { capacity, extensions: { capacityStage: "preflight", capacity } },
895
+ );
896
+ }
897
+ const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
898
+
796
899
  // Assembly order = precedence: the family's sampling DEFAULTS
797
900
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
798
901
  // paths and the name promises every request) < the caller's `sampling`
@@ -805,9 +908,9 @@ export default class AiSdkProvider implements Provider {
805
908
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
806
909
  model: this.#model,
807
910
  messages,
808
- ...this.#reasoningBody(preserveGrammarSentence),
911
+ ...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
809
912
  ...this.#grammarBody(sendGrammar),
810
- ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
913
+ ...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
811
914
  // Request per-token logprobs only when enabled (managed field —
812
915
  // reserved from caller sampling; the env flag is the single control).
813
916
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
@@ -905,7 +1008,7 @@ export default class AiSdkProvider implements Provider {
905
1008
  : await executeAiSdkModel({
906
1009
  languageModel: this.#languageModel,
907
1010
  headers: requestHeaders,
908
- providerOptions: this.#requestProviderOptions(workerId),
1011
+ providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
909
1012
  systemProviderOptions: this.#systemCacheProviderOptions,
910
1013
  messages,
911
1014
  signal: operationSignal,
@@ -929,12 +1032,17 @@ export default class AiSdkProvider implements Provider {
929
1032
  ? sampling.stop
930
1033
  : undefined,
931
1034
  seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
932
- maxOutputTokens: maxTokens,
1035
+ maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
933
1036
  reasoning: this.#reasoning.mode === "off"
934
1037
  ? "none"
935
1038
  : this.#reasoning.mode === "adaptive"
936
1039
  ? "provider-default"
937
- : effortFromReasoning(this.#reasoning),
1040
+ : this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
1041
+ ? "provider-default"
1042
+ : effortFromReasoning({
1043
+ mode: this.#reasoning.mode,
1044
+ budget: capacity.reasoningBudget,
1045
+ }),
938
1046
  });
939
1047
  } catch (error) {
940
1048
  const failure = transportFailureEvidence(error);
@@ -976,14 +1084,16 @@ export default class AiSdkProvider implements Provider {
976
1084
  timeoutMs: timeout.timeoutMs,
977
1085
  },
978
1086
  accounting,
1087
+ capacity,
979
1088
  });
980
1089
  }
981
- const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
1090
+ const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
982
1091
  if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
983
1092
  throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
984
1093
  status: pe.status,
985
1094
  cause: err,
986
1095
  accounting,
1096
+ capacity,
987
1097
  });
988
1098
  }
989
1099
  pe.prependAccounting(accounting);
@@ -1091,11 +1201,34 @@ export default class AiSdkProvider implements Provider {
1091
1201
  const evidence = {
1092
1202
  assistantRaw: raw,
1093
1203
  accounting,
1204
+ capacity,
1094
1205
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
1095
1206
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
1096
1207
  ...(meta !== undefined ? { meta } : {}),
1097
1208
  ...(notices !== undefined ? { notices } : {}),
1098
1209
  };
1210
+ if (capacity.outputBudget !== null
1211
+ && usage?.outputTokens !== undefined
1212
+ && usage.outputTokens > capacity.outputBudget) {
1213
+ const attempt: ProviderAttempt = {
1214
+ assistant: { ...assistant, finishReason: raw.finishReason },
1215
+ ...evidence,
1216
+ };
1217
+ throw new ProviderError(
1218
+ this.#source,
1219
+ "invalid_response",
1220
+ `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`,
1221
+ {
1222
+ attempt,
1223
+ accounting,
1224
+ extensions: {
1225
+ stage: "provider-response",
1226
+ outputBudget: capacity.outputBudget,
1227
+ reportedOutputTokens: usage.outputTokens,
1228
+ },
1229
+ },
1230
+ );
1231
+ }
1099
1232
  if (raw.finishReason === "resource_interrupted") {
1100
1233
  const attempt: ProviderResponse<"resource_interrupted"> = {
1101
1234
  assistant: { ...assistant, finishReason: raw.finishReason },
package/src/Mock.test.ts CHANGED
@@ -13,6 +13,7 @@ const build = (responses: MockResponse[] = [{ assistant: { content: "hi", reason
13
13
  test("Mock: contextWindow and model are stable across reads", () => {
14
14
  const m = build();
15
15
  assert.equal(m.contextWindow, 100000);
16
+ assert.equal(m.inputCapacity, null);
16
17
  assert.equal(m.contextWindow, 100000);
17
18
  assert.equal(m.model, "mock");
18
19
  assert.equal(m.model, "mock");
@@ -148,35 +149,48 @@ test("Mock: exhausted queue throws a ProviderError carrying its settled accounti
148
149
 
149
150
  // -- {§provider-generation-envelope} --
150
151
 
151
- test("the reserve getters are on the Provider interface (not just the concrete class)", () => {
152
- // Typing against the contract catches a getter-only concrete surface.
153
- const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
152
+ test("the generation-envelope getters are on the Provider interface", () => {
153
+ const previous = {
154
+ output: process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET,
155
+ reasoning: process.env.PLURNK_PROVIDERS_REASONING_BUDGET,
156
+ };
154
157
  try {
155
- process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
158
+ process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = "35%";
159
+ process.env.PLURNK_PROVIDERS_REASONING_BUDGET = "10%";
156
160
  const p: Provider = new Mock({ contextWindow: 49152, responses: [] });
157
- assert.equal(p.reasoningReserve, 4915); // 10% of 49152, read through the interface type
161
+ assert.equal(p.outputBudget, 17_203);
162
+ assert.equal(p.reasoningBudget, 4_915);
163
+ assert.equal(p.inputCapacity, 31_949);
158
164
  } finally {
159
- if (prevR === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_RESERVE; else process.env.PLURNK_PROVIDERS_REASONING_RESERVE = prevR;
165
+ if (previous.output === undefined) delete process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET;
166
+ else process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = previous.output;
167
+ if (previous.reasoning === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_BUDGET;
168
+ else process.env.PLURNK_PROVIDERS_REASONING_BUDGET = previous.reasoning;
160
169
  }
161
170
  });
162
171
 
163
- test("Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its window (the service partition path)", () => {
164
- const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
165
- const prevC = process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE;
172
+ test("Mock resolves percentage and absolute generation budgets against its window", () => {
173
+ const previous = {
174
+ output: process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET,
175
+ reasoning: process.env.PLURNK_PROVIDERS_REASONING_BUDGET,
176
+ };
166
177
  try {
167
- process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
168
- process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE = "8192"; // mixed pct + absolute
178
+ process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = "35%";
179
+ process.env.PLURNK_PROVIDERS_REASONING_BUDGET = "8192";
169
180
  const m = new Mock({ contextWindow: 49152, responses: [] });
170
- assert.equal(m.reasoningReserve, 4915); // 10% of 49152
171
- assert.equal(m.completionReserve, 8192); // absolute stands
181
+ assert.equal(m.outputBudget, 17_203);
182
+ assert.equal(m.reasoningBudget, 8_192);
172
183
  } finally {
173
- if (prevR === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_RESERVE; else process.env.PLURNK_PROVIDERS_REASONING_RESERVE = prevR;
174
- if (prevC === undefined) delete process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE; else process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE = prevC;
184
+ if (previous.output === undefined) delete process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET;
185
+ else process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = previous.output;
186
+ if (previous.reasoning === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_BUDGET;
187
+ else process.env.PLURNK_PROVIDERS_REASONING_BUDGET = previous.reasoning;
175
188
  }
176
189
  });
177
190
 
178
- test("no reserve env → null (the no-cap path; bare Mocks unaffected)", () => {
191
+ test("no generation-budget env leaves a bare Mock unbounded", () => {
179
192
  const m = new Mock({ contextWindow: 49152, responses: [] });
180
- assert.equal(m.reasoningReserve, null);
181
- assert.equal(m.completionReserve, null);
193
+ assert.equal(m.outputBudget, null);
194
+ assert.equal(m.reasoningBudget, null);
195
+ assert.equal(m.inputCapacity, null);
182
196
  });