@plurnk/plurnk-providers 1.14.1 → 1.15.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/.env.defaults +21 -9
  2. package/SPEC.md +68 -12
  3. package/dist/AiSdkProvider.d.ts +5 -2
  4. package/dist/AiSdkProvider.d.ts.map +1 -1
  5. package/dist/AiSdkProvider.js +34 -342
  6. package/dist/AiSdkProvider.js.map +1 -1
  7. package/dist/AiSdkRequestBody.d.ts +40 -0
  8. package/dist/AiSdkRequestBody.d.ts.map +1 -0
  9. package/dist/AiSdkRequestBody.js +361 -0
  10. package/dist/AiSdkRequestBody.js.map +1 -0
  11. package/dist/Mock.d.ts +5 -1
  12. package/dist/Mock.d.ts.map +1 -1
  13. package/dist/Mock.js +9 -2
  14. package/dist/Mock.js.map +1 -1
  15. package/dist/Pool.d.ts +2 -1
  16. package/dist/Pool.d.ts.map +1 -1
  17. package/dist/Pool.js +7 -0
  18. package/dist/Pool.js.map +1 -1
  19. package/dist/aiSdkTransport.d.ts +1 -1
  20. package/dist/aiSdkTransport.d.ts.map +1 -1
  21. package/dist/aiSdkTransport.js +50 -34
  22. package/dist/aiSdkTransport.js.map +1 -1
  23. package/dist/capacity.d.ts +7 -0
  24. package/dist/capacity.d.ts.map +1 -1
  25. package/dist/capacity.js +21 -0
  26. package/dist/capacity.js.map +1 -1
  27. package/dist/catalogProvider.d.ts +2 -1
  28. package/dist/catalogProvider.d.ts.map +1 -1
  29. package/dist/catalogProvider.js +14 -3
  30. package/dist/catalogProvider.js.map +1 -1
  31. package/dist/compatibleProvider.d.ts.map +1 -1
  32. package/dist/compatibleProvider.js +6 -3
  33. package/dist/compatibleProvider.js.map +1 -1
  34. package/dist/errors.d.ts.map +1 -1
  35. package/dist/errors.js +23 -2
  36. package/dist/errors.js.map +1 -1
  37. package/dist/index.d.ts +3 -0
  38. package/dist/index.d.ts.map +1 -1
  39. package/dist/index.js +2 -0
  40. package/dist/index.js.map +1 -1
  41. package/dist/promptTokens.d.ts.map +1 -1
  42. package/dist/promptTokens.js +2 -1
  43. package/dist/promptTokens.js.map +1 -1
  44. package/dist/reasoning-effort.d.ts +4 -0
  45. package/dist/reasoning-effort.d.ts.map +1 -0
  46. package/dist/reasoning-effort.js +14 -0
  47. package/dist/reasoning-effort.js.map +1 -0
  48. package/dist/types.d.ts +18 -1
  49. package/dist/types.d.ts.map +1 -1
  50. package/dist/types.js +14 -1
  51. package/dist/types.js.map +1 -1
  52. package/package.json +6 -6
  53. package/src/AiSdkProvider.test.ts +142 -115
  54. package/src/AiSdkProvider.ts +41 -382
  55. package/src/AiSdkRequestBody.ts +419 -0
  56. package/src/Mock.ts +10 -2
  57. package/src/Pool.test.ts +3 -0
  58. package/src/Pool.ts +8 -1
  59. package/src/aiSdkTransport.test.ts +31 -16
  60. package/src/aiSdkTransport.ts +55 -38
  61. package/src/boundaries.test.ts +2 -0
  62. package/src/capacity.test.ts +33 -1
  63. package/src/capacity.ts +34 -0
  64. package/src/catalogProvider.test.ts +22 -0
  65. package/src/catalogProvider.ts +16 -3
  66. package/src/compatibleProvider.test.ts +12 -0
  67. package/src/compatibleProvider.ts +6 -3
  68. package/src/errors.test.ts +1 -0
  69. package/src/errors.ts +22 -1
  70. package/src/index.ts +3 -0
  71. package/src/inputModalities.test.ts +53 -0
  72. package/src/promptTokens.ts +2 -1
  73. package/src/reasoning-effort.ts +15 -0
  74. package/src/types.ts +30 -2
@@ -8,27 +8,17 @@
8
8
  import { REASONING_POLICIES } from "@plurnk/plurnk-contracts";
9
9
  import { MAX_PROVIDER_TIMEOUT_MS } from "./env.js";
10
10
  import { UnsupportedReasoningPolicyError } from "./types.js";
11
- import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence, } from "./aiSdkTransport.js";
11
+ import { executeAiSdkModel, executeOpenAICompatible, transportFailureOutputObserved, transportFailureEvidence } from "./aiSdkTransport.js";
12
12
  import { prepareRetries } from "ai/internal";
13
13
  import { toProviderError, ProviderError, ProviderTimeoutError } from "./errors.js";
14
- import { validateGbnf } from "@plurnk/gbnf";
15
14
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.js";
16
15
  import { emitWarningOnce } from "./warnings.js";
17
16
  import { resolveProviderCost } from "./cost.js";
18
17
  import { validateProviderRequestAccounting } from "./accounting.js";
19
18
  import { validateProviderUsage } from "./usage.js";
20
19
  import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.js";
21
- const isJsonObject = (value) => typeof value === "object" && value !== null && !Array.isArray(value);
22
- const mergeJsonObjects = (left, right) => Object.fromEntries([...new Set([...Object.keys(left), ...Object.keys(right)])].map((key) => {
23
- const leftValue = left[key];
24
- const rightValue = right[key];
25
- return [
26
- key,
27
- isJsonObject(leftValue) && isJsonObject(rightValue)
28
- ? mergeJsonObjects(leftValue, rightValue)
29
- : rightValue ?? leftValue,
30
- ];
31
- }));
20
+ import { nativeFixedEffort } from "./reasoning-effort.js";
21
+ import AiSdkRequestBody from "./AiSdkRequestBody.js";
32
22
  class ProviderRequestObserverError extends Error {
33
23
  constructor(cause) {
34
24
  super("provider request accounting could not be durably settled", { cause });
@@ -101,32 +91,6 @@ const projectTemplateReasoning = (content) => {
101
91
  }
102
92
  return { content, reasoning: "", projected: false, contentStart: 0 };
103
93
  };
104
- const fixedEffort = (mode) => {
105
- if (mode === "low" || mode === "medium" || mode === "high" || mode === "xhigh" || mode === "max")
106
- return mode;
107
- throw new TypeError(`reasoning policy '${mode}' is not a fixed effort`);
108
- };
109
- // The native SDK effort surface tops at xhigh; admission never grants a native
110
- // route "max", so reaching it here is a contract violation, not a fallback site.
111
- const nativeFixedEffort = (mode) => {
112
- const effort = fixedEffort(mode);
113
- if (effort === "max")
114
- throw new TypeError(`reasoning policy 'max' has no native SDK effort surface`);
115
- return effort;
116
- };
117
- // Anthropic's older manual-reasoning protocol needs an absolute allowance while
118
- // PLURNK's durable contract names an effort. These fractions match the native
119
- // SDK's policy projection, but apply to PLURNK's total envelope rather than the
120
- // model's physical maximum. The minimum is imposed by the provider protocol.
121
- const MANUAL_REASONING_FRACTIONS = Object.freeze({
122
- adaptive: 0.6,
123
- low: 0.1,
124
- medium: 0.3,
125
- high: 0.6,
126
- xhigh: 0.75,
127
- max: 0.85,
128
- });
129
- const MANUAL_REASONING_MINIMUM = 1024;
130
94
  const providerWarningMessage = (warning) => {
131
95
  switch (warning.type) {
132
96
  case "unsupported":
@@ -136,27 +100,6 @@ const providerWarningMessage = (warning) => {
136
100
  case "other": return warning.message;
137
101
  }
138
102
  };
139
- // Body keys the provider owns — a caller's `sampling` passthrough may not set
140
- // these. Two families:
141
- // transport/managed — grammar transport, the stream/JSON choice, slot pinning,
142
- // data capture ({§provider-evidence}: backend-specific fields never cross the contract);
143
- // contract invariants — `n` (atomic single completion: choices[0] is the
144
- // response; n>1 = paid, dropped output), the tool-calling family (tools-in-
145
- // body doctrine, §2: native tool_calls return null content = a broken turn),
146
- // modalities/audio (text-only contract), prediction (decode semantics, not
147
- // sampling), and the token caps (the envelope is the managed maxOutputTokens —
148
- // sampling must not bypass the consumer's cap).
149
- // Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
150
- // platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
151
- // metadata, store, verbosity) pass through; the managed floors spread UNDER
152
- // sampling stay deliberately caller-overridable.
153
- const RESERVED_BODY_KEYS = new Set([
154
- "model", "messages", "stream", "stream_options", "grammar", "response_format", "id_slot", "logprobs", "top_logprobs",
155
- "reasoning_format", "reasoning_effort", "thinking", "think", "include_reasoning", "chat_template_kwargs", "thinking_budget_tokens", // lexicon-allow: backend wire fields
156
- "n", "tools", "tool_choice", "functions", "function_call", "parallel_tool_calls",
157
- "modalities", "audio", "prediction", "max_tokens", "max_completion_tokens",
158
- "prompt_cache_key",
159
- ]);
160
103
  export default class AiSdkProvider {
161
104
  #model;
162
105
  #url;
@@ -171,6 +114,7 @@ export default class AiSdkProvider {
171
114
  #apiKeyRejectedMessage;
172
115
  #eosText;
173
116
  #contextWindow;
117
+ #inputModalities;
174
118
  #maxInputTokens;
175
119
  #maxOutputTokens;
176
120
  #outputBudget;
@@ -220,6 +164,7 @@ export default class AiSdkProvider {
220
164
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
221
165
  // the honest capability signal for every other backend.
222
166
  tokenize;
167
+ #requestBody;
223
168
  constructor(config) {
224
169
  this.#model = config.model;
225
170
  this.#url = config.url;
@@ -245,6 +190,7 @@ export default class AiSdkProvider {
245
190
  this.#headers = config.headers ?? {};
246
191
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
247
192
  this.#contextWindow = config.contextWindow ?? null;
193
+ this.#inputModalities = config.inputModalities ?? new Set();
248
194
  this.#maxInputTokens = config.maxInputTokens ?? null;
249
195
  this.#maxOutputTokens = config.maxOutputTokens ?? null;
250
196
  this.#outputBudget = config.outputBudget ?? null;
@@ -265,8 +211,8 @@ export default class AiSdkProvider {
265
211
  // Loud guard: an out-of-date consumer (stale plugin dist) omitting the
266
212
  // required tuning fields must fail at construction, not silently send
267
213
  // undefined sampling on every grammar request.
268
- if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
269
- throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
214
+ if (config.temperature === undefined || config.repeatPenalty === undefined) {
215
+ throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty declared (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY; null = provider default)`);
270
216
  }
271
217
  this.#temperature = config.temperature;
272
218
  this.#repeatPenalty = config.repeatPenalty;
@@ -377,8 +323,10 @@ export default class AiSdkProvider {
377
323
  return tokens;
378
324
  };
379
325
  }
326
+ this.#requestBody = new AiSdkRequestBody({ reasoningBudget: this.#reasoningBudget, additiveReasoningProvider: this.#additiveReasoningProvider, reasoning: this.#reasoning, reasoningToggle: this.#reasoningToggle, compatibleAdaptiveReasoning: this.#compatibleAdaptiveReasoning, compatibleOffReasoning: this.#compatibleOffReasoning, adaptiveReasoningProviderOptions: this.#adaptiveReasoningProviderOptions, repeatPenalty: this.#repeatPenalty, frequencyPenalty: this.#frequencyPenalty, dryMultiplier: this.#dryMultiplier, dryBase: this.#dryBase, dryAllowedLength: this.#dryAllowedLength, repeatLastN: this.#repeatLastN, reasoningStyle: this.#reasoningStyle, source: this.#source, grammarStyle: this.#grammarStyle, cacheAffinity: this.#cacheAffinity, reasoningResponseProviderOptions: this.#reasoningResponseProviderOptions, firstPartyMetadata: this.#firstPartyMetadata, supportsSlotPinning: this.#supportsSlotPinning, slotCount: this.#slotCount });
380
327
  }
381
328
  get contextWindow() { return this.#contextWindow; }
329
+ get inputModalities() { return this.#inputModalities; }
382
330
  get maxInputTokens() { return this.#maxInputTokens; }
383
331
  get maxOutputTokens() { return this.#maxOutputTokens; }
384
332
  get outputBudget() { return this.#outputBudget; }
@@ -420,7 +368,7 @@ export default class AiSdkProvider {
420
368
  body: JSON.stringify({
421
369
  model: this.#model,
422
370
  messages,
423
- ...this.#reasoningBody(),
371
+ ...this.#requestBody.reasoningBody(),
424
372
  }),
425
373
  ...(requestSignal === undefined ? {} : { signal: requestSignal }),
426
374
  });
@@ -462,205 +410,6 @@ export default class AiSdkProvider {
462
410
  measurement: await this.countPromptTokens(messages, signal),
463
411
  });
464
412
  }
465
- // Reasoning activation and allowance are independent of grammar transport;
466
- // only the response representation becomes lossless when evidence is needed.
467
- // The llama-server template mapping is owned by {§llama-reasoning-request}.
468
- #reasoningBody(preserveGrammarSentence = false, reasoningBudget = this.#reasoningBudget) {
469
- const { mode } = this.#reasoning;
470
- const budget = reasoningBudget;
471
- const on = mode !== "off";
472
- switch (this.#reasoningStyle) {
473
- case "template": {
474
- const allowance = mode === "off"
475
- ? 0
476
- : budget;
477
- return {
478
- chat_template_kwargs: { enable_thinking: on },
479
- reasoning_format: preserveGrammarSentence ? "none" : "auto",
480
- ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
481
- };
482
- }
483
- case "think": return on ? { think: true } : {};
484
- case "include_reasoning": return on ? { include_reasoning: true } : {};
485
- case "effort": return mode === "off"
486
- ? this.#compatibleOffReasoning === undefined
487
- ? {}
488
- : { reasoning_effort: this.#compatibleOffReasoning }
489
- : mode === "adaptive"
490
- ? this.#compatibleAdaptiveReasoning === "provider-default"
491
- ? {}
492
- : { reasoning_effort: this.#compatibleAdaptiveReasoning }
493
- : { reasoning_effort: fixedEffort(mode) };
494
- // Graded reasoning is mandatory when the route advertises an effort
495
- // value. Cataloged routes supply the exact strongest legal value;
496
- // construction rejects an unsupported off or fixed policy.
497
- case "effort_required": {
498
- if (mode === "off") {
499
- if (this.#compatibleOffReasoning === undefined) {
500
- throw new TypeError(`${this.#source}: required reasoning effort has no off projection`);
501
- }
502
- return { reasoning_effort: this.#compatibleOffReasoning };
503
- }
504
- if (mode === "adaptive") {
505
- return this.#compatibleAdaptiveReasoning === "provider-default"
506
- ? {}
507
- : { reasoning_effort: this.#compatibleAdaptiveReasoning };
508
- }
509
- return { reasoning_effort: fixedEffort(mode) };
510
- }
511
- // Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
512
- // reason-by-default model (DeepSeek V4: default 'high') reasoning.
513
- // ADAPTIVE omits the field UNLESS the catalog declares a toggle control:
514
- // toggle routes (nemotron-lightning) default reasoning OFF, so adaptive
515
- // sends the documented Fireworks Boolean enable (#457). The literal
516
- // "adaptive" is MiniMax-M3-only — Fireworks 400s it for every other
517
- // model (wire-verified; the 1.0.2 adaptive default refused to boot on
518
- // it). V4 gotcha: integer efforts 400.
519
- case "effort_explicit": return mode === "off"
520
- ? { reasoning_effort: "none" }
521
- : mode === "adaptive"
522
- ? this.#reasoningToggle ? { reasoning_effort: true } : {}
523
- : { reasoning_effort: fixedEffort(mode) };
524
- // {§deepseek-reasoning-request}
525
- case "thinking_effort": return mode === "off"
526
- ? { thinking: { type: "disabled" } }
527
- : mode === "adaptive" ? { thinking: { type: "enabled" } } : {
528
- thinking: { type: "enabled" },
529
- reasoning_effort: fixedEffort(mode),
530
- };
531
- // Anthropic-compatible native dynamic or manual budget mode.
532
- case "anthropic": return mode === "off"
533
- ? { thinking: { type: "disabled" } }
534
- : mode === "adaptive" ? { thinking: { type: "adaptive" } } : {
535
- thinking: {
536
- type: "enabled",
537
- budget_tokens: budget,
538
- },
539
- };
540
- case "none": return {};
541
- }
542
- }
543
- // Per-worker slot affinity: the consumer passes which worker this is; the
544
- // provider owns WHICH slot serves it. Sticky per workerId, round-robin across
545
- // new runs (distinct runs → distinct slots while slots last), LRU-bounded
546
- // bookkeeping so a long-lived daemon never grows the map unboundedly —
547
- // an evicted-and-returning run simply re-pins, worst case one cold prefill.
548
- #runSlots = new Map();
549
- #nextSlot = 0;
550
- #slotBody(workerId) {
551
- if (!this.#supportsSlotPinning || this.#slotCount === null || this.#slotCount < 1)
552
- return {};
553
- let slot = this.#runSlots.get(workerId);
554
- if (slot === undefined) {
555
- slot = this.#nextSlot++ % this.#slotCount;
556
- if (this.#runSlots.size >= this.#slotCount * 8) {
557
- this.#runSlots.delete(this.#runSlots.keys().next().value);
558
- }
559
- }
560
- else {
561
- this.#runSlots.delete(workerId); // re-insert to refresh LRU recency
562
- }
563
- this.#runSlots.set(workerId, slot);
564
- return { id_slot: slot };
565
- }
566
- // Optional local llama-server GBNF transport ({§gbnf-response-observation}). Unsupported
567
- // backends receive no grammar-related field.
568
- #grammarBody(grammar) {
569
- if (grammar === undefined)
570
- return {};
571
- switch (this.#grammarStyle) {
572
- // Greedy decoding under hard constraint loops without a repeat-penalty
573
- // floor — llama.cpp spells it `repeat_penalty`.
574
- case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
575
- case "none": return {};
576
- }
577
- }
578
- // Anti-degeneration default on every request, keyed to the backend's wire
579
- // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
580
- // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
581
- // straight to the token cap on pure looped repetition (run52). Ships next to
582
- // temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
583
- // managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
584
- // MULTIPLIER; the plain cloud path ("none") can't, so it gets
585
- // frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
586
- // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
587
- #repetitionPenaltyBody() {
588
- switch (this.#grammarStyle) {
589
- // repeat_penalty + optional DRY (repeated-sequence penalty) + a wider
590
- // repeat_last_n window — the loop-breaking tools a llama.cpp backend serves.
591
- // Each rides only when its operator knob is set; absent = the box's default.
592
- case "llamacpp": return {
593
- repeat_penalty: this.#repeatPenalty,
594
- ...(this.#repeatLastN !== undefined ? { repeat_last_n: this.#repeatLastN } : {}),
595
- ...(this.#dryMultiplier !== undefined && this.#dryMultiplier > 0 ? {
596
- dry_multiplier: this.#dryMultiplier,
597
- ...(this.#dryBase !== undefined ? { dry_base: this.#dryBase } : {}),
598
- ...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
599
- } : {}),
600
- };
601
- case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
602
- }
603
- }
604
- // First-party telemetry headers ({§provider-request-authority} {§provider-call-kind}): forwarded only when the spec
605
- // opted in (the plurnk endpoint). The gate is here, not at the call site, so
606
- // attributions/client/strikes can never reach a third-party backend even if
607
- // the consumer passes them to the wrong provider. Empty values emit no header
608
- // — EXCEPT strikes, where 0 is a real value (clean streak) distinct from
609
- // absent (consumer didn't report); contract {§strikes-first-party-metadata}. Strikes
610
- // ride HTTP headers only — the packet never carries them (the model must
611
- // never see strike state; engine accounting is not a metric to game).
612
- #metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind) {
613
- if (!this.#firstPartyMetadata)
614
- return {};
615
- const h = {};
616
- if (attributions !== undefined && attributions.length > 0)
617
- h["Plurnk-Attribution"] = JSON.stringify(attributions);
618
- if (client !== undefined && client.length > 0)
619
- h["Plurnk-Client"] = client;
620
- if (strikes !== undefined && Number.isInteger(strikes) && strikes >= 0)
621
- h["Plurnk-Strikes"] = String(strikes);
622
- // Worker identity: the opaque workerId
623
- // the consumer already supplies, forwarded so the endpoint can key
624
- // per-worker affinity/telemetry — same gate as every first-party signal.
625
- h["Plurnk-Worker-Id"] = workerId;
626
- // Root worker of the lineage ({§worker-primary}): the no-parent ancestor of this turn's
627
- // worker tree. The consumer classifies primary-vs-spawned by equality
628
- // (primaryWorkerId == workerId ⇒ the primary/root worker). The provider
629
- // EMITS what the consumer supplies and never invents a primary; the
630
- // consumer's contract is to stamp it EVERY turn (including the primary's
631
- // own, where it equals workerId). Absence is the consumer's violation for
632
- // the endpoint to surface, not a provider default.
633
- if (primaryWorkerId !== undefined && primaryWorkerId.length > 0)
634
- h["Plurnk-Worker-Primary"] = primaryWorkerId;
635
- // Turn coordinate ({§lifecycle-terms}): workspace/loop/turn, the
636
- // daemon-side sequence the endpoint can never scrape from the wire.
637
- // Coordinates are 1-based — 0 is not a real value, so no strikes-style
638
- // zero exception; absent/empty/0 emits no header.
639
- if (workspaceId !== undefined && workspaceId.length > 0)
640
- h["Plurnk-Workspace-Id"] = workspaceId;
641
- if (loop !== undefined && Number.isInteger(loop) && loop >= 1)
642
- h["Plurnk-Loop"] = String(loop);
643
- if (turn !== undefined && Number.isInteger(turn) && turn >= 1)
644
- h["Plurnk-Turn"] = String(turn);
645
- if (callKind !== undefined)
646
- h["Plurnk-Call-Kind"] = callKind;
647
- return h;
648
- }
649
- // PLURNK_PROVIDERS_GBNF_DEBUG ({§gbnf-response-observation}): validate the supplied GBNF locally and fail
650
- // hard if it's malformed, BEFORE any wire call — and the grammar is NOT
651
- // transported, so the request runs unconstrained. A debug aid to catch invalid
652
- // grammars (e.g. while editing the plurnk grammar) without a model round-trip;
653
- // off in production. `validateGbnf(grammar, "")` parses the grammar + resolves
654
- // its root, throwing iff the grammar itself is invalid (the empty input's
655
- // verdict is irrelevant — we only care that parsing succeeded).
656
- #assertGrammarValid(grammar) {
657
- try {
658
- validateGbnf(grammar, "");
659
- }
660
- catch (cause) {
661
- throw new Error(`grammar validation (PLURNK_PROVIDERS_GBNF_DEBUG): invalid GBNF — ${cause.message}`, { cause });
662
- }
663
- }
664
413
  // Per-turn metadata bag: pass the backend's non-standard top-level fields
665
414
  // through verbatim. Providers do not reinterpret vendor currency or account
666
415
  // metadata; a monetary value carries its own amount and currency.
@@ -668,71 +417,6 @@ export default class AiSdkProvider {
668
417
  const meta = { ...chunkMetadata };
669
418
  return Object.keys(meta).length > 0 ? meta : undefined;
670
419
  }
671
- // Caller-supplied OpenAI-compat sampling params (temperature, top_p, top_k,
672
- // penalties, stop, seed, …) merged UNDER the managed body: model, messages,
673
- // reasoning, grammar (+ its repeat-penalty floor), max_tokens and slot always
674
- // win, and reserved transport/protocol keys are stripped so the passthrough
675
- // can't smuggle a grammar, a stream toggle, or a backend slot
676
- // ({§provider-request-authority}).
677
- #samplingBody(sampling) {
678
- if (sampling === undefined)
679
- return {};
680
- const out = {};
681
- for (const [k, v] of Object.entries(sampling))
682
- if (!RESERVED_BODY_KEYS.has(k))
683
- out[k] = v;
684
- return out;
685
- }
686
- #requestProviderOptions(workerId, nativeReasoningBudget) {
687
- const responseOptions = this.#reasoning.mode === "off"
688
- ? undefined
689
- : this.#reasoningResponseProviderOptions;
690
- const adaptiveOptions = this.#reasoning.mode === "adaptive"
691
- && nativeReasoningBudget === null
692
- ? this.#adaptiveReasoningProviderOptions
693
- : undefined;
694
- const nativeReasoning = nativeReasoningBudget !== null
695
- ? this.#additiveReasoningProvider === "anthropic"
696
- ? { anthropic: { thinking: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
697
- : this.#additiveReasoningProvider === "bedrock"
698
- ? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: nativeReasoningBudget } } }
699
- : undefined
700
- : undefined;
701
- const options = {};
702
- for (const part of [responseOptions, adaptiveOptions, nativeReasoning]) {
703
- for (const [provider, values] of Object.entries(part ?? {})) {
704
- options[provider] = mergeJsonObjects(options[provider] ?? {}, values);
705
- }
706
- }
707
- if (this.#cacheAffinity?.target === "provider-option") {
708
- const { provider, name } = this.#cacheAffinity;
709
- options[provider] = { ...options[provider], [name]: workerId };
710
- }
711
- return Object.keys(options).length === 0 ? undefined : options;
712
- }
713
- #nativeMaxOutputTokens(outputBudget, nativeReasoningBudget) {
714
- if (outputBudget === null)
715
- return undefined;
716
- return nativeReasoningBudget !== null
717
- ? outputBudget - nativeReasoningBudget
718
- : outputBudget;
719
- }
720
- #nativeReasoningBudget(outputBudget, configuredReasoningBudget) {
721
- if (this.#additiveReasoningProvider === undefined || this.#reasoning.mode === "off")
722
- return null;
723
- if (configuredReasoningBudget !== null)
724
- return configuredReasoningBudget;
725
- if (this.#adaptiveReasoningProviderOptions !== undefined)
726
- return null;
727
- if (outputBudget === null) {
728
- throw new TypeError(`${this.#source}: manual provider reasoning requires a resolved total output budget`);
729
- }
730
- if (outputBudget <= MANUAL_REASONING_MINIMUM) {
731
- throw new TypeError(`${this.#source}: total output budget must exceed the provider's ${MANUAL_REASONING_MINIMUM}-token minimum reasoning allowance`);
732
- }
733
- const fraction = MANUAL_REASONING_FRACTIONS[this.#reasoning.mode];
734
- return Math.min(outputBudget - 1, Math.max(MANUAL_REASONING_MINIMUM, Math.round(outputBudget * fraction)));
735
- }
736
420
  #accounting(outcome, usage, evidence, status) {
737
421
  const knownUsage = usage === undefined ? undefined : validateProviderUsage(usage);
738
422
  const direct = this.#normalizeCost?.(evidence);
@@ -759,7 +443,7 @@ export default class AiSdkProvider {
759
443
  // supplied grammar before the call but withholds it from the backend.
760
444
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
761
445
  if (wantGrammar && this.#gbnfDebug)
762
- this.#assertGrammarValid(grammar);
446
+ this.#requestBody.assertGrammarValid(grammar);
763
447
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
764
448
  const preserveGrammarSentence = wantGrammar
765
449
  && this.#reasoningStyle === "template";
@@ -770,8 +454,10 @@ export default class AiSdkProvider {
770
454
  }
771
455
  throw new ProviderError(this.#source, "capacity_exceeded", `The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`, { capacity, extensions: { capacityStage: "preflight", capacity } });
772
456
  }
773
- const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
774
- const nativeReasoningBudget = this.#nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
457
+ // {§provider-flexed-allowance} (#482): the wire grants the flexed
458
+ // allowance — the floor, or the exactly-measured slack above it.
459
+ const effectiveMaxOutputTokens = capacity.responseMax ?? capacity.outputBudget ?? undefined;
460
+ const nativeReasoningBudget = this.#requestBody.nativeReasoningBudget(capacity.outputBudget, capacity.reasoningBudget);
775
461
  // Assembly order = precedence: the family's sampling DEFAULTS
776
462
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
777
463
  // paths and the name promises every request) < the caller's `sampling`
@@ -779,24 +465,24 @@ export default class AiSdkProvider {
779
465
  const body = {
780
466
  // Floors are suppressed on router-owned-tuning providers (plurnk) —
781
467
  // the router's per-model tuning must not be overridden by client floors.
782
- ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
783
- ...this.#samplingBody(sampling),
468
+ ...(this.#tuningFloors ? { ...(this.#temperature !== null ? { temperature: this.#temperature } : {}), ...this.#requestBody.repetitionPenaltyBody() } : {}),
469
+ ...this.#requestBody.samplingBody(sampling),
784
470
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
785
471
  model: this.#model,
786
472
  messages,
787
- ...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
788
- ...this.#grammarBody(sendGrammar),
473
+ ...this.#requestBody.reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
474
+ ...this.#requestBody.grammarBody(sendGrammar),
789
475
  ...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
790
476
  // Request per-token logprobs only when enabled (managed field —
791
477
  // reserved from caller sampling; the env flag is the single control).
792
478
  ...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
793
- ...this.#slotBody(workerId),
479
+ ...this.#requestBody.slotBody(workerId),
794
480
  ...(this.#cacheAffinity?.target === "body"
795
481
  ? { [this.#cacheAffinity.name]: workerId }
796
482
  : {}),
797
483
  };
798
484
  // Per-request headers = static auth/routing + any first-party telemetry.
799
- const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
485
+ const metaHeaders = this.#requestBody.metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn, callKind);
800
486
  const headers = new Headers(this.#headers);
801
487
  if (this.#cacheAffinity?.target === "header") {
802
488
  headers.set(this.#cacheAffinity.name, workerId);
@@ -905,7 +591,7 @@ export default class AiSdkProvider {
905
591
  : await executeAiSdkModel({
906
592
  languageModel: this.#languageModel,
907
593
  headers: requestHeaders,
908
- providerOptions: this.#requestProviderOptions(workerId, nativeReasoningBudget),
594
+ providerOptions: this.#requestBody.requestProviderOptions(workerId, nativeReasoningBudget),
909
595
  systemProviderOptions: this.#systemCacheProviderOptions,
910
596
  messages,
911
597
  signal: operationSignal,
@@ -916,7 +602,7 @@ export default class AiSdkProvider {
916
602
  captureRawBody: this.#rawBody,
917
603
  ...(observeRequestReasoning === undefined ? {} : { observeReasoning: observeRequestReasoning }),
918
604
  temperature: this.#tuningFloors
919
- ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
605
+ ? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature ?? undefined)
920
606
  : typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
921
607
  topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
922
608
  topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
@@ -930,7 +616,7 @@ export default class AiSdkProvider {
930
616
  ? sampling.stop
931
617
  : undefined,
932
618
  seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
933
- maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
619
+ maxOutputTokens: this.#requestBody.nativeMaxOutputTokens(capacity.outputBudget, nativeReasoningBudget),
934
620
  reasoning: this.#reasoning.mode === "off"
935
621
  ? "none"
936
622
  : this.#reasoning.mode === "adaptive"
@@ -1116,19 +802,25 @@ export default class AiSdkProvider {
1116
802
  ...(meta !== undefined ? { meta } : {}),
1117
803
  ...(notices !== undefined ? { notices } : {}),
1118
804
  };
1119
- if (capacity.outputBudget !== null
805
+ // {§provider-flexed-allowance} (#482): conformance judges the GRANT the
806
+ // wire actually sent, not the configured floor — output between the two
807
+ // is overflow tolerance working, never a provider fault. Run7 loop-death
808
+ // was this guard still holding the floor after the flex landed.
809
+ const grantedOutput = capacity.responseMax ?? capacity.outputBudget;
810
+ if (grantedOutput !== null
1120
811
  && usage?.outputTokens !== undefined
1121
- && usage.outputTokens > capacity.outputBudget) {
812
+ && usage.outputTokens > grantedOutput) {
1122
813
  const attempt = {
1123
814
  assistant: { ...assistant, finishReason: raw.finishReason },
1124
815
  ...evidence,
1125
816
  };
1126
- throw new ProviderError(this.#source, "invalid_response", `The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`, {
817
+ throw new ProviderError(this.#source, "invalid_response", `The provider reported ${usage.outputTokens} output tokens after receiving a granted output allowance of ${grantedOutput}.`, {
1127
818
  attempt,
1128
819
  accounting,
1129
820
  extensions: {
1130
821
  stage: "provider-response",
1131
822
  outputBudget: capacity.outputBudget,
823
+ grantedOutput,
1132
824
  reportedOutputTokens: usage.outputTokens,
1133
825
  },
1134
826
  });