@plurnk/plurnk-providers 1.6.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +9 -16
- package/README.md +17 -0
- package/SPEC.md +128 -45
- package/dist/AiSdkProvider.d.ts +16 -9
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +151 -54
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +8 -4
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +53 -18
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -4
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +68 -12
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +18 -20
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +10 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/env.d.ts +8 -10
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +54 -37
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +6 -22
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +30 -91
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/providerError.d.ts +25 -0
- package/dist/providerError.d.ts.map +1 -0
- package/dist/providerError.js +91 -0
- package/dist/providerError.js.map +1 -0
- package/dist/sdkModels.d.ts +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +5 -8
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +24 -4
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +7 -2
- package/dist/usage.js.map +1 -1
- package/package.json +22 -7
- package/src/AiSdkProvider.test.ts +198 -37
- package/src/AiSdkProvider.ts +192 -59
- package/src/Mock.test.ts +32 -18
- package/src/Mock.ts +58 -19
- package/src/Pool.test.ts +71 -13
- package/src/Pool.ts +78 -13
- package/src/ProviderRegistry.test.ts +1 -1
- package/src/accounting.ts +0 -1
- package/src/accountingPublic.ts +9 -0
- package/src/boundaries.test.ts +28 -15
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +82 -9
- package/src/catalogProvider.ts +24 -21
- package/src/compatibleProvider.test.ts +1 -2
- package/src/compatibleProvider.ts +10 -7
- package/src/cost.test.ts +31 -0
- package/src/env.test.ts +49 -20
- package/src/env.ts +114 -51
- package/src/errors.test.ts +33 -0
- package/src/errors.ts +38 -134
- package/src/index.ts +5 -2
- package/src/ollama.test.ts +1 -2
- package/src/promptTokens.ts +8 -5
- package/src/providerError.ts +139 -0
- package/src/sdkModels.test.ts +2 -5
- package/src/sdkModels.ts +6 -8
- package/src/types.ts +41 -19
- package/src/usage.ts +7 -2
package/src/AiSdkProvider.ts
CHANGED
|
@@ -11,11 +11,12 @@ import type {
|
|
|
11
11
|
GrammarEvidence,
|
|
12
12
|
PromptTokenMeasurement,
|
|
13
13
|
Provider,
|
|
14
|
+
ProviderAttempt,
|
|
14
15
|
ProviderCostNormalizer,
|
|
15
16
|
ProviderCallKind,
|
|
16
17
|
ProviderGenerateArgs,
|
|
17
18
|
ProviderRequestAccounting,
|
|
18
|
-
|
|
19
|
+
ProviderRequestCapacity,
|
|
19
20
|
ProviderRequestSettlement,
|
|
20
21
|
ProviderResponse,
|
|
21
22
|
ProviderUsage,
|
|
@@ -23,7 +24,7 @@ import type {
|
|
|
23
24
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
24
25
|
import type { JSONValue } from "ai";
|
|
25
26
|
import { MAX_PROVIDER_TIMEOUT_MS } from "./env.ts";
|
|
26
|
-
import type { Reasoning, ReasoningResponseStyle
|
|
27
|
+
import type { Reasoning, ReasoningResponseStyle } from "./env.ts";
|
|
27
28
|
import {
|
|
28
29
|
executeAiSdkModel,
|
|
29
30
|
executeOpenAICompatible,
|
|
@@ -40,6 +41,7 @@ import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk
|
|
|
40
41
|
import { resolveProviderCost } from "./cost.ts";
|
|
41
42
|
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
42
43
|
import { validateProviderUsage } from "./usage.ts";
|
|
44
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
43
45
|
|
|
44
46
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
45
47
|
|
|
@@ -69,6 +71,15 @@ export type AiSdkProviderConfig = {
|
|
|
69
71
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
70
72
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
71
73
|
contextWindow?: number | null; // default null; caller resolves-or-fails, narrows to required with the interface
|
|
74
|
+
maxInputTokens?: number | null;
|
|
75
|
+
maxOutputTokens?: number | null;
|
|
76
|
+
outputBudget?: number | null;
|
|
77
|
+
reasoningBudget?: number | null;
|
|
78
|
+
// Native Anthropic and Bedrock SDKs interpret generic maxOutputTokens as
|
|
79
|
+
// visible output and add an explicit provider reasoning budget. This marker lets the
|
|
80
|
+
// adapter subtract that subset so the resulting wire cap remains PLURNK's
|
|
81
|
+
// one total output budget.
|
|
82
|
+
additiveReasoningProvider?: "anthropic" | "bedrock";
|
|
72
83
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
73
84
|
reasoningResponseStyle?: ReasoningResponseStyle; // {§provider-tagged-reasoning}; default "verbatim"
|
|
74
85
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
@@ -109,9 +120,9 @@ export type AiSdkProviderConfig = {
|
|
|
109
120
|
// when no probe ran or it read no row.
|
|
110
121
|
servedModel?: string;
|
|
111
122
|
// Backend decodes unbounded without a caller cap (llama-server n_predict
|
|
112
|
-
// to the wall) — surfaced as Provider.
|
|
123
|
+
// to the wall) — surfaced as Provider.requiresOutputBudget so consumers can
|
|
113
124
|
// boot-refuse an envelope-less local alias. Default unset (no claim).
|
|
114
|
-
|
|
125
|
+
requiresOutputBudget?: boolean;
|
|
115
126
|
// The side-channel reasoning intent — REQUIRED, no in-code default
|
|
116
127
|
// (PLURNK_PROVIDERS_REASONING + _BUDGET, read via reasoningFromEnv):
|
|
117
128
|
// { mode: off|adaptive|on, budget: optional when on }. The provider maps it
|
|
@@ -157,14 +168,12 @@ export type AiSdkProviderConfig = {
|
|
|
157
168
|
// gated per-alias.
|
|
158
169
|
topLogprobs?: number | null;
|
|
159
170
|
rawBody?: boolean;
|
|
160
|
-
// {§provider-generation-envelope} The generation
|
|
161
|
-
//
|
|
171
|
+
// {§provider-generation-envelope} The generation budgets, env-read via the
|
|
172
|
+
// common envelope parser — a percentage of the detected window or an absolute token
|
|
162
173
|
// count. Optional so an out-of-date sibling keeps constructing (no claim);
|
|
163
174
|
// the standard factory always supplies them. Resolved against contextWindow
|
|
164
175
|
// at read time (getters), so a probe that lands after config assembly still
|
|
165
176
|
// derives correctly.
|
|
166
|
-
reasoningReserve?: ReserveSpec;
|
|
167
|
-
completionReserve?: ReserveSpec;
|
|
168
177
|
// The plurnk.ai router owns tuning — false suppresses the
|
|
169
178
|
// client-side temperature/penalty FLOORS on this provider (caller `sampling`
|
|
170
179
|
// still passes through verbatim). Default true (floors ride).
|
|
@@ -265,7 +274,7 @@ export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
|
265
274
|
|
|
266
275
|
// AI SDK's portable reasoning control has no boolean-enabled value. `medium`
|
|
267
276
|
// is the neutral activation projection for an explicit, unqualified `on`; it
|
|
268
|
-
// changes no PLURNK
|
|
277
|
+
// changes no PLURNK output budget. An operator reasoning subset, when present, remains
|
|
269
278
|
// the only input to the existing magnitude-to-tier projection.
|
|
270
279
|
const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
|
|
271
280
|
reasoning.budget === null ? "medium" : effortFromBudget(reasoning.budget);
|
|
@@ -278,7 +287,7 @@ const effortFromReasoning = (reasoning: Reasoning): "low" | "medium" | "high" =>
|
|
|
278
287
|
// response; n>1 = paid, dropped output), the tool-calling family (tools-in-
|
|
279
288
|
// body doctrine, §2: native tool_calls return null content = a broken turn),
|
|
280
289
|
// modalities/audio (text-only contract), prediction (decode semantics, not
|
|
281
|
-
// sampling), and the token caps (the envelope is the managed
|
|
290
|
+
// sampling), and the token caps (the envelope is the managed maxOutputTokens —
|
|
282
291
|
// sampling must not bypass the consumer's cap).
|
|
283
292
|
// Sampling intent (temperature, top_p, penalties, stop, seed, logit_bias) and
|
|
284
293
|
// platform knobs (user, service_tier, prompt_cache_*, safety_identifier,
|
|
@@ -306,6 +315,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
306
315
|
#apiKeyRejectedMessage: string | undefined;
|
|
307
316
|
#eosText: string | undefined;
|
|
308
317
|
#contextWindow: number | null;
|
|
318
|
+
#maxInputTokens: number | null;
|
|
319
|
+
#maxOutputTokens: number | null;
|
|
320
|
+
#outputBudget: number | null;
|
|
321
|
+
#reasoningBudget: number | null;
|
|
322
|
+
#additiveReasoningProvider: "anthropic" | "bedrock" | undefined;
|
|
309
323
|
#reasoning: Reasoning;
|
|
310
324
|
#temperature: number;
|
|
311
325
|
#repeatPenalty: number;
|
|
@@ -334,12 +348,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
334
348
|
#retryAttempts: number;
|
|
335
349
|
#errorDetailLimit: number | undefined;
|
|
336
350
|
#topLogprobs: number | null;
|
|
337
|
-
#reasoningReserve: ReserveSpec | undefined;
|
|
338
|
-
#completionReserve: ReserveSpec | undefined;
|
|
339
351
|
#tuningFloors: boolean;
|
|
340
352
|
#rawBody: boolean;
|
|
341
353
|
#servedModel: string | undefined;
|
|
342
|
-
#
|
|
354
|
+
#requiresOutputBudget: boolean | undefined;
|
|
343
355
|
readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
344
356
|
|
|
345
357
|
// Optional capability ({§provider-local-capabilities}): exact tokenization served by the backend's
|
|
@@ -372,6 +384,11 @@ export default class AiSdkProvider implements Provider {
|
|
|
372
384
|
this.#headers = config.headers ?? {};
|
|
373
385
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
374
386
|
this.#contextWindow = config.contextWindow ?? null;
|
|
387
|
+
this.#maxInputTokens = config.maxInputTokens ?? null;
|
|
388
|
+
this.#maxOutputTokens = config.maxOutputTokens ?? null;
|
|
389
|
+
this.#outputBudget = config.outputBudget ?? null;
|
|
390
|
+
this.#reasoningBudget = config.reasoningBudget ?? null;
|
|
391
|
+
this.#additiveReasoningProvider = config.additiveReasoningProvider;
|
|
375
392
|
this.#reasoning = config.reasoning;
|
|
376
393
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
377
394
|
// required tuning fields must fail at construction, not silently send
|
|
@@ -435,25 +452,42 @@ export default class AiSdkProvider implements Provider {
|
|
|
435
452
|
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
436
453
|
this.#slotCount = config.slotCount ?? null;
|
|
437
454
|
this.#topLogprobs = config.topLogprobs ?? null;
|
|
438
|
-
this.#reasoningReserve = config.reasoningReserve;
|
|
439
|
-
this.#completionReserve = config.completionReserve;
|
|
440
455
|
this.#tuningFloors = config.tuningFloors ?? true;
|
|
441
456
|
this.#rawBody = config.rawBody ?? false;
|
|
442
457
|
this.#servedModel = config.servedModel;
|
|
443
|
-
this.#
|
|
444
|
-
const
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
458
|
+
this.#requiresOutputBudget = config.requiresOutputBudget;
|
|
459
|
+
for (const [name, value] of [
|
|
460
|
+
["contextWindow", this.#contextWindow],
|
|
461
|
+
["maxInputTokens", this.#maxInputTokens],
|
|
462
|
+
["maxOutputTokens", this.#maxOutputTokens],
|
|
463
|
+
["outputBudget", this.#outputBudget],
|
|
464
|
+
["reasoningBudget", this.#reasoningBudget],
|
|
465
|
+
] as const) {
|
|
466
|
+
if (value !== null && (!Number.isSafeInteger(value) || value <= 0)) {
|
|
467
|
+
throw new Error(`${this.#source}: ${name} must be a positive safe integer or null`);
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
if (this.#reasoningBudget !== null
|
|
471
|
+
&& this.#outputBudget !== null
|
|
472
|
+
&& this.#reasoningBudget >= this.#outputBudget) {
|
|
473
|
+
throw new Error(`${this.#source}: reasoningBudget must be smaller than the total outputBudget`);
|
|
474
|
+
}
|
|
475
|
+
if (this.#reasoning.budget !== this.#reasoningBudget) {
|
|
476
|
+
throw new Error(`${this.#source}: reasoning intent and generation envelope disagree on reasoningBudget`);
|
|
451
477
|
}
|
|
452
478
|
if (this.#reasoningStyle === "anthropic"
|
|
453
479
|
&& this.#reasoning.mode === "on"
|
|
454
480
|
&& this.#reasoning.budget === null
|
|
455
|
-
&&
|
|
456
|
-
throw new Error(`${this.#source}: explicit Anthropic reasoning requires
|
|
481
|
+
&& this.#reasoningBudget === null) {
|
|
482
|
+
throw new Error(`${this.#source}: explicit Anthropic reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET`);
|
|
483
|
+
}
|
|
484
|
+
if (this.#additiveReasoningProvider !== undefined
|
|
485
|
+
&& this.#reasoning.mode === "on"
|
|
486
|
+
&& this.#reasoningBudget === null) {
|
|
487
|
+
throw new Error(`${this.#source}: explicit ${this.#additiveReasoningProvider} reasoning requires PLURNK_PROVIDERS_REASONING_BUDGET so the total output budget remains bounded`);
|
|
488
|
+
}
|
|
489
|
+
if (this.#requiresOutputBudget === true && this.#outputBudget === null) {
|
|
490
|
+
throw new Error(`${this.#source}: this backend requires a resolved PLURNK_PROVIDERS_OUTPUT_BUDGET`);
|
|
457
491
|
}
|
|
458
492
|
const { tokenizeUrl } = config;
|
|
459
493
|
if (tokenizeUrl !== undefined) {
|
|
@@ -477,20 +511,22 @@ export default class AiSdkProvider implements Provider {
|
|
|
477
511
|
}
|
|
478
512
|
|
|
479
513
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
return
|
|
514
|
+
get maxInputTokens(): number | null { return this.#maxInputTokens; }
|
|
515
|
+
get maxOutputTokens(): number | null { return this.#maxOutputTokens; }
|
|
516
|
+
get outputBudget(): number | null { return this.#outputBudget; }
|
|
517
|
+
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
518
|
+
get inputCapacity(): number | null {
|
|
519
|
+
return effectiveInputCapacity({
|
|
520
|
+
contextWindow: this.#contextWindow,
|
|
521
|
+
maxInputTokens: this.#maxInputTokens,
|
|
522
|
+
outputBudget: this.#outputBudget,
|
|
523
|
+
});
|
|
486
524
|
}
|
|
487
|
-
get reasoningReserve(): number | null { return this.#resolveReserve(this.#reasoningReserve); }
|
|
488
|
-
get completionReserve(): number | null { return this.#resolveReserve(this.#completionReserve); }
|
|
489
525
|
get model(): string { return this.#model; }
|
|
490
526
|
// Backend's self-reported served id; undefined when unprobed/unknown.
|
|
491
527
|
get servedModel(): string | undefined { return this.#servedModel; }
|
|
492
528
|
// Resolved "decodes unbounded without a cap" fact; undefined = no claim.
|
|
493
|
-
get
|
|
529
|
+
get requiresOutputBudget(): boolean | undefined { return this.#requiresOutputBudget; }
|
|
494
530
|
// Resolved capability: will a transported grammar actually constrain
|
|
495
531
|
// this backend's decode? Introspectable so a consumer can verify the rails
|
|
496
532
|
// are LIVE without spending a generation on a forcing-grammar probe.
|
|
@@ -553,17 +589,46 @@ export default class AiSdkProvider implements Provider {
|
|
|
553
589
|
);
|
|
554
590
|
}
|
|
555
591
|
}
|
|
592
|
+
|
|
593
|
+
async assessRequestCapacity(
|
|
594
|
+
messages: readonly ChatMessage[],
|
|
595
|
+
maxOutputTokens?: number,
|
|
596
|
+
signal?: AbortSignal,
|
|
597
|
+
): Promise<ProviderRequestCapacity> {
|
|
598
|
+
const outputBudget = effectiveOutputBudget({
|
|
599
|
+
requested: maxOutputTokens,
|
|
600
|
+
configured: this.#outputBudget,
|
|
601
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
602
|
+
contextWindow: this.#contextWindow,
|
|
603
|
+
});
|
|
604
|
+
const reasoningBudget = effectiveReasoningBudget({
|
|
605
|
+
configured: this.#reasoningBudget,
|
|
606
|
+
outputBudget,
|
|
607
|
+
});
|
|
608
|
+
return assessRequestCapacity({
|
|
609
|
+
contextWindow: this.#contextWindow,
|
|
610
|
+
maxInputTokens: this.#maxInputTokens,
|
|
611
|
+
maxOutputTokens: this.#maxOutputTokens,
|
|
612
|
+
outputBudget,
|
|
613
|
+
reasoningBudget,
|
|
614
|
+
measurement: await this.countPromptTokens(messages, signal),
|
|
615
|
+
});
|
|
616
|
+
}
|
|
556
617
|
// Reasoning activation and allowance are independent of grammar transport;
|
|
557
618
|
// only the response representation becomes lossless when evidence is needed.
|
|
558
619
|
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
559
|
-
#reasoningBody(
|
|
560
|
-
|
|
620
|
+
#reasoningBody(
|
|
621
|
+
preserveGrammarSentence = false,
|
|
622
|
+
reasoningBudget = this.#reasoningBudget,
|
|
623
|
+
): Record<string, unknown> {
|
|
624
|
+
const { mode } = this.#reasoning;
|
|
625
|
+
const budget = reasoningBudget;
|
|
561
626
|
const on = mode !== "off";
|
|
562
627
|
switch (this.#reasoningStyle) {
|
|
563
628
|
case "template": {
|
|
564
629
|
const allowance = mode === "off"
|
|
565
630
|
? 0
|
|
566
|
-
: mode === "on" && budget !== null ? budget : this
|
|
631
|
+
: mode === "on" && budget !== null ? budget : this.#reasoningBudget;
|
|
567
632
|
return {
|
|
568
633
|
chat_template_kwargs: { enable_thinking: on },
|
|
569
634
|
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
@@ -574,7 +639,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
574
639
|
case "include_reasoning": return on ? { include_reasoning: true } : {};
|
|
575
640
|
// Explicit on uses the portable enabled posture or a tier derived
|
|
576
641
|
// from an explicit budget; off/adaptive omit the field.
|
|
577
|
-
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning(
|
|
642
|
+
case "effort": return mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
578
643
|
// Fireworks enum: OFF is sent EXPLICITLY ("none") — omission leaves a
|
|
579
644
|
// reason-by-default model (DeepSeek V4: default 'high') reasoning.
|
|
580
645
|
// ADAPTIVE omits the field: the backend's own default posture IS the
|
|
@@ -584,7 +649,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
584
649
|
// efforts 400.
|
|
585
650
|
case "effort_explicit": return mode === "off"
|
|
586
651
|
? { reasoning_effort: "none" }
|
|
587
|
-
: mode === "on" ? { reasoning_effort: effortFromReasoning(
|
|
652
|
+
: mode === "on" ? { reasoning_effort: effortFromReasoning({ mode, budget }) } : {};
|
|
588
653
|
// {§deepseek-reasoning-request}
|
|
589
654
|
case "thinking_effort": return mode === "off"
|
|
590
655
|
? { thinking: { type: "disabled" } }
|
|
@@ -593,14 +658,14 @@ export default class AiSdkProvider implements Provider {
|
|
|
593
658
|
...(budget === null ? {} : { reasoning_effort: effortFromBudget(budget) }),
|
|
594
659
|
} : {};
|
|
595
660
|
// Anthropic compat: explicit thinking object. off → disabled; on →
|
|
596
|
-
// enabled with the explicit
|
|
661
|
+
// enabled with the explicit reasoning subset; adaptive →
|
|
597
662
|
// omit (the API default).
|
|
598
663
|
case "anthropic": return mode === "off"
|
|
599
664
|
? { thinking: { type: "disabled" } }
|
|
600
665
|
: mode === "on" ? {
|
|
601
666
|
thinking: {
|
|
602
667
|
type: "enabled",
|
|
603
|
-
budget_tokens: budget
|
|
668
|
+
budget_tokens: budget!,
|
|
604
669
|
},
|
|
605
670
|
} : {};
|
|
606
671
|
case "none": return {};
|
|
@@ -742,19 +807,43 @@ export default class AiSdkProvider implements Provider {
|
|
|
742
807
|
return out;
|
|
743
808
|
}
|
|
744
809
|
|
|
745
|
-
#requestProviderOptions(
|
|
746
|
-
|
|
810
|
+
#requestProviderOptions(
|
|
811
|
+
workerId: string,
|
|
812
|
+
reasoningBudget: number | null,
|
|
813
|
+
): AiSdkProviderOptions | undefined {
|
|
814
|
+
const responseOptions = this.#reasoning.mode === "off"
|
|
747
815
|
? undefined
|
|
748
816
|
: this.#reasoningResponseProviderOptions;
|
|
749
|
-
|
|
750
|
-
|
|
751
|
-
|
|
752
|
-
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
757
|
-
|
|
817
|
+
const nativeReasoning = this.#reasoning.mode === "on" && reasoningBudget !== null
|
|
818
|
+
? this.#additiveReasoningProvider === "anthropic"
|
|
819
|
+
? { anthropic: { thinking: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
820
|
+
: this.#additiveReasoningProvider === "bedrock"
|
|
821
|
+
? { bedrock: { reasoningConfig: { type: "enabled", budgetTokens: reasoningBudget } } }
|
|
822
|
+
: undefined
|
|
823
|
+
: undefined;
|
|
824
|
+
const options: AiSdkProviderOptions = {};
|
|
825
|
+
for (const part of [responseOptions, nativeReasoning]) {
|
|
826
|
+
for (const [provider, values] of Object.entries(part ?? {})) {
|
|
827
|
+
options[provider] = { ...options[provider], ...values };
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
if (this.#cacheAffinity?.target === "provider-option") {
|
|
831
|
+
const { provider, name } = this.#cacheAffinity;
|
|
832
|
+
options[provider] = { ...options[provider], [name]: workerId };
|
|
833
|
+
}
|
|
834
|
+
return Object.keys(options).length === 0 ? undefined : options;
|
|
835
|
+
}
|
|
836
|
+
|
|
837
|
+
#nativeMaxOutputTokens(
|
|
838
|
+
outputBudget: number | null,
|
|
839
|
+
reasoningBudget: number | null,
|
|
840
|
+
): number | undefined {
|
|
841
|
+
if (outputBudget === null) return undefined;
|
|
842
|
+
return this.#additiveReasoningProvider !== undefined
|
|
843
|
+
&& this.#reasoning.mode === "on"
|
|
844
|
+
&& reasoningBudget !== null
|
|
845
|
+
? outputBudget - reasoningBudget
|
|
846
|
+
: outputBudget;
|
|
758
847
|
}
|
|
759
848
|
|
|
760
849
|
#accounting(
|
|
@@ -775,7 +864,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
775
864
|
});
|
|
776
865
|
}
|
|
777
866
|
|
|
778
|
-
async generate({ messages, workerId, primaryWorkerId, signal, grammar,
|
|
867
|
+
async generate({ messages, workerId, primaryWorkerId, signal, grammar, maxOutputTokens, attributions, client, strikes, workspaceId, loop, turn, sampling, observeRequest, callKind }: ProviderGenerateArgs): Promise<ProviderResponse> {
|
|
779
868
|
// {§provider-interface} The worker identity is required.
|
|
780
869
|
if (workerId === undefined || workerId.length === 0) throw new Error("generate: workerId is required — the worker's stable, opaque identity");
|
|
781
870
|
if (callKind !== undefined && callKind !== "emission" && callKind !== "bare") {
|
|
@@ -793,6 +882,20 @@ export default class AiSdkProvider implements Provider {
|
|
|
793
882
|
const preserveGrammarSentence = wantGrammar
|
|
794
883
|
&& this.#reasoningStyle === "template";
|
|
795
884
|
|
|
885
|
+
const capacity = await this.assessRequestCapacity(messages, maxOutputTokens, signal);
|
|
886
|
+
if (capacity.decision === "reject") {
|
|
887
|
+
if (capacity.prompt.kind !== "exact") {
|
|
888
|
+
throw new TypeError(`${this.#source}: only an exact prompt measurement may reject capacity`);
|
|
889
|
+
}
|
|
890
|
+
throw new ProviderError(
|
|
891
|
+
this.#source,
|
|
892
|
+
"capacity_exceeded",
|
|
893
|
+
`The exact provider request uses ${capacity.prompt.tokens} input tokens, exceeding its ${capacity.inputCapacity} token input capacity.`,
|
|
894
|
+
{ capacity, extensions: { capacityStage: "preflight", capacity } },
|
|
895
|
+
);
|
|
896
|
+
}
|
|
897
|
+
const effectiveMaxOutputTokens = capacity.outputBudget ?? undefined;
|
|
898
|
+
|
|
796
899
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
797
900
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
798
901
|
// paths and the name promises every request) < the caller's `sampling`
|
|
@@ -805,9 +908,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
805
908
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
806
909
|
model: this.#model,
|
|
807
910
|
messages,
|
|
808
|
-
...this.#reasoningBody(preserveGrammarSentence),
|
|
911
|
+
...this.#reasoningBody(preserveGrammarSentence, capacity.reasoningBudget),
|
|
809
912
|
...this.#grammarBody(sendGrammar),
|
|
810
|
-
...(
|
|
913
|
+
...(effectiveMaxOutputTokens !== undefined ? { max_tokens: effectiveMaxOutputTokens } : {}),
|
|
811
914
|
// Request per-token logprobs only when enabled (managed field —
|
|
812
915
|
// reserved from caller sampling; the env flag is the single control).
|
|
813
916
|
...(this.#topLogprobs !== null ? { logprobs: true, top_logprobs: this.#topLogprobs } : {}),
|
|
@@ -905,7 +1008,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
905
1008
|
: await executeAiSdkModel({
|
|
906
1009
|
languageModel: this.#languageModel,
|
|
907
1010
|
headers: requestHeaders,
|
|
908
|
-
providerOptions: this.#requestProviderOptions(workerId),
|
|
1011
|
+
providerOptions: this.#requestProviderOptions(workerId, capacity.reasoningBudget),
|
|
909
1012
|
systemProviderOptions: this.#systemCacheProviderOptions,
|
|
910
1013
|
messages,
|
|
911
1014
|
signal: operationSignal,
|
|
@@ -929,12 +1032,17 @@ export default class AiSdkProvider implements Provider {
|
|
|
929
1032
|
? sampling.stop
|
|
930
1033
|
: undefined,
|
|
931
1034
|
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
932
|
-
maxOutputTokens:
|
|
1035
|
+
maxOutputTokens: this.#nativeMaxOutputTokens(capacity.outputBudget, capacity.reasoningBudget),
|
|
933
1036
|
reasoning: this.#reasoning.mode === "off"
|
|
934
1037
|
? "none"
|
|
935
1038
|
: this.#reasoning.mode === "adaptive"
|
|
936
1039
|
? "provider-default"
|
|
937
|
-
:
|
|
1040
|
+
: this.#additiveReasoningProvider !== undefined && capacity.reasoningBudget !== null
|
|
1041
|
+
? "provider-default"
|
|
1042
|
+
: effortFromReasoning({
|
|
1043
|
+
mode: this.#reasoning.mode,
|
|
1044
|
+
budget: capacity.reasoningBudget,
|
|
1045
|
+
}),
|
|
938
1046
|
});
|
|
939
1047
|
} catch (error) {
|
|
940
1048
|
const failure = transportFailureEvidence(error);
|
|
@@ -976,14 +1084,16 @@ export default class AiSdkProvider implements Provider {
|
|
|
976
1084
|
timeoutMs: timeout.timeoutMs,
|
|
977
1085
|
},
|
|
978
1086
|
accounting,
|
|
1087
|
+
capacity,
|
|
979
1088
|
});
|
|
980
1089
|
}
|
|
981
|
-
const pe = toProviderError(err, this.#source, this.#errorDetailLimit);
|
|
1090
|
+
const pe = toProviderError(err, this.#source, this.#errorDetailLimit, capacity);
|
|
982
1091
|
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
983
1092
|
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, {
|
|
984
1093
|
status: pe.status,
|
|
985
1094
|
cause: err,
|
|
986
1095
|
accounting,
|
|
1096
|
+
capacity,
|
|
987
1097
|
});
|
|
988
1098
|
}
|
|
989
1099
|
pe.prependAccounting(accounting);
|
|
@@ -1091,11 +1201,34 @@ export default class AiSdkProvider implements Provider {
|
|
|
1091
1201
|
const evidence = {
|
|
1092
1202
|
assistantRaw: raw,
|
|
1093
1203
|
accounting,
|
|
1204
|
+
capacity,
|
|
1094
1205
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
1095
1206
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
1096
1207
|
...(meta !== undefined ? { meta } : {}),
|
|
1097
1208
|
...(notices !== undefined ? { notices } : {}),
|
|
1098
1209
|
};
|
|
1210
|
+
if (capacity.outputBudget !== null
|
|
1211
|
+
&& usage?.outputTokens !== undefined
|
|
1212
|
+
&& usage.outputTokens > capacity.outputBudget) {
|
|
1213
|
+
const attempt: ProviderAttempt = {
|
|
1214
|
+
assistant: { ...assistant, finishReason: raw.finishReason },
|
|
1215
|
+
...evidence,
|
|
1216
|
+
};
|
|
1217
|
+
throw new ProviderError(
|
|
1218
|
+
this.#source,
|
|
1219
|
+
"invalid_response",
|
|
1220
|
+
`The provider reported ${usage.outputTokens} output tokens after receiving a total output budget of ${capacity.outputBudget}.`,
|
|
1221
|
+
{
|
|
1222
|
+
attempt,
|
|
1223
|
+
accounting,
|
|
1224
|
+
extensions: {
|
|
1225
|
+
stage: "provider-response",
|
|
1226
|
+
outputBudget: capacity.outputBudget,
|
|
1227
|
+
reportedOutputTokens: usage.outputTokens,
|
|
1228
|
+
},
|
|
1229
|
+
},
|
|
1230
|
+
);
|
|
1231
|
+
}
|
|
1099
1232
|
if (raw.finishReason === "resource_interrupted") {
|
|
1100
1233
|
const attempt: ProviderResponse<"resource_interrupted"> = {
|
|
1101
1234
|
assistant: { ...assistant, finishReason: raw.finishReason },
|
package/src/Mock.test.ts
CHANGED
|
@@ -13,6 +13,7 @@ const build = (responses: MockResponse[] = [{ assistant: { content: "hi", reason
|
|
|
13
13
|
test("Mock: contextWindow and model are stable across reads", () => {
|
|
14
14
|
const m = build();
|
|
15
15
|
assert.equal(m.contextWindow, 100000);
|
|
16
|
+
assert.equal(m.inputCapacity, null);
|
|
16
17
|
assert.equal(m.contextWindow, 100000);
|
|
17
18
|
assert.equal(m.model, "mock");
|
|
18
19
|
assert.equal(m.model, "mock");
|
|
@@ -148,35 +149,48 @@ test("Mock: exhausted queue throws a ProviderError carrying its settled accounti
|
|
|
148
149
|
|
|
149
150
|
// -- {§provider-generation-envelope} --
|
|
150
151
|
|
|
151
|
-
test("the
|
|
152
|
-
|
|
153
|
-
|
|
152
|
+
test("the generation-envelope getters are on the Provider interface", () => {
|
|
153
|
+
const previous = {
|
|
154
|
+
output: process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET,
|
|
155
|
+
reasoning: process.env.PLURNK_PROVIDERS_REASONING_BUDGET,
|
|
156
|
+
};
|
|
154
157
|
try {
|
|
155
|
-
process.env.
|
|
158
|
+
process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = "35%";
|
|
159
|
+
process.env.PLURNK_PROVIDERS_REASONING_BUDGET = "10%";
|
|
156
160
|
const p: Provider = new Mock({ contextWindow: 49152, responses: [] });
|
|
157
|
-
assert.equal(p.
|
|
161
|
+
assert.equal(p.outputBudget, 17_203);
|
|
162
|
+
assert.equal(p.reasoningBudget, 4_915);
|
|
163
|
+
assert.equal(p.inputCapacity, 31_949);
|
|
158
164
|
} finally {
|
|
159
|
-
if (
|
|
165
|
+
if (previous.output === undefined) delete process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET;
|
|
166
|
+
else process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = previous.output;
|
|
167
|
+
if (previous.reasoning === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_BUDGET;
|
|
168
|
+
else process.env.PLURNK_PROVIDERS_REASONING_BUDGET = previous.reasoning;
|
|
160
169
|
}
|
|
161
170
|
});
|
|
162
171
|
|
|
163
|
-
test("Mock resolves
|
|
164
|
-
const
|
|
165
|
-
|
|
172
|
+
test("Mock resolves percentage and absolute generation budgets against its window", () => {
|
|
173
|
+
const previous = {
|
|
174
|
+
output: process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET,
|
|
175
|
+
reasoning: process.env.PLURNK_PROVIDERS_REASONING_BUDGET,
|
|
176
|
+
};
|
|
166
177
|
try {
|
|
167
|
-
process.env.
|
|
168
|
-
process.env.
|
|
178
|
+
process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = "35%";
|
|
179
|
+
process.env.PLURNK_PROVIDERS_REASONING_BUDGET = "8192";
|
|
169
180
|
const m = new Mock({ contextWindow: 49152, responses: [] });
|
|
170
|
-
assert.equal(m.
|
|
171
|
-
assert.equal(m.
|
|
181
|
+
assert.equal(m.outputBudget, 17_203);
|
|
182
|
+
assert.equal(m.reasoningBudget, 8_192);
|
|
172
183
|
} finally {
|
|
173
|
-
if (
|
|
174
|
-
|
|
184
|
+
if (previous.output === undefined) delete process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET;
|
|
185
|
+
else process.env.PLURNK_PROVIDERS_OUTPUT_BUDGET = previous.output;
|
|
186
|
+
if (previous.reasoning === undefined) delete process.env.PLURNK_PROVIDERS_REASONING_BUDGET;
|
|
187
|
+
else process.env.PLURNK_PROVIDERS_REASONING_BUDGET = previous.reasoning;
|
|
175
188
|
}
|
|
176
189
|
});
|
|
177
190
|
|
|
178
|
-
test("no
|
|
191
|
+
test("no generation-budget env leaves a bare Mock unbounded", () => {
|
|
179
192
|
const m = new Mock({ contextWindow: 49152, responses: [] });
|
|
180
|
-
assert.equal(m.
|
|
181
|
-
assert.equal(m.
|
|
193
|
+
assert.equal(m.outputBudget, null);
|
|
194
|
+
assert.equal(m.reasoningBudget, null);
|
|
195
|
+
assert.equal(m.inputCapacity, null);
|
|
182
196
|
});
|