@plurnk/plurnk-providers 1.6.0 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +9 -16
- package/README.md +17 -0
- package/SPEC.md +128 -45
- package/dist/AiSdkProvider.d.ts +16 -9
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +151 -54
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +8 -4
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +53 -18
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -4
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +68 -12
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js.map +1 -1
- package/dist/accountingPublic.d.ts +5 -0
- package/dist/accountingPublic.d.ts.map +1 -0
- package/dist/accountingPublic.js +3 -0
- package/dist/accountingPublic.js.map +1 -0
- package/dist/capacity.d.ts +26 -0
- package/dist/capacity.d.ts.map +1 -0
- package/dist/capacity.js +90 -0
- package/dist/capacity.js.map +1 -0
- package/dist/catalogProvider.d.ts +2 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +18 -20
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +10 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/env.d.ts +8 -10
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +54 -37
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +6 -22
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +30 -91
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -1
- package/dist/index.js.map +1 -1
- package/dist/promptTokens.d.ts.map +1 -1
- package/dist/promptTokens.js +7 -4
- package/dist/promptTokens.js.map +1 -1
- package/dist/providerError.d.ts +25 -0
- package/dist/providerError.d.ts.map +1 -0
- package/dist/providerError.js +91 -0
- package/dist/providerError.js.map +1 -0
- package/dist/sdkModels.d.ts +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +5 -8
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +24 -4
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +7 -2
- package/dist/usage.js.map +1 -1
- package/package.json +22 -7
- package/src/AiSdkProvider.test.ts +198 -37
- package/src/AiSdkProvider.ts +192 -59
- package/src/Mock.test.ts +32 -18
- package/src/Mock.ts +58 -19
- package/src/Pool.test.ts +71 -13
- package/src/Pool.ts +78 -13
- package/src/ProviderRegistry.test.ts +1 -1
- package/src/accounting.ts +0 -1
- package/src/accountingPublic.ts +9 -0
- package/src/boundaries.test.ts +28 -15
- package/src/capacity.test.ts +92 -0
- package/src/capacity.ts +140 -0
- package/src/catalogProvider.test.ts +82 -9
- package/src/catalogProvider.ts +24 -21
- package/src/compatibleProvider.test.ts +1 -2
- package/src/compatibleProvider.ts +10 -7
- package/src/cost.test.ts +31 -0
- package/src/env.test.ts +49 -20
- package/src/env.ts +114 -51
- package/src/errors.test.ts +33 -0
- package/src/errors.ts +38 -134
- package/src/index.ts +5 -2
- package/src/ollama.test.ts +1 -2
- package/src/promptTokens.ts +8 -5
- package/src/providerError.ts +139 -0
- package/src/sdkModels.test.ts +2 -5
- package/src/sdkModels.ts +6 -8
- package/src/types.ts +41 -19
- package/src/usage.ts +7 -2
package/src/Mock.ts
CHANGED
|
@@ -5,10 +5,11 @@
|
|
|
5
5
|
// Provider contract. Production providers don't expose the `ops` escape
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
7
|
|
|
8
|
-
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
|
-
import {
|
|
8
|
+
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderCost, ProviderEncryptedReasoningItem, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
|
+
import { resolveGenerationEnvelopeFromEnv } from "./env.ts";
|
|
10
10
|
import { validateProviderRequestAccounting } from "./accounting.ts";
|
|
11
11
|
import { ProviderError } from "./errors.ts";
|
|
12
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
12
13
|
|
|
13
14
|
export type MockAssistant = {
|
|
14
15
|
content: string;
|
|
@@ -46,29 +47,35 @@ type MockGenerateArgs = Omit<Parameters<Provider["generate"]>[0], "workerId"> &
|
|
|
46
47
|
|
|
47
48
|
export default class Mock implements Provider {
|
|
48
49
|
#contextWindow: number | null;
|
|
49
|
-
#
|
|
50
|
-
#
|
|
50
|
+
#outputBudget: number | null;
|
|
51
|
+
#reasoningBudget: number | null;
|
|
51
52
|
#queue: MockResponse[];
|
|
52
53
|
|
|
53
|
-
// {§provider-generation-envelope}
|
|
54
|
-
//
|
|
55
|
-
//
|
|
56
|
-
//
|
|
57
|
-
//
|
|
58
|
-
// without the reserves in most base tests, so absent env → null → the
|
|
59
|
-
// consumer's no-cap path, and the ~100 `new Mock({ contextWindow, responses })`
|
|
60
|
-
// sites stay untouched. Mock has no alias identity, so it reads the BARE knobs.
|
|
54
|
+
// {§provider-generation-envelope} Mock resolves the same generation
|
|
55
|
+
// envelope as a real provider, but tolerates absent policy because it is
|
|
56
|
+
// also the universal test fixture. No output budget means its context
|
|
57
|
+
// window alone cannot determine an input capacity. Mock has no alias
|
|
58
|
+
// identity, so it reads the bare knobs.
|
|
61
59
|
constructor({ contextWindow, responses }: { contextWindow: number | null; responses: MockResponse[] }) {
|
|
62
60
|
this.#contextWindow = contextWindow;
|
|
63
|
-
const
|
|
64
|
-
this.#
|
|
65
|
-
this.#
|
|
61
|
+
const envelope = resolveGenerationEnvelopeFromEnv(process.env, contextWindow);
|
|
62
|
+
this.#outputBudget = envelope.outputBudget;
|
|
63
|
+
this.#reasoningBudget = envelope.reasoningBudget;
|
|
66
64
|
this.#queue = [...responses];
|
|
67
65
|
}
|
|
68
66
|
|
|
69
67
|
get contextWindow(): number | null { return this.#contextWindow; }
|
|
70
|
-
get
|
|
71
|
-
get
|
|
68
|
+
get maxInputTokens(): number | null { return null; }
|
|
69
|
+
get maxOutputTokens(): number | null { return null; }
|
|
70
|
+
get outputBudget(): number | null { return this.#outputBudget; }
|
|
71
|
+
get reasoningBudget(): number | null { return this.#reasoningBudget; }
|
|
72
|
+
get inputCapacity(): number | null {
|
|
73
|
+
return effectiveInputCapacity({
|
|
74
|
+
contextWindow: this.#contextWindow,
|
|
75
|
+
maxInputTokens: this.maxInputTokens,
|
|
76
|
+
outputBudget: this.#outputBudget,
|
|
77
|
+
});
|
|
78
|
+
}
|
|
72
79
|
get model(): string { return "mock"; }
|
|
73
80
|
|
|
74
81
|
// Mock's deliberately simple vocabulary defines each two content code units
|
|
@@ -82,11 +89,42 @@ export default class Mock implements Provider {
|
|
|
82
89
|
};
|
|
83
90
|
}
|
|
84
91
|
|
|
85
|
-
async
|
|
92
|
+
async assessRequestCapacity(
|
|
93
|
+
messages: readonly ChatMessage[],
|
|
94
|
+
maxOutputTokens?: number,
|
|
95
|
+
): Promise<ProviderRequestCapacity> {
|
|
96
|
+
const outputBudget = effectiveOutputBudget({
|
|
97
|
+
requested: maxOutputTokens,
|
|
98
|
+
configured: this.#outputBudget,
|
|
99
|
+
maxOutputTokens: null,
|
|
100
|
+
contextWindow: this.#contextWindow,
|
|
101
|
+
});
|
|
102
|
+
const reasoningBudget = effectiveReasoningBudget({
|
|
103
|
+
configured: this.#reasoningBudget,
|
|
104
|
+
outputBudget,
|
|
105
|
+
});
|
|
106
|
+
return assessRequestCapacity({
|
|
107
|
+
contextWindow: this.#contextWindow,
|
|
108
|
+
maxInputTokens: null,
|
|
109
|
+
maxOutputTokens: null,
|
|
110
|
+
outputBudget,
|
|
111
|
+
reasoningBudget,
|
|
112
|
+
measurement: await this.countPromptTokens(messages),
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
async generate({ messages, maxOutputTokens, signal, grammar, observeRequest }: MockGenerateArgs): Promise<MockReturnedResponse> {
|
|
86
117
|
// Honor abort before consuming the queue — an aborted call makes no
|
|
87
118
|
// "wire call" and must not exhaust a queued response
|
|
88
119
|
// ({§provider-failure-normalization}).
|
|
89
120
|
signal?.throwIfAborted();
|
|
121
|
+
const capacity = await this.assessRequestCapacity(messages, maxOutputTokens);
|
|
122
|
+
if (capacity.decision === "reject") {
|
|
123
|
+
throw new ProviderError("mock", "capacity_exceeded", "Mock request exceeds its exact input capacity.", {
|
|
124
|
+
capacity,
|
|
125
|
+
extensions: { capacityStage: "preflight", capacity },
|
|
126
|
+
});
|
|
127
|
+
}
|
|
90
128
|
const settle = await observeRequest?.({ provider: "provider:mock", model: this.model });
|
|
91
129
|
const next = this.#queue.shift();
|
|
92
130
|
if (next === undefined) {
|
|
@@ -101,7 +139,7 @@ export default class Mock implements Provider {
|
|
|
101
139
|
"mock",
|
|
102
140
|
"invalid_response",
|
|
103
141
|
"Mock provider exhausted: no more queued responses",
|
|
104
|
-
{ accounting: [accounting] },
|
|
142
|
+
{ accounting: [accounting], capacity },
|
|
105
143
|
);
|
|
106
144
|
}
|
|
107
145
|
const a = next.assistant;
|
|
@@ -137,6 +175,7 @@ export default class Mock implements Provider {
|
|
|
137
175
|
assistant,
|
|
138
176
|
assistantRaw: next.assistantRaw ?? null,
|
|
139
177
|
accounting: [requestAccounting],
|
|
178
|
+
capacity,
|
|
140
179
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
141
180
|
};
|
|
142
181
|
}
|
package/src/Pool.test.ts
CHANGED
|
@@ -4,6 +4,7 @@ import Pool from "./Pool.ts";
|
|
|
4
4
|
import { ProviderError } from "./errors.ts";
|
|
5
5
|
import type { PromptTokenMeasurement, Provider, ProviderResponse } from "./types.ts";
|
|
6
6
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
7
|
+
import { effectiveInputCapacity } from "./capacity.ts";
|
|
7
8
|
|
|
8
9
|
test.afterEach(() => { resetEmittedWarnings(); });
|
|
9
10
|
|
|
@@ -22,12 +23,23 @@ const RESP: ProviderResponse = {
|
|
|
22
23
|
usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
|
|
23
24
|
cost: { kind: "unknown", reason: "test fixture has no cost" },
|
|
24
25
|
}],
|
|
26
|
+
capacity: {
|
|
27
|
+
decision: "admit",
|
|
28
|
+
contextWindow: 48_000,
|
|
29
|
+
maxInputTokens: null,
|
|
30
|
+
maxOutputTokens: null,
|
|
31
|
+
outputBudget: 12_000,
|
|
32
|
+
reasoningBudget: null,
|
|
33
|
+
inputCapacity: 36_000,
|
|
34
|
+
prompt: { kind: "exact", tokens: 0, source: "test:exact" },
|
|
35
|
+
},
|
|
25
36
|
};
|
|
26
37
|
|
|
27
38
|
type FakeOpts = {
|
|
28
39
|
model?: string; window?: number | null; servedModel?: string;
|
|
29
|
-
constrainsOutput?: boolean;
|
|
30
|
-
|
|
40
|
+
constrainsOutput?: boolean; requiresOutputBudget?: boolean;
|
|
41
|
+
maxInputTokens?: number | null; maxOutputTokens?: number | null;
|
|
42
|
+
outputBudget?: number | null; reasoningBudget?: number | null;
|
|
31
43
|
tokenize?: boolean; throws?: Error;
|
|
32
44
|
promptMeasurement?: PromptTokenMeasurement;
|
|
33
45
|
};
|
|
@@ -37,17 +49,38 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
37
49
|
const b: Provider = {
|
|
38
50
|
model: opts.model ?? "gemma",
|
|
39
51
|
contextWindow: opts.window === undefined ? 48000 : opts.window,
|
|
52
|
+
maxInputTokens: opts.maxInputTokens ?? null,
|
|
53
|
+
maxOutputTokens: opts.maxOutputTokens ?? null,
|
|
54
|
+
outputBudget: opts.outputBudget ?? null,
|
|
55
|
+
reasoningBudget: opts.reasoningBudget ?? null,
|
|
56
|
+
inputCapacity: effectiveInputCapacity({
|
|
57
|
+
contextWindow: opts.window === undefined ? 48_000 : opts.window,
|
|
58
|
+
maxInputTokens: opts.maxInputTokens ?? null,
|
|
59
|
+
outputBudget: opts.outputBudget ?? null,
|
|
60
|
+
}),
|
|
40
61
|
...(opts.servedModel !== undefined ? { servedModel: opts.servedModel } : {}),
|
|
41
62
|
...(opts.constrainsOutput !== undefined ? { constrainsOutput: opts.constrainsOutput } : {}),
|
|
42
|
-
...(opts.
|
|
43
|
-
...(opts.reasoningReserve !== undefined ? { reasoningReserve: opts.reasoningReserve } : {}),
|
|
44
|
-
...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
|
|
63
|
+
...(opts.requiresOutputBudget !== undefined ? { requiresOutputBudget: opts.requiresOutputBudget } : {}),
|
|
45
64
|
...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
|
|
46
65
|
countPromptTokens: async (messages) => opts.promptMeasurement ?? ({
|
|
47
66
|
kind: "exact",
|
|
48
67
|
tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
|
|
49
68
|
source: "test:exact",
|
|
50
69
|
}),
|
|
70
|
+
assessRequestCapacity: async (messages, maxOutputTokens) => ({
|
|
71
|
+
decision: "admit",
|
|
72
|
+
contextWindow: opts.window === undefined ? 48_000 : opts.window,
|
|
73
|
+
maxInputTokens: opts.maxInputTokens ?? null,
|
|
74
|
+
maxOutputTokens: opts.maxOutputTokens ?? null,
|
|
75
|
+
outputBudget: maxOutputTokens ?? opts.outputBudget ?? null,
|
|
76
|
+
reasoningBudget: opts.reasoningBudget ?? null,
|
|
77
|
+
inputCapacity: null,
|
|
78
|
+
prompt: opts.promptMeasurement ?? {
|
|
79
|
+
kind: "exact",
|
|
80
|
+
tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
|
|
81
|
+
source: "test:exact",
|
|
82
|
+
},
|
|
83
|
+
}),
|
|
51
84
|
generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
|
|
52
85
|
served.push(args.workerId);
|
|
53
86
|
if (opts.throws !== undefined) throw opts.throws;
|
|
@@ -82,20 +115,22 @@ test("Pool: any unknown (null) window makes the pool null - no improvised cap",
|
|
|
82
115
|
assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: null }).b]).contextWindow, null);
|
|
83
116
|
});
|
|
84
117
|
|
|
85
|
-
test("Pool:
|
|
86
|
-
const big = backend({ window: 48000,
|
|
87
|
-
const small = backend({ window: 32000,
|
|
118
|
+
test("Pool: physical limits and budgets aggregate to independent safe floors", () => {
|
|
119
|
+
const big = backend({ window: 48000, maxInputTokens: 40_000, maxOutputTokens: 16_000, outputBudget: 12_000, reasoningBudget: 4_800 }).b;
|
|
120
|
+
const small = backend({ window: 32000, maxInputTokens: 24_000, maxOutputTokens: 12_000, outputBudget: 8_000, reasoningBudget: 3_200 }).b;
|
|
88
121
|
const p = new Pool([big, small]);
|
|
89
122
|
assert.equal(p.contextWindow, 32000);
|
|
90
|
-
assert.equal(p.
|
|
91
|
-
assert.equal(p.
|
|
123
|
+
assert.equal(p.maxInputTokens, 24_000);
|
|
124
|
+
assert.equal(p.maxOutputTokens, 12_000);
|
|
125
|
+
assert.equal(p.outputBudget, 8_000);
|
|
126
|
+
assert.equal(p.reasoningBudget, 3_200);
|
|
92
127
|
});
|
|
93
128
|
|
|
94
|
-
test("Pool: capabilities aggregate conservatively (constrainsOutput all-true,
|
|
129
|
+
test("Pool: capabilities aggregate conservatively (constrainsOutput all-true, requiresOutputBudget any-true)", () => {
|
|
95
130
|
assert.equal(new Pool([backend({ constrainsOutput: true }).b, backend({ constrainsOutput: true }).b]).constrainsOutput, true);
|
|
96
131
|
assert.equal(new Pool([backend({ constrainsOutput: true }).b, backend({ constrainsOutput: false }).b]).constrainsOutput, undefined);
|
|
97
|
-
assert.equal(new Pool([backend({
|
|
98
|
-
assert.equal(new Pool([backend({}).b]).
|
|
132
|
+
assert.equal(new Pool([backend({ requiresOutputBudget: false }).b, backend({ requiresOutputBudget: true }).b]).requiresOutputBudget, true);
|
|
133
|
+
assert.equal(new Pool([backend({}).b]).requiresOutputBudget, undefined);
|
|
99
134
|
});
|
|
100
135
|
|
|
101
136
|
test("Pool: servedModel is the common id, undefined when they differ", () => {
|
|
@@ -131,6 +166,29 @@ test("Pool: prompt evidence is conservative across every routable backend", asyn
|
|
|
131
166
|
source: "pool:test:a,heuristic:chars2",
|
|
132
167
|
detail: "at least one interchangeable backend has only an estimate: unknown framing",
|
|
133
168
|
});
|
|
169
|
+
|
|
170
|
+
const unavailable = backend({ promptMeasurement: {
|
|
171
|
+
kind: "unavailable", source: "test:none", detail: "counter offline",
|
|
172
|
+
} }).b;
|
|
173
|
+
assert.deepEqual(await new Pool([exact, unavailable]).countPromptTokens([]), {
|
|
174
|
+
kind: "unavailable",
|
|
175
|
+
source: "pool:test:a,test:none",
|
|
176
|
+
detail: "at least one interchangeable backend cannot measure the request: counter offline",
|
|
177
|
+
});
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
test("Pool: request capacity uses the smallest complete backend envelope", async () => {
|
|
181
|
+
const prompt = { kind: "exact", tokens: 11, source: "test:exact" } as const;
|
|
182
|
+
const narrowInput = backend({ window: 100, outputBudget: 90, promptMeasurement: prompt }).b;
|
|
183
|
+
const narrowContext = backend({ window: 50, outputBudget: 10, promptMeasurement: prompt }).b;
|
|
184
|
+
const pool = new Pool([narrowInput, narrowContext]);
|
|
185
|
+
|
|
186
|
+
assert.equal(pool.inputCapacity, 10);
|
|
187
|
+
const capacity = await pool.assessRequestCapacity([]);
|
|
188
|
+
assert.equal(capacity.contextWindow, 50);
|
|
189
|
+
assert.equal(capacity.outputBudget, 10);
|
|
190
|
+
assert.equal(capacity.inputCapacity, 10, "independent minima do not synthesize a nonexistent 40-token envelope");
|
|
191
|
+
assert.equal(capacity.decision, "reject");
|
|
134
192
|
});
|
|
135
193
|
|
|
136
194
|
// --- dispatch: round-robin + affinity ---
|
package/src/Pool.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Provider, ProviderRequestAccounting, ProviderResponse, ChatMessage, PromptTokenMeasurement } from "./types.ts";
|
|
1
|
+
import type { Provider, ProviderRequestAccounting, ProviderRequestCapacity, ProviderResponse, ChatMessage, PromptTokenMeasurement } from "./types.ts";
|
|
2
2
|
import { ProviderError, type ProviderErrorKind } from "./errors.ts";
|
|
3
3
|
import { emitWarningOnce } from "./warnings.ts";
|
|
4
4
|
import { assertPromptTokenMeasurement } from "./promptTokens.ts";
|
|
@@ -6,6 +6,7 @@ import Meta, {
|
|
|
6
6
|
type PluginAttribution,
|
|
7
7
|
type PluginAttributionContext,
|
|
8
8
|
} from "@plurnk/plurnk-meta";
|
|
9
|
+
import { effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget, requestCapacityDecision } from "./capacity.ts";
|
|
9
10
|
|
|
10
11
|
// A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
|
|
11
12
|
// transient retries before throwing one of these, so re-hitting the same
|
|
@@ -33,7 +34,7 @@ export default class Pool implements Provider {
|
|
|
33
34
|
readonly #affinity = new Map<string, number>(); // workerId -> backend index; sticky, LRU-bounded
|
|
34
35
|
readonly #cap: number; // LRU ceiling on the affinity map
|
|
35
36
|
#next = 0; // round-robin cursor for NEW workers
|
|
36
|
-
readonly #floor: Provider; // the min-window backend: the safe
|
|
37
|
+
readonly #floor: Provider; // the min-window backend: the safe context floor
|
|
37
38
|
|
|
38
39
|
// Optional exact tokenizer: present iff every backend exposes one (same
|
|
39
40
|
// vocab, since interchangeable). Delegated; absent when the fleet can't.
|
|
@@ -52,9 +53,9 @@ export default class Pool implements Provider {
|
|
|
52
53
|
this.#cap = backends.length * 8;
|
|
53
54
|
|
|
54
55
|
// The exposed window is the SAFE FLOOR across backends (a packet that fits the
|
|
55
|
-
// smallest fits all)
|
|
56
|
-
//
|
|
57
|
-
//
|
|
56
|
+
// smallest fits all). Every other capacity fact aggregates independently
|
|
57
|
+
// below. ANY unknown (null) window makes the pool null - a worker might
|
|
58
|
+
// route to it, and the consumer must not improvise a cap
|
|
58
59
|
// ({§model-fact-resolution}).
|
|
59
60
|
const anyUnknown = backends.find((b) => b.contextWindow === null);
|
|
60
61
|
this.#floor = anyUnknown ?? backends.reduce((lo, b) => (b.contextWindow! < lo.contextWindow! ? b : lo));
|
|
@@ -78,12 +79,21 @@ export default class Pool implements Provider {
|
|
|
78
79
|
|
|
79
80
|
get model(): string { return this.#backends[0].model; }
|
|
80
81
|
get contextWindow(): number | null { return this.#floor.contextWindow; }
|
|
81
|
-
|
|
82
|
-
|
|
82
|
+
#minimumKnown(project: (provider: Provider) => number | null): number | null {
|
|
83
|
+
const values = this.#backends.map(project);
|
|
84
|
+
return values.some((value) => value === null)
|
|
85
|
+
? null
|
|
86
|
+
: Math.min(...values as number[]);
|
|
87
|
+
}
|
|
88
|
+
get maxInputTokens(): number | null { return this.#minimumKnown((provider) => provider.maxInputTokens); }
|
|
89
|
+
get maxOutputTokens(): number | null { return this.#minimumKnown((provider) => provider.maxOutputTokens); }
|
|
90
|
+
get outputBudget(): number | null { return this.#minimumKnown((provider) => provider.outputBudget); }
|
|
91
|
+
get reasoningBudget(): number | null { return this.#minimumKnown((provider) => provider.reasoningBudget); }
|
|
92
|
+
get inputCapacity(): number | null { return this.#minimumKnown((provider) => provider.inputCapacity); }
|
|
83
93
|
|
|
84
94
|
// Served id / capabilities aggregate CONSERVATIVELY: a worker could land on any
|
|
85
95
|
// backend, so claim `constrainsOutput` only if EVERY backend does, and
|
|
86
|
-
// `
|
|
96
|
+
// `requiresOutputBudget` if ANY does (bring a cap if even one backend needs it).
|
|
87
97
|
get servedModel(): string | undefined {
|
|
88
98
|
const s = this.#backends[0].servedModel;
|
|
89
99
|
return this.#backends.every((b) => b.servedModel === s) ? s : undefined;
|
|
@@ -91,8 +101,8 @@ export default class Pool implements Provider {
|
|
|
91
101
|
get constrainsOutput(): boolean | undefined {
|
|
92
102
|
return this.#backends.every((b) => b.constrainsOutput === true) ? true : undefined;
|
|
93
103
|
}
|
|
94
|
-
get
|
|
95
|
-
return this.#backends.some((b) => b.
|
|
104
|
+
get requiresOutputBudget(): boolean | undefined {
|
|
105
|
+
return this.#backends.some((b) => b.requiresOutputBudget === true) ? true : undefined;
|
|
96
106
|
}
|
|
97
107
|
|
|
98
108
|
async countPromptTokens(
|
|
@@ -104,8 +114,20 @@ export default class Pool implements Provider {
|
|
|
104
114
|
await backend.countPromptTokens(messages, signal),
|
|
105
115
|
`Pool backend ${backend.model}`,
|
|
106
116
|
)));
|
|
107
|
-
const tokens = Math.max(...measurements.map((measurement) => measurement.tokens));
|
|
108
117
|
const sources = [...new Set(measurements.map(({ source }) => source))].join(",");
|
|
118
|
+
const unavailable = measurements.find((measurement) => measurement.kind === "unavailable");
|
|
119
|
+
if (unavailable?.kind === "unavailable") {
|
|
120
|
+
return {
|
|
121
|
+
kind: "unavailable",
|
|
122
|
+
source: `pool:${sources}`,
|
|
123
|
+
detail: `at least one interchangeable backend cannot measure the request: ${unavailable.detail}`,
|
|
124
|
+
};
|
|
125
|
+
}
|
|
126
|
+
const available = measurements.filter(
|
|
127
|
+
(measurement): measurement is Exclude<PromptTokenMeasurement, { readonly kind: "unavailable" }> =>
|
|
128
|
+
measurement.kind !== "unavailable",
|
|
129
|
+
);
|
|
130
|
+
const tokens = Math.max(...available.map((measurement) => measurement.tokens));
|
|
109
131
|
const estimate = measurements.find((measurement) => measurement.kind === "estimate");
|
|
110
132
|
if (estimate?.kind === "estimate") {
|
|
111
133
|
return {
|
|
@@ -115,14 +137,57 @@ export default class Pool implements Provider {
|
|
|
115
137
|
detail: `at least one interchangeable backend has only an estimate: ${estimate.detail}`,
|
|
116
138
|
};
|
|
117
139
|
}
|
|
118
|
-
const exact =
|
|
119
|
-
&&
|
|
140
|
+
const exact = available.every((measurement) => measurement.kind === "exact")
|
|
141
|
+
&& available.every((measurement) => measurement.tokens === tokens);
|
|
120
142
|
return {
|
|
121
143
|
kind: exact ? "exact" : "upper_bound",
|
|
122
144
|
tokens,
|
|
123
145
|
source: `pool:${sources}`,
|
|
124
146
|
};
|
|
125
147
|
}
|
|
148
|
+
|
|
149
|
+
async assessRequestCapacity(
|
|
150
|
+
messages: readonly ChatMessage[],
|
|
151
|
+
maxOutputTokens?: number,
|
|
152
|
+
signal?: AbortSignal,
|
|
153
|
+
): Promise<ProviderRequestCapacity> {
|
|
154
|
+
const envelopes = this.#backends.map((provider) => {
|
|
155
|
+
const outputBudget = effectiveOutputBudget({
|
|
156
|
+
requested: maxOutputTokens,
|
|
157
|
+
configured: provider.outputBudget,
|
|
158
|
+
maxOutputTokens: provider.maxOutputTokens,
|
|
159
|
+
contextWindow: provider.contextWindow,
|
|
160
|
+
});
|
|
161
|
+
return {
|
|
162
|
+
outputBudget,
|
|
163
|
+
reasoningBudget: effectiveReasoningBudget({
|
|
164
|
+
configured: provider.reasoningBudget,
|
|
165
|
+
outputBudget,
|
|
166
|
+
}),
|
|
167
|
+
inputCapacity: effectiveInputCapacity({
|
|
168
|
+
contextWindow: provider.contextWindow,
|
|
169
|
+
maxInputTokens: provider.maxInputTokens,
|
|
170
|
+
outputBudget,
|
|
171
|
+
}),
|
|
172
|
+
};
|
|
173
|
+
});
|
|
174
|
+
const minimum = (values: readonly (number | null)[]): number | null =>
|
|
175
|
+
values.some((value) => value === null)
|
|
176
|
+
? null
|
|
177
|
+
: Math.min(...values as number[]);
|
|
178
|
+
const measurement = await this.countPromptTokens(messages, signal);
|
|
179
|
+
const inputCapacity = minimum(envelopes.map((envelope) => envelope.inputCapacity));
|
|
180
|
+
return {
|
|
181
|
+
decision: requestCapacityDecision(inputCapacity, measurement),
|
|
182
|
+
contextWindow: this.contextWindow,
|
|
183
|
+
maxInputTokens: this.maxInputTokens,
|
|
184
|
+
maxOutputTokens: this.maxOutputTokens,
|
|
185
|
+
outputBudget: minimum(envelopes.map((envelope) => envelope.outputBudget)),
|
|
186
|
+
reasoningBudget: minimum(envelopes.map((envelope) => envelope.reasoningBudget)),
|
|
187
|
+
inputCapacity,
|
|
188
|
+
prompt: measurement,
|
|
189
|
+
};
|
|
190
|
+
}
|
|
126
191
|
// --- dispatch ---
|
|
127
192
|
|
|
128
193
|
async generate(args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> {
|
|
@@ -17,7 +17,7 @@ const fullEnv = Object.freeze({
|
|
|
17
17
|
PLURNK_PROVIDERS_OPERATION_TIMEOUT: "2700000",
|
|
18
18
|
PLURNK_PROVIDERS_FIRST_CONTENT_TIMEOUT: "600000",
|
|
19
19
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
20
|
-
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4",
|
|
20
|
+
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_OUTPUT_BUDGET: "35%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_ERROR_DETAIL_LIMIT: "512", PLURNK_PROVIDERS_CACHE_AFFINITY: "1", PLURNK_PROVIDERS_CACHE_WRITE_POLICY: "stable-system",
|
|
21
21
|
OPENAI_BASE_URL: "http://x",
|
|
22
22
|
});
|
|
23
23
|
|
package/src/accounting.ts
CHANGED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
export { aggregateProviderAccounting } from "./accounting.ts";
|
|
2
|
+
export { estimateProviderCost } from "./cost.ts";
|
|
3
|
+
export type {
|
|
4
|
+
ProviderAccounting,
|
|
5
|
+
ProviderCost,
|
|
6
|
+
ProviderRequestAccounting,
|
|
7
|
+
ProviderUsage,
|
|
8
|
+
} from "./types.ts";
|
|
9
|
+
export type { TokenRates } from "./usage.ts";
|
package/src/boundaries.test.ts
CHANGED
|
@@ -26,32 +26,45 @@ test("provider source does not import the PLURNK parser", () => {
|
|
|
26
26
|
}
|
|
27
27
|
});
|
|
28
28
|
|
|
29
|
+
const assertRuntimeNeutralGraph = (entrypoint: string, allowed: ReadonlySet<string>): void => {
|
|
30
|
+
const pending = [entrypoint];
|
|
31
|
+
const visited = new Set<string>();
|
|
32
|
+
while (pending.length > 0) {
|
|
33
|
+
const file = pending.pop()!;
|
|
34
|
+
if (visited.has(file)) continue;
|
|
35
|
+
visited.add(file);
|
|
36
|
+
assert.ok(allowed.has(file), `${entrypoint} reaches ${file}`);
|
|
37
|
+
const source = readFileSync(join(root, file), "utf8");
|
|
38
|
+
assert.doesNotMatch(source, /from\s+["']node:/, `${file} imports a Node built-in`);
|
|
39
|
+
for (const match of source.matchAll(/from\s+["']\.\/([^"']+)\.ts["']/g)) {
|
|
40
|
+
pending.push(`${match[1]}.ts`);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
};
|
|
44
|
+
|
|
29
45
|
test("the OpenAI-compatible entrypoint excludes Node-owned provider machinery", () => {
|
|
30
|
-
|
|
46
|
+
assertRuntimeNeutralGraph("openai.ts", new Set([
|
|
31
47
|
"accounting.ts",
|
|
32
48
|
"AiSdkProvider.ts",
|
|
33
49
|
"aiSdkTransport.ts",
|
|
50
|
+
"capacity.ts",
|
|
34
51
|
"cost.ts",
|
|
35
52
|
"env.ts",
|
|
36
53
|
"errors.ts",
|
|
37
54
|
"notices.ts",
|
|
38
55
|
"openai.ts",
|
|
39
56
|
"promptTokens.ts",
|
|
57
|
+
"providerError.ts",
|
|
40
58
|
"types.ts",
|
|
41
59
|
"usage.ts",
|
|
42
60
|
"warnings.ts",
|
|
43
|
-
]);
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
assert.doesNotMatch(source, /from\s+["']node:/, `${file} imports a Node built-in`);
|
|
53
|
-
for (const match of source.matchAll(/from\s+["']\.\/([^"']+)\.ts["']/g)) {
|
|
54
|
-
pending.push(`${match[1]}.ts`);
|
|
55
|
-
}
|
|
56
|
-
}
|
|
61
|
+
]));
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test("the normalized-error entrypoint excludes transport and Node machinery", () => {
|
|
65
|
+
assertRuntimeNeutralGraph("providerError.ts", new Set([
|
|
66
|
+
"notices.ts",
|
|
67
|
+
"providerError.ts",
|
|
68
|
+
"types.ts",
|
|
69
|
+
]));
|
|
57
70
|
});
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import test from "node:test";
|
|
2
|
+
import assert from "node:assert/strict";
|
|
3
|
+
import { assessRequestCapacity, effectiveInputCapacity, effectiveOutputBudget, effectiveReasoningBudget } from "./capacity.ts";
|
|
4
|
+
|
|
5
|
+
test("effective output budget is caller-tightenable and physically capped", () => {
|
|
6
|
+
assert.equal(effectiveOutputBudget({
|
|
7
|
+
requested: undefined,
|
|
8
|
+
configured: 40_000,
|
|
9
|
+
maxOutputTokens: 32_000,
|
|
10
|
+
contextWindow: 128_000,
|
|
11
|
+
}), 32_000);
|
|
12
|
+
assert.equal(effectiveOutputBudget({
|
|
13
|
+
requested: 8_000,
|
|
14
|
+
configured: 40_000,
|
|
15
|
+
maxOutputTokens: 32_000,
|
|
16
|
+
contextWindow: 128_000,
|
|
17
|
+
}), 8_000);
|
|
18
|
+
});
|
|
19
|
+
|
|
20
|
+
test("call-specific output tightening also tightens its reasoning subset", () => {
|
|
21
|
+
assert.equal(effectiveReasoningBudget({ configured: 8_000, outputBudget: 4_000 }), 3_999);
|
|
22
|
+
assert.equal(effectiveReasoningBudget({ configured: 2_000, outputBudget: 4_000 }), 2_000);
|
|
23
|
+
assert.equal(effectiveReasoningBudget({ configured: null, outputBudget: 1 }), null);
|
|
24
|
+
assert.throws(
|
|
25
|
+
() => effectiveReasoningBudget({ configured: 1, outputBudget: 1 }),
|
|
26
|
+
/leave at least one token outside the reasoning budget/,
|
|
27
|
+
);
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
test("capacity applies independent input and combined-context limits", () => {
|
|
31
|
+
assert.equal(effectiveInputCapacity({
|
|
32
|
+
contextWindow: 100_000,
|
|
33
|
+
maxInputTokens: 70_000,
|
|
34
|
+
outputBudget: 20_000,
|
|
35
|
+
}), 70_000);
|
|
36
|
+
const capacity = assessRequestCapacity({
|
|
37
|
+
contextWindow: 100_000,
|
|
38
|
+
maxInputTokens: 70_000,
|
|
39
|
+
maxOutputTokens: 40_000,
|
|
40
|
+
outputBudget: 20_000,
|
|
41
|
+
reasoningBudget: null,
|
|
42
|
+
measurement: { kind: "exact", tokens: 70_001, source: "fixture" },
|
|
43
|
+
});
|
|
44
|
+
assert.equal(capacity.inputCapacity, 70_000);
|
|
45
|
+
assert.equal(capacity.decision, "reject");
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
test("a known combined context must leave positive input capacity", () => {
|
|
49
|
+
assert.equal(effectiveInputCapacity({
|
|
50
|
+
contextWindow: 2,
|
|
51
|
+
maxInputTokens: null,
|
|
52
|
+
outputBudget: 1,
|
|
53
|
+
}), 1);
|
|
54
|
+
assert.throws(
|
|
55
|
+
() => effectiveInputCapacity({
|
|
56
|
+
contextWindow: 1,
|
|
57
|
+
maxInputTokens: null,
|
|
58
|
+
outputBudget: 1,
|
|
59
|
+
}),
|
|
60
|
+
/must leave positive input capacity/,
|
|
61
|
+
);
|
|
62
|
+
});
|
|
63
|
+
|
|
64
|
+
test("only exact overflow rejects before provider I/O", () => {
|
|
65
|
+
const base = {
|
|
66
|
+
contextWindow: 100,
|
|
67
|
+
maxInputTokens: null,
|
|
68
|
+
maxOutputTokens: 40,
|
|
69
|
+
outputBudget: 40,
|
|
70
|
+
reasoningBudget: null,
|
|
71
|
+
} as const;
|
|
72
|
+
assert.equal(assessRequestCapacity({
|
|
73
|
+
...base,
|
|
74
|
+
measurement: { kind: "exact", tokens: 61, source: "exact" },
|
|
75
|
+
}).decision, "reject");
|
|
76
|
+
assert.equal(assessRequestCapacity({
|
|
77
|
+
...base,
|
|
78
|
+
measurement: { kind: "upper_bound", tokens: 61, source: "bound" },
|
|
79
|
+
}).decision, "defer");
|
|
80
|
+
assert.equal(assessRequestCapacity({
|
|
81
|
+
...base,
|
|
82
|
+
measurement: { kind: "estimate", tokens: 61, source: "estimate", detail: "heuristic" },
|
|
83
|
+
}).decision, "defer");
|
|
84
|
+
assert.equal(assessRequestCapacity({
|
|
85
|
+
...base,
|
|
86
|
+
measurement: { kind: "unavailable", source: "fixture", detail: "no request tokenizer" },
|
|
87
|
+
}).decision, "defer");
|
|
88
|
+
assert.equal(assessRequestCapacity({
|
|
89
|
+
...base,
|
|
90
|
+
measurement: { kind: "upper_bound", tokens: 60, source: "bound" },
|
|
91
|
+
}).decision, "admit");
|
|
92
|
+
});
|