@plurnk/plurnk-providers 1.3.12 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +39 -22
- package/README.md +65 -4
- package/SPEC.md +222 -56
- package/dist/AiSdkProvider.d.ts +18 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +240 -153
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +14 -15
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +26 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +18 -3
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +3 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +20 -7
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +64 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +29 -9
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +150 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -4
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +4 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +43 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +3 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +26 -14
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +13 -9
- package/src/AiSdkProvider.test.ts +480 -159
- package/src/AiSdkProvider.ts +320 -196
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +33 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/aiSdkTransport.ts +25 -6
- package/src/boundaries.test.ts +8 -3
- package/src/catalogProvider.test.ts +17 -0
- package/src/catalogProvider.ts +25 -10
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +63 -0
- package/src/cost.ts +83 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -23
- package/src/env.ts +45 -18
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +207 -0
- package/src/index.ts +29 -7
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +7 -0
- package/src/sdkModels.ts +4 -8
- package/src/types.ts +106 -64
- package/src/usage.test.ts +15 -4
- package/src/usage.ts +32 -14
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
package/src/Mock.test.ts
CHANGED
|
@@ -7,7 +7,7 @@ import type { MockResponse } from "./Mock.ts";
|
|
|
7
7
|
const build = (responses: MockResponse[] = [{ assistant: { content: "hi", reasoning: null } }]) =>
|
|
8
8
|
new Mock({ contextWindow: 100000, responses });
|
|
9
9
|
|
|
10
|
-
// — Identity (
|
|
10
|
+
// — Identity ({§provider-interface}) —
|
|
11
11
|
|
|
12
12
|
test("Mock: contextWindow and model are stable across reads", () => {
|
|
13
13
|
const m = build();
|
|
@@ -22,21 +22,24 @@ test("Mock: contextWindow passes null through", () => {
|
|
|
22
22
|
assert.equal(m.contextWindow, null);
|
|
23
23
|
});
|
|
24
24
|
|
|
25
|
-
// — Tokenomics (
|
|
25
|
+
// — Tokenomics ({§provider-interface}) —
|
|
26
26
|
|
|
27
|
-
test("Mock:
|
|
27
|
+
test("Mock: prompt counting is exact for its declared mock vocabulary", async () => {
|
|
28
28
|
const m = build();
|
|
29
|
-
assert.
|
|
30
|
-
|
|
31
|
-
|
|
29
|
+
assert.deepEqual(await m.countPromptTokens([]), {
|
|
30
|
+
kind: "exact", tokens: 0, source: "mock:chars2",
|
|
31
|
+
});
|
|
32
|
+
assert.deepEqual(await m.countPromptTokens([{ role: "user", content: "four" }]), {
|
|
33
|
+
kind: "exact", tokens: 2, source: "mock:chars2",
|
|
34
|
+
});
|
|
32
35
|
});
|
|
33
36
|
|
|
34
|
-
test("Mock: calculateCost
|
|
37
|
+
test("Mock: calculateCost returns its deliberate zero estimate", () => {
|
|
35
38
|
const m = build();
|
|
36
|
-
assert.equal(m.calculateCost({ prompt:
|
|
39
|
+
assert.equal(m.calculateCost({ prompt: 100, completion: 20, reasoning: 10, cached: 5, total: 130 }), 0);
|
|
37
40
|
});
|
|
38
41
|
|
|
39
|
-
// — Transport (
|
|
42
|
+
// — Transport ({§provider-interface}) —
|
|
40
43
|
|
|
41
44
|
test("Mock: generate resolves a valid ProviderResponse shape", async () => {
|
|
42
45
|
const m = build([{ assistant: { content: "hello", reasoning: "cot" } }]);
|
|
@@ -67,6 +70,18 @@ test("Mock: generate applies caller-supplied overrides", async () => {
|
|
|
67
70
|
assert.deepEqual(assistantRaw, { wire: true });
|
|
68
71
|
});
|
|
69
72
|
|
|
73
|
+
test("Mock: a supplied grammar produces unsplit evidence unless the fixture supplies exact evidence", async () => {
|
|
74
|
+
const explicit = { input: "prefixx", contentStart: 6, transported: false } as const;
|
|
75
|
+
const m = build([
|
|
76
|
+
{ assistant: { content: "x", reasoning: null } },
|
|
77
|
+
{ assistant: { content: "x", reasoning: null }, grammarEvidence: explicit },
|
|
78
|
+
]);
|
|
79
|
+
const inferred = await m.generate({ messages: [], grammar: 'root ::= "x"' });
|
|
80
|
+
const supplied = await m.generate({ messages: [], grammar: 'root ::= "prefixx"' });
|
|
81
|
+
assert.deepEqual(inferred.grammarEvidence, { input: "x", contentStart: 0, transported: true });
|
|
82
|
+
assert.deepEqual(supplied.grammarEvidence, explicit);
|
|
83
|
+
});
|
|
84
|
+
|
|
70
85
|
test("Mock: ops escape hatch passes through when provided", async () => {
|
|
71
86
|
const ops = [{ kind: "send" }] as never;
|
|
72
87
|
const m = build([{ assistant: { content: "x", reasoning: null, ops } }]);
|
|
@@ -80,7 +95,7 @@ test("Mock: ops absent when not provided", async () => {
|
|
|
80
95
|
assert.equal("ops" in assistant, false);
|
|
81
96
|
});
|
|
82
97
|
|
|
83
|
-
// — Abort (
|
|
98
|
+
// — Abort ({§provider-failure-normalization}) —
|
|
84
99
|
|
|
85
100
|
test("Mock: generate with a pre-aborted signal rejects and consumes no response", async () => {
|
|
86
101
|
const m = build([{ assistant: { content: "untouched", reasoning: null } }]);
|
|
@@ -106,9 +121,9 @@ test("Mock: exhausted queue throws a specific error", async () => {
|
|
|
106
121
|
await assert.rejects(() => m.generate({ messages: [] }), /exhausted/);
|
|
107
122
|
});
|
|
108
123
|
|
|
109
|
-
// --
|
|
124
|
+
// -- {§provider-generation-envelope} --
|
|
110
125
|
|
|
111
|
-
test("
|
|
126
|
+
test("the reserve getters are on the Provider interface (not just the concrete class)", () => {
|
|
112
127
|
// Typing against the contract catches a getter-only concrete surface.
|
|
113
128
|
const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
|
|
114
129
|
try {
|
|
@@ -120,7 +135,7 @@ test("#507 the reserve getters are on the Provider interface (not just the concr
|
|
|
120
135
|
}
|
|
121
136
|
});
|
|
122
137
|
|
|
123
|
-
test("
|
|
138
|
+
test("Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its window (the service partition path)", () => {
|
|
124
139
|
const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
|
|
125
140
|
const prevC = process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE;
|
|
126
141
|
try {
|
|
@@ -135,7 +150,7 @@ test("#507 Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its wi
|
|
|
135
150
|
}
|
|
136
151
|
});
|
|
137
152
|
|
|
138
|
-
test("
|
|
153
|
+
test("no reserve env → null (the no-cap path; bare Mocks unaffected)", () => {
|
|
139
154
|
const m = new Mock({ contextWindow: 49152, responses: [] });
|
|
140
155
|
assert.equal(m.reasoningReserve, null);
|
|
141
156
|
assert.equal(m.completionReserve, null);
|
package/src/Mock.ts
CHANGED
|
@@ -5,7 +5,8 @@
|
|
|
5
5
|
// Provider contract. Production providers don't expose the `ops` escape
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
7
|
|
|
8
|
-
import type { ChatMessage, FinishReason, Provider, ProviderAssistant, ProviderUsage } from "./types.ts";
|
|
8
|
+
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
9
10
|
import { resolveEnvelopeFromEnv } from "./env.ts";
|
|
10
11
|
|
|
11
12
|
export type MockAssistant = {
|
|
@@ -15,11 +16,10 @@ export type MockAssistant = {
|
|
|
15
16
|
usage?: Partial<ProviderUsage>;
|
|
16
17
|
finishReason?: FinishReason;
|
|
17
18
|
model?: string;
|
|
18
|
-
//
|
|
19
|
-
|
|
20
|
-
reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
|
|
19
|
+
// Provider-normalized encrypted reasoning fixture.
|
|
20
|
+
reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
|
|
21
21
|
// Pre-parsed ops — intg-only escape hatch. Typed `unknown[]` so the
|
|
22
|
-
// framework carries
|
|
22
|
+
// framework carries no parser dependency; plurnk-service
|
|
23
23
|
// casts these to PlurnkStatement[] on its side. Production providers never
|
|
24
24
|
// include this field.
|
|
25
25
|
ops?: unknown[];
|
|
@@ -28,10 +28,12 @@ export type MockAssistant = {
|
|
|
28
28
|
export type MockResponse = {
|
|
29
29
|
assistant: MockAssistant;
|
|
30
30
|
assistantRaw?: unknown;
|
|
31
|
+
grammarEvidence?: GrammarEvidence;
|
|
31
32
|
};
|
|
32
33
|
|
|
33
34
|
// Returned shape: ProviderAssistant + pre-parsed ops visible for tests.
|
|
34
35
|
export type MockReturnedAssistant = ProviderAssistant & { ops?: unknown[] };
|
|
36
|
+
export type MockReturnedResponse = ProviderResponse & { assistant: MockReturnedAssistant };
|
|
35
37
|
|
|
36
38
|
const DEFAULT_USAGE: ProviderUsage = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
|
|
37
39
|
|
|
@@ -41,8 +43,9 @@ export default class Mock implements Provider {
|
|
|
41
43
|
#completionReserve: number | null;
|
|
42
44
|
#queue: MockResponse[];
|
|
43
45
|
|
|
44
|
-
//
|
|
45
|
-
// read (the service's partition/budget
|
|
46
|
+
// {§provider-generation-envelope} Reserves resolve the same way a real
|
|
47
|
+
// provider's do — the tolerant env read (the service's partition/budget
|
|
48
|
+
// suite sets PLURNK_PROVIDERS_*_RESERVE
|
|
46
49
|
// and Mock reflects it, resolved against contextWindow). Tolerant, NOT the
|
|
47
50
|
// fail-hard envelopeFromEnv: Mock is the universal fixture, constructed
|
|
48
51
|
// without the reserves in most base tests, so absent env → null → the
|
|
@@ -61,18 +64,25 @@ export default class Mock implements Provider {
|
|
|
61
64
|
get completionReserve(): number | null { return this.#completionReserve; }
|
|
62
65
|
get model(): string { return "mock"; }
|
|
63
66
|
|
|
64
|
-
//
|
|
65
|
-
//
|
|
66
|
-
|
|
67
|
-
|
|
67
|
+
// Mock's deliberately simple vocabulary defines each two content code units
|
|
68
|
+
// as one token and has no hidden request framing. This is exact for the mock,
|
|
69
|
+
// unlike a production adapter applying the same arithmetic to an unknown model.
|
|
70
|
+
async countPromptTokens(messages: readonly ChatMessage[]): Promise<PromptTokenMeasurement> {
|
|
71
|
+
return {
|
|
72
|
+
kind: "exact",
|
|
73
|
+
tokens: messages.reduce((sum, { content }) => sum + Math.ceil(content.length / 2), 0),
|
|
74
|
+
source: "mock:chars2",
|
|
75
|
+
};
|
|
68
76
|
}
|
|
69
77
|
|
|
70
78
|
// Mock is free.
|
|
71
79
|
calculateCost(_usage: ProviderUsage): number { return 0; }
|
|
80
|
+
calculateCharge(_usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> { return { kind: "free", source: "mock provider" }; }
|
|
72
81
|
|
|
73
|
-
async generate({ signal }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal }): Promise<
|
|
82
|
+
async generate({ signal, grammar }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal; grammar?: string }): Promise<MockReturnedResponse> {
|
|
74
83
|
// Honor abort before consuming the queue — an aborted call makes no
|
|
75
|
-
// "wire call" and must not exhaust a queued response
|
|
84
|
+
// "wire call" and must not exhaust a queued response
|
|
85
|
+
// ({§provider-failure-normalization}).
|
|
76
86
|
signal?.throwIfAborted();
|
|
77
87
|
const next = this.#queue.shift();
|
|
78
88
|
if (next === undefined) throw new Error("Mock provider exhausted: no more queued responses");
|
|
@@ -81,12 +91,20 @@ export default class Mock implements Provider {
|
|
|
81
91
|
content: a.content,
|
|
82
92
|
reasoning: a.reasoning,
|
|
83
93
|
usage: { ...DEFAULT_USAGE, ...a.usage },
|
|
84
|
-
...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}),
|
|
94
|
+
...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}),
|
|
85
95
|
finishReason: a.finishReason ?? "stop",
|
|
86
96
|
model: a.model ?? "mock",
|
|
87
97
|
...(a.ops !== undefined ? { ops: a.ops } : {}),
|
|
88
98
|
};
|
|
89
|
-
|
|
99
|
+
const grammarEvidence = next.grammarEvidence
|
|
100
|
+
?? (grammar === undefined
|
|
101
|
+
? undefined
|
|
102
|
+
: { input: assistant.content, contentStart: 0, transported: true });
|
|
103
|
+
return {
|
|
104
|
+
assistant,
|
|
105
|
+
assistantRaw: next.assistantRaw ?? null,
|
|
106
|
+
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
107
|
+
};
|
|
90
108
|
}
|
|
91
109
|
|
|
92
110
|
get remaining(): number { return this.#queue.length; }
|
package/src/Pool.test.ts
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
import test from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import Pool from "./Pool.ts";
|
|
4
|
-
import { ProviderError } from "./
|
|
5
|
-
import type { Provider, ProviderResponse } from "./types.ts";
|
|
4
|
+
import { ProviderError } from "./errors.ts";
|
|
5
|
+
import type { PromptTokenMeasurement, Provider, ProviderResponse } from "./types.ts";
|
|
6
6
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
7
7
|
|
|
8
8
|
test.afterEach(() => { resetEmittedWarnings(); });
|
|
@@ -14,6 +14,7 @@ type FakeOpts = {
|
|
|
14
14
|
constrainsOutput?: boolean; requiresMaxTokens?: boolean;
|
|
15
15
|
reasoningReserve?: number | null; completionReserve?: number | null;
|
|
16
16
|
tokenize?: boolean; cost?: number; throws?: Error;
|
|
17
|
+
promptMeasurement?: PromptTokenMeasurement;
|
|
17
18
|
};
|
|
18
19
|
// A fake backend that records which workers it served, and optionally throws.
|
|
19
20
|
const backend = (opts: FakeOpts = {}) => {
|
|
@@ -27,7 +28,11 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
27
28
|
...(opts.reasoningReserve !== undefined ? { reasoningReserve: opts.reasoningReserve } : {}),
|
|
28
29
|
...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
|
|
29
30
|
...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
|
|
30
|
-
|
|
31
|
+
countPromptTokens: async (messages) => opts.promptMeasurement ?? ({
|
|
32
|
+
kind: "exact",
|
|
33
|
+
tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
|
|
34
|
+
source: "test:exact",
|
|
35
|
+
}),
|
|
31
36
|
calculateCost: () => opts.cost ?? 0,
|
|
32
37
|
generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
|
|
33
38
|
served.push(args.workerId);
|
|
@@ -39,6 +44,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
39
44
|
};
|
|
40
45
|
const netErr = () => new ProviderError("provider:x", "network_failure", "down", { status: 503 });
|
|
41
46
|
const authErr = () => new ProviderError("provider:x", "unauthorized", "no key", { status: 401 });
|
|
47
|
+
const interruptedErr = () => new ProviderError("provider:x", "resource_interrupted", "generation interrupted");
|
|
42
48
|
const gen = (p: Pool, workerId: string, extra: Record<string, unknown> = {}) =>
|
|
43
49
|
p.generate({ messages: [], workerId, ...extra } as Parameters<Provider["generate"]>[0]);
|
|
44
50
|
|
|
@@ -58,7 +64,7 @@ test("Pool: contextWindow is the safe floor (min) across backends", () => {
|
|
|
58
64
|
assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: 32000 }).b]).contextWindow, 32000);
|
|
59
65
|
});
|
|
60
66
|
|
|
61
|
-
test("Pool: any unknown (null) window makes the pool null - no improvised cap
|
|
67
|
+
test("Pool: any unknown (null) window makes the pool null - no improvised cap", () => {
|
|
62
68
|
assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: null }).b]).contextWindow, null);
|
|
63
69
|
});
|
|
64
70
|
|
|
@@ -88,12 +94,32 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
|
|
|
88
94
|
assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
|
|
89
95
|
});
|
|
90
96
|
|
|
91
|
-
test("Pool:
|
|
97
|
+
test("Pool: prompt counting + calculateCost delegate to a backend", async () => {
|
|
92
98
|
const p = new Pool([backend({ cost: 42 }).b]);
|
|
93
|
-
assert.
|
|
99
|
+
assert.deepEqual(await p.countPromptTokens([{ role: "user", content: "abcd" }]), {
|
|
100
|
+
kind: "exact", tokens: 4, source: "pool:test:exact",
|
|
101
|
+
});
|
|
94
102
|
assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
|
|
95
103
|
});
|
|
96
104
|
|
|
105
|
+
test("Pool: prompt evidence is conservative across every routable backend", async () => {
|
|
106
|
+
const exact = backend({ promptMeasurement: { kind: "exact", tokens: 8, source: "test:a" } }).b;
|
|
107
|
+
const larger = backend({ promptMeasurement: { kind: "exact", tokens: 11, source: "test:b" } }).b;
|
|
108
|
+
assert.deepEqual(await new Pool([exact, larger]).countPromptTokens([]), {
|
|
109
|
+
kind: "upper_bound", tokens: 11, source: "pool:test:a,test:b",
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
const estimate = backend({ promptMeasurement: {
|
|
113
|
+
kind: "estimate", tokens: 3, source: "heuristic:chars2", detail: "unknown framing",
|
|
114
|
+
} }).b;
|
|
115
|
+
assert.deepEqual(await new Pool([exact, estimate]).countPromptTokens([]), {
|
|
116
|
+
kind: "estimate",
|
|
117
|
+
tokens: 8,
|
|
118
|
+
source: "pool:test:a,heuristic:chars2",
|
|
119
|
+
detail: "at least one interchangeable backend has only an estimate: unknown framing",
|
|
120
|
+
});
|
|
121
|
+
});
|
|
122
|
+
|
|
97
123
|
// --- dispatch: round-robin + affinity ---
|
|
98
124
|
|
|
99
125
|
test("Pool: NEW workers round-robin across backends", async () => {
|
|
@@ -134,6 +160,17 @@ test("Pool: a terminal (auth) failure does NOT overflow", async () => {
|
|
|
134
160
|
assert.deepEqual(other.served, []); // never tried - auth fails the same on a peer
|
|
135
161
|
});
|
|
136
162
|
|
|
163
|
+
test("#161: Pool does not overflow a resource-interrupted attempt and discard its evidence", async () => {
|
|
164
|
+
const interrupted = backend({ throws: interruptedErr() }), other = backend();
|
|
165
|
+
const pool = new Pool([interrupted.b, other.b]);
|
|
166
|
+
await assert.rejects(
|
|
167
|
+
() => gen(pool, "w1"),
|
|
168
|
+
(error: unknown) => error instanceof ProviderError && error.kind === "resource_interrupted",
|
|
169
|
+
);
|
|
170
|
+
assert.deepEqual(interrupted.served, ["w1"]);
|
|
171
|
+
assert.deepEqual(other.served, []);
|
|
172
|
+
});
|
|
173
|
+
|
|
137
174
|
test("Pool: whole fleet unavailable throws the last error, each backend tried once", async () => {
|
|
138
175
|
const a = backend({ throws: netErr() }), b = backend({ throws: netErr() });
|
|
139
176
|
const p = new Pool([a.b, b.b]);
|
package/src/Pool.ts
CHANGED
|
@@ -1,13 +1,20 @@
|
|
|
1
|
-
import type { Provider, ProviderResponse, ProviderUsage, ChatMessage } from "./types.ts";
|
|
2
|
-
import {
|
|
1
|
+
import type { Provider, ProviderResponse, ProviderUsage, ChatMessage, PromptTokenMeasurement } from "./types.ts";
|
|
2
|
+
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
3
|
+
import { resolveProviderCost } from "./cost.ts";
|
|
4
|
+
import { ProviderError, type ProviderErrorKind } from "./errors.ts";
|
|
3
5
|
import { emitWarningOnce } from "./warnings.ts";
|
|
6
|
+
import { assertPromptTokenMeasurement } from "./promptTokens.ts";
|
|
7
|
+
import Meta, {
|
|
8
|
+
type PluginAttribution,
|
|
9
|
+
type PluginAttributionContext,
|
|
10
|
+
} from "@plurnk/plurnk-meta";
|
|
4
11
|
|
|
5
12
|
// A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
|
|
6
|
-
// transient retries
|
|
13
|
+
// transient retries before throwing one of these, so re-hitting the same
|
|
7
14
|
// backend is pointless - a sibling may still serve, so the pool overflows to it.
|
|
8
15
|
// Auth/quota/content kinds are deliberately absent: a peer backend fails them
|
|
9
16
|
// identically, and failing over would only multiply the damage (and the spend).
|
|
10
|
-
const OVERFLOW_KINDS: ReadonlySet<
|
|
17
|
+
const OVERFLOW_KINDS: ReadonlySet<ProviderErrorKind> = new Set(["network_failure", "rate_limit"]);
|
|
11
18
|
|
|
12
19
|
// Pool - fronts N INTERCHANGEABLE backends as one Provider. This is CAPACITY, not
|
|
13
20
|
// blend: the mechanism (round-robin across workers, sticky within a worker for
|
|
@@ -15,11 +22,12 @@ const OVERFLOW_KINDS: ReadonlySet<ProviderTelemetryKind> = new Set(["network_fai
|
|
|
15
22
|
// blend/escalation DECISION (which SKU, when to switch models) stays the
|
|
16
23
|
// consumer's, one level up, by choosing WHICH pool to call. Backends MUST be
|
|
17
24
|
// interchangeable - same served model, compatible window - so the pool presents ONE
|
|
18
|
-
// honest Provider surface instead of pretending N models are one
|
|
25
|
+
// honest Provider surface instead of pretending N models are one
|
|
26
|
+
// ({§provider-capacity-pool}).
|
|
19
27
|
//
|
|
20
28
|
// Affinity is the load-bearing part. A worker's turns stick to one backend so its
|
|
21
29
|
// stable prompt prefix keeps hitting the same KV cache; scattering a worker across
|
|
22
|
-
// backends shreds the prefix cache
|
|
30
|
+
// backends shreds the prefix cache. It is the same slot-affinity pattern one
|
|
23
31
|
// level up: worker -> slot within a llama-server becomes worker -> backend across a
|
|
24
32
|
// fleet - round-robin ACROSS workers, sticky WITHIN one.
|
|
25
33
|
export default class Pool implements Provider {
|
|
@@ -29,9 +37,10 @@ export default class Pool implements Provider {
|
|
|
29
37
|
#next = 0; // round-robin cursor for NEW workers
|
|
30
38
|
readonly #floor: Provider; // the min-window backend: the safe budget floor
|
|
31
39
|
|
|
32
|
-
//
|
|
40
|
+
// Optional exact tokenizer: present iff every backend exposes one (same
|
|
33
41
|
// vocab, since interchangeable). Delegated; absent when the fleet can't.
|
|
34
42
|
readonly tokenize?: (text: string) => Promise<number[]>;
|
|
43
|
+
readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
|
|
35
44
|
|
|
36
45
|
constructor(backends: readonly Provider[]) {
|
|
37
46
|
if (backends.length === 0) throw new Error("Pool: at least one backend is required");
|
|
@@ -40,14 +49,15 @@ export default class Pool implements Provider {
|
|
|
40
49
|
// SELECTION job, not this primitive - fail at construction, never mid-turn.
|
|
41
50
|
const model = backends[0].model;
|
|
42
51
|
const mixed = backends.find((b) => b.model !== model);
|
|
43
|
-
if (mixed !== undefined) throw new Error(`Pool: backends must be interchangeable, got mixed models "${model}" and "${mixed.model}" - heterogeneous blend is the consumer's selection, not a pool
|
|
52
|
+
if (mixed !== undefined) throw new Error(`Pool: backends must be interchangeable, got mixed models "${model}" and "${mixed.model}" - heterogeneous blend is the consumer's selection, not a pool`);
|
|
44
53
|
this.#backends = backends;
|
|
45
54
|
this.#cap = backends.length * 8;
|
|
46
55
|
|
|
47
56
|
// The exposed window is the SAFE FLOOR across backends (a packet that fits the
|
|
48
57
|
// smallest fits all); its reserves travel with it so the budget stays
|
|
49
58
|
// self-consistent. ANY unknown (null) window makes the pool null - a worker
|
|
50
|
-
// might route to it, and the consumer must not improvise a cap
|
|
59
|
+
// might route to it, and the consumer must not improvise a cap
|
|
60
|
+
// ({§model-fact-resolution}).
|
|
51
61
|
const anyUnknown = backends.find((b) => b.contextWindow === null);
|
|
52
62
|
this.#floor = anyUnknown ?? backends.reduce((lo, b) => (b.contextWindow! < lo.contextWindow! ? b : lo));
|
|
53
63
|
if (new Set(backends.map((b) => b.contextWindow)).size > 1) {
|
|
@@ -59,6 +69,11 @@ export default class Pool implements Provider {
|
|
|
59
69
|
|
|
60
70
|
// tokenize is a per-instance optional method; expose it iff the whole fleet has it.
|
|
61
71
|
if (backends.every((b) => typeof b.tokenize === "function")) this.tokenize = (text) => this.#backends[0].tokenize!(text);
|
|
72
|
+
if (backends.some((backend) => typeof backend.attributions === "function")) {
|
|
73
|
+
this.attributions = (context) => Meta.composeAttributions(
|
|
74
|
+
...backends.map((backend) => backend.attributions?.(context) ?? []),
|
|
75
|
+
);
|
|
76
|
+
}
|
|
62
77
|
}
|
|
63
78
|
|
|
64
79
|
// --- Provider surface: interchangeable backends collapse to one honest face ---
|
|
@@ -82,8 +97,39 @@ export default class Pool implements Provider {
|
|
|
82
97
|
return this.#backends.some((b) => b.requiresMaxTokens === true) ? true : undefined;
|
|
83
98
|
}
|
|
84
99
|
|
|
85
|
-
|
|
100
|
+
async countPromptTokens(
|
|
101
|
+
messages: readonly ChatMessage[],
|
|
102
|
+
signal?: AbortSignal,
|
|
103
|
+
): Promise<PromptTokenMeasurement> {
|
|
104
|
+
const measurements = await Promise.all(this.#backends.map(async (backend) =>
|
|
105
|
+
assertPromptTokenMeasurement(
|
|
106
|
+
await backend.countPromptTokens(messages, signal),
|
|
107
|
+
`Pool backend ${backend.model}`,
|
|
108
|
+
)));
|
|
109
|
+
const tokens = Math.max(...measurements.map((measurement) => measurement.tokens));
|
|
110
|
+
const sources = [...new Set(measurements.map(({ source }) => source))].join(",");
|
|
111
|
+
const estimate = measurements.find((measurement) => measurement.kind === "estimate");
|
|
112
|
+
if (estimate?.kind === "estimate") {
|
|
113
|
+
return {
|
|
114
|
+
kind: "estimate",
|
|
115
|
+
tokens,
|
|
116
|
+
source: `pool:${sources}`,
|
|
117
|
+
detail: `at least one interchangeable backend has only an estimate: ${estimate.detail}`,
|
|
118
|
+
};
|
|
119
|
+
}
|
|
120
|
+
const exact = measurements.every((measurement) => measurement.kind === "exact")
|
|
121
|
+
&& measurements.every((measurement) => measurement.tokens === tokens);
|
|
122
|
+
return {
|
|
123
|
+
kind: exact ? "exact" : "upper_bound",
|
|
124
|
+
tokens,
|
|
125
|
+
source: `pool:${sources}`,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
86
128
|
calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
|
|
129
|
+
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
|
|
130
|
+
const backend = this.#backends[0];
|
|
131
|
+
return resolveProviderCost(undefined, backend.calculateCharge?.(usage), () => backend.calculateCost(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
132
|
+
}
|
|
87
133
|
|
|
88
134
|
// --- dispatch ---
|
|
89
135
|
|