@plurnk/plurnk-providers 1.3.12 → 1.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/.env.defaults +39 -22
  2. package/README.md +65 -4
  3. package/SPEC.md +222 -56
  4. package/dist/AiSdkProvider.d.ts +18 -5
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +240 -153
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +14 -15
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +26 -10
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.d.ts +8 -2
  13. package/dist/Pool.d.ts.map +1 -1
  14. package/dist/Pool.js +41 -8
  15. package/dist/Pool.js.map +1 -1
  16. package/dist/ProviderRegistry.d.ts +4 -1
  17. package/dist/ProviderRegistry.d.ts.map +1 -1
  18. package/dist/ProviderRegistry.js +7 -3
  19. package/dist/ProviderRegistry.js.map +1 -1
  20. package/dist/aiSdkTransport.d.ts +4 -2
  21. package/dist/aiSdkTransport.d.ts.map +1 -1
  22. package/dist/aiSdkTransport.js +18 -3
  23. package/dist/aiSdkTransport.js.map +1 -1
  24. package/dist/catalogProvider.d.ts +3 -1
  25. package/dist/catalogProvider.d.ts.map +1 -1
  26. package/dist/catalogProvider.js +20 -7
  27. package/dist/catalogProvider.js.map +1 -1
  28. package/dist/compatibleProvider.d.ts.map +1 -1
  29. package/dist/compatibleProvider.js +11 -4
  30. package/dist/compatibleProvider.js.map +1 -1
  31. package/dist/cost.d.ts +11 -0
  32. package/dist/cost.d.ts.map +1 -0
  33. package/dist/cost.js +64 -0
  34. package/dist/cost.js.map +1 -0
  35. package/dist/discover.d.ts +2 -0
  36. package/dist/discover.d.ts.map +1 -1
  37. package/dist/discover.js +15 -9
  38. package/dist/discover.js.map +1 -1
  39. package/dist/env.d.ts +4 -0
  40. package/dist/env.d.ts.map +1 -1
  41. package/dist/env.js +29 -9
  42. package/dist/env.js.map +1 -1
  43. package/dist/errors.d.ts +27 -0
  44. package/dist/errors.d.ts.map +1 -0
  45. package/dist/errors.js +150 -0
  46. package/dist/errors.js.map +1 -0
  47. package/dist/index.d.ts +11 -5
  48. package/dist/index.d.ts.map +1 -1
  49. package/dist/index.js +7 -4
  50. package/dist/index.js.map +1 -1
  51. package/dist/notices.d.ts +10 -0
  52. package/dist/notices.d.ts.map +1 -0
  53. package/dist/notices.js +11 -0
  54. package/dist/notices.js.map +1 -0
  55. package/dist/ollama.d.ts.map +1 -1
  56. package/dist/ollama.js +3 -3
  57. package/dist/ollama.js.map +1 -1
  58. package/dist/openai.d.ts +1 -1
  59. package/dist/openai.d.ts.map +1 -1
  60. package/dist/promptTokens.d.ts +4 -0
  61. package/dist/promptTokens.d.ts.map +1 -0
  62. package/dist/promptTokens.js +32 -0
  63. package/dist/promptTokens.js.map +1 -0
  64. package/dist/sdkModels.d.ts.map +1 -1
  65. package/dist/sdkModels.js +4 -3
  66. package/dist/sdkModels.js.map +1 -1
  67. package/dist/types.d.ts +43 -16
  68. package/dist/types.d.ts.map +1 -1
  69. package/dist/types.js +1 -1
  70. package/dist/types.js.map +1 -1
  71. package/dist/usage.d.ts +3 -0
  72. package/dist/usage.d.ts.map +1 -1
  73. package/dist/usage.js +26 -14
  74. package/dist/usage.js.map +1 -1
  75. package/dist/warnings.js +0 -0
  76. package/dist/warnings.js.map +1 -1
  77. package/package.json +13 -9
  78. package/src/AiSdkProvider.test.ts +480 -159
  79. package/src/AiSdkProvider.ts +320 -196
  80. package/src/Mock.test.ts +29 -14
  81. package/src/Mock.ts +33 -15
  82. package/src/Pool.test.ts +43 -6
  83. package/src/Pool.ts +56 -10
  84. package/src/ProviderRegistry.test.ts +158 -9
  85. package/src/ProviderRegistry.ts +19 -6
  86. package/src/aiSdkTransport.ts +25 -6
  87. package/src/boundaries.test.ts +8 -3
  88. package/src/catalogProvider.test.ts +17 -0
  89. package/src/catalogProvider.ts +25 -10
  90. package/src/compatibleProvider.test.ts +96 -0
  91. package/src/compatibleProvider.ts +15 -6
  92. package/src/cost.test.ts +63 -0
  93. package/src/cost.ts +83 -0
  94. package/src/defaults.test.ts +1 -0
  95. package/src/discover.test.ts +48 -7
  96. package/src/discover.ts +31 -21
  97. package/src/env.test.ts +38 -23
  98. package/src/env.ts +45 -18
  99. package/src/errors.test.ts +148 -0
  100. package/src/errors.ts +207 -0
  101. package/src/index.ts +29 -7
  102. package/src/lexicon-guard.test.ts +6 -6
  103. package/src/notices.ts +22 -0
  104. package/src/ollama.test.ts +64 -0
  105. package/src/ollama.ts +6 -3
  106. package/src/openai.ts +3 -0
  107. package/src/promptTokens.ts +41 -0
  108. package/src/sdkModels.test.ts +7 -0
  109. package/src/sdkModels.ts +4 -8
  110. package/src/types.ts +106 -64
  111. package/src/usage.test.ts +15 -4
  112. package/src/usage.ts +32 -14
  113. package/src/warnings.test.ts +10 -10
  114. package/src/warnings.ts +0 -0
  115. package/dist/OpenAICompat.d.ts +0 -76
  116. package/dist/OpenAICompat.d.ts.map +0 -1
  117. package/dist/OpenAICompat.js +0 -555
  118. package/dist/OpenAICompat.js.map +0 -1
  119. package/dist/openaiStream.d.ts +0 -47
  120. package/dist/openaiStream.d.ts.map +0 -1
  121. package/dist/openaiStream.js +0 -280
  122. package/dist/openaiStream.js.map +0 -1
  123. package/dist/standardProviders.d.ts +0 -31
  124. package/dist/standardProviders.d.ts.map +0 -1
  125. package/dist/standardProviders.js +0 -518
  126. package/dist/standardProviders.js.map +0 -1
  127. package/dist/telemetry.d.ts +0 -24
  128. package/dist/telemetry.d.ts.map +0 -1
  129. package/dist/telemetry.js +0 -85
  130. package/dist/telemetry.js.map +0 -1
  131. package/src/telemetry.test.ts +0 -69
  132. package/src/telemetry.ts +0 -116
package/src/Mock.test.ts CHANGED
@@ -7,7 +7,7 @@ import type { MockResponse } from "./Mock.ts";
7
7
  const build = (responses: MockResponse[] = [{ assistant: { content: "hi", reasoning: null } }]) =>
8
8
  new Mock({ contextWindow: 100000, responses });
9
9
 
10
- // — Identity (SPEC §10.2, §10.6) —
10
+ // — Identity ({§provider-interface}) —
11
11
 
12
12
  test("Mock: contextWindow and model are stable across reads", () => {
13
13
  const m = build();
@@ -22,21 +22,24 @@ test("Mock: contextWindow passes null through", () => {
22
22
  assert.equal(m.contextWindow, null);
23
23
  });
24
24
 
25
- // — Tokenomics (SPEC §10.3, §10.4, §10.5) —
25
+ // — Tokenomics ({§provider-interface}) —
26
26
 
27
- test("Mock: countTokens('') is 0; non-empty is a positive integer", () => {
27
+ test("Mock: prompt counting is exact for its declared mock vocabulary", async () => {
28
28
  const m = build();
29
- assert.equal(m.countTokens(""), 0);
30
- const n = m.countTokens("four");
31
- assert.ok(Number.isInteger(n) && n > 0);
29
+ assert.deepEqual(await m.countPromptTokens([]), {
30
+ kind: "exact", tokens: 0, source: "mock:chars2",
31
+ });
32
+ assert.deepEqual(await m.countPromptTokens([{ role: "user", content: "four" }]), {
33
+ kind: "exact", tokens: 2, source: "mock:chars2",
34
+ });
32
35
  });
33
36
 
34
- test("Mock: calculateCost zero usage is 0 (free)", () => {
37
+ test("Mock: calculateCost returns its deliberate zero estimate", () => {
35
38
  const m = build();
36
- assert.equal(m.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 0);
39
+ assert.equal(m.calculateCost({ prompt: 100, completion: 20, reasoning: 10, cached: 5, total: 130 }), 0);
37
40
  });
38
41
 
39
- // — Transport (SPEC §10.7, §10.10) —
42
+ // — Transport ({§provider-interface}) —
40
43
 
41
44
  test("Mock: generate resolves a valid ProviderResponse shape", async () => {
42
45
  const m = build([{ assistant: { content: "hello", reasoning: "cot" } }]);
@@ -67,6 +70,18 @@ test("Mock: generate applies caller-supplied overrides", async () => {
67
70
  assert.deepEqual(assistantRaw, { wire: true });
68
71
  });
69
72
 
73
+ test("Mock: a supplied grammar produces unsplit evidence unless the fixture supplies exact evidence", async () => {
74
+ const explicit = { input: "prefixx", contentStart: 6, transported: false } as const;
75
+ const m = build([
76
+ { assistant: { content: "x", reasoning: null } },
77
+ { assistant: { content: "x", reasoning: null }, grammarEvidence: explicit },
78
+ ]);
79
+ const inferred = await m.generate({ messages: [], grammar: 'root ::= "x"' });
80
+ const supplied = await m.generate({ messages: [], grammar: 'root ::= "prefixx"' });
81
+ assert.deepEqual(inferred.grammarEvidence, { input: "x", contentStart: 0, transported: true });
82
+ assert.deepEqual(supplied.grammarEvidence, explicit);
83
+ });
84
+
70
85
  test("Mock: ops escape hatch passes through when provided", async () => {
71
86
  const ops = [{ kind: "send" }] as never;
72
87
  const m = build([{ assistant: { content: "x", reasoning: null, ops } }]);
@@ -80,7 +95,7 @@ test("Mock: ops absent when not provided", async () => {
80
95
  assert.equal("ops" in assistant, false);
81
96
  });
82
97
 
83
- // — Abort (SPEC §10.8) —
98
+ // — Abort ({§provider-failure-normalization}) —
84
99
 
85
100
  test("Mock: generate with a pre-aborted signal rejects and consumes no response", async () => {
86
101
  const m = build([{ assistant: { content: "untouched", reasoning: null } }]);
@@ -106,9 +121,9 @@ test("Mock: exhausted queue throws a specific error", async () => {
106
121
  await assert.rejects(() => m.generate({ messages: [] }), /exhausted/);
107
122
  });
108
123
 
109
- // -- #507: the reserve surface lives on the Provider CONTRACT, and Mock drives core's partition suite --
124
+ // -- {§provider-generation-envelope} --
110
125
 
111
- test("#507 the reserve getters are on the Provider interface (not just the concrete class)", () => {
126
+ test("the reserve getters are on the Provider interface (not just the concrete class)", () => {
112
127
  // Typing against the contract catches a getter-only concrete surface.
113
128
  const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
114
129
  try {
@@ -120,7 +135,7 @@ test("#507 the reserve getters are on the Provider interface (not just the concr
120
135
  }
121
136
  });
122
137
 
123
- test("#507 Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its window (the service partition path)", () => {
138
+ test("Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its window (the service partition path)", () => {
124
139
  const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
125
140
  const prevC = process.env.PLURNK_PROVIDERS_COMPLETION_RESERVE;
126
141
  try {
@@ -135,7 +150,7 @@ test("#507 Mock resolves reserves from PLURNK_PROVIDERS_*_RESERVE against its wi
135
150
  }
136
151
  });
137
152
 
138
- test("#507 no reserve env → null (the #421 no-cap path; the ~100 bare Mocks unaffected)", () => {
153
+ test("no reserve env → null (the no-cap path; bare Mocks unaffected)", () => {
139
154
  const m = new Mock({ contextWindow: 49152, responses: [] });
140
155
  assert.equal(m.reasoningReserve, null);
141
156
  assert.equal(m.completionReserve, null);
package/src/Mock.ts CHANGED
@@ -5,7 +5,8 @@
5
5
  // Provider contract. Production providers don't expose the `ops` escape
6
6
  // hatch — that's an intg-only convenience.
7
7
 
8
- import type { ChatMessage, FinishReason, Provider, ProviderAssistant, ProviderUsage } from "./types.ts";
8
+ import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
9
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
9
10
  import { resolveEnvelopeFromEnv } from "./env.ts";
10
11
 
11
12
  export type MockAssistant = {
@@ -15,11 +16,10 @@ export type MockAssistant = {
15
16
  usage?: Partial<ProviderUsage>;
16
17
  finishReason?: FinishReason;
17
18
  model?: string;
18
- // #482: sealed relay reasoning items (id-keyed, widened per client), so core's
19
- // sealed-reasoning conformance test can drive the shape through Mock.
20
- reasoningEncrypted?: ReadonlyArray<{ id: string | null; subtype: string; encrypted: ReadonlyArray<{ data: string; format: string | null }> }>;
19
+ // Provider-normalized encrypted reasoning fixture.
20
+ reasoningEncrypted?: ReadonlyArray<ProviderEncryptedReasoningItem>;
21
21
  // Pre-parsed ops — intg-only escape hatch. Typed `unknown[]` so the
22
- // framework carries NO @plurnk/plurnk-grammar dependency; plurnk-service
22
+ // framework carries no parser dependency; plurnk-service
23
23
  // casts these to PlurnkStatement[] on its side. Production providers never
24
24
  // include this field.
25
25
  ops?: unknown[];
@@ -28,10 +28,12 @@ export type MockAssistant = {
28
28
  export type MockResponse = {
29
29
  assistant: MockAssistant;
30
30
  assistantRaw?: unknown;
31
+ grammarEvidence?: GrammarEvidence;
31
32
  };
32
33
 
33
34
  // Returned shape: ProviderAssistant + pre-parsed ops visible for tests.
34
35
  export type MockReturnedAssistant = ProviderAssistant & { ops?: unknown[] };
36
+ export type MockReturnedResponse = ProviderResponse & { assistant: MockReturnedAssistant };
35
37
 
36
38
  const DEFAULT_USAGE: ProviderUsage = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
37
39
 
@@ -41,8 +43,9 @@ export default class Mock implements Provider {
41
43
  #completionReserve: number | null;
42
44
  #queue: MockResponse[];
43
45
 
44
- // #507: reserves resolve the SAME way a real provider's do — the TOLERANT env
45
- // read (the service's partition/budget suite sets PLURNK_PROVIDERS_*_RESERVE
46
+ // {§provider-generation-envelope} Reserves resolve the same way a real
47
+ // provider's do — the tolerant env read (the service's partition/budget
48
+ // suite sets PLURNK_PROVIDERS_*_RESERVE
46
49
  // and Mock reflects it, resolved against contextWindow). Tolerant, NOT the
47
50
  // fail-hard envelopeFromEnv: Mock is the universal fixture, constructed
48
51
  // without the reserves in most base tests, so absent env → null → the
@@ -61,18 +64,25 @@ export default class Mock implements Provider {
61
64
  get completionReserve(): number | null { return this.#completionReserve; }
62
65
  get model(): string { return "mock"; }
63
66
 
64
- // Heuristic tokenizer (chars/2 upper bound, matching the framework's
65
- // fallback). Mock is test-only; real provider siblings ship exact counts.
66
- countTokens(text: string): number {
67
- return text.length === 0 ? 0 : Math.ceil(text.length / 2);
67
+ // Mock's deliberately simple vocabulary defines each two content code units
68
+ // as one token and has no hidden request framing. This is exact for the mock,
69
+ // unlike a production adapter applying the same arithmetic to an unknown model.
70
+ async countPromptTokens(messages: readonly ChatMessage[]): Promise<PromptTokenMeasurement> {
71
+ return {
72
+ kind: "exact",
73
+ tokens: messages.reduce((sum, { content }) => sum + Math.ceil(content.length / 2), 0),
74
+ source: "mock:chars2",
75
+ };
68
76
  }
69
77
 
70
78
  // Mock is free.
71
79
  calculateCost(_usage: ProviderUsage): number { return 0; }
80
+ calculateCharge(_usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> { return { kind: "free", source: "mock provider" }; }
72
81
 
73
- async generate({ signal }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal }): Promise<{ assistant: MockReturnedAssistant; assistantRaw: unknown }> {
82
+ async generate({ signal, grammar }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal; grammar?: string }): Promise<MockReturnedResponse> {
74
83
  // Honor abort before consuming the queue — an aborted call makes no
75
- // "wire call" and must not exhaust a queued response (SPEC §10.8).
84
+ // "wire call" and must not exhaust a queued response
85
+ // ({§provider-failure-normalization}).
76
86
  signal?.throwIfAborted();
77
87
  const next = this.#queue.shift();
78
88
  if (next === undefined) throw new Error("Mock provider exhausted: no more queued responses");
@@ -81,12 +91,20 @@ export default class Mock implements Provider {
81
91
  content: a.content,
82
92
  reasoning: a.reasoning,
83
93
  usage: { ...DEFAULT_USAGE, ...a.usage },
84
- ...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}), // #482 — sealed blobs ride the contract through Mock too
94
+ ...(a.reasoningEncrypted !== undefined ? { reasoningEncrypted: a.reasoningEncrypted } : {}),
85
95
  finishReason: a.finishReason ?? "stop",
86
96
  model: a.model ?? "mock",
87
97
  ...(a.ops !== undefined ? { ops: a.ops } : {}),
88
98
  };
89
- return { assistant, assistantRaw: next.assistantRaw ?? null };
99
+ const grammarEvidence = next.grammarEvidence
100
+ ?? (grammar === undefined
101
+ ? undefined
102
+ : { input: assistant.content, contentStart: 0, transported: true });
103
+ return {
104
+ assistant,
105
+ assistantRaw: next.assistantRaw ?? null,
106
+ ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
107
+ };
90
108
  }
91
109
 
92
110
  get remaining(): number { return this.#queue.length; }
package/src/Pool.test.ts CHANGED
@@ -1,8 +1,8 @@
1
1
  import test from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import Pool from "./Pool.ts";
4
- import { ProviderError } from "./telemetry.ts";
5
- import type { Provider, ProviderResponse } from "./types.ts";
4
+ import { ProviderError } from "./errors.ts";
5
+ import type { PromptTokenMeasurement, Provider, ProviderResponse } from "./types.ts";
6
6
  import { resetEmittedWarnings } from "./warnings.ts";
7
7
 
8
8
  test.afterEach(() => { resetEmittedWarnings(); });
@@ -14,6 +14,7 @@ type FakeOpts = {
14
14
  constrainsOutput?: boolean; requiresMaxTokens?: boolean;
15
15
  reasoningReserve?: number | null; completionReserve?: number | null;
16
16
  tokenize?: boolean; cost?: number; throws?: Error;
17
+ promptMeasurement?: PromptTokenMeasurement;
17
18
  };
18
19
  // A fake backend that records which workers it served, and optionally throws.
19
20
  const backend = (opts: FakeOpts = {}) => {
@@ -27,7 +28,11 @@ const backend = (opts: FakeOpts = {}) => {
27
28
  ...(opts.reasoningReserve !== undefined ? { reasoningReserve: opts.reasoningReserve } : {}),
28
29
  ...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
29
30
  ...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
30
- countTokens: (t: string) => t.length,
31
+ countPromptTokens: async (messages) => opts.promptMeasurement ?? ({
32
+ kind: "exact",
33
+ tokens: messages.reduce((sum, { content }) => sum + content.length, 0),
34
+ source: "test:exact",
35
+ }),
31
36
  calculateCost: () => opts.cost ?? 0,
32
37
  generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
33
38
  served.push(args.workerId);
@@ -39,6 +44,7 @@ const backend = (opts: FakeOpts = {}) => {
39
44
  };
40
45
  const netErr = () => new ProviderError("provider:x", "network_failure", "down", { status: 503 });
41
46
  const authErr = () => new ProviderError("provider:x", "unauthorized", "no key", { status: 401 });
47
+ const interruptedErr = () => new ProviderError("provider:x", "resource_interrupted", "generation interrupted");
42
48
  const gen = (p: Pool, workerId: string, extra: Record<string, unknown> = {}) =>
43
49
  p.generate({ messages: [], workerId, ...extra } as Parameters<Provider["generate"]>[0]);
44
50
 
@@ -58,7 +64,7 @@ test("Pool: contextWindow is the safe floor (min) across backends", () => {
58
64
  assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: 32000 }).b]).contextWindow, 32000);
59
65
  });
60
66
 
61
- test("Pool: any unknown (null) window makes the pool null - no improvised cap (#421)", () => {
67
+ test("Pool: any unknown (null) window makes the pool null - no improvised cap", () => {
62
68
  assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: null }).b]).contextWindow, null);
63
69
  });
64
70
 
@@ -88,12 +94,32 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
88
94
  assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
89
95
  });
90
96
 
91
- test("Pool: countTokens + calculateCost delegate to a backend", () => {
97
+ test("Pool: prompt counting + calculateCost delegate to a backend", async () => {
92
98
  const p = new Pool([backend({ cost: 42 }).b]);
93
- assert.equal(p.countTokens("abcd"), 4);
99
+ assert.deepEqual(await p.countPromptTokens([{ role: "user", content: "abcd" }]), {
100
+ kind: "exact", tokens: 4, source: "pool:test:exact",
101
+ });
94
102
  assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
95
103
  });
96
104
 
105
+ test("Pool: prompt evidence is conservative across every routable backend", async () => {
106
+ const exact = backend({ promptMeasurement: { kind: "exact", tokens: 8, source: "test:a" } }).b;
107
+ const larger = backend({ promptMeasurement: { kind: "exact", tokens: 11, source: "test:b" } }).b;
108
+ assert.deepEqual(await new Pool([exact, larger]).countPromptTokens([]), {
109
+ kind: "upper_bound", tokens: 11, source: "pool:test:a,test:b",
110
+ });
111
+
112
+ const estimate = backend({ promptMeasurement: {
113
+ kind: "estimate", tokens: 3, source: "heuristic:chars2", detail: "unknown framing",
114
+ } }).b;
115
+ assert.deepEqual(await new Pool([exact, estimate]).countPromptTokens([]), {
116
+ kind: "estimate",
117
+ tokens: 8,
118
+ source: "pool:test:a,heuristic:chars2",
119
+ detail: "at least one interchangeable backend has only an estimate: unknown framing",
120
+ });
121
+ });
122
+
97
123
  // --- dispatch: round-robin + affinity ---
98
124
 
99
125
  test("Pool: NEW workers round-robin across backends", async () => {
@@ -134,6 +160,17 @@ test("Pool: a terminal (auth) failure does NOT overflow", async () => {
134
160
  assert.deepEqual(other.served, []); // never tried - auth fails the same on a peer
135
161
  });
136
162
 
163
+ test("#161: Pool does not overflow a resource-interrupted attempt and discard its evidence", async () => {
164
+ const interrupted = backend({ throws: interruptedErr() }), other = backend();
165
+ const pool = new Pool([interrupted.b, other.b]);
166
+ await assert.rejects(
167
+ () => gen(pool, "w1"),
168
+ (error: unknown) => error instanceof ProviderError && error.kind === "resource_interrupted",
169
+ );
170
+ assert.deepEqual(interrupted.served, ["w1"]);
171
+ assert.deepEqual(other.served, []);
172
+ });
173
+
137
174
  test("Pool: whole fleet unavailable throws the last error, each backend tried once", async () => {
138
175
  const a = backend({ throws: netErr() }), b = backend({ throws: netErr() });
139
176
  const p = new Pool([a.b, b.b]);
package/src/Pool.ts CHANGED
@@ -1,13 +1,20 @@
1
- import type { Provider, ProviderResponse, ProviderUsage, ChatMessage } from "./types.ts";
2
- import { ProviderError, type ProviderTelemetryKind } from "./telemetry.ts";
1
+ import type { Provider, ProviderResponse, ProviderUsage, ChatMessage, PromptTokenMeasurement } from "./types.ts";
2
+ import type { ProviderCost } from "@plurnk/plurnk-contracts";
3
+ import { resolveProviderCost } from "./cost.ts";
4
+ import { ProviderError, type ProviderErrorKind } from "./errors.ts";
3
5
  import { emitWarningOnce } from "./warnings.ts";
6
+ import { assertPromptTokenMeasurement } from "./promptTokens.ts";
7
+ import Meta, {
8
+ type PluginAttribution,
9
+ type PluginAttributionContext,
10
+ } from "@plurnk/plurnk-meta";
4
11
 
5
12
  // A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
6
- // transient retries (#18) before throwing one of these, so re-hitting the same
13
+ // transient retries before throwing one of these, so re-hitting the same
7
14
  // backend is pointless - a sibling may still serve, so the pool overflows to it.
8
15
  // Auth/quota/content kinds are deliberately absent: a peer backend fails them
9
16
  // identically, and failing over would only multiply the damage (and the spend).
10
- const OVERFLOW_KINDS: ReadonlySet<ProviderTelemetryKind> = new Set(["network_failure", "rate_limit"]);
17
+ const OVERFLOW_KINDS: ReadonlySet<ProviderErrorKind> = new Set(["network_failure", "rate_limit"]);
11
18
 
12
19
  // Pool - fronts N INTERCHANGEABLE backends as one Provider. This is CAPACITY, not
13
20
  // blend: the mechanism (round-robin across workers, sticky within a worker for
@@ -15,11 +22,12 @@ const OVERFLOW_KINDS: ReadonlySet<ProviderTelemetryKind> = new Set(["network_fai
15
22
  // blend/escalation DECISION (which SKU, when to switch models) stays the
16
23
  // consumer's, one level up, by choosing WHICH pool to call. Backends MUST be
17
24
  // interchangeable - same served model, compatible window - so the pool presents ONE
18
- // honest Provider surface instead of pretending N models are one (#533).
25
+ // honest Provider surface instead of pretending N models are one
26
+ // ({§provider-capacity-pool}).
19
27
  //
20
28
  // Affinity is the load-bearing part. A worker's turns stick to one backend so its
21
29
  // stable prompt prefix keeps hitting the same KV cache; scattering a worker across
22
- // backends shreds the prefix cache (#531). It is the #11 slot-affinity pattern one
30
+ // backends shreds the prefix cache. It is the same slot-affinity pattern one
23
31
  // level up: worker -> slot within a llama-server becomes worker -> backend across a
24
32
  // fleet - round-robin ACROSS workers, sticky WITHIN one.
25
33
  export default class Pool implements Provider {
@@ -29,9 +37,10 @@ export default class Pool implements Provider {
29
37
  #next = 0; // round-robin cursor for NEW workers
30
38
  readonly #floor: Provider; // the min-window backend: the safe budget floor
31
39
 
32
- // OPTIONAL exact tokenizer (#37): present iff EVERY backend exposes one (same
40
+ // Optional exact tokenizer: present iff every backend exposes one (same
33
41
  // vocab, since interchangeable). Delegated; absent when the fleet can't.
34
42
  readonly tokenize?: (text: string) => Promise<number[]>;
43
+ readonly attributions?: (context: PluginAttributionContext) => PluginAttribution;
35
44
 
36
45
  constructor(backends: readonly Provider[]) {
37
46
  if (backends.length === 0) throw new Error("Pool: at least one backend is required");
@@ -40,14 +49,15 @@ export default class Pool implements Provider {
40
49
  // SELECTION job, not this primitive - fail at construction, never mid-turn.
41
50
  const model = backends[0].model;
42
51
  const mixed = backends.find((b) => b.model !== model);
43
- if (mixed !== undefined) throw new Error(`Pool: backends must be interchangeable, got mixed models "${model}" and "${mixed.model}" - heterogeneous blend is the consumer's selection, not a pool (#533)`);
52
+ if (mixed !== undefined) throw new Error(`Pool: backends must be interchangeable, got mixed models "${model}" and "${mixed.model}" - heterogeneous blend is the consumer's selection, not a pool`);
44
53
  this.#backends = backends;
45
54
  this.#cap = backends.length * 8;
46
55
 
47
56
  // The exposed window is the SAFE FLOOR across backends (a packet that fits the
48
57
  // smallest fits all); its reserves travel with it so the budget stays
49
58
  // self-consistent. ANY unknown (null) window makes the pool null - a worker
50
- // might route to it, and the consumer must not improvise a cap (#421).
59
+ // might route to it, and the consumer must not improvise a cap
60
+ // ({§model-fact-resolution}).
51
61
  const anyUnknown = backends.find((b) => b.contextWindow === null);
52
62
  this.#floor = anyUnknown ?? backends.reduce((lo, b) => (b.contextWindow! < lo.contextWindow! ? b : lo));
53
63
  if (new Set(backends.map((b) => b.contextWindow)).size > 1) {
@@ -59,6 +69,11 @@ export default class Pool implements Provider {
59
69
 
60
70
  // tokenize is a per-instance optional method; expose it iff the whole fleet has it.
61
71
  if (backends.every((b) => typeof b.tokenize === "function")) this.tokenize = (text) => this.#backends[0].tokenize!(text);
72
+ if (backends.some((backend) => typeof backend.attributions === "function")) {
73
+ this.attributions = (context) => Meta.composeAttributions(
74
+ ...backends.map((backend) => backend.attributions?.(context) ?? []),
75
+ );
76
+ }
62
77
  }
63
78
 
64
79
  // --- Provider surface: interchangeable backends collapse to one honest face ---
@@ -82,8 +97,39 @@ export default class Pool implements Provider {
82
97
  return this.#backends.some((b) => b.requiresMaxTokens === true) ? true : undefined;
83
98
  }
84
99
 
85
- countTokens(text: string): number { return this.#backends[0].countTokens(text); }
100
+ async countPromptTokens(
101
+ messages: readonly ChatMessage[],
102
+ signal?: AbortSignal,
103
+ ): Promise<PromptTokenMeasurement> {
104
+ const measurements = await Promise.all(this.#backends.map(async (backend) =>
105
+ assertPromptTokenMeasurement(
106
+ await backend.countPromptTokens(messages, signal),
107
+ `Pool backend ${backend.model}`,
108
+ )));
109
+ const tokens = Math.max(...measurements.map((measurement) => measurement.tokens));
110
+ const sources = [...new Set(measurements.map(({ source }) => source))].join(",");
111
+ const estimate = measurements.find((measurement) => measurement.kind === "estimate");
112
+ if (estimate?.kind === "estimate") {
113
+ return {
114
+ kind: "estimate",
115
+ tokens,
116
+ source: `pool:${sources}`,
117
+ detail: `at least one interchangeable backend has only an estimate: ${estimate.detail}`,
118
+ };
119
+ }
120
+ const exact = measurements.every((measurement) => measurement.kind === "exact")
121
+ && measurements.every((measurement) => measurement.tokens === tokens);
122
+ return {
123
+ kind: exact ? "exact" : "upper_bound",
124
+ tokens,
125
+ source: `pool:${sources}`,
126
+ };
127
+ }
86
128
  calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
129
+ calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
130
+ const backend = this.#backends[0];
131
+ return resolveProviderCost(undefined, backend.calculateCharge?.(usage), () => backend.calculateCost(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
132
+ }
87
133
 
88
134
  // --- dispatch ---
89
135