@plurnk/plurnk-providers 1.1.1 → 1.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/.env.defaults +10 -0
  2. package/README.md +50 -60
  3. package/SPEC.md +7 -6
  4. package/dist/OpenAICompat.d.ts +4 -0
  5. package/dist/OpenAICompat.d.ts.map +1 -1
  6. package/dist/OpenAICompat.js +22 -3
  7. package/dist/OpenAICompat.js.map +1 -1
  8. package/dist/ProviderRegistry.js +2 -2
  9. package/dist/ProviderRegistry.js.map +1 -1
  10. package/dist/env.d.ts +1 -0
  11. package/dist/env.d.ts.map +1 -1
  12. package/dist/env.js +15 -3
  13. package/dist/env.js.map +1 -1
  14. package/dist/index.d.ts +1 -1
  15. package/dist/index.d.ts.map +1 -1
  16. package/dist/index.js +1 -1
  17. package/dist/index.js.map +1 -1
  18. package/dist/openaiStream.d.ts.map +1 -1
  19. package/dist/openaiStream.js +12 -0
  20. package/dist/openaiStream.js.map +1 -1
  21. package/dist/standardProviders.d.ts +1 -0
  22. package/dist/standardProviders.d.ts.map +1 -1
  23. package/dist/standardProviders.js +47 -7
  24. package/dist/standardProviders.js.map +1 -1
  25. package/package.json +8 -6
  26. package/src/Mock.test.ts +142 -0
  27. package/src/Mock.ts +95 -0
  28. package/src/OpenAICompat.test.ts +1107 -0
  29. package/src/OpenAICompat.ts +756 -0
  30. package/src/Pool.test.ts +155 -0
  31. package/src/Pool.ts +134 -0
  32. package/src/ProviderRegistry.test.ts +176 -0
  33. package/src/ProviderRegistry.ts +93 -0
  34. package/src/boundaries.test.ts +24 -0
  35. package/src/discover.test.ts +123 -0
  36. package/src/discover.ts +112 -0
  37. package/src/env.test.ts +190 -0
  38. package/src/env.ts +211 -0
  39. package/src/index.ts +51 -0
  40. package/src/lexicon-guard.test.ts +58 -0
  41. package/src/openaiStream.ts +279 -0
  42. package/src/standardProviders.test.ts +925 -0
  43. package/src/standardProviders.ts +618 -0
  44. package/src/telemetry.test.ts +62 -0
  45. package/src/telemetry.ts +108 -0
  46. package/src/types.ts +219 -0
  47. package/src/usage.test.ts +136 -0
  48. package/src/usage.ts +82 -0
  49. package/src/warnings.test.ts +31 -0
  50. package/src/warnings.ts +0 -0
@@ -0,0 +1,155 @@
1
+ import test from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import Pool from "./Pool.ts";
4
+ import { ProviderError } from "./telemetry.ts";
5
+ import type { Provider, ProviderResponse } from "./types.ts";
6
+ import { resetEmittedWarnings } from "./warnings.ts";
7
+
8
+ test.afterEach(() => { resetEmittedWarnings(); });
9
+
10
+ const RESP = { assistant: { usage: { prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 } }, assistantRaw: null } as unknown as ProviderResponse;
11
+
12
+ type FakeOpts = {
13
+ model?: string; window?: number | null; servedModel?: string;
14
+ constrainsOutput?: boolean; requiresMaxTokens?: boolean;
15
+ reasoningReserve?: number | null; completionReserve?: number | null;
16
+ tokenize?: boolean; cost?: number; throws?: Error;
17
+ };
18
+ // A fake backend that records which workers it served, and optionally throws.
19
+ const backend = (opts: FakeOpts = {}) => {
20
+ const served: string[] = [];
21
+ const b: Provider = {
22
+ model: opts.model ?? "gemma",
23
+ contextWindow: opts.window === undefined ? 48000 : opts.window,
24
+ ...(opts.servedModel !== undefined ? { servedModel: opts.servedModel } : {}),
25
+ ...(opts.constrainsOutput !== undefined ? { constrainsOutput: opts.constrainsOutput } : {}),
26
+ ...(opts.requiresMaxTokens !== undefined ? { requiresMaxTokens: opts.requiresMaxTokens } : {}),
27
+ ...(opts.reasoningReserve !== undefined ? { reasoningReserve: opts.reasoningReserve } : {}),
28
+ ...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
29
+ ...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
30
+ countTokens: (t: string) => t.length,
31
+ costFor: () => opts.cost ?? 0,
32
+ generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
33
+ served.push(args.workerId);
34
+ if (opts.throws !== undefined) throw opts.throws;
35
+ return RESP;
36
+ },
37
+ };
38
+ return { b, served };
39
+ };
40
+ const netErr = () => new ProviderError("provider:x", "network_failure", "down", { status: 503 });
41
+ const authErr = () => new ProviderError("provider:x", "unauthorized", "no key", { status: 401 });
42
+ const gen = (p: Pool, workerId: string, extra: Record<string, unknown> = {}) =>
43
+ p.generate({ messages: [], workerId, ...extra } as Parameters<Provider["generate"]>[0]);
44
+
45
+ // --- construction / homogeneity ---
46
+
47
+ test("Pool: empty backend list is rejected", () => {
48
+ assert.throws(() => new Pool([]), /at least one backend/);
49
+ });
50
+
51
+ test("Pool: mixed models are rejected at construction (heterogeneous blend is not a pool)", () => {
52
+ assert.throws(() => new Pool([backend({ model: "gemma" }).b, backend({ model: "grok" }).b]), /interchangeable/);
53
+ });
54
+
55
+ // --- surface aggregation ---
56
+
57
+ test("Pool: contextWindow is the safe floor (min) across backends", () => {
58
+ assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: 32000 }).b]).contextWindow, 32000);
59
+ });
60
+
61
+ test("Pool: any unknown (null) window makes the pool null - no improvised cap (#421)", () => {
62
+ assert.equal(new Pool([backend({ window: 48000 }).b, backend({ window: null }).b]).contextWindow, null);
63
+ });
64
+
65
+ test("Pool: reserves travel with the floor window", () => {
66
+ const big = backend({ window: 48000, reasoningReserve: 4800, completionReserve: 12000 }).b;
67
+ const small = backend({ window: 32000, reasoningReserve: 3200, completionReserve: 8000 }).b;
68
+ const p = new Pool([big, small]);
69
+ assert.equal(p.contextWindow, 32000);
70
+ assert.equal(p.reasoningReserve, 3200); // the floor backend's, not the larger one's
71
+ assert.equal(p.completionReserve, 8000);
72
+ });
73
+
74
+ test("Pool: capabilities aggregate conservatively (constrainsOutput all-true, requiresMaxTokens any-true)", () => {
75
+ assert.equal(new Pool([backend({ constrainsOutput: true }).b, backend({ constrainsOutput: true }).b]).constrainsOutput, true);
76
+ assert.equal(new Pool([backend({ constrainsOutput: true }).b, backend({ constrainsOutput: false }).b]).constrainsOutput, undefined);
77
+ assert.equal(new Pool([backend({ requiresMaxTokens: false }).b, backend({ requiresMaxTokens: true }).b]).requiresMaxTokens, true);
78
+ assert.equal(new Pool([backend({}).b]).requiresMaxTokens, undefined);
79
+ });
80
+
81
+ test("Pool: servedModel is the common id, undefined when they differ", () => {
82
+ assert.equal(new Pool([backend({ servedModel: "g.gguf" }).b, backend({ servedModel: "g.gguf" }).b]).servedModel, "g.gguf");
83
+ assert.equal(new Pool([backend({ servedModel: "g.gguf" }).b, backend({ servedModel: "h.gguf" }).b]).servedModel, undefined);
84
+ });
85
+
86
+ test("Pool: tokenize is exposed iff every backend has it", () => {
87
+ assert.equal(typeof new Pool([backend({ tokenize: true }).b, backend({ tokenize: true }).b]).tokenize, "function");
88
+ assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
89
+ });
90
+
91
+ test("Pool: countTokens + costFor delegate to a backend", () => {
92
+ const p = new Pool([backend({ cost: 42 }).b]);
93
+ assert.equal(p.countTokens("abcd"), 4);
94
+ assert.equal(p.costFor({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
95
+ });
96
+
97
+ // --- dispatch: round-robin + affinity ---
98
+
99
+ test("Pool: NEW workers round-robin across backends", async () => {
100
+ const b0 = backend(), b1 = backend(), b2 = backend();
101
+ const p = new Pool([b0.b, b1.b, b2.b]);
102
+ await gen(p, "w1"); await gen(p, "w2"); await gen(p, "w3"); await gen(p, "w4");
103
+ assert.deepEqual(b0.served, ["w1", "w4"]); // 0, then wraps at 3 % 3
104
+ assert.deepEqual(b1.served, ["w2"]);
105
+ assert.deepEqual(b2.served, ["w3"]);
106
+ });
107
+
108
+ test("Pool: a worker STICKS to its backend across turns (KV-cache affinity)", async () => {
109
+ const b0 = backend(), b1 = backend();
110
+ const p = new Pool([b0.b, b1.b]);
111
+ await gen(p, "w1"); await gen(p, "w2"); await gen(p, "w1"); await gen(p, "w1");
112
+ assert.deepEqual(b0.served, ["w1", "w1", "w1"]); // w1 always backend 0
113
+ assert.deepEqual(b1.served, ["w2"]);
114
+ });
115
+
116
+ // --- dispatch: overflow ---
117
+
118
+ test("Pool: a backend-availability failure overflows to a healthy sibling and re-sticks", async () => {
119
+ const down = backend({ throws: netErr() }), up = backend();
120
+ const p = new Pool([down.b, up.b]);
121
+ await gen(p, "w1"); // w1 -> backend 0 (down) -> overflow to backend 1 (up)
122
+ assert.deepEqual(down.served, ["w1"]);
123
+ assert.deepEqual(up.served, ["w1"]);
124
+ await gen(p, "w1"); // re-stuck: w1 now routes straight to backend 1
125
+ assert.deepEqual(down.served, ["w1"]); // NOT hit again
126
+ assert.deepEqual(up.served, ["w1", "w1"]);
127
+ });
128
+
129
+ test("Pool: a terminal (auth) failure does NOT overflow", async () => {
130
+ const bad = backend({ throws: authErr() }), other = backend();
131
+ const p = new Pool([bad.b, other.b]);
132
+ await assert.rejects(() => gen(p, "w1"), /no key/);
133
+ assert.deepEqual(bad.served, ["w1"]);
134
+ assert.deepEqual(other.served, []); // never tried - auth fails the same on a peer
135
+ });
136
+
137
+ test("Pool: whole fleet unavailable throws the last error, each backend tried once", async () => {
138
+ const a = backend({ throws: netErr() }), b = backend({ throws: netErr() });
139
+ const p = new Pool([a.b, b.b]);
140
+ await assert.rejects(() => gen(p, "w1"), (e: unknown) => e instanceof ProviderError && e.kind === "network_failure");
141
+ assert.deepEqual(a.served, ["w1"]);
142
+ assert.deepEqual(b.served, ["w1"]);
143
+ });
144
+
145
+ test("Pool: an aborted signal propagates, never overflows", async () => {
146
+ const down = backend({ throws: netErr() }), up = backend();
147
+ const p = new Pool([down.b, up.b]);
148
+ const ac = new AbortController(); ac.abort();
149
+ await assert.rejects(() => gen(p, "w1", { signal: ac.signal }), /down/);
150
+ assert.deepEqual(up.served, []); // cancellation is not a failover
151
+ });
152
+
153
+ test("Pool: generate requires a workerId (affinity keys on it)", async () => {
154
+ await assert.rejects(() => gen(new Pool([backend().b]), ""), /workerId is required/);
155
+ });
package/src/Pool.ts ADDED
@@ -0,0 +1,134 @@
1
+ import type { Provider, ProviderResponse, ProviderUsage, ChatMessage } from "./types.ts";
2
+ import { ProviderError, type ProviderTelemetryKind } from "./telemetry.ts";
3
+ import { emitWarningOnce } from "./warnings.ts";
4
+
5
+ // A backend-AVAILABILITY failure: the sub-provider already exhausted its OWN
6
+ // transient retries (#18) before throwing one of these, so re-hitting the same
7
+ // backend is pointless - a sibling may still serve, so the pool overflows to it.
8
+ // Auth/quota/content kinds are deliberately absent: a peer backend fails them
9
+ // identically, and failing over would only multiply the damage (and the spend).
10
+ const OVERFLOW_KINDS: ReadonlySet<ProviderTelemetryKind> = new Set(["network_failure", "rate_limit"]);
11
+
12
+ // Pool - fronts N INTERCHANGEABLE backends as one Provider. This is CAPACITY, not
13
+ // blend: the mechanism (round-robin across workers, sticky within a worker for
14
+ // KV-cache reuse, overflow to a healthy sibling) is public and lives here; the
15
+ // blend/escalation DECISION (which SKU, when to switch models) stays the
16
+ // consumer's, one level up, by choosing WHICH pool to call. Backends MUST be
17
+ // interchangeable - same served model, compatible window - so the pool presents ONE
18
+ // honest Provider surface instead of pretending N models are one (#533).
19
+ //
20
+ // Affinity is the load-bearing part. A worker's turns stick to one backend so its
21
+ // stable prompt prefix keeps hitting the same KV cache; scattering a worker across
22
+ // backends shreds the prefix cache (#531). It is the #11 slot-affinity pattern one
23
+ // level up: worker -> slot within a llama-server becomes worker -> backend across a
24
+ // fleet - round-robin ACROSS workers, sticky WITHIN one.
25
+ export default class Pool implements Provider {
26
+ readonly #backends: readonly Provider[];
27
+ readonly #affinity = new Map<string, number>(); // workerId -> backend index; sticky, LRU-bounded
28
+ readonly #cap: number; // LRU ceiling on the affinity map
29
+ #next = 0; // round-robin cursor for NEW workers
30
+ readonly #floor: Provider; // the min-window backend: the safe budget floor
31
+
32
+ // OPTIONAL exact tokenizer (#37): present iff EVERY backend exposes one (same
33
+ // vocab, since interchangeable). Delegated; absent when the fleet can't.
34
+ readonly tokenize?: (text: string) => Promise<number[]>;
35
+
36
+ constructor(backends: readonly Provider[]) {
37
+ if (backends.length === 0) throw new Error("Pool: at least one backend is required");
38
+ // Homogeneity is a CONTRACT, not a hope: interchangeable backends share a
39
+ // served identity. A heterogeneous "pool" is the consumer's per-turn
40
+ // SELECTION job, not this primitive - fail at construction, never mid-turn.
41
+ const model = backends[0].model;
42
+ const mixed = backends.find((b) => b.model !== model);
43
+ if (mixed !== undefined) throw new Error(`Pool: backends must be interchangeable, got mixed models "${model}" and "${mixed.model}" - heterogeneous blend is the consumer's selection, not a pool (#533)`);
44
+ this.#backends = backends;
45
+ this.#cap = backends.length * 8;
46
+
47
+ // The exposed window is the SAFE FLOOR across backends (a packet that fits the
48
+ // smallest fits all); its reserves travel with it so the budget stays
49
+ // self-consistent. ANY unknown (null) window makes the pool null - a worker
50
+ // might route to it, and the consumer must not improvise a cap (#421).
51
+ const anyUnknown = backends.find((b) => b.contextWindow === null);
52
+ this.#floor = anyUnknown ?? backends.reduce((lo, b) => (b.contextWindow! < lo.contextWindow! ? b : lo));
53
+ if (new Set(backends.map((b) => b.contextWindow)).size > 1) {
54
+ emitWarningOnce(
55
+ `Pool: backends report different context windows (${backends.map((b) => b.contextWindow).join(", ")}); using the safe floor ${this.#floor.contextWindow}. Interchangeable backends should match.`,
56
+ "PLURNK_POOL_WINDOW_DRIFT",
57
+ );
58
+ }
59
+
60
+ // tokenize is a per-instance optional method; expose it iff the whole fleet has it.
61
+ if (backends.every((b) => typeof b.tokenize === "function")) this.tokenize = (text) => this.#backends[0].tokenize!(text);
62
+ }
63
+
64
+ // --- Provider surface: interchangeable backends collapse to one honest face ---
65
+
66
+ get model(): string { return this.#backends[0].model; }
67
+ get contextWindow(): number | null { return this.#floor.contextWindow; }
68
+ get reasoningReserve(): number | null | undefined { return this.#floor.reasoningReserve; }
69
+ get completionReserve(): number | null | undefined { return this.#floor.completionReserve; }
70
+
71
+ // Served id / capabilities aggregate CONSERVATIVELY: a worker could land on any
72
+ // backend, so claim `constrainsOutput` only if EVERY backend does, and
73
+ // `requiresMaxTokens` if ANY does (bring a cap if even one backend needs it).
74
+ get servedModel(): string | undefined {
75
+ const s = this.#backends[0].servedModel;
76
+ return this.#backends.every((b) => b.servedModel === s) ? s : undefined;
77
+ }
78
+ get constrainsOutput(): boolean | undefined {
79
+ return this.#backends.every((b) => b.constrainsOutput === true) ? true : undefined;
80
+ }
81
+ get requiresMaxTokens(): boolean | undefined {
82
+ return this.#backends.some((b) => b.requiresMaxTokens === true) ? true : undefined;
83
+ }
84
+
85
+ countTokens(text: string): number { return this.#backends[0].countTokens(text); }
86
+ costFor(usage: ProviderUsage): number { return this.#backends[0].costFor(usage); }
87
+
88
+ // --- dispatch ---
89
+
90
+ async generate(args: { messages: ChatMessage[]; workerId: string; primaryWorkerId?: string; signal?: AbortSignal; grammar?: string; maxTokens?: number; attributions?: string[]; client?: string; strikes?: number; workspaceId?: string; loop?: number; turn?: number; sampling?: Record<string, unknown> }): Promise<ProviderResponse> {
91
+ const { workerId, signal } = args;
92
+ if (workerId === undefined || workerId.length === 0) throw new Error("Pool.generate: workerId is required - affinity keys on it");
93
+ const tried = new Set<number>();
94
+ let idx = this.#route(workerId);
95
+ let lastErr: unknown;
96
+ for (;;) {
97
+ tried.add(idx);
98
+ try {
99
+ return await this.#backends[idx].generate(args);
100
+ } catch (err) {
101
+ lastErr = err;
102
+ if (signal?.aborted) throw err; // caller cancellation is never a failover
103
+ // Only a backend-AVAILABILITY failure overflows; auth/quota/content/
104
+ // malformed fail the same on a peer, so they propagate.
105
+ if (!(err instanceof ProviderError) || !OVERFLOW_KINDS.has(err.kind)) throw err;
106
+ const next = this.#nextUntried(tried);
107
+ if (next === null) throw lastErr; // the whole fleet is unavailable
108
+ this.#affinity.set(workerId, next); // re-stick: the worker's cache moves with it
109
+ idx = next;
110
+ }
111
+ }
112
+ }
113
+
114
+ // Route a worker to its affine backend, assigning a fresh one round-robin on
115
+ // first sight. LRU-bounded so a long-lived daemon's map never grows unbounded -
116
+ // an evicted-and-returning worker simply re-pins (one cold prefill, worst case).
117
+ #route(workerId: string): number {
118
+ const pinned = this.#affinity.get(workerId);
119
+ if (pinned !== undefined) {
120
+ this.#affinity.delete(workerId); // refresh LRU recency
121
+ this.#affinity.set(workerId, pinned);
122
+ return pinned;
123
+ }
124
+ const idx = this.#next++ % this.#backends.length;
125
+ if (this.#affinity.size >= this.#cap) this.#affinity.delete(this.#affinity.keys().next().value as string);
126
+ this.#affinity.set(workerId, idx);
127
+ return idx;
128
+ }
129
+
130
+ #nextUntried(tried: Set<number>): number | null {
131
+ for (let i = 0; i < this.#backends.length; i++) if (!tried.has(i)) return i;
132
+ return null;
133
+ }
134
+ }
@@ -0,0 +1,176 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
4
+
5
+ const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, costFor: () => 0, generate: async () => { throw new Error("unused"); } };
6
+ const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
7
+ async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
8
+
9
+ // Alias PARSING is tested in @plurnk/plurnk-aliases (its owner, #27). Here we
10
+ // exercise the resolution + two-tier instantiation this module owns; the active
11
+ // alias is driven end-to-end by loadActiveProvider below.
12
+
13
+ // — two-tier instantiation (SPEC §5) —
14
+
15
+ const fullEnv = Object.freeze({
16
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
+ PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
18
+ OPENAI_BASE_URL: "http://x",
19
+ });
20
+
21
+ test("instantiateProvider: standard name resolves in-framework, no scan, no import", async () => {
22
+ resetDiscoveryCache();
23
+ mock.method(globalThis, "fetch", async (url: string) => {
24
+ if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
25
+ throw new Error("unexpected fetch");
26
+ });
27
+ const imports: string[] = [];
28
+ let scanned = false;
29
+ const p = await instantiateProvider("openai", { ...fullEnv }, "m",
30
+ async (s) => { imports.push(s); return {}; },
31
+ async () => { scanned = true; return { registry: new Map(), skipped: new Map(), attributions: new Map() }; });
32
+ assert.equal(p.model, "m");
33
+ assert.deepEqual(imports, []); // tier 1 never touches the importer…
34
+ assert.equal(scanned, false); // …nor the scan
35
+ mock.restoreAll();
36
+ });
37
+
38
+ test("instantiateProvider: bespoke name resolves via the scan and imports the discovered package", async () => {
39
+ resetDiscoveryCache();
40
+ const calls: unknown[] = [];
41
+ const p = await instantiateProvider("openrouter", { ...fullEnv }, "anthropic/claude-opus-latest",
42
+ async (specifier) => { calls.push(specifier); return { default: { fromEnv: async (_e: NodeJS.ProcessEnv, model: string) => { calls.push(model); return fakeProvider; } } }; },
43
+ mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
44
+ assert.equal(p, fakeProvider);
45
+ assert.deepEqual(calls, ["@plurnk/plurnk-providers-openrouter", "anthropic/claude-opus-latest"]);
46
+ });
47
+
48
+ test("instantiateProvider: a per-alias baseUrl is passed to a bespoke factory as the 3rd-arg option", async () => {
49
+ resetDiscoveryCache();
50
+ let received: unknown;
51
+ await instantiateProvider("ollama", { ...fullEnv }, "qwen2.5-coder",
52
+ async () => ({ default: { fromEnv: async (_e: NodeJS.ProcessEnv, _m: string, options?: unknown) => { received = options; return fakeProvider; } } }),
53
+ mapOf({ ollama: "@plurnk/plurnk-providers-ollama" }),
54
+ "http://nook:11434");
55
+ assert.deepEqual(received, { baseUrl: "http://nook:11434" });
56
+ });
57
+
58
+ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe to the override host", async () => {
59
+ resetDiscoveryCache();
60
+ const probed: string[] = [];
61
+ mock.method(globalThis, "fetch", async (url: string) => {
62
+ probed.push(String(url));
63
+ return new Response(JSON.stringify({ data: [] }), { status: 200 });
64
+ });
65
+ await instantiateProvider("openai", { ...fullEnv }, "m", // fullEnv.OPENAI_BASE_URL is http://x — the override must win
66
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
67
+ "http://hazel2:8080/v1");
68
+ assert.ok(probed.some((u) => u === "http://hazel2:8080/v1/models"), `probe hit the override host; saw ${probed.join(", ")}`);
69
+ assert.equal(probed.some((u) => u.startsWith("http://x")), false); // never the per-name OPENAI_BASE_URL
70
+ mock.restoreAll();
71
+ });
72
+
73
+ test("instantiateProvider: a THIRD-PARTY scope is discovered — name maps to its package", async () => {
74
+ resetDiscoveryCache();
75
+ const imports: string[] = [];
76
+ const p = await instantiateProvider("foo", { ...fullEnv }, "m",
77
+ async (specifier) => { imports.push(specifier); return { default: { fromEnv: async () => fakeProvider } }; },
78
+ mapOf({ foo: "@acme/acme-provider-foo" }));
79
+ assert.equal(p, fakeProvider);
80
+ assert.deepEqual(imports, ["@acme/acme-provider-foo"]); // not an @plurnk/ specifier
81
+ });
82
+
83
+ test("instantiateProvider: a standard name is authoritative — a scanned package of the same name is shadowed", async () => {
84
+ resetDiscoveryCache();
85
+ mock.method(globalThis, "fetch", async (url: string) => {
86
+ if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
87
+ throw new Error("unexpected fetch");
88
+ });
89
+ const imports: string[] = [];
90
+ const p = await instantiateProvider("openai", { ...fullEnv }, "m",
91
+ async (s) => { imports.push(s); return {}; },
92
+ mapOf({ openai: "@acme/acme-provider-openai" })); // shadowed by tier 1
93
+ assert.equal(p.model, "m");
94
+ assert.deepEqual(imports, []); // the scanned same-name package is never imported
95
+ mock.restoreAll();
96
+ });
97
+
98
+ test("instantiateProvider: unknown provider throws — no standard, no discovered package", async () => {
99
+ resetDiscoveryCache();
100
+ await assert.rejects(
101
+ () => instantiateProvider("nope", { ...fullEnv }, "m", async () => ({}), mapOf({})),
102
+ /unknown provider "nope": not a standard provider, and no installed package declares plurnk\.kind:"provider" with name "nope"/,
103
+ );
104
+ });
105
+
106
+ test("instantiateProvider: an untrusted (skipped) provider gives a precise error, not 'unknown' (#15)", async () => {
107
+ resetDiscoveryCache();
108
+ const imports: string[] = [];
109
+ await assert.rejects(
110
+ () => instantiateProvider("foo", { ...fullEnv }, "m",
111
+ async (s) => { imports.push(s); return {}; },
112
+ mapOf({}, { foo: "@acme/acme-provider-foo" })), // discovered but trust-declined
113
+ /provider "foo" resolves to @acme\/acme-provider-foo, but it is untrusted under PLURNK_PLUGINS_TRUSTED_ONLY/,
114
+ );
115
+ assert.deepEqual(imports, []); // never imported an untrusted package
116
+ });
117
+
118
+ test("instantiateProvider: discovered package without a fromEnv factory throws (factory shape, SPEC )", async () => {
119
+ resetDiscoveryCache();
120
+ await assert.rejects(
121
+ () => instantiateProvider("broken", { ...fullEnv }, "m",
122
+ async () => ({ default: {} }),
123
+ mapOf({ broken: "@acme/acme-provider-broken" })),
124
+ /@acme\/acme-provider-broken default export is not a Provider factory/,
125
+ );
126
+ });
127
+
128
+ test("instantiateProvider: a discovered package whose import throws fails hard, naming the specifier and preserving the cause", async () => {
129
+ resetDiscoveryCache();
130
+ const cause = new Error("ERR_MODULE_NOT_FOUND");
131
+ await assert.rejects(
132
+ () => instantiateProvider("foo", { ...fullEnv }, "m",
133
+ async () => { throw cause; },
134
+ mapOf({ foo: "@acme/acme-provider-foo" })),
135
+ (err: Error) => {
136
+ assert.match(err.message, /provider "foo" resolves to @acme\/acme-provider-foo, but importing it failed/);
137
+ assert.equal(err.cause, cause); // original error preserved
138
+ return true;
139
+ },
140
+ );
141
+ });
142
+
143
+ test("instantiateProvider: per-alias knobs scope through to the provider (per-alias scoping doctrine, user 2026-07-03)", async () => {
144
+ resetDiscoveryCache();
145
+ mock.method(globalThis, "fetch", async (url: string) => {
146
+ if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
147
+ if (String(url).endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
148
+ return new Response(JSON.stringify({ data: [] }), { status: 200 });
149
+ });
150
+ const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW_turbo: "12345", PLURNK_PROVIDERS_LLAMA_SERVER_turbo: "1" };
151
+ const p = await instantiateProvider("openai", env, "m",
152
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
153
+ undefined, "turbo");
154
+ assert.equal(p.contextWindow, 12345); // _turbo CONTEXT_WINDOW reached the provider
155
+ assert.equal(p.constrainsOutput, true); // _turbo LLAMA_SERVER pin reached it too
156
+ // same env, DIFFERENT alias: neither override applies
157
+ const q = await instantiateProvider("openai", env, "m",
158
+ async () => ({}), async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() }),
159
+ undefined, "plain");
160
+ assert.equal(q.contextWindow, null);
161
+ assert.equal(q.constrainsOutput, false);
162
+ mock.restoreAll();
163
+ });
164
+
165
+ test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
166
+ resetDiscoveryCache();
167
+ const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
168
+ const p = await loadActiveProvider(env,
169
+ async () => ({ default: { fromEnv: async () => fakeProvider } }),
170
+ mapOf({ openrouter: "@plurnk/plurnk-providers-openrouter" }));
171
+ assert.equal(p, fakeProvider);
172
+ });
173
+
174
+ test("loadActiveProvider: throws a named error when no alias is active", async () => {
175
+ await assert.rejects(() => loadActiveProvider({ ...fullEnv }), /set PLURNK_MODEL to an alias/);
176
+ });
@@ -0,0 +1,93 @@
1
+ // Provider instantiation + active-alias resolution. Alias PARSING (the
2
+ // PLURNK_MODEL_<alias>=<provider>/<model> cascade + PLURNK_BASEURL_<alias>
3
+ // overrides) lives in @plurnk/plurnk-aliases — the zero-dep parser shared with
4
+ // thin clients (#27); this module resolves the active alias to a Provider.
5
+ //
6
+ // Two-tier resolution (SPEC §5): tier 1 is the closed standard-provider table;
7
+ // tier 2 is a SCOPE-AGNOSTIC node_modules scan (discover()) for packages
8
+ // declaring `plurnk.kind:"provider"` — first-party plugins (installed flat
9
+ // via @plurnk/plurnk-providers-all) AND third-party providers under any scope.
10
+ // The framework is contract-only — it does NOT depend on its plugins; the
11
+ // scan is what surfaces them (#12/#14).
12
+
13
+ import type { Provider, ProviderFactory } from "./types.ts";
14
+ import { isStandardProvider, standardProviderFromEnv } from "./standardProviders.ts";
15
+ import { discover, type DiscoverOptions, type Discovery } from "./discover.ts";
16
+ import { resolveActiveAlias } from "@plurnk/plurnk-aliases";
17
+ import { scopeEnvToAlias } from "./env.ts";
18
+
19
+ // Two injectable seams, both defaulting to production behavior and never passed
20
+ // by real callers: the module importer (tests exercise the bespoke path without
21
+ // a real package on disk) and the discovery scan (tests inject a fixed map).
22
+ type ImportModule = (specifier: string) => Promise<unknown>;
23
+ const importModule: ImportModule = (specifier) => import(specifier);
24
+ type DiscoverFn = (options?: DiscoverOptions) => Promise<Discovery>;
25
+
26
+ // The node_modules scan is filesystem work that never changes within a process,
27
+ // so it runs once and the result is memoized. A long-lived daemon pays one scan
28
+ // at first bespoke instantiation; every later run reuses it. The trust gate is
29
+ // boot config, so the env from the first scan stands for the process.
30
+ let discoveredCache: Discovery | null = null;
31
+ const providerPackages = async (discoverFn: DiscoverFn, env: NodeJS.ProcessEnv): Promise<Discovery> => {
32
+ discoveredCache ??= await discoverFn({ env });
33
+ return discoveredCache;
34
+ };
35
+
36
+ // Two-tier resolution (SPEC §5): tier 1 standard table → tier 2 discovered
37
+ // package (scope-agnostic scan, trust-gated) → fail-hard. The standard table is
38
+ // authoritative — a scanned package whose name duplicates a standard one is
39
+ // shadowed here (never reached), since tier 1 returns first.
40
+ export const instantiateProvider = async (
41
+ name: string,
42
+ env: NodeJS.ProcessEnv,
43
+ model: string,
44
+ importImpl: ImportModule = importModule,
45
+ discoverFn: DiscoverFn = discover,
46
+ baseUrl?: string, // per-alias endpoint override (PLURNK_BASEURL_<alias>); threaded to both tiers
47
+ alias?: string, // the alias this instantiation serves — scopes PLURNK_PROVIDERS_<KNOB>_<alias> overrides
48
+ ): Promise<Provider> => {
49
+ // Per-alias knob scoping: overlay any _<alias>-suffixed knob onto its bare
50
+ // name so both tiers (and every fromEnv) read plain vars, per-alias-resolved.
51
+ if (alias !== undefined) env = scopeEnvToAlias(env, alias);
52
+ if (isStandardProvider(name)) {
53
+ const standard = await standardProviderFromEnv(name, env, model, baseUrl);
54
+ if (standard === null) throw new Error(`provider "${name}": standard registry resolution failed`);
55
+ return standard;
56
+ }
57
+ const { registry, skipped } = await providerPackages(discoverFn, env);
58
+ const specifier = registry.get(name);
59
+ if (specifier === undefined) {
60
+ const declined = skipped.get(name);
61
+ if (declined !== undefined) {
62
+ throw new Error(`provider "${name}" resolves to ${declined}, but it is untrusted under PLURNK_PLUGINS_TRUSTED_ONLY — add it to the allowlist (or publish under @plurnk/)`);
63
+ }
64
+ throw new Error(`unknown provider "${name}": not a standard provider, and no installed package declares plurnk.kind:"provider" with name "${name}"`);
65
+ }
66
+ let mod: unknown;
67
+ try {
68
+ mod = await importImpl(specifier);
69
+ } catch (cause) {
70
+ throw new Error(`provider "${name}" resolves to ${specifier}, but importing it failed`, { cause });
71
+ }
72
+ const factory = (mod as { default?: ProviderFactory }).default;
73
+ if (factory === undefined || typeof factory.fromEnv !== "function") {
74
+ throw new Error(`${specifier} default export is not a Provider factory (missing static fromEnv)`);
75
+ }
76
+ return await factory.fromEnv(env, model, baseUrl !== undefined ? { baseUrl } : undefined);
77
+ };
78
+
79
+ // Test-only: drop the memoized discovery so a fresh scan/injection runs next.
80
+ export const resetDiscoveryCache = (): void => { discoveredCache = null; };
81
+
82
+ // Boot convenience: resolve the active alias cascade and instantiate it.
83
+ export const loadActiveProvider = async (
84
+ env: NodeJS.ProcessEnv = process.env,
85
+ importImpl: ImportModule = importModule,
86
+ discoverFn: DiscoverFn = discover,
87
+ ): Promise<Provider> => {
88
+ const alias = resolveActiveAlias(env);
89
+ if (alias === null) {
90
+ throw new Error("no active provider: set PLURNK_MODEL to an alias declared via PLURNK_MODEL_<alias>=<provider>/<model>");
91
+ }
92
+ return instantiateProvider(alias.provider, env, alias.model, importImpl, discoverFn, alias.baseUrl, alias.alias);
93
+ };
@@ -0,0 +1,24 @@
1
+ import test from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import { readFileSync, readdirSync } from "node:fs";
4
+ import { join, dirname } from "node:path";
5
+ import { fileURLToPath } from "node:url";
6
+
7
+ const root = join(dirname(fileURLToPath(import.meta.url)), "..", "src");
8
+ const sourceFiles = (): string[] =>
9
+ readdirSync(root).filter((name) => name.endsWith(".ts") && !name.endsWith(".test.ts"));
10
+
11
+ test("provider source does not import the service or database", () => {
12
+ for (const file of sourceFiles()) {
13
+ const source = readFileSync(join(root, file), "utf8");
14
+ assert.ok(!/from\s+["']@plurnk\/plurnk-service/.test(source), `${file} imports the service`);
15
+ assert.ok(!/from\s+["']node:sqlite/.test(source), `${file} imports node:sqlite`);
16
+ }
17
+ });
18
+
19
+ test("provider source does not import the PLURNK parser", () => {
20
+ for (const file of sourceFiles()) {
21
+ const source = readFileSync(join(root, file), "utf8");
22
+ assert.ok(!/from\s+["']@plurnk\/plurnk-grammar/.test(source), `${file} imports the parser`);
23
+ }
24
+ });