@plurnk/plurnk-providers 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -60
- package/SPEC.md +6 -6
- package/dist/OpenAICompat.js +2 -2
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.js +2 -2
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/env.js +3 -3
- package/dist/env.js.map +1 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +12 -0
- package/dist/openaiStream.js.map +1 -1
- package/package.json +7 -6
- package/src/Mock.test.ts +142 -0
- package/src/Mock.ts +95 -0
- package/src/OpenAICompat.test.ts +1107 -0
- package/src/OpenAICompat.ts +756 -0
- package/src/Pool.test.ts +155 -0
- package/src/Pool.ts +134 -0
- package/src/ProviderRegistry.test.ts +176 -0
- package/src/ProviderRegistry.ts +93 -0
- package/src/boundaries.test.ts +24 -0
- package/src/discover.test.ts +123 -0
- package/src/discover.ts +112 -0
- package/src/env.test.ts +190 -0
- package/src/env.ts +211 -0
- package/src/index.ts +51 -0
- package/src/lexicon-guard.test.ts +58 -0
- package/src/openaiStream.ts +279 -0
- package/src/standardProviders.test.ts +925 -0
- package/src/standardProviders.ts +618 -0
- package/src/telemetry.test.ts +62 -0
- package/src/telemetry.ts +108 -0
- package/src/types.ts +219 -0
- package/src/usage.test.ts +136 -0
- package/src/usage.ts +82 -0
- package/src/warnings.test.ts +31 -0
- package/src/warnings.ts +0 -0
|
@@ -0,0 +1,925 @@
|
|
|
1
|
+
import test, { mock } from "node:test";
|
|
2
|
+
import { strict as assert } from "node:assert";
|
|
3
|
+
import { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from "./standardProviders.ts";
|
|
4
|
+
|
|
5
|
+
// Base URLs are REQUIRED (no in-code default) — the fixture supplies the vendor
|
|
6
|
+
// defaults for the providers exercised outside the coverage loop. `openai` is
|
|
7
|
+
// deliberately omitted so its missing-base fail-hard test still fires.
|
|
8
|
+
const baseEnv = Object.freeze({
|
|
9
|
+
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
10
|
+
GROQ_BASE_URL: "https://api.groq.com/openai/v1",
|
|
11
|
+
DEEPINFRA_BASE_URL: "https://api.deepinfra.com/v1/openai",
|
|
12
|
+
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
13
|
+
ANTHROPIC_BASE_URL: "https://api.anthropic.com/v1",
|
|
14
|
+
PLURNK_BASE_URL: "https://plurnk.ai/v1",
|
|
15
|
+
});
|
|
16
|
+
|
|
17
|
+
// Mock fetch: serves GET /v1/models (the n_ctx probe) and a [DONE] stream for
|
|
18
|
+
// /chat/completions (generate). `nctx` controls the probed window. Records URLs.
|
|
19
|
+
const mockEndpoint = ({ nctx, metaNctx, modelId = "m" }: { nctx?: number; metaNctx?: number; modelId?: string } = {}) => {
|
|
20
|
+
const calls: string[] = [];
|
|
21
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
22
|
+
const u = String(url);
|
|
23
|
+
calls.push(u);
|
|
24
|
+
if (u.endsWith("/models")) {
|
|
25
|
+
const row = {
|
|
26
|
+
id: modelId,
|
|
27
|
+
...(nctx !== undefined ? { n_ctx: nctx } : {}),
|
|
28
|
+
...(metaNctx !== undefined ? { meta: { n_vocab: 262144, n_ctx: metaNctx } } : {}),
|
|
29
|
+
};
|
|
30
|
+
return new Response(JSON.stringify({ data: [row] }), { status: 200 });
|
|
31
|
+
}
|
|
32
|
+
// Honor the request's transport: SSE when stream:true, one JSON otherwise.
|
|
33
|
+
const streamed = init?.body !== undefined && JSON.parse(String(init.body)).stream === true;
|
|
34
|
+
if (streamed) return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
35
|
+
return new Response(JSON.stringify({ model: modelId, choices: [{ message: { content: "" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
36
|
+
});
|
|
37
|
+
return calls;
|
|
38
|
+
};
|
|
39
|
+
const chatCall = (calls: string[]) => calls.find((u) => u.endsWith("/chat/completions"));
|
|
40
|
+
import { resetEmittedWarnings } from "./warnings.ts";
|
|
41
|
+
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); }); // #40: warning-asserting tests stay order-independent
|
|
42
|
+
|
|
43
|
+
test("isStandardProvider: known vs unknown", () => {
|
|
44
|
+
assert.equal(isStandardProvider("openai"), true);
|
|
45
|
+
assert.equal(isStandardProvider("groq"), true);
|
|
46
|
+
assert.equal(isStandardProvider("openrouter"), false); // bespoke sibling
|
|
47
|
+
assert.equal(isStandardProvider("nope"), false);
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
test("standardProviderFromEnv: returns null for a non-standard name", async () => {
|
|
51
|
+
assert.equal(await standardProviderFromEnv("openrouter", { ...baseEnv }, "m"), null);
|
|
52
|
+
});
|
|
53
|
+
|
|
54
|
+
test("openai: throws a named error when OPENAI_BASE_URL is unset", async () => {
|
|
55
|
+
await assert.rejects(standardProviderFromEnv("openai", { ...baseEnv }, "m"), /OPENAI_BASE_URL or OPENAI_API_BASE must be set/);
|
|
56
|
+
});
|
|
57
|
+
|
|
58
|
+
test("openai: a still-set tokenizer var fails hard with the migration pointer (tokenizer shed)", async () => {
|
|
59
|
+
await assert.rejects(
|
|
60
|
+
standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x", OPENAI_TOKENIZER: "cl100k_base" }, "m"),
|
|
61
|
+
/OPENAI_TOKENIZER was removed — exact counting moved to the @plurnk\/plurnk-mimetypes tokenizer seam/,
|
|
62
|
+
);
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
test("openai: defaults to the chars/2 heuristic upper bound, and SURFACES it", async () => {
|
|
66
|
+
mockEndpoint();
|
|
67
|
+
const warned: Array<string | Error> = [];
|
|
68
|
+
mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
|
|
69
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
70
|
+
const s = "The quick brown fox.";
|
|
71
|
+
assert.equal(p!.countTokens(s), Math.ceil(s.length / 2));
|
|
72
|
+
// Never a silent fallback: the heuristic announces itself at construction.
|
|
73
|
+
assert.ok(warned.some((w) => String(w).includes("chars/2 upper bound")), `expected heuristic warning; got ${warned.join("; ")}`);
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
// — context-window resolution (issue #6) —
|
|
77
|
+
|
|
78
|
+
test("openai: derives contextWindow from endpoint n_ctx when env unset", async () => {
|
|
79
|
+
mockEndpoint({ nctx: 49152, modelId: "macher.gguf" });
|
|
80
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "macher.gguf");
|
|
81
|
+
assert.equal(p!.contextWindow, 49152);
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test("openai: derives contextWindow from llama-server's nested meta.n_ctx (issue #7)", async () => {
|
|
85
|
+
mockEndpoint({ metaNctx: 49152, modelId: "macher.gguf" });
|
|
86
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "macher.gguf");
|
|
87
|
+
assert.equal(p!.contextWindow, 49152);
|
|
88
|
+
});
|
|
89
|
+
|
|
90
|
+
test("openai: meta.n_ctx wins over a top-level n_ctx", async () => {
|
|
91
|
+
mockEndpoint({ nctx: 8192, metaNctx: 49152 });
|
|
92
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
93
|
+
assert.equal(p!.contextWindow, 49152);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
// — served model identity (#37) —
|
|
97
|
+
|
|
98
|
+
test("#37: servedModel resolves the backend's real served id when the wire model is an alias", async () => {
|
|
99
|
+
const served = "gemma-4-26B-A4B-it-qat-UD-Q4_K_XL.gguf";
|
|
100
|
+
mockEndpoint({ metaNctx: 49152, modelId: served }); // llama-server row reports the real name
|
|
101
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "turboderp"); // wire model is the alias
|
|
102
|
+
assert.equal(p!.servedModel, served); // seam maps this exactly
|
|
103
|
+
assert.equal(p!.model, "turboderp"); // the wire model stays the alias
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
test("#37: servedModel absent when the probe reads no row (consumer falls back to model)", async () => {
|
|
107
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
108
|
+
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
109
|
+
const streamed = init?.body !== undefined && JSON.parse(String(init.body)).stream === true;
|
|
110
|
+
if (streamed) return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
111
|
+
return new Response(JSON.stringify({ choices: [{ message: { content: "" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
112
|
+
});
|
|
113
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
114
|
+
assert.equal(p!.servedModel, undefined);
|
|
115
|
+
});
|
|
116
|
+
|
|
117
|
+
test("openai: explicit PLURNK_PROVIDERS_CONTEXT_WINDOW wins over n_ctx", async () => {
|
|
118
|
+
mockEndpoint({ nctx: 49152 });
|
|
119
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local", PLURNK_PROVIDERS_CONTEXT_WINDOW: "400000" }, "m");
|
|
120
|
+
assert.equal(p!.contextWindow, 400000);
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
test("openai: contextWindow null when the endpoint reports no n_ctx (e.g. real OpenAI)", async () => {
|
|
124
|
+
mockEndpoint({}); // models response without n_ctx
|
|
125
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
126
|
+
assert.equal(p!.contextWindow, null);
|
|
127
|
+
});
|
|
128
|
+
|
|
129
|
+
test("underivable context is SURFACED, never silent: PLURNK_CONTEXT_UNKNOWN names model + remediation", async () => {
|
|
130
|
+
mockEndpoint({}); // env unset, probe reports nothing, no catalog entry for "m"
|
|
131
|
+
const warned: Array<string | Error> = [];
|
|
132
|
+
mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
|
|
133
|
+
await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
134
|
+
const w = warned.map(String).find((x) => x.includes("context window underivable"));
|
|
135
|
+
assert.ok(w, `expected PLURNK_CONTEXT_UNKNOWN; got: ${warned.join("; ")}`);
|
|
136
|
+
assert.ok(w!.includes('"m"'), "names the model");
|
|
137
|
+
assert.ok(w!.includes("PLURNK_PROVIDERS_CONTEXT_WINDOW"), "names the remediation var");
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
test("derivable context emits NO context-unknown warning", async () => {
|
|
141
|
+
mockEndpoint({ nctx: 49152 });
|
|
142
|
+
const warned: Array<string | Error> = [];
|
|
143
|
+
mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
|
|
144
|
+
await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
145
|
+
assert.equal(warned.map(String).filter((x) => x.includes("context window underivable")).length, 0);
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
test("openai: probe failure degrades to null, never throws", async () => {
|
|
149
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
150
|
+
if (String(url).endsWith("/models")) return new Response("nope", { status: 503 });
|
|
151
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
152
|
+
return new Response(body, { status: 200 });
|
|
153
|
+
});
|
|
154
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
155
|
+
assert.equal(p!.contextWindow, null);
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
test("openai: a probe network error (fetch rejects) degrades to null context and no grammar, never throws", async () => {
|
|
159
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
160
|
+
if (String(url).endsWith("/models")) throw new TypeError("network down"); // fetch itself rejects
|
|
161
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
162
|
+
return new Response(body, { status: 200 });
|
|
163
|
+
});
|
|
164
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
165
|
+
assert.equal(p!.contextWindow, null);
|
|
166
|
+
await p!.generate({ workerId: "r", messages: [] }); // no fingerprint → no grammar capability, generate still works
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
test("openai: a slot-probe network error (fetch rejects on /props) degrades slotCount to null, never throws", async () => {
|
|
170
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
171
|
+
const u = String(url);
|
|
172
|
+
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "m", meta: { n_ctx: 4096 } }] }), { status: 200 }); // llama fingerprint → slot probe runs
|
|
173
|
+
if (u.endsWith("/props")) throw new TypeError("network down"); // the /props slot probe rejects
|
|
174
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
175
|
+
return new Response(body, { status: 200 });
|
|
176
|
+
});
|
|
177
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
178
|
+
assert.equal(p!.contextWindow, 4096); // meta.n_ctx still resolved despite the slot-probe failure
|
|
179
|
+
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"?' }); // grammar still transports; no id_slot (slotCount null)
|
|
180
|
+
});
|
|
181
|
+
|
|
182
|
+
test("cloud standard providers do not probe (no n_ctx fetch)", async () => {
|
|
183
|
+
const calls = mockEndpoint({ nctx: 99999 });
|
|
184
|
+
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
185
|
+
assert.equal(p!.contextWindow, 8192); // from the pin, NOT the mocked probe (groq has no probeNctx)
|
|
186
|
+
assert.equal(calls.some((u) => u.endsWith("/models")), false); // never queried /models
|
|
187
|
+
});
|
|
188
|
+
|
|
189
|
+
test("standard provider tags failures with provider:<name> telemetry source", async () => {
|
|
190
|
+
const { ProviderError } = await import("./telemetry.ts");
|
|
191
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
192
|
+
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
193
|
+
return new Response("forbidden", { status: 403 }); // chat/completions fails
|
|
194
|
+
});
|
|
195
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
196
|
+
await assert.rejects(() => p!.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
197
|
+
assert.ok(err instanceof ProviderError);
|
|
198
|
+
assert.equal(err.toTelemetryEvent().source, "provider:openai");
|
|
199
|
+
assert.equal(err.kind, "unauthorized");
|
|
200
|
+
return true;
|
|
201
|
+
});
|
|
202
|
+
});
|
|
203
|
+
|
|
204
|
+
test("groq: requires its API key", async () => {
|
|
205
|
+
await assert.rejects(standardProviderFromEnv("groq", { ...baseEnv }, "m"), /GROQ_API_KEY must be set/);
|
|
206
|
+
});
|
|
207
|
+
|
|
208
|
+
test("groq: applies PLURNK_PROVIDERS_CONTEXT_WINDOW", async () => {
|
|
209
|
+
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "131072" }, "m");
|
|
210
|
+
assert.equal(p!.contextWindow, 131072);
|
|
211
|
+
});
|
|
212
|
+
|
|
213
|
+
// — grammar capability detection (SPEC §13, issue #8) —
|
|
214
|
+
|
|
215
|
+
test("openai: llama-server fingerprint (meta block) enables grammar transport", async () => {
|
|
216
|
+
const bodies: string[] = [];
|
|
217
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
218
|
+
if (String(url).endsWith("/models")) {
|
|
219
|
+
return new Response(JSON.stringify({ data: [{ id: "m", meta: { n_vocab: 262144, n_ctx: 49152 } }] }), { status: 200 });
|
|
220
|
+
}
|
|
221
|
+
if (String(url).endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
|
|
222
|
+
bodies.push(String(init?.body));
|
|
223
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
224
|
+
return new Response(body, { status: 200 });
|
|
225
|
+
});
|
|
226
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
227
|
+
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"?' });
|
|
228
|
+
const sent = JSON.parse(bodies[0]);
|
|
229
|
+
assert.equal(sent.grammar, 'root ::= "x"?');
|
|
230
|
+
assert.equal(sent.repeat_penalty, 1.15);
|
|
231
|
+
assert.equal(sent.id_slot, 0); // fingerprint wires internal slot affinity too
|
|
232
|
+
});
|
|
233
|
+
|
|
234
|
+
// — #34: detection must not silently decide capability —
|
|
235
|
+
|
|
236
|
+
const llamaModels = () => new Response(JSON.stringify({ data: [{ id: "m", meta: { n_vocab: 262144, n_ctx: 49152 } }] }), { status: 200 });
|
|
237
|
+
const sse = () => new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
238
|
+
|
|
239
|
+
test("openai: a transient /models failure is RETRIED — capability survives one hiccup (#34)", async () => {
|
|
240
|
+
let modelCalls = 0;
|
|
241
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
242
|
+
const u = String(url);
|
|
243
|
+
if (u.endsWith("/models")) { modelCalls++; if (modelCalls < 3) return new Response("busy", { status: 503 }); return llamaModels(); }
|
|
244
|
+
if (u.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
|
|
245
|
+
return sse();
|
|
246
|
+
});
|
|
247
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
248
|
+
assert.equal(modelCalls, 3); // two failures retried, third answered
|
|
249
|
+
assert.equal(p!.constrainsOutput, true); // rails LIVE despite the hiccups
|
|
250
|
+
});
|
|
251
|
+
|
|
252
|
+
test("openai: detection exhaustion is SURFACED, never silent (#34)", async () => {
|
|
253
|
+
const warned: Array<string | Error> = [];
|
|
254
|
+
mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
|
|
255
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
256
|
+
if (String(url).endsWith("/models")) return new Response("down", { status: 503 });
|
|
257
|
+
return sse();
|
|
258
|
+
});
|
|
259
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
260
|
+
assert.equal(p!.constrainsOutput, false);
|
|
261
|
+
assert.ok(warned.some((w) => String(w).includes("llama-server detection failed") && String(w).includes("PLURNK_PROVIDERS_LLAMA_SERVER=1")), `expected PLURNK_PROBE_FAILED warning; got ${warned.join(" | ")}`);
|
|
262
|
+
});
|
|
263
|
+
|
|
264
|
+
test("openai: PLURNK_PROVIDERS_LLAMA_SERVER=1 pins llamacpp capabilities WITHOUT trusting the probe (#34)", async () => {
|
|
265
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
266
|
+
const u = String(url);
|
|
267
|
+
if (u.endsWith("/models")) return new Response("down", { status: 503 }); // probe dead the whole time
|
|
268
|
+
if (u.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 2 }), { status: 200 });
|
|
269
|
+
return sse();
|
|
270
|
+
});
|
|
271
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local", PLURNK_PROVIDERS_LLAMA_SERVER: "1" }, "m");
|
|
272
|
+
assert.equal(p!.constrainsOutput, true); // pinned: rails live, no probe dependency
|
|
273
|
+
assert.notEqual(p!.tokenize, undefined); // full capability set rides the pin
|
|
274
|
+
assert.equal(p!.requiresMaxTokens, true); // #43: the unbounded-decode fact rides the pin too
|
|
275
|
+
});
|
|
276
|
+
|
|
277
|
+
test("#43: llama-server fingerprint surfaces requiresMaxTokens=true; cloud stays undefined (no claim)", async () => {
|
|
278
|
+
mockEndpoint({ metaNctx: 49152 }); // llama-server fingerprint
|
|
279
|
+
const local = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
280
|
+
assert.equal(local!.requiresMaxTokens, true); // decodes to the wall (providers#10) — consumer must bring an envelope
|
|
281
|
+
mock.restoreAll();
|
|
282
|
+
mockEndpoint({}); // plain remote: no fingerprint
|
|
283
|
+
const cloud = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k" }, "deepseek-v4-flash");
|
|
284
|
+
assert.equal(cloud!.requiresMaxTokens, undefined); // self-clamping cloud: no claim, no refusal trigger
|
|
285
|
+
});
|
|
286
|
+
|
|
287
|
+
test("openai: PLURNK_PROVIDERS_LLAMA_SERVER=0 forces plain-remote even when the fingerprint matches (#34)", async () => {
|
|
288
|
+
mockEndpoint({ metaNctx: 49152 }); // fingerprint SAYS llama-server
|
|
289
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local", PLURNK_PROVIDERS_LLAMA_SERVER: "0" }, "m");
|
|
290
|
+
assert.equal(p!.constrainsOutput, false);
|
|
291
|
+
assert.equal(p!.tokenize, undefined);
|
|
292
|
+
});
|
|
293
|
+
|
|
294
|
+
test("openai: a garbage PLURNK_PROVIDERS_LLAMA_SERVER value fails hard", async () => {
|
|
295
|
+
await assert.rejects(
|
|
296
|
+
standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local", PLURNK_PROVIDERS_LLAMA_SERVER: "yes" }, "m"),
|
|
297
|
+
/PLURNK_PROVIDERS_LLAMA_SERVER must be "1" .* "0" .* or unset/,
|
|
298
|
+
);
|
|
299
|
+
});
|
|
300
|
+
|
|
301
|
+
test("constrainsOutput: fireworks (static response_format) reports true; groq reports false", async () => {
|
|
302
|
+
mockEndpoint();
|
|
303
|
+
const fw = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
304
|
+
assert.equal(fw!.constrainsOutput, true);
|
|
305
|
+
const gq = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
306
|
+
assert.equal(gq!.constrainsOutput, false);
|
|
307
|
+
});
|
|
308
|
+
|
|
309
|
+
test("openai: llama-server fingerprint surfaces the tokenize() capability (native /tokenize, model's own vocab)", async () => {
|
|
310
|
+
const tokenizeCalls: Array<{ url: string; body: string }> = [];
|
|
311
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
312
|
+
const u = String(url);
|
|
313
|
+
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "m", meta: { n_vocab: 262144, n_ctx: 49152 } }] }), { status: 200 });
|
|
314
|
+
if (u.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
|
|
315
|
+
if (u.endsWith("/tokenize")) { tokenizeCalls.push({ url: u, body: String(init?.body) }); return new Response(JSON.stringify({ tokens: [101, 7, 42] }), { status: 200 }); }
|
|
316
|
+
throw new Error(`unexpected fetch ${u}`);
|
|
317
|
+
});
|
|
318
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
319
|
+
assert.notEqual(p!.tokenize, undefined);
|
|
320
|
+
const ids = await p!.tokenize!("hello");
|
|
321
|
+
assert.deepEqual(ids, [101, 7, 42]);
|
|
322
|
+
assert.equal(tokenizeCalls[0].url, "http://local/tokenize"); // native root endpoint, not /v1
|
|
323
|
+
assert.deepEqual(JSON.parse(tokenizeCalls[0].body), { content: "hello" });
|
|
324
|
+
});
|
|
325
|
+
|
|
326
|
+
test("openai: non-llama-server endpoint has NO tokenize capability (undefined is the honest signal)", async () => {
|
|
327
|
+
mockEndpoint({ nctx: 8192 }); // top-level n_ctx, no meta → vLLM-ish, not llama-server
|
|
328
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
329
|
+
assert.equal(p!.tokenize, undefined);
|
|
330
|
+
});
|
|
331
|
+
|
|
332
|
+
test("plurnk: detectLlamaServer=false never surfaces tokenize, even when the endpoint fingerprints", async () => {
|
|
333
|
+
mockEndpoint({ metaNctx: 32768, modelId: "plurnk" });
|
|
334
|
+
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
335
|
+
assert.equal(p!.tokenize, undefined);
|
|
336
|
+
});
|
|
337
|
+
|
|
338
|
+
test("openai: top-level n_ctx without meta (vLLM) does NOT enable grammar", async () => {
|
|
339
|
+
const bodies: string[] = [];
|
|
340
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
341
|
+
if (String(url).endsWith("/models")) {
|
|
342
|
+
return new Response(JSON.stringify({ data: [{ id: "m", n_ctx: 8192 }] }), { status: 200 });
|
|
343
|
+
}
|
|
344
|
+
bodies.push(String(init?.body));
|
|
345
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
346
|
+
return new Response(body, { status: 200 });
|
|
347
|
+
});
|
|
348
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
349
|
+
assert.equal(p!.contextWindow, 8192); // window still read
|
|
350
|
+
await p!.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
|
|
351
|
+
assert.equal("grammar" in JSON.parse(bodies[0]), false);
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
test("openai: llama-server upgrades 'think'→'template'; enable_thinking mirrors PLURNK_PROVIDERS_REASONING", async () => {
|
|
355
|
+
const mk = (reasoning: string) => {
|
|
356
|
+
const bodies: string[] = [];
|
|
357
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
358
|
+
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "m", meta: { n_ctx: 49152 } }] }), { status: 200 });
|
|
359
|
+
if (String(url).endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
|
|
360
|
+
bodies.push(String(init?.body));
|
|
361
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
362
|
+
return new Response(body, { status: 200 });
|
|
363
|
+
});
|
|
364
|
+
return { bodies, env: { ...baseEnv, PLURNK_PROVIDERS_REASONING: reasoning, OPENAI_BASE_URL: "http://local" } };
|
|
365
|
+
};
|
|
366
|
+
const off = mk("off");
|
|
367
|
+
const pOff = await standardProviderFromEnv("openai", off.env, "m");
|
|
368
|
+
await pOff!.generate({ workerId: "r", messages: [] });
|
|
369
|
+
assert.deepEqual(JSON.parse(off.bodies[0]).chat_template_kwargs, { enable_thinking: false });
|
|
370
|
+
assert.equal("think" in JSON.parse(off.bodies[0]), false); // think→template, never raw think
|
|
371
|
+
|
|
372
|
+
mock.restoreAll();
|
|
373
|
+
const on = mk("adaptive");
|
|
374
|
+
const pOn = await standardProviderFromEnv("openai", on.env, "m");
|
|
375
|
+
await pOn!.generate({ workerId: "r", messages: [] });
|
|
376
|
+
assert.deepEqual(JSON.parse(on.bodies[0]).chat_template_kwargs, { enable_thinking: true });
|
|
377
|
+
});
|
|
378
|
+
|
|
379
|
+
test("openai: non-llama-server endpoint keeps the 'think' style (no template kwargs)", async () => {
|
|
380
|
+
const bodies: string[] = [];
|
|
381
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
382
|
+
if (String(url).endsWith("/models")) {
|
|
383
|
+
return new Response(JSON.stringify({ data: [{ id: "m" }] }), { status: 200 }); // no meta → not llama-server
|
|
384
|
+
}
|
|
385
|
+
bodies.push(String(init?.body));
|
|
386
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
387
|
+
return new Response(body, { status: 200 });
|
|
388
|
+
});
|
|
389
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
390
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
391
|
+
assert.equal("chat_template_kwargs" in JSON.parse(bodies[0]), false);
|
|
392
|
+
});
|
|
393
|
+
|
|
394
|
+
test("openai: probed total_slots drives internal run→slot affinity; never surfaces", async () => {
|
|
395
|
+
const bodies: string[] = [];
|
|
396
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
397
|
+
const u = String(url);
|
|
398
|
+
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "m", meta: { n_ctx: 16384 } }] }), { status: 200 });
|
|
399
|
+
if (u.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 2 }), { status: 200 });
|
|
400
|
+
bodies.push(String(init?.body));
|
|
401
|
+
const body = new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } });
|
|
402
|
+
return new Response(body, { status: 200 });
|
|
403
|
+
});
|
|
404
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "m");
|
|
405
|
+
assert.equal(p!.contextWindow, 16384); // per-slot window, as the server reports it
|
|
406
|
+
assert.equal("slotCount" in p!, false); // resource internals never on the surface
|
|
407
|
+
await p!.generate({ workerId: "run-A", messages: [] });
|
|
408
|
+
await p!.generate({ workerId: "run-B", messages: [] });
|
|
409
|
+
await p!.generate({ workerId: "run-A", messages: [] });
|
|
410
|
+
assert.deepEqual(bodies.map((b) => JSON.parse(b).id_slot), [0, 1, 0]);
|
|
411
|
+
|
|
412
|
+
mock.restoreAll();
|
|
413
|
+
mockEndpoint({ nctx: 8192 }); // top-level n_ctx, no meta → not llama-server
|
|
414
|
+
const vllm = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x" }, "m");
|
|
415
|
+
const calls = mockEndpoint({ nctx: 8192 });
|
|
416
|
+
await vllm!.generate({ workerId: "run-A", messages: [] });
|
|
417
|
+
assert.equal(calls.some((u) => u.endsWith("/props")), false); // no fingerprint → no props probe
|
|
418
|
+
});
|
|
419
|
+
|
|
420
|
+
test("openai: env-pinned context size does not disable grammar detection (probe still runs)", async () => {
|
|
421
|
+
const calls = mockEndpoint({ metaNctx: 49152 });
|
|
422
|
+
const p = await standardProviderFromEnv(
|
|
423
|
+
"openai",
|
|
424
|
+
{ ...baseEnv, OPENAI_BASE_URL: "http://local", PLURNK_PROVIDERS_CONTEXT_WINDOW: "400000" },
|
|
425
|
+
"m",
|
|
426
|
+
);
|
|
427
|
+
assert.equal(p!.contextWindow, 400000); // env wins for the window
|
|
428
|
+
assert.equal(calls.some((u) => u.endsWith("/models")), true); // probe still fired for capability
|
|
429
|
+
});
|
|
430
|
+
|
|
431
|
+
// — URL resolution —
|
|
432
|
+
|
|
433
|
+
test("openai flexBaseStrip: base with trailing /v1 yields a single /v1/chat/completions", async () => {
|
|
434
|
+
const calls = mockEndpoint();
|
|
435
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x/v1" }, "m");
|
|
436
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
437
|
+
assert.equal(chatCall(calls), "http://x/v1/chat/completions");
|
|
438
|
+
});
|
|
439
|
+
|
|
440
|
+
test("a provider appends chatPath to its base URL", async () => {
|
|
441
|
+
const calls = mockEndpoint();
|
|
442
|
+
const p = await standardProviderFromEnv("deepinfra", { ...baseEnv, DEEPINFRA_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
443
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
444
|
+
assert.equal(chatCall(calls), "https://api.deepinfra.com/v1/openai/chat/completions");
|
|
445
|
+
});
|
|
446
|
+
|
|
447
|
+
test("baseUrlVar supplies the base URL (no in-code default)", async () => {
|
|
448
|
+
const calls = mockEndpoint();
|
|
449
|
+
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", GROQ_BASE_URL: "http://proxy/openai/v1", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
450
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
451
|
+
assert.equal(chatCall(calls), "http://proxy/openai/v1/chat/completions");
|
|
452
|
+
});
|
|
453
|
+
|
|
454
|
+
test("a standard provider fails hard when its base URL is unset (no in-code default)", async () => {
|
|
455
|
+
await assert.rejects(
|
|
456
|
+
standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", GROQ_BASE_URL: "" }, "m"),
|
|
457
|
+
/groq provider: GROQ_BASE_URL must be set/,
|
|
458
|
+
);
|
|
459
|
+
});
|
|
460
|
+
|
|
461
|
+
test("every registry entry resolves the chat URL the spec encodes", async () => {
|
|
462
|
+
const first = (v: string | readonly string[] | undefined): string | undefined =>
|
|
463
|
+
v === undefined ? undefined : typeof v === "string" ? v : v[0];
|
|
464
|
+
const envFor = (name: string): NodeJS.ProcessEnv => {
|
|
465
|
+
const spec = STANDARD_PROVIDERS[name];
|
|
466
|
+
// Pin the window: this sweep asserts the chat URL, not budgeting, and the
|
|
467
|
+
// cloud specs fail-hard on an uncataloged model absent a pin (#419).
|
|
468
|
+
const e: NodeJS.ProcessEnv = { ...baseEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" };
|
|
469
|
+
// Specs whose auth rides a custom headersFromEnv builder have no apiKeyVar.
|
|
470
|
+
const keyVar = first(spec.apiKeyVar);
|
|
471
|
+
if (keyVar !== undefined) e[keyVar] = "k";
|
|
472
|
+
// Every entry requires its base URL (no in-code default); bedrock also
|
|
473
|
+
// accepts its BASE_URL var, so setting it here covers all specs.
|
|
474
|
+
const baseVar = first(spec.baseUrlVar);
|
|
475
|
+
if (baseVar !== undefined) e[baseVar] = "http://x/v1";
|
|
476
|
+
return e;
|
|
477
|
+
};
|
|
478
|
+
for (const name of Object.keys(STANDARD_PROVIDERS)) {
|
|
479
|
+
const calls = mockEndpoint();
|
|
480
|
+
const p = await standardProviderFromEnv(name, envFor(name), "m");
|
|
481
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
482
|
+
const u = chatCall(calls)!;
|
|
483
|
+
assert.ok(u.endsWith("/chat/completions"), `${name} → ${u}`);
|
|
484
|
+
assert.ok(u.startsWith("http"), `${name} → ${u}`);
|
|
485
|
+
mock.restoreAll();
|
|
486
|
+
}
|
|
487
|
+
});
|
|
488
|
+
|
|
489
|
+
// — accepted env-var aliases & derived bases (audit; web-sourced wild conventions) —
|
|
490
|
+
|
|
491
|
+
test("deepinfra: resolves auth via the DEEPINFRA_TOKEN alias (not only DEEPINFRA_API_KEY)", async () => {
|
|
492
|
+
const seen = plurnkMock();
|
|
493
|
+
const p = await standardProviderFromEnv("deepinfra", { ...baseEnv, DEEPINFRA_TOKEN: "di-tok", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
494
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
495
|
+
assert.equal(chatHeaders(seen).Authorization, "Bearer di-tok");
|
|
496
|
+
mock.restoreAll();
|
|
497
|
+
});
|
|
498
|
+
|
|
499
|
+
test("deepinfra: a required key unset across ALL aliases fails hard, naming each", async () => {
|
|
500
|
+
await assert.rejects(
|
|
501
|
+
standardProviderFromEnv("deepinfra", { ...baseEnv }, "m"),
|
|
502
|
+
/DEEPINFRA_API_KEY or DEEPINFRA_API_TOKEN or DEEPINFRA_TOKEN must be set/,
|
|
503
|
+
);
|
|
504
|
+
});
|
|
505
|
+
|
|
506
|
+
test("openai: base URL via the legacy OPENAI_API_BASE alias", async () => {
|
|
507
|
+
const calls = mockEndpoint();
|
|
508
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_API_BASE: "http://legacy/v1" }, "m");
|
|
509
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
510
|
+
assert.equal(chatCall(calls), "http://legacy/v1/chat/completions");
|
|
511
|
+
mock.restoreAll();
|
|
512
|
+
});
|
|
513
|
+
|
|
514
|
+
test("bedrock: derives the base from AWS_REGION (.../openai/v1), no BEDROCK_BASE_URL needed", async () => {
|
|
515
|
+
const calls = mockEndpoint();
|
|
516
|
+
const p = await standardProviderFromEnv("bedrock", { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok", AWS_REGION: "us-west-2", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
517
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
518
|
+
assert.equal(chatCall(calls), "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions");
|
|
519
|
+
mock.restoreAll();
|
|
520
|
+
});
|
|
521
|
+
|
|
522
|
+
test("bedrock: AWS_DEFAULT_REGION is accepted when AWS_REGION is unset", async () => {
|
|
523
|
+
const calls = mockEndpoint();
|
|
524
|
+
const p = await standardProviderFromEnv("bedrock", { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok", AWS_DEFAULT_REGION: "eu-west-1", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
525
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
526
|
+
assert.equal(chatCall(calls), "https://bedrock-runtime.eu-west-1.amazonaws.com/openai/v1/chat/completions");
|
|
527
|
+
mock.restoreAll();
|
|
528
|
+
});
|
|
529
|
+
|
|
530
|
+
test("bedrock: an explicit BEDROCK_BASE_URL overrides region derivation", async () => {
|
|
531
|
+
const calls = mockEndpoint();
|
|
532
|
+
const env = { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok", AWS_REGION: "us-west-2", BEDROCK_BASE_URL: "https://gw.internal/openai/v1", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" };
|
|
533
|
+
const p = await standardProviderFromEnv("bedrock", env, "m");
|
|
534
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
535
|
+
assert.equal(chatCall(calls), "https://gw.internal/openai/v1/chat/completions");
|
|
536
|
+
mock.restoreAll();
|
|
537
|
+
});
|
|
538
|
+
|
|
539
|
+
test("bedrock: neither BEDROCK_BASE_URL nor a region fails hard, naming the region vars", async () => {
|
|
540
|
+
await assert.rejects(
|
|
541
|
+
standardProviderFromEnv("bedrock", { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok" }, "m"),
|
|
542
|
+
/BEDROCK_BASE_URL must be set, or AWS_REGION \/ AWS_DEFAULT_REGION/,
|
|
543
|
+
);
|
|
544
|
+
});
|
|
545
|
+
|
|
546
|
+
test("bedrock: contextWindow resolves from the catalog via the inference-profile's publisher (#22)", async () => {
|
|
547
|
+
const env = { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok", AWS_REGION: "us-west-2" };
|
|
548
|
+
const p = await standardProviderFromEnv("bedrock", env, "us.anthropic.claude-sonnet-4-5");
|
|
549
|
+
// Resolves a REAL window from the anthropic catalog via publisher-stripping (#22). Assert the
|
|
550
|
+
// MECHANISM (non-null, positive), never the literal - a catalog refresh must not break the build.
|
|
551
|
+
assert.ok(p!.contextWindow !== null && p!.contextWindow > 0, `expected a catalog-resolved window, got ${p!.contextWindow}`);
|
|
552
|
+
// cost is NOT taken from the native anthropic rate (bedrock marks up) — stays 0
|
|
553
|
+
assert.equal(p!.costFor({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
|
|
554
|
+
});
|
|
555
|
+
|
|
556
|
+
test("bedrock: a publisher the catalog lacks (meta) fails hard (cloud, no probe); PLURNK_PROVIDERS_CONTEXT_WINDOW still wins", async () => {
|
|
557
|
+
const base = { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok", AWS_REGION: "us-east-1" };
|
|
558
|
+
await assert.rejects(
|
|
559
|
+
standardProviderFromEnv("bedrock", base, "us.meta.llama-3-70b"),
|
|
560
|
+
/context window unresolved/, // #419: cloud provider, uncataloged publisher, unpinned
|
|
561
|
+
);
|
|
562
|
+
const pinned = await standardProviderFromEnv("bedrock", { ...base, PLURNK_PROVIDERS_CONTEXT_WINDOW: "128000" }, "us.meta.llama-3-70b");
|
|
563
|
+
assert.equal(pinned!.contextWindow, 128000);
|
|
564
|
+
});
|
|
565
|
+
|
|
566
|
+
test("PLURNK_PROVIDERS_GBNF_DEBUG=1 wires through: an invalid grammar throws without a chat call", async () => {
|
|
567
|
+
const calls = mockEndpoint({ metaNctx: 4096 }); // llama fingerprint → grammarStyle llamacpp
|
|
568
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://x", PLURNK_PROVIDERS_GBNF_DEBUG: "1" }, "m");
|
|
569
|
+
await assert.rejects(
|
|
570
|
+
() => p!.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // invalid GBNF
|
|
571
|
+
/PLURNK_PROVIDERS_GBNF_DEBUG/,
|
|
572
|
+
);
|
|
573
|
+
assert.equal(chatCall(calls), undefined); // probes only — the grammar never reached /chat/completions
|
|
574
|
+
mock.restoreAll();
|
|
575
|
+
});
|
|
576
|
+
|
|
577
|
+
// — vendored-snapshot fallback (#19): live wins, catalog fills the gap —
|
|
578
|
+
|
|
579
|
+
import { catalogSnapshot } from "@plurnk/plurnk-models";
|
|
580
|
+
|
|
581
|
+
test("standard provider: a catalog hit fills contextWindow + cost when there's no live source", async () => {
|
|
582
|
+
// groq doesn't probe; pick a real model id from the vendored snapshot.
|
|
583
|
+
const [modelId, info] = Object.entries(catalogSnapshot().groq)[0];
|
|
584
|
+
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k" }, modelId);
|
|
585
|
+
assert.ok(p !== null);
|
|
586
|
+
assert.equal(p.contextWindow, info.contextWindow); // catalog window, no probe needed
|
|
587
|
+
if (info.cost !== undefined) {
|
|
588
|
+
// 1M output tokens → outputPer1M USD, in pico-USD (per-1M ×1e6 per token).
|
|
589
|
+
const c = p.costFor({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
|
|
590
|
+
assert.equal(c, Math.round(info.cost.outputPer1M * 1e6 * 1_000_000));
|
|
591
|
+
} else {
|
|
592
|
+
assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
|
|
593
|
+
}
|
|
594
|
+
});
|
|
595
|
+
|
|
596
|
+
test("standard provider: completionReserve is capped to min(catalog.maxOutput, 25%*ctx) when catalog wins (#507)", async () => {
|
|
597
|
+
// claude-opus-4-8: ctx=1,000,000, maxOutput=128,000 — 25% = 250,000 (catalog wins: 128K < 250K).
|
|
598
|
+
// anthropic has no probe — context + maxOutput come cleanly from catalog.
|
|
599
|
+
const opus = catalogSnapshot().anthropic?.["claude-opus-4-8"];
|
|
600
|
+
if (opus?.maxOutput === undefined) return; // skip if catalog snapshot lacks the model
|
|
601
|
+
const p = await standardProviderFromEnv("anthropic", { ...baseEnv, ANTHROPIC_API_KEY: "k" }, "claude-opus-4-8");
|
|
602
|
+
assert.ok(p !== null);
|
|
603
|
+
assert.equal(p.completionReserve, opus.maxOutput); // 128000: catalog wins over 25%=250000
|
|
604
|
+
});
|
|
605
|
+
|
|
606
|
+
test("standard provider: completionReserve uses percentage floor when maxOutput >= 25%*ctx (#507)", async () => {
|
|
607
|
+
// claude-haiku-4-5-20251001: ctx=200,000, maxOutput=64,000 — 25% = 50,000 (floor wins: min(64K,50K)=50K).
|
|
608
|
+
const haiku = catalogSnapshot().anthropic?.["claude-haiku-4-5-20251001"];
|
|
609
|
+
if (haiku?.maxOutput === undefined) return;
|
|
610
|
+
const p = await standardProviderFromEnv("anthropic", { ...baseEnv, ANTHROPIC_API_KEY: "k" }, "claude-haiku-4-5-20251001");
|
|
611
|
+
assert.ok(p !== null);
|
|
612
|
+
const expected = Math.round(haiku.contextWindow * 0.25); // 50000: floor wins over maxOutput=64K
|
|
613
|
+
assert.equal(p.completionReserve, expected);
|
|
614
|
+
});
|
|
615
|
+
|
|
616
|
+
test("standard provider: operator absolute COMPLETION_RESERVE overrides the catalog cap (#507)", async () => {
|
|
617
|
+
// Even though catalog gives 128K for claude-opus, an absolute env pin always wins.
|
|
618
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "k", PLURNK_PROVIDERS_COMPLETION_RESERVE: "8192" };
|
|
619
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-opus-4-8");
|
|
620
|
+
assert.ok(p !== null);
|
|
621
|
+
assert.equal(p.completionReserve, 8192); // absolute pin beats catalog
|
|
622
|
+
});
|
|
623
|
+
|
|
624
|
+
test("standard provider: a local (non-cataloged) model misses the fallback — probe owns it, contextWindow null", async () => {
|
|
625
|
+
mock.method(globalThis, "fetch", async (url: string) =>
|
|
626
|
+
String(url).endsWith("/models") ? new Response(JSON.stringify({ data: [] }), { status: 200 }) : new Response("{}", { status: 200 }));
|
|
627
|
+
const p = await standardProviderFromEnv("openai", { ...baseEnv, OPENAI_BASE_URL: "http://local" }, "macher.gguf");
|
|
628
|
+
assert.ok(p !== null);
|
|
629
|
+
assert.equal(p.contextWindow, null); // empty probe + catalog miss → null; live owns the local case
|
|
630
|
+
mock.restoreAll();
|
|
631
|
+
});
|
|
632
|
+
|
|
633
|
+
// — anthropic standard entry (first-party Claude, #18) —
|
|
634
|
+
|
|
635
|
+
test("anthropic: standard entry sends bearer auth + the thinking param to the compat endpoint", async () => {
|
|
636
|
+
let body = "";
|
|
637
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
638
|
+
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 }); }
|
|
639
|
+
return new Response("{}", { status: 200 });
|
|
640
|
+
});
|
|
641
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "sk-ant-xyz", PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "3000" };
|
|
642
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-opus-4-8");
|
|
643
|
+
assert.ok(p !== null);
|
|
644
|
+
await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
645
|
+
assert.deepEqual(JSON.parse(body).thinking, { type: "enabled", budget_tokens: 3000 });
|
|
646
|
+
mock.restoreAll();
|
|
647
|
+
});
|
|
648
|
+
|
|
649
|
+
test("anthropic: context + cost come from the catalog (no probe)", async () => {
|
|
650
|
+
const [modelId, info] = Object.entries(catalogSnapshot().anthropic)[0];
|
|
651
|
+
const p = await standardProviderFromEnv("anthropic", { ...baseEnv, ANTHROPIC_API_KEY: "k" }, modelId);
|
|
652
|
+
assert.equal(p!.contextWindow, info.contextWindow);
|
|
653
|
+
});
|
|
654
|
+
|
|
655
|
+
// — reasoning reserve derivation (#568) —
|
|
656
|
+
|
|
657
|
+
test("standard provider: reasoningReserve is half of completionReserve when reasoning=adaptive (#568)", async () => {
|
|
658
|
+
// claude-opus-4-8: ctx=1M, maxOutput=128K → completionReserve=128K (catalog wins over 25%=250K).
|
|
659
|
+
// Half-completion heuristic: reasoningReserve=64K. anthropic has no probe.
|
|
660
|
+
const opus = catalogSnapshot().anthropic?.["claude-opus-4-8"];
|
|
661
|
+
if (opus?.maxOutput === undefined) return;
|
|
662
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "k", PLURNK_PROVIDERS_REASONING: "adaptive" };
|
|
663
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-opus-4-8");
|
|
664
|
+
assert.ok(p !== null);
|
|
665
|
+
assert.equal(p.completionReserve, opus.maxOutput);
|
|
666
|
+
assert.equal(p.reasoningReserve, Math.round(opus.maxOutput / 2));
|
|
667
|
+
});
|
|
668
|
+
|
|
669
|
+
test("standard provider: reasoningReserve is exact REASONING_BUDGET when reasoning=on (#568)", async () => {
|
|
670
|
+
// Even though completionReserve/2 would be 64K, the explicit budget (8K) wins.
|
|
671
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "k", PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "8192" };
|
|
672
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-opus-4-8");
|
|
673
|
+
assert.ok(p !== null);
|
|
674
|
+
assert.equal(p.reasoningReserve, 8192);
|
|
675
|
+
});
|
|
676
|
+
|
|
677
|
+
test("standard provider: operator absolute REASONING_RESERVE overrides all derived paths (#568)", async () => {
|
|
678
|
+
// Absolute token pin beats both the budget and the half-completion heuristic.
|
|
679
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "k", PLURNK_PROVIDERS_REASONING: "on", PLURNK_PROVIDERS_REASONING_BUDGET: "8192", PLURNK_PROVIDERS_REASONING_RESERVE: "4096" };
|
|
680
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-opus-4-8");
|
|
681
|
+
assert.ok(p !== null);
|
|
682
|
+
assert.equal(p.reasoningReserve, 4096);
|
|
683
|
+
});
|
|
684
|
+
|
|
685
|
+
test("anthropic: frequency_penalty is NOT sent (Claude native API has no such param, DOC #568)", async () => {
|
|
686
|
+
let body = "";
|
|
687
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
688
|
+
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 }); }
|
|
689
|
+
return new Response("{}", { status: 200 });
|
|
690
|
+
});
|
|
691
|
+
const env = { ...baseEnv, ANTHROPIC_API_KEY: "k", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4" };
|
|
692
|
+
const p = await standardProviderFromEnv("anthropic", env, "claude-haiku-4-5-20251001");
|
|
693
|
+
await p!.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
694
|
+
assert.equal(JSON.parse(body).frequency_penalty, undefined);
|
|
695
|
+
mock.restoreAll();
|
|
696
|
+
});
|
|
697
|
+
|
|
698
|
+
// — bedrock standard entry (AWS, bearer API key, #19) —
|
|
699
|
+
|
|
700
|
+
test("bedrock: requires BEDROCK_BASE_URL (region-templated) and AWS_BEARER_TOKEN_BEDROCK", async () => {
|
|
701
|
+
await assert.rejects(
|
|
702
|
+
standardProviderFromEnv("bedrock", { ...baseEnv, AWS_BEARER_TOKEN_BEDROCK: "tok" }, "us.anthropic.claude-sonnet-4-6"),
|
|
703
|
+
/BEDROCK_BASE_URL must be set/,
|
|
704
|
+
);
|
|
705
|
+
await assert.rejects(
|
|
706
|
+
standardProviderFromEnv("bedrock", { ...baseEnv, BEDROCK_BASE_URL: "https://bedrock-runtime.us-east-1.amazonaws.com/v1" }, "m"),
|
|
707
|
+
/AWS_BEARER_TOKEN_BEDROCK must be set/,
|
|
708
|
+
);
|
|
709
|
+
});
|
|
710
|
+
|
|
711
|
+
test("bedrock: an explicit base is used verbatim and sends the Bedrock API key as bearer", async () => {
|
|
712
|
+
const calls: { url: string; auth: string }[] = [];
|
|
713
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
714
|
+
calls.push({ url: String(url), auth: String((init?.headers as Record<string, string>)?.Authorization ?? "") });
|
|
715
|
+
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
716
|
+
});
|
|
717
|
+
const env = { ...baseEnv, BEDROCK_BASE_URL: "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1", AWS_BEARER_TOKEN_BEDROCK: "bedrock-key", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" };
|
|
718
|
+
const p = await standardProviderFromEnv("bedrock", env, "us.anthropic.claude-sonnet-4-6");
|
|
719
|
+
await p!.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
720
|
+
assert.equal(calls[0].url, "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions");
|
|
721
|
+
assert.equal(calls[0].auth, "Bearer bedrock-key");
|
|
722
|
+
mock.restoreAll();
|
|
723
|
+
});
|
|
724
|
+
|
|
725
|
+
// — plurnk hosted model: two optional credentials via headersFromEnv —
|
|
726
|
+
|
|
727
|
+
// Mock that serves the /models probe + captures the chat-completions request
|
|
728
|
+
// headers; `chatStatus` lets a test force a rejection.
|
|
729
|
+
const plurnkMock = (chatStatus = 200) => {
|
|
730
|
+
const seen: { url: string; headers: Record<string, string>; body: string }[] = [];
|
|
731
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
732
|
+
const u = String(url);
|
|
733
|
+
const headers = (init?.headers ?? {}) as Record<string, string>;
|
|
734
|
+
seen.push({ url: u, headers, body: String(init?.body ?? "") });
|
|
735
|
+
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
|
|
736
|
+
if (u.endsWith("/props")) return new Response(JSON.stringify({ total_slots: 1 }), { status: 200 });
|
|
737
|
+
if (chatStatus !== 200) return new Response("denied", { status: chatStatus });
|
|
738
|
+
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
739
|
+
});
|
|
740
|
+
return seen;
|
|
741
|
+
};
|
|
742
|
+
const chatHeaders = (seen: { url: string; headers: Record<string, string> }[]) =>
|
|
743
|
+
seen.find((s) => s.url.endsWith("/chat/completions"))!.headers;
|
|
744
|
+
|
|
745
|
+
test("plurnk: unset PLURNK_API_KEY throws the friendly #537 guidance pre-call, not a raw upstream 401", async () => {
|
|
746
|
+
await assert.rejects(
|
|
747
|
+
standardProviderFromEnv("plurnk", { ...baseEnv }, "plurnk"), // base from PLURNK_BASE_URL, no key
|
|
748
|
+
/PLURNK_API_KEY not found\. Acquire one at https:\/\/plurnk\.ai \. Plurnk also supports local models and alternative cloud provider configurations\./,
|
|
749
|
+
);
|
|
750
|
+
});
|
|
751
|
+
|
|
752
|
+
test("plurnk: PLURNK_API_KEY sends the bearer; no separate account header (the key identifies the account)", async () => {
|
|
753
|
+
const seen = plurnkMock();
|
|
754
|
+
const env = { ...baseEnv, PLURNK_API_KEY: "pk-live-123" };
|
|
755
|
+
const p = await standardProviderFromEnv("plurnk", env, "plurnk");
|
|
756
|
+
await p!.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
757
|
+
const h = chatHeaders(seen);
|
|
758
|
+
assert.equal(h.Authorization, "Bearer pk-live-123");
|
|
759
|
+
assert.equal("Plurnk-Account" in h, false); // retired — the key carries account identity
|
|
760
|
+
mock.restoreAll();
|
|
761
|
+
});
|
|
762
|
+
|
|
763
|
+
test("plurnk: forwards attributions + client as Plurnk-* telemetry headers (firstPartyMetadata)", async () => {
|
|
764
|
+
const seen = plurnkMock();
|
|
765
|
+
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
766
|
+
await p!.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.0.0"], client: "plurnk-tui/0.9.0" });
|
|
767
|
+
const h = chatHeaders(seen);
|
|
768
|
+
assert.equal(h["Plurnk-Attribution"], '["@acme/x@1.0.0"]');
|
|
769
|
+
assert.equal(h["Plurnk-Client"], "plurnk-tui/0.9.0");
|
|
770
|
+
assert.equal(h["Plurnk-Worker-Id"], "r"); // worker identity rides the same gate (#26/#511)
|
|
771
|
+
mock.restoreAll();
|
|
772
|
+
});
|
|
773
|
+
|
|
774
|
+
test("plurnk: forwards strikes as Plurnk-Strikes — and 0 is a real value, distinct from absent (#313)", async () => {
|
|
775
|
+
let seen = plurnkMock();
|
|
776
|
+
let p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
777
|
+
await p!.generate({ workerId: "r", messages: [], strikes: 3 });
|
|
778
|
+
assert.equal(chatHeaders(seen)["Plurnk-Strikes"], "3");
|
|
779
|
+
mock.restoreAll();
|
|
780
|
+
seen = plurnkMock();
|
|
781
|
+
p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
782
|
+
await p!.generate({ workerId: "r", messages: [], strikes: 0 }); // clean streak reported explicitly
|
|
783
|
+
assert.equal(chatHeaders(seen)["Plurnk-Strikes"], "0");
|
|
784
|
+
mock.restoreAll();
|
|
785
|
+
seen = plurnkMock();
|
|
786
|
+
p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
787
|
+
await p!.generate({ workerId: "r", messages: [] }); // not reported → no header
|
|
788
|
+
assert.equal("Plurnk-Strikes" in chatHeaders(seen), false);
|
|
789
|
+
});
|
|
790
|
+
|
|
791
|
+
test("fireworks: does NOT forward attributions/client — first-party telemetry can't leak to a third party", async () => {
|
|
792
|
+
let seenHeaders: Record<string, string> = {};
|
|
793
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
794
|
+
if (String(url).endsWith("/chat/completions")) {
|
|
795
|
+
seenHeaders = (init?.headers ?? {}) as Record<string, string>;
|
|
796
|
+
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
797
|
+
}
|
|
798
|
+
return new Response("{}", { status: 200 });
|
|
799
|
+
});
|
|
800
|
+
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "deepseek-v4-flash");
|
|
801
|
+
await p!.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.0.0"], client: "plurnk-tui/0.9.0", strikes: 2 });
|
|
802
|
+
assert.equal("Plurnk-Attribution" in seenHeaders, false);
|
|
803
|
+
assert.equal("Plurnk-Client" in seenHeaders, false);
|
|
804
|
+
assert.equal("Plurnk-Strikes" in seenHeaders, false); // strikes gated identically
|
|
805
|
+
assert.equal("Plurnk-Worker-Id" in seenHeaders, false); // worker identity gated identically (#26/#511)
|
|
806
|
+
mock.restoreAll();
|
|
807
|
+
});
|
|
808
|
+
|
|
809
|
+
test("plurnk: a 401 is classified unauthorized — terminal, never retried", async () => {
|
|
810
|
+
const { ProviderError } = await import("./telemetry.ts");
|
|
811
|
+
const seen = plurnkMock(401);
|
|
812
|
+
const env = { ...baseEnv, PLURNK_PROVIDERS_RETRY_ATTEMPTS: "3", PLURNK_API_KEY: "expired" };
|
|
813
|
+
const p = await standardProviderFromEnv("plurnk", env, "plurnk");
|
|
814
|
+
await assert.rejects(() => p!.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
815
|
+
assert.ok(err instanceof ProviderError); assert.equal(err.kind, "unauthorized"); return true;
|
|
816
|
+
});
|
|
817
|
+
assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried
|
|
818
|
+
mock.restoreAll();
|
|
819
|
+
});
|
|
820
|
+
|
|
821
|
+
test("plurnk: a set-but-rejected key (401) surfaces the distinct #537 rejected hint, not the missing-key wording", async () => {
|
|
822
|
+
const seen = plurnkMock(401);
|
|
823
|
+
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-bad" }, "plurnk");
|
|
824
|
+
await assert.rejects(
|
|
825
|
+
() => p!.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] }),
|
|
826
|
+
/PLURNK_API_KEY was rejected by plurnk\.ai \(invalid or expired\)\. Verify it at https:\/\/plurnk\.ai \./,
|
|
827
|
+
);
|
|
828
|
+
assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried, distinct message
|
|
829
|
+
});
|
|
830
|
+
|
|
831
|
+
test("plurnk: normalizes the endpoint's balance_pico into meta.balancePico (#23)", async () => {
|
|
832
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
833
|
+
const u = String(url);
|
|
834
|
+
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
|
|
835
|
+
// plurnk has detectLlamaServer:false → streams; balance rides as a top-level field on a chunk.
|
|
836
|
+
const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance_pico":880000000}\n\ndata: [DONE]';
|
|
837
|
+
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode(sse)); c.close(); } }), { status: 200 });
|
|
838
|
+
});
|
|
839
|
+
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
840
|
+
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
841
|
+
assert.equal(res.meta?.balancePico, 880000000);
|
|
842
|
+
mock.restoreAll();
|
|
843
|
+
});
|
|
844
|
+
|
|
845
|
+
test("a third-party (non-plurnk) provider never NORMALIZES balancePico — only plurnk holds that contract", async () => {
|
|
846
|
+
mock.method(globalThis, "fetch", async () =>
|
|
847
|
+
new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"hi"}}],"balance_pico":880000000}\n\ndata: [DONE]')); c.close(); } }), { status: 200 }));
|
|
848
|
+
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
849
|
+
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
850
|
+
assert.equal("balancePico" in (res.meta ?? {}), false); // groq has no balanceMetaKey — no normalization
|
|
851
|
+
assert.equal(res.meta?.balance_pico, 880000000); // but the raw field still passes through (every-provider meta)
|
|
852
|
+
mock.restoreAll();
|
|
853
|
+
});
|
|
854
|
+
|
|
855
|
+
test("plurnk: reads its window from upstream but stays a plain OpenAI client — no grammar, no slot pinning, despite a meta block", async () => {
|
|
856
|
+
// The mock's /models returns a meta block (a llama-server fingerprint), yet
|
|
857
|
+
// detectLlamaServer:false means plurnk reads only the window and refuses every
|
|
858
|
+
// capability it could otherwise be talked into.
|
|
859
|
+
const seen = plurnkMock();
|
|
860
|
+
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
861
|
+
assert.equal(p!.contextWindow, 49152); // window STILL read from upstream — a 32k→48k change is a server decision
|
|
862
|
+
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
863
|
+
const body = JSON.parse(seen.find((s) => s.url.endsWith("/chat/completions"))!.body);
|
|
864
|
+
assert.equal("grammar" in body, false); // never forwards GBNF — the router injects its own
|
|
865
|
+
assert.equal("response_format" in body, false);
|
|
866
|
+
assert.equal("id_slot" in body, false); // never slot-pinned
|
|
867
|
+
assert.equal("think" in body, false); // reasoningStyle "none" → no reasoning param leaks
|
|
868
|
+
mock.restoreAll();
|
|
869
|
+
});
|
|
870
|
+
|
|
871
|
+
// — fireworks carries GBNF via response_format.grammar (cloud GBNF, #grammarStyle) —
|
|
872
|
+
|
|
873
|
+
test("fireworks: a grammar transports as response_format.grammar (not the llama.cpp top-level field)", async () => {
|
|
874
|
+
let body = "";
|
|
875
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
876
|
+
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
877
|
+
return new Response("{}", { status: 200 });
|
|
878
|
+
});
|
|
879
|
+
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "accounts/fireworks/models/deepseek-v4-pro");
|
|
880
|
+
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
881
|
+
const b = JSON.parse(body);
|
|
882
|
+
assert.deepEqual(b.response_format, { type: "grammar", grammar: 'root ::= "ok"' });
|
|
883
|
+
assert.equal("grammar" in b, false);
|
|
884
|
+
mock.restoreAll();
|
|
885
|
+
});
|
|
886
|
+
|
|
887
|
+
// — fireworks modelPrefix: the alias carries only the distinctive tail —
|
|
888
|
+
|
|
889
|
+
const fireworksWireModel = async (alias: string): Promise<string> => {
|
|
890
|
+
let body = "";
|
|
891
|
+
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
892
|
+
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
893
|
+
return new Response("{}", { status: 200 });
|
|
894
|
+
});
|
|
895
|
+
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, alias);
|
|
896
|
+
await p!.generate({ workerId: "r", messages: [] });
|
|
897
|
+
mock.restoreAll();
|
|
898
|
+
return JSON.parse(body).model;
|
|
899
|
+
};
|
|
900
|
+
|
|
901
|
+
test("fireworks: a bare alias is prefixed with accounts/fireworks/models/ on the wire", async () => {
|
|
902
|
+
assert.equal(await fireworksWireModel("deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
|
|
903
|
+
});
|
|
904
|
+
|
|
905
|
+
test("fireworks: an already-prefixed id is left unchanged (idempotent prepend)", async () => {
|
|
906
|
+
assert.equal(await fireworksWireModel("accounts/fireworks/models/deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
|
|
907
|
+
});
|
|
908
|
+
|
|
909
|
+
test("#518 prompt_cache_key: default-ON for a standard provider (workerId), OFF for anthropic (cache_control)", async () => {
|
|
910
|
+
const bodies: string[] = [];
|
|
911
|
+
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
912
|
+
bodies.push(String(init?.body ?? ""));
|
|
913
|
+
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode("data: [DONE]")); c.close(); } }), { status: 200 });
|
|
914
|
+
});
|
|
915
|
+
const env = { ...baseEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192", TOGETHER_BASE_URL: "https://api.together.xyz/v1", ANTHROPIC_BASE_URL: "https://api.anthropic.com/v1" };
|
|
916
|
+
|
|
917
|
+
const together = await standardProviderFromEnv("together", { ...env, TOGETHER_API_KEY: "k" }, "moonshotai/Kimi-K2.7-Code");
|
|
918
|
+
await together!.generate({ workerId: "worker-xyz", messages: [] });
|
|
919
|
+
assert.equal(JSON.parse(bodies.at(-1)!).prompt_cache_key, "worker-xyz"); // default-on for standard providers
|
|
920
|
+
|
|
921
|
+
const anthropic = await standardProviderFromEnv("anthropic", { ...env, ANTHROPIC_API_KEY: "k" }, "claude-x");
|
|
922
|
+
await anthropic!.generate({ workerId: "worker-xyz", messages: [] });
|
|
923
|
+
assert.equal("prompt_cache_key" in JSON.parse(bodies.at(-1)!), false); // opted out (cache_control mechanism)
|
|
924
|
+
mock.restoreAll();
|
|
925
|
+
});
|