@plurnk/plurnk-providers 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1107 @@
1
+ import test, { mock } from "node:test";
2
+ import { strict as assert } from "node:assert";
3
+ import OpenAICompatProvider, { effortFromBudget } from "./OpenAICompat.ts";
4
+ import { OpenAiHttpError } from "./openaiStream.ts";
5
+ import { ProviderError } from "./telemetry.ts";
6
+
7
+ // Build a fake fetch returning a one-chunk SSE stream, capturing the request
8
+ // so tests can assert what the spine sent on the wire.
9
+ const sseStream = (chunks: unknown[]) => {
10
+ const lines = [...chunks.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
11
+ return new ReadableStream({
12
+ start(controller) {
13
+ controller.enqueue(new TextEncoder().encode(lines));
14
+ controller.close();
15
+ },
16
+ });
17
+ };
18
+
19
+ const installFetch = (chunks: unknown[]) => {
20
+ const calls: { url: string; init: RequestInit }[] = [];
21
+ mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
22
+ calls.push({ url, init });
23
+ return new Response(sseStream(chunks), { status: 200 });
24
+ });
25
+ return calls;
26
+ };
27
+
28
+ // Fake fetch returning one non-streamed JSON body — for the paths the spine
29
+ // demotes off SSE (a response_format grammar). Captures the request the same way.
30
+ const installFetchJson = (payload: unknown) => {
31
+ const calls: { url: string; init: RequestInit }[] = [];
32
+ mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
33
+ calls.push({ url, init });
34
+ return new Response(JSON.stringify(payload), { status: 200, headers: { "Content-Type": "application/json" } });
35
+ });
36
+ return calls;
37
+ };
38
+
39
+ const jsonChoice = { model: "m", choices: [{ message: { content: "x" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
40
+
41
+ // Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
42
+ // streams its chunks; any other status returns that error (with an optional
43
+ // retry-after header). The last entry repeats once the script runs out.
44
+ type ScriptedResponse = { status: number; chunks?: unknown[]; retryAfter?: number | string; body?: string };
45
+ const installFetchScript = (responses: ScriptedResponse[]) => {
46
+ const calls: { url: string; init: RequestInit }[] = [];
47
+ let i = 0;
48
+ mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
49
+ calls.push({ url, init });
50
+ const r = responses[Math.min(i, responses.length - 1)];
51
+ i++;
52
+ if (r.status === 200) return new Response(sseStream(r.chunks ?? []), { status: 200 });
53
+ const headers = r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {};
54
+ return new Response(r.body ?? "err", { status: r.status, headers });
55
+ });
56
+ return calls;
57
+ };
58
+
59
+ // Let the pending request + its catch/backoff scheduling drain before asserting.
60
+ const flush = () => new Promise<void>((r) => setImmediate(r));
61
+
62
+ import { resetEmittedWarnings } from "./warnings.ts";
63
+ test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); }); // #40: warning-asserting tests stay order-independent
64
+
65
+ test("effortFromBudget: maps budget to tiers", () => {
66
+ assert.equal(effortFromBudget(1), "low");
67
+ assert.equal(effortFromBudget(1000), "low");
68
+ assert.equal(effortFromBudget(1001), "medium");
69
+ assert.equal(effortFromBudget(4000), "medium");
70
+ assert.equal(effortFromBudget(4001), "high");
71
+ });
72
+
73
+ test("#543: OpenAiHttpError distills a non-JSON (edge/CDN HTML) body and drops the OpenAI prefix", () => {
74
+ const cf = new OpenAiHttpError(524, "<!DOCTYPE html><html><body>Error code 524</body></html>", null);
75
+ assert.equal(cf.message, "524 origin timeout"); // distilled: no raw HTML, no "OpenAI" prefix
76
+ assert.ok(cf.body.length > 20); // raw body retained on the field for forensics
77
+ const api = new OpenAiHttpError(400, '{"error":{"message":"bad param"}}', null);
78
+ assert.match(api.message, /^OpenAI 400 - \{/); // JSON API error passes through verbatim
79
+ });
80
+
81
+ test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
82
+ const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
83
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
84
+ await assert.rejects(p.generate({ workerId: "r", messages: [] }));
85
+ await flush();
86
+ assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
87
+ mock.restoreAll();
88
+ });
89
+
90
+ test("#548: a 422 grammar_invalid is transient — retried on the budget, surfaces as grammar_invalid", async () => {
91
+ const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
92
+ const calls = installFetchScript([{ status: 422, body }]);
93
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
94
+ await assert.rejects(
95
+ p.generate({ workerId: "r", messages: [] }),
96
+ (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
97
+ );
98
+ await flush();
99
+ assert.equal(calls.length, 3); // initial + 2 retries: rode the bounded budget, unlike a terminal 422
100
+ mock.restoreAll();
101
+ });
102
+
103
+ test("an SSE error frame is a failed exchange, not an empty completion", async () => {
104
+ const calls = installFetch([{
105
+ status: 422,
106
+ error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
107
+ }]);
108
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
109
+ await assert.rejects(
110
+ p.generate({ workerId: "r", messages: [] }),
111
+ (e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
112
+ );
113
+ assert.equal(calls.length, 1);
114
+ });
115
+
116
+ test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
117
+ installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
118
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
119
+ const res = await p.generate({ workerId: "r", messages: [] });
120
+ assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
121
+ });
122
+
123
+ test("#539: without a probed eos_token the content passes through untouched", async () => {
124
+ installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
125
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
126
+ const res = await p.generate({ workerId: "r", messages: [] });
127
+ assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
128
+ });
129
+
130
+ test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
131
+ installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
132
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
133
+ const res = await p.generate({ workerId: "r", messages: [] });
134
+ assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
135
+ });
136
+
137
+ test("identity getters and defaults", () => {
138
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
139
+ assert.equal(p.model, "m");
140
+ assert.equal(p.contextWindow, null); // default
141
+ assert.equal(p.countTokens(""), 0);
142
+ assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
143
+ assert.equal(p.costFor({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
144
+ });
145
+
146
+ test("injected countTokens and costFor are used", () => {
147
+ const p = new OpenAICompatProvider({
148
+ model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
149
+ countTokens: (t) => t.length,
150
+ costFor: (u) => u.total * 2,
151
+ });
152
+ assert.equal(p.countTokens("abc"), 3);
153
+ assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
154
+ });
155
+
156
+ test("generate maps a streamed response into ProviderResponse", async () => {
157
+ const p = new OpenAICompatProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
158
+ installFetch([
159
+ { model: "wire-model", choices: [{ delta: { content: "hel" } }] },
160
+ { choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
161
+ { usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5, cached_tokens: 1 } },
162
+ ]);
163
+ const { assistant, assistantRaw } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
164
+ assert.equal(assistant.content, "hello");
165
+ assert.equal(assistant.model, "wire-model"); // wire-reported wins
166
+ assert.equal(assistant.finishReason, "stop");
167
+ assert.deepEqual(assistant.usage, { prompt: 3, completion: 2, reasoning: 0, cached: 1, total: 5 });
168
+ assert.equal(assistant.reasoning, null); // none emitted
169
+ assert.notEqual(assistantRaw, undefined);
170
+ });
171
+
172
+ test("generate normalizes an out-of-set finish_reason to null", async () => {
173
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
174
+ installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
175
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
176
+ assert.equal(assistant.finishReason, null);
177
+ });
178
+
179
+ test("generate translates a backend cap synonym to canonical length (#425)", async () => {
180
+ // gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
181
+ // "length" so its truncation check (=== "length") is a cross-backend invariant.
182
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
183
+ installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
184
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
185
+ assert.equal(assistant.finishReason, "length");
186
+ });
187
+
188
+ test("generate translates end_turn to canonical stop (#425)", async () => {
189
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
190
+ installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
191
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
192
+ assert.equal(assistant.finishReason, "stop");
193
+ });
194
+
195
+ test("generate aggregates reasoning deltas under multiple field names", async () => {
196
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
197
+ installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
198
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
199
+ assert.equal(assistant.reasoning, "because");
200
+ assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent (#482)
201
+ });
202
+
203
+ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details surface verbatim, text entries do not", async () => {
204
+ // The live o4-mini-via-OpenRouter shape: reasoning null, one encrypted entry.
205
+ installFetchJson({ model: "m", choices: [{ message: {
206
+ content: "4", reasoning: null,
207
+ reasoning_details: [
208
+ { type: "reasoning.encrypted", data: "gAAAAABqBLOB", format: "openai-responses-v1", id: "rs_1", index: 0 },
209
+ { type: "reasoning.text", text: "never surfaced here" },
210
+ ],
211
+ }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
212
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
213
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
214
+ // item shape: wire `id` preserved, subtype from position (#482 widening)
215
+ assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
216
+ assert.equal(assistant.reasoning, null); // sealed turn: nothing readable
217
+ assert.equal(assistant.content, "4");
218
+ });
219
+
220
+ test("#482 widening: distinct wire ids stay distinct items (a single-object shape would collide them)", async () => {
221
+ installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning: null, reasoning_details: [
222
+ { type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
223
+ { type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
224
+ ] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
225
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
226
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
227
+ assert.equal(assistant.reasoningEncrypted?.length, 2);
228
+ assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
229
+ });
230
+
231
+ test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
232
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
233
+ installFetch([
234
+ { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
235
+ { choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
236
+ { choices: [{ delta: { content: "4" }, finish_reason: "stop" }] },
237
+ ]);
238
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
239
+ assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAABqXYZ", format: "openai-responses-v1" }] }]);
240
+ assert.equal(assistant.content, "4");
241
+ });
242
+
243
+ test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
244
+ const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
245
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
246
+ await on.generate({ workerId: "r", messages: [] });
247
+ assert.equal(JSON.parse(calls[0].init.body as string).think, true);
248
+
249
+ mock.restoreAll();
250
+ const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
251
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
252
+ await off.generate({ workerId: "r", messages: [] });
253
+ assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
254
+ });
255
+
256
+ test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
257
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
258
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
259
+ await p.generate({ workerId: "r", messages: [] });
260
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
261
+ });
262
+
263
+ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 — literal is MiniMax-only), on sends the tier", async () => {
264
+ // expected === null → the field must be ABSENT from the wire body. Fireworks
265
+ // 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
266
+ // #403): adaptive = the backend's own default posture = omission.
267
+ for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
268
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
269
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
270
+ await p.generate({ workerId: "r", messages: [] });
271
+ const body = JSON.parse(calls[0].init.body as string);
272
+ if (expected === null) assert.equal("reasoning_effort" in body, false, `mode ${reasoning.mode}: field must be omitted`);
273
+ else assert.equal(body.reasoning_effort, expected, `mode ${reasoning.mode}`);
274
+ mock.restoreAll();
275
+ }
276
+ });
277
+
278
+ test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
279
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
280
+ // default rides with the grammar (non-streamed demotion path)
281
+ let calls = installFetchJson(jsonChoice);
282
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
283
+ assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
284
+ mock.restoreAll();
285
+ // explicit caller sampling wins over the default
286
+ calls = installFetchJson(jsonChoice);
287
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"', sampling: { temperature: 0.7 } });
288
+ assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.7);
289
+ mock.restoreAll();
290
+ // temperature is now the UNIVERSAL default: present without a grammar too
291
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
292
+ await p.generate({ workerId: "r", messages: [] });
293
+ assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
294
+ });
295
+
296
+ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
297
+ const base = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
298
+ // set + llamacpp -> the loop-breakers ride the wire
299
+ const p = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
300
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
301
+ await p.generate({ workerId: "r", messages: [] });
302
+ let body = JSON.parse(calls[0].init.body as string);
303
+ assert.equal(body.dry_multiplier, 0.8);
304
+ assert.equal(body.dry_base, 1.75);
305
+ assert.equal(body.dry_allowed_length, 2);
306
+ assert.equal(body.repeat_last_n, 512);
307
+ assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
308
+ mock.restoreAll();
309
+ // unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
310
+ const p2 = new OpenAICompatProvider({ ...base, grammarStyle: "llamacpp" });
311
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
312
+ await p2.generate({ workerId: "r", messages: [] });
313
+ body = JSON.parse(calls[0].init.body as string);
314
+ assert.equal("dry_multiplier" in body, false);
315
+ assert.equal("repeat_last_n" in body, false);
316
+ mock.restoreAll();
317
+ // DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
318
+ const p3 = new OpenAICompatProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
319
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
320
+ await p3.generate({ workerId: "r", messages: [] });
321
+ body = JSON.parse(calls[0].init.body as string);
322
+ assert.equal("dry_multiplier" in body, false);
323
+ assert.equal("repeat_last_n" in body, false);
324
+ mock.restoreAll();
325
+ });
326
+
327
+ test("effort_explicit: intent maps IDENTICALLY with and without a grammar — the #32 clamp is lifted (reasoning+rails coexist)", async () => {
328
+ const warned: Array<string | Error> = [];
329
+ mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
330
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format", source: "provider:test" });
331
+ // grammar transported → intent STILL flows through (canary-verified: the mask
332
+ // covers only content; clamping to "none" was the plan-less #331 regression)
333
+ let calls = installFetchJson(jsonChoice);
334
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
335
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
336
+ assert.equal(warned.filter((w) => String(w).includes("clamped")).length, 0, "no clamp warning — the clamp is gone");
337
+ mock.restoreAll();
338
+ // no grammar → same mapping
339
+ const p2 = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format" });
340
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
341
+ await p2.generate({ workerId: "r", messages: [] });
342
+ assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
343
+ });
344
+
345
+ test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
346
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
347
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
348
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
349
+ const body = JSON.parse(calls[0].init.body as string);
350
+ assert.equal(body.temperature, 0.2);
351
+ assert.equal(body.repeat_penalty, 1.15);
352
+ });
353
+
354
+ test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
355
+ // response_format cloud (fireworks) with NO grammar - the firefast case that went out bare
356
+ const fw = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
357
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
358
+ await fw.generate({ workerId: "r", messages: [] });
359
+ assert.equal(JSON.parse(calls[0].init.body as string).repetition_penalty, 1.15);
360
+ mock.restoreAll();
361
+ // llama.cpp with NO grammar carries its key too (unconstrained local is guarded now)
362
+ const llama = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
363
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
364
+ await llama.generate({ workerId: "r", messages: [] });
365
+ assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
366
+ mock.restoreAll();
367
+ // a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
368
+ const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
369
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
370
+ await cloud.generate({ workerId: "r", messages: [] });
371
+ const cloudBody = JSON.parse(calls[0].init.body as string);
372
+ assert.equal(cloudBody.frequency_penalty, 0.4); // the OpenAI-standard additive, not the multiplier
373
+ assert.equal("repetition_penalty" in cloudBody, false);
374
+ assert.equal("repeat_penalty" in cloudBody, false);
375
+ mock.restoreAll();
376
+ // frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
377
+ const bare = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
378
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
379
+ await bare.generate({ workerId: "r", messages: [] });
380
+ assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
381
+ });
382
+
383
+ test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
384
+ const p = new OpenAICompatProvider({ model: "managed-model", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
385
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
386
+ await p.generate({
387
+ workerId: "r",
388
+ messages: [{ role: "user", content: "hi" }],
389
+ maxTokens: 100,
390
+ sampling: {
391
+ temperature: 0.2, top_p: 0.9, top_k: 40, stop: ["\n"], // real sampling → passthrough
392
+ model: "hijack", response_format: { type: "grammar", grammar: "x" }, id_slot: 7, // reserved → stripped
393
+ },
394
+ });
395
+ const body = JSON.parse(calls[0].init.body as string);
396
+ assert.equal(body.temperature, 0.2);
397
+ assert.equal(body.top_p, 0.9);
398
+ assert.equal(body.top_k, 40);
399
+ assert.deepEqual(body.stop, ["\n"]);
400
+ assert.equal(body.model, "managed-model"); // managed field wins over a hijack attempt
401
+ assert.equal(body.max_tokens, 100);
402
+ assert.equal("response_format" in body, false); // reserved transport key stripped
403
+ assert.equal("id_slot" in body, false); // reserved slot key stripped
404
+ });
405
+
406
+ test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
407
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
408
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
409
+ await p.generate({
410
+ workerId: "r",
411
+ messages: [{ role: "user", content: "hi" }],
412
+ maxTokens: 100,
413
+ sampling: {
414
+ n: 3, // breaks choices[0] atomicity -> stripped
415
+ tools: [{ type: "function" }], tool_choice: "auto", // tools-in-body doctrine -> stripped
416
+ modalities: ["text", "audio"], prediction: { type: "content" }, // text-only / decode semantics -> stripped
417
+ max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass (#425 cap) -> stripped
418
+ seed: 42, user: "acct-7", service_tier: "flex", // platform/sampling intent -> pass
419
+ },
420
+ });
421
+ const body = JSON.parse(calls[0].init.body as string);
422
+ for (const k of ["n", "tools", "tool_choice", "modalities", "prediction", "max_completion_tokens"]) {
423
+ assert.equal(k in body, false, `${k} must be stripped`);
424
+ }
425
+ assert.equal(body.max_tokens, 100); // the managed envelope, not the smuggled 999999
426
+ assert.equal(body.seed, 42);
427
+ assert.equal(body.user, "acct-7");
428
+ assert.equal(body.service_tier, "flex");
429
+ });
430
+
431
+ test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
432
+ // The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
433
+ // reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
434
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
435
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
436
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
437
+ const body = JSON.parse(calls[0].init.body as string);
438
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true }); // channel stays sanctioned under the grammar
439
+ assert.equal(typeof body.grammar, "string"); // rails ride beside it
440
+ // #488 per-request loud state: rail attachment + verdict on meta, drill-readable per turn
441
+ assert.equal(res.meta?.railsAttached, true);
442
+ assert.equal(res.meta?.railsVerdict, "accept");
443
+ });
444
+
445
+ test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
446
+ // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
447
+ // escaped into a discarded reasoning block, unconstrained.
448
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
449
+ installFetch([
450
+ { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
451
+ { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
452
+ ]);
453
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
454
+ assert.equal(res.meta?.railsVerdict, "accept"); // the visible fragment conforms...
455
+ const escape = res.telemetry?.find((e) => e.message?.includes("escaped the grammar") === true);
456
+ assert.ok(escape, "escape telemetry attached");
457
+ assert.equal(escape!.kind, "grammar_unenforced");
458
+ assert.match(escape!.message ?? "", /5000 completion tokens billed/);
459
+ });
460
+
461
+ test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
462
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
463
+ installFetch([
464
+ { choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
465
+ { usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
466
+ ]);
467
+ const res = await p.generate({ workerId: "r", messages: [] }); // no grammar arg
468
+ assert.equal(res.meta?.railsAttached, undefined);
469
+ assert.equal(res.telemetry, undefined);
470
+ });
471
+
472
+ test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
473
+ const on = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
474
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
475
+ await on.generate({ workerId: "r", messages: [] });
476
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
477
+
478
+ mock.restoreAll();
479
+ const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
480
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
481
+ await off.generate({ workerId: "r", messages: [] });
482
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
483
+ });
484
+
485
+ test("budget 0 suppresses effort and include_reasoning", async () => {
486
+ const effort = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
487
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
488
+ await effort.generate({ workerId: "r", messages: [] });
489
+ assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
490
+
491
+ mock.restoreAll();
492
+ const relay = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
493
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
494
+ await relay.generate({ workerId: "r", messages: [] });
495
+ assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
496
+ });
497
+
498
+ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
499
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
500
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
501
+ await p.generate({ workerId: "r", messages: [] });
502
+ assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
503
+ });
504
+
505
+ // — grammar-constrained sampling (SPEC §13, issues #8/#9) —
506
+
507
+ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
508
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
509
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
510
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
511
+ const body = JSON.parse(calls[0].init.body as string);
512
+ assert.equal(body.grammar, 'root ::= "x"');
513
+ assert.equal(body.repeat_penalty, 1.15);
514
+ assert.equal("response_format" in body, false);
515
+ });
516
+
517
+ test("grammar transport 'response_format': response_format.grammar, no top-level grammar (Fireworks)", async () => {
518
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
519
+ const calls = installFetchJson(jsonChoice); // response_format grammar demotes off SSE
520
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
521
+ const body = JSON.parse(calls[0].init.body as string);
522
+ assert.deepEqual(body.response_format, { type: "grammar", grammar: 'root ::= "x"' });
523
+ assert.equal("grammar" in body, false); // not the llama.cpp shape
524
+ assert.equal("repeat_penalty" in body, false); // llama.cpp spelling not used here
525
+ assert.equal(body.repetition_penalty, 1.15); // the floor still rides (OpenAI-compat spelling, #20)
526
+ });
527
+
528
+ // A response_format grammar is the one case the spine drops streaming for, even
529
+ // with streaming on (default): fireworks mislabels the streamed grammar output
530
+ // as reasoning_content but returns it as content non-streamed (§13). The demotion
531
+ // is per-request — a grammarless call on the same provider still streams.
532
+ test("response_format grammar demotes THIS request off SSE; grammarless calls still stream", async () => {
533
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
534
+ const jsonCalls = installFetchJson(jsonChoice);
535
+ await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
536
+ assert.equal("stream" in JSON.parse(jsonCalls[0].init.body as string), false); // no SSE flag
537
+ mock.restoreAll();
538
+ const sseCalls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
539
+ await p.generate({ workerId: "r", messages: [] }); // no grammar → streams
540
+ assert.equal(JSON.parse(sseCalls[0].init.body as string).stream, true); // SSE flag present
541
+ });
542
+
543
+ test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
544
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
545
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
546
+ await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
547
+ const body = JSON.parse(calls[0].init.body as string);
548
+ assert.equal("grammar" in body, false);
549
+ assert.equal("response_format" in body, false);
550
+ });
551
+
552
+ // — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
553
+ // returns; bytes flow; a non-accept verdict rides response.telemetry —
554
+
555
+ const grammarProvider = () => new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
556
+ const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
557
+
558
+ test("enforcement: conforming output passes through unchanged", async () => {
559
+ const p = grammarProvider();
560
+ streamingContent("ok");
561
+ const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
562
+ assert.equal(assistant.content, "ok");
563
+ });
564
+
565
+ test("observation: REJECTED output still returns — bytes present, verdict attached with position", async () => {
566
+ const p = grammarProvider();
567
+ streamingContent("no");
568
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
569
+ assert.equal(res.assistant.content, "no"); // bytes ALWAYS flow
570
+ assert.equal(res.telemetry?.length, 1);
571
+ const ev = res.telemetry![0];
572
+ assert.equal(ev.kind, "grammar_unenforced");
573
+ assert.equal(ev.source, "provider:test");
574
+ assert.match(String(ev.message), /grammar not enforced: output rejected .* at code point 0/);
575
+ assert.equal(ev.position, 0); // divergence offset for consumer policy
576
+ });
577
+
578
+ test("observation: an incomplete (valid prefix, never terminated) also returns with the verdict", async () => {
579
+ const p = grammarProvider();
580
+ streamingContent("ok");
581
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok" "!"' });
582
+ assert.equal(res.assistant.content, "ok");
583
+ assert.equal(res.telemetry?.length, 1);
584
+ assert.match(String(res.telemetry![0].message), /incomplete match .* never terminated/);
585
+ assert.equal(res.telemetry![0].position, 2);
586
+ });
587
+
588
+ test("observation: conforming output attaches NO telemetry", async () => {
589
+ const p = grammarProvider();
590
+ streamingContent("ok");
591
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
592
+ assert.equal(res.telemetry, undefined);
593
+ });
594
+
595
+ test("observation: empty content under a non-empty grammar returns with the verdict (the 'content never arrives' leak, observed)", async () => {
596
+ const p = grammarProvider();
597
+ installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]); // no content delta → ""
598
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
599
+ assert.equal(res.assistant.content, "");
600
+ assert.equal(res.telemetry?.[0].kind, "grammar_unenforced");
601
+ });
602
+
603
+ test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
604
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
605
+ streamingContent("anything goes");
606
+ const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
607
+ assert.equal(assistant.content, "anything goes"); // no enforcement check
608
+ });
609
+
610
+ test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap — warn, return content", async () => {
611
+ const p = grammarProvider();
612
+ streamingContent("whatever");
613
+ const warnings: Error[] = [];
614
+ const onWarn = (w: Error) => warnings.push(w);
615
+ process.on("warning", onWarn);
616
+ const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }); // no `root` rule → validateGbnf throws
617
+ await flush();
618
+ process.off("warning", onWarn);
619
+ assert.equal(assistant.content, "whatever"); // transport not failed
620
+ assert.ok(warnings.some((w) => (w as Error & { code?: string }).code === "PLURNK_GRAMMAR_UNVERIFIABLE"), "emitted the verify-gap warning");
621
+ });
622
+
623
+ // — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
624
+
625
+ test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
626
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
627
+ const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
628
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
629
+ const body = JSON.parse(calls[0].init.body as string);
630
+ assert.equal("grammar" in body, false); // grammar never sent — model ran unconstrained
631
+ assert.equal(body.repeat_penalty, 1.15); // #426: penalty rides even rail-off - unconstrained decode needs it MORE
632
+ assert.equal(res.assistant.content, "ok"); // free output happens to conform → returned
633
+ assert.equal("telemetry" in res, false); // conforming → no event
634
+ });
635
+
636
+ test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
637
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
638
+ const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
639
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
640
+ // The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
641
+ assert.equal(res.assistant.content, "xon-conforming output");
642
+ assert.equal(res.assistant.reasoning, "let me think about ok");
643
+ // Non-fatal telemetry carries the divergence so the consumer can self-correct.
644
+ assert.equal(res.telemetry?.length, 1);
645
+ const [event] = res.telemetry ?? [];
646
+ assert.equal(event.source, "provider:test");
647
+ assert.equal(event.kind, "grammar_unenforced");
648
+ assert.equal(event.position, 0); // 'x' rejected at code point 0
649
+ assert.match(event.message ?? "", /output rejected by the transported grammar at code point 0 \("x"\)/);
650
+ const body = JSON.parse(calls[0].init.body as string);
651
+ assert.equal("grammar" in body, false); // still never sent — diagnosed, not enforced
652
+ });
653
+
654
+ test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
655
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
656
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
657
+ await assert.rejects(
658
+ () => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
659
+ /grammar validation \(PLURNK_PROVIDERS_GBNF_DEBUG\): invalid GBNF/,
660
+ );
661
+ assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
662
+ });
663
+
664
+ // — meta bag: pass-through extras + validated known keys (#23) —
665
+
666
+ test("meta: the spec's balance field is normalized to a validated meta.balancePico", async () => {
667
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
668
+ installFetchJson({ ...jsonChoice, balance_pico: 4_200_000 });
669
+ const res = await p.generate({ workerId: "r", messages: [] });
670
+ assert.equal(res.meta?.balancePico, 4_200_000);
671
+ assert.equal("balance_pico" in (res.meta ?? {}), false); // raw key renamed to the canonical balancePico
672
+ });
673
+
674
+ test("meta: passes the backend's extra top-level fields through verbatim (every provider)", async () => {
675
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false }); // no balanceMetaKey
676
+ installFetchJson({ ...jsonChoice, balance_pico: 4_200_000, system_fingerprint: "fp_abc" });
677
+ const res = await p.generate({ workerId: "r", messages: [] });
678
+ assert.equal(res.meta?.balance_pico, 4_200_000); // passed through raw — no balance contract on this provider
679
+ assert.equal(res.meta?.system_fingerprint, "fp_abc");
680
+ assert.equal("balancePico" in (res.meta ?? {}), false); // not normalized without the key
681
+ });
682
+
683
+ test("meta: a non-numeric balance is dropped, never surfaced as balancePico (null-honest)", async () => {
684
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
685
+ installFetchJson({ ...jsonChoice, balance_pico: "lots" });
686
+ const res = await p.generate({ workerId: "r", messages: [] });
687
+ assert.equal("balancePico" in (res.meta ?? {}), false);
688
+ assert.equal("balance_pico" in (res.meta ?? {}), false); // raw dropped too — the known key is validated away
689
+ });
690
+
691
+ // — first-party telemetry headers (attribution + client, SPEC §5) —
692
+
693
+ const headerVal = (init: RequestInit, name: string): string | undefined =>
694
+ new Headers(init.headers).get(name) ?? undefined;
695
+
696
+ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
697
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
698
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
699
+ await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
700
+ assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
701
+ assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
702
+ });
703
+
704
+ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
705
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
706
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
707
+ await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
708
+ assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
709
+ mock.restoreAll();
710
+
711
+ // the primary worker's own turn: Primary == Worker-Id, still stamped (never skipped on equality)
712
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
713
+ await p.generate({ workerId: "w-root", primaryWorkerId: "w-root", messages: [] });
714
+ assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root");
715
+ mock.restoreAll();
716
+
717
+ // absent when the consumer supplies none — the provider never invents a primary
718
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
719
+ await p.generate({ workerId: "w-root", messages: [] });
720
+ assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined);
721
+ });
722
+
723
+ test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
724
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
725
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
726
+ await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
727
+ assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
728
+ });
729
+
730
+ test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
731
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
732
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
733
+ await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
734
+ assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
735
+ assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
736
+ });
737
+
738
+ test("firstPartyMetadata on but empty values: no header emitted", async () => {
739
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
740
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
741
+ await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
742
+ assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
743
+ assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
744
+ });
745
+
746
+ test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
747
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
748
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
749
+ await p.generate({ workerId: "r", messages: [] });
750
+ const body = JSON.parse(calls[0].init.body as string);
751
+ assert.equal("grammar" in body, false);
752
+ assert.equal(body.repeat_penalty, 1.15); // #426: penalty is no longer grammar-gated - it rides rail-off
753
+ });
754
+
755
+ test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
756
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
757
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
758
+ await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
759
+ assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
760
+
761
+ mock.restoreAll();
762
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
763
+ await p.generate({ workerId: "r", messages: [] });
764
+ assert.equal("max_tokens" in JSON.parse(calls[0].init.body as string), false);
765
+ });
766
+
767
+ test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
768
+ const pinning = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
769
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
770
+ await pinning.generate({ workerId: "run-A", messages: [] });
771
+ await pinning.generate({ workerId: "run-B", messages: [] });
772
+ await pinning.generate({ workerId: "run-A", messages: [] }); // sticky
773
+ await pinning.generate({ workerId: "run-C", messages: [] }); // wraps round-robin
774
+ const slots = calls.map((c) => JSON.parse(c.init.body as string).id_slot);
775
+ assert.deepEqual(slots, [0, 1, 0, 0]);
776
+ });
777
+
778
+ test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
779
+ const cloud = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
780
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
781
+ await cloud.generate({ workerId: "run-A", messages: [] });
782
+ assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
783
+
784
+ mock.restoreAll();
785
+ const noCount = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
786
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
787
+ await noCount.generate({ workerId: "run-A", messages: [] });
788
+ assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
789
+ });
790
+
791
+ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
792
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
793
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
794
+ const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
795
+ for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
796
+ await p.generate({ workerId: "r16", messages: [] }); // call 16: size==cap → evicts the oldest (r0), itself → slot 0
797
+ await p.generate({ workerId: "r0", messages: [] }); // call 17: r0 was evicted → treated as NEW, re-slotted
798
+ await p.generate({ workerId: "r16", messages: [] }); // call 18: r16 still resident → sticky to its slot
799
+ assert.equal(slotOf(0), 0); // r0's original pin
800
+ assert.notEqual(slotOf(17), slotOf(0)); // …lost after eviction (would equal 0 if it had stayed sticky)
801
+ assert.equal(slotOf(18), slotOf(16)); // r16 kept its slot — recent run survives the window
802
+ });
803
+
804
+ test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
805
+ const { ProviderError } = await import("./telemetry.ts");
806
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
807
+ mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
808
+ await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
809
+ assert.ok(err instanceof ProviderError);
810
+ assert.equal(err.kind, "network_failure"); // ≥500 → network_failure
811
+ assert.equal(err.status, 500);
812
+ return true;
813
+ });
814
+ });
815
+
816
+ test("generate fail-hards on a missing or empty workerId", async () => {
817
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
818
+ installFetch([{ choices: [{ delta: { content: "x" } }] }]);
819
+ await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
820
+ await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
821
+ });
822
+
823
+ test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
824
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
825
+ const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
826
+ const input = [{ role: "user" as const, content: "hi" }];
827
+ const res = await p.generate({ workerId: "r", messages: input });
828
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).messages, input); // no extra assistant turn
829
+ assert.equal(res.assistant.content, "out"); // content returned verbatim
830
+ });
831
+
832
+ test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
833
+ const { ProviderError } = await import("./telemetry.ts");
834
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
835
+ mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
836
+ await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
837
+ assert.ok(err instanceof ProviderError);
838
+ assert.equal(err.kind, "rate_limit");
839
+ assert.equal(err.status, 429);
840
+ assert.deepEqual(err.toTelemetryEvent(), { source: "provider:test", kind: "rate_limit", message: err.message, position: null });
841
+ return true;
842
+ });
843
+ });
844
+
845
+ test("generate rejects on a pre-aborted external signal", async () => {
846
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
847
+ installFetch([{ choices: [{ delta: { content: "x" } }] }]);
848
+ const signal = AbortSignal.abort(new Error("nope"));
849
+ await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
850
+ });
851
+
852
+ test("configured headers and url are sent verbatim", async () => {
853
+ const p = new OpenAICompatProvider({
854
+ model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
855
+ headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
856
+ });
857
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
858
+ await p.generate({ workerId: "r", messages: [] });
859
+ assert.equal(calls[0].url, "http://host/custom/chat/completions");
860
+ assert.equal((calls[0].init.headers as Record<string, string>).Authorization, "Bearer secret");
861
+ assert.equal((calls[0].init.headers as Record<string, string>)["X-Title"], "plurnk");
862
+ });
863
+
864
+ // — transient-failure retry (#18) —
865
+
866
+ const retryCfg = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const };
867
+
868
+ test("retry: a transient failure retries and a later success resolves", async () => {
869
+ const calls = installFetchScript([
870
+ { status: 429, retryAfter: 0 },
871
+ { status: 503, retryAfter: 0 },
872
+ { status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
873
+ ]);
874
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
875
+ const res = await p.generate({ workerId: "r", messages: [] });
876
+ assert.equal(res.assistant.content, "ok");
877
+ assert.equal(calls.length, 3); // 429 → 503 → 200
878
+ });
879
+
880
+ test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
881
+ const { ProviderError } = await import("./telemetry.ts");
882
+ const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
883
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 2 });
884
+ await assert.rejects(
885
+ () => p.generate({ workerId: "r", messages: [] }),
886
+ (err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
887
+ );
888
+ assert.equal(calls.length, 3); // 1 initial + 2 retries
889
+ });
890
+
891
+ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms wait, then retries", async () => {
892
+ const calls = installFetchScript([
893
+ { status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
894
+ { status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
895
+ ]);
896
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 1 });
897
+ const { assistant } = await p.generate({ workerId: "r", messages: [] });
898
+ assert.equal(assistant.content, "ok");
899
+ assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
900
+ });
901
+
902
+ test("retry: a terminal error (401 unauthorized) is never retried", async () => {
903
+ const calls = installFetchScript([{ status: 401 }]);
904
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 5 });
905
+ await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
906
+ assert.equal(calls.length, 1); // terminal — no retry despite budget
907
+ });
908
+
909
+ test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
910
+ const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
911
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 0 });
912
+ await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
913
+ assert.equal(calls.length, 1); // no retry budget
914
+ });
915
+
916
+ test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
917
+ const ac = new AbortController();
918
+ const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
919
+ const p = new OpenAICompatProvider({ ...retryCfg, retryAttempts: 3 });
920
+ const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
921
+ await flush(); // attempt 0 fails, enters the backoff sleep
922
+ assert.equal(calls.length, 1);
923
+ ac.abort(new Error("cancelled"));
924
+ await assert.rejects(() => promise); // abort cuts through the backoff
925
+ assert.equal(calls.length, 1); // never retried after cancellation
926
+ });
927
+
928
+ // — anthropic reasoning style (thinking param, #18) —
929
+
930
+ test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
931
+ // N>0 → enabled with budget_tokens
932
+ const capped = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
933
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
934
+ await capped.generate({ workerId: "r", messages: [] });
935
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
936
+
937
+ mock.restoreAll();
938
+ // 0 → explicit disabled
939
+ const off = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
940
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
941
+ await off.generate({ workerId: "r", messages: [] });
942
+ assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
943
+
944
+ mock.restoreAll();
945
+ // -1 adaptive → omit (API default depth)
946
+ const adaptive = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
947
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
948
+ await adaptive.generate({ workerId: "r", messages: [] });
949
+ assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
950
+ });
951
+
952
+ // — non-streaming transport (streaming:false) —
953
+
954
+ test("streaming:false posts without stream and parses the single JSON response", async () => {
955
+ const calls: { body: string }[] = [];
956
+ mock.method(globalThis, "fetch", async (_url: string, init: RequestInit) => {
957
+ calls.push({ body: String(init.body) });
958
+ return new Response(JSON.stringify({
959
+ model: "wire-model",
960
+ choices: [{ message: { content: "hello", reasoning_content: "because" }, finish_reason: "stop" }],
961
+ usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
962
+ }), { status: 200, headers: { "Content-Type": "application/json" } });
963
+ });
964
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
965
+ const res = await p.generate({ workerId: "r", messages: [] });
966
+ const sent = JSON.parse(calls[0].body);
967
+ assert.equal("stream" in sent, false); // no streaming flag
968
+ assert.equal(res.assistant.content, "hello"); // content from message.content
969
+ assert.equal(res.assistant.reasoning, "because"); // reasoning_content mapped
970
+ assert.equal(res.assistant.finishReason, "stop");
971
+ assert.equal(res.assistant.usage.total, 4);
972
+ mock.restoreAll();
973
+ });
974
+
975
+ // ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
976
+ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
977
+
978
+ test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
979
+ const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
980
+ const p = new OpenAICompatProvider({ ...captureBase });
981
+ const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
982
+ const body = JSON.parse((calls[0].init.body as string));
983
+ assert.equal("logprobs" in body, false);
984
+ assert.equal("top_logprobs" in body, false);
985
+ assert.equal(res.assistant.logprobs, undefined);
986
+ assert.equal(res.assistant.meanLogprob, undefined);
987
+ assert.equal(res.rawBody, undefined);
988
+ mock.restoreAll();
989
+ });
990
+
991
+ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
992
+ const chunk = { model: "m", usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 }, choices: [{ delta: { content: "yesno" }, finish_reason: "stop", logprobs: { content: [
993
+ { token: "yes", logprob: -0.5, sampling_logprob: -0.5, top_logprobs: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] },
994
+ { token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
995
+ ] } }] };
996
+ const calls = installFetch([chunk]);
997
+ const p = new OpenAICompatProvider({ ...captureBase, topLogprobs: 2 });
998
+ const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
999
+ const body = JSON.parse((calls[0].init.body as string));
1000
+ assert.equal(body.logprobs, true);
1001
+ assert.equal(body.top_logprobs, 2);
1002
+ assert.equal(res.assistant.logprobs?.length, 2);
1003
+ assert.deepEqual(res.assistant.logprobs?.[0], { token: "yes", logprob: -0.5, top: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] });
1004
+ assert.equal(res.assistant.meanLogprob, -0.3); // (-0.5 + -0.1) / 2
1005
+ mock.restoreAll();
1006
+ });
1007
+
1008
+ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
1009
+ const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
1010
+ installFetchJson(wire);
1011
+ const p = new OpenAICompatProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
1012
+ const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
1013
+ assert.deepEqual(res.rawBody, wire); // verbatim
1014
+ assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
1015
+ assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].token_id, 42);
1016
+ assert.equal(res.assistant.logprobs?.[0].token, "no"); // structured view still uses raw logprob
1017
+ mock.restoreAll();
1018
+ });
1019
+
1020
+ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
1021
+ const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
1022
+ const p = new OpenAICompatProvider({ ...captureBase }); // logprobs OFF
1023
+ await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
1024
+ const body = JSON.parse((calls[0].init.body as string));
1025
+ assert.equal("logprobs" in body, false); // sampling passthrough stripped it
1026
+ assert.equal("top_logprobs" in body, false);
1027
+ mock.restoreAll();
1028
+ });
1029
+
1030
+ // — turn coordinate headers (#404, per #391): same gate as every first-party signal —
1031
+
1032
+ test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
1033
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1034
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1035
+ await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1036
+ const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1037
+ assert.equal(h["Plurnk-Workspace-Id"], "s-9");
1038
+ assert.equal(h["Plurnk-Loop"], "3");
1039
+ assert.equal(h["Plurnk-Turn"], "41");
1040
+ });
1041
+
1042
+ test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
1043
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1044
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1045
+ await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
1046
+ const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1047
+ assert.equal("Plurnk-Workspace-Id" in h, false);
1048
+ assert.equal("Plurnk-Loop" in h, false);
1049
+ assert.equal("Plurnk-Turn" in h, false);
1050
+ });
1051
+
1052
+ test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
1053
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
1054
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1055
+ await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
1056
+ const h = (calls[0].init.headers ?? {}) as Record<string, string>;
1057
+ assert.equal("Plurnk-Workspace-Id" in h, false);
1058
+ assert.equal("Plurnk-Loop" in h, false);
1059
+ assert.equal("Plurnk-Turn" in h, false);
1060
+ assert.equal(typeof h["Plurnk-Strikes"], "undefined"); // and absent strikes stays absent
1061
+ });
1062
+
1063
+ // -- #507: envelope surface + router-owned tuning --
1064
+
1065
+ test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
1066
+ const base = { model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
1067
+ const derived = new OpenAICompatProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
1068
+ assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
1069
+ assert.equal(derived.completionReserve, 12288); // 25% of 49152
1070
+ const pinned = new OpenAICompatProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
1071
+ assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
1072
+ assert.equal(pinned.completionReserve, null); // percent without a window = underivable
1073
+ const legacy = new OpenAICompatProvider({ ...base, contextWindow: 49152 });
1074
+ assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
1075
+ });
1076
+
1077
+ test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
1078
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
1079
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1080
+ await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
1081
+ const body = JSON.parse(calls[0].init.body as string);
1082
+ assert.equal(body.temperature, 0.9); // caller intent passes verbatim
1083
+ assert.equal("frequency_penalty" in body, false); // the floor is suppressed (router owns tuning, SPEC §5)
1084
+ });
1085
+
1086
+ // -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
1087
+
1088
+ test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
1089
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1090
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1091
+ await p.generate({ workerId: "worker-abc", messages: [] });
1092
+ assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
1093
+ });
1094
+
1095
+ test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
1096
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
1097
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1098
+ await p.generate({ workerId: "worker-abc", messages: [] });
1099
+ assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
1100
+ });
1101
+
1102
+ test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
1103
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
1104
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1105
+ await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
1106
+ assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins
1107
+ });