@plurnk/plurnk-providers 1.3.5 → 1.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +35 -45
- package/README.md +44 -53
- package/SPEC.md +215 -371
- package/dist/AiSdkProvider.d.ts +78 -0
- package/dist/AiSdkProvider.d.ts.map +1 -0
- package/dist/AiSdkProvider.js +591 -0
- package/dist/AiSdkProvider.js.map +1 -0
- package/dist/OpenAICompat.d.ts +1 -2
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +39 -117
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +37 -24
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +52 -0
- package/dist/aiSdkTransport.d.ts.map +1 -0
- package/dist/aiSdkTransport.js +294 -0
- package/dist/aiSdkTransport.js.map +1 -0
- package/dist/catalogProvider.d.ts +15 -0
- package/dist/catalogProvider.d.ts.map +1 -0
- package/dist/catalogProvider.js +103 -0
- package/dist/catalogProvider.js.map +1 -0
- package/dist/compatibleProvider.d.ts +3 -0
- package/dist/compatibleProvider.d.ts.map +1 -0
- package/dist/compatibleProvider.js +146 -0
- package/dist/compatibleProvider.js.map +1 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +1 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +13 -6
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +4 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -7
- package/dist/index.js.map +1 -1
- package/dist/ollama.d.ts +3 -0
- package/dist/ollama.d.ts.map +1 -0
- package/dist/ollama.js +39 -0
- package/dist/ollama.js.map +1 -0
- package/dist/openai.d.ts +2 -4
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -2
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -0
- package/dist/sdkModels.d.ts.map +1 -0
- package/dist/sdkModels.js +153 -0
- package/dist/sdkModels.js.map +1 -0
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +0 -1
- package/dist/standardProviders.js.map +1 -1
- package/dist/telemetry.d.ts.map +1 -1
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/types.d.ts +3 -2
- package/dist/types.d.ts.map +1 -1
- package/package.json +18 -10
- package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
- package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
- package/src/Mock.test.ts +1 -1
- package/src/ProviderRegistry.test.ts +40 -27
- package/src/ProviderRegistry.ts +35 -24
- package/src/aiSdkTransport.test.ts +253 -0
- package/src/aiSdkTransport.ts +369 -0
- package/src/boundaries.test.ts +2 -2
- package/src/catalogProvider.test.ts +100 -0
- package/src/catalogProvider.ts +151 -0
- package/src/compatibleProvider.test.ts +44 -0
- package/src/compatibleProvider.ts +205 -0
- package/src/discover.test.ts +12 -12
- package/src/discover.ts +3 -6
- package/src/env.ts +14 -6
- package/src/index.ts +6 -10
- package/src/ollama.ts +63 -0
- package/src/openai.ts +2 -8
- package/src/sdkModels.test.ts +47 -0
- package/src/sdkModels.ts +194 -0
- package/src/telemetry.test.ts +17 -10
- package/src/telemetry.ts +22 -14
- package/src/types.ts +5 -8
- package/src/aiSdkAdapter.spike.test.ts +0 -242
- package/src/openaiStream.ts +0 -310
- package/src/standardProviders.test.ts +0 -939
- package/src/standardProviders.ts +0 -631
|
@@ -1,13 +1,37 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import
|
|
4
|
-
import { OpenAiHttpError } from "./openaiStream.ts";
|
|
3
|
+
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
5
4
|
import { ProviderError } from "./telemetry.ts";
|
|
6
5
|
|
|
7
6
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
8
7
|
// so tests can assert what the spine sent on the wire.
|
|
9
8
|
const sseStream = (chunks: unknown[]) => {
|
|
10
|
-
const
|
|
9
|
+
const normalized = chunks.map((value, index) => {
|
|
10
|
+
const chunk = value as Record<string, any>;
|
|
11
|
+
const usage = chunk.usage !== undefined
|
|
12
|
+
? {
|
|
13
|
+
...chunk.usage,
|
|
14
|
+
...(chunk.usage.cached_tokens !== undefined
|
|
15
|
+
? {
|
|
16
|
+
prompt_tokens_details: {
|
|
17
|
+
cached_tokens: chunk.usage.cached_tokens,
|
|
18
|
+
},
|
|
19
|
+
}
|
|
20
|
+
: {}),
|
|
21
|
+
}
|
|
22
|
+
: undefined;
|
|
23
|
+
if (usage !== undefined) delete usage.cached_tokens;
|
|
24
|
+
return {
|
|
25
|
+
id: "test-completion",
|
|
26
|
+
object: "chat.completion.chunk",
|
|
27
|
+
created: index + 1,
|
|
28
|
+
model: "m",
|
|
29
|
+
...chunk,
|
|
30
|
+
...(chunk.choices === undefined && usage !== undefined ? { choices: [] } : {}),
|
|
31
|
+
...(usage !== undefined ? { usage } : {}),
|
|
32
|
+
};
|
|
33
|
+
});
|
|
34
|
+
const lines = [...normalized.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
|
|
11
35
|
return new ReadableStream({
|
|
12
36
|
start(controller) {
|
|
13
37
|
controller.enqueue(new TextEncoder().encode(lines));
|
|
@@ -43,7 +67,6 @@ const injectedBase = {
|
|
|
43
67
|
fetchTimeoutMs: 5000,
|
|
44
68
|
temperature: 0.2,
|
|
45
69
|
repeatPenalty: 1.15,
|
|
46
|
-
retryDelayMs: 1,
|
|
47
70
|
retryAttempts: 0,
|
|
48
71
|
reasoning: { mode: "off" as const, budget: null },
|
|
49
72
|
};
|
|
@@ -65,9 +88,9 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
65
88
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
66
89
|
};
|
|
67
90
|
|
|
68
|
-
const streamed = await new
|
|
91
|
+
const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
|
|
69
92
|
.generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
|
|
70
|
-
const buffered = await new
|
|
93
|
+
const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
|
|
71
94
|
.generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
|
|
72
95
|
|
|
73
96
|
assert.equal(streamed.assistant.content, "streamed");
|
|
@@ -83,18 +106,20 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
83
106
|
});
|
|
84
107
|
|
|
85
108
|
test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
|
|
86
|
-
const pendingFetch: typeof globalThis.fetch = async (_input, init) =>
|
|
87
|
-
|
|
109
|
+
const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
|
|
110
|
+
init?.signal?.throwIfAborted();
|
|
111
|
+
return new Promise((_resolve, reject) => {
|
|
88
112
|
const signal = init?.signal;
|
|
89
113
|
signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
90
114
|
});
|
|
115
|
+
};
|
|
91
116
|
const caller = new AbortController();
|
|
92
|
-
const callerProvider = new
|
|
117
|
+
const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
|
|
93
118
|
const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
|
|
94
119
|
caller.abort(new Error("operator cancelled"));
|
|
95
120
|
await assert.rejects(callerRequest, /operator cancelled/);
|
|
96
121
|
|
|
97
|
-
const timeoutProvider = new
|
|
122
|
+
const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
|
|
98
123
|
await assert.rejects(
|
|
99
124
|
timeoutProvider.generate({ workerId: "timeout", messages: [] }),
|
|
100
125
|
(error: ProviderError) => error.kind === "network_failure",
|
|
@@ -116,7 +141,7 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
|
|
|
116
141
|
{ choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
|
|
117
142
|
]), { status: 200 });
|
|
118
143
|
};
|
|
119
|
-
const provider = new
|
|
144
|
+
const provider = new AiSdkProvider({
|
|
120
145
|
...injectedBase,
|
|
121
146
|
fetch: providerFetch,
|
|
122
147
|
retryAttempts: 1,
|
|
@@ -135,7 +160,13 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
|
|
|
135
160
|
// Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
|
|
136
161
|
// streams its chunks; any other status returns that error (with an optional
|
|
137
162
|
// retry-after header). The last entry repeats once the script runs out.
|
|
138
|
-
type ScriptedResponse = {
|
|
163
|
+
type ScriptedResponse = {
|
|
164
|
+
status: number;
|
|
165
|
+
chunks?: unknown[];
|
|
166
|
+
retryAfter?: number | string;
|
|
167
|
+
shouldRetry?: boolean;
|
|
168
|
+
body?: string;
|
|
169
|
+
};
|
|
139
170
|
const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
140
171
|
const calls: { url: string; init: RequestInit }[] = [];
|
|
141
172
|
let i = 0;
|
|
@@ -144,8 +175,15 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
|
144
175
|
const r = responses[Math.min(i, responses.length - 1)];
|
|
145
176
|
i++;
|
|
146
177
|
if (r.status === 200) return new Response(sseStream(r.chunks ?? []), { status: 200 });
|
|
147
|
-
const headers =
|
|
148
|
-
|
|
178
|
+
const headers = {
|
|
179
|
+
"content-type": "application/json",
|
|
180
|
+
...(r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {}),
|
|
181
|
+
...(r.shouldRetry !== undefined ? { "x-should-retry": String(r.shouldRetry) } : {}),
|
|
182
|
+
};
|
|
183
|
+
return new Response(
|
|
184
|
+
r.body ?? JSON.stringify({ error: { message: `HTTP ${r.status}` } }),
|
|
185
|
+
{ status: r.status, headers },
|
|
186
|
+
);
|
|
149
187
|
});
|
|
150
188
|
return calls;
|
|
151
189
|
};
|
|
@@ -164,33 +202,25 @@ test("effortFromBudget: maps budget to tiers", () => {
|
|
|
164
202
|
assert.equal(effortFromBudget(4001), "high");
|
|
165
203
|
});
|
|
166
204
|
|
|
167
|
-
test("#543: OpenAiHttpError distills a non-JSON (edge/CDN HTML) body and drops the OpenAI prefix", () => {
|
|
168
|
-
const cf = new OpenAiHttpError(524, "<!DOCTYPE html><html><body>Error code 524</body></html>", null);
|
|
169
|
-
assert.equal(cf.message, "524 origin timeout"); // distilled: no raw HTML, no "OpenAI" prefix
|
|
170
|
-
assert.ok(cf.body.length > 20); // raw body retained on the field for forensics
|
|
171
|
-
const api = new OpenAiHttpError(400, '{"error":{"message":"bad param"}}', null);
|
|
172
|
-
assert.match(api.message, /^OpenAI 400 - \{/); // JSON API error passes through verbatim
|
|
173
|
-
});
|
|
174
|
-
|
|
175
205
|
test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
176
206
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
177
|
-
const p = new
|
|
207
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
178
208
|
await assert.rejects(p.generate({ workerId: "r", messages: [] }));
|
|
179
209
|
await flush();
|
|
180
210
|
assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
|
|
181
211
|
mock.restoreAll();
|
|
182
212
|
});
|
|
183
213
|
|
|
184
|
-
test("#548: a 422 grammar_invalid is
|
|
214
|
+
test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
|
|
185
215
|
const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
|
|
186
216
|
const calls = installFetchScript([{ status: 422, body }]);
|
|
187
|
-
const p = new
|
|
217
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
|
|
188
218
|
await assert.rejects(
|
|
189
219
|
p.generate({ workerId: "r", messages: [] }),
|
|
190
220
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
191
221
|
);
|
|
192
222
|
await flush();
|
|
193
|
-
assert.equal(calls.length,
|
|
223
|
+
assert.equal(calls.length, 1);
|
|
194
224
|
mock.restoreAll();
|
|
195
225
|
});
|
|
196
226
|
|
|
@@ -199,7 +229,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
199
229
|
status: 422,
|
|
200
230
|
error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
|
|
201
231
|
}]);
|
|
202
|
-
const p = new
|
|
232
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
203
233
|
await assert.rejects(
|
|
204
234
|
p.generate({ workerId: "r", messages: [] }),
|
|
205
235
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
@@ -209,27 +239,27 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
209
239
|
|
|
210
240
|
test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
|
|
211
241
|
installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
212
|
-
const p = new
|
|
242
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
213
243
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
214
244
|
assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
|
|
215
245
|
});
|
|
216
246
|
|
|
217
247
|
test("#539: without a probed eos_token the content passes through untouched", async () => {
|
|
218
248
|
installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
219
|
-
const p = new
|
|
249
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
220
250
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
221
251
|
assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
|
|
222
252
|
});
|
|
223
253
|
|
|
224
254
|
test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
|
|
225
255
|
installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
|
|
226
|
-
const p = new
|
|
256
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
227
257
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
228
258
|
assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
|
|
229
259
|
});
|
|
230
260
|
|
|
231
261
|
test("identity getters and defaults", () => {
|
|
232
|
-
const p = new
|
|
262
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
233
263
|
assert.equal(p.model, "m");
|
|
234
264
|
assert.equal(p.contextWindow, null); // default
|
|
235
265
|
assert.equal(p.countTokens(""), 0);
|
|
@@ -238,8 +268,8 @@ test("identity getters and defaults", () => {
|
|
|
238
268
|
});
|
|
239
269
|
|
|
240
270
|
test("injected countTokens and calculateCost are used", () => {
|
|
241
|
-
const p = new
|
|
242
|
-
model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
271
|
+
const p = new AiSdkProvider({
|
|
272
|
+
model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
243
273
|
countTokens: (t) => t.length,
|
|
244
274
|
calculateCost: (u) => u.total * 2,
|
|
245
275
|
});
|
|
@@ -248,7 +278,7 @@ test("injected countTokens and calculateCost are used", () => {
|
|
|
248
278
|
});
|
|
249
279
|
|
|
250
280
|
test("generate maps a streamed response into ProviderResponse", async () => {
|
|
251
|
-
const p = new
|
|
281
|
+
const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
252
282
|
installFetch([
|
|
253
283
|
{ model: "wire-model", choices: [{ delta: { content: "hel" } }] },
|
|
254
284
|
{ choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
|
|
@@ -263,31 +293,42 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
263
293
|
assert.notEqual(assistantRaw, undefined);
|
|
264
294
|
});
|
|
265
295
|
|
|
266
|
-
test("generate normalizes an out-of-set finish_reason
|
|
267
|
-
const
|
|
296
|
+
test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
|
|
297
|
+
const warnings: Array<{ message: string; code?: string }> = [];
|
|
298
|
+
mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
|
|
299
|
+
warnings.push({
|
|
300
|
+
message: String(message),
|
|
301
|
+
...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
|
|
302
|
+
});
|
|
303
|
+
});
|
|
304
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
268
305
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
|
|
269
306
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
270
307
|
assert.equal(assistant.finishReason, null);
|
|
308
|
+
assert.deepEqual(warnings, [{
|
|
309
|
+
message: 'unrecognized finish_reason "function_call"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core\'s length-cap detection will miss it.',
|
|
310
|
+
code: "PLURNK_FINISH_REASON_UNKNOWN",
|
|
311
|
+
}]);
|
|
271
312
|
});
|
|
272
313
|
|
|
273
314
|
test("generate translates a backend cap synonym to canonical length (#425)", async () => {
|
|
274
315
|
// gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
|
|
275
316
|
// "length" so its truncation check (=== "length") is a cross-backend invariant.
|
|
276
|
-
const p = new
|
|
317
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
277
318
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
|
|
278
319
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
279
320
|
assert.equal(assistant.finishReason, "length");
|
|
280
321
|
});
|
|
281
322
|
|
|
282
323
|
test("generate translates end_turn to canonical stop (#425)", async () => {
|
|
283
|
-
const p = new
|
|
324
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
284
325
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
|
|
285
326
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
286
327
|
assert.equal(assistant.finishReason, "stop");
|
|
287
328
|
});
|
|
288
329
|
|
|
289
330
|
test("generate aggregates reasoning deltas under multiple field names", async () => {
|
|
290
|
-
const p = new
|
|
331
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
291
332
|
installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
|
|
292
333
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
293
334
|
assert.equal(assistant.reasoning, "because");
|
|
@@ -303,7 +344,7 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
|
|
|
303
344
|
{ type: "reasoning.text", text: "never surfaced here" },
|
|
304
345
|
],
|
|
305
346
|
}, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
306
|
-
const p = new
|
|
347
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
307
348
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
308
349
|
// item shape: wire `id` preserved, subtype from position (#482 widening)
|
|
309
350
|
assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
|
|
@@ -316,14 +357,14 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
|
|
|
316
357
|
{ type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
|
|
317
358
|
{ type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
|
|
318
359
|
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
319
|
-
const p = new
|
|
360
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
320
361
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
321
362
|
assert.equal(assistant.reasoningEncrypted?.length, 2);
|
|
322
363
|
assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
|
|
323
364
|
});
|
|
324
365
|
|
|
325
366
|
test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
|
|
326
|
-
const p = new
|
|
367
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
327
368
|
installFetch([
|
|
328
369
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
|
|
329
370
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
|
|
@@ -335,20 +376,20 @@ test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entr
|
|
|
335
376
|
});
|
|
336
377
|
|
|
337
378
|
test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
|
|
338
|
-
const on = new
|
|
379
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
339
380
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
340
381
|
await on.generate({ workerId: "r", messages: [] });
|
|
341
382
|
assert.equal(JSON.parse(calls[0].init.body as string).think, true);
|
|
342
383
|
|
|
343
384
|
mock.restoreAll();
|
|
344
|
-
const off = new
|
|
385
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
345
386
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
346
387
|
await off.generate({ workerId: "r", messages: [] });
|
|
347
388
|
assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
|
|
348
389
|
});
|
|
349
390
|
|
|
350
391
|
test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
|
|
351
|
-
const p = new
|
|
392
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
352
393
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
353
394
|
await p.generate({ workerId: "r", messages: [] });
|
|
354
395
|
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
@@ -359,7 +400,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
359
400
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
360
401
|
// #403): adaptive = the backend's own default posture = omission.
|
|
361
402
|
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
362
|
-
const p = new
|
|
403
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
363
404
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
364
405
|
await p.generate({ workerId: "r", messages: [] });
|
|
365
406
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -370,7 +411,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
370
411
|
});
|
|
371
412
|
|
|
372
413
|
test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
|
|
373
|
-
const p = new
|
|
414
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
374
415
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
375
416
|
await p.generate({ workerId: "r", messages: [] });
|
|
376
417
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
@@ -387,9 +428,9 @@ test("the family temperature default rides every request; caller sampling overri
|
|
|
387
428
|
});
|
|
388
429
|
|
|
389
430
|
test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
|
|
390
|
-
const base = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
431
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
|
|
391
432
|
// set + llamacpp -> the loop-breakers ride the wire
|
|
392
|
-
const p = new
|
|
433
|
+
const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
|
|
393
434
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
394
435
|
await p.generate({ workerId: "r", messages: [] });
|
|
395
436
|
let body = JSON.parse(calls[0].init.body as string);
|
|
@@ -400,7 +441,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
400
441
|
assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
|
|
401
442
|
mock.restoreAll();
|
|
402
443
|
// unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
|
|
403
|
-
const p2 = new
|
|
444
|
+
const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
|
|
404
445
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
405
446
|
await p2.generate({ workerId: "r", messages: [] });
|
|
406
447
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -408,7 +449,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
408
449
|
assert.equal("repeat_last_n" in body, false);
|
|
409
450
|
mock.restoreAll();
|
|
410
451
|
// DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
|
|
411
|
-
const p3 = new
|
|
452
|
+
const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
|
|
412
453
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
413
454
|
await p3.generate({ workerId: "r", messages: [] });
|
|
414
455
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -418,7 +459,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
418
459
|
});
|
|
419
460
|
|
|
420
461
|
test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
|
|
421
|
-
const p = new
|
|
462
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
422
463
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
423
464
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
424
465
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -428,13 +469,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
428
469
|
|
|
429
470
|
test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
|
|
430
471
|
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
431
|
-
const llama = new
|
|
472
|
+
const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
432
473
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
433
474
|
await llama.generate({ workerId: "r", messages: [] });
|
|
434
475
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
435
476
|
mock.restoreAll();
|
|
436
477
|
// a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
|
|
437
|
-
const cloud = new
|
|
478
|
+
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
438
479
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
439
480
|
await cloud.generate({ workerId: "r", messages: [] });
|
|
440
481
|
const cloudBody = JSON.parse(calls[0].init.body as string);
|
|
@@ -443,14 +484,14 @@ test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (
|
|
|
443
484
|
assert.equal("repeat_penalty" in cloudBody, false);
|
|
444
485
|
mock.restoreAll();
|
|
445
486
|
// frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
|
|
446
|
-
const bare = new
|
|
487
|
+
const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
447
488
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
448
489
|
await bare.generate({ workerId: "r", messages: [] });
|
|
449
490
|
assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
|
|
450
491
|
});
|
|
451
492
|
|
|
452
493
|
test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
|
|
453
|
-
const p = new
|
|
494
|
+
const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
454
495
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
455
496
|
await p.generate({
|
|
456
497
|
workerId: "r",
|
|
@@ -473,7 +514,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
|
|
|
473
514
|
});
|
|
474
515
|
|
|
475
516
|
test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
|
|
476
|
-
const p = new
|
|
517
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
477
518
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
478
519
|
await p.generate({
|
|
479
520
|
workerId: "r",
|
|
@@ -500,7 +541,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
500
541
|
test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
|
|
501
542
|
// The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
|
|
502
543
|
// reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
|
|
503
|
-
const p = new
|
|
544
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
504
545
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
505
546
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
506
547
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -514,7 +555,7 @@ test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — s
|
|
|
514
555
|
test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
|
|
515
556
|
// The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
|
|
516
557
|
// escaped into a discarded reasoning block, unconstrained.
|
|
517
|
-
const p = new
|
|
558
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
518
559
|
installFetch([
|
|
519
560
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
520
561
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
@@ -528,7 +569,7 @@ test("#488 channel-escape detector: billed completion tokens vastly beyond visib
|
|
|
528
569
|
});
|
|
529
570
|
|
|
530
571
|
test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
|
|
531
|
-
const p = new
|
|
572
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
532
573
|
installFetch([
|
|
533
574
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
534
575
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
@@ -539,33 +580,33 @@ test("#488 loud state absent on grammarless calls; no escape event without a tra
|
|
|
539
580
|
});
|
|
540
581
|
|
|
541
582
|
test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
|
|
542
|
-
const on = new
|
|
583
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
543
584
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
544
585
|
await on.generate({ workerId: "r", messages: [] });
|
|
545
586
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
|
|
546
587
|
|
|
547
588
|
mock.restoreAll();
|
|
548
|
-
const off = new
|
|
589
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
549
590
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
550
591
|
await off.generate({ workerId: "r", messages: [] });
|
|
551
592
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
|
|
552
593
|
});
|
|
553
594
|
|
|
554
595
|
test("budget 0 suppresses effort and include_reasoning", async () => {
|
|
555
|
-
const effort = new
|
|
596
|
+
const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
556
597
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
557
598
|
await effort.generate({ workerId: "r", messages: [] });
|
|
558
599
|
assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
|
|
559
600
|
|
|
560
601
|
mock.restoreAll();
|
|
561
|
-
const relay = new
|
|
602
|
+
const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
562
603
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
563
604
|
await relay.generate({ workerId: "r", messages: [] });
|
|
564
605
|
assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
|
|
565
606
|
});
|
|
566
607
|
|
|
567
608
|
test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
|
|
568
|
-
const p = new
|
|
609
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
569
610
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
570
611
|
await p.generate({ workerId: "r", messages: [] });
|
|
571
612
|
assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
|
|
@@ -574,7 +615,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
|
|
|
574
615
|
// — grammar-constrained sampling (SPEC §13, issues #8/#9) —
|
|
575
616
|
|
|
576
617
|
test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
|
|
577
|
-
const p = new
|
|
618
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
578
619
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
579
620
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
580
621
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -584,7 +625,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
|
|
|
584
625
|
});
|
|
585
626
|
|
|
586
627
|
test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
|
|
587
|
-
const p = new
|
|
628
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
588
629
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
589
630
|
await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
|
|
590
631
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -595,7 +636,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
|
|
|
595
636
|
// — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
|
|
596
637
|
// returns; bytes flow; a non-accept verdict rides response.telemetry —
|
|
597
638
|
|
|
598
|
-
const grammarProvider = () => new
|
|
639
|
+
const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
|
|
599
640
|
const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
600
641
|
|
|
601
642
|
test("enforcement: conforming output passes through unchanged", async () => {
|
|
@@ -644,7 +685,7 @@ test("observation: empty content under a non-empty grammar returns with the verd
|
|
|
644
685
|
});
|
|
645
686
|
|
|
646
687
|
test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
|
|
647
|
-
const p = new
|
|
688
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
|
|
648
689
|
streamingContent("anything goes");
|
|
649
690
|
const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
|
|
650
691
|
assert.equal(assistant.content, "anything goes"); // no enforcement check
|
|
@@ -666,7 +707,7 @@ test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap
|
|
|
666
707
|
// — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
|
|
667
708
|
|
|
668
709
|
test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
|
|
669
|
-
const p = new
|
|
710
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
670
711
|
const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
|
|
671
712
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
672
713
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -677,7 +718,7 @@ test("gbnfDebug: the grammar is NOT transported; conforming free output passes t
|
|
|
677
718
|
});
|
|
678
719
|
|
|
679
720
|
test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
|
|
680
|
-
const p = new
|
|
721
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
681
722
|
const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
|
|
682
723
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
683
724
|
// The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
|
|
@@ -695,7 +736,7 @@ test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a gramm
|
|
|
695
736
|
});
|
|
696
737
|
|
|
697
738
|
test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
|
|
698
|
-
const p = new
|
|
739
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
|
|
699
740
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
700
741
|
await assert.rejects(
|
|
701
742
|
() => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
|
|
@@ -707,7 +748,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
707
748
|
// — meta bag: verbatim provider metadata (#23) —
|
|
708
749
|
|
|
709
750
|
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
710
|
-
const p = new
|
|
751
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
711
752
|
const balance = { amount: "0.0000042", currency: "XMR" };
|
|
712
753
|
installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
|
|
713
754
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -721,7 +762,7 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
|
|
|
721
762
|
new Headers(init.headers).get(name) ?? undefined;
|
|
722
763
|
|
|
723
764
|
test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
|
|
724
|
-
const p = new
|
|
765
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
725
766
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
726
767
|
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
|
|
727
768
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
|
|
@@ -729,7 +770,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
|
|
|
729
770
|
});
|
|
730
771
|
|
|
731
772
|
test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
|
|
732
|
-
const p = new
|
|
773
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
733
774
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
734
775
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
735
776
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
|
|
@@ -748,14 +789,14 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
|
|
|
748
789
|
});
|
|
749
790
|
|
|
750
791
|
test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
|
|
751
|
-
const p = new
|
|
792
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
752
793
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
753
794
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
754
795
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
|
|
755
796
|
});
|
|
756
797
|
|
|
757
798
|
test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
|
|
758
|
-
const p = new
|
|
799
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
759
800
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
760
801
|
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
|
|
761
802
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
|
|
@@ -763,7 +804,7 @@ test("firstPartyMetadata off (default): the headers are structurally dropped eve
|
|
|
763
804
|
});
|
|
764
805
|
|
|
765
806
|
test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
766
|
-
const p = new
|
|
807
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
767
808
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
768
809
|
await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
|
|
769
810
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
|
|
@@ -771,7 +812,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
|
771
812
|
});
|
|
772
813
|
|
|
773
814
|
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
|
|
774
|
-
const p = new
|
|
815
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
775
816
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
776
817
|
await p.generate({ workerId: "r", messages: [] });
|
|
777
818
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -780,7 +821,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
|
|
|
780
821
|
});
|
|
781
822
|
|
|
782
823
|
test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
|
|
783
|
-
const p = new
|
|
824
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
784
825
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
785
826
|
await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
|
|
786
827
|
assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
|
|
@@ -792,7 +833,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
|
|
|
792
833
|
});
|
|
793
834
|
|
|
794
835
|
test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
|
|
795
|
-
const pinning = new
|
|
836
|
+
const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
796
837
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
797
838
|
await pinning.generate({ workerId: "run-A", messages: [] });
|
|
798
839
|
await pinning.generate({ workerId: "run-B", messages: [] });
|
|
@@ -803,20 +844,20 @@ test("slot affinity is internal: sticky per workerId, distinct runs spread acros
|
|
|
803
844
|
});
|
|
804
845
|
|
|
805
846
|
test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
|
|
806
|
-
const cloud = new
|
|
847
|
+
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
|
|
807
848
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
808
849
|
await cloud.generate({ workerId: "run-A", messages: [] });
|
|
809
850
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
810
851
|
|
|
811
852
|
mock.restoreAll();
|
|
812
|
-
const noCount = new
|
|
853
|
+
const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
|
|
813
854
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
814
855
|
await noCount.generate({ workerId: "run-A", messages: [] });
|
|
815
856
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
816
857
|
});
|
|
817
858
|
|
|
818
859
|
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
|
|
819
|
-
const p = new
|
|
860
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
820
861
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
821
862
|
const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
|
|
822
863
|
for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
|
|
@@ -830,7 +871,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
|
|
|
830
871
|
|
|
831
872
|
test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
|
|
832
873
|
const { ProviderError } = await import("./telemetry.ts");
|
|
833
|
-
const p = new
|
|
874
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
|
|
834
875
|
mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
|
|
835
876
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
836
877
|
assert.ok(err instanceof ProviderError);
|
|
@@ -841,14 +882,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
|
|
|
841
882
|
});
|
|
842
883
|
|
|
843
884
|
test("generate fail-hards on a missing or empty workerId", async () => {
|
|
844
|
-
const p = new
|
|
885
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
845
886
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
846
887
|
await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
|
|
847
888
|
await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
|
|
848
889
|
});
|
|
849
890
|
|
|
850
891
|
test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
|
|
851
|
-
const p = new
|
|
892
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
852
893
|
const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
|
|
853
894
|
const input = [{ role: "user" as const, content: "hi" }];
|
|
854
895
|
const res = await p.generate({ workerId: "r", messages: input });
|
|
@@ -858,7 +899,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
|
|
|
858
899
|
|
|
859
900
|
test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
|
|
860
901
|
const { ProviderError } = await import("./telemetry.ts");
|
|
861
|
-
const p = new
|
|
902
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
|
|
862
903
|
mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
|
|
863
904
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
864
905
|
assert.ok(err instanceof ProviderError);
|
|
@@ -870,27 +911,28 @@ test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEven
|
|
|
870
911
|
});
|
|
871
912
|
|
|
872
913
|
test("generate rejects on a pre-aborted external signal", async () => {
|
|
873
|
-
const p = new
|
|
914
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
874
915
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
875
916
|
const signal = AbortSignal.abort(new Error("nope"));
|
|
876
917
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
|
|
877
918
|
});
|
|
878
919
|
|
|
879
920
|
test("configured headers and url are sent verbatim", async () => {
|
|
880
|
-
const p = new
|
|
881
|
-
model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
921
|
+
const p = new AiSdkProvider({
|
|
922
|
+
model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
882
923
|
headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
|
|
883
924
|
});
|
|
884
925
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
885
926
|
await p.generate({ workerId: "r", messages: [] });
|
|
886
927
|
assert.equal(calls[0].url, "http://host/custom/chat/completions");
|
|
887
|
-
|
|
888
|
-
assert.equal((
|
|
928
|
+
const headers = new Headers(calls[0].init.headers);
|
|
929
|
+
assert.equal(headers.get("authorization"), "Bearer secret");
|
|
930
|
+
assert.equal(headers.get("x-title"), "plurnk");
|
|
889
931
|
});
|
|
890
932
|
|
|
891
933
|
// — transient-failure retry (#18) —
|
|
892
934
|
|
|
893
|
-
const retryCfg = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
935
|
+
const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
|
|
894
936
|
|
|
895
937
|
test("retry: a transient failure retries and a later success resolves", async () => {
|
|
896
938
|
const calls = installFetchScript([
|
|
@@ -898,51 +940,52 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
898
940
|
{ status: 503, retryAfter: 0 },
|
|
899
941
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
|
|
900
942
|
]);
|
|
901
|
-
const p = new
|
|
943
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
902
944
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
903
945
|
assert.equal(res.assistant.content, "ok");
|
|
904
946
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
905
947
|
});
|
|
906
948
|
|
|
907
|
-
test("#559: streamed-body silence
|
|
949
|
+
test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
|
|
908
950
|
let calls = 0;
|
|
909
951
|
mock.method(globalThis, "fetch", async () => {
|
|
910
952
|
calls++;
|
|
911
953
|
if (calls === 1) {
|
|
912
954
|
return new Response(new ReadableStream({
|
|
913
955
|
start(controller) {
|
|
914
|
-
controller.enqueue(new TextEncoder().encode(
|
|
956
|
+
controller.enqueue(new TextEncoder().encode(
|
|
957
|
+
'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
958
|
+
));
|
|
959
|
+
setTimeout(() => controller.close(), 100);
|
|
915
960
|
},
|
|
916
961
|
}), { status: 200 });
|
|
917
962
|
}
|
|
918
963
|
return new Response(new ReadableStream({
|
|
919
964
|
start(controller) {
|
|
920
|
-
controller.enqueue(new TextEncoder().encode(
|
|
965
|
+
controller.enqueue(new TextEncoder().encode(
|
|
966
|
+
'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
|
|
967
|
+
));
|
|
921
968
|
controller.close();
|
|
922
969
|
},
|
|
923
970
|
}), { status: 200 });
|
|
924
971
|
});
|
|
925
|
-
const p = new
|
|
972
|
+
const p = new AiSdkProvider({
|
|
926
973
|
model: "m",
|
|
927
|
-
url: "http://x",
|
|
974
|
+
url: "http://x/v1/chat/completions",
|
|
928
975
|
fetchTimeoutMs: 1000,
|
|
929
976
|
streamIdleTimeoutMs: 10,
|
|
930
977
|
temperature: 0.2,
|
|
931
978
|
repeatPenalty: 1.15,
|
|
932
|
-
retryDelayMs: 1,
|
|
933
979
|
reasoning: { mode: "off", budget: null },
|
|
934
980
|
retryAttempts: 1,
|
|
935
981
|
source: "provider:test",
|
|
936
982
|
});
|
|
937
|
-
|
|
938
|
-
|
|
939
|
-
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
assert.equal(
|
|
943
|
-
assert.equal(retries[0].kind, "network_failure");
|
|
944
|
-
assert.equal(typeof retries[0].elapsedMs, "number");
|
|
945
|
-
assert.match(String(retries[0].message), /no body bytes for 10ms/);
|
|
983
|
+
await assert.rejects(
|
|
984
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
985
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
986
|
+
&& /chunk timeout/i.test(error.message),
|
|
987
|
+
);
|
|
988
|
+
assert.equal(calls, 1);
|
|
946
989
|
mock.restoreAll();
|
|
947
990
|
});
|
|
948
991
|
|
|
@@ -955,27 +998,25 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
|
|
|
955
998
|
controller.close();
|
|
956
999
|
},
|
|
957
1000
|
}), { status: 200 }));
|
|
958
|
-
const p = new
|
|
1001
|
+
const p = new AiSdkProvider({
|
|
959
1002
|
model: "m",
|
|
960
|
-
url: "http://x",
|
|
1003
|
+
url: "http://x/v1/chat/completions",
|
|
961
1004
|
fetchTimeoutMs: 1000,
|
|
962
1005
|
streamIdleTimeoutMs: 0,
|
|
963
1006
|
temperature: 0.2,
|
|
964
1007
|
repeatPenalty: 1.15,
|
|
965
|
-
retryDelayMs: 1,
|
|
966
1008
|
reasoning: { mode: "off", budget: null },
|
|
967
1009
|
retryAttempts: 0,
|
|
968
1010
|
});
|
|
969
1011
|
const result = await p.generate({ workerId: "r", messages: [] });
|
|
970
1012
|
assert.equal(result.assistant.content, "slow is valid");
|
|
971
|
-
assert.equal(result.meta?.transportRetries, undefined);
|
|
972
1013
|
mock.restoreAll();
|
|
973
1014
|
});
|
|
974
1015
|
|
|
975
1016
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
976
1017
|
const { ProviderError } = await import("./telemetry.ts");
|
|
977
1018
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
|
978
|
-
const p = new
|
|
1019
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
|
|
979
1020
|
await assert.rejects(
|
|
980
1021
|
() => p.generate({ workerId: "r", messages: [] }),
|
|
981
1022
|
(err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
|
|
@@ -988,7 +1029,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
988
1029
|
{ status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
|
|
989
1030
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
|
|
990
1031
|
]);
|
|
991
|
-
const p = new
|
|
1032
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
|
|
992
1033
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
993
1034
|
assert.equal(assistant.content, "ok");
|
|
994
1035
|
assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
|
|
@@ -996,14 +1037,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
996
1037
|
|
|
997
1038
|
test("retry: a terminal error (401 unauthorized) is never retried", async () => {
|
|
998
1039
|
const calls = installFetchScript([{ status: 401 }]);
|
|
999
|
-
const p = new
|
|
1040
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
|
|
1000
1041
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
|
|
1001
1042
|
assert.equal(calls.length, 1); // terminal — no retry despite budget
|
|
1002
1043
|
});
|
|
1003
1044
|
|
|
1004
1045
|
test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
|
|
1005
1046
|
const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
|
|
1006
|
-
const p = new
|
|
1047
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
|
|
1007
1048
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
|
|
1008
1049
|
assert.equal(calls.length, 1); // no retry budget
|
|
1009
1050
|
});
|
|
@@ -1011,7 +1052,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
|
|
|
1011
1052
|
test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
|
|
1012
1053
|
const ac = new AbortController();
|
|
1013
1054
|
const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
|
|
1014
|
-
const p = new
|
|
1055
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
1015
1056
|
const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
|
|
1016
1057
|
await flush(); // attempt 0 fails, enters the backoff sleep
|
|
1017
1058
|
assert.equal(calls.length, 1);
|
|
@@ -1024,21 +1065,21 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
1024
1065
|
|
|
1025
1066
|
test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
|
|
1026
1067
|
// N>0 → enabled with budget_tokens
|
|
1027
|
-
const capped = new
|
|
1068
|
+
const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
|
|
1028
1069
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1029
1070
|
await capped.generate({ workerId: "r", messages: [] });
|
|
1030
1071
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
|
|
1031
1072
|
|
|
1032
1073
|
mock.restoreAll();
|
|
1033
1074
|
// 0 → explicit disabled
|
|
1034
|
-
const off = new
|
|
1075
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
|
|
1035
1076
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1036
1077
|
await off.generate({ workerId: "r", messages: [] });
|
|
1037
1078
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
|
|
1038
1079
|
|
|
1039
1080
|
mock.restoreAll();
|
|
1040
1081
|
// -1 adaptive → omit (API default depth)
|
|
1041
|
-
const adaptive = new
|
|
1082
|
+
const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
|
|
1042
1083
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1043
1084
|
await adaptive.generate({ workerId: "r", messages: [] });
|
|
1044
1085
|
assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
|
|
@@ -1056,7 +1097,7 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1056
1097
|
usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
|
|
1057
1098
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
1058
1099
|
});
|
|
1059
|
-
const p = new
|
|
1100
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
1060
1101
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
1061
1102
|
const sent = JSON.parse(calls[0].body);
|
|
1062
1103
|
assert.equal("stream" in sent, false); // no streaming flag
|
|
@@ -1068,11 +1109,11 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1068
1109
|
});
|
|
1069
1110
|
|
|
1070
1111
|
// ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
|
|
1071
|
-
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
1112
|
+
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1072
1113
|
|
|
1073
1114
|
test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
|
|
1074
1115
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1075
|
-
const p = new
|
|
1116
|
+
const p = new AiSdkProvider({ ...captureBase });
|
|
1076
1117
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1077
1118
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1078
1119
|
assert.equal("logprobs" in body, false);
|
|
@@ -1089,7 +1130,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1089
1130
|
{ token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
|
|
1090
1131
|
] } }] };
|
|
1091
1132
|
const calls = installFetch([chunk]);
|
|
1092
|
-
const p = new
|
|
1133
|
+
const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
|
|
1093
1134
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1094
1135
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1095
1136
|
assert.equal(body.logprobs, true);
|
|
@@ -1103,7 +1144,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1103
1144
|
test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
|
|
1104
1145
|
const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
1105
1146
|
installFetchJson(wire);
|
|
1106
|
-
const p = new
|
|
1147
|
+
const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
|
|
1107
1148
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1108
1149
|
assert.deepEqual(res.rawBody, wire); // verbatim
|
|
1109
1150
|
assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
|
|
@@ -1114,7 +1155,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
|
|
|
1114
1155
|
|
|
1115
1156
|
test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
|
|
1116
1157
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1117
|
-
const p = new
|
|
1158
|
+
const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
|
|
1118
1159
|
await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
|
|
1119
1160
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1120
1161
|
assert.equal("logprobs" in body, false); // sampling passthrough stripped it
|
|
@@ -1125,52 +1166,52 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
|
|
|
1125
1166
|
// — turn coordinate headers (#404, per #391): same gate as every first-party signal —
|
|
1126
1167
|
|
|
1127
1168
|
test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
|
|
1128
|
-
const p = new
|
|
1169
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1129
1170
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1130
1171
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1131
|
-
const
|
|
1132
|
-
assert.equal(
|
|
1133
|
-
assert.equal(
|
|
1134
|
-
assert.equal(
|
|
1172
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1173
|
+
assert.equal(headers.get("plurnk-workspace-id"), "s-9");
|
|
1174
|
+
assert.equal(headers.get("plurnk-loop"), "3");
|
|
1175
|
+
assert.equal(headers.get("plurnk-turn"), "41");
|
|
1135
1176
|
});
|
|
1136
1177
|
|
|
1137
1178
|
test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
|
|
1138
|
-
const p = new
|
|
1179
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1139
1180
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1140
1181
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1141
|
-
const
|
|
1142
|
-
assert.equal("
|
|
1143
|
-
assert.equal("
|
|
1144
|
-
assert.equal("
|
|
1182
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1183
|
+
assert.equal(headers.has("plurnk-workspace-id"), false);
|
|
1184
|
+
assert.equal(headers.has("plurnk-loop"), false);
|
|
1185
|
+
assert.equal(headers.has("plurnk-turn"), false);
|
|
1145
1186
|
});
|
|
1146
1187
|
|
|
1147
1188
|
test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
|
|
1148
|
-
const p = new
|
|
1189
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1149
1190
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1150
1191
|
await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
|
|
1151
|
-
const
|
|
1152
|
-
assert.equal("
|
|
1153
|
-
assert.equal("
|
|
1154
|
-
assert.equal("
|
|
1155
|
-
assert.equal(
|
|
1192
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1193
|
+
assert.equal(headers.has("plurnk-workspace-id"), false);
|
|
1194
|
+
assert.equal(headers.has("plurnk-loop"), false);
|
|
1195
|
+
assert.equal(headers.has("plurnk-turn"), false);
|
|
1196
|
+
assert.equal(headers.has("plurnk-strikes"), false);
|
|
1156
1197
|
});
|
|
1157
1198
|
|
|
1158
1199
|
// -- #507: envelope surface + router-owned tuning --
|
|
1159
1200
|
|
|
1160
1201
|
test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
|
|
1161
|
-
const base = { model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
1162
|
-
const derived = new
|
|
1202
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1203
|
+
const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
|
|
1163
1204
|
assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
|
|
1164
1205
|
assert.equal(derived.completionReserve, 12288); // 25% of 49152
|
|
1165
|
-
const pinned = new
|
|
1206
|
+
const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
|
|
1166
1207
|
assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
|
|
1167
1208
|
assert.equal(pinned.completionReserve, null); // percent without a window = underivable
|
|
1168
|
-
const legacy = new
|
|
1209
|
+
const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
|
|
1169
1210
|
assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
|
|
1170
1211
|
});
|
|
1171
1212
|
|
|
1172
1213
|
test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
|
|
1173
|
-
const p = new
|
|
1214
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
|
|
1174
1215
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1175
1216
|
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
|
|
1176
1217
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1181,21 +1222,21 @@ test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty
|
|
|
1181
1222
|
// -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
|
|
1182
1223
|
|
|
1183
1224
|
test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
|
|
1184
|
-
const p = new
|
|
1225
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1185
1226
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1186
1227
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1187
1228
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
|
|
1188
1229
|
});
|
|
1189
1230
|
|
|
1190
1231
|
test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
|
|
1191
|
-
const p = new
|
|
1232
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1192
1233
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1193
1234
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1194
1235
|
assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
|
|
1195
1236
|
});
|
|
1196
1237
|
|
|
1197
1238
|
test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
|
|
1198
|
-
const p = new
|
|
1239
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1199
1240
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1200
1241
|
await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
|
|
1201
1242
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins
|