@plurnk/plurnk-providers 1.3.4 → 1.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -52
- package/README.md +44 -53
- package/SPEC.md +215 -354
- package/dist/AiSdkProvider.d.ts +78 -0
- package/dist/AiSdkProvider.d.ts.map +1 -0
- package/dist/AiSdkProvider.js +591 -0
- package/dist/AiSdkProvider.js.map +1 -0
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +3 -5
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +44 -133
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +37 -24
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +52 -0
- package/dist/aiSdkTransport.d.ts.map +1 -0
- package/dist/aiSdkTransport.js +294 -0
- package/dist/aiSdkTransport.js.map +1 -0
- package/dist/catalogProvider.d.ts +15 -0
- package/dist/catalogProvider.d.ts.map +1 -0
- package/dist/catalogProvider.js +103 -0
- package/dist/catalogProvider.js.map +1 -0
- package/dist/compatibleProvider.d.ts +3 -0
- package/dist/compatibleProvider.d.ts.map +1 -0
- package/dist/compatibleProvider.js +146 -0
- package/dist/compatibleProvider.js.map +1 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +1 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +13 -6
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +5 -7
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -8
- package/dist/index.js.map +1 -1
- package/dist/ollama.d.ts +3 -0
- package/dist/ollama.d.ts.map +1 -0
- package/dist/ollama.js +39 -0
- package/dist/ollama.js.map +1 -0
- package/dist/openai.d.ts +2 -4
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -2
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -0
- package/dist/sdkModels.d.ts.map +1 -0
- package/dist/sdkModels.js +153 -0
- package/dist/sdkModels.js.map +1 -0
- package/dist/standardProviders.d.ts +0 -1
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +9 -11
- package/dist/standardProviders.js.map +1 -1
- package/dist/telemetry.d.ts.map +1 -1
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/types.d.ts +4 -3
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +19 -8
- package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +202 -177
- package/src/{OpenAICompat.ts → AiSdkProvider.ts} +89 -144
- package/src/Mock.test.ts +3 -3
- package/src/Mock.ts +1 -1
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +40 -27
- package/src/ProviderRegistry.ts +35 -24
- package/src/aiSdkTransport.test.ts +253 -0
- package/src/aiSdkTransport.ts +369 -0
- package/src/boundaries.test.ts +2 -2
- package/src/catalogProvider.test.ts +100 -0
- package/src/catalogProvider.ts +151 -0
- package/src/compatibleProvider.test.ts +44 -0
- package/src/compatibleProvider.ts +205 -0
- package/src/discover.test.ts +12 -12
- package/src/discover.ts +3 -6
- package/src/env.ts +14 -6
- package/src/index.ts +7 -11
- package/src/ollama.ts +63 -0
- package/src/openai.ts +2 -8
- package/src/sdkModels.test.ts +47 -0
- package/src/sdkModels.ts +194 -0
- package/src/telemetry.test.ts +17 -10
- package/src/telemetry.ts +22 -14
- package/src/types.ts +10 -13
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
- package/src/openaiStream.ts +0 -310
- package/src/standardProviders.test.ts +0 -949
- package/src/standardProviders.ts +0 -635
|
@@ -1,13 +1,37 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import
|
|
4
|
-
import { OpenAiHttpError } from "./openaiStream.ts";
|
|
3
|
+
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
5
4
|
import { ProviderError } from "./telemetry.ts";
|
|
6
5
|
|
|
7
6
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
8
7
|
// so tests can assert what the spine sent on the wire.
|
|
9
8
|
const sseStream = (chunks: unknown[]) => {
|
|
10
|
-
const
|
|
9
|
+
const normalized = chunks.map((value, index) => {
|
|
10
|
+
const chunk = value as Record<string, any>;
|
|
11
|
+
const usage = chunk.usage !== undefined
|
|
12
|
+
? {
|
|
13
|
+
...chunk.usage,
|
|
14
|
+
...(chunk.usage.cached_tokens !== undefined
|
|
15
|
+
? {
|
|
16
|
+
prompt_tokens_details: {
|
|
17
|
+
cached_tokens: chunk.usage.cached_tokens,
|
|
18
|
+
},
|
|
19
|
+
}
|
|
20
|
+
: {}),
|
|
21
|
+
}
|
|
22
|
+
: undefined;
|
|
23
|
+
if (usage !== undefined) delete usage.cached_tokens;
|
|
24
|
+
return {
|
|
25
|
+
id: "test-completion",
|
|
26
|
+
object: "chat.completion.chunk",
|
|
27
|
+
created: index + 1,
|
|
28
|
+
model: "m",
|
|
29
|
+
...chunk,
|
|
30
|
+
...(chunk.choices === undefined && usage !== undefined ? { choices: [] } : {}),
|
|
31
|
+
...(usage !== undefined ? { usage } : {}),
|
|
32
|
+
};
|
|
33
|
+
});
|
|
34
|
+
const lines = [...normalized.map((c) => `data: ${JSON.stringify(c)}`), "data: [DONE]"].join("\n\n");
|
|
11
35
|
return new ReadableStream({
|
|
12
36
|
start(controller) {
|
|
13
37
|
controller.enqueue(new TextEncoder().encode(lines));
|
|
@@ -43,7 +67,6 @@ const injectedBase = {
|
|
|
43
67
|
fetchTimeoutMs: 5000,
|
|
44
68
|
temperature: 0.2,
|
|
45
69
|
repeatPenalty: 1.15,
|
|
46
|
-
retryDelayMs: 1,
|
|
47
70
|
retryAttempts: 0,
|
|
48
71
|
reasoning: { mode: "off" as const, budget: null },
|
|
49
72
|
};
|
|
@@ -65,9 +88,9 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
65
88
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
66
89
|
};
|
|
67
90
|
|
|
68
|
-
const streamed = await new
|
|
91
|
+
const streamed = await new AiSdkProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
|
|
69
92
|
.generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
|
|
70
|
-
const buffered = await new
|
|
93
|
+
const buffered = await new AiSdkProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
|
|
71
94
|
.generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
|
|
72
95
|
|
|
73
96
|
assert.equal(streamed.assistant.content, "streamed");
|
|
@@ -83,18 +106,20 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
83
106
|
});
|
|
84
107
|
|
|
85
108
|
test("#608: caller cancellation and provider timeout reach an injected fetch", async () => {
|
|
86
|
-
const pendingFetch: typeof globalThis.fetch = async (_input, init) =>
|
|
87
|
-
|
|
109
|
+
const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
|
|
110
|
+
init?.signal?.throwIfAborted();
|
|
111
|
+
return new Promise((_resolve, reject) => {
|
|
88
112
|
const signal = init?.signal;
|
|
89
113
|
signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
90
114
|
});
|
|
115
|
+
};
|
|
91
116
|
const caller = new AbortController();
|
|
92
|
-
const callerProvider = new
|
|
117
|
+
const callerProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch });
|
|
93
118
|
const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
|
|
94
119
|
caller.abort(new Error("operator cancelled"));
|
|
95
120
|
await assert.rejects(callerRequest, /operator cancelled/);
|
|
96
121
|
|
|
97
|
-
const timeoutProvider = new
|
|
122
|
+
const timeoutProvider = new AiSdkProvider({ ...injectedBase, fetch: pendingFetch, fetchTimeoutMs: 1 });
|
|
98
123
|
await assert.rejects(
|
|
99
124
|
timeoutProvider.generate({ workerId: "timeout", messages: [] }),
|
|
100
125
|
(error: ProviderError) => error.kind === "network_failure",
|
|
@@ -116,7 +141,7 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
|
|
|
116
141
|
{ choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
|
|
117
142
|
]), { status: 200 });
|
|
118
143
|
};
|
|
119
|
-
const provider = new
|
|
144
|
+
const provider = new AiSdkProvider({
|
|
120
145
|
...injectedBase,
|
|
121
146
|
fetch: providerFetch,
|
|
122
147
|
retryAttempts: 1,
|
|
@@ -135,7 +160,13 @@ test("#608: per-instance fetch owns tokenization and retry attempts", async () =
|
|
|
135
160
|
// Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
|
|
136
161
|
// streams its chunks; any other status returns that error (with an optional
|
|
137
162
|
// retry-after header). The last entry repeats once the script runs out.
|
|
138
|
-
type ScriptedResponse = {
|
|
163
|
+
type ScriptedResponse = {
|
|
164
|
+
status: number;
|
|
165
|
+
chunks?: unknown[];
|
|
166
|
+
retryAfter?: number | string;
|
|
167
|
+
shouldRetry?: boolean;
|
|
168
|
+
body?: string;
|
|
169
|
+
};
|
|
139
170
|
const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
140
171
|
const calls: { url: string; init: RequestInit }[] = [];
|
|
141
172
|
let i = 0;
|
|
@@ -144,8 +175,15 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
|
144
175
|
const r = responses[Math.min(i, responses.length - 1)];
|
|
145
176
|
i++;
|
|
146
177
|
if (r.status === 200) return new Response(sseStream(r.chunks ?? []), { status: 200 });
|
|
147
|
-
const headers =
|
|
148
|
-
|
|
178
|
+
const headers = {
|
|
179
|
+
"content-type": "application/json",
|
|
180
|
+
...(r.retryAfter !== undefined ? { "retry-after": String(r.retryAfter) } : {}),
|
|
181
|
+
...(r.shouldRetry !== undefined ? { "x-should-retry": String(r.shouldRetry) } : {}),
|
|
182
|
+
};
|
|
183
|
+
return new Response(
|
|
184
|
+
r.body ?? JSON.stringify({ error: { message: `HTTP ${r.status}` } }),
|
|
185
|
+
{ status: r.status, headers },
|
|
186
|
+
);
|
|
149
187
|
});
|
|
150
188
|
return calls;
|
|
151
189
|
};
|
|
@@ -164,33 +202,25 @@ test("effortFromBudget: maps budget to tiers", () => {
|
|
|
164
202
|
assert.equal(effortFromBudget(4001), "high");
|
|
165
203
|
});
|
|
166
204
|
|
|
167
|
-
test("#543: OpenAiHttpError distills a non-JSON (edge/CDN HTML) body and drops the OpenAI prefix", () => {
|
|
168
|
-
const cf = new OpenAiHttpError(524, "<!DOCTYPE html><html><body>Error code 524</body></html>", null);
|
|
169
|
-
assert.equal(cf.message, "524 origin timeout"); // distilled: no raw HTML, no "OpenAI" prefix
|
|
170
|
-
assert.ok(cf.body.length > 20); // raw body retained on the field for forensics
|
|
171
|
-
const api = new OpenAiHttpError(400, '{"error":{"message":"bad param"}}', null);
|
|
172
|
-
assert.match(api.message, /^OpenAI 400 - \{/); // JSON API error passes through verbatim
|
|
173
|
-
});
|
|
174
|
-
|
|
175
205
|
test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
176
206
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
177
|
-
const p = new
|
|
207
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
178
208
|
await assert.rejects(p.generate({ workerId: "r", messages: [] }));
|
|
179
209
|
await flush();
|
|
180
210
|
assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
|
|
181
211
|
mock.restoreAll();
|
|
182
212
|
});
|
|
183
213
|
|
|
184
|
-
test("#548: a 422 grammar_invalid is
|
|
214
|
+
test("#548: a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
|
|
185
215
|
const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
|
|
186
216
|
const calls = installFetchScript([{ status: 422, body }]);
|
|
187
|
-
const p = new
|
|
217
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
|
|
188
218
|
await assert.rejects(
|
|
189
219
|
p.generate({ workerId: "r", messages: [] }),
|
|
190
220
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
191
221
|
);
|
|
192
222
|
await flush();
|
|
193
|
-
assert.equal(calls.length,
|
|
223
|
+
assert.equal(calls.length, 1);
|
|
194
224
|
mock.restoreAll();
|
|
195
225
|
});
|
|
196
226
|
|
|
@@ -199,7 +229,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
199
229
|
status: 422,
|
|
200
230
|
error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
|
|
201
231
|
}]);
|
|
202
|
-
const p = new
|
|
232
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
203
233
|
await assert.rejects(
|
|
204
234
|
p.generate({ workerId: "r", messages: [] }),
|
|
205
235
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
@@ -209,46 +239,46 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
209
239
|
|
|
210
240
|
test("#539: a trailing eos_token (--special EOG leak) is stripped from content", async () => {
|
|
211
241
|
installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
212
|
-
const p = new
|
|
242
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
213
243
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
214
244
|
assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
|
|
215
245
|
});
|
|
216
246
|
|
|
217
247
|
test("#539: without a probed eos_token the content passes through untouched", async () => {
|
|
218
248
|
installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
219
|
-
const p = new
|
|
249
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
220
250
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
221
251
|
assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
|
|
222
252
|
});
|
|
223
253
|
|
|
224
254
|
test("#539: only the TRAILING eos_token is stripped; a quoted one mid-body survives", async () => {
|
|
225
255
|
installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
|
|
226
|
-
const p = new
|
|
256
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
227
257
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
228
258
|
assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
|
|
229
259
|
});
|
|
230
260
|
|
|
231
261
|
test("identity getters and defaults", () => {
|
|
232
|
-
const p = new
|
|
262
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
233
263
|
assert.equal(p.model, "m");
|
|
234
264
|
assert.equal(p.contextWindow, null); // default
|
|
235
265
|
assert.equal(p.countTokens(""), 0);
|
|
236
266
|
assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
|
|
237
|
-
assert.equal(p.
|
|
267
|
+
assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
|
|
238
268
|
});
|
|
239
269
|
|
|
240
|
-
test("injected countTokens and
|
|
241
|
-
const p = new
|
|
242
|
-
model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
270
|
+
test("injected countTokens and calculateCost are used", () => {
|
|
271
|
+
const p = new AiSdkProvider({
|
|
272
|
+
model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
243
273
|
countTokens: (t) => t.length,
|
|
244
|
-
|
|
274
|
+
calculateCost: (u) => u.total * 2,
|
|
245
275
|
});
|
|
246
276
|
assert.equal(p.countTokens("abc"), 3);
|
|
247
|
-
assert.equal(p.
|
|
277
|
+
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
|
|
248
278
|
});
|
|
249
279
|
|
|
250
280
|
test("generate maps a streamed response into ProviderResponse", async () => {
|
|
251
|
-
const p = new
|
|
281
|
+
const p = new AiSdkProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
252
282
|
installFetch([
|
|
253
283
|
{ model: "wire-model", choices: [{ delta: { content: "hel" } }] },
|
|
254
284
|
{ choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
|
|
@@ -263,31 +293,42 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
263
293
|
assert.notEqual(assistantRaw, undefined);
|
|
264
294
|
});
|
|
265
295
|
|
|
266
|
-
test("generate normalizes an out-of-set finish_reason
|
|
267
|
-
const
|
|
296
|
+
test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
|
|
297
|
+
const warnings: Array<{ message: string; code?: string }> = [];
|
|
298
|
+
mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
|
|
299
|
+
warnings.push({
|
|
300
|
+
message: String(message),
|
|
301
|
+
...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
|
|
302
|
+
});
|
|
303
|
+
});
|
|
304
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
268
305
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
|
|
269
306
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
270
307
|
assert.equal(assistant.finishReason, null);
|
|
308
|
+
assert.deepEqual(warnings, [{
|
|
309
|
+
message: 'unrecognized finish_reason "function_call"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core\'s length-cap detection will miss it.',
|
|
310
|
+
code: "PLURNK_FINISH_REASON_UNKNOWN",
|
|
311
|
+
}]);
|
|
271
312
|
});
|
|
272
313
|
|
|
273
314
|
test("generate translates a backend cap synonym to canonical length (#425)", async () => {
|
|
274
315
|
// gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
|
|
275
316
|
// "length" so its truncation check (=== "length") is a cross-backend invariant.
|
|
276
|
-
const p = new
|
|
317
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
277
318
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
|
|
278
319
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
279
320
|
assert.equal(assistant.finishReason, "length");
|
|
280
321
|
});
|
|
281
322
|
|
|
282
323
|
test("generate translates end_turn to canonical stop (#425)", async () => {
|
|
283
|
-
const p = new
|
|
324
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
284
325
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
|
|
285
326
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
286
327
|
assert.equal(assistant.finishReason, "stop");
|
|
287
328
|
});
|
|
288
329
|
|
|
289
330
|
test("generate aggregates reasoning deltas under multiple field names", async () => {
|
|
290
|
-
const p = new
|
|
331
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
291
332
|
installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
|
|
292
333
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
293
334
|
assert.equal(assistant.reasoning, "because");
|
|
@@ -303,7 +344,7 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
|
|
|
303
344
|
{ type: "reasoning.text", text: "never surfaced here" },
|
|
304
345
|
],
|
|
305
346
|
}, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
306
|
-
const p = new
|
|
347
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
307
348
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
308
349
|
// item shape: wire `id` preserved, subtype from position (#482 widening)
|
|
309
350
|
assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
|
|
@@ -316,14 +357,14 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
|
|
|
316
357
|
{ type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
|
|
317
358
|
{ type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
|
|
318
359
|
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
319
|
-
const p = new
|
|
360
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
320
361
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
321
362
|
assert.equal(assistant.reasoningEncrypted?.length, 2);
|
|
322
363
|
assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
|
|
323
364
|
});
|
|
324
365
|
|
|
325
366
|
test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entry index", async () => {
|
|
326
|
-
const p = new
|
|
367
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
327
368
|
installFetch([
|
|
328
369
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
|
|
329
370
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
|
|
@@ -335,20 +376,20 @@ test("#482 sealed relay reasoning (streamed): chunked blob concatenates per entr
|
|
|
335
376
|
});
|
|
336
377
|
|
|
337
378
|
test("reasoningStyle 'think' gates on budget != 0 (magnitude irrelevant for native)", async () => {
|
|
338
|
-
const on = new
|
|
379
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
339
380
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
340
381
|
await on.generate({ workerId: "r", messages: [] });
|
|
341
382
|
assert.equal(JSON.parse(calls[0].init.body as string).think, true);
|
|
342
383
|
|
|
343
384
|
mock.restoreAll();
|
|
344
|
-
const off = new
|
|
385
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
345
386
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
346
387
|
await off.generate({ workerId: "r", messages: [] });
|
|
347
388
|
assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
|
|
348
389
|
});
|
|
349
390
|
|
|
350
391
|
test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", async () => {
|
|
351
|
-
const p = new
|
|
392
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 5000 }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
352
393
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
353
394
|
await p.generate({ workerId: "r", messages: [] });
|
|
354
395
|
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
@@ -359,7 +400,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
359
400
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
360
401
|
// #403): adaptive = the backend's own default posture = omission.
|
|
361
402
|
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
362
|
-
const p = new
|
|
403
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
363
404
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
364
405
|
await p.generate({ workerId: "r", messages: [] });
|
|
365
406
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -370,7 +411,7 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
370
411
|
});
|
|
371
412
|
|
|
372
413
|
test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
|
|
373
|
-
const p = new
|
|
414
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
374
415
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
375
416
|
await p.generate({ workerId: "r", messages: [] });
|
|
376
417
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
@@ -387,9 +428,9 @@ test("the family temperature default rides every request; caller sampling overri
|
|
|
387
428
|
});
|
|
388
429
|
|
|
389
430
|
test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
|
|
390
|
-
const base = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
431
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
|
|
391
432
|
// set + llamacpp -> the loop-breakers ride the wire
|
|
392
|
-
const p = new
|
|
433
|
+
const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
|
|
393
434
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
394
435
|
await p.generate({ workerId: "r", messages: [] });
|
|
395
436
|
let body = JSON.parse(calls[0].init.body as string);
|
|
@@ -400,7 +441,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
400
441
|
assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
|
|
401
442
|
mock.restoreAll();
|
|
402
443
|
// unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
|
|
403
|
-
const p2 = new
|
|
444
|
+
const p2 = new AiSdkProvider({ ...base, grammarStyle: "llamacpp" });
|
|
404
445
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
405
446
|
await p2.generate({ workerId: "r", messages: [] });
|
|
406
447
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -408,7 +449,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
408
449
|
assert.equal("repeat_last_n" in body, false);
|
|
409
450
|
mock.restoreAll();
|
|
410
451
|
// DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
|
|
411
|
-
const p3 = new
|
|
452
|
+
const p3 = new AiSdkProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
|
|
412
453
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
413
454
|
await p3.generate({ workerId: "r", messages: [] });
|
|
414
455
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -418,7 +459,7 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
418
459
|
});
|
|
419
460
|
|
|
420
461
|
test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
|
|
421
|
-
const p = new
|
|
462
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
422
463
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
423
464
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
424
465
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -428,13 +469,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
428
469
|
|
|
429
470
|
test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
|
|
430
471
|
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
431
|
-
const llama = new
|
|
472
|
+
const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
432
473
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
433
474
|
await llama.generate({ workerId: "r", messages: [] });
|
|
434
475
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
435
476
|
mock.restoreAll();
|
|
436
477
|
// a `none`-style cloud backend WITH a frequency penalty gets frequency_penalty (OpenAI-standard, #426)
|
|
437
|
-
const cloud = new
|
|
478
|
+
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
438
479
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
439
480
|
await cloud.generate({ workerId: "r", messages: [] });
|
|
440
481
|
const cloudBody = JSON.parse(calls[0].init.body as string);
|
|
@@ -443,14 +484,14 @@ test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (
|
|
|
443
484
|
assert.equal("repeat_penalty" in cloudBody, false);
|
|
444
485
|
mock.restoreAll();
|
|
445
486
|
// frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
|
|
446
|
-
const bare = new
|
|
487
|
+
const bare = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
447
488
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
448
489
|
await bare.generate({ workerId: "r", messages: [] });
|
|
449
490
|
assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
|
|
450
491
|
});
|
|
451
492
|
|
|
452
493
|
test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
|
|
453
|
-
const p = new
|
|
494
|
+
const p = new AiSdkProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
454
495
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
455
496
|
await p.generate({
|
|
456
497
|
workerId: "r",
|
|
@@ -473,7 +514,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
|
|
|
473
514
|
});
|
|
474
515
|
|
|
475
516
|
test("#477 sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
|
|
476
|
-
const p = new
|
|
517
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
477
518
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
478
519
|
await p.generate({
|
|
479
520
|
workerId: "r",
|
|
@@ -500,7 +541,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
500
541
|
test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — sanctioned channel coexists with rails", async () => {
|
|
501
542
|
// The brief rails-win-the-channel clamp is REVERTED: closing the channel starved a
|
|
502
543
|
// reasoning-tuned model into escaping mid-content (unconstrained, discarded, billed).
|
|
503
|
-
const p = new
|
|
544
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
504
545
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
505
546
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
506
547
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -514,7 +555,7 @@ test("#488 postmortem: intent maps IDENTICALLY under a transported grammar — s
|
|
|
514
555
|
test("#488 channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
|
|
515
556
|
// The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
|
|
516
557
|
// escaped into a discarded reasoning block, unconstrained.
|
|
517
|
-
const p = new
|
|
558
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
518
559
|
installFetch([
|
|
519
560
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
520
561
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
@@ -528,7 +569,7 @@ test("#488 channel-escape detector: billed completion tokens vastly beyond visib
|
|
|
528
569
|
});
|
|
529
570
|
|
|
530
571
|
test("#488 loud state absent on grammarless calls; no escape event without a transported grammar", async () => {
|
|
531
|
-
const p = new
|
|
572
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
532
573
|
installFetch([
|
|
533
574
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
534
575
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
@@ -539,33 +580,33 @@ test("#488 loud state absent on grammarless calls; no escape event without a tra
|
|
|
539
580
|
});
|
|
540
581
|
|
|
541
582
|
test("reasoningStyle 'template' always emits enable_thinking mirroring budget != 0 — explicit false, never omitted", async () => {
|
|
542
|
-
const on = new
|
|
583
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
543
584
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
544
585
|
await on.generate({ workerId: "r", messages: [] });
|
|
545
586
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: true });
|
|
546
587
|
|
|
547
588
|
mock.restoreAll();
|
|
548
|
-
const off = new
|
|
589
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
549
590
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
550
591
|
await off.generate({ workerId: "r", messages: [] });
|
|
551
592
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).chat_template_kwargs, { enable_thinking: false });
|
|
552
593
|
});
|
|
553
594
|
|
|
554
595
|
test("budget 0 suppresses effort and include_reasoning", async () => {
|
|
555
|
-
const effort = new
|
|
596
|
+
const effort = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
556
597
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
557
598
|
await effort.generate({ workerId: "r", messages: [] });
|
|
558
599
|
assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
|
|
559
600
|
|
|
560
601
|
mock.restoreAll();
|
|
561
|
-
const relay = new
|
|
602
|
+
const relay = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
562
603
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
563
604
|
await relay.generate({ workerId: "r", messages: [] });
|
|
564
605
|
assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
|
|
565
606
|
});
|
|
566
607
|
|
|
567
608
|
test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
|
|
568
|
-
const p = new
|
|
609
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
569
610
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
570
611
|
await p.generate({ workerId: "r", messages: [] });
|
|
571
612
|
assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
|
|
@@ -574,7 +615,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
|
|
|
574
615
|
// — grammar-constrained sampling (SPEC §13, issues #8/#9) —
|
|
575
616
|
|
|
576
617
|
test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
|
|
577
|
-
const p = new
|
|
618
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
578
619
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
579
620
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
580
621
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -584,7 +625,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
|
|
|
584
625
|
});
|
|
585
626
|
|
|
586
627
|
test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
|
|
587
|
-
const p = new
|
|
628
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
588
629
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
589
630
|
await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
|
|
590
631
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -595,7 +636,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
|
|
|
595
636
|
// — grammar conformance OBSERVATION (SPEC §10.14, §13): a completed exchange always
|
|
596
637
|
// returns; bytes flow; a non-accept verdict rides response.telemetry —
|
|
597
638
|
|
|
598
|
-
const grammarProvider = () => new
|
|
639
|
+
const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
|
|
599
640
|
const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
600
641
|
|
|
601
642
|
test("enforcement: conforming output passes through unchanged", async () => {
|
|
@@ -644,7 +685,7 @@ test("observation: empty content under a non-empty grammar returns with the verd
|
|
|
644
685
|
});
|
|
645
686
|
|
|
646
687
|
test("enforcement: when no grammar is sent (grammarStyle 'none'), output is NOT validated — no wire fields, no error (SPEC )", async () => {
|
|
647
|
-
const p = new
|
|
688
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
|
|
648
689
|
streamingContent("anything goes");
|
|
649
690
|
const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' }); // grammar passed but never transported
|
|
650
691
|
assert.equal(assistant.content, "anything goes"); // no enforcement check
|
|
@@ -666,7 +707,7 @@ test("enforcement: a grammar our validator can't parse is a NON-FATAL verify gap
|
|
|
666
707
|
// — PLURNK_PROVIDERS_GBNF_DEBUG: run unconstrained, then verify the free output against the grammar —
|
|
667
708
|
|
|
668
709
|
test("gbnfDebug: the grammar is NOT transported; conforming free output passes through with NO telemetry", async () => {
|
|
669
|
-
const p = new
|
|
710
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
670
711
|
const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
|
|
671
712
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
672
713
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -677,7 +718,7 @@ test("gbnfDebug: the grammar is NOT transported; conforming free output passes t
|
|
|
677
718
|
});
|
|
678
719
|
|
|
679
720
|
test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a grammar_unenforced telemetry event with the divergence position (#24)", async () => {
|
|
680
|
-
const p = new
|
|
721
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
681
722
|
const calls = installFetch([{ choices: [{ delta: { reasoning_content: "let me think about ok", content: "xon-conforming output" }, finish_reason: "stop" }] }]);
|
|
682
723
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
683
724
|
// The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
|
|
@@ -695,7 +736,7 @@ test("gbnfDebug: a conflict does NOT throw — it returns the bytes plus a gramm
|
|
|
695
736
|
});
|
|
696
737
|
|
|
697
738
|
test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
|
|
698
|
-
const p = new
|
|
739
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
|
|
699
740
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
700
741
|
await assert.rejects(
|
|
701
742
|
() => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
|
|
@@ -704,31 +745,15 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
704
745
|
assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
|
|
705
746
|
});
|
|
706
747
|
|
|
707
|
-
// — meta bag:
|
|
748
|
+
// — meta bag: verbatim provider metadata (#23) —
|
|
708
749
|
|
|
709
|
-
test("meta:
|
|
710
|
-
const p = new
|
|
711
|
-
|
|
750
|
+
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
751
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
752
|
+
const balance = { amount: "0.0000042", currency: "XMR" };
|
|
753
|
+
installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
|
|
712
754
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
713
|
-
assert.
|
|
714
|
-
assert.equal("balance_pico" in (res.meta ?? {}), false); // raw key renamed to the canonical balancePico
|
|
715
|
-
});
|
|
716
|
-
|
|
717
|
-
test("meta: passes the backend's extra top-level fields through verbatim (every provider)", async () => {
|
|
718
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false }); // no balanceMetaKey
|
|
719
|
-
installFetchJson({ ...jsonChoice, balance_pico: 4_200_000, system_fingerprint: "fp_abc" });
|
|
720
|
-
const res = await p.generate({ workerId: "r", messages: [] });
|
|
721
|
-
assert.equal(res.meta?.balance_pico, 4_200_000); // passed through raw — no balance contract on this provider
|
|
755
|
+
assert.deepEqual(res.meta?.balance, balance);
|
|
722
756
|
assert.equal(res.meta?.system_fingerprint, "fp_abc");
|
|
723
|
-
assert.equal("balancePico" in (res.meta ?? {}), false); // not normalized without the key
|
|
724
|
-
});
|
|
725
|
-
|
|
726
|
-
test("meta: a non-numeric balance is dropped, never surfaced as balancePico (null-honest)", async () => {
|
|
727
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
|
|
728
|
-
installFetchJson({ ...jsonChoice, balance_pico: "lots" });
|
|
729
|
-
const res = await p.generate({ workerId: "r", messages: [] });
|
|
730
|
-
assert.equal("balancePico" in (res.meta ?? {}), false);
|
|
731
|
-
assert.equal("balance_pico" in (res.meta ?? {}), false); // raw dropped too — the known key is validated away
|
|
732
757
|
});
|
|
733
758
|
|
|
734
759
|
// — first-party telemetry headers (attribution + client, SPEC §5) —
|
|
@@ -737,7 +762,7 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
|
|
|
737
762
|
new Headers(init.headers).get(name) ?? undefined;
|
|
738
763
|
|
|
739
764
|
test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
|
|
740
|
-
const p = new
|
|
765
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
741
766
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
742
767
|
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
|
|
743
768
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
|
|
@@ -745,7 +770,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
|
|
|
745
770
|
});
|
|
746
771
|
|
|
747
772
|
test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
|
|
748
|
-
const p = new
|
|
773
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
749
774
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
750
775
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
751
776
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
|
|
@@ -764,14 +789,14 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
|
|
|
764
789
|
});
|
|
765
790
|
|
|
766
791
|
test("#522 Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
|
|
767
|
-
const p = new
|
|
792
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
768
793
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
769
794
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
770
795
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
|
|
771
796
|
});
|
|
772
797
|
|
|
773
798
|
test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
|
|
774
|
-
const p = new
|
|
799
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
775
800
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
776
801
|
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
|
|
777
802
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
|
|
@@ -779,7 +804,7 @@ test("firstPartyMetadata off (default): the headers are structurally dropped eve
|
|
|
779
804
|
});
|
|
780
805
|
|
|
781
806
|
test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
782
|
-
const p = new
|
|
807
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
783
808
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
784
809
|
await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
|
|
785
810
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
|
|
@@ -787,7 +812,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
|
787
812
|
});
|
|
788
813
|
|
|
789
814
|
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides (#426)", async () => {
|
|
790
|
-
const p = new
|
|
815
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
791
816
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
792
817
|
await p.generate({ workerId: "r", messages: [] });
|
|
793
818
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -796,7 +821,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
|
|
|
796
821
|
});
|
|
797
822
|
|
|
798
823
|
test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
|
|
799
|
-
const p = new
|
|
824
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
800
825
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
801
826
|
await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
|
|
802
827
|
assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
|
|
@@ -808,7 +833,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
|
|
|
808
833
|
});
|
|
809
834
|
|
|
810
835
|
test("slot affinity is internal: sticky per workerId, distinct runs spread across slots (#11)", async () => {
|
|
811
|
-
const pinning = new
|
|
836
|
+
const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
812
837
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
813
838
|
await pinning.generate({ workerId: "run-A", messages: [] });
|
|
814
839
|
await pinning.generate({ workerId: "run-B", messages: [] });
|
|
@@ -819,20 +844,20 @@ test("slot affinity is internal: sticky per workerId, distinct runs spread acros
|
|
|
819
844
|
});
|
|
820
845
|
|
|
821
846
|
test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
|
|
822
|
-
const cloud = new
|
|
847
|
+
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
|
|
823
848
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
824
849
|
await cloud.generate({ workerId: "run-A", messages: [] });
|
|
825
850
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
826
851
|
|
|
827
852
|
mock.restoreAll();
|
|
828
|
-
const noCount = new
|
|
853
|
+
const noCount = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
|
|
829
854
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
830
855
|
await noCount.generate({ workerId: "run-A", messages: [] });
|
|
831
856
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
832
857
|
});
|
|
833
858
|
|
|
834
859
|
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent runs stay sticky (#11)", async () => {
|
|
835
|
-
const p = new
|
|
860
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
836
861
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
837
862
|
const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
|
|
838
863
|
for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
|
|
@@ -846,7 +871,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
|
|
|
846
871
|
|
|
847
872
|
test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
|
|
848
873
|
const { ProviderError } = await import("./telemetry.ts");
|
|
849
|
-
const p = new
|
|
874
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
|
|
850
875
|
mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
|
|
851
876
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
852
877
|
assert.ok(err instanceof ProviderError);
|
|
@@ -857,14 +882,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
|
|
|
857
882
|
});
|
|
858
883
|
|
|
859
884
|
test("generate fail-hards on a missing or empty workerId", async () => {
|
|
860
|
-
const p = new
|
|
885
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
861
886
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
862
887
|
await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
|
|
863
888
|
await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
|
|
864
889
|
});
|
|
865
890
|
|
|
866
891
|
test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
|
|
867
|
-
const p = new
|
|
892
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
868
893
|
const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
|
|
869
894
|
const input = [{ role: "user" as const, content: "hi" }];
|
|
870
895
|
const res = await p.generate({ workerId: "r", messages: input });
|
|
@@ -874,7 +899,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
|
|
|
874
899
|
|
|
875
900
|
test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEvent", async () => {
|
|
876
901
|
const { ProviderError } = await import("./telemetry.ts");
|
|
877
|
-
const p = new
|
|
902
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
|
|
878
903
|
mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
|
|
879
904
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
880
905
|
assert.ok(err instanceof ProviderError);
|
|
@@ -886,27 +911,28 @@ test("generate wraps an HTTP failure as a ProviderError carrying a TelemetryEven
|
|
|
886
911
|
});
|
|
887
912
|
|
|
888
913
|
test("generate rejects on a pre-aborted external signal", async () => {
|
|
889
|
-
const p = new
|
|
914
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
890
915
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
891
916
|
const signal = AbortSignal.abort(new Error("nope"));
|
|
892
917
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
|
|
893
918
|
});
|
|
894
919
|
|
|
895
920
|
test("configured headers and url are sent verbatim", async () => {
|
|
896
|
-
const p = new
|
|
897
|
-
model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
921
|
+
const p = new AiSdkProvider({
|
|
922
|
+
model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
898
923
|
headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
|
|
899
924
|
});
|
|
900
925
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
901
926
|
await p.generate({ workerId: "r", messages: [] });
|
|
902
927
|
assert.equal(calls[0].url, "http://host/custom/chat/completions");
|
|
903
|
-
|
|
904
|
-
assert.equal((
|
|
928
|
+
const headers = new Headers(calls[0].init.headers);
|
|
929
|
+
assert.equal(headers.get("authorization"), "Bearer secret");
|
|
930
|
+
assert.equal(headers.get("x-title"), "plurnk");
|
|
905
931
|
});
|
|
906
932
|
|
|
907
933
|
// — transient-failure retry (#18) —
|
|
908
934
|
|
|
909
|
-
const retryCfg = { model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15,
|
|
935
|
+
const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
|
|
910
936
|
|
|
911
937
|
test("retry: a transient failure retries and a later success resolves", async () => {
|
|
912
938
|
const calls = installFetchScript([
|
|
@@ -914,51 +940,52 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
914
940
|
{ status: 503, retryAfter: 0 },
|
|
915
941
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
|
|
916
942
|
]);
|
|
917
|
-
const p = new
|
|
943
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
918
944
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
919
945
|
assert.equal(res.assistant.content, "ok");
|
|
920
946
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
921
947
|
});
|
|
922
948
|
|
|
923
|
-
test("#559: streamed-body silence
|
|
949
|
+
test("#559: streamed-body silence fails the exchange without replaying partial output", async () => {
|
|
924
950
|
let calls = 0;
|
|
925
951
|
mock.method(globalThis, "fetch", async () => {
|
|
926
952
|
calls++;
|
|
927
953
|
if (calls === 1) {
|
|
928
954
|
return new Response(new ReadableStream({
|
|
929
955
|
start(controller) {
|
|
930
|
-
controller.enqueue(new TextEncoder().encode(
|
|
956
|
+
controller.enqueue(new TextEncoder().encode(
|
|
957
|
+
'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
958
|
+
));
|
|
959
|
+
setTimeout(() => controller.close(), 100);
|
|
931
960
|
},
|
|
932
961
|
}), { status: 200 });
|
|
933
962
|
}
|
|
934
963
|
return new Response(new ReadableStream({
|
|
935
964
|
start(controller) {
|
|
936
|
-
controller.enqueue(new TextEncoder().encode(
|
|
965
|
+
controller.enqueue(new TextEncoder().encode(
|
|
966
|
+
'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
|
|
967
|
+
));
|
|
937
968
|
controller.close();
|
|
938
969
|
},
|
|
939
970
|
}), { status: 200 });
|
|
940
971
|
});
|
|
941
|
-
const p = new
|
|
972
|
+
const p = new AiSdkProvider({
|
|
942
973
|
model: "m",
|
|
943
|
-
url: "http://x",
|
|
974
|
+
url: "http://x/v1/chat/completions",
|
|
944
975
|
fetchTimeoutMs: 1000,
|
|
945
976
|
streamIdleTimeoutMs: 10,
|
|
946
977
|
temperature: 0.2,
|
|
947
978
|
repeatPenalty: 1.15,
|
|
948
|
-
retryDelayMs: 1,
|
|
949
979
|
reasoning: { mode: "off", budget: null },
|
|
950
980
|
retryAttempts: 1,
|
|
951
981
|
source: "provider:test",
|
|
952
982
|
});
|
|
953
|
-
|
|
954
|
-
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
assert.equal(
|
|
959
|
-
assert.equal(retries[0].kind, "network_failure");
|
|
960
|
-
assert.equal(typeof retries[0].elapsedMs, "number");
|
|
961
|
-
assert.match(String(retries[0].message), /no body bytes for 10ms/);
|
|
983
|
+
await assert.rejects(
|
|
984
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
985
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
986
|
+
&& /chunk timeout/i.test(error.message),
|
|
987
|
+
);
|
|
988
|
+
assert.equal(calls, 1);
|
|
962
989
|
mock.restoreAll();
|
|
963
990
|
});
|
|
964
991
|
|
|
@@ -971,27 +998,25 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
|
|
|
971
998
|
controller.close();
|
|
972
999
|
},
|
|
973
1000
|
}), { status: 200 }));
|
|
974
|
-
const p = new
|
|
1001
|
+
const p = new AiSdkProvider({
|
|
975
1002
|
model: "m",
|
|
976
|
-
url: "http://x",
|
|
1003
|
+
url: "http://x/v1/chat/completions",
|
|
977
1004
|
fetchTimeoutMs: 1000,
|
|
978
1005
|
streamIdleTimeoutMs: 0,
|
|
979
1006
|
temperature: 0.2,
|
|
980
1007
|
repeatPenalty: 1.15,
|
|
981
|
-
retryDelayMs: 1,
|
|
982
1008
|
reasoning: { mode: "off", budget: null },
|
|
983
1009
|
retryAttempts: 0,
|
|
984
1010
|
});
|
|
985
1011
|
const result = await p.generate({ workerId: "r", messages: [] });
|
|
986
1012
|
assert.equal(result.assistant.content, "slow is valid");
|
|
987
|
-
assert.equal(result.meta?.transportRetries, undefined);
|
|
988
1013
|
mock.restoreAll();
|
|
989
1014
|
});
|
|
990
1015
|
|
|
991
1016
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
992
1017
|
const { ProviderError } = await import("./telemetry.ts");
|
|
993
1018
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
|
994
|
-
const p = new
|
|
1019
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
|
|
995
1020
|
await assert.rejects(
|
|
996
1021
|
() => p.generate({ workerId: "r", messages: [] }),
|
|
997
1022
|
(err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
|
|
@@ -1004,7 +1029,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
1004
1029
|
{ status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
|
|
1005
1030
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
|
|
1006
1031
|
]);
|
|
1007
|
-
const p = new
|
|
1032
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 1 });
|
|
1008
1033
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
1009
1034
|
assert.equal(assistant.content, "ok");
|
|
1010
1035
|
assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
|
|
@@ -1012,14 +1037,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
1012
1037
|
|
|
1013
1038
|
test("retry: a terminal error (401 unauthorized) is never retried", async () => {
|
|
1014
1039
|
const calls = installFetchScript([{ status: 401 }]);
|
|
1015
|
-
const p = new
|
|
1040
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 5 });
|
|
1016
1041
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
|
|
1017
1042
|
assert.equal(calls.length, 1); // terminal — no retry despite budget
|
|
1018
1043
|
});
|
|
1019
1044
|
|
|
1020
1045
|
test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
|
|
1021
1046
|
const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
|
|
1022
|
-
const p = new
|
|
1047
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 0 });
|
|
1023
1048
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
|
|
1024
1049
|
assert.equal(calls.length, 1); // no retry budget
|
|
1025
1050
|
});
|
|
@@ -1027,7 +1052,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
|
|
|
1027
1052
|
test("retry: a caller abort during backoff rejects promptly with no further attempt (mid-flight abort, SPEC )", async () => {
|
|
1028
1053
|
const ac = new AbortController();
|
|
1029
1054
|
const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
|
|
1030
|
-
const p = new
|
|
1055
|
+
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
1031
1056
|
const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
|
|
1032
1057
|
await flush(); // attempt 0 fails, enters the backoff sleep
|
|
1033
1058
|
assert.equal(calls.length, 1);
|
|
@@ -1040,21 +1065,21 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
1040
1065
|
|
|
1041
1066
|
test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
|
|
1042
1067
|
// N>0 → enabled with budget_tokens
|
|
1043
|
-
const capped = new
|
|
1068
|
+
const capped = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
|
|
1044
1069
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1045
1070
|
await capped.generate({ workerId: "r", messages: [] });
|
|
1046
1071
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
|
|
1047
1072
|
|
|
1048
1073
|
mock.restoreAll();
|
|
1049
1074
|
// 0 → explicit disabled
|
|
1050
|
-
const off = new
|
|
1075
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
|
|
1051
1076
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1052
1077
|
await off.generate({ workerId: "r", messages: [] });
|
|
1053
1078
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
|
|
1054
1079
|
|
|
1055
1080
|
mock.restoreAll();
|
|
1056
1081
|
// -1 adaptive → omit (API default depth)
|
|
1057
|
-
const adaptive = new
|
|
1082
|
+
const adaptive = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
|
|
1058
1083
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1059
1084
|
await adaptive.generate({ workerId: "r", messages: [] });
|
|
1060
1085
|
assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
|
|
@@ -1072,7 +1097,7 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1072
1097
|
usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
|
|
1073
1098
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
1074
1099
|
});
|
|
1075
|
-
const p = new
|
|
1100
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
1076
1101
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
1077
1102
|
const sent = JSON.parse(calls[0].body);
|
|
1078
1103
|
assert.equal("stream" in sent, false); // no streaming flag
|
|
@@ -1084,11 +1109,11 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1084
1109
|
});
|
|
1085
1110
|
|
|
1086
1111
|
// ── Data capture (#36): logprobs + verbatim rawBody, opt-in, off by default ──
|
|
1087
|
-
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
1112
|
+
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1088
1113
|
|
|
1089
1114
|
test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
|
|
1090
1115
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1091
|
-
const p = new
|
|
1116
|
+
const p = new AiSdkProvider({ ...captureBase });
|
|
1092
1117
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1093
1118
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1094
1119
|
assert.equal("logprobs" in body, false);
|
|
@@ -1105,7 +1130,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1105
1130
|
{ token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
|
|
1106
1131
|
] } }] };
|
|
1107
1132
|
const calls = installFetch([chunk]);
|
|
1108
|
-
const p = new
|
|
1133
|
+
const p = new AiSdkProvider({ ...captureBase, topLogprobs: 2 });
|
|
1109
1134
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1110
1135
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1111
1136
|
assert.equal(body.logprobs, true);
|
|
@@ -1119,7 +1144,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1119
1144
|
test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
|
|
1120
1145
|
const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
1121
1146
|
installFetchJson(wire);
|
|
1122
|
-
const p = new
|
|
1147
|
+
const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
|
|
1123
1148
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1124
1149
|
assert.deepEqual(res.rawBody, wire); // verbatim
|
|
1125
1150
|
assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
|
|
@@ -1130,7 +1155,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
|
|
|
1130
1155
|
|
|
1131
1156
|
test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
|
|
1132
1157
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1133
|
-
const p = new
|
|
1158
|
+
const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
|
|
1134
1159
|
await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
|
|
1135
1160
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1136
1161
|
assert.equal("logprobs" in body, false); // sampling passthrough stripped it
|
|
@@ -1141,52 +1166,52 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
|
|
|
1141
1166
|
// — turn coordinate headers (#404, per #391): same gate as every first-party signal —
|
|
1142
1167
|
|
|
1143
1168
|
test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
|
|
1144
|
-
const p = new
|
|
1169
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1145
1170
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1146
1171
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1147
|
-
const
|
|
1148
|
-
assert.equal(
|
|
1149
|
-
assert.equal(
|
|
1150
|
-
assert.equal(
|
|
1172
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1173
|
+
assert.equal(headers.get("plurnk-workspace-id"), "s-9");
|
|
1174
|
+
assert.equal(headers.get("plurnk-loop"), "3");
|
|
1175
|
+
assert.equal(headers.get("plurnk-turn"), "41");
|
|
1151
1176
|
});
|
|
1152
1177
|
|
|
1153
1178
|
test("#404: third-party providers structurally DROP the coordinate (gate off by default)", async () => {
|
|
1154
|
-
const p = new
|
|
1179
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1155
1180
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1156
1181
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1157
|
-
const
|
|
1158
|
-
assert.equal("
|
|
1159
|
-
assert.equal("
|
|
1160
|
-
assert.equal("
|
|
1182
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1183
|
+
assert.equal(headers.has("plurnk-workspace-id"), false);
|
|
1184
|
+
assert.equal(headers.has("plurnk-loop"), false);
|
|
1185
|
+
assert.equal(headers.has("plurnk-turn"), false);
|
|
1161
1186
|
});
|
|
1162
1187
|
|
|
1163
1188
|
test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strikes-style zero exception)", async () => {
|
|
1164
|
-
const p = new
|
|
1189
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1165
1190
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1166
1191
|
await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
|
|
1167
|
-
const
|
|
1168
|
-
assert.equal("
|
|
1169
|
-
assert.equal("
|
|
1170
|
-
assert.equal("
|
|
1171
|
-
assert.equal(
|
|
1192
|
+
const headers = new Headers(calls[0].init.headers);
|
|
1193
|
+
assert.equal(headers.has("plurnk-workspace-id"), false);
|
|
1194
|
+
assert.equal(headers.has("plurnk-loop"), false);
|
|
1195
|
+
assert.equal(headers.has("plurnk-turn"), false);
|
|
1196
|
+
assert.equal(headers.has("plurnk-strikes"), false);
|
|
1172
1197
|
});
|
|
1173
1198
|
|
|
1174
1199
|
// -- #507: envelope surface + router-owned tuning --
|
|
1175
1200
|
|
|
1176
1201
|
test("#507 reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
|
|
1177
|
-
const base = { model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15,
|
|
1178
|
-
const derived = new
|
|
1202
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1203
|
+
const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
|
|
1179
1204
|
assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
|
|
1180
1205
|
assert.equal(derived.completionReserve, 12288); // 25% of 49152
|
|
1181
|
-
const pinned = new
|
|
1206
|
+
const pinned = new AiSdkProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
|
|
1182
1207
|
assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
|
|
1183
1208
|
assert.equal(pinned.completionReserve, null); // percent without a window = underivable
|
|
1184
|
-
const legacy = new
|
|
1209
|
+
const legacy = new AiSdkProvider({ ...base, contextWindow: 49152 });
|
|
1185
1210
|
assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
|
|
1186
1211
|
});
|
|
1187
1212
|
|
|
1188
1213
|
test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
|
|
1189
|
-
const p = new
|
|
1214
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
|
|
1190
1215
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1191
1216
|
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
|
|
1192
1217
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1197,21 +1222,21 @@ test("#507 router-owned tuning: tuningFloors:false drops the temperature/penalty
|
|
|
1197
1222
|
// -- #518: prompt-cache affinity (workerId -> prompt_cache_key) --
|
|
1198
1223
|
|
|
1199
1224
|
test("#518 promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
|
|
1200
|
-
const p = new
|
|
1225
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1201
1226
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1202
1227
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1203
1228
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
|
|
1204
1229
|
});
|
|
1205
1230
|
|
|
1206
1231
|
test("#518 promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
|
|
1207
|
-
const p = new
|
|
1232
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1208
1233
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1209
1234
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1210
1235
|
assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
|
|
1211
1236
|
});
|
|
1212
1237
|
|
|
1213
1238
|
test("#518 prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
|
|
1214
|
-
const p = new
|
|
1239
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1215
1240
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1216
1241
|
await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
|
|
1217
1242
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc"); // managed wins
|