@plurnk/plurnk-providers 1.5.0 → 1.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +36 -22
- package/SPEC.md +133 -59
- package/dist/AiSdkProvider.d.ts +19 -26
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +318 -106
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +4 -9
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +36 -9
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +2 -21
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +19 -14
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +5 -2
- package/dist/accounting.d.ts.map +1 -1
- package/dist/accounting.js +100 -16
- package/dist/accounting.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +9 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +160 -62
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +7 -3
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +30 -24
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +18 -7
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +10 -10
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +90 -42
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +5 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +30 -10
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +14 -2
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +58 -2
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +4 -4
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -2
- package/dist/index.js.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/sdkModels.d.ts +6 -2
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +38 -5
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +33 -31
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +21 -5
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +164 -83
- package/dist/usage.js.map +1 -1
- package/package.json +7 -6
- package/src/AiSdkProvider.test.ts +788 -191
- package/src/AiSdkProvider.ts +381 -124
- package/src/Mock.test.ts +37 -12
- package/src/Mock.ts +45 -14
- package/src/Pool.test.ts +19 -6
- package/src/Pool.ts +20 -16
- package/src/ProviderRegistry.test.ts +16 -11
- package/src/accounting.test.ts +58 -22
- package/src/accounting.ts +120 -18
- package/src/aiSdkTransport.test.ts +42 -49
- package/src/aiSdkTransport.ts +174 -62
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +258 -22
- package/src/catalogProvider.ts +42 -27
- package/src/compatibleProvider.test.ts +6 -3
- package/src/compatibleProvider.ts +20 -7
- package/src/cost.test.ts +55 -36
- package/src/cost.ts +111 -50
- package/src/defaults.test.ts +13 -3
- package/src/env.test.ts +54 -5
- package/src/env.ts +43 -18
- package/src/errors.test.ts +47 -2
- package/src/errors.ts +67 -3
- package/src/index.ts +21 -5
- package/src/ollama.test.ts +4 -1
- package/src/ollama.ts +3 -3
- package/src/sdkModels.test.ts +76 -4
- package/src/sdkModels.ts +45 -7
- package/src/types.ts +77 -38
- package/src/usage.test.ts +112 -116
- package/src/usage.ts +209 -93
|
@@ -1,10 +1,26 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
|
-
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
3
|
+
import AiSdkProvider, { effortFromBudget, type AiSdkProviderConfig } from "./AiSdkProvider.ts";
|
|
4
4
|
import { ProviderError } from "./errors.ts";
|
|
5
|
-
import {
|
|
5
|
+
import { providerCostNormalizer } from "./accounting.ts";
|
|
6
6
|
import type { LanguageModel } from "ai";
|
|
7
7
|
|
|
8
|
+
type TestProviderConfig = Omit<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">
|
|
9
|
+
& Partial<Pick<AiSdkProviderConfig, "operationTimeoutMs" | "firstContentTimeoutMs">>;
|
|
10
|
+
|
|
11
|
+
const testProvider = (config: TestProviderConfig): AiSdkProvider => {
|
|
12
|
+
const {
|
|
13
|
+
operationTimeoutMs = config.fetchTimeoutMs,
|
|
14
|
+
firstContentTimeoutMs = 0,
|
|
15
|
+
...rest
|
|
16
|
+
} = config;
|
|
17
|
+
return new AiSdkProvider({
|
|
18
|
+
...rest,
|
|
19
|
+
operationTimeoutMs,
|
|
20
|
+
firstContentTimeoutMs,
|
|
21
|
+
});
|
|
22
|
+
};
|
|
23
|
+
|
|
8
24
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
9
25
|
// so tests can assert what the spine sent on the wire.
|
|
10
26
|
const sseStream = (chunks: unknown[]) => {
|
|
@@ -61,6 +77,31 @@ const installFetchJson = (payload: unknown) => {
|
|
|
61
77
|
return calls;
|
|
62
78
|
};
|
|
63
79
|
|
|
80
|
+
const settledCharge = {
|
|
81
|
+
kind: "charged",
|
|
82
|
+
amount: { amount: "0.00000042", currency: "XMR" },
|
|
83
|
+
usdEquivalent: "0.000071",
|
|
84
|
+
source: "plurnk endpoint settlement",
|
|
85
|
+
} as const;
|
|
86
|
+
|
|
87
|
+
const billedErrorBody = {
|
|
88
|
+
status: 422,
|
|
89
|
+
error: {
|
|
90
|
+
message: "non-conforming emission rejected",
|
|
91
|
+
type: "grammar_invalid",
|
|
92
|
+
},
|
|
93
|
+
usage: {
|
|
94
|
+
prompt_tokens: 8,
|
|
95
|
+
completion_tokens: 3,
|
|
96
|
+
reasoning_tokens: 0,
|
|
97
|
+
prompt_tokens_details: { cached_tokens: 2 },
|
|
98
|
+
total_tokens: 11,
|
|
99
|
+
},
|
|
100
|
+
charge: settledCharge,
|
|
101
|
+
};
|
|
102
|
+
|
|
103
|
+
const directCost = ({ charge }: { charge?: unknown }) => charge as typeof settledCharge | undefined;
|
|
104
|
+
|
|
64
105
|
const jsonChoice = { model: "m", choices: [{ message: { content: "x" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
65
106
|
|
|
66
107
|
const injectedBase = {
|
|
@@ -90,9 +131,9 @@ test("per-instance fetch owns streaming and buffered requests", async () => {
|
|
|
90
131
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
91
132
|
};
|
|
92
133
|
|
|
93
|
-
const streamed = await
|
|
134
|
+
const streamed = await testProvider({ ...injectedBase, fetch: streamingFetch, rawBody: true })
|
|
94
135
|
.generate({ workerId: "stream", messages: [{ role: "user", content: "hello" }] });
|
|
95
|
-
const buffered = await
|
|
136
|
+
const buffered = await testProvider({ ...injectedBase, fetch: bufferedFetch, streaming: false })
|
|
96
137
|
.generate({ workerId: "buffer", messages: [{ role: "user", content: "hello" }] });
|
|
97
138
|
|
|
98
139
|
assert.equal(streamed.assistant.content, "streamed");
|
|
@@ -116,12 +157,17 @@ test("caller cancellation and provider timeout reach an injected fetch", async (
|
|
|
116
157
|
});
|
|
117
158
|
};
|
|
118
159
|
const caller = new AbortController();
|
|
119
|
-
const callerProvider =
|
|
160
|
+
const callerProvider = testProvider({ ...injectedBase, fetch: pendingFetch });
|
|
120
161
|
const callerRequest = callerProvider.generate({ workerId: "cancel", messages: [], signal: caller.signal });
|
|
121
162
|
caller.abort(new Error("operator cancelled"));
|
|
122
163
|
await assert.rejects(callerRequest, /operator cancelled/);
|
|
123
164
|
|
|
124
|
-
const timeoutProvider =
|
|
165
|
+
const timeoutProvider = testProvider({
|
|
166
|
+
...injectedBase,
|
|
167
|
+
fetch: pendingFetch,
|
|
168
|
+
fetchTimeoutMs: 1,
|
|
169
|
+
operationTimeoutMs: 100,
|
|
170
|
+
});
|
|
125
171
|
await assert.rejects(
|
|
126
172
|
timeoutProvider.generate({ workerId: "timeout", messages: [] }),
|
|
127
173
|
(error: ProviderError) => error.kind === "network_failure",
|
|
@@ -143,7 +189,7 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
|
143
189
|
{ choices: [], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } },
|
|
144
190
|
]), { status: 200 });
|
|
145
191
|
};
|
|
146
|
-
const provider =
|
|
192
|
+
const provider = testProvider({
|
|
147
193
|
...injectedBase,
|
|
148
194
|
fetch: providerFetch,
|
|
149
195
|
retryAttempts: 1,
|
|
@@ -159,6 +205,60 @@ test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
|
159
205
|
]);
|
|
160
206
|
});
|
|
161
207
|
|
|
208
|
+
test("request-observer open failures preserve the durability cause and issue no provider I/O", async () => {
|
|
209
|
+
const root = new Error("durable request open failed");
|
|
210
|
+
let calls = 0;
|
|
211
|
+
const provider = testProvider({
|
|
212
|
+
...injectedBase,
|
|
213
|
+
retryAttempts: 3,
|
|
214
|
+
fetch: async () => {
|
|
215
|
+
calls++;
|
|
216
|
+
return new Response(JSON.stringify(jsonChoice), {
|
|
217
|
+
status: 200,
|
|
218
|
+
headers: { "Content-Type": "application/json" },
|
|
219
|
+
});
|
|
220
|
+
},
|
|
221
|
+
streaming: false,
|
|
222
|
+
});
|
|
223
|
+
|
|
224
|
+
await assert.rejects(
|
|
225
|
+
provider.generate({
|
|
226
|
+
workerId: "observer-open",
|
|
227
|
+
messages: [],
|
|
228
|
+
observeRequest: async () => { throw root; },
|
|
229
|
+
}),
|
|
230
|
+
(error: unknown) => error === root,
|
|
231
|
+
);
|
|
232
|
+
assert.equal(calls, 0);
|
|
233
|
+
});
|
|
234
|
+
|
|
235
|
+
test("request-observer settlement failures preserve the durability cause without retrying I/O", async () => {
|
|
236
|
+
const root = new Error("durable request settlement failed");
|
|
237
|
+
let calls = 0;
|
|
238
|
+
const provider = testProvider({
|
|
239
|
+
...injectedBase,
|
|
240
|
+
retryAttempts: 3,
|
|
241
|
+
fetch: async () => {
|
|
242
|
+
calls++;
|
|
243
|
+
return new Response(JSON.stringify(jsonChoice), {
|
|
244
|
+
status: 200,
|
|
245
|
+
headers: { "Content-Type": "application/json" },
|
|
246
|
+
});
|
|
247
|
+
},
|
|
248
|
+
streaming: false,
|
|
249
|
+
});
|
|
250
|
+
|
|
251
|
+
await assert.rejects(
|
|
252
|
+
provider.generate({
|
|
253
|
+
workerId: "observer-settle",
|
|
254
|
+
messages: [],
|
|
255
|
+
observeRequest: async () => async () => { throw root; },
|
|
256
|
+
}),
|
|
257
|
+
(error: unknown) => error === root,
|
|
258
|
+
);
|
|
259
|
+
assert.equal(calls, 1);
|
|
260
|
+
});
|
|
261
|
+
|
|
162
262
|
// Sequenced fetch mock for retry tests: each entry is one HTTP response. A 200
|
|
163
263
|
// streams its chunks; any other status returns that error (with an optional
|
|
164
264
|
// retry-after header). The last entry repeats once the script runs out.
|
|
@@ -206,8 +306,13 @@ test("effortFromBudget: maps budget to tiers", () => {
|
|
|
206
306
|
|
|
207
307
|
test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
208
308
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
209
|
-
const p =
|
|
210
|
-
await assert.rejects(
|
|
309
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
310
|
+
await assert.rejects(
|
|
311
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
312
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
313
|
+
&& error.status === 524
|
|
314
|
+
&& error.problem.retryable === false,
|
|
315
|
+
);
|
|
211
316
|
await flush();
|
|
212
317
|
assert.equal(calls.length, 1); // edge code: one attempt, no retry despite retryAttempts: 3
|
|
213
318
|
mock.restoreAll();
|
|
@@ -216,7 +321,7 @@ test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttemp
|
|
|
216
321
|
test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
|
|
217
322
|
const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
|
|
218
323
|
const calls = installFetchScript([{ status: 422, body }]);
|
|
219
|
-
const p =
|
|
324
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
|
|
220
325
|
await assert.rejects(
|
|
221
326
|
p.generate({ workerId: "r", messages: [] }),
|
|
222
327
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
@@ -231,7 +336,7 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
231
336
|
status: 422,
|
|
232
337
|
error: { message: "non-conforming emission rejected", type: "grammar_invalid" },
|
|
233
338
|
}]);
|
|
234
|
-
const p =
|
|
339
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
235
340
|
await assert.rejects(
|
|
236
341
|
p.generate({ workerId: "r", messages: [] }),
|
|
237
342
|
(e: unknown) => e instanceof ProviderError && e.kind === "grammar_invalid",
|
|
@@ -239,29 +344,138 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
239
344
|
assert.equal(calls.length, 1);
|
|
240
345
|
});
|
|
241
346
|
|
|
347
|
+
test("a buffered classified error retains normalized usage and settled charge", async () => {
|
|
348
|
+
const calls = installFetchScript([{ status: 422, body: JSON.stringify(billedErrorBody) }]);
|
|
349
|
+
const p = testProvider({
|
|
350
|
+
...injectedBase,
|
|
351
|
+
streaming: false,
|
|
352
|
+
normalizeCost: directCost,
|
|
353
|
+
});
|
|
354
|
+
await assert.rejects(
|
|
355
|
+
p.generate({ workerId: "billed-json-error", messages: [] }),
|
|
356
|
+
(error: unknown) => {
|
|
357
|
+
assert.ok(error instanceof ProviderError);
|
|
358
|
+
assert.equal(error.kind, "grammar_invalid");
|
|
359
|
+
assert.deepEqual(error.accounting, [{
|
|
360
|
+
provider: "provider",
|
|
361
|
+
model: "m",
|
|
362
|
+
outcome: "error",
|
|
363
|
+
status: 422,
|
|
364
|
+
usage: {
|
|
365
|
+
inputTokens: 8,
|
|
366
|
+
outputTokens: 3,
|
|
367
|
+
totalTokens: 11,
|
|
368
|
+
inputTokenDetails: { cacheReadTokens: 2 },
|
|
369
|
+
outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
|
|
370
|
+
},
|
|
371
|
+
cost: settledCharge,
|
|
372
|
+
}]);
|
|
373
|
+
assert.equal(error.attempt, undefined, "accounting evidence does not fabricate an assistant response");
|
|
374
|
+
return true;
|
|
375
|
+
},
|
|
376
|
+
);
|
|
377
|
+
assert.equal(calls.length, 1);
|
|
378
|
+
});
|
|
379
|
+
|
|
380
|
+
test("an SSE classified error retains the same normalized usage and settled charge", async () => {
|
|
381
|
+
const calls = installFetch([billedErrorBody]);
|
|
382
|
+
const p = testProvider({
|
|
383
|
+
...injectedBase,
|
|
384
|
+
normalizeCost: directCost,
|
|
385
|
+
});
|
|
386
|
+
await assert.rejects(
|
|
387
|
+
p.generate({ workerId: "billed-sse-error", messages: [] }),
|
|
388
|
+
(error: unknown) => {
|
|
389
|
+
assert.ok(error instanceof ProviderError);
|
|
390
|
+
assert.equal(error.kind, "grammar_invalid");
|
|
391
|
+
assert.deepEqual(error.accounting, [{
|
|
392
|
+
provider: "provider",
|
|
393
|
+
model: "m",
|
|
394
|
+
outcome: "error",
|
|
395
|
+
status: 422,
|
|
396
|
+
usage: {
|
|
397
|
+
inputTokens: 8,
|
|
398
|
+
outputTokens: 3,
|
|
399
|
+
totalTokens: 11,
|
|
400
|
+
inputTokenDetails: { cacheReadTokens: 2 },
|
|
401
|
+
outputTokenDetails: { textTokens: 3, reasoningTokens: 0 },
|
|
402
|
+
},
|
|
403
|
+
cost: settledCharge,
|
|
404
|
+
}]);
|
|
405
|
+
assert.equal(error.attempt, undefined);
|
|
406
|
+
return true;
|
|
407
|
+
},
|
|
408
|
+
);
|
|
409
|
+
assert.equal(calls.length, 1);
|
|
410
|
+
});
|
|
411
|
+
|
|
412
|
+
test("a successful response normalizes direct charge without duplicating it as metadata", async () => {
|
|
413
|
+
installFetchJson({ ...jsonChoice, charge: settledCharge });
|
|
414
|
+
const p = testProvider({
|
|
415
|
+
...injectedBase,
|
|
416
|
+
streaming: false,
|
|
417
|
+
normalizeCost: directCost,
|
|
418
|
+
});
|
|
419
|
+
const response = await p.generate({ workerId: "billed-json-success", messages: [] });
|
|
420
|
+
assert.deepEqual(response.accounting[0]?.cost, settledCharge);
|
|
421
|
+
assert.equal(response.meta?.charge, undefined);
|
|
422
|
+
});
|
|
423
|
+
|
|
424
|
+
test("malformed monetary evidence closes the physical request before surfacing the normalization failure", async () => {
|
|
425
|
+
const root = new TypeError("direct charge is malformed");
|
|
426
|
+
const settled: unknown[] = [];
|
|
427
|
+
const calls = installFetchJson({ ...jsonChoice, charge: { malformed: true } });
|
|
428
|
+
const provider = testProvider({
|
|
429
|
+
...injectedBase,
|
|
430
|
+
retryAttempts: 3,
|
|
431
|
+
streaming: false,
|
|
432
|
+
normalizeCost: () => { throw root; },
|
|
433
|
+
});
|
|
434
|
+
|
|
435
|
+
await assert.rejects(
|
|
436
|
+
provider.generate({
|
|
437
|
+
workerId: "malformed-charge",
|
|
438
|
+
messages: [],
|
|
439
|
+
observeRequest: async () => async (accounting) => { settled.push(accounting); },
|
|
440
|
+
}),
|
|
441
|
+
(error: unknown) => error === root,
|
|
442
|
+
);
|
|
443
|
+
assert.equal(calls.length, 1);
|
|
444
|
+
assert.deepEqual(settled, [{
|
|
445
|
+
provider: "provider",
|
|
446
|
+
model: "m",
|
|
447
|
+
outcome: "response",
|
|
448
|
+
usage: { inputTokens: 1, outputTokens: 1, totalTokens: 2 },
|
|
449
|
+
cost: {
|
|
450
|
+
kind: "unknown",
|
|
451
|
+
reason: "provider request accounting could not be normalized after physical I/O",
|
|
452
|
+
},
|
|
453
|
+
}]);
|
|
454
|
+
});
|
|
455
|
+
|
|
242
456
|
test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
|
|
243
457
|
installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
244
|
-
const p =
|
|
458
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
245
459
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
246
460
|
assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
|
|
247
461
|
});
|
|
248
462
|
|
|
249
463
|
test("without a probed eos_token the content passes through untouched", async () => {
|
|
250
464
|
installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
251
|
-
const p =
|
|
465
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
252
466
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
253
467
|
assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
|
|
254
468
|
});
|
|
255
469
|
|
|
256
470
|
test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
|
|
257
471
|
installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
|
|
258
|
-
const p =
|
|
472
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
259
473
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
260
474
|
assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
|
|
261
475
|
});
|
|
262
476
|
|
|
263
477
|
test("identity getters and default prompt estimate", async () => {
|
|
264
|
-
const p =
|
|
478
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
265
479
|
assert.equal(p.model, "m");
|
|
266
480
|
assert.equal(p.contextWindow, null); // default
|
|
267
481
|
assert.deepEqual(
|
|
@@ -274,39 +488,54 @@ test("identity getters and default prompt estimate", async () => {
|
|
|
274
488
|
},
|
|
275
489
|
"chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
|
|
276
490
|
);
|
|
277
|
-
assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
|
|
278
491
|
});
|
|
279
492
|
|
|
280
|
-
test("injected prompt measurement preserves provenance and
|
|
493
|
+
test("injected prompt measurement preserves provenance and request cost estimation stays internal", async () => {
|
|
281
494
|
const seen: string[] = [];
|
|
282
|
-
|
|
495
|
+
installFetchJson(jsonChoice);
|
|
496
|
+
const p = testProvider({
|
|
283
497
|
model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
284
498
|
countPromptTokens: (messages) => {
|
|
285
499
|
seen.push(...messages.map(({ content }) => content));
|
|
286
500
|
return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
|
|
287
501
|
},
|
|
288
|
-
|
|
502
|
+
streaming: false,
|
|
503
|
+
estimateCost: (usage) => ({
|
|
504
|
+
kind: "estimated",
|
|
505
|
+
amount: { amount: String((usage?.totalTokens ?? 0) * 2), currency: "USD" },
|
|
506
|
+
source: "test estimator",
|
|
507
|
+
}),
|
|
289
508
|
});
|
|
290
509
|
assert.deepEqual(
|
|
291
510
|
await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
|
|
292
511
|
{ kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
|
|
293
512
|
);
|
|
294
513
|
assert.deepEqual(seen, ["system", "user"]);
|
|
295
|
-
|
|
514
|
+
const response = await p.generate({ workerId: "accounted", messages: [] });
|
|
515
|
+
assert.deepEqual(response.accounting[0]?.cost, {
|
|
516
|
+
kind: "estimated",
|
|
517
|
+
amount: { amount: "4", currency: "USD" },
|
|
518
|
+
source: "test estimator",
|
|
519
|
+
});
|
|
296
520
|
});
|
|
297
521
|
|
|
298
522
|
test("generate maps a streamed response into ProviderResponse", async () => {
|
|
299
|
-
const p =
|
|
523
|
+
const p = testProvider({ model: "req-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
300
524
|
installFetch([
|
|
301
525
|
{ model: "wire-model", choices: [{ delta: { content: "hel" } }] },
|
|
302
526
|
{ choices: [{ delta: { content: "lo" }, finish_reason: "stop" }] },
|
|
303
527
|
{ usage: { prompt_tokens: 3, completion_tokens: 2, total_tokens: 5, cached_tokens: 1 } },
|
|
304
528
|
]);
|
|
305
|
-
const { assistant, assistantRaw } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
529
|
+
const { assistant, assistantRaw, accounting } = await p.generate({ workerId: "r", messages: [{ role: "user", content: "hi" }] });
|
|
306
530
|
assert.equal(assistant.content, "hello");
|
|
307
531
|
assert.equal(assistant.model, "wire-model"); // wire-reported wins
|
|
308
532
|
assert.equal(assistant.finishReason, "stop");
|
|
309
|
-
assert.deepEqual(
|
|
533
|
+
assert.deepEqual(accounting[0]?.usage, {
|
|
534
|
+
inputTokens: 3,
|
|
535
|
+
outputTokens: 2,
|
|
536
|
+
totalTokens: 5,
|
|
537
|
+
inputTokenDetails: { cacheReadTokens: 1 },
|
|
538
|
+
});
|
|
310
539
|
assert.equal(assistant.reasoning, null); // none emitted
|
|
311
540
|
assert.notEqual(assistantRaw, undefined);
|
|
312
541
|
});
|
|
@@ -318,9 +547,8 @@ test("native SDK accounting metadata becomes a normalized charge in buffered and
|
|
|
318
547
|
};
|
|
319
548
|
const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
|
|
320
549
|
const charge = {
|
|
321
|
-
kind: "
|
|
550
|
+
kind: "charged",
|
|
322
551
|
amount: { amount: "0.00154935", currency: "USD" },
|
|
323
|
-
usdEquivalent: "0.00154935",
|
|
324
552
|
source: "OpenRouter response usage.cost",
|
|
325
553
|
};
|
|
326
554
|
const languageModel = {
|
|
@@ -364,23 +592,91 @@ test("native SDK accounting metadata becomes a normalized charge in buffered and
|
|
|
364
592
|
repeatPenalty: 1.15,
|
|
365
593
|
reasoning: { mode: "off" as const, budget: null },
|
|
366
594
|
retryAttempts: 0,
|
|
367
|
-
|
|
595
|
+
normalizeCost: providerCostNormalizer("@openrouter/ai-sdk-provider"),
|
|
368
596
|
};
|
|
369
597
|
|
|
370
598
|
await t.test("buffered", async () => {
|
|
371
|
-
const response = await
|
|
599
|
+
const response = await testProvider({ ...config, streaming: false })
|
|
372
600
|
.generate({ workerId: "buffered", messages: [] });
|
|
373
|
-
assert.deepEqual(response.
|
|
601
|
+
assert.deepEqual(response.accounting[0]?.cost, charge);
|
|
374
602
|
});
|
|
375
603
|
await t.test("streamed", async () => {
|
|
376
|
-
const response = await
|
|
604
|
+
const response = await testProvider(config)
|
|
377
605
|
.generate({ workerId: "streamed", messages: [] });
|
|
378
|
-
assert.deepEqual(response.
|
|
606
|
+
assert.deepEqual(response.accounting[0]?.cost, charge);
|
|
607
|
+
});
|
|
608
|
+
});
|
|
609
|
+
|
|
610
|
+
test("native SDK providers share the first-content retry contract", async () => {
|
|
611
|
+
let calls = 0;
|
|
612
|
+
const usage = {
|
|
613
|
+
inputTokens: { total: 1, noCache: 1, cacheRead: 0, cacheWrite: 0 },
|
|
614
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
615
|
+
};
|
|
616
|
+
const languageModel = {
|
|
617
|
+
specificationVersion: "v4",
|
|
618
|
+
provider: "native.test",
|
|
619
|
+
modelId: "native-timeout",
|
|
620
|
+
supportedUrls: {},
|
|
621
|
+
doGenerate: async () => { throw new Error("buffered generation is not under test"); },
|
|
622
|
+
doStream: async ({ abortSignal }: { abortSignal?: AbortSignal }) => {
|
|
623
|
+
calls++;
|
|
624
|
+
if (calls > 1) {
|
|
625
|
+
return {
|
|
626
|
+
stream: new ReadableStream({
|
|
627
|
+
start(controller) {
|
|
628
|
+
controller.enqueue({ type: "stream-start", warnings: [] });
|
|
629
|
+
controller.enqueue({ type: "response-metadata", id: "native-retry", modelId: "native-timeout" });
|
|
630
|
+
controller.enqueue({ type: "text-start", id: "text-1" });
|
|
631
|
+
controller.enqueue({ type: "text-delta", id: "text-1", delta: "recovered" });
|
|
632
|
+
controller.enqueue({ type: "text-end", id: "text-1" });
|
|
633
|
+
controller.enqueue({
|
|
634
|
+
type: "finish",
|
|
635
|
+
finishReason: { unified: "stop", raw: "completed" },
|
|
636
|
+
usage,
|
|
637
|
+
});
|
|
638
|
+
controller.close();
|
|
639
|
+
},
|
|
640
|
+
}),
|
|
641
|
+
response: {},
|
|
642
|
+
};
|
|
643
|
+
}
|
|
644
|
+
return {
|
|
645
|
+
stream: new ReadableStream({
|
|
646
|
+
start(controller) {
|
|
647
|
+
controller.enqueue({ type: "stream-start", warnings: [] });
|
|
648
|
+
const timer = setTimeout(() => controller.close(), 100);
|
|
649
|
+
abortSignal?.addEventListener("abort", () => {
|
|
650
|
+
clearTimeout(timer);
|
|
651
|
+
controller.error(abortSignal.reason);
|
|
652
|
+
}, { once: true });
|
|
653
|
+
},
|
|
654
|
+
}),
|
|
655
|
+
response: {},
|
|
656
|
+
};
|
|
657
|
+
},
|
|
658
|
+
} as unknown as LanguageModel;
|
|
659
|
+
const provider = testProvider({
|
|
660
|
+
model: "native-timeout",
|
|
661
|
+
languageModel,
|
|
662
|
+
fetchTimeoutMs: 5_000,
|
|
663
|
+
operationTimeoutMs: 5_000,
|
|
664
|
+
firstContentTimeoutMs: 10,
|
|
665
|
+
temperature: 0.2,
|
|
666
|
+
repeatPenalty: 1.15,
|
|
667
|
+
reasoning: { mode: "off", budget: null },
|
|
668
|
+
retryAttempts: 1,
|
|
669
|
+
source: "provider:test-native",
|
|
379
670
|
});
|
|
671
|
+
|
|
672
|
+
const result = await provider.generate({ workerId: "native-retry", messages: [] });
|
|
673
|
+
assert.equal(result.assistant.content, "recovered");
|
|
674
|
+
assert.equal(calls, 2);
|
|
675
|
+
assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
|
|
380
676
|
});
|
|
381
677
|
|
|
382
678
|
test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
|
|
383
|
-
const p =
|
|
679
|
+
const p = testProvider({
|
|
384
680
|
model: "grok-test",
|
|
385
681
|
url: "http://x/v1/chat/completions",
|
|
386
682
|
fetchTimeoutMs: 5_000,
|
|
@@ -389,7 +685,7 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
|
|
|
389
685
|
reasoning: { mode: "off", budget: null },
|
|
390
686
|
retryAttempts: 0,
|
|
391
687
|
streaming: false,
|
|
392
|
-
|
|
688
|
+
normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
|
|
393
689
|
});
|
|
394
690
|
installFetchJson({
|
|
395
691
|
id: "response-1",
|
|
@@ -403,8 +699,8 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
|
|
|
403
699
|
},
|
|
404
700
|
});
|
|
405
701
|
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
406
|
-
assert.deepEqual(response.
|
|
407
|
-
kind: "
|
|
702
|
+
assert.deepEqual(response.accounting[0]?.cost, {
|
|
703
|
+
kind: "charged",
|
|
408
704
|
amount: { amount: "15493500", currency: "USDTICK" },
|
|
409
705
|
usdEquivalent: "0.00154935",
|
|
410
706
|
source: "xAI response usage.cost_in_usd_ticks",
|
|
@@ -413,7 +709,7 @@ test("compatible xAI wire usage becomes an exact tick charge without raw-body ca
|
|
|
413
709
|
});
|
|
414
710
|
|
|
415
711
|
test("streamed xAI final usage retains its exact tick charge", async () => {
|
|
416
|
-
const p =
|
|
712
|
+
const p = testProvider({
|
|
417
713
|
model: "grok-test",
|
|
418
714
|
url: "http://x/v1/chat/completions",
|
|
419
715
|
fetchTimeoutMs: 5_000,
|
|
@@ -421,7 +717,7 @@ test("streamed xAI final usage retains its exact tick charge", async () => {
|
|
|
421
717
|
repeatPenalty: 1.15,
|
|
422
718
|
reasoning: { mode: "off", budget: null },
|
|
423
719
|
retryAttempts: 0,
|
|
424
|
-
|
|
720
|
+
normalizeCost: providerCostNormalizer("@ai-sdk/xai"),
|
|
425
721
|
});
|
|
426
722
|
installFetch([
|
|
427
723
|
{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
@@ -436,8 +732,8 @@ test("streamed xAI final usage retains its exact tick charge", async () => {
|
|
|
436
732
|
},
|
|
437
733
|
]);
|
|
438
734
|
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
439
|
-
assert.deepEqual(response.
|
|
440
|
-
kind: "
|
|
735
|
+
assert.deepEqual(response.accounting[0]?.cost, {
|
|
736
|
+
kind: "charged",
|
|
441
737
|
amount: { amount: "15493500", currency: "USDTICK" },
|
|
442
738
|
usdEquivalent: "0.00154935",
|
|
443
739
|
source: "xAI response usage.cost_in_usd_ticks",
|
|
@@ -452,7 +748,7 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
|
|
|
452
748
|
...(typeof options === "object" && options.code !== undefined ? { code: options.code } : {}),
|
|
453
749
|
});
|
|
454
750
|
});
|
|
455
|
-
const p =
|
|
751
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
456
752
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "function_call" }] }]);
|
|
457
753
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
458
754
|
assert.equal(assistant.finishReason, null);
|
|
@@ -470,7 +766,7 @@ test("#161: a streamed resource interruption is a failed exchange with complete
|
|
|
470
766
|
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
471
767
|
},
|
|
472
768
|
]);
|
|
473
|
-
const provider =
|
|
769
|
+
const provider = testProvider({
|
|
474
770
|
...injectedBase,
|
|
475
771
|
retryAttempts: 2,
|
|
476
772
|
rawBody: true,
|
|
@@ -489,12 +785,10 @@ test("#161: a streamed resource interruption is a failed exchange with complete
|
|
|
489
785
|
assert.equal(error.attempt?.assistant.content, "partial answer");
|
|
490
786
|
assert.equal(error.attempt?.assistant.reasoning, "partial thought");
|
|
491
787
|
assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
|
|
492
|
-
assert.deepEqual(error.
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
cached: 0,
|
|
497
|
-
total: 12,
|
|
788
|
+
assert.deepEqual(error.accounting[0]?.usage, {
|
|
789
|
+
inputTokens: 7,
|
|
790
|
+
outputTokens: 5,
|
|
791
|
+
totalTokens: 12,
|
|
498
792
|
});
|
|
499
793
|
assert.equal(
|
|
500
794
|
(error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
|
|
@@ -517,7 +811,7 @@ test("#161: a buffered resource interruption preserves the successful wire respo
|
|
|
517
811
|
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
518
812
|
};
|
|
519
813
|
const calls = installFetchJson(wire);
|
|
520
|
-
const provider =
|
|
814
|
+
const provider = testProvider({
|
|
521
815
|
...injectedBase,
|
|
522
816
|
streaming: false,
|
|
523
817
|
retryAttempts: 2,
|
|
@@ -546,37 +840,37 @@ test("#161: a buffered resource interruption preserves the successful wire respo
|
|
|
546
840
|
test("generate translates a backend cap synonym to canonical length", async () => {
|
|
547
841
|
// gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
|
|
548
842
|
// "length" so its truncation check (=== "length") is a cross-backend invariant.
|
|
549
|
-
const p =
|
|
843
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
550
844
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "MAX_TOKENS" }] }]);
|
|
551
845
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
552
846
|
assert.equal(assistant.finishReason, "length");
|
|
553
847
|
});
|
|
554
848
|
|
|
555
849
|
test("generate translates end_turn to canonical stop", async () => {
|
|
556
|
-
const p =
|
|
850
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
557
851
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
|
|
558
852
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
559
853
|
assert.equal(assistant.finishReason, "stop");
|
|
560
854
|
});
|
|
561
855
|
|
|
562
856
|
test("generate translates xAI completed to canonical stop", async () => {
|
|
563
|
-
const p =
|
|
857
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
564
858
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "completed" }] }]);
|
|
565
859
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
566
860
|
assert.equal(assistant.finishReason, "stop");
|
|
567
861
|
});
|
|
568
862
|
|
|
569
863
|
test("generate aggregates reasoning deltas under multiple field names", async () => {
|
|
570
|
-
const p =
|
|
864
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
571
865
|
installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
|
|
572
866
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
573
867
|
assert.equal(assistant.reasoning, "because");
|
|
574
868
|
assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
|
|
575
869
|
});
|
|
576
870
|
|
|
577
|
-
test("{§provider-tagged-reasoning} explicit think-tags
|
|
871
|
+
test("{§provider-tagged-reasoning} explicit think-tags project content without estimating token attribution", async () => {
|
|
578
872
|
const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
|
|
579
|
-
const p =
|
|
873
|
+
const p = testProvider(config);
|
|
580
874
|
installFetch([
|
|
581
875
|
{ choices: [{ delta: { content: "<thi" } }] },
|
|
582
876
|
{ choices: [{ delta: { content: "nk>12345</th" } }] },
|
|
@@ -588,12 +882,10 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one streamed le
|
|
|
588
882
|
|
|
589
883
|
assert.equal(response.assistant.reasoning, "12345");
|
|
590
884
|
assert.equal(response.assistant.content, "abcde");
|
|
591
|
-
assert.deepEqual(response.
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
cached: 0,
|
|
596
|
-
total: 13,
|
|
885
|
+
assert.deepEqual(response.accounting[0]?.usage, {
|
|
886
|
+
inputTokens: 3,
|
|
887
|
+
outputTokens: 10,
|
|
888
|
+
totalTokens: 13,
|
|
597
889
|
});
|
|
598
890
|
assert.deepEqual(
|
|
599
891
|
((response.assistantRaw as { content: string; reasoning: string }).content),
|
|
@@ -611,12 +903,12 @@ test("{§provider-tagged-reasoning} explicit think-tags projects one buffered le
|
|
|
611
903
|
usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
|
|
612
904
|
});
|
|
613
905
|
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
614
|
-
const response = await
|
|
906
|
+
const response = await testProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
|
|
615
907
|
|
|
616
908
|
assert.equal(response.assistant.reasoning, "12345");
|
|
617
909
|
assert.equal(response.assistant.content, "abcde");
|
|
618
|
-
assert.equal(response.
|
|
619
|
-
assert.equal(response.
|
|
910
|
+
assert.equal(response.accounting[0]?.usage?.outputTokens, 10);
|
|
911
|
+
assert.equal(response.accounting[0]?.usage?.outputTokenDetails, undefined);
|
|
620
912
|
});
|
|
621
913
|
|
|
622
914
|
test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
|
|
@@ -631,12 +923,14 @@ test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reas
|
|
|
631
923
|
},
|
|
632
924
|
});
|
|
633
925
|
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
634
|
-
const response = await
|
|
926
|
+
const response = await testProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
|
|
635
927
|
|
|
636
928
|
assert.equal(response.assistant.reasoning, "12345");
|
|
637
929
|
assert.equal(response.assistant.content, "abcde");
|
|
638
|
-
assert.
|
|
639
|
-
|
|
930
|
+
assert.deepEqual(response.accounting[0]?.usage?.outputTokenDetails, {
|
|
931
|
+
textTokens: 7,
|
|
932
|
+
reasoningTokens: 3,
|
|
933
|
+
});
|
|
640
934
|
});
|
|
641
935
|
|
|
642
936
|
test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
|
|
@@ -645,15 +939,13 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
|
|
|
645
939
|
{ choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
|
|
646
940
|
{ usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
|
|
647
941
|
]);
|
|
648
|
-
const streamed = await
|
|
942
|
+
const streamed = await testProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
|
|
649
943
|
assert.equal(streamed.assistant.reasoning, "unfinished");
|
|
650
944
|
assert.equal(streamed.assistant.content, "");
|
|
651
|
-
assert.deepEqual(streamed.
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
cached: 0,
|
|
656
|
-
total: 11,
|
|
945
|
+
assert.deepEqual(streamed.accounting[0]?.usage, {
|
|
946
|
+
inputTokens: 3,
|
|
947
|
+
outputTokens: 8,
|
|
948
|
+
totalTokens: 11,
|
|
657
949
|
});
|
|
658
950
|
|
|
659
951
|
mock.restoreAll();
|
|
@@ -663,11 +955,11 @@ test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reason
|
|
|
663
955
|
usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
|
|
664
956
|
});
|
|
665
957
|
const bufferedConfig = { ...config, streaming: false };
|
|
666
|
-
const buffered = await
|
|
958
|
+
const buffered = await testProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
|
|
667
959
|
assert.equal(buffered.assistant.reasoning, "unfinished");
|
|
668
960
|
assert.equal(buffered.assistant.content, "");
|
|
669
|
-
assert.equal(buffered.
|
|
670
|
-
assert.equal(buffered.
|
|
961
|
+
assert.equal(buffered.accounting[0]?.usage?.outputTokens, 8);
|
|
962
|
+
assert.equal(buffered.accounting[0]?.usage?.outputTokenDetails, undefined);
|
|
671
963
|
});
|
|
672
964
|
|
|
673
965
|
test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
|
|
@@ -676,11 +968,11 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
|
|
|
676
968
|
choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
|
|
677
969
|
usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
|
|
678
970
|
});
|
|
679
|
-
const verbatim = await
|
|
971
|
+
const verbatim = await testProvider({ ...injectedBase, streaming: false })
|
|
680
972
|
.generate({ workerId: "verbatim", messages: [] });
|
|
681
973
|
assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
|
|
682
974
|
assert.equal(verbatim.assistant.reasoning, null);
|
|
683
|
-
assert.equal(verbatim.
|
|
975
|
+
assert.equal(verbatim.accounting[0]?.usage?.outputTokens, 4);
|
|
684
976
|
|
|
685
977
|
mock.restoreAll();
|
|
686
978
|
installFetchJson({
|
|
@@ -689,7 +981,7 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
|
|
|
689
981
|
usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
|
|
690
982
|
});
|
|
691
983
|
const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
692
|
-
const nonLeading = await
|
|
984
|
+
const nonLeading = await testProvider(taggedConfig)
|
|
693
985
|
.generate({ workerId: "non-leading", messages: [] });
|
|
694
986
|
assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
|
|
695
987
|
assert.equal(nonLeading.assistant.reasoning, null);
|
|
@@ -703,14 +995,14 @@ test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reason
|
|
|
703
995
|
}, finish_reason: "stop" }],
|
|
704
996
|
usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
|
|
705
997
|
});
|
|
706
|
-
const structured = await
|
|
998
|
+
const structured = await testProvider(taggedConfig)
|
|
707
999
|
.generate({ workerId: "structured", messages: [] });
|
|
708
1000
|
assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
|
|
709
1001
|
assert.equal(structured.assistant.reasoning, "structured reasoning");
|
|
710
1002
|
});
|
|
711
1003
|
|
|
712
1004
|
test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
|
|
713
|
-
const content = "<think>🧠reason</think
|
|
1005
|
+
const content = "<think>🧠reason</think># PLAN0\n\n## SEND0 [200]\ndone";
|
|
714
1006
|
const config = {
|
|
715
1007
|
...injectedBase,
|
|
716
1008
|
contextWindow: 640,
|
|
@@ -721,14 +1013,14 @@ test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-proje
|
|
|
721
1013
|
};
|
|
722
1014
|
installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
723
1015
|
|
|
724
|
-
const response = await
|
|
1016
|
+
const response = await testProvider(config).generate({
|
|
725
1017
|
workerId: "tagged-grammar",
|
|
726
1018
|
messages: [],
|
|
727
1019
|
grammar: `root ::= ${JSON.stringify(content)}`,
|
|
728
1020
|
});
|
|
729
1021
|
|
|
730
1022
|
assert.equal(response.assistant.reasoning, "🧠reason");
|
|
731
|
-
assert.equal(response.assistant.content, "
|
|
1023
|
+
assert.equal(response.assistant.content, "# PLAN0\n\n## SEND0 [200]\ndone");
|
|
732
1024
|
assert.deepEqual(response.grammarEvidence, {
|
|
733
1025
|
input: content,
|
|
734
1026
|
contentStart: [..."<think>🧠reason</think>"].length,
|
|
@@ -745,7 +1037,7 @@ test("encrypted reasoning (non-streamed): encrypted entries normalize and text e
|
|
|
745
1037
|
{ type: "reasoning.text", text: "never surfaced here" },
|
|
746
1038
|
],
|
|
747
1039
|
}, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
748
|
-
const p =
|
|
1040
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
749
1041
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
750
1042
|
// Wire detail ID is preserved; the assistant-message location supports the
|
|
751
1043
|
// derived classification but supplies no downstream client entity ID.
|
|
@@ -759,7 +1051,7 @@ test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
|
|
|
759
1051
|
{ type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
|
|
760
1052
|
{ type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
|
|
761
1053
|
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
762
|
-
const p =
|
|
1054
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
763
1055
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
764
1056
|
assert.equal(assistant.reasoningEncrypted?.length, 2);
|
|
765
1057
|
assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
|
|
@@ -769,7 +1061,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
|
|
|
769
1061
|
installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
|
|
770
1062
|
{ type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
|
|
771
1063
|
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
772
|
-
const p =
|
|
1064
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
773
1065
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
774
1066
|
assert.deepEqual(assistant.reasoningEncrypted, [{
|
|
775
1067
|
id: null,
|
|
@@ -779,7 +1071,7 @@ test("assistant-message location classifies encrypted reasoning without inventin
|
|
|
779
1071
|
});
|
|
780
1072
|
|
|
781
1073
|
test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
|
|
782
|
-
const p =
|
|
1074
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
783
1075
|
installFetch([
|
|
784
1076
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
|
|
785
1077
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "BqXYZ", id: "rs_1", index: 0 }] } }] },
|
|
@@ -790,32 +1082,35 @@ test("encrypted reasoning (streamed): chunked blob concatenates per entry index"
|
|
|
790
1082
|
assert.equal(assistant.content, "4");
|
|
791
1083
|
});
|
|
792
1084
|
|
|
793
|
-
test("reasoningStyle 'think'
|
|
794
|
-
const on =
|
|
1085
|
+
test("reasoningStyle 'think' follows activation (magnitude is irrelevant to the boolean wire control)", async () => {
|
|
1086
|
+
const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
795
1087
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
796
1088
|
await on.generate({ workerId: "r", messages: [] });
|
|
797
1089
|
assert.equal(JSON.parse(calls[0].init.body as string).think, true);
|
|
798
1090
|
|
|
799
1091
|
mock.restoreAll();
|
|
800
|
-
const off =
|
|
1092
|
+
const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "think" });
|
|
801
1093
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
802
1094
|
await off.generate({ workerId: "r", messages: [] });
|
|
803
1095
|
assert.equal("think" in JSON.parse(calls[0].init.body as string), false);
|
|
804
1096
|
});
|
|
805
1097
|
|
|
806
|
-
test("reasoningStyle 'effort'
|
|
807
|
-
|
|
808
|
-
|
|
809
|
-
|
|
810
|
-
|
|
1098
|
+
test("reasoningStyle 'effort' enables at the portable default without inventing a budget", async () => {
|
|
1099
|
+
for (const [budget, expected] of [[null, "medium"], [5000, "high"]] as const) {
|
|
1100
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
1101
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1102
|
+
await p.generate({ workerId: "r", messages: [] });
|
|
1103
|
+
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, expected);
|
|
1104
|
+
mock.restoreAll();
|
|
1105
|
+
}
|
|
811
1106
|
});
|
|
812
1107
|
|
|
813
1108
|
test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
|
|
814
1109
|
// expected === null → the field must be ABSENT from the wire body. Fireworks
|
|
815
1110
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
816
1111
|
// Adaptive = the backend's own default posture = omission.
|
|
817
|
-
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
818
|
-
const p =
|
|
1112
|
+
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: null }, "medium"], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
1113
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
819
1114
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
820
1115
|
await p.generate({ workerId: "r", messages: [] });
|
|
821
1116
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -829,10 +1124,11 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
|
|
|
829
1124
|
const cases = [
|
|
830
1125
|
[{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
|
|
831
1126
|
[{ mode: "adaptive", budget: null }, {}],
|
|
1127
|
+
[{ mode: "on", budget: null }, { thinking: { type: "enabled" } }],
|
|
832
1128
|
[{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
833
1129
|
] as const;
|
|
834
1130
|
for (const [reasoning, expected] of cases) {
|
|
835
|
-
const p =
|
|
1131
|
+
const p = testProvider({
|
|
836
1132
|
model: "m",
|
|
837
1133
|
url: "http://x/v1/chat/completions",
|
|
838
1134
|
fetchTimeoutMs: 5000,
|
|
@@ -858,7 +1154,7 @@ test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete Dee
|
|
|
858
1154
|
});
|
|
859
1155
|
|
|
860
1156
|
test("the family temperature default rides every request; caller sampling overrides it", async () => {
|
|
861
|
-
const p =
|
|
1157
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
862
1158
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
863
1159
|
await p.generate({ workerId: "r", messages: [] });
|
|
864
1160
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
@@ -877,7 +1173,7 @@ test("the family temperature default rides every request; caller sampling overri
|
|
|
877
1173
|
test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
|
|
878
1174
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
|
|
879
1175
|
// set + llamacpp -> the loop-breakers ride the wire
|
|
880
|
-
const p =
|
|
1176
|
+
const p = testProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
|
|
881
1177
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
882
1178
|
await p.generate({ workerId: "r", messages: [] });
|
|
883
1179
|
let body = JSON.parse(calls[0].init.body as string);
|
|
@@ -888,7 +1184,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
|
|
|
888
1184
|
assert.equal(body.repeat_penalty, 1.15); // repeat_penalty always rides the llamacpp path
|
|
889
1185
|
mock.restoreAll();
|
|
890
1186
|
// unset -> no dry_*/repeat_last_n on the wire (box keeps its own defaults)
|
|
891
|
-
const p2 =
|
|
1187
|
+
const p2 = testProvider({ ...base, grammarStyle: "llamacpp" });
|
|
892
1188
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
893
1189
|
await p2.generate({ workerId: "r", messages: [] });
|
|
894
1190
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -896,7 +1192,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
|
|
|
896
1192
|
assert.equal("repeat_last_n" in body, false);
|
|
897
1193
|
mock.restoreAll();
|
|
898
1194
|
// DRY is a llama.cpp sampler: a cloud ("none") provider never emits it, even if configured
|
|
899
|
-
const p3 =
|
|
1195
|
+
const p3 = testProvider({ ...base, grammarStyle: "none", dryMultiplier: 0.8, repeatLastN: 512 });
|
|
900
1196
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
901
1197
|
await p3.generate({ workerId: "r", messages: [] });
|
|
902
1198
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -906,7 +1202,7 @@ test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box
|
|
|
906
1202
|
});
|
|
907
1203
|
|
|
908
1204
|
test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
|
|
909
|
-
const p =
|
|
1205
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
910
1206
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
911
1207
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
912
1208
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -916,13 +1212,13 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
916
1212
|
|
|
917
1213
|
test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
|
|
918
1214
|
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
919
|
-
const llama =
|
|
1215
|
+
const llama = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
920
1216
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
921
1217
|
await llama.generate({ workerId: "r", messages: [] });
|
|
922
1218
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
923
1219
|
mock.restoreAll();
|
|
924
1220
|
// A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
|
|
925
|
-
const cloud =
|
|
1221
|
+
const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
926
1222
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
927
1223
|
await cloud.generate({ workerId: "r", messages: [] });
|
|
928
1224
|
const cloudBody = JSON.parse(calls[0].init.body as string);
|
|
@@ -931,14 +1227,14 @@ test("the repeat penalty rides every request rail-off, keyed per backend", async
|
|
|
931
1227
|
assert.equal("repeat_penalty" in cloudBody, false);
|
|
932
1228
|
mock.restoreAll();
|
|
933
1229
|
// frequencyPenalty unset (default 0) opts out cleanly - sends nothing (an out-of-date plugin runs unguarded, never breaks)
|
|
934
|
-
const bare =
|
|
1230
|
+
const bare = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
935
1231
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
936
1232
|
await bare.generate({ workerId: "r", messages: [] });
|
|
937
1233
|
assert.equal("frequency_penalty" in JSON.parse(calls[0].init.body as string), false);
|
|
938
1234
|
});
|
|
939
1235
|
|
|
940
1236
|
test("sampling passthrough forwards caller params; managed + reserved keys win", async () => {
|
|
941
|
-
const p =
|
|
1237
|
+
const p = testProvider({ model: "managed-model", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
942
1238
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
943
1239
|
await p.generate({
|
|
944
1240
|
workerId: "r",
|
|
@@ -961,7 +1257,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
|
|
|
961
1257
|
});
|
|
962
1258
|
|
|
963
1259
|
test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
|
|
964
|
-
const p =
|
|
1260
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
965
1261
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
966
1262
|
await p.generate({
|
|
967
1263
|
workerId: "r",
|
|
@@ -986,7 +1282,7 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
|
|
|
986
1282
|
});
|
|
987
1283
|
|
|
988
1284
|
test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
|
|
989
|
-
const p =
|
|
1285
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
990
1286
|
const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
|
|
991
1287
|
const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
|
|
992
1288
|
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
|
|
@@ -1005,8 +1301,24 @@ test("template reasoning returns the exact pre-projection grammar sentence ({§g
|
|
|
1005
1301
|
assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
|
|
1006
1302
|
});
|
|
1007
1303
|
|
|
1304
|
+
test("template reasoning projects a leading think envelope without losing grammar evidence", async () => {
|
|
1305
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1306
|
+
const input = "<think>\ncon🙂sider</think>x";
|
|
1307
|
+
const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
|
|
1308
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
|
|
1309
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1310
|
+
assert.equal(body.reasoning_format, "none");
|
|
1311
|
+
assert.equal(res.assistant.reasoning, "con🙂sider");
|
|
1312
|
+
assert.equal(res.assistant.content, "x");
|
|
1313
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
1314
|
+
input,
|
|
1315
|
+
contentStart: [..."<think>\ncon🙂sider</think>"].length,
|
|
1316
|
+
transported: true,
|
|
1317
|
+
});
|
|
1318
|
+
});
|
|
1319
|
+
|
|
1008
1320
|
test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
|
|
1009
|
-
const p =
|
|
1321
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1010
1322
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1011
1323
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1012
1324
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1015,7 +1327,7 @@ test("a verbatim template response remains exact evidence when it has no channel
|
|
|
1015
1327
|
});
|
|
1016
1328
|
|
|
1017
1329
|
test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
|
|
1018
|
-
const p =
|
|
1330
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1019
1331
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1020
1332
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1021
1333
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1025,14 +1337,14 @@ test("a template grammar preserves exact evidence when reasoning is disabled", a
|
|
|
1025
1337
|
});
|
|
1026
1338
|
|
|
1027
1339
|
test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
|
|
1028
|
-
const p =
|
|
1340
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1029
1341
|
installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
|
|
1030
1342
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1031
1343
|
assert.equal(res.grammarEvidence, undefined);
|
|
1032
1344
|
});
|
|
1033
1345
|
|
|
1034
1346
|
test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
|
|
1035
|
-
const p =
|
|
1347
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1036
1348
|
const input = "<|channel>thought\n<channel|>x";
|
|
1037
1349
|
const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
|
|
1038
1350
|
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
|
|
@@ -1063,16 +1375,16 @@ test("channel-escape detector: billed completion tokens vastly beyond visible ch
|
|
|
1063
1375
|
}
|
|
1064
1376
|
return new Response(sseStream(chunks), { status: 200 });
|
|
1065
1377
|
};
|
|
1066
|
-
const p =
|
|
1378
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1067
1379
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1068
1380
|
const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
|
|
1069
1381
|
assert.ok(escape, "escape notice attached");
|
|
1070
1382
|
assert.equal(escape!.kind, "grammar_unenforced");
|
|
1071
|
-
assert.match(escape!.message ?? "", /5000
|
|
1383
|
+
assert.match(escape!.message ?? "", /5000 output tokens billed/);
|
|
1072
1384
|
});
|
|
1073
1385
|
|
|
1074
1386
|
test("channel-escape state is absent without a transported grammar", async () => {
|
|
1075
|
-
const p =
|
|
1387
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1076
1388
|
installFetch([
|
|
1077
1389
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
1078
1390
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
@@ -1083,7 +1395,7 @@ test("channel-escape state is absent without a transported grammar", async () =>
|
|
|
1083
1395
|
});
|
|
1084
1396
|
|
|
1085
1397
|
test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
|
|
1086
|
-
const on =
|
|
1398
|
+
const on = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
1087
1399
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1088
1400
|
await on.generate({ workerId: "r", messages: [] });
|
|
1089
1401
|
let body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1092,7 +1404,7 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
|
|
|
1092
1404
|
assert.equal(body.thinking_budget_tokens, 64);
|
|
1093
1405
|
|
|
1094
1406
|
mock.restoreAll();
|
|
1095
|
-
const off =
|
|
1407
|
+
const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
1096
1408
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1097
1409
|
await off.generate({ workerId: "r", messages: [] });
|
|
1098
1410
|
body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1103,33 +1415,54 @@ test("reasoningStyle 'template' sends llama-server activation, parser, and respo
|
|
|
1103
1415
|
|
|
1104
1416
|
test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
|
|
1105
1417
|
const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
|
|
1106
|
-
const p =
|
|
1418
|
+
const p = testProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
|
|
1107
1419
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1108
1420
|
await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
|
|
1109
1421
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1110
1422
|
assert.equal(body.thinking_budget_tokens, 32);
|
|
1111
1423
|
assert.equal(body.reasoning_format, "auto");
|
|
1112
1424
|
assert.throws(
|
|
1113
|
-
() =>
|
|
1425
|
+
() => testProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
|
|
1114
1426
|
/REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
|
|
1115
1427
|
);
|
|
1116
1428
|
});
|
|
1117
1429
|
|
|
1118
|
-
test("budget
|
|
1119
|
-
const
|
|
1430
|
+
test("reasoningStyle 'template' explicit activation without a budget uses the resolved reserve", async () => {
|
|
1431
|
+
const p = testProvider({
|
|
1432
|
+
model: "m",
|
|
1433
|
+
url: "http://x/v1/chat/completions",
|
|
1434
|
+
contextWindow: 640,
|
|
1435
|
+
reasoningReserve: { tokens: 64 },
|
|
1436
|
+
completionReserve: { tokens: 160 },
|
|
1437
|
+
fetchTimeoutMs: 5000,
|
|
1438
|
+
temperature: 0.2,
|
|
1439
|
+
repeatPenalty: 1.15,
|
|
1440
|
+
reasoning: { mode: "on", budget: null },
|
|
1441
|
+
retryAttempts: 0,
|
|
1442
|
+
reasoningStyle: "template",
|
|
1443
|
+
});
|
|
1444
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1445
|
+
await p.generate({ workerId: "r", messages: [] });
|
|
1446
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1447
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
1448
|
+
assert.equal(body.thinking_budget_tokens, 64);
|
|
1449
|
+
});
|
|
1450
|
+
|
|
1451
|
+
test("reasoning off suppresses effort and include_reasoning controls", async () => {
|
|
1452
|
+
const effort = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "effort" });
|
|
1120
1453
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1121
1454
|
await effort.generate({ workerId: "r", messages: [] });
|
|
1122
1455
|
assert.equal("reasoning_effort" in JSON.parse(calls[0].init.body as string), false);
|
|
1123
1456
|
|
|
1124
1457
|
mock.restoreAll();
|
|
1125
|
-
const relay =
|
|
1458
|
+
const relay = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
1126
1459
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1127
1460
|
await relay.generate({ workerId: "r", messages: [] });
|
|
1128
1461
|
assert.equal("include_reasoning" in JSON.parse(calls[0].init.body as string), false);
|
|
1129
1462
|
});
|
|
1130
1463
|
|
|
1131
1464
|
test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", async () => {
|
|
1132
|
-
const p =
|
|
1465
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "include_reasoning" });
|
|
1133
1466
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1134
1467
|
await p.generate({ workerId: "r", messages: [] });
|
|
1135
1468
|
assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
|
|
@@ -1138,7 +1471,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
|
|
|
1138
1471
|
// — grammar-constrained sampling —
|
|
1139
1472
|
|
|
1140
1473
|
test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
|
|
1141
|
-
const p =
|
|
1474
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
1142
1475
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1143
1476
|
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1144
1477
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1148,7 +1481,7 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
|
|
|
1148
1481
|
});
|
|
1149
1482
|
|
|
1150
1483
|
test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
|
|
1151
|
-
const p =
|
|
1484
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1152
1485
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1153
1486
|
await p.generate({ workerId: "r", messages: [], grammar: "root ::= statement" });
|
|
1154
1487
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1158,7 +1491,7 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
|
|
|
1158
1491
|
|
|
1159
1492
|
// — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
|
|
1160
1493
|
|
|
1161
|
-
const grammarProvider = () =>
|
|
1494
|
+
const grammarProvider = () => testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
|
|
1162
1495
|
const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
1163
1496
|
|
|
1164
1497
|
test("an unsplit grammar response carries the exact observed sentence", async () => {
|
|
@@ -1192,7 +1525,7 @@ test("empty unsplit content remains exact grammar evidence", async () => {
|
|
|
1192
1525
|
});
|
|
1193
1526
|
|
|
1194
1527
|
test("grammarStyle 'none' produces no grammar observation", async () => {
|
|
1195
|
-
const p =
|
|
1528
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
|
|
1196
1529
|
streamingContent("anything goes");
|
|
1197
1530
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
1198
1531
|
assert.equal(res.assistant.content, "anything goes");
|
|
@@ -1211,7 +1544,7 @@ test("provider evidence does not depend on the local validator understanding the
|
|
|
1211
1544
|
// — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
|
|
1212
1545
|
|
|
1213
1546
|
test("gbnfDebug marks an unconstrained observation as not transported", async () => {
|
|
1214
|
-
const p =
|
|
1547
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
1215
1548
|
const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
|
|
1216
1549
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
1217
1550
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1223,7 +1556,7 @@ test("gbnfDebug marks an unconstrained observation as not transported", async ()
|
|
|
1223
1556
|
});
|
|
1224
1557
|
|
|
1225
1558
|
test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
|
|
1226
|
-
const p =
|
|
1559
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
1227
1560
|
const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
|
|
1228
1561
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
1229
1562
|
assert.equal(res.assistant.content, "xon-conforming output");
|
|
@@ -1234,7 +1567,7 @@ test("gbnfDebug preserves conflicting bytes without a provider verdict", async (
|
|
|
1234
1567
|
});
|
|
1235
1568
|
|
|
1236
1569
|
test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
|
|
1237
|
-
const p =
|
|
1570
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true });
|
|
1238
1571
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1239
1572
|
await assert.rejects(
|
|
1240
1573
|
() => p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' }), // no `root` rule → invalid GBNF
|
|
@@ -1246,7 +1579,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
1246
1579
|
// — meta bag: verbatim provider metadata —
|
|
1247
1580
|
|
|
1248
1581
|
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
1249
|
-
const p =
|
|
1582
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
1250
1583
|
const balance = { amount: "0.0000042", currency: "XMR" };
|
|
1251
1584
|
installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
|
|
1252
1585
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -1260,15 +1593,34 @@ const headerVal = (init: RequestInit, name: string): string | undefined =>
|
|
|
1260
1593
|
new Headers(init.headers).get(name) ?? undefined;
|
|
1261
1594
|
|
|
1262
1595
|
test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async () => {
|
|
1263
|
-
const p =
|
|
1596
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1264
1597
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1265
1598
|
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0", "@foo/y@0.3.1"], client: "plurnk.nvim/1.4.0" });
|
|
1266
1599
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), '["@acme/x@1.2.0","@foo/y@0.3.1"]');
|
|
1267
1600
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
|
|
1268
1601
|
});
|
|
1269
1602
|
|
|
1603
|
+
test("Plurnk-Call-Kind carries the caller's emission or bare output contract under the first-party gate", async () => {
|
|
1604
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1605
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1606
|
+
await p.generate({ workerId: "emission", messages: [], callKind: "emission" });
|
|
1607
|
+
await p.generate({ workerId: "bare", messages: [], callKind: "bare" });
|
|
1608
|
+
assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), "emission");
|
|
1609
|
+
assert.equal(headerVal(calls[1].init, "Plurnk-Call-Kind"), "bare");
|
|
1610
|
+
});
|
|
1611
|
+
|
|
1612
|
+
test("generate rejects an unknown call kind before provider I/O", async () => {
|
|
1613
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1614
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1615
|
+
await assert.rejects(
|
|
1616
|
+
p.generate({ workerId: "invalid", messages: [], callKind: "unknown" as never }),
|
|
1617
|
+
/unsupported callKind "unknown"/,
|
|
1618
|
+
);
|
|
1619
|
+
assert.equal(calls.length, 0);
|
|
1620
|
+
});
|
|
1621
|
+
|
|
1270
1622
|
test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
|
|
1271
|
-
const p =
|
|
1623
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1272
1624
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1273
1625
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
1274
1626
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), "w-root"); // a descendant: Primary != Worker-Id
|
|
@@ -1287,22 +1639,23 @@ test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even
|
|
|
1287
1639
|
});
|
|
1288
1640
|
|
|
1289
1641
|
test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
|
|
1290
|
-
const p =
|
|
1642
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1291
1643
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1292
1644
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
1293
1645
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined); // never reaches a third-party backend
|
|
1294
1646
|
});
|
|
1295
1647
|
|
|
1296
1648
|
test("firstPartyMetadata off (default): the headers are structurally dropped even when values are passed", async () => {
|
|
1297
|
-
const p =
|
|
1649
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1298
1650
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1299
|
-
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0" });
|
|
1651
|
+
await p.generate({ workerId: "r", messages: [], attributions: ["@acme/x@1.2.0"], client: "plurnk-cli/2.0.0", callKind: "bare" });
|
|
1300
1652
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined); // never leaks to a non-first-party backend
|
|
1301
1653
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
|
|
1654
|
+
assert.equal(headerVal(calls[0].init, "Plurnk-Call-Kind"), undefined);
|
|
1302
1655
|
});
|
|
1303
1656
|
|
|
1304
1657
|
test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
1305
|
-
const p =
|
|
1658
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1306
1659
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1307
1660
|
await p.generate({ workerId: "r", messages: [], attributions: [], client: "" });
|
|
1308
1661
|
assert.equal(headerVal(calls[0].init, "Plurnk-Attribution"), undefined);
|
|
@@ -1310,7 +1663,7 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
|
1310
1663
|
});
|
|
1311
1664
|
|
|
1312
1665
|
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
|
|
1313
|
-
const p =
|
|
1666
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
1314
1667
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1315
1668
|
await p.generate({ workerId: "r", messages: [] });
|
|
1316
1669
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1319,7 +1672,7 @@ test("grammar transport: no grammar passed sends no grammar field, but the penal
|
|
|
1319
1672
|
});
|
|
1320
1673
|
|
|
1321
1674
|
test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
|
|
1322
|
-
const p =
|
|
1675
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1323
1676
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1324
1677
|
await p.generate({ workerId: "r", messages: [], maxTokens: 2048 });
|
|
1325
1678
|
assert.equal(JSON.parse(calls[0].init.body as string).max_tokens, 2048);
|
|
@@ -1331,7 +1684,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
|
|
|
1331
1684
|
});
|
|
1332
1685
|
|
|
1333
1686
|
test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
|
|
1334
|
-
const pinning =
|
|
1687
|
+
const pinning = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
1335
1688
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1336
1689
|
await pinning.generate({ workerId: "run-A", messages: [] });
|
|
1337
1690
|
await pinning.generate({ workerId: "run-B", messages: [] });
|
|
@@ -1342,20 +1695,20 @@ test("slot affinity is internal: sticky per workerId, distinct workers spread ac
|
|
|
1342
1695
|
});
|
|
1343
1696
|
|
|
1344
1697
|
test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever", async () => {
|
|
1345
|
-
const cloud =
|
|
1698
|
+
const cloud = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // default: no pinning
|
|
1346
1699
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1347
1700
|
await cloud.generate({ workerId: "run-A", messages: [] });
|
|
1348
1701
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
1349
1702
|
|
|
1350
1703
|
mock.restoreAll();
|
|
1351
|
-
const noCount =
|
|
1704
|
+
const noCount = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true }); // slotCount null
|
|
1352
1705
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1353
1706
|
await noCount.generate({ workerId: "run-A", messages: [] });
|
|
1354
1707
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
1355
1708
|
});
|
|
1356
1709
|
|
|
1357
1710
|
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
|
|
1358
|
-
const p =
|
|
1711
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
1359
1712
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1360
1713
|
const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
|
|
1361
1714
|
for (let i = 0; i < 16; i++) await p.generate({ workerId: `r${i}`, messages: [] }); // fills the 16-entry window {r0..r15}
|
|
@@ -1369,7 +1722,7 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
|
|
|
1369
1722
|
|
|
1370
1723
|
test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
|
|
1371
1724
|
const { ProviderError } = await import("./errors.ts");
|
|
1372
|
-
const p =
|
|
1725
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
|
|
1373
1726
|
mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
|
|
1374
1727
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
1375
1728
|
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
@@ -1380,14 +1733,14 @@ test("streaming:false: a non-ok response rejects as a classified ProviderError (
|
|
|
1380
1733
|
});
|
|
1381
1734
|
|
|
1382
1735
|
test("generate fail-hards on a missing or empty workerId", async () => {
|
|
1383
|
-
const p =
|
|
1736
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1384
1737
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1385
1738
|
await assert.rejects(() => p.generate({ workerId: "", messages: [] }), /workerId is required/);
|
|
1386
1739
|
await assert.rejects(() => (p.generate as (a: object) => Promise<unknown>)({ messages: [] }), /workerId is required/);
|
|
1387
1740
|
});
|
|
1388
1741
|
|
|
1389
1742
|
test("messages pass through verbatim — the provider injects no turn (PLAN lives in the grammar, never a provider prefill)", async () => {
|
|
1390
|
-
const p =
|
|
1743
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1391
1744
|
const calls = installFetch([{ choices: [{ delta: { content: "out" } }] }]);
|
|
1392
1745
|
const input = [{ role: "user" as const, content: "hi" }];
|
|
1393
1746
|
const res = await p.generate({ workerId: "r", messages: input });
|
|
@@ -1397,7 +1750,7 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
|
|
|
1397
1750
|
|
|
1398
1751
|
test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
|
|
1399
1752
|
const { ProviderError } = await import("./errors.ts");
|
|
1400
|
-
const p =
|
|
1753
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
|
|
1401
1754
|
mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
|
|
1402
1755
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
1403
1756
|
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
@@ -1411,14 +1764,14 @@ test("generate wraps an HTTP failure as a ProviderError carrying Problem Details
|
|
|
1411
1764
|
});
|
|
1412
1765
|
|
|
1413
1766
|
test("generate rejects on a pre-aborted external signal", async () => {
|
|
1414
|
-
const p =
|
|
1767
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1415
1768
|
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1416
1769
|
const signal = AbortSignal.abort(new Error("nope"));
|
|
1417
1770
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [], signal }));
|
|
1418
1771
|
});
|
|
1419
1772
|
|
|
1420
1773
|
test("configured headers and url are sent verbatim", async () => {
|
|
1421
|
-
const p =
|
|
1774
|
+
const p = testProvider({
|
|
1422
1775
|
model: "m", url: "http://host/custom/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
1423
1776
|
headers: { Authorization: "Bearer secret", "X-Title": "plurnk" },
|
|
1424
1777
|
});
|
|
@@ -1445,14 +1798,16 @@ const stalledStreamResponse = (): Response => new Response(new ReadableStream({
|
|
|
1445
1798
|
|
|
1446
1799
|
test("retry: a transient failure retries and a later success resolves", async () => {
|
|
1447
1800
|
const calls = installFetchScript([
|
|
1801
|
+
{ status: 408, retryAfter: 0 },
|
|
1802
|
+
{ status: 409, retryAfter: 0 },
|
|
1448
1803
|
{ status: 429, retryAfter: 0 },
|
|
1449
1804
|
{ status: 503, retryAfter: 0 },
|
|
1450
1805
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" } }] }] },
|
|
1451
1806
|
]);
|
|
1452
|
-
const p =
|
|
1807
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 4 });
|
|
1453
1808
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
1454
1809
|
assert.equal(res.assistant.content, "ok");
|
|
1455
|
-
assert.equal(calls.length,
|
|
1810
|
+
assert.equal(calls.length, 5); // 408 → 409 → 429 → 503 → 200
|
|
1456
1811
|
});
|
|
1457
1812
|
|
|
1458
1813
|
test("streamed-body silence retries and returns the retry's complete output", async () => {
|
|
@@ -1469,7 +1824,7 @@ test("streamed-body silence retries and returns the retry's complete output", as
|
|
|
1469
1824
|
},
|
|
1470
1825
|
}), { status: 200 });
|
|
1471
1826
|
});
|
|
1472
|
-
const p =
|
|
1827
|
+
const p = testProvider({
|
|
1473
1828
|
model: "m",
|
|
1474
1829
|
url: "http://x/v1/chat/completions",
|
|
1475
1830
|
fetchTimeoutMs: 5000,
|
|
@@ -1492,7 +1847,7 @@ test("streamed-body silence does not replay when retries are disabled", async ()
|
|
|
1492
1847
|
calls++;
|
|
1493
1848
|
return stalledStreamResponse();
|
|
1494
1849
|
});
|
|
1495
|
-
const p =
|
|
1850
|
+
const p = testProvider({
|
|
1496
1851
|
model: "m",
|
|
1497
1852
|
url: "http://x/v1/chat/completions",
|
|
1498
1853
|
fetchTimeoutMs: 1000,
|
|
@@ -1506,7 +1861,8 @@ test("streamed-body silence does not replay when retries are disabled", async ()
|
|
|
1506
1861
|
await assert.rejects(
|
|
1507
1862
|
p.generate({ workerId: "r", messages: [] }),
|
|
1508
1863
|
(error: ProviderError) => error.kind === "network_failure"
|
|
1509
|
-
&&
|
|
1864
|
+
&& error.problem.timeoutPhase === "stream_idle"
|
|
1865
|
+
&& error.problem.timeoutMs === 10,
|
|
1510
1866
|
);
|
|
1511
1867
|
assert.equal(calls, 1, "zero retries permits exactly one provider request");
|
|
1512
1868
|
mock.restoreAll();
|
|
@@ -1518,7 +1874,7 @@ test("streamed-body silence exhausts the configured retry budget once", async ()
|
|
|
1518
1874
|
calls++;
|
|
1519
1875
|
return stalledStreamResponse();
|
|
1520
1876
|
});
|
|
1521
|
-
const p =
|
|
1877
|
+
const p = testProvider({
|
|
1522
1878
|
model: "m",
|
|
1523
1879
|
url: "http://x/v1/chat/completions",
|
|
1524
1880
|
fetchTimeoutMs: 5000,
|
|
@@ -1540,6 +1896,116 @@ test("streamed-body silence exhausts the configured retry budget once", async ()
|
|
|
1540
1896
|
mock.restoreAll();
|
|
1541
1897
|
});
|
|
1542
1898
|
|
|
1899
|
+
test("an attempt timeout retries within the larger operation deadline and settles every physical request", async () => {
|
|
1900
|
+
let calls = 0;
|
|
1901
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
1902
|
+
calls++;
|
|
1903
|
+
if (calls > 1) {
|
|
1904
|
+
return new Response(sseStream([
|
|
1905
|
+
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
1906
|
+
]), { status: 200 });
|
|
1907
|
+
}
|
|
1908
|
+
return await new Promise<Response>((_resolve, reject) => {
|
|
1909
|
+
const signal = init?.signal;
|
|
1910
|
+
signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
1911
|
+
});
|
|
1912
|
+
});
|
|
1913
|
+
const connectivity = { operationTimeoutMs: 5_000 };
|
|
1914
|
+
const settled: Array<{ outcome: string }> = [];
|
|
1915
|
+
const p = testProvider({
|
|
1916
|
+
model: "m",
|
|
1917
|
+
url: "http://x/v1/chat/completions",
|
|
1918
|
+
fetchTimeoutMs: 10,
|
|
1919
|
+
streamIdleTimeoutMs: 0,
|
|
1920
|
+
temperature: 0.2,
|
|
1921
|
+
repeatPenalty: 1.15,
|
|
1922
|
+
reasoning: { mode: "off", budget: null },
|
|
1923
|
+
retryAttempts: 1,
|
|
1924
|
+
source: "provider:test",
|
|
1925
|
+
...connectivity,
|
|
1926
|
+
});
|
|
1927
|
+
const result = await p.generate({
|
|
1928
|
+
workerId: "r",
|
|
1929
|
+
messages: [],
|
|
1930
|
+
observeRequest: async () => async (accounting) => { settled.push(accounting); },
|
|
1931
|
+
});
|
|
1932
|
+
assert.equal(result.assistant.content, "recovered");
|
|
1933
|
+
assert.equal(calls, 2);
|
|
1934
|
+
assert.deepEqual(settled.map(({ outcome }) => outcome), ["error", "response"]);
|
|
1935
|
+
assert.deepEqual(result.accounting.map(({ outcome }) => outcome), ["error", "response"]);
|
|
1936
|
+
mock.restoreAll();
|
|
1937
|
+
});
|
|
1938
|
+
|
|
1939
|
+
test("first-content silence retries independently of the stream-idle deadline", async () => {
|
|
1940
|
+
let calls = 0;
|
|
1941
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1942
|
+
calls++;
|
|
1943
|
+
if (calls > 1) {
|
|
1944
|
+
return new Response(sseStream([
|
|
1945
|
+
{ choices: [{ delta: { content: "recovered" }, finish_reason: "stop" }] },
|
|
1946
|
+
]), { status: 200 });
|
|
1947
|
+
}
|
|
1948
|
+
return new Response(new ReadableStream({
|
|
1949
|
+
start(controller) {
|
|
1950
|
+
setTimeout(() => controller.close(), 100);
|
|
1951
|
+
},
|
|
1952
|
+
}), { status: 200 });
|
|
1953
|
+
});
|
|
1954
|
+
const connectivity = { operationTimeoutMs: 5_000, firstContentTimeoutMs: 10 };
|
|
1955
|
+
const p = testProvider({
|
|
1956
|
+
model: "m",
|
|
1957
|
+
url: "http://x/v1/chat/completions",
|
|
1958
|
+
fetchTimeoutMs: 5_000,
|
|
1959
|
+
streamIdleTimeoutMs: 0,
|
|
1960
|
+
temperature: 0.2,
|
|
1961
|
+
repeatPenalty: 1.15,
|
|
1962
|
+
reasoning: { mode: "off", budget: null },
|
|
1963
|
+
retryAttempts: 1,
|
|
1964
|
+
source: "provider:test",
|
|
1965
|
+
...connectivity,
|
|
1966
|
+
});
|
|
1967
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
1968
|
+
assert.equal(result.assistant.content, "recovered");
|
|
1969
|
+
assert.equal(calls, 2);
|
|
1970
|
+
mock.restoreAll();
|
|
1971
|
+
});
|
|
1972
|
+
|
|
1973
|
+
test("operation-deadline exhaustion is a distinct non-retryable failure", async () => {
|
|
1974
|
+
let calls = 0;
|
|
1975
|
+
mock.method(globalThis, "fetch", async (_input: string | URL | Request, init?: RequestInit) => {
|
|
1976
|
+
calls++;
|
|
1977
|
+
return await new Promise<Response>((_resolve, reject) => {
|
|
1978
|
+
const signal = init?.signal;
|
|
1979
|
+
signal?.addEventListener("abort", () => reject(signal.reason), { once: true });
|
|
1980
|
+
});
|
|
1981
|
+
});
|
|
1982
|
+
const connectivity = { operationTimeoutMs: 10 };
|
|
1983
|
+
const p = testProvider({
|
|
1984
|
+
model: "m",
|
|
1985
|
+
url: "http://x/v1/chat/completions",
|
|
1986
|
+
fetchTimeoutMs: 50,
|
|
1987
|
+
streamIdleTimeoutMs: 0,
|
|
1988
|
+
temperature: 0.2,
|
|
1989
|
+
repeatPenalty: 1.15,
|
|
1990
|
+
reasoning: { mode: "off", budget: null },
|
|
1991
|
+
retryAttempts: 3,
|
|
1992
|
+
source: "provider:test",
|
|
1993
|
+
...connectivity,
|
|
1994
|
+
});
|
|
1995
|
+
await assert.rejects(
|
|
1996
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
1997
|
+
(error: ProviderError) => error.kind === "deadline_exceeded"
|
|
1998
|
+
&& error.status === 504
|
|
1999
|
+
&& error.problem.retryable === false
|
|
2000
|
+
&& error.problem.timeoutPhase === "operation"
|
|
2001
|
+
&& error.problem.timeoutMs === 10
|
|
2002
|
+
&& error.accounting.length === 1
|
|
2003
|
+
&& error.accounting[0]?.outcome === "error",
|
|
2004
|
+
);
|
|
2005
|
+
assert.equal(calls, 1);
|
|
2006
|
+
mock.restoreAll();
|
|
2007
|
+
});
|
|
2008
|
+
|
|
1543
2009
|
test("the total generation deadline spans stalled-stream retry scheduling", async () => {
|
|
1544
2010
|
let calls = 0;
|
|
1545
2011
|
mock.method(globalThis, "fetch", async () => {
|
|
@@ -1556,10 +2022,11 @@ test("the total generation deadline spans stalled-stream retry scheduling", asyn
|
|
|
1556
2022
|
}
|
|
1557
2023
|
return stalledStreamResponse();
|
|
1558
2024
|
});
|
|
1559
|
-
const p =
|
|
2025
|
+
const p = testProvider({
|
|
1560
2026
|
model: "m",
|
|
1561
2027
|
url: "http://x/v1/chat/completions",
|
|
1562
|
-
fetchTimeoutMs:
|
|
2028
|
+
fetchTimeoutMs: 5000,
|
|
2029
|
+
operationTimeoutMs: 50,
|
|
1563
2030
|
streamIdleTimeoutMs: 10,
|
|
1564
2031
|
temperature: 0.2,
|
|
1565
2032
|
repeatPenalty: 1.15,
|
|
@@ -1570,7 +2037,8 @@ test("the total generation deadline spans stalled-stream retry scheduling", asyn
|
|
|
1570
2037
|
const started = Date.now();
|
|
1571
2038
|
await assert.rejects(
|
|
1572
2039
|
p.generate({ workerId: "r", messages: [] }),
|
|
1573
|
-
(error: ProviderError) => error.kind === "
|
|
2040
|
+
(error: ProviderError) => error.kind === "deadline_exceeded"
|
|
2041
|
+
&& error.problem.timeoutPhase === "operation",
|
|
1574
2042
|
);
|
|
1575
2043
|
assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
|
|
1576
2044
|
assert.equal(calls, 1, "the total deadline expires before another request begins");
|
|
@@ -1586,7 +2054,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
|
|
|
1586
2054
|
controller.close();
|
|
1587
2055
|
},
|
|
1588
2056
|
}), { status: 200 }));
|
|
1589
|
-
const p =
|
|
2057
|
+
const p = testProvider({
|
|
1590
2058
|
model: "m",
|
|
1591
2059
|
url: "http://x/v1/chat/completions",
|
|
1592
2060
|
fetchTimeoutMs: 1000,
|
|
@@ -1604,7 +2072,7 @@ test("a zero stream-idle timeout permits a slow inter-chunk pause", async () =>
|
|
|
1604
2072
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
1605
2073
|
const { ProviderError } = await import("./errors.ts");
|
|
1606
2074
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
|
1607
|
-
const p =
|
|
2075
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 2 });
|
|
1608
2076
|
await assert.rejects(
|
|
1609
2077
|
() => p.generate({ workerId: "r", messages: [] }),
|
|
1610
2078
|
(err: unknown) => { assert.ok(err instanceof ProviderError); assert.equal(err.kind, "rate_limit"); return true; },
|
|
@@ -1617,7 +2085,7 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
1617
2085
|
{ status: 503, retryAfter: "Wed, 21 Oct 2015 07:28:00 GMT" }, // date form, in the past → max(0, past−now) = 0
|
|
1618
2086
|
{ status: 200, chunks: [{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }] },
|
|
1619
2087
|
]);
|
|
1620
|
-
const p =
|
|
2088
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 1 });
|
|
1621
2089
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
1622
2090
|
assert.equal(assistant.content, "ok");
|
|
1623
2091
|
assert.equal(calls.length, 2); // initial 503 + one retry, no real wall-clock wait
|
|
@@ -1625,14 +2093,14 @@ test("retry: a Retry-After HTTP-date is honored — a past date parses to a 0ms
|
|
|
1625
2093
|
|
|
1626
2094
|
test("retry: a terminal error (401 unauthorized) is never retried", async () => {
|
|
1627
2095
|
const calls = installFetchScript([{ status: 401 }]);
|
|
1628
|
-
const p =
|
|
2096
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 5 });
|
|
1629
2097
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), /401/);
|
|
1630
2098
|
assert.equal(calls.length, 1); // terminal — no retry despite budget
|
|
1631
2099
|
});
|
|
1632
2100
|
|
|
1633
2101
|
test("retry: retryAttempts 0 surfaces the first transient failure immediately", async () => {
|
|
1634
2102
|
const calls = installFetchScript([{ status: 503, retryAfter: 0 }]);
|
|
1635
|
-
const p =
|
|
2103
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 0 });
|
|
1636
2104
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }));
|
|
1637
2105
|
assert.equal(calls.length, 1); // no retry budget
|
|
1638
2106
|
});
|
|
@@ -1640,7 +2108,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
|
|
|
1640
2108
|
test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
|
|
1641
2109
|
const ac = new AbortController();
|
|
1642
2110
|
const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
|
|
1643
|
-
const p =
|
|
2111
|
+
const p = testProvider({ ...retryCfg, retryAttempts: 3 });
|
|
1644
2112
|
const promise = p.generate({ workerId: "r", messages: [], signal: ac.signal });
|
|
1645
2113
|
await flush(); // attempt 0 fails, enters the backoff sleep
|
|
1646
2114
|
assert.equal(calls.length, 1);
|
|
@@ -1651,23 +2119,29 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
1651
2119
|
|
|
1652
2120
|
// — Anthropic reasoning style (wire `thinking` parameter) —
|
|
1653
2121
|
|
|
1654
|
-
test("reasoningStyle 'anthropic' maps
|
|
2122
|
+
test("reasoningStyle 'anthropic' maps an optional budget or the resolved reserve to the thinking param", async () => {
|
|
1655
2123
|
// N>0 → enabled with budget_tokens
|
|
1656
|
-
const capped =
|
|
2124
|
+
const capped = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: 4096 }, reasoningStyle: "anthropic" });
|
|
1657
2125
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1658
2126
|
await capped.generate({ workerId: "r", messages: [] });
|
|
1659
2127
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 4096 });
|
|
1660
2128
|
|
|
2129
|
+
mock.restoreAll();
|
|
2130
|
+
const unbudgeted = testProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 8192, reasoningReserve: { tokens: 2048 }, fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "on", budget: null }, reasoningStyle: "anthropic" });
|
|
2131
|
+
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
2132
|
+
await unbudgeted.generate({ workerId: "r", messages: [] });
|
|
2133
|
+
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "enabled", budget_tokens: 2048 });
|
|
2134
|
+
|
|
1661
2135
|
mock.restoreAll();
|
|
1662
2136
|
// 0 → explicit disabled
|
|
1663
|
-
const off =
|
|
2137
|
+
const off = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, reasoningStyle: "anthropic" });
|
|
1664
2138
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1665
2139
|
await off.generate({ workerId: "r", messages: [] });
|
|
1666
2140
|
assert.deepEqual(JSON.parse(calls[0].init.body as string).thinking, { type: "disabled" });
|
|
1667
2141
|
|
|
1668
2142
|
mock.restoreAll();
|
|
1669
2143
|
// -1 adaptive → omit (API default depth)
|
|
1670
|
-
const adaptive =
|
|
2144
|
+
const adaptive = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, retryAttempts: 0, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, reasoningStyle: "anthropic" });
|
|
1671
2145
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1672
2146
|
await adaptive.generate({ workerId: "r", messages: [] });
|
|
1673
2147
|
assert.equal("thinking" in JSON.parse(calls[0].init.body as string), false);
|
|
@@ -1685,14 +2159,14 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1685
2159
|
usage: { prompt_tokens: 3, completion_tokens: 1, total_tokens: 4 },
|
|
1686
2160
|
}), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
1687
2161
|
});
|
|
1688
|
-
const p =
|
|
2162
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
1689
2163
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
1690
2164
|
const sent = JSON.parse(calls[0].body);
|
|
1691
2165
|
assert.equal("stream" in sent, false); // no streaming flag
|
|
1692
2166
|
assert.equal(res.assistant.content, "hello"); // content from message.content
|
|
1693
2167
|
assert.equal(res.assistant.reasoning, "because"); // reasoning_content mapped
|
|
1694
2168
|
assert.equal(res.assistant.finishReason, "stop");
|
|
1695
|
-
assert.equal(res.
|
|
2169
|
+
assert.equal(res.accounting[0]?.usage?.totalTokens, 4);
|
|
1696
2170
|
mock.restoreAll();
|
|
1697
2171
|
});
|
|
1698
2172
|
|
|
@@ -1701,7 +2175,7 @@ const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTime
|
|
|
1701
2175
|
|
|
1702
2176
|
test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
|
|
1703
2177
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1704
|
-
const p =
|
|
2178
|
+
const p = testProvider({ ...captureBase });
|
|
1705
2179
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1706
2180
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1707
2181
|
assert.equal("logprobs" in body, false);
|
|
@@ -1718,7 +2192,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
|
|
|
1718
2192
|
{ token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
|
|
1719
2193
|
] } }] };
|
|
1720
2194
|
const calls = installFetch([chunk]);
|
|
1721
|
-
const p =
|
|
2195
|
+
const p = testProvider({ ...captureBase, topLogprobs: 2 });
|
|
1722
2196
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1723
2197
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1724
2198
|
assert.equal(body.logprobs, true);
|
|
@@ -1732,7 +2206,7 @@ test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logpr
|
|
|
1732
2206
|
test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
|
|
1733
2207
|
const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
1734
2208
|
installFetchJson(wire);
|
|
1735
|
-
const p =
|
|
2209
|
+
const p = testProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
|
|
1736
2210
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
1737
2211
|
assert.deepEqual(res.rawBody, wire); // verbatim
|
|
1738
2212
|
assert.equal((res.rawBody as typeof wire).choices[0].logprobs.content[0].sampling_logprob, -0.1);
|
|
@@ -1743,7 +2217,7 @@ test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob prese
|
|
|
1743
2217
|
|
|
1744
2218
|
test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
|
|
1745
2219
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1746
|
-
const p =
|
|
2220
|
+
const p = testProvider({ ...captureBase }); // logprobs OFF
|
|
1747
2221
|
await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
|
|
1748
2222
|
const body = JSON.parse((calls[0].init.body as string));
|
|
1749
2223
|
assert.equal("logprobs" in body, false); // sampling passthrough stripped it
|
|
@@ -1754,7 +2228,7 @@ test("caller sampling cannot forge logprobs (reserved keys): the env flag is the
|
|
|
1754
2228
|
// — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
|
|
1755
2229
|
|
|
1756
2230
|
test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
|
|
1757
|
-
const p =
|
|
2231
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1758
2232
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1759
2233
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1760
2234
|
const headers = new Headers(calls[0].init.headers);
|
|
@@ -1764,7 +2238,7 @@ test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the firs
|
|
|
1764
2238
|
});
|
|
1765
2239
|
|
|
1766
2240
|
test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
|
|
1767
|
-
const p =
|
|
2241
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1768
2242
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1769
2243
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
1770
2244
|
const headers = new Headers(calls[0].init.headers);
|
|
@@ -1774,7 +2248,7 @@ test("third-party providers structurally DROP the coordinate (gate off by defaul
|
|
|
1774
2248
|
});
|
|
1775
2249
|
|
|
1776
2250
|
test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
|
|
1777
|
-
const p =
|
|
2251
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1778
2252
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1779
2253
|
await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
|
|
1780
2254
|
const headers = new Headers(calls[0].init.headers);
|
|
@@ -1788,18 +2262,18 @@ test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
|
|
|
1788
2262
|
|
|
1789
2263
|
test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
|
|
1790
2264
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1791
|
-
const derived =
|
|
2265
|
+
const derived = testProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
|
|
1792
2266
|
assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
|
|
1793
2267
|
assert.equal(derived.completionReserve, 12288); // 25% of 49152
|
|
1794
|
-
const pinned =
|
|
2268
|
+
const pinned = testProvider({ ...base, contextWindow: null, reasoningReserve: { tokens: 4096 }, completionReserve: { percent: 0.25 } });
|
|
1795
2269
|
assert.equal(pinned.reasoningReserve, 4096); // absolute pin needs no window
|
|
1796
2270
|
assert.equal(pinned.completionReserve, null); // percent without a window = underivable
|
|
1797
|
-
const legacy =
|
|
2271
|
+
const legacy = testProvider({ ...base, contextWindow: 49152 });
|
|
1798
2272
|
assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
|
|
1799
2273
|
});
|
|
1800
2274
|
|
|
1801
2275
|
test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
|
|
1802
|
-
const p =
|
|
2276
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
|
|
1803
2277
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1804
2278
|
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
|
|
1805
2279
|
const body = JSON.parse(calls[0].init.body as string);
|
|
@@ -1807,25 +2281,148 @@ test("router-owned tuning: tuningFloors:false drops the temperature/penalty floo
|
|
|
1807
2281
|
assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
|
|
1808
2282
|
});
|
|
1809
2283
|
|
|
1810
|
-
// --
|
|
2284
|
+
// -- {§provider-cache-affinity} / {§provider-cache-write-policy} --
|
|
1811
2285
|
|
|
1812
|
-
test("
|
|
1813
|
-
const p =
|
|
2286
|
+
test("a compatible route's declared body affinity is managed by workerId", async () => {
|
|
2287
|
+
const p = testProvider({
|
|
2288
|
+
model: "m",
|
|
2289
|
+
url: "http://x/v1/chat/completions",
|
|
2290
|
+
fetchTimeoutMs: 5000,
|
|
2291
|
+
temperature: 0.2,
|
|
2292
|
+
repeatPenalty: 1.15,
|
|
2293
|
+
reasoning: { mode: "off", budget: null },
|
|
2294
|
+
retryAttempts: 0,
|
|
2295
|
+
cacheAffinity: { target: "body", name: "prompt_cache_key" },
|
|
2296
|
+
});
|
|
1814
2297
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1815
|
-
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
2298
|
+
await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
|
|
1816
2299
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
|
|
1817
2300
|
});
|
|
1818
2301
|
|
|
1819
|
-
test("
|
|
1820
|
-
const p =
|
|
2302
|
+
test("an undeclared compatible route receives no guessed cache field", async () => {
|
|
2303
|
+
const p = testProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1821
2304
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1822
2305
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1823
2306
|
assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
|
|
1824
2307
|
});
|
|
1825
2308
|
|
|
1826
|
-
test("
|
|
1827
|
-
const p =
|
|
2309
|
+
test("a compatible route's declared header affinity composes with static headers", async () => {
|
|
2310
|
+
const p = testProvider({
|
|
2311
|
+
model: "m",
|
|
2312
|
+
url: "http://x/v1/chat/completions",
|
|
2313
|
+
headers: { Authorization: "Bearer key" },
|
|
2314
|
+
fetchTimeoutMs: 5000,
|
|
2315
|
+
temperature: 0.2,
|
|
2316
|
+
repeatPenalty: 1.15,
|
|
2317
|
+
reasoning: { mode: "off", budget: null },
|
|
2318
|
+
retryAttempts: 0,
|
|
2319
|
+
cacheAffinity: { target: "header", name: "x-grok-conv-id" },
|
|
2320
|
+
});
|
|
1828
2321
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1829
|
-
await p.generate({ workerId: "worker-abc", messages: []
|
|
1830
|
-
|
|
2322
|
+
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
2323
|
+
const headers = new Headers(calls[0].init.headers);
|
|
2324
|
+
assert.equal(headers.get("authorization"), "Bearer key");
|
|
2325
|
+
assert.equal(headers.get("x-grok-conv-id"), "worker-abc");
|
|
2326
|
+
});
|
|
2327
|
+
|
|
2328
|
+
test("native request projections compose reasoning visibility, affinity, and system cache control", async () => {
|
|
2329
|
+
let request: Record<string, unknown> | undefined;
|
|
2330
|
+
const usage = {
|
|
2331
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
2332
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
2333
|
+
};
|
|
2334
|
+
const languageModel = {
|
|
2335
|
+
specificationVersion: "v4",
|
|
2336
|
+
provider: "native.test",
|
|
2337
|
+
modelId: "native-cache",
|
|
2338
|
+
supportedUrls: {},
|
|
2339
|
+
doGenerate: async (options: Record<string, unknown>) => {
|
|
2340
|
+
request = options;
|
|
2341
|
+
return {
|
|
2342
|
+
content: [{ type: "text", text: "ok" }],
|
|
2343
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
2344
|
+
usage,
|
|
2345
|
+
response: { id: "response", modelId: "native-cache" },
|
|
2346
|
+
warnings: [],
|
|
2347
|
+
};
|
|
2348
|
+
},
|
|
2349
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2350
|
+
} as unknown as LanguageModel;
|
|
2351
|
+
const p = testProvider({
|
|
2352
|
+
model: "native-cache",
|
|
2353
|
+
languageModel,
|
|
2354
|
+
fetchTimeoutMs: 5000,
|
|
2355
|
+
temperature: 0.2,
|
|
2356
|
+
repeatPenalty: 1.15,
|
|
2357
|
+
reasoning: { mode: "adaptive", budget: null },
|
|
2358
|
+
retryAttempts: 0,
|
|
2359
|
+
streaming: false,
|
|
2360
|
+
cacheAffinity: { target: "provider-option", provider: "openai", name: "promptCacheKey" },
|
|
2361
|
+
reasoningResponseProviderOptions: {
|
|
2362
|
+
google: { thinkingConfig: { includeThoughts: true } },
|
|
2363
|
+
},
|
|
2364
|
+
systemCacheProviderOptions: {
|
|
2365
|
+
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
2366
|
+
},
|
|
2367
|
+
});
|
|
2368
|
+
await p.generate({
|
|
2369
|
+
workerId: "worker-native",
|
|
2370
|
+
messages: [
|
|
2371
|
+
{ role: "system", content: "stable definition" },
|
|
2372
|
+
{ role: "system", content: "stable policy" },
|
|
2373
|
+
{ role: "user", content: "changing packet" },
|
|
2374
|
+
],
|
|
2375
|
+
});
|
|
2376
|
+
|
|
2377
|
+
assert.deepEqual(request?.providerOptions, {
|
|
2378
|
+
google: { thinkingConfig: { includeThoughts: true } },
|
|
2379
|
+
openai: { promptCacheKey: "worker-native" },
|
|
2380
|
+
});
|
|
2381
|
+
assert.deepEqual(request?.prompt, [
|
|
2382
|
+
{ role: "system", content: "stable definition", providerOptions: undefined },
|
|
2383
|
+
{
|
|
2384
|
+
role: "system",
|
|
2385
|
+
content: "stable policy",
|
|
2386
|
+
providerOptions: { anthropic: { cacheControl: { type: "ephemeral" } } },
|
|
2387
|
+
},
|
|
2388
|
+
{ role: "user", content: [{ type: "text", text: "changing packet" }], providerOptions: undefined },
|
|
2389
|
+
]);
|
|
2390
|
+
});
|
|
2391
|
+
|
|
2392
|
+
test("native AI SDK reasoning turns on without an operator token budget", async () => {
|
|
2393
|
+
let request: Record<string, unknown> | undefined;
|
|
2394
|
+
const languageModel = {
|
|
2395
|
+
specificationVersion: "v4",
|
|
2396
|
+
provider: "native.test",
|
|
2397
|
+
modelId: "native-reasoning",
|
|
2398
|
+
supportedUrls: {},
|
|
2399
|
+
doGenerate: async (options: Record<string, unknown>) => {
|
|
2400
|
+
request = options;
|
|
2401
|
+
return {
|
|
2402
|
+
content: [{ type: "reasoning", text: "consider" }, { type: "text", text: "ok" }],
|
|
2403
|
+
finishReason: { unified: "stop", raw: "stop" },
|
|
2404
|
+
usage: {
|
|
2405
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
2406
|
+
outputTokens: { total: 2, text: 1, reasoning: 1 },
|
|
2407
|
+
},
|
|
2408
|
+
response: { id: "response", modelId: "native-reasoning" },
|
|
2409
|
+
warnings: [],
|
|
2410
|
+
};
|
|
2411
|
+
},
|
|
2412
|
+
doStream: async () => { throw new Error("streaming is not under test"); },
|
|
2413
|
+
} as unknown as LanguageModel;
|
|
2414
|
+
const p = testProvider({
|
|
2415
|
+
model: "native-reasoning",
|
|
2416
|
+
languageModel,
|
|
2417
|
+
fetchTimeoutMs: 5000,
|
|
2418
|
+
temperature: 0.2,
|
|
2419
|
+
repeatPenalty: 1.15,
|
|
2420
|
+
reasoning: { mode: "on", budget: null },
|
|
2421
|
+
retryAttempts: 0,
|
|
2422
|
+
streaming: false,
|
|
2423
|
+
});
|
|
2424
|
+
const response = await p.generate({ workerId: "worker-native", messages: [{ role: "user", content: "hello" }] });
|
|
2425
|
+
|
|
2426
|
+
assert.equal(request?.reasoning, "medium");
|
|
2427
|
+
assert.equal(response.assistant.reasoning, "consider");
|
|
1831
2428
|
});
|