@plurnk/plurnk-providers 1.3.12 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +47 -38
- package/README.md +68 -4
- package/SPEC.md +245 -62
- package/dist/AiSdkProvider.d.ts +19 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +264 -156
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +15 -17
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +27 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/accounting.d.ts +3 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +84 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +5 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +91 -5
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +5 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +24 -11
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +61 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -6
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +27 -29
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +152 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +12 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +8 -5
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +17 -6
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +52 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +4 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +64 -19
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +15 -10
- package/src/AiSdkProvider.test.ts +750 -169
- package/src/AiSdkProvider.ts +354 -200
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +36 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/accounting.test.ts +58 -0
- package/src/accounting.ts +88 -0
- package/src/aiSdkTransport.ts +101 -8
- package/src/boundaries.test.ts +9 -3
- package/src/catalogProvider.test.ts +43 -15
- package/src/catalogProvider.ts +32 -16
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +64 -0
- package/src/cost.ts +78 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -48
- package/src/env.ts +43 -40
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +208 -0
- package/src/index.ts +30 -8
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +29 -3
- package/src/sdkModels.ts +19 -11
- package/src/types.ts +125 -64
- package/src/usage.test.ts +24 -5
- package/src/usage.ts +72 -21
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
4
|
-
import { ProviderError } from "./
|
|
4
|
+
import { ProviderError } from "./errors.ts";
|
|
5
|
+
import { authoritativeChargeNormalizer } from "./accounting.ts";
|
|
6
|
+
import type { LanguageModel } from "ai";
|
|
5
7
|
|
|
6
8
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
7
9
|
// so tests can assert what the spine sent on the wire.
|
|
@@ -71,7 +73,7 @@ const injectedBase = {
|
|
|
71
73
|
reasoning: { mode: "off" as const, budget: null },
|
|
72
74
|
};
|
|
73
75
|
|
|
74
|
-
test("
|
|
76
|
+
test("per-instance fetch owns streaming and buffered requests", async () => {
|
|
75
77
|
const calls: Array<{ input: string | URL | Request; init?: RequestInit }> = [];
|
|
76
78
|
const streamingFetch: typeof globalThis.fetch = async (input, init) => {
|
|
77
79
|
calls.push({ input, init });
|
|
@@ -105,7 +107,7 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
105
107
|
assert.equal(JSON.parse(String(calls[1].init?.body)).stream, undefined);
|
|
106
108
|
});
|
|
107
109
|
|
|
108
|
-
test("
|
|
110
|
+
test("caller cancellation and provider timeout reach an injected fetch", async () => {
|
|
109
111
|
const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
|
|
110
112
|
init?.signal?.throwIfAborted();
|
|
111
113
|
return new Promise((_resolve, reject) => {
|
|
@@ -126,7 +128,7 @@ test("#608: caller cancellation and provider timeout reach an injected fetch", a
|
|
|
126
128
|
);
|
|
127
129
|
});
|
|
128
130
|
|
|
129
|
-
test("
|
|
131
|
+
test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
130
132
|
const calls: string[] = [];
|
|
131
133
|
let generationAttempts = 0;
|
|
132
134
|
const providerFetch: typeof globalThis.fetch = async (input) => {
|
|
@@ -192,7 +194,7 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
|
192
194
|
const flush = () => new Promise<void>((r) => setImmediate(r));
|
|
193
195
|
|
|
194
196
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
195
|
-
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
197
|
+
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
196
198
|
|
|
197
199
|
test("effortFromBudget: maps budget to tiers", () => {
|
|
198
200
|
assert.equal(effortFromBudget(1), "low");
|
|
@@ -202,7 +204,7 @@ test("effortFromBudget: maps budget to tiers", () => {
|
|
|
202
204
|
assert.equal(effortFromBudget(4001), "high");
|
|
203
205
|
});
|
|
204
206
|
|
|
205
|
-
test("
|
|
207
|
+
test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
206
208
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
207
209
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
208
210
|
await assert.rejects(p.generate({ workerId: "r", messages: [] }));
|
|
@@ -211,7 +213,7 @@ test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retry
|
|
|
211
213
|
mock.restoreAll();
|
|
212
214
|
});
|
|
213
215
|
|
|
214
|
-
test("
|
|
216
|
+
test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
|
|
215
217
|
const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
|
|
216
218
|
const calls = installFetchScript([{ status: 422, body }]);
|
|
217
219
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
|
|
@@ -237,43 +239,59 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
237
239
|
assert.equal(calls.length, 1);
|
|
238
240
|
});
|
|
239
241
|
|
|
240
|
-
test("
|
|
242
|
+
test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
|
|
241
243
|
installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
242
244
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
243
245
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
244
246
|
assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
|
|
245
247
|
});
|
|
246
248
|
|
|
247
|
-
test("
|
|
249
|
+
test("without a probed eos_token the content passes through untouched", async () => {
|
|
248
250
|
installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
249
251
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
250
252
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
251
253
|
assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
|
|
252
254
|
});
|
|
253
255
|
|
|
254
|
-
test("
|
|
256
|
+
test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
|
|
255
257
|
installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
|
|
256
258
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
257
259
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
258
260
|
assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
|
|
259
261
|
});
|
|
260
262
|
|
|
261
|
-
test("identity getters and
|
|
263
|
+
test("identity getters and default prompt estimate", async () => {
|
|
262
264
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
263
265
|
assert.equal(p.model, "m");
|
|
264
266
|
assert.equal(p.contextWindow, null); // default
|
|
265
|
-
assert.
|
|
266
|
-
|
|
267
|
-
|
|
267
|
+
assert.deepEqual(
|
|
268
|
+
await p.countPromptTokens([{ role: "user", content: "漢漢漢" }]),
|
|
269
|
+
{
|
|
270
|
+
kind: "estimate",
|
|
271
|
+
tokens: 2,
|
|
272
|
+
source: "heuristic:chars2",
|
|
273
|
+
detail: "chars/2 over message content; provider request framing is unknown",
|
|
274
|
+
},
|
|
275
|
+
"chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
|
|
276
|
+
);
|
|
277
|
+
assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
|
|
268
278
|
});
|
|
269
279
|
|
|
270
|
-
test("injected
|
|
280
|
+
test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
|
|
281
|
+
const seen: string[] = [];
|
|
271
282
|
const p = new AiSdkProvider({
|
|
272
283
|
model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
273
|
-
|
|
284
|
+
countPromptTokens: (messages) => {
|
|
285
|
+
seen.push(...messages.map(({ content }) => content));
|
|
286
|
+
return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
|
|
287
|
+
},
|
|
274
288
|
calculateCost: (u) => u.total * 2,
|
|
275
289
|
});
|
|
276
|
-
assert.
|
|
290
|
+
assert.deepEqual(
|
|
291
|
+
await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
|
|
292
|
+
{ kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
|
|
293
|
+
);
|
|
294
|
+
assert.deepEqual(seen, ["system", "user"]);
|
|
277
295
|
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
|
|
278
296
|
});
|
|
279
297
|
|
|
@@ -293,6 +311,139 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
293
311
|
assert.notEqual(assistantRaw, undefined);
|
|
294
312
|
});
|
|
295
313
|
|
|
314
|
+
test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
|
|
315
|
+
const usage = {
|
|
316
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
317
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
318
|
+
};
|
|
319
|
+
const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
|
|
320
|
+
const charge = {
|
|
321
|
+
kind: "authoritative",
|
|
322
|
+
amount: { amount: "0.00154935", currency: "USD" },
|
|
323
|
+
usdEquivalent: "0.00154935",
|
|
324
|
+
source: "OpenRouter response usage.cost",
|
|
325
|
+
};
|
|
326
|
+
const languageModel = {
|
|
327
|
+
specificationVersion: "v4",
|
|
328
|
+
provider: "openrouter.chat",
|
|
329
|
+
modelId: "router-test",
|
|
330
|
+
supportedUrls: {},
|
|
331
|
+
doGenerate: async () => ({
|
|
332
|
+
content: [{ type: "text", text: "ok" }],
|
|
333
|
+
finishReason: { unified: "stop", raw: "completed" },
|
|
334
|
+
usage,
|
|
335
|
+
providerMetadata,
|
|
336
|
+
response: { id: "response-buffered", modelId: "router-test" },
|
|
337
|
+
warnings: [],
|
|
338
|
+
}),
|
|
339
|
+
doStream: async () => ({
|
|
340
|
+
stream: new ReadableStream({
|
|
341
|
+
start(controller) {
|
|
342
|
+
controller.enqueue({ type: "stream-start", warnings: [] });
|
|
343
|
+
controller.enqueue({ type: "response-metadata", id: "response-streamed", modelId: "router-test" });
|
|
344
|
+
controller.enqueue({ type: "text-start", id: "text-1" });
|
|
345
|
+
controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
|
|
346
|
+
controller.enqueue({ type: "text-end", id: "text-1" });
|
|
347
|
+
controller.enqueue({
|
|
348
|
+
type: "finish",
|
|
349
|
+
finishReason: { unified: "stop", raw: "completed" },
|
|
350
|
+
usage,
|
|
351
|
+
providerMetadata,
|
|
352
|
+
});
|
|
353
|
+
controller.close();
|
|
354
|
+
},
|
|
355
|
+
}),
|
|
356
|
+
response: {},
|
|
357
|
+
}),
|
|
358
|
+
} as unknown as LanguageModel;
|
|
359
|
+
const config = {
|
|
360
|
+
model: "router-test",
|
|
361
|
+
languageModel,
|
|
362
|
+
fetchTimeoutMs: 5_000,
|
|
363
|
+
temperature: 0.2,
|
|
364
|
+
repeatPenalty: 1.15,
|
|
365
|
+
reasoning: { mode: "off" as const, budget: null },
|
|
366
|
+
retryAttempts: 0,
|
|
367
|
+
normalizeCharge: authoritativeChargeNormalizer("@openrouter/ai-sdk-provider"),
|
|
368
|
+
};
|
|
369
|
+
|
|
370
|
+
await t.test("buffered", async () => {
|
|
371
|
+
const response = await new AiSdkProvider({ ...config, streaming: false })
|
|
372
|
+
.generate({ workerId: "buffered", messages: [] });
|
|
373
|
+
assert.deepEqual(response.charge, charge);
|
|
374
|
+
});
|
|
375
|
+
await t.test("streamed", async () => {
|
|
376
|
+
const response = await new AiSdkProvider(config)
|
|
377
|
+
.generate({ workerId: "streamed", messages: [] });
|
|
378
|
+
assert.deepEqual(response.charge, charge);
|
|
379
|
+
});
|
|
380
|
+
});
|
|
381
|
+
|
|
382
|
+
test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
|
|
383
|
+
const p = new AiSdkProvider({
|
|
384
|
+
model: "grok-test",
|
|
385
|
+
url: "http://x/v1/chat/completions",
|
|
386
|
+
fetchTimeoutMs: 5_000,
|
|
387
|
+
temperature: 0.2,
|
|
388
|
+
repeatPenalty: 1.15,
|
|
389
|
+
reasoning: { mode: "off", budget: null },
|
|
390
|
+
retryAttempts: 0,
|
|
391
|
+
streaming: false,
|
|
392
|
+
normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
|
|
393
|
+
});
|
|
394
|
+
installFetchJson({
|
|
395
|
+
id: "response-1",
|
|
396
|
+
model: "grok-test",
|
|
397
|
+
choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
|
|
398
|
+
usage: {
|
|
399
|
+
prompt_tokens: 2,
|
|
400
|
+
completion_tokens: 1,
|
|
401
|
+
total_tokens: 3,
|
|
402
|
+
cost_in_usd_ticks: 15_493_500,
|
|
403
|
+
},
|
|
404
|
+
});
|
|
405
|
+
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
406
|
+
assert.deepEqual(response.charge, {
|
|
407
|
+
kind: "authoritative",
|
|
408
|
+
amount: { amount: "15493500", currency: "USDTICK" },
|
|
409
|
+
usdEquivalent: "0.00154935",
|
|
410
|
+
source: "xAI response usage.cost_in_usd_ticks",
|
|
411
|
+
});
|
|
412
|
+
assert.equal(response.rawBody, undefined);
|
|
413
|
+
});
|
|
414
|
+
|
|
415
|
+
test("streamed xAI final usage retains its exact tick charge", async () => {
|
|
416
|
+
const p = new AiSdkProvider({
|
|
417
|
+
model: "grok-test",
|
|
418
|
+
url: "http://x/v1/chat/completions",
|
|
419
|
+
fetchTimeoutMs: 5_000,
|
|
420
|
+
temperature: 0.2,
|
|
421
|
+
repeatPenalty: 1.15,
|
|
422
|
+
reasoning: { mode: "off", budget: null },
|
|
423
|
+
retryAttempts: 0,
|
|
424
|
+
normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
|
|
425
|
+
});
|
|
426
|
+
installFetch([
|
|
427
|
+
{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
428
|
+
{
|
|
429
|
+
choices: [],
|
|
430
|
+
usage: {
|
|
431
|
+
prompt_tokens: 2,
|
|
432
|
+
completion_tokens: 1,
|
|
433
|
+
total_tokens: 3,
|
|
434
|
+
cost_in_usd_ticks: 15_493_500,
|
|
435
|
+
},
|
|
436
|
+
},
|
|
437
|
+
]);
|
|
438
|
+
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
439
|
+
assert.deepEqual(response.charge, {
|
|
440
|
+
kind: "authoritative",
|
|
441
|
+
amount: { amount: "15493500", currency: "USDTICK" },
|
|
442
|
+
usdEquivalent: "0.00154935",
|
|
443
|
+
source: "xAI response usage.cost_in_usd_ticks",
|
|
444
|
+
});
|
|
445
|
+
});
|
|
446
|
+
|
|
296
447
|
test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
|
|
297
448
|
const warnings: Array<{ message: string; code?: string }> = [];
|
|
298
449
|
mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
|
|
@@ -311,7 +462,88 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
|
|
|
311
462
|
}]);
|
|
312
463
|
});
|
|
313
464
|
|
|
314
|
-
test("
|
|
465
|
+
test("#161: a streamed resource interruption is a failed exchange with complete attempt evidence", async () => {
|
|
466
|
+
const calls = installFetch([
|
|
467
|
+
{ model: "served-model", choices: [{ delta: { reasoning_content: "partial thought", content: "partial answer" } }] },
|
|
468
|
+
{
|
|
469
|
+
choices: [{ delta: {}, finish_reason: "insufficient_system_resource" }],
|
|
470
|
+
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
471
|
+
},
|
|
472
|
+
]);
|
|
473
|
+
const provider = new AiSdkProvider({
|
|
474
|
+
...injectedBase,
|
|
475
|
+
retryAttempts: 2,
|
|
476
|
+
rawBody: true,
|
|
477
|
+
});
|
|
478
|
+
|
|
479
|
+
await assert.rejects(
|
|
480
|
+
provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
|
|
481
|
+
(error: unknown) => {
|
|
482
|
+
assert.ok(error instanceof ProviderError);
|
|
483
|
+
assert.equal(error.kind, "resource_interrupted");
|
|
484
|
+
assert.equal(error.status, 503);
|
|
485
|
+
assert.equal(error.problem.stage, "provider-response");
|
|
486
|
+
assert.equal(error.problem.retryable, false);
|
|
487
|
+
assert.equal(error.problem.finishReason, "resource_interrupted");
|
|
488
|
+
assert.equal(error.problem.rawFinishReason, "insufficient_system_resource");
|
|
489
|
+
assert.equal(error.attempt?.assistant.content, "partial answer");
|
|
490
|
+
assert.equal(error.attempt?.assistant.reasoning, "partial thought");
|
|
491
|
+
assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
|
|
492
|
+
assert.deepEqual(error.attempt?.assistant.usage, {
|
|
493
|
+
prompt: 7,
|
|
494
|
+
completion: 2,
|
|
495
|
+
reasoning: 3,
|
|
496
|
+
cached: 0,
|
|
497
|
+
total: 12,
|
|
498
|
+
});
|
|
499
|
+
assert.equal(
|
|
500
|
+
(error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
|
|
501
|
+
"insufficient_system_resource",
|
|
502
|
+
);
|
|
503
|
+
assert.ok(Array.isArray(error.attempt?.rawBody));
|
|
504
|
+
return true;
|
|
505
|
+
},
|
|
506
|
+
);
|
|
507
|
+
assert.equal(calls.length, 1, "a semantic interruption is not replayed as an HTTP failure");
|
|
508
|
+
});
|
|
509
|
+
|
|
510
|
+
test("#161: a buffered resource interruption preserves the successful wire response as failed-attempt evidence", async () => {
|
|
511
|
+
const wire = {
|
|
512
|
+
model: "served-model",
|
|
513
|
+
choices: [{
|
|
514
|
+
message: { content: "partial answer", reasoning_content: "partial thought" },
|
|
515
|
+
finish_reason: "insufficient_system_resource",
|
|
516
|
+
}],
|
|
517
|
+
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
518
|
+
};
|
|
519
|
+
const calls = installFetchJson(wire);
|
|
520
|
+
const provider = new AiSdkProvider({
|
|
521
|
+
...injectedBase,
|
|
522
|
+
streaming: false,
|
|
523
|
+
retryAttempts: 2,
|
|
524
|
+
rawBody: true,
|
|
525
|
+
});
|
|
526
|
+
|
|
527
|
+
await assert.rejects(
|
|
528
|
+
provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
|
|
529
|
+
(error: unknown) => {
|
|
530
|
+
assert.ok(error instanceof ProviderError);
|
|
531
|
+
assert.equal(error.kind, "resource_interrupted");
|
|
532
|
+
assert.equal(error.attempt?.assistant.content, "partial answer");
|
|
533
|
+
assert.equal(error.attempt?.assistant.reasoning, "partial thought");
|
|
534
|
+
assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
|
|
535
|
+
assert.deepEqual(error.attempt?.rawBody, wire);
|
|
536
|
+
assert.equal(
|
|
537
|
+
(error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
|
|
538
|
+
"insufficient_system_resource",
|
|
539
|
+
);
|
|
540
|
+
return true;
|
|
541
|
+
},
|
|
542
|
+
);
|
|
543
|
+
assert.equal(calls.length, 1);
|
|
544
|
+
});
|
|
545
|
+
|
|
546
|
+
test("generate translates a backend cap synonym to canonical length", async () => {
|
|
315
547
|
// gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
|
|
316
548
|
// "length" so its truncation check (=== "length") is a cross-backend invariant.
|
|
317
549
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
@@ -320,7 +552,7 @@ test("generate translates a backend cap synonym to canonical length (#425)", asy
|
|
|
320
552
|
assert.equal(assistant.finishReason, "length");
|
|
321
553
|
});
|
|
322
554
|
|
|
323
|
-
test("generate translates end_turn to canonical stop
|
|
555
|
+
test("generate translates end_turn to canonical stop", async () => {
|
|
324
556
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
325
557
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
|
|
326
558
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -339,10 +571,172 @@ test("generate aggregates reasoning deltas under multiple field names", async ()
|
|
|
339
571
|
installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
|
|
340
572
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
341
573
|
assert.equal(assistant.reasoning, "because");
|
|
342
|
-
assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
|
|
574
|
+
assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
|
|
575
|
+
});
|
|
576
|
+
|
|
577
|
+
test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
|
|
578
|
+
const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
|
|
579
|
+
const p = new AiSdkProvider(config);
|
|
580
|
+
installFetch([
|
|
581
|
+
{ choices: [{ delta: { content: "<thi" } }] },
|
|
582
|
+
{ choices: [{ delta: { content: "nk>12345</th" } }] },
|
|
583
|
+
{ choices: [{ delta: { content: "ink>abcde" }, finish_reason: "stop" }] },
|
|
584
|
+
{ usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 } },
|
|
585
|
+
]);
|
|
586
|
+
|
|
587
|
+
const response = await p.generate({ workerId: "tagged-stream", messages: [] });
|
|
588
|
+
|
|
589
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
590
|
+
assert.equal(response.assistant.content, "abcde");
|
|
591
|
+
assert.deepEqual(response.assistant.usage, {
|
|
592
|
+
prompt: 3,
|
|
593
|
+
completion: 5,
|
|
594
|
+
reasoning: 5,
|
|
595
|
+
cached: 0,
|
|
596
|
+
total: 13,
|
|
597
|
+
});
|
|
598
|
+
assert.deepEqual(
|
|
599
|
+
((response.assistantRaw as { content: string; reasoning: string }).content),
|
|
600
|
+
"abcde",
|
|
601
|
+
);
|
|
602
|
+
assert.equal((response.assistantRaw as { reasoning: string }).reasoning, "12345");
|
|
603
|
+
assert.match(JSON.stringify(response.rawBody), /<thi/);
|
|
604
|
+
assert.match(JSON.stringify(response.rawBody), /nk>12345/);
|
|
605
|
+
});
|
|
606
|
+
|
|
607
|
+
test("{§provider-tagged-reasoning} explicit think-tags projects one buffered leading envelope", async () => {
|
|
608
|
+
installFetchJson({
|
|
609
|
+
model: "m",
|
|
610
|
+
choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
|
|
611
|
+
usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
|
|
612
|
+
});
|
|
613
|
+
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
614
|
+
const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
|
|
615
|
+
|
|
616
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
617
|
+
assert.equal(response.assistant.content, "abcde");
|
|
618
|
+
assert.equal(response.assistant.usage.completion, 5);
|
|
619
|
+
assert.equal(response.assistant.usage.reasoning, 5);
|
|
620
|
+
});
|
|
621
|
+
|
|
622
|
+
test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
|
|
623
|
+
installFetchJson({
|
|
624
|
+
model: "m",
|
|
625
|
+
choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
|
|
626
|
+
usage: {
|
|
627
|
+
prompt_tokens: 3,
|
|
628
|
+
completion_tokens: 10,
|
|
629
|
+
total_tokens: 13,
|
|
630
|
+
completion_tokens_details: { reasoning_tokens: 3 },
|
|
631
|
+
},
|
|
632
|
+
});
|
|
633
|
+
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
634
|
+
const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
|
|
635
|
+
|
|
636
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
637
|
+
assert.equal(response.assistant.content, "abcde");
|
|
638
|
+
assert.equal(response.assistant.usage.completion, 7);
|
|
639
|
+
assert.equal(response.assistant.usage.reasoning, 3);
|
|
640
|
+
});
|
|
641
|
+
|
|
642
|
+
test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
|
|
643
|
+
const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const };
|
|
644
|
+
installFetch([
|
|
645
|
+
{ choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
|
|
646
|
+
{ usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
|
|
647
|
+
]);
|
|
648
|
+
const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
|
|
649
|
+
assert.equal(streamed.assistant.reasoning, "unfinished");
|
|
650
|
+
assert.equal(streamed.assistant.content, "");
|
|
651
|
+
assert.deepEqual(streamed.assistant.usage, {
|
|
652
|
+
prompt: 3,
|
|
653
|
+
completion: 0,
|
|
654
|
+
reasoning: 8,
|
|
655
|
+
cached: 0,
|
|
656
|
+
total: 11,
|
|
657
|
+
});
|
|
658
|
+
|
|
659
|
+
mock.restoreAll();
|
|
660
|
+
installFetchJson({
|
|
661
|
+
model: "m",
|
|
662
|
+
choices: [{ message: { content: "<think>unfinished" }, finish_reason: "length" }],
|
|
663
|
+
usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
|
|
664
|
+
});
|
|
665
|
+
const bufferedConfig = { ...config, streaming: false };
|
|
666
|
+
const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
|
|
667
|
+
assert.equal(buffered.assistant.reasoning, "unfinished");
|
|
668
|
+
assert.equal(buffered.assistant.content, "");
|
|
669
|
+
assert.equal(buffered.assistant.usage.completion, 0);
|
|
670
|
+
assert.equal(buffered.assistant.usage.reasoning, 8);
|
|
343
671
|
});
|
|
344
672
|
|
|
345
|
-
test("
|
|
673
|
+
test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
|
|
674
|
+
installFetchJson({
|
|
675
|
+
model: "m",
|
|
676
|
+
choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
|
|
677
|
+
usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
|
|
678
|
+
});
|
|
679
|
+
const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
|
|
680
|
+
.generate({ workerId: "verbatim", messages: [] });
|
|
681
|
+
assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
|
|
682
|
+
assert.equal(verbatim.assistant.reasoning, null);
|
|
683
|
+
assert.equal(verbatim.assistant.usage.completion, 4);
|
|
684
|
+
|
|
685
|
+
mock.restoreAll();
|
|
686
|
+
installFetchJson({
|
|
687
|
+
model: "m",
|
|
688
|
+
choices: [{ message: { content: "show <think>literal</think> exactly" }, finish_reason: "stop" }],
|
|
689
|
+
usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
|
|
690
|
+
});
|
|
691
|
+
const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
692
|
+
const nonLeading = await new AiSdkProvider(taggedConfig)
|
|
693
|
+
.generate({ workerId: "non-leading", messages: [] });
|
|
694
|
+
assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
|
|
695
|
+
assert.equal(nonLeading.assistant.reasoning, null);
|
|
696
|
+
|
|
697
|
+
mock.restoreAll();
|
|
698
|
+
installFetchJson({
|
|
699
|
+
model: "m",
|
|
700
|
+
choices: [{ message: {
|
|
701
|
+
content: "<think>literal visible bytes</think>",
|
|
702
|
+
reasoning_content: "structured reasoning",
|
|
703
|
+
}, finish_reason: "stop" }],
|
|
704
|
+
usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
|
|
705
|
+
});
|
|
706
|
+
const structured = await new AiSdkProvider(taggedConfig)
|
|
707
|
+
.generate({ workerId: "structured", messages: [] });
|
|
708
|
+
assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
|
|
709
|
+
assert.equal(structured.assistant.reasoning, "structured reasoning");
|
|
710
|
+
});
|
|
711
|
+
|
|
712
|
+
test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
|
|
713
|
+
const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
|
|
714
|
+
const config = {
|
|
715
|
+
...injectedBase,
|
|
716
|
+
contextWindow: 640,
|
|
717
|
+
reasoning: { mode: "adaptive" as const, budget: null },
|
|
718
|
+
reasoningResponseStyle: "think-tags" as const,
|
|
719
|
+
reasoningStyle: "think" as const,
|
|
720
|
+
grammarStyle: "llamacpp" as const,
|
|
721
|
+
};
|
|
722
|
+
installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
723
|
+
|
|
724
|
+
const response = await new AiSdkProvider(config).generate({
|
|
725
|
+
workerId: "tagged-grammar",
|
|
726
|
+
messages: [],
|
|
727
|
+
grammar: `root ::= ${JSON.stringify(content)}`,
|
|
728
|
+
});
|
|
729
|
+
|
|
730
|
+
assert.equal(response.assistant.reasoning, "🧠reason");
|
|
731
|
+
assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
|
|
732
|
+
assert.deepEqual(response.grammarEvidence, {
|
|
733
|
+
input: content,
|
|
734
|
+
contentStart: [..."<think>🧠reason</think>"].length,
|
|
735
|
+
transported: true,
|
|
736
|
+
});
|
|
737
|
+
});
|
|
738
|
+
|
|
739
|
+
test("encrypted reasoning (non-streamed): encrypted entries normalize and text entries stay separate", async () => {
|
|
346
740
|
// The live o4-mini-via-OpenRouter shape: reasoning null, one encrypted entry.
|
|
347
741
|
installFetchJson({ model: "m", choices: [{ message: {
|
|
348
742
|
content: "4", reasoning: null,
|
|
@@ -353,13 +747,14 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
|
|
|
353
747
|
}, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
354
748
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
355
749
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
356
|
-
//
|
|
750
|
+
// Wire detail ID is preserved; the assistant-message location supports the
|
|
751
|
+
// derived classification but supplies no downstream client entity ID.
|
|
357
752
|
assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
|
|
358
|
-
assert.equal(assistant.reasoning, null); //
|
|
753
|
+
assert.equal(assistant.reasoning, null); // Encrypted turn: nothing readable.
|
|
359
754
|
assert.equal(assistant.content, "4");
|
|
360
755
|
});
|
|
361
756
|
|
|
362
|
-
test("
|
|
757
|
+
test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
|
|
363
758
|
installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning: null, reasoning_details: [
|
|
364
759
|
{ type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
|
|
365
760
|
{ type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
|
|
@@ -370,7 +765,20 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
|
|
|
370
765
|
assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
|
|
371
766
|
});
|
|
372
767
|
|
|
373
|
-
test("
|
|
768
|
+
test("assistant-message location classifies encrypted reasoning without inventing a missing detail id", async () => {
|
|
769
|
+
installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
|
|
770
|
+
{ type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
|
|
771
|
+
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
772
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
773
|
+
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
774
|
+
assert.deepEqual(assistant.reasoningEncrypted, [{
|
|
775
|
+
id: null,
|
|
776
|
+
subtype: "message",
|
|
777
|
+
encrypted: [{ data: "OPAQUE", format: "openai-responses-v1" }],
|
|
778
|
+
}]);
|
|
779
|
+
});
|
|
780
|
+
|
|
781
|
+
test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
|
|
374
782
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
375
783
|
installFetch([
|
|
376
784
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
|
|
@@ -402,10 +810,10 @@ test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", as
|
|
|
402
810
|
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
403
811
|
});
|
|
404
812
|
|
|
405
|
-
test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS
|
|
813
|
+
test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
|
|
406
814
|
// expected === null → the field must be ABSENT from the wire body. Fireworks
|
|
407
815
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
408
|
-
//
|
|
816
|
+
// Adaptive = the backend's own default posture = omission.
|
|
409
817
|
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
410
818
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
411
819
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
@@ -417,7 +825,39 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
417
825
|
}
|
|
418
826
|
});
|
|
419
827
|
|
|
420
|
-
test("
|
|
828
|
+
test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
|
|
829
|
+
const cases = [
|
|
830
|
+
[{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
|
|
831
|
+
[{ mode: "adaptive", budget: null }, {}],
|
|
832
|
+
[{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
833
|
+
] as const;
|
|
834
|
+
for (const [reasoning, expected] of cases) {
|
|
835
|
+
const p = new AiSdkProvider({
|
|
836
|
+
model: "m",
|
|
837
|
+
url: "http://x/v1/chat/completions",
|
|
838
|
+
fetchTimeoutMs: 5000,
|
|
839
|
+
temperature: 0.2,
|
|
840
|
+
repeatPenalty: 1.15,
|
|
841
|
+
reasoning,
|
|
842
|
+
retryAttempts: 0,
|
|
843
|
+
reasoningStyle: "thinking_effort",
|
|
844
|
+
});
|
|
845
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
846
|
+
await p.generate({
|
|
847
|
+
workerId: "r",
|
|
848
|
+
messages: [],
|
|
849
|
+
sampling: { thinking: { type: "disabled" }, reasoning_effort: "max" },
|
|
850
|
+
});
|
|
851
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
852
|
+
assert.deepEqual(
|
|
853
|
+
Object.fromEntries(Object.entries(body).filter(([key]) => key === "thinking" || key === "reasoning_effort")),
|
|
854
|
+
expected,
|
|
855
|
+
);
|
|
856
|
+
mock.restoreAll();
|
|
857
|
+
}
|
|
858
|
+
});
|
|
859
|
+
|
|
860
|
+
test("the family temperature default rides every request; caller sampling overrides it", async () => {
|
|
421
861
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
422
862
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
423
863
|
await p.generate({ workerId: "r", messages: [] });
|
|
@@ -434,7 +874,7 @@ test("the family temperature default rides every request; caller sampling overri
|
|
|
434
874
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
435
875
|
});
|
|
436
876
|
|
|
437
|
-
test("
|
|
877
|
+
test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
|
|
438
878
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
|
|
439
879
|
// set + llamacpp -> the loop-breakers ride the wire
|
|
440
880
|
const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
|
|
@@ -474,14 +914,14 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
474
914
|
assert.equal(body.repeat_penalty, 1.15);
|
|
475
915
|
});
|
|
476
916
|
|
|
477
|
-
test("
|
|
917
|
+
test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
|
|
478
918
|
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
479
919
|
const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
480
920
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
481
921
|
await llama.generate({ workerId: "r", messages: [] });
|
|
482
922
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
483
923
|
mock.restoreAll();
|
|
484
|
-
//
|
|
924
|
+
// A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
|
|
485
925
|
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
486
926
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
487
927
|
await cloud.generate({ workerId: "r", messages: [] });
|
|
@@ -520,7 +960,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
|
|
|
520
960
|
assert.equal("id_slot" in body, false); // reserved slot key stripped
|
|
521
961
|
});
|
|
522
962
|
|
|
523
|
-
test("
|
|
963
|
+
test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
|
|
524
964
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
525
965
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
526
966
|
await p.generate({
|
|
@@ -531,7 +971,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
531
971
|
n: 3, // breaks choices[0] atomicity -> stripped
|
|
532
972
|
tools: [{ type: "function" }], tool_choice: "auto", // tools-in-body doctrine -> stripped
|
|
533
973
|
modalities: ["text", "audio"], prediction: { type: "content" }, // text-only / decode semantics -> stripped
|
|
534
|
-
max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass
|
|
974
|
+
max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass -> stripped
|
|
535
975
|
seed: 42, user: "acct-7", service_tier: "flex", // platform/sampling intent -> pass
|
|
536
976
|
},
|
|
537
977
|
});
|
|
@@ -545,58 +985,134 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
545
985
|
assert.equal(body.service_tier, "flex");
|
|
546
986
|
});
|
|
547
987
|
|
|
548
|
-
test("
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
const
|
|
988
|
+
test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
|
|
989
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
990
|
+
const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
|
|
991
|
+
const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
|
|
992
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
|
|
993
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
994
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
995
|
+
assert.equal(body.reasoning_format, "none");
|
|
996
|
+
assert.equal(body.thinking_budget_tokens, 64);
|
|
997
|
+
assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
|
|
998
|
+
assert.equal(res.assistant.reasoning, "con🙂sider");
|
|
999
|
+
assert.equal(res.assistant.content, "x");
|
|
1000
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
1001
|
+
input: grammarInput,
|
|
1002
|
+
contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
|
|
1003
|
+
transported: true,
|
|
1004
|
+
});
|
|
1005
|
+
assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
|
|
1006
|
+
});
|
|
1007
|
+
|
|
1008
|
+
test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
|
|
1009
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1010
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1011
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1012
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1013
|
+
assert.equal(body.reasoning_format, "none");
|
|
1014
|
+
assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
|
|
1015
|
+
});
|
|
1016
|
+
|
|
1017
|
+
test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
|
|
1018
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
552
1019
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
553
1020
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
554
1021
|
const body = JSON.parse(calls[0].init.body as string);
|
|
555
|
-
assert.deepEqual(body.chat_template_kwargs, { enable_thinking:
|
|
556
|
-
assert.equal(
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
1022
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
|
|
1023
|
+
assert.equal(body.reasoning_format, "none");
|
|
1024
|
+
assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
|
|
1025
|
+
});
|
|
1026
|
+
|
|
1027
|
+
test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
|
|
1028
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1029
|
+
installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
|
|
1030
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1031
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
1032
|
+
});
|
|
1033
|
+
|
|
1034
|
+
test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
|
|
1035
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1036
|
+
const input = "<|channel>thought\n<channel|>x";
|
|
1037
|
+
const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
|
|
1038
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
|
|
1039
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1040
|
+
assert.equal(body.reasoning_format, "none");
|
|
1041
|
+
assert.equal(res.assistant.reasoning, null);
|
|
1042
|
+
assert.equal(res.assistant.content, "x");
|
|
1043
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
1044
|
+
input,
|
|
1045
|
+
contentStart: [..."<|channel>thought\n<channel|>"].length,
|
|
1046
|
+
transported: true,
|
|
1047
|
+
});
|
|
560
1048
|
});
|
|
561
1049
|
|
|
562
|
-
test("
|
|
1050
|
+
test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
|
|
563
1051
|
// The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
|
|
564
1052
|
// escaped into a discarded reasoning block, unconstrained.
|
|
565
|
-
const
|
|
566
|
-
installFetch([
|
|
1053
|
+
const chunks = [
|
|
567
1054
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
568
1055
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
569
|
-
]
|
|
1056
|
+
];
|
|
1057
|
+
const fetch: typeof globalThis.fetch = async (input, init) => {
|
|
1058
|
+
if (String(input).endsWith("/tokenize")) {
|
|
1059
|
+
const body = JSON.parse(String(init?.body)) as { content: string };
|
|
1060
|
+
return new Response(JSON.stringify({
|
|
1061
|
+
tokens: body.content.length === 0 ? [] : [1],
|
|
1062
|
+
}), { headers: { "content-type": "application/json" } });
|
|
1063
|
+
}
|
|
1064
|
+
return new Response(sseStream(chunks), { status: 200 });
|
|
1065
|
+
};
|
|
1066
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
570
1067
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
assert.ok(escape, "escape telemetry attached");
|
|
1068
|
+
const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
|
|
1069
|
+
assert.ok(escape, "escape notice attached");
|
|
574
1070
|
assert.equal(escape!.kind, "grammar_unenforced");
|
|
575
1071
|
assert.match(escape!.message ?? "", /5000 completion tokens billed/);
|
|
576
1072
|
});
|
|
577
1073
|
|
|
578
|
-
test("
|
|
1074
|
+
test("channel-escape state is absent without a transported grammar", async () => {
|
|
579
1075
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
580
1076
|
installFetch([
|
|
581
1077
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
582
1078
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
583
1079
|
]);
|
|
584
1080
|
const res = await p.generate({ workerId: "r", messages: [] }); // no grammar arg
|
|
585
|
-
assert.equal(res.
|
|
586
|
-
assert.equal(res.
|
|
1081
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
1082
|
+
assert.equal(res.notices, undefined);
|
|
587
1083
|
});
|
|
588
1084
|
|
|
589
|
-
test("reasoningStyle 'template'
|
|
590
|
-
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
1085
|
+
test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
|
|
1086
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
591
1087
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
592
1088
|
await on.generate({ workerId: "r", messages: [] });
|
|
593
|
-
|
|
1089
|
+
let body = JSON.parse(calls[0].init.body as string);
|
|
1090
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
1091
|
+
assert.equal(body.reasoning_format, "auto");
|
|
1092
|
+
assert.equal(body.thinking_budget_tokens, 64);
|
|
594
1093
|
|
|
595
1094
|
mock.restoreAll();
|
|
596
|
-
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
1095
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
597
1096
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
598
1097
|
await off.generate({ workerId: "r", messages: [] });
|
|
599
|
-
|
|
1098
|
+
body = JSON.parse(calls[0].init.body as string);
|
|
1099
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
|
|
1100
|
+
assert.equal(body.reasoning_format, "auto");
|
|
1101
|
+
assert.equal(body.thinking_budget_tokens, 0);
|
|
1102
|
+
});
|
|
1103
|
+
|
|
1104
|
+
test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
|
|
1105
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
|
|
1106
|
+
const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
|
|
1107
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1108
|
+
await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
|
|
1109
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1110
|
+
assert.equal(body.thinking_budget_tokens, 32);
|
|
1111
|
+
assert.equal(body.reasoning_format, "auto");
|
|
1112
|
+
assert.throws(
|
|
1113
|
+
() => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
|
|
1114
|
+
/REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
|
|
1115
|
+
);
|
|
600
1116
|
});
|
|
601
1117
|
|
|
602
1118
|
test("budget 0 suppresses effort and include_reasoning", async () => {
|
|
@@ -619,7 +1135,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
|
|
|
619
1135
|
assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
|
|
620
1136
|
});
|
|
621
1137
|
|
|
622
|
-
// — grammar-constrained sampling
|
|
1138
|
+
// — grammar-constrained sampling —
|
|
623
1139
|
|
|
624
1140
|
test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
|
|
625
1141
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
@@ -640,106 +1156,81 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
|
|
|
640
1156
|
assert.equal("response_format" in body, false);
|
|
641
1157
|
});
|
|
642
1158
|
|
|
643
|
-
// — grammar
|
|
644
|
-
// returns; bytes flow; a non-accept verdict rides response.telemetry —
|
|
1159
|
+
// — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
|
|
645
1160
|
|
|
646
1161
|
const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
|
|
647
1162
|
const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
648
1163
|
|
|
649
|
-
test("
|
|
1164
|
+
test("an unsplit grammar response carries the exact observed sentence", async () => {
|
|
650
1165
|
const p = grammarProvider();
|
|
651
1166
|
streamingContent("ok");
|
|
652
|
-
const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
653
|
-
assert.equal(assistant.content, "ok");
|
|
654
|
-
});
|
|
655
|
-
|
|
656
|
-
test("observation: REJECTED output still returns — bytes present, verdict attached with position", async () => {
|
|
657
|
-
const p = grammarProvider();
|
|
658
|
-
streamingContent("no");
|
|
659
1167
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
660
|
-
assert.equal(res.assistant.content, "no"); // bytes ALWAYS flow
|
|
661
|
-
assert.equal(res.telemetry?.length, 1);
|
|
662
|
-
const ev = res.telemetry![0];
|
|
663
|
-
assert.equal(ev.kind, "grammar_unenforced");
|
|
664
|
-
assert.equal(ev.source, "provider:test");
|
|
665
|
-
assert.match(String(ev.message), /grammar not enforced: output rejected .* at code point 0/);
|
|
666
|
-
assert.equal(ev.position, 0); // divergence offset for consumer policy
|
|
667
|
-
});
|
|
668
|
-
|
|
669
|
-
test("observation: an incomplete (valid prefix, never terminated) also returns with the verdict", async () => {
|
|
670
|
-
const p = grammarProvider();
|
|
671
|
-
streamingContent("ok");
|
|
672
|
-
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok" "!"' });
|
|
673
1168
|
assert.equal(res.assistant.content, "ok");
|
|
674
|
-
assert.
|
|
675
|
-
|
|
676
|
-
|
|
1169
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
1170
|
+
input: "ok",
|
|
1171
|
+
contentStart: 0,
|
|
1172
|
+
transported: true,
|
|
1173
|
+
});
|
|
677
1174
|
});
|
|
678
1175
|
|
|
679
|
-
test("
|
|
1176
|
+
test("the provider returns rejected or incomplete bytes as evidence without grading them", async () => {
|
|
680
1177
|
const p = grammarProvider();
|
|
681
|
-
streamingContent("
|
|
1178
|
+
streamingContent("no");
|
|
682
1179
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
683
|
-
assert.equal(res.
|
|
1180
|
+
assert.equal(res.assistant.content, "no");
|
|
1181
|
+
assert.deepEqual(res.grammarEvidence, { input: "no", contentStart: 0, transported: true });
|
|
1182
|
+
assert.equal(res.notices, undefined);
|
|
1183
|
+
assert.equal(res.meta?.railsVerdict, undefined);
|
|
684
1184
|
});
|
|
685
1185
|
|
|
686
|
-
test("
|
|
1186
|
+
test("empty unsplit content remains exact grammar evidence", async () => {
|
|
687
1187
|
const p = grammarProvider();
|
|
688
|
-
installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
|
|
1188
|
+
installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
|
|
689
1189
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
690
1190
|
assert.equal(res.assistant.content, "");
|
|
691
|
-
assert.
|
|
1191
|
+
assert.deepEqual(res.grammarEvidence, { input: "", contentStart: 0, transported: true });
|
|
692
1192
|
});
|
|
693
1193
|
|
|
694
|
-
test("
|
|
1194
|
+
test("grammarStyle 'none' produces no grammar observation", async () => {
|
|
695
1195
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
|
|
696
1196
|
streamingContent("anything goes");
|
|
697
|
-
const
|
|
698
|
-
assert.equal(assistant.content, "anything goes");
|
|
1197
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
1198
|
+
assert.equal(res.assistant.content, "anything goes");
|
|
1199
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
699
1200
|
});
|
|
700
1201
|
|
|
701
|
-
test("
|
|
1202
|
+
test("provider evidence does not depend on the local validator understanding the grammar", async () => {
|
|
702
1203
|
const p = grammarProvider();
|
|
703
1204
|
streamingContent("whatever");
|
|
704
|
-
const
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
await flush();
|
|
709
|
-
process.off("warning", onWarn);
|
|
710
|
-
assert.equal(assistant.content, "whatever"); // transport not failed
|
|
711
|
-
assert.ok(warnings.some((w) => (w as Error & { code?: string }).code === "PLURNK_GRAMMAR_UNVERIFIABLE"), "emitted the verify-gap warning");
|
|
1205
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' });
|
|
1206
|
+
assert.equal(res.assistant.content, "whatever");
|
|
1207
|
+
assert.deepEqual(res.grammarEvidence, { input: "whatever", contentStart: 0, transported: true });
|
|
1208
|
+
assert.equal(res.notices, undefined);
|
|
712
1209
|
});
|
|
713
1210
|
|
|
714
|
-
// — PLURNK_PROVIDERS_GBNF_DEBUG:
|
|
1211
|
+
// — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
|
|
715
1212
|
|
|
716
|
-
test("gbnfDebug
|
|
1213
|
+
test("gbnfDebug marks an unconstrained observation as not transported", async () => {
|
|
717
1214
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
718
1215
|
const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
|
|
719
1216
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
720
1217
|
const body = JSON.parse(calls[0].init.body as string);
|
|
721
|
-
assert.equal("grammar" in body, false);
|
|
722
|
-
assert.equal(body.repeat_penalty, 1.15);
|
|
723
|
-
assert.equal(res.assistant.content, "ok");
|
|
724
|
-
assert.
|
|
1218
|
+
assert.equal("grammar" in body, false);
|
|
1219
|
+
assert.equal(body.repeat_penalty, 1.15);
|
|
1220
|
+
assert.equal(res.assistant.content, "ok");
|
|
1221
|
+
assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: false });
|
|
1222
|
+
assert.equal(res.notices, undefined);
|
|
725
1223
|
});
|
|
726
1224
|
|
|
727
|
-
test("gbnfDebug
|
|
1225
|
+
test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
|
|
728
1226
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
729
|
-
const calls = installFetch([{ choices: [{ delta: {
|
|
1227
|
+
const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
|
|
730
1228
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
731
|
-
// The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
|
|
732
1229
|
assert.equal(res.assistant.content, "xon-conforming output");
|
|
733
|
-
assert.
|
|
734
|
-
|
|
735
|
-
assert.equal(res.telemetry?.length, 1);
|
|
736
|
-
const [event] = res.telemetry ?? [];
|
|
737
|
-
assert.equal(event.source, "provider:test");
|
|
738
|
-
assert.equal(event.kind, "grammar_unenforced");
|
|
739
|
-
assert.equal(event.position, 0); // 'x' rejected at code point 0
|
|
740
|
-
assert.match(event.message ?? "", /output rejected by the transported grammar at code point 0 \("x"\)/);
|
|
1230
|
+
assert.deepEqual(res.grammarEvidence, { input: "xon-conforming output", contentStart: 0, transported: false });
|
|
1231
|
+
assert.equal(res.notices, undefined);
|
|
741
1232
|
const body = JSON.parse(calls[0].init.body as string);
|
|
742
|
-
assert.equal("grammar" in body, false);
|
|
1233
|
+
assert.equal("grammar" in body, false);
|
|
743
1234
|
});
|
|
744
1235
|
|
|
745
1236
|
test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
|
|
@@ -752,7 +1243,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
752
1243
|
assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
|
|
753
1244
|
});
|
|
754
1245
|
|
|
755
|
-
// — meta bag: verbatim provider metadata
|
|
1246
|
+
// — meta bag: verbatim provider metadata —
|
|
756
1247
|
|
|
757
1248
|
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
758
1249
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
@@ -763,7 +1254,7 @@ test("meta: passes backend fields through without reinterpreting monetary values
|
|
|
763
1254
|
assert.equal(res.meta?.system_fingerprint, "fp_abc");
|
|
764
1255
|
});
|
|
765
1256
|
|
|
766
|
-
// — first-party telemetry headers (
|
|
1257
|
+
// — first-party telemetry headers ({§provider-request-authority}) —
|
|
767
1258
|
|
|
768
1259
|
const headerVal = (init: RequestInit, name: string): string | undefined =>
|
|
769
1260
|
new Headers(init.headers).get(name) ?? undefined;
|
|
@@ -776,7 +1267,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
|
|
|
776
1267
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
|
|
777
1268
|
});
|
|
778
1269
|
|
|
779
|
-
test("
|
|
1270
|
+
test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
|
|
780
1271
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
781
1272
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
782
1273
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
@@ -795,7 +1286,7 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
|
|
|
795
1286
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined);
|
|
796
1287
|
});
|
|
797
1288
|
|
|
798
|
-
test("
|
|
1289
|
+
test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
|
|
799
1290
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
800
1291
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
801
1292
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
@@ -818,13 +1309,13 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
|
818
1309
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
|
|
819
1310
|
});
|
|
820
1311
|
|
|
821
|
-
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides
|
|
1312
|
+
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
|
|
822
1313
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
823
1314
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
824
1315
|
await p.generate({ workerId: "r", messages: [] });
|
|
825
1316
|
const body = JSON.parse(calls[0].init.body as string);
|
|
826
1317
|
assert.equal("grammar" in body, false);
|
|
827
|
-
assert.equal(body.repeat_penalty, 1.15); //
|
|
1318
|
+
assert.equal(body.repeat_penalty, 1.15); // penalty is not grammar-gated
|
|
828
1319
|
});
|
|
829
1320
|
|
|
830
1321
|
test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
|
|
@@ -839,7 +1330,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
|
|
|
839
1330
|
assert.equal("max_tokens" in JSON.parse(calls[0].init.body as string), false);
|
|
840
1331
|
});
|
|
841
1332
|
|
|
842
|
-
test("slot affinity is internal: sticky per workerId, distinct
|
|
1333
|
+
test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
|
|
843
1334
|
const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
844
1335
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
845
1336
|
await pinning.generate({ workerId: "run-A", messages: [] });
|
|
@@ -863,7 +1354,7 @@ test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever
|
|
|
863
1354
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
864
1355
|
});
|
|
865
1356
|
|
|
866
|
-
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent
|
|
1357
|
+
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
|
|
867
1358
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
868
1359
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
869
1360
|
const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
|
|
@@ -877,11 +1368,11 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
|
|
|
877
1368
|
});
|
|
878
1369
|
|
|
879
1370
|
test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
|
|
880
|
-
const { ProviderError } = await import("./
|
|
1371
|
+
const { ProviderError } = await import("./errors.ts");
|
|
881
1372
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
|
|
882
1373
|
mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
|
|
883
1374
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
884
|
-
assert.ok(err instanceof ProviderError);
|
|
1375
|
+
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
885
1376
|
assert.equal(err.kind, "network_failure"); // ≥500 → network_failure
|
|
886
1377
|
assert.equal(err.status, 500);
|
|
887
1378
|
return true;
|
|
@@ -904,15 +1395,17 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
|
|
|
904
1395
|
assert.equal(res.assistant.content, "out"); // content returned verbatim
|
|
905
1396
|
});
|
|
906
1397
|
|
|
907
|
-
test("generate wraps an HTTP failure as a ProviderError carrying
|
|
908
|
-
const { ProviderError } = await import("./
|
|
1398
|
+
test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
|
|
1399
|
+
const { ProviderError } = await import("./errors.ts");
|
|
909
1400
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
|
|
910
1401
|
mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
|
|
911
1402
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
912
|
-
assert.ok(err instanceof ProviderError);
|
|
1403
|
+
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
913
1404
|
assert.equal(err.kind, "rate_limit");
|
|
914
1405
|
assert.equal(err.status, 429);
|
|
915
|
-
assert.
|
|
1406
|
+
assert.equal(err.problem.status, 429);
|
|
1407
|
+
assert.equal(err.problem.detail, err.message);
|
|
1408
|
+
assert.equal(err.problem.type, "https://problems.plurnk.dev/provider/test/rate-limit");
|
|
916
1409
|
return true;
|
|
917
1410
|
});
|
|
918
1411
|
});
|
|
@@ -937,10 +1430,19 @@ test("configured headers and url are sent verbatim", async () => {
|
|
|
937
1430
|
assert.equal(headers.get("x-title"), "plurnk");
|
|
938
1431
|
});
|
|
939
1432
|
|
|
940
|
-
// — transient-failure retry
|
|
1433
|
+
// — transient-failure retry —
|
|
941
1434
|
|
|
942
1435
|
const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
|
|
943
1436
|
|
|
1437
|
+
const stalledStreamResponse = (): Response => new Response(new ReadableStream({
|
|
1438
|
+
start(controller) {
|
|
1439
|
+
controller.enqueue(new TextEncoder().encode(
|
|
1440
|
+
'data: {"id":"stalled","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
1441
|
+
));
|
|
1442
|
+
setTimeout(() => controller.close(), 100);
|
|
1443
|
+
},
|
|
1444
|
+
}), { status: 200 });
|
|
1445
|
+
|
|
944
1446
|
test("retry: a transient failure retries and a later success resolves", async () => {
|
|
945
1447
|
const calls = installFetchScript([
|
|
946
1448
|
{ status: 429, retryAfter: 0 },
|
|
@@ -953,20 +1455,11 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
953
1455
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
954
1456
|
});
|
|
955
1457
|
|
|
956
|
-
test("
|
|
1458
|
+
test("streamed-body silence retries and returns the retry's complete output", async () => {
|
|
957
1459
|
let calls = 0;
|
|
958
1460
|
mock.method(globalThis, "fetch", async () => {
|
|
959
1461
|
calls++;
|
|
960
|
-
if (calls === 1)
|
|
961
|
-
return new Response(new ReadableStream({
|
|
962
|
-
start(controller) {
|
|
963
|
-
controller.enqueue(new TextEncoder().encode(
|
|
964
|
-
'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
965
|
-
));
|
|
966
|
-
setTimeout(() => controller.close(), 100);
|
|
967
|
-
},
|
|
968
|
-
}), { status: 200 });
|
|
969
|
-
}
|
|
1462
|
+
if (calls === 1) return stalledStreamResponse();
|
|
970
1463
|
return new Response(new ReadableStream({
|
|
971
1464
|
start(controller) {
|
|
972
1465
|
controller.enqueue(new TextEncoder().encode(
|
|
@@ -979,7 +1472,7 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
|
|
|
979
1472
|
const p = new AiSdkProvider({
|
|
980
1473
|
model: "m",
|
|
981
1474
|
url: "http://x/v1/chat/completions",
|
|
982
|
-
fetchTimeoutMs:
|
|
1475
|
+
fetchTimeoutMs: 5000,
|
|
983
1476
|
streamIdleTimeoutMs: 10,
|
|
984
1477
|
temperature: 0.2,
|
|
985
1478
|
repeatPenalty: 1.15,
|
|
@@ -987,16 +1480,104 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
|
|
|
987
1480
|
retryAttempts: 1,
|
|
988
1481
|
source: "provider:test",
|
|
989
1482
|
});
|
|
1483
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
1484
|
+
assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the stalled partial");
|
|
1485
|
+
assert.equal(calls, 2, "the stall retried once and the retry succeeded");
|
|
1486
|
+
mock.restoreAll();
|
|
1487
|
+
});
|
|
1488
|
+
|
|
1489
|
+
test("streamed-body silence does not replay when retries are disabled", async () => {
|
|
1490
|
+
let calls = 0;
|
|
1491
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1492
|
+
calls++;
|
|
1493
|
+
return stalledStreamResponse();
|
|
1494
|
+
});
|
|
1495
|
+
const p = new AiSdkProvider({
|
|
1496
|
+
model: "m",
|
|
1497
|
+
url: "http://x/v1/chat/completions",
|
|
1498
|
+
fetchTimeoutMs: 1000,
|
|
1499
|
+
streamIdleTimeoutMs: 10,
|
|
1500
|
+
temperature: 0.2,
|
|
1501
|
+
repeatPenalty: 1.15,
|
|
1502
|
+
reasoning: { mode: "off", budget: null },
|
|
1503
|
+
retryAttempts: 0,
|
|
1504
|
+
source: "provider:test",
|
|
1505
|
+
});
|
|
990
1506
|
await assert.rejects(
|
|
991
1507
|
p.generate({ workerId: "r", messages: [] }),
|
|
992
1508
|
(error: ProviderError) => error.kind === "network_failure"
|
|
993
1509
|
&& /chunk timeout/i.test(error.message),
|
|
994
1510
|
);
|
|
995
|
-
assert.equal(calls, 1);
|
|
1511
|
+
assert.equal(calls, 1, "zero retries permits exactly one provider request");
|
|
1512
|
+
mock.restoreAll();
|
|
1513
|
+
});
|
|
1514
|
+
|
|
1515
|
+
test("streamed-body silence exhausts the configured retry budget once", async () => {
|
|
1516
|
+
let calls = 0;
|
|
1517
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1518
|
+
calls++;
|
|
1519
|
+
return stalledStreamResponse();
|
|
1520
|
+
});
|
|
1521
|
+
const p = new AiSdkProvider({
|
|
1522
|
+
model: "m",
|
|
1523
|
+
url: "http://x/v1/chat/completions",
|
|
1524
|
+
fetchTimeoutMs: 5000,
|
|
1525
|
+
streamIdleTimeoutMs: 10,
|
|
1526
|
+
temperature: 0.2,
|
|
1527
|
+
repeatPenalty: 1.15,
|
|
1528
|
+
reasoning: { mode: "off", budget: null },
|
|
1529
|
+
retryAttempts: 1,
|
|
1530
|
+
source: "provider:test",
|
|
1531
|
+
});
|
|
1532
|
+
await assert.rejects(
|
|
1533
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
1534
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
1535
|
+
&& error.problem.attempts === 2
|
|
1536
|
+
&& error.problem.retryExhausted === true
|
|
1537
|
+
&& error.problem.retryable === false,
|
|
1538
|
+
);
|
|
1539
|
+
assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
|
|
1540
|
+
mock.restoreAll();
|
|
1541
|
+
});
|
|
1542
|
+
|
|
1543
|
+
test("the total generation deadline spans stalled-stream retry scheduling", async () => {
|
|
1544
|
+
let calls = 0;
|
|
1545
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1546
|
+
calls++;
|
|
1547
|
+
if (calls > 1) {
|
|
1548
|
+
return new Response(new ReadableStream({
|
|
1549
|
+
start(controller) {
|
|
1550
|
+
controller.enqueue(new TextEncoder().encode(
|
|
1551
|
+
'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"late"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
|
|
1552
|
+
));
|
|
1553
|
+
controller.close();
|
|
1554
|
+
},
|
|
1555
|
+
}), { status: 200 });
|
|
1556
|
+
}
|
|
1557
|
+
return stalledStreamResponse();
|
|
1558
|
+
});
|
|
1559
|
+
const p = new AiSdkProvider({
|
|
1560
|
+
model: "m",
|
|
1561
|
+
url: "http://x/v1/chat/completions",
|
|
1562
|
+
fetchTimeoutMs: 50,
|
|
1563
|
+
streamIdleTimeoutMs: 10,
|
|
1564
|
+
temperature: 0.2,
|
|
1565
|
+
repeatPenalty: 1.15,
|
|
1566
|
+
reasoning: { mode: "off", budget: null },
|
|
1567
|
+
retryAttempts: 3,
|
|
1568
|
+
source: "provider:test",
|
|
1569
|
+
});
|
|
1570
|
+
const started = Date.now();
|
|
1571
|
+
await assert.rejects(
|
|
1572
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
1573
|
+
(error: ProviderError) => error.kind === "network_failure",
|
|
1574
|
+
);
|
|
1575
|
+
assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
|
|
1576
|
+
assert.equal(calls, 1, "the total deadline expires before another request begins");
|
|
996
1577
|
mock.restoreAll();
|
|
997
1578
|
});
|
|
998
1579
|
|
|
999
|
-
test("
|
|
1580
|
+
test("a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
|
|
1000
1581
|
mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
|
|
1001
1582
|
async start(controller) {
|
|
1002
1583
|
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
|
|
@@ -1021,7 +1602,7 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
|
|
|
1021
1602
|
});
|
|
1022
1603
|
|
|
1023
1604
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
1024
|
-
const { ProviderError } = await import("./
|
|
1605
|
+
const { ProviderError } = await import("./errors.ts");
|
|
1025
1606
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
|
1026
1607
|
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
|
|
1027
1608
|
await assert.rejects(
|
|
@@ -1056,7 +1637,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
|
|
|
1056
1637
|
assert.equal(calls.length, 1); // no retry budget
|
|
1057
1638
|
});
|
|
1058
1639
|
|
|
1059
|
-
test("retry: a caller abort during backoff rejects promptly with no further attempt
|
|
1640
|
+
test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
|
|
1060
1641
|
const ac = new AbortController();
|
|
1061
1642
|
const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
|
|
1062
1643
|
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
@@ -1068,7 +1649,7 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
1068
1649
|
assert.equal(calls.length, 1); // never retried after cancellation
|
|
1069
1650
|
});
|
|
1070
1651
|
|
|
1071
|
-
// —
|
|
1652
|
+
// — Anthropic reasoning style (wire `thinking` parameter) —
|
|
1072
1653
|
|
|
1073
1654
|
test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
|
|
1074
1655
|
// N>0 → enabled with budget_tokens
|
|
@@ -1115,10 +1696,10 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1115
1696
|
mock.restoreAll();
|
|
1116
1697
|
});
|
|
1117
1698
|
|
|
1118
|
-
// ── Data capture (
|
|
1699
|
+
// ── Data capture ({§provider-evidence}): logprobs + verbatim rawBody, opt-in, off by default ──
|
|
1119
1700
|
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1120
1701
|
|
|
1121
|
-
test("
|
|
1702
|
+
test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
|
|
1122
1703
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1123
1704
|
const p = new AiSdkProvider({ ...captureBase });
|
|
1124
1705
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
@@ -1131,7 +1712,7 @@ test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no ra
|
|
|
1131
1712
|
mock.restoreAll();
|
|
1132
1713
|
});
|
|
1133
1714
|
|
|
1134
|
-
test("
|
|
1715
|
+
test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
|
|
1135
1716
|
const chunk = { model: "m", usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 }, choices: [{ delta: { content: "yesno" }, finish_reason: "stop", logprobs: { content: [
|
|
1136
1717
|
{ token: "yes", logprob: -0.5, sampling_logprob: -0.5, top_logprobs: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] },
|
|
1137
1718
|
{ token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
|
|
@@ -1148,7 +1729,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1148
1729
|
mock.restoreAll();
|
|
1149
1730
|
});
|
|
1150
1731
|
|
|
1151
|
-
test("
|
|
1732
|
+
test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
|
|
1152
1733
|
const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
1153
1734
|
installFetchJson(wire);
|
|
1154
1735
|
const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
|
|
@@ -1160,7 +1741,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
|
|
|
1160
1741
|
mock.restoreAll();
|
|
1161
1742
|
});
|
|
1162
1743
|
|
|
1163
|
-
test("
|
|
1744
|
+
test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
|
|
1164
1745
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1165
1746
|
const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
|
|
1166
1747
|
await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
|
|
@@ -1170,9 +1751,9 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
|
|
|
1170
1751
|
mock.restoreAll();
|
|
1171
1752
|
});
|
|
1172
1753
|
|
|
1173
|
-
// — turn coordinate headers (
|
|
1754
|
+
// — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
|
|
1174
1755
|
|
|
1175
|
-
test("
|
|
1756
|
+
test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
|
|
1176
1757
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1177
1758
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1178
1759
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
@@ -1182,7 +1763,7 @@ test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under th
|
|
|
1182
1763
|
assert.equal(headers.get("plurnk-turn"), "41");
|
|
1183
1764
|
});
|
|
1184
1765
|
|
|
1185
|
-
test("
|
|
1766
|
+
test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
|
|
1186
1767
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1187
1768
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1188
1769
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
@@ -1192,7 +1773,7 @@ test("#404: third-party providers structurally DROP the coordinate (gate off by
|
|
|
1192
1773
|
assert.equal(headers.has("plurnk-turn"), false);
|
|
1193
1774
|
});
|
|
1194
1775
|
|
|
1195
|
-
test("
|
|
1776
|
+
test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
|
|
1196
1777
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1197
1778
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1198
1779
|
await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
|
|
@@ -1203,9 +1784,9 @@ test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strike
|
|
|
1203
1784
|
assert.equal(headers.has("plurnk-strikes"), false);
|
|
1204
1785
|
});
|
|
1205
1786
|
|
|
1206
|
-
// --
|
|
1787
|
+
// -- {§provider-generation-envelope} --
|
|
1207
1788
|
|
|
1208
|
-
test("
|
|
1789
|
+
test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
|
|
1209
1790
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1210
1791
|
const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
|
|
1211
1792
|
assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
|
|
@@ -1217,32 +1798,32 @@ test("#507 reserves derive from the detected window; absolutes stand alone; null
|
|
|
1217
1798
|
assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
|
|
1218
1799
|
});
|
|
1219
1800
|
|
|
1220
|
-
test("
|
|
1801
|
+
test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
|
|
1221
1802
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
|
|
1222
1803
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1223
1804
|
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
|
|
1224
1805
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1225
1806
|
assert.equal(body.temperature, 0.9); // caller intent passes verbatim
|
|
1226
|
-
assert.equal("frequency_penalty" in body, false); // the floor is suppressed
|
|
1807
|
+
assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
|
|
1227
1808
|
});
|
|
1228
1809
|
|
|
1229
|
-
// --
|
|
1810
|
+
// -- prompt-cache affinity (workerId -> prompt_cache_key) --
|
|
1230
1811
|
|
|
1231
|
-
test("
|
|
1812
|
+
test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
|
|
1232
1813
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1233
1814
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1234
1815
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1235
1816
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
|
|
1236
1817
|
});
|
|
1237
1818
|
|
|
1238
|
-
test("
|
|
1819
|
+
test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
|
|
1239
1820
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1240
1821
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1241
1822
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1242
1823
|
assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
|
|
1243
1824
|
});
|
|
1244
1825
|
|
|
1245
|
-
test("
|
|
1826
|
+
test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
|
|
1246
1827
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1247
1828
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1248
1829
|
await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
|