@plurnk/plurnk-providers 1.3.11 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +40 -23
- package/README.md +65 -4
- package/SPEC.md +222 -56
- package/dist/AiSdkProvider.d.ts +18 -5
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +240 -153
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +14 -15
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +26 -10
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.d.ts +8 -2
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +41 -8
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts +4 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +7 -3
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +4 -2
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +18 -3
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +3 -1
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +20 -7
- package/dist/catalogProvider.js.map +1 -1
- package/dist/compatibleProvider.d.ts.map +1 -1
- package/dist/compatibleProvider.js +11 -4
- package/dist/compatibleProvider.js.map +1 -1
- package/dist/cost.d.ts +11 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +64 -0
- package/dist/cost.js.map +1 -0
- package/dist/discover.d.ts +2 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js +15 -9
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +4 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +29 -9
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts +27 -0
- package/dist/errors.d.ts.map +1 -0
- package/dist/errors.js +150 -0
- package/dist/errors.js.map +1 -0
- package/dist/index.d.ts +11 -5
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7 -4
- package/dist/index.js.map +1 -1
- package/dist/notices.d.ts +10 -0
- package/dist/notices.d.ts.map +1 -0
- package/dist/notices.js +11 -0
- package/dist/notices.js.map +1 -0
- package/dist/ollama.d.ts.map +1 -1
- package/dist/ollama.js +3 -3
- package/dist/ollama.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/promptTokens.d.ts +4 -0
- package/dist/promptTokens.d.ts.map +1 -0
- package/dist/promptTokens.js +32 -0
- package/dist/promptTokens.js.map +1 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +4 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +43 -16
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js +1 -1
- package/dist/types.js.map +1 -1
- package/dist/usage.d.ts +3 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +26 -14
- package/dist/usage.js.map +1 -1
- package/dist/warnings.js +0 -0
- package/dist/warnings.js.map +1 -1
- package/package.json +13 -9
- package/src/AiSdkProvider.test.ts +480 -159
- package/src/AiSdkProvider.ts +320 -196
- package/src/Mock.test.ts +29 -14
- package/src/Mock.ts +33 -15
- package/src/Pool.test.ts +43 -6
- package/src/Pool.ts +56 -10
- package/src/ProviderRegistry.test.ts +158 -9
- package/src/ProviderRegistry.ts +19 -6
- package/src/aiSdkTransport.ts +25 -6
- package/src/boundaries.test.ts +8 -3
- package/src/catalogProvider.test.ts +17 -0
- package/src/catalogProvider.ts +25 -10
- package/src/compatibleProvider.test.ts +96 -0
- package/src/compatibleProvider.ts +15 -6
- package/src/cost.test.ts +63 -0
- package/src/cost.ts +83 -0
- package/src/defaults.test.ts +1 -0
- package/src/discover.test.ts +48 -7
- package/src/discover.ts +31 -21
- package/src/env.test.ts +38 -23
- package/src/env.ts +45 -18
- package/src/errors.test.ts +148 -0
- package/src/errors.ts +207 -0
- package/src/index.ts +29 -7
- package/src/lexicon-guard.test.ts +6 -6
- package/src/notices.ts +22 -0
- package/src/ollama.test.ts +64 -0
- package/src/ollama.ts +6 -3
- package/src/openai.ts +3 -0
- package/src/promptTokens.ts +41 -0
- package/src/sdkModels.test.ts +7 -0
- package/src/sdkModels.ts +4 -8
- package/src/types.ts +106 -64
- package/src/usage.test.ts +15 -4
- package/src/usage.ts +32 -14
- package/src/warnings.test.ts +10 -10
- package/src/warnings.ts +0 -0
- package/dist/OpenAICompat.d.ts +0 -76
- package/dist/OpenAICompat.d.ts.map +0 -1
- package/dist/OpenAICompat.js +0 -555
- package/dist/OpenAICompat.js.map +0 -1
- package/dist/openaiStream.d.ts +0 -47
- package/dist/openaiStream.d.ts.map +0 -1
- package/dist/openaiStream.js +0 -280
- package/dist/openaiStream.js.map +0 -1
- package/dist/standardProviders.d.ts +0 -31
- package/dist/standardProviders.d.ts.map +0 -1
- package/dist/standardProviders.js +0 -518
- package/dist/standardProviders.js.map +0 -1
- package/dist/telemetry.d.ts +0 -24
- package/dist/telemetry.d.ts.map +0 -1
- package/dist/telemetry.js +0 -85
- package/dist/telemetry.js.map +0 -1
- package/src/telemetry.test.ts +0 -69
- package/src/telemetry.ts +0 -116
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
import test, { mock } from "node:test";
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
4
|
-
import { ProviderError } from "./
|
|
4
|
+
import { ProviderError } from "./errors.ts";
|
|
5
5
|
|
|
6
6
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
7
7
|
// so tests can assert what the spine sent on the wire.
|
|
@@ -71,7 +71,7 @@ const injectedBase = {
|
|
|
71
71
|
reasoning: { mode: "off" as const, budget: null },
|
|
72
72
|
};
|
|
73
73
|
|
|
74
|
-
test("
|
|
74
|
+
test("per-instance fetch owns streaming and buffered requests", async () => {
|
|
75
75
|
const calls: Array<{ input: string | URL | Request; init?: RequestInit }> = [];
|
|
76
76
|
const streamingFetch: typeof globalThis.fetch = async (input, init) => {
|
|
77
77
|
calls.push({ input, init });
|
|
@@ -105,7 +105,7 @@ test("#608: per-instance fetch owns streaming and buffered requests", async () =
|
|
|
105
105
|
assert.equal(JSON.parse(String(calls[1].init?.body)).stream, undefined);
|
|
106
106
|
});
|
|
107
107
|
|
|
108
|
-
test("
|
|
108
|
+
test("caller cancellation and provider timeout reach an injected fetch", async () => {
|
|
109
109
|
const pendingFetch: typeof globalThis.fetch = async (_input, init) => {
|
|
110
110
|
init?.signal?.throwIfAborted();
|
|
111
111
|
return new Promise((_resolve, reject) => {
|
|
@@ -126,7 +126,7 @@ test("#608: caller cancellation and provider timeout reach an injected fetch", a
|
|
|
126
126
|
);
|
|
127
127
|
});
|
|
128
128
|
|
|
129
|
-
test("
|
|
129
|
+
test("per-instance fetch owns tokenization and retry attempts", async () => {
|
|
130
130
|
const calls: string[] = [];
|
|
131
131
|
let generationAttempts = 0;
|
|
132
132
|
const providerFetch: typeof globalThis.fetch = async (input) => {
|
|
@@ -192,7 +192,7 @@ const installFetchScript = (responses: ScriptedResponse[]) => {
|
|
|
192
192
|
const flush = () => new Promise<void>((r) => setImmediate(r));
|
|
193
193
|
|
|
194
194
|
import { resetEmittedWarnings } from "./warnings.ts";
|
|
195
|
-
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
195
|
+
test.afterEach(() => { mock.restoreAll(); resetEmittedWarnings(); });
|
|
196
196
|
|
|
197
197
|
test("effortFromBudget: maps budget to tiers", () => {
|
|
198
198
|
assert.equal(effortFromBudget(1), "low");
|
|
@@ -202,7 +202,7 @@ test("effortFromBudget: maps budget to tiers", () => {
|
|
|
202
202
|
assert.equal(effortFromBudget(4001), "high");
|
|
203
203
|
});
|
|
204
204
|
|
|
205
|
-
test("
|
|
205
|
+
test("a 524 Cloudflare edge timeout fails fast - not retried despite retryAttempts", async () => {
|
|
206
206
|
const calls = installFetchScript([{ status: 524, retryAfter: 120 }]);
|
|
207
207
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 3 });
|
|
208
208
|
await assert.rejects(p.generate({ workerId: "r", messages: [] }));
|
|
@@ -211,7 +211,7 @@ test("#543: a 524 Cloudflare edge timeout fails fast - not retried despite retry
|
|
|
211
211
|
mock.restoreAll();
|
|
212
212
|
});
|
|
213
213
|
|
|
214
|
-
test("
|
|
214
|
+
test("a 422 grammar_invalid is a failed exchange, not transport replay policy", async () => {
|
|
215
215
|
const body = JSON.stringify({ error: { message: "non-conforming emission rejected: ...", type: "grammar_invalid" } });
|
|
216
216
|
const calls = installFetchScript([{ status: 422, body }]);
|
|
217
217
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 2 });
|
|
@@ -237,43 +237,59 @@ test("an SSE error frame is a failed exchange, not an empty completion", async (
|
|
|
237
237
|
assert.equal(calls.length, 1);
|
|
238
238
|
});
|
|
239
239
|
|
|
240
|
-
test("
|
|
240
|
+
test("a trailing eos_token (--special EOG leak) is stripped from content", async () => {
|
|
241
241
|
installFetchJson({ model: "m", choices: [{ message: { content: "the answer<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
242
242
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
243
243
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
244
244
|
assert.equal(res.assistant.content, "the answer"); // trailing <eos> gone; packet + verdict see clean bytes
|
|
245
245
|
});
|
|
246
246
|
|
|
247
|
-
test("
|
|
247
|
+
test("without a probed eos_token the content passes through untouched", async () => {
|
|
248
248
|
installFetchJson({ model: "m", choices: [{ message: { content: "keeps <eos> literally" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 3, total_tokens: 4 } });
|
|
249
249
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
250
250
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
251
251
|
assert.equal(res.assistant.content, "keeps <eos> literally"); // no eosText (a cloud backend) -> no strip
|
|
252
252
|
});
|
|
253
253
|
|
|
254
|
-
test("
|
|
254
|
+
test("only the trailing eos_token is stripped; a quoted one mid-body survives", async () => {
|
|
255
255
|
installFetchJson({ model: "m", choices: [{ message: { content: "quotes <eos> in the body<eos>" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 } });
|
|
256
256
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, eosText: "<eos>" });
|
|
257
257
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
258
258
|
assert.equal(res.assistant.content, "quotes <eos> in the body"); // only the tail goes
|
|
259
259
|
});
|
|
260
260
|
|
|
261
|
-
test("identity getters and
|
|
261
|
+
test("identity getters and default prompt estimate", async () => {
|
|
262
262
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
263
263
|
assert.equal(p.model, "m");
|
|
264
264
|
assert.equal(p.contextWindow, null); // default
|
|
265
|
-
assert.
|
|
266
|
-
|
|
267
|
-
|
|
265
|
+
assert.deepEqual(
|
|
266
|
+
await p.countPromptTokens([{ role: "user", content: "漢漢漢" }]),
|
|
267
|
+
{
|
|
268
|
+
kind: "estimate",
|
|
269
|
+
tokens: 2,
|
|
270
|
+
source: "heuristic:chars2",
|
|
271
|
+
detail: "chars/2 over message content; provider request framing is unknown",
|
|
272
|
+
},
|
|
273
|
+
"chars/2 is explicitly an estimate; high-token-density Unicode prevents an upper-bound claim",
|
|
274
|
+
);
|
|
275
|
+
assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // current unknown-rate sentinel
|
|
268
276
|
});
|
|
269
277
|
|
|
270
|
-
test("injected
|
|
278
|
+
test("injected prompt measurement preserves provenance and calculateCost is used", async () => {
|
|
279
|
+
const seen: string[] = [];
|
|
271
280
|
const p = new AiSdkProvider({
|
|
272
281
|
model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
273
|
-
|
|
282
|
+
countPromptTokens: (messages) => {
|
|
283
|
+
seen.push(...messages.map(({ content }) => content));
|
|
284
|
+
return { kind: "upper_bound", tokens: 7, source: "test:proven-bound" };
|
|
285
|
+
},
|
|
274
286
|
calculateCost: (u) => u.total * 2,
|
|
275
287
|
});
|
|
276
|
-
assert.
|
|
288
|
+
assert.deepEqual(
|
|
289
|
+
await p.countPromptTokens([{ role: "system", content: "system" }, { role: "user", content: "user" }]),
|
|
290
|
+
{ kind: "upper_bound", tokens: 7, source: "test:proven-bound" },
|
|
291
|
+
);
|
|
292
|
+
assert.deepEqual(seen, ["system", "user"]);
|
|
277
293
|
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
|
|
278
294
|
});
|
|
279
295
|
|
|
@@ -311,7 +327,88 @@ test("generate surfaces and normalizes an out-of-set finish_reason", async () =>
|
|
|
311
327
|
}]);
|
|
312
328
|
});
|
|
313
329
|
|
|
314
|
-
test("
|
|
330
|
+
test("#161: a streamed resource interruption is a failed exchange with complete attempt evidence", async () => {
|
|
331
|
+
const calls = installFetch([
|
|
332
|
+
{ model: "served-model", choices: [{ delta: { reasoning_content: "partial thought", content: "partial answer" } }] },
|
|
333
|
+
{
|
|
334
|
+
choices: [{ delta: {}, finish_reason: "insufficient_system_resource" }],
|
|
335
|
+
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
336
|
+
},
|
|
337
|
+
]);
|
|
338
|
+
const provider = new AiSdkProvider({
|
|
339
|
+
...injectedBase,
|
|
340
|
+
retryAttempts: 2,
|
|
341
|
+
rawBody: true,
|
|
342
|
+
});
|
|
343
|
+
|
|
344
|
+
await assert.rejects(
|
|
345
|
+
provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
|
|
346
|
+
(error: unknown) => {
|
|
347
|
+
assert.ok(error instanceof ProviderError);
|
|
348
|
+
assert.equal(error.kind, "resource_interrupted");
|
|
349
|
+
assert.equal(error.status, 503);
|
|
350
|
+
assert.equal(error.problem.stage, "provider-response");
|
|
351
|
+
assert.equal(error.problem.retryable, false);
|
|
352
|
+
assert.equal(error.problem.finishReason, "resource_interrupted");
|
|
353
|
+
assert.equal(error.problem.rawFinishReason, "insufficient_system_resource");
|
|
354
|
+
assert.equal(error.attempt?.assistant.content, "partial answer");
|
|
355
|
+
assert.equal(error.attempt?.assistant.reasoning, "partial thought");
|
|
356
|
+
assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
|
|
357
|
+
assert.deepEqual(error.attempt?.assistant.usage, {
|
|
358
|
+
prompt: 7,
|
|
359
|
+
completion: 2,
|
|
360
|
+
reasoning: 3,
|
|
361
|
+
cached: 0,
|
|
362
|
+
total: 12,
|
|
363
|
+
});
|
|
364
|
+
assert.equal(
|
|
365
|
+
(error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
|
|
366
|
+
"insufficient_system_resource",
|
|
367
|
+
);
|
|
368
|
+
assert.ok(Array.isArray(error.attempt?.rawBody));
|
|
369
|
+
return true;
|
|
370
|
+
},
|
|
371
|
+
);
|
|
372
|
+
assert.equal(calls.length, 1, "a semantic interruption is not replayed as an HTTP failure");
|
|
373
|
+
});
|
|
374
|
+
|
|
375
|
+
test("#161: a buffered resource interruption preserves the successful wire response as failed-attempt evidence", async () => {
|
|
376
|
+
const wire = {
|
|
377
|
+
model: "served-model",
|
|
378
|
+
choices: [{
|
|
379
|
+
message: { content: "partial answer", reasoning_content: "partial thought" },
|
|
380
|
+
finish_reason: "insufficient_system_resource",
|
|
381
|
+
}],
|
|
382
|
+
usage: { prompt_tokens: 7, completion_tokens: 5, total_tokens: 12 },
|
|
383
|
+
};
|
|
384
|
+
const calls = installFetchJson(wire);
|
|
385
|
+
const provider = new AiSdkProvider({
|
|
386
|
+
...injectedBase,
|
|
387
|
+
streaming: false,
|
|
388
|
+
retryAttempts: 2,
|
|
389
|
+
rawBody: true,
|
|
390
|
+
});
|
|
391
|
+
|
|
392
|
+
await assert.rejects(
|
|
393
|
+
provider.generate({ workerId: "interrupted", messages: [{ role: "user", content: "hello" }] }),
|
|
394
|
+
(error: unknown) => {
|
|
395
|
+
assert.ok(error instanceof ProviderError);
|
|
396
|
+
assert.equal(error.kind, "resource_interrupted");
|
|
397
|
+
assert.equal(error.attempt?.assistant.content, "partial answer");
|
|
398
|
+
assert.equal(error.attempt?.assistant.reasoning, "partial thought");
|
|
399
|
+
assert.equal(error.attempt?.assistant.finishReason, "resource_interrupted");
|
|
400
|
+
assert.deepEqual(error.attempt?.rawBody, wire);
|
|
401
|
+
assert.equal(
|
|
402
|
+
(error.attempt?.assistantRaw as { rawFinishReason?: string }).rawFinishReason,
|
|
403
|
+
"insufficient_system_resource",
|
|
404
|
+
);
|
|
405
|
+
return true;
|
|
406
|
+
},
|
|
407
|
+
);
|
|
408
|
+
assert.equal(calls.length, 1);
|
|
409
|
+
});
|
|
410
|
+
|
|
411
|
+
test("generate translates a backend cap synonym to canonical length", async () => {
|
|
315
412
|
// gemini shouts MAX_TOKENS, anthropic says max_tokens -- both must reach core as
|
|
316
413
|
// "length" so its truncation check (=== "length") is a cross-backend invariant.
|
|
317
414
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
@@ -320,7 +417,7 @@ test("generate translates a backend cap synonym to canonical length (#425)", asy
|
|
|
320
417
|
assert.equal(assistant.finishReason, "length");
|
|
321
418
|
});
|
|
322
419
|
|
|
323
|
-
test("generate translates end_turn to canonical stop
|
|
420
|
+
test("generate translates end_turn to canonical stop", async () => {
|
|
324
421
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
325
422
|
installFetch([{ choices: [{ delta: { content: "x" }, finish_reason: "end_turn" }] }]);
|
|
326
423
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
@@ -339,10 +436,172 @@ test("generate aggregates reasoning deltas under multiple field names", async ()
|
|
|
339
436
|
installFetch([{ choices: [{ delta: { reasoning_content: "be", thinking: "cause" } }] }]);
|
|
340
437
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
341
438
|
assert.equal(assistant.reasoning, "because");
|
|
342
|
-
assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
|
|
439
|
+
assert.equal("reasoningEncrypted" in assistant, false); // open reasoning only -> field absent
|
|
440
|
+
});
|
|
441
|
+
|
|
442
|
+
test("{§provider-tagged-reasoning} explicit think-tags projects one streamed leading envelope and reclassifies usage", async () => {
|
|
443
|
+
const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const, rawBody: true };
|
|
444
|
+
const p = new AiSdkProvider(config);
|
|
445
|
+
installFetch([
|
|
446
|
+
{ choices: [{ delta: { content: "<thi" } }] },
|
|
447
|
+
{ choices: [{ delta: { content: "nk>12345</th" } }] },
|
|
448
|
+
{ choices: [{ delta: { content: "ink>abcde" }, finish_reason: "stop" }] },
|
|
449
|
+
{ usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 } },
|
|
450
|
+
]);
|
|
451
|
+
|
|
452
|
+
const response = await p.generate({ workerId: "tagged-stream", messages: [] });
|
|
453
|
+
|
|
454
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
455
|
+
assert.equal(response.assistant.content, "abcde");
|
|
456
|
+
assert.deepEqual(response.assistant.usage, {
|
|
457
|
+
prompt: 3,
|
|
458
|
+
completion: 5,
|
|
459
|
+
reasoning: 5,
|
|
460
|
+
cached: 0,
|
|
461
|
+
total: 13,
|
|
462
|
+
});
|
|
463
|
+
assert.deepEqual(
|
|
464
|
+
((response.assistantRaw as { content: string; reasoning: string }).content),
|
|
465
|
+
"abcde",
|
|
466
|
+
);
|
|
467
|
+
assert.equal((response.assistantRaw as { reasoning: string }).reasoning, "12345");
|
|
468
|
+
assert.match(JSON.stringify(response.rawBody), /<thi/);
|
|
469
|
+
assert.match(JSON.stringify(response.rawBody), /nk>12345/);
|
|
470
|
+
});
|
|
471
|
+
|
|
472
|
+
test("{§provider-tagged-reasoning} explicit think-tags projects one buffered leading envelope", async () => {
|
|
473
|
+
installFetchJson({
|
|
474
|
+
model: "m",
|
|
475
|
+
choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
|
|
476
|
+
usage: { prompt_tokens: 3, completion_tokens: 10, total_tokens: 13 },
|
|
477
|
+
});
|
|
478
|
+
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
479
|
+
const response = await new AiSdkProvider(config).generate({ workerId: "tagged-buffer", messages: [] });
|
|
480
|
+
|
|
481
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
482
|
+
assert.equal(response.assistant.content, "abcde");
|
|
483
|
+
assert.equal(response.assistant.usage.completion, 5);
|
|
484
|
+
assert.equal(response.assistant.usage.reasoning, 5);
|
|
485
|
+
});
|
|
486
|
+
|
|
487
|
+
test("{§provider-tagged-reasoning} tagged text does not overwrite itemized reasoning usage", async () => {
|
|
488
|
+
installFetchJson({
|
|
489
|
+
model: "m",
|
|
490
|
+
choices: [{ message: { content: "<think>12345</think>abcde" }, finish_reason: "stop" }],
|
|
491
|
+
usage: {
|
|
492
|
+
prompt_tokens: 3,
|
|
493
|
+
completion_tokens: 10,
|
|
494
|
+
total_tokens: 13,
|
|
495
|
+
completion_tokens_details: { reasoning_tokens: 3 },
|
|
496
|
+
},
|
|
497
|
+
});
|
|
498
|
+
const config = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
499
|
+
const response = await new AiSdkProvider(config).generate({ workerId: "tagged-itemized", messages: [] });
|
|
500
|
+
|
|
501
|
+
assert.equal(response.assistant.reasoning, "12345");
|
|
502
|
+
assert.equal(response.assistant.content, "abcde");
|
|
503
|
+
assert.equal(response.assistant.usage.completion, 7);
|
|
504
|
+
assert.equal(response.assistant.usage.reasoning, 3);
|
|
505
|
+
});
|
|
506
|
+
|
|
507
|
+
test("{§provider-tagged-reasoning} an unclosed capped envelope is wholly reasoning in streamed and buffered responses", async () => {
|
|
508
|
+
const config = { ...injectedBase, reasoningResponseStyle: "think-tags" as const };
|
|
509
|
+
installFetch([
|
|
510
|
+
{ choices: [{ delta: { content: "<think>unfinished" }, finish_reason: "length" }] },
|
|
511
|
+
{ usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 } },
|
|
512
|
+
]);
|
|
513
|
+
const streamed = await new AiSdkProvider(config).generate({ workerId: "tagged-capped-stream", messages: [] });
|
|
514
|
+
assert.equal(streamed.assistant.reasoning, "unfinished");
|
|
515
|
+
assert.equal(streamed.assistant.content, "");
|
|
516
|
+
assert.deepEqual(streamed.assistant.usage, {
|
|
517
|
+
prompt: 3,
|
|
518
|
+
completion: 0,
|
|
519
|
+
reasoning: 8,
|
|
520
|
+
cached: 0,
|
|
521
|
+
total: 11,
|
|
522
|
+
});
|
|
523
|
+
|
|
524
|
+
mock.restoreAll();
|
|
525
|
+
installFetchJson({
|
|
526
|
+
model: "m",
|
|
527
|
+
choices: [{ message: { content: "<think>unfinished" }, finish_reason: "length" }],
|
|
528
|
+
usage: { prompt_tokens: 3, completion_tokens: 8, total_tokens: 11 },
|
|
529
|
+
});
|
|
530
|
+
const bufferedConfig = { ...config, streaming: false };
|
|
531
|
+
const buffered = await new AiSdkProvider(bufferedConfig).generate({ workerId: "tagged-capped-buffer", messages: [] });
|
|
532
|
+
assert.equal(buffered.assistant.reasoning, "unfinished");
|
|
533
|
+
assert.equal(buffered.assistant.content, "");
|
|
534
|
+
assert.equal(buffered.assistant.usage.completion, 0);
|
|
535
|
+
assert.equal(buffered.assistant.usage.reasoning, 8);
|
|
536
|
+
});
|
|
537
|
+
|
|
538
|
+
test("{§provider-tagged-reasoning} verbatim, non-leading, and structured-reasoning controls preserve literal tags", async () => {
|
|
539
|
+
installFetchJson({
|
|
540
|
+
model: "m",
|
|
541
|
+
choices: [{ message: { content: "<think>literal</think>answer" }, finish_reason: "stop" }],
|
|
542
|
+
usage: { prompt_tokens: 1, completion_tokens: 4, total_tokens: 5 },
|
|
543
|
+
});
|
|
544
|
+
const verbatim = await new AiSdkProvider({ ...injectedBase, streaming: false })
|
|
545
|
+
.generate({ workerId: "verbatim", messages: [] });
|
|
546
|
+
assert.equal(verbatim.assistant.content, "<think>literal</think>answer");
|
|
547
|
+
assert.equal(verbatim.assistant.reasoning, null);
|
|
548
|
+
assert.equal(verbatim.assistant.usage.completion, 4);
|
|
549
|
+
|
|
550
|
+
mock.restoreAll();
|
|
551
|
+
installFetchJson({
|
|
552
|
+
model: "m",
|
|
553
|
+
choices: [{ message: { content: "show <think>literal</think> exactly" }, finish_reason: "stop" }],
|
|
554
|
+
usage: { prompt_tokens: 1, completion_tokens: 5, total_tokens: 6 },
|
|
555
|
+
});
|
|
556
|
+
const taggedConfig = { ...injectedBase, streaming: false, reasoningResponseStyle: "think-tags" as const };
|
|
557
|
+
const nonLeading = await new AiSdkProvider(taggedConfig)
|
|
558
|
+
.generate({ workerId: "non-leading", messages: [] });
|
|
559
|
+
assert.equal(nonLeading.assistant.content, "show <think>literal</think> exactly");
|
|
560
|
+
assert.equal(nonLeading.assistant.reasoning, null);
|
|
561
|
+
|
|
562
|
+
mock.restoreAll();
|
|
563
|
+
installFetchJson({
|
|
564
|
+
model: "m",
|
|
565
|
+
choices: [{ message: {
|
|
566
|
+
content: "<think>literal visible bytes</think>",
|
|
567
|
+
reasoning_content: "structured reasoning",
|
|
568
|
+
}, finish_reason: "stop" }],
|
|
569
|
+
usage: { prompt_tokens: 1, completion_tokens: 7, total_tokens: 8 },
|
|
570
|
+
});
|
|
571
|
+
const structured = await new AiSdkProvider(taggedConfig)
|
|
572
|
+
.generate({ workerId: "structured", messages: [] });
|
|
573
|
+
assert.equal(structured.assistant.content, "<think>literal visible bytes</think>");
|
|
574
|
+
assert.equal(structured.assistant.reasoning, "structured reasoning");
|
|
575
|
+
});
|
|
576
|
+
|
|
577
|
+
test("{§provider-tagged-reasoning} grammar evidence retains the exact pre-projection tagged sentence", async () => {
|
|
578
|
+
const content = "<think>🧠reason</think><<PLAN::PLAN\n<<SEND[200]:done:SEND";
|
|
579
|
+
const config = {
|
|
580
|
+
...injectedBase,
|
|
581
|
+
contextWindow: 640,
|
|
582
|
+
reasoning: { mode: "adaptive" as const, budget: null },
|
|
583
|
+
reasoningResponseStyle: "think-tags" as const,
|
|
584
|
+
reasoningStyle: "think" as const,
|
|
585
|
+
grammarStyle: "llamacpp" as const,
|
|
586
|
+
};
|
|
587
|
+
installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
588
|
+
|
|
589
|
+
const response = await new AiSdkProvider(config).generate({
|
|
590
|
+
workerId: "tagged-grammar",
|
|
591
|
+
messages: [],
|
|
592
|
+
grammar: `root ::= ${JSON.stringify(content)}`,
|
|
593
|
+
});
|
|
594
|
+
|
|
595
|
+
assert.equal(response.assistant.reasoning, "🧠reason");
|
|
596
|
+
assert.equal(response.assistant.content, "<<PLAN::PLAN\n<<SEND[200]:done:SEND");
|
|
597
|
+
assert.deepEqual(response.grammarEvidence, {
|
|
598
|
+
input: content,
|
|
599
|
+
contentStart: [..."<think>🧠reason</think>"].length,
|
|
600
|
+
transported: true,
|
|
601
|
+
});
|
|
343
602
|
});
|
|
344
603
|
|
|
345
|
-
test("
|
|
604
|
+
test("encrypted reasoning (non-streamed): encrypted entries normalize and text entries stay separate", async () => {
|
|
346
605
|
// The live o4-mini-via-OpenRouter shape: reasoning null, one encrypted entry.
|
|
347
606
|
installFetchJson({ model: "m", choices: [{ message: {
|
|
348
607
|
content: "4", reasoning: null,
|
|
@@ -353,13 +612,14 @@ test("#482 sealed relay reasoning (non-streamed): encrypted reasoning_details su
|
|
|
353
612
|
}, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
354
613
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
355
614
|
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
356
|
-
//
|
|
615
|
+
// Wire detail ID is preserved; the assistant-message location supports the
|
|
616
|
+
// derived classification but supplies no downstream client entity ID.
|
|
357
617
|
assert.deepEqual(assistant.reasoningEncrypted, [{ id: "rs_1", subtype: "message", encrypted: [{ data: "gAAAAABqBLOB", format: "openai-responses-v1" }] }]);
|
|
358
|
-
assert.equal(assistant.reasoning, null); //
|
|
618
|
+
assert.equal(assistant.reasoning, null); // Encrypted turn: nothing readable.
|
|
359
619
|
assert.equal(assistant.content, "4");
|
|
360
620
|
});
|
|
361
621
|
|
|
362
|
-
test("
|
|
622
|
+
test("distinct encrypted-reasoning wire ids stay distinct items", async () => {
|
|
363
623
|
installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning: null, reasoning_details: [
|
|
364
624
|
{ type: "reasoning.encrypted", data: "AAA", format: "openai-responses-v1", id: "rs_1" },
|
|
365
625
|
{ type: "reasoning.encrypted", data: "BBB", format: "openai-responses-v1", id: "rs_2" },
|
|
@@ -370,7 +630,20 @@ test("#482 widening: distinct wire ids stay distinct items (a single-object shap
|
|
|
370
630
|
assert.deepEqual(assistant.reasoningEncrypted?.map((i) => i.id), ["rs_1", "rs_2"]);
|
|
371
631
|
});
|
|
372
632
|
|
|
373
|
-
test("
|
|
633
|
+
test("assistant-message location classifies encrypted reasoning without inventing a missing detail id", async () => {
|
|
634
|
+
installFetchJson({ model: "m", choices: [{ message: { content: "ok", reasoning_details: [
|
|
635
|
+
{ type: "reasoning.encrypted", data: "OPAQUE", format: "openai-responses-v1", id: null, index: 0 },
|
|
636
|
+
] }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } });
|
|
637
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
638
|
+
const { assistant } = await p.generate({ workerId: "r", messages: [] });
|
|
639
|
+
assert.deepEqual(assistant.reasoningEncrypted, [{
|
|
640
|
+
id: null,
|
|
641
|
+
subtype: "message",
|
|
642
|
+
encrypted: [{ data: "OPAQUE", format: "openai-responses-v1" }],
|
|
643
|
+
}]);
|
|
644
|
+
});
|
|
645
|
+
|
|
646
|
+
test("encrypted reasoning (streamed): chunked blob concatenates per entry index", async () => {
|
|
374
647
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
375
648
|
installFetch([
|
|
376
649
|
{ choices: [{ delta: { reasoning_details: [{ type: "reasoning.encrypted", data: "gAAAA", format: "openai-responses-v1", id: "rs_1", index: 0 }] } }] },
|
|
@@ -402,10 +675,10 @@ test("reasoningStyle 'effort' sends a reasoning_effort tier from the budget", as
|
|
|
402
675
|
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
403
676
|
});
|
|
404
677
|
|
|
405
|
-
test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS
|
|
678
|
+
test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS, on sends the tier", async () => {
|
|
406
679
|
// expected === null → the field must be ABSENT from the wire body. Fireworks
|
|
407
680
|
// 400s reasoning_effort='adaptive' for non-MiniMax models (wire-verified,
|
|
408
|
-
//
|
|
681
|
+
// Adaptive = the backend's own default posture = omission.
|
|
409
682
|
for (const [reasoning, expected] of [[{ mode: "off", budget: null }, "none"], [{ mode: "adaptive", budget: null }, null], [{ mode: "on", budget: 5000 }, "high"]] as Array<[{ mode: "off" | "adaptive" | "on"; budget: number | null }, string | null]>) {
|
|
410
683
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning, retryAttempts: 0, reasoningStyle: "effort_explicit" });
|
|
411
684
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
@@ -417,7 +690,39 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
417
690
|
}
|
|
418
691
|
});
|
|
419
692
|
|
|
420
|
-
test("
|
|
693
|
+
test("{§deepseek-reasoning-request} #157: thinking_effort maps the complete DeepSeek reasoning contract", async () => {
|
|
694
|
+
const cases = [
|
|
695
|
+
[{ mode: "off", budget: null }, { thinking: { type: "disabled" } }],
|
|
696
|
+
[{ mode: "adaptive", budget: null }, {}],
|
|
697
|
+
[{ mode: "on", budget: 5000 }, { thinking: { type: "enabled" }, reasoning_effort: "high" }],
|
|
698
|
+
] as const;
|
|
699
|
+
for (const [reasoning, expected] of cases) {
|
|
700
|
+
const p = new AiSdkProvider({
|
|
701
|
+
model: "m",
|
|
702
|
+
url: "http://x/v1/chat/completions",
|
|
703
|
+
fetchTimeoutMs: 5000,
|
|
704
|
+
temperature: 0.2,
|
|
705
|
+
repeatPenalty: 1.15,
|
|
706
|
+
reasoning,
|
|
707
|
+
retryAttempts: 0,
|
|
708
|
+
reasoningStyle: "thinking_effort",
|
|
709
|
+
});
|
|
710
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
711
|
+
await p.generate({
|
|
712
|
+
workerId: "r",
|
|
713
|
+
messages: [],
|
|
714
|
+
sampling: { thinking: { type: "disabled" }, reasoning_effort: "max" },
|
|
715
|
+
});
|
|
716
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
717
|
+
assert.deepEqual(
|
|
718
|
+
Object.fromEntries(Object.entries(body).filter(([key]) => key === "thinking" || key === "reasoning_effort")),
|
|
719
|
+
expected,
|
|
720
|
+
);
|
|
721
|
+
mock.restoreAll();
|
|
722
|
+
}
|
|
723
|
+
});
|
|
724
|
+
|
|
725
|
+
test("the family temperature default rides every request; caller sampling overrides it", async () => {
|
|
421
726
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
422
727
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
423
728
|
await p.generate({ workerId: "r", messages: [] });
|
|
@@ -434,7 +739,7 @@ test("the family temperature default rides every request; caller sampling overri
|
|
|
434
739
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
435
740
|
});
|
|
436
741
|
|
|
437
|
-
test("
|
|
742
|
+
test("DRY + repeat_last_n ride the llamacpp path when set; unset leaves the box default; never on cloud", async () => {
|
|
438
743
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off" as const, budget: null }, retryAttempts: 0 };
|
|
439
744
|
// set + llamacpp -> the loop-breakers ride the wire
|
|
440
745
|
const p = new AiSdkProvider({ ...base, grammarStyle: "llamacpp", dryMultiplier: 0.8, dryBase: 1.75, dryAllowedLength: 2, repeatLastN: 512 });
|
|
@@ -474,14 +779,14 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
474
779
|
assert.equal(body.repeat_penalty, 1.15);
|
|
475
780
|
});
|
|
476
781
|
|
|
477
|
-
test("
|
|
782
|
+
test("the repeat penalty rides every request rail-off, keyed per backend", async () => {
|
|
478
783
|
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
479
784
|
const llama = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
480
785
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
481
786
|
await llama.generate({ workerId: "r", messages: [] });
|
|
482
787
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
483
788
|
mock.restoreAll();
|
|
484
|
-
//
|
|
789
|
+
// A `none`-style cloud backend with a frequency penalty gets frequency_penalty.
|
|
485
790
|
const cloud = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
486
791
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
487
792
|
await cloud.generate({ workerId: "r", messages: [] });
|
|
@@ -520,7 +825,7 @@ test("sampling passthrough forwards caller params; managed + reserved keys win",
|
|
|
520
825
|
assert.equal("id_slot" in body, false); // reserved slot key stripped
|
|
521
826
|
});
|
|
522
827
|
|
|
523
|
-
test("
|
|
828
|
+
test("sampling passthrough guards contract invariants: n/tools/caps stripped, platform knobs pass", async () => {
|
|
524
829
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
525
830
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
526
831
|
await p.generate({
|
|
@@ -531,7 +836,7 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
531
836
|
n: 3, // breaks choices[0] atomicity -> stripped
|
|
532
837
|
tools: [{ type: "function" }], tool_choice: "auto", // tools-in-body doctrine -> stripped
|
|
533
838
|
modalities: ["text", "audio"], prediction: { type: "content" }, // text-only / decode semantics -> stripped
|
|
534
|
-
max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass
|
|
839
|
+
max_tokens: 999999, max_completion_tokens: 999999, // envelope bypass -> stripped
|
|
535
840
|
seed: 42, user: "acct-7", service_tier: "flex", // platform/sampling intent -> pass
|
|
536
841
|
},
|
|
537
842
|
});
|
|
@@ -545,58 +850,97 @@ test("#477 sampling passthrough guards contract invariants: n/tools/caps strippe
|
|
|
545
850
|
assert.equal(body.service_tier, "flex");
|
|
546
851
|
});
|
|
547
852
|
|
|
548
|
-
test("
|
|
549
|
-
|
|
550
|
-
|
|
551
|
-
const
|
|
552
|
-
const
|
|
553
|
-
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
853
|
+
test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
|
|
854
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
855
|
+
const calls = installFetch([{ choices: [{ delta: { reasoning_content: "con🙂sider", content: "x" } }] }]);
|
|
856
|
+
const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
|
|
857
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
|
|
554
858
|
const body = JSON.parse(calls[0].init.body as string);
|
|
555
|
-
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
556
|
-
assert.equal(
|
|
557
|
-
|
|
558
|
-
assert.equal(
|
|
559
|
-
assert.
|
|
859
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
860
|
+
assert.equal(body.reasoning_format, "auto");
|
|
861
|
+
assert.equal(body.thinking_budget_tokens, 64);
|
|
862
|
+
assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
|
|
863
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
864
|
+
input: grammarInput,
|
|
865
|
+
contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
|
|
866
|
+
transported: true,
|
|
867
|
+
});
|
|
868
|
+
assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
|
|
869
|
+
});
|
|
870
|
+
|
|
871
|
+
test("template reasoning does not invent pre-projection evidence when the wire omits its reasoning field", async () => {
|
|
872
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
873
|
+
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
874
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
875
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
560
876
|
});
|
|
561
877
|
|
|
562
|
-
test("
|
|
878
|
+
test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
|
|
563
879
|
// The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
|
|
564
880
|
// escaped into a discarded reasoning block, unconstrained.
|
|
565
|
-
const
|
|
566
|
-
installFetch([
|
|
881
|
+
const chunks = [
|
|
567
882
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
568
883
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
569
|
-
]
|
|
884
|
+
];
|
|
885
|
+
const fetch: typeof globalThis.fetch = async (input, init) => {
|
|
886
|
+
if (String(input).endsWith("/tokenize")) {
|
|
887
|
+
const body = JSON.parse(String(init?.body)) as { content: string };
|
|
888
|
+
return new Response(JSON.stringify({
|
|
889
|
+
tokens: body.content.length === 0 ? [] : [1],
|
|
890
|
+
}), { headers: { "content-type": "application/json" } });
|
|
891
|
+
}
|
|
892
|
+
return new Response(sseStream(chunks), { status: 200 });
|
|
893
|
+
};
|
|
894
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetch, tokenizeUrl: "http://x/tokenize", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
570
895
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
assert.ok(escape, "escape telemetry attached");
|
|
896
|
+
const escape = res.notices?.find((e) => e.message.includes("escaped the grammar"));
|
|
897
|
+
assert.ok(escape, "escape notice attached");
|
|
574
898
|
assert.equal(escape!.kind, "grammar_unenforced");
|
|
575
899
|
assert.match(escape!.message ?? "", /5000 completion tokens billed/);
|
|
576
900
|
});
|
|
577
901
|
|
|
578
|
-
test("
|
|
902
|
+
test("channel-escape state is absent without a transported grammar", async () => {
|
|
579
903
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
580
904
|
installFetch([
|
|
581
905
|
{ choices: [{ delta: { content: "x" }, finish_reason: "length" }] },
|
|
582
906
|
{ usage: { prompt_tokens: 10, completion_tokens: 5000, total_tokens: 5010 } },
|
|
583
907
|
]);
|
|
584
908
|
const res = await p.generate({ workerId: "r", messages: [] }); // no grammar arg
|
|
585
|
-
assert.equal(res.
|
|
586
|
-
assert.equal(res.
|
|
909
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
910
|
+
assert.equal(res.notices, undefined);
|
|
587
911
|
});
|
|
588
912
|
|
|
589
|
-
test("reasoningStyle 'template'
|
|
590
|
-
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
913
|
+
test("reasoningStyle 'template' sends llama-server activation, parser, and response-wide allowance", async () => {
|
|
914
|
+
const on = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
591
915
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
592
916
|
await on.generate({ workerId: "r", messages: [] });
|
|
593
|
-
|
|
917
|
+
let body = JSON.parse(calls[0].init.body as string);
|
|
918
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
919
|
+
assert.equal(body.reasoning_format, "auto");
|
|
920
|
+
assert.equal(body.thinking_budget_tokens, 64);
|
|
594
921
|
|
|
595
922
|
mock.restoreAll();
|
|
596
|
-
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
923
|
+
const off = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template" });
|
|
597
924
|
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
598
925
|
await off.generate({ workerId: "r", messages: [] });
|
|
599
|
-
|
|
926
|
+
body = JSON.parse(calls[0].init.body as string);
|
|
927
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
|
|
928
|
+
assert.equal(body.reasoning_format, "auto");
|
|
929
|
+
assert.equal(body.thinking_budget_tokens, 0);
|
|
930
|
+
});
|
|
931
|
+
|
|
932
|
+
test("reasoningStyle 'template' explicit budget tightens the reserve and cannot exceed it", async () => {
|
|
933
|
+
const base = { model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 } as const, completionReserve: { tokens: 160 } as const, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryAttempts: 0, reasoningStyle: "template" as const };
|
|
934
|
+
const p = new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 32 } });
|
|
935
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
936
|
+
await p.generate({ workerId: "r", messages: [], sampling: { thinking_budget_tokens: 999, reasoning_format: "none" } });
|
|
937
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
938
|
+
assert.equal(body.thinking_budget_tokens, 32);
|
|
939
|
+
assert.equal(body.reasoning_format, "auto");
|
|
940
|
+
assert.throws(
|
|
941
|
+
() => new AiSdkProvider({ ...base, reasoning: { mode: "on", budget: 65 } }),
|
|
942
|
+
/REASONING_BUDGET \(65\) exceeds the resolved PLURNK_PROVIDERS_REASONING_RESERVE \(64\)/,
|
|
943
|
+
);
|
|
600
944
|
});
|
|
601
945
|
|
|
602
946
|
test("budget 0 suppresses effort and include_reasoning", async () => {
|
|
@@ -619,7 +963,7 @@ test("reasoningStyle 'include_reasoning' sets the relay passthrough toggle", asy
|
|
|
619
963
|
assert.equal(JSON.parse(calls[0].init.body as string).include_reasoning, true);
|
|
620
964
|
});
|
|
621
965
|
|
|
622
|
-
// — grammar-constrained sampling
|
|
966
|
+
// — grammar-constrained sampling —
|
|
623
967
|
|
|
624
968
|
test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor", async () => {
|
|
625
969
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
@@ -640,106 +984,81 @@ test("grammar transport 'none' (default): the grammar is never sent — no silen
|
|
|
640
984
|
assert.equal("response_format" in body, false);
|
|
641
985
|
});
|
|
642
986
|
|
|
643
|
-
// — grammar
|
|
644
|
-
// returns; bytes flow; a non-accept verdict rides response.telemetry —
|
|
987
|
+
// — exact pre-projection grammar evidence ({§gbnf-response-observation}) —
|
|
645
988
|
|
|
646
989
|
const grammarProvider = () => new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", source: "provider:test" });
|
|
647
990
|
const streamingContent = (content: string) => installFetch([{ choices: [{ delta: { content }, finish_reason: "stop" }] }]);
|
|
648
991
|
|
|
649
|
-
test("
|
|
992
|
+
test("an unsplit grammar response carries the exact observed sentence", async () => {
|
|
650
993
|
const p = grammarProvider();
|
|
651
994
|
streamingContent("ok");
|
|
652
|
-
const { assistant } = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
653
|
-
assert.equal(assistant.content, "ok");
|
|
654
|
-
});
|
|
655
|
-
|
|
656
|
-
test("observation: REJECTED output still returns — bytes present, verdict attached with position", async () => {
|
|
657
|
-
const p = grammarProvider();
|
|
658
|
-
streamingContent("no");
|
|
659
995
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
660
|
-
assert.equal(res.assistant.content, "no"); // bytes ALWAYS flow
|
|
661
|
-
assert.equal(res.telemetry?.length, 1);
|
|
662
|
-
const ev = res.telemetry![0];
|
|
663
|
-
assert.equal(ev.kind, "grammar_unenforced");
|
|
664
|
-
assert.equal(ev.source, "provider:test");
|
|
665
|
-
assert.match(String(ev.message), /grammar not enforced: output rejected .* at code point 0/);
|
|
666
|
-
assert.equal(ev.position, 0); // divergence offset for consumer policy
|
|
667
|
-
});
|
|
668
|
-
|
|
669
|
-
test("observation: an incomplete (valid prefix, never terminated) also returns with the verdict", async () => {
|
|
670
|
-
const p = grammarProvider();
|
|
671
|
-
streamingContent("ok");
|
|
672
|
-
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok" "!"' });
|
|
673
996
|
assert.equal(res.assistant.content, "ok");
|
|
674
|
-
assert.
|
|
675
|
-
|
|
676
|
-
|
|
997
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
998
|
+
input: "ok",
|
|
999
|
+
contentStart: 0,
|
|
1000
|
+
transported: true,
|
|
1001
|
+
});
|
|
677
1002
|
});
|
|
678
1003
|
|
|
679
|
-
test("
|
|
1004
|
+
test("the provider returns rejected or incomplete bytes as evidence without grading them", async () => {
|
|
680
1005
|
const p = grammarProvider();
|
|
681
|
-
streamingContent("
|
|
1006
|
+
streamingContent("no");
|
|
682
1007
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
683
|
-
assert.equal(res.
|
|
1008
|
+
assert.equal(res.assistant.content, "no");
|
|
1009
|
+
assert.deepEqual(res.grammarEvidence, { input: "no", contentStart: 0, transported: true });
|
|
1010
|
+
assert.equal(res.notices, undefined);
|
|
1011
|
+
assert.equal(res.meta?.railsVerdict, undefined);
|
|
684
1012
|
});
|
|
685
1013
|
|
|
686
|
-
test("
|
|
1014
|
+
test("empty unsplit content remains exact grammar evidence", async () => {
|
|
687
1015
|
const p = grammarProvider();
|
|
688
|
-
installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
|
|
1016
|
+
installFetch([{ choices: [{ delta: {}, finish_reason: "stop" }] }]);
|
|
689
1017
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
690
1018
|
assert.equal(res.assistant.content, "");
|
|
691
|
-
assert.
|
|
1019
|
+
assert.deepEqual(res.grammarEvidence, { input: "", contentStart: 0, transported: true });
|
|
692
1020
|
});
|
|
693
1021
|
|
|
694
|
-
test("
|
|
1022
|
+
test("grammarStyle 'none' produces no grammar observation", async () => {
|
|
695
1023
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 }); // grammarStyle defaults to "none"
|
|
696
1024
|
streamingContent("anything goes");
|
|
697
|
-
const
|
|
698
|
-
assert.equal(assistant.content, "anything goes");
|
|
1025
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
1026
|
+
assert.equal(res.assistant.content, "anything goes");
|
|
1027
|
+
assert.equal(res.grammarEvidence, undefined);
|
|
699
1028
|
});
|
|
700
1029
|
|
|
701
|
-
test("
|
|
1030
|
+
test("provider evidence does not depend on the local validator understanding the grammar", async () => {
|
|
702
1031
|
const p = grammarProvider();
|
|
703
1032
|
streamingContent("whatever");
|
|
704
|
-
const
|
|
705
|
-
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
await flush();
|
|
709
|
-
process.off("warning", onWarn);
|
|
710
|
-
assert.equal(assistant.content, "whatever"); // transport not failed
|
|
711
|
-
assert.ok(warnings.some((w) => (w as Error & { code?: string }).code === "PLURNK_GRAMMAR_UNVERIFIABLE"), "emitted the verify-gap warning");
|
|
1033
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'foo ::= "a"' });
|
|
1034
|
+
assert.equal(res.assistant.content, "whatever");
|
|
1035
|
+
assert.deepEqual(res.grammarEvidence, { input: "whatever", contentStart: 0, transported: true });
|
|
1036
|
+
assert.equal(res.notices, undefined);
|
|
712
1037
|
});
|
|
713
1038
|
|
|
714
|
-
// — PLURNK_PROVIDERS_GBNF_DEBUG:
|
|
1039
|
+
// — PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar, withhold it, and preserve the observation —
|
|
715
1040
|
|
|
716
|
-
test("gbnfDebug
|
|
1041
|
+
test("gbnfDebug marks an unconstrained observation as not transported", async () => {
|
|
717
1042
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
718
1043
|
const calls = installFetch([{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] }]);
|
|
719
1044
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
720
1045
|
const body = JSON.parse(calls[0].init.body as string);
|
|
721
|
-
assert.equal("grammar" in body, false);
|
|
722
|
-
assert.equal(body.repeat_penalty, 1.15);
|
|
723
|
-
assert.equal(res.assistant.content, "ok");
|
|
724
|
-
assert.
|
|
1046
|
+
assert.equal("grammar" in body, false);
|
|
1047
|
+
assert.equal(body.repeat_penalty, 1.15);
|
|
1048
|
+
assert.equal(res.assistant.content, "ok");
|
|
1049
|
+
assert.deepEqual(res.grammarEvidence, { input: "ok", contentStart: 0, transported: false });
|
|
1050
|
+
assert.equal(res.notices, undefined);
|
|
725
1051
|
});
|
|
726
1052
|
|
|
727
|
-
test("gbnfDebug
|
|
1053
|
+
test("gbnfDebug preserves conflicting bytes without a provider verdict", async () => {
|
|
728
1054
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp", gbnfDebug: true, source: "provider:test" });
|
|
729
|
-
const calls = installFetch([{ choices: [{ delta: {
|
|
1055
|
+
const calls = installFetch([{ choices: [{ delta: { content: "xon-conforming output" }, finish_reason: "stop" }] }]);
|
|
730
1056
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
731
|
-
// The model's bytes survive — not discarded by a throw (the empty-turn cascade root cause).
|
|
732
1057
|
assert.equal(res.assistant.content, "xon-conforming output");
|
|
733
|
-
assert.
|
|
734
|
-
|
|
735
|
-
assert.equal(res.telemetry?.length, 1);
|
|
736
|
-
const [event] = res.telemetry ?? [];
|
|
737
|
-
assert.equal(event.source, "provider:test");
|
|
738
|
-
assert.equal(event.kind, "grammar_unenforced");
|
|
739
|
-
assert.equal(event.position, 0); // 'x' rejected at code point 0
|
|
740
|
-
assert.match(event.message ?? "", /output rejected by the transported grammar at code point 0 \("x"\)/);
|
|
1058
|
+
assert.deepEqual(res.grammarEvidence, { input: "xon-conforming output", contentStart: 0, transported: false });
|
|
1059
|
+
assert.equal(res.notices, undefined);
|
|
741
1060
|
const body = JSON.parse(calls[0].init.body as string);
|
|
742
|
-
assert.equal("grammar" in body, false);
|
|
1061
|
+
assert.equal("grammar" in body, false);
|
|
743
1062
|
});
|
|
744
1063
|
|
|
745
1064
|
test("gbnfDebug: an INVALID grammar throws before any wire call — it never reaches the model", async () => {
|
|
@@ -752,7 +1071,7 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
752
1071
|
assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
|
|
753
1072
|
});
|
|
754
1073
|
|
|
755
|
-
// — meta bag: verbatim provider metadata
|
|
1074
|
+
// — meta bag: verbatim provider metadata —
|
|
756
1075
|
|
|
757
1076
|
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
758
1077
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
@@ -763,7 +1082,7 @@ test("meta: passes backend fields through without reinterpreting monetary values
|
|
|
763
1082
|
assert.equal(res.meta?.system_fingerprint, "fp_abc");
|
|
764
1083
|
});
|
|
765
1084
|
|
|
766
|
-
// — first-party telemetry headers (
|
|
1085
|
+
// — first-party telemetry headers ({§provider-request-authority}) —
|
|
767
1086
|
|
|
768
1087
|
const headerVal = (init: RequestInit, name: string): string | undefined =>
|
|
769
1088
|
new Headers(init.headers).get(name) ?? undefined;
|
|
@@ -776,7 +1095,7 @@ test("firstPartyMetadata: attributions + client ride as Plurnk-* headers", async
|
|
|
776
1095
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), "plurnk.nvim/1.4.0");
|
|
777
1096
|
});
|
|
778
1097
|
|
|
779
|
-
test("
|
|
1098
|
+
test("Plurnk-Worker-Primary: the lineage root rides under the gate; emitted even when it equals workerId", async () => {
|
|
780
1099
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
781
1100
|
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
782
1101
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
@@ -795,7 +1114,7 @@ test("#522 Plurnk-Worker-Primary: the lineage root rides under the gate; emitted
|
|
|
795
1114
|
assert.equal(headerVal(calls[0].init, "Plurnk-Worker-Primary"), undefined);
|
|
796
1115
|
});
|
|
797
1116
|
|
|
798
|
-
test("
|
|
1117
|
+
test("Plurnk-Worker-Primary is structurally dropped when firstPartyMetadata is off", async () => {
|
|
799
1118
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
800
1119
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
801
1120
|
await p.generate({ workerId: "w-child", primaryWorkerId: "w-root", messages: [] });
|
|
@@ -818,13 +1137,13 @@ test("firstPartyMetadata on but empty values: no header emitted", async () => {
|
|
|
818
1137
|
assert.equal(headerVal(calls[0].init, "Plurnk-Client"), undefined);
|
|
819
1138
|
});
|
|
820
1139
|
|
|
821
|
-
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides
|
|
1140
|
+
test("grammar transport: no grammar passed sends no grammar field, but the penalty rides", async () => {
|
|
822
1141
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
823
1142
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
824
1143
|
await p.generate({ workerId: "r", messages: [] });
|
|
825
1144
|
const body = JSON.parse(calls[0].init.body as string);
|
|
826
1145
|
assert.equal("grammar" in body, false);
|
|
827
|
-
assert.equal(body.repeat_penalty, 1.15); //
|
|
1146
|
+
assert.equal(body.repeat_penalty, 1.15); // penalty is not grammar-gated
|
|
828
1147
|
});
|
|
829
1148
|
|
|
830
1149
|
test("maxTokens transports as max_tokens; absent → no wire field (server default)", async () => {
|
|
@@ -839,7 +1158,7 @@ test("maxTokens transports as max_tokens; absent → no wire field (server defau
|
|
|
839
1158
|
assert.equal("max_tokens" in JSON.parse(calls[0].init.body as string), false);
|
|
840
1159
|
});
|
|
841
1160
|
|
|
842
|
-
test("slot affinity is internal: sticky per workerId, distinct
|
|
1161
|
+
test("slot affinity is internal: sticky per workerId, distinct workers spread across slots", async () => {
|
|
843
1162
|
const pinning = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
844
1163
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
845
1164
|
await pinning.generate({ workerId: "run-A", messages: [] });
|
|
@@ -863,7 +1182,7 @@ test("slot affinity: no pinning backend or unknown slotCount → no id_slot ever
|
|
|
863
1182
|
assert.equal("id_slot" in JSON.parse(calls[0].init.body as string), false);
|
|
864
1183
|
});
|
|
865
1184
|
|
|
866
|
-
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent
|
|
1185
|
+
test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; recent workers stay sticky", async () => {
|
|
867
1186
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, supportsSlotPinning: true, slotCount: 2 });
|
|
868
1187
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
869
1188
|
const slotOf = (i: number) => JSON.parse(calls[i].init.body as string).id_slot;
|
|
@@ -877,11 +1196,11 @@ test("slot affinity: a worker past the LRU window (slotCount*8) loses its pin; r
|
|
|
877
1196
|
});
|
|
878
1197
|
|
|
879
1198
|
test("streaming:false: a non-ok response rejects as a classified ProviderError (covers the non-streamed transport)", async () => {
|
|
880
|
-
const { ProviderError } = await import("./
|
|
1199
|
+
const { ProviderError } = await import("./errors.ts");
|
|
881
1200
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, source: "provider:test" });
|
|
882
1201
|
mock.method(globalThis, "fetch", async () => new Response("boom", { status: 500 }));
|
|
883
1202
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
884
|
-
assert.ok(err instanceof ProviderError);
|
|
1203
|
+
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
885
1204
|
assert.equal(err.kind, "network_failure"); // ≥500 → network_failure
|
|
886
1205
|
assert.equal(err.status, 500);
|
|
887
1206
|
return true;
|
|
@@ -904,15 +1223,17 @@ test("messages pass through verbatim — the provider injects no turn (PLAN live
|
|
|
904
1223
|
assert.equal(res.assistant.content, "out"); // content returned verbatim
|
|
905
1224
|
});
|
|
906
1225
|
|
|
907
|
-
test("generate wraps an HTTP failure as a ProviderError carrying
|
|
908
|
-
const { ProviderError } = await import("./
|
|
1226
|
+
test("generate wraps an HTTP failure as a ProviderError carrying Problem Details", async () => {
|
|
1227
|
+
const { ProviderError } = await import("./errors.ts");
|
|
909
1228
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, source: "provider:test" });
|
|
910
1229
|
mock.method(globalThis, "fetch", async () => new Response("rate limited", { status: 429 }));
|
|
911
1230
|
await assert.rejects(() => p.generate({ workerId: "r", messages: [] }), (err: unknown) => {
|
|
912
|
-
assert.ok(err instanceof ProviderError);
|
|
1231
|
+
assert.ok(err instanceof ProviderError, `expected ProviderError, got ${String(err)}`);
|
|
913
1232
|
assert.equal(err.kind, "rate_limit");
|
|
914
1233
|
assert.equal(err.status, 429);
|
|
915
|
-
assert.
|
|
1234
|
+
assert.equal(err.problem.status, 429);
|
|
1235
|
+
assert.equal(err.problem.detail, err.message);
|
|
1236
|
+
assert.equal(err.problem.type, "https://problems.plurnk.dev/provider/test/rate-limit");
|
|
916
1237
|
return true;
|
|
917
1238
|
});
|
|
918
1239
|
});
|
|
@@ -937,7 +1258,7 @@ test("configured headers and url are sent verbatim", async () => {
|
|
|
937
1258
|
assert.equal(headers.get("x-title"), "plurnk");
|
|
938
1259
|
});
|
|
939
1260
|
|
|
940
|
-
// — transient-failure retry
|
|
1261
|
+
// — transient-failure retry —
|
|
941
1262
|
|
|
942
1263
|
const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
|
|
943
1264
|
|
|
@@ -953,7 +1274,7 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
953
1274
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
954
1275
|
});
|
|
955
1276
|
|
|
956
|
-
test("
|
|
1277
|
+
test("streamed-body silence fails the exchange without replaying partial output", async () => {
|
|
957
1278
|
let calls = 0;
|
|
958
1279
|
mock.method(globalThis, "fetch", async () => {
|
|
959
1280
|
calls++;
|
|
@@ -996,7 +1317,7 @@ test("#559: streamed-body silence fails the exchange without replaying partial o
|
|
|
996
1317
|
mock.restoreAll();
|
|
997
1318
|
});
|
|
998
1319
|
|
|
999
|
-
test("
|
|
1320
|
+
test("a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
|
|
1000
1321
|
mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
|
|
1001
1322
|
async start(controller) {
|
|
1002
1323
|
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
|
|
@@ -1021,7 +1342,7 @@ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async
|
|
|
1021
1342
|
});
|
|
1022
1343
|
|
|
1023
1344
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
1024
|
-
const { ProviderError } = await import("./
|
|
1345
|
+
const { ProviderError } = await import("./errors.ts");
|
|
1025
1346
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
|
1026
1347
|
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 2 });
|
|
1027
1348
|
await assert.rejects(
|
|
@@ -1056,7 +1377,7 @@ test("retry: retryAttempts 0 surfaces the first transient failure immediately",
|
|
|
1056
1377
|
assert.equal(calls.length, 1); // no retry budget
|
|
1057
1378
|
});
|
|
1058
1379
|
|
|
1059
|
-
test("retry: a caller abort during backoff rejects promptly with no further attempt
|
|
1380
|
+
test("retry: a caller abort during backoff rejects promptly with no further attempt", async () => {
|
|
1060
1381
|
const ac = new AbortController();
|
|
1061
1382
|
const calls = installFetchScript([{ status: 503, retryAfter: 5 }]); // 5s backoff we never wait out
|
|
1062
1383
|
const p = new AiSdkProvider({ ...retryCfg, retryAttempts: 3 });
|
|
@@ -1068,7 +1389,7 @@ test("retry: a caller abort during backoff rejects promptly with no further atte
|
|
|
1068
1389
|
assert.equal(calls.length, 1); // never retried after cancellation
|
|
1069
1390
|
});
|
|
1070
1391
|
|
|
1071
|
-
// —
|
|
1392
|
+
// — Anthropic reasoning style (wire `thinking` parameter) —
|
|
1072
1393
|
|
|
1073
1394
|
test("reasoningStyle 'anthropic' maps the budget to the thinking param", async () => {
|
|
1074
1395
|
// N>0 → enabled with budget_tokens
|
|
@@ -1115,10 +1436,10 @@ test("streaming:false posts without stream and parses the single JSON response",
|
|
|
1115
1436
|
mock.restoreAll();
|
|
1116
1437
|
});
|
|
1117
1438
|
|
|
1118
|
-
// ── Data capture (
|
|
1439
|
+
// ── Data capture ({§provider-evidence}): logprobs + verbatim rawBody, opt-in, off by default ──
|
|
1119
1440
|
const captureBase = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1120
1441
|
|
|
1121
|
-
test("
|
|
1442
|
+
test("logprobs OFF by default: no wire request, no assistant.logprobs, no rawBody", async () => {
|
|
1122
1443
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1123
1444
|
const p = new AiSdkProvider({ ...captureBase });
|
|
1124
1445
|
const res = await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }] });
|
|
@@ -1131,7 +1452,7 @@ test("#36 logprobs OFF by default: no wire request, no assistant.logprobs, no ra
|
|
|
1131
1452
|
mock.restoreAll();
|
|
1132
1453
|
});
|
|
1133
1454
|
|
|
1134
|
-
test("
|
|
1455
|
+
test("logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw logprob + meanLogprob", async () => {
|
|
1135
1456
|
const chunk = { model: "m", usage: { prompt_tokens: 1, completion_tokens: 2, total_tokens: 3 }, choices: [{ delta: { content: "yesno" }, finish_reason: "stop", logprobs: { content: [
|
|
1136
1457
|
{ token: "yes", logprob: -0.5, sampling_logprob: -0.5, top_logprobs: [{ token: "yes", logprob: -0.5 }, { token: "no", logprob: -1.0 }] },
|
|
1137
1458
|
{ token: "no", logprob: -0.1, sampling_logprob: -0.1, top_logprobs: [{ token: "no", logprob: -0.1 }] },
|
|
@@ -1148,7 +1469,7 @@ test("#36 logprobs ON (streamed): requests logprobs+top_logprobs, surfaces raw l
|
|
|
1148
1469
|
mock.restoreAll();
|
|
1149
1470
|
});
|
|
1150
1471
|
|
|
1151
|
-
test("
|
|
1472
|
+
test("rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob preserved", async () => {
|
|
1152
1473
|
const wire = { model: "m", extra_top_level: "kept", choices: [{ message: { content: "no" }, finish_reason: "stop", logprobs: { content: [{ token: "no", logprob: -0.1, sampling_logprob: -0.1, token_id: 42 }] } }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } };
|
|
1153
1474
|
installFetchJson(wire);
|
|
1154
1475
|
const p = new AiSdkProvider({ ...captureBase, streaming: false, topLogprobs: 0, rawBody: true });
|
|
@@ -1160,7 +1481,7 @@ test("#36 rawBody ON (non-streamed): verbatim wire body incl. sampling_logprob p
|
|
|
1160
1481
|
mock.restoreAll();
|
|
1161
1482
|
});
|
|
1162
1483
|
|
|
1163
|
-
test("
|
|
1484
|
+
test("caller sampling cannot forge logprobs (reserved keys): the env flag is the only control", async () => {
|
|
1164
1485
|
const calls = installFetch([{ model: "m", choices: [{ delta: { content: "hi" }, finish_reason: "stop" }], usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 } }]);
|
|
1165
1486
|
const p = new AiSdkProvider({ ...captureBase }); // logprobs OFF
|
|
1166
1487
|
await p.generate({ workerId: "r", messages: [{ role: "user", content: "q" }], sampling: { logprobs: true, top_logprobs: 5 } });
|
|
@@ -1170,9 +1491,9 @@ test("#36 caller sampling cannot forge logprobs (reserved keys): the env flag is
|
|
|
1170
1491
|
mock.restoreAll();
|
|
1171
1492
|
});
|
|
1172
1493
|
|
|
1173
|
-
// — turn coordinate headers (
|
|
1494
|
+
// — turn coordinate headers ({§lifecycle-terms}): same gate as every first-party signal —
|
|
1174
1495
|
|
|
1175
|
-
test("
|
|
1496
|
+
test("workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under the first-party gate", async () => {
|
|
1176
1497
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1177
1498
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1178
1499
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
@@ -1182,7 +1503,7 @@ test("#404: workspaceId/loop/turn ride as Plurnk-Workspace-Id/Loop/Turn under th
|
|
|
1182
1503
|
assert.equal(headers.get("plurnk-turn"), "41");
|
|
1183
1504
|
});
|
|
1184
1505
|
|
|
1185
|
-
test("
|
|
1506
|
+
test("third-party providers structurally DROP the coordinate (gate off by default)", async () => {
|
|
1186
1507
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1187
1508
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1188
1509
|
await p.generate({ workerId: "r", messages: [], workspaceId: "s-9", loop: 3, turn: 41 });
|
|
@@ -1192,7 +1513,7 @@ test("#404: third-party providers structurally DROP the coordinate (gate off by
|
|
|
1192
1513
|
assert.equal(headers.has("plurnk-turn"), false);
|
|
1193
1514
|
});
|
|
1194
1515
|
|
|
1195
|
-
test("
|
|
1516
|
+
test("coordinates are 1-based — 0/absent/empty emit no header", async () => {
|
|
1196
1517
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, firstPartyMetadata: true });
|
|
1197
1518
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1198
1519
|
await p.generate({ workerId: "r", messages: [], workspaceId: "", loop: 0, turn: 0 });
|
|
@@ -1203,9 +1524,9 @@ test("#404: coordinates are 1-based — 0/absent/empty emit no header (no strike
|
|
|
1203
1524
|
assert.equal(headers.has("plurnk-strikes"), false);
|
|
1204
1525
|
});
|
|
1205
1526
|
|
|
1206
|
-
// --
|
|
1527
|
+
// -- {§provider-generation-envelope} --
|
|
1207
1528
|
|
|
1208
|
-
test("
|
|
1529
|
+
test("reserves derive from the detected window; absolutes stand alone; null window + percent = no claim", () => {
|
|
1209
1530
|
const base = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const, retryAttempts: 0 };
|
|
1210
1531
|
const derived = new AiSdkProvider({ ...base, contextWindow: 49152, reasoningReserve: { percent: 0.1 }, completionReserve: { percent: 0.25 } });
|
|
1211
1532
|
assert.equal(derived.reasoningReserve, 4915); // jennifer/turboderp: 10% of 49152
|
|
@@ -1217,32 +1538,32 @@ test("#507 reserves derive from the detected window; absolutes stand alone; null
|
|
|
1217
1538
|
assert.equal(legacy.reasoningReserve, null); // out-of-date sibling: no claim
|
|
1218
1539
|
});
|
|
1219
1540
|
|
|
1220
|
-
test("
|
|
1541
|
+
test("router-owned tuning: tuningFloors:false drops the temperature/penalty floors, caller sampling still rides", async () => {
|
|
1221
1542
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, frequencyPenalty: 0.4, reasoning: { mode: "off", budget: null }, retryAttempts: 0, tuningFloors: false });
|
|
1222
1543
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1223
1544
|
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.9 } });
|
|
1224
1545
|
const body = JSON.parse(calls[0].init.body as string);
|
|
1225
1546
|
assert.equal(body.temperature, 0.9); // caller intent passes verbatim
|
|
1226
|
-
assert.equal("frequency_penalty" in body, false); // the floor is suppressed
|
|
1547
|
+
assert.equal("frequency_penalty" in body, false); // the floor is suppressed; the router owns tuning
|
|
1227
1548
|
});
|
|
1228
1549
|
|
|
1229
|
-
// --
|
|
1550
|
+
// -- prompt-cache affinity (workerId -> prompt_cache_key) --
|
|
1230
1551
|
|
|
1231
|
-
test("
|
|
1552
|
+
test("promptCacheKey on: body sends prompt_cache_key = workerId (serverless replica affinity)", async () => {
|
|
1232
1553
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1233
1554
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1234
1555
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1235
1556
|
assert.equal(JSON.parse(calls[0].init.body as string).prompt_cache_key, "worker-abc");
|
|
1236
1557
|
});
|
|
1237
1558
|
|
|
1238
|
-
test("
|
|
1559
|
+
test("promptCacheKey off (default): no prompt_cache_key on the wire", async () => {
|
|
1239
1560
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
1240
1561
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1241
1562
|
await p.generate({ workerId: "worker-abc", messages: [] });
|
|
1242
1563
|
assert.equal("prompt_cache_key" in JSON.parse(calls[0].init.body as string), false);
|
|
1243
1564
|
});
|
|
1244
1565
|
|
|
1245
|
-
test("
|
|
1566
|
+
test("prompt_cache_key is managed: caller sampling cannot forge/override the affinity key", async () => {
|
|
1246
1567
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, promptCacheKey: true });
|
|
1247
1568
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1248
1569
|
await p.generate({ workerId: "worker-abc", messages: [], sampling: { prompt_cache_key: "hijack" } });
|