@plurnk/plurnk-providers 1.4.0 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +9 -17
- package/README.md +3 -0
- package/SPEC.md +40 -23
- package/dist/AiSdkProvider.d.ts +2 -1
- package/dist/AiSdkProvider.d.ts.map +1 -1
- package/dist/AiSdkProvider.js +55 -34
- package/dist/AiSdkProvider.js.map +1 -1
- package/dist/Mock.d.ts +6 -7
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -0
- package/dist/Mock.js.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/accounting.d.ts +3 -0
- package/dist/accounting.d.ts.map +1 -0
- package/dist/accounting.js +84 -0
- package/dist/accounting.js.map +1 -0
- package/dist/aiSdkTransport.d.ts +2 -1
- package/dist/aiSdkTransport.d.ts.map +1 -1
- package/dist/aiSdkTransport.js +73 -2
- package/dist/aiSdkTransport.js.map +1 -1
- package/dist/catalogProvider.d.ts +3 -2
- package/dist/catalogProvider.d.ts.map +1 -1
- package/dist/catalogProvider.js +11 -11
- package/dist/catalogProvider.js.map +1 -1
- package/dist/cost.d.ts +1 -1
- package/dist/cost.d.ts.map +1 -1
- package/dist/cost.js +6 -9
- package/dist/cost.js.map +1 -1
- package/dist/env.d.ts +0 -6
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +0 -22
- package/dist/env.js.map +1 -1
- package/dist/errors.d.ts.map +1 -1
- package/dist/errors.js +2 -0
- package/dist/errors.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/sdkModels.d.ts +2 -0
- package/dist/sdkModels.d.ts.map +1 -1
- package/dist/sdkModels.js +13 -3
- package/dist/sdkModels.js.map +1 -1
- package/dist/types.d.ts +9 -0
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -0
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +38 -5
- package/dist/usage.js.map +1 -1
- package/package.json +8 -7
- package/src/AiSdkProvider.test.ts +277 -17
- package/src/AiSdkProvider.ts +70 -40
- package/src/Mock.ts +5 -2
- package/src/Pool.ts +1 -1
- package/src/accounting.test.ts +58 -0
- package/src/accounting.ts +88 -0
- package/src/aiSdkTransport.ts +77 -3
- package/src/boundaries.test.ts +1 -0
- package/src/catalogProvider.test.ts +26 -15
- package/src/catalogProvider.ts +12 -11
- package/src/cost.test.ts +7 -6
- package/src/cost.ts +4 -9
- package/src/env.test.ts +1 -26
- package/src/env.ts +0 -24
- package/src/errors.ts +1 -0
- package/src/index.ts +1 -1
- package/src/sdkModels.test.ts +22 -3
- package/src/sdkModels.ts +15 -3
- package/src/types.ts +22 -3
- package/src/usage.test.ts +9 -1
- package/src/usage.ts +40 -7
|
@@ -2,6 +2,8 @@ import test, { mock } from "node:test";
|
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
|
|
4
4
|
import { ProviderError } from "./errors.ts";
|
|
5
|
+
import { authoritativeChargeNormalizer } from "./accounting.ts";
|
|
6
|
+
import type { LanguageModel } from "ai";
|
|
5
7
|
|
|
6
8
|
// Build a fake fetch returning a one-chunk SSE stream, capturing the request
|
|
7
9
|
// so tests can assert what the spine sent on the wire.
|
|
@@ -309,6 +311,139 @@ test("generate maps a streamed response into ProviderResponse", async () => {
|
|
|
309
311
|
assert.notEqual(assistantRaw, undefined);
|
|
310
312
|
});
|
|
311
313
|
|
|
314
|
+
test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
|
|
315
|
+
const usage = {
|
|
316
|
+
inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
|
|
317
|
+
outputTokens: { total: 1, text: 1, reasoning: 0 },
|
|
318
|
+
};
|
|
319
|
+
const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
|
|
320
|
+
const charge = {
|
|
321
|
+
kind: "authoritative",
|
|
322
|
+
amount: { amount: "0.00154935", currency: "USD" },
|
|
323
|
+
usdEquivalent: "0.00154935",
|
|
324
|
+
source: "OpenRouter response usage.cost",
|
|
325
|
+
};
|
|
326
|
+
const languageModel = {
|
|
327
|
+
specificationVersion: "v4",
|
|
328
|
+
provider: "openrouter.chat",
|
|
329
|
+
modelId: "router-test",
|
|
330
|
+
supportedUrls: {},
|
|
331
|
+
doGenerate: async () => ({
|
|
332
|
+
content: [{ type: "text", text: "ok" }],
|
|
333
|
+
finishReason: { unified: "stop", raw: "completed" },
|
|
334
|
+
usage,
|
|
335
|
+
providerMetadata,
|
|
336
|
+
response: { id: "response-buffered", modelId: "router-test" },
|
|
337
|
+
warnings: [],
|
|
338
|
+
}),
|
|
339
|
+
doStream: async () => ({
|
|
340
|
+
stream: new ReadableStream({
|
|
341
|
+
start(controller) {
|
|
342
|
+
controller.enqueue({ type: "stream-start", warnings: [] });
|
|
343
|
+
controller.enqueue({ type: "response-metadata", id: "response-streamed", modelId: "router-test" });
|
|
344
|
+
controller.enqueue({ type: "text-start", id: "text-1" });
|
|
345
|
+
controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
|
|
346
|
+
controller.enqueue({ type: "text-end", id: "text-1" });
|
|
347
|
+
controller.enqueue({
|
|
348
|
+
type: "finish",
|
|
349
|
+
finishReason: { unified: "stop", raw: "completed" },
|
|
350
|
+
usage,
|
|
351
|
+
providerMetadata,
|
|
352
|
+
});
|
|
353
|
+
controller.close();
|
|
354
|
+
},
|
|
355
|
+
}),
|
|
356
|
+
response: {},
|
|
357
|
+
}),
|
|
358
|
+
} as unknown as LanguageModel;
|
|
359
|
+
const config = {
|
|
360
|
+
model: "router-test",
|
|
361
|
+
languageModel,
|
|
362
|
+
fetchTimeoutMs: 5_000,
|
|
363
|
+
temperature: 0.2,
|
|
364
|
+
repeatPenalty: 1.15,
|
|
365
|
+
reasoning: { mode: "off" as const, budget: null },
|
|
366
|
+
retryAttempts: 0,
|
|
367
|
+
normalizeCharge: authoritativeChargeNormalizer("@openrouter/ai-sdk-provider"),
|
|
368
|
+
};
|
|
369
|
+
|
|
370
|
+
await t.test("buffered", async () => {
|
|
371
|
+
const response = await new AiSdkProvider({ ...config, streaming: false })
|
|
372
|
+
.generate({ workerId: "buffered", messages: [] });
|
|
373
|
+
assert.deepEqual(response.charge, charge);
|
|
374
|
+
});
|
|
375
|
+
await t.test("streamed", async () => {
|
|
376
|
+
const response = await new AiSdkProvider(config)
|
|
377
|
+
.generate({ workerId: "streamed", messages: [] });
|
|
378
|
+
assert.deepEqual(response.charge, charge);
|
|
379
|
+
});
|
|
380
|
+
});
|
|
381
|
+
|
|
382
|
+
test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
|
|
383
|
+
const p = new AiSdkProvider({
|
|
384
|
+
model: "grok-test",
|
|
385
|
+
url: "http://x/v1/chat/completions",
|
|
386
|
+
fetchTimeoutMs: 5_000,
|
|
387
|
+
temperature: 0.2,
|
|
388
|
+
repeatPenalty: 1.15,
|
|
389
|
+
reasoning: { mode: "off", budget: null },
|
|
390
|
+
retryAttempts: 0,
|
|
391
|
+
streaming: false,
|
|
392
|
+
normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
|
|
393
|
+
});
|
|
394
|
+
installFetchJson({
|
|
395
|
+
id: "response-1",
|
|
396
|
+
model: "grok-test",
|
|
397
|
+
choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
|
|
398
|
+
usage: {
|
|
399
|
+
prompt_tokens: 2,
|
|
400
|
+
completion_tokens: 1,
|
|
401
|
+
total_tokens: 3,
|
|
402
|
+
cost_in_usd_ticks: 15_493_500,
|
|
403
|
+
},
|
|
404
|
+
});
|
|
405
|
+
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
406
|
+
assert.deepEqual(response.charge, {
|
|
407
|
+
kind: "authoritative",
|
|
408
|
+
amount: { amount: "15493500", currency: "USDTICK" },
|
|
409
|
+
usdEquivalent: "0.00154935",
|
|
410
|
+
source: "xAI response usage.cost_in_usd_ticks",
|
|
411
|
+
});
|
|
412
|
+
assert.equal(response.rawBody, undefined);
|
|
413
|
+
});
|
|
414
|
+
|
|
415
|
+
test("streamed xAI final usage retains its exact tick charge", async () => {
|
|
416
|
+
const p = new AiSdkProvider({
|
|
417
|
+
model: "grok-test",
|
|
418
|
+
url: "http://x/v1/chat/completions",
|
|
419
|
+
fetchTimeoutMs: 5_000,
|
|
420
|
+
temperature: 0.2,
|
|
421
|
+
repeatPenalty: 1.15,
|
|
422
|
+
reasoning: { mode: "off", budget: null },
|
|
423
|
+
retryAttempts: 0,
|
|
424
|
+
normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
|
|
425
|
+
});
|
|
426
|
+
installFetch([
|
|
427
|
+
{ choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
|
|
428
|
+
{
|
|
429
|
+
choices: [],
|
|
430
|
+
usage: {
|
|
431
|
+
prompt_tokens: 2,
|
|
432
|
+
completion_tokens: 1,
|
|
433
|
+
total_tokens: 3,
|
|
434
|
+
cost_in_usd_ticks: 15_493_500,
|
|
435
|
+
},
|
|
436
|
+
},
|
|
437
|
+
]);
|
|
438
|
+
const response = await p.generate({ workerId: "xai", messages: [] });
|
|
439
|
+
assert.deepEqual(response.charge, {
|
|
440
|
+
kind: "authoritative",
|
|
441
|
+
amount: { amount: "15493500", currency: "USDTICK" },
|
|
442
|
+
usdEquivalent: "0.00154935",
|
|
443
|
+
source: "xAI response usage.cost_in_usd_ticks",
|
|
444
|
+
});
|
|
445
|
+
});
|
|
446
|
+
|
|
312
447
|
test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
|
|
313
448
|
const warnings: Array<{ message: string; code?: string }> = [];
|
|
314
449
|
mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
|
|
@@ -852,14 +987,16 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
|
|
|
852
987
|
|
|
853
988
|
test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
|
|
854
989
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
855
|
-
const calls = installFetch([{ choices: [{ delta: { reasoning_content: "con🙂sider", content: "x" } }] }]);
|
|
856
990
|
const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
|
|
991
|
+
const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
|
|
857
992
|
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
|
|
858
993
|
const body = JSON.parse(calls[0].init.body as string);
|
|
859
994
|
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
|
|
860
|
-
assert.equal(body.reasoning_format, "
|
|
995
|
+
assert.equal(body.reasoning_format, "none");
|
|
861
996
|
assert.equal(body.thinking_budget_tokens, 64);
|
|
862
997
|
assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
|
|
998
|
+
assert.equal(res.assistant.reasoning, "con🙂sider");
|
|
999
|
+
assert.equal(res.assistant.content, "x");
|
|
863
1000
|
assert.deepEqual(res.grammarEvidence, {
|
|
864
1001
|
input: grammarInput,
|
|
865
1002
|
contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
|
|
@@ -868,13 +1005,48 @@ test("template reasoning returns the exact pre-projection grammar sentence ({§g
|
|
|
868
1005
|
assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
|
|
869
1006
|
});
|
|
870
1007
|
|
|
871
|
-
test("
|
|
1008
|
+
test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
|
|
872
1009
|
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
873
|
-
installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1010
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1011
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1012
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1013
|
+
assert.equal(body.reasoning_format, "none");
|
|
1014
|
+
assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
|
|
1015
|
+
});
|
|
1016
|
+
|
|
1017
|
+
test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
|
|
1018
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1019
|
+
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
1020
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
1021
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1022
|
+
assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
|
|
1023
|
+
assert.equal(body.reasoning_format, "none");
|
|
1024
|
+
assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
|
|
1025
|
+
});
|
|
1026
|
+
|
|
1027
|
+
test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
|
|
1028
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1029
|
+
installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
|
|
874
1030
|
const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
875
1031
|
assert.equal(res.grammarEvidence, undefined);
|
|
876
1032
|
});
|
|
877
1033
|
|
|
1034
|
+
test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
|
|
1035
|
+
const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
|
|
1036
|
+
const input = "<|channel>thought\n<channel|>x";
|
|
1037
|
+
const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
|
|
1038
|
+
const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
|
|
1039
|
+
const body = JSON.parse(calls[0].init.body as string);
|
|
1040
|
+
assert.equal(body.reasoning_format, "none");
|
|
1041
|
+
assert.equal(res.assistant.reasoning, null);
|
|
1042
|
+
assert.equal(res.assistant.content, "x");
|
|
1043
|
+
assert.deepEqual(res.grammarEvidence, {
|
|
1044
|
+
input,
|
|
1045
|
+
contentStart: [..."<|channel>thought\n<channel|>"].length,
|
|
1046
|
+
transported: true,
|
|
1047
|
+
});
|
|
1048
|
+
});
|
|
1049
|
+
|
|
878
1050
|
test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
|
|
879
1051
|
// The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
|
|
880
1052
|
// escaped into a discarded reasoning block, unconstrained.
|
|
@@ -1262,6 +1434,15 @@ test("configured headers and url are sent verbatim", async () => {
|
|
|
1262
1434
|
|
|
1263
1435
|
const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
|
|
1264
1436
|
|
|
1437
|
+
const stalledStreamResponse = (): Response => new Response(new ReadableStream({
|
|
1438
|
+
start(controller) {
|
|
1439
|
+
controller.enqueue(new TextEncoder().encode(
|
|
1440
|
+
'data: {"id":"stalled","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
1441
|
+
));
|
|
1442
|
+
setTimeout(() => controller.close(), 100);
|
|
1443
|
+
},
|
|
1444
|
+
}), { status: 200 });
|
|
1445
|
+
|
|
1265
1446
|
test("retry: a transient failure retries and a later success resolves", async () => {
|
|
1266
1447
|
const calls = installFetchScript([
|
|
1267
1448
|
{ status: 429, retryAfter: 0 },
|
|
@@ -1274,20 +1455,11 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
1274
1455
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
1275
1456
|
});
|
|
1276
1457
|
|
|
1277
|
-
test("streamed-body silence
|
|
1458
|
+
test("streamed-body silence retries and returns the retry's complete output", async () => {
|
|
1278
1459
|
let calls = 0;
|
|
1279
1460
|
mock.method(globalThis, "fetch", async () => {
|
|
1280
1461
|
calls++;
|
|
1281
|
-
if (calls === 1)
|
|
1282
|
-
return new Response(new ReadableStream({
|
|
1283
|
-
start(controller) {
|
|
1284
|
-
controller.enqueue(new TextEncoder().encode(
|
|
1285
|
-
'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
|
|
1286
|
-
));
|
|
1287
|
-
setTimeout(() => controller.close(), 100);
|
|
1288
|
-
},
|
|
1289
|
-
}), { status: 200 });
|
|
1290
|
-
}
|
|
1462
|
+
if (calls === 1) return stalledStreamResponse();
|
|
1291
1463
|
return new Response(new ReadableStream({
|
|
1292
1464
|
start(controller) {
|
|
1293
1465
|
controller.enqueue(new TextEncoder().encode(
|
|
@@ -1300,7 +1472,7 @@ test("streamed-body silence fails the exchange without replaying partial output"
|
|
|
1300
1472
|
const p = new AiSdkProvider({
|
|
1301
1473
|
model: "m",
|
|
1302
1474
|
url: "http://x/v1/chat/completions",
|
|
1303
|
-
fetchTimeoutMs:
|
|
1475
|
+
fetchTimeoutMs: 5000,
|
|
1304
1476
|
streamIdleTimeoutMs: 10,
|
|
1305
1477
|
temperature: 0.2,
|
|
1306
1478
|
repeatPenalty: 1.15,
|
|
@@ -1308,12 +1480,100 @@ test("streamed-body silence fails the exchange without replaying partial output"
|
|
|
1308
1480
|
retryAttempts: 1,
|
|
1309
1481
|
source: "provider:test",
|
|
1310
1482
|
});
|
|
1483
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
1484
|
+
assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the stalled partial");
|
|
1485
|
+
assert.equal(calls, 2, "the stall retried once and the retry succeeded");
|
|
1486
|
+
mock.restoreAll();
|
|
1487
|
+
});
|
|
1488
|
+
|
|
1489
|
+
test("streamed-body silence does not replay when retries are disabled", async () => {
|
|
1490
|
+
let calls = 0;
|
|
1491
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1492
|
+
calls++;
|
|
1493
|
+
return stalledStreamResponse();
|
|
1494
|
+
});
|
|
1495
|
+
const p = new AiSdkProvider({
|
|
1496
|
+
model: "m",
|
|
1497
|
+
url: "http://x/v1/chat/completions",
|
|
1498
|
+
fetchTimeoutMs: 1000,
|
|
1499
|
+
streamIdleTimeoutMs: 10,
|
|
1500
|
+
temperature: 0.2,
|
|
1501
|
+
repeatPenalty: 1.15,
|
|
1502
|
+
reasoning: { mode: "off", budget: null },
|
|
1503
|
+
retryAttempts: 0,
|
|
1504
|
+
source: "provider:test",
|
|
1505
|
+
});
|
|
1311
1506
|
await assert.rejects(
|
|
1312
1507
|
p.generate({ workerId: "r", messages: [] }),
|
|
1313
1508
|
(error: ProviderError) => error.kind === "network_failure"
|
|
1314
1509
|
&& /chunk timeout/i.test(error.message),
|
|
1315
1510
|
);
|
|
1316
|
-
assert.equal(calls, 1);
|
|
1511
|
+
assert.equal(calls, 1, "zero retries permits exactly one provider request");
|
|
1512
|
+
mock.restoreAll();
|
|
1513
|
+
});
|
|
1514
|
+
|
|
1515
|
+
test("streamed-body silence exhausts the configured retry budget once", async () => {
|
|
1516
|
+
let calls = 0;
|
|
1517
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1518
|
+
calls++;
|
|
1519
|
+
return stalledStreamResponse();
|
|
1520
|
+
});
|
|
1521
|
+
const p = new AiSdkProvider({
|
|
1522
|
+
model: "m",
|
|
1523
|
+
url: "http://x/v1/chat/completions",
|
|
1524
|
+
fetchTimeoutMs: 5000,
|
|
1525
|
+
streamIdleTimeoutMs: 10,
|
|
1526
|
+
temperature: 0.2,
|
|
1527
|
+
repeatPenalty: 1.15,
|
|
1528
|
+
reasoning: { mode: "off", budget: null },
|
|
1529
|
+
retryAttempts: 1,
|
|
1530
|
+
source: "provider:test",
|
|
1531
|
+
});
|
|
1532
|
+
await assert.rejects(
|
|
1533
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
1534
|
+
(error: ProviderError) => error.kind === "network_failure"
|
|
1535
|
+
&& error.problem.attempts === 2
|
|
1536
|
+
&& error.problem.retryExhausted === true
|
|
1537
|
+
&& error.problem.retryable === false,
|
|
1538
|
+
);
|
|
1539
|
+
assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
|
|
1540
|
+
mock.restoreAll();
|
|
1541
|
+
});
|
|
1542
|
+
|
|
1543
|
+
test("the total generation deadline spans stalled-stream retry scheduling", async () => {
|
|
1544
|
+
let calls = 0;
|
|
1545
|
+
mock.method(globalThis, "fetch", async () => {
|
|
1546
|
+
calls++;
|
|
1547
|
+
if (calls > 1) {
|
|
1548
|
+
return new Response(new ReadableStream({
|
|
1549
|
+
start(controller) {
|
|
1550
|
+
controller.enqueue(new TextEncoder().encode(
|
|
1551
|
+
'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"late"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
|
|
1552
|
+
));
|
|
1553
|
+
controller.close();
|
|
1554
|
+
},
|
|
1555
|
+
}), { status: 200 });
|
|
1556
|
+
}
|
|
1557
|
+
return stalledStreamResponse();
|
|
1558
|
+
});
|
|
1559
|
+
const p = new AiSdkProvider({
|
|
1560
|
+
model: "m",
|
|
1561
|
+
url: "http://x/v1/chat/completions",
|
|
1562
|
+
fetchTimeoutMs: 50,
|
|
1563
|
+
streamIdleTimeoutMs: 10,
|
|
1564
|
+
temperature: 0.2,
|
|
1565
|
+
repeatPenalty: 1.15,
|
|
1566
|
+
reasoning: { mode: "off", budget: null },
|
|
1567
|
+
retryAttempts: 3,
|
|
1568
|
+
source: "provider:test",
|
|
1569
|
+
});
|
|
1570
|
+
const started = Date.now();
|
|
1571
|
+
await assert.rejects(
|
|
1572
|
+
p.generate({ workerId: "r", messages: [] }),
|
|
1573
|
+
(error: ProviderError) => error.kind === "network_failure",
|
|
1574
|
+
);
|
|
1575
|
+
assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
|
|
1576
|
+
assert.equal(calls, 1, "the total deadline expires before another request begins");
|
|
1317
1577
|
mock.restoreAll();
|
|
1318
1578
|
});
|
|
1319
1579
|
|
package/src/AiSdkProvider.ts
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
7
|
// extensions and local endpoint probes the SDK cannot represent.
|
|
8
8
|
|
|
9
|
-
import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
|
+
import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
10
10
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
11
11
|
import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
|
|
12
12
|
import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
|
|
@@ -18,6 +18,7 @@ import { validateGbnf } from "@plurnk/gbnf";
|
|
|
18
18
|
import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
|
|
19
19
|
import { emitWarningOnce } from "./warnings.ts";
|
|
20
20
|
import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
|
|
21
|
+
import { validateAuthoritativeCharge } from "./cost.ts";
|
|
21
22
|
|
|
22
23
|
export type ProviderFetch = typeof globalThis.fetch;
|
|
23
24
|
|
|
@@ -44,6 +45,7 @@ export type AiSdkProviderConfig = {
|
|
|
44
45
|
countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
|
|
45
46
|
calculateCost?: (usage: ProviderUsage) => number; // default () => 0
|
|
46
47
|
calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
48
|
+
normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
47
49
|
source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
|
|
48
50
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
49
51
|
// Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
@@ -157,19 +159,15 @@ type TaggedReasoningProjection = {
|
|
|
157
159
|
readonly contentStart: number;
|
|
158
160
|
};
|
|
159
161
|
|
|
160
|
-
|
|
161
|
-
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
162
|
-
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
163
|
-
const projectTaggedReasoning = (
|
|
162
|
+
const projectLeadingReasoning = (
|
|
164
163
|
content: string,
|
|
165
164
|
structuredReasoning: string,
|
|
166
|
-
|
|
165
|
+
opening: string,
|
|
166
|
+
closing: string,
|
|
167
167
|
): TaggedReasoningProjection => {
|
|
168
|
-
|
|
169
|
-
if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
168
|
+
if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
|
|
170
169
|
return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
171
170
|
}
|
|
172
|
-
const closing = "</think>";
|
|
173
171
|
const closingIndex = content.indexOf(closing, opening.length);
|
|
174
172
|
if (closingIndex === -1) {
|
|
175
173
|
return {
|
|
@@ -188,6 +186,24 @@ const projectTaggedReasoning = (
|
|
|
188
186
|
};
|
|
189
187
|
};
|
|
190
188
|
|
|
189
|
+
// {§provider-tagged-reasoning} Only the model-contract position is structural:
|
|
190
|
+
// one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
|
|
191
|
+
// on one path and leaves later literal tags in the visible suffix untouched.
|
|
192
|
+
const projectTaggedReasoning = (
|
|
193
|
+
content: string,
|
|
194
|
+
structuredReasoning: string,
|
|
195
|
+
style: ReasoningResponseStyle,
|
|
196
|
+
): TaggedReasoningProjection => style === "think-tags"
|
|
197
|
+
? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
|
|
198
|
+
: { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
|
|
199
|
+
|
|
200
|
+
// llama-server's template reasoning parser can project this leading channel out
|
|
201
|
+
// of the OpenAI-compatible response. Grammar evidence needs the sentence before
|
|
202
|
+
// that lossy projection, so constrained template turns request it verbatim and
|
|
203
|
+
// split the observed enclosure here.
|
|
204
|
+
const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
|
|
205
|
+
projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
|
|
206
|
+
|
|
191
207
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
192
208
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
193
209
|
if (budget <= 1000) return "low";
|
|
@@ -243,6 +259,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
243
259
|
#promptTokensUrl: string | undefined;
|
|
244
260
|
#calculateCost: (usage: ProviderUsage) => number;
|
|
245
261
|
#calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
262
|
+
#normalizeCharge?: AuthoritativeChargeNormalizer;
|
|
246
263
|
#source: string;
|
|
247
264
|
#grammarStyle: GrammarStyle;
|
|
248
265
|
#promptCacheKey: boolean;
|
|
@@ -268,7 +285,6 @@ export default class AiSdkProvider implements Provider {
|
|
|
268
285
|
// tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
|
|
269
286
|
// the honest capability signal for every other backend.
|
|
270
287
|
tokenize?: (text: string) => Promise<number[]>;
|
|
271
|
-
|
|
272
288
|
constructor(config: AiSdkProviderConfig) {
|
|
273
289
|
this.#model = config.model;
|
|
274
290
|
this.#url = config.url;
|
|
@@ -307,6 +323,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
307
323
|
this.#promptTokensUrl = config.promptTokensUrl;
|
|
308
324
|
this.#calculateCost = config.calculateCost ?? (() => 0);
|
|
309
325
|
this.#calculateCharge = config.calculateCharge;
|
|
326
|
+
this.#normalizeCharge = config.normalizeCharge;
|
|
310
327
|
this.#source = config.source ?? "provider";
|
|
311
328
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
312
329
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
@@ -428,9 +445,10 @@ export default class AiSdkProvider implements Provider {
|
|
|
428
445
|
?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
|
|
429
446
|
}
|
|
430
447
|
|
|
431
|
-
// Reasoning
|
|
432
|
-
//
|
|
433
|
-
|
|
448
|
+
// Reasoning activation and allowance are independent of grammar transport;
|
|
449
|
+
// only the response representation becomes lossless when evidence is needed.
|
|
450
|
+
// The llama-server template mapping is owned by {§llama-reasoning-request}.
|
|
451
|
+
#reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
|
|
434
452
|
const { mode, budget } = this.#reasoning;
|
|
435
453
|
const on = mode !== "off";
|
|
436
454
|
switch (this.#reasoningStyle) {
|
|
@@ -440,7 +458,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
440
458
|
: mode === "on" ? budget : this.reasoningReserve;
|
|
441
459
|
return {
|
|
442
460
|
chat_template_kwargs: { enable_thinking: on },
|
|
443
|
-
reasoning_format: "auto",
|
|
461
|
+
reasoning_format: preserveGrammarSentence ? "none" : "auto",
|
|
444
462
|
...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
|
|
445
463
|
};
|
|
446
464
|
}
|
|
@@ -621,6 +639,8 @@ export default class AiSdkProvider implements Provider {
|
|
|
621
639
|
const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
|
|
622
640
|
if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
|
|
623
641
|
const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
|
|
642
|
+
const preserveGrammarSentence = wantGrammar
|
|
643
|
+
&& this.#reasoningStyle === "template";
|
|
624
644
|
|
|
625
645
|
// Assembly order = precedence: the family's sampling DEFAULTS
|
|
626
646
|
// (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
|
|
@@ -634,7 +654,7 @@ export default class AiSdkProvider implements Provider {
|
|
|
634
654
|
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
635
655
|
model: this.#model,
|
|
636
656
|
messages,
|
|
637
|
-
...this.#reasoningBody(),
|
|
657
|
+
...this.#reasoningBody(preserveGrammarSentence),
|
|
638
658
|
...this.#grammarBody(sendGrammar),
|
|
639
659
|
...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
|
|
640
660
|
// Request per-token logprobs only when enabled (managed field —
|
|
@@ -649,7 +669,9 @@ export default class AiSdkProvider implements Provider {
|
|
|
649
669
|
|
|
650
670
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
651
671
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
652
|
-
const headers = Object.keys(metaHeaders).length
|
|
672
|
+
const headers = Object.keys(metaHeaders).length === 0
|
|
673
|
+
? this.#headers
|
|
674
|
+
: { ...this.#headers, ...metaHeaders };
|
|
653
675
|
let raw;
|
|
654
676
|
try {
|
|
655
677
|
raw = this.#languageModel === undefined
|
|
@@ -716,45 +738,47 @@ export default class AiSdkProvider implements Provider {
|
|
|
716
738
|
// wire text for forensics.
|
|
717
739
|
if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
|
|
718
740
|
|
|
719
|
-
const
|
|
720
|
-
|
|
721
|
-
raw.
|
|
722
|
-
|
|
723
|
-
|
|
741
|
+
const grammarInput = raw.content;
|
|
742
|
+
const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
|
|
743
|
+
? projectTemplateReasoning(raw.content)
|
|
744
|
+
: projectTaggedReasoning(
|
|
745
|
+
raw.content,
|
|
746
|
+
raw.reasoning,
|
|
747
|
+
this.#reasoningResponseStyle,
|
|
748
|
+
);
|
|
724
749
|
|
|
725
|
-
// Preserve the exact sentence seen at the grammar boundary.
|
|
726
|
-
// `reasoning_format: "
|
|
727
|
-
//
|
|
728
|
-
//
|
|
750
|
+
// Preserve the exact sentence seen at the grammar boundary. Constrained
|
|
751
|
+
// template turns request `reasoning_format: "none"`, so even an empty
|
|
752
|
+
// channel remains observable. An unexpectedly projected response cannot
|
|
753
|
+
// supply independent pre-projection evidence.
|
|
729
754
|
let grammarEvidence: GrammarEvidence | undefined;
|
|
730
755
|
if (wantGrammar) {
|
|
731
|
-
if (
|
|
732
|
-
|
|
733
|
-
input: raw.content,
|
|
734
|
-
contentStart: taggedReasoning.contentStart,
|
|
735
|
-
transported: sendGrammar !== undefined,
|
|
736
|
-
};
|
|
737
|
-
} else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
|
|
738
|
-
if (raw.reasoningProjected) {
|
|
739
|
-
const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
|
|
756
|
+
if (preserveGrammarSentence) {
|
|
757
|
+
if (!raw.reasoningProjected) {
|
|
740
758
|
grammarEvidence = {
|
|
741
|
-
input:
|
|
742
|
-
contentStart:
|
|
759
|
+
input: grammarInput,
|
|
760
|
+
contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
|
|
743
761
|
transported: sendGrammar !== undefined,
|
|
744
762
|
};
|
|
745
763
|
}
|
|
764
|
+
} else if (projectedReasoning.projected) {
|
|
765
|
+
grammarEvidence = {
|
|
766
|
+
input: grammarInput,
|
|
767
|
+
contentStart: projectedReasoning.contentStart,
|
|
768
|
+
transported: sendGrammar !== undefined,
|
|
769
|
+
};
|
|
746
770
|
} else {
|
|
747
771
|
grammarEvidence = {
|
|
748
|
-
input:
|
|
772
|
+
input: grammarInput,
|
|
749
773
|
contentStart: 0,
|
|
750
774
|
transported: sendGrammar !== undefined,
|
|
751
775
|
};
|
|
752
776
|
}
|
|
753
777
|
}
|
|
754
778
|
|
|
755
|
-
if (
|
|
756
|
-
raw.content =
|
|
757
|
-
raw.reasoning =
|
|
779
|
+
if (projectedReasoning.projected) {
|
|
780
|
+
raw.content = projectedReasoning.content;
|
|
781
|
+
raw.reasoning = projectedReasoning.reasoning;
|
|
758
782
|
raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
|
|
759
783
|
}
|
|
760
784
|
|
|
@@ -804,8 +828,13 @@ export default class AiSdkProvider implements Provider {
|
|
|
804
828
|
model: raw.model,
|
|
805
829
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
806
830
|
};
|
|
831
|
+
const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
|
|
832
|
+
const charge = normalizedCharge === undefined
|
|
833
|
+
? undefined
|
|
834
|
+
: validateAuthoritativeCharge(normalizedCharge);
|
|
807
835
|
const evidence = {
|
|
808
836
|
assistantRaw: raw,
|
|
837
|
+
...(charge === undefined ? {} : { charge }),
|
|
809
838
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
810
839
|
...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
|
|
811
840
|
...(meta !== undefined ? { meta } : {}),
|
|
@@ -837,4 +866,5 @@ export default class AiSdkProvider implements Provider {
|
|
|
837
866
|
...evidence,
|
|
838
867
|
};
|
|
839
868
|
}
|
|
869
|
+
|
|
840
870
|
}
|
package/src/Mock.ts
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
// Provider contract. Production providers don't expose the `ops` escape
|
|
6
6
|
// hatch — that's an intg-only convenience.
|
|
7
7
|
|
|
8
|
-
import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
8
|
+
import type { AuthoritativeCharge, ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
9
9
|
import type { ProviderCost } from "@plurnk/plurnk-contracts";
|
|
10
10
|
import { resolveEnvelopeFromEnv } from "./env.ts";
|
|
11
11
|
|
|
@@ -28,6 +28,7 @@ export type MockAssistant = {
|
|
|
28
28
|
export type MockResponse = {
|
|
29
29
|
assistant: MockAssistant;
|
|
30
30
|
assistantRaw?: unknown;
|
|
31
|
+
charge?: AuthoritativeCharge;
|
|
31
32
|
grammarEvidence?: GrammarEvidence;
|
|
32
33
|
};
|
|
33
34
|
|
|
@@ -36,6 +37,7 @@ export type MockReturnedAssistant = ProviderAssistant & { ops?: unknown[] };
|
|
|
36
37
|
export type MockReturnedResponse = ProviderResponse & { assistant: MockReturnedAssistant };
|
|
37
38
|
|
|
38
39
|
const DEFAULT_USAGE: ProviderUsage = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
|
|
40
|
+
type MockGenerateArgs = Omit<Parameters<Provider["generate"]>[0], "workerId"> & { workerId?: string };
|
|
39
41
|
|
|
40
42
|
export default class Mock implements Provider {
|
|
41
43
|
#contextWindow: number | null;
|
|
@@ -79,7 +81,7 @@ export default class Mock implements Provider {
|
|
|
79
81
|
calculateCost(_usage: ProviderUsage): number { return 0; }
|
|
80
82
|
calculateCharge(_usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> { return { kind: "free", source: "mock provider" }; }
|
|
81
83
|
|
|
82
|
-
async generate({ signal, grammar }:
|
|
84
|
+
async generate({ signal, grammar }: MockGenerateArgs): Promise<MockReturnedResponse> {
|
|
83
85
|
// Honor abort before consuming the queue — an aborted call makes no
|
|
84
86
|
// "wire call" and must not exhaust a queued response
|
|
85
87
|
// ({§provider-failure-normalization}).
|
|
@@ -103,6 +105,7 @@ export default class Mock implements Provider {
|
|
|
103
105
|
return {
|
|
104
106
|
assistant,
|
|
105
107
|
assistantRaw: next.assistantRaw ?? null,
|
|
108
|
+
...(next.charge === undefined ? {} : { charge: next.charge }),
|
|
106
109
|
...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
|
|
107
110
|
};
|
|
108
111
|
}
|
package/src/Pool.ts
CHANGED
|
@@ -128,7 +128,7 @@ export default class Pool implements Provider {
|
|
|
128
128
|
calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
|
|
129
129
|
calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
|
|
130
130
|
const backend = this.#backends[0];
|
|
131
|
-
return resolveProviderCost(undefined, backend.calculateCharge?.(usage)
|
|
131
|
+
return resolveProviderCost(undefined, backend.calculateCharge?.(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
|
|
132
132
|
}
|
|
133
133
|
|
|
134
134
|
// --- dispatch ---
|