@plurnk/plurnk-providers 1.4.0 → 1.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. package/.env.defaults +9 -17
  2. package/README.md +3 -0
  3. package/SPEC.md +40 -23
  4. package/dist/AiSdkProvider.d.ts +2 -1
  5. package/dist/AiSdkProvider.d.ts.map +1 -1
  6. package/dist/AiSdkProvider.js +55 -34
  7. package/dist/AiSdkProvider.js.map +1 -1
  8. package/dist/Mock.d.ts +6 -7
  9. package/dist/Mock.d.ts.map +1 -1
  10. package/dist/Mock.js +1 -0
  11. package/dist/Mock.js.map +1 -1
  12. package/dist/Pool.js +1 -1
  13. package/dist/Pool.js.map +1 -1
  14. package/dist/accounting.d.ts +3 -0
  15. package/dist/accounting.d.ts.map +1 -0
  16. package/dist/accounting.js +84 -0
  17. package/dist/accounting.js.map +1 -0
  18. package/dist/aiSdkTransport.d.ts +2 -1
  19. package/dist/aiSdkTransport.d.ts.map +1 -1
  20. package/dist/aiSdkTransport.js +73 -2
  21. package/dist/aiSdkTransport.js.map +1 -1
  22. package/dist/catalogProvider.d.ts +3 -2
  23. package/dist/catalogProvider.d.ts.map +1 -1
  24. package/dist/catalogProvider.js +11 -11
  25. package/dist/catalogProvider.js.map +1 -1
  26. package/dist/cost.d.ts +1 -1
  27. package/dist/cost.d.ts.map +1 -1
  28. package/dist/cost.js +6 -9
  29. package/dist/cost.js.map +1 -1
  30. package/dist/env.d.ts +0 -6
  31. package/dist/env.d.ts.map +1 -1
  32. package/dist/env.js +0 -22
  33. package/dist/env.js.map +1 -1
  34. package/dist/errors.d.ts.map +1 -1
  35. package/dist/errors.js +2 -0
  36. package/dist/errors.js.map +1 -1
  37. package/dist/index.d.ts +1 -1
  38. package/dist/index.d.ts.map +1 -1
  39. package/dist/index.js +1 -1
  40. package/dist/index.js.map +1 -1
  41. package/dist/sdkModels.d.ts +2 -0
  42. package/dist/sdkModels.d.ts.map +1 -1
  43. package/dist/sdkModels.js +13 -3
  44. package/dist/sdkModels.js.map +1 -1
  45. package/dist/types.d.ts +9 -0
  46. package/dist/types.d.ts.map +1 -1
  47. package/dist/usage.d.ts +1 -0
  48. package/dist/usage.d.ts.map +1 -1
  49. package/dist/usage.js +38 -5
  50. package/dist/usage.js.map +1 -1
  51. package/package.json +8 -7
  52. package/src/AiSdkProvider.test.ts +277 -17
  53. package/src/AiSdkProvider.ts +70 -40
  54. package/src/Mock.ts +5 -2
  55. package/src/Pool.ts +1 -1
  56. package/src/accounting.test.ts +58 -0
  57. package/src/accounting.ts +88 -0
  58. package/src/aiSdkTransport.ts +77 -3
  59. package/src/boundaries.test.ts +1 -0
  60. package/src/catalogProvider.test.ts +26 -15
  61. package/src/catalogProvider.ts +12 -11
  62. package/src/cost.test.ts +7 -6
  63. package/src/cost.ts +4 -9
  64. package/src/env.test.ts +1 -26
  65. package/src/env.ts +0 -24
  66. package/src/errors.ts +1 -0
  67. package/src/index.ts +1 -1
  68. package/src/sdkModels.test.ts +22 -3
  69. package/src/sdkModels.ts +15 -3
  70. package/src/types.ts +22 -3
  71. package/src/usage.test.ts +9 -1
  72. package/src/usage.ts +40 -7
@@ -2,6 +2,8 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import AiSdkProvider, { effortFromBudget } from "./AiSdkProvider.ts";
4
4
  import { ProviderError } from "./errors.ts";
5
+ import { authoritativeChargeNormalizer } from "./accounting.ts";
6
+ import type { LanguageModel } from "ai";
5
7
 
6
8
  // Build a fake fetch returning a one-chunk SSE stream, capturing the request
7
9
  // so tests can assert what the spine sent on the wire.
@@ -309,6 +311,139 @@ test("generate maps a streamed response into ProviderResponse", async () => {
309
311
  assert.notEqual(assistantRaw, undefined);
310
312
  });
311
313
 
314
+ test("native SDK accounting metadata becomes a normalized charge in buffered and streamed responses", async (t) => {
315
+ const usage = {
316
+ inputTokens: { total: 2, noCache: 2, cacheRead: 0, cacheWrite: 0 },
317
+ outputTokens: { total: 1, text: 1, reasoning: 0 },
318
+ };
319
+ const providerMetadata = { openrouter: { usage: { cost: 0.00154935 } } };
320
+ const charge = {
321
+ kind: "authoritative",
322
+ amount: { amount: "0.00154935", currency: "USD" },
323
+ usdEquivalent: "0.00154935",
324
+ source: "OpenRouter response usage.cost",
325
+ };
326
+ const languageModel = {
327
+ specificationVersion: "v4",
328
+ provider: "openrouter.chat",
329
+ modelId: "router-test",
330
+ supportedUrls: {},
331
+ doGenerate: async () => ({
332
+ content: [{ type: "text", text: "ok" }],
333
+ finishReason: { unified: "stop", raw: "completed" },
334
+ usage,
335
+ providerMetadata,
336
+ response: { id: "response-buffered", modelId: "router-test" },
337
+ warnings: [],
338
+ }),
339
+ doStream: async () => ({
340
+ stream: new ReadableStream({
341
+ start(controller) {
342
+ controller.enqueue({ type: "stream-start", warnings: [] });
343
+ controller.enqueue({ type: "response-metadata", id: "response-streamed", modelId: "router-test" });
344
+ controller.enqueue({ type: "text-start", id: "text-1" });
345
+ controller.enqueue({ type: "text-delta", id: "text-1", delta: "ok" });
346
+ controller.enqueue({ type: "text-end", id: "text-1" });
347
+ controller.enqueue({
348
+ type: "finish",
349
+ finishReason: { unified: "stop", raw: "completed" },
350
+ usage,
351
+ providerMetadata,
352
+ });
353
+ controller.close();
354
+ },
355
+ }),
356
+ response: {},
357
+ }),
358
+ } as unknown as LanguageModel;
359
+ const config = {
360
+ model: "router-test",
361
+ languageModel,
362
+ fetchTimeoutMs: 5_000,
363
+ temperature: 0.2,
364
+ repeatPenalty: 1.15,
365
+ reasoning: { mode: "off" as const, budget: null },
366
+ retryAttempts: 0,
367
+ normalizeCharge: authoritativeChargeNormalizer("@openrouter/ai-sdk-provider"),
368
+ };
369
+
370
+ await t.test("buffered", async () => {
371
+ const response = await new AiSdkProvider({ ...config, streaming: false })
372
+ .generate({ workerId: "buffered", messages: [] });
373
+ assert.deepEqual(response.charge, charge);
374
+ });
375
+ await t.test("streamed", async () => {
376
+ const response = await new AiSdkProvider(config)
377
+ .generate({ workerId: "streamed", messages: [] });
378
+ assert.deepEqual(response.charge, charge);
379
+ });
380
+ });
381
+
382
+ test("compatible xAI wire usage becomes an exact tick charge without raw-body capture", async () => {
383
+ const p = new AiSdkProvider({
384
+ model: "grok-test",
385
+ url: "http://x/v1/chat/completions",
386
+ fetchTimeoutMs: 5_000,
387
+ temperature: 0.2,
388
+ repeatPenalty: 1.15,
389
+ reasoning: { mode: "off", budget: null },
390
+ retryAttempts: 0,
391
+ streaming: false,
392
+ normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
393
+ });
394
+ installFetchJson({
395
+ id: "response-1",
396
+ model: "grok-test",
397
+ choices: [{ message: { content: "ok" }, finish_reason: "stop" }],
398
+ usage: {
399
+ prompt_tokens: 2,
400
+ completion_tokens: 1,
401
+ total_tokens: 3,
402
+ cost_in_usd_ticks: 15_493_500,
403
+ },
404
+ });
405
+ const response = await p.generate({ workerId: "xai", messages: [] });
406
+ assert.deepEqual(response.charge, {
407
+ kind: "authoritative",
408
+ amount: { amount: "15493500", currency: "USDTICK" },
409
+ usdEquivalent: "0.00154935",
410
+ source: "xAI response usage.cost_in_usd_ticks",
411
+ });
412
+ assert.equal(response.rawBody, undefined);
413
+ });
414
+
415
+ test("streamed xAI final usage retains its exact tick charge", async () => {
416
+ const p = new AiSdkProvider({
417
+ model: "grok-test",
418
+ url: "http://x/v1/chat/completions",
419
+ fetchTimeoutMs: 5_000,
420
+ temperature: 0.2,
421
+ repeatPenalty: 1.15,
422
+ reasoning: { mode: "off", budget: null },
423
+ retryAttempts: 0,
424
+ normalizeCharge: authoritativeChargeNormalizer("@ai-sdk/xai"),
425
+ });
426
+ installFetch([
427
+ { choices: [{ delta: { content: "ok" }, finish_reason: "stop" }] },
428
+ {
429
+ choices: [],
430
+ usage: {
431
+ prompt_tokens: 2,
432
+ completion_tokens: 1,
433
+ total_tokens: 3,
434
+ cost_in_usd_ticks: 15_493_500,
435
+ },
436
+ },
437
+ ]);
438
+ const response = await p.generate({ workerId: "xai", messages: [] });
439
+ assert.deepEqual(response.charge, {
440
+ kind: "authoritative",
441
+ amount: { amount: "15493500", currency: "USDTICK" },
442
+ usdEquivalent: "0.00154935",
443
+ source: "xAI response usage.cost_in_usd_ticks",
444
+ });
445
+ });
446
+
312
447
  test("generate surfaces and normalizes an out-of-set finish_reason", async () => {
313
448
  const warnings: Array<{ message: string; code?: string }> = [];
314
449
  mock.method(process, "emitWarning", (message: string | Error, options?: string | { code?: string }) => {
@@ -852,14 +987,16 @@ test("sampling passthrough guards contract invariants: n/tools/caps stripped, pl
852
987
 
853
988
  test("template reasoning returns the exact pre-projection grammar sentence ({§gbnf-response-observation})", async () => {
854
989
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
855
- const calls = installFetch([{ choices: [{ delta: { reasoning_content: "con🙂sider", content: "x" } }] }]);
856
990
  const grammarInput = "<|channel>thought\ncon🙂sider<channel|>x";
991
+ const calls = installFetch([{ choices: [{ delta: { content: grammarInput } }] }]);
857
992
  const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(grammarInput)}` });
858
993
  const body = JSON.parse(calls[0].init.body as string);
859
994
  assert.deepEqual(body.chat_template_kwargs, { enable_thinking: true });
860
- assert.equal(body.reasoning_format, "auto");
995
+ assert.equal(body.reasoning_format, "none");
861
996
  assert.equal(body.thinking_budget_tokens, 64);
862
997
  assert.equal(body.grammar, `root ::= ${JSON.stringify(grammarInput)}`);
998
+ assert.equal(res.assistant.reasoning, "con🙂sider");
999
+ assert.equal(res.assistant.content, "x");
863
1000
  assert.deepEqual(res.grammarEvidence, {
864
1001
  input: grammarInput,
865
1002
  contentStart: [..."<|channel>thought\ncon🙂sider<channel|>"].length,
@@ -868,13 +1005,48 @@ test("template reasoning returns the exact pre-projection grammar sentence ({§g
868
1005
  assert.equal(res.meta?.railsVerdict, undefined, "the provider represents evidence but does not grade itself");
869
1006
  });
870
1007
 
871
- test("template reasoning does not invent pre-projection evidence when the wire omits its reasoning field", async () => {
1008
+ test("a verbatim template response remains exact evidence when it has no channel envelope", async () => {
872
1009
  const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
873
- installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1010
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1011
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1012
+ const body = JSON.parse(calls[0].init.body as string);
1013
+ assert.equal(body.reasoning_format, "none");
1014
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1015
+ });
1016
+
1017
+ test("a template grammar preserves exact evidence when reasoning is disabled", async () => {
1018
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1019
+ const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
1020
+ const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
1021
+ const body = JSON.parse(calls[0].init.body as string);
1022
+ assert.deepEqual(body.chat_template_kwargs, { enable_thinking: false });
1023
+ assert.equal(body.reasoning_format, "none");
1024
+ assert.deepEqual(res.grammarEvidence, { input: "x", contentStart: 0, transported: true });
1025
+ });
1026
+
1027
+ test("an unexpectedly projected template response cannot claim pre-projection evidence", async () => {
1028
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1029
+ installFetch([{ choices: [{ delta: { reasoning_content: "reason", content: "x" } }] }]);
874
1030
  const res = await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
875
1031
  assert.equal(res.grammarEvidence, undefined);
876
1032
  });
877
1033
 
1034
+ test("template reasoning preserves an empty grammar-required channel as exact evidence", async () => {
1035
+ const p = new AiSdkProvider({ model: "m", url: "http://x/v1/chat/completions", contextWindow: 640, reasoningReserve: { tokens: 64 }, completionReserve: { tokens: 160 }, fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "adaptive", budget: null }, retryAttempts: 0, reasoningStyle: "template", grammarStyle: "llamacpp" });
1036
+ const input = "<|channel>thought\n<channel|>x";
1037
+ const calls = installFetch([{ choices: [{ delta: { content: input } }] }]);
1038
+ const res = await p.generate({ workerId: "r", messages: [], grammar: `root ::= ${JSON.stringify(input)}` });
1039
+ const body = JSON.parse(calls[0].init.body as string);
1040
+ assert.equal(body.reasoning_format, "none");
1041
+ assert.equal(res.assistant.reasoning, null);
1042
+ assert.equal(res.assistant.content, "x");
1043
+ assert.deepEqual(res.grammarEvidence, {
1044
+ input,
1045
+ contentStart: [..."<|channel>thought\n<channel|>"].length,
1046
+ transported: true,
1047
+ });
1048
+ });
1049
+
878
1050
  test("channel-escape detector: billed completion tokens vastly beyond visible channels attach grammar_unenforced", async () => {
879
1051
  // The run105 shape: tiny visible content, no reasoning, thousands billed — the decode
880
1052
  // escaped into a discarded reasoning block, unconstrained.
@@ -1262,6 +1434,15 @@ test("configured headers and url are sent verbatim", async () => {
1262
1434
 
1263
1435
  const retryCfg = { model: "m", url: "http://x/v1/chat/completions", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, reasoning: { mode: "off", budget: null } as const };
1264
1436
 
1437
+ const stalledStreamResponse = (): Response => new Response(new ReadableStream({
1438
+ start(controller) {
1439
+ controller.enqueue(new TextEncoder().encode(
1440
+ 'data: {"id":"stalled","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
1441
+ ));
1442
+ setTimeout(() => controller.close(), 100);
1443
+ },
1444
+ }), { status: 200 });
1445
+
1265
1446
  test("retry: a transient failure retries and a later success resolves", async () => {
1266
1447
  const calls = installFetchScript([
1267
1448
  { status: 429, retryAfter: 0 },
@@ -1274,20 +1455,11 @@ test("retry: a transient failure retries and a later success resolves", async ()
1274
1455
  assert.equal(calls.length, 3); // 429 → 503 → 200
1275
1456
  });
1276
1457
 
1277
- test("streamed-body silence fails the exchange without replaying partial output", async () => {
1458
+ test("streamed-body silence retries and returns the retry's complete output", async () => {
1278
1459
  let calls = 0;
1279
1460
  mock.method(globalThis, "fetch", async () => {
1280
1461
  calls++;
1281
- if (calls === 1) {
1282
- return new Response(new ReadableStream({
1283
- start(controller) {
1284
- controller.enqueue(new TextEncoder().encode(
1285
- 'data: {"id":"first","object":"chat.completion.chunk","created":1,"model":"m","choices":[{"index":0,"delta":{"content":"partial"},"finish_reason":null}]}\n\n',
1286
- ));
1287
- setTimeout(() => controller.close(), 100);
1288
- },
1289
- }), { status: 200 });
1290
- }
1462
+ if (calls === 1) return stalledStreamResponse();
1291
1463
  return new Response(new ReadableStream({
1292
1464
  start(controller) {
1293
1465
  controller.enqueue(new TextEncoder().encode(
@@ -1300,7 +1472,7 @@ test("streamed-body silence fails the exchange without replaying partial output"
1300
1472
  const p = new AiSdkProvider({
1301
1473
  model: "m",
1302
1474
  url: "http://x/v1/chat/completions",
1303
- fetchTimeoutMs: 1000,
1475
+ fetchTimeoutMs: 5000,
1304
1476
  streamIdleTimeoutMs: 10,
1305
1477
  temperature: 0.2,
1306
1478
  repeatPenalty: 1.15,
@@ -1308,12 +1480,100 @@ test("streamed-body silence fails the exchange without replaying partial output"
1308
1480
  retryAttempts: 1,
1309
1481
  source: "provider:test",
1310
1482
  });
1483
+ const result = await p.generate({ workerId: "r", messages: [] });
1484
+ assert.equal(result.assistant.content, "recovered", "the retry's complete output, not the stalled partial");
1485
+ assert.equal(calls, 2, "the stall retried once and the retry succeeded");
1486
+ mock.restoreAll();
1487
+ });
1488
+
1489
+ test("streamed-body silence does not replay when retries are disabled", async () => {
1490
+ let calls = 0;
1491
+ mock.method(globalThis, "fetch", async () => {
1492
+ calls++;
1493
+ return stalledStreamResponse();
1494
+ });
1495
+ const p = new AiSdkProvider({
1496
+ model: "m",
1497
+ url: "http://x/v1/chat/completions",
1498
+ fetchTimeoutMs: 1000,
1499
+ streamIdleTimeoutMs: 10,
1500
+ temperature: 0.2,
1501
+ repeatPenalty: 1.15,
1502
+ reasoning: { mode: "off", budget: null },
1503
+ retryAttempts: 0,
1504
+ source: "provider:test",
1505
+ });
1311
1506
  await assert.rejects(
1312
1507
  p.generate({ workerId: "r", messages: [] }),
1313
1508
  (error: ProviderError) => error.kind === "network_failure"
1314
1509
  && /chunk timeout/i.test(error.message),
1315
1510
  );
1316
- assert.equal(calls, 1);
1511
+ assert.equal(calls, 1, "zero retries permits exactly one provider request");
1512
+ mock.restoreAll();
1513
+ });
1514
+
1515
+ test("streamed-body silence exhausts the configured retry budget once", async () => {
1516
+ let calls = 0;
1517
+ mock.method(globalThis, "fetch", async () => {
1518
+ calls++;
1519
+ return stalledStreamResponse();
1520
+ });
1521
+ const p = new AiSdkProvider({
1522
+ model: "m",
1523
+ url: "http://x/v1/chat/completions",
1524
+ fetchTimeoutMs: 5000,
1525
+ streamIdleTimeoutMs: 10,
1526
+ temperature: 0.2,
1527
+ repeatPenalty: 1.15,
1528
+ reasoning: { mode: "off", budget: null },
1529
+ retryAttempts: 1,
1530
+ source: "provider:test",
1531
+ });
1532
+ await assert.rejects(
1533
+ p.generate({ workerId: "r", messages: [] }),
1534
+ (error: ProviderError) => error.kind === "network_failure"
1535
+ && error.problem.attempts === 2
1536
+ && error.problem.retryExhausted === true
1537
+ && error.problem.retryable === false,
1538
+ );
1539
+ assert.equal(calls, 2, "one configured retry permits exactly two provider requests");
1540
+ mock.restoreAll();
1541
+ });
1542
+
1543
+ test("the total generation deadline spans stalled-stream retry scheduling", async () => {
1544
+ let calls = 0;
1545
+ mock.method(globalThis, "fetch", async () => {
1546
+ calls++;
1547
+ if (calls > 1) {
1548
+ return new Response(new ReadableStream({
1549
+ start(controller) {
1550
+ controller.enqueue(new TextEncoder().encode(
1551
+ 'data: {"id":"second","object":"chat.completion.chunk","created":2,"model":"m","choices":[{"index":0,"delta":{"content":"late"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n',
1552
+ ));
1553
+ controller.close();
1554
+ },
1555
+ }), { status: 200 });
1556
+ }
1557
+ return stalledStreamResponse();
1558
+ });
1559
+ const p = new AiSdkProvider({
1560
+ model: "m",
1561
+ url: "http://x/v1/chat/completions",
1562
+ fetchTimeoutMs: 50,
1563
+ streamIdleTimeoutMs: 10,
1564
+ temperature: 0.2,
1565
+ repeatPenalty: 1.15,
1566
+ reasoning: { mode: "off", budget: null },
1567
+ retryAttempts: 3,
1568
+ source: "provider:test",
1569
+ });
1570
+ const started = Date.now();
1571
+ await assert.rejects(
1572
+ p.generate({ workerId: "r", messages: [] }),
1573
+ (error: ProviderError) => error.kind === "network_failure",
1574
+ );
1575
+ assert.ok(Date.now() - started < 500, "the configured total deadline ends retry scheduling");
1576
+ assert.equal(calls, 1, "the total deadline expires before another request begins");
1317
1577
  mock.restoreAll();
1318
1578
  });
1319
1579
 
@@ -6,7 +6,7 @@
6
6
  // ordinary vendor protocol. The compatible URL path remains only for PLURNK
7
7
  // extensions and local endpoint probes the SDK cannot represent.
8
8
 
9
- import type { ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
9
+ import type { AuthoritativeChargeNormalizer, ChatMessage, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
10
10
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
11
11
  import type { Reasoning, ReasoningResponseStyle, ReserveSpec } from "./env.ts";
12
12
  import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
@@ -18,6 +18,7 @@ import { validateGbnf } from "@plurnk/gbnf";
18
18
  import { assertPromptTokenMeasurement, estimatePromptTokens } from "./promptTokens.ts";
19
19
  import { emitWarningOnce } from "./warnings.ts";
20
20
  import type { PluginAttribution, PluginAttributionContext } from "@plurnk/plurnk-meta";
21
+ import { validateAuthoritativeCharge } from "./cost.ts";
21
22
 
22
23
  export type ProviderFetch = typeof globalThis.fetch;
23
24
 
@@ -44,6 +45,7 @@ export type AiSdkProviderConfig = {
44
45
  countPromptTokens?: (messages: readonly ChatMessage[], signal?: AbortSignal) => PromptTokenMeasurement | Promise<PromptTokenMeasurement>;
45
46
  calculateCost?: (usage: ProviderUsage) => number; // default () => 0
46
47
  calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
48
+ normalizeCharge?: AuthoritativeChargeNormalizer;
47
49
  source?: string; // notice/problem source, e.g. "provider:openai"; default "provider"
48
50
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
49
51
  // Send the OpenAI-standard `prompt_cache_key` set to workerId, so a
@@ -157,19 +159,15 @@ type TaggedReasoningProjection = {
157
159
  readonly contentStart: number;
158
160
  };
159
161
 
160
- // {§provider-tagged-reasoning} Only the model-contract position is structural:
161
- // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
162
- // on one path and leaves later literal tags in the visible suffix untouched.
163
- const projectTaggedReasoning = (
162
+ const projectLeadingReasoning = (
164
163
  content: string,
165
164
  structuredReasoning: string,
166
- style: ReasoningResponseStyle,
165
+ opening: string,
166
+ closing: string,
167
167
  ): TaggedReasoningProjection => {
168
- const opening = "<think>";
169
- if (style !== "think-tags" || structuredReasoning.length > 0 || !content.startsWith(opening)) {
168
+ if (structuredReasoning.length > 0 || !content.startsWith(opening)) {
170
169
  return { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
171
170
  }
172
- const closing = "</think>";
173
171
  const closingIndex = content.indexOf(closing, opening.length);
174
172
  if (closingIndex === -1) {
175
173
  return {
@@ -188,6 +186,24 @@ const projectTaggedReasoning = (
188
186
  };
189
187
  };
190
188
 
189
+ // {§provider-tagged-reasoning} Only the model-contract position is structural:
190
+ // one exact leading envelope. Parsing after stream assembly keeps SSE and JSON
191
+ // on one path and leaves later literal tags in the visible suffix untouched.
192
+ const projectTaggedReasoning = (
193
+ content: string,
194
+ structuredReasoning: string,
195
+ style: ReasoningResponseStyle,
196
+ ): TaggedReasoningProjection => style === "think-tags"
197
+ ? projectLeadingReasoning(content, structuredReasoning, "<think>", "</think>")
198
+ : { content, reasoning: structuredReasoning, projected: false, contentStart: 0 };
199
+
200
+ // llama-server's template reasoning parser can project this leading channel out
201
+ // of the OpenAI-compatible response. Grammar evidence needs the sentence before
202
+ // that lossy projection, so constrained template turns request it verbatim and
203
+ // split the observed enclosure here.
204
+ const projectTemplateReasoning = (content: string): TaggedReasoningProjection =>
205
+ projectLeadingReasoning(content, "", "<|channel>thought\n", "<channel|>");
206
+
191
207
  // Shared budget→effort breakpoints (xai and google had identical copies).
192
208
  export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
193
209
  if (budget <= 1000) return "low";
@@ -243,6 +259,7 @@ export default class AiSdkProvider implements Provider {
243
259
  #promptTokensUrl: string | undefined;
244
260
  #calculateCost: (usage: ProviderUsage) => number;
245
261
  #calculateCharge?: (usage: ProviderUsage) => Exclude<ProviderCost, { kind: "authoritative" }>;
262
+ #normalizeCharge?: AuthoritativeChargeNormalizer;
246
263
  #source: string;
247
264
  #grammarStyle: GrammarStyle;
248
265
  #promptCacheKey: boolean;
@@ -268,7 +285,6 @@ export default class AiSdkProvider implements Provider {
268
285
  // tokenizeUrl (llama-server), so `provider.tokenize === undefined` remains
269
286
  // the honest capability signal for every other backend.
270
287
  tokenize?: (text: string) => Promise<number[]>;
271
-
272
288
  constructor(config: AiSdkProviderConfig) {
273
289
  this.#model = config.model;
274
290
  this.#url = config.url;
@@ -307,6 +323,7 @@ export default class AiSdkProvider implements Provider {
307
323
  this.#promptTokensUrl = config.promptTokensUrl;
308
324
  this.#calculateCost = config.calculateCost ?? (() => 0);
309
325
  this.#calculateCharge = config.calculateCharge;
326
+ this.#normalizeCharge = config.normalizeCharge;
310
327
  this.#source = config.source ?? "provider";
311
328
  this.#grammarStyle = config.grammarStyle ?? "none";
312
329
  this.#promptCacheKey = config.promptCacheKey ?? false;
@@ -428,9 +445,10 @@ export default class AiSdkProvider implements Provider {
428
445
  ?? { kind: "unknown", reason: "no provider rate or settled charge is available" };
429
446
  }
430
447
 
431
- // Reasoning intent maps independently of grammar transport. The llama-server
432
- // template mapping is owned by {§llama-reasoning-request}.
433
- #reasoningBody(): Record<string, unknown> {
448
+ // Reasoning activation and allowance are independent of grammar transport;
449
+ // only the response representation becomes lossless when evidence is needed.
450
+ // The llama-server template mapping is owned by {§llama-reasoning-request}.
451
+ #reasoningBody(preserveGrammarSentence = false): Record<string, unknown> {
434
452
  const { mode, budget } = this.#reasoning;
435
453
  const on = mode !== "off";
436
454
  switch (this.#reasoningStyle) {
@@ -440,7 +458,7 @@ export default class AiSdkProvider implements Provider {
440
458
  : mode === "on" ? budget : this.reasoningReserve;
441
459
  return {
442
460
  chat_template_kwargs: { enable_thinking: on },
443
- reasoning_format: "auto",
461
+ reasoning_format: preserveGrammarSentence ? "none" : "auto",
444
462
  ...(allowance === null ? {} : { thinking_budget_tokens: allowance }),
445
463
  };
446
464
  }
@@ -621,6 +639,8 @@ export default class AiSdkProvider implements Provider {
621
639
  const wantGrammar = grammar !== undefined && this.#grammarStyle !== "none";
622
640
  if (wantGrammar && this.#gbnfDebug) this.#assertGrammarValid(grammar!);
623
641
  const sendGrammar = wantGrammar && !this.#gbnfDebug ? grammar : undefined;
642
+ const preserveGrammarSentence = wantGrammar
643
+ && this.#reasoningStyle === "template";
624
644
 
625
645
  // Assembly order = precedence: the family's sampling DEFAULTS
626
646
  // (PLURNK_PROVIDERS_TEMPERATURE — universal, measured on grammar
@@ -634,7 +654,7 @@ export default class AiSdkProvider implements Provider {
634
654
  ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
635
655
  model: this.#model,
636
656
  messages,
637
- ...this.#reasoningBody(),
657
+ ...this.#reasoningBody(preserveGrammarSentence),
638
658
  ...this.#grammarBody(sendGrammar),
639
659
  ...(maxTokens !== undefined ? { max_tokens: maxTokens } : {}),
640
660
  // Request per-token logprobs only when enabled (managed field —
@@ -649,7 +669,9 @@ export default class AiSdkProvider implements Provider {
649
669
 
650
670
  // Per-request headers = static auth/routing + any first-party telemetry.
651
671
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
652
- const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
672
+ const headers = Object.keys(metaHeaders).length === 0
673
+ ? this.#headers
674
+ : { ...this.#headers, ...metaHeaders };
653
675
  let raw;
654
676
  try {
655
677
  raw = this.#languageModel === undefined
@@ -716,45 +738,47 @@ export default class AiSdkProvider implements Provider {
716
738
  // wire text for forensics.
717
739
  if (this.#eosText !== undefined) raw.content = stripTrailingSpecial(raw.content, this.#eosText);
718
740
 
719
- const taggedReasoning = projectTaggedReasoning(
720
- raw.content,
721
- raw.reasoning,
722
- this.#reasoningResponseStyle,
723
- );
741
+ const grammarInput = raw.content;
742
+ const projectedReasoning = preserveGrammarSentence && !raw.reasoningProjected
743
+ ? projectTemplateReasoning(raw.content)
744
+ : projectTaggedReasoning(
745
+ raw.content,
746
+ raw.reasoning,
747
+ this.#reasoningResponseStyle,
748
+ );
724
749
 
725
- // Preserve the exact sentence seen at the grammar boundary. llama-server's
726
- // `reasoning_format: "auto"` projects one raw Harmony enclosure into the
727
- // reasoning/content fields; the wire field's presence is the proof that the
728
- // projection occurred. The provider represents this evidence and never grades it.
750
+ // Preserve the exact sentence seen at the grammar boundary. Constrained
751
+ // template turns request `reasoning_format: "none"`, so even an empty
752
+ // channel remains observable. An unexpectedly projected response cannot
753
+ // supply independent pre-projection evidence.
729
754
  let grammarEvidence: GrammarEvidence | undefined;
730
755
  if (wantGrammar) {
731
- if (taggedReasoning.projected) {
732
- grammarEvidence = {
733
- input: raw.content,
734
- contentStart: taggedReasoning.contentStart,
735
- transported: sendGrammar !== undefined,
736
- };
737
- } else if (this.#reasoningStyle === "template" && this.#reasoning.mode !== "off") {
738
- if (raw.reasoningProjected) {
739
- const prefix = `<|channel>thought\n${raw.reasoning}<channel|>`;
756
+ if (preserveGrammarSentence) {
757
+ if (!raw.reasoningProjected) {
740
758
  grammarEvidence = {
741
- input: `${prefix}${raw.content}`,
742
- contentStart: [...prefix].length,
759
+ input: grammarInput,
760
+ contentStart: projectedReasoning.projected ? projectedReasoning.contentStart : 0,
743
761
  transported: sendGrammar !== undefined,
744
762
  };
745
763
  }
764
+ } else if (projectedReasoning.projected) {
765
+ grammarEvidence = {
766
+ input: grammarInput,
767
+ contentStart: projectedReasoning.contentStart,
768
+ transported: sendGrammar !== undefined,
769
+ };
746
770
  } else {
747
771
  grammarEvidence = {
748
- input: raw.content,
772
+ input: grammarInput,
749
773
  contentStart: 0,
750
774
  transported: sendGrammar !== undefined,
751
775
  };
752
776
  }
753
777
  }
754
778
 
755
- if (taggedReasoning.projected) {
756
- raw.content = taggedReasoning.content;
757
- raw.reasoning = taggedReasoning.reasoning;
779
+ if (projectedReasoning.projected) {
780
+ raw.content = projectedReasoning.content;
781
+ raw.reasoning = projectedReasoning.reasoning;
758
782
  raw.usage = attributeUnitemizedReasoning(raw.usage, raw.reasoning, raw.content);
759
783
  }
760
784
 
@@ -804,8 +828,13 @@ export default class AiSdkProvider implements Provider {
804
828
  model: raw.model,
805
829
  ...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
806
830
  };
831
+ const normalizedCharge = this.#normalizeCharge?.(raw.chargeEvidence);
832
+ const charge = normalizedCharge === undefined
833
+ ? undefined
834
+ : validateAuthoritativeCharge(normalizedCharge);
807
835
  const evidence = {
808
836
  assistantRaw: raw,
837
+ ...(charge === undefined ? {} : { charge }),
809
838
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
810
839
  ...(raw.rawBody !== undefined ? { rawBody: raw.rawBody } : {}),
811
840
  ...(meta !== undefined ? { meta } : {}),
@@ -837,4 +866,5 @@ export default class AiSdkProvider implements Provider {
837
866
  ...evidence,
838
867
  };
839
868
  }
869
+
840
870
  }
package/src/Mock.ts CHANGED
@@ -5,7 +5,7 @@
5
5
  // Provider contract. Production providers don't expose the `ops` escape
6
6
  // hatch — that's an intg-only convenience.
7
7
 
8
- import type { ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
8
+ import type { AuthoritativeCharge, ChatMessage, FinishReason, GrammarEvidence, PromptTokenMeasurement, Provider, ProviderAssistant, ProviderEncryptedReasoningItem, ProviderResponse, ProviderUsage } from "./types.ts";
9
9
  import type { ProviderCost } from "@plurnk/plurnk-contracts";
10
10
  import { resolveEnvelopeFromEnv } from "./env.ts";
11
11
 
@@ -28,6 +28,7 @@ export type MockAssistant = {
28
28
  export type MockResponse = {
29
29
  assistant: MockAssistant;
30
30
  assistantRaw?: unknown;
31
+ charge?: AuthoritativeCharge;
31
32
  grammarEvidence?: GrammarEvidence;
32
33
  };
33
34
 
@@ -36,6 +37,7 @@ export type MockReturnedAssistant = ProviderAssistant & { ops?: unknown[] };
36
37
  export type MockReturnedResponse = ProviderResponse & { assistant: MockReturnedAssistant };
37
38
 
38
39
  const DEFAULT_USAGE: ProviderUsage = { prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 };
40
+ type MockGenerateArgs = Omit<Parameters<Provider["generate"]>[0], "workerId"> & { workerId?: string };
39
41
 
40
42
  export default class Mock implements Provider {
41
43
  #contextWindow: number | null;
@@ -79,7 +81,7 @@ export default class Mock implements Provider {
79
81
  calculateCost(_usage: ProviderUsage): number { return 0; }
80
82
  calculateCharge(_usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> { return { kind: "free", source: "mock provider" }; }
81
83
 
82
- async generate({ signal, grammar }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal; grammar?: string }): Promise<MockReturnedResponse> {
84
+ async generate({ signal, grammar }: MockGenerateArgs): Promise<MockReturnedResponse> {
83
85
  // Honor abort before consuming the queue — an aborted call makes no
84
86
  // "wire call" and must not exhaust a queued response
85
87
  // ({§provider-failure-normalization}).
@@ -103,6 +105,7 @@ export default class Mock implements Provider {
103
105
  return {
104
106
  assistant,
105
107
  assistantRaw: next.assistantRaw ?? null,
108
+ ...(next.charge === undefined ? {} : { charge: next.charge }),
106
109
  ...(grammarEvidence !== undefined ? { grammarEvidence } : {}),
107
110
  };
108
111
  }
package/src/Pool.ts CHANGED
@@ -128,7 +128,7 @@ export default class Pool implements Provider {
128
128
  calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
129
129
  calculateCharge(usage: ProviderUsage): Exclude<ProviderCost, { kind: "authoritative" }> {
130
130
  const backend = this.#backends[0];
131
- return resolveProviderCost(undefined, backend.calculateCharge?.(usage), () => backend.calculateCost(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
131
+ return resolveProviderCost(undefined, backend.calculateCharge?.(usage)) as Exclude<ProviderCost, { kind: "authoritative" }>;
132
132
  }
133
133
 
134
134
  // --- dispatch ---