@plurnk/plurnk-providers 1.3.3 → 1.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.env.defaults +20 -21
  2. package/SPEC.md +65 -44
  3. package/dist/Mock.d.ts +1 -1
  4. package/dist/Mock.d.ts.map +1 -1
  5. package/dist/Mock.js +1 -1
  6. package/dist/Mock.js.map +1 -1
  7. package/dist/OpenAICompat.d.ts +5 -4
  8. package/dist/OpenAICompat.d.ts.map +1 -1
  9. package/dist/OpenAICompat.js +38 -38
  10. package/dist/OpenAICompat.js.map +1 -1
  11. package/dist/Pool.d.ts +1 -1
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +1 -1
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/env.d.ts.map +1 -1
  16. package/dist/env.js +2 -0
  17. package/dist/env.js.map +1 -1
  18. package/dist/index.d.ts +2 -2
  19. package/dist/index.d.ts.map +1 -1
  20. package/dist/index.js +2 -2
  21. package/dist/index.js.map +1 -1
  22. package/dist/openai.d.ts +1 -1
  23. package/dist/openai.d.ts.map +1 -1
  24. package/dist/openai.js +1 -1
  25. package/dist/openai.js.map +1 -1
  26. package/dist/openaiStream.d.ts +6 -1
  27. package/dist/openaiStream.d.ts.map +1 -1
  28. package/dist/openaiStream.js +30 -2
  29. package/dist/openaiStream.js.map +1 -1
  30. package/dist/standardProviders.d.ts +3 -3
  31. package/dist/standardProviders.d.ts.map +1 -1
  32. package/dist/standardProviders.js +36 -17
  33. package/dist/standardProviders.js.map +1 -1
  34. package/dist/types.d.ts +1 -1
  35. package/dist/types.d.ts.map +1 -1
  36. package/dist/usage.d.ts +1 -1
  37. package/dist/usage.d.ts.map +1 -1
  38. package/dist/usage.js +4 -2
  39. package/dist/usage.js.map +1 -1
  40. package/package.json +10 -7
  41. package/src/Mock.test.ts +2 -2
  42. package/src/Mock.ts +1 -1
  43. package/src/OpenAICompat.test.ts +86 -86
  44. package/src/OpenAICompat.ts +46 -45
  45. package/src/Pool.test.ts +3 -3
  46. package/src/Pool.ts +1 -1
  47. package/src/ProviderRegistry.test.ts +30 -1
  48. package/src/aiSdkAdapter.spike.test.ts +242 -0
  49. package/src/env.test.ts +8 -0
  50. package/src/env.ts +2 -0
  51. package/src/index.ts +2 -2
  52. package/src/openai.ts +1 -1
  53. package/src/openaiStream.ts +30 -2
  54. package/src/standardProviders.test.ts +45 -31
  55. package/src/standardProviders.ts +42 -29
  56. package/src/types.ts +5 -5
  57. package/src/usage.test.ts +8 -10
  58. package/src/usage.ts +7 -3
@@ -25,8 +25,7 @@ const installFetch = (chunks: unknown[]) => {
25
25
  return calls;
26
26
  };
27
27
 
28
- // Fake fetch returning one non-streamed JSON body — for the paths the spine
29
- // demotes off SSE (a response_format grammar). Captures the request the same way.
28
+ // Fake fetch returning one non-streamed JSON body. Captures the request too.
30
29
  const installFetchJson = (payload: unknown) => {
31
30
  const calls: { url: string; init: RequestInit }[] = [];
32
31
  mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
@@ -235,17 +234,17 @@ test("identity getters and defaults", () => {
235
234
  assert.equal(p.contextWindow, null); // default
236
235
  assert.equal(p.countTokens(""), 0);
237
236
  assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
238
- assert.equal(p.costFor({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
237
+ assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
239
238
  });
240
239
 
241
- test("injected countTokens and costFor are used", () => {
240
+ test("injected countTokens and calculateCost are used", () => {
242
241
  const p = new OpenAICompatProvider({
243
242
  model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
244
243
  countTokens: (t) => t.length,
245
- costFor: (u) => u.total * 2,
244
+ calculateCost: (u) => u.total * 2,
246
245
  });
247
246
  assert.equal(p.countTokens("abc"), 3);
248
- assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
247
+ assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
249
248
  });
250
249
 
251
250
  test("generate maps a streamed response into ProviderResponse", async () => {
@@ -371,15 +370,14 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
371
370
  });
372
371
 
373
372
  test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
374
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
375
- // default rides with the grammar (non-streamed demotion path)
376
- let calls = installFetchJson(jsonChoice);
377
- await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
373
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
374
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
375
+ await p.generate({ workerId: "r", messages: [] });
378
376
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
379
377
  mock.restoreAll();
380
378
  // explicit caller sampling wins over the default
381
- calls = installFetchJson(jsonChoice);
382
- await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"', sampling: { temperature: 0.7 } });
379
+ calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
380
+ await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.7 } });
383
381
  assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.7);
384
382
  mock.restoreAll();
385
383
  // temperature is now the UNIVERSAL default: present without a grammar too
@@ -419,24 +417,6 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
419
417
  mock.restoreAll();
420
418
  });
421
419
 
422
- test("effort_explicit: intent maps IDENTICALLY with and without a grammar — the #32 clamp is lifted (reasoning+rails coexist)", async () => {
423
- const warned: Array<string | Error> = [];
424
- mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
425
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format", source: "provider:test" });
426
- // grammar transported → intent STILL flows through (canary-verified: the mask
427
- // covers only content; clamping to "none" was the plan-less #331 regression)
428
- let calls = installFetchJson(jsonChoice);
429
- await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
430
- assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
431
- assert.equal(warned.filter((w) => String(w).includes("clamped")).length, 0, "no clamp warning — the clamp is gone");
432
- mock.restoreAll();
433
- // no grammar → same mapping
434
- const p2 = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format" });
435
- calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
436
- await p2.generate({ workerId: "r", messages: [] });
437
- assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
438
- });
439
-
440
420
  test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
441
421
  const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
442
422
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
@@ -447,15 +427,9 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
447
427
  });
448
428
 
449
429
  test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
450
- // response_format cloud (fireworks) with NO grammar - the firefast case that went out bare
451
- const fw = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
452
- let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
453
- await fw.generate({ workerId: "r", messages: [] });
454
- assert.equal(JSON.parse(calls[0].init.body as string).repetition_penalty, 1.15);
455
- mock.restoreAll();
456
- // llama.cpp with NO grammar carries its key too (unconstrained local is guarded now)
430
+ // llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
457
431
  const llama = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
458
- calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
432
+ let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
459
433
  await llama.generate({ workerId: "r", messages: [] });
460
434
  assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
461
435
  mock.restoreAll();
@@ -609,32 +583,6 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
609
583
  assert.equal("response_format" in body, false);
610
584
  });
611
585
 
612
- test("grammar transport 'response_format': response_format.grammar, no top-level grammar (Fireworks)", async () => {
613
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
614
- const calls = installFetchJson(jsonChoice); // response_format grammar demotes off SSE
615
- await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
616
- const body = JSON.parse(calls[0].init.body as string);
617
- assert.deepEqual(body.response_format, { type: "grammar", grammar: 'root ::= "x"' });
618
- assert.equal("grammar" in body, false); // not the llama.cpp shape
619
- assert.equal("repeat_penalty" in body, false); // llama.cpp spelling not used here
620
- assert.equal(body.repetition_penalty, 1.15); // the floor still rides (OpenAI-compat spelling, #20)
621
- });
622
-
623
- // A response_format grammar is the one case the spine drops streaming for, even
624
- // with streaming on (default): fireworks mislabels the streamed grammar output
625
- // as reasoning_content but returns it as content non-streamed (§13). The demotion
626
- // is per-request — a grammarless call on the same provider still streams.
627
- test("response_format grammar demotes THIS request off SSE; grammarless calls still stream", async () => {
628
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
629
- const jsonCalls = installFetchJson(jsonChoice);
630
- await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
631
- assert.equal("stream" in JSON.parse(jsonCalls[0].init.body as string), false); // no SSE flag
632
- mock.restoreAll();
633
- const sseCalls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
634
- await p.generate({ workerId: "r", messages: [] }); // no grammar → streams
635
- assert.equal(JSON.parse(sseCalls[0].init.body as string).stream, true); // SSE flag present
636
- });
637
-
638
586
  test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
639
587
  const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
640
588
  const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
@@ -756,31 +704,15 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
756
704
  assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
757
705
  });
758
706
 
759
- // — meta bag: pass-through extras + validated known keys (#23) —
707
+ // — meta bag: verbatim provider metadata (#23) —
760
708
 
761
- test("meta: the spec's balance field is normalized to a validated meta.balancePico", async () => {
762
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
763
- installFetchJson({ ...jsonChoice, balance_pico: 4_200_000 });
764
- const res = await p.generate({ workerId: "r", messages: [] });
765
- assert.equal(res.meta?.balancePico, 4_200_000);
766
- assert.equal("balance_pico" in (res.meta ?? {}), false); // raw key renamed to the canonical balancePico
767
- });
768
-
769
- test("meta: passes the backend's extra top-level fields through verbatim (every provider)", async () => {
770
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false }); // no balanceMetaKey
771
- installFetchJson({ ...jsonChoice, balance_pico: 4_200_000, system_fingerprint: "fp_abc" });
709
+ test("meta: passes backend fields through without reinterpreting monetary values", async () => {
710
+ const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
711
+ const balance = { amount: "0.0000042", currency: "XMR" };
712
+ installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
772
713
  const res = await p.generate({ workerId: "r", messages: [] });
773
- assert.equal(res.meta?.balance_pico, 4_200_000); // passed through raw — no balance contract on this provider
714
+ assert.deepEqual(res.meta?.balance, balance);
774
715
  assert.equal(res.meta?.system_fingerprint, "fp_abc");
775
- assert.equal("balancePico" in (res.meta ?? {}), false); // not normalized without the key
776
- });
777
-
778
- test("meta: a non-numeric balance is dropped, never surfaced as balancePico (null-honest)", async () => {
779
- const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
780
- installFetchJson({ ...jsonChoice, balance_pico: "lots" });
781
- const res = await p.generate({ workerId: "r", messages: [] });
782
- assert.equal("balancePico" in (res.meta ?? {}), false);
783
- assert.equal("balance_pico" in (res.meta ?? {}), false); // raw dropped too — the known key is validated away
784
716
  });
785
717
 
786
718
  // — first-party telemetry headers (attribution + client, SPEC §5) —
@@ -972,6 +904,74 @@ test("retry: a transient failure retries and a later success resolves", async ()
972
904
  assert.equal(calls.length, 3); // 429 → 503 → 200
973
905
  });
974
906
 
907
+ test("#559: streamed-body silence retries and records the recovery in durable response metadata", async () => {
908
+ let calls = 0;
909
+ mock.method(globalThis, "fetch", async () => {
910
+ calls++;
911
+ if (calls === 1) {
912
+ return new Response(new ReadableStream({
913
+ start(controller) {
914
+ controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"partial"}}]}\n\n'));
915
+ },
916
+ }), { status: 200 });
917
+ }
918
+ return new Response(new ReadableStream({
919
+ start(controller) {
920
+ controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
921
+ controller.close();
922
+ },
923
+ }), { status: 200 });
924
+ });
925
+ const p = new OpenAICompatProvider({
926
+ model: "m",
927
+ url: "http://x",
928
+ fetchTimeoutMs: 1000,
929
+ streamIdleTimeoutMs: 10,
930
+ temperature: 0.2,
931
+ repeatPenalty: 1.15,
932
+ retryDelayMs: 1,
933
+ reasoning: { mode: "off", budget: null },
934
+ retryAttempts: 1,
935
+ source: "provider:test",
936
+ });
937
+ const result = await p.generate({ workerId: "r", messages: [] });
938
+ assert.equal(result.assistant.content, "recovered");
939
+ assert.equal(calls, 2);
940
+ const retries = result.meta?.transportRetries as Array<Record<string, unknown>>;
941
+ assert.equal(retries.length, 1);
942
+ assert.equal(retries[0].attempt, 1);
943
+ assert.equal(retries[0].kind, "network_failure");
944
+ assert.equal(typeof retries[0].elapsedMs, "number");
945
+ assert.match(String(retries[0].message), /no body bytes for 10ms/);
946
+ mock.restoreAll();
947
+ });
948
+
949
+ test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
950
+ mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
951
+ async start(controller) {
952
+ controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
953
+ await new Promise((resolve) => setTimeout(resolve, 20));
954
+ controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"is valid"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
955
+ controller.close();
956
+ },
957
+ }), { status: 200 }));
958
+ const p = new OpenAICompatProvider({
959
+ model: "m",
960
+ url: "http://x",
961
+ fetchTimeoutMs: 1000,
962
+ streamIdleTimeoutMs: 0,
963
+ temperature: 0.2,
964
+ repeatPenalty: 1.15,
965
+ retryDelayMs: 1,
966
+ reasoning: { mode: "off", budget: null },
967
+ retryAttempts: 0,
968
+ });
969
+ const result = await p.generate({ workerId: "r", messages: [] });
970
+ assert.equal(result.assistant.content, "slow is valid");
971
+ assert.equal(result.meta?.transportRetries, undefined);
972
+ mock.restoreAll();
973
+ });
974
+
975
975
  test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
976
976
  const { ProviderError } = await import("./telemetry.ts");
977
977
  const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
@@ -29,23 +29,21 @@ import { emitWarningOnce } from "./warnings.ts";
29
29
  // (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
30
30
  export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
31
31
 
32
- // How a caller-supplied GBNF grammar is carried on the wire — backends accept
33
- // different shapes for the SAME GBNF (probed/configured, never guessed; §13); the
34
- // wire shape per style lives in #grammarBody. "none" means the grammar is NOT sent
35
- // (never silently — so a constrained consumer can't mistake unconstrained output
36
- // for enforced).
37
- export type GrammarStyle = "none" | "llamacpp" | "response_format";
32
+ // GBNF transport is a local llama-server capability. "none" means no
33
+ // service-managed constrained sampling; endpoint-owned settings are not inferred.
34
+ export type GrammarStyle = "none" | "llamacpp";
38
35
 
39
36
  export type OpenAICompatConfig = {
40
37
  model: string;
41
38
  url: string; // fully-resolved chat-completions URL
42
39
  fetchTimeoutMs: number;
40
+ streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
43
41
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
44
42
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
45
43
  contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
46
44
  reasoningStyle?: ReasoningStyle; // default "none"
47
45
  countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
48
- costFor?: (usage: ProviderUsage) => number; // default () => 0
46
+ calculateCost?: (usage: ProviderUsage) => number; // default () => 0
49
47
  source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
50
48
  grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
51
49
  // #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
@@ -54,12 +52,14 @@ export type OpenAICompatConfig = {
54
52
  // false -- a backend that strict-validates unknown fields 400s, so enable only
55
53
  // where the field is accepted. Same identity that already drives slot affinity.
56
54
  promptCacheKey?: boolean;
55
+ // Optional provider-configured service tier. Unlike caller sampling, this is
56
+ // a fixed deployment choice and therefore wins on every request.
57
+ serviceTier?: string;
57
58
  gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
58
59
  streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
59
60
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
60
61
  apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
61
62
  eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
62
- balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
63
63
  // Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
64
64
  supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
65
65
  slotCount?: number | null; // probed slot count for pinning backends; default null
@@ -238,6 +238,7 @@ export default class OpenAICompatProvider implements Provider {
238
238
  #model: string;
239
239
  #url: string;
240
240
  #fetchTimeoutMs: number;
241
+ #streamIdleTimeoutMs: number | undefined;
241
242
  #headers: Record<string, string>;
242
243
  #fetch: ProviderFetch;
243
244
  #hasApiKey = false;
@@ -255,14 +256,14 @@ export default class OpenAICompatProvider implements Provider {
255
256
  #retryDelayMs: number;
256
257
  #reasoningStyle: ReasoningStyle;
257
258
  #countTokens: (text: string) => number;
258
- #costFor: (usage: ProviderUsage) => number;
259
+ #calculateCost: (usage: ProviderUsage) => number;
259
260
  #source: string;
260
261
  #grammarStyle: GrammarStyle;
261
262
  #promptCacheKey: boolean;
263
+ #serviceTier: string | undefined;
262
264
  #gbnfDebug: boolean;
263
265
  #streaming: boolean;
264
266
  #firstPartyMetadata: boolean;
265
- #balanceMetaKey: string | undefined;
266
267
  #supportsSlotPinning: boolean;
267
268
  #slotCount: number | null;
268
269
  #retryAttempts: number;
@@ -284,6 +285,7 @@ export default class OpenAICompatProvider implements Provider {
284
285
  this.#model = config.model;
285
286
  this.#url = config.url;
286
287
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
288
+ this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
287
289
  this.#headers = config.headers ?? {};
288
290
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
289
291
  this.#contextWindow = config.contextWindow ?? null;
@@ -305,17 +307,17 @@ export default class OpenAICompatProvider implements Provider {
305
307
  this.#retryAttempts = config.retryAttempts;
306
308
  this.#reasoningStyle = config.reasoningStyle ?? "none";
307
309
  this.#countTokens = config.countTokens ?? heuristicTokens;
308
- this.#costFor = config.costFor ?? (() => 0);
310
+ this.#calculateCost = config.calculateCost ?? (() => 0);
309
311
  this.#source = config.source ?? "provider";
310
312
  this.#grammarStyle = config.grammarStyle ?? "none";
311
313
  this.#promptCacheKey = config.promptCacheKey ?? false;
314
+ this.#serviceTier = config.serviceTier;
312
315
  this.#gbnfDebug = config.gbnfDebug ?? false;
313
316
  this.#streaming = config.streaming ?? true;
314
317
  this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
315
318
  this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
316
319
  this.#eosText = config.eosText;
317
320
  this.#hasApiKey = "Authorization" in this.#headers;
318
- this.#balanceMetaKey = config.balanceMetaKey;
319
321
  this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
320
322
  this.#slotCount = config.slotCount ?? null;
321
323
  this.#topLogprobs = config.topLogprobs ?? null;
@@ -365,7 +367,7 @@ export default class OpenAICompatProvider implements Provider {
365
367
  get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
366
368
 
367
369
  countTokens(text: string): number { return this.#countTokens(text); }
368
- costFor(usage: ProviderUsage): number { return this.#costFor(usage); }
370
+ calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
369
371
 
370
372
  // Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
371
373
  // backend's wire mechanism — including under a transported grammar. The #32
@@ -444,31 +446,25 @@ export default class OpenAICompatProvider implements Provider {
444
446
  return { id_slot: slot };
445
447
  }
446
448
 
447
- // Grammar transport (SPEC §13): carry the caller-supplied GBNF in the shape
448
- // the backend accepts. Same grammar, different wire field per backend; an
449
- // unsupported/unknown backend sends NO field at all (cloud APIs 400 on
450
- // unknowns, and a silent send would let a constrained consumer mistake
451
- // unconstrained output for enforced).
449
+ // Optional local llama-server GBNF transport (SPEC §13). Unsupported
450
+ // backends receive no grammar-related field.
452
451
  #grammarBody(grammar: string | undefined): Record<string, unknown> {
453
452
  if (grammar === undefined) return {};
454
453
  switch (this.#grammarStyle) {
455
454
  // Greedy decoding under hard constraint loops without a repeat-penalty
456
- // floor (#9, SPEC §13) — every grammar path carries it. llama.cpp spells
457
- // it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
458
- // (verified honored live, #20).
455
+ // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
459
456
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
460
- case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
461
457
  case "none": return {};
462
458
  }
463
459
  }
464
460
 
465
461
  // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
466
- // convention - NOT grammar-bound. GBNF is a local rail (off for cloud), so a cloud
462
+ // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
467
463
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
468
464
  // straight to the token cap on pure looped repetition (run52). Ships next to
469
465
  // temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
470
- // managed FLOOR in #grammarBody. Per backend: llama.cpp/response_format take the
471
- // repeat_penalty MULTIPLIER; the plain cloud path ("none") can't, so it gets
466
+ // managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
467
+ // MULTIPLIER; the plain cloud path ("none") can't, so it gets
472
468
  // frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
473
469
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
474
470
  #repetitionPenaltyBody(): Record<string, unknown> {
@@ -485,7 +481,6 @@ export default class OpenAICompatProvider implements Provider {
485
481
  ...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
486
482
  } : {}),
487
483
  };
488
- case "response_format": return { repetition_penalty: this.#repeatPenalty };
489
484
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
490
485
  }
491
486
  }
@@ -568,18 +563,10 @@ export default class OpenAICompatProvider implements Provider {
568
563
  }
569
564
 
570
565
  // Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
571
- // (the transport's `chunkMetadata`) through VERBATIM, then normalize the known
572
- // keys we hold a contract for — the spec's balance field → a validated
573
- // `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
574
- // key so the consumer reads one canonical name. Undefined when nothing's there;
575
- // the service merges this into its Turn metadata and filters what reaches clients.
566
+ // through verbatim. Providers do not reinterpret vendor currency or account
567
+ // metadata; a monetary value carries its own amount and currency.
576
568
  #buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
577
569
  const meta: Record<string, unknown> = { ...chunkMetadata };
578
- if (this.#balanceMetaKey !== undefined) {
579
- const raw = meta[this.#balanceMetaKey];
580
- delete meta[this.#balanceMetaKey];
581
- if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
582
- }
583
570
  return Object.keys(meta).length > 0 ? meta : undefined;
584
571
  }
585
572
 
@@ -620,6 +607,7 @@ export default class OpenAICompatProvider implements Provider {
620
607
  // the router's per-model tuning must not be overridden by client floors.
621
608
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
622
609
  ...this.#samplingBody(sampling),
610
+ ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
623
611
  model: this.#model,
624
612
  messages,
625
613
  ...this.#reasoningBody(),
@@ -640,23 +628,27 @@ export default class OpenAICompatProvider implements Provider {
640
628
  // signal spans them all. Retry only the transient classifications, prefer
641
629
  // a server Retry-After over the backoff, and let the caller's abort cut
642
630
  // through both the in-flight request and the backoff sleep.
643
- // Stream by default, but fall back to one non-streamed JSON for the one
644
- // case it breaks: a response_format grammar (fireworks) streams its
645
- // constrained output mislabeled as reasoning_content, yet returns it as
646
- // content non-streamed. The atomic dump is correct either way, so the
647
- // demotion is scoped to exactly that request, not the whole provider.
648
- const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
649
- const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
631
+ const transport = this.#streaming ? chatCompletionStream : chatCompletion;
650
632
 
651
633
  // Per-request headers = static auth/routing + any first-party telemetry.
652
634
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
653
635
  const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
636
+ const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
654
637
  let raw;
655
638
  for (let attempt = 0; ; attempt++) {
639
+ const attemptStarted = performance.now();
656
640
  const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
657
641
  const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
658
642
  try {
659
- raw = await transport({ url: this.#url, headers, body, signal: effectiveSignal, fetch: this.#fetch, captureRawBody: this.#rawBody });
643
+ raw = await transport({
644
+ url: this.#url,
645
+ headers,
646
+ body,
647
+ signal: effectiveSignal,
648
+ fetch: this.#fetch,
649
+ captureRawBody: this.#rawBody,
650
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
651
+ });
660
652
  break;
661
653
  } catch (err) {
662
654
  // Caller-initiated abort is cancellation — never retried or wrapped.
@@ -677,6 +669,12 @@ export default class OpenAICompatProvider implements Provider {
677
669
  }
678
670
  throw pe;
679
671
  }
672
+ transportRetries.push({
673
+ attempt: attempt + 1,
674
+ kind,
675
+ elapsedMs: Math.round(performance.now() - attemptStarted),
676
+ message: err instanceof Error ? err.message : String(err),
677
+ });
680
678
  const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
681
679
  await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
682
680
  }
@@ -728,7 +726,10 @@ export default class OpenAICompatProvider implements Provider {
728
726
  }
729
727
 
730
728
  const builtMeta = this.#buildMeta(raw.chunkMetadata);
731
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
729
+ const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
730
+ const meta = railsMeta !== undefined || retryMeta !== undefined
731
+ ? { ...builtMeta, ...railsMeta, ...retryMeta }
732
+ : builtMeta;
732
733
 
733
734
  // #36: surface per-token logprobs + their mean when the backend returned
734
735
  // them (only possible when the flag requested them). Absent otherwise —
package/src/Pool.test.ts CHANGED
@@ -28,7 +28,7 @@ const backend = (opts: FakeOpts = {}) => {
28
28
  ...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
29
29
  ...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
30
30
  countTokens: (t: string) => t.length,
31
- costFor: () => opts.cost ?? 0,
31
+ calculateCost: () => opts.cost ?? 0,
32
32
  generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
33
33
  served.push(args.workerId);
34
34
  if (opts.throws !== undefined) throw opts.throws;
@@ -88,10 +88,10 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
88
88
  assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
89
89
  });
90
90
 
91
- test("Pool: countTokens + costFor delegate to a backend", () => {
91
+ test("Pool: countTokens + calculateCost delegate to a backend", () => {
92
92
  const p = new Pool([backend({ cost: 42 }).b]);
93
93
  assert.equal(p.countTokens("abcd"), 4);
94
- assert.equal(p.costFor({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
94
+ assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
95
95
  });
96
96
 
97
97
  // --- dispatch: round-robin + affinity ---
package/src/Pool.ts CHANGED
@@ -83,7 +83,7 @@ export default class Pool implements Provider {
83
83
  }
84
84
 
85
85
  countTokens(text: string): number { return this.#backends[0].countTokens(text); }
86
- costFor(usage: ProviderUsage): number { return this.#backends[0].costFor(usage); }
86
+ calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
87
87
 
88
88
  // --- dispatch ---
89
89
 
@@ -2,7 +2,7 @@ import test, { mock } from "node:test";
2
2
  import { strict as assert } from "node:assert";
3
3
  import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
4
4
 
5
- const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, costFor: () => 0, generate: async () => { throw new Error("unused"); } };
5
+ const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, calculateCost: () => 0, generate: async () => { throw new Error("unused"); } };
6
6
  const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
7
7
  async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
8
8
 
@@ -14,6 +14,7 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
14
14
 
15
15
  const fullEnv = Object.freeze({
16
16
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
+ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
17
18
  PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
18
19
  OPENAI_BASE_URL: "http://x",
19
20
  });
@@ -162,6 +163,34 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
162
163
  mock.restoreAll();
163
164
  });
164
165
 
166
+ test("#622: two Fireworks aliases independently select default and priority service tiers", async () => {
167
+ const bodies: Record<string, unknown>[] = [];
168
+ mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
169
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
170
+ return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
171
+ });
172
+ const env = {
173
+ ...fullEnv,
174
+ FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
175
+ FIREWORKS_API_KEY: "fw",
176
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
177
+ PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
178
+ PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
179
+ };
180
+ const imports = async () => ({});
181
+ const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
182
+ const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
183
+ const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
184
+ await fast.generate({ workerId: "fast-worker", messages: [] });
185
+ await standard.generate({ workerId: "standard-worker", messages: [] });
186
+ assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
187
+ assert.deepEqual(bodies.map((body) => body.model), [
188
+ "accounts/fireworks/routers/glm-5p2-fast",
189
+ "accounts/fireworks/models/deepseek-v4-pro",
190
+ ]);
191
+ mock.restoreAll();
192
+ });
193
+
165
194
  test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
166
195
  resetDiscoveryCache();
167
196
  const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;