@plurnk/plurnk-providers 1.3.3 → 1.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +20 -21
- package/SPEC.md +65 -44
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +5 -4
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +38 -38
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +2 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/openaiStream.d.ts +6 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +30 -2
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +3 -3
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +36 -17
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +10 -7
- package/src/Mock.test.ts +2 -2
- package/src/Mock.ts +1 -1
- package/src/OpenAICompat.test.ts +86 -86
- package/src/OpenAICompat.ts +46 -45
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +30 -1
- package/src/aiSdkAdapter.spike.test.ts +242 -0
- package/src/env.test.ts +8 -0
- package/src/env.ts +2 -0
- package/src/index.ts +2 -2
- package/src/openai.ts +1 -1
- package/src/openaiStream.ts +30 -2
- package/src/standardProviders.test.ts +45 -31
- package/src/standardProviders.ts +42 -29
- package/src/types.ts +5 -5
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
package/src/OpenAICompat.test.ts
CHANGED
|
@@ -25,8 +25,7 @@ const installFetch = (chunks: unknown[]) => {
|
|
|
25
25
|
return calls;
|
|
26
26
|
};
|
|
27
27
|
|
|
28
|
-
// Fake fetch returning one non-streamed JSON body
|
|
29
|
-
// demotes off SSE (a response_format grammar). Captures the request the same way.
|
|
28
|
+
// Fake fetch returning one non-streamed JSON body. Captures the request too.
|
|
30
29
|
const installFetchJson = (payload: unknown) => {
|
|
31
30
|
const calls: { url: string; init: RequestInit }[] = [];
|
|
32
31
|
mock.method(globalThis, "fetch", async (url: string, init: RequestInit) => {
|
|
@@ -235,17 +234,17 @@ test("identity getters and defaults", () => {
|
|
|
235
234
|
assert.equal(p.contextWindow, null); // default
|
|
236
235
|
assert.equal(p.countTokens(""), 0);
|
|
237
236
|
assert.equal(p.countTokens("four"), 2); // default heuristic ceil(4/2) upper bound
|
|
238
|
-
assert.equal(p.
|
|
237
|
+
assert.equal(p.calculateCost({ prompt: 9, completion: 9, reasoning: 0, cached: 0, total: 18 }), 0); // default free
|
|
239
238
|
});
|
|
240
239
|
|
|
241
|
-
test("injected countTokens and
|
|
240
|
+
test("injected countTokens and calculateCost are used", () => {
|
|
242
241
|
const p = new OpenAICompatProvider({
|
|
243
242
|
model: "m", url: "http://x", fetchTimeoutMs: 1000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0,
|
|
244
243
|
countTokens: (t) => t.length,
|
|
245
|
-
|
|
244
|
+
calculateCost: (u) => u.total * 2,
|
|
246
245
|
});
|
|
247
246
|
assert.equal(p.countTokens("abc"), 3);
|
|
248
|
-
assert.equal(p.
|
|
247
|
+
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 5 }), 10);
|
|
249
248
|
});
|
|
250
249
|
|
|
251
250
|
test("generate maps a streamed response into ProviderResponse", async () => {
|
|
@@ -371,15 +370,14 @@ test("reasoningStyle 'effort_explicit': off SENDS none, adaptive OMITS (#403 —
|
|
|
371
370
|
});
|
|
372
371
|
|
|
373
372
|
test("the family temperature default rides every request; caller sampling overrides it (#30)", async () => {
|
|
374
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
373
|
+
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
374
|
+
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
375
|
+
await p.generate({ workerId: "r", messages: [] });
|
|
378
376
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.2);
|
|
379
377
|
mock.restoreAll();
|
|
380
378
|
// explicit caller sampling wins over the default
|
|
381
|
-
calls =
|
|
382
|
-
await p.generate({ workerId: "r", messages: [],
|
|
379
|
+
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
380
|
+
await p.generate({ workerId: "r", messages: [], sampling: { temperature: 0.7 } });
|
|
383
381
|
assert.equal(JSON.parse(calls[0].init.body as string).temperature, 0.7);
|
|
384
382
|
mock.restoreAll();
|
|
385
383
|
// temperature is now the UNIVERSAL default: present without a grammar too
|
|
@@ -419,24 +417,6 @@ test("#567: DRY + repeat_last_n ride the llamacpp path when set; unset leaves th
|
|
|
419
417
|
mock.restoreAll();
|
|
420
418
|
});
|
|
421
419
|
|
|
422
|
-
test("effort_explicit: intent maps IDENTICALLY with and without a grammar — the #32 clamp is lifted (reasoning+rails coexist)", async () => {
|
|
423
|
-
const warned: Array<string | Error> = [];
|
|
424
|
-
mock.method(process, "emitWarning", (msg: string | Error) => { warned.push(msg); });
|
|
425
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format", source: "provider:test" });
|
|
426
|
-
// grammar transported → intent STILL flows through (canary-verified: the mask
|
|
427
|
-
// covers only content; clamping to "none" was the plan-less #331 regression)
|
|
428
|
-
let calls = installFetchJson(jsonChoice);
|
|
429
|
-
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
430
|
-
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
431
|
-
assert.equal(warned.filter((w) => String(w).includes("clamped")).length, 0, "no clamp warning — the clamp is gone");
|
|
432
|
-
mock.restoreAll();
|
|
433
|
-
// no grammar → same mapping
|
|
434
|
-
const p2 = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "on", budget: 8192 }, retryAttempts: 0, reasoningStyle: "effort_explicit", grammarStyle: "response_format" });
|
|
435
|
-
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
436
|
-
await p2.generate({ workerId: "r", messages: [] });
|
|
437
|
-
assert.equal(JSON.parse(calls[0].init.body as string).reasoning_effort, "high");
|
|
438
|
-
});
|
|
439
|
-
|
|
440
420
|
test("llamacpp grammar path: temperature default + the managed repeat-penalty floor", async () => {
|
|
441
421
|
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
442
422
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
@@ -447,15 +427,9 @@ test("llamacpp grammar path: temperature default + the managed repeat-penalty fl
|
|
|
447
427
|
});
|
|
448
428
|
|
|
449
429
|
test("#426: the repeat penalty rides EVERY request rail-off, keyed per backend (cloud degeneration guard)", async () => {
|
|
450
|
-
//
|
|
451
|
-
const fw = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
|
|
452
|
-
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
453
|
-
await fw.generate({ workerId: "r", messages: [] });
|
|
454
|
-
assert.equal(JSON.parse(calls[0].init.body as string).repetition_penalty, 1.15);
|
|
455
|
-
mock.restoreAll();
|
|
456
|
-
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded now)
|
|
430
|
+
// llama.cpp with NO grammar carries its key too (unconstrained local is guarded)
|
|
457
431
|
const llama = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "llamacpp" });
|
|
458
|
-
calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
432
|
+
let calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
459
433
|
await llama.generate({ workerId: "r", messages: [] });
|
|
460
434
|
assert.equal(JSON.parse(calls[0].init.body as string).repeat_penalty, 1.15);
|
|
461
435
|
mock.restoreAll();
|
|
@@ -609,32 +583,6 @@ test("grammar transport 'llamacpp': top-level grammar + the repeat-penalty floor
|
|
|
609
583
|
assert.equal("response_format" in body, false);
|
|
610
584
|
});
|
|
611
585
|
|
|
612
|
-
test("grammar transport 'response_format': response_format.grammar, no top-level grammar (Fireworks)", async () => {
|
|
613
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
|
|
614
|
-
const calls = installFetchJson(jsonChoice); // response_format grammar demotes off SSE
|
|
615
|
-
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
616
|
-
const body = JSON.parse(calls[0].init.body as string);
|
|
617
|
-
assert.deepEqual(body.response_format, { type: "grammar", grammar: 'root ::= "x"' });
|
|
618
|
-
assert.equal("grammar" in body, false); // not the llama.cpp shape
|
|
619
|
-
assert.equal("repeat_penalty" in body, false); // llama.cpp spelling not used here
|
|
620
|
-
assert.equal(body.repetition_penalty, 1.15); // the floor still rides (OpenAI-compat spelling, #20)
|
|
621
|
-
});
|
|
622
|
-
|
|
623
|
-
// A response_format grammar is the one case the spine drops streaming for, even
|
|
624
|
-
// with streaming on (default): fireworks mislabels the streamed grammar output
|
|
625
|
-
// as reasoning_content but returns it as content non-streamed (§13). The demotion
|
|
626
|
-
// is per-request — a grammarless call on the same provider still streams.
|
|
627
|
-
test("response_format grammar demotes THIS request off SSE; grammarless calls still stream", async () => {
|
|
628
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, grammarStyle: "response_format" });
|
|
629
|
-
const jsonCalls = installFetchJson(jsonChoice);
|
|
630
|
-
await p.generate({ workerId: "r", messages: [], grammar: 'root ::= "x"' });
|
|
631
|
-
assert.equal("stream" in JSON.parse(jsonCalls[0].init.body as string), false); // no SSE flag
|
|
632
|
-
mock.restoreAll();
|
|
633
|
-
const sseCalls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
634
|
-
await p.generate({ workerId: "r", messages: [] }); // no grammar → streams
|
|
635
|
-
assert.equal(JSON.parse(sseCalls[0].init.body as string).stream, true); // SSE flag present
|
|
636
|
-
});
|
|
637
|
-
|
|
638
586
|
test("grammar transport 'none' (default): the grammar is never sent — no silent unconstrained", async () => {
|
|
639
587
|
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0 });
|
|
640
588
|
const calls = installFetch([{ choices: [{ delta: { content: "x" } }] }]);
|
|
@@ -756,31 +704,15 @@ test("gbnfDebug: an INVALID grammar throws before any wire call — it never rea
|
|
|
756
704
|
assert.equal(calls.length, 0); // fail-hard before the fetch — grammar never transported
|
|
757
705
|
});
|
|
758
706
|
|
|
759
|
-
// — meta bag:
|
|
707
|
+
// — meta bag: verbatim provider metadata (#23) —
|
|
760
708
|
|
|
761
|
-
test("meta:
|
|
762
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
assert.equal(res.meta?.balancePico, 4_200_000);
|
|
766
|
-
assert.equal("balance_pico" in (res.meta ?? {}), false); // raw key renamed to the canonical balancePico
|
|
767
|
-
});
|
|
768
|
-
|
|
769
|
-
test("meta: passes the backend's extra top-level fields through verbatim (every provider)", async () => {
|
|
770
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false }); // no balanceMetaKey
|
|
771
|
-
installFetchJson({ ...jsonChoice, balance_pico: 4_200_000, system_fingerprint: "fp_abc" });
|
|
709
|
+
test("meta: passes backend fields through without reinterpreting monetary values", async () => {
|
|
710
|
+
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false });
|
|
711
|
+
const balance = { amount: "0.0000042", currency: "XMR" };
|
|
712
|
+
installFetchJson({ ...jsonChoice, balance, system_fingerprint: "fp_abc" });
|
|
772
713
|
const res = await p.generate({ workerId: "r", messages: [] });
|
|
773
|
-
assert.
|
|
714
|
+
assert.deepEqual(res.meta?.balance, balance);
|
|
774
715
|
assert.equal(res.meta?.system_fingerprint, "fp_abc");
|
|
775
|
-
assert.equal("balancePico" in (res.meta ?? {}), false); // not normalized without the key
|
|
776
|
-
});
|
|
777
|
-
|
|
778
|
-
test("meta: a non-numeric balance is dropped, never surfaced as balancePico (null-honest)", async () => {
|
|
779
|
-
const p = new OpenAICompatProvider({ model: "m", url: "http://x", fetchTimeoutMs: 5000, temperature: 0.2, repeatPenalty: 1.15, retryDelayMs: 1, reasoning: { mode: "off", budget: null }, retryAttempts: 0, streaming: false, balanceMetaKey: "balance_pico" });
|
|
780
|
-
installFetchJson({ ...jsonChoice, balance_pico: "lots" });
|
|
781
|
-
const res = await p.generate({ workerId: "r", messages: [] });
|
|
782
|
-
assert.equal("balancePico" in (res.meta ?? {}), false);
|
|
783
|
-
assert.equal("balance_pico" in (res.meta ?? {}), false); // raw dropped too — the known key is validated away
|
|
784
716
|
});
|
|
785
717
|
|
|
786
718
|
// — first-party telemetry headers (attribution + client, SPEC §5) —
|
|
@@ -972,6 +904,74 @@ test("retry: a transient failure retries and a later success resolves", async ()
|
|
|
972
904
|
assert.equal(calls.length, 3); // 429 → 503 → 200
|
|
973
905
|
});
|
|
974
906
|
|
|
907
|
+
test("#559: streamed-body silence retries and records the recovery in durable response metadata", async () => {
|
|
908
|
+
let calls = 0;
|
|
909
|
+
mock.method(globalThis, "fetch", async () => {
|
|
910
|
+
calls++;
|
|
911
|
+
if (calls === 1) {
|
|
912
|
+
return new Response(new ReadableStream({
|
|
913
|
+
start(controller) {
|
|
914
|
+
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"partial"}}]}\n\n'));
|
|
915
|
+
},
|
|
916
|
+
}), { status: 200 });
|
|
917
|
+
}
|
|
918
|
+
return new Response(new ReadableStream({
|
|
919
|
+
start(controller) {
|
|
920
|
+
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"recovered"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
|
|
921
|
+
controller.close();
|
|
922
|
+
},
|
|
923
|
+
}), { status: 200 });
|
|
924
|
+
});
|
|
925
|
+
const p = new OpenAICompatProvider({
|
|
926
|
+
model: "m",
|
|
927
|
+
url: "http://x",
|
|
928
|
+
fetchTimeoutMs: 1000,
|
|
929
|
+
streamIdleTimeoutMs: 10,
|
|
930
|
+
temperature: 0.2,
|
|
931
|
+
repeatPenalty: 1.15,
|
|
932
|
+
retryDelayMs: 1,
|
|
933
|
+
reasoning: { mode: "off", budget: null },
|
|
934
|
+
retryAttempts: 1,
|
|
935
|
+
source: "provider:test",
|
|
936
|
+
});
|
|
937
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
938
|
+
assert.equal(result.assistant.content, "recovered");
|
|
939
|
+
assert.equal(calls, 2);
|
|
940
|
+
const retries = result.meta?.transportRetries as Array<Record<string, unknown>>;
|
|
941
|
+
assert.equal(retries.length, 1);
|
|
942
|
+
assert.equal(retries[0].attempt, 1);
|
|
943
|
+
assert.equal(retries[0].kind, "network_failure");
|
|
944
|
+
assert.equal(typeof retries[0].elapsedMs, "number");
|
|
945
|
+
assert.match(String(retries[0].message), /no body bytes for 10ms/);
|
|
946
|
+
mock.restoreAll();
|
|
947
|
+
});
|
|
948
|
+
|
|
949
|
+
test("#559: a zero stream-idle timeout permits a slow inter-chunk pause", async () => {
|
|
950
|
+
mock.method(globalThis, "fetch", async () => new Response(new ReadableStream({
|
|
951
|
+
async start(controller) {
|
|
952
|
+
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"slow "}}]}\n\n'));
|
|
953
|
+
await new Promise((resolve) => setTimeout(resolve, 20));
|
|
954
|
+
controller.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"is valid"},"finish_reason":"stop"}]}\n\ndata: [DONE]\n\n'));
|
|
955
|
+
controller.close();
|
|
956
|
+
},
|
|
957
|
+
}), { status: 200 }));
|
|
958
|
+
const p = new OpenAICompatProvider({
|
|
959
|
+
model: "m",
|
|
960
|
+
url: "http://x",
|
|
961
|
+
fetchTimeoutMs: 1000,
|
|
962
|
+
streamIdleTimeoutMs: 0,
|
|
963
|
+
temperature: 0.2,
|
|
964
|
+
repeatPenalty: 1.15,
|
|
965
|
+
retryDelayMs: 1,
|
|
966
|
+
reasoning: { mode: "off", budget: null },
|
|
967
|
+
retryAttempts: 0,
|
|
968
|
+
});
|
|
969
|
+
const result = await p.generate({ workerId: "r", messages: [] });
|
|
970
|
+
assert.equal(result.assistant.content, "slow is valid");
|
|
971
|
+
assert.equal(result.meta?.transportRetries, undefined);
|
|
972
|
+
mock.restoreAll();
|
|
973
|
+
});
|
|
974
|
+
|
|
975
975
|
test("retry: exhausting the budget surfaces the classified ProviderError", async () => {
|
|
976
976
|
const { ProviderError } = await import("./telemetry.ts");
|
|
977
977
|
const calls = installFetchScript([{ status: 429, retryAfter: 0 }]); // always rate-limited
|
package/src/OpenAICompat.ts
CHANGED
|
@@ -29,23 +29,21 @@ import { emitWarningOnce } from "./warnings.ts";
|
|
|
29
29
|
// (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
|
|
30
30
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
31
31
|
|
|
32
|
-
//
|
|
33
|
-
//
|
|
34
|
-
|
|
35
|
-
// (never silently — so a constrained consumer can't mistake unconstrained output
|
|
36
|
-
// for enforced).
|
|
37
|
-
export type GrammarStyle = "none" | "llamacpp" | "response_format";
|
|
32
|
+
// GBNF transport is a local llama-server capability. "none" means no
|
|
33
|
+
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
34
|
+
export type GrammarStyle = "none" | "llamacpp";
|
|
38
35
|
|
|
39
36
|
export type OpenAICompatConfig = {
|
|
40
37
|
model: string;
|
|
41
38
|
url: string; // fully-resolved chat-completions URL
|
|
42
39
|
fetchTimeoutMs: number;
|
|
40
|
+
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
43
41
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
44
42
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
45
43
|
contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
|
|
46
44
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
47
45
|
countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
|
|
48
|
-
|
|
46
|
+
calculateCost?: (usage: ProviderUsage) => number; // default () => 0
|
|
49
47
|
source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
|
|
50
48
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
51
49
|
// #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
@@ -54,12 +52,14 @@ export type OpenAICompatConfig = {
|
|
|
54
52
|
// false -- a backend that strict-validates unknown fields 400s, so enable only
|
|
55
53
|
// where the field is accepted. Same identity that already drives slot affinity.
|
|
56
54
|
promptCacheKey?: boolean;
|
|
55
|
+
// Optional provider-configured service tier. Unlike caller sampling, this is
|
|
56
|
+
// a fixed deployment choice and therefore wins on every request.
|
|
57
|
+
serviceTier?: string;
|
|
57
58
|
gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
|
|
58
59
|
streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
|
|
59
60
|
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
60
61
|
apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
|
|
61
62
|
eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
|
|
62
|
-
balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
|
|
63
63
|
// Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
|
|
64
64
|
supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
|
|
65
65
|
slotCount?: number | null; // probed slot count for pinning backends; default null
|
|
@@ -238,6 +238,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
238
238
|
#model: string;
|
|
239
239
|
#url: string;
|
|
240
240
|
#fetchTimeoutMs: number;
|
|
241
|
+
#streamIdleTimeoutMs: number | undefined;
|
|
241
242
|
#headers: Record<string, string>;
|
|
242
243
|
#fetch: ProviderFetch;
|
|
243
244
|
#hasApiKey = false;
|
|
@@ -255,14 +256,14 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
255
256
|
#retryDelayMs: number;
|
|
256
257
|
#reasoningStyle: ReasoningStyle;
|
|
257
258
|
#countTokens: (text: string) => number;
|
|
258
|
-
#
|
|
259
|
+
#calculateCost: (usage: ProviderUsage) => number;
|
|
259
260
|
#source: string;
|
|
260
261
|
#grammarStyle: GrammarStyle;
|
|
261
262
|
#promptCacheKey: boolean;
|
|
263
|
+
#serviceTier: string | undefined;
|
|
262
264
|
#gbnfDebug: boolean;
|
|
263
265
|
#streaming: boolean;
|
|
264
266
|
#firstPartyMetadata: boolean;
|
|
265
|
-
#balanceMetaKey: string | undefined;
|
|
266
267
|
#supportsSlotPinning: boolean;
|
|
267
268
|
#slotCount: number | null;
|
|
268
269
|
#retryAttempts: number;
|
|
@@ -284,6 +285,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
284
285
|
this.#model = config.model;
|
|
285
286
|
this.#url = config.url;
|
|
286
287
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
288
|
+
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
287
289
|
this.#headers = config.headers ?? {};
|
|
288
290
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
289
291
|
this.#contextWindow = config.contextWindow ?? null;
|
|
@@ -305,17 +307,17 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
305
307
|
this.#retryAttempts = config.retryAttempts;
|
|
306
308
|
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
307
309
|
this.#countTokens = config.countTokens ?? heuristicTokens;
|
|
308
|
-
this.#
|
|
310
|
+
this.#calculateCost = config.calculateCost ?? (() => 0);
|
|
309
311
|
this.#source = config.source ?? "provider";
|
|
310
312
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
311
313
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
314
|
+
this.#serviceTier = config.serviceTier;
|
|
312
315
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
313
316
|
this.#streaming = config.streaming ?? true;
|
|
314
317
|
this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
|
|
315
318
|
this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
|
|
316
319
|
this.#eosText = config.eosText;
|
|
317
320
|
this.#hasApiKey = "Authorization" in this.#headers;
|
|
318
|
-
this.#balanceMetaKey = config.balanceMetaKey;
|
|
319
321
|
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
320
322
|
this.#slotCount = config.slotCount ?? null;
|
|
321
323
|
this.#topLogprobs = config.topLogprobs ?? null;
|
|
@@ -365,7 +367,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
365
367
|
get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
|
|
366
368
|
|
|
367
369
|
countTokens(text: string): number { return this.#countTokens(text); }
|
|
368
|
-
|
|
370
|
+
calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
|
|
369
371
|
|
|
370
372
|
// Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
|
|
371
373
|
// backend's wire mechanism — including under a transported grammar. The #32
|
|
@@ -444,31 +446,25 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
444
446
|
return { id_slot: slot };
|
|
445
447
|
}
|
|
446
448
|
|
|
447
|
-
//
|
|
448
|
-
//
|
|
449
|
-
// unsupported/unknown backend sends NO field at all (cloud APIs 400 on
|
|
450
|
-
// unknowns, and a silent send would let a constrained consumer mistake
|
|
451
|
-
// unconstrained output for enforced).
|
|
449
|
+
// Optional local llama-server GBNF transport (SPEC §13). Unsupported
|
|
450
|
+
// backends receive no grammar-related field.
|
|
452
451
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
453
452
|
if (grammar === undefined) return {};
|
|
454
453
|
switch (this.#grammarStyle) {
|
|
455
454
|
// Greedy decoding under hard constraint loops without a repeat-penalty
|
|
456
|
-
// floor (#9, SPEC §13) —
|
|
457
|
-
// it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
|
|
458
|
-
// (verified honored live, #20).
|
|
455
|
+
// floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
|
|
459
456
|
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
460
|
-
case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
|
|
461
457
|
case "none": return {};
|
|
462
458
|
}
|
|
463
459
|
}
|
|
464
460
|
|
|
465
461
|
// Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
|
|
466
|
-
// convention - NOT grammar-bound. GBNF is a local
|
|
462
|
+
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
467
463
|
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
468
464
|
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
469
465
|
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
470
|
-
// managed FLOOR in #grammarBody.
|
|
471
|
-
//
|
|
466
|
+
// managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
|
|
467
|
+
// MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
472
468
|
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
473
469
|
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
474
470
|
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
@@ -485,7 +481,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
485
481
|
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
486
482
|
} : {}),
|
|
487
483
|
};
|
|
488
|
-
case "response_format": return { repetition_penalty: this.#repeatPenalty };
|
|
489
484
|
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
490
485
|
}
|
|
491
486
|
}
|
|
@@ -568,18 +563,10 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
568
563
|
}
|
|
569
564
|
|
|
570
565
|
// Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
|
|
571
|
-
//
|
|
572
|
-
//
|
|
573
|
-
// `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
|
|
574
|
-
// key so the consumer reads one canonical name. Undefined when nothing's there;
|
|
575
|
-
// the service merges this into its Turn metadata and filters what reaches clients.
|
|
566
|
+
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
567
|
+
// metadata; a monetary value carries its own amount and currency.
|
|
576
568
|
#buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
577
569
|
const meta: Record<string, unknown> = { ...chunkMetadata };
|
|
578
|
-
if (this.#balanceMetaKey !== undefined) {
|
|
579
|
-
const raw = meta[this.#balanceMetaKey];
|
|
580
|
-
delete meta[this.#balanceMetaKey];
|
|
581
|
-
if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
|
|
582
|
-
}
|
|
583
570
|
return Object.keys(meta).length > 0 ? meta : undefined;
|
|
584
571
|
}
|
|
585
572
|
|
|
@@ -620,6 +607,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
620
607
|
// the router's per-model tuning must not be overridden by client floors.
|
|
621
608
|
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
622
609
|
...this.#samplingBody(sampling),
|
|
610
|
+
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
623
611
|
model: this.#model,
|
|
624
612
|
messages,
|
|
625
613
|
...this.#reasoningBody(),
|
|
@@ -640,23 +628,27 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
640
628
|
// signal spans them all. Retry only the transient classifications, prefer
|
|
641
629
|
// a server Retry-After over the backoff, and let the caller's abort cut
|
|
642
630
|
// through both the in-flight request and the backoff sleep.
|
|
643
|
-
|
|
644
|
-
// case it breaks: a response_format grammar (fireworks) streams its
|
|
645
|
-
// constrained output mislabeled as reasoning_content, yet returns it as
|
|
646
|
-
// content non-streamed. The atomic dump is correct either way, so the
|
|
647
|
-
// demotion is scoped to exactly that request, not the whole provider.
|
|
648
|
-
const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
|
|
649
|
-
const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
|
|
631
|
+
const transport = this.#streaming ? chatCompletionStream : chatCompletion;
|
|
650
632
|
|
|
651
633
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
652
634
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
653
635
|
const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
|
|
636
|
+
const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
|
|
654
637
|
let raw;
|
|
655
638
|
for (let attempt = 0; ; attempt++) {
|
|
639
|
+
const attemptStarted = performance.now();
|
|
656
640
|
const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
|
|
657
641
|
const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
|
|
658
642
|
try {
|
|
659
|
-
raw = await transport({
|
|
643
|
+
raw = await transport({
|
|
644
|
+
url: this.#url,
|
|
645
|
+
headers,
|
|
646
|
+
body,
|
|
647
|
+
signal: effectiveSignal,
|
|
648
|
+
fetch: this.#fetch,
|
|
649
|
+
captureRawBody: this.#rawBody,
|
|
650
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
651
|
+
});
|
|
660
652
|
break;
|
|
661
653
|
} catch (err) {
|
|
662
654
|
// Caller-initiated abort is cancellation — never retried or wrapped.
|
|
@@ -677,6 +669,12 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
677
669
|
}
|
|
678
670
|
throw pe;
|
|
679
671
|
}
|
|
672
|
+
transportRetries.push({
|
|
673
|
+
attempt: attempt + 1,
|
|
674
|
+
kind,
|
|
675
|
+
elapsedMs: Math.round(performance.now() - attemptStarted),
|
|
676
|
+
message: err instanceof Error ? err.message : String(err),
|
|
677
|
+
});
|
|
680
678
|
const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
|
|
681
679
|
await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
|
|
682
680
|
}
|
|
@@ -728,7 +726,10 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
728
726
|
}
|
|
729
727
|
|
|
730
728
|
const builtMeta = this.#buildMeta(raw.chunkMetadata);
|
|
731
|
-
const
|
|
729
|
+
const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
|
|
730
|
+
const meta = railsMeta !== undefined || retryMeta !== undefined
|
|
731
|
+
? { ...builtMeta, ...railsMeta, ...retryMeta }
|
|
732
|
+
: builtMeta;
|
|
732
733
|
|
|
733
734
|
// #36: surface per-token logprobs + their mean when the backend returned
|
|
734
735
|
// them (only possible when the flag requested them). Absent otherwise —
|
package/src/Pool.test.ts
CHANGED
|
@@ -28,7 +28,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
28
28
|
...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
|
|
29
29
|
...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
|
|
30
30
|
countTokens: (t: string) => t.length,
|
|
31
|
-
|
|
31
|
+
calculateCost: () => opts.cost ?? 0,
|
|
32
32
|
generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
|
|
33
33
|
served.push(args.workerId);
|
|
34
34
|
if (opts.throws !== undefined) throw opts.throws;
|
|
@@ -88,10 +88,10 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
|
|
|
88
88
|
assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
|
|
89
89
|
});
|
|
90
90
|
|
|
91
|
-
test("Pool: countTokens +
|
|
91
|
+
test("Pool: countTokens + calculateCost delegate to a backend", () => {
|
|
92
92
|
const p = new Pool([backend({ cost: 42 }).b]);
|
|
93
93
|
assert.equal(p.countTokens("abcd"), 4);
|
|
94
|
-
assert.equal(p.
|
|
94
|
+
assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
|
|
95
95
|
});
|
|
96
96
|
|
|
97
97
|
// --- dispatch: round-robin + affinity ---
|
package/src/Pool.ts
CHANGED
|
@@ -83,7 +83,7 @@ export default class Pool implements Provider {
|
|
|
83
83
|
}
|
|
84
84
|
|
|
85
85
|
countTokens(text: string): number { return this.#backends[0].countTokens(text); }
|
|
86
|
-
|
|
86
|
+
calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
|
|
87
87
|
|
|
88
88
|
// --- dispatch ---
|
|
89
89
|
|
|
@@ -2,7 +2,7 @@ import test, { mock } from "node:test";
|
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
|
|
4
4
|
|
|
5
|
-
const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0,
|
|
5
|
+
const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, calculateCost: () => 0, generate: async () => { throw new Error("unused"); } };
|
|
6
6
|
const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
|
|
7
7
|
async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
|
|
8
8
|
|
|
@@ -14,6 +14,7 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
|
|
|
14
14
|
|
|
15
15
|
const fullEnv = Object.freeze({
|
|
16
16
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
|
|
17
|
+
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
17
18
|
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
18
19
|
OPENAI_BASE_URL: "http://x",
|
|
19
20
|
});
|
|
@@ -162,6 +163,34 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
|
|
|
162
163
|
mock.restoreAll();
|
|
163
164
|
});
|
|
164
165
|
|
|
166
|
+
test("#622: two Fireworks aliases independently select default and priority service tiers", async () => {
|
|
167
|
+
const bodies: Record<string, unknown>[] = [];
|
|
168
|
+
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
169
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
170
|
+
return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
171
|
+
});
|
|
172
|
+
const env = {
|
|
173
|
+
...fullEnv,
|
|
174
|
+
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
175
|
+
FIREWORKS_API_KEY: "fw",
|
|
176
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
177
|
+
PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
|
|
178
|
+
PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
|
|
179
|
+
};
|
|
180
|
+
const imports = async () => ({});
|
|
181
|
+
const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
|
|
182
|
+
const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
|
|
183
|
+
const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
|
|
184
|
+
await fast.generate({ workerId: "fast-worker", messages: [] });
|
|
185
|
+
await standard.generate({ workerId: "standard-worker", messages: [] });
|
|
186
|
+
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
|
|
187
|
+
assert.deepEqual(bodies.map((body) => body.model), [
|
|
188
|
+
"accounts/fireworks/routers/glm-5p2-fast",
|
|
189
|
+
"accounts/fireworks/models/deepseek-v4-pro",
|
|
190
|
+
]);
|
|
191
|
+
mock.restoreAll();
|
|
192
|
+
});
|
|
193
|
+
|
|
165
194
|
test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
|
|
166
195
|
resetDiscoveryCache();
|
|
167
196
|
const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
|