@plurnk/plurnk-providers 1.3.2 → 1.3.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +14 -14
- package/SPEC.md +36 -32
- package/dist/OpenAICompat.d.ts +3 -1
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +33 -22
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +2 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1 -1
- package/dist/index.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/openaiStream.d.ts +6 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +30 -2
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +3 -2
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +27 -7
- package/dist/standardProviders.js.map +1 -1
- package/package.json +6 -6
- package/src/OpenAICompat.test.ts +76 -60
- package/src/OpenAICompat.ts +40 -28
- package/src/ProviderRegistry.test.ts +29 -0
- package/src/env.test.ts +8 -0
- package/src/env.ts +2 -0
- package/src/index.ts +1 -1
- package/src/openai.ts +1 -1
- package/src/openaiStream.ts +30 -2
- package/src/standardProviders.test.ts +37 -13
- package/src/standardProviders.ts +33 -16
package/src/OpenAICompat.ts
CHANGED
|
@@ -29,17 +29,15 @@ import { emitWarningOnce } from "./warnings.ts";
|
|
|
29
29
|
// (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
|
|
30
30
|
export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
|
|
31
31
|
|
|
32
|
-
//
|
|
33
|
-
//
|
|
34
|
-
|
|
35
|
-
// (never silently — so a constrained consumer can't mistake unconstrained output
|
|
36
|
-
// for enforced).
|
|
37
|
-
export type GrammarStyle = "none" | "llamacpp" | "response_format";
|
|
32
|
+
// GBNF transport is a local llama-server capability. "none" means no
|
|
33
|
+
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
34
|
+
export type GrammarStyle = "none" | "llamacpp";
|
|
38
35
|
|
|
39
36
|
export type OpenAICompatConfig = {
|
|
40
37
|
model: string;
|
|
41
38
|
url: string; // fully-resolved chat-completions URL
|
|
42
39
|
fetchTimeoutMs: number;
|
|
40
|
+
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
43
41
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
44
42
|
fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
|
|
45
43
|
contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
|
|
@@ -54,6 +52,9 @@ export type OpenAICompatConfig = {
|
|
|
54
52
|
// false -- a backend that strict-validates unknown fields 400s, so enable only
|
|
55
53
|
// where the field is accepted. Same identity that already drives slot affinity.
|
|
56
54
|
promptCacheKey?: boolean;
|
|
55
|
+
// Optional provider-configured service tier. Unlike caller sampling, this is
|
|
56
|
+
// a fixed deployment choice and therefore wins on every request.
|
|
57
|
+
serviceTier?: string;
|
|
57
58
|
gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
|
|
58
59
|
streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
|
|
59
60
|
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
@@ -238,6 +239,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
238
239
|
#model: string;
|
|
239
240
|
#url: string;
|
|
240
241
|
#fetchTimeoutMs: number;
|
|
242
|
+
#streamIdleTimeoutMs: number | undefined;
|
|
241
243
|
#headers: Record<string, string>;
|
|
242
244
|
#fetch: ProviderFetch;
|
|
243
245
|
#hasApiKey = false;
|
|
@@ -259,6 +261,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
259
261
|
#source: string;
|
|
260
262
|
#grammarStyle: GrammarStyle;
|
|
261
263
|
#promptCacheKey: boolean;
|
|
264
|
+
#serviceTier: string | undefined;
|
|
262
265
|
#gbnfDebug: boolean;
|
|
263
266
|
#streaming: boolean;
|
|
264
267
|
#firstPartyMetadata: boolean;
|
|
@@ -284,6 +287,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
284
287
|
this.#model = config.model;
|
|
285
288
|
this.#url = config.url;
|
|
286
289
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
290
|
+
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
287
291
|
this.#headers = config.headers ?? {};
|
|
288
292
|
this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
|
|
289
293
|
this.#contextWindow = config.contextWindow ?? null;
|
|
@@ -309,6 +313,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
309
313
|
this.#source = config.source ?? "provider";
|
|
310
314
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
311
315
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
316
|
+
this.#serviceTier = config.serviceTier;
|
|
312
317
|
this.#gbnfDebug = config.gbnfDebug ?? false;
|
|
313
318
|
this.#streaming = config.streaming ?? true;
|
|
314
319
|
this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
|
|
@@ -444,31 +449,25 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
444
449
|
return { id_slot: slot };
|
|
445
450
|
}
|
|
446
451
|
|
|
447
|
-
//
|
|
448
|
-
//
|
|
449
|
-
// unsupported/unknown backend sends NO field at all (cloud APIs 400 on
|
|
450
|
-
// unknowns, and a silent send would let a constrained consumer mistake
|
|
451
|
-
// unconstrained output for enforced).
|
|
452
|
+
// Optional local llama-server GBNF transport (SPEC §13). Unsupported
|
|
453
|
+
// backends receive no grammar-related field.
|
|
452
454
|
#grammarBody(grammar: string | undefined): Record<string, unknown> {
|
|
453
455
|
if (grammar === undefined) return {};
|
|
454
456
|
switch (this.#grammarStyle) {
|
|
455
457
|
// Greedy decoding under hard constraint loops without a repeat-penalty
|
|
456
|
-
// floor (#9, SPEC §13) —
|
|
457
|
-
// it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
|
|
458
|
-
// (verified honored live, #20).
|
|
458
|
+
// floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
|
|
459
459
|
case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
|
|
460
|
-
case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
|
|
461
460
|
case "none": return {};
|
|
462
461
|
}
|
|
463
462
|
}
|
|
464
463
|
|
|
465
464
|
// Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
|
|
466
|
-
// convention - NOT grammar-bound. GBNF is a local
|
|
465
|
+
// convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
|
|
467
466
|
// alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
|
|
468
467
|
// straight to the token cap on pure looped repetition (run52). Ships next to
|
|
469
468
|
// temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
|
|
470
|
-
// managed FLOOR in #grammarBody.
|
|
471
|
-
//
|
|
469
|
+
// managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
|
|
470
|
+
// MULTIPLIER; the plain cloud path ("none") can't, so it gets
|
|
472
471
|
// frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
|
|
473
472
|
// live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
|
|
474
473
|
#repetitionPenaltyBody(): Record<string, unknown> {
|
|
@@ -485,7 +484,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
485
484
|
...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
|
|
486
485
|
} : {}),
|
|
487
486
|
};
|
|
488
|
-
case "response_format": return { repetition_penalty: this.#repeatPenalty };
|
|
489
487
|
case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
|
|
490
488
|
}
|
|
491
489
|
}
|
|
@@ -620,6 +618,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
620
618
|
// the router's per-model tuning must not be overridden by client floors.
|
|
621
619
|
...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
|
|
622
620
|
...this.#samplingBody(sampling),
|
|
621
|
+
...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
|
|
623
622
|
model: this.#model,
|
|
624
623
|
messages,
|
|
625
624
|
...this.#reasoningBody(),
|
|
@@ -640,23 +639,27 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
640
639
|
// signal spans them all. Retry only the transient classifications, prefer
|
|
641
640
|
// a server Retry-After over the backoff, and let the caller's abort cut
|
|
642
641
|
// through both the in-flight request and the backoff sleep.
|
|
643
|
-
|
|
644
|
-
// case it breaks: a response_format grammar (fireworks) streams its
|
|
645
|
-
// constrained output mislabeled as reasoning_content, yet returns it as
|
|
646
|
-
// content non-streamed. The atomic dump is correct either way, so the
|
|
647
|
-
// demotion is scoped to exactly that request, not the whole provider.
|
|
648
|
-
const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
|
|
649
|
-
const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
|
|
642
|
+
const transport = this.#streaming ? chatCompletionStream : chatCompletion;
|
|
650
643
|
|
|
651
644
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
652
645
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
653
646
|
const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
|
|
647
|
+
const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
|
|
654
648
|
let raw;
|
|
655
649
|
for (let attempt = 0; ; attempt++) {
|
|
650
|
+
const attemptStarted = performance.now();
|
|
656
651
|
const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
|
|
657
652
|
const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
|
|
658
653
|
try {
|
|
659
|
-
raw = await transport({
|
|
654
|
+
raw = await transport({
|
|
655
|
+
url: this.#url,
|
|
656
|
+
headers,
|
|
657
|
+
body,
|
|
658
|
+
signal: effectiveSignal,
|
|
659
|
+
fetch: this.#fetch,
|
|
660
|
+
captureRawBody: this.#rawBody,
|
|
661
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
662
|
+
});
|
|
660
663
|
break;
|
|
661
664
|
} catch (err) {
|
|
662
665
|
// Caller-initiated abort is cancellation — never retried or wrapped.
|
|
@@ -677,6 +680,12 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
677
680
|
}
|
|
678
681
|
throw pe;
|
|
679
682
|
}
|
|
683
|
+
transportRetries.push({
|
|
684
|
+
attempt: attempt + 1,
|
|
685
|
+
kind,
|
|
686
|
+
elapsedMs: Math.round(performance.now() - attemptStarted),
|
|
687
|
+
message: err instanceof Error ? err.message : String(err),
|
|
688
|
+
});
|
|
680
689
|
const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
|
|
681
690
|
await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
|
|
682
691
|
}
|
|
@@ -728,7 +737,10 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
728
737
|
}
|
|
729
738
|
|
|
730
739
|
const builtMeta = this.#buildMeta(raw.chunkMetadata);
|
|
731
|
-
const
|
|
740
|
+
const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
|
|
741
|
+
const meta = railsMeta !== undefined || retryMeta !== undefined
|
|
742
|
+
? { ...builtMeta, ...railsMeta, ...retryMeta }
|
|
743
|
+
: builtMeta;
|
|
732
744
|
|
|
733
745
|
// #36: surface per-token logprobs + their mean when the backend returned
|
|
734
746
|
// them (only possible when the flag requested them). Absent otherwise —
|
|
@@ -14,6 +14,7 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
|
|
|
14
14
|
|
|
15
15
|
const fullEnv = Object.freeze({
|
|
16
16
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
|
|
17
|
+
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
17
18
|
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
18
19
|
OPENAI_BASE_URL: "http://x",
|
|
19
20
|
});
|
|
@@ -162,6 +163,34 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
|
|
|
162
163
|
mock.restoreAll();
|
|
163
164
|
});
|
|
164
165
|
|
|
166
|
+
test("#622: two Fireworks aliases independently select default and priority service tiers", async () => {
|
|
167
|
+
const bodies: Record<string, unknown>[] = [];
|
|
168
|
+
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
169
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
170
|
+
return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
171
|
+
});
|
|
172
|
+
const env = {
|
|
173
|
+
...fullEnv,
|
|
174
|
+
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
175
|
+
FIREWORKS_API_KEY: "fw",
|
|
176
|
+
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
177
|
+
PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
|
|
178
|
+
PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
|
|
179
|
+
};
|
|
180
|
+
const imports = async () => ({});
|
|
181
|
+
const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
|
|
182
|
+
const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
|
|
183
|
+
const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
|
|
184
|
+
await fast.generate({ workerId: "fast-worker", messages: [] });
|
|
185
|
+
await standard.generate({ workerId: "standard-worker", messages: [] });
|
|
186
|
+
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
|
|
187
|
+
assert.deepEqual(bodies.map((body) => body.model), [
|
|
188
|
+
"accounts/fireworks/routers/glm-5p2-fast",
|
|
189
|
+
"accounts/fireworks/models/deepseek-v4-pro",
|
|
190
|
+
]);
|
|
191
|
+
mock.restoreAll();
|
|
192
|
+
});
|
|
193
|
+
|
|
165
194
|
test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
|
|
166
195
|
resetDiscoveryCache();
|
|
167
196
|
const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
|
package/src/env.test.ts
CHANGED
|
@@ -166,6 +166,14 @@ test("#399: the shipped floor activates reasoning by default (adaptive — owner
|
|
|
166
166
|
assert.ok(!defaults.match(/^PLURNK_PROVIDERS_REASONING_BUDGET=/m), "no shipped magnitude — budget is on-mode only");
|
|
167
167
|
});
|
|
168
168
|
|
|
169
|
+
test("#567: the shipped DRY floor stays off while retaining the measured alias-safe shape", async () => {
|
|
170
|
+
const { readFileSync } = await import("node:fs");
|
|
171
|
+
const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
|
|
172
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "one-model tuning is not a universal sampler floor");
|
|
173
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_BASE=1\.75$/m);
|
|
174
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32$/m, "the measured identifier-safe shape remains available for alias opt-in");
|
|
175
|
+
});
|
|
176
|
+
|
|
169
177
|
// -- #507: envelope reserves (owner-ruled migration from PLURNK_SERVICE_*) --
|
|
170
178
|
|
|
171
179
|
test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
|
package/src/env.ts
CHANGED
|
@@ -163,10 +163,12 @@ export const PROVIDERS_KNOBS = Object.freeze([
|
|
|
163
163
|
"PLURNK_PROVIDERS_CONTEXT_WINDOW",
|
|
164
164
|
"PLURNK_PROVIDERS_RETRY_ATTEMPTS",
|
|
165
165
|
"PLURNK_PROVIDERS_FETCH_TIMEOUT",
|
|
166
|
+
"PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
|
|
166
167
|
"PLURNK_PROVIDERS_LLAMA_SERVER",
|
|
167
168
|
"PLURNK_PROVIDERS_TEMPERATURE",
|
|
168
169
|
"PLURNK_PROVIDERS_REPEAT_PENALTY",
|
|
169
170
|
"PLURNK_PROVIDERS_FREQUENCY_PENALTY",
|
|
171
|
+
"PLURNK_PROVIDERS_SERVICE_TIER",
|
|
170
172
|
"PLURNK_PROVIDERS_REPEAT_LAST_N",
|
|
171
173
|
"PLURNK_PROVIDERS_DRY_MULTIPLIER",
|
|
172
174
|
"PLURNK_PROVIDERS_DRY_BASE",
|
package/src/index.ts
CHANGED
|
@@ -36,7 +36,7 @@ export type { OpenAICompatConfig, ReasoningStyle, GrammarStyle } from "./OpenAIC
|
|
|
36
36
|
// worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
|
|
37
37
|
// DECISION stays the consumer's, by choosing which pool to call.
|
|
38
38
|
export { default as Pool } from "./Pool.ts";
|
|
39
|
-
export { chatCompletionStream, chatCompletion, OpenAiHttpError } from "./openaiStream.ts";
|
|
39
|
+
export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
|
|
40
40
|
export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
|
|
41
41
|
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
|
|
42
42
|
export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
|
package/src/openai.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
export { default as OpenAICompatProvider, effortFromBudget } from "./OpenAICompat.ts";
|
|
2
2
|
export type { GrammarStyle, OpenAICompatConfig, ReasoningStyle } from "./OpenAICompat.ts";
|
|
3
|
-
export { chatCompletion, chatCompletionStream, OpenAiHttpError } from "./openaiStream.ts";
|
|
3
|
+
export { chatCompletion, chatCompletionStream, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
|
|
4
4
|
export type {
|
|
5
5
|
EncryptedReasoningItem,
|
|
6
6
|
ProviderFetch,
|
package/src/openaiStream.ts
CHANGED
|
@@ -13,6 +13,9 @@ type StreamRequest = {
|
|
|
13
13
|
// #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
|
|
14
14
|
// default so a serving turn never pays the reassembly/retention cost.
|
|
15
15
|
captureRawBody?: boolean;
|
|
16
|
+
// Maximum silence between streamed response-body chunks. Undefined/zero
|
|
17
|
+
// disables this clock; the caller's signal still owns the total deadline.
|
|
18
|
+
streamIdleTimeoutMs?: number;
|
|
16
19
|
};
|
|
17
20
|
|
|
18
21
|
import type { RawUsage } from "./usage.ts";
|
|
@@ -122,6 +125,15 @@ export class OpenAiHttpError extends Error {
|
|
|
122
125
|
}
|
|
123
126
|
}
|
|
124
127
|
|
|
128
|
+
export class StreamIdleError extends Error {
|
|
129
|
+
readonly timeoutMs: number;
|
|
130
|
+
constructor(timeoutMs: number) {
|
|
131
|
+
super(`stream received no body bytes for ${timeoutMs}ms`);
|
|
132
|
+
this.name = "StreamIdleError";
|
|
133
|
+
this.timeoutMs = timeoutMs;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
125
137
|
const parseRetryAfter = (header: string | null): number | null => {
|
|
126
138
|
if (header === null) return null;
|
|
127
139
|
const asInt = Number.parseInt(header, 10);
|
|
@@ -170,7 +182,7 @@ export const chatCompletion = async ({ url, headers, body, signal, fetch, captur
|
|
|
170
182
|
};
|
|
171
183
|
};
|
|
172
184
|
|
|
173
|
-
export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
|
|
185
|
+
export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
|
|
174
186
|
const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
|
|
175
187
|
|
|
176
188
|
const response = await fetch(url, {
|
|
@@ -206,7 +218,23 @@ export const chatCompletionStream = async ({ url, headers, body, signal, fetch,
|
|
|
206
218
|
let encryptedNoKey = 0;
|
|
207
219
|
|
|
208
220
|
while (true) {
|
|
209
|
-
const
|
|
221
|
+
const read = reader.read();
|
|
222
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
223
|
+
const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
|
|
224
|
+
? new Promise<never>((_resolve, reject) => {
|
|
225
|
+
timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
|
|
226
|
+
})
|
|
227
|
+
: null;
|
|
228
|
+
let result: Awaited<ReturnType<typeof reader.read>>;
|
|
229
|
+
try {
|
|
230
|
+
result = idle === null ? await read : await Promise.race([read, idle]);
|
|
231
|
+
} catch (err) {
|
|
232
|
+
if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
|
|
233
|
+
throw err;
|
|
234
|
+
} finally {
|
|
235
|
+
if (timer !== undefined) clearTimeout(timer);
|
|
236
|
+
}
|
|
237
|
+
const { done, value } = result;
|
|
210
238
|
if (done) break;
|
|
211
239
|
buffer += decoder.decode(value, { stream: true });
|
|
212
240
|
const lines = buffer.split("\n");
|
|
@@ -6,7 +6,7 @@ import { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from
|
|
|
6
6
|
// defaults for the providers exercised outside the coverage loop. `openai` is
|
|
7
7
|
// deliberately omitted so its missing-base fail-hard test still fires.
|
|
8
8
|
const baseEnv = Object.freeze({
|
|
9
|
-
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
9
|
+
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
10
10
|
GROQ_BASE_URL: "https://api.groq.com/openai/v1",
|
|
11
11
|
DEEPINFRA_BASE_URL: "https://api.deepinfra.com/v1/openai",
|
|
12
12
|
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
@@ -298,10 +298,10 @@ test("openai: a garbage PLURNK_PROVIDERS_LLAMA_SERVER value fails hard", async (
|
|
|
298
298
|
);
|
|
299
299
|
});
|
|
300
300
|
|
|
301
|
-
test("constrainsOutput:
|
|
301
|
+
test("constrainsOutput: cloud providers do not claim local GBNF transport", async () => {
|
|
302
302
|
mockEndpoint();
|
|
303
303
|
const fw = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
304
|
-
assert.equal(fw!.constrainsOutput,
|
|
304
|
+
assert.equal(fw!.constrainsOutput, false);
|
|
305
305
|
const gq = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
306
306
|
assert.equal(gq!.constrainsOutput, false);
|
|
307
307
|
});
|
|
@@ -868,9 +868,7 @@ test("plurnk: reads its window from upstream but stays a plain OpenAI client —
|
|
|
868
868
|
mock.restoreAll();
|
|
869
869
|
});
|
|
870
870
|
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
test("fireworks: a grammar transports as response_format.grammar (not the llama.cpp top-level field)", async () => {
|
|
871
|
+
test("fireworks: caller GBNF is not transported to the cloud API", async () => {
|
|
874
872
|
let body = "";
|
|
875
873
|
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
876
874
|
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
@@ -879,31 +877,57 @@ test("fireworks: a grammar transports as response_format.grammar (not the llama.
|
|
|
879
877
|
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "accounts/fireworks/models/deepseek-v4-pro");
|
|
880
878
|
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
881
879
|
const b = JSON.parse(body);
|
|
882
|
-
assert.
|
|
880
|
+
assert.equal("response_format" in b, false);
|
|
883
881
|
assert.equal("grammar" in b, false);
|
|
884
882
|
mock.restoreAll();
|
|
885
883
|
});
|
|
886
884
|
|
|
887
885
|
// — fireworks modelPrefix: the alias carries only the distinctive tail —
|
|
888
886
|
|
|
889
|
-
const
|
|
887
|
+
const fireworksWireBody = async (model: string, env: NodeJS.ProcessEnv = {}): Promise<Record<string, unknown>> => {
|
|
890
888
|
let body = "";
|
|
891
889
|
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
892
890
|
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
893
891
|
return new Response("{}", { status: 200 });
|
|
894
892
|
});
|
|
895
|
-
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" },
|
|
893
|
+
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", ...env }, model);
|
|
896
894
|
await p!.generate({ workerId: "r", messages: [] });
|
|
897
895
|
mock.restoreAll();
|
|
898
|
-
return JSON.parse(body)
|
|
896
|
+
return JSON.parse(body) as Record<string, unknown>;
|
|
899
897
|
};
|
|
900
898
|
|
|
901
899
|
test("fireworks: a bare alias is prefixed with accounts/fireworks/models/ on the wire", async () => {
|
|
902
|
-
assert.equal(await
|
|
900
|
+
assert.equal((await fireworksWireBody("deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
|
|
901
|
+
});
|
|
902
|
+
|
|
903
|
+
test("fireworks: fully qualified model and router ids are preserved verbatim", async () => {
|
|
904
|
+
assert.equal((await fireworksWireBody("accounts/fireworks/models/deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
|
|
905
|
+
assert.equal((await fireworksWireBody("accounts/fireworks/routers/glm-5p2-fast")).model, "accounts/fireworks/routers/glm-5p2-fast");
|
|
906
|
+
});
|
|
907
|
+
|
|
908
|
+
test("fireworks: configured service tier is fixed on the wire and invalid values fail hard", async () => {
|
|
909
|
+
const body = await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "priority" });
|
|
910
|
+
assert.equal(body.service_tier, "priority");
|
|
911
|
+
assert.equal((await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "flex" })).service_tier, "flex");
|
|
912
|
+
await assert.rejects(
|
|
913
|
+
standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "urgent" }, "deepseek-v4-pro"),
|
|
914
|
+
/PLURNK_PROVIDERS_SERVICE_TIER must be one of "auto", "default", "flex", "priority"/,
|
|
915
|
+
);
|
|
903
916
|
});
|
|
904
917
|
|
|
905
|
-
test("fireworks:
|
|
906
|
-
|
|
918
|
+
test("fireworks: configured tier wins over per-call sampling; unset retains per-call intent", async () => {
|
|
919
|
+
let bodies: Record<string, unknown>[] = [];
|
|
920
|
+
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
921
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
922
|
+
return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
923
|
+
});
|
|
924
|
+
const fixed = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "priority" }, "deepseek-v4-pro");
|
|
925
|
+
await fixed!.generate({ workerId: "r", messages: [], sampling: { service_tier: "default" } });
|
|
926
|
+
const flexible = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "deepseek-v4-pro");
|
|
927
|
+
await flexible!.generate({ workerId: "r", messages: [], sampling: { service_tier: "priority" } });
|
|
928
|
+
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "priority"]);
|
|
929
|
+
bodies = [];
|
|
930
|
+
mock.restoreAll();
|
|
907
931
|
});
|
|
908
932
|
|
|
909
933
|
test("#518 prompt_cache_key: default-ON for a standard provider (workerId), OFF for anthropic (cache_control)", async () => {
|
package/src/standardProviders.ts
CHANGED
|
@@ -59,21 +59,19 @@ type StandardProviderSpec = {
|
|
|
59
59
|
// not already include it.
|
|
60
60
|
flexBaseStrip?: boolean;
|
|
61
61
|
reasoningStyle: ReasoningStyle;
|
|
62
|
-
//
|
|
63
|
-
// probeNctx entry is upgraded to "llamacpp" when the probe sees a
|
|
64
|
-
// llama-server; cloud backends that support GBNF set their shape statically
|
|
65
|
-
// (fireworks → "response_format", verified live).
|
|
66
|
-
grammarStyle?: GrammarStyle;
|
|
67
|
-
// SSE streaming (default true). The streaming transport is dropped
|
|
68
|
-
// per-request only when it would break a feature (a response_format grammar
|
|
69
|
-
// arrives mislabeled as reasoning_content under fireworks' stream); leave
|
|
70
|
-
// unset to keep streaming on for every other call. See OpenAICompat.generate.
|
|
62
|
+
// SSE streaming (default true).
|
|
71
63
|
streaming?: boolean;
|
|
72
64
|
// Constant model-id prefix the backend requires but the alias shouldn't
|
|
73
65
|
// repeat (fireworks → "accounts/fireworks/models/", so the alias is just
|
|
74
66
|
// `fireworks/deepseek-v4-pro`). Prepended idempotently to form the wire id,
|
|
75
67
|
// which is ALSO the catalog key (models.dev keys fireworks-ai on the full id).
|
|
76
68
|
modelPrefix?: string;
|
|
69
|
+
// A fully qualified model namespace that bypasses modelPrefix. Fireworks
|
|
70
|
+
// uses sibling models/, routers/, and deployments/ resource collections.
|
|
71
|
+
qualifiedModelPrefix?: string;
|
|
72
|
+
// Fixed request tiers supported by this provider. An unset knob means the
|
|
73
|
+
// provider default; configured values are validated and sent every call.
|
|
74
|
+
serviceTiers?: readonly string[];
|
|
77
75
|
// First-party telemetry forwarding. ONLY the plurnk hosted endpoint sets
|
|
78
76
|
// this — it forwards the consumer's per-turn `attributions` (contributor
|
|
79
77
|
// credit) and `client` (originating frontend) as `Plurnk-Attribution` /
|
|
@@ -155,7 +153,11 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
|
|
|
155
153
|
fireworks: {
|
|
156
154
|
apiKeyVar: "FIREWORKS_API_KEY", apiKeyRequired: true,
|
|
157
155
|
baseUrlVar: "FIREWORKS_BASE_URL", chatPath: "/chat/completions",
|
|
158
|
-
reasoningStyle: "effort_explicit",
|
|
156
|
+
reasoningStyle: "effort_explicit",
|
|
157
|
+
modelPrefix: "accounts/fireworks/models/",
|
|
158
|
+
qualifiedModelPrefix: "accounts/fireworks/",
|
|
159
|
+
serviceTiers: ["auto", "default", "flex", "priority"],
|
|
160
|
+
tokenizerEnvVar: "FIREWORKS_TOKENIZER",
|
|
159
161
|
},
|
|
160
162
|
deepinfra: {
|
|
161
163
|
apiKeyVar: ["DEEPINFRA_API_KEY", "DEEPINFRA_API_TOKEN", "DEEPINFRA_TOKEN"], apiKeyRequired: true,
|
|
@@ -289,7 +291,7 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
|
|
|
289
291
|
apiKeyVar: "PLURNK_API_KEY", apiKeyRequired: true,
|
|
290
292
|
apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
|
|
291
293
|
apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
|
|
292
|
-
reasoningStyle: "none",
|
|
294
|
+
reasoningStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
|
|
293
295
|
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, balanceMetaKey: "balance_pico", suppressTuningFloors: true,
|
|
294
296
|
},
|
|
295
297
|
});
|
|
@@ -417,7 +419,8 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
417
419
|
// The on-the-wire model id: a backend-required constant prefix (fireworks)
|
|
418
420
|
// prepended idempotently, so the operator's alias carries only the distinctive
|
|
419
421
|
// tail. This id is what the backend, the probe, AND the catalog key on.
|
|
420
|
-
const
|
|
422
|
+
const qualified = spec.qualifiedModelPrefix !== undefined && model.startsWith(spec.qualifiedModelPrefix);
|
|
423
|
+
const wireModel = spec.modelPrefix !== undefined && !qualified && !model.startsWith(spec.modelPrefix)
|
|
421
424
|
? `${spec.modelPrefix}${model}`
|
|
422
425
|
: model;
|
|
423
426
|
|
|
@@ -444,17 +447,29 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
444
447
|
);
|
|
445
448
|
const url = resolveUrl(spec, env, name, baseUrlOverride);
|
|
446
449
|
const fetchTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name);
|
|
450
|
+
const streamIdleTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name);
|
|
451
|
+
const serviceTier = (() => {
|
|
452
|
+
const raw = env.PLURNK_PROVIDERS_SERVICE_TIER;
|
|
453
|
+
if (raw === undefined || raw.length === 0) return undefined;
|
|
454
|
+
if (spec.serviceTiers === undefined) {
|
|
455
|
+
throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER is not supported`);
|
|
456
|
+
}
|
|
457
|
+
if (!spec.serviceTiers.includes(raw)) {
|
|
458
|
+
throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER must be one of ${spec.serviceTiers.map((tier) => JSON.stringify(tier)).join(", ")} (got "${raw}")`);
|
|
459
|
+
}
|
|
460
|
+
return raw;
|
|
461
|
+
})();
|
|
447
462
|
|
|
448
463
|
// The probe always runs for probeNctx specs — grammar capability must not
|
|
449
464
|
// hinge on whether the operator pinned PLURNK_PROVIDERS_CONTEXT_WINDOW. For
|
|
450
465
|
// contextWindow itself, explicit env still wins over the probed n_ctx.
|
|
451
466
|
let contextWindow = contextWindowFromEnv(env, name);
|
|
452
|
-
//
|
|
453
|
-
//
|
|
467
|
+
// GBNF transport is upgraded to "llamacpp" only when the probe fingerprints
|
|
468
|
+
// a llama-server. Slot
|
|
454
469
|
// pinning is llama-server-only, so it keys on that same fingerprint. A spec
|
|
455
470
|
// can opt out of the fingerprint entirely (detectLlamaServer: false → plurnk)
|
|
456
471
|
// to read the window but stay a plain remote OpenAI server.
|
|
457
|
-
let grammarStyle: GrammarStyle =
|
|
472
|
+
let grammarStyle: GrammarStyle = "none";
|
|
458
473
|
let supportsSlotPinning = false;
|
|
459
474
|
let slotCount: number | null = null;
|
|
460
475
|
let eosText: string | undefined;
|
|
@@ -500,7 +515,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
500
515
|
// un-upgraded, but NEVER silently (#34) — rails going dark without a
|
|
501
516
|
// signal cost the consumer weeks of misattributed rambles.
|
|
502
517
|
emitWarningOnce(
|
|
503
|
-
`${name} provider: llama-server detection failed after ${probeAttempts} attempts —
|
|
518
|
+
`${name} provider: llama-server detection failed after ${probeAttempts} attempts — local capabilities are unknown. If this endpoint is a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
|
|
504
519
|
"PLURNK_PROBE_FAILED",
|
|
505
520
|
);
|
|
506
521
|
}
|
|
@@ -577,6 +592,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
577
592
|
headers,
|
|
578
593
|
contextWindow,
|
|
579
594
|
fetchTimeoutMs,
|
|
595
|
+
streamIdleTimeoutMs,
|
|
580
596
|
reasoning,
|
|
581
597
|
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
582
598
|
repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
@@ -607,6 +623,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
|
|
|
607
623
|
firstPartyMetadata: spec.firstPartyMetadata,
|
|
608
624
|
apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
|
|
609
625
|
promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
|
|
626
|
+
serviceTier,
|
|
610
627
|
balanceMetaKey: spec.balanceMetaKey,
|
|
611
628
|
supportsSlotPinning,
|
|
612
629
|
slotCount,
|