@plurnk/plurnk-providers 1.3.3 → 1.3.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -29,17 +29,15 @@ import { emitWarningOnce } from "./warnings.ts";
29
29
  // (canary-verified; the #32 clamp is lifted — it caused the plan-less service#331).
30
30
  export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" | "effort_explicit" | "template" | "anthropic";
31
31
 
32
- // How a caller-supplied GBNF grammar is carried on the wire — backends accept
33
- // different shapes for the SAME GBNF (probed/configured, never guessed; §13); the
34
- // wire shape per style lives in #grammarBody. "none" means the grammar is NOT sent
35
- // (never silently — so a constrained consumer can't mistake unconstrained output
36
- // for enforced).
37
- export type GrammarStyle = "none" | "llamacpp" | "response_format";
32
+ // GBNF transport is a local llama-server capability. "none" means no
33
+ // service-managed constrained sampling; endpoint-owned settings are not inferred.
34
+ export type GrammarStyle = "none" | "llamacpp";
38
35
 
39
36
  export type OpenAICompatConfig = {
40
37
  model: string;
41
38
  url: string; // fully-resolved chat-completions URL
42
39
  fetchTimeoutMs: number;
40
+ streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
43
41
  headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
44
42
  fetch?: ProviderFetch; // per-instance request executor; default globalThis.fetch
45
43
  contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
@@ -54,6 +52,9 @@ export type OpenAICompatConfig = {
54
52
  // false -- a backend that strict-validates unknown fields 400s, so enable only
55
53
  // where the field is accepted. Same identity that already drives slot affinity.
56
54
  promptCacheKey?: boolean;
55
+ // Optional provider-configured service tier. Unlike caller sampling, this is
56
+ // a fixed deployment choice and therefore wins on every request.
57
+ serviceTier?: string;
57
58
  gbnfDebug?: boolean; // PLURNK_PROVIDERS_GBNF_DEBUG: validate the grammar locally + throw on invalid, but DON'T transport it (run unconstrained); default false
58
59
  streaming?: boolean; // SSE transport (default true); false → one non-streamed JSON
59
60
  firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
@@ -238,6 +239,7 @@ export default class OpenAICompatProvider implements Provider {
238
239
  #model: string;
239
240
  #url: string;
240
241
  #fetchTimeoutMs: number;
242
+ #streamIdleTimeoutMs: number | undefined;
241
243
  #headers: Record<string, string>;
242
244
  #fetch: ProviderFetch;
243
245
  #hasApiKey = false;
@@ -259,6 +261,7 @@ export default class OpenAICompatProvider implements Provider {
259
261
  #source: string;
260
262
  #grammarStyle: GrammarStyle;
261
263
  #promptCacheKey: boolean;
264
+ #serviceTier: string | undefined;
262
265
  #gbnfDebug: boolean;
263
266
  #streaming: boolean;
264
267
  #firstPartyMetadata: boolean;
@@ -284,6 +287,7 @@ export default class OpenAICompatProvider implements Provider {
284
287
  this.#model = config.model;
285
288
  this.#url = config.url;
286
289
  this.#fetchTimeoutMs = config.fetchTimeoutMs;
290
+ this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
287
291
  this.#headers = config.headers ?? {};
288
292
  this.#fetch = config.fetch ?? ((input, init) => globalThis.fetch(input, init));
289
293
  this.#contextWindow = config.contextWindow ?? null;
@@ -309,6 +313,7 @@ export default class OpenAICompatProvider implements Provider {
309
313
  this.#source = config.source ?? "provider";
310
314
  this.#grammarStyle = config.grammarStyle ?? "none";
311
315
  this.#promptCacheKey = config.promptCacheKey ?? false;
316
+ this.#serviceTier = config.serviceTier;
312
317
  this.#gbnfDebug = config.gbnfDebug ?? false;
313
318
  this.#streaming = config.streaming ?? true;
314
319
  this.#firstPartyMetadata = config.firstPartyMetadata ?? false;
@@ -444,31 +449,25 @@ export default class OpenAICompatProvider implements Provider {
444
449
  return { id_slot: slot };
445
450
  }
446
451
 
447
- // Grammar transport (SPEC §13): carry the caller-supplied GBNF in the shape
448
- // the backend accepts. Same grammar, different wire field per backend; an
449
- // unsupported/unknown backend sends NO field at all (cloud APIs 400 on
450
- // unknowns, and a silent send would let a constrained consumer mistake
451
- // unconstrained output for enforced).
452
+ // Optional local llama-server GBNF transport (SPEC §13). Unsupported
453
+ // backends receive no grammar-related field.
452
454
  #grammarBody(grammar: string | undefined): Record<string, unknown> {
453
455
  if (grammar === undefined) return {};
454
456
  switch (this.#grammarStyle) {
455
457
  // Greedy decoding under hard constraint loops without a repeat-penalty
456
- // floor (#9, SPEC §13) — every grammar path carries it. llama.cpp spells
457
- // it `repeat_penalty`; the OpenAI-compat (Fireworks) shape is `repetition_penalty`
458
- // (verified honored live, #20).
458
+ // floor (#9, SPEC §13) — llama.cpp spells it `repeat_penalty`.
459
459
  case "llamacpp": return { grammar, repeat_penalty: this.#repeatPenalty };
460
- case "response_format": return { response_format: { type: "grammar", grammar }, repetition_penalty: this.#repeatPenalty };
461
460
  case "none": return {};
462
461
  }
463
462
  }
464
463
 
465
464
  // Anti-degeneration DEFAULT on EVERY request (#426), keyed to the backend's wire
466
- // convention - NOT grammar-bound. GBNF is a local rail (off for cloud), so a cloud
465
+ // convention - NOT grammar-bound. GBNF is a local constraint, so a cloud
467
466
  // alias runs the sampler bare: firefast (deepseek/fireworks) ran 4/86 bench turns
468
467
  // straight to the token cap on pure looped repetition (run52). Ships next to
469
468
  // temperature so caller `sampling` can tune it; the grammar path re-asserts it as a
470
- // managed FLOOR in #grammarBody. Per backend: llama.cpp/response_format take the
471
- // repeat_penalty MULTIPLIER; the plain cloud path ("none") can't, so it gets
469
+ // managed FLOOR in #grammarBody. llama.cpp takes the repeat_penalty
470
+ // MULTIPLIER; the plain cloud path ("none") can't, so it gets
472
471
  // frequency_penalty - OpenAI-standard, accepted by every OpenAI-compat backend (verified
473
472
  // live: together/deepinfra/fireworks; it is OpenAI's own param, so real OpenAI takes it too).
474
473
  #repetitionPenaltyBody(): Record<string, unknown> {
@@ -485,7 +484,6 @@ export default class OpenAICompatProvider implements Provider {
485
484
  ...(this.#dryAllowedLength !== undefined ? { dry_allowed_length: this.#dryAllowedLength } : {}),
486
485
  } : {}),
487
486
  };
488
- case "response_format": return { repetition_penalty: this.#repeatPenalty };
489
487
  case "none": return this.#frequencyPenalty > 0 ? { frequency_penalty: this.#frequencyPenalty } : {};
490
488
  }
491
489
  }
@@ -620,6 +618,7 @@ export default class OpenAICompatProvider implements Provider {
620
618
  // the router's per-model tuning must not be overridden by client floors.
621
619
  ...(this.#tuningFloors ? { temperature: this.#temperature, ...this.#repetitionPenaltyBody() } : {}),
622
620
  ...this.#samplingBody(sampling),
621
+ ...(this.#serviceTier !== undefined ? { service_tier: this.#serviceTier } : {}),
623
622
  model: this.#model,
624
623
  messages,
625
624
  ...this.#reasoningBody(),
@@ -640,23 +639,27 @@ export default class OpenAICompatProvider implements Provider {
640
639
  // signal spans them all. Retry only the transient classifications, prefer
641
640
  // a server Retry-After over the backoff, and let the caller's abort cut
642
641
  // through both the in-flight request and the backoff sleep.
643
- // Stream by default, but fall back to one non-streamed JSON for the one
644
- // case it breaks: a response_format grammar (fireworks) streams its
645
- // constrained output mislabeled as reasoning_content, yet returns it as
646
- // content non-streamed. The atomic dump is correct either way, so the
647
- // demotion is scoped to exactly that request, not the whole provider.
648
- const grammarBreaksStream = sendGrammar !== undefined && this.#grammarStyle === "response_format";
649
- const transport = this.#streaming && !grammarBreaksStream ? chatCompletionStream : chatCompletion;
642
+ const transport = this.#streaming ? chatCompletionStream : chatCompletion;
650
643
 
651
644
  // Per-request headers = static auth/routing + any first-party telemetry.
652
645
  const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
653
646
  const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
647
+ const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
654
648
  let raw;
655
649
  for (let attempt = 0; ; attempt++) {
650
+ const attemptStarted = performance.now();
656
651
  const timeoutSignal = AbortSignal.timeout(this.#fetchTimeoutMs);
657
652
  const effectiveSignal = signal !== undefined ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
658
653
  try {
659
- raw = await transport({ url: this.#url, headers, body, signal: effectiveSignal, fetch: this.#fetch, captureRawBody: this.#rawBody });
654
+ raw = await transport({
655
+ url: this.#url,
656
+ headers,
657
+ body,
658
+ signal: effectiveSignal,
659
+ fetch: this.#fetch,
660
+ captureRawBody: this.#rawBody,
661
+ streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
662
+ });
660
663
  break;
661
664
  } catch (err) {
662
665
  // Caller-initiated abort is cancellation — never retried or wrapped.
@@ -677,6 +680,12 @@ export default class OpenAICompatProvider implements Provider {
677
680
  }
678
681
  throw pe;
679
682
  }
683
+ transportRetries.push({
684
+ attempt: attempt + 1,
685
+ kind,
686
+ elapsedMs: Math.round(performance.now() - attemptStarted),
687
+ message: err instanceof Error ? err.message : String(err),
688
+ });
680
689
  const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
681
690
  await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
682
691
  }
@@ -728,7 +737,10 @@ export default class OpenAICompatProvider implements Provider {
728
737
  }
729
738
 
730
739
  const builtMeta = this.#buildMeta(raw.chunkMetadata);
731
- const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
740
+ const retryMeta = transportRetries.length > 0 ? { transportRetries } : undefined;
741
+ const meta = railsMeta !== undefined || retryMeta !== undefined
742
+ ? { ...builtMeta, ...railsMeta, ...retryMeta }
743
+ : builtMeta;
732
744
 
733
745
  // #36: surface per-token logprobs + their mean when the backend returned
734
746
  // them (only possible when the flag requested them). Absent otherwise —
@@ -14,6 +14,7 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
14
14
 
15
15
  const fullEnv = Object.freeze({
16
16
  PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
17
+ PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
17
18
  PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
18
19
  OPENAI_BASE_URL: "http://x",
19
20
  });
@@ -162,6 +163,34 @@ test("instantiateProvider: per-alias knobs scope through to the provider (per-al
162
163
  mock.restoreAll();
163
164
  });
164
165
 
166
+ test("#622: two Fireworks aliases independently select default and priority service tiers", async () => {
167
+ const bodies: Record<string, unknown>[] = [];
168
+ mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
169
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
170
+ return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
171
+ });
172
+ const env = {
173
+ ...fullEnv,
174
+ FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
175
+ FIREWORKS_API_KEY: "fw",
176
+ PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
177
+ PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
178
+ PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
179
+ };
180
+ const imports = async () => ({});
181
+ const discover = async () => ({ registry: new Map(), skipped: new Map(), attributions: new Map() });
182
+ const fast = await instantiateProvider("fireworks", env, "accounts/fireworks/routers/glm-5p2-fast", imports, discover, undefined, "fast");
183
+ const standard = await instantiateProvider("fireworks", env, "deepseek-v4-pro", imports, discover, undefined, "standard");
184
+ await fast.generate({ workerId: "fast-worker", messages: [] });
185
+ await standard.generate({ workerId: "standard-worker", messages: [] });
186
+ assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
187
+ assert.deepEqual(bodies.map((body) => body.model), [
188
+ "accounts/fireworks/routers/glm-5p2-fast",
189
+ "accounts/fireworks/models/deepseek-v4-pro",
190
+ ]);
191
+ mock.restoreAll();
192
+ });
193
+
165
194
  test("loadActiveProvider: resolves the alias cascade end-to-end via the scan", async () => {
166
195
  resetDiscoveryCache();
167
196
  const env = { ...fullEnv, PLURNK_MODEL: "opus", PLURNK_MODEL_opus: "openrouter/anthropic/claude-opus-latest" } as NodeJS.ProcessEnv;
package/src/env.test.ts CHANGED
@@ -166,6 +166,14 @@ test("#399: the shipped floor activates reasoning by default (adaptive — owner
166
166
  assert.ok(!defaults.match(/^PLURNK_PROVIDERS_REASONING_BUDGET=/m), "no shipped magnitude — budget is on-mode only");
167
167
  });
168
168
 
169
+ test("#567: the shipped DRY floor stays off while retaining the measured alias-safe shape", async () => {
170
+ const { readFileSync } = await import("node:fs");
171
+ const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
172
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "one-model tuning is not a universal sampler floor");
173
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_BASE=1\.75$/m);
174
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32$/m, "the measured identifier-safe shape remains available for alias opt-in");
175
+ });
176
+
169
177
  // -- #507: envelope reserves (owner-ruled migration from PLURNK_SERVICE_*) --
170
178
 
171
179
  test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
package/src/env.ts CHANGED
@@ -163,10 +163,12 @@ export const PROVIDERS_KNOBS = Object.freeze([
163
163
  "PLURNK_PROVIDERS_CONTEXT_WINDOW",
164
164
  "PLURNK_PROVIDERS_RETRY_ATTEMPTS",
165
165
  "PLURNK_PROVIDERS_FETCH_TIMEOUT",
166
+ "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
166
167
  "PLURNK_PROVIDERS_LLAMA_SERVER",
167
168
  "PLURNK_PROVIDERS_TEMPERATURE",
168
169
  "PLURNK_PROVIDERS_REPEAT_PENALTY",
169
170
  "PLURNK_PROVIDERS_FREQUENCY_PENALTY",
171
+ "PLURNK_PROVIDERS_SERVICE_TIER",
170
172
  "PLURNK_PROVIDERS_REPEAT_LAST_N",
171
173
  "PLURNK_PROVIDERS_DRY_MULTIPLIER",
172
174
  "PLURNK_PROVIDERS_DRY_BASE",
package/src/index.ts CHANGED
@@ -36,7 +36,7 @@ export type { OpenAICompatConfig, ReasoningStyle, GrammarStyle } from "./OpenAIC
36
36
  // worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
37
37
  // DECISION stays the consumer's, by choosing which pool to call.
38
38
  export { default as Pool } from "./Pool.ts";
39
- export { chatCompletionStream, chatCompletion, OpenAiHttpError } from "./openaiStream.ts";
39
+ export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
40
40
  export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
41
41
  export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
42
42
  export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
package/src/openai.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  export { default as OpenAICompatProvider, effortFromBudget } from "./OpenAICompat.ts";
2
2
  export type { GrammarStyle, OpenAICompatConfig, ReasoningStyle } from "./OpenAICompat.ts";
3
- export { chatCompletion, chatCompletionStream, OpenAiHttpError } from "./openaiStream.ts";
3
+ export { chatCompletion, chatCompletionStream, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
4
4
  export type {
5
5
  EncryptedReasoningItem,
6
6
  ProviderFetch,
@@ -13,6 +13,9 @@ type StreamRequest = {
13
13
  // #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
14
14
  // default so a serving turn never pays the reassembly/retention cost.
15
15
  captureRawBody?: boolean;
16
+ // Maximum silence between streamed response-body chunks. Undefined/zero
17
+ // disables this clock; the caller's signal still owns the total deadline.
18
+ streamIdleTimeoutMs?: number;
16
19
  };
17
20
 
18
21
  import type { RawUsage } from "./usage.ts";
@@ -122,6 +125,15 @@ export class OpenAiHttpError extends Error {
122
125
  }
123
126
  }
124
127
 
128
+ export class StreamIdleError extends Error {
129
+ readonly timeoutMs: number;
130
+ constructor(timeoutMs: number) {
131
+ super(`stream received no body bytes for ${timeoutMs}ms`);
132
+ this.name = "StreamIdleError";
133
+ this.timeoutMs = timeoutMs;
134
+ }
135
+ }
136
+
125
137
  const parseRetryAfter = (header: string | null): number | null => {
126
138
  if (header === null) return null;
127
139
  const asInt = Number.parseInt(header, 10);
@@ -170,7 +182,7 @@ export const chatCompletion = async ({ url, headers, body, signal, fetch, captur
170
182
  };
171
183
  };
172
184
 
173
- export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
185
+ export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
174
186
  const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
175
187
 
176
188
  const response = await fetch(url, {
@@ -206,7 +218,23 @@ export const chatCompletionStream = async ({ url, headers, body, signal, fetch,
206
218
  let encryptedNoKey = 0;
207
219
 
208
220
  while (true) {
209
- const { done, value } = await reader.read();
221
+ const read = reader.read();
222
+ let timer: ReturnType<typeof setTimeout> | undefined;
223
+ const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
224
+ ? new Promise<never>((_resolve, reject) => {
225
+ timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
226
+ })
227
+ : null;
228
+ let result: Awaited<ReturnType<typeof reader.read>>;
229
+ try {
230
+ result = idle === null ? await read : await Promise.race([read, idle]);
231
+ } catch (err) {
232
+ if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
233
+ throw err;
234
+ } finally {
235
+ if (timer !== undefined) clearTimeout(timer);
236
+ }
237
+ const { done, value } = result;
210
238
  if (done) break;
211
239
  buffer += decoder.decode(value, { stream: true });
212
240
  const lines = buffer.split("\n");
@@ -6,7 +6,7 @@ import { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from
6
6
  // defaults for the providers exercised outside the coverage loop. `openai` is
7
7
  // deliberately omitted so its missing-base fail-hard test still fires.
8
8
  const baseEnv = Object.freeze({
9
- PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
9
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
10
10
  GROQ_BASE_URL: "https://api.groq.com/openai/v1",
11
11
  DEEPINFRA_BASE_URL: "https://api.deepinfra.com/v1/openai",
12
12
  FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
@@ -298,10 +298,10 @@ test("openai: a garbage PLURNK_PROVIDERS_LLAMA_SERVER value fails hard", async (
298
298
  );
299
299
  });
300
300
 
301
- test("constrainsOutput: fireworks (static response_format) reports true; groq reports false", async () => {
301
+ test("constrainsOutput: cloud providers do not claim local GBNF transport", async () => {
302
302
  mockEndpoint();
303
303
  const fw = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
304
- assert.equal(fw!.constrainsOutput, true);
304
+ assert.equal(fw!.constrainsOutput, false);
305
305
  const gq = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
306
306
  assert.equal(gq!.constrainsOutput, false);
307
307
  });
@@ -868,9 +868,7 @@ test("plurnk: reads its window from upstream but stays a plain OpenAI client —
868
868
  mock.restoreAll();
869
869
  });
870
870
 
871
- // — fireworks carries GBNF via response_format.grammar (cloud GBNF, #grammarStyle) —
872
-
873
- test("fireworks: a grammar transports as response_format.grammar (not the llama.cpp top-level field)", async () => {
871
+ test("fireworks: caller GBNF is not transported to the cloud API", async () => {
874
872
  let body = "";
875
873
  mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
876
874
  if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
@@ -879,31 +877,57 @@ test("fireworks: a grammar transports as response_format.grammar (not the llama.
879
877
  const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "accounts/fireworks/models/deepseek-v4-pro");
880
878
  await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
881
879
  const b = JSON.parse(body);
882
- assert.deepEqual(b.response_format, { type: "grammar", grammar: 'root ::= "ok"' });
880
+ assert.equal("response_format" in b, false);
883
881
  assert.equal("grammar" in b, false);
884
882
  mock.restoreAll();
885
883
  });
886
884
 
887
885
  // — fireworks modelPrefix: the alias carries only the distinctive tail —
888
886
 
889
- const fireworksWireModel = async (alias: string): Promise<string> => {
887
+ const fireworksWireBody = async (model: string, env: NodeJS.ProcessEnv = {}): Promise<Record<string, unknown>> => {
890
888
  let body = "";
891
889
  mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
892
890
  if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
893
891
  return new Response("{}", { status: 200 });
894
892
  });
895
- const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, alias);
893
+ const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", ...env }, model);
896
894
  await p!.generate({ workerId: "r", messages: [] });
897
895
  mock.restoreAll();
898
- return JSON.parse(body).model;
896
+ return JSON.parse(body) as Record<string, unknown>;
899
897
  };
900
898
 
901
899
  test("fireworks: a bare alias is prefixed with accounts/fireworks/models/ on the wire", async () => {
902
- assert.equal(await fireworksWireModel("deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
900
+ assert.equal((await fireworksWireBody("deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
901
+ });
902
+
903
+ test("fireworks: fully qualified model and router ids are preserved verbatim", async () => {
904
+ assert.equal((await fireworksWireBody("accounts/fireworks/models/deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
905
+ assert.equal((await fireworksWireBody("accounts/fireworks/routers/glm-5p2-fast")).model, "accounts/fireworks/routers/glm-5p2-fast");
906
+ });
907
+
908
+ test("fireworks: configured service tier is fixed on the wire and invalid values fail hard", async () => {
909
+ const body = await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "priority" });
910
+ assert.equal(body.service_tier, "priority");
911
+ assert.equal((await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "flex" })).service_tier, "flex");
912
+ await assert.rejects(
913
+ standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "urgent" }, "deepseek-v4-pro"),
914
+ /PLURNK_PROVIDERS_SERVICE_TIER must be one of "auto", "default", "flex", "priority"/,
915
+ );
903
916
  });
904
917
 
905
- test("fireworks: an already-prefixed id is left unchanged (idempotent prepend)", async () => {
906
- assert.equal(await fireworksWireModel("accounts/fireworks/models/deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
918
+ test("fireworks: configured tier wins over per-call sampling; unset retains per-call intent", async () => {
919
+ let bodies: Record<string, unknown>[] = [];
920
+ mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
921
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
922
+ return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
923
+ });
924
+ const fixed = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "priority" }, "deepseek-v4-pro");
925
+ await fixed!.generate({ workerId: "r", messages: [], sampling: { service_tier: "default" } });
926
+ const flexible = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "deepseek-v4-pro");
927
+ await flexible!.generate({ workerId: "r", messages: [], sampling: { service_tier: "priority" } });
928
+ assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "priority"]);
929
+ bodies = [];
930
+ mock.restoreAll();
907
931
  });
908
932
 
909
933
  test("#518 prompt_cache_key: default-ON for a standard provider (workerId), OFF for anthropic (cache_control)", async () => {
@@ -59,21 +59,19 @@ type StandardProviderSpec = {
59
59
  // not already include it.
60
60
  flexBaseStrip?: boolean;
61
61
  reasoningStyle: ReasoningStyle;
62
- // How this backend carries a GBNF grammar (default "none" — not sent). A
63
- // probeNctx entry is upgraded to "llamacpp" when the probe sees a
64
- // llama-server; cloud backends that support GBNF set their shape statically
65
- // (fireworks → "response_format", verified live).
66
- grammarStyle?: GrammarStyle;
67
- // SSE streaming (default true). The streaming transport is dropped
68
- // per-request only when it would break a feature (a response_format grammar
69
- // arrives mislabeled as reasoning_content under fireworks' stream); leave
70
- // unset to keep streaming on for every other call. See OpenAICompat.generate.
62
+ // SSE streaming (default true).
71
63
  streaming?: boolean;
72
64
  // Constant model-id prefix the backend requires but the alias shouldn't
73
65
  // repeat (fireworks → "accounts/fireworks/models/", so the alias is just
74
66
  // `fireworks/deepseek-v4-pro`). Prepended idempotently to form the wire id,
75
67
  // which is ALSO the catalog key (models.dev keys fireworks-ai on the full id).
76
68
  modelPrefix?: string;
69
+ // A fully qualified model namespace that bypasses modelPrefix. Fireworks
70
+ // uses sibling models/, routers/, and deployments/ resource collections.
71
+ qualifiedModelPrefix?: string;
72
+ // Fixed request tiers supported by this provider. An unset knob means the
73
+ // provider default; configured values are validated and sent every call.
74
+ serviceTiers?: readonly string[];
77
75
  // First-party telemetry forwarding. ONLY the plurnk hosted endpoint sets
78
76
  // this — it forwards the consumer's per-turn `attributions` (contributor
79
77
  // credit) and `client` (originating frontend) as `Plurnk-Attribution` /
@@ -155,7 +153,11 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
155
153
  fireworks: {
156
154
  apiKeyVar: "FIREWORKS_API_KEY", apiKeyRequired: true,
157
155
  baseUrlVar: "FIREWORKS_BASE_URL", chatPath: "/chat/completions",
158
- reasoningStyle: "effort_explicit", grammarStyle: "response_format", modelPrefix: "accounts/fireworks/models/", tokenizerEnvVar: "FIREWORKS_TOKENIZER",
156
+ reasoningStyle: "effort_explicit",
157
+ modelPrefix: "accounts/fireworks/models/",
158
+ qualifiedModelPrefix: "accounts/fireworks/",
159
+ serviceTiers: ["auto", "default", "flex", "priority"],
160
+ tokenizerEnvVar: "FIREWORKS_TOKENIZER",
159
161
  },
160
162
  deepinfra: {
161
163
  apiKeyVar: ["DEEPINFRA_API_KEY", "DEEPINFRA_API_TOKEN", "DEEPINFRA_TOKEN"], apiKeyRequired: true,
@@ -289,7 +291,7 @@ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>>
289
291
  apiKeyVar: "PLURNK_API_KEY", apiKeyRequired: true,
290
292
  apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
291
293
  apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
292
- reasoningStyle: "none", grammarStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
294
+ reasoningStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
293
295
  probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, balanceMetaKey: "balance_pico", suppressTuningFloors: true,
294
296
  },
295
297
  });
@@ -417,7 +419,8 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
417
419
  // The on-the-wire model id: a backend-required constant prefix (fireworks)
418
420
  // prepended idempotently, so the operator's alias carries only the distinctive
419
421
  // tail. This id is what the backend, the probe, AND the catalog key on.
420
- const wireModel = spec.modelPrefix !== undefined && !model.startsWith(spec.modelPrefix)
422
+ const qualified = spec.qualifiedModelPrefix !== undefined && model.startsWith(spec.qualifiedModelPrefix);
423
+ const wireModel = spec.modelPrefix !== undefined && !qualified && !model.startsWith(spec.modelPrefix)
421
424
  ? `${spec.modelPrefix}${model}`
422
425
  : model;
423
426
 
@@ -444,17 +447,29 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
444
447
  );
445
448
  const url = resolveUrl(spec, env, name, baseUrlOverride);
446
449
  const fetchTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name);
450
+ const streamIdleTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT, "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT", name);
451
+ const serviceTier = (() => {
452
+ const raw = env.PLURNK_PROVIDERS_SERVICE_TIER;
453
+ if (raw === undefined || raw.length === 0) return undefined;
454
+ if (spec.serviceTiers === undefined) {
455
+ throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER is not supported`);
456
+ }
457
+ if (!spec.serviceTiers.includes(raw)) {
458
+ throw new Error(`${name} provider: PLURNK_PROVIDERS_SERVICE_TIER must be one of ${spec.serviceTiers.map((tier) => JSON.stringify(tier)).join(", ")} (got "${raw}")`);
459
+ }
460
+ return raw;
461
+ })();
447
462
 
448
463
  // The probe always runs for probeNctx specs — grammar capability must not
449
464
  // hinge on whether the operator pinned PLURNK_PROVIDERS_CONTEXT_WINDOW. For
450
465
  // contextWindow itself, explicit env still wins over the probed n_ctx.
451
466
  let contextWindow = contextWindowFromEnv(env, name);
452
- // Grammar shape: a static spec choice (e.g. fireworks → "response_format"),
453
- // upgraded to "llamacpp" when the probe fingerprints a llama-server. Slot
467
+ // GBNF transport is upgraded to "llamacpp" only when the probe fingerprints
468
+ // a llama-server. Slot
454
469
  // pinning is llama-server-only, so it keys on that same fingerprint. A spec
455
470
  // can opt out of the fingerprint entirely (detectLlamaServer: false → plurnk)
456
471
  // to read the window but stay a plain remote OpenAI server.
457
- let grammarStyle: GrammarStyle = spec.grammarStyle ?? "none";
472
+ let grammarStyle: GrammarStyle = "none";
458
473
  let supportsSlotPinning = false;
459
474
  let slotCount: number | null = null;
460
475
  let eosText: string | undefined;
@@ -500,7 +515,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
500
515
  // un-upgraded, but NEVER silently (#34) — rails going dark without a
501
516
  // signal cost the consumer weeks of misattributed rambles.
502
517
  emitWarningOnce(
503
- `${name} provider: llama-server detection failed after ${probeAttempts} attempts — grammar transport stays OFF (grammarStyle "none"). If this endpoint IS a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
518
+ `${name} provider: llama-server detection failed after ${probeAttempts} attempts — local capabilities are unknown. If this endpoint is a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
504
519
  "PLURNK_PROBE_FAILED",
505
520
  );
506
521
  }
@@ -577,6 +592,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
577
592
  headers,
578
593
  contextWindow,
579
594
  fetchTimeoutMs,
595
+ streamIdleTimeoutMs,
580
596
  reasoning,
581
597
  temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
582
598
  repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
@@ -607,6 +623,7 @@ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessE
607
623
  firstPartyMetadata: spec.firstPartyMetadata,
608
624
  apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
609
625
  promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
626
+ serviceTier,
610
627
  balanceMetaKey: spec.balanceMetaKey,
611
628
  supportsSlotPinning,
612
629
  slotCount,