@plurnk/plurnk-providers 1.3.3 → 1.3.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.env.defaults +20 -21
  2. package/SPEC.md +65 -44
  3. package/dist/Mock.d.ts +1 -1
  4. package/dist/Mock.d.ts.map +1 -1
  5. package/dist/Mock.js +1 -1
  6. package/dist/Mock.js.map +1 -1
  7. package/dist/OpenAICompat.d.ts +5 -4
  8. package/dist/OpenAICompat.d.ts.map +1 -1
  9. package/dist/OpenAICompat.js +38 -38
  10. package/dist/OpenAICompat.js.map +1 -1
  11. package/dist/Pool.d.ts +1 -1
  12. package/dist/Pool.d.ts.map +1 -1
  13. package/dist/Pool.js +1 -1
  14. package/dist/Pool.js.map +1 -1
  15. package/dist/env.d.ts.map +1 -1
  16. package/dist/env.js +2 -0
  17. package/dist/env.js.map +1 -1
  18. package/dist/index.d.ts +2 -2
  19. package/dist/index.d.ts.map +1 -1
  20. package/dist/index.js +2 -2
  21. package/dist/index.js.map +1 -1
  22. package/dist/openai.d.ts +1 -1
  23. package/dist/openai.d.ts.map +1 -1
  24. package/dist/openai.js +1 -1
  25. package/dist/openai.js.map +1 -1
  26. package/dist/openaiStream.d.ts +6 -1
  27. package/dist/openaiStream.d.ts.map +1 -1
  28. package/dist/openaiStream.js +30 -2
  29. package/dist/openaiStream.js.map +1 -1
  30. package/dist/standardProviders.d.ts +3 -3
  31. package/dist/standardProviders.d.ts.map +1 -1
  32. package/dist/standardProviders.js +36 -17
  33. package/dist/standardProviders.js.map +1 -1
  34. package/dist/types.d.ts +1 -1
  35. package/dist/types.d.ts.map +1 -1
  36. package/dist/usage.d.ts +1 -1
  37. package/dist/usage.d.ts.map +1 -1
  38. package/dist/usage.js +4 -2
  39. package/dist/usage.js.map +1 -1
  40. package/package.json +10 -7
  41. package/src/Mock.test.ts +2 -2
  42. package/src/Mock.ts +1 -1
  43. package/src/OpenAICompat.test.ts +86 -86
  44. package/src/OpenAICompat.ts +46 -45
  45. package/src/Pool.test.ts +3 -3
  46. package/src/Pool.ts +1 -1
  47. package/src/ProviderRegistry.test.ts +30 -1
  48. package/src/aiSdkAdapter.spike.test.ts +242 -0
  49. package/src/env.test.ts +8 -0
  50. package/src/env.ts +2 -0
  51. package/src/index.ts +2 -2
  52. package/src/openai.ts +1 -1
  53. package/src/openaiStream.ts +30 -2
  54. package/src/standardProviders.test.ts +45 -31
  55. package/src/standardProviders.ts +42 -29
  56. package/src/types.ts +5 -5
  57. package/src/usage.test.ts +8 -10
  58. package/src/usage.ts +7 -3
@@ -0,0 +1,242 @@
1
+ import assert from "node:assert/strict";
2
+ import { describe, it } from "node:test";
3
+ import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
4
+ import { APICallError, streamText } from "ai";
5
+
6
+ const encoder = new TextEncoder();
7
+
8
+ const sseResponse = (...chunks: object[]): Response => new Response(
9
+ new ReadableStream({
10
+ start(controller) {
11
+ for (const chunk of chunks) {
12
+ controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
13
+ }
14
+ controller.enqueue(encoder.encode("data: [DONE]\n\n"));
15
+ controller.close();
16
+ },
17
+ }),
18
+ { headers: { "content-type": "text/event-stream" } },
19
+ );
20
+
21
+ describe("AI SDK adapter spike", () => {
22
+ it("preserves PLURNK request extensions and complete stream evidence", async () => {
23
+ let requestBody: Record<string, unknown> | undefined;
24
+ const rawChunks = [
25
+ {
26
+ id: "response-1",
27
+ object: "chat.completion.chunk",
28
+ created: 1,
29
+ model: "test-model",
30
+ choices: [{
31
+ index: 0,
32
+ delta: { role: "assistant", reasoning_content: "because " },
33
+ finish_reason: null,
34
+ }],
35
+ },
36
+ {
37
+ id: "response-1",
38
+ object: "chat.completion.chunk",
39
+ created: 1,
40
+ model: "test-model",
41
+ choices: [{
42
+ index: 0,
43
+ delta: { content: "answer" },
44
+ finish_reason: null,
45
+ }],
46
+ },
47
+ {
48
+ id: "response-1",
49
+ object: "chat.completion.chunk",
50
+ created: 1,
51
+ model: "test-model",
52
+ choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
53
+ usage: {
54
+ prompt_tokens: 3,
55
+ completion_tokens: 5,
56
+ total_tokens: 8,
57
+ completion_tokens_details: { reasoning_tokens: 2 },
58
+ },
59
+ },
60
+ ];
61
+ const provider = createOpenAICompatible({
62
+ name: "spike",
63
+ baseURL: "https://example.test/v1",
64
+ apiKey: "test-key",
65
+ includeUsage: true,
66
+ fetch: async (_input, init) => {
67
+ requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
68
+ return sseResponse(...rawChunks);
69
+ },
70
+ });
71
+
72
+ const result = streamText({
73
+ model: provider("test-model"),
74
+ messages: [{ role: "user", content: "question" }],
75
+ maxOutputTokens: 64,
76
+ temperature: 0.25,
77
+ maxRetries: 0,
78
+ timeout: { totalMs: 1_000, chunkMs: 500 },
79
+ includeRawChunks: true,
80
+ providerOptions: {
81
+ spike: {
82
+ grammar: "root ::= \"answer\"",
83
+ id_slot: 2,
84
+ enable_thinking: true,
85
+ service_tier: "flex",
86
+ },
87
+ },
88
+ });
89
+ const parts = [];
90
+ for await (const part of result.fullStream) parts.push(part);
91
+
92
+ assert.equal(requestBody?.model, "test-model");
93
+ assert.equal(requestBody?.max_tokens, 64);
94
+ assert.equal(requestBody?.temperature, 0.25);
95
+ assert.equal(requestBody?.grammar, "root ::= \"answer\"");
96
+ assert.equal(requestBody?.id_slot, 2);
97
+ assert.equal(requestBody?.enable_thinking, true);
98
+ assert.equal(requestBody?.service_tier, "flex");
99
+ assert.equal(requestBody?.stream, true);
100
+ assert.deepEqual(requestBody?.stream_options, { include_usage: true });
101
+
102
+ assert.equal(await result.text, "answer");
103
+ assert.equal(await result.reasoningText, "because ");
104
+ assert.equal(await result.finishReason, "stop");
105
+ assert.equal(await result.rawFinishReason, "stop");
106
+ const usage = await result.usage;
107
+ assert.equal(usage.inputTokens, 3);
108
+ assert.equal(usage.outputTokens, 5);
109
+ assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
110
+ assert.equal(usage.outputTokenDetails.textTokens, 3);
111
+ assert.equal(usage.totalTokens, 8);
112
+ assert.deepEqual(
113
+ parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
114
+ rawChunks,
115
+ );
116
+ });
117
+
118
+ it("surfaces typed HTTP failures without retrying", async () => {
119
+ let requests = 0;
120
+ const provider = createOpenAICompatible({
121
+ name: "spike",
122
+ baseURL: "https://example.test/v1",
123
+ apiKey: "test-key",
124
+ fetch: async () => {
125
+ requests += 1;
126
+ return new Response(
127
+ JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
128
+ {
129
+ status: 429,
130
+ headers: {
131
+ "content-type": "application/json",
132
+ "retry-after": "3",
133
+ "x-request-id": "request-1",
134
+ },
135
+ },
136
+ );
137
+ },
138
+ });
139
+ const result = streamText({
140
+ model: provider("test-model"),
141
+ prompt: "question",
142
+ maxRetries: 0,
143
+ onError: () => {},
144
+ });
145
+
146
+ const parts = [];
147
+ for await (const part of result.fullStream) parts.push(part);
148
+ const errorPart = parts.find((part) => part.type === "error");
149
+ assert.ok(errorPart?.type === "error");
150
+ assert.ok(APICallError.isInstance(errorPart.error));
151
+ assert.equal(errorPart.error.statusCode, 429);
152
+ assert.equal(errorPart.error.isRetryable, true);
153
+ assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
154
+ assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
155
+ assert.match(errorPart.error.responseBody ?? "", /rate limited/);
156
+ assert.equal(requests, 1);
157
+ });
158
+
159
+ it("enforces stream-idle timeout through the transport contract", async () => {
160
+ const provider = createOpenAICompatible({
161
+ name: "spike",
162
+ baseURL: "https://example.test/v1",
163
+ apiKey: "test-key",
164
+ fetch: async () => new Response(
165
+ new ReadableStream({
166
+ start(controller) {
167
+ controller.enqueue(encoder.encode(`data: ${JSON.stringify({
168
+ id: "response-1",
169
+ object: "chat.completion.chunk",
170
+ created: 1,
171
+ model: "test-model",
172
+ choices: [{
173
+ index: 0,
174
+ delta: { role: "assistant", content: "started" },
175
+ finish_reason: null,
176
+ }],
177
+ })}\n\n`));
178
+ setTimeout(() => controller.close(), 100);
179
+ },
180
+ }),
181
+ { headers: { "content-type": "text/event-stream" } },
182
+ ),
183
+ });
184
+ const result = streamText({
185
+ model: provider("test-model"),
186
+ prompt: "question",
187
+ maxRetries: 0,
188
+ timeout: { totalMs: 500, chunkMs: 25 },
189
+ onError: () => {},
190
+ });
191
+
192
+ await assert.rejects(
193
+ () => Promise.resolve(result.text),
194
+ (error: unknown) => {
195
+ assert.match(String(error), /timed out|timeout/i);
196
+ return true;
197
+ },
198
+ );
199
+ });
200
+
201
+ it("propagates caller cancellation into the injected transport", async () => {
202
+ const caller = new AbortController();
203
+ let transportAborted = false;
204
+ const provider = createOpenAICompatible({
205
+ name: "spike",
206
+ baseURL: "https://example.test/v1",
207
+ apiKey: "test-key",
208
+ fetch: async (_input, init) => {
209
+ const transportSignal = init?.signal as AbortSignal;
210
+ return await new Promise<Response>((_resolve, reject) => {
211
+ transportSignal.addEventListener(
212
+ "abort",
213
+ () => {
214
+ transportAborted = true;
215
+ reject(transportSignal.reason);
216
+ },
217
+ { once: true },
218
+ );
219
+ });
220
+ },
221
+ });
222
+ const result = streamText({
223
+ model: provider("test-model"),
224
+ prompt: "question",
225
+ maxRetries: 0,
226
+ abortSignal: caller.signal,
227
+ onError: () => {},
228
+ });
229
+ const partsPromise = (async () => {
230
+ const parts = [];
231
+ for await (const part of result.fullStream) parts.push(part);
232
+ return parts;
233
+ })();
234
+
235
+ await new Promise((resolve) => setTimeout(resolve, 0));
236
+ caller.abort(new Error("caller stopped"));
237
+ const parts = await partsPromise;
238
+
239
+ assert.equal(transportAborted, true);
240
+ assert.ok(parts.some((part) => part.type === "abort"));
241
+ });
242
+ });
package/src/env.test.ts CHANGED
@@ -166,6 +166,14 @@ test("#399: the shipped floor activates reasoning by default (adaptive — owner
166
166
  assert.ok(!defaults.match(/^PLURNK_PROVIDERS_REASONING_BUDGET=/m), "no shipped magnitude — budget is on-mode only");
167
167
  });
168
168
 
169
+ test("#567: the shipped DRY floor stays off while retaining the measured alias-safe shape", async () => {
170
+ const { readFileSync } = await import("node:fs");
171
+ const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
172
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "one-model tuning is not a universal sampler floor");
173
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_BASE=1\.75$/m);
174
+ assert.match(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32$/m, "the measured identifier-safe shape remains available for alias opt-in");
175
+ });
176
+
169
177
  // -- #507: envelope reserves (owner-ruled migration from PLURNK_SERVICE_*) --
170
178
 
171
179
  test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
package/src/env.ts CHANGED
@@ -163,10 +163,12 @@ export const PROVIDERS_KNOBS = Object.freeze([
163
163
  "PLURNK_PROVIDERS_CONTEXT_WINDOW",
164
164
  "PLURNK_PROVIDERS_RETRY_ATTEMPTS",
165
165
  "PLURNK_PROVIDERS_FETCH_TIMEOUT",
166
+ "PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
166
167
  "PLURNK_PROVIDERS_LLAMA_SERVER",
167
168
  "PLURNK_PROVIDERS_TEMPERATURE",
168
169
  "PLURNK_PROVIDERS_REPEAT_PENALTY",
169
170
  "PLURNK_PROVIDERS_FREQUENCY_PENALTY",
171
+ "PLURNK_PROVIDERS_SERVICE_TIER",
170
172
  "PLURNK_PROVIDERS_REPEAT_LAST_N",
171
173
  "PLURNK_PROVIDERS_DRY_MULTIPLIER",
172
174
  "PLURNK_PROVIDERS_DRY_BASE",
package/src/index.ts CHANGED
@@ -36,11 +36,11 @@ export type { OpenAICompatConfig, ReasoningStyle, GrammarStyle } from "./OpenAIC
36
36
  // worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
37
37
  // DECISION stays the consumer's, by choosing which pool to call.
38
38
  export { default as Pool } from "./Pool.ts";
39
- export { chatCompletionStream, chatCompletion, OpenAiHttpError } from "./openaiStream.ts";
39
+ export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
40
40
  export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
41
41
  export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
42
42
  export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
43
- export { normalizeUsage, computeCost } from "./usage.ts";
43
+ export { normalizeUsage, calculateCostUsd } from "./usage.ts";
44
44
  export type { RawUsage, TokenRates } from "./usage.ts";
45
45
  export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
46
46
  export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
package/src/openai.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  export { default as OpenAICompatProvider, effortFromBudget } from "./OpenAICompat.ts";
2
2
  export type { GrammarStyle, OpenAICompatConfig, ReasoningStyle } from "./OpenAICompat.ts";
3
- export { chatCompletion, chatCompletionStream, OpenAiHttpError } from "./openaiStream.ts";
3
+ export { chatCompletion, chatCompletionStream, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
4
4
  export type {
5
5
  EncryptedReasoningItem,
6
6
  ProviderFetch,
@@ -13,6 +13,9 @@ type StreamRequest = {
13
13
  // #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
14
14
  // default so a serving turn never pays the reassembly/retention cost.
15
15
  captureRawBody?: boolean;
16
+ // Maximum silence between streamed response-body chunks. Undefined/zero
17
+ // disables this clock; the caller's signal still owns the total deadline.
18
+ streamIdleTimeoutMs?: number;
16
19
  };
17
20
 
18
21
  import type { RawUsage } from "./usage.ts";
@@ -122,6 +125,15 @@ export class OpenAiHttpError extends Error {
122
125
  }
123
126
  }
124
127
 
128
+ export class StreamIdleError extends Error {
129
+ readonly timeoutMs: number;
130
+ constructor(timeoutMs: number) {
131
+ super(`stream received no body bytes for ${timeoutMs}ms`);
132
+ this.name = "StreamIdleError";
133
+ this.timeoutMs = timeoutMs;
134
+ }
135
+ }
136
+
125
137
  const parseRetryAfter = (header: string | null): number | null => {
126
138
  if (header === null) return null;
127
139
  const asInt = Number.parseInt(header, 10);
@@ -170,7 +182,7 @@ export const chatCompletion = async ({ url, headers, body, signal, fetch, captur
170
182
  };
171
183
  };
172
184
 
173
- export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
185
+ export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
174
186
  const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
175
187
 
176
188
  const response = await fetch(url, {
@@ -206,7 +218,23 @@ export const chatCompletionStream = async ({ url, headers, body, signal, fetch,
206
218
  let encryptedNoKey = 0;
207
219
 
208
220
  while (true) {
209
- const { done, value } = await reader.read();
221
+ const read = reader.read();
222
+ let timer: ReturnType<typeof setTimeout> | undefined;
223
+ const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
224
+ ? new Promise<never>((_resolve, reject) => {
225
+ timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
226
+ })
227
+ : null;
228
+ let result: Awaited<ReturnType<typeof reader.read>>;
229
+ try {
230
+ result = idle === null ? await read : await Promise.race([read, idle]);
231
+ } catch (err) {
232
+ if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
233
+ throw err;
234
+ } finally {
235
+ if (timer !== undefined) clearTimeout(timer);
236
+ }
237
+ const { done, value } = result;
210
238
  if (done) break;
211
239
  buffer += decoder.decode(value, { stream: true });
212
240
  const lines = buffer.split("\n");
@@ -6,7 +6,7 @@ import { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from
6
6
  // defaults for the providers exercised outside the coverage loop. `openai` is
7
7
  // deliberately omitted so its missing-base fail-hard test still fires.
8
8
  const baseEnv = Object.freeze({
9
- PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
9
+ PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
10
10
  GROQ_BASE_URL: "https://api.groq.com/openai/v1",
11
11
  DEEPINFRA_BASE_URL: "https://api.deepinfra.com/v1/openai",
12
12
  FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
@@ -298,10 +298,10 @@ test("openai: a garbage PLURNK_PROVIDERS_LLAMA_SERVER value fails hard", async (
298
298
  );
299
299
  });
300
300
 
301
- test("constrainsOutput: fireworks (static response_format) reports true; groq reports false", async () => {
301
+ test("constrainsOutput: cloud providers do not claim local GBNF transport", async () => {
302
302
  mockEndpoint();
303
303
  const fw = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
304
- assert.equal(fw!.constrainsOutput, true);
304
+ assert.equal(fw!.constrainsOutput, false);
305
305
  const gq = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
306
306
  assert.equal(gq!.constrainsOutput, false);
307
307
  });
@@ -550,7 +550,7 @@ test("bedrock: contextWindow resolves from the catalog via the inference-profile
550
550
  // MECHANISM (non-null, positive), never the literal - a catalog refresh must not break the build.
551
551
  assert.ok(p!.contextWindow !== null && p!.contextWindow > 0, `expected a catalog-resolved window, got ${p!.contextWindow}`);
552
552
  // cost is NOT taken from the native anthropic rate (bedrock marks up) — stays 0
553
- assert.equal(p!.costFor({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
553
+ assert.equal(p!.calculateCost({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
554
554
  });
555
555
 
556
556
  test("bedrock: a publisher the catalog lacks (meta) fails hard (cloud, no probe); PLURNK_PROVIDERS_CONTEXT_WINDOW still wins", async () => {
@@ -585,11 +585,11 @@ test("standard provider: a catalog hit fills contextWindow + cost when there's n
585
585
  assert.ok(p !== null);
586
586
  assert.equal(p.contextWindow, info.contextWindow); // catalog window, no probe needed
587
587
  if (info.cost !== undefined) {
588
- // 1M output tokens → outputPer1M USD, in pico-USD (per-1M ×1e6 per token).
589
- const c = p.costFor({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
590
- assert.equal(c, Math.round(info.cost.outputPer1M * 1e6 * 1_000_000));
588
+ // One million output tokens cost exactly the catalog's per-million USD rate.
589
+ const c = p.calculateCost({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
590
+ assert.equal(c, info.cost.outputPer1M);
591
591
  } else {
592
- assert.equal(p.costFor({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
592
+ assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
593
593
  }
594
594
  });
595
595
 
@@ -828,27 +828,17 @@ test("plurnk: a set-but-rejected key (401) surfaces the distinct #537 rejected h
828
828
  assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried, distinct message
829
829
  });
830
830
 
831
- test("plurnk: normalizes the endpoint's balance_pico into meta.balancePico (#23)", async () => {
831
+ test("plurnk: passes currency-explicit balance metadata through (#23)", async () => {
832
832
  mock.method(globalThis, "fetch", async (url: string) => {
833
833
  const u = String(url);
834
834
  if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
835
835
  // plurnk has detectLlamaServer:false → streams; balance rides as a top-level field on a chunk.
836
- const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance_pico":880000000}\n\ndata: [DONE]';
836
+ const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance":{"amount":"0.00000088","currency":"XMR"}}\n\ndata: [DONE]';
837
837
  return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode(sse)); c.close(); } }), { status: 200 });
838
838
  });
839
839
  const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
840
840
  const res = await p!.generate({ workerId: "r", messages: [] });
841
- assert.equal(res.meta?.balancePico, 880000000);
842
- mock.restoreAll();
843
- });
844
-
845
- test("a third-party (non-plurnk) provider never NORMALIZES balancePico — only plurnk holds that contract", async () => {
846
- mock.method(globalThis, "fetch", async () =>
847
- new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"hi"}}],"balance_pico":880000000}\n\ndata: [DONE]')); c.close(); } }), { status: 200 }));
848
- const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
849
- const res = await p!.generate({ workerId: "r", messages: [] });
850
- assert.equal("balancePico" in (res.meta ?? {}), false); // groq has no balanceMetaKey — no normalization
851
- assert.equal(res.meta?.balance_pico, 880000000); // but the raw field still passes through (every-provider meta)
841
+ assert.deepEqual(res.meta?.balance, { amount: "0.00000088", currency: "XMR" });
852
842
  mock.restoreAll();
853
843
  });
854
844
 
@@ -868,9 +858,7 @@ test("plurnk: reads its window from upstream but stays a plain OpenAI client —
868
858
  mock.restoreAll();
869
859
  });
870
860
 
871
- // — fireworks carries GBNF via response_format.grammar (cloud GBNF, #grammarStyle) —
872
-
873
- test("fireworks: a grammar transports as response_format.grammar (not the llama.cpp top-level field)", async () => {
861
+ test("fireworks: caller GBNF is not transported to the cloud API", async () => {
874
862
  let body = "";
875
863
  mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
876
864
  if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
@@ -879,31 +867,57 @@ test("fireworks: a grammar transports as response_format.grammar (not the llama.
879
867
  const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "accounts/fireworks/models/deepseek-v4-pro");
880
868
  await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
881
869
  const b = JSON.parse(body);
882
- assert.deepEqual(b.response_format, { type: "grammar", grammar: 'root ::= "ok"' });
870
+ assert.equal("response_format" in b, false);
883
871
  assert.equal("grammar" in b, false);
884
872
  mock.restoreAll();
885
873
  });
886
874
 
887
875
  // — fireworks modelPrefix: the alias carries only the distinctive tail —
888
876
 
889
- const fireworksWireModel = async (alias: string): Promise<string> => {
877
+ const fireworksWireBody = async (model: string, env: NodeJS.ProcessEnv = {}): Promise<Record<string, unknown>> => {
890
878
  let body = "";
891
879
  mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
892
880
  if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
893
881
  return new Response("{}", { status: 200 });
894
882
  });
895
- const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, alias);
883
+ const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", ...env }, model);
896
884
  await p!.generate({ workerId: "r", messages: [] });
897
885
  mock.restoreAll();
898
- return JSON.parse(body).model;
886
+ return JSON.parse(body) as Record<string, unknown>;
899
887
  };
900
888
 
901
889
  test("fireworks: a bare alias is prefixed with accounts/fireworks/models/ on the wire", async () => {
902
- assert.equal(await fireworksWireModel("deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
890
+ assert.equal((await fireworksWireBody("deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
891
+ });
892
+
893
+ test("fireworks: fully qualified model and router ids are preserved verbatim", async () => {
894
+ assert.equal((await fireworksWireBody("accounts/fireworks/models/deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
895
+ assert.equal((await fireworksWireBody("accounts/fireworks/routers/glm-5p2-fast")).model, "accounts/fireworks/routers/glm-5p2-fast");
903
896
  });
904
897
 
905
- test("fireworks: an already-prefixed id is left unchanged (idempotent prepend)", async () => {
906
- assert.equal(await fireworksWireModel("accounts/fireworks/models/deepseek-v4-pro"), "accounts/fireworks/models/deepseek-v4-pro");
898
+ test("fireworks: configured service tier is fixed on the wire and invalid values fail hard", async () => {
899
+ const body = await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "priority" });
900
+ assert.equal(body.service_tier, "priority");
901
+ assert.equal((await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "flex" })).service_tier, "flex");
902
+ await assert.rejects(
903
+ standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "urgent" }, "deepseek-v4-pro"),
904
+ /PLURNK_PROVIDERS_SERVICE_TIER must be one of "auto", "default", "flex", "priority"/,
905
+ );
906
+ });
907
+
908
+ test("fireworks: configured tier wins over per-call sampling; unset retains per-call intent", async () => {
909
+ let bodies: Record<string, unknown>[] = [];
910
+ mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
911
+ bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
912
+ return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
913
+ });
914
+ const fixed = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "priority" }, "deepseek-v4-pro");
915
+ await fixed!.generate({ workerId: "r", messages: [], sampling: { service_tier: "default" } });
916
+ const flexible = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "deepseek-v4-pro");
917
+ await flexible!.generate({ workerId: "r", messages: [], sampling: { service_tier: "priority" } });
918
+ assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "priority"]);
919
+ bodies = [];
920
+ mock.restoreAll();
907
921
  });
908
922
 
909
923
  test("#518 prompt_cache_key: default-ON for a standard provider (workerId), OFF for anthropic (cache_control)", async () => {