@plurnk/plurnk-providers 1.3.3 → 1.3.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +20 -21
- package/SPEC.md +65 -44
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +5 -4
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +38 -38
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +2 -0
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/openai.d.ts +1 -1
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -1
- package/dist/openai.js.map +1 -1
- package/dist/openaiStream.d.ts +6 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +30 -2
- package/dist/openaiStream.js.map +1 -1
- package/dist/standardProviders.d.ts +3 -3
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +36 -17
- package/dist/standardProviders.js.map +1 -1
- package/dist/types.d.ts +1 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +10 -7
- package/src/Mock.test.ts +2 -2
- package/src/Mock.ts +1 -1
- package/src/OpenAICompat.test.ts +86 -86
- package/src/OpenAICompat.ts +46 -45
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +30 -1
- package/src/aiSdkAdapter.spike.test.ts +242 -0
- package/src/env.test.ts +8 -0
- package/src/env.ts +2 -0
- package/src/index.ts +2 -2
- package/src/openai.ts +1 -1
- package/src/openaiStream.ts +30 -2
- package/src/standardProviders.test.ts +45 -31
- package/src/standardProviders.ts +42 -29
- package/src/types.ts +5 -5
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
|
@@ -0,0 +1,242 @@
|
|
|
1
|
+
import assert from "node:assert/strict";
|
|
2
|
+
import { describe, it } from "node:test";
|
|
3
|
+
import { createOpenAICompatible } from "@ai-sdk/openai-compatible";
|
|
4
|
+
import { APICallError, streamText } from "ai";
|
|
5
|
+
|
|
6
|
+
const encoder = new TextEncoder();
|
|
7
|
+
|
|
8
|
+
const sseResponse = (...chunks: object[]): Response => new Response(
|
|
9
|
+
new ReadableStream({
|
|
10
|
+
start(controller) {
|
|
11
|
+
for (const chunk of chunks) {
|
|
12
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify(chunk)}\n\n`));
|
|
13
|
+
}
|
|
14
|
+
controller.enqueue(encoder.encode("data: [DONE]\n\n"));
|
|
15
|
+
controller.close();
|
|
16
|
+
},
|
|
17
|
+
}),
|
|
18
|
+
{ headers: { "content-type": "text/event-stream" } },
|
|
19
|
+
);
|
|
20
|
+
|
|
21
|
+
describe("AI SDK adapter spike", () => {
|
|
22
|
+
it("preserves PLURNK request extensions and complete stream evidence", async () => {
|
|
23
|
+
let requestBody: Record<string, unknown> | undefined;
|
|
24
|
+
const rawChunks = [
|
|
25
|
+
{
|
|
26
|
+
id: "response-1",
|
|
27
|
+
object: "chat.completion.chunk",
|
|
28
|
+
created: 1,
|
|
29
|
+
model: "test-model",
|
|
30
|
+
choices: [{
|
|
31
|
+
index: 0,
|
|
32
|
+
delta: { role: "assistant", reasoning_content: "because " },
|
|
33
|
+
finish_reason: null,
|
|
34
|
+
}],
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
id: "response-1",
|
|
38
|
+
object: "chat.completion.chunk",
|
|
39
|
+
created: 1,
|
|
40
|
+
model: "test-model",
|
|
41
|
+
choices: [{
|
|
42
|
+
index: 0,
|
|
43
|
+
delta: { content: "answer" },
|
|
44
|
+
finish_reason: null,
|
|
45
|
+
}],
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
id: "response-1",
|
|
49
|
+
object: "chat.completion.chunk",
|
|
50
|
+
created: 1,
|
|
51
|
+
model: "test-model",
|
|
52
|
+
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
|
|
53
|
+
usage: {
|
|
54
|
+
prompt_tokens: 3,
|
|
55
|
+
completion_tokens: 5,
|
|
56
|
+
total_tokens: 8,
|
|
57
|
+
completion_tokens_details: { reasoning_tokens: 2 },
|
|
58
|
+
},
|
|
59
|
+
},
|
|
60
|
+
];
|
|
61
|
+
const provider = createOpenAICompatible({
|
|
62
|
+
name: "spike",
|
|
63
|
+
baseURL: "https://example.test/v1",
|
|
64
|
+
apiKey: "test-key",
|
|
65
|
+
includeUsage: true,
|
|
66
|
+
fetch: async (_input, init) => {
|
|
67
|
+
requestBody = JSON.parse(String(init?.body)) as Record<string, unknown>;
|
|
68
|
+
return sseResponse(...rawChunks);
|
|
69
|
+
},
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
const result = streamText({
|
|
73
|
+
model: provider("test-model"),
|
|
74
|
+
messages: [{ role: "user", content: "question" }],
|
|
75
|
+
maxOutputTokens: 64,
|
|
76
|
+
temperature: 0.25,
|
|
77
|
+
maxRetries: 0,
|
|
78
|
+
timeout: { totalMs: 1_000, chunkMs: 500 },
|
|
79
|
+
includeRawChunks: true,
|
|
80
|
+
providerOptions: {
|
|
81
|
+
spike: {
|
|
82
|
+
grammar: "root ::= \"answer\"",
|
|
83
|
+
id_slot: 2,
|
|
84
|
+
enable_thinking: true,
|
|
85
|
+
service_tier: "flex",
|
|
86
|
+
},
|
|
87
|
+
},
|
|
88
|
+
});
|
|
89
|
+
const parts = [];
|
|
90
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
91
|
+
|
|
92
|
+
assert.equal(requestBody?.model, "test-model");
|
|
93
|
+
assert.equal(requestBody?.max_tokens, 64);
|
|
94
|
+
assert.equal(requestBody?.temperature, 0.25);
|
|
95
|
+
assert.equal(requestBody?.grammar, "root ::= \"answer\"");
|
|
96
|
+
assert.equal(requestBody?.id_slot, 2);
|
|
97
|
+
assert.equal(requestBody?.enable_thinking, true);
|
|
98
|
+
assert.equal(requestBody?.service_tier, "flex");
|
|
99
|
+
assert.equal(requestBody?.stream, true);
|
|
100
|
+
assert.deepEqual(requestBody?.stream_options, { include_usage: true });
|
|
101
|
+
|
|
102
|
+
assert.equal(await result.text, "answer");
|
|
103
|
+
assert.equal(await result.reasoningText, "because ");
|
|
104
|
+
assert.equal(await result.finishReason, "stop");
|
|
105
|
+
assert.equal(await result.rawFinishReason, "stop");
|
|
106
|
+
const usage = await result.usage;
|
|
107
|
+
assert.equal(usage.inputTokens, 3);
|
|
108
|
+
assert.equal(usage.outputTokens, 5);
|
|
109
|
+
assert.equal(usage.outputTokenDetails.reasoningTokens, 2);
|
|
110
|
+
assert.equal(usage.outputTokenDetails.textTokens, 3);
|
|
111
|
+
assert.equal(usage.totalTokens, 8);
|
|
112
|
+
assert.deepEqual(
|
|
113
|
+
parts.filter((part) => part.type === "raw").map((part) => part.rawValue),
|
|
114
|
+
rawChunks,
|
|
115
|
+
);
|
|
116
|
+
});
|
|
117
|
+
|
|
118
|
+
it("surfaces typed HTTP failures without retrying", async () => {
|
|
119
|
+
let requests = 0;
|
|
120
|
+
const provider = createOpenAICompatible({
|
|
121
|
+
name: "spike",
|
|
122
|
+
baseURL: "https://example.test/v1",
|
|
123
|
+
apiKey: "test-key",
|
|
124
|
+
fetch: async () => {
|
|
125
|
+
requests += 1;
|
|
126
|
+
return new Response(
|
|
127
|
+
JSON.stringify({ error: { message: "rate limited", type: "rate_limit" } }),
|
|
128
|
+
{
|
|
129
|
+
status: 429,
|
|
130
|
+
headers: {
|
|
131
|
+
"content-type": "application/json",
|
|
132
|
+
"retry-after": "3",
|
|
133
|
+
"x-request-id": "request-1",
|
|
134
|
+
},
|
|
135
|
+
},
|
|
136
|
+
);
|
|
137
|
+
},
|
|
138
|
+
});
|
|
139
|
+
const result = streamText({
|
|
140
|
+
model: provider("test-model"),
|
|
141
|
+
prompt: "question",
|
|
142
|
+
maxRetries: 0,
|
|
143
|
+
onError: () => {},
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
const parts = [];
|
|
147
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
148
|
+
const errorPart = parts.find((part) => part.type === "error");
|
|
149
|
+
assert.ok(errorPart?.type === "error");
|
|
150
|
+
assert.ok(APICallError.isInstance(errorPart.error));
|
|
151
|
+
assert.equal(errorPart.error.statusCode, 429);
|
|
152
|
+
assert.equal(errorPart.error.isRetryable, true);
|
|
153
|
+
assert.equal(errorPart.error.responseHeaders?.["retry-after"], "3");
|
|
154
|
+
assert.equal(errorPart.error.responseHeaders?.["x-request-id"], "request-1");
|
|
155
|
+
assert.match(errorPart.error.responseBody ?? "", /rate limited/);
|
|
156
|
+
assert.equal(requests, 1);
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it("enforces stream-idle timeout through the transport contract", async () => {
|
|
160
|
+
const provider = createOpenAICompatible({
|
|
161
|
+
name: "spike",
|
|
162
|
+
baseURL: "https://example.test/v1",
|
|
163
|
+
apiKey: "test-key",
|
|
164
|
+
fetch: async () => new Response(
|
|
165
|
+
new ReadableStream({
|
|
166
|
+
start(controller) {
|
|
167
|
+
controller.enqueue(encoder.encode(`data: ${JSON.stringify({
|
|
168
|
+
id: "response-1",
|
|
169
|
+
object: "chat.completion.chunk",
|
|
170
|
+
created: 1,
|
|
171
|
+
model: "test-model",
|
|
172
|
+
choices: [{
|
|
173
|
+
index: 0,
|
|
174
|
+
delta: { role: "assistant", content: "started" },
|
|
175
|
+
finish_reason: null,
|
|
176
|
+
}],
|
|
177
|
+
})}\n\n`));
|
|
178
|
+
setTimeout(() => controller.close(), 100);
|
|
179
|
+
},
|
|
180
|
+
}),
|
|
181
|
+
{ headers: { "content-type": "text/event-stream" } },
|
|
182
|
+
),
|
|
183
|
+
});
|
|
184
|
+
const result = streamText({
|
|
185
|
+
model: provider("test-model"),
|
|
186
|
+
prompt: "question",
|
|
187
|
+
maxRetries: 0,
|
|
188
|
+
timeout: { totalMs: 500, chunkMs: 25 },
|
|
189
|
+
onError: () => {},
|
|
190
|
+
});
|
|
191
|
+
|
|
192
|
+
await assert.rejects(
|
|
193
|
+
() => Promise.resolve(result.text),
|
|
194
|
+
(error: unknown) => {
|
|
195
|
+
assert.match(String(error), /timed out|timeout/i);
|
|
196
|
+
return true;
|
|
197
|
+
},
|
|
198
|
+
);
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
it("propagates caller cancellation into the injected transport", async () => {
|
|
202
|
+
const caller = new AbortController();
|
|
203
|
+
let transportAborted = false;
|
|
204
|
+
const provider = createOpenAICompatible({
|
|
205
|
+
name: "spike",
|
|
206
|
+
baseURL: "https://example.test/v1",
|
|
207
|
+
apiKey: "test-key",
|
|
208
|
+
fetch: async (_input, init) => {
|
|
209
|
+
const transportSignal = init?.signal as AbortSignal;
|
|
210
|
+
return await new Promise<Response>((_resolve, reject) => {
|
|
211
|
+
transportSignal.addEventListener(
|
|
212
|
+
"abort",
|
|
213
|
+
() => {
|
|
214
|
+
transportAborted = true;
|
|
215
|
+
reject(transportSignal.reason);
|
|
216
|
+
},
|
|
217
|
+
{ once: true },
|
|
218
|
+
);
|
|
219
|
+
});
|
|
220
|
+
},
|
|
221
|
+
});
|
|
222
|
+
const result = streamText({
|
|
223
|
+
model: provider("test-model"),
|
|
224
|
+
prompt: "question",
|
|
225
|
+
maxRetries: 0,
|
|
226
|
+
abortSignal: caller.signal,
|
|
227
|
+
onError: () => {},
|
|
228
|
+
});
|
|
229
|
+
const partsPromise = (async () => {
|
|
230
|
+
const parts = [];
|
|
231
|
+
for await (const part of result.fullStream) parts.push(part);
|
|
232
|
+
return parts;
|
|
233
|
+
})();
|
|
234
|
+
|
|
235
|
+
await new Promise((resolve) => setTimeout(resolve, 0));
|
|
236
|
+
caller.abort(new Error("caller stopped"));
|
|
237
|
+
const parts = await partsPromise;
|
|
238
|
+
|
|
239
|
+
assert.equal(transportAborted, true);
|
|
240
|
+
assert.ok(parts.some((part) => part.type === "abort"));
|
|
241
|
+
});
|
|
242
|
+
});
|
package/src/env.test.ts
CHANGED
|
@@ -166,6 +166,14 @@ test("#399: the shipped floor activates reasoning by default (adaptive — owner
|
|
|
166
166
|
assert.ok(!defaults.match(/^PLURNK_PROVIDERS_REASONING_BUDGET=/m), "no shipped magnitude — budget is on-mode only");
|
|
167
167
|
});
|
|
168
168
|
|
|
169
|
+
test("#567: the shipped DRY floor stays off while retaining the measured alias-safe shape", async () => {
|
|
170
|
+
const { readFileSync } = await import("node:fs");
|
|
171
|
+
const defaults = readFileSync(new URL("../.env.defaults", import.meta.url), "utf8");
|
|
172
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_MULTIPLIER=0$/m, "one-model tuning is not a universal sampler floor");
|
|
173
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_BASE=1\.75$/m);
|
|
174
|
+
assert.match(defaults, /^PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH=32$/m, "the measured identifier-safe shape remains available for alias opt-in");
|
|
175
|
+
});
|
|
176
|
+
|
|
169
177
|
// -- #507: envelope reserves (owner-ruled migration from PLURNK_SERVICE_*) --
|
|
170
178
|
|
|
171
179
|
test("#507 envelopeFromEnv: percentages and absolutes parse; missing/invalid fail hard", async () => {
|
package/src/env.ts
CHANGED
|
@@ -163,10 +163,12 @@ export const PROVIDERS_KNOBS = Object.freeze([
|
|
|
163
163
|
"PLURNK_PROVIDERS_CONTEXT_WINDOW",
|
|
164
164
|
"PLURNK_PROVIDERS_RETRY_ATTEMPTS",
|
|
165
165
|
"PLURNK_PROVIDERS_FETCH_TIMEOUT",
|
|
166
|
+
"PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT",
|
|
166
167
|
"PLURNK_PROVIDERS_LLAMA_SERVER",
|
|
167
168
|
"PLURNK_PROVIDERS_TEMPERATURE",
|
|
168
169
|
"PLURNK_PROVIDERS_REPEAT_PENALTY",
|
|
169
170
|
"PLURNK_PROVIDERS_FREQUENCY_PENALTY",
|
|
171
|
+
"PLURNK_PROVIDERS_SERVICE_TIER",
|
|
170
172
|
"PLURNK_PROVIDERS_REPEAT_LAST_N",
|
|
171
173
|
"PLURNK_PROVIDERS_DRY_MULTIPLIER",
|
|
172
174
|
"PLURNK_PROVIDERS_DRY_BASE",
|
package/src/index.ts
CHANGED
|
@@ -36,11 +36,11 @@ export type { OpenAICompatConfig, ReasoningStyle, GrammarStyle } from "./OpenAIC
|
|
|
36
36
|
// worker-sticky for KV-cache reuse, overflow to a healthy sibling; the blend
|
|
37
37
|
// DECISION stays the consumer's, by choosing which pool to call.
|
|
38
38
|
export { default as Pool } from "./Pool.ts";
|
|
39
|
-
export { chatCompletionStream, chatCompletion, OpenAiHttpError } from "./openaiStream.ts";
|
|
39
|
+
export { chatCompletionStream, chatCompletion, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
|
|
40
40
|
export type { StreamResponse, EncryptedReasoningItem, ProviderFetch } from "./openaiStream.ts";
|
|
41
41
|
export { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, requireEnv, reasoningFromEnv, scopeEnvToAlias, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve } from "./env.ts";
|
|
42
42
|
export type { Reasoning, ReasoningMode, ReserveSpec } from "./env.ts";
|
|
43
|
-
export { normalizeUsage,
|
|
43
|
+
export { normalizeUsage, calculateCostUsd } from "./usage.ts";
|
|
44
44
|
export type { RawUsage, TokenRates } from "./usage.ts";
|
|
45
45
|
export { ProviderError, classifyProviderError, toProviderError, providerSource } from "./telemetry.ts";
|
|
46
46
|
export type { TelemetryEvent, ProviderTelemetryKind } from "./telemetry.ts";
|
package/src/openai.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
export { default as OpenAICompatProvider, effortFromBudget } from "./OpenAICompat.ts";
|
|
2
2
|
export type { GrammarStyle, OpenAICompatConfig, ReasoningStyle } from "./OpenAICompat.ts";
|
|
3
|
-
export { chatCompletion, chatCompletionStream, OpenAiHttpError } from "./openaiStream.ts";
|
|
3
|
+
export { chatCompletion, chatCompletionStream, OpenAiHttpError, StreamIdleError } from "./openaiStream.ts";
|
|
4
4
|
export type {
|
|
5
5
|
EncryptedReasoningItem,
|
|
6
6
|
ProviderFetch,
|
package/src/openaiStream.ts
CHANGED
|
@@ -13,6 +13,9 @@ type StreamRequest = {
|
|
|
13
13
|
// #36: assemble the verbatim wire body onto StreamResponse.rawBody. Off by
|
|
14
14
|
// default so a serving turn never pays the reassembly/retention cost.
|
|
15
15
|
captureRawBody?: boolean;
|
|
16
|
+
// Maximum silence between streamed response-body chunks. Undefined/zero
|
|
17
|
+
// disables this clock; the caller's signal still owns the total deadline.
|
|
18
|
+
streamIdleTimeoutMs?: number;
|
|
16
19
|
};
|
|
17
20
|
|
|
18
21
|
import type { RawUsage } from "./usage.ts";
|
|
@@ -122,6 +125,15 @@ export class OpenAiHttpError extends Error {
|
|
|
122
125
|
}
|
|
123
126
|
}
|
|
124
127
|
|
|
128
|
+
export class StreamIdleError extends Error {
|
|
129
|
+
readonly timeoutMs: number;
|
|
130
|
+
constructor(timeoutMs: number) {
|
|
131
|
+
super(`stream received no body bytes for ${timeoutMs}ms`);
|
|
132
|
+
this.name = "StreamIdleError";
|
|
133
|
+
this.timeoutMs = timeoutMs;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
125
137
|
const parseRetryAfter = (header: string | null): number | null => {
|
|
126
138
|
if (header === null) return null;
|
|
127
139
|
const asInt = Number.parseInt(header, 10);
|
|
@@ -170,7 +182,7 @@ export const chatCompletion = async ({ url, headers, body, signal, fetch, captur
|
|
|
170
182
|
};
|
|
171
183
|
};
|
|
172
184
|
|
|
173
|
-
export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody }: StreamRequest): Promise<StreamResponse> => {
|
|
185
|
+
export const chatCompletionStream = async ({ url, headers, body, signal, fetch, captureRawBody, streamIdleTimeoutMs }: StreamRequest): Promise<StreamResponse> => {
|
|
174
186
|
const requestBody = { ...body, stream: true, stream_options: { include_usage: true } };
|
|
175
187
|
|
|
176
188
|
const response = await fetch(url, {
|
|
@@ -206,7 +218,23 @@ export const chatCompletionStream = async ({ url, headers, body, signal, fetch,
|
|
|
206
218
|
let encryptedNoKey = 0;
|
|
207
219
|
|
|
208
220
|
while (true) {
|
|
209
|
-
const
|
|
221
|
+
const read = reader.read();
|
|
222
|
+
let timer: ReturnType<typeof setTimeout> | undefined;
|
|
223
|
+
const idle = streamIdleTimeoutMs !== undefined && streamIdleTimeoutMs > 0
|
|
224
|
+
? new Promise<never>((_resolve, reject) => {
|
|
225
|
+
timer = setTimeout(() => reject(new StreamIdleError(streamIdleTimeoutMs)), streamIdleTimeoutMs);
|
|
226
|
+
})
|
|
227
|
+
: null;
|
|
228
|
+
let result: Awaited<ReturnType<typeof reader.read>>;
|
|
229
|
+
try {
|
|
230
|
+
result = idle === null ? await read : await Promise.race([read, idle]);
|
|
231
|
+
} catch (err) {
|
|
232
|
+
if (err instanceof StreamIdleError) void reader.cancel(err).catch(() => undefined);
|
|
233
|
+
throw err;
|
|
234
|
+
} finally {
|
|
235
|
+
if (timer !== undefined) clearTimeout(timer);
|
|
236
|
+
}
|
|
237
|
+
const { done, value } = result;
|
|
210
238
|
if (done) break;
|
|
211
239
|
buffer += decoder.decode(value, { stream: true });
|
|
212
240
|
const lines = buffer.split("\n");
|
|
@@ -6,7 +6,7 @@ import { STANDARD_PROVIDERS, isStandardProvider, standardProviderFromEnv } from
|
|
|
6
6
|
// defaults for the providers exercised outside the coverage loop. `openai` is
|
|
7
7
|
// deliberately omitted so its missing-base fail-hard test still fires.
|
|
8
8
|
const baseEnv = Object.freeze({
|
|
9
|
-
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
9
|
+
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000", PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0", PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_RETRY_DELAY: "1", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0",
|
|
10
10
|
GROQ_BASE_URL: "https://api.groq.com/openai/v1",
|
|
11
11
|
DEEPINFRA_BASE_URL: "https://api.deepinfra.com/v1/openai",
|
|
12
12
|
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
@@ -298,10 +298,10 @@ test("openai: a garbage PLURNK_PROVIDERS_LLAMA_SERVER value fails hard", async (
|
|
|
298
298
|
);
|
|
299
299
|
});
|
|
300
300
|
|
|
301
|
-
test("constrainsOutput:
|
|
301
|
+
test("constrainsOutput: cloud providers do not claim local GBNF transport", async () => {
|
|
302
302
|
mockEndpoint();
|
|
303
303
|
const fw = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
304
|
-
assert.equal(fw!.constrainsOutput,
|
|
304
|
+
assert.equal(fw!.constrainsOutput, false);
|
|
305
305
|
const gq = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
306
306
|
assert.equal(gq!.constrainsOutput, false);
|
|
307
307
|
});
|
|
@@ -550,7 +550,7 @@ test("bedrock: contextWindow resolves from the catalog via the inference-profile
|
|
|
550
550
|
// MECHANISM (non-null, positive), never the literal - a catalog refresh must not break the build.
|
|
551
551
|
assert.ok(p!.contextWindow !== null && p!.contextWindow > 0, `expected a catalog-resolved window, got ${p!.contextWindow}`);
|
|
552
552
|
// cost is NOT taken from the native anthropic rate (bedrock marks up) — stays 0
|
|
553
|
-
assert.equal(p!.
|
|
553
|
+
assert.equal(p!.calculateCost({ prompt: 1_000_000, completion: 1_000_000, reasoning: 0, cached: 0, total: 2_000_000 }), 0);
|
|
554
554
|
});
|
|
555
555
|
|
|
556
556
|
test("bedrock: a publisher the catalog lacks (meta) fails hard (cloud, no probe); PLURNK_PROVIDERS_CONTEXT_WINDOW still wins", async () => {
|
|
@@ -585,11 +585,11 @@ test("standard provider: a catalog hit fills contextWindow + cost when there's n
|
|
|
585
585
|
assert.ok(p !== null);
|
|
586
586
|
assert.equal(p.contextWindow, info.contextWindow); // catalog window, no probe needed
|
|
587
587
|
if (info.cost !== undefined) {
|
|
588
|
-
//
|
|
589
|
-
const c = p.
|
|
590
|
-
assert.equal(c,
|
|
588
|
+
// One million output tokens cost exactly the catalog's per-million USD rate.
|
|
589
|
+
const c = p.calculateCost({ prompt: 0, completion: 1_000_000, reasoning: 0, cached: 0, total: 1_000_000 });
|
|
590
|
+
assert.equal(c, info.cost.outputPer1M);
|
|
591
591
|
} else {
|
|
592
|
-
assert.equal(p.
|
|
592
|
+
assert.equal(p.calculateCost({ prompt: 1, completion: 1, reasoning: 0, cached: 0, total: 2 }), 0);
|
|
593
593
|
}
|
|
594
594
|
});
|
|
595
595
|
|
|
@@ -828,27 +828,17 @@ test("plurnk: a set-but-rejected key (401) surfaces the distinct #537 rejected h
|
|
|
828
828
|
assert.equal(seen.filter((s) => s.url.endsWith("/chat/completions")).length, 1); // terminal — never retried, distinct message
|
|
829
829
|
});
|
|
830
830
|
|
|
831
|
-
test("plurnk:
|
|
831
|
+
test("plurnk: passes currency-explicit balance metadata through (#23)", async () => {
|
|
832
832
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
833
833
|
const u = String(url);
|
|
834
834
|
if (u.endsWith("/models")) return new Response(JSON.stringify({ data: [{ id: "plurnk", meta: { n_ctx: 49152 } }] }), { status: 200 });
|
|
835
835
|
// plurnk has detectLlamaServer:false → streams; balance rides as a top-level field on a chunk.
|
|
836
|
-
const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"
|
|
836
|
+
const sse = 'data: {"choices":[{"delta":{"content":"hi"},"finish_reason":"stop"}],"balance":{"amount":"0.00000088","currency":"XMR"}}\n\ndata: [DONE]';
|
|
837
837
|
return new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode(sse)); c.close(); } }), { status: 200 });
|
|
838
838
|
});
|
|
839
839
|
const p = await standardProviderFromEnv("plurnk", { ...baseEnv, PLURNK_API_KEY: "pk-test" }, "plurnk");
|
|
840
840
|
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
841
|
-
assert.
|
|
842
|
-
mock.restoreAll();
|
|
843
|
-
});
|
|
844
|
-
|
|
845
|
-
test("a third-party (non-plurnk) provider never NORMALIZES balancePico — only plurnk holds that contract", async () => {
|
|
846
|
-
mock.method(globalThis, "fetch", async () =>
|
|
847
|
-
new Response(new ReadableStream({ start(c) { c.enqueue(new TextEncoder().encode('data: {"choices":[{"delta":{"content":"hi"}}],"balance_pico":880000000}\n\ndata: [DONE]')); c.close(); } }), { status: 200 }));
|
|
848
|
-
const p = await standardProviderFromEnv("groq", { ...baseEnv, GROQ_API_KEY: "k", PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "m");
|
|
849
|
-
const res = await p!.generate({ workerId: "r", messages: [] });
|
|
850
|
-
assert.equal("balancePico" in (res.meta ?? {}), false); // groq has no balanceMetaKey — no normalization
|
|
851
|
-
assert.equal(res.meta?.balance_pico, 880000000); // but the raw field still passes through (every-provider meta)
|
|
841
|
+
assert.deepEqual(res.meta?.balance, { amount: "0.00000088", currency: "XMR" });
|
|
852
842
|
mock.restoreAll();
|
|
853
843
|
});
|
|
854
844
|
|
|
@@ -868,9 +858,7 @@ test("plurnk: reads its window from upstream but stays a plain OpenAI client —
|
|
|
868
858
|
mock.restoreAll();
|
|
869
859
|
});
|
|
870
860
|
|
|
871
|
-
|
|
872
|
-
|
|
873
|
-
test("fireworks: a grammar transports as response_format.grammar (not the llama.cpp top-level field)", async () => {
|
|
861
|
+
test("fireworks: caller GBNF is not transported to the cloud API", async () => {
|
|
874
862
|
let body = "";
|
|
875
863
|
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
876
864
|
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
@@ -879,31 +867,57 @@ test("fireworks: a grammar transports as response_format.grammar (not the llama.
|
|
|
879
867
|
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "accounts/fireworks/models/deepseek-v4-pro");
|
|
880
868
|
await p!.generate({ workerId: "r", messages: [], grammar: 'root ::= "ok"' });
|
|
881
869
|
const b = JSON.parse(body);
|
|
882
|
-
assert.
|
|
870
|
+
assert.equal("response_format" in b, false);
|
|
883
871
|
assert.equal("grammar" in b, false);
|
|
884
872
|
mock.restoreAll();
|
|
885
873
|
});
|
|
886
874
|
|
|
887
875
|
// — fireworks modelPrefix: the alias carries only the distinctive tail —
|
|
888
876
|
|
|
889
|
-
const
|
|
877
|
+
const fireworksWireBody = async (model: string, env: NodeJS.ProcessEnv = {}): Promise<Record<string, unknown>> => {
|
|
890
878
|
let body = "";
|
|
891
879
|
mock.method(globalThis, "fetch", async (url: string, init?: RequestInit) => {
|
|
892
880
|
if (String(url).endsWith("/chat/completions")) { body = String(init?.body); return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } }); }
|
|
893
881
|
return new Response("{}", { status: 200 });
|
|
894
882
|
});
|
|
895
|
-
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" },
|
|
883
|
+
const p = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", ...env }, model);
|
|
896
884
|
await p!.generate({ workerId: "r", messages: [] });
|
|
897
885
|
mock.restoreAll();
|
|
898
|
-
return JSON.parse(body)
|
|
886
|
+
return JSON.parse(body) as Record<string, unknown>;
|
|
899
887
|
};
|
|
900
888
|
|
|
901
889
|
test("fireworks: a bare alias is prefixed with accounts/fireworks/models/ on the wire", async () => {
|
|
902
|
-
assert.equal(await
|
|
890
|
+
assert.equal((await fireworksWireBody("deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
|
|
891
|
+
});
|
|
892
|
+
|
|
893
|
+
test("fireworks: fully qualified model and router ids are preserved verbatim", async () => {
|
|
894
|
+
assert.equal((await fireworksWireBody("accounts/fireworks/models/deepseek-v4-pro")).model, "accounts/fireworks/models/deepseek-v4-pro");
|
|
895
|
+
assert.equal((await fireworksWireBody("accounts/fireworks/routers/glm-5p2-fast")).model, "accounts/fireworks/routers/glm-5p2-fast");
|
|
903
896
|
});
|
|
904
897
|
|
|
905
|
-
test("fireworks:
|
|
906
|
-
|
|
898
|
+
test("fireworks: configured service tier is fixed on the wire and invalid values fail hard", async () => {
|
|
899
|
+
const body = await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "priority" });
|
|
900
|
+
assert.equal(body.service_tier, "priority");
|
|
901
|
+
assert.equal((await fireworksWireBody("deepseek-v4-pro", { PLURNK_PROVIDERS_SERVICE_TIER: "flex" })).service_tier, "flex");
|
|
902
|
+
await assert.rejects(
|
|
903
|
+
standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "urgent" }, "deepseek-v4-pro"),
|
|
904
|
+
/PLURNK_PROVIDERS_SERVICE_TIER must be one of "auto", "default", "flex", "priority"/,
|
|
905
|
+
);
|
|
906
|
+
});
|
|
907
|
+
|
|
908
|
+
test("fireworks: configured tier wins over per-call sampling; unset retains per-call intent", async () => {
|
|
909
|
+
let bodies: Record<string, unknown>[] = [];
|
|
910
|
+
mock.method(globalThis, "fetch", async (_url: string, init?: RequestInit) => {
|
|
911
|
+
bodies.push(JSON.parse(String(init?.body)) as Record<string, unknown>);
|
|
912
|
+
return new Response(JSON.stringify({ choices: [{ message: { content: "ok" }, finish_reason: "stop" }] }), { status: 200, headers: { "Content-Type": "application/json" } });
|
|
913
|
+
});
|
|
914
|
+
const fixed = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw", PLURNK_PROVIDERS_SERVICE_TIER: "priority" }, "deepseek-v4-pro");
|
|
915
|
+
await fixed!.generate({ workerId: "r", messages: [], sampling: { service_tier: "default" } });
|
|
916
|
+
const flexible = await standardProviderFromEnv("fireworks", { ...baseEnv, FIREWORKS_API_KEY: "fw" }, "deepseek-v4-pro");
|
|
917
|
+
await flexible!.generate({ workerId: "r", messages: [], sampling: { service_tier: "priority" } });
|
|
918
|
+
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "priority"]);
|
|
919
|
+
bodies = [];
|
|
920
|
+
mock.restoreAll();
|
|
907
921
|
});
|
|
908
922
|
|
|
909
923
|
test("#518 prompt_cache_key: default-ON for a standard provider (workerId), OFF for anthropic (cache_control)", async () => {
|