@plurnk/plurnk-providers 1.3.4 → 1.3.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +41 -52
- package/README.md +44 -53
- package/SPEC.md +215 -354
- package/dist/AiSdkProvider.d.ts +78 -0
- package/dist/AiSdkProvider.d.ts.map +1 -0
- package/dist/AiSdkProvider.js +591 -0
- package/dist/AiSdkProvider.js.map +1 -0
- package/dist/Mock.d.ts +1 -1
- package/dist/Mock.d.ts.map +1 -1
- package/dist/Mock.js +1 -1
- package/dist/Mock.js.map +1 -1
- package/dist/OpenAICompat.d.ts +3 -5
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +44 -133
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/Pool.d.ts +1 -1
- package/dist/Pool.d.ts.map +1 -1
- package/dist/Pool.js +1 -1
- package/dist/Pool.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +37 -24
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +52 -0
- package/dist/aiSdkTransport.d.ts.map +1 -0
- package/dist/aiSdkTransport.js +294 -0
- package/dist/aiSdkTransport.js.map +1 -0
- package/dist/catalogProvider.d.ts +15 -0
- package/dist/catalogProvider.d.ts.map +1 -0
- package/dist/catalogProvider.js +103 -0
- package/dist/catalogProvider.js.map +1 -0
- package/dist/compatibleProvider.d.ts +3 -0
- package/dist/compatibleProvider.d.ts.map +1 -0
- package/dist/compatibleProvider.js +146 -0
- package/dist/compatibleProvider.js.map +1 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +1 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +13 -6
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +5 -7
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -8
- package/dist/index.js.map +1 -1
- package/dist/ollama.d.ts +3 -0
- package/dist/ollama.d.ts.map +1 -0
- package/dist/ollama.js +39 -0
- package/dist/ollama.js.map +1 -0
- package/dist/openai.d.ts +2 -4
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -2
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -0
- package/dist/sdkModels.d.ts.map +1 -0
- package/dist/sdkModels.js +153 -0
- package/dist/sdkModels.js.map +1 -0
- package/dist/standardProviders.d.ts +0 -1
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +9 -11
- package/dist/standardProviders.js.map +1 -1
- package/dist/telemetry.d.ts.map +1 -1
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/types.d.ts +4 -3
- package/dist/types.d.ts.map +1 -1
- package/dist/usage.d.ts +1 -1
- package/dist/usage.d.ts.map +1 -1
- package/dist/usage.js +4 -2
- package/dist/usage.js.map +1 -1
- package/package.json +19 -8
- package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +202 -177
- package/src/{OpenAICompat.ts → AiSdkProvider.ts} +89 -144
- package/src/Mock.test.ts +3 -3
- package/src/Mock.ts +1 -1
- package/src/Pool.test.ts +3 -3
- package/src/Pool.ts +1 -1
- package/src/ProviderRegistry.test.ts +40 -27
- package/src/ProviderRegistry.ts +35 -24
- package/src/aiSdkTransport.test.ts +253 -0
- package/src/aiSdkTransport.ts +369 -0
- package/src/boundaries.test.ts +2 -2
- package/src/catalogProvider.test.ts +100 -0
- package/src/catalogProvider.ts +151 -0
- package/src/compatibleProvider.test.ts +44 -0
- package/src/compatibleProvider.ts +205 -0
- package/src/discover.test.ts +12 -12
- package/src/discover.ts +3 -6
- package/src/env.ts +14 -6
- package/src/index.ts +7 -11
- package/src/ollama.ts +63 -0
- package/src/openai.ts +2 -8
- package/src/sdkModels.test.ts +47 -0
- package/src/sdkModels.ts +194 -0
- package/src/telemetry.test.ts +17 -10
- package/src/telemetry.ts +22 -14
- package/src/types.ts +10 -13
- package/src/usage.test.ts +8 -10
- package/src/usage.ts +7 -3
- package/src/openaiStream.ts +0 -310
- package/src/standardProviders.test.ts +0 -949
- package/src/standardProviders.ts +0 -635
|
@@ -1,21 +1,21 @@
|
|
|
1
|
-
//
|
|
1
|
+
// PLURNK adapter over an AI SDK language model. Implements the universal generate()
|
|
2
2
|
// spine — signal merging, the SSE call, usage mapping, finishReason
|
|
3
3
|
// normalization, response assembly — that every sibling had duplicated.
|
|
4
4
|
//
|
|
5
|
-
// Composition, not inheritance:
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
// window), builds the config, and returns `new OpenAICompatProvider(config)`.
|
|
9
|
-
// Pure-config providers come from ./standardProviders.ts with no sibling at all.
|
|
5
|
+
// Composition, not inheritance: an official AI SDK language model supplies the
|
|
6
|
+
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
|
+
// extensions and local endpoint probes the SDK cannot represent.
|
|
10
8
|
|
|
11
9
|
import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
12
10
|
import type { Reasoning, ReserveSpec } from "./env.ts";
|
|
13
|
-
import {
|
|
14
|
-
import {
|
|
15
|
-
import { toProviderError,
|
|
11
|
+
import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
|
|
12
|
+
import type { LanguageModel } from "ai";
|
|
13
|
+
import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
|
|
16
14
|
import { validateGbnf, type Verdict } from "@plurnk/gbnf";
|
|
17
15
|
import { emitWarningOnce } from "./warnings.ts";
|
|
18
16
|
|
|
17
|
+
export type ProviderFetch = typeof globalThis.fetch;
|
|
18
|
+
|
|
19
19
|
// How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
|
|
20
20
|
// REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
|
|
21
21
|
// (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
|
|
@@ -33,9 +33,10 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
|
|
|
33
33
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
34
34
|
export type GrammarStyle = "none" | "llamacpp";
|
|
35
35
|
|
|
36
|
-
export type
|
|
36
|
+
export type AiSdkProviderConfig = {
|
|
37
37
|
model: string;
|
|
38
|
-
url
|
|
38
|
+
url?: string; // OpenAI-compatible chat-completions URL
|
|
39
|
+
languageModel?: LanguageModel; // native AI SDK provider model
|
|
39
40
|
fetchTimeoutMs: number;
|
|
40
41
|
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
41
42
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
@@ -43,7 +44,7 @@ export type OpenAICompatConfig = {
|
|
|
43
44
|
contextWindow?: number | null; // default null; caller resolves-or-fails (#419), narrows to required with the interface
|
|
44
45
|
reasoningStyle?: ReasoningStyle; // default "none"
|
|
45
46
|
countTokens?: (text: string) => number; // default chars/2 upper-bound heuristic
|
|
46
|
-
|
|
47
|
+
calculateCost?: (usage: ProviderUsage) => number; // default () => 0
|
|
47
48
|
source?: string; // telemetry source, e.g. "provider:openai"; default "provider"
|
|
48
49
|
grammarStyle?: GrammarStyle; // how a GBNF grammar is carried; default "none" (not sent)
|
|
49
50
|
// #518: send the OpenAI-standard `prompt_cache_key` set to workerId, so a
|
|
@@ -60,7 +61,6 @@ export type OpenAICompatConfig = {
|
|
|
60
61
|
firstPartyMetadata?: boolean; // forward per-turn attributions + client as Plurnk-* headers (plurnk only); default false
|
|
61
62
|
apiKeyRejectedMessage?: string; // #537: friendly hint when a PRESENT key is 401/403-rejected (distinct from unset); default undefined
|
|
62
63
|
eosText?: string; // #539: server-reported eos_token, stripped from the content tail (--special renders it as text); default undefined
|
|
63
|
-
balanceMetaKey?: string; // top-level response field carrying account balance (pico-USD) → validated meta.balancePico (plurnk only, #23); default unset
|
|
64
64
|
// Slot affinity wiring (provider-INTERNAL — never consumer-facing, #11).
|
|
65
65
|
supportsSlotPinning?: boolean; // backend accepts an `id_slot` body field (llama-server); default false
|
|
66
66
|
slotCount?: number | null; // probed slot count for pinning backends; default null
|
|
@@ -88,8 +88,7 @@ export type OpenAICompatConfig = {
|
|
|
88
88
|
// DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
|
|
89
89
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
90
90
|
// (greedy-under-mask loops without it, #9) — the VALUE is operator config;
|
|
91
|
-
// WHERE it applies stays mechanism.
|
|
92
|
-
// backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
|
|
91
|
+
// WHERE it applies stays mechanism.
|
|
93
92
|
temperature: number;
|
|
94
93
|
repeatPenalty: number;
|
|
95
94
|
// #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
|
|
@@ -107,7 +106,6 @@ export type OpenAICompatConfig = {
|
|
|
107
106
|
dryBase?: number;
|
|
108
107
|
dryAllowedLength?: number;
|
|
109
108
|
repeatLastN?: number;
|
|
110
|
-
retryDelayMs: number;
|
|
111
109
|
// Transient-failure retry budget — REQUIRED, no in-code default
|
|
112
110
|
// (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
|
|
113
111
|
// first failure; N = up to N retries on a transient error (§4, #18).
|
|
@@ -135,13 +133,6 @@ export type OpenAICompatConfig = {
|
|
|
135
133
|
tuningFloors?: boolean;
|
|
136
134
|
};
|
|
137
135
|
|
|
138
|
-
// Transient classifications worth retrying: rate_limit (429) and network_failure
|
|
139
|
-
// (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
|
|
140
|
-
// reject a fresh sample may satisfy, #548) rides the same bounded budget.
|
|
141
|
-
// unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
|
|
142
|
-
// retrying just burns time and budget.
|
|
143
|
-
const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
|
|
144
|
-
|
|
145
136
|
// #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
146
137
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
147
138
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -154,42 +145,6 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
|
|
|
154
145
|
return out;
|
|
155
146
|
};
|
|
156
147
|
|
|
157
|
-
// Sleep that rejects the moment `signal` aborts (caller cancellation must not
|
|
158
|
-
// wait out a backoff). Resolves normally on timeout.
|
|
159
|
-
const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
|
|
160
|
-
new Promise((resolve, reject) => {
|
|
161
|
-
if (signal?.aborted) { reject(signal.reason); return; }
|
|
162
|
-
const timer = setTimeout(resolve, ms);
|
|
163
|
-
signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
|
|
164
|
-
});
|
|
165
|
-
|
|
166
|
-
// SPEC §2 closed set. The four canonical values pass through; known per-backend
|
|
167
|
-
// synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
|
|
168
|
-
// SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
|
|
169
|
-
// backend names it -- core's `finishReason === "length"` truncation check (#425)
|
|
170
|
-
// is then an invariant by construction, not a convention each backend must
|
|
171
|
-
// independently honor. A non-empty value outside both the set and the table
|
|
172
|
-
// collapses to null AND warns once, so a new backend's unmapped cap string
|
|
173
|
-
// surfaces instead of silently becoming "no signal" (which would make core miss
|
|
174
|
-
// the truncation entirely). Case-folded: gemini shouts its reasons.
|
|
175
|
-
const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
|
|
176
|
-
["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
|
|
177
|
-
["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
|
|
178
|
-
["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
|
|
179
|
-
["tool_use", "tool_calls"],
|
|
180
|
-
["safety", "content_filter"], ["recitation", "content_filter"],
|
|
181
|
-
]);
|
|
182
|
-
const normalizeFinishReason = (raw: string | null): FinishReason => {
|
|
183
|
-
if (raw === null || raw.length === 0) return null;
|
|
184
|
-
const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
|
|
185
|
-
if (hit !== undefined) return hit;
|
|
186
|
-
emitWarningOnce(
|
|
187
|
-
`unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
|
|
188
|
-
"PLURNK_FINISH_REASON_UNKNOWN",
|
|
189
|
-
);
|
|
190
|
-
return null;
|
|
191
|
-
};
|
|
192
|
-
|
|
193
148
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
194
149
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
195
150
|
if (budget <= 1000) return "low";
|
|
@@ -235,9 +190,10 @@ const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string =
|
|
|
235
190
|
return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
|
|
236
191
|
};
|
|
237
192
|
|
|
238
|
-
export default class
|
|
193
|
+
export default class AiSdkProvider implements Provider {
|
|
239
194
|
#model: string;
|
|
240
|
-
#url: string;
|
|
195
|
+
#url: string | undefined;
|
|
196
|
+
#languageModel: LanguageModel | undefined;
|
|
241
197
|
#fetchTimeoutMs: number;
|
|
242
198
|
#streamIdleTimeoutMs: number | undefined;
|
|
243
199
|
#headers: Record<string, string>;
|
|
@@ -254,10 +210,9 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
254
210
|
#dryBase: number | undefined;
|
|
255
211
|
#dryAllowedLength: number | undefined;
|
|
256
212
|
#repeatLastN: number | undefined;
|
|
257
|
-
#retryDelayMs: number;
|
|
258
213
|
#reasoningStyle: ReasoningStyle;
|
|
259
214
|
#countTokens: (text: string) => number;
|
|
260
|
-
#
|
|
215
|
+
#calculateCost: (usage: ProviderUsage) => number;
|
|
261
216
|
#source: string;
|
|
262
217
|
#grammarStyle: GrammarStyle;
|
|
263
218
|
#promptCacheKey: boolean;
|
|
@@ -265,7 +220,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
265
220
|
#gbnfDebug: boolean;
|
|
266
221
|
#streaming: boolean;
|
|
267
222
|
#firstPartyMetadata: boolean;
|
|
268
|
-
#balanceMetaKey: string | undefined;
|
|
269
223
|
#supportsSlotPinning: boolean;
|
|
270
224
|
#slotCount: number | null;
|
|
271
225
|
#retryAttempts: number;
|
|
@@ -283,9 +237,13 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
283
237
|
// the honest capability signal for every other backend.
|
|
284
238
|
tokenize?: (text: string) => Promise<number[]>;
|
|
285
239
|
|
|
286
|
-
constructor(config:
|
|
240
|
+
constructor(config: AiSdkProviderConfig) {
|
|
287
241
|
this.#model = config.model;
|
|
288
242
|
this.#url = config.url;
|
|
243
|
+
this.#languageModel = config.languageModel;
|
|
244
|
+
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
245
|
+
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
246
|
+
}
|
|
289
247
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
290
248
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
291
249
|
this.#headers = config.headers ?? {};
|
|
@@ -295,8 +253,8 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
295
253
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
296
254
|
// required tuning fields must fail at construction, not silently send
|
|
297
255
|
// undefined sampling on every grammar request.
|
|
298
|
-
if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number"
|
|
299
|
-
throw new Error(`${config.source ?? "provider"}:
|
|
256
|
+
if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
|
|
257
|
+
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
|
|
300
258
|
}
|
|
301
259
|
this.#temperature = config.temperature;
|
|
302
260
|
this.#repeatPenalty = config.repeatPenalty;
|
|
@@ -305,11 +263,10 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
305
263
|
this.#dryBase = config.dryBase;
|
|
306
264
|
this.#dryAllowedLength = config.dryAllowedLength;
|
|
307
265
|
this.#repeatLastN = config.repeatLastN;
|
|
308
|
-
this.#retryDelayMs = config.retryDelayMs;
|
|
309
266
|
this.#retryAttempts = config.retryAttempts;
|
|
310
267
|
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
311
268
|
this.#countTokens = config.countTokens ?? heuristicTokens;
|
|
312
|
-
this.#
|
|
269
|
+
this.#calculateCost = config.calculateCost ?? (() => 0);
|
|
313
270
|
this.#source = config.source ?? "provider";
|
|
314
271
|
this.#grammarStyle = config.grammarStyle ?? "none";
|
|
315
272
|
this.#promptCacheKey = config.promptCacheKey ?? false;
|
|
@@ -320,7 +277,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
320
277
|
this.#apiKeyRejectedMessage = config.apiKeyRejectedMessage;
|
|
321
278
|
this.#eosText = config.eosText;
|
|
322
279
|
this.#hasApiKey = "Authorization" in this.#headers;
|
|
323
|
-
this.#balanceMetaKey = config.balanceMetaKey;
|
|
324
280
|
this.#supportsSlotPinning = config.supportsSlotPinning ?? false;
|
|
325
281
|
this.#slotCount = config.slotCount ?? null;
|
|
326
282
|
this.#topLogprobs = config.topLogprobs ?? null;
|
|
@@ -370,7 +326,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
370
326
|
get constrainsOutput(): boolean { return this.#grammarStyle !== "none"; }
|
|
371
327
|
|
|
372
328
|
countTokens(text: string): number { return this.#countTokens(text); }
|
|
373
|
-
|
|
329
|
+
calculateCost(usage: ProviderUsage): number { return this.#calculateCost(usage); }
|
|
374
330
|
|
|
375
331
|
// Maps the reasoning INTENT (off | adaptive | on+budget, #33) to the
|
|
376
332
|
// backend's wire mechanism — including under a transported grammar. The #32
|
|
@@ -566,18 +522,10 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
566
522
|
}
|
|
567
523
|
|
|
568
524
|
// Per-turn metadata bag (#23): pass the backend's non-standard top-level fields
|
|
569
|
-
//
|
|
570
|
-
//
|
|
571
|
-
// `balancePico` (finite pico-USD; dropped if non-numeric), renamed off its raw
|
|
572
|
-
// key so the consumer reads one canonical name. Undefined when nothing's there;
|
|
573
|
-
// the service merges this into its Turn metadata and filters what reaches clients.
|
|
525
|
+
// through verbatim. Providers do not reinterpret vendor currency or account
|
|
526
|
+
// metadata; a monetary value carries its own amount and currency.
|
|
574
527
|
#buildMeta(chunkMetadata: Record<string, unknown>): Record<string, unknown> | undefined {
|
|
575
528
|
const meta: Record<string, unknown> = { ...chunkMetadata };
|
|
576
|
-
if (this.#balanceMetaKey !== undefined) {
|
|
577
|
-
const raw = meta[this.#balanceMetaKey];
|
|
578
|
-
delete meta[this.#balanceMetaKey];
|
|
579
|
-
if (typeof raw === "number" && Number.isFinite(raw)) meta.balancePico = raw;
|
|
580
|
-
}
|
|
581
529
|
return Object.keys(meta).length > 0 ? meta : undefined;
|
|
582
530
|
}
|
|
583
531
|
|
|
@@ -634,61 +582,65 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
634
582
|
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
635
583
|
};
|
|
636
584
|
|
|
637
|
-
// Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
|
|
638
|
-
// (the budget is per-request, not shared across retries); the caller's
|
|
639
|
-
// signal spans them all. Retry only the transient classifications, prefer
|
|
640
|
-
// a server Retry-After over the backoff, and let the caller's abort cut
|
|
641
|
-
// through both the in-flight request and the backoff sleep.
|
|
642
|
-
const transport = this.#streaming ? chatCompletionStream : chatCompletion;
|
|
643
|
-
|
|
644
585
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
645
586
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
646
587
|
const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
|
|
647
|
-
const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
|
|
648
588
|
let raw;
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
raw = await transport({
|
|
655
|
-
url: this.#url,
|
|
589
|
+
try {
|
|
590
|
+
raw = this.#languageModel === undefined
|
|
591
|
+
? await executeOpenAICompatible({
|
|
592
|
+
url: this.#url!,
|
|
593
|
+
model: this.#model,
|
|
656
594
|
headers,
|
|
657
595
|
body,
|
|
658
|
-
|
|
596
|
+
messages,
|
|
597
|
+
signal,
|
|
659
598
|
fetch: this.#fetch,
|
|
599
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
600
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
601
|
+
retryAttempts: this.#retryAttempts,
|
|
602
|
+
streaming: this.#streaming,
|
|
660
603
|
captureRawBody: this.#rawBody,
|
|
604
|
+
})
|
|
605
|
+
: await executeAiSdkModel({
|
|
606
|
+
languageModel: this.#languageModel,
|
|
607
|
+
headers,
|
|
608
|
+
messages,
|
|
609
|
+
signal,
|
|
610
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
661
611
|
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
612
|
+
retryAttempts: this.#retryAttempts,
|
|
613
|
+
streaming: this.#streaming,
|
|
614
|
+
captureRawBody: this.#rawBody,
|
|
615
|
+
temperature: this.#tuningFloors
|
|
616
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
617
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
618
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
619
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
620
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
621
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
622
|
+
? sampling.frequency_penalty
|
|
623
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
624
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
625
|
+
? [sampling.stop]
|
|
626
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
627
|
+
? sampling.stop
|
|
628
|
+
: undefined,
|
|
629
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
630
|
+
maxOutputTokens: maxTokens,
|
|
631
|
+
reasoning: this.#reasoning.mode === "off"
|
|
632
|
+
? "none"
|
|
633
|
+
: this.#reasoning.mode === "adaptive"
|
|
634
|
+
? "provider-default"
|
|
635
|
+
: effortFromBudget(this.#reasoning.budget!),
|
|
662
636
|
});
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
// #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
|
|
669
|
-
// but fail-fast - a retry just re-incurs the same origin/edge timeout.
|
|
670
|
-
const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
|
|
671
|
-
// Terminal kind, edge failure, or budget spent -> surface the failure.
|
|
672
|
-
if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
|
|
673
|
-
const pe = toProviderError(err, this.#source);
|
|
674
|
-
// #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
|
|
675
|
-
// transport failure — surface the distinct, actionable hint (kind stays
|
|
676
|
-
// "unauthorized", so core's routing is unchanged) rather than the raw
|
|
677
|
-
// upstream JSON. Distinct from the unset-key throw at construction.
|
|
678
|
-
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
679
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
680
|
-
}
|
|
681
|
-
throw pe;
|
|
682
|
-
}
|
|
683
|
-
transportRetries.push({
|
|
684
|
-
attempt: attempt + 1,
|
|
685
|
-
kind,
|
|
686
|
-
elapsedMs: Math.round(performance.now() - attemptStarted),
|
|
687
|
-
message: err instanceof Error ? err.message : String(err),
|
|
688
|
-
});
|
|
689
|
-
const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
|
|
690
|
-
await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
|
|
637
|
+
} catch (err) {
|
|
638
|
+
if (signal?.aborted) throw err;
|
|
639
|
+
const pe = toProviderError(err, this.#source);
|
|
640
|
+
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
641
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
691
642
|
}
|
|
643
|
+
throw pe;
|
|
692
644
|
}
|
|
693
645
|
|
|
694
646
|
// #539: llama-server --special renders EOG tokens as text, so a turn ending
|
|
@@ -707,7 +659,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
707
659
|
// Discard/retry/escalate/self-correct is the consumer's policy.
|
|
708
660
|
let telemetry: TelemetryEvent[] | undefined;
|
|
709
661
|
let railsMeta: Record<string, unknown> | undefined;
|
|
710
|
-
const usage =
|
|
662
|
+
const usage = raw.usage;
|
|
711
663
|
const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
|
|
712
664
|
if (observedGrammar !== undefined) {
|
|
713
665
|
const verdict = this.#grammarVerdict(observedGrammar, raw.content);
|
|
@@ -725,7 +677,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
725
677
|
// invisible, billed (12,288 billed vs 1,033 chars visible, live).
|
|
726
678
|
// countTokens OVERCOUNTS text (chars/2 upper bound), so billed
|
|
727
679
|
// exceeding visible-plus-slack is real vanishing, not estimator noise.
|
|
728
|
-
const visible = this.#countTokens(raw.content) + this.#countTokens(raw.
|
|
680
|
+
const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
|
|
729
681
|
if (sendGrammar !== undefined && usage.completion > visible + 64) {
|
|
730
682
|
(telemetry ??= []).push({
|
|
731
683
|
source: this.#source,
|
|
@@ -736,30 +688,23 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
736
688
|
}
|
|
737
689
|
}
|
|
738
690
|
|
|
739
|
-
const builtMeta = this.#buildMeta(raw.
|
|
740
|
-
const
|
|
741
|
-
const
|
|
742
|
-
? { ...builtMeta, ...railsMeta, ...retryMeta }
|
|
743
|
-
: builtMeta;
|
|
744
|
-
|
|
745
|
-
// #36: surface per-token logprobs + their mean when the backend returned
|
|
746
|
-
// them (only possible when the flag requested them). Absent otherwise —
|
|
747
|
-
// never synthesized.
|
|
748
|
-
const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
691
|
+
const builtMeta = this.#buildMeta(raw.metadata);
|
|
692
|
+
const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
|
|
693
|
+
const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
749
694
|
const meanLogprob = logprobs !== undefined
|
|
750
|
-
? logprobs.reduce((sum,
|
|
695
|
+
? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
|
|
751
696
|
: undefined;
|
|
752
697
|
|
|
753
698
|
return {
|
|
754
699
|
assistant: {
|
|
755
700
|
content: raw.content,
|
|
756
|
-
reasoning: raw.
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
|
|
701
|
+
reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
|
|
702
|
+
...(raw.reasoningEncrypted.length > 0
|
|
703
|
+
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
704
|
+
: {}),
|
|
760
705
|
usage,
|
|
761
|
-
finishReason:
|
|
762
|
-
model: raw.model
|
|
706
|
+
finishReason: raw.finishReason,
|
|
707
|
+
model: raw.model,
|
|
763
708
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
764
709
|
},
|
|
765
710
|
assistantRaw: raw,
|
package/src/Mock.test.ts
CHANGED
|
@@ -31,9 +31,9 @@ test("Mock: countTokens('') is 0; non-empty is a positive integer", () => {
|
|
|
31
31
|
assert.ok(Number.isInteger(n) && n > 0);
|
|
32
32
|
});
|
|
33
33
|
|
|
34
|
-
test("Mock:
|
|
34
|
+
test("Mock: calculateCost zero usage is 0 (free)", () => {
|
|
35
35
|
const m = build();
|
|
36
|
-
assert.equal(m.
|
|
36
|
+
assert.equal(m.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 0);
|
|
37
37
|
});
|
|
38
38
|
|
|
39
39
|
// — Transport (SPEC §10.7, §10.10) —
|
|
@@ -109,7 +109,7 @@ test("Mock: exhausted queue throws a specific error", async () => {
|
|
|
109
109
|
// -- #507: the reserve surface lives on the Provider CONTRACT, and Mock drives core's partition suite --
|
|
110
110
|
|
|
111
111
|
test("#507 the reserve getters are on the Provider interface (not just the concrete class)", () => {
|
|
112
|
-
// Typing against the
|
|
112
|
+
// Typing against the contract catches a getter-only concrete surface.
|
|
113
113
|
const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
|
|
114
114
|
try {
|
|
115
115
|
process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
|
package/src/Mock.ts
CHANGED
|
@@ -68,7 +68,7 @@ export default class Mock implements Provider {
|
|
|
68
68
|
}
|
|
69
69
|
|
|
70
70
|
// Mock is free.
|
|
71
|
-
|
|
71
|
+
calculateCost(_usage: ProviderUsage): number { return 0; }
|
|
72
72
|
|
|
73
73
|
async generate({ signal }: { messages: ChatMessage[]; workerId?: string; signal?: AbortSignal }): Promise<{ assistant: MockReturnedAssistant; assistantRaw: unknown }> {
|
|
74
74
|
// Honor abort before consuming the queue — an aborted call makes no
|
package/src/Pool.test.ts
CHANGED
|
@@ -28,7 +28,7 @@ const backend = (opts: FakeOpts = {}) => {
|
|
|
28
28
|
...(opts.completionReserve !== undefined ? { completionReserve: opts.completionReserve } : {}),
|
|
29
29
|
...(opts.tokenize ? { tokenize: async (t: string) => [t.length] } : {}),
|
|
30
30
|
countTokens: (t: string) => t.length,
|
|
31
|
-
|
|
31
|
+
calculateCost: () => opts.cost ?? 0,
|
|
32
32
|
generate: async (args: Parameters<Provider["generate"]>[0]): Promise<ProviderResponse> => {
|
|
33
33
|
served.push(args.workerId);
|
|
34
34
|
if (opts.throws !== undefined) throw opts.throws;
|
|
@@ -88,10 +88,10 @@ test("Pool: tokenize is exposed iff every backend has it", () => {
|
|
|
88
88
|
assert.equal(new Pool([backend({ tokenize: true }).b, backend({ tokenize: false }).b]).tokenize, undefined);
|
|
89
89
|
});
|
|
90
90
|
|
|
91
|
-
test("Pool: countTokens +
|
|
91
|
+
test("Pool: countTokens + calculateCost delegate to a backend", () => {
|
|
92
92
|
const p = new Pool([backend({ cost: 42 }).b]);
|
|
93
93
|
assert.equal(p.countTokens("abcd"), 4);
|
|
94
|
-
assert.equal(p.
|
|
94
|
+
assert.equal(p.calculateCost({ prompt: 0, completion: 0, reasoning: 0, cached: 0, total: 0 }), 42);
|
|
95
95
|
});
|
|
96
96
|
|
|
97
97
|
// --- dispatch: round-robin + affinity ---
|
package/src/Pool.ts
CHANGED
|
@@ -83,7 +83,7 @@ export default class Pool implements Provider {
|
|
|
83
83
|
}
|
|
84
84
|
|
|
85
85
|
countTokens(text: string): number { return this.#backends[0].countTokens(text); }
|
|
86
|
-
|
|
86
|
+
calculateCost(usage: ProviderUsage): number { return this.#backends[0].calculateCost(usage); }
|
|
87
87
|
|
|
88
88
|
// --- dispatch ---
|
|
89
89
|
|
|
@@ -2,7 +2,6 @@ import test, { mock } from "node:test";
|
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
|
|
4
4
|
|
|
5
|
-
const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, costFor: () => 0, generate: async () => { throw new Error("unused"); } };
|
|
6
5
|
const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
|
|
7
6
|
async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
|
|
8
7
|
|
|
@@ -10,16 +9,16 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
|
|
|
10
9
|
// exercise the resolution + two-tier instantiation this module owns; the active
|
|
11
10
|
// alias is driven end-to-end by loadActiveProvider below.
|
|
12
11
|
|
|
13
|
-
// —
|
|
12
|
+
// — provider resolution (SPEC §5) —
|
|
14
13
|
|
|
15
14
|
const fullEnv = Object.freeze({
|
|
16
15
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
|
|
17
16
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
18
|
-
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
17
|
+
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
|
|
19
18
|
OPENAI_BASE_URL: "http://x",
|
|
20
19
|
});
|
|
21
20
|
|
|
22
|
-
test("instantiateProvider:
|
|
21
|
+
test("instantiateProvider: cataloged name resolves in-framework, no scan, no import", async () => {
|
|
23
22
|
resetDiscoveryCache();
|
|
24
23
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
25
24
|
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
@@ -36,24 +35,33 @@ test("instantiateProvider: standard name resolves in-framework, no scan, no impo
|
|
|
36
35
|
mock.restoreAll();
|
|
37
36
|
});
|
|
38
37
|
|
|
39
|
-
test("instantiateProvider:
|
|
38
|
+
test("instantiateProvider: an installed AI SDK provider resolves through discovery", async () => {
|
|
40
39
|
resetDiscoveryCache();
|
|
41
40
|
const calls: unknown[] = [];
|
|
42
|
-
const p = await instantiateProvider("
|
|
43
|
-
async (specifier) => {
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
41
|
+
const p = await instantiateProvider("acme", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "model-a",
|
|
42
|
+
async (specifier) => {
|
|
43
|
+
calls.push(specifier);
|
|
44
|
+
return { default: { languageModel: (model: string) => { calls.push(model); return {} as never; } } };
|
|
45
|
+
},
|
|
46
|
+
mapOf({ acme: "@acme/ai-provider" }));
|
|
47
|
+
assert.equal(p.model, "model-a");
|
|
48
|
+
assert.equal(p.contextWindow, 8192);
|
|
49
|
+
assert.deepEqual(calls, ["@acme/ai-provider", "model-a"]);
|
|
47
50
|
});
|
|
48
51
|
|
|
49
|
-
test("instantiateProvider: a per-alias baseUrl
|
|
52
|
+
test("instantiateProvider: a per-alias baseUrl drives the built-in Ollama probe", async () => {
|
|
50
53
|
resetDiscoveryCache();
|
|
51
|
-
|
|
54
|
+
const calls: string[] = [];
|
|
55
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
56
|
+
calls.push(String(url));
|
|
57
|
+
return new Response(JSON.stringify({ model_info: { "qwen.context_length": 32768 } }));
|
|
58
|
+
});
|
|
52
59
|
await instantiateProvider("ollama", { ...fullEnv }, "qwen2.5-coder",
|
|
53
|
-
async () => ({
|
|
54
|
-
mapOf({
|
|
60
|
+
async () => ({}),
|
|
61
|
+
mapOf({}),
|
|
55
62
|
"http://nook:11434");
|
|
56
|
-
assert.deepEqual(
|
|
63
|
+
assert.deepEqual(calls, ["http://nook:11434/api/show"]);
|
|
64
|
+
mock.restoreAll();
|
|
57
65
|
});
|
|
58
66
|
|
|
59
67
|
test("instantiateProvider: a per-alias baseUrl drives the standard openai probe to the override host", async () => {
|
|
@@ -74,14 +82,14 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
|
|
|
74
82
|
test("instantiateProvider: a THIRD-PARTY scope is discovered — name maps to its package", async () => {
|
|
75
83
|
resetDiscoveryCache();
|
|
76
84
|
const imports: string[] = [];
|
|
77
|
-
const p = await instantiateProvider("foo", { ...fullEnv }, "m",
|
|
78
|
-
async (specifier) => { imports.push(specifier); return { default: {
|
|
85
|
+
const p = await instantiateProvider("foo", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096" }, "m",
|
|
86
|
+
async (specifier) => { imports.push(specifier); return { default: { languageModel: () => ({} as never) } }; },
|
|
79
87
|
mapOf({ foo: "@acme/acme-provider-foo" }));
|
|
80
|
-
assert.equal(p,
|
|
88
|
+
assert.equal(p.model, "m");
|
|
81
89
|
assert.deepEqual(imports, ["@acme/acme-provider-foo"]); // not an @plurnk/ specifier
|
|
82
90
|
});
|
|
83
91
|
|
|
84
|
-
test("instantiateProvider: a
|
|
92
|
+
test("instantiateProvider: a cataloged name is authoritative — a scanned package of the same name is shadowed", async () => {
|
|
85
93
|
resetDiscoveryCache();
|
|
86
94
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
87
95
|
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
@@ -100,7 +108,7 @@ test("instantiateProvider: unknown provider throws — no standard, no discovere
|
|
|
100
108
|
resetDiscoveryCache();
|
|
101
109
|
await assert.rejects(
|
|
102
110
|
() => instantiateProvider("nope", { ...fullEnv }, "m", async () => ({}), mapOf({})),
|
|
103
|
-
/unknown provider "nope"
|
|
111
|
+
/unknown provider "nope"/,
|
|
104
112
|
);
|
|
105
113
|
});
|
|
106
114
|
|
|
@@ -116,13 +124,13 @@ test("instantiateProvider: an untrusted (skipped) provider gives a precise error
|
|
|
116
124
|
assert.deepEqual(imports, []); // never imported an untrusted package
|
|
117
125
|
});
|
|
118
126
|
|
|
119
|
-
test("instantiateProvider: discovered package
|
|
127
|
+
test("instantiateProvider: discovered package must export an AI SDK provider", async () => {
|
|
120
128
|
resetDiscoveryCache();
|
|
121
129
|
await assert.rejects(
|
|
122
130
|
() => instantiateProvider("broken", { ...fullEnv }, "m",
|
|
123
131
|
async () => ({ default: {} }),
|
|
124
132
|
mapOf({ broken: "@acme/acme-provider-broken" })),
|
|
125
|
-
/@acme\/acme-provider-broken default export is not
|
|
133
|
+
/@acme\/acme-provider-broken default export is not an AI SDK provider/,
|
|
126
134
|
);
|
|
127
135
|
});
|
|
128
136
|
|
|
@@ -174,6 +182,8 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
174
182
|
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
175
183
|
FIREWORKS_API_KEY: "fw",
|
|
176
184
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
185
|
+
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_REASONING_STYLE: "effort_explicit",
|
|
186
|
+
PLURNK_PROVIDERS_TOP_LOGPROBS: "2",
|
|
177
187
|
PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
|
|
178
188
|
PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
|
|
179
189
|
};
|
|
@@ -184,6 +194,9 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
184
194
|
await fast.generate({ workerId: "fast-worker", messages: [] });
|
|
185
195
|
await standard.generate({ workerId: "standard-worker", messages: [] });
|
|
186
196
|
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
|
|
197
|
+
assert.deepEqual(bodies.map((body) => body.prompt_cache_key), ["fast-worker", "standard-worker"]);
|
|
198
|
+
assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["none", "none"]);
|
|
199
|
+
assert.deepEqual(bodies.map((body) => body.top_logprobs), [2, 2]);
|
|
187
200
|
assert.deepEqual(bodies.map((body) => body.model), [
|
|
188
201
|
"accounts/fireworks/routers/glm-5p2-fast",
|
|
189
202
|
"accounts/fireworks/models/deepseek-v4-pro",
|
|
@@ -191,13 +204,13 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
191
204
|
mock.restoreAll();
|
|
192
205
|
});
|
|
193
206
|
|
|
194
|
-
test("loadActiveProvider: resolves the alias cascade
|
|
207
|
+
test("loadActiveProvider: resolves the alias cascade to an installed AI SDK provider", async () => {
|
|
195
208
|
resetDiscoveryCache();
|
|
196
|
-
const env = { ...fullEnv, PLURNK_MODEL: "
|
|
209
|
+
const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096", PLURNK_MODEL: "custom", PLURNK_MODEL_custom: "acme/model-a" } as NodeJS.ProcessEnv;
|
|
197
210
|
const p = await loadActiveProvider(env,
|
|
198
|
-
async () => ({ default: {
|
|
199
|
-
mapOf({
|
|
200
|
-
assert.equal(p,
|
|
211
|
+
async () => ({ default: { languageModel: () => ({} as never) } }),
|
|
212
|
+
mapOf({ acme: "@acme/ai-provider" }));
|
|
213
|
+
assert.equal(p.model, "model-a");
|
|
201
214
|
});
|
|
202
215
|
|
|
203
216
|
test("loadActiveProvider: throws a named error when no alias is active", async () => {
|