@plurnk/plurnk-providers 1.3.5 → 1.3.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.defaults +35 -45
- package/README.md +44 -53
- package/SPEC.md +215 -371
- package/dist/AiSdkProvider.d.ts +78 -0
- package/dist/AiSdkProvider.d.ts.map +1 -0
- package/dist/AiSdkProvider.js +591 -0
- package/dist/AiSdkProvider.js.map +1 -0
- package/dist/OpenAICompat.d.ts +1 -2
- package/dist/OpenAICompat.d.ts.map +1 -1
- package/dist/OpenAICompat.js +39 -117
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.d.ts.map +1 -1
- package/dist/ProviderRegistry.js +37 -24
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/aiSdkTransport.d.ts +52 -0
- package/dist/aiSdkTransport.d.ts.map +1 -0
- package/dist/aiSdkTransport.js +294 -0
- package/dist/aiSdkTransport.js.map +1 -0
- package/dist/catalogProvider.d.ts +15 -0
- package/dist/catalogProvider.d.ts.map +1 -0
- package/dist/catalogProvider.js +103 -0
- package/dist/catalogProvider.js.map +1 -0
- package/dist/compatibleProvider.d.ts +3 -0
- package/dist/compatibleProvider.d.ts.map +1 -0
- package/dist/compatibleProvider.js +146 -0
- package/dist/compatibleProvider.js.map +1 -0
- package/dist/discover.d.ts.map +1 -1
- package/dist/discover.js.map +1 -1
- package/dist/env.d.ts +1 -0
- package/dist/env.d.ts.map +1 -1
- package/dist/env.js +13 -6
- package/dist/env.js.map +1 -1
- package/dist/index.d.ts +4 -6
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +3 -7
- package/dist/index.js.map +1 -1
- package/dist/ollama.d.ts +3 -0
- package/dist/ollama.d.ts.map +1 -0
- package/dist/ollama.js +39 -0
- package/dist/ollama.js.map +1 -0
- package/dist/openai.d.ts +2 -4
- package/dist/openai.d.ts.map +1 -1
- package/dist/openai.js +1 -2
- package/dist/openai.js.map +1 -1
- package/dist/sdkModels.d.ts +13 -0
- package/dist/sdkModels.d.ts.map +1 -0
- package/dist/sdkModels.js +153 -0
- package/dist/sdkModels.js.map +1 -0
- package/dist/standardProviders.d.ts.map +1 -1
- package/dist/standardProviders.js +0 -1
- package/dist/standardProviders.js.map +1 -1
- package/dist/telemetry.d.ts.map +1 -1
- package/dist/telemetry.js +20 -9
- package/dist/telemetry.js.map +1 -1
- package/dist/types.d.ts +3 -2
- package/dist/types.d.ts.map +1 -1
- package/package.json +18 -10
- package/src/{OpenAICompat.test.ts → AiSdkProvider.test.ts} +193 -152
- package/src/{OpenAICompat.ts → AiSdkProvider.ts} +83 -127
- package/src/Mock.test.ts +1 -1
- package/src/ProviderRegistry.test.ts +40 -27
- package/src/ProviderRegistry.ts +35 -24
- package/src/aiSdkTransport.test.ts +253 -0
- package/src/aiSdkTransport.ts +369 -0
- package/src/boundaries.test.ts +2 -2
- package/src/catalogProvider.test.ts +100 -0
- package/src/catalogProvider.ts +151 -0
- package/src/compatibleProvider.test.ts +44 -0
- package/src/compatibleProvider.ts +205 -0
- package/src/discover.test.ts +12 -12
- package/src/discover.ts +3 -6
- package/src/env.ts +14 -6
- package/src/index.ts +6 -10
- package/src/ollama.ts +63 -0
- package/src/openai.ts +2 -8
- package/src/sdkModels.test.ts +47 -0
- package/src/sdkModels.ts +194 -0
- package/src/telemetry.test.ts +17 -10
- package/src/telemetry.ts +22 -14
- package/src/types.ts +5 -8
- package/src/aiSdkAdapter.spike.test.ts +0 -242
- package/src/openaiStream.ts +0 -310
- package/src/standardProviders.test.ts +0 -939
- package/src/standardProviders.ts +0 -631
|
@@ -1,21 +1,21 @@
|
|
|
1
|
-
//
|
|
1
|
+
// PLURNK adapter over an AI SDK language model. Implements the universal generate()
|
|
2
2
|
// spine — signal merging, the SSE call, usage mapping, finishReason
|
|
3
3
|
// normalization, response assembly — that every sibling had duplicated.
|
|
4
4
|
//
|
|
5
|
-
// Composition, not inheritance:
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
// window), builds the config, and returns `new OpenAICompatProvider(config)`.
|
|
9
|
-
// Pure-config providers come from ./standardProviders.ts with no sibling at all.
|
|
5
|
+
// Composition, not inheritance: an official AI SDK language model supplies the
|
|
6
|
+
// ordinary vendor protocol. The compatible URL path remains only for PLURNK
|
|
7
|
+
// extensions and local endpoint probes the SDK cannot represent.
|
|
10
8
|
|
|
11
9
|
import type { ChatMessage, FinishReason, Provider, ProviderResponse, ProviderUsage } from "./types.ts";
|
|
12
10
|
import type { Reasoning, ReserveSpec } from "./env.ts";
|
|
13
|
-
import {
|
|
14
|
-
import {
|
|
15
|
-
import { toProviderError,
|
|
11
|
+
import { executeAiSdkModel, executeOpenAICompatible } from "./aiSdkTransport.ts";
|
|
12
|
+
import type { LanguageModel } from "ai";
|
|
13
|
+
import { toProviderError, ProviderError, type TelemetryEvent } from "./telemetry.ts";
|
|
16
14
|
import { validateGbnf, type Verdict } from "@plurnk/gbnf";
|
|
17
15
|
import { emitWarningOnce } from "./warnings.ts";
|
|
18
16
|
|
|
17
|
+
export type ProviderFetch = typeof globalThis.fetch;
|
|
18
|
+
|
|
19
19
|
// How the reasoning intent (PLURNK_PROVIDERS_REASONING: off | adaptive | on, plus
|
|
20
20
|
// REASONING_BUDGET iff on — #32/#33) translates to each backend's wire mechanism
|
|
21
21
|
// (SPEC §4); the per-style mapping lives in #reasoningBody. Non-obvious ones:
|
|
@@ -33,9 +33,10 @@ export type ReasoningStyle = "none" | "think" | "include_reasoning" | "effort" |
|
|
|
33
33
|
// service-managed constrained sampling; endpoint-owned settings are not inferred.
|
|
34
34
|
export type GrammarStyle = "none" | "llamacpp";
|
|
35
35
|
|
|
36
|
-
export type
|
|
36
|
+
export type AiSdkProviderConfig = {
|
|
37
37
|
model: string;
|
|
38
|
-
url
|
|
38
|
+
url?: string; // OpenAI-compatible chat-completions URL
|
|
39
|
+
languageModel?: LanguageModel; // native AI SDK provider model
|
|
39
40
|
fetchTimeoutMs: number;
|
|
40
41
|
streamIdleTimeoutMs?: number; // streamed body inter-chunk deadline; zero/unset disables
|
|
41
42
|
headers?: Record<string, string>; // fully-resolved request headers (incl. auth); default {}
|
|
@@ -87,8 +88,7 @@ export type OpenAICompatConfig = {
|
|
|
87
88
|
// DEFAULT for EVERY request, spread UNDER caller sampling (#30/endpoint#7).
|
|
88
89
|
// `repeatPenalty` is the FLOOR the provider manages wherever a grammar rides
|
|
89
90
|
// (greedy-under-mask loops without it, #9) — the VALUE is operator config;
|
|
90
|
-
// WHERE it applies stays mechanism.
|
|
91
|
-
// backoff base (attempt N waits retryDelayMs * 2^(N-1); Retry-After wins).
|
|
91
|
+
// WHERE it applies stays mechanism.
|
|
92
92
|
temperature: number;
|
|
93
93
|
repeatPenalty: number;
|
|
94
94
|
// #426: anti-degeneration guard on the CLOUD path (grammarStyle "none"), where the
|
|
@@ -106,7 +106,6 @@ export type OpenAICompatConfig = {
|
|
|
106
106
|
dryBase?: number;
|
|
107
107
|
dryAllowedLength?: number;
|
|
108
108
|
repeatLastN?: number;
|
|
109
|
-
retryDelayMs: number;
|
|
110
109
|
// Transient-failure retry budget — REQUIRED, no in-code default
|
|
111
110
|
// (PLURNK_PROVIDERS_RETRY_ATTEMPTS, a non-negative int): 0 = surface the
|
|
112
111
|
// first failure; N = up to N retries on a transient error (§4, #18).
|
|
@@ -134,13 +133,6 @@ export type OpenAICompatConfig = {
|
|
|
134
133
|
tuningFloors?: boolean;
|
|
135
134
|
};
|
|
136
135
|
|
|
137
|
-
// Transient classifications worth retrying: rate_limit (429) and network_failure
|
|
138
|
-
// (5xx, timeout, connection reset) are transport; grammar_invalid (a 422 output
|
|
139
|
-
// reject a fresh sample may satisfy, #548) rides the same bounded budget.
|
|
140
|
-
// unauthorized, quota_exceeded, invalid_response, model_refused are terminal —
|
|
141
|
-
// retrying just burns time and budget.
|
|
142
|
-
const RETRYABLE: ReadonlySet<string> = new Set(["rate_limit", "network_failure", "grammar_invalid"]);
|
|
143
|
-
|
|
144
136
|
// #539: drop trailing occurrences of a server-rendered EOG marker. llama-server
|
|
145
137
|
// under --special renders EOS as literal text, so a raw-EOS-ended turn carries a
|
|
146
138
|
// trailing <eos> the grammar never sanctioned. Trailing-only + exact-match, so it
|
|
@@ -153,42 +145,6 @@ const stripTrailingSpecial = (content: string, marker: string): string => {
|
|
|
153
145
|
return out;
|
|
154
146
|
};
|
|
155
147
|
|
|
156
|
-
// Sleep that rejects the moment `signal` aborts (caller cancellation must not
|
|
157
|
-
// wait out a backoff). Resolves normally on timeout.
|
|
158
|
-
const sleepWithAbort = (ms: number, signal: AbortSignal | undefined): Promise<void> =>
|
|
159
|
-
new Promise((resolve, reject) => {
|
|
160
|
-
if (signal?.aborted) { reject(signal.reason); return; }
|
|
161
|
-
const timer = setTimeout(resolve, ms);
|
|
162
|
-
signal?.addEventListener("abort", () => { clearTimeout(timer); reject(signal.reason); }, { once: true });
|
|
163
|
-
});
|
|
164
|
-
|
|
165
|
-
// SPEC §2 closed set. The four canonical values pass through; known per-backend
|
|
166
|
-
// synonyms translate INTO them (anthropic max_tokens/end_turn, gemini MAX_TOKENS/
|
|
167
|
-
// SAFETY/RECITATION) so a token-cap hit canonicalizes to "length" whatever the
|
|
168
|
-
// backend names it -- core's `finishReason === "length"` truncation check (#425)
|
|
169
|
-
// is then an invariant by construction, not a convention each backend must
|
|
170
|
-
// independently honor. A non-empty value outside both the set and the table
|
|
171
|
-
// collapses to null AND warns once, so a new backend's unmapped cap string
|
|
172
|
-
// surfaces instead of silently becoming "no signal" (which would make core miss
|
|
173
|
-
// the truncation entirely). Case-folded: gemini shouts its reasons.
|
|
174
|
-
const FINISH_SYNONYMS = new Map<string, Exclude<FinishReason, null>>([
|
|
175
|
-
["stop", "stop"], ["length", "length"], ["tool_calls", "tool_calls"], ["content_filter", "content_filter"],
|
|
176
|
-
["max_tokens", "length"], ["model_length", "length"], ["max_completion_tokens", "length"],
|
|
177
|
-
["end_turn", "stop"], ["stop_sequence", "stop"], ["eos_token", "stop"],
|
|
178
|
-
["tool_use", "tool_calls"],
|
|
179
|
-
["safety", "content_filter"], ["recitation", "content_filter"],
|
|
180
|
-
]);
|
|
181
|
-
const normalizeFinishReason = (raw: string | null): FinishReason => {
|
|
182
|
-
if (raw === null || raw.length === 0) return null;
|
|
183
|
-
const hit = FINISH_SYNONYMS.get(raw.toLowerCase());
|
|
184
|
-
if (hit !== undefined) return hit;
|
|
185
|
-
emitWarningOnce(
|
|
186
|
-
`unrecognized finish_reason "${raw}"; treated as no-signal (finishReason=null). If it denotes a token-cap hit, core's length-cap detection will miss it -- add it to FINISH_SYNONYMS.`,
|
|
187
|
-
"PLURNK_FINISH_REASON_UNKNOWN",
|
|
188
|
-
);
|
|
189
|
-
return null;
|
|
190
|
-
};
|
|
191
|
-
|
|
192
148
|
// Shared budget→effort breakpoints (xai and google had identical copies).
|
|
193
149
|
export const effortFromBudget = (budget: number): "low" | "medium" | "high" => {
|
|
194
150
|
if (budget <= 1000) return "low";
|
|
@@ -234,9 +190,10 @@ const describeUnenforced = (v: Exclude<Verdict, { status: "accept" }>): string =
|
|
|
234
190
|
return `grammar not enforced: output is an incomplete match of the transported grammar — a valid prefix of ${v.pos} code points that never terminated`;
|
|
235
191
|
};
|
|
236
192
|
|
|
237
|
-
export default class
|
|
193
|
+
export default class AiSdkProvider implements Provider {
|
|
238
194
|
#model: string;
|
|
239
|
-
#url: string;
|
|
195
|
+
#url: string | undefined;
|
|
196
|
+
#languageModel: LanguageModel | undefined;
|
|
240
197
|
#fetchTimeoutMs: number;
|
|
241
198
|
#streamIdleTimeoutMs: number | undefined;
|
|
242
199
|
#headers: Record<string, string>;
|
|
@@ -253,7 +210,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
253
210
|
#dryBase: number | undefined;
|
|
254
211
|
#dryAllowedLength: number | undefined;
|
|
255
212
|
#repeatLastN: number | undefined;
|
|
256
|
-
#retryDelayMs: number;
|
|
257
213
|
#reasoningStyle: ReasoningStyle;
|
|
258
214
|
#countTokens: (text: string) => number;
|
|
259
215
|
#calculateCost: (usage: ProviderUsage) => number;
|
|
@@ -281,9 +237,13 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
281
237
|
// the honest capability signal for every other backend.
|
|
282
238
|
tokenize?: (text: string) => Promise<number[]>;
|
|
283
239
|
|
|
284
|
-
constructor(config:
|
|
240
|
+
constructor(config: AiSdkProviderConfig) {
|
|
285
241
|
this.#model = config.model;
|
|
286
242
|
this.#url = config.url;
|
|
243
|
+
this.#languageModel = config.languageModel;
|
|
244
|
+
if ((this.#url === undefined) === (this.#languageModel === undefined)) {
|
|
245
|
+
throw new Error(`${config.source ?? "provider"}: configure exactly one AI SDK model or OpenAI-compatible URL`);
|
|
246
|
+
}
|
|
287
247
|
this.#fetchTimeoutMs = config.fetchTimeoutMs;
|
|
288
248
|
this.#streamIdleTimeoutMs = config.streamIdleTimeoutMs;
|
|
289
249
|
this.#headers = config.headers ?? {};
|
|
@@ -293,8 +253,8 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
293
253
|
// Loud guard: an out-of-date consumer (stale plugin dist) omitting the
|
|
294
254
|
// required tuning fields must fail at construction, not silently send
|
|
295
255
|
// undefined sampling on every grammar request.
|
|
296
|
-
if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number"
|
|
297
|
-
throw new Error(`${config.source ?? "provider"}:
|
|
256
|
+
if (typeof config.temperature !== "number" || typeof config.repeatPenalty !== "number") {
|
|
257
|
+
throw new Error(`${config.source ?? "provider"}: AiSdkProviderConfig requires temperature + repeatPenalty (PLURNK_PROVIDERS_TEMPERATURE / _REPEAT_PENALTY)`);
|
|
298
258
|
}
|
|
299
259
|
this.#temperature = config.temperature;
|
|
300
260
|
this.#repeatPenalty = config.repeatPenalty;
|
|
@@ -303,7 +263,6 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
303
263
|
this.#dryBase = config.dryBase;
|
|
304
264
|
this.#dryAllowedLength = config.dryAllowedLength;
|
|
305
265
|
this.#repeatLastN = config.repeatLastN;
|
|
306
|
-
this.#retryDelayMs = config.retryDelayMs;
|
|
307
266
|
this.#retryAttempts = config.retryAttempts;
|
|
308
267
|
this.#reasoningStyle = config.reasoningStyle ?? "none";
|
|
309
268
|
this.#countTokens = config.countTokens ?? heuristicTokens;
|
|
@@ -623,61 +582,65 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
623
582
|
...(this.#promptCacheKey ? { prompt_cache_key: workerId } : {}),
|
|
624
583
|
};
|
|
625
584
|
|
|
626
|
-
// Transient-failure retry (#18). Each attempt gets a FRESH fetch timeout
|
|
627
|
-
// (the budget is per-request, not shared across retries); the caller's
|
|
628
|
-
// signal spans them all. Retry only the transient classifications, prefer
|
|
629
|
-
// a server Retry-After over the backoff, and let the caller's abort cut
|
|
630
|
-
// through both the in-flight request and the backoff sleep.
|
|
631
|
-
const transport = this.#streaming ? chatCompletionStream : chatCompletion;
|
|
632
|
-
|
|
633
585
|
// Per-request headers = static auth/routing + any first-party telemetry.
|
|
634
586
|
const metaHeaders = this.#metadataHeaders(attributions, client, strikes, workerId, primaryWorkerId, workspaceId, loop, turn);
|
|
635
587
|
const headers = Object.keys(metaHeaders).length > 0 ? { ...this.#headers, ...metaHeaders } : this.#headers;
|
|
636
|
-
const transportRetries: Array<{ attempt: number; kind: string; elapsedMs: number; message: string }> = [];
|
|
637
588
|
let raw;
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
641
|
-
|
|
642
|
-
|
|
643
|
-
raw = await transport({
|
|
644
|
-
url: this.#url,
|
|
589
|
+
try {
|
|
590
|
+
raw = this.#languageModel === undefined
|
|
591
|
+
? await executeOpenAICompatible({
|
|
592
|
+
url: this.#url!,
|
|
593
|
+
model: this.#model,
|
|
645
594
|
headers,
|
|
646
595
|
body,
|
|
647
|
-
|
|
596
|
+
messages,
|
|
597
|
+
signal,
|
|
648
598
|
fetch: this.#fetch,
|
|
599
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
600
|
+
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
601
|
+
retryAttempts: this.#retryAttempts,
|
|
602
|
+
streaming: this.#streaming,
|
|
649
603
|
captureRawBody: this.#rawBody,
|
|
604
|
+
})
|
|
605
|
+
: await executeAiSdkModel({
|
|
606
|
+
languageModel: this.#languageModel,
|
|
607
|
+
headers,
|
|
608
|
+
messages,
|
|
609
|
+
signal,
|
|
610
|
+
fetchTimeoutMs: this.#fetchTimeoutMs,
|
|
650
611
|
streamIdleTimeoutMs: this.#streamIdleTimeoutMs,
|
|
612
|
+
retryAttempts: this.#retryAttempts,
|
|
613
|
+
streaming: this.#streaming,
|
|
614
|
+
captureRawBody: this.#rawBody,
|
|
615
|
+
temperature: this.#tuningFloors
|
|
616
|
+
? (typeof sampling?.temperature === "number" ? sampling.temperature : this.#temperature)
|
|
617
|
+
: typeof sampling?.temperature === "number" ? sampling.temperature : undefined,
|
|
618
|
+
topP: typeof sampling?.top_p === "number" ? sampling.top_p : undefined,
|
|
619
|
+
topK: typeof sampling?.top_k === "number" ? sampling.top_k : undefined,
|
|
620
|
+
presencePenalty: typeof sampling?.presence_penalty === "number" ? sampling.presence_penalty : undefined,
|
|
621
|
+
frequencyPenalty: typeof sampling?.frequency_penalty === "number"
|
|
622
|
+
? sampling.frequency_penalty
|
|
623
|
+
: this.#tuningFloors && this.#frequencyPenalty > 0 ? this.#frequencyPenalty : undefined,
|
|
624
|
+
stopSequences: typeof sampling?.stop === "string"
|
|
625
|
+
? [sampling.stop]
|
|
626
|
+
: Array.isArray(sampling?.stop) && sampling.stop.every((value) => typeof value === "string")
|
|
627
|
+
? sampling.stop
|
|
628
|
+
: undefined,
|
|
629
|
+
seed: typeof sampling?.seed === "number" ? sampling.seed : undefined,
|
|
630
|
+
maxOutputTokens: maxTokens,
|
|
631
|
+
reasoning: this.#reasoning.mode === "off"
|
|
632
|
+
? "none"
|
|
633
|
+
: this.#reasoning.mode === "adaptive"
|
|
634
|
+
? "provider-default"
|
|
635
|
+
: effortFromBudget(this.#reasoning.budget!),
|
|
651
636
|
});
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
// #543: Cloudflare/CDN edge codes (520-527) classify as network_failure
|
|
658
|
-
// but fail-fast - a retry just re-incurs the same origin/edge timeout.
|
|
659
|
-
const edgeTimeout = err instanceof OpenAiHttpError && isEdgeStatus(err.status);
|
|
660
|
-
// Terminal kind, edge failure, or budget spent -> surface the failure.
|
|
661
|
-
if (!RETRYABLE.has(kind) || edgeTimeout || attempt >= this.#retryAttempts) {
|
|
662
|
-
const pe = toProviderError(err, this.#source);
|
|
663
|
-
// #537 case 2: a 401/403 with a key PRESENT is a rejected key, not a
|
|
664
|
-
// transport failure — surface the distinct, actionable hint (kind stays
|
|
665
|
-
// "unauthorized", so core's routing is unchanged) rather than the raw
|
|
666
|
-
// upstream JSON. Distinct from the unset-key throw at construction.
|
|
667
|
-
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
668
|
-
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
669
|
-
}
|
|
670
|
-
throw pe;
|
|
671
|
-
}
|
|
672
|
-
transportRetries.push({
|
|
673
|
-
attempt: attempt + 1,
|
|
674
|
-
kind,
|
|
675
|
-
elapsedMs: Math.round(performance.now() - attemptStarted),
|
|
676
|
-
message: err instanceof Error ? err.message : String(err),
|
|
677
|
-
});
|
|
678
|
-
const retryAfter = err instanceof OpenAiHttpError ? err.retryAfter : null;
|
|
679
|
-
await sleepWithAbort(retryAfter ?? this.#retryDelayMs * 2 ** attempt, signal);
|
|
637
|
+
} catch (err) {
|
|
638
|
+
if (signal?.aborted) throw err;
|
|
639
|
+
const pe = toProviderError(err, this.#source);
|
|
640
|
+
if ((pe.status === 401 || pe.status === 403) && this.#hasApiKey && this.#apiKeyRejectedMessage !== undefined) {
|
|
641
|
+
throw new ProviderError(this.#source, "unauthorized", this.#apiKeyRejectedMessage, { status: pe.status, cause: err });
|
|
680
642
|
}
|
|
643
|
+
throw pe;
|
|
681
644
|
}
|
|
682
645
|
|
|
683
646
|
// #539: llama-server --special renders EOG tokens as text, so a turn ending
|
|
@@ -696,7 +659,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
696
659
|
// Discard/retry/escalate/self-correct is the consumer's policy.
|
|
697
660
|
let telemetry: TelemetryEvent[] | undefined;
|
|
698
661
|
let railsMeta: Record<string, unknown> | undefined;
|
|
699
|
-
const usage =
|
|
662
|
+
const usage = raw.usage;
|
|
700
663
|
const observedGrammar = sendGrammar ?? (wantGrammar && this.#gbnfDebug ? grammar : undefined);
|
|
701
664
|
if (observedGrammar !== undefined) {
|
|
702
665
|
const verdict = this.#grammarVerdict(observedGrammar, raw.content);
|
|
@@ -714,7 +677,7 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
714
677
|
// invisible, billed (12,288 billed vs 1,033 chars visible, live).
|
|
715
678
|
// countTokens OVERCOUNTS text (chars/2 upper bound), so billed
|
|
716
679
|
// exceeding visible-plus-slack is real vanishing, not estimator noise.
|
|
717
|
-
const visible = this.#countTokens(raw.content) + this.#countTokens(raw.
|
|
680
|
+
const visible = this.#countTokens(raw.content) + this.#countTokens(raw.reasoning);
|
|
718
681
|
if (sendGrammar !== undefined && usage.completion > visible + 64) {
|
|
719
682
|
(telemetry ??= []).push({
|
|
720
683
|
source: this.#source,
|
|
@@ -725,30 +688,23 @@ export default class OpenAICompatProvider implements Provider {
|
|
|
725
688
|
}
|
|
726
689
|
}
|
|
727
690
|
|
|
728
|
-
const builtMeta = this.#buildMeta(raw.
|
|
729
|
-
const
|
|
730
|
-
const
|
|
731
|
-
? { ...builtMeta, ...railsMeta, ...retryMeta }
|
|
732
|
-
: builtMeta;
|
|
733
|
-
|
|
734
|
-
// #36: surface per-token logprobs + their mean when the backend returned
|
|
735
|
-
// them (only possible when the flag requested them). Absent otherwise —
|
|
736
|
-
// never synthesized.
|
|
737
|
-
const logprobs = raw.logprobs !== null && raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
691
|
+
const builtMeta = this.#buildMeta(raw.metadata);
|
|
692
|
+
const meta = railsMeta !== undefined ? { ...builtMeta, ...railsMeta } : builtMeta;
|
|
693
|
+
const logprobs = raw.logprobs.length > 0 ? raw.logprobs : undefined;
|
|
738
694
|
const meanLogprob = logprobs !== undefined
|
|
739
|
-
? logprobs.reduce((sum,
|
|
695
|
+
? logprobs.reduce((sum, token) => sum + token.logprob, 0) / logprobs.length
|
|
740
696
|
: undefined;
|
|
741
697
|
|
|
742
698
|
return {
|
|
743
699
|
assistant: {
|
|
744
700
|
content: raw.content,
|
|
745
|
-
reasoning: raw.
|
|
746
|
-
|
|
747
|
-
|
|
748
|
-
|
|
701
|
+
reasoning: raw.reasoning.length > 0 ? raw.reasoning : null,
|
|
702
|
+
...(raw.reasoningEncrypted.length > 0
|
|
703
|
+
? { reasoningEncrypted: raw.reasoningEncrypted }
|
|
704
|
+
: {}),
|
|
749
705
|
usage,
|
|
750
|
-
finishReason:
|
|
751
|
-
model: raw.model
|
|
706
|
+
finishReason: raw.finishReason,
|
|
707
|
+
model: raw.model,
|
|
752
708
|
...(logprobs !== undefined ? { logprobs, meanLogprob } : {}),
|
|
753
709
|
},
|
|
754
710
|
assistantRaw: raw,
|
package/src/Mock.test.ts
CHANGED
|
@@ -109,7 +109,7 @@ test("Mock: exhausted queue throws a specific error", async () => {
|
|
|
109
109
|
// -- #507: the reserve surface lives on the Provider CONTRACT, and Mock drives core's partition suite --
|
|
110
110
|
|
|
111
111
|
test("#507 the reserve getters are on the Provider interface (not just the concrete class)", () => {
|
|
112
|
-
// Typing against the
|
|
112
|
+
// Typing against the contract catches a getter-only concrete surface.
|
|
113
113
|
const prevR = process.env.PLURNK_PROVIDERS_REASONING_RESERVE;
|
|
114
114
|
try {
|
|
115
115
|
process.env.PLURNK_PROVIDERS_REASONING_RESERVE = "10%";
|
|
@@ -2,7 +2,6 @@ import test, { mock } from "node:test";
|
|
|
2
2
|
import { strict as assert } from "node:assert";
|
|
3
3
|
import { instantiateProvider, loadActiveProvider, resetDiscoveryCache } from "./ProviderRegistry.ts";
|
|
4
4
|
|
|
5
|
-
const fakeProvider = { contextWindow: 1, model: "m", countTokens: () => 0, calculateCost: () => 0, generate: async () => { throw new Error("unused"); } };
|
|
6
5
|
const mapOf = (entries: Record<string, string>, skipped: Record<string, string> = {}) =>
|
|
7
6
|
async () => ({ registry: new Map(Object.entries(entries)), skipped: new Map(Object.entries(skipped)), attributions: new Map<string, string | string[]>() });
|
|
8
7
|
|
|
@@ -10,16 +9,16 @@ const mapOf = (entries: Record<string, string>, skipped: Record<string, string>
|
|
|
10
9
|
// exercise the resolution + two-tier instantiation this module owns; the active
|
|
11
10
|
// alias is driven end-to-end by loadActiveProvider below.
|
|
12
11
|
|
|
13
|
-
// —
|
|
12
|
+
// — provider resolution (SPEC §5) —
|
|
14
13
|
|
|
15
14
|
const fullEnv = Object.freeze({
|
|
16
15
|
PLURNK_PROVIDERS_FETCH_TIMEOUT: "600000",
|
|
17
16
|
PLURNK_PROVIDERS_STREAM_IDLE_TIMEOUT: "0",
|
|
18
|
-
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%",
|
|
17
|
+
PLURNK_PROVIDERS_REASONING: "off", PLURNK_PROVIDERS_TEMPERATURE: "0.2", PLURNK_PROVIDERS_REPEAT_PENALTY: "1.15", PLURNK_PROVIDERS_FREQUENCY_PENALTY: "0.4", PLURNK_PROVIDERS_REASONING_RESERVE: "10%", PLURNK_PROVIDERS_COMPLETION_RESERVE: "25%", PLURNK_PROVIDERS_PROBE_ATTEMPTS: "3", PLURNK_PROVIDERS_PROBE_DELAY: "1", PLURNK_PROVIDERS_RETRY_ATTEMPTS: "0", PLURNK_PROVIDERS_PROMPT_CACHE_KEY: "1",
|
|
19
18
|
OPENAI_BASE_URL: "http://x",
|
|
20
19
|
});
|
|
21
20
|
|
|
22
|
-
test("instantiateProvider:
|
|
21
|
+
test("instantiateProvider: cataloged name resolves in-framework, no scan, no import", async () => {
|
|
23
22
|
resetDiscoveryCache();
|
|
24
23
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
25
24
|
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
@@ -36,24 +35,33 @@ test("instantiateProvider: standard name resolves in-framework, no scan, no impo
|
|
|
36
35
|
mock.restoreAll();
|
|
37
36
|
});
|
|
38
37
|
|
|
39
|
-
test("instantiateProvider:
|
|
38
|
+
test("instantiateProvider: an installed AI SDK provider resolves through discovery", async () => {
|
|
40
39
|
resetDiscoveryCache();
|
|
41
40
|
const calls: unknown[] = [];
|
|
42
|
-
const p = await instantiateProvider("
|
|
43
|
-
async (specifier) => {
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
41
|
+
const p = await instantiateProvider("acme", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192" }, "model-a",
|
|
42
|
+
async (specifier) => {
|
|
43
|
+
calls.push(specifier);
|
|
44
|
+
return { default: { languageModel: (model: string) => { calls.push(model); return {} as never; } } };
|
|
45
|
+
},
|
|
46
|
+
mapOf({ acme: "@acme/ai-provider" }));
|
|
47
|
+
assert.equal(p.model, "model-a");
|
|
48
|
+
assert.equal(p.contextWindow, 8192);
|
|
49
|
+
assert.deepEqual(calls, ["@acme/ai-provider", "model-a"]);
|
|
47
50
|
});
|
|
48
51
|
|
|
49
|
-
test("instantiateProvider: a per-alias baseUrl
|
|
52
|
+
test("instantiateProvider: a per-alias baseUrl drives the built-in Ollama probe", async () => {
|
|
50
53
|
resetDiscoveryCache();
|
|
51
|
-
|
|
54
|
+
const calls: string[] = [];
|
|
55
|
+
mock.method(globalThis, "fetch", async (url: string) => {
|
|
56
|
+
calls.push(String(url));
|
|
57
|
+
return new Response(JSON.stringify({ model_info: { "qwen.context_length": 32768 } }));
|
|
58
|
+
});
|
|
52
59
|
await instantiateProvider("ollama", { ...fullEnv }, "qwen2.5-coder",
|
|
53
|
-
async () => ({
|
|
54
|
-
mapOf({
|
|
60
|
+
async () => ({}),
|
|
61
|
+
mapOf({}),
|
|
55
62
|
"http://nook:11434");
|
|
56
|
-
assert.deepEqual(
|
|
63
|
+
assert.deepEqual(calls, ["http://nook:11434/api/show"]);
|
|
64
|
+
mock.restoreAll();
|
|
57
65
|
});
|
|
58
66
|
|
|
59
67
|
test("instantiateProvider: a per-alias baseUrl drives the standard openai probe to the override host", async () => {
|
|
@@ -74,14 +82,14 @@ test("instantiateProvider: a per-alias baseUrl drives the standard openai probe
|
|
|
74
82
|
test("instantiateProvider: a THIRD-PARTY scope is discovered — name maps to its package", async () => {
|
|
75
83
|
resetDiscoveryCache();
|
|
76
84
|
const imports: string[] = [];
|
|
77
|
-
const p = await instantiateProvider("foo", { ...fullEnv }, "m",
|
|
78
|
-
async (specifier) => { imports.push(specifier); return { default: {
|
|
85
|
+
const p = await instantiateProvider("foo", { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096" }, "m",
|
|
86
|
+
async (specifier) => { imports.push(specifier); return { default: { languageModel: () => ({} as never) } }; },
|
|
79
87
|
mapOf({ foo: "@acme/acme-provider-foo" }));
|
|
80
|
-
assert.equal(p,
|
|
88
|
+
assert.equal(p.model, "m");
|
|
81
89
|
assert.deepEqual(imports, ["@acme/acme-provider-foo"]); // not an @plurnk/ specifier
|
|
82
90
|
});
|
|
83
91
|
|
|
84
|
-
test("instantiateProvider: a
|
|
92
|
+
test("instantiateProvider: a cataloged name is authoritative — a scanned package of the same name is shadowed", async () => {
|
|
85
93
|
resetDiscoveryCache();
|
|
86
94
|
mock.method(globalThis, "fetch", async (url: string) => {
|
|
87
95
|
if (String(url).endsWith("/models")) return new Response(JSON.stringify({ data: [] }), { status: 200 });
|
|
@@ -100,7 +108,7 @@ test("instantiateProvider: unknown provider throws — no standard, no discovere
|
|
|
100
108
|
resetDiscoveryCache();
|
|
101
109
|
await assert.rejects(
|
|
102
110
|
() => instantiateProvider("nope", { ...fullEnv }, "m", async () => ({}), mapOf({})),
|
|
103
|
-
/unknown provider "nope"
|
|
111
|
+
/unknown provider "nope"/,
|
|
104
112
|
);
|
|
105
113
|
});
|
|
106
114
|
|
|
@@ -116,13 +124,13 @@ test("instantiateProvider: an untrusted (skipped) provider gives a precise error
|
|
|
116
124
|
assert.deepEqual(imports, []); // never imported an untrusted package
|
|
117
125
|
});
|
|
118
126
|
|
|
119
|
-
test("instantiateProvider: discovered package
|
|
127
|
+
test("instantiateProvider: discovered package must export an AI SDK provider", async () => {
|
|
120
128
|
resetDiscoveryCache();
|
|
121
129
|
await assert.rejects(
|
|
122
130
|
() => instantiateProvider("broken", { ...fullEnv }, "m",
|
|
123
131
|
async () => ({ default: {} }),
|
|
124
132
|
mapOf({ broken: "@acme/acme-provider-broken" })),
|
|
125
|
-
/@acme\/acme-provider-broken default export is not
|
|
133
|
+
/@acme\/acme-provider-broken default export is not an AI SDK provider/,
|
|
126
134
|
);
|
|
127
135
|
});
|
|
128
136
|
|
|
@@ -174,6 +182,8 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
174
182
|
FIREWORKS_BASE_URL: "https://api.fireworks.ai/inference/v1",
|
|
175
183
|
FIREWORKS_API_KEY: "fw",
|
|
176
184
|
PLURNK_PROVIDERS_CONTEXT_WINDOW: "8192",
|
|
185
|
+
PLURNK_PROVIDERS_PROVIDER_FIREWORKS_REASONING_STYLE: "effort_explicit",
|
|
186
|
+
PLURNK_PROVIDERS_TOP_LOGPROBS: "2",
|
|
177
187
|
PLURNK_PROVIDERS_SERVICE_TIER_fast: "priority",
|
|
178
188
|
PLURNK_PROVIDERS_SERVICE_TIER_standard: "default",
|
|
179
189
|
};
|
|
@@ -184,6 +194,9 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
184
194
|
await fast.generate({ workerId: "fast-worker", messages: [] });
|
|
185
195
|
await standard.generate({ workerId: "standard-worker", messages: [] });
|
|
186
196
|
assert.deepEqual(bodies.map((body) => body.service_tier), ["priority", "default"]);
|
|
197
|
+
assert.deepEqual(bodies.map((body) => body.prompt_cache_key), ["fast-worker", "standard-worker"]);
|
|
198
|
+
assert.deepEqual(bodies.map((body) => body.reasoning_effort), ["none", "none"]);
|
|
199
|
+
assert.deepEqual(bodies.map((body) => body.top_logprobs), [2, 2]);
|
|
187
200
|
assert.deepEqual(bodies.map((body) => body.model), [
|
|
188
201
|
"accounts/fireworks/routers/glm-5p2-fast",
|
|
189
202
|
"accounts/fireworks/models/deepseek-v4-pro",
|
|
@@ -191,13 +204,13 @@ test("#622: two Fireworks aliases independently select default and priority serv
|
|
|
191
204
|
mock.restoreAll();
|
|
192
205
|
});
|
|
193
206
|
|
|
194
|
-
test("loadActiveProvider: resolves the alias cascade
|
|
207
|
+
test("loadActiveProvider: resolves the alias cascade to an installed AI SDK provider", async () => {
|
|
195
208
|
resetDiscoveryCache();
|
|
196
|
-
const env = { ...fullEnv, PLURNK_MODEL: "
|
|
209
|
+
const env = { ...fullEnv, PLURNK_PROVIDERS_CONTEXT_WINDOW: "4096", PLURNK_MODEL: "custom", PLURNK_MODEL_custom: "acme/model-a" } as NodeJS.ProcessEnv;
|
|
197
210
|
const p = await loadActiveProvider(env,
|
|
198
|
-
async () => ({ default: {
|
|
199
|
-
mapOf({
|
|
200
|
-
assert.equal(p,
|
|
211
|
+
async () => ({ default: { languageModel: () => ({} as never) } }),
|
|
212
|
+
mapOf({ acme: "@acme/ai-provider" }));
|
|
213
|
+
assert.equal(p.model, "model-a");
|
|
201
214
|
});
|
|
202
215
|
|
|
203
216
|
test("loadActiveProvider: throws a named error when no alias is active", async () => {
|
package/src/ProviderRegistry.ts
CHANGED
|
@@ -3,18 +3,19 @@
|
|
|
3
3
|
// overrides) lives in @plurnk/plurnk-aliases — the zero-dep parser shared with
|
|
4
4
|
// thin clients (#27); this module resolves the active alias to a Provider.
|
|
5
5
|
//
|
|
6
|
-
//
|
|
7
|
-
//
|
|
8
|
-
//
|
|
9
|
-
//
|
|
10
|
-
// The framework is contract-only — it does NOT depend on its plugins; the
|
|
11
|
-
// scan is what surfaces them (#12/#14).
|
|
6
|
+
// Resolution order (SPEC §5): models.dev catalog → PLURNK provider declaration
|
|
7
|
+
// → local protocol adapter → scope-agnostic AI SDK plugin discovery. Generic
|
|
8
|
+
// provider facts belong to models.dev or operator config; PLURNK owns only the
|
|
9
|
+
// stable Provider contract and product-specific local behavior.
|
|
12
10
|
|
|
13
|
-
import type {
|
|
14
|
-
import {
|
|
11
|
+
import type { AiSdkProviderPlugin, Provider } from "./types.ts";
|
|
12
|
+
import { catalogProviderFromEnv, providerFromSdkModel } from "./catalogProvider.ts";
|
|
15
13
|
import { discover, type DiscoverOptions, type Discovery } from "./discover.ts";
|
|
16
14
|
import { resolveActiveAlias } from "@plurnk/plurnk-aliases";
|
|
17
15
|
import { scopeEnvToAlias } from "./env.ts";
|
|
16
|
+
import { ollamaProviderFromEnv } from "./ollama.ts";
|
|
17
|
+
import { compatibleProviderFromEnv } from "./compatibleProvider.ts";
|
|
18
|
+
import { contextWindowFromEnv } from "./env.ts";
|
|
18
19
|
|
|
19
20
|
// Two injectable seams, both defaulting to production behavior and never passed
|
|
20
21
|
// by real callers: the module importer (tests exercise the bespoke path without
|
|
@@ -33,10 +34,8 @@ const providerPackages = async (discoverFn: DiscoverFn, env: NodeJS.ProcessEnv):
|
|
|
33
34
|
return discoveredCache;
|
|
34
35
|
};
|
|
35
36
|
|
|
36
|
-
//
|
|
37
|
-
//
|
|
38
|
-
// authoritative — a scanned package whose name duplicates a standard one is
|
|
39
|
-
// shadowed here (never reached), since tier 1 returns first.
|
|
37
|
+
// Catalog and explicit declarations are authoritative. Discovery is the
|
|
38
|
+
// extensibility seam for an AI SDK provider that neither source describes.
|
|
40
39
|
export const instantiateProvider = async (
|
|
41
40
|
name: string,
|
|
42
41
|
env: NodeJS.ProcessEnv,
|
|
@@ -46,14 +45,13 @@ export const instantiateProvider = async (
|
|
|
46
45
|
baseUrl?: string, // per-alias endpoint override (PLURNK_BASEURL_<alias>); threaded to both tiers
|
|
47
46
|
alias?: string, // the alias this instantiation serves — scopes PLURNK_PROVIDERS_<KNOB>_<alias> overrides
|
|
48
47
|
): Promise<Provider> => {
|
|
49
|
-
// Per-alias knob scoping
|
|
50
|
-
//
|
|
48
|
+
// Per-alias knob scoping overlays _<alias>-suffixed knobs onto their bare
|
|
49
|
+
// names before any resolver reads them.
|
|
51
50
|
if (alias !== undefined) env = scopeEnvToAlias(env, alias);
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
}
|
|
51
|
+
const catalog = catalogProviderFromEnv(name, env, model, baseUrl);
|
|
52
|
+
if (catalog !== null) return catalog;
|
|
53
|
+
if (name === "ollama") return ollamaProviderFromEnv(env, model, baseUrl === undefined ? undefined : { baseUrl });
|
|
54
|
+
if (name === "openai" || name === "plurnk") return compatibleProviderFromEnv(name, env, model, baseUrl);
|
|
57
55
|
const { registry, skipped } = await providerPackages(discoverFn, env);
|
|
58
56
|
const specifier = registry.get(name);
|
|
59
57
|
if (specifier === undefined) {
|
|
@@ -61,7 +59,7 @@ export const instantiateProvider = async (
|
|
|
61
59
|
if (declined !== undefined) {
|
|
62
60
|
throw new Error(`provider "${name}" resolves to ${declined}, but it is untrusted under PLURNK_PLUGINS_TRUSTED_ONLY — add it to the allowlist (or publish under @plurnk/)`);
|
|
63
61
|
}
|
|
64
|
-
throw new Error(`unknown provider "${name}":
|
|
62
|
+
throw new Error(`unknown provider "${name}": absent from models.dev, operator declarations, local adapters, and installed AI SDK provider plugins`);
|
|
65
63
|
}
|
|
66
64
|
let mod: unknown;
|
|
67
65
|
try {
|
|
@@ -69,11 +67,24 @@ export const instantiateProvider = async (
|
|
|
69
67
|
} catch (cause) {
|
|
70
68
|
throw new Error(`provider "${name}" resolves to ${specifier}, but importing it failed`, { cause });
|
|
71
69
|
}
|
|
72
|
-
const
|
|
73
|
-
if (
|
|
74
|
-
throw new Error(`${specifier} default export is not
|
|
70
|
+
const sdkProvider = (mod as { default?: AiSdkProviderPlugin }).default;
|
|
71
|
+
if (sdkProvider === undefined || typeof sdkProvider.languageModel !== "function") {
|
|
72
|
+
throw new Error(`${specifier} default export is not an AI SDK provider (missing languageModel)`);
|
|
73
|
+
}
|
|
74
|
+
if (baseUrl !== undefined) {
|
|
75
|
+
throw new Error(`${specifier}: PLURNK_BASEURL_${alias ?? "<alias>"} cannot reconfigure an installed AI SDK provider; declare the provider through PLURNK_PROVIDERS_PROVIDER_* instead`);
|
|
76
|
+
}
|
|
77
|
+
const contextWindow = contextWindowFromEnv(env, name);
|
|
78
|
+
if (contextWindow === null) {
|
|
79
|
+
throw new Error(`${specifier}: PLURNK_PROVIDERS_CONTEXT_WINDOW must be set because Models.dev has no metadata for provider "${name}"`);
|
|
75
80
|
}
|
|
76
|
-
return
|
|
81
|
+
return providerFromSdkModel({
|
|
82
|
+
name,
|
|
83
|
+
env,
|
|
84
|
+
model,
|
|
85
|
+
languageModel: sdkProvider.languageModel(model),
|
|
86
|
+
contextWindow,
|
|
87
|
+
});
|
|
77
88
|
};
|
|
78
89
|
|
|
79
90
|
// Test-only: drop the memoized discovery so a fresh scan/injection runs next.
|