@plurnk/plurnk-providers 1.2.0 → 1.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,618 @@
1
+ // Pure-config OpenAI-compatible providers. A provider qualifies as "standard"
2
+ // when it has no unique runtime surface — no catalog probe, no pricing fetch,
3
+ // no bespoke wire shape — so it reduces to: an env var for the key, a base
4
+ // URL, a reasoning-translation style, and a tokenizer. Such providers need NO
5
+ // sibling package; the framework instantiates them directly.
6
+ //
7
+ // Two-tier resolution (SPEC §5): the consumer tries standardProviderFromEnv
8
+ // first, then falls back to the discover() node_modules scan for the bespoke
9
+ // ones (openrouter, ollama, google, xai, cloudflare, third-party). The scan
10
+ // resolves the package specifier — it is NOT a hardcoded @plurnk/ pattern.
11
+
12
+ import type { Provider, ProviderUsage } from "./types.ts";
13
+ import OpenAICompatProvider, { type ReasoningStyle, type GrammarStyle } from "./OpenAICompat.ts";
14
+ import { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, reasoningFromEnv, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve, type ReserveSpec } from "./env.ts";
15
+ import { emitWarningOnce } from "./warnings.ts";
16
+ import { providerSource } from "./telemetry.ts";
17
+ import { computeCost } from "./usage.ts";
18
+ import { lookup } from "@plurnk/plurnk-models";
19
+
20
+ type StandardProviderSpec = {
21
+ // Bearer-auth env var(s), and whether the key is mandatory (local
22
+ // OpenAI-compat servers run without auth, so the generic "openai" entry
23
+ // leaves it optional). A LIST accepts the conventional aliases the wild uses
24
+ // for one credential (e.g. deepinfra's API_KEY / API_TOKEN / TOKEN) — first
25
+ // non-empty wins, the required-but-unset error names them all. Omit when
26
+ // supplying a custom `headersFromEnv` builder.
27
+ apiKeyVar?: string | readonly string[];
28
+ apiKeyRequired?: boolean;
29
+ // Custom missing-key throw message (replaces the generic "<var> must be set")
30
+ // — friendly, actionable guidance shown ONLY on the unset path (#537). A
31
+ // present-but-rejected key is a wire 401 downstream, never this.
32
+ apiKeyMessage?: string;
33
+ // Custom message when a PRESENT key is rejected by the backend (a live 401/403
34
+ // with a bearer sent) — distinct from the unset-key apiKeyMessage (#537 case 2).
35
+ apiKeyRejectedMessage?: string;
36
+ // Custom request-header builder for auth the single-var bearer can't express
37
+ // (multiple optional credentials, vendor routing headers). Returns the
38
+ // headers built from env; an empty object means no auth headers are sent.
39
+ // When set, it REPLACES the apiKeyVar bearer logic.
40
+ headersFromEnv?: (env: NodeJS.ProcessEnv) => Record<string, string>;
41
+ // Base URL env var(s): no in-code default; the canonical endpoint ships as a
42
+ // floored default in .env.defaults (overridable in the operator's env or
43
+ // per-alias), never a baked constant. A list accepts conventional aliases
44
+ // (openai's BASE_URL / API_BASE). Either this or baseUrlFromEnv must resolve.
45
+ baseUrlVar?: string | readonly string[];
46
+ // Derive the base from env when no override var is set — for endpoints whose
47
+ // URL is templated from standard env (bedrock builds it from AWS_REGION).
48
+ // Throws a named error if it can't derive.
49
+ baseUrlFromEnv?: (env: NodeJS.ProcessEnv) => string | undefined;
50
+ // Custom catalog context-window resolver for a relay whose model id the
51
+ // models.dev catalog doesn't key directly (bedrock inference profiles, #22).
52
+ // Returns the model's context window or undefined. When set, the catalog COST
53
+ // path is skipped — the model's native rate isn't this relay's rate.
54
+ catalogContextLookup?: (model: string) => number | undefined;
55
+ // Path appended to the (slash-trimmed) base to reach chat-completions.
56
+ chatPath: string;
57
+ // When true (generic "openai" only), strip a trailing /v1 from the
58
+ // operator-supplied base before appending chatPath — the base may or may
59
+ // not already include it.
60
+ flexBaseStrip?: boolean;
61
+ reasoningStyle: ReasoningStyle;
62
+ // How this backend carries a GBNF grammar (default "none" — not sent). A
63
+ // probeNctx entry is upgraded to "llamacpp" when the probe sees a
64
+ // llama-server; cloud backends that support GBNF set their shape statically
65
+ // (fireworks → "response_format", verified live).
66
+ grammarStyle?: GrammarStyle;
67
+ // SSE streaming (default true). The streaming transport is dropped
68
+ // per-request only when it would break a feature (a response_format grammar
69
+ // arrives mislabeled as reasoning_content under fireworks' stream); leave
70
+ // unset to keep streaming on for every other call. See OpenAICompat.generate.
71
+ streaming?: boolean;
72
+ // Constant model-id prefix the backend requires but the alias shouldn't
73
+ // repeat (fireworks → "accounts/fireworks/models/", so the alias is just
74
+ // `fireworks/deepseek-v4-pro`). Prepended idempotently to form the wire id,
75
+ // which is ALSO the catalog key (models.dev keys fireworks-ai on the full id).
76
+ modelPrefix?: string;
77
+ // First-party telemetry forwarding. ONLY the plurnk hosted endpoint sets
78
+ // this — it forwards the consumer's per-turn `attributions` (contributor
79
+ // credit) and `client` (originating frontend) as `Plurnk-Attribution` /
80
+ // `Plurnk-Client` headers. Absent everywhere else, so those signals are
81
+ // structurally incapable of reaching a third-party backend (never sold,
82
+ // never leaked — the destination is the consent boundary).
83
+ firstPartyMetadata?: boolean;
84
+ // #507: the plurnk.ai router owns tuning (SPEC §5) — suppress the client-side
85
+ // temperature/penalty floors on this provider; caller sampling still passes.
86
+ suppressTuningFloors?: boolean;
87
+ // When true, omit frequency_penalty from the request body — the backend's
88
+ // native param set excludes it and will reject it (DOC-verified). Sets
89
+ // frequencyPenalty to 0 so #repetitionPenaltyBody's "> 0" gate suppresses
90
+ // the field. Mirrors the plugin omissions (dba4300 / providers-xai#2).
91
+ suppressFrequencyPenalty?: boolean;
92
+ // #518: send prompt_cache_key=workerId (serverless replica-cache affinity).
93
+ // Default ON for standard providers (OpenAI-standard field, broadly accepted);
94
+ // set false to opt a backend out (e.g. anthropic's cache_control mechanism).
95
+ promptCacheKey?: boolean;
96
+ // Top-level response field the endpoint reports account balance (pico-USD) in,
97
+ // surfaced as ProviderResponse.balancePico (plurnk only, #23). Absent elsewhere.
98
+ balanceMetaKey?: string;
99
+ // RETIRED knob (#27→mimetypes#44): exact client-side tokenizer families were
100
+ // removed with the tokenizer shed — the var is kept ONLY to fail hard with a
101
+ // migration pointer when an operator still sets it.
102
+ tokenizerEnvVar: string;
103
+ // When true, probe GET /v1/models at construction. Two reads off one call:
104
+ // the endpoint-reported context window (`n_ctx`, used when
105
+ // PLURNK_PROVIDERS_CONTEXT_WINDOW is unset) and the llama-server fingerprint
106
+ // (a `meta` block on the model row) that enables grammar-constrained
107
+ // sampling (SPEC §13). Set for providers that may front a local
108
+ // OpenAI-compat server; cloud endpoints report neither → null / false.
109
+ probeNctx?: boolean;
110
+ // Whether a probeNctx spec may infer LOCAL llama-server capabilities (grammar
111
+ // transport → "llamacpp", slot pinning, template reasoning) from the probe's
112
+ // `meta` fingerprint. Default true. Set FALSE for an endpoint that reports a
113
+ // window but must be treated as a plain remote OpenAI server — `plurnk` reads
114
+ // its (server-controlled) window from upstream yet must NEVER be talked into
115
+ // grammar/slot behavior, so its capabilities can't be flipped by what the
116
+ // endpoint happens to return.
117
+ detectLlamaServer?: boolean;
118
+ };
119
+
120
+ // Frozen so a downstream can't mutate the shared table.
121
+ export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>> = Object.freeze({
122
+ // Generic OpenAI-compatible endpoint (OpenAI proper, llama-server, vLLM,
123
+ // LM Studio, or any chat-completions shim). Operator supplies the base.
124
+ // Replaces the former @plurnk/plurnk-providers-openai sibling verbatim.
125
+ openai: {
126
+ apiKeyVar: "OPENAI_API_KEY", apiKeyRequired: false,
127
+ baseUrlVar: ["OPENAI_BASE_URL", "OPENAI_API_BASE"], chatPath: "/v1/chat/completions", flexBaseStrip: true,
128
+ reasoningStyle: "think", tokenizerEnvVar: "OPENAI_TOKENIZER",
129
+ probeNctx: true,
130
+ },
131
+ groq: {
132
+ apiKeyVar: "GROQ_API_KEY", apiKeyRequired: true,
133
+ baseUrlVar: "GROQ_BASE_URL", chatPath: "/chat/completions",
134
+ reasoningStyle: "effort", tokenizerEnvVar: "GROQ_TOKENIZER",
135
+ },
136
+ deepseek: {
137
+ apiKeyVar: "DEEPSEEK_API_KEY", apiKeyRequired: true,
138
+ baseUrlVar: "DEEPSEEK_BASE_URL", chatPath: "/chat/completions",
139
+ reasoningStyle: "none", tokenizerEnvVar: "DEEPSEEK_TOKENIZER",
140
+ },
141
+ mistral: {
142
+ apiKeyVar: "MISTRAL_API_KEY", apiKeyRequired: true,
143
+ baseUrlVar: "MISTRAL_BASE_URL", chatPath: "/chat/completions",
144
+ reasoningStyle: "none", tokenizerEnvVar: "MISTRAL_TOKENIZER",
145
+ },
146
+ together: {
147
+ apiKeyVar: "TOGETHER_API_KEY", apiKeyRequired: true,
148
+ baseUrlVar: "TOGETHER_BASE_URL", chatPath: "/chat/completions",
149
+ reasoningStyle: "none", tokenizerEnvVar: "TOGETHER_TOKENIZER",
150
+ },
151
+ // reasoningStyle "effort_explicit", NOT "none": fireworks serves reason-by-
152
+ // DEFAULT models (DeepSeek V4 defaults 'high'), so budget 0 must SEND
153
+ // reasoning_effort:"none" — omitting the field leaves the reasoner live inside
154
+ // a constrained decode until max_tokens (#30; 0/5 → 30/30 measured).
155
+ fireworks: {
156
+ apiKeyVar: "FIREWORKS_API_KEY", apiKeyRequired: true,
157
+ baseUrlVar: "FIREWORKS_BASE_URL", chatPath: "/chat/completions",
158
+ reasoningStyle: "effort_explicit", grammarStyle: "response_format", modelPrefix: "accounts/fireworks/models/", tokenizerEnvVar: "FIREWORKS_TOKENIZER",
159
+ },
160
+ deepinfra: {
161
+ apiKeyVar: ["DEEPINFRA_API_KEY", "DEEPINFRA_API_TOKEN", "DEEPINFRA_TOKEN"], apiKeyRequired: true,
162
+ baseUrlVar: "DEEPINFRA_BASE_URL", chatPath: "/chat/completions",
163
+ reasoningStyle: "none", tokenizerEnvVar: "DEEPINFRA_TOKENIZER",
164
+ },
165
+ // — Chinese cloud hosts (all OpenAI-compat, plain bearer, doc-verified 2026). —
166
+ // The .env.defaults base is the INTERNATIONAL endpoint; mainland operators
167
+ // point the override var at the `.cn` twin noted per entry. Reasoning is left
168
+ // "none" for the whole cohort: each host's reasoning toggle is either
169
+ // model-selected or a non-standard param that rides in `extra_body` (the
170
+ // `effort`/`anthropic` styles don't reach it) — same posture as deepseek.
171
+ // None are in the @plurnk/plurnk-models snapshot, so context comes from
172
+ // PLURNK_PROVIDERS_CONTEXT_WINDOW and cost stays unknown until the catalog adds
173
+ // them (a plurnk-models issue, not this repo's).
174
+ moonshot: {
175
+ apiKeyVar: "MOONSHOT_API_KEY", apiKeyRequired: true,
176
+ baseUrlVar: "MOONSHOT_BASE_URL", chatPath: "/chat/completions",
177
+ reasoningStyle: "none", tokenizerEnvVar: "MOONSHOT_TOKENIZER",
178
+ },
179
+ // Alibaba Qwen via DashScope "compatible-mode". Mainland: dashscope.aliyuncs.com.
180
+ dashscope: {
181
+ apiKeyVar: "DASHSCOPE_API_KEY", apiKeyRequired: true,
182
+ baseUrlVar: "DASHSCOPE_BASE_URL", chatPath: "/chat/completions",
183
+ reasoningStyle: "none", tokenizerEnvVar: "DASHSCOPE_TOKENIZER",
184
+ },
185
+ // Zhipu GLM. Base carries /api/paas/v4 (non-/v1). Mainland: open.bigmodel.cn/api/paas/v4.
186
+ zhipu: {
187
+ apiKeyVar: ["ZHIPUAI_API_KEY", "ZAI_API_KEY"], apiKeyRequired: true,
188
+ baseUrlVar: "ZHIPU_BASE_URL", chatPath: "/chat/completions",
189
+ reasoningStyle: "none", tokenizerEnvVar: "ZHIPU_TOKENIZER",
190
+ },
191
+ // ByteDance Doubao via BytePlus ModelArk (base carries /api/v3; `model` is an
192
+ // inference-endpoint/model id). Mainland (Volcengine): ark.cn-beijing.volces.com/api/v3.
193
+ volcengine: {
194
+ apiKeyVar: "ARK_API_KEY", apiKeyRequired: true,
195
+ baseUrlVar: "ARK_BASE_URL", chatPath: "/chat/completions",
196
+ reasoningStyle: "none", tokenizerEnvVar: "ARK_TOKENIZER",
197
+ },
198
+ // Tencent Hunyuan — single global host (no intl/mainland split).
199
+ hunyuan: {
200
+ apiKeyVar: "HUNYUAN_API_KEY", apiKeyRequired: true,
201
+ baseUrlVar: "HUNYUAN_BASE_URL", chatPath: "/chat/completions",
202
+ reasoningStyle: "none", tokenizerEnvVar: "HUNYUAN_TOKENIZER",
203
+ },
204
+ // MiniMax. Mainland twin is api.minimaxi.com (note the extra "i").
205
+ minimax: {
206
+ apiKeyVar: "MINIMAX_API_KEY", apiKeyRequired: true,
207
+ baseUrlVar: "MINIMAX_BASE_URL", chatPath: "/chat/completions",
208
+ reasoningStyle: "none", tokenizerEnvVar: "MINIMAX_TOKENIZER",
209
+ },
210
+ // StepFun. Intl twin is api.stepfun.ai (.ai vs the .com mainland host).
211
+ stepfun: {
212
+ apiKeyVar: "STEP_API_KEY", apiKeyRequired: true,
213
+ baseUrlVar: "STEPFUN_BASE_URL", chatPath: "/chat/completions",
214
+ reasoningStyle: "none", tokenizerEnvVar: "STEPFUN_TOKENIZER",
215
+ },
216
+ // Baichuan — single host.
217
+ baichuan: {
218
+ apiKeyVar: "BAICHUAN_API_KEY", apiKeyRequired: true,
219
+ baseUrlVar: "BAICHUAN_BASE_URL", chatPath: "/chat/completions",
220
+ reasoningStyle: "none", tokenizerEnvVar: "BAICHUAN_TOKENIZER",
221
+ },
222
+ // Baidu ERNIE via Qianfan v2 (key is a bce-v3/ALTAK-… bearer; base carries /v2).
223
+ qianfan: {
224
+ apiKeyVar: "QIANFAN_API_KEY", apiKeyRequired: true,
225
+ baseUrlVar: "QIANFAN_BASE_URL", chatPath: "/chat/completions",
226
+ reasoningStyle: "none", tokenizerEnvVar: "QIANFAN_TOKENIZER",
227
+ },
228
+ // SiliconFlow aggregator. Mainland twin is api.siliconflow.cn.
229
+ siliconflow: {
230
+ apiKeyVar: "SILICONFLOW_API_KEY", apiKeyRequired: true,
231
+ baseUrlVar: "SILICONFLOW_BASE_URL", chatPath: "/chat/completions",
232
+ reasoningStyle: "none", tokenizerEnvVar: "SILICONFLOW_TOKENIZER",
233
+ },
234
+ // ModelScope API-Inference aggregator — single host (.cn).
235
+ modelscope: {
236
+ apiKeyVar: ["MODELSCOPE_API_KEY", "MODELSCOPE_TOKEN"], apiKeyRequired: true,
237
+ baseUrlVar: "MODELSCOPE_BASE_URL", chatPath: "/chat/completions",
238
+ reasoningStyle: "none", tokenizerEnvVar: "MODELSCOPE_TOKENIZER",
239
+ },
240
+ // First-party Claude via Anthropic's OpenAI-compat endpoint: bearer auth,
241
+ // OpenAI SSE, the `thinking` reasoning param (reasoning_effort is ignored).
242
+ // No probe — context/cost come from the @plurnk/plurnk-models catalog.
243
+ anthropic: {
244
+ apiKeyVar: "ANTHROPIC_API_KEY", apiKeyRequired: true,
245
+ baseUrlVar: "ANTHROPIC_BASE_URL", chatPath: "/chat/completions",
246
+ reasoningStyle: "anthropic", tokenizerEnvVar: "ANTHROPIC_TOKENIZER",
247
+ promptCacheKey: false, // #518: anthropic caches via cache_control breakpoints, not prompt_cache_key (unverified) — opt out
248
+ suppressFrequencyPenalty: true, // Claude's native API has no frequency_penalty; the compat endpoint rejects it (DOC #568)
249
+ },
250
+ // AWS Bedrock via its OpenAI-compat endpoint (path is /openai/v1, NOT /v1),
251
+ // bearer-authed with a Bedrock API key (SigV4 optional). Region-templated base
252
+ // (see baseUrlFromEnv); model ids are inference profiles like
253
+ // `us.anthropic.claude-sonnet-4-6` (see catalogContextLookup). Cost stays unknown —
254
+ // bedrock marks up over the native rate, so set PLURNK_PROVIDERS_CONTEXT_WINDOW for
255
+ // a publisher the catalog lacks (#22). frequency_penalty suppressed: Bedrock's
256
+ // OpenAI-compat surfaces Claude's native param set, which excludes it (DOC #568).
257
+ bedrock: {
258
+ apiKeyVar: "AWS_BEARER_TOKEN_BEDROCK", apiKeyRequired: true,
259
+ baseUrlVar: "BEDROCK_BASE_URL", chatPath: "/chat/completions",
260
+ baseUrlFromEnv: (env) => {
261
+ const region = firstSet(env, ["AWS_REGION", "AWS_DEFAULT_REGION"]);
262
+ if (region === undefined) throw new Error("bedrock provider: BEDROCK_BASE_URL must be set, or AWS_REGION / AWS_DEFAULT_REGION to derive it");
263
+ return `https://bedrock-runtime.${region}.amazonaws.com/openai/v1`;
264
+ },
265
+ // Inference profile <region>.<publisher>.<model> — the catalog has no
266
+ // `bedrock` provider, so strip the region scope and look the model up under
267
+ // its PUBLISHER (anthropic, …). Only the context window rides; bedrock marks
268
+ // up, so the native cost is NOT used (cost stays unknown, #22).
269
+ catalogContextLookup: (model) => {
270
+ const stripped = model.replace(/^(?:us-gov|us|eu|apac)\./, "");
271
+ const dot = stripped.indexOf(".");
272
+ if (dot < 0) return undefined;
273
+ return lookup(stripped.slice(0, dot), stripped.slice(dot + 1))?.contextWindow;
274
+ },
275
+ reasoningStyle: "none", tokenizerEnvVar: "BEDROCK_TOKENIZER",
276
+ suppressFrequencyPenalty: true, // Bedrock surfaces Claude's native param set, which has no frequency_penalty (DOC #568)
277
+ },
278
+ // The plurnk hosted model — deliberately the most boring OpenAI-compatible
279
+ // client we can ship: the ecosystem must not know what sits behind
280
+ // plurnk.ai (model/window/grammar/tuning are the router's business).
281
+ // probeNctx reads the window from upstream (a 32k→48k change is a server
282
+ // decision, not a client release), but detectLlamaServer:false keeps it a
283
+ // plain remote server that can NOT be flipped into grammar/slot behavior.
284
+ // PLURNK_API_KEY is REQUIRED — plurnk.ai rejects keyless requests (live 401
285
+ // error_invalid_key, #537). The unset path throws the friendly apiKeyMessage
286
+ // pre-call instead of letting a raw upstream 401 terminate the loop.
287
+ plurnk: {
288
+ baseUrlVar: "PLURNK_BASE_URL", chatPath: "/chat/completions",
289
+ apiKeyVar: "PLURNK_API_KEY", apiKeyRequired: true,
290
+ apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
291
+ apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
292
+ reasoningStyle: "none", grammarStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
293
+ probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, balanceMetaKey: "balance_pico", suppressTuningFloors: true,
294
+ },
295
+ });
296
+
297
+ export const isStandardProvider = (name: string): boolean => name in STANDARD_PROVIDERS;
298
+
299
+ // Normalize a single-or-list env-var spec to a list, and return the first env
300
+ // var that is set non-empty (the accepted-alias resolution — first wins).
301
+ const asList = (v: string | readonly string[] | undefined): readonly string[] =>
302
+ v === undefined ? [] : typeof v === "string" ? [v] : v;
303
+ const firstSet = (env: NodeJS.ProcessEnv, names: readonly string[]): string | undefined => {
304
+ for (const name of names) {
305
+ const value = env[name];
306
+ if (value !== undefined && value.length > 0) return value;
307
+ }
308
+ return undefined;
309
+ };
310
+
311
+ const resolveUrl = (spec: StandardProviderSpec, env: NodeJS.ProcessEnv, label: string, override?: string): string => {
312
+ // per-alias override (PLURNK_BASEURL_<alias>) → base-URL var(s) → env-derived
313
+ // template. No in-code default — the endpoint is required operator config.
314
+ // The override goes through the same normalization (flexBaseStrip + chatPath),
315
+ // so an operator pastes the box URL verbatim.
316
+ const base = override ?? firstSet(env, asList(spec.baseUrlVar)) ?? spec.baseUrlFromEnv?.(env);
317
+ if (base === undefined || base.length === 0) {
318
+ const names = asList(spec.baseUrlVar);
319
+ throw new Error(`${label} provider: ${names.length > 0 ? names.join(" or ") : "base URL"} must be set`);
320
+ }
321
+ const trimmed = spec.flexBaseStrip === true ? base.replace(/\/v1\/?$/, "") : base.replace(/\/$/, "");
322
+ return `${trimmed}${spec.chatPath}`;
323
+ };
324
+
325
+ // Auth/routing headers. A custom builder (multi-credential auth) wins; otherwise
326
+ // the bearer from the first accepted alias that is set: required → fail-hard
327
+ // naming every accepted var, optional → omitted when none set (a keyless server
328
+ // then receives no Authorization header).
329
+ const resolveHeaders = (spec: StandardProviderSpec, env: NodeJS.ProcessEnv, label: string): Record<string, string> => {
330
+ if (spec.headersFromEnv !== undefined) return spec.headersFromEnv(env);
331
+ const names = asList(spec.apiKeyVar);
332
+ if (names.length === 0) return {};
333
+ const apiKey = firstSet(env, names);
334
+ if (apiKey === undefined) {
335
+ if (spec.apiKeyRequired === true) throw new Error(spec.apiKeyMessage ?? `${label} provider: ${names.join(" or ")} must be set`);
336
+ return {};
337
+ }
338
+ return { Authorization: `Bearer ${apiKey}` };
339
+ };
340
+
341
+ // GET /v1/models probe. Yields the reported context window (llama-server nests
342
+ // it under `meta`, vLLM reports it top-level, cloud endpoints omit it) and the
343
+ // llama-server fingerprint — only llama-server rows carry a `meta` block, and
344
+ // llama-server is the backend whose chat-completions accepts a `grammar` field.
345
+ // Best-effort: any failure (unreachable, no field, non-2xx) degrades to
346
+ // { null, false } — a legitimate "unknown", not a swallowed contract violation.
347
+ type EndpointProbe = { nCtx: number | null; llamaServer: boolean; servedModel: string | null; failed: boolean };
348
+
349
+ // One probe failure must never decide capability (#34). Attempts/delay are
350
+ // operator knobs (PLURNK_PROVIDERS_PROBE_ATTEMPTS / _PROBE_DELAY, canonical
351
+ // 3 / 250ms in .env.defaults) — the full-sweep rule: no magic numbers in code.
352
+
353
+ const probeModels = async (chatUrl: string, headers: Record<string, string>, model: string, fetchTimeoutMs: number): Promise<EndpointProbe> => {
354
+ const modelsUrl = chatUrl.replace(/\/chat\/completions$/, "/models");
355
+ try {
356
+ const res = await fetch(modelsUrl, { headers, signal: AbortSignal.timeout(fetchTimeoutMs) });
357
+ if (!res.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
358
+ const data = (await res.json()) as { data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }> };
359
+ const rows = data.data ?? [];
360
+ const row = rows.find((r) => r.id === model) ?? rows[0];
361
+ const n = row?.meta?.n_ctx ?? row?.n_ctx;
362
+ // A clean 200 without the meta block is a CONFIRMED non-llama-server —
363
+ // a valid answer, not a failure; no retry.
364
+ return {
365
+ nCtx: typeof n === "number" && n > 0 ? n : null,
366
+ llamaServer: row?.meta !== undefined,
367
+ // #37: the backend's self-reported id — for a local llama-server the
368
+ // wire model is the ALIAS, but this row carries the real served name
369
+ // (the .gguf) the tokenizer seam maps. Absent when the probe read no row.
370
+ servedModel: typeof row?.id === "string" && row.id.length > 0 ? row.id : null,
371
+ failed: false,
372
+ };
373
+ } catch {
374
+ return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
375
+ }
376
+ };
377
+
378
+ // Retry wrapper (#34): only FAILED probes (non-200 / thrown fetch / timeout)
379
+ // retry — a confirmed answer returns immediately. Exhaustion returns the last
380
+ // failed result; the CALLER decides what a still-unknown capability means.
381
+ const probeModelsRetrying = async (chatUrl: string, headers: Record<string, string>, model: string, fetchTimeoutMs: number, attempts: number, delayMs: number): Promise<EndpointProbe> => {
382
+ let probe: EndpointProbe = { nCtx: null, llamaServer: false, servedModel: null, failed: true };
383
+ for (let attempt = 0; attempt < attempts; attempt++) {
384
+ if (attempt > 0) await new Promise((r) => setTimeout(r, delayMs * 2 ** (attempt - 1)));
385
+ probe = await probeModels(chatUrl, headers, model, fetchTimeoutMs);
386
+ if (!probe.failed) return probe;
387
+ }
388
+ return probe;
389
+ };
390
+
391
+ // llama-server /props: total_slots (the valid id_slot range for slot pinning) +
392
+ // eos_token (the EOG the server renders as TEXT under --special, #539). One fetch,
393
+ // both facts. Only queried after the llama-server fingerprint confirms; same
394
+ // best-effort posture as the models probe.
395
+ const probeServerProps = async (chatUrl: string, headers: Record<string, string>, fetchTimeoutMs: number): Promise<{ slotCount: number | null; eosToken: string | null }> => {
396
+ const propsUrl = chatUrl.replace(/\/v1\/chat\/completions$/, "/props");
397
+ try {
398
+ const res = await fetch(propsUrl, { headers, signal: AbortSignal.timeout(fetchTimeoutMs) });
399
+ if (!res.ok) return { slotCount: null, eosToken: null };
400
+ const data = (await res.json()) as { total_slots?: number; eos_token?: string };
401
+ return {
402
+ slotCount: typeof data.total_slots === "number" && data.total_slots > 0 ? data.total_slots : null,
403
+ eosToken: typeof data.eos_token === "string" && data.eos_token.length > 0 ? data.eos_token : null,
404
+ };
405
+ } catch {
406
+ return { slotCount: null, eosToken: null };
407
+ }
408
+ };
409
+
410
+ // Returns a configured Provider, or null when `name` is not a standard
411
+ // provider (so the consumer falls through to dynamic import). Async because a
412
+ // probeNctx-enabled provider queries /v1/models at construction.
413
+ export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessEnv, model: string, baseUrlOverride?: string): Promise<Provider | null> => {
414
+ const spec = STANDARD_PROVIDERS[name];
415
+ if (spec === undefined) return null;
416
+
417
+ // The on-the-wire model id: a backend-required constant prefix (fireworks)
418
+ // prepended idempotently, so the operator's alias carries only the distinctive
419
+ // tail. This id is what the backend, the probe, AND the catalog key on.
420
+ const wireModel = spec.modelPrefix !== undefined && !model.startsWith(spec.modelPrefix)
421
+ ? `${spec.modelPrefix}${model}`
422
+ : model;
423
+
424
+ const headers = resolveHeaders(spec, env, name);
425
+
426
+ // Tokenizer shed (mimetypes#44 landed): client-side families are GONE — exact
427
+ // counting lives in the mimetypes tokenizer seam + the tokenize() capability.
428
+ // A still-set family var fails hard with the migration pointer; countTokens is
429
+ // the chars/2 upper bound, surfaced so window math is never silently inexact.
430
+ // Namespace seizure (fail-forward): the debug knob moved into the owned
431
+ // PLURNK_PROVIDERS_ namespace (alias-scopable). The old name fails hard.
432
+ if (env.PLURNK_GBNF_DEBUG !== undefined && env.PLURNK_GBNF_DEBUG.length > 0) {
433
+ throw new Error(`${name} provider: PLURNK_GBNF_DEBUG was renamed to PLURNK_PROVIDERS_GBNF_DEBUG (alias-scopable as _<alias>); update the env`);
434
+ }
435
+ const staleTokenizer = env[spec.tokenizerEnvVar];
436
+ if (staleTokenizer !== undefined && staleTokenizer.length > 0) {
437
+ throw new Error(`${name} provider: ${spec.tokenizerEnvVar} was removed — exact counting moved to the @plurnk/plurnk-mimetypes tokenizer seam (mimetypes#44) and Provider.tokenize(); unset the var (countTokens is a chars/2 upper bound)`);
438
+ }
439
+ // Once per process per (code, message) — #40: construction-per-worker daemons
440
+ // repeated this until operators tuned warnings out.
441
+ emitWarningOnce(
442
+ `${name} provider: countTokens is a chars/2 upper bound — exact counts come from the mimetypes tokenizer seam or tokenize()`,
443
+ "PLURNK_TOKENIZER_HEURISTIC",
444
+ );
445
+ const url = resolveUrl(spec, env, name, baseUrlOverride);
446
+ const fetchTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name);
447
+
448
+ // The probe always runs for probeNctx specs — grammar capability must not
449
+ // hinge on whether the operator pinned PLURNK_PROVIDERS_CONTEXT_WINDOW. For
450
+ // contextWindow itself, explicit env still wins over the probed n_ctx.
451
+ let contextWindow = contextWindowFromEnv(env, name);
452
+ // Grammar shape: a static spec choice (e.g. fireworks → "response_format"),
453
+ // upgraded to "llamacpp" when the probe fingerprints a llama-server. Slot
454
+ // pinning is llama-server-only, so it keys on that same fingerprint. A spec
455
+ // can opt out of the fingerprint entirely (detectLlamaServer: false → plurnk)
456
+ // to read the window but stay a plain remote OpenAI server.
457
+ let grammarStyle: GrammarStyle = spec.grammarStyle ?? "none";
458
+ let supportsSlotPinning = false;
459
+ let slotCount: number | null = null;
460
+ let eosText: string | undefined;
461
+ let tokenizeUrl: string | undefined;
462
+ let servedModel: string | undefined;
463
+ let requiresMaxTokens: boolean | undefined;
464
+ let reasoningStyle = spec.reasoningStyle;
465
+ if (spec.probeNctx === true) {
466
+ // Operator pin (#34): "1" → llama-server capabilities WITHOUT trusting
467
+ // the probe; "0" → plain remote, skip the fingerprint; unset → detect.
468
+ // A plurnk-owned UNIVERSAL knob (alias-scopable: _<alias> suffix wins).
469
+ const pinRaw = env.PLURNK_PROVIDERS_LLAMA_SERVER;
470
+ if (pinRaw !== undefined && pinRaw !== "" && pinRaw !== "0" && pinRaw !== "1") {
471
+ throw new Error(`${name} provider: PLURNK_PROVIDERS_LLAMA_SERVER must be "1" (pin llama-server capabilities), "0" (force plain remote), or unset (auto-detect) (got "${pinRaw}")`);
472
+ }
473
+ const pin = pinRaw === undefined || pinRaw === "" ? null : pinRaw === "1";
474
+ const probeAttempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", name);
475
+ const probe = await probeModelsRetrying(url, headers, wireModel, fetchTimeoutMs, probeAttempts, parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_DELAY, "PLURNK_PROVIDERS_PROBE_DELAY", name));
476
+ contextWindow ??= probe.nCtx;
477
+ // #37: the backend's self-reported served id (from the same probe), so an
478
+ // alias-fronted local model resolves its exact tokenizer. Absent otherwise.
479
+ servedModel = probe.servedModel ?? undefined;
480
+ const isLlama = pin ?? (probe.llamaServer && spec.detectLlamaServer !== false);
481
+ if (isLlama && spec.detectLlamaServer !== false) {
482
+ grammarStyle = "llamacpp";
483
+ supportsSlotPinning = true;
484
+ const serverProps = await probeServerProps(url, headers, fetchTimeoutMs);
485
+ slotCount = serverProps.slotCount;
486
+ eosText = serverProps.eosToken ?? undefined;
487
+ // llama-server serves its NATIVE /tokenize at the root (like /props):
488
+ // the model's own vocab, exact — surfaced as the tokenize() capability.
489
+ tokenizeUrl = url.replace(/\/v1\/chat\/completions$/, "/tokenize");
490
+ // llama-server ignores `think` — its working reasoning toggle is the
491
+ // jinja chat_template_kwargs.enable_thinking, including the explicit
492
+ // FALSE at REASONING=off that grammar-constrained loops require (§13).
493
+ if (reasoningStyle === "think") reasoningStyle = "template";
494
+ // #43: llama-server honors n_predict to the CONTEXT WALL (providers#10)
495
+ // — no self-clamp. Surface the fact so a consumer can boot-refuse a
496
+ // local alias whose output envelope was never declared.
497
+ requiresMaxTokens = true;
498
+ } else if (pin === null && probe.failed && spec.detectLlamaServer !== false) {
499
+ // Detection exhausted its retries with NO answer: capability stays
500
+ // un-upgraded, but NEVER silently (#34) — rails going dark without a
501
+ // signal cost the consumer weeks of misattributed rambles.
502
+ emitWarningOnce(
503
+ `${name} provider: llama-server detection failed after ${probeAttempts} attempts — grammar transport stays OFF (grammarStyle "none"). If this endpoint IS a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
504
+ "PLURNK_PROBE_FAILED",
505
+ );
506
+ }
507
+ }
508
+
509
+ // Vendored-snapshot FALLBACK (#19) — live always wins. contextWindow already
510
+ // preferred env then the live probe; the catalog only fills a still-null
511
+ // window for a known cloud model (groq/deepseek/mistral/…, which don't
512
+ // probe). A local llama-server model misses the catalog and keeps its
513
+ // probed n_ctx. Standard providers carry NO live pricing, so the catalog is
514
+ // the sole — never shadowing — cost source; per-1M USD → pico-USD/token (×1e6).
515
+ // A relay with a catalogContextLookup (bedrock) resolves its window via the
516
+ // underlying model's publisher and carries NO catalog cost (native rate ≠ relay
517
+ // rate, #22); everyone else keys the catalog directly on (name, wireModel).
518
+ const fallback = spec.catalogContextLookup === undefined ? lookup(name, wireModel) : undefined;
519
+ contextWindow ??= (spec.catalogContextLookup !== undefined ? spec.catalogContextLookup(wireModel) : fallback?.contextWindow) ?? null;
520
+ // Unresolved window (env, probe, catalog ALL missed) - two paths (#419 hybrid).
521
+ // A CLOUD provider (no probe) has no other window source, so an uncataloged,
522
+ // unpinned model is a config error we FAIL-HARD on (the #417 kimi case) rather
523
+ // than budget against a wrong number. A PROBING provider (openai/llama-server)
524
+ // DEGRADES to null instead: a probe blip or a box that doesn't report n_ctx must
525
+ // not crash construction (#34 robustness); the consumer treats null as "no cap",
526
+ // never a wrong CTX stand-in. Either way the unknown is SURFACED, never silent.
527
+ if (contextWindow === null) {
528
+ if (spec.probeNctx !== true) {
529
+ throw new Error(
530
+ `${name} provider: context window unresolved for "${wireModel}" - a cloud provider with no probe, absent from the @plurnk/plurnk-models catalog and unpinned. Pin PLURNK_PROVIDERS_CONTEXT_WINDOW (alias-scopable as _<alias>) or add the model to the catalog (#419)`,
531
+ );
532
+ }
533
+ emitWarningOnce(
534
+ `${name} provider: context window underivable for "${wireModel}" (probe returned no n_ctx, catalog miss) - contextWindow=null; set PLURNK_PROVIDERS_CONTEXT_WINDOW (alias-scopable as _<alias>) so window budgets are deliberate`,
535
+ "PLURNK_CONTEXT_UNKNOWN",
536
+ );
537
+ }
538
+ const cost = fallback?.cost;
539
+ const costFor = cost === undefined
540
+ ? undefined
541
+ : (usage: ProviderUsage): number => computeCost(usage, {
542
+ input: cost.inputPer1M * 1e6,
543
+ output: cost.outputPer1M * 1e6,
544
+ cached: (cost.cacheReadPer1M ?? cost.inputPer1M) * 1e6,
545
+ });
546
+
547
+ // #507: completion cap — when the catalog reports a maxOutput, use
548
+ // min(maxOutput, percentage_amount) as an absolute reserve instead of the
549
+ // raw percentage. The 25% floor dramatically over-reserves on large-context
550
+ // models (gpt-4.1-mini: 1M ctx → 262K reserved, real cap 32K; claude-sonnet:
551
+ // 1M ctx → 250K reserved, real cap 128K). The min() preserves the floor on
552
+ // models where maxOutput > 25%*ctx (deepseek 1M/384K, llama 131K/131K).
553
+ // An operator absolute pin (COMPLETION_RESERVE=8192 not "25%") always wins.
554
+ const reasoning = reasoningFromEnv(env, name);
555
+ const { reasoningReserve: envReasoning, completionReserve: envCompletion } = envelopeFromEnv(env, name);
556
+ const completionReserve: ReserveSpec = (() => {
557
+ if ("tokens" in envCompletion) return envCompletion; // operator absolute always wins
558
+ if (fallback?.maxOutput === undefined || contextWindow === null) return envCompletion;
559
+ return { tokens: Math.min(fallback.maxOutput, Math.round(envCompletion.percent * contextWindow)) };
560
+ })();
561
+ // #568: reasoning reserve — exact REASONING_BUDGET > half-completion > env%.
562
+ // Using the budget directly prevents double-counting (half of 250K = 125K when
563
+ // the actual budget is 8K). Half-completion is the adaptive heuristic when
564
+ // reasoning=adaptive (budget=null). The env% is the last resort (only reached
565
+ // when contextWindow is null AND reasoning=adaptive).
566
+ const resolvedCompletion = resolveReserve(completionReserve, contextWindow);
567
+ const reasoningReserve: ReserveSpec = (() => {
568
+ if ("tokens" in envReasoning) return envReasoning; // operator absolute always wins
569
+ if (reasoning.budget !== null) return { tokens: reasoning.budget }; // exact budget when reasoning=on
570
+ if (resolvedCompletion !== null) return { tokens: Math.round(resolvedCompletion / 2) }; // half-completion for adaptive
571
+ return envReasoning; // last resort: env percentage
572
+ })();
573
+
574
+ return new OpenAICompatProvider({
575
+ model: wireModel,
576
+ url,
577
+ headers,
578
+ contextWindow,
579
+ fetchTimeoutMs,
580
+ reasoning,
581
+ temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
582
+ repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
583
+ frequencyPenalty: spec.suppressFrequencyPenalty === true ? 0 : parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
584
+ // #567: llama.cpp loop-breakers — optional, off by default (absent = the box's own
585
+ // default). The provider sends them only on the llamacpp path; values are operator config.
586
+ dryMultiplier: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_MULTIPLIER, "PLURNK_PROVIDERS_DRY_MULTIPLIER", name, 0) ?? undefined,
587
+ dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", name, 0) ?? undefined,
588
+ dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", name) ?? undefined,
589
+ repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", name) ?? undefined,
590
+ // #507: resolved envelope (catalog-capped completion above) + router-owned-tuning suppression (plurnk).
591
+ reasoningReserve,
592
+ completionReserve,
593
+ tuningFloors: spec.suppressTuningFloors !== true,
594
+ retryDelayMs: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_DELAY, "PLURNK_PROVIDERS_RETRY_DELAY", name),
595
+ retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
596
+ reasoningStyle,
597
+ costFor,
598
+ source: providerSource(name),
599
+ grammarStyle,
600
+ // Optional debug toggle (off by default): validate a transported grammar
601
+ // locally and throw on invalid, without sending it to the model (§13).
602
+ gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "" && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
603
+ // Opt-in data capture (#36), off by default, per-alias-scopable — universal
604
+ // across every standard provider (any backend that returns logprobs).
605
+ ...dataCaptureFromEnv(env, name),
606
+ streaming: spec.streaming,
607
+ firstPartyMetadata: spec.firstPartyMetadata,
608
+ apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
609
+ promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
610
+ balanceMetaKey: spec.balanceMetaKey,
611
+ supportsSlotPinning,
612
+ slotCount,
613
+ eosText,
614
+ tokenizeUrl,
615
+ servedModel,
616
+ requiresMaxTokens,
617
+ });
618
+ };