@plurnk/plurnk-providers 1.2.0 → 1.3.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -60
- package/SPEC.md +6 -6
- package/dist/OpenAICompat.js +2 -2
- package/dist/OpenAICompat.js.map +1 -1
- package/dist/ProviderRegistry.js +2 -2
- package/dist/ProviderRegistry.js.map +1 -1
- package/dist/env.js +3 -3
- package/dist/env.js.map +1 -1
- package/dist/openaiStream.d.ts.map +1 -1
- package/dist/openaiStream.js +12 -0
- package/dist/openaiStream.js.map +1 -1
- package/package.json +7 -6
- package/src/Mock.test.ts +142 -0
- package/src/Mock.ts +95 -0
- package/src/OpenAICompat.test.ts +1107 -0
- package/src/OpenAICompat.ts +756 -0
- package/src/Pool.test.ts +155 -0
- package/src/Pool.ts +134 -0
- package/src/ProviderRegistry.test.ts +176 -0
- package/src/ProviderRegistry.ts +93 -0
- package/src/boundaries.test.ts +24 -0
- package/src/discover.test.ts +123 -0
- package/src/discover.ts +112 -0
- package/src/env.test.ts +190 -0
- package/src/env.ts +211 -0
- package/src/index.ts +51 -0
- package/src/lexicon-guard.test.ts +58 -0
- package/src/openaiStream.ts +279 -0
- package/src/standardProviders.test.ts +925 -0
- package/src/standardProviders.ts +618 -0
- package/src/telemetry.test.ts +62 -0
- package/src/telemetry.ts +108 -0
- package/src/types.ts +219 -0
- package/src/usage.test.ts +136 -0
- package/src/usage.ts +82 -0
- package/src/warnings.test.ts +31 -0
- package/src/warnings.ts +0 -0
|
@@ -0,0 +1,618 @@
|
|
|
1
|
+
// Pure-config OpenAI-compatible providers. A provider qualifies as "standard"
|
|
2
|
+
// when it has no unique runtime surface — no catalog probe, no pricing fetch,
|
|
3
|
+
// no bespoke wire shape — so it reduces to: an env var for the key, a base
|
|
4
|
+
// URL, a reasoning-translation style, and a tokenizer. Such providers need NO
|
|
5
|
+
// sibling package; the framework instantiates them directly.
|
|
6
|
+
//
|
|
7
|
+
// Two-tier resolution (SPEC §5): the consumer tries standardProviderFromEnv
|
|
8
|
+
// first, then falls back to the discover() node_modules scan for the bespoke
|
|
9
|
+
// ones (openrouter, ollama, google, xai, cloudflare, third-party). The scan
|
|
10
|
+
// resolves the package specifier — it is NOT a hardcoded @plurnk/ pattern.
|
|
11
|
+
|
|
12
|
+
import type { Provider, ProviderUsage } from "./types.ts";
|
|
13
|
+
import OpenAICompatProvider, { type ReasoningStyle, type GrammarStyle } from "./OpenAICompat.ts";
|
|
14
|
+
import { parseRequiredInt, parseOptionalInt, parseRequiredFloat, parseOptionalFloat, reasoningFromEnv, dataCaptureFromEnv, contextWindowFromEnv, envelopeFromEnv, resolveReserve, type ReserveSpec } from "./env.ts";
|
|
15
|
+
import { emitWarningOnce } from "./warnings.ts";
|
|
16
|
+
import { providerSource } from "./telemetry.ts";
|
|
17
|
+
import { computeCost } from "./usage.ts";
|
|
18
|
+
import { lookup } from "@plurnk/plurnk-models";
|
|
19
|
+
|
|
20
|
+
type StandardProviderSpec = {
|
|
21
|
+
// Bearer-auth env var(s), and whether the key is mandatory (local
|
|
22
|
+
// OpenAI-compat servers run without auth, so the generic "openai" entry
|
|
23
|
+
// leaves it optional). A LIST accepts the conventional aliases the wild uses
|
|
24
|
+
// for one credential (e.g. deepinfra's API_KEY / API_TOKEN / TOKEN) — first
|
|
25
|
+
// non-empty wins, the required-but-unset error names them all. Omit when
|
|
26
|
+
// supplying a custom `headersFromEnv` builder.
|
|
27
|
+
apiKeyVar?: string | readonly string[];
|
|
28
|
+
apiKeyRequired?: boolean;
|
|
29
|
+
// Custom missing-key throw message (replaces the generic "<var> must be set")
|
|
30
|
+
// — friendly, actionable guidance shown ONLY on the unset path (#537). A
|
|
31
|
+
// present-but-rejected key is a wire 401 downstream, never this.
|
|
32
|
+
apiKeyMessage?: string;
|
|
33
|
+
// Custom message when a PRESENT key is rejected by the backend (a live 401/403
|
|
34
|
+
// with a bearer sent) — distinct from the unset-key apiKeyMessage (#537 case 2).
|
|
35
|
+
apiKeyRejectedMessage?: string;
|
|
36
|
+
// Custom request-header builder for auth the single-var bearer can't express
|
|
37
|
+
// (multiple optional credentials, vendor routing headers). Returns the
|
|
38
|
+
// headers built from env; an empty object means no auth headers are sent.
|
|
39
|
+
// When set, it REPLACES the apiKeyVar bearer logic.
|
|
40
|
+
headersFromEnv?: (env: NodeJS.ProcessEnv) => Record<string, string>;
|
|
41
|
+
// Base URL env var(s): no in-code default; the canonical endpoint ships as a
|
|
42
|
+
// floored default in .env.defaults (overridable in the operator's env or
|
|
43
|
+
// per-alias), never a baked constant. A list accepts conventional aliases
|
|
44
|
+
// (openai's BASE_URL / API_BASE). Either this or baseUrlFromEnv must resolve.
|
|
45
|
+
baseUrlVar?: string | readonly string[];
|
|
46
|
+
// Derive the base from env when no override var is set — for endpoints whose
|
|
47
|
+
// URL is templated from standard env (bedrock builds it from AWS_REGION).
|
|
48
|
+
// Throws a named error if it can't derive.
|
|
49
|
+
baseUrlFromEnv?: (env: NodeJS.ProcessEnv) => string | undefined;
|
|
50
|
+
// Custom catalog context-window resolver for a relay whose model id the
|
|
51
|
+
// models.dev catalog doesn't key directly (bedrock inference profiles, #22).
|
|
52
|
+
// Returns the model's context window or undefined. When set, the catalog COST
|
|
53
|
+
// path is skipped — the model's native rate isn't this relay's rate.
|
|
54
|
+
catalogContextLookup?: (model: string) => number | undefined;
|
|
55
|
+
// Path appended to the (slash-trimmed) base to reach chat-completions.
|
|
56
|
+
chatPath: string;
|
|
57
|
+
// When true (generic "openai" only), strip a trailing /v1 from the
|
|
58
|
+
// operator-supplied base before appending chatPath — the base may or may
|
|
59
|
+
// not already include it.
|
|
60
|
+
flexBaseStrip?: boolean;
|
|
61
|
+
reasoningStyle: ReasoningStyle;
|
|
62
|
+
// How this backend carries a GBNF grammar (default "none" — not sent). A
|
|
63
|
+
// probeNctx entry is upgraded to "llamacpp" when the probe sees a
|
|
64
|
+
// llama-server; cloud backends that support GBNF set their shape statically
|
|
65
|
+
// (fireworks → "response_format", verified live).
|
|
66
|
+
grammarStyle?: GrammarStyle;
|
|
67
|
+
// SSE streaming (default true). The streaming transport is dropped
|
|
68
|
+
// per-request only when it would break a feature (a response_format grammar
|
|
69
|
+
// arrives mislabeled as reasoning_content under fireworks' stream); leave
|
|
70
|
+
// unset to keep streaming on for every other call. See OpenAICompat.generate.
|
|
71
|
+
streaming?: boolean;
|
|
72
|
+
// Constant model-id prefix the backend requires but the alias shouldn't
|
|
73
|
+
// repeat (fireworks → "accounts/fireworks/models/", so the alias is just
|
|
74
|
+
// `fireworks/deepseek-v4-pro`). Prepended idempotently to form the wire id,
|
|
75
|
+
// which is ALSO the catalog key (models.dev keys fireworks-ai on the full id).
|
|
76
|
+
modelPrefix?: string;
|
|
77
|
+
// First-party telemetry forwarding. ONLY the plurnk hosted endpoint sets
|
|
78
|
+
// this — it forwards the consumer's per-turn `attributions` (contributor
|
|
79
|
+
// credit) and `client` (originating frontend) as `Plurnk-Attribution` /
|
|
80
|
+
// `Plurnk-Client` headers. Absent everywhere else, so those signals are
|
|
81
|
+
// structurally incapable of reaching a third-party backend (never sold,
|
|
82
|
+
// never leaked — the destination is the consent boundary).
|
|
83
|
+
firstPartyMetadata?: boolean;
|
|
84
|
+
// #507: the plurnk.ai router owns tuning (SPEC §5) — suppress the client-side
|
|
85
|
+
// temperature/penalty floors on this provider; caller sampling still passes.
|
|
86
|
+
suppressTuningFloors?: boolean;
|
|
87
|
+
// When true, omit frequency_penalty from the request body — the backend's
|
|
88
|
+
// native param set excludes it and will reject it (DOC-verified). Sets
|
|
89
|
+
// frequencyPenalty to 0 so #repetitionPenaltyBody's "> 0" gate suppresses
|
|
90
|
+
// the field. Mirrors the plugin omissions (dba4300 / providers-xai#2).
|
|
91
|
+
suppressFrequencyPenalty?: boolean;
|
|
92
|
+
// #518: send prompt_cache_key=workerId (serverless replica-cache affinity).
|
|
93
|
+
// Default ON for standard providers (OpenAI-standard field, broadly accepted);
|
|
94
|
+
// set false to opt a backend out (e.g. anthropic's cache_control mechanism).
|
|
95
|
+
promptCacheKey?: boolean;
|
|
96
|
+
// Top-level response field the endpoint reports account balance (pico-USD) in,
|
|
97
|
+
// surfaced as ProviderResponse.balancePico (plurnk only, #23). Absent elsewhere.
|
|
98
|
+
balanceMetaKey?: string;
|
|
99
|
+
// RETIRED knob (#27→mimetypes#44): exact client-side tokenizer families were
|
|
100
|
+
// removed with the tokenizer shed — the var is kept ONLY to fail hard with a
|
|
101
|
+
// migration pointer when an operator still sets it.
|
|
102
|
+
tokenizerEnvVar: string;
|
|
103
|
+
// When true, probe GET /v1/models at construction. Two reads off one call:
|
|
104
|
+
// the endpoint-reported context window (`n_ctx`, used when
|
|
105
|
+
// PLURNK_PROVIDERS_CONTEXT_WINDOW is unset) and the llama-server fingerprint
|
|
106
|
+
// (a `meta` block on the model row) that enables grammar-constrained
|
|
107
|
+
// sampling (SPEC §13). Set for providers that may front a local
|
|
108
|
+
// OpenAI-compat server; cloud endpoints report neither → null / false.
|
|
109
|
+
probeNctx?: boolean;
|
|
110
|
+
// Whether a probeNctx spec may infer LOCAL llama-server capabilities (grammar
|
|
111
|
+
// transport → "llamacpp", slot pinning, template reasoning) from the probe's
|
|
112
|
+
// `meta` fingerprint. Default true. Set FALSE for an endpoint that reports a
|
|
113
|
+
// window but must be treated as a plain remote OpenAI server — `plurnk` reads
|
|
114
|
+
// its (server-controlled) window from upstream yet must NEVER be talked into
|
|
115
|
+
// grammar/slot behavior, so its capabilities can't be flipped by what the
|
|
116
|
+
// endpoint happens to return.
|
|
117
|
+
detectLlamaServer?: boolean;
|
|
118
|
+
};
|
|
119
|
+
|
|
120
|
+
// Frozen so a downstream can't mutate the shared table.
|
|
121
|
+
export const STANDARD_PROVIDERS: Readonly<Record<string, StandardProviderSpec>> = Object.freeze({
|
|
122
|
+
// Generic OpenAI-compatible endpoint (OpenAI proper, llama-server, vLLM,
|
|
123
|
+
// LM Studio, or any chat-completions shim). Operator supplies the base.
|
|
124
|
+
// Replaces the former @plurnk/plurnk-providers-openai sibling verbatim.
|
|
125
|
+
openai: {
|
|
126
|
+
apiKeyVar: "OPENAI_API_KEY", apiKeyRequired: false,
|
|
127
|
+
baseUrlVar: ["OPENAI_BASE_URL", "OPENAI_API_BASE"], chatPath: "/v1/chat/completions", flexBaseStrip: true,
|
|
128
|
+
reasoningStyle: "think", tokenizerEnvVar: "OPENAI_TOKENIZER",
|
|
129
|
+
probeNctx: true,
|
|
130
|
+
},
|
|
131
|
+
groq: {
|
|
132
|
+
apiKeyVar: "GROQ_API_KEY", apiKeyRequired: true,
|
|
133
|
+
baseUrlVar: "GROQ_BASE_URL", chatPath: "/chat/completions",
|
|
134
|
+
reasoningStyle: "effort", tokenizerEnvVar: "GROQ_TOKENIZER",
|
|
135
|
+
},
|
|
136
|
+
deepseek: {
|
|
137
|
+
apiKeyVar: "DEEPSEEK_API_KEY", apiKeyRequired: true,
|
|
138
|
+
baseUrlVar: "DEEPSEEK_BASE_URL", chatPath: "/chat/completions",
|
|
139
|
+
reasoningStyle: "none", tokenizerEnvVar: "DEEPSEEK_TOKENIZER",
|
|
140
|
+
},
|
|
141
|
+
mistral: {
|
|
142
|
+
apiKeyVar: "MISTRAL_API_KEY", apiKeyRequired: true,
|
|
143
|
+
baseUrlVar: "MISTRAL_BASE_URL", chatPath: "/chat/completions",
|
|
144
|
+
reasoningStyle: "none", tokenizerEnvVar: "MISTRAL_TOKENIZER",
|
|
145
|
+
},
|
|
146
|
+
together: {
|
|
147
|
+
apiKeyVar: "TOGETHER_API_KEY", apiKeyRequired: true,
|
|
148
|
+
baseUrlVar: "TOGETHER_BASE_URL", chatPath: "/chat/completions",
|
|
149
|
+
reasoningStyle: "none", tokenizerEnvVar: "TOGETHER_TOKENIZER",
|
|
150
|
+
},
|
|
151
|
+
// reasoningStyle "effort_explicit", NOT "none": fireworks serves reason-by-
|
|
152
|
+
// DEFAULT models (DeepSeek V4 defaults 'high'), so budget 0 must SEND
|
|
153
|
+
// reasoning_effort:"none" — omitting the field leaves the reasoner live inside
|
|
154
|
+
// a constrained decode until max_tokens (#30; 0/5 → 30/30 measured).
|
|
155
|
+
fireworks: {
|
|
156
|
+
apiKeyVar: "FIREWORKS_API_KEY", apiKeyRequired: true,
|
|
157
|
+
baseUrlVar: "FIREWORKS_BASE_URL", chatPath: "/chat/completions",
|
|
158
|
+
reasoningStyle: "effort_explicit", grammarStyle: "response_format", modelPrefix: "accounts/fireworks/models/", tokenizerEnvVar: "FIREWORKS_TOKENIZER",
|
|
159
|
+
},
|
|
160
|
+
deepinfra: {
|
|
161
|
+
apiKeyVar: ["DEEPINFRA_API_KEY", "DEEPINFRA_API_TOKEN", "DEEPINFRA_TOKEN"], apiKeyRequired: true,
|
|
162
|
+
baseUrlVar: "DEEPINFRA_BASE_URL", chatPath: "/chat/completions",
|
|
163
|
+
reasoningStyle: "none", tokenizerEnvVar: "DEEPINFRA_TOKENIZER",
|
|
164
|
+
},
|
|
165
|
+
// — Chinese cloud hosts (all OpenAI-compat, plain bearer, doc-verified 2026). —
|
|
166
|
+
// The .env.defaults base is the INTERNATIONAL endpoint; mainland operators
|
|
167
|
+
// point the override var at the `.cn` twin noted per entry. Reasoning is left
|
|
168
|
+
// "none" for the whole cohort: each host's reasoning toggle is either
|
|
169
|
+
// model-selected or a non-standard param that rides in `extra_body` (the
|
|
170
|
+
// `effort`/`anthropic` styles don't reach it) — same posture as deepseek.
|
|
171
|
+
// None are in the @plurnk/plurnk-models snapshot, so context comes from
|
|
172
|
+
// PLURNK_PROVIDERS_CONTEXT_WINDOW and cost stays unknown until the catalog adds
|
|
173
|
+
// them (a plurnk-models issue, not this repo's).
|
|
174
|
+
moonshot: {
|
|
175
|
+
apiKeyVar: "MOONSHOT_API_KEY", apiKeyRequired: true,
|
|
176
|
+
baseUrlVar: "MOONSHOT_BASE_URL", chatPath: "/chat/completions",
|
|
177
|
+
reasoningStyle: "none", tokenizerEnvVar: "MOONSHOT_TOKENIZER",
|
|
178
|
+
},
|
|
179
|
+
// Alibaba Qwen via DashScope "compatible-mode". Mainland: dashscope.aliyuncs.com.
|
|
180
|
+
dashscope: {
|
|
181
|
+
apiKeyVar: "DASHSCOPE_API_KEY", apiKeyRequired: true,
|
|
182
|
+
baseUrlVar: "DASHSCOPE_BASE_URL", chatPath: "/chat/completions",
|
|
183
|
+
reasoningStyle: "none", tokenizerEnvVar: "DASHSCOPE_TOKENIZER",
|
|
184
|
+
},
|
|
185
|
+
// Zhipu GLM. Base carries /api/paas/v4 (non-/v1). Mainland: open.bigmodel.cn/api/paas/v4.
|
|
186
|
+
zhipu: {
|
|
187
|
+
apiKeyVar: ["ZHIPUAI_API_KEY", "ZAI_API_KEY"], apiKeyRequired: true,
|
|
188
|
+
baseUrlVar: "ZHIPU_BASE_URL", chatPath: "/chat/completions",
|
|
189
|
+
reasoningStyle: "none", tokenizerEnvVar: "ZHIPU_TOKENIZER",
|
|
190
|
+
},
|
|
191
|
+
// ByteDance Doubao via BytePlus ModelArk (base carries /api/v3; `model` is an
|
|
192
|
+
// inference-endpoint/model id). Mainland (Volcengine): ark.cn-beijing.volces.com/api/v3.
|
|
193
|
+
volcengine: {
|
|
194
|
+
apiKeyVar: "ARK_API_KEY", apiKeyRequired: true,
|
|
195
|
+
baseUrlVar: "ARK_BASE_URL", chatPath: "/chat/completions",
|
|
196
|
+
reasoningStyle: "none", tokenizerEnvVar: "ARK_TOKENIZER",
|
|
197
|
+
},
|
|
198
|
+
// Tencent Hunyuan — single global host (no intl/mainland split).
|
|
199
|
+
hunyuan: {
|
|
200
|
+
apiKeyVar: "HUNYUAN_API_KEY", apiKeyRequired: true,
|
|
201
|
+
baseUrlVar: "HUNYUAN_BASE_URL", chatPath: "/chat/completions",
|
|
202
|
+
reasoningStyle: "none", tokenizerEnvVar: "HUNYUAN_TOKENIZER",
|
|
203
|
+
},
|
|
204
|
+
// MiniMax. Mainland twin is api.minimaxi.com (note the extra "i").
|
|
205
|
+
minimax: {
|
|
206
|
+
apiKeyVar: "MINIMAX_API_KEY", apiKeyRequired: true,
|
|
207
|
+
baseUrlVar: "MINIMAX_BASE_URL", chatPath: "/chat/completions",
|
|
208
|
+
reasoningStyle: "none", tokenizerEnvVar: "MINIMAX_TOKENIZER",
|
|
209
|
+
},
|
|
210
|
+
// StepFun. Intl twin is api.stepfun.ai (.ai vs the .com mainland host).
|
|
211
|
+
stepfun: {
|
|
212
|
+
apiKeyVar: "STEP_API_KEY", apiKeyRequired: true,
|
|
213
|
+
baseUrlVar: "STEPFUN_BASE_URL", chatPath: "/chat/completions",
|
|
214
|
+
reasoningStyle: "none", tokenizerEnvVar: "STEPFUN_TOKENIZER",
|
|
215
|
+
},
|
|
216
|
+
// Baichuan — single host.
|
|
217
|
+
baichuan: {
|
|
218
|
+
apiKeyVar: "BAICHUAN_API_KEY", apiKeyRequired: true,
|
|
219
|
+
baseUrlVar: "BAICHUAN_BASE_URL", chatPath: "/chat/completions",
|
|
220
|
+
reasoningStyle: "none", tokenizerEnvVar: "BAICHUAN_TOKENIZER",
|
|
221
|
+
},
|
|
222
|
+
// Baidu ERNIE via Qianfan v2 (key is a bce-v3/ALTAK-… bearer; base carries /v2).
|
|
223
|
+
qianfan: {
|
|
224
|
+
apiKeyVar: "QIANFAN_API_KEY", apiKeyRequired: true,
|
|
225
|
+
baseUrlVar: "QIANFAN_BASE_URL", chatPath: "/chat/completions",
|
|
226
|
+
reasoningStyle: "none", tokenizerEnvVar: "QIANFAN_TOKENIZER",
|
|
227
|
+
},
|
|
228
|
+
// SiliconFlow aggregator. Mainland twin is api.siliconflow.cn.
|
|
229
|
+
siliconflow: {
|
|
230
|
+
apiKeyVar: "SILICONFLOW_API_KEY", apiKeyRequired: true,
|
|
231
|
+
baseUrlVar: "SILICONFLOW_BASE_URL", chatPath: "/chat/completions",
|
|
232
|
+
reasoningStyle: "none", tokenizerEnvVar: "SILICONFLOW_TOKENIZER",
|
|
233
|
+
},
|
|
234
|
+
// ModelScope API-Inference aggregator — single host (.cn).
|
|
235
|
+
modelscope: {
|
|
236
|
+
apiKeyVar: ["MODELSCOPE_API_KEY", "MODELSCOPE_TOKEN"], apiKeyRequired: true,
|
|
237
|
+
baseUrlVar: "MODELSCOPE_BASE_URL", chatPath: "/chat/completions",
|
|
238
|
+
reasoningStyle: "none", tokenizerEnvVar: "MODELSCOPE_TOKENIZER",
|
|
239
|
+
},
|
|
240
|
+
// First-party Claude via Anthropic's OpenAI-compat endpoint: bearer auth,
|
|
241
|
+
// OpenAI SSE, the `thinking` reasoning param (reasoning_effort is ignored).
|
|
242
|
+
// No probe — context/cost come from the @plurnk/plurnk-models catalog.
|
|
243
|
+
anthropic: {
|
|
244
|
+
apiKeyVar: "ANTHROPIC_API_KEY", apiKeyRequired: true,
|
|
245
|
+
baseUrlVar: "ANTHROPIC_BASE_URL", chatPath: "/chat/completions",
|
|
246
|
+
reasoningStyle: "anthropic", tokenizerEnvVar: "ANTHROPIC_TOKENIZER",
|
|
247
|
+
promptCacheKey: false, // #518: anthropic caches via cache_control breakpoints, not prompt_cache_key (unverified) — opt out
|
|
248
|
+
suppressFrequencyPenalty: true, // Claude's native API has no frequency_penalty; the compat endpoint rejects it (DOC #568)
|
|
249
|
+
},
|
|
250
|
+
// AWS Bedrock via its OpenAI-compat endpoint (path is /openai/v1, NOT /v1),
|
|
251
|
+
// bearer-authed with a Bedrock API key (SigV4 optional). Region-templated base
|
|
252
|
+
// (see baseUrlFromEnv); model ids are inference profiles like
|
|
253
|
+
// `us.anthropic.claude-sonnet-4-6` (see catalogContextLookup). Cost stays unknown —
|
|
254
|
+
// bedrock marks up over the native rate, so set PLURNK_PROVIDERS_CONTEXT_WINDOW for
|
|
255
|
+
// a publisher the catalog lacks (#22). frequency_penalty suppressed: Bedrock's
|
|
256
|
+
// OpenAI-compat surfaces Claude's native param set, which excludes it (DOC #568).
|
|
257
|
+
bedrock: {
|
|
258
|
+
apiKeyVar: "AWS_BEARER_TOKEN_BEDROCK", apiKeyRequired: true,
|
|
259
|
+
baseUrlVar: "BEDROCK_BASE_URL", chatPath: "/chat/completions",
|
|
260
|
+
baseUrlFromEnv: (env) => {
|
|
261
|
+
const region = firstSet(env, ["AWS_REGION", "AWS_DEFAULT_REGION"]);
|
|
262
|
+
if (region === undefined) throw new Error("bedrock provider: BEDROCK_BASE_URL must be set, or AWS_REGION / AWS_DEFAULT_REGION to derive it");
|
|
263
|
+
return `https://bedrock-runtime.${region}.amazonaws.com/openai/v1`;
|
|
264
|
+
},
|
|
265
|
+
// Inference profile <region>.<publisher>.<model> — the catalog has no
|
|
266
|
+
// `bedrock` provider, so strip the region scope and look the model up under
|
|
267
|
+
// its PUBLISHER (anthropic, …). Only the context window rides; bedrock marks
|
|
268
|
+
// up, so the native cost is NOT used (cost stays unknown, #22).
|
|
269
|
+
catalogContextLookup: (model) => {
|
|
270
|
+
const stripped = model.replace(/^(?:us-gov|us|eu|apac)\./, "");
|
|
271
|
+
const dot = stripped.indexOf(".");
|
|
272
|
+
if (dot < 0) return undefined;
|
|
273
|
+
return lookup(stripped.slice(0, dot), stripped.slice(dot + 1))?.contextWindow;
|
|
274
|
+
},
|
|
275
|
+
reasoningStyle: "none", tokenizerEnvVar: "BEDROCK_TOKENIZER",
|
|
276
|
+
suppressFrequencyPenalty: true, // Bedrock surfaces Claude's native param set, which has no frequency_penalty (DOC #568)
|
|
277
|
+
},
|
|
278
|
+
// The plurnk hosted model — deliberately the most boring OpenAI-compatible
|
|
279
|
+
// client we can ship: the ecosystem must not know what sits behind
|
|
280
|
+
// plurnk.ai (model/window/grammar/tuning are the router's business).
|
|
281
|
+
// probeNctx reads the window from upstream (a 32k→48k change is a server
|
|
282
|
+
// decision, not a client release), but detectLlamaServer:false keeps it a
|
|
283
|
+
// plain remote server that can NOT be flipped into grammar/slot behavior.
|
|
284
|
+
// PLURNK_API_KEY is REQUIRED — plurnk.ai rejects keyless requests (live 401
|
|
285
|
+
// error_invalid_key, #537). The unset path throws the friendly apiKeyMessage
|
|
286
|
+
// pre-call instead of letting a raw upstream 401 terminate the loop.
|
|
287
|
+
plurnk: {
|
|
288
|
+
baseUrlVar: "PLURNK_BASE_URL", chatPath: "/chat/completions",
|
|
289
|
+
apiKeyVar: "PLURNK_API_KEY", apiKeyRequired: true,
|
|
290
|
+
apiKeyMessage: "PLURNK_API_KEY not found. Acquire one at https://plurnk.ai . Plurnk also supports local models and alternative cloud provider configurations.",
|
|
291
|
+
apiKeyRejectedMessage: "PLURNK_API_KEY was rejected by plurnk.ai (invalid or expired). Verify it at https://plurnk.ai .",
|
|
292
|
+
reasoningStyle: "none", grammarStyle: "none", tokenizerEnvVar: "PLURNK_TOKENIZER",
|
|
293
|
+
probeNctx: true, detectLlamaServer: false, firstPartyMetadata: true, balanceMetaKey: "balance_pico", suppressTuningFloors: true,
|
|
294
|
+
},
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
export const isStandardProvider = (name: string): boolean => name in STANDARD_PROVIDERS;
|
|
298
|
+
|
|
299
|
+
// Normalize a single-or-list env-var spec to a list, and return the first env
|
|
300
|
+
// var that is set non-empty (the accepted-alias resolution — first wins).
|
|
301
|
+
const asList = (v: string | readonly string[] | undefined): readonly string[] =>
|
|
302
|
+
v === undefined ? [] : typeof v === "string" ? [v] : v;
|
|
303
|
+
const firstSet = (env: NodeJS.ProcessEnv, names: readonly string[]): string | undefined => {
|
|
304
|
+
for (const name of names) {
|
|
305
|
+
const value = env[name];
|
|
306
|
+
if (value !== undefined && value.length > 0) return value;
|
|
307
|
+
}
|
|
308
|
+
return undefined;
|
|
309
|
+
};
|
|
310
|
+
|
|
311
|
+
const resolveUrl = (spec: StandardProviderSpec, env: NodeJS.ProcessEnv, label: string, override?: string): string => {
|
|
312
|
+
// per-alias override (PLURNK_BASEURL_<alias>) → base-URL var(s) → env-derived
|
|
313
|
+
// template. No in-code default — the endpoint is required operator config.
|
|
314
|
+
// The override goes through the same normalization (flexBaseStrip + chatPath),
|
|
315
|
+
// so an operator pastes the box URL verbatim.
|
|
316
|
+
const base = override ?? firstSet(env, asList(spec.baseUrlVar)) ?? spec.baseUrlFromEnv?.(env);
|
|
317
|
+
if (base === undefined || base.length === 0) {
|
|
318
|
+
const names = asList(spec.baseUrlVar);
|
|
319
|
+
throw new Error(`${label} provider: ${names.length > 0 ? names.join(" or ") : "base URL"} must be set`);
|
|
320
|
+
}
|
|
321
|
+
const trimmed = spec.flexBaseStrip === true ? base.replace(/\/v1\/?$/, "") : base.replace(/\/$/, "");
|
|
322
|
+
return `${trimmed}${spec.chatPath}`;
|
|
323
|
+
};
|
|
324
|
+
|
|
325
|
+
// Auth/routing headers. A custom builder (multi-credential auth) wins; otherwise
|
|
326
|
+
// the bearer from the first accepted alias that is set: required → fail-hard
|
|
327
|
+
// naming every accepted var, optional → omitted when none set (a keyless server
|
|
328
|
+
// then receives no Authorization header).
|
|
329
|
+
const resolveHeaders = (spec: StandardProviderSpec, env: NodeJS.ProcessEnv, label: string): Record<string, string> => {
|
|
330
|
+
if (spec.headersFromEnv !== undefined) return spec.headersFromEnv(env);
|
|
331
|
+
const names = asList(spec.apiKeyVar);
|
|
332
|
+
if (names.length === 0) return {};
|
|
333
|
+
const apiKey = firstSet(env, names);
|
|
334
|
+
if (apiKey === undefined) {
|
|
335
|
+
if (spec.apiKeyRequired === true) throw new Error(spec.apiKeyMessage ?? `${label} provider: ${names.join(" or ")} must be set`);
|
|
336
|
+
return {};
|
|
337
|
+
}
|
|
338
|
+
return { Authorization: `Bearer ${apiKey}` };
|
|
339
|
+
};
|
|
340
|
+
|
|
341
|
+
// GET /v1/models probe. Yields the reported context window (llama-server nests
|
|
342
|
+
// it under `meta`, vLLM reports it top-level, cloud endpoints omit it) and the
|
|
343
|
+
// llama-server fingerprint — only llama-server rows carry a `meta` block, and
|
|
344
|
+
// llama-server is the backend whose chat-completions accepts a `grammar` field.
|
|
345
|
+
// Best-effort: any failure (unreachable, no field, non-2xx) degrades to
|
|
346
|
+
// { null, false } — a legitimate "unknown", not a swallowed contract violation.
|
|
347
|
+
type EndpointProbe = { nCtx: number | null; llamaServer: boolean; servedModel: string | null; failed: boolean };
|
|
348
|
+
|
|
349
|
+
// One probe failure must never decide capability (#34). Attempts/delay are
|
|
350
|
+
// operator knobs (PLURNK_PROVIDERS_PROBE_ATTEMPTS / _PROBE_DELAY, canonical
|
|
351
|
+
// 3 / 250ms in .env.defaults) — the full-sweep rule: no magic numbers in code.
|
|
352
|
+
|
|
353
|
+
const probeModels = async (chatUrl: string, headers: Record<string, string>, model: string, fetchTimeoutMs: number): Promise<EndpointProbe> => {
|
|
354
|
+
const modelsUrl = chatUrl.replace(/\/chat\/completions$/, "/models");
|
|
355
|
+
try {
|
|
356
|
+
const res = await fetch(modelsUrl, { headers, signal: AbortSignal.timeout(fetchTimeoutMs) });
|
|
357
|
+
if (!res.ok) return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
|
|
358
|
+
const data = (await res.json()) as { data?: Array<{ id?: string; n_ctx?: number; meta?: { n_ctx?: number } }> };
|
|
359
|
+
const rows = data.data ?? [];
|
|
360
|
+
const row = rows.find((r) => r.id === model) ?? rows[0];
|
|
361
|
+
const n = row?.meta?.n_ctx ?? row?.n_ctx;
|
|
362
|
+
// A clean 200 without the meta block is a CONFIRMED non-llama-server —
|
|
363
|
+
// a valid answer, not a failure; no retry.
|
|
364
|
+
return {
|
|
365
|
+
nCtx: typeof n === "number" && n > 0 ? n : null,
|
|
366
|
+
llamaServer: row?.meta !== undefined,
|
|
367
|
+
// #37: the backend's self-reported id — for a local llama-server the
|
|
368
|
+
// wire model is the ALIAS, but this row carries the real served name
|
|
369
|
+
// (the .gguf) the tokenizer seam maps. Absent when the probe read no row.
|
|
370
|
+
servedModel: typeof row?.id === "string" && row.id.length > 0 ? row.id : null,
|
|
371
|
+
failed: false,
|
|
372
|
+
};
|
|
373
|
+
} catch {
|
|
374
|
+
return { nCtx: null, llamaServer: false, servedModel: null, failed: true };
|
|
375
|
+
}
|
|
376
|
+
};
|
|
377
|
+
|
|
378
|
+
// Retry wrapper (#34): only FAILED probes (non-200 / thrown fetch / timeout)
|
|
379
|
+
// retry — a confirmed answer returns immediately. Exhaustion returns the last
|
|
380
|
+
// failed result; the CALLER decides what a still-unknown capability means.
|
|
381
|
+
const probeModelsRetrying = async (chatUrl: string, headers: Record<string, string>, model: string, fetchTimeoutMs: number, attempts: number, delayMs: number): Promise<EndpointProbe> => {
|
|
382
|
+
let probe: EndpointProbe = { nCtx: null, llamaServer: false, servedModel: null, failed: true };
|
|
383
|
+
for (let attempt = 0; attempt < attempts; attempt++) {
|
|
384
|
+
if (attempt > 0) await new Promise((r) => setTimeout(r, delayMs * 2 ** (attempt - 1)));
|
|
385
|
+
probe = await probeModels(chatUrl, headers, model, fetchTimeoutMs);
|
|
386
|
+
if (!probe.failed) return probe;
|
|
387
|
+
}
|
|
388
|
+
return probe;
|
|
389
|
+
};
|
|
390
|
+
|
|
391
|
+
// llama-server /props: total_slots (the valid id_slot range for slot pinning) +
|
|
392
|
+
// eos_token (the EOG the server renders as TEXT under --special, #539). One fetch,
|
|
393
|
+
// both facts. Only queried after the llama-server fingerprint confirms; same
|
|
394
|
+
// best-effort posture as the models probe.
|
|
395
|
+
const probeServerProps = async (chatUrl: string, headers: Record<string, string>, fetchTimeoutMs: number): Promise<{ slotCount: number | null; eosToken: string | null }> => {
|
|
396
|
+
const propsUrl = chatUrl.replace(/\/v1\/chat\/completions$/, "/props");
|
|
397
|
+
try {
|
|
398
|
+
const res = await fetch(propsUrl, { headers, signal: AbortSignal.timeout(fetchTimeoutMs) });
|
|
399
|
+
if (!res.ok) return { slotCount: null, eosToken: null };
|
|
400
|
+
const data = (await res.json()) as { total_slots?: number; eos_token?: string };
|
|
401
|
+
return {
|
|
402
|
+
slotCount: typeof data.total_slots === "number" && data.total_slots > 0 ? data.total_slots : null,
|
|
403
|
+
eosToken: typeof data.eos_token === "string" && data.eos_token.length > 0 ? data.eos_token : null,
|
|
404
|
+
};
|
|
405
|
+
} catch {
|
|
406
|
+
return { slotCount: null, eosToken: null };
|
|
407
|
+
}
|
|
408
|
+
};
|
|
409
|
+
|
|
410
|
+
// Returns a configured Provider, or null when `name` is not a standard
|
|
411
|
+
// provider (so the consumer falls through to dynamic import). Async because a
|
|
412
|
+
// probeNctx-enabled provider queries /v1/models at construction.
|
|
413
|
+
export const standardProviderFromEnv = async (name: string, env: NodeJS.ProcessEnv, model: string, baseUrlOverride?: string): Promise<Provider | null> => {
|
|
414
|
+
const spec = STANDARD_PROVIDERS[name];
|
|
415
|
+
if (spec === undefined) return null;
|
|
416
|
+
|
|
417
|
+
// The on-the-wire model id: a backend-required constant prefix (fireworks)
|
|
418
|
+
// prepended idempotently, so the operator's alias carries only the distinctive
|
|
419
|
+
// tail. This id is what the backend, the probe, AND the catalog key on.
|
|
420
|
+
const wireModel = spec.modelPrefix !== undefined && !model.startsWith(spec.modelPrefix)
|
|
421
|
+
? `${spec.modelPrefix}${model}`
|
|
422
|
+
: model;
|
|
423
|
+
|
|
424
|
+
const headers = resolveHeaders(spec, env, name);
|
|
425
|
+
|
|
426
|
+
// Tokenizer shed (mimetypes#44 landed): client-side families are GONE — exact
|
|
427
|
+
// counting lives in the mimetypes tokenizer seam + the tokenize() capability.
|
|
428
|
+
// A still-set family var fails hard with the migration pointer; countTokens is
|
|
429
|
+
// the chars/2 upper bound, surfaced so window math is never silently inexact.
|
|
430
|
+
// Namespace seizure (fail-forward): the debug knob moved into the owned
|
|
431
|
+
// PLURNK_PROVIDERS_ namespace (alias-scopable). The old name fails hard.
|
|
432
|
+
if (env.PLURNK_GBNF_DEBUG !== undefined && env.PLURNK_GBNF_DEBUG.length > 0) {
|
|
433
|
+
throw new Error(`${name} provider: PLURNK_GBNF_DEBUG was renamed to PLURNK_PROVIDERS_GBNF_DEBUG (alias-scopable as _<alias>); update the env`);
|
|
434
|
+
}
|
|
435
|
+
const staleTokenizer = env[spec.tokenizerEnvVar];
|
|
436
|
+
if (staleTokenizer !== undefined && staleTokenizer.length > 0) {
|
|
437
|
+
throw new Error(`${name} provider: ${spec.tokenizerEnvVar} was removed — exact counting moved to the @plurnk/plurnk-mimetypes tokenizer seam (mimetypes#44) and Provider.tokenize(); unset the var (countTokens is a chars/2 upper bound)`);
|
|
438
|
+
}
|
|
439
|
+
// Once per process per (code, message) — #40: construction-per-worker daemons
|
|
440
|
+
// repeated this until operators tuned warnings out.
|
|
441
|
+
emitWarningOnce(
|
|
442
|
+
`${name} provider: countTokens is a chars/2 upper bound — exact counts come from the mimetypes tokenizer seam or tokenize()`,
|
|
443
|
+
"PLURNK_TOKENIZER_HEURISTIC",
|
|
444
|
+
);
|
|
445
|
+
const url = resolveUrl(spec, env, name, baseUrlOverride);
|
|
446
|
+
const fetchTimeoutMs = parseRequiredInt(env.PLURNK_PROVIDERS_FETCH_TIMEOUT, "PLURNK_PROVIDERS_FETCH_TIMEOUT", name);
|
|
447
|
+
|
|
448
|
+
// The probe always runs for probeNctx specs — grammar capability must not
|
|
449
|
+
// hinge on whether the operator pinned PLURNK_PROVIDERS_CONTEXT_WINDOW. For
|
|
450
|
+
// contextWindow itself, explicit env still wins over the probed n_ctx.
|
|
451
|
+
let contextWindow = contextWindowFromEnv(env, name);
|
|
452
|
+
// Grammar shape: a static spec choice (e.g. fireworks → "response_format"),
|
|
453
|
+
// upgraded to "llamacpp" when the probe fingerprints a llama-server. Slot
|
|
454
|
+
// pinning is llama-server-only, so it keys on that same fingerprint. A spec
|
|
455
|
+
// can opt out of the fingerprint entirely (detectLlamaServer: false → plurnk)
|
|
456
|
+
// to read the window but stay a plain remote OpenAI server.
|
|
457
|
+
let grammarStyle: GrammarStyle = spec.grammarStyle ?? "none";
|
|
458
|
+
let supportsSlotPinning = false;
|
|
459
|
+
let slotCount: number | null = null;
|
|
460
|
+
let eosText: string | undefined;
|
|
461
|
+
let tokenizeUrl: string | undefined;
|
|
462
|
+
let servedModel: string | undefined;
|
|
463
|
+
let requiresMaxTokens: boolean | undefined;
|
|
464
|
+
let reasoningStyle = spec.reasoningStyle;
|
|
465
|
+
if (spec.probeNctx === true) {
|
|
466
|
+
// Operator pin (#34): "1" → llama-server capabilities WITHOUT trusting
|
|
467
|
+
// the probe; "0" → plain remote, skip the fingerprint; unset → detect.
|
|
468
|
+
// A plurnk-owned UNIVERSAL knob (alias-scopable: _<alias> suffix wins).
|
|
469
|
+
const pinRaw = env.PLURNK_PROVIDERS_LLAMA_SERVER;
|
|
470
|
+
if (pinRaw !== undefined && pinRaw !== "" && pinRaw !== "0" && pinRaw !== "1") {
|
|
471
|
+
throw new Error(`${name} provider: PLURNK_PROVIDERS_LLAMA_SERVER must be "1" (pin llama-server capabilities), "0" (force plain remote), or unset (auto-detect) (got "${pinRaw}")`);
|
|
472
|
+
}
|
|
473
|
+
const pin = pinRaw === undefined || pinRaw === "" ? null : pinRaw === "1";
|
|
474
|
+
const probeAttempts = parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_ATTEMPTS, "PLURNK_PROVIDERS_PROBE_ATTEMPTS", name);
|
|
475
|
+
const probe = await probeModelsRetrying(url, headers, wireModel, fetchTimeoutMs, probeAttempts, parseRequiredInt(env.PLURNK_PROVIDERS_PROBE_DELAY, "PLURNK_PROVIDERS_PROBE_DELAY", name));
|
|
476
|
+
contextWindow ??= probe.nCtx;
|
|
477
|
+
// #37: the backend's self-reported served id (from the same probe), so an
|
|
478
|
+
// alias-fronted local model resolves its exact tokenizer. Absent otherwise.
|
|
479
|
+
servedModel = probe.servedModel ?? undefined;
|
|
480
|
+
const isLlama = pin ?? (probe.llamaServer && spec.detectLlamaServer !== false);
|
|
481
|
+
if (isLlama && spec.detectLlamaServer !== false) {
|
|
482
|
+
grammarStyle = "llamacpp";
|
|
483
|
+
supportsSlotPinning = true;
|
|
484
|
+
const serverProps = await probeServerProps(url, headers, fetchTimeoutMs);
|
|
485
|
+
slotCount = serverProps.slotCount;
|
|
486
|
+
eosText = serverProps.eosToken ?? undefined;
|
|
487
|
+
// llama-server serves its NATIVE /tokenize at the root (like /props):
|
|
488
|
+
// the model's own vocab, exact — surfaced as the tokenize() capability.
|
|
489
|
+
tokenizeUrl = url.replace(/\/v1\/chat\/completions$/, "/tokenize");
|
|
490
|
+
// llama-server ignores `think` — its working reasoning toggle is the
|
|
491
|
+
// jinja chat_template_kwargs.enable_thinking, including the explicit
|
|
492
|
+
// FALSE at REASONING=off that grammar-constrained loops require (§13).
|
|
493
|
+
if (reasoningStyle === "think") reasoningStyle = "template";
|
|
494
|
+
// #43: llama-server honors n_predict to the CONTEXT WALL (providers#10)
|
|
495
|
+
// — no self-clamp. Surface the fact so a consumer can boot-refuse a
|
|
496
|
+
// local alias whose output envelope was never declared.
|
|
497
|
+
requiresMaxTokens = true;
|
|
498
|
+
} else if (pin === null && probe.failed && spec.detectLlamaServer !== false) {
|
|
499
|
+
// Detection exhausted its retries with NO answer: capability stays
|
|
500
|
+
// un-upgraded, but NEVER silently (#34) — rails going dark without a
|
|
501
|
+
// signal cost the consumer weeks of misattributed rambles.
|
|
502
|
+
emitWarningOnce(
|
|
503
|
+
`${name} provider: llama-server detection failed after ${probeAttempts} attempts — grammar transport stays OFF (grammarStyle "none"). If this endpoint IS a llama-server, pin PLURNK_PROVIDERS_LLAMA_SERVER=1 (or its _<alias> form).`,
|
|
504
|
+
"PLURNK_PROBE_FAILED",
|
|
505
|
+
);
|
|
506
|
+
}
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
// Vendored-snapshot FALLBACK (#19) — live always wins. contextWindow already
|
|
510
|
+
// preferred env then the live probe; the catalog only fills a still-null
|
|
511
|
+
// window for a known cloud model (groq/deepseek/mistral/…, which don't
|
|
512
|
+
// probe). A local llama-server model misses the catalog and keeps its
|
|
513
|
+
// probed n_ctx. Standard providers carry NO live pricing, so the catalog is
|
|
514
|
+
// the sole — never shadowing — cost source; per-1M USD → pico-USD/token (×1e6).
|
|
515
|
+
// A relay with a catalogContextLookup (bedrock) resolves its window via the
|
|
516
|
+
// underlying model's publisher and carries NO catalog cost (native rate ≠ relay
|
|
517
|
+
// rate, #22); everyone else keys the catalog directly on (name, wireModel).
|
|
518
|
+
const fallback = spec.catalogContextLookup === undefined ? lookup(name, wireModel) : undefined;
|
|
519
|
+
contextWindow ??= (spec.catalogContextLookup !== undefined ? spec.catalogContextLookup(wireModel) : fallback?.contextWindow) ?? null;
|
|
520
|
+
// Unresolved window (env, probe, catalog ALL missed) - two paths (#419 hybrid).
|
|
521
|
+
// A CLOUD provider (no probe) has no other window source, so an uncataloged,
|
|
522
|
+
// unpinned model is a config error we FAIL-HARD on (the #417 kimi case) rather
|
|
523
|
+
// than budget against a wrong number. A PROBING provider (openai/llama-server)
|
|
524
|
+
// DEGRADES to null instead: a probe blip or a box that doesn't report n_ctx must
|
|
525
|
+
// not crash construction (#34 robustness); the consumer treats null as "no cap",
|
|
526
|
+
// never a wrong CTX stand-in. Either way the unknown is SURFACED, never silent.
|
|
527
|
+
if (contextWindow === null) {
|
|
528
|
+
if (spec.probeNctx !== true) {
|
|
529
|
+
throw new Error(
|
|
530
|
+
`${name} provider: context window unresolved for "${wireModel}" - a cloud provider with no probe, absent from the @plurnk/plurnk-models catalog and unpinned. Pin PLURNK_PROVIDERS_CONTEXT_WINDOW (alias-scopable as _<alias>) or add the model to the catalog (#419)`,
|
|
531
|
+
);
|
|
532
|
+
}
|
|
533
|
+
emitWarningOnce(
|
|
534
|
+
`${name} provider: context window underivable for "${wireModel}" (probe returned no n_ctx, catalog miss) - contextWindow=null; set PLURNK_PROVIDERS_CONTEXT_WINDOW (alias-scopable as _<alias>) so window budgets are deliberate`,
|
|
535
|
+
"PLURNK_CONTEXT_UNKNOWN",
|
|
536
|
+
);
|
|
537
|
+
}
|
|
538
|
+
const cost = fallback?.cost;
|
|
539
|
+
const costFor = cost === undefined
|
|
540
|
+
? undefined
|
|
541
|
+
: (usage: ProviderUsage): number => computeCost(usage, {
|
|
542
|
+
input: cost.inputPer1M * 1e6,
|
|
543
|
+
output: cost.outputPer1M * 1e6,
|
|
544
|
+
cached: (cost.cacheReadPer1M ?? cost.inputPer1M) * 1e6,
|
|
545
|
+
});
|
|
546
|
+
|
|
547
|
+
// #507: completion cap — when the catalog reports a maxOutput, use
|
|
548
|
+
// min(maxOutput, percentage_amount) as an absolute reserve instead of the
|
|
549
|
+
// raw percentage. The 25% floor dramatically over-reserves on large-context
|
|
550
|
+
// models (gpt-4.1-mini: 1M ctx → 262K reserved, real cap 32K; claude-sonnet:
|
|
551
|
+
// 1M ctx → 250K reserved, real cap 128K). The min() preserves the floor on
|
|
552
|
+
// models where maxOutput > 25%*ctx (deepseek 1M/384K, llama 131K/131K).
|
|
553
|
+
// An operator absolute pin (COMPLETION_RESERVE=8192 not "25%") always wins.
|
|
554
|
+
const reasoning = reasoningFromEnv(env, name);
|
|
555
|
+
const { reasoningReserve: envReasoning, completionReserve: envCompletion } = envelopeFromEnv(env, name);
|
|
556
|
+
const completionReserve: ReserveSpec = (() => {
|
|
557
|
+
if ("tokens" in envCompletion) return envCompletion; // operator absolute always wins
|
|
558
|
+
if (fallback?.maxOutput === undefined || contextWindow === null) return envCompletion;
|
|
559
|
+
return { tokens: Math.min(fallback.maxOutput, Math.round(envCompletion.percent * contextWindow)) };
|
|
560
|
+
})();
|
|
561
|
+
// #568: reasoning reserve — exact REASONING_BUDGET > half-completion > env%.
|
|
562
|
+
// Using the budget directly prevents double-counting (half of 250K = 125K when
|
|
563
|
+
// the actual budget is 8K). Half-completion is the adaptive heuristic when
|
|
564
|
+
// reasoning=adaptive (budget=null). The env% is the last resort (only reached
|
|
565
|
+
// when contextWindow is null AND reasoning=adaptive).
|
|
566
|
+
const resolvedCompletion = resolveReserve(completionReserve, contextWindow);
|
|
567
|
+
const reasoningReserve: ReserveSpec = (() => {
|
|
568
|
+
if ("tokens" in envReasoning) return envReasoning; // operator absolute always wins
|
|
569
|
+
if (reasoning.budget !== null) return { tokens: reasoning.budget }; // exact budget when reasoning=on
|
|
570
|
+
if (resolvedCompletion !== null) return { tokens: Math.round(resolvedCompletion / 2) }; // half-completion for adaptive
|
|
571
|
+
return envReasoning; // last resort: env percentage
|
|
572
|
+
})();
|
|
573
|
+
|
|
574
|
+
return new OpenAICompatProvider({
|
|
575
|
+
model: wireModel,
|
|
576
|
+
url,
|
|
577
|
+
headers,
|
|
578
|
+
contextWindow,
|
|
579
|
+
fetchTimeoutMs,
|
|
580
|
+
reasoning,
|
|
581
|
+
temperature: parseRequiredFloat(env.PLURNK_PROVIDERS_TEMPERATURE, "PLURNK_PROVIDERS_TEMPERATURE", name, 0),
|
|
582
|
+
repeatPenalty: parseRequiredFloat(env.PLURNK_PROVIDERS_REPEAT_PENALTY, "PLURNK_PROVIDERS_REPEAT_PENALTY", name, 0),
|
|
583
|
+
frequencyPenalty: spec.suppressFrequencyPenalty === true ? 0 : parseRequiredFloat(env.PLURNK_PROVIDERS_FREQUENCY_PENALTY, "PLURNK_PROVIDERS_FREQUENCY_PENALTY", name, 0),
|
|
584
|
+
// #567: llama.cpp loop-breakers — optional, off by default (absent = the box's own
|
|
585
|
+
// default). The provider sends them only on the llamacpp path; values are operator config.
|
|
586
|
+
dryMultiplier: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_MULTIPLIER, "PLURNK_PROVIDERS_DRY_MULTIPLIER", name, 0) ?? undefined,
|
|
587
|
+
dryBase: parseOptionalFloat(env.PLURNK_PROVIDERS_DRY_BASE, "PLURNK_PROVIDERS_DRY_BASE", name, 0) ?? undefined,
|
|
588
|
+
dryAllowedLength: parseOptionalInt(env.PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH, "PLURNK_PROVIDERS_DRY_ALLOWED_LENGTH", name) ?? undefined,
|
|
589
|
+
repeatLastN: parseOptionalInt(env.PLURNK_PROVIDERS_REPEAT_LAST_N, "PLURNK_PROVIDERS_REPEAT_LAST_N", name) ?? undefined,
|
|
590
|
+
// #507: resolved envelope (catalog-capped completion above) + router-owned-tuning suppression (plurnk).
|
|
591
|
+
reasoningReserve,
|
|
592
|
+
completionReserve,
|
|
593
|
+
tuningFloors: spec.suppressTuningFloors !== true,
|
|
594
|
+
retryDelayMs: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_DELAY, "PLURNK_PROVIDERS_RETRY_DELAY", name),
|
|
595
|
+
retryAttempts: parseRequiredInt(env.PLURNK_PROVIDERS_RETRY_ATTEMPTS, "PLURNK_PROVIDERS_RETRY_ATTEMPTS", name),
|
|
596
|
+
reasoningStyle,
|
|
597
|
+
costFor,
|
|
598
|
+
source: providerSource(name),
|
|
599
|
+
grammarStyle,
|
|
600
|
+
// Optional debug toggle (off by default): validate a transported grammar
|
|
601
|
+
// locally and throw on invalid, without sending it to the model (§13).
|
|
602
|
+
gbnfDebug: env.PLURNK_PROVIDERS_GBNF_DEBUG !== undefined && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "" && env.PLURNK_PROVIDERS_GBNF_DEBUG !== "0",
|
|
603
|
+
// Opt-in data capture (#36), off by default, per-alias-scopable — universal
|
|
604
|
+
// across every standard provider (any backend that returns logprobs).
|
|
605
|
+
...dataCaptureFromEnv(env, name),
|
|
606
|
+
streaming: spec.streaming,
|
|
607
|
+
firstPartyMetadata: spec.firstPartyMetadata,
|
|
608
|
+
apiKeyRejectedMessage: spec.apiKeyRejectedMessage,
|
|
609
|
+
promptCacheKey: spec.promptCacheKey ?? true, // #518: default-on for standard providers (OpenAI-standard field, 6/6 backends verified accept it); per-spec opt-out below
|
|
610
|
+
balanceMetaKey: spec.balanceMetaKey,
|
|
611
|
+
supportsSlotPinning,
|
|
612
|
+
slotCount,
|
|
613
|
+
eosText,
|
|
614
|
+
tokenizeUrl,
|
|
615
|
+
servedModel,
|
|
616
|
+
requiresMaxTokens,
|
|
617
|
+
});
|
|
618
|
+
};
|