@agentproto/llm-endpoint 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,427 @@
1
+ import * as node_http from 'node:http';
2
+ import { IncomingMessage, ServerResponse } from 'http';
3
+
4
+ /**
5
+ * Capability tier of a route, coarse enough to map onto the four Claude
6
+ * families. Used by {@link toAnthropicStyle} to pick a family prefix for the
7
+ * generated Anthropic-shaped id (see {@link TIER_TO_FAMILY}).
8
+ */
9
+ type ModelTier = 'extra-high' | 'high' | 'medium' | 'small';
10
+ interface ModelRoute {
11
+ provider: string;
12
+ model: string;
13
+ /**
14
+ * Optional Claude-shaped compatibility alias. Committed packs never hardcode
15
+ * this — it is populated either by a local pack (loaded from
16
+ * `packs.local.json`) or on demand by {@link toAnthropicStyle} when a request
17
+ * opts into the Anthropic-style format. Committed packs use
18
+ * provider-transparent model IDs and never pretend to be real Claude models.
19
+ */
20
+ equivalentClaudeName?: string;
21
+ /**
22
+ * Coarse capability tier. Only consumed by {@link toAnthropicStyle} to select
23
+ * a Claude family for the generated alias; ignored on every routing path.
24
+ */
25
+ tier?: ModelTier;
26
+ /**
27
+ * Verified context window of the upstream route, in tokens. Surfaced in the
28
+ * /v1/models metadata when present; never guessed — leave absent when the
29
+ * limit for this specific route is not verified.
30
+ */
31
+ contextWindow?: number;
32
+ /**
33
+ * Verified max output tokens for the upstream route, in tokens. Same
34
+ * honesty rule as {@link ModelRoute.contextWindow}.
35
+ */
36
+ maxOutputTokens?: number;
37
+ }
38
+ /**
39
+ * A pack is a curated set of model routes. Packs solve the collision problem
40
+ * when multiple providers share the same display name, and they let users opt
41
+ * into local compatibility aliases for clients that only speak the Anthropic
42
+ * model namespace.
43
+ *
44
+ * Selection methods:
45
+ * - URL path: /v1/{packId}/messages → pack = {packId}
46
+ * - Query param: /v1/messages?pack={packId}
47
+ * - Header: X-Proxy-Pack: {packId}
48
+ * - Default: no pack specified → uses 'default' pack
49
+ */
50
+ interface ModelPack {
51
+ id: string;
52
+ label: string;
53
+ description: string;
54
+ models: Record<string, ModelRoute>;
55
+ /**
56
+ * Exclude-list of tool-name patterns applied to every request routed through
57
+ * this pack (wildcards allowed, e.g. "mcp__*"). Cuts upstream prefill cost
58
+ * when a client (e.g. Claude Desktop) sends hundreds of tool definitions.
59
+ * Applied after per-request header/query trims so an explicit client
60
+ * X-Proxy-Tools allow-list still wins.
61
+ */
62
+ toolsExclude?: string[];
63
+ /**
64
+ * Allow-list of tool-name patterns applied to every request routed through
65
+ * this pack (wildcards allowed). Kept only when no toolsAllow is set on the
66
+ * request itself; a request-level ?tools=/X-Proxy-Tools always wins.
67
+ */
68
+ toolsAllow?: string[];
69
+ }
70
+
71
+ /**
72
+ * Named OpenAI-compatible endpoints — N local/LAN model servers (Ollama,
73
+ * llama-server, vLLM, …) configured from a JSON file instead of one-off env
74
+ * vars. Mirrors the field names of `openagentik/router`'s `providers[]`
75
+ * schema (`kind`, `baseUrl`, `apiKeyEnv`, `defaultRequestFields`, `timeoutMs`)
76
+ * so a config is portable between the two — see
77
+ * `projects/openagentik/router/router.example.yaml` and
78
+ * `packages/core/src/config/schema.ts` in that repo.
79
+ *
80
+ * Deliberately dependency-free (no zod, no fs writes) and side-effect-free at
81
+ * import time, like packs.ts — this module is imported by both the proxy
82
+ * (src/index.ts) and the CLI (`agentproto llm endpoints …`), which must NOT
83
+ * pull in an HTTP server just to read/validate the config file.
84
+ */
85
+ /** A JSON value — what `JSON.parse` can ever produce. */
86
+ type JsonValue = string | number | boolean | null | JsonValue[] | {
87
+ [key: string]: JsonValue;
88
+ };
89
+ /** Arbitrary top-level OpenAI-compatible request fields, merged UNDER the
90
+ * client's own request (see index.ts's applyDefaultRequestFields) —
91
+ * everything from vLLM's `chat_template_kwargs` extension to a plain field
92
+ * like LM Studio's `reasoning_effort`. See README's "Adding an
93
+ * OpenAI-compatible upstream provider" section. */
94
+ type EndpointDefaultRequestFields = Record<string, JsonValue>;
95
+ interface EndpointTimeoutConfig {
96
+ /** Applied as the outbound socket timeout (time-to-first-byte). */
97
+ firstTokenMs?: number;
98
+ }
99
+ /**
100
+ * One configured endpoint. `kind` is carried (mirroring the router schema)
101
+ * but only `"openai"` is meaningful here — every routable endpoint in
102
+ * llm-endpoint speaks the OpenAI chat/completions wire shape, same as forge.
103
+ */
104
+ interface EndpointConfig {
105
+ id: string;
106
+ kind: 'openai';
107
+ baseUrl: string;
108
+ /** Name of the env var holding the key — never the key itself. Absent ⇒
109
+ * the endpoint is always keyless (a private/LAN server with no auth). */
110
+ apiKeyEnv?: string;
111
+ defaultRequestFields?: EndpointDefaultRequestFields;
112
+ timeoutMs?: EndpointTimeoutConfig;
113
+ }
114
+ interface ParsedEndpointsResult {
115
+ endpoints: EndpointConfig[];
116
+ errors: string[];
117
+ }
118
+ /**
119
+ * Validate the `{ endpoints: [...] }` envelope. Checks each entry's shape,
120
+ * plus duplicate ids — including against the implicit `forge` id, which a
121
+ * file entry may never reuse.
122
+ */
123
+ declare function parseEndpointsConfig(parsed: unknown): ParsedEndpointsResult;
124
+ /** `~/.agentproto/llm-endpoints.json`, overridable via `LLM_ENDPOINT_ENDPOINTS_FILE`. */
125
+ declare function resolveEndpointsFilePath(): string;
126
+ interface EndpointsFileLoad extends ParsedEndpointsResult {
127
+ path: string;
128
+ }
129
+ /**
130
+ * Read + validate the endpoints file from disk. A missing file is NOT an
131
+ * error (fail-soft, like packs.local.json) — it means "no named endpoints
132
+ * configured", which is the normal case. Malformed JSON or a shape error IS
133
+ * an error, and yields an empty endpoint list (fail closed on that file
134
+ * rather than partially trusting it).
135
+ */
136
+ declare function readEndpointsFromDisk(path?: string): EndpointsFileLoad;
137
+ /** Cached, validated endpoint list — re-read from disk on first call after
138
+ * boot or after {@link resetConfiguredEndpointsCache}. A file-load error is
139
+ * logged once and treated as "no endpoints configured" (fail-soft), mirroring
140
+ * index.ts's getLocalPacks(). */
141
+ declare function getConfiguredEndpoints(): EndpointConfig[];
142
+ /** Drop the cached endpoint list so the next call re-reads the file. */
143
+ declare function resetConfiguredEndpointsCache(): void;
144
+ /** A resolved OpenAI-compatible upstream — host/port/protocol/path-prefix
145
+ * parsed from a base URL. Shared shape for forge, nebius, and every
146
+ * file-configured endpoint (see index.ts's ConfigurableProviderSpec). */
147
+ interface ConfigurableUpstream {
148
+ hostname: string;
149
+ port: number;
150
+ protocol: 'http' | 'https';
151
+ /** URL pathname with any trailing slash stripped, e.g. "/v1" or "" for root. */
152
+ pathPrefix: string;
153
+ }
154
+
155
+ interface ToolTrimOptions {
156
+ provider: string;
157
+ queryTools: string | null;
158
+ queryNoTools: string | null;
159
+ headerTools: string | null;
160
+ headerNoTools: string | null;
161
+ headerExcludeTools: string | null;
162
+ /** Pack-level exclude patterns (pack.toolsExclude) — applied last. */
163
+ packToolsExclude?: string[];
164
+ /** Pack-level allow patterns (pack.toolsAllow) — applied last. */
165
+ packToolsAllow?: string[];
166
+ }
167
+ declare function trimTools(payload: any, opts: ToolTrimOptions): void;
168
+ interface ProviderKeys {
169
+ anthropic: string;
170
+ moonshot: string;
171
+ openrouter: string;
172
+ requesty: string;
173
+ zai: string;
174
+ groq: string;
175
+ xai: string;
176
+ openai: string;
177
+ }
178
+
179
+ /** Back-compat name — see {@link ConfigurableUpstream}. */
180
+ type ForgeUpstream = ConfigurableUpstream;
181
+ /**
182
+ * Parse `FORGE_BASE_URL` into a routable upstream. Unlike every other
183
+ * provider here (fixed https hostname, implicit :443), forge points at a
184
+ * private, self-hosted OpenAI-compatible server (vLLM `--enable-lora`) so the
185
+ * scheme, host, port, and path prefix are all env-configured — e.g.
186
+ * `http://10.0.10.20:8000/v1`. Returns null when unset or malformed, which
187
+ * callers treat as "forge provider not configured" (a 400, never a crash).
188
+ */
189
+ declare function resolveForgeBaseUrl(raw?: string | undefined): ConfigurableUpstream | null;
190
+ /**
191
+ * Resolve the Nebius AI Studio upstream. Defaults to the public
192
+ * `https://api.studio.nebius.com/v1` endpoint — nebius works with just
193
+ * `NEBIUS_API_KEY` set. `NEBIUS_BASE_URL` overrides it, e.g. to point at a
194
+ * Nebius *dedicated* endpoint on a different host, with no code change.
195
+ * Returns null only when the override is set but malformed — an unset
196
+ * `NEBIUS_BASE_URL` is the normal case and resolves to the default.
197
+ */
198
+ declare function resolveNebiusBaseUrl(raw?: string | undefined): ConfigurableUpstream | null;
199
+ interface ModelRouteContext {
200
+ activePack: ModelPack;
201
+ queryModelCode: string | null;
202
+ queryProvider: string | null;
203
+ forcedAliasCode: string | null;
204
+ /**
205
+ * When true, the Messages path is allowed to match local-pack
206
+ * `equivalentClaudeName` aliases. The OpenAI chat/completions and Responses
207
+ * surfaces are always transparent and never resolve default alias packs.
208
+ */
209
+ allowAliases: boolean;
210
+ /**
211
+ * When true, the active pack has been run through {@link toAnthropicStyle}
212
+ * for this request, so its `equivalentClaudeName` aliases are resolvable even
213
+ * though it is an official (non-local) pack.
214
+ */
215
+ anthropicFormat?: boolean;
216
+ }
217
+ /**
218
+ * Resolves the upstream provider/model for a request.
219
+ *
220
+ * Messages path (`allowAliases: true`):
221
+ * - X-Proxy-Model-Alias header / PROXY_MODEL_ALIAS env forces a pack code.
222
+ * - Explicit ?m=<code> selects a pack code.
223
+ * - Local packs may define `equivalentClaudeName` aliases.
224
+ * - Public pack codes match transparent model IDs.
225
+ * - `provider/model` references are parsed transparently.
226
+ * - `?p=<provider>` with a bare model id routes to that provider.
227
+ *
228
+ * Chat/Responses paths (`allowAliases: false`):
229
+ * - Only `provider/model` references or `?p=<provider>` with a bare model id.
230
+ * - Pack aliases and regex fallbacks are intentionally not used.
231
+ *
232
+ * Throws a plain Error for unknown explicit codes/providers.
233
+ */
234
+ declare function resolveModelRoute(payload: {
235
+ model?: string;
236
+ }, ctx: ModelRouteContext, localPacks?: Record<string, ModelPack>): {
237
+ provider: string;
238
+ model: string;
239
+ };
240
+ type UpstreamAuthMethod = 'api-key' | 'oauth-bearer';
241
+ /** A resolved upstream credential: the secret plus the header shape to use. */
242
+ interface UpstreamCredential {
243
+ value: string;
244
+ method: UpstreamAuthMethod;
245
+ }
246
+ /**
247
+ * Resolve the outbound credential for `provider`. Mirrors getApiKey()'s
248
+ * provider-string contract but returns the credential AND the header method.
249
+ *
250
+ * - `LLM_ENDPOINT_PROFILE_<P>` set → resolve the named profile:
251
+ * • missing / disabled → undefined (caller 401s, as today)
252
+ * • source-backed (no credRef) → undefined + a clear follow-up log
253
+ * (self-refresh not supported here yet)
254
+ * • credentialRef-backed → { value, method: profile.method }
255
+ * • keychain read returns null (credential absent / present-but-unreadable
256
+ * on a supported host, e.g. a locked Keychain) → undefined (fail-closed;
257
+ * caller 401s — we do NOT silently downgrade a mapped profile to the env
258
+ * key)
259
+ * • keychain read throws (platform-unsupported backend, e.g. non-darwin
260
+ * host) → env-key fallback + a one-time log; never crashes the request
261
+ * - no mapping → env-key path: method is derived from the credential's own
262
+ * shape, not hardcoded — an anthropic env key that is actually a
263
+ * subscription OAT (`sk-ant-oat…`, e.g. injected by the runtime's
264
+ * billing-auth resolver for a modelDerivedApiKey adapter with no
265
+ * `authSubscription`, such as pi) resolves to "oauth-bearer" so
266
+ * {@link buildUpstreamAuthHeaders} sends it as `Authorization: Bearer`
267
+ * instead of `x-api-key` — Anthropic hard-401s an OAT presented as
268
+ * `x-api-key` ("invalid x-api-key"). Any other anthropic key, and every
269
+ * other provider, keeps "api-key" exactly as before.
270
+ */
271
+ declare function resolveUpstreamCredential(provider: string): Promise<UpstreamCredential | undefined>;
272
+ /**
273
+ * Single source of truth for the outbound upstream-auth header shape. Given a
274
+ * resolved credential, returns the auth-related headers to merge into the
275
+ * request. For an `api-key` credential each provider keeps its existing header
276
+ * exactly; for `oauth-bearer` (only meaningful on the anthropic upstream) it
277
+ * emits the Bearer + anthropic-version + anthropic-beta triple.
278
+ *
279
+ * Fail-closed: an `oauth-bearer` credential (e.g. a Claude subscription OAT)
280
+ * resolved for a NON-anthropic provider returns `null` — the token is never
281
+ * emitted to a third-party upstream. Callers MUST treat `null` as a hard 401
282
+ * and send no request.
283
+ */
284
+ declare function buildUpstreamAuthHeaders(provider: string, cred: UpstreamCredential): Record<string, string> | null;
285
+ /**
286
+ * Fail-closed guard for the OpenAI-compatible surfaces (/v1/responses,
287
+ * /v1/chat/completions). Those surfaces are ALWAYS non-anthropic
288
+ * (getChatCompletionsEndpoint returns null for anthropic) and forward the
289
+ * resolved credential as `Authorization: Bearer <value>` — so only an
290
+ * `api-key` credential may be used. A subscription/oauth credential (e.g. a
291
+ * Claude OAT) must be rejected, never leaked to a third-party host.
292
+ *
293
+ * Returns true when the credential may proceed on this surface. An absent
294
+ * credential returns true here (the caller's missing-key check 401s it).
295
+ */
296
+ declare function isCredentialAllowedOnOpenAiSurface(cred: UpstreamCredential | undefined): boolean;
297
+ declare const CANONICAL_UPSTREAMS: (keyof ProviderKeys)[];
298
+ /** Narrow an arbitrary provider string to one of the canonical upstreams. */
299
+ declare function isCanonicalUpstream(provider: string): provider is keyof ProviderKeys;
300
+ /** How a credential WOULD resolve for an upstream (never leaks the value). */
301
+ type UpstreamSource = 'profile' | 'env' | 'none';
302
+ /**
303
+ * Non-secret status of one upstream's outbound credential:
304
+ * - `linkedProfile`: the `LLM_ENDPOINT_PROFILE_<P>` profile id, or null.
305
+ * - `source`: how a credential would resolve — a mapped profile ("profile",
306
+ * even if that profile is missing/disabled), else a non-empty per-provider
307
+ * env key ("env"), else "none".
308
+ * - `method`: the outbound auth shape — the mapped profile's method, or
309
+ * "api-key" for the env path, or null when nothing is configured.
310
+ * - `present`: whether a credential actually resolves. Known for free for the
311
+ * env ("env" ⇒ true) and none ("none" ⇒ false) sources; for a profile source
312
+ * it is `null` unless `?probe=1` was requested, because confirming it reads
313
+ * the OS keychain (one read per mapped profile).
314
+ */
315
+ interface UpstreamStatus {
316
+ provider: string;
317
+ linkedProfile: string | null;
318
+ source: UpstreamSource;
319
+ method: UpstreamAuthMethod | null;
320
+ present: boolean | null;
321
+ }
322
+ /**
323
+ * Describe one upstream's credential status without returning a secret. The
324
+ * default (mapping-only) view is cheap — an env-var read plus, for a mapped
325
+ * provider, one auth-profiles.json read for the method. `present` for a profile
326
+ * source is filled only when `opts.probe` is set (it costs a keychain read).
327
+ */
328
+ declare function describeUpstreamStatus(provider: string, opts: {
329
+ probe: boolean;
330
+ }): Promise<UpstreamStatus>;
331
+ /** Status for all 8 canonical upstreams, in ProviderKeys order. */
332
+ declare function collectUpstreamStatuses(opts: {
333
+ probe: boolean;
334
+ }): Promise<UpstreamStatus[]>;
335
+ /** The result of a per-upstream live test — a verdict, or "no cheap probe". */
336
+ type UpstreamTestResult = {
337
+ ok: boolean;
338
+ status: number;
339
+ detail: string;
340
+ } | {
341
+ ok: null;
342
+ reason: 'no-probe';
343
+ };
344
+ /**
345
+ * Run the cheapest authenticated call for `provider` and return a verdict.
346
+ * Resolves the credential through the same precedence as a real request, then
347
+ * uses buildUpstreamAuthHeaders for the exact per-provider header shape — so an
348
+ * oauth-bearer credential mis-mapped to a non-anthropic upstream is refused
349
+ * (never forwarded) rather than tested.
350
+ */
351
+ declare function testUpstream(provider: string): Promise<UpstreamTestResult>;
352
+ declare function stripThinkingFromAnthropicJson(jsonStr: string): string;
353
+ declare function resolveEmptyTurnRetries(): number;
354
+ /**
355
+ * True when an Anthropic Messages JSON body is a "tour vide": ended normally
356
+ * (`end_turn`) but carries no client-visible content once thinking is stripped.
357
+ */
358
+ declare function isEmptyAnthropicTurn(jsonStr: string): boolean;
359
+ declare function openaiJsonToAnthropic(jsonStr: string): string;
360
+ /** Parse a comma-separated token allow-list into a Set (trimmed, non-empty). */
361
+ declare function parseAccessTokens(raw: string | undefined): Set<string>;
362
+ /**
363
+ * Extract a presented access token from `Authorization: Bearer <t>` or the
364
+ * `X-Proxy-Access: <t>` header (so a client that reserves Authorization for
365
+ * something else can still pass one). Returns null when neither is present.
366
+ */
367
+ declare function extractInboundToken(headers: IncomingMessage['headers']): string | null;
368
+ /**
369
+ * True when the request may proceed: gate disabled (empty allow-list), or the
370
+ * presented token is in the allow-list. A high-entropy random token makes the
371
+ * Set membership check's non-constant time immaterial.
372
+ */
373
+ declare function isAuthorized(headers: IncomingMessage['headers'], tokens: Set<string>): boolean;
374
+ /**
375
+ * Extract a presented edge token from the `X-Edge-Auth: <t>` header. Returns
376
+ * null when absent or blank.
377
+ */
378
+ declare function extractEdgeToken(headers: IncomingMessage['headers']): string | null;
379
+ /**
380
+ * True when the request may proceed through the edge/WAF layer: layer
381
+ * disabled (empty allow-list), or the presented `X-Edge-Auth` token is in the
382
+ * allow-list.
383
+ */
384
+ declare function isEdgeAuthorized(headers: IncomingMessage['headers'], edgeTokensSet: Set<string>): boolean;
385
+ /**
386
+ * Builds a Cloudflare custom-rule (wirefilter) expression that BLOCKS any
387
+ * request lacking a valid token, for the header (`authorization` or
388
+ * `x-edge-auth`) matching the layer being enforced at the edge. Intended to
389
+ * be pasted into a Cloudflare custom rule so the token check happens before
390
+ * traffic ever reaches this process — the `edgeTokens()`/`isEdgeAuthorized`
391
+ * pair above is the same policy enforced in-process as a fallback.
392
+ *
393
+ * CORS preflight (`OPTIONS`) is always allowed through so the browser's
394
+ * preflight — which cannot carry the token — isn't blocked.
395
+ */
396
+ declare function buildWafRuleExpression(opts: {
397
+ host?: string;
398
+ tokens: string[];
399
+ header: 'authorization' | 'x-edge-auth';
400
+ }): string;
401
+ /**
402
+ * Normalise an incoming request path so a redundant `/v1` an Anthropic client
403
+ * appends to a base URL that already ends in `/v1` (optionally with a pack
404
+ * segment) doesn't break routing.
405
+ *
406
+ * /v1/v1/messages → /v1/messages
407
+ * /v1/stealth-requesty/v1/messages → /v1/stealth-requesty/messages
408
+ * /v1/stealth-requesty/v1/models → /v1/stealth-requesty/models
409
+ *
410
+ * Without the pack-segment case the extra `/v1` breaks the `/v1/<pack>/<verb>`
411
+ * match, the pack is silently lost, and the request falls back to the default
412
+ * pack — the "no usable models" / "Unable to resolve model" failure a client
413
+ * configured with `base_url = .../v1/<pack>` hits.
414
+ */
415
+ declare function normalizeProxyPath(pathname: string): string;
416
+ /**
417
+ * True only for the DEFAULT model-discovery path — `/v1/models` or `/models`,
418
+ * never a pack-scoped list (`/v1/<pack>/models`). Lets opt-in public discovery
419
+ * expose the default codename list without disclosing pack names.
420
+ */
421
+ declare function isPublicModelListPath(pathname: string): boolean;
422
+ declare const server: node_http.Server<typeof IncomingMessage, typeof ServerResponse>;
423
+
424
+ /** Démarre le proxy sur `port` (défaut : {@link PORT}). Renvoie le serveur en écoute. */
425
+ declare function start(port?: number): node_http.Server<typeof IncomingMessage, typeof ServerResponse>;
426
+
427
+ export { CANONICAL_UPSTREAMS, type ConfigurableUpstream, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, describeUpstreamStatus, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };