@agentproto/llm-endpoint 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -1
- package/dist/cli.mjs +97 -32
- package/dist/cli.mjs.map +1 -1
- package/dist/index.d.ts +78 -1
- package/dist/index.mjs +243 -33
- package/dist/index.mjs.map +1 -1
- package/package.json +2 -2
package/dist/index.d.ts
CHANGED
|
@@ -68,6 +68,74 @@ interface ModelPack {
|
|
|
68
68
|
toolsAllow?: string[];
|
|
69
69
|
}
|
|
70
70
|
|
|
71
|
+
/**
|
|
72
|
+
* Connectors — small, declarative probes for the local/LAN inference
|
|
73
|
+
* runtimes a named endpoint (`endpoints.ts`) commonly points at: LM Studio,
|
|
74
|
+
* Ollama, vLLM, llama-server, plus a generic OpenAI-compatible fallback for
|
|
75
|
+
* anything else. A connector answers three questions about a `baseUrl`: is
|
|
76
|
+
* this runtime actually running there (`probe`), what models does it have
|
|
77
|
+
* ({@link listModels}), and does it need special request handling
|
|
78
|
+
* ({@link ConnectorQuirks}).
|
|
79
|
+
*
|
|
80
|
+
* Used by `agentproto llm endpoints add/detect/test` and `doctor`'s
|
|
81
|
+
* inference-endpoints check — never by the proxy's own request routing
|
|
82
|
+
* (`index.ts`). In particular, LM Studio's GGUF Qwen builds reject a
|
|
83
|
+
* non-leading system message once tool defs are present; the generic fix
|
|
84
|
+
* for that lives in the adapter layer (a separate PR) — `quirks` is only a
|
|
85
|
+
* hook so callers can find out which connector needs it, not a duplicate of
|
|
86
|
+
* that fix.
|
|
87
|
+
*
|
|
88
|
+
* Deliberately dependency-free and side-effect-free at import time, like
|
|
89
|
+
* `endpoints.ts` and `packs.ts`.
|
|
90
|
+
*/
|
|
91
|
+
type ConnectorId = 'lmstudio' | 'ollama' | 'vllm' | 'llama-server' | 'openai-compatible';
|
|
92
|
+
declare const CONNECTOR_IDS: readonly ConnectorId[];
|
|
93
|
+
declare function isConnectorId(value: unknown): value is ConnectorId;
|
|
94
|
+
type ConnectorModelState = 'loaded' | 'not-loaded' | 'unknown';
|
|
95
|
+
/** One model as reported by a connector's `listModels`. `loadedCtx`/`maxCtx`
|
|
96
|
+
* are omitted (not `undefined`-filled) when the runtime's API doesn't
|
|
97
|
+
* expose them — e.g. Ollama's `/api/tags`/`/api/ps` carry no context size. */
|
|
98
|
+
interface ConnectorModel {
|
|
99
|
+
id: string;
|
|
100
|
+
loadedCtx?: number;
|
|
101
|
+
maxCtx?: number;
|
|
102
|
+
device?: string;
|
|
103
|
+
state: ConnectorModelState;
|
|
104
|
+
}
|
|
105
|
+
interface ConnectorQuirks {
|
|
106
|
+
/** This runtime's chat template can reject a system message that isn't
|
|
107
|
+
* the first message once tool definitions are present (observed on LM
|
|
108
|
+
* Studio GGUF Qwen builds) — callers that build the outbound message
|
|
109
|
+
* array should keep a single leading system message for it. */
|
|
110
|
+
requiresLeadingSystemMessage?: boolean;
|
|
111
|
+
}
|
|
112
|
+
interface Connector {
|
|
113
|
+
id: ConnectorId;
|
|
114
|
+
label: string;
|
|
115
|
+
/** Default local port this runtime listens on; `null` for the
|
|
116
|
+
* OpenAI-compatible fallback, which has no default port of its own and
|
|
117
|
+
* is never probed by default-port detection. */
|
|
118
|
+
defaultPort: number | null;
|
|
119
|
+
quirks: ConnectorQuirks;
|
|
120
|
+
/** Is this runtime actually answering at `baseUrl`? Never throws — a
|
|
121
|
+
* network error, timeout, or unexpected shape resolves `false`. */
|
|
122
|
+
probe(baseUrl: string, fetchImpl?: typeof fetch): Promise<boolean>;
|
|
123
|
+
/** Best-effort model listing. Never throws — a failed call resolves `[]`. */
|
|
124
|
+
listModels(baseUrl: string, fetchImpl?: typeof fetch): Promise<ConnectorModel[]>;
|
|
125
|
+
}
|
|
126
|
+
declare const CONNECTORS: Record<ConnectorId, Connector>;
|
|
127
|
+
declare function connectorById(id: string): Connector | undefined;
|
|
128
|
+
/** Identify which connector is serving `baseUrl`, trying the most specific
|
|
129
|
+
* probes first and falling back to the generic OpenAI-compatible one.
|
|
130
|
+
* Returns `null` only if even the fallback probe fails (nothing OpenAI-
|
|
131
|
+
* compatible is reachable there at all). */
|
|
132
|
+
declare function detectConnector(baseUrl: string, fetchImpl?: typeof fetch): Promise<ConnectorId | null>;
|
|
133
|
+
/** Default local ports probed by `agentproto llm endpoints detect`, one per
|
|
134
|
+
* runtime that has a real default port (`openai-compatible` doesn't — it's
|
|
135
|
+
* only ever selected explicitly, or as the fallback identity for a
|
|
136
|
+
* positive probe at some other port/runtime). */
|
|
137
|
+
declare const DEFAULT_LOCAL_PORTS: Readonly<Partial<Record<ConnectorId, number>>>;
|
|
138
|
+
|
|
71
139
|
/**
|
|
72
140
|
* Named OpenAI-compatible endpoints — N local/LAN model servers (Ollama,
|
|
73
141
|
* llama-server, vLLM, …) configured from a JSON file instead of one-off env
|
|
@@ -82,6 +150,7 @@ interface ModelPack {
|
|
|
82
150
|
* (src/index.ts) and the CLI (`agentproto llm endpoints …`), which must NOT
|
|
83
151
|
* pull in an HTTP server just to read/validate the config file.
|
|
84
152
|
*/
|
|
153
|
+
|
|
85
154
|
/** A JSON value — what `JSON.parse` can ever produce. */
|
|
86
155
|
type JsonValue = string | number | boolean | null | JsonValue[] | {
|
|
87
156
|
[key: string]: JsonValue;
|
|
@@ -105,6 +174,13 @@ interface EndpointConfig {
|
|
|
105
174
|
id: string;
|
|
106
175
|
kind: 'openai';
|
|
107
176
|
baseUrl: string;
|
|
177
|
+
/** Which local/LAN runtime this points at (`lmstudio`, `ollama`, `vllm`,
|
|
178
|
+
* `llama-server`) or the generic `openai-compatible` fallback — see
|
|
179
|
+
* `connectors.ts`. Absent on entries written before connectors existed;
|
|
180
|
+
* callers that need one (`endpoints test`, `doctor`) treat a missing
|
|
181
|
+
* value as `openai-compatible`. Never affects request routing — only
|
|
182
|
+
* which connector's `listModels`/`quirks` apply. */
|
|
183
|
+
connector?: ConnectorId;
|
|
108
184
|
/** Name of the env var holding the key — never the key itself. Absent ⇒
|
|
109
185
|
* the endpoint is always keyless (a private/LAN server with no auth). */
|
|
110
186
|
apiKeyEnv?: string;
|
|
@@ -349,6 +425,7 @@ type UpstreamTestResult = {
|
|
|
349
425
|
* (never forwarded) rather than tested.
|
|
350
426
|
*/
|
|
351
427
|
declare function testUpstream(provider: string): Promise<UpstreamTestResult>;
|
|
428
|
+
declare function adaptAnthropicToOpenAI(payload: any): void;
|
|
352
429
|
declare function stripThinkingFromAnthropicJson(jsonStr: string): string;
|
|
353
430
|
declare function resolveEmptyTurnRetries(): number;
|
|
354
431
|
/**
|
|
@@ -424,4 +501,4 @@ declare const server: node_http.Server<typeof IncomingMessage, typeof ServerResp
|
|
|
424
501
|
/** Démarre le proxy sur `port` (défaut : {@link PORT}). Renvoie le serveur en écoute. */
|
|
425
502
|
declare function start(port?: number): node_http.Server<typeof IncomingMessage, typeof ServerResponse>;
|
|
426
503
|
|
|
427
|
-
export { CANONICAL_UPSTREAMS, type ConfigurableUpstream, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, describeUpstreamStatus, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
|
|
504
|
+
export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, type ConfigurableUpstream, type Connector, type ConnectorId, type ConnectorModel, type ConnectorModelState, type ConnectorQuirks, DEFAULT_LOCAL_PORTS, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
|
package/dist/index.mjs
CHANGED
|
@@ -573,28 +573,28 @@ function flattenContent(content) {
|
|
|
573
573
|
return content.map((c) => c.text).join("");
|
|
574
574
|
}
|
|
575
575
|
function translateInputToMessages(input, instructions) {
|
|
576
|
-
const
|
|
577
|
-
if (instructions)
|
|
578
|
-
|
|
579
|
-
}
|
|
576
|
+
const sysParts = [];
|
|
577
|
+
if (instructions) sysParts.push(instructions);
|
|
578
|
+
const rest = [];
|
|
580
579
|
if (typeof input === "string") {
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
role: "tool",
|
|
593
|
-
|
|
594
|
-
content: item.output
|
|
595
|
-
});
|
|
580
|
+
rest.push({ role: "user", content: input });
|
|
581
|
+
} else {
|
|
582
|
+
for (const item of input) {
|
|
583
|
+
if (item.type === "message") {
|
|
584
|
+
if (item.role === "developer" || item.role === "system") {
|
|
585
|
+
const text = flattenContent(item.content);
|
|
586
|
+
if (text) sysParts.push(text);
|
|
587
|
+
continue;
|
|
588
|
+
}
|
|
589
|
+
rest.push({ role: item.role, content: flattenContent(item.content) });
|
|
590
|
+
} else if (item.type === "function_call_output") {
|
|
591
|
+
rest.push({ role: "tool", tool_call_id: item.call_id, content: item.output });
|
|
592
|
+
}
|
|
596
593
|
}
|
|
597
594
|
}
|
|
595
|
+
const messages = [];
|
|
596
|
+
if (sysParts.length) messages.push({ role: "system", content: sysParts.join("\n\n") });
|
|
597
|
+
messages.push(...rest);
|
|
598
598
|
return messages;
|
|
599
599
|
}
|
|
600
600
|
function translateToolToChatCompletions(tool) {
|
|
@@ -1534,6 +1534,159 @@ async function resumeIncompleteLocalQueueBatches() {
|
|
|
1534
1534
|
}
|
|
1535
1535
|
}
|
|
1536
1536
|
}
|
|
1537
|
+
|
|
1538
|
+
// src/connectors.ts
|
|
1539
|
+
var CONNECTOR_IDS = ["lmstudio", "ollama", "vllm", "llama-server", "openai-compatible"];
|
|
1540
|
+
function isConnectorId(value) {
|
|
1541
|
+
return typeof value === "string" && CONNECTOR_IDS.includes(value);
|
|
1542
|
+
}
|
|
1543
|
+
var PROBE_TIMEOUT_MS = 2500;
|
|
1544
|
+
async function fetchJsonSafe(url, fetchImpl, timeoutMs = PROBE_TIMEOUT_MS) {
|
|
1545
|
+
const controller = new AbortController();
|
|
1546
|
+
const timer = setTimeout(() => controller.abort(), timeoutMs);
|
|
1547
|
+
try {
|
|
1548
|
+
const res = await fetchImpl(url, { signal: controller.signal });
|
|
1549
|
+
if (!res.ok) return { ok: false };
|
|
1550
|
+
const body = await res.json().catch(() => null);
|
|
1551
|
+
return { ok: true, body };
|
|
1552
|
+
} catch {
|
|
1553
|
+
return { ok: false };
|
|
1554
|
+
} finally {
|
|
1555
|
+
clearTimeout(timer);
|
|
1556
|
+
}
|
|
1557
|
+
}
|
|
1558
|
+
function runtimeRoot(baseUrl) {
|
|
1559
|
+
return baseUrl.replace(/\/v1\/?$/, "");
|
|
1560
|
+
}
|
|
1561
|
+
function stripTrailingSlash(url) {
|
|
1562
|
+
return url.replace(/\/+$/, "");
|
|
1563
|
+
}
|
|
1564
|
+
var lmstudio = {
|
|
1565
|
+
id: "lmstudio",
|
|
1566
|
+
label: "LM Studio",
|
|
1567
|
+
defaultPort: 1234,
|
|
1568
|
+
quirks: { requiresLeadingSystemMessage: true },
|
|
1569
|
+
async probe(baseUrl, fetchImpl = fetch) {
|
|
1570
|
+
const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/v0/models`, fetchImpl);
|
|
1571
|
+
return result.ok && isRecord(result.body) && Array.isArray(result.body.data);
|
|
1572
|
+
},
|
|
1573
|
+
async listModels(baseUrl, fetchImpl = fetch) {
|
|
1574
|
+
const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/v0/models`, fetchImpl);
|
|
1575
|
+
if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
|
|
1576
|
+
return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({
|
|
1577
|
+
id: m.id,
|
|
1578
|
+
...typeof m.loaded_context_length === "number" ? { loadedCtx: m.loaded_context_length } : {},
|
|
1579
|
+
...typeof m.max_context_length === "number" ? { maxCtx: m.max_context_length } : {},
|
|
1580
|
+
state: m.state === "loaded" ? "loaded" : m.state === "not-loaded" ? "not-loaded" : "unknown"
|
|
1581
|
+
}));
|
|
1582
|
+
}
|
|
1583
|
+
};
|
|
1584
|
+
var ollama = {
|
|
1585
|
+
id: "ollama",
|
|
1586
|
+
label: "Ollama",
|
|
1587
|
+
defaultPort: 11434,
|
|
1588
|
+
quirks: {},
|
|
1589
|
+
async probe(baseUrl, fetchImpl = fetch) {
|
|
1590
|
+
const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/tags`, fetchImpl);
|
|
1591
|
+
return result.ok && isRecord(result.body) && Array.isArray(result.body.models);
|
|
1592
|
+
},
|
|
1593
|
+
async listModels(baseUrl, fetchImpl = fetch) {
|
|
1594
|
+
const root = runtimeRoot(baseUrl);
|
|
1595
|
+
const [tags, ps] = await Promise.all([fetchJsonSafe(`${root}/api/tags`, fetchImpl), fetchJsonSafe(`${root}/api/ps`, fetchImpl)]);
|
|
1596
|
+
const loadedNames = /* @__PURE__ */ new Set();
|
|
1597
|
+
if (ps.ok && isRecord(ps.body) && Array.isArray(ps.body.models)) {
|
|
1598
|
+
for (const m of ps.body.models) {
|
|
1599
|
+
if (isRecord(m) && typeof m.name === "string") loadedNames.add(m.name);
|
|
1600
|
+
}
|
|
1601
|
+
}
|
|
1602
|
+
if (!tags.ok || !isRecord(tags.body) || !Array.isArray(tags.body.models)) return [];
|
|
1603
|
+
return tags.body.models.filter(isRecord).filter((m) => typeof m.name === "string").map((m) => ({ id: m.name, state: loadedNames.has(m.name) ? "loaded" : "not-loaded" }));
|
|
1604
|
+
}
|
|
1605
|
+
};
|
|
1606
|
+
var llamaServer = {
|
|
1607
|
+
id: "llama-server",
|
|
1608
|
+
label: "llama-server",
|
|
1609
|
+
defaultPort: 8080,
|
|
1610
|
+
quirks: {},
|
|
1611
|
+
async probe(baseUrl, fetchImpl = fetch) {
|
|
1612
|
+
const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/props`, fetchImpl);
|
|
1613
|
+
return result.ok && isRecord(result.body) && ("default_generation_settings" in result.body || "model_path" in result.body);
|
|
1614
|
+
},
|
|
1615
|
+
async listModels(baseUrl, fetchImpl = fetch) {
|
|
1616
|
+
const [props, models] = await Promise.all([
|
|
1617
|
+
fetchJsonSafe(`${runtimeRoot(baseUrl)}/props`, fetchImpl),
|
|
1618
|
+
fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl)
|
|
1619
|
+
]);
|
|
1620
|
+
let ctx;
|
|
1621
|
+
if (props.ok && isRecord(props.body)) {
|
|
1622
|
+
const gen = props.body.default_generation_settings;
|
|
1623
|
+
if (isRecord(gen) && typeof gen.n_ctx === "number") ctx = gen.n_ctx;
|
|
1624
|
+
else if (typeof props.body.n_ctx === "number") ctx = props.body.n_ctx;
|
|
1625
|
+
}
|
|
1626
|
+
if (!models.ok || !isRecord(models.body) || !Array.isArray(models.body.data)) return [];
|
|
1627
|
+
return models.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({ id: m.id, ...ctx !== void 0 ? { loadedCtx: ctx, maxCtx: ctx } : {}, state: "loaded" }));
|
|
1628
|
+
}
|
|
1629
|
+
};
|
|
1630
|
+
var vllm = {
|
|
1631
|
+
id: "vllm",
|
|
1632
|
+
label: "vLLM",
|
|
1633
|
+
defaultPort: 8e3,
|
|
1634
|
+
quirks: {},
|
|
1635
|
+
async probe(baseUrl, fetchImpl = fetch) {
|
|
1636
|
+
const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
|
|
1637
|
+
if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return false;
|
|
1638
|
+
return result.body.data.some((m) => isRecord(m) && typeof m.max_model_len === "number");
|
|
1639
|
+
},
|
|
1640
|
+
async listModels(baseUrl, fetchImpl = fetch) {
|
|
1641
|
+
const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
|
|
1642
|
+
if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
|
|
1643
|
+
return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({
|
|
1644
|
+
id: m.id,
|
|
1645
|
+
...typeof m.max_model_len === "number" ? { loadedCtx: m.max_model_len, maxCtx: m.max_model_len } : {},
|
|
1646
|
+
state: "loaded"
|
|
1647
|
+
}));
|
|
1648
|
+
}
|
|
1649
|
+
};
|
|
1650
|
+
var openaiCompatible = {
|
|
1651
|
+
id: "openai-compatible",
|
|
1652
|
+
label: "OpenAI-compatible",
|
|
1653
|
+
defaultPort: null,
|
|
1654
|
+
quirks: {},
|
|
1655
|
+
async probe(baseUrl, fetchImpl = fetch) {
|
|
1656
|
+
const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
|
|
1657
|
+
return result.ok && isRecord(result.body) && Array.isArray(result.body.data);
|
|
1658
|
+
},
|
|
1659
|
+
async listModels(baseUrl, fetchImpl = fetch) {
|
|
1660
|
+
const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
|
|
1661
|
+
if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
|
|
1662
|
+
return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({ id: m.id, state: "unknown" }));
|
|
1663
|
+
}
|
|
1664
|
+
};
|
|
1665
|
+
var CONNECTORS = {
|
|
1666
|
+
lmstudio,
|
|
1667
|
+
ollama,
|
|
1668
|
+
vllm,
|
|
1669
|
+
"llama-server": llamaServer,
|
|
1670
|
+
"openai-compatible": openaiCompatible
|
|
1671
|
+
};
|
|
1672
|
+
function connectorById(id) {
|
|
1673
|
+
return isConnectorId(id) ? CONNECTORS[id] : void 0;
|
|
1674
|
+
}
|
|
1675
|
+
var DETECTION_ORDER = ["lmstudio", "ollama", "llama-server", "vllm", "openai-compatible"];
|
|
1676
|
+
async function detectConnector(baseUrl, fetchImpl = fetch) {
|
|
1677
|
+
for (const id of DETECTION_ORDER) {
|
|
1678
|
+
if (await CONNECTORS[id].probe(baseUrl, fetchImpl)) return id;
|
|
1679
|
+
}
|
|
1680
|
+
return null;
|
|
1681
|
+
}
|
|
1682
|
+
var DEFAULT_LOCAL_PORTS = {
|
|
1683
|
+
lmstudio: 1234,
|
|
1684
|
+
ollama: 11434,
|
|
1685
|
+
"llama-server": 8080,
|
|
1686
|
+
vllm: 8e3
|
|
1687
|
+
};
|
|
1688
|
+
|
|
1689
|
+
// src/endpoints.ts
|
|
1537
1690
|
var FORBIDDEN_DEFAULT_REQUEST_FIELD_KEYS = /* @__PURE__ */ new Set(["model", "messages", "stream", "tools", "input"]);
|
|
1538
1691
|
var RESERVED_ENDPOINT_IDS = /* @__PURE__ */ new Set(["forge"]);
|
|
1539
1692
|
function validateEndpointConfig(raw, where, errors) {
|
|
@@ -1541,7 +1694,7 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1541
1694
|
errors.push(`${where}: expected an object, got ${raw === null ? "null" : typeof raw}`);
|
|
1542
1695
|
return null;
|
|
1543
1696
|
}
|
|
1544
|
-
const { id, kind, baseUrl, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
|
|
1697
|
+
const { id, kind, baseUrl, connector, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
|
|
1545
1698
|
let ok = true;
|
|
1546
1699
|
if (typeof id !== "string" || id.length === 0) {
|
|
1547
1700
|
errors.push(`${where}.id: required non-empty string`);
|
|
@@ -1569,6 +1722,10 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1569
1722
|
ok = false;
|
|
1570
1723
|
}
|
|
1571
1724
|
}
|
|
1725
|
+
if (connector !== void 0 && !isConnectorId(connector)) {
|
|
1726
|
+
errors.push(`${where}.connector: must be one of ${CONNECTOR_IDS.join(", ")} when present (got ${JSON.stringify(connector)})`);
|
|
1727
|
+
ok = false;
|
|
1728
|
+
}
|
|
1572
1729
|
if (apiKeyEnv !== void 0 && (typeof apiKeyEnv !== "string" || apiKeyEnv.length === 0)) {
|
|
1573
1730
|
errors.push(`${where}.apiKeyEnv: must be a non-empty string when present`);
|
|
1574
1731
|
ok = false;
|
|
@@ -1605,6 +1762,7 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1605
1762
|
}
|
|
1606
1763
|
if (!ok || typeof id !== "string" || typeof baseUrl !== "string" || !validUrl) return null;
|
|
1607
1764
|
const built = { id, kind: "openai", baseUrl };
|
|
1765
|
+
if (isConnectorId(connector)) built.connector = connector;
|
|
1608
1766
|
if (typeof apiKeyEnv === "string") built.apiKeyEnv = apiKeyEnv;
|
|
1609
1767
|
if (builtFields) built.defaultRequestFields = builtFields;
|
|
1610
1768
|
if (builtTimeout) built.timeoutMs = builtTimeout;
|
|
@@ -1883,6 +2041,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
|
|
|
1883
2041
|
const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
|
|
1884
2042
|
return parseConfigurableUpstreamUrl(value, "nebius");
|
|
1885
2043
|
}
|
|
2044
|
+
var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
|
|
2045
|
+
function parseDeviceProvider(provider) {
|
|
2046
|
+
const m = DEVICE_PROVIDER_RE.exec(provider);
|
|
2047
|
+
return m ? { endpointId: m[1], device: m[2] } : null;
|
|
2048
|
+
}
|
|
2049
|
+
function resolveDeviceProviderSpec(device) {
|
|
2050
|
+
const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
|
|
2051
|
+
return {
|
|
2052
|
+
keyRequired: true,
|
|
2053
|
+
apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
|
|
2054
|
+
resolveUpstream: () => {
|
|
2055
|
+
const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
|
|
2056
|
+
if (!daemonUrl) return null;
|
|
2057
|
+
const base = parseUpstreamUrl(daemonUrl);
|
|
2058
|
+
if (!base) return null;
|
|
2059
|
+
return {
|
|
2060
|
+
...base,
|
|
2061
|
+
pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
|
|
2062
|
+
};
|
|
2063
|
+
},
|
|
2064
|
+
unavailableMessage
|
|
2065
|
+
};
|
|
2066
|
+
}
|
|
1886
2067
|
function getConfigurableProviderSpec(provider) {
|
|
1887
2068
|
const staticSpec = CONFIGURABLE_PROVIDERS[provider];
|
|
1888
2069
|
if (staticSpec) {
|
|
@@ -1904,6 +2085,8 @@ function getConfigurableProviderSpec(provider) {
|
|
|
1904
2085
|
unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
|
|
1905
2086
|
};
|
|
1906
2087
|
}
|
|
2088
|
+
const deviceRoute = parseDeviceProvider(provider);
|
|
2089
|
+
if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
|
|
1907
2090
|
return void 0;
|
|
1908
2091
|
}
|
|
1909
2092
|
function applyDefaultRequestFields(payload, defaults, skipKeys) {
|
|
@@ -2005,7 +2188,7 @@ async function probeFileEndpointModels(endpoint) {
|
|
|
2005
2188
|
return result;
|
|
2006
2189
|
}
|
|
2007
2190
|
function isKnownProvider(provider) {
|
|
2008
|
-
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
|
|
2191
|
+
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
|
|
2009
2192
|
}
|
|
2010
2193
|
function applyProviderOverride(target, providerOverride) {
|
|
2011
2194
|
const route = { provider: target.provider, model: target.model };
|
|
@@ -2022,7 +2205,7 @@ function parseAnyTransparentModel(model) {
|
|
|
2022
2205
|
const slashIdx = model.indexOf("/");
|
|
2023
2206
|
if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
|
|
2024
2207
|
const provider = model.slice(0, slashIdx);
|
|
2025
|
-
if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
|
|
2208
|
+
if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
|
|
2026
2209
|
return { provider, model: model.slice(slashIdx + 1) };
|
|
2027
2210
|
}
|
|
2028
2211
|
function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
|
|
@@ -2494,6 +2677,8 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2494
2677
|
return;
|
|
2495
2678
|
}
|
|
2496
2679
|
payload.model = resolvedTarget.model;
|
|
2680
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
2681
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
2497
2682
|
trimTools(payload, {
|
|
2498
2683
|
provider: resolvedTarget.provider,
|
|
2499
2684
|
queryTools: opts.queryTools,
|
|
@@ -2533,7 +2718,11 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2533
2718
|
method: "POST",
|
|
2534
2719
|
headers: {
|
|
2535
2720
|
"Content-Type": "application/json",
|
|
2536
|
-
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
|
|
2721
|
+
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
|
|
2722
|
+
// See the /v1/messages handler's identical header — the daemon's
|
|
2723
|
+
// /devices/:id/exec-stream 400s without it, always POST outer verb
|
|
2724
|
+
// notwithstanding.
|
|
2725
|
+
...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
|
|
2537
2726
|
}
|
|
2538
2727
|
};
|
|
2539
2728
|
const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
|
|
@@ -2559,21 +2748,28 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2559
2748
|
});
|
|
2560
2749
|
}
|
|
2561
2750
|
function adaptAnthropicToOpenAI(payload) {
|
|
2751
|
+
const sysParts = [];
|
|
2562
2752
|
if (payload.system != null) {
|
|
2563
|
-
let sysText = "";
|
|
2564
2753
|
if (typeof payload.system === "string") {
|
|
2565
|
-
|
|
2754
|
+
if (payload.system) sysParts.push(payload.system);
|
|
2566
2755
|
} else if (Array.isArray(payload.system)) {
|
|
2567
|
-
|
|
2568
|
-
|
|
2569
|
-
if (sysText) {
|
|
2570
|
-
if (!Array.isArray(payload.messages)) payload.messages = [];
|
|
2571
|
-
if (!payload.messages[0] || payload.messages[0].role !== "system") {
|
|
2572
|
-
payload.messages.unshift({ role: "system", content: sysText });
|
|
2573
|
-
}
|
|
2756
|
+
const text = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
|
|
2757
|
+
if (text) sysParts.push(text);
|
|
2574
2758
|
}
|
|
2575
2759
|
delete payload.system;
|
|
2576
2760
|
}
|
|
2761
|
+
if (Array.isArray(payload.messages)) {
|
|
2762
|
+
payload.messages = payload.messages.filter((m) => {
|
|
2763
|
+
if (m?.role !== "system" && m?.role !== "developer") return true;
|
|
2764
|
+
const text = typeof m.content === "string" ? m.content : Array.isArray(m.content) ? m.content.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n") : "";
|
|
2765
|
+
if (text) sysParts.push(text);
|
|
2766
|
+
return false;
|
|
2767
|
+
});
|
|
2768
|
+
}
|
|
2769
|
+
if (sysParts.length) {
|
|
2770
|
+
if (!Array.isArray(payload.messages)) payload.messages = [];
|
|
2771
|
+
payload.messages.unshift({ role: "system", content: sysParts.join("\n\n") });
|
|
2772
|
+
}
|
|
2577
2773
|
if (payload.tool_choice && typeof payload.tool_choice === "object") {
|
|
2578
2774
|
const tc = payload.tool_choice;
|
|
2579
2775
|
if (tc.type === "any") {
|
|
@@ -3363,6 +3559,8 @@ var server = createServer((req, res) => {
|
|
|
3363
3559
|
let cred;
|
|
3364
3560
|
let headers = { "Content-Type": "application/json" };
|
|
3365
3561
|
payload.model = resolvedTarget.model;
|
|
3562
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
3563
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
3366
3564
|
trimTools(payload, {
|
|
3367
3565
|
provider: resolvedTarget.provider,
|
|
3368
3566
|
queryTools,
|
|
@@ -3388,6 +3586,7 @@ var server = createServer((req, res) => {
|
|
|
3388
3586
|
cred = await resolveUpstreamCredential(resolvedTarget.provider);
|
|
3389
3587
|
targetApiKey = cred?.value ?? "";
|
|
3390
3588
|
if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
|
|
3589
|
+
if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
|
|
3391
3590
|
const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
|
|
3392
3591
|
adaptAnthropicToOpenAI(payload);
|
|
3393
3592
|
applyDefaultRequestFields(
|
|
@@ -3584,6 +3783,17 @@ var server = createServer((req, res) => {
|
|
|
3584
3783
|
);
|
|
3585
3784
|
}
|
|
3586
3785
|
});
|
|
3786
|
+
if (status < 200 || status >= 300) {
|
|
3787
|
+
let errBody = "";
|
|
3788
|
+
proxyRes.on("data", (c) => {
|
|
3789
|
+
if (errBody.length < 300) errBody += c.toString("utf8");
|
|
3790
|
+
});
|
|
3791
|
+
proxyRes.on("end", () => {
|
|
3792
|
+
console.error(
|
|
3793
|
+
`[Proxy] upstream error: ${resolvedTarget.provider}:${resolvedTarget.model} status=${status} body=${JSON.stringify(errBody.slice(0, 300))}`
|
|
3794
|
+
);
|
|
3795
|
+
});
|
|
3796
|
+
}
|
|
3587
3797
|
const contentType = proxyRes.headers["content-type"] || "";
|
|
3588
3798
|
const isStreaming = payload.stream === true && /text\/event-stream/i.test(contentType);
|
|
3589
3799
|
const retryEmptyTurn = () => {
|
|
@@ -3709,6 +3919,6 @@ function start(port = PORT) {
|
|
|
3709
3919
|
});
|
|
3710
3920
|
}
|
|
3711
3921
|
|
|
3712
|
-
export { CANONICAL_UPSTREAMS, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, describeUpstreamStatus, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
|
|
3922
|
+
export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, DEFAULT_LOCAL_PORTS, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
|
|
3713
3923
|
//# sourceMappingURL=index.mjs.map
|
|
3714
3924
|
//# sourceMappingURL=index.mjs.map
|