@agentproto/llm-endpoint 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -1
- package/dist/cli.mjs +97 -32
- package/dist/cli.mjs.map +1 -1
- package/dist/index.d.ts +78 -1
- package/dist/index.mjs +243 -33
- package/dist/index.mjs.map +1 -1
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -150,11 +150,12 @@ configured from a JSON file instead of a pair of env vars per server:
|
|
|
150
150
|
"defaultRequestFields": { "chat_template_kwargs": { "enable_thinking": false } },
|
|
151
151
|
"timeoutMs": { "firstTokenMs": 180000 }
|
|
152
152
|
},
|
|
153
|
-
{ "id": "ollama", "kind": "openai", "baseUrl": "http://192.168.1.20:11434/v1" },
|
|
153
|
+
{ "id": "ollama", "kind": "openai", "baseUrl": "http://192.168.1.20:11434/v1", "connector": "ollama" },
|
|
154
154
|
{
|
|
155
155
|
"id": "lmstudio",
|
|
156
156
|
"kind": "openai",
|
|
157
157
|
"baseUrl": "http://127.0.0.1:1234/v1",
|
|
158
|
+
"connector": "lmstudio",
|
|
158
159
|
"defaultRequestFields": { "reasoning_effort": "none" }
|
|
159
160
|
}
|
|
160
161
|
]
|
|
@@ -172,6 +173,13 @@ a 500 for the whole listing). `forge` itself keeps working unchanged — it's
|
|
|
172
173
|
the implicit endpoint that `FORGE_BASE_URL`/`FORGE_API_KEY` configure; a file
|
|
173
174
|
entry may not reuse the id `"forge"`.
|
|
174
175
|
|
|
176
|
+
- **`connector`** names which local/LAN runtime `baseUrl` points at —
|
|
177
|
+
`lmstudio`, `ollama`, `vllm`, `llama-server`, or the generic
|
|
178
|
+
`openai-compatible` fallback (see `src/connectors.ts`). Optional: it never
|
|
179
|
+
affects request routing, only which connector's `listModels`/request quirks
|
|
180
|
+
`agentproto llm endpoints test`/`detect`/`doctor` use to show loaded models
|
|
181
|
+
and context size. An entry with no `connector` is treated as
|
|
182
|
+
`openai-compatible`.
|
|
175
183
|
- **`apiKeyEnv`** names an env var (never the key itself). Absent, or the env
|
|
176
184
|
var unset, means the endpoint is always keyless — no `Authorization` header
|
|
177
185
|
sent, same as `forge` with no `FORGE_API_KEY`.
|
package/dist/cli.mjs
CHANGED
|
@@ -575,28 +575,28 @@ function flattenContent(content) {
|
|
|
575
575
|
return content.map((c) => c.text).join("");
|
|
576
576
|
}
|
|
577
577
|
function translateInputToMessages(input, instructions) {
|
|
578
|
-
const
|
|
579
|
-
if (instructions)
|
|
580
|
-
|
|
581
|
-
}
|
|
578
|
+
const sysParts = [];
|
|
579
|
+
if (instructions) sysParts.push(instructions);
|
|
580
|
+
const rest = [];
|
|
582
581
|
if (typeof input === "string") {
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
role: "tool",
|
|
595
|
-
|
|
596
|
-
content: item.output
|
|
597
|
-
});
|
|
582
|
+
rest.push({ role: "user", content: input });
|
|
583
|
+
} else {
|
|
584
|
+
for (const item of input) {
|
|
585
|
+
if (item.type === "message") {
|
|
586
|
+
if (item.role === "developer" || item.role === "system") {
|
|
587
|
+
const text = flattenContent(item.content);
|
|
588
|
+
if (text) sysParts.push(text);
|
|
589
|
+
continue;
|
|
590
|
+
}
|
|
591
|
+
rest.push({ role: item.role, content: flattenContent(item.content) });
|
|
592
|
+
} else if (item.type === "function_call_output") {
|
|
593
|
+
rest.push({ role: "tool", tool_call_id: item.call_id, content: item.output });
|
|
594
|
+
}
|
|
598
595
|
}
|
|
599
596
|
}
|
|
597
|
+
const messages = [];
|
|
598
|
+
if (sysParts.length) messages.push({ role: "system", content: sysParts.join("\n\n") });
|
|
599
|
+
messages.push(...rest);
|
|
600
600
|
return messages;
|
|
601
601
|
}
|
|
602
602
|
function translateToolToChatCompletions(tool) {
|
|
@@ -1536,6 +1536,14 @@ async function resumeIncompleteLocalQueueBatches() {
|
|
|
1536
1536
|
}
|
|
1537
1537
|
}
|
|
1538
1538
|
}
|
|
1539
|
+
|
|
1540
|
+
// src/connectors.ts
|
|
1541
|
+
var CONNECTOR_IDS = ["lmstudio", "ollama", "vllm", "llama-server", "openai-compatible"];
|
|
1542
|
+
function isConnectorId(value) {
|
|
1543
|
+
return typeof value === "string" && CONNECTOR_IDS.includes(value);
|
|
1544
|
+
}
|
|
1545
|
+
|
|
1546
|
+
// src/endpoints.ts
|
|
1539
1547
|
var FORBIDDEN_DEFAULT_REQUEST_FIELD_KEYS = /* @__PURE__ */ new Set(["model", "messages", "stream", "tools", "input"]);
|
|
1540
1548
|
var RESERVED_ENDPOINT_IDS = /* @__PURE__ */ new Set(["forge"]);
|
|
1541
1549
|
function validateEndpointConfig(raw, where, errors) {
|
|
@@ -1543,7 +1551,7 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1543
1551
|
errors.push(`${where}: expected an object, got ${raw === null ? "null" : typeof raw}`);
|
|
1544
1552
|
return null;
|
|
1545
1553
|
}
|
|
1546
|
-
const { id, kind, baseUrl, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
|
|
1554
|
+
const { id, kind, baseUrl, connector, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
|
|
1547
1555
|
let ok = true;
|
|
1548
1556
|
if (typeof id !== "string" || id.length === 0) {
|
|
1549
1557
|
errors.push(`${where}.id: required non-empty string`);
|
|
@@ -1571,6 +1579,10 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1571
1579
|
ok = false;
|
|
1572
1580
|
}
|
|
1573
1581
|
}
|
|
1582
|
+
if (connector !== void 0 && !isConnectorId(connector)) {
|
|
1583
|
+
errors.push(`${where}.connector: must be one of ${CONNECTOR_IDS.join(", ")} when present (got ${JSON.stringify(connector)})`);
|
|
1584
|
+
ok = false;
|
|
1585
|
+
}
|
|
1574
1586
|
if (apiKeyEnv !== void 0 && (typeof apiKeyEnv !== "string" || apiKeyEnv.length === 0)) {
|
|
1575
1587
|
errors.push(`${where}.apiKeyEnv: must be a non-empty string when present`);
|
|
1576
1588
|
ok = false;
|
|
@@ -1607,6 +1619,7 @@ function validateEndpointConfig(raw, where, errors) {
|
|
|
1607
1619
|
}
|
|
1608
1620
|
if (!ok || typeof id !== "string" || typeof baseUrl !== "string" || !validUrl) return null;
|
|
1609
1621
|
const built = { id, kind: "openai", baseUrl };
|
|
1622
|
+
if (isConnectorId(connector)) built.connector = connector;
|
|
1610
1623
|
if (typeof apiKeyEnv === "string") built.apiKeyEnv = apiKeyEnv;
|
|
1611
1624
|
if (builtFields) built.defaultRequestFields = builtFields;
|
|
1612
1625
|
if (builtTimeout) built.timeoutMs = builtTimeout;
|
|
@@ -1882,6 +1895,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
|
|
|
1882
1895
|
const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
|
|
1883
1896
|
return parseConfigurableUpstreamUrl(value, "nebius");
|
|
1884
1897
|
}
|
|
1898
|
+
var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
|
|
1899
|
+
function parseDeviceProvider(provider) {
|
|
1900
|
+
const m = DEVICE_PROVIDER_RE.exec(provider);
|
|
1901
|
+
return m ? { endpointId: m[1], device: m[2] } : null;
|
|
1902
|
+
}
|
|
1903
|
+
function resolveDeviceProviderSpec(device) {
|
|
1904
|
+
const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
|
|
1905
|
+
return {
|
|
1906
|
+
keyRequired: true,
|
|
1907
|
+
apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
|
|
1908
|
+
resolveUpstream: () => {
|
|
1909
|
+
const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
|
|
1910
|
+
if (!daemonUrl) return null;
|
|
1911
|
+
const base = parseUpstreamUrl(daemonUrl);
|
|
1912
|
+
if (!base) return null;
|
|
1913
|
+
return {
|
|
1914
|
+
...base,
|
|
1915
|
+
pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
|
|
1916
|
+
};
|
|
1917
|
+
},
|
|
1918
|
+
unavailableMessage
|
|
1919
|
+
};
|
|
1920
|
+
}
|
|
1885
1921
|
function getConfigurableProviderSpec(provider) {
|
|
1886
1922
|
const staticSpec = CONFIGURABLE_PROVIDERS[provider];
|
|
1887
1923
|
if (staticSpec) {
|
|
@@ -1903,6 +1939,8 @@ function getConfigurableProviderSpec(provider) {
|
|
|
1903
1939
|
unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
|
|
1904
1940
|
};
|
|
1905
1941
|
}
|
|
1942
|
+
const deviceRoute = parseDeviceProvider(provider);
|
|
1943
|
+
if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
|
|
1906
1944
|
return void 0;
|
|
1907
1945
|
}
|
|
1908
1946
|
function applyDefaultRequestFields(payload, defaults, skipKeys) {
|
|
@@ -2004,7 +2042,7 @@ async function probeFileEndpointModels(endpoint) {
|
|
|
2004
2042
|
return result;
|
|
2005
2043
|
}
|
|
2006
2044
|
function isKnownProvider(provider) {
|
|
2007
|
-
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
|
|
2045
|
+
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
|
|
2008
2046
|
}
|
|
2009
2047
|
function applyProviderOverride(target, providerOverride) {
|
|
2010
2048
|
const route = { provider: target.provider, model: target.model };
|
|
@@ -2021,7 +2059,7 @@ function parseAnyTransparentModel(model) {
|
|
|
2021
2059
|
const slashIdx = model.indexOf("/");
|
|
2022
2060
|
if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
|
|
2023
2061
|
const provider = model.slice(0, slashIdx);
|
|
2024
|
-
if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
|
|
2062
|
+
if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
|
|
2025
2063
|
return { provider, model: model.slice(slashIdx + 1) };
|
|
2026
2064
|
}
|
|
2027
2065
|
function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
|
|
@@ -2493,6 +2531,8 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2493
2531
|
return;
|
|
2494
2532
|
}
|
|
2495
2533
|
payload.model = resolvedTarget.model;
|
|
2534
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
2535
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
2496
2536
|
trimTools(payload, {
|
|
2497
2537
|
provider: resolvedTarget.provider,
|
|
2498
2538
|
queryTools: opts.queryTools,
|
|
@@ -2532,7 +2572,11 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2532
2572
|
method: "POST",
|
|
2533
2573
|
headers: {
|
|
2534
2574
|
"Content-Type": "application/json",
|
|
2535
|
-
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
|
|
2575
|
+
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
|
|
2576
|
+
// See the /v1/messages handler's identical header — the daemon's
|
|
2577
|
+
// /devices/:id/exec-stream 400s without it, always POST outer verb
|
|
2578
|
+
// notwithstanding.
|
|
2579
|
+
...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
|
|
2536
2580
|
}
|
|
2537
2581
|
};
|
|
2538
2582
|
const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
|
|
@@ -2558,21 +2602,28 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2558
2602
|
});
|
|
2559
2603
|
}
|
|
2560
2604
|
function adaptAnthropicToOpenAI(payload) {
|
|
2605
|
+
const sysParts = [];
|
|
2561
2606
|
if (payload.system != null) {
|
|
2562
|
-
let sysText = "";
|
|
2563
2607
|
if (typeof payload.system === "string") {
|
|
2564
|
-
|
|
2608
|
+
if (payload.system) sysParts.push(payload.system);
|
|
2565
2609
|
} else if (Array.isArray(payload.system)) {
|
|
2566
|
-
|
|
2567
|
-
|
|
2568
|
-
if (sysText) {
|
|
2569
|
-
if (!Array.isArray(payload.messages)) payload.messages = [];
|
|
2570
|
-
if (!payload.messages[0] || payload.messages[0].role !== "system") {
|
|
2571
|
-
payload.messages.unshift({ role: "system", content: sysText });
|
|
2572
|
-
}
|
|
2610
|
+
const text = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
|
|
2611
|
+
if (text) sysParts.push(text);
|
|
2573
2612
|
}
|
|
2574
2613
|
delete payload.system;
|
|
2575
2614
|
}
|
|
2615
|
+
if (Array.isArray(payload.messages)) {
|
|
2616
|
+
payload.messages = payload.messages.filter((m) => {
|
|
2617
|
+
if (m?.role !== "system" && m?.role !== "developer") return true;
|
|
2618
|
+
const text = typeof m.content === "string" ? m.content : Array.isArray(m.content) ? m.content.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n") : "";
|
|
2619
|
+
if (text) sysParts.push(text);
|
|
2620
|
+
return false;
|
|
2621
|
+
});
|
|
2622
|
+
}
|
|
2623
|
+
if (sysParts.length) {
|
|
2624
|
+
if (!Array.isArray(payload.messages)) payload.messages = [];
|
|
2625
|
+
payload.messages.unshift({ role: "system", content: sysParts.join("\n\n") });
|
|
2626
|
+
}
|
|
2576
2627
|
if (payload.tool_choice && typeof payload.tool_choice === "object") {
|
|
2577
2628
|
const tc = payload.tool_choice;
|
|
2578
2629
|
if (tc.type === "any") {
|
|
@@ -3362,6 +3413,8 @@ var server = createServer((req, res) => {
|
|
|
3362
3413
|
let cred;
|
|
3363
3414
|
let headers = { "Content-Type": "application/json" };
|
|
3364
3415
|
payload.model = resolvedTarget.model;
|
|
3416
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
3417
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
3365
3418
|
trimTools(payload, {
|
|
3366
3419
|
provider: resolvedTarget.provider,
|
|
3367
3420
|
queryTools,
|
|
@@ -3387,6 +3440,7 @@ var server = createServer((req, res) => {
|
|
|
3387
3440
|
cred = await resolveUpstreamCredential(resolvedTarget.provider);
|
|
3388
3441
|
targetApiKey = cred?.value ?? "";
|
|
3389
3442
|
if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
|
|
3443
|
+
if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
|
|
3390
3444
|
const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
|
|
3391
3445
|
adaptAnthropicToOpenAI(payload);
|
|
3392
3446
|
applyDefaultRequestFields(
|
|
@@ -3583,6 +3637,17 @@ var server = createServer((req, res) => {
|
|
|
3583
3637
|
);
|
|
3584
3638
|
}
|
|
3585
3639
|
});
|
|
3640
|
+
if (status < 200 || status >= 300) {
|
|
3641
|
+
let errBody = "";
|
|
3642
|
+
proxyRes.on("data", (c) => {
|
|
3643
|
+
if (errBody.length < 300) errBody += c.toString("utf8");
|
|
3644
|
+
});
|
|
3645
|
+
proxyRes.on("end", () => {
|
|
3646
|
+
console.error(
|
|
3647
|
+
`[Proxy] upstream error: ${resolvedTarget.provider}:${resolvedTarget.model} status=${status} body=${JSON.stringify(errBody.slice(0, 300))}`
|
|
3648
|
+
);
|
|
3649
|
+
});
|
|
3650
|
+
}
|
|
3586
3651
|
const contentType = proxyRes.headers["content-type"] || "";
|
|
3587
3652
|
const isStreaming = payload.stream === true && /text\/event-stream/i.test(contentType);
|
|
3588
3653
|
const retryEmptyTurn = () => {
|