@agentproto/llm-endpoint 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -150,11 +150,12 @@ configured from a JSON file instead of a pair of env vars per server:
150
150
  "defaultRequestFields": { "chat_template_kwargs": { "enable_thinking": false } },
151
151
  "timeoutMs": { "firstTokenMs": 180000 }
152
152
  },
153
- { "id": "ollama", "kind": "openai", "baseUrl": "http://192.168.1.20:11434/v1" },
153
+ { "id": "ollama", "kind": "openai", "baseUrl": "http://192.168.1.20:11434/v1", "connector": "ollama" },
154
154
  {
155
155
  "id": "lmstudio",
156
156
  "kind": "openai",
157
157
  "baseUrl": "http://127.0.0.1:1234/v1",
158
+ "connector": "lmstudio",
158
159
  "defaultRequestFields": { "reasoning_effort": "none" }
159
160
  }
160
161
  ]
@@ -172,6 +173,13 @@ a 500 for the whole listing). `forge` itself keeps working unchanged — it's
172
173
  the implicit endpoint that `FORGE_BASE_URL`/`FORGE_API_KEY` configure; a file
173
174
  entry may not reuse the id `"forge"`.
174
175
 
176
+ - **`connector`** names which local/LAN runtime `baseUrl` points at —
177
+ `lmstudio`, `ollama`, `vllm`, `llama-server`, or the generic
178
+ `openai-compatible` fallback (see `src/connectors.ts`). Optional: it never
179
+ affects request routing, only which connector's `listModels`/request quirks
180
+ `agentproto llm endpoints test`/`detect`/`doctor` use to show loaded models
181
+ and context size. An entry with no `connector` is treated as
182
+ `openai-compatible`.
175
183
  - **`apiKeyEnv`** names an env var (never the key itself). Absent, or the env
176
184
  var unset, means the endpoint is always keyless — no `Authorization` header
177
185
  sent, same as `forge` with no `FORGE_API_KEY`.
package/dist/cli.mjs CHANGED
@@ -575,28 +575,28 @@ function flattenContent(content) {
575
575
  return content.map((c) => c.text).join("");
576
576
  }
577
577
  function translateInputToMessages(input, instructions) {
578
- const messages = [];
579
- if (instructions) {
580
- messages.push({ role: "system", content: instructions });
581
- }
578
+ const sysParts = [];
579
+ if (instructions) sysParts.push(instructions);
580
+ const rest = [];
582
581
  if (typeof input === "string") {
583
- messages.push({ role: "user", content: input });
584
- return messages;
585
- }
586
- for (const item of input) {
587
- if (item.type === "message") {
588
- messages.push({
589
- role: item.role === "developer" ? "system" : item.role,
590
- content: flattenContent(item.content)
591
- });
592
- } else if (item.type === "function_call_output") {
593
- messages.push({
594
- role: "tool",
595
- tool_call_id: item.call_id,
596
- content: item.output
597
- });
582
+ rest.push({ role: "user", content: input });
583
+ } else {
584
+ for (const item of input) {
585
+ if (item.type === "message") {
586
+ if (item.role === "developer" || item.role === "system") {
587
+ const text = flattenContent(item.content);
588
+ if (text) sysParts.push(text);
589
+ continue;
590
+ }
591
+ rest.push({ role: item.role, content: flattenContent(item.content) });
592
+ } else if (item.type === "function_call_output") {
593
+ rest.push({ role: "tool", tool_call_id: item.call_id, content: item.output });
594
+ }
598
595
  }
599
596
  }
597
+ const messages = [];
598
+ if (sysParts.length) messages.push({ role: "system", content: sysParts.join("\n\n") });
599
+ messages.push(...rest);
600
600
  return messages;
601
601
  }
602
602
  function translateToolToChatCompletions(tool) {
@@ -1536,6 +1536,14 @@ async function resumeIncompleteLocalQueueBatches() {
1536
1536
  }
1537
1537
  }
1538
1538
  }
1539
+
1540
+ // src/connectors.ts
1541
+ var CONNECTOR_IDS = ["lmstudio", "ollama", "vllm", "llama-server", "openai-compatible"];
1542
+ function isConnectorId(value) {
1543
+ return typeof value === "string" && CONNECTOR_IDS.includes(value);
1544
+ }
1545
+
1546
+ // src/endpoints.ts
1539
1547
  var FORBIDDEN_DEFAULT_REQUEST_FIELD_KEYS = /* @__PURE__ */ new Set(["model", "messages", "stream", "tools", "input"]);
1540
1548
  var RESERVED_ENDPOINT_IDS = /* @__PURE__ */ new Set(["forge"]);
1541
1549
  function validateEndpointConfig(raw, where, errors) {
@@ -1543,7 +1551,7 @@ function validateEndpointConfig(raw, where, errors) {
1543
1551
  errors.push(`${where}: expected an object, got ${raw === null ? "null" : typeof raw}`);
1544
1552
  return null;
1545
1553
  }
1546
- const { id, kind, baseUrl, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
1554
+ const { id, kind, baseUrl, connector, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
1547
1555
  let ok = true;
1548
1556
  if (typeof id !== "string" || id.length === 0) {
1549
1557
  errors.push(`${where}.id: required non-empty string`);
@@ -1571,6 +1579,10 @@ function validateEndpointConfig(raw, where, errors) {
1571
1579
  ok = false;
1572
1580
  }
1573
1581
  }
1582
+ if (connector !== void 0 && !isConnectorId(connector)) {
1583
+ errors.push(`${where}.connector: must be one of ${CONNECTOR_IDS.join(", ")} when present (got ${JSON.stringify(connector)})`);
1584
+ ok = false;
1585
+ }
1574
1586
  if (apiKeyEnv !== void 0 && (typeof apiKeyEnv !== "string" || apiKeyEnv.length === 0)) {
1575
1587
  errors.push(`${where}.apiKeyEnv: must be a non-empty string when present`);
1576
1588
  ok = false;
@@ -1607,6 +1619,7 @@ function validateEndpointConfig(raw, where, errors) {
1607
1619
  }
1608
1620
  if (!ok || typeof id !== "string" || typeof baseUrl !== "string" || !validUrl) return null;
1609
1621
  const built = { id, kind: "openai", baseUrl };
1622
+ if (isConnectorId(connector)) built.connector = connector;
1610
1623
  if (typeof apiKeyEnv === "string") built.apiKeyEnv = apiKeyEnv;
1611
1624
  if (builtFields) built.defaultRequestFields = builtFields;
1612
1625
  if (builtTimeout) built.timeoutMs = builtTimeout;
@@ -1882,6 +1895,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
1882
1895
  const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
1883
1896
  return parseConfigurableUpstreamUrl(value, "nebius");
1884
1897
  }
1898
+ var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
1899
+ function parseDeviceProvider(provider) {
1900
+ const m = DEVICE_PROVIDER_RE.exec(provider);
1901
+ return m ? { endpointId: m[1], device: m[2] } : null;
1902
+ }
1903
+ function resolveDeviceProviderSpec(device) {
1904
+ const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
1905
+ return {
1906
+ keyRequired: true,
1907
+ apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
1908
+ resolveUpstream: () => {
1909
+ const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
1910
+ if (!daemonUrl) return null;
1911
+ const base = parseUpstreamUrl(daemonUrl);
1912
+ if (!base) return null;
1913
+ return {
1914
+ ...base,
1915
+ pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
1916
+ };
1917
+ },
1918
+ unavailableMessage
1919
+ };
1920
+ }
1885
1921
  function getConfigurableProviderSpec(provider) {
1886
1922
  const staticSpec = CONFIGURABLE_PROVIDERS[provider];
1887
1923
  if (staticSpec) {
@@ -1903,6 +1939,8 @@ function getConfigurableProviderSpec(provider) {
1903
1939
  unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
1904
1940
  };
1905
1941
  }
1942
+ const deviceRoute = parseDeviceProvider(provider);
1943
+ if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
1906
1944
  return void 0;
1907
1945
  }
1908
1946
  function applyDefaultRequestFields(payload, defaults, skipKeys) {
@@ -2004,7 +2042,7 @@ async function probeFileEndpointModels(endpoint) {
2004
2042
  return result;
2005
2043
  }
2006
2044
  function isKnownProvider(provider) {
2007
- return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
2045
+ return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
2008
2046
  }
2009
2047
  function applyProviderOverride(target, providerOverride) {
2010
2048
  const route = { provider: target.provider, model: target.model };
@@ -2021,7 +2059,7 @@ function parseAnyTransparentModel(model) {
2021
2059
  const slashIdx = model.indexOf("/");
2022
2060
  if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
2023
2061
  const provider = model.slice(0, slashIdx);
2024
- if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
2062
+ if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
2025
2063
  return { provider, model: model.slice(slashIdx + 1) };
2026
2064
  }
2027
2065
  function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
@@ -2493,6 +2531,8 @@ function handleChatCompletionsRequest(req, res, opts) {
2493
2531
  return;
2494
2532
  }
2495
2533
  payload.model = resolvedTarget.model;
2534
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
2535
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
2496
2536
  trimTools(payload, {
2497
2537
  provider: resolvedTarget.provider,
2498
2538
  queryTools: opts.queryTools,
@@ -2532,7 +2572,11 @@ function handleChatCompletionsRequest(req, res, opts) {
2532
2572
  method: "POST",
2533
2573
  headers: {
2534
2574
  "Content-Type": "application/json",
2535
- ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
2575
+ ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
2576
+ // See the /v1/messages handler's identical header — the daemon's
2577
+ // /devices/:id/exec-stream 400s without it, always POST outer verb
2578
+ // notwithstanding.
2579
+ ...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
2536
2580
  }
2537
2581
  };
2538
2582
  const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
@@ -2558,21 +2602,28 @@ function handleChatCompletionsRequest(req, res, opts) {
2558
2602
  });
2559
2603
  }
2560
2604
  function adaptAnthropicToOpenAI(payload) {
2605
+ const sysParts = [];
2561
2606
  if (payload.system != null) {
2562
- let sysText = "";
2563
2607
  if (typeof payload.system === "string") {
2564
- sysText = payload.system;
2608
+ if (payload.system) sysParts.push(payload.system);
2565
2609
  } else if (Array.isArray(payload.system)) {
2566
- sysText = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
2567
- }
2568
- if (sysText) {
2569
- if (!Array.isArray(payload.messages)) payload.messages = [];
2570
- if (!payload.messages[0] || payload.messages[0].role !== "system") {
2571
- payload.messages.unshift({ role: "system", content: sysText });
2572
- }
2610
+ const text = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
2611
+ if (text) sysParts.push(text);
2573
2612
  }
2574
2613
  delete payload.system;
2575
2614
  }
2615
+ if (Array.isArray(payload.messages)) {
2616
+ payload.messages = payload.messages.filter((m) => {
2617
+ if (m?.role !== "system" && m?.role !== "developer") return true;
2618
+ const text = typeof m.content === "string" ? m.content : Array.isArray(m.content) ? m.content.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n") : "";
2619
+ if (text) sysParts.push(text);
2620
+ return false;
2621
+ });
2622
+ }
2623
+ if (sysParts.length) {
2624
+ if (!Array.isArray(payload.messages)) payload.messages = [];
2625
+ payload.messages.unshift({ role: "system", content: sysParts.join("\n\n") });
2626
+ }
2576
2627
  if (payload.tool_choice && typeof payload.tool_choice === "object") {
2577
2628
  const tc = payload.tool_choice;
2578
2629
  if (tc.type === "any") {
@@ -3362,6 +3413,8 @@ var server = createServer((req, res) => {
3362
3413
  let cred;
3363
3414
  let headers = { "Content-Type": "application/json" };
3364
3415
  payload.model = resolvedTarget.model;
3416
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
3417
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
3365
3418
  trimTools(payload, {
3366
3419
  provider: resolvedTarget.provider,
3367
3420
  queryTools,
@@ -3387,6 +3440,7 @@ var server = createServer((req, res) => {
3387
3440
  cred = await resolveUpstreamCredential(resolvedTarget.provider);
3388
3441
  targetApiKey = cred?.value ?? "";
3389
3442
  if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
3443
+ if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
3390
3444
  const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
3391
3445
  adaptAnthropicToOpenAI(payload);
3392
3446
  applyDefaultRequestFields(
@@ -3583,6 +3637,17 @@ var server = createServer((req, res) => {
3583
3637
  );
3584
3638
  }
3585
3639
  });
3640
+ if (status < 200 || status >= 300) {
3641
+ let errBody = "";
3642
+ proxyRes.on("data", (c) => {
3643
+ if (errBody.length < 300) errBody += c.toString("utf8");
3644
+ });
3645
+ proxyRes.on("end", () => {
3646
+ console.error(
3647
+ `[Proxy] upstream error: ${resolvedTarget.provider}:${resolvedTarget.model} status=${status} body=${JSON.stringify(errBody.slice(0, 300))}`
3648
+ );
3649
+ });
3650
+ }
3586
3651
  const contentType = proxyRes.headers["content-type"] || "";
3587
3652
  const isStreaming = payload.stream === true && /text\/event-stream/i.test(contentType);
3588
3653
  const retryEmptyTurn = () => {