@agentproto/llm-endpoint 0.8.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -68,6 +68,74 @@ interface ModelPack {
68
68
  toolsAllow?: string[];
69
69
  }
70
70
 
71
+ /**
72
+ * Connectors — small, declarative probes for the local/LAN inference
73
+ * runtimes a named endpoint (`endpoints.ts`) commonly points at: LM Studio,
74
+ * Ollama, vLLM, llama-server, plus a generic OpenAI-compatible fallback for
75
+ * anything else. A connector answers three questions about a `baseUrl`: is
76
+ * this runtime actually running there (`probe`), what models does it have
77
+ * ({@link listModels}), and does it need special request handling
78
+ * ({@link ConnectorQuirks}).
79
+ *
80
+ * Used by `agentproto llm endpoints add/detect/test` and `doctor`'s
81
+ * inference-endpoints check — never by the proxy's own request routing
82
+ * (`index.ts`). In particular, LM Studio's GGUF Qwen builds reject a
83
+ * non-leading system message once tool defs are present; the generic fix
84
+ * for that lives in the adapter layer (a separate PR) — `quirks` is only a
85
+ * hook so callers can find out which connector needs it, not a duplicate of
86
+ * that fix.
87
+ *
88
+ * Deliberately dependency-free and side-effect-free at import time, like
89
+ * `endpoints.ts` and `packs.ts`.
90
+ */
91
+ type ConnectorId = 'lmstudio' | 'ollama' | 'vllm' | 'llama-server' | 'openai-compatible';
92
+ declare const CONNECTOR_IDS: readonly ConnectorId[];
93
+ declare function isConnectorId(value: unknown): value is ConnectorId;
94
+ type ConnectorModelState = 'loaded' | 'not-loaded' | 'unknown';
95
+ /** One model as reported by a connector's `listModels`. `loadedCtx`/`maxCtx`
96
+ * are omitted (not `undefined`-filled) when the runtime's API doesn't
97
+ * expose them — e.g. Ollama's `/api/tags`/`/api/ps` carry no context size. */
98
+ interface ConnectorModel {
99
+ id: string;
100
+ loadedCtx?: number;
101
+ maxCtx?: number;
102
+ device?: string;
103
+ state: ConnectorModelState;
104
+ }
105
+ interface ConnectorQuirks {
106
+ /** This runtime's chat template can reject a system message that isn't
107
+ * the first message once tool definitions are present (observed on LM
108
+ * Studio GGUF Qwen builds) — callers that build the outbound message
109
+ * array should keep a single leading system message for it. */
110
+ requiresLeadingSystemMessage?: boolean;
111
+ }
112
+ interface Connector {
113
+ id: ConnectorId;
114
+ label: string;
115
+ /** Default local port this runtime listens on; `null` for the
116
+ * OpenAI-compatible fallback, which has no default port of its own and
117
+ * is never probed by default-port detection. */
118
+ defaultPort: number | null;
119
+ quirks: ConnectorQuirks;
120
+ /** Is this runtime actually answering at `baseUrl`? Never throws — a
121
+ * network error, timeout, or unexpected shape resolves `false`. */
122
+ probe(baseUrl: string, fetchImpl?: typeof fetch): Promise<boolean>;
123
+ /** Best-effort model listing. Never throws — a failed call resolves `[]`. */
124
+ listModels(baseUrl: string, fetchImpl?: typeof fetch): Promise<ConnectorModel[]>;
125
+ }
126
+ declare const CONNECTORS: Record<ConnectorId, Connector>;
127
+ declare function connectorById(id: string): Connector | undefined;
128
+ /** Identify which connector is serving `baseUrl`, trying the most specific
129
+ * probes first and falling back to the generic OpenAI-compatible one.
130
+ * Returns `null` only if even the fallback probe fails (nothing OpenAI-
131
+ * compatible is reachable there at all). */
132
+ declare function detectConnector(baseUrl: string, fetchImpl?: typeof fetch): Promise<ConnectorId | null>;
133
+ /** Default local ports probed by `agentproto llm endpoints detect`, one per
134
+ * runtime that has a real default port (`openai-compatible` doesn't — it's
135
+ * only ever selected explicitly, or as the fallback identity for a
136
+ * positive probe at some other port/runtime). */
137
+ declare const DEFAULT_LOCAL_PORTS: Readonly<Partial<Record<ConnectorId, number>>>;
138
+
71
139
  /**
72
140
  * Named OpenAI-compatible endpoints — N local/LAN model servers (Ollama,
73
141
  * llama-server, vLLM, …) configured from a JSON file instead of one-off env
@@ -82,6 +150,7 @@ interface ModelPack {
82
150
  * (src/index.ts) and the CLI (`agentproto llm endpoints …`), which must NOT
83
151
  * pull in an HTTP server just to read/validate the config file.
84
152
  */
153
+
85
154
  /** A JSON value — what `JSON.parse` can ever produce. */
86
155
  type JsonValue = string | number | boolean | null | JsonValue[] | {
87
156
  [key: string]: JsonValue;
@@ -105,6 +174,13 @@ interface EndpointConfig {
105
174
  id: string;
106
175
  kind: 'openai';
107
176
  baseUrl: string;
177
+ /** Which local/LAN runtime this points at (`lmstudio`, `ollama`, `vllm`,
178
+ * `llama-server`) or the generic `openai-compatible` fallback — see
179
+ * `connectors.ts`. Absent on entries written before connectors existed;
180
+ * callers that need one (`endpoints test`, `doctor`) treat a missing
181
+ * value as `openai-compatible`. Never affects request routing — only
182
+ * which connector's `listModels`/`quirks` apply. */
183
+ connector?: ConnectorId;
108
184
  /** Name of the env var holding the key — never the key itself. Absent ⇒
109
185
  * the endpoint is always keyless (a private/LAN server with no auth). */
110
186
  apiKeyEnv?: string;
@@ -349,6 +425,7 @@ type UpstreamTestResult = {
349
425
  * (never forwarded) rather than tested.
350
426
  */
351
427
  declare function testUpstream(provider: string): Promise<UpstreamTestResult>;
428
+ declare function adaptAnthropicToOpenAI(payload: any): void;
352
429
  declare function stripThinkingFromAnthropicJson(jsonStr: string): string;
353
430
  declare function resolveEmptyTurnRetries(): number;
354
431
  /**
@@ -424,4 +501,4 @@ declare const server: node_http.Server<typeof IncomingMessage, typeof ServerResp
424
501
  /** Démarre le proxy sur `port` (défaut : {@link PORT}). Renvoie le serveur en écoute. */
425
502
  declare function start(port?: number): node_http.Server<typeof IncomingMessage, typeof ServerResponse>;
426
503
 
427
- export { CANONICAL_UPSTREAMS, type ConfigurableUpstream, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, describeUpstreamStatus, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
504
+ export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, type ConfigurableUpstream, type Connector, type ConnectorId, type ConnectorModel, type ConnectorModelState, type ConnectorQuirks, DEFAULT_LOCAL_PORTS, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
package/dist/index.mjs CHANGED
@@ -573,28 +573,28 @@ function flattenContent(content) {
573
573
  return content.map((c) => c.text).join("");
574
574
  }
575
575
  function translateInputToMessages(input, instructions) {
576
- const messages = [];
577
- if (instructions) {
578
- messages.push({ role: "system", content: instructions });
579
- }
576
+ const sysParts = [];
577
+ if (instructions) sysParts.push(instructions);
578
+ const rest = [];
580
579
  if (typeof input === "string") {
581
- messages.push({ role: "user", content: input });
582
- return messages;
583
- }
584
- for (const item of input) {
585
- if (item.type === "message") {
586
- messages.push({
587
- role: item.role === "developer" ? "system" : item.role,
588
- content: flattenContent(item.content)
589
- });
590
- } else if (item.type === "function_call_output") {
591
- messages.push({
592
- role: "tool",
593
- tool_call_id: item.call_id,
594
- content: item.output
595
- });
580
+ rest.push({ role: "user", content: input });
581
+ } else {
582
+ for (const item of input) {
583
+ if (item.type === "message") {
584
+ if (item.role === "developer" || item.role === "system") {
585
+ const text = flattenContent(item.content);
586
+ if (text) sysParts.push(text);
587
+ continue;
588
+ }
589
+ rest.push({ role: item.role, content: flattenContent(item.content) });
590
+ } else if (item.type === "function_call_output") {
591
+ rest.push({ role: "tool", tool_call_id: item.call_id, content: item.output });
592
+ }
596
593
  }
597
594
  }
595
+ const messages = [];
596
+ if (sysParts.length) messages.push({ role: "system", content: sysParts.join("\n\n") });
597
+ messages.push(...rest);
598
598
  return messages;
599
599
  }
600
600
  function translateToolToChatCompletions(tool) {
@@ -1534,6 +1534,159 @@ async function resumeIncompleteLocalQueueBatches() {
1534
1534
  }
1535
1535
  }
1536
1536
  }
1537
+
1538
+ // src/connectors.ts
1539
+ var CONNECTOR_IDS = ["lmstudio", "ollama", "vllm", "llama-server", "openai-compatible"];
1540
+ function isConnectorId(value) {
1541
+ return typeof value === "string" && CONNECTOR_IDS.includes(value);
1542
+ }
1543
+ var PROBE_TIMEOUT_MS = 2500;
1544
+ async function fetchJsonSafe(url, fetchImpl, timeoutMs = PROBE_TIMEOUT_MS) {
1545
+ const controller = new AbortController();
1546
+ const timer = setTimeout(() => controller.abort(), timeoutMs);
1547
+ try {
1548
+ const res = await fetchImpl(url, { signal: controller.signal });
1549
+ if (!res.ok) return { ok: false };
1550
+ const body = await res.json().catch(() => null);
1551
+ return { ok: true, body };
1552
+ } catch {
1553
+ return { ok: false };
1554
+ } finally {
1555
+ clearTimeout(timer);
1556
+ }
1557
+ }
1558
+ function runtimeRoot(baseUrl) {
1559
+ return baseUrl.replace(/\/v1\/?$/, "");
1560
+ }
1561
+ function stripTrailingSlash(url) {
1562
+ return url.replace(/\/+$/, "");
1563
+ }
1564
+ var lmstudio = {
1565
+ id: "lmstudio",
1566
+ label: "LM Studio",
1567
+ defaultPort: 1234,
1568
+ quirks: { requiresLeadingSystemMessage: true },
1569
+ async probe(baseUrl, fetchImpl = fetch) {
1570
+ const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/v0/models`, fetchImpl);
1571
+ return result.ok && isRecord(result.body) && Array.isArray(result.body.data);
1572
+ },
1573
+ async listModels(baseUrl, fetchImpl = fetch) {
1574
+ const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/v0/models`, fetchImpl);
1575
+ if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
1576
+ return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({
1577
+ id: m.id,
1578
+ ...typeof m.loaded_context_length === "number" ? { loadedCtx: m.loaded_context_length } : {},
1579
+ ...typeof m.max_context_length === "number" ? { maxCtx: m.max_context_length } : {},
1580
+ state: m.state === "loaded" ? "loaded" : m.state === "not-loaded" ? "not-loaded" : "unknown"
1581
+ }));
1582
+ }
1583
+ };
1584
+ var ollama = {
1585
+ id: "ollama",
1586
+ label: "Ollama",
1587
+ defaultPort: 11434,
1588
+ quirks: {},
1589
+ async probe(baseUrl, fetchImpl = fetch) {
1590
+ const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/api/tags`, fetchImpl);
1591
+ return result.ok && isRecord(result.body) && Array.isArray(result.body.models);
1592
+ },
1593
+ async listModels(baseUrl, fetchImpl = fetch) {
1594
+ const root = runtimeRoot(baseUrl);
1595
+ const [tags, ps] = await Promise.all([fetchJsonSafe(`${root}/api/tags`, fetchImpl), fetchJsonSafe(`${root}/api/ps`, fetchImpl)]);
1596
+ const loadedNames = /* @__PURE__ */ new Set();
1597
+ if (ps.ok && isRecord(ps.body) && Array.isArray(ps.body.models)) {
1598
+ for (const m of ps.body.models) {
1599
+ if (isRecord(m) && typeof m.name === "string") loadedNames.add(m.name);
1600
+ }
1601
+ }
1602
+ if (!tags.ok || !isRecord(tags.body) || !Array.isArray(tags.body.models)) return [];
1603
+ return tags.body.models.filter(isRecord).filter((m) => typeof m.name === "string").map((m) => ({ id: m.name, state: loadedNames.has(m.name) ? "loaded" : "not-loaded" }));
1604
+ }
1605
+ };
1606
+ var llamaServer = {
1607
+ id: "llama-server",
1608
+ label: "llama-server",
1609
+ defaultPort: 8080,
1610
+ quirks: {},
1611
+ async probe(baseUrl, fetchImpl = fetch) {
1612
+ const result = await fetchJsonSafe(`${runtimeRoot(baseUrl)}/props`, fetchImpl);
1613
+ return result.ok && isRecord(result.body) && ("default_generation_settings" in result.body || "model_path" in result.body);
1614
+ },
1615
+ async listModels(baseUrl, fetchImpl = fetch) {
1616
+ const [props, models] = await Promise.all([
1617
+ fetchJsonSafe(`${runtimeRoot(baseUrl)}/props`, fetchImpl),
1618
+ fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl)
1619
+ ]);
1620
+ let ctx;
1621
+ if (props.ok && isRecord(props.body)) {
1622
+ const gen = props.body.default_generation_settings;
1623
+ if (isRecord(gen) && typeof gen.n_ctx === "number") ctx = gen.n_ctx;
1624
+ else if (typeof props.body.n_ctx === "number") ctx = props.body.n_ctx;
1625
+ }
1626
+ if (!models.ok || !isRecord(models.body) || !Array.isArray(models.body.data)) return [];
1627
+ return models.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({ id: m.id, ...ctx !== void 0 ? { loadedCtx: ctx, maxCtx: ctx } : {}, state: "loaded" }));
1628
+ }
1629
+ };
1630
+ var vllm = {
1631
+ id: "vllm",
1632
+ label: "vLLM",
1633
+ defaultPort: 8e3,
1634
+ quirks: {},
1635
+ async probe(baseUrl, fetchImpl = fetch) {
1636
+ const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
1637
+ if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return false;
1638
+ return result.body.data.some((m) => isRecord(m) && typeof m.max_model_len === "number");
1639
+ },
1640
+ async listModels(baseUrl, fetchImpl = fetch) {
1641
+ const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
1642
+ if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
1643
+ return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({
1644
+ id: m.id,
1645
+ ...typeof m.max_model_len === "number" ? { loadedCtx: m.max_model_len, maxCtx: m.max_model_len } : {},
1646
+ state: "loaded"
1647
+ }));
1648
+ }
1649
+ };
1650
+ var openaiCompatible = {
1651
+ id: "openai-compatible",
1652
+ label: "OpenAI-compatible",
1653
+ defaultPort: null,
1654
+ quirks: {},
1655
+ async probe(baseUrl, fetchImpl = fetch) {
1656
+ const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
1657
+ return result.ok && isRecord(result.body) && Array.isArray(result.body.data);
1658
+ },
1659
+ async listModels(baseUrl, fetchImpl = fetch) {
1660
+ const result = await fetchJsonSafe(`${stripTrailingSlash(baseUrl)}/models`, fetchImpl);
1661
+ if (!result.ok || !isRecord(result.body) || !Array.isArray(result.body.data)) return [];
1662
+ return result.body.data.filter(isRecord).filter((m) => typeof m.id === "string").map((m) => ({ id: m.id, state: "unknown" }));
1663
+ }
1664
+ };
1665
+ var CONNECTORS = {
1666
+ lmstudio,
1667
+ ollama,
1668
+ vllm,
1669
+ "llama-server": llamaServer,
1670
+ "openai-compatible": openaiCompatible
1671
+ };
1672
+ function connectorById(id) {
1673
+ return isConnectorId(id) ? CONNECTORS[id] : void 0;
1674
+ }
1675
+ var DETECTION_ORDER = ["lmstudio", "ollama", "llama-server", "vllm", "openai-compatible"];
1676
+ async function detectConnector(baseUrl, fetchImpl = fetch) {
1677
+ for (const id of DETECTION_ORDER) {
1678
+ if (await CONNECTORS[id].probe(baseUrl, fetchImpl)) return id;
1679
+ }
1680
+ return null;
1681
+ }
1682
+ var DEFAULT_LOCAL_PORTS = {
1683
+ lmstudio: 1234,
1684
+ ollama: 11434,
1685
+ "llama-server": 8080,
1686
+ vllm: 8e3
1687
+ };
1688
+
1689
+ // src/endpoints.ts
1537
1690
  var FORBIDDEN_DEFAULT_REQUEST_FIELD_KEYS = /* @__PURE__ */ new Set(["model", "messages", "stream", "tools", "input"]);
1538
1691
  var RESERVED_ENDPOINT_IDS = /* @__PURE__ */ new Set(["forge"]);
1539
1692
  function validateEndpointConfig(raw, where, errors) {
@@ -1541,7 +1694,7 @@ function validateEndpointConfig(raw, where, errors) {
1541
1694
  errors.push(`${where}: expected an object, got ${raw === null ? "null" : typeof raw}`);
1542
1695
  return null;
1543
1696
  }
1544
- const { id, kind, baseUrl, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
1697
+ const { id, kind, baseUrl, connector, apiKeyEnv, defaultRequestFields, timeoutMs } = raw;
1545
1698
  let ok = true;
1546
1699
  if (typeof id !== "string" || id.length === 0) {
1547
1700
  errors.push(`${where}.id: required non-empty string`);
@@ -1569,6 +1722,10 @@ function validateEndpointConfig(raw, where, errors) {
1569
1722
  ok = false;
1570
1723
  }
1571
1724
  }
1725
+ if (connector !== void 0 && !isConnectorId(connector)) {
1726
+ errors.push(`${where}.connector: must be one of ${CONNECTOR_IDS.join(", ")} when present (got ${JSON.stringify(connector)})`);
1727
+ ok = false;
1728
+ }
1572
1729
  if (apiKeyEnv !== void 0 && (typeof apiKeyEnv !== "string" || apiKeyEnv.length === 0)) {
1573
1730
  errors.push(`${where}.apiKeyEnv: must be a non-empty string when present`);
1574
1731
  ok = false;
@@ -1605,6 +1762,7 @@ function validateEndpointConfig(raw, where, errors) {
1605
1762
  }
1606
1763
  if (!ok || typeof id !== "string" || typeof baseUrl !== "string" || !validUrl) return null;
1607
1764
  const built = { id, kind: "openai", baseUrl };
1765
+ if (isConnectorId(connector)) built.connector = connector;
1608
1766
  if (typeof apiKeyEnv === "string") built.apiKeyEnv = apiKeyEnv;
1609
1767
  if (builtFields) built.defaultRequestFields = builtFields;
1610
1768
  if (builtTimeout) built.timeoutMs = builtTimeout;
@@ -1883,6 +2041,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
1883
2041
  const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
1884
2042
  return parseConfigurableUpstreamUrl(value, "nebius");
1885
2043
  }
2044
+ var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
2045
+ function parseDeviceProvider(provider) {
2046
+ const m = DEVICE_PROVIDER_RE.exec(provider);
2047
+ return m ? { endpointId: m[1], device: m[2] } : null;
2048
+ }
2049
+ function resolveDeviceProviderSpec(device) {
2050
+ const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
2051
+ return {
2052
+ keyRequired: true,
2053
+ apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
2054
+ resolveUpstream: () => {
2055
+ const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
2056
+ if (!daemonUrl) return null;
2057
+ const base = parseUpstreamUrl(daemonUrl);
2058
+ if (!base) return null;
2059
+ return {
2060
+ ...base,
2061
+ pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
2062
+ };
2063
+ },
2064
+ unavailableMessage
2065
+ };
2066
+ }
1886
2067
  function getConfigurableProviderSpec(provider) {
1887
2068
  const staticSpec = CONFIGURABLE_PROVIDERS[provider];
1888
2069
  if (staticSpec) {
@@ -1904,6 +2085,8 @@ function getConfigurableProviderSpec(provider) {
1904
2085
  unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
1905
2086
  };
1906
2087
  }
2088
+ const deviceRoute = parseDeviceProvider(provider);
2089
+ if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
1907
2090
  return void 0;
1908
2091
  }
1909
2092
  function applyDefaultRequestFields(payload, defaults, skipKeys) {
@@ -2005,7 +2188,7 @@ async function probeFileEndpointModels(endpoint) {
2005
2188
  return result;
2006
2189
  }
2007
2190
  function isKnownProvider(provider) {
2008
- return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
2191
+ return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
2009
2192
  }
2010
2193
  function applyProviderOverride(target, providerOverride) {
2011
2194
  const route = { provider: target.provider, model: target.model };
@@ -2022,7 +2205,7 @@ function parseAnyTransparentModel(model) {
2022
2205
  const slashIdx = model.indexOf("/");
2023
2206
  if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
2024
2207
  const provider = model.slice(0, slashIdx);
2025
- if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
2208
+ if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
2026
2209
  return { provider, model: model.slice(slashIdx + 1) };
2027
2210
  }
2028
2211
  function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
@@ -2494,6 +2677,8 @@ function handleChatCompletionsRequest(req, res, opts) {
2494
2677
  return;
2495
2678
  }
2496
2679
  payload.model = resolvedTarget.model;
2680
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
2681
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
2497
2682
  trimTools(payload, {
2498
2683
  provider: resolvedTarget.provider,
2499
2684
  queryTools: opts.queryTools,
@@ -2533,7 +2718,11 @@ function handleChatCompletionsRequest(req, res, opts) {
2533
2718
  method: "POST",
2534
2719
  headers: {
2535
2720
  "Content-Type": "application/json",
2536
- ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
2721
+ ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
2722
+ // See the /v1/messages handler's identical header — the daemon's
2723
+ // /devices/:id/exec-stream 400s without it, always POST outer verb
2724
+ // notwithstanding.
2725
+ ...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
2537
2726
  }
2538
2727
  };
2539
2728
  const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
@@ -2559,21 +2748,28 @@ function handleChatCompletionsRequest(req, res, opts) {
2559
2748
  });
2560
2749
  }
2561
2750
  function adaptAnthropicToOpenAI(payload) {
2751
+ const sysParts = [];
2562
2752
  if (payload.system != null) {
2563
- let sysText = "";
2564
2753
  if (typeof payload.system === "string") {
2565
- sysText = payload.system;
2754
+ if (payload.system) sysParts.push(payload.system);
2566
2755
  } else if (Array.isArray(payload.system)) {
2567
- sysText = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
2568
- }
2569
- if (sysText) {
2570
- if (!Array.isArray(payload.messages)) payload.messages = [];
2571
- if (!payload.messages[0] || payload.messages[0].role !== "system") {
2572
- payload.messages.unshift({ role: "system", content: sysText });
2573
- }
2756
+ const text = payload.system.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n");
2757
+ if (text) sysParts.push(text);
2574
2758
  }
2575
2759
  delete payload.system;
2576
2760
  }
2761
+ if (Array.isArray(payload.messages)) {
2762
+ payload.messages = payload.messages.filter((m) => {
2763
+ if (m?.role !== "system" && m?.role !== "developer") return true;
2764
+ const text = typeof m.content === "string" ? m.content : Array.isArray(m.content) ? m.content.map((b) => typeof b === "string" ? b : b?.text ?? "").filter(Boolean).join("\n\n") : "";
2765
+ if (text) sysParts.push(text);
2766
+ return false;
2767
+ });
2768
+ }
2769
+ if (sysParts.length) {
2770
+ if (!Array.isArray(payload.messages)) payload.messages = [];
2771
+ payload.messages.unshift({ role: "system", content: sysParts.join("\n\n") });
2772
+ }
2577
2773
  if (payload.tool_choice && typeof payload.tool_choice === "object") {
2578
2774
  const tc = payload.tool_choice;
2579
2775
  if (tc.type === "any") {
@@ -3363,6 +3559,8 @@ var server = createServer((req, res) => {
3363
3559
  let cred;
3364
3560
  let headers = { "Content-Type": "application/json" };
3365
3561
  payload.model = resolvedTarget.model;
3562
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
3563
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
3366
3564
  trimTools(payload, {
3367
3565
  provider: resolvedTarget.provider,
3368
3566
  queryTools,
@@ -3388,6 +3586,7 @@ var server = createServer((req, res) => {
3388
3586
  cred = await resolveUpstreamCredential(resolvedTarget.provider);
3389
3587
  targetApiKey = cred?.value ?? "";
3390
3588
  if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
3589
+ if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
3391
3590
  const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
3392
3591
  adaptAnthropicToOpenAI(payload);
3393
3592
  applyDefaultRequestFields(
@@ -3584,6 +3783,17 @@ var server = createServer((req, res) => {
3584
3783
  );
3585
3784
  }
3586
3785
  });
3786
+ if (status < 200 || status >= 300) {
3787
+ let errBody = "";
3788
+ proxyRes.on("data", (c) => {
3789
+ if (errBody.length < 300) errBody += c.toString("utf8");
3790
+ });
3791
+ proxyRes.on("end", () => {
3792
+ console.error(
3793
+ `[Proxy] upstream error: ${resolvedTarget.provider}:${resolvedTarget.model} status=${status} body=${JSON.stringify(errBody.slice(0, 300))}`
3794
+ );
3795
+ });
3796
+ }
3587
3797
  const contentType = proxyRes.headers["content-type"] || "";
3588
3798
  const isStreaming = payload.stream === true && /text\/event-stream/i.test(contentType);
3589
3799
  const retryEmptyTurn = () => {
@@ -3709,6 +3919,6 @@ function start(port = PORT) {
3709
3919
  });
3710
3920
  }
3711
3921
 
3712
- export { CANONICAL_UPSTREAMS, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, describeUpstreamStatus, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
3922
+ export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, DEFAULT_LOCAL_PORTS, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
3713
3923
  //# sourceMappingURL=index.mjs.map
3714
3924
  //# sourceMappingURL=index.mjs.map