@agentproto/llm-endpoint 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs CHANGED
@@ -2041,6 +2041,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
2041
2041
  const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
2042
2042
  return parseConfigurableUpstreamUrl(value, "nebius");
2043
2043
  }
2044
+ var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
2045
+ function parseDeviceProvider(provider) {
2046
+ const m = DEVICE_PROVIDER_RE.exec(provider);
2047
+ return m ? { endpointId: m[1], device: m[2] } : null;
2048
+ }
2049
+ function resolveDeviceProviderSpec(device) {
2050
+ const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
2051
+ return {
2052
+ keyRequired: true,
2053
+ apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
2054
+ resolveUpstream: () => {
2055
+ const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
2056
+ if (!daemonUrl) return null;
2057
+ const base = parseUpstreamUrl(daemonUrl);
2058
+ if (!base) return null;
2059
+ return {
2060
+ ...base,
2061
+ pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
2062
+ };
2063
+ },
2064
+ unavailableMessage
2065
+ };
2066
+ }
2044
2067
  function getConfigurableProviderSpec(provider) {
2045
2068
  const staticSpec = CONFIGURABLE_PROVIDERS[provider];
2046
2069
  if (staticSpec) {
@@ -2062,6 +2085,8 @@ function getConfigurableProviderSpec(provider) {
2062
2085
  unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
2063
2086
  };
2064
2087
  }
2088
+ const deviceRoute = parseDeviceProvider(provider);
2089
+ if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
2065
2090
  return void 0;
2066
2091
  }
2067
2092
  function applyDefaultRequestFields(payload, defaults, skipKeys) {
@@ -2163,7 +2188,7 @@ async function probeFileEndpointModels(endpoint) {
2163
2188
  return result;
2164
2189
  }
2165
2190
  function isKnownProvider(provider) {
2166
- return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
2191
+ return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
2167
2192
  }
2168
2193
  function applyProviderOverride(target, providerOverride) {
2169
2194
  const route = { provider: target.provider, model: target.model };
@@ -2180,7 +2205,7 @@ function parseAnyTransparentModel(model) {
2180
2205
  const slashIdx = model.indexOf("/");
2181
2206
  if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
2182
2207
  const provider = model.slice(0, slashIdx);
2183
- if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
2208
+ if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
2184
2209
  return { provider, model: model.slice(slashIdx + 1) };
2185
2210
  }
2186
2211
  function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
@@ -2652,6 +2677,8 @@ function handleChatCompletionsRequest(req, res, opts) {
2652
2677
  return;
2653
2678
  }
2654
2679
  payload.model = resolvedTarget.model;
2680
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
2681
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
2655
2682
  trimTools(payload, {
2656
2683
  provider: resolvedTarget.provider,
2657
2684
  queryTools: opts.queryTools,
@@ -2691,7 +2718,11 @@ function handleChatCompletionsRequest(req, res, opts) {
2691
2718
  method: "POST",
2692
2719
  headers: {
2693
2720
  "Content-Type": "application/json",
2694
- ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
2721
+ ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
2722
+ // See the /v1/messages handler's identical header — the daemon's
2723
+ // /devices/:id/exec-stream 400s without it, always POST outer verb
2724
+ // notwithstanding.
2725
+ ...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
2695
2726
  }
2696
2727
  };
2697
2728
  const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
@@ -3528,6 +3559,8 @@ var server = createServer((req, res) => {
3528
3559
  let cred;
3529
3560
  let headers = { "Content-Type": "application/json" };
3530
3561
  payload.model = resolvedTarget.model;
3562
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
3563
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
3531
3564
  trimTools(payload, {
3532
3565
  provider: resolvedTarget.provider,
3533
3566
  queryTools,
@@ -3553,6 +3586,7 @@ var server = createServer((req, res) => {
3553
3586
  cred = await resolveUpstreamCredential(resolvedTarget.provider);
3554
3587
  targetApiKey = cred?.value ?? "";
3555
3588
  if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
3589
+ if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
3556
3590
  const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
3557
3591
  adaptAnthropicToOpenAI(payload);
3558
3592
  applyDefaultRequestFields(