@agentproto/llm-endpoint 0.9.0 → 0.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.mjs CHANGED
@@ -1895,6 +1895,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
1895
1895
  const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
1896
1896
  return parseConfigurableUpstreamUrl(value, "nebius");
1897
1897
  }
1898
+ var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
1899
+ function parseDeviceProvider(provider) {
1900
+ const m = DEVICE_PROVIDER_RE.exec(provider);
1901
+ return m ? { endpointId: m[1], device: m[2] } : null;
1902
+ }
1903
+ function resolveDeviceProviderSpec(device) {
1904
+ const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
1905
+ return {
1906
+ keyRequired: true,
1907
+ apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
1908
+ resolveUpstream: () => {
1909
+ const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
1910
+ if (!daemonUrl) return null;
1911
+ const base = parseUpstreamUrl(daemonUrl);
1912
+ if (!base) return null;
1913
+ return {
1914
+ ...base,
1915
+ pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
1916
+ };
1917
+ },
1918
+ unavailableMessage
1919
+ };
1920
+ }
1898
1921
  function getConfigurableProviderSpec(provider) {
1899
1922
  const staticSpec = CONFIGURABLE_PROVIDERS[provider];
1900
1923
  if (staticSpec) {
@@ -1916,6 +1939,8 @@ function getConfigurableProviderSpec(provider) {
1916
1939
  unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
1917
1940
  };
1918
1941
  }
1942
+ const deviceRoute = parseDeviceProvider(provider);
1943
+ if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
1919
1944
  return void 0;
1920
1945
  }
1921
1946
  function applyDefaultRequestFields(payload, defaults, skipKeys) {
@@ -2017,7 +2042,7 @@ async function probeFileEndpointModels(endpoint) {
2017
2042
  return result;
2018
2043
  }
2019
2044
  function isKnownProvider(provider) {
2020
- return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
2045
+ return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
2021
2046
  }
2022
2047
  function applyProviderOverride(target, providerOverride) {
2023
2048
  const route = { provider: target.provider, model: target.model };
@@ -2034,7 +2059,7 @@ function parseAnyTransparentModel(model) {
2034
2059
  const slashIdx = model.indexOf("/");
2035
2060
  if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
2036
2061
  const provider = model.slice(0, slashIdx);
2037
- if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
2062
+ if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
2038
2063
  return { provider, model: model.slice(slashIdx + 1) };
2039
2064
  }
2040
2065
  function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
@@ -2506,6 +2531,8 @@ function handleChatCompletionsRequest(req, res, opts) {
2506
2531
  return;
2507
2532
  }
2508
2533
  payload.model = resolvedTarget.model;
2534
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
2535
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
2509
2536
  trimTools(payload, {
2510
2537
  provider: resolvedTarget.provider,
2511
2538
  queryTools: opts.queryTools,
@@ -2545,7 +2572,11 @@ function handleChatCompletionsRequest(req, res, opts) {
2545
2572
  method: "POST",
2546
2573
  headers: {
2547
2574
  "Content-Type": "application/json",
2548
- ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
2575
+ ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
2576
+ // See the /v1/messages handler's identical header — the daemon's
2577
+ // /devices/:id/exec-stream 400s without it, always POST outer verb
2578
+ // notwithstanding.
2579
+ ...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
2549
2580
  }
2550
2581
  };
2551
2582
  const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
@@ -3382,6 +3413,8 @@ var server = createServer((req, res) => {
3382
3413
  let cred;
3383
3414
  let headers = { "Content-Type": "application/json" };
3384
3415
  payload.model = resolvedTarget.model;
3416
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
3417
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
3385
3418
  trimTools(payload, {
3386
3419
  provider: resolvedTarget.provider,
3387
3420
  queryTools,
@@ -3407,6 +3440,7 @@ var server = createServer((req, res) => {
3407
3440
  cred = await resolveUpstreamCredential(resolvedTarget.provider);
3408
3441
  targetApiKey = cred?.value ?? "";
3409
3442
  if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
3443
+ if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
3410
3444
  const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
3411
3445
  adaptAnthropicToOpenAI(payload);
3412
3446
  applyDefaultRequestFields(