@agentproto/llm-endpoint 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.mjs +37 -3
- package/dist/cli.mjs.map +1 -1
- package/dist/index.mjs +37 -3
- package/dist/index.mjs.map +1 -1
- package/package.json +1 -1
package/dist/index.mjs
CHANGED
|
@@ -2041,6 +2041,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
|
|
|
2041
2041
|
const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
|
|
2042
2042
|
return parseConfigurableUpstreamUrl(value, "nebius");
|
|
2043
2043
|
}
|
|
2044
|
+
var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
|
|
2045
|
+
function parseDeviceProvider(provider) {
|
|
2046
|
+
const m = DEVICE_PROVIDER_RE.exec(provider);
|
|
2047
|
+
return m ? { endpointId: m[1], device: m[2] } : null;
|
|
2048
|
+
}
|
|
2049
|
+
function resolveDeviceProviderSpec(device) {
|
|
2050
|
+
const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
|
|
2051
|
+
return {
|
|
2052
|
+
keyRequired: true,
|
|
2053
|
+
apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
|
|
2054
|
+
resolveUpstream: () => {
|
|
2055
|
+
const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
|
|
2056
|
+
if (!daemonUrl) return null;
|
|
2057
|
+
const base = parseUpstreamUrl(daemonUrl);
|
|
2058
|
+
if (!base) return null;
|
|
2059
|
+
return {
|
|
2060
|
+
...base,
|
|
2061
|
+
pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
|
|
2062
|
+
};
|
|
2063
|
+
},
|
|
2064
|
+
unavailableMessage
|
|
2065
|
+
};
|
|
2066
|
+
}
|
|
2044
2067
|
function getConfigurableProviderSpec(provider) {
|
|
2045
2068
|
const staticSpec = CONFIGURABLE_PROVIDERS[provider];
|
|
2046
2069
|
if (staticSpec) {
|
|
@@ -2062,6 +2085,8 @@ function getConfigurableProviderSpec(provider) {
|
|
|
2062
2085
|
unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
|
|
2063
2086
|
};
|
|
2064
2087
|
}
|
|
2088
|
+
const deviceRoute = parseDeviceProvider(provider);
|
|
2089
|
+
if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
|
|
2065
2090
|
return void 0;
|
|
2066
2091
|
}
|
|
2067
2092
|
function applyDefaultRequestFields(payload, defaults, skipKeys) {
|
|
@@ -2163,7 +2188,7 @@ async function probeFileEndpointModels(endpoint) {
|
|
|
2163
2188
|
return result;
|
|
2164
2189
|
}
|
|
2165
2190
|
function isKnownProvider(provider) {
|
|
2166
|
-
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
|
|
2191
|
+
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
|
|
2167
2192
|
}
|
|
2168
2193
|
function applyProviderOverride(target, providerOverride) {
|
|
2169
2194
|
const route = { provider: target.provider, model: target.model };
|
|
@@ -2180,7 +2205,7 @@ function parseAnyTransparentModel(model) {
|
|
|
2180
2205
|
const slashIdx = model.indexOf("/");
|
|
2181
2206
|
if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
|
|
2182
2207
|
const provider = model.slice(0, slashIdx);
|
|
2183
|
-
if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
|
|
2208
|
+
if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
|
|
2184
2209
|
return { provider, model: model.slice(slashIdx + 1) };
|
|
2185
2210
|
}
|
|
2186
2211
|
function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
|
|
@@ -2652,6 +2677,8 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2652
2677
|
return;
|
|
2653
2678
|
}
|
|
2654
2679
|
payload.model = resolvedTarget.model;
|
|
2680
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
2681
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
2655
2682
|
trimTools(payload, {
|
|
2656
2683
|
provider: resolvedTarget.provider,
|
|
2657
2684
|
queryTools: opts.queryTools,
|
|
@@ -2691,7 +2718,11 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2691
2718
|
method: "POST",
|
|
2692
2719
|
headers: {
|
|
2693
2720
|
"Content-Type": "application/json",
|
|
2694
|
-
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
|
|
2721
|
+
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
|
|
2722
|
+
// See the /v1/messages handler's identical header — the daemon's
|
|
2723
|
+
// /devices/:id/exec-stream 400s without it, always POST outer verb
|
|
2724
|
+
// notwithstanding.
|
|
2725
|
+
...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
|
|
2695
2726
|
}
|
|
2696
2727
|
};
|
|
2697
2728
|
const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
|
|
@@ -3528,6 +3559,8 @@ var server = createServer((req, res) => {
|
|
|
3528
3559
|
let cred;
|
|
3529
3560
|
let headers = { "Content-Type": "application/json" };
|
|
3530
3561
|
payload.model = resolvedTarget.model;
|
|
3562
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
3563
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
3531
3564
|
trimTools(payload, {
|
|
3532
3565
|
provider: resolvedTarget.provider,
|
|
3533
3566
|
queryTools,
|
|
@@ -3553,6 +3586,7 @@ var server = createServer((req, res) => {
|
|
|
3553
3586
|
cred = await resolveUpstreamCredential(resolvedTarget.provider);
|
|
3554
3587
|
targetApiKey = cred?.value ?? "";
|
|
3555
3588
|
if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
|
|
3589
|
+
if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
|
|
3556
3590
|
const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
|
|
3557
3591
|
adaptAnthropicToOpenAI(payload);
|
|
3558
3592
|
applyDefaultRequestFields(
|