@agentproto/llm-endpoint 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.mjs +37 -3
- package/dist/cli.mjs.map +1 -1
- package/dist/index.mjs +37 -3
- package/dist/index.mjs.map +1 -1
- package/package.json +1 -1
package/dist/cli.mjs
CHANGED
|
@@ -1895,6 +1895,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
|
|
|
1895
1895
|
const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
|
|
1896
1896
|
return parseConfigurableUpstreamUrl(value, "nebius");
|
|
1897
1897
|
}
|
|
1898
|
+
var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
|
|
1899
|
+
function parseDeviceProvider(provider) {
|
|
1900
|
+
const m = DEVICE_PROVIDER_RE.exec(provider);
|
|
1901
|
+
return m ? { endpointId: m[1], device: m[2] } : null;
|
|
1902
|
+
}
|
|
1903
|
+
function resolveDeviceProviderSpec(device) {
|
|
1904
|
+
const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
|
|
1905
|
+
return {
|
|
1906
|
+
keyRequired: true,
|
|
1907
|
+
apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
|
|
1908
|
+
resolveUpstream: () => {
|
|
1909
|
+
const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
|
|
1910
|
+
if (!daemonUrl) return null;
|
|
1911
|
+
const base = parseUpstreamUrl(daemonUrl);
|
|
1912
|
+
if (!base) return null;
|
|
1913
|
+
return {
|
|
1914
|
+
...base,
|
|
1915
|
+
pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
|
|
1916
|
+
};
|
|
1917
|
+
},
|
|
1918
|
+
unavailableMessage
|
|
1919
|
+
};
|
|
1920
|
+
}
|
|
1898
1921
|
function getConfigurableProviderSpec(provider) {
|
|
1899
1922
|
const staticSpec = CONFIGURABLE_PROVIDERS[provider];
|
|
1900
1923
|
if (staticSpec) {
|
|
@@ -1916,6 +1939,8 @@ function getConfigurableProviderSpec(provider) {
|
|
|
1916
1939
|
unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
|
|
1917
1940
|
};
|
|
1918
1941
|
}
|
|
1942
|
+
const deviceRoute = parseDeviceProvider(provider);
|
|
1943
|
+
if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
|
|
1919
1944
|
return void 0;
|
|
1920
1945
|
}
|
|
1921
1946
|
function applyDefaultRequestFields(payload, defaults, skipKeys) {
|
|
@@ -2017,7 +2042,7 @@ async function probeFileEndpointModels(endpoint) {
|
|
|
2017
2042
|
return result;
|
|
2018
2043
|
}
|
|
2019
2044
|
function isKnownProvider(provider) {
|
|
2020
|
-
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
|
|
2045
|
+
return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
|
|
2021
2046
|
}
|
|
2022
2047
|
function applyProviderOverride(target, providerOverride) {
|
|
2023
2048
|
const route = { provider: target.provider, model: target.model };
|
|
@@ -2034,7 +2059,7 @@ function parseAnyTransparentModel(model) {
|
|
|
2034
2059
|
const slashIdx = model.indexOf("/");
|
|
2035
2060
|
if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
|
|
2036
2061
|
const provider = model.slice(0, slashIdx);
|
|
2037
|
-
if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
|
|
2062
|
+
if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
|
|
2038
2063
|
return { provider, model: model.slice(slashIdx + 1) };
|
|
2039
2064
|
}
|
|
2040
2065
|
function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
|
|
@@ -2506,6 +2531,8 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2506
2531
|
return;
|
|
2507
2532
|
}
|
|
2508
2533
|
payload.model = resolvedTarget.model;
|
|
2534
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
2535
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
2509
2536
|
trimTools(payload, {
|
|
2510
2537
|
provider: resolvedTarget.provider,
|
|
2511
2538
|
queryTools: opts.queryTools,
|
|
@@ -2545,7 +2572,11 @@ function handleChatCompletionsRequest(req, res, opts) {
|
|
|
2545
2572
|
method: "POST",
|
|
2546
2573
|
headers: {
|
|
2547
2574
|
"Content-Type": "application/json",
|
|
2548
|
-
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
|
|
2575
|
+
...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
|
|
2576
|
+
// See the /v1/messages handler's identical header — the daemon's
|
|
2577
|
+
// /devices/:id/exec-stream 400s without it, always POST outer verb
|
|
2578
|
+
// notwithstanding.
|
|
2579
|
+
...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
|
|
2549
2580
|
}
|
|
2550
2581
|
};
|
|
2551
2582
|
const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
|
|
@@ -3382,6 +3413,8 @@ var server = createServer((req, res) => {
|
|
|
3382
3413
|
let cred;
|
|
3383
3414
|
let headers = { "Content-Type": "application/json" };
|
|
3384
3415
|
payload.model = resolvedTarget.model;
|
|
3416
|
+
const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
|
|
3417
|
+
if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
|
|
3385
3418
|
trimTools(payload, {
|
|
3386
3419
|
provider: resolvedTarget.provider,
|
|
3387
3420
|
queryTools,
|
|
@@ -3407,6 +3440,7 @@ var server = createServer((req, res) => {
|
|
|
3407
3440
|
cred = await resolveUpstreamCredential(resolvedTarget.provider);
|
|
3408
3441
|
targetApiKey = cred?.value ?? "";
|
|
3409
3442
|
if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
|
|
3443
|
+
if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
|
|
3410
3444
|
const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
|
|
3411
3445
|
adaptAnthropicToOpenAI(payload);
|
|
3412
3446
|
applyDefaultRequestFields(
|