@agentproto/llm-endpoint 0.9.0 → 0.11.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -228,6 +228,146 @@ interface ConfigurableUpstream {
228
228
  pathPrefix: string;
229
229
  }
230
230
 
231
+ /**
232
+ * Harness fit check — whether an agent harness's known first-request token
233
+ * size fits inside an inference endpoint's LOADED context, checked BEFORE a
234
+ * session is spawned against it. A harness with a big tool-preamble (e.g.
235
+ * claude-code) can otherwise be spawned blind against a small local model and
236
+ * die with an opaque upstream 400/500 minutes later.
237
+ *
238
+ * Dependency-free and side-effect-free at import time, like `connectors.ts`/
239
+ * `endpoints.ts`/`packs.ts` — both `@agentproto/cli` and `@agentproto/runtime`
240
+ * import this directly, so it stays a pure leaf module.
241
+ */
242
+ type FitVerdict = 'fits' | 'no-fit' | 'unknown';
243
+ interface HarnessFirstRequestSize {
244
+ /** Measured (or best-known) first-request token count for this harness. */
245
+ tokens: number;
246
+ /** Where the number came from — printed alongside it in `doctor`/PR bodies
247
+ * so a stale or unmeasured figure is never mistaken for a fresh one. */
248
+ source: string;
249
+ }
250
+ /**
251
+ * First-request token size per harness/adapter slug — the size of the FIRST
252
+ * request a fresh session of that harness sends (tool defs + system preamble
253
+ * + the first user turn), before any real conversation has accumulated.
254
+ *
255
+ * `claude-sdk`'s figure predates SPIKE-A (not re-measured there, carried over
256
+ * from the design doc's "deferred mode" measurement). `pi`'s figure
257
+ * SUPERSEDES an earlier ~2k design-doc estimate once SPIKE-A actually
258
+ * measured it against a real endpoint. Harnesses absent from this table
259
+ * (e.g. `opencode`) have never been measured — callers must treat that as
260
+ * `unknown` (see {@link checkHarnessFit}), never guess a number.
261
+ */
262
+ declare const HARNESS_FIRST_REQUEST_SIZE: Readonly<Record<string, HarnessFirstRequestSize>>;
263
+ /** Default headroom: the endpoint's loaded ctx must be at least
264
+ * `requiredTokens / (1 - DEFAULT_HEADROOM_RATIO)` — i.e. the first request
265
+ * may use at most `1 - DEFAULT_HEADROOM_RATIO` of the loaded ctx. Matches
266
+ * the plan's own worked example (36k required, 25% headroom ⇒ "ctx >= 48k",
267
+ * since 36000 / 0.75 = 48000). */
268
+ declare const DEFAULT_HEADROOM_RATIO = 0.25;
269
+ interface FitCheckInput {
270
+ /** Adapter/harness slug, e.g. "claude-code", "pi". */
271
+ harness: string;
272
+ /** The endpoint's loaded ctx (from a connector's `listModels().loadedCtx`),
273
+ * or `undefined` when the runtime doesn't expose one (e.g. Ollama). */
274
+ loadedCtx: number | undefined;
275
+ /** Fraction of the loaded ctx to hold back as headroom. Default 25%. */
276
+ headroomRatio?: number;
277
+ /** Human label for the endpoint, used in the actionable message — e.g.
278
+ * `"lmstudio@win-pc"`. Defaults to a generic phrase when omitted. */
279
+ endpointLabel?: string;
280
+ }
281
+ interface FitCheckResult {
282
+ verdict: FitVerdict;
283
+ /** The harness's known first-request size, when known. */
284
+ requiredTokens?: number;
285
+ /** The endpoint's loaded ctx, echoed back, when known. */
286
+ loadedCtx?: number;
287
+ /** The effective minimum loaded ctx this harness needs at the applied
288
+ * headroom — `requiredTokens / (1 - headroomRatio)`, rounded up. */
289
+ thresholdTokens?: number;
290
+ headroomRatio: number;
291
+ /** Actionable, human-readable explanation — present on `no-fit` and
292
+ * `unknown` (absent on `fits`, nothing to explain). */
293
+ message?: string;
294
+ }
295
+ /**
296
+ * Compare a harness's known first-request size against an endpoint's loaded
297
+ * ctx. Never throws. Returns `unknown` (never blocks a caller on its own —
298
+ * see the plan's "`unknown` must not block but must warn" rule) when either
299
+ * the harness's size or the endpoint's loaded ctx isn't known; `no-fit` with
300
+ * an actionable message when the harness's first request would not fit
301
+ * inside the loaded ctx at the requested headroom; `fits` otherwise.
302
+ */
303
+ declare function checkHarnessFit(input: FitCheckInput): FitCheckResult;
304
+
305
+ /**
306
+ * pi wiring — generate/update the matching provider entries in
307
+ * `~/.pi/agent/models.json` (the `pi` harness's own model registry) from the
308
+ * configured named endpoints + their connectors, so a locally loaded model
309
+ * is usable from `pi` without a hand edit.
310
+ *
311
+ * Lives in `@agentproto/llm-endpoint` (not `@agentproto/cli`, where it
312
+ * started) so BOTH the CLI (`agentproto llm endpoints sync-pi`) and the
313
+ * daemon (`packages/runtime`'s `inference` session-binding, which must
314
+ * sync before spawning a `pi` session against a local endpoint) can call it
315
+ * directly — `packages/runtime` never depends on `@agentproto/cli`.
316
+ *
317
+ * Two rules drive the shape of this: `contextWindow` must be the model's
318
+ * LOADED context (what's actually usable right now), never its max — a
319
+ * stale max value is what caused `~/.pi/agent/models.json` to drift out of
320
+ * sync by hand in practice; and this must never clobber a user's own
321
+ * hand-added provider/model entries. One connector (Ollama) never reports a
322
+ * loaded context size at all — `DEFAULT_CONTEXT_FALLBACK` below covers that
323
+ * gap with a conservative guess rather than treating a genuinely loaded
324
+ * model as nothing-to-sync. Ownership of what THIS module wrote is
325
+ * tracked in a side ledger (`~/.agentproto/pi-models-managed.json`), never
326
+ * inferred from `models.json`'s own content — that keeps "is this ours" a
327
+ * recorded fact instead of a guess, and means no extra marker field has to
328
+ * ride along in a file `pi` itself parses.
329
+ */
330
+ interface PiModelEntry {
331
+ id: string;
332
+ name: string;
333
+ reasoning: boolean;
334
+ input: string[];
335
+ contextWindow: number;
336
+ maxTokens: number;
337
+ cost: {
338
+ input: number;
339
+ output: number;
340
+ cacheRead: number;
341
+ cacheWrite: number;
342
+ };
343
+ }
344
+ declare function resolvePiModelsFilePath(): string;
345
+ declare function resolvePiLedgerFilePath(): string;
346
+ interface PiSyncEntry {
347
+ providerId: string;
348
+ baseUrl: string;
349
+ action: 'added' | 'updated' | 'unchanged' | 'skipped-no-loaded-models';
350
+ modelIds: string[];
351
+ }
352
+ interface PiSyncResult {
353
+ modelsPath: string;
354
+ ledgerPath: string;
355
+ entries: PiSyncEntry[];
356
+ }
357
+ interface PiSyncOptions {
358
+ dryRun?: boolean;
359
+ fetchImpl?: typeof fetch;
360
+ }
361
+ /**
362
+ * Regenerate every configured named endpoint's provider entry in pi's
363
+ * `models.json` from its connector's LIVE loaded models. Never touches a
364
+ * provider id with no configured endpoint (nothing to regenerate it from),
365
+ * and within a touched provider, only ever adds/updates/removes the model
366
+ * ids THIS module previously wrote (per the ledger) — any other model in
367
+ * that same provider (hand-added by the user) is left exactly as-is.
368
+ */
369
+ declare function syncPiModels(opts?: PiSyncOptions): Promise<PiSyncResult>;
370
+
231
371
  interface ToolTrimOptions {
232
372
  provider: string;
233
373
  queryTools: string | null;
@@ -501,4 +641,4 @@ declare const server: node_http.Server<typeof IncomingMessage, typeof ServerResp
501
641
  /** Démarre le proxy sur `port` (défaut : {@link PORT}). Renvoie le serveur en écoute. */
502
642
  declare function start(port?: number): node_http.Server<typeof IncomingMessage, typeof ServerResponse>;
503
643
 
504
- export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, type ConfigurableUpstream, type Connector, type ConnectorId, type ConnectorModel, type ConnectorModelState, type ConnectorQuirks, DEFAULT_LOCAL_PORTS, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type ForgeUpstream, type ModelRouteContext, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
644
+ export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, type ConfigurableUpstream, type Connector, type ConnectorId, type ConnectorModel, type ConnectorModelState, type ConnectorQuirks, DEFAULT_HEADROOM_RATIO, DEFAULT_LOCAL_PORTS, type EndpointConfig, type EndpointDefaultRequestFields, type EndpointsFileLoad, type FitCheckInput, type FitCheckResult, type FitVerdict, type ForgeUpstream, HARNESS_FIRST_REQUEST_SIZE, type HarnessFirstRequestSize, type ModelRouteContext, type PiModelEntry, type PiSyncEntry, type PiSyncOptions, type PiSyncResult, type ToolTrimOptions, type UpstreamCredential, type UpstreamSource, type UpstreamStatus, type UpstreamTestResult, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, checkHarnessFit, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolvePiLedgerFilePath, resolvePiModelsFilePath, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, syncPiModels, testUpstream, trimTools };
package/dist/index.mjs CHANGED
@@ -1,7 +1,7 @@
1
1
  import { createServer, request } from 'http';
2
2
  import { request as request$1 } from 'https';
3
3
  import { readFileSync } from 'fs';
4
- import path, { resolve } from 'path';
4
+ import path, { resolve, join, dirname } from 'path';
5
5
  import { fileURLToPath } from 'url';
6
6
  import { randomBytes, createHash } from 'crypto';
7
7
  import { getAuthProfile, KeychainStore } from '@agentproto/auth';
@@ -1838,6 +1838,176 @@ function parseUpstreamUrl(value) {
1838
1838
  return { hostname: parsed.hostname, port, protocol, pathPrefix };
1839
1839
  }
1840
1840
 
1841
+ // src/harness-fit.ts
1842
+ var HARNESS_FIRST_REQUEST_SIZE = {
1843
+ "claude-code": {
1844
+ tokens: 35925,
1845
+ source: "SPIKE-A-RESULTS.md \u2014 two real runs against a local endpoint, 35,925/35,924 tokens (2026-09-27)"
1846
+ },
1847
+ "claude-sdk": {
1848
+ tokens: 18200,
1849
+ source: 'INFERENCE-ENDPOINTS-DESIGN.md \u2014 "deferred mode" measurement (2026-09-27); not re-measured in SPIKE-A, may be stale'
1850
+ },
1851
+ pi: {
1852
+ tokens: 7504,
1853
+ source: "SPIKE-A-RESULTS.md \u2014 real cross-device runs, 7,466-7,504 tokens (2026-09-27); supersedes an earlier ~2k design-doc estimate"
1854
+ }
1855
+ };
1856
+ var DEFAULT_HEADROOM_RATIO = 0.25;
1857
+ function checkHarnessFit(input) {
1858
+ const headroomRatio = input.headroomRatio ?? DEFAULT_HEADROOM_RATIO;
1859
+ const label = input.endpointLabel ?? "this endpoint";
1860
+ const size = HARNESS_FIRST_REQUEST_SIZE[input.harness];
1861
+ if (!size) {
1862
+ return {
1863
+ verdict: "unknown",
1864
+ headroomRatio,
1865
+ message: `"${input.harness}"'s first-request size has not been measured \u2014 proceeding without a fit check. If the spawn fails with a context-overflow error against ${label}, that is why.`
1866
+ };
1867
+ }
1868
+ const requiredTokens = size.tokens;
1869
+ const thresholdTokens = Math.ceil(requiredTokens / (1 - headroomRatio));
1870
+ if (input.loadedCtx === void 0) {
1871
+ return {
1872
+ verdict: "unknown",
1873
+ requiredTokens,
1874
+ thresholdTokens,
1875
+ headroomRatio,
1876
+ message: `${label}'s loaded context size is unknown (this runtime doesn't report it) \u2014 proceeding without a fit check. "${input.harness}" needs ~${formatK(requiredTokens)} tokens for its first request.`
1877
+ };
1878
+ }
1879
+ if (input.loadedCtx >= thresholdTokens) {
1880
+ return { verdict: "fits", requiredTokens, loadedCtx: input.loadedCtx, thresholdTokens, headroomRatio };
1881
+ }
1882
+ return {
1883
+ verdict: "no-fit",
1884
+ requiredTokens,
1885
+ loadedCtx: input.loadedCtx,
1886
+ thresholdTokens,
1887
+ headroomRatio,
1888
+ message: `"${input.harness}" needs ~${formatK(requiredTokens)} tokens for its first request, ${label} has ${formatK(input.loadedCtx)} loaded: use a harness with a smaller first request (e.g. pi), or reload the model with ctx >= ${formatK(thresholdTokens)}.`
1889
+ };
1890
+ }
1891
+ function formatK(tokens) {
1892
+ return `${Math.round(tokens / 1e3)}k`;
1893
+ }
1894
+ var DEFAULT_MAX_TOKENS = 8192;
1895
+ var DEFAULT_CONTEXT_FALLBACK = 4096;
1896
+ function resolvePiModelsFilePath() {
1897
+ const override = process.env.AGENTPROTO_PI_MODELS_FILE?.trim();
1898
+ if (override) return override;
1899
+ return join(homedir(), ".pi", "agent", "models.json");
1900
+ }
1901
+ function resolvePiLedgerFilePath() {
1902
+ const override = process.env.AGENTPROTO_PI_LEDGER_FILE?.trim();
1903
+ if (override) return override;
1904
+ return join(homedir(), ".agentproto", "pi-models-managed.json");
1905
+ }
1906
+ function isRecord2(v) {
1907
+ return typeof v === "object" && v !== null && !Array.isArray(v);
1908
+ }
1909
+ async function readJsonFile(path2, fallback) {
1910
+ let raw;
1911
+ try {
1912
+ raw = await readFile(path2, "utf-8");
1913
+ } catch {
1914
+ return fallback;
1915
+ }
1916
+ try {
1917
+ return JSON.parse(raw);
1918
+ } catch {
1919
+ return fallback;
1920
+ }
1921
+ }
1922
+ async function readPiModelsFile(path2) {
1923
+ const parsed = await readJsonFile(path2, { providers: {} });
1924
+ if (!isRecord2(parsed) || !isRecord2(parsed.providers)) return { providers: {} };
1925
+ const providers = {};
1926
+ for (const [id, raw] of Object.entries(parsed.providers)) {
1927
+ if (!isRecord2(raw) || typeof raw.baseUrl !== "string" || typeof raw.api !== "string" || typeof raw.apiKey !== "string") continue;
1928
+ providers[id] = { baseUrl: raw.baseUrl, api: raw.api, apiKey: raw.apiKey, models: Array.isArray(raw.models) ? raw.models : [] };
1929
+ }
1930
+ return { providers };
1931
+ }
1932
+ async function readLedger(path2) {
1933
+ const parsed = await readJsonFile(path2, { managed: {} });
1934
+ if (!isRecord2(parsed) || !isRecord2(parsed.managed)) return { managed: {} };
1935
+ const managed = {};
1936
+ for (const [id, ids] of Object.entries(parsed.managed)) {
1937
+ if (Array.isArray(ids)) managed[id] = ids.filter((x) => typeof x === "string");
1938
+ }
1939
+ return { managed };
1940
+ }
1941
+ async function writeJsonFile(path2, value) {
1942
+ await mkdir(dirname(path2), { recursive: true });
1943
+ await writeFile(path2, `${JSON.stringify(value, null, 2)}
1944
+ `);
1945
+ }
1946
+ function resolveApiKeyValue(endpoint) {
1947
+ if (endpoint.apiKeyEnv) {
1948
+ const value = process.env[endpoint.apiKeyEnv];
1949
+ if (value) return value;
1950
+ }
1951
+ return "not-needed";
1952
+ }
1953
+ async function buildDesiredModels(endpoint, fetchImpl) {
1954
+ const connectorId = endpoint.connector ?? "openai-compatible";
1955
+ const connector = connectorById(connectorId);
1956
+ if (!connector) return [];
1957
+ const models = await connector.listModels(endpoint.baseUrl, fetchImpl);
1958
+ return models.filter((m) => m.state === "loaded").map((m) => ({
1959
+ id: m.id,
1960
+ name: `${m.id} (${connector.label})`,
1961
+ reasoning: false,
1962
+ input: ["text"],
1963
+ // `loadedCtx` is only absent for a connector (Ollama) whose API never
1964
+ // reports it — see DEFAULT_CONTEXT_FALLBACK above.
1965
+ contextWindow: typeof m.loadedCtx === "number" ? m.loadedCtx : DEFAULT_CONTEXT_FALLBACK,
1966
+ maxTokens: DEFAULT_MAX_TOKENS,
1967
+ cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }
1968
+ }));
1969
+ }
1970
+ async function syncPiModels(opts = {}) {
1971
+ const fetchImpl = opts.fetchImpl ?? fetch;
1972
+ const modelsPath = resolvePiModelsFilePath();
1973
+ const ledgerPath = resolvePiLedgerFilePath();
1974
+ const endpoints = getConfiguredEndpoints();
1975
+ const piFile = await readPiModelsFile(modelsPath);
1976
+ const ledger = await readLedger(ledgerPath);
1977
+ const entries = [];
1978
+ const nextManaged = { ...ledger.managed };
1979
+ for (const endpoint of endpoints) {
1980
+ const desired = await buildDesiredModels(endpoint, fetchImpl);
1981
+ const previouslyManaged = new Set(ledger.managed[endpoint.id] ?? []);
1982
+ const existingProvider = piFile.providers[endpoint.id];
1983
+ const keptUserModels = (existingProvider?.models ?? []).filter((m) => !previouslyManaged.has(m.id));
1984
+ if (desired.length === 0 && !existingProvider) {
1985
+ entries.push({ providerId: endpoint.id, baseUrl: endpoint.baseUrl, action: "skipped-no-loaded-models", modelIds: [] });
1986
+ continue;
1987
+ }
1988
+ const newModels = [...keptUserModels, ...desired];
1989
+ const unchanged = existingProvider !== void 0 && existingProvider.baseUrl === endpoint.baseUrl && JSON.stringify([...existingProvider.models].sort((a, b) => a.id.localeCompare(b.id))) === JSON.stringify([...newModels].sort((a, b) => a.id.localeCompare(b.id)));
1990
+ piFile.providers[endpoint.id] = {
1991
+ baseUrl: endpoint.baseUrl,
1992
+ api: "openai-completions",
1993
+ apiKey: resolveApiKeyValue(endpoint),
1994
+ models: newModels
1995
+ };
1996
+ nextManaged[endpoint.id] = desired.map((m) => m.id);
1997
+ entries.push({
1998
+ providerId: endpoint.id,
1999
+ baseUrl: endpoint.baseUrl,
2000
+ action: unchanged ? "unchanged" : existingProvider ? "updated" : "added",
2001
+ modelIds: desired.map((m) => m.id)
2002
+ });
2003
+ }
2004
+ if (!opts.dryRun) {
2005
+ await writeJsonFile(modelsPath, piFile);
2006
+ await writeJsonFile(ledgerPath, { managed: nextManaged });
2007
+ }
2008
+ return { modelsPath, ledgerPath, entries };
2009
+ }
2010
+
1841
2011
  // src/index.ts
1842
2012
  var PORT = Number(process.env.LLM_ENDPOINT_PORT ?? process.env.PORT ?? 18090);
1843
2013
  var _mergedPackIdsCache = null;
@@ -2041,6 +2211,29 @@ function resolveNebiusBaseUrl(raw = process.env.NEBIUS_BASE_URL) {
2041
2211
  const value = raw?.trim() || CONFIGURABLE_PROVIDERS.nebius.defaultBaseUrl;
2042
2212
  return parseConfigurableUpstreamUrl(value, "nebius");
2043
2213
  }
2214
+ var DEVICE_PROVIDER_RE = /^(.+)@([^@/]+)$/;
2215
+ function parseDeviceProvider(provider) {
2216
+ const m = DEVICE_PROVIDER_RE.exec(provider);
2217
+ return m ? { endpointId: m[1], device: m[2] } : null;
2218
+ }
2219
+ function resolveDeviceProviderSpec(device) {
2220
+ const unavailableMessage = () => `"@${device}" device-inference routing requires this llm-endpoint sidecar to be started by an agentproto daemon (LLM_ENDPOINT_DAEMON_URL is unset) \u2014 run it via \`agentproto serve\` with features.llmEndpoint on, not as a bare standalone process.`;
2221
+ return {
2222
+ keyRequired: true,
2223
+ apiKeyEnv: "LLM_ENDPOINT_DAEMON_TOKEN",
2224
+ resolveUpstream: () => {
2225
+ const daemonUrl = process.env.LLM_ENDPOINT_DAEMON_URL?.trim();
2226
+ if (!daemonUrl) return null;
2227
+ const base = parseUpstreamUrl(daemonUrl);
2228
+ if (!base) return null;
2229
+ return {
2230
+ ...base,
2231
+ pathPrefix: `${base.pathPrefix}/devices/${encodeURIComponent(device)}/exec-stream/device-inference/v1`
2232
+ };
2233
+ },
2234
+ unavailableMessage
2235
+ };
2236
+ }
2044
2237
  function getConfigurableProviderSpec(provider) {
2045
2238
  const staticSpec = CONFIGURABLE_PROVIDERS[provider];
2046
2239
  if (staticSpec) {
@@ -2062,6 +2255,8 @@ function getConfigurableProviderSpec(provider) {
2062
2255
  unavailableMessage: () => `"${provider}" endpoint is misconfigured (invalid baseUrl).`
2063
2256
  };
2064
2257
  }
2258
+ const deviceRoute = parseDeviceProvider(provider);
2259
+ if (deviceRoute) return resolveDeviceProviderSpec(deviceRoute.device);
2065
2260
  return void 0;
2066
2261
  }
2067
2262
  function applyDefaultRequestFields(payload, defaults, skipKeys) {
@@ -2163,7 +2358,7 @@ async function probeFileEndpointModels(endpoint) {
2163
2358
  return result;
2164
2359
  }
2165
2360
  function isKnownProvider(provider) {
2166
- return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider);
2361
+ return KNOWN_PROVIDERS.has(provider) || getConfiguredEndpoints().some((e) => e.id === provider) || DEVICE_PROVIDER_RE.test(provider);
2167
2362
  }
2168
2363
  function applyProviderOverride(target, providerOverride) {
2169
2364
  const route = { provider: target.provider, model: target.model };
@@ -2180,7 +2375,7 @@ function parseAnyTransparentModel(model) {
2180
2375
  const slashIdx = model.indexOf("/");
2181
2376
  if (slashIdx <= 0 || slashIdx === model.length - 1) return null;
2182
2377
  const provider = model.slice(0, slashIdx);
2183
- if (!getConfiguredEndpoints().some((e) => e.id === provider)) return null;
2378
+ if (!getConfiguredEndpoints().some((e) => e.id === provider) && !DEVICE_PROVIDER_RE.test(provider)) return null;
2184
2379
  return { provider, model: model.slice(slashIdx + 1) };
2185
2380
  }
2186
2381
  function resolveModelRoute(payload, ctx, localPacks = getLocalPacks()) {
@@ -2652,6 +2847,8 @@ function handleChatCompletionsRequest(req, res, opts) {
2652
2847
  return;
2653
2848
  }
2654
2849
  payload.model = resolvedTarget.model;
2850
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
2851
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
2655
2852
  trimTools(payload, {
2656
2853
  provider: resolvedTarget.provider,
2657
2854
  queryTools: opts.queryTools,
@@ -2691,7 +2888,11 @@ function handleChatCompletionsRequest(req, res, opts) {
2691
2888
  method: "POST",
2692
2889
  headers: {
2693
2890
  "Content-Type": "application/json",
2694
- ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {}
2891
+ ...targetApiKey ? { "Authorization": `Bearer ${targetApiKey}` } : {},
2892
+ // See the /v1/messages handler's identical header — the daemon's
2893
+ // /devices/:id/exec-stream 400s without it, always POST outer verb
2894
+ // notwithstanding.
2895
+ ...deviceRoute ? { "x-agentproto-forward-method": "POST" } : {}
2695
2896
  }
2696
2897
  };
2697
2898
  const proxyReq = sendUpstreamRequest(protocol, options, (proxyRes) => {
@@ -3528,6 +3729,8 @@ var server = createServer((req, res) => {
3528
3729
  let cred;
3529
3730
  let headers = { "Content-Type": "application/json" };
3530
3731
  payload.model = resolvedTarget.model;
3732
+ const deviceRoute = parseDeviceProvider(resolvedTarget.provider);
3733
+ if (deviceRoute) payload.model = `${deviceRoute.endpointId}/${resolvedTarget.model}`;
3531
3734
  trimTools(payload, {
3532
3735
  provider: resolvedTarget.provider,
3533
3736
  queryTools,
@@ -3553,6 +3756,7 @@ var server = createServer((req, res) => {
3553
3756
  cred = await resolveUpstreamCredential(resolvedTarget.provider);
3554
3757
  targetApiKey = cred?.value ?? "";
3555
3758
  if (cred && cred.value) Object.assign(headers, buildUpstreamAuthHeaders(resolvedTarget.provider, cred));
3759
+ if (deviceRoute) headers["x-agentproto-forward-method"] = "POST";
3556
3760
  const clientThinkingEnabled = isRecord(payload.thinking) && payload.thinking.type === "enabled";
3557
3761
  adaptAnthropicToOpenAI(payload);
3558
3762
  applyDefaultRequestFields(
@@ -3885,6 +4089,6 @@ function start(port = PORT) {
3885
4089
  });
3886
4090
  }
3887
4091
 
3888
- export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, DEFAULT_LOCAL_PORTS, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, testUpstream, trimTools };
4092
+ export { CANONICAL_UPSTREAMS, CONNECTORS, CONNECTOR_IDS, DEFAULT_HEADROOM_RATIO, DEFAULT_LOCAL_PORTS, HARNESS_FIRST_REQUEST_SIZE, adaptAnthropicToOpenAI, buildUpstreamAuthHeaders, buildWafRuleExpression, checkHarnessFit, collectUpstreamStatuses, connectorById, describeUpstreamStatus, detectConnector, extractEdgeToken, extractInboundToken, getConfiguredEndpoints, isAuthorized, isCanonicalUpstream, isConnectorId, isCredentialAllowedOnOpenAiSurface, isEdgeAuthorized, isEmptyAnthropicTurn, isPublicModelListPath, normalizeProxyPath, openaiJsonToAnthropic, parseAccessTokens, parseEndpointsConfig, readEndpointsFromDisk, resetConfiguredEndpointsCache, resolveEmptyTurnRetries, resolveEndpointsFilePath, resolveForgeBaseUrl, resolveModelRoute, resolveNebiusBaseUrl, resolvePiLedgerFilePath, resolvePiModelsFilePath, resolveUpstreamCredential, server, start, stripThinkingFromAnthropicJson, syncPiModels, testUpstream, trimTools };
3889
4093
  //# sourceMappingURL=index.mjs.map
3890
4094
  //# sourceMappingURL=index.mjs.map