@infersec/conduit 1.75.0 → 1.76.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -9,7 +9,7 @@ export declare function provisionResources(options: {
9
9
  apiUrl: string;
10
10
  apiKey: string;
11
11
  contextLength: number;
12
- engine: "llama.cpp" | "vllm";
12
+ engine: "exllamav3" | "llama.cpp" | "mlx-lm" | "sglang" | "tensorrt-llm" | "vllm";
13
13
  format: string;
14
14
  parallelism: number;
15
15
  quantization: string | null;
@@ -17,7 +17,7 @@ export interface TestEntry {
17
17
  };
18
18
  concurrency: number;
19
19
  contextLength: number;
20
- engine: "llama.cpp" | "vllm";
20
+ engine: "exllamav3" | "llama.cpp" | "mlx-lm" | "sglang" | "tensorrt-llm" | "vllm";
21
21
  format: string;
22
22
  generate: boolean;
23
23
  parallelism: number;
package/dist/cli.js CHANGED
@@ -4348,13 +4348,29 @@ function getEffectiveContextLength({ contextLength, engineConfig, engineType })
4348
4348
  if (contextLength === null || contextLength <= 0) {
4349
4349
  return null;
4350
4350
  }
4351
- if (engineType === "llama.cpp" && engineConfig) {
4352
- const parallelism = engineConfig?.parallelism;
4353
- if (typeof parallelism === "number" && parallelism > 0) {
4354
- return contextLength / parallelism;
4351
+ if (!engineConfig) {
4352
+ return contextLength;
4353
+ }
4354
+ switch (engineType) {
4355
+ case "llama.cpp": {
4356
+ const parallelism = engineConfig.parallelism;
4357
+ if (typeof parallelism === "number" && parallelism > 0) {
4358
+ return contextLength / parallelism;
4359
+ }
4360
+ return contextLength;
4361
+ }
4362
+ case "sglang":
4363
+ case "tensorrt-llm":
4364
+ case "vllm": {
4365
+ const tensorParallelSize = engineConfig.tensorParallelSize;
4366
+ if (typeof tensorParallelSize === "number" && tensorParallelSize > 0) {
4367
+ return contextLength / tensorParallelSize;
4368
+ }
4369
+ return contextLength;
4355
4370
  }
4371
+ default:
4372
+ return contextLength;
4356
4373
  }
4357
- return contextLength;
4358
4374
  }
4359
4375
 
4360
4376
  function asError(error) {
@@ -19889,7 +19905,14 @@ object({
19889
19905
  PATCH: APIEndpointSchema.optional()
19890
19906
  });
19891
19907
 
19892
- const LLMEngineSchema = _enum(["llama.cpp", "vllm"]);
19908
+ const LLMEngineSchema = _enum([
19909
+ "exllamav3",
19910
+ "llama.cpp",
19911
+ "mlx-lm",
19912
+ "sglang",
19913
+ "tensorrt-llm",
19914
+ "vllm"
19915
+ ]);
19893
19916
  const LlamacppEngineConfigSchema = object({
19894
19917
  batchSize: number$1().int().positive().nullable().optional(),
19895
19918
  cacheTypeK: string$1().nullable().optional(),
@@ -19908,19 +19931,59 @@ const VLLMEngineConfigSchema = object({
19908
19931
  extraArgs: array(string$1()).optional(),
19909
19932
  tensorParallelSize: number$1().int().positive().optional()
19910
19933
  });
19934
+ const SGLangEngineConfigSchema = object({
19935
+ device: string$1().optional(),
19936
+ dtype: string$1().optional(),
19937
+ extraArgs: array(string$1()).optional(),
19938
+ tensorParallelSize: number$1().int().positive().optional()
19939
+ });
19940
+ const TensorRTLLMEngineConfigSchema = object({
19941
+ backend: _enum(["_autodeploy", "pytorch", "tensorrt"]).optional(),
19942
+ dtype: string$1().optional(),
19943
+ extraArgs: array(string$1()).optional(),
19944
+ tensorParallelSize: number$1().int().positive().optional()
19945
+ });
19946
+ const Exllamav3EngineConfigSchema = object({
19947
+ cacheMode: _enum(["fp16", "q4", "q6", "q8"]).optional(),
19948
+ extraArgs: array(string$1()).optional(),
19949
+ gpuSplit: string$1().optional(),
19950
+ maxSeqLen: number$1().int().positive().optional()
19951
+ });
19952
+ const MLXLMEngineConfigSchema = object({
19953
+ extraArgs: array(string$1()).optional(),
19954
+ maxKvSize: number$1().int().positive().optional(),
19955
+ trustRemoteCode: boolean$1().optional()
19956
+ });
19911
19957
  const EngineConfigSchema = discriminatedUnion("type", [
19958
+ object({ config: Exllamav3EngineConfigSchema, type: literal("exllamav3") }),
19912
19959
  object({ config: LlamacppEngineConfigSchema, type: literal("llama.cpp") }),
19960
+ object({ config: MLXLMEngineConfigSchema, type: literal("mlx-lm") }),
19961
+ object({ config: SGLangEngineConfigSchema, type: literal("sglang") }),
19962
+ object({
19963
+ config: TensorRTLLMEngineConfigSchema,
19964
+ type: literal("tensorrt-llm")
19965
+ }),
19913
19966
  object({ config: VLLMEngineConfigSchema, type: literal("vllm") })
19914
19967
  ]);
19915
19968
  const LLMModelFormatSchema = _enum([
19916
- // VLLM
19969
+ // VLLM / SGLang / TensorRT-LLM
19917
19970
  "safetensors",
19918
19971
  "pytorch",
19919
19972
  "awq",
19920
19973
  "gptq",
19921
19974
  // Llama.cpp
19922
- "gguf"
19975
+ "gguf",
19976
+ // ExLlamaV3
19977
+ "exl3",
19978
+ "exl2",
19979
+ // MLX-LM
19980
+ "mlx"
19923
19981
  ]);
19982
+ object({
19983
+ nativeAnthropicMessages: boolean$1(),
19984
+ supportsEmbeddings: boolean$1(),
19985
+ supportsVision: boolean$1()
19986
+ });
19924
19987
  const LLMModelTaskTypeSchema = _enum(["text-generation", "embeddings"]);
19925
19988
  const LLMModelSchema = object({
19926
19989
  format: LLMModelFormatSchema,
@@ -19980,6 +20043,7 @@ const InferenceAgentMachineMetadataSchema = object({
19980
20043
  model: string$1().nullable(),
19981
20044
  physicalCores: number$1().int().positive().nullable()
19982
20045
  }),
20046
+ exllamav3Version: string$1().nullable(),
19983
20047
  gpus: array(InferenceAgentMachineGPUSchema),
19984
20048
  hostname: string$1(),
19985
20049
  llamaCppVersion: string$1().nullable(),
@@ -19988,6 +20052,7 @@ const InferenceAgentMachineMetadataSchema = object({
19988
20052
  availableBytes: number$1().int().nonnegative().nullable(),
19989
20053
  totalBytes: number$1().int().nonnegative().nullable()
19990
20054
  }),
20055
+ mlxlmVersion: string$1().nullable(),
19991
20056
  os: object({
19992
20057
  arch: string$1(),
19993
20058
  platform: string$1(),
@@ -19995,6 +20060,8 @@ const InferenceAgentMachineMetadataSchema = object({
19995
20060
  type: string$1().nullable(),
19996
20061
  version: string$1().nullable()
19997
20062
  }),
20063
+ sglangVersion: string$1().nullable(),
20064
+ tensorrtLlmVersion: string$1().nullable(),
19998
20065
  vllmVersion: string$1().nullable()
19999
20066
  });
20000
20067
  const InferenceAgentMachineReportPayloadSchema = object({
@@ -20969,6 +21036,10 @@ object({
20969
21036
  });
20970
21037
  const EngineOutputSchema = object({
20971
21038
  created: string$1(),
21039
+ exllamav3CacheMode: string$1().nullable(),
21040
+ exllamav3ExtraArgs: array(string$1()),
21041
+ exllamav3GpuSplit: string$1().nullable(),
21042
+ exllamav3MaxSeqLen: number$1().nullable(),
20972
21043
  id: ULIDSchema,
20973
21044
  llamacppBatchSize: number$1().nullable(),
20974
21045
  llamacppCacheTypeK: string$1().nullable(),
@@ -20980,7 +21051,18 @@ const EngineOutputSchema = object({
20980
21051
  llamacppParallelism: number$1(),
20981
21052
  llamacppTensorSplit: string$1().nullable(),
20982
21053
  llamacppUbatchSize: number$1().nullable(),
21054
+ mlxlmExtraArgs: array(string$1()),
21055
+ mlxlmMaxKvSize: number$1().nullable(),
21056
+ mlxlmTrustRemoteCode: boolean$1(),
20983
21057
  name: string$1(),
21058
+ sglangDevice: string$1().nullable(),
21059
+ sglangDtype: string$1().nullable(),
21060
+ sglangExtraArgs: array(string$1()),
21061
+ sglangTensorParallelSize: number$1(),
21062
+ trtllmBackend: string$1().nullable(),
21063
+ trtllmDtype: string$1().nullable(),
21064
+ trtllmExtraArgs: array(string$1()),
21065
+ trtllmTensorParallelSize: number$1(),
20984
21066
  type: LLMEngineSchema,
20985
21067
  updated: string$1(),
20986
21068
  vllmDevice: string$1().nullable(),
@@ -20989,6 +21071,10 @@ const EngineOutputSchema = object({
20989
21071
  vllmTensorParallelSize: number$1()
20990
21072
  });
20991
21073
  object({
21074
+ exllamav3CacheMode: string$1().nullable().optional(),
21075
+ exllamav3ExtraArgs: array(string$1()).optional(),
21076
+ exllamav3GpuSplit: string$1().nullable().optional(),
21077
+ exllamav3MaxSeqLen: number$1().int().positive().nullable().optional(),
20992
21078
  llamacppBatchSize: number$1().int().positive().nullable().optional(),
20993
21079
  llamacppCacheTypeK: string$1().nullable().optional(),
20994
21080
  llamacppCacheTypeV: string$1().nullable().optional(),
@@ -20999,7 +21085,18 @@ object({
20999
21085
  llamacppParallelism: number$1().int().positive().optional(),
21000
21086
  llamacppTensorSplit: string$1().nullable().optional(),
21001
21087
  llamacppUbatchSize: number$1().int().positive().nullable().optional(),
21088
+ mlxlmExtraArgs: array(string$1()).optional(),
21089
+ mlxlmMaxKvSize: number$1().int().positive().nullable().optional(),
21090
+ mlxlmTrustRemoteCode: boolean$1().optional(),
21002
21091
  name: ResourceNameSchema,
21092
+ sglangDevice: string$1().nullable().optional(),
21093
+ sglangDtype: string$1().nullable().optional(),
21094
+ sglangExtraArgs: array(string$1()).optional(),
21095
+ sglangTensorParallelSize: number$1().int().positive().optional(),
21096
+ trtllmBackend: string$1().nullable().optional(),
21097
+ trtllmDtype: string$1().nullable().optional(),
21098
+ trtllmExtraArgs: array(string$1()).optional(),
21099
+ trtllmTensorParallelSize: number$1().int().positive().optional(),
21003
21100
  type: LLMEngineSchema,
21004
21101
  vllmDevice: string$1().nullable().optional(),
21005
21102
  vllmDtype: string$1().nullable().optional(),
@@ -21007,6 +21104,10 @@ object({
21007
21104
  vllmTensorParallelSize: number$1().int().positive().optional()
21008
21105
  });
21009
21106
  object({
21107
+ exllamav3CacheMode: string$1().nullable().optional(),
21108
+ exllamav3ExtraArgs: array(string$1()).optional(),
21109
+ exllamav3GpuSplit: string$1().nullable().optional(),
21110
+ exllamav3MaxSeqLen: number$1().int().positive().nullable().optional(),
21010
21111
  llamacppBatchSize: number$1().int().positive().nullable().optional(),
21011
21112
  llamacppCacheTypeK: string$1().nullable().optional(),
21012
21113
  llamacppCacheTypeV: string$1().nullable().optional(),
@@ -21017,7 +21118,18 @@ object({
21017
21118
  llamacppParallelism: number$1().int().positive().optional(),
21018
21119
  llamacppTensorSplit: string$1().nullable().optional(),
21019
21120
  llamacppUbatchSize: number$1().int().positive().nullable().optional(),
21121
+ mlxlmExtraArgs: array(string$1()).optional(),
21122
+ mlxlmMaxKvSize: number$1().int().positive().nullable().optional(),
21123
+ mlxlmTrustRemoteCode: boolean$1().optional(),
21020
21124
  name: ResourceNameSchema.optional(),
21125
+ sglangDevice: string$1().nullable().optional(),
21126
+ sglangDtype: string$1().nullable().optional(),
21127
+ sglangExtraArgs: array(string$1()).optional(),
21128
+ sglangTensorParallelSize: number$1().int().positive().optional(),
21129
+ trtllmBackend: string$1().nullable().optional(),
21130
+ trtllmDtype: string$1().nullable().optional(),
21131
+ trtllmExtraArgs: array(string$1()).optional(),
21132
+ trtllmTensorParallelSize: number$1().int().positive().optional(),
21021
21133
  type: LLMEngineSchema.optional(),
21022
21134
  vllmDevice: string$1().nullable().optional(),
21023
21135
  vllmDtype: string$1().nullable().optional(),
@@ -21090,6 +21202,39 @@ object({
21090
21202
  }
21091
21203
  });
21092
21204
 
21205
+ const ENGINE_API_COMPATIBILITY = {
21206
+ exllamav3: {
21207
+ nativeAnthropicMessages: false,
21208
+ supportsEmbeddings: false,
21209
+ supportsVision: true
21210
+ },
21211
+ "llama.cpp": {
21212
+ nativeAnthropicMessages: true,
21213
+ supportsEmbeddings: true,
21214
+ supportsVision: true
21215
+ },
21216
+ "mlx-lm": {
21217
+ nativeAnthropicMessages: false,
21218
+ supportsEmbeddings: false,
21219
+ supportsVision: false
21220
+ },
21221
+ sglang: {
21222
+ nativeAnthropicMessages: false,
21223
+ supportsEmbeddings: true,
21224
+ supportsVision: true
21225
+ },
21226
+ "tensorrt-llm": {
21227
+ nativeAnthropicMessages: false,
21228
+ supportsEmbeddings: true,
21229
+ supportsVision: true
21230
+ },
21231
+ vllm: {
21232
+ nativeAnthropicMessages: true,
21233
+ supportsEmbeddings: true,
21234
+ supportsVision: true
21235
+ }
21236
+ };
21237
+
21093
21238
  object({
21094
21239
  accountID: ULIDSchema.optional(),
21095
21240
  email: string$1().email(),
@@ -116639,6 +116784,43 @@ async function downloadFileWithRange({ accessToken, filePath, fileSize, modelSlu
116639
116784
  throw new Error(errorMessage);
116640
116785
  }
116641
116786
 
116787
+ const EXLLAMAV3_EXECUTABLE = process.env.EXLLAMAV3_EXECUTABLE ?? "python3";
116788
+ const SERVER_SCRIPT = join(import.meta.dirname, "server.py");
116789
+ const DEFAULT_EXLLAMAV3_CONTEXT_LENGTH = 4096;
116790
+ async function startExllamav3({ enginePort, targetDirectory }) {
116791
+ const contextLength = Math.max(1, this.contextLength ?? DEFAULT_EXLLAMAV3_CONTEXT_LENGTH);
116792
+ const engineConfig = this.engineConfig;
116793
+ const cacheMode = typeof engineConfig?.cacheMode === "string" ? engineConfig.cacheMode : "q4";
116794
+ const gpuSplit = typeof engineConfig?.gpuSplit === "string" ? engineConfig.gpuSplit : null;
116795
+ const maxSeqLen = typeof engineConfig?.maxSeqLen === "number" ? engineConfig.maxSeqLen : contextLength;
116796
+ const args = [
116797
+ SERVER_SCRIPT,
116798
+ "--model",
116799
+ targetDirectory,
116800
+ "--host",
116801
+ "127.0.0.1",
116802
+ "--port",
116803
+ String(enginePort),
116804
+ "--cache-mode",
116805
+ cacheMode,
116806
+ "--max-seq-len",
116807
+ String(maxSeqLen)
116808
+ ];
116809
+ if (gpuSplit) {
116810
+ args.push("--gpu-split", gpuSplit);
116811
+ }
116812
+ const extraArgs = engineConfig?.extraArgs;
116813
+ if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
116814
+ args.push(...extraArgs);
116815
+ }
116816
+ const processManager = new ProcessManager({
116817
+ command: EXLLAMAV3_EXECUTABLE,
116818
+ args
116819
+ });
116820
+ await processManager.start();
116821
+ return processManager;
116822
+ }
116823
+
116642
116824
  const DEFAULT_LLAMACPP_GPU_LAYERS = 999;
116643
116825
  const LLAMACPP_START_ARGS = ["--host", "0.0.0.0", "--jinja"];
116644
116826
  const LLAMACPP_EXECUTABLE = process.env.LLAMACPP_EXECUTABLE ?? "llama-server";
@@ -116738,6 +116920,46 @@ async function startLlamacpp({ enginePort, targetDirectory }) {
116738
116920
  return processManager;
116739
116921
  }
116740
116922
 
116923
+ const MLXLM_EXECUTABLE = process.env.MLXLM_EXECUTABLE ?? "python3";
116924
+ const DEFAULT_MLXLM_CONTEXT_LENGTH = 4096;
116925
+ async function startMLXLM({ enginePort, targetDirectory }) {
116926
+ if (this.model.taskType === "embeddings") {
116927
+ throw new Error("MLX-LM engine does not support embeddings task type");
116928
+ }
116929
+ const contextLength = Math.max(1, this.contextLength ?? DEFAULT_MLXLM_CONTEXT_LENGTH);
116930
+ const engineConfig = this.engineConfig;
116931
+ const args = [
116932
+ "-m",
116933
+ "mlx_lm.server",
116934
+ "--model",
116935
+ targetDirectory,
116936
+ "--host",
116937
+ "127.0.0.1",
116938
+ "--port",
116939
+ String(enginePort),
116940
+ "--context-length",
116941
+ String(contextLength)
116942
+ ];
116943
+ const maxKvSize = typeof engineConfig?.maxKvSize === "number" ? engineConfig.maxKvSize : null;
116944
+ if (maxKvSize !== null) {
116945
+ args.push("--max-kv-size", String(maxKvSize));
116946
+ }
116947
+ const trustRemoteCode = engineConfig?.trustRemoteCode === true;
116948
+ if (trustRemoteCode || process.env.MLXLM_TRUST_REMOTE_CODE === "true") {
116949
+ args.push("--trust-remote-code");
116950
+ }
116951
+ const extraArgs = engineConfig?.extraArgs;
116952
+ if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
116953
+ args.push(...extraArgs);
116954
+ }
116955
+ const processManager = new ProcessManager({
116956
+ command: MLXLM_EXECUTABLE,
116957
+ args
116958
+ });
116959
+ await processManager.start();
116960
+ return processManager;
116961
+ }
116962
+
116741
116963
  const SAFE_CHARS = /[a-zA-Z0-9\-_.]/;
116742
116964
  const SEPARATOR = "__";
116743
116965
  function sanitizeSegment(value) {
@@ -116762,6 +116984,95 @@ function createModelStorageKey(model) {
116762
116984
  return `${model.source.type}${SEPARATOR}${sanitizeSegment(identifier)}`;
116763
116985
  }
116764
116986
 
116987
+ const SGLANG_START_ARGS = ["-m", "sglang.launch_server", "--host", "127.0.0.1"];
116988
+ const SGLANG_EXECUTABLE = process.env.SGLANG_EXECUTABLE ?? "python3";
116989
+ const DEFAULT_SGLANG_CONTEXT_LENGTH = 2048;
116990
+ async function startSGLang({ enginePort, targetDirectory }) {
116991
+ const contextLength = Math.max(1, this.contextLength ?? DEFAULT_SGLANG_CONTEXT_LENGTH);
116992
+ const engineConfig = this.engineConfig;
116993
+ const device = typeof engineConfig?.device === "string" ? engineConfig.device : undefined;
116994
+ const dtype = typeof engineConfig?.dtype === "string" ? engineConfig.dtype : undefined;
116995
+ const tensorParallelSize = typeof engineConfig?.tensorParallelSize === "number" ? engineConfig.tensorParallelSize : 1;
116996
+ const args = [
116997
+ ...SGLANG_START_ARGS,
116998
+ "--port",
116999
+ String(enginePort),
117000
+ "--model-path",
117001
+ targetDirectory,
117002
+ "--served-model-name",
117003
+ this.model.id,
117004
+ "--context-length",
117005
+ String(contextLength),
117006
+ "--tp-size",
117007
+ String(tensorParallelSize)
117008
+ ];
117009
+ if (this.model.taskType === "embeddings") {
117010
+ args.push("--task", "embed");
117011
+ }
117012
+ if (device) {
117013
+ args.push("--device", device);
117014
+ }
117015
+ if (dtype) {
117016
+ args.push("--dtype", dtype);
117017
+ }
117018
+ const extraArgs = engineConfig?.extraArgs;
117019
+ if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
117020
+ args.push(...extraArgs);
117021
+ }
117022
+ if (this.model.multimodalEnabled) {
117023
+ args.push("--limit-mm-per-prompt", process.env.SGLANG_MM_LIMIT ?? '{"image":5}');
117024
+ }
117025
+ if (process.env.SGLANG_TRUST_REMOTE_CODE === "true") {
117026
+ args.push("--trust-remote-code");
117027
+ }
117028
+ const processManager = new ProcessManager({
117029
+ command: SGLANG_EXECUTABLE,
117030
+ args
117031
+ });
117032
+ await processManager.start();
117033
+ return processManager;
117034
+ }
117035
+
117036
+ const TRTLLM_EXECUTABLE = process.env.TRTLLM_EXECUTABLE ?? "trtllm-serve";
117037
+ const DEFAULT_TRTLLM_CONTEXT_LENGTH = 2048;
117038
+ async function startTensorRTLLM({ enginePort, targetDirectory }) {
117039
+ const contextLength = Math.max(1, this.contextLength ?? DEFAULT_TRTLLM_CONTEXT_LENGTH);
117040
+ const engineConfig = this.engineConfig;
117041
+ const backend = typeof engineConfig?.backend === "string" ? engineConfig.backend : "pytorch";
117042
+ const dtype = typeof engineConfig?.dtype === "string" ? engineConfig.dtype : undefined;
117043
+ const tensorParallelSize = typeof engineConfig?.tensorParallelSize === "number" ? engineConfig.tensorParallelSize : 1;
117044
+ const args = [
117045
+ "serve",
117046
+ targetDirectory,
117047
+ "--host",
117048
+ "127.0.0.1",
117049
+ "--port",
117050
+ String(enginePort),
117051
+ "--backend",
117052
+ backend,
117053
+ "--max-seq-len",
117054
+ String(contextLength),
117055
+ "--tp-size",
117056
+ String(tensorParallelSize)
117057
+ ];
117058
+ if (this.model.taskType === "embeddings") {
117059
+ args.push("--task", "embed");
117060
+ }
117061
+ if (dtype) {
117062
+ args.push("--dtype", dtype);
117063
+ }
117064
+ const extraArgs = engineConfig?.extraArgs;
117065
+ if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
117066
+ args.push(...extraArgs);
117067
+ }
117068
+ const processManager = new ProcessManager({
117069
+ command: TRTLLM_EXECUTABLE,
117070
+ args
117071
+ });
117072
+ await processManager.start();
117073
+ return processManager;
117074
+ }
117075
+
116765
117076
  const ENGINE_FETCH_TIMEOUT_MS$1 = 7200000;
116766
117077
  const DOWNLOAD_LOCK_TIMEOUT_MS = 20 * 60 * 1000;
116767
117078
  const DOWNLOAD_LOCK_POLL_INTERVAL_MS = 5000;
@@ -116795,7 +117106,11 @@ class ModelManager extends EventEmitter {
116795
117106
  }
116796
117107
  async fetchOpenAI(path, opts) {
116797
117108
  switch (this.engine) {
117109
+ case "exllamav3":
116798
117110
  case "llama.cpp":
117111
+ case "mlx-lm":
117112
+ case "sglang":
117113
+ case "tensorrt-llm":
116799
117114
  case "vllm": {
116800
117115
  this.logger.debug(`Fetching from engine: ${path}`);
116801
117116
  const callerSignal = opts?.signal;
@@ -116843,7 +117158,11 @@ class ModelManager extends EventEmitter {
116843
117158
  modelID: this.model.id
116844
117159
  });
116845
117160
  switch (this.engine) {
117161
+ case "exllamav3":
116846
117162
  case "llama.cpp":
117163
+ case "mlx-lm":
117164
+ case "sglang":
117165
+ case "tensorrt-llm":
116847
117166
  case "vllm":
116848
117167
  if (this.model.source.type !== "huggingface") {
116849
117168
  throw new Error(`Model source not implemented: ${this.model.source.type}`);
@@ -116950,10 +117269,34 @@ class ModelManager extends EventEmitter {
116950
117269
  case "vllm": {
116951
117270
  return this.checkVLLMReadiness();
116952
117271
  }
117272
+ case "exllamav3":
117273
+ case "mlx-lm":
117274
+ case "sglang":
117275
+ case "tensorrt-llm": {
117276
+ return this.checkGenericHealthReadiness();
117277
+ }
116953
117278
  default:
116954
117279
  return "ready";
116955
117280
  }
116956
117281
  }
117282
+ async checkGenericHealthReadiness() {
117283
+ try {
117284
+ const response = await undiciExports.fetch(joinURL(`http://localhost:${this.enginePort}`, "/health"), {
117285
+ method: "GET",
117286
+ signal: AbortSignal.timeout(5000)
117287
+ });
117288
+ if (response.status === 503) {
117289
+ return "loading";
117290
+ }
117291
+ if (response.ok) {
117292
+ return "ready";
117293
+ }
117294
+ return "loading";
117295
+ }
117296
+ catch (_error) {
117297
+ return "unreachable";
117298
+ }
117299
+ }
116957
117300
  async checkLlamacppReadiness() {
116958
117301
  try {
116959
117302
  const response = await undiciExports.fetch(joinURL(`http://localhost:${this.enginePort}`, "/health"), {
@@ -117150,16 +117493,37 @@ class ModelManager extends EventEmitter {
117150
117493
  });
117151
117494
  }
117152
117495
  async startEngineProcess() {
117496
+ const targetDir = join(this.modelsDirectory, this.uniqueName);
117153
117497
  switch (this.engine) {
117498
+ case "exllamav3":
117499
+ return startExllamav3.call(this, {
117500
+ enginePort: this.enginePort,
117501
+ targetDirectory: targetDir
117502
+ });
117154
117503
  case "llama.cpp":
117155
117504
  return startLlamacpp.call(this, {
117156
117505
  enginePort: this.enginePort,
117157
- targetDirectory: join(this.modelsDirectory, this.uniqueName)
117506
+ targetDirectory: targetDir
117507
+ });
117508
+ case "mlx-lm":
117509
+ return startMLXLM.call(this, {
117510
+ enginePort: this.enginePort,
117511
+ targetDirectory: targetDir
117512
+ });
117513
+ case "sglang":
117514
+ return startSGLang.call(this, {
117515
+ enginePort: this.enginePort,
117516
+ targetDirectory: targetDir
117517
+ });
117518
+ case "tensorrt-llm":
117519
+ return startTensorRTLLM.call(this, {
117520
+ enginePort: this.enginePort,
117521
+ targetDirectory: targetDir
117158
117522
  });
117159
117523
  case "vllm":
117160
117524
  return startVLLM.call(this, {
117161
117525
  enginePort: this.enginePort,
117162
- targetDirectory: join(this.modelsDirectory, this.uniqueName)
117526
+ targetDirectory: targetDir
117163
117527
  });
117164
117528
  default: {
117165
117529
  const engineType = this.engine;
@@ -118227,6 +118591,189 @@ function createPostEmbeddingsHandler(options) {
118227
118591
  return createConduitOpenAIAPIReferenceHandlers(options)["/v1/embeddings"].POST;
118228
118592
  }
118229
118593
 
118594
+ function engineSupportsNativeAnthropic(engineType) {
118595
+ if (!engineType)
118596
+ return false;
118597
+ return ENGINE_API_COMPATIBILITY[engineType]?.nativeAnthropicMessages ?? false;
118598
+ }
118599
+ function translateAnthropicRequestToOpenAI(body) {
118600
+ const parsed = JSON.parse(body);
118601
+ const messages = Array.isArray(parsed.messages) ? parsed.messages : [];
118602
+ const openaiMessages = [];
118603
+ const system = parsed.system;
118604
+ if (typeof system === "string" && system.length > 0) {
118605
+ openaiMessages.push({ content: system, role: "system" });
118606
+ }
118607
+ else if (Array.isArray(system)) {
118608
+ const textParts = system
118609
+ .filter((b) => {
118610
+ const block = b;
118611
+ return block?.type === "text" && typeof block.text === "string";
118612
+ })
118613
+ .map((b) => b.text);
118614
+ if (textParts.length > 0) {
118615
+ openaiMessages.push({ content: textParts.join("\n"), role: "system" });
118616
+ }
118617
+ }
118618
+ for (const msg of messages) {
118619
+ const m = msg;
118620
+ const role = m.role;
118621
+ const content = m.content;
118622
+ if (typeof content === "string") {
118623
+ openaiMessages.push({ content, role });
118624
+ continue;
118625
+ }
118626
+ if (!Array.isArray(content)) {
118627
+ openaiMessages.push({ content: content ?? "", role });
118628
+ continue;
118629
+ }
118630
+ if (role === "assistant") {
118631
+ const toolCalls = [];
118632
+ const textParts = [];
118633
+ for (const block of content) {
118634
+ const b = block;
118635
+ if (b.type === "text" && typeof b.text === "string") {
118636
+ textParts.push(b.text);
118637
+ }
118638
+ else if (b.type === "tool_use") {
118639
+ toolCalls.push({
118640
+ function: {
118641
+ arguments: JSON.stringify(b.input ?? {}),
118642
+ name: b.name
118643
+ },
118644
+ id: b.id,
118645
+ type: "function"
118646
+ });
118647
+ }
118648
+ }
118649
+ openaiMessages.push({
118650
+ content: textParts.join("") || null,
118651
+ role,
118652
+ ...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {})
118653
+ });
118654
+ }
118655
+ else if (role === "user") {
118656
+ const textParts = [];
118657
+ const toolResults = [];
118658
+ for (const block of content) {
118659
+ const b = block;
118660
+ if (b.type === "text" && typeof b.text === "string") {
118661
+ textParts.push(b.text);
118662
+ }
118663
+ else if (b.type === "tool_result") {
118664
+ toolResults.push({
118665
+ content: typeof b.content === "string"
118666
+ ? b.content
118667
+ : JSON.stringify(b.content ?? ""),
118668
+ role: "tool",
118669
+ tool_call_id: b.tool_use_id
118670
+ });
118671
+ }
118672
+ }
118673
+ if (textParts.length > 0) {
118674
+ openaiMessages.push({ content: textParts.join("\n"), role });
118675
+ }
118676
+ for (const tr of toolResults) {
118677
+ openaiMessages.push(tr);
118678
+ }
118679
+ }
118680
+ else {
118681
+ openaiMessages.push({ content: JSON.stringify(content), role });
118682
+ }
118683
+ }
118684
+ const result = {
118685
+ max_tokens: parsed.max_tokens ?? 4096,
118686
+ messages: openaiMessages,
118687
+ model: parsed.model,
118688
+ stream: parsed.stream ?? false
118689
+ };
118690
+ if (typeof parsed.temperature === "number")
118691
+ result.temperature = parsed.temperature;
118692
+ if (typeof parsed.top_p === "number")
118693
+ result.top_p = parsed.top_p;
118694
+ if (Array.isArray(parsed.stop_sequences))
118695
+ result.stop = parsed.stop_sequences;
118696
+ if (Array.isArray(parsed.tools)) {
118697
+ result.tools = parsed.tools.map((tool) => {
118698
+ const t = tool;
118699
+ return {
118700
+ function: {
118701
+ ...(typeof t.description === "string" ? { description: t.description } : {}),
118702
+ name: t.name,
118703
+ parameters: t.input_schema ?? {}
118704
+ },
118705
+ type: "function"
118706
+ };
118707
+ });
118708
+ }
118709
+ if (parsed.tool_choice && typeof parsed.tool_choice === "object") {
118710
+ const tc = parsed.tool_choice;
118711
+ if (tc.type === "auto")
118712
+ result.tool_choice = "auto";
118713
+ else if (tc.type === "any")
118714
+ result.tool_choice = "required";
118715
+ else if (tc.type === "tool" && typeof tc.name === "string") {
118716
+ result.tool_choice = { function: { name: tc.name }, type: "function" };
118717
+ }
118718
+ }
118719
+ return { body: JSON.stringify(result), path: "/v1/chat/completions" };
118720
+ }
118721
+ function translateOpenAIResponseToAnthropic({ body, model }) {
118722
+ const parsed = JSON.parse(body);
118723
+ const choices = Array.isArray(parsed.choices) ? parsed.choices : [];
118724
+ const choice = choices[0];
118725
+ const message = choice?.message;
118726
+ const content = [];
118727
+ if (message) {
118728
+ if (typeof message.content === "string" && message.content.length > 0) {
118729
+ content.push({ text: message.content, type: "text" });
118730
+ }
118731
+ if (Array.isArray(message.tool_calls)) {
118732
+ for (const tc of message.tool_calls) {
118733
+ const call = tc;
118734
+ const fn = call.function;
118735
+ const argsRaw = typeof fn?.arguments === "string" ? fn.arguments : "{}";
118736
+ let parsedInput = {};
118737
+ try {
118738
+ parsedInput = JSON.parse(argsRaw);
118739
+ }
118740
+ catch {
118741
+ parsedInput = {};
118742
+ }
118743
+ content.push({
118744
+ id: call.id,
118745
+ input: parsedInput,
118746
+ name: fn?.name,
118747
+ type: "tool_use"
118748
+ });
118749
+ }
118750
+ }
118751
+ }
118752
+ const finishReason = choice?.finish_reason;
118753
+ let stopReason = "end_turn";
118754
+ if (finishReason === "length")
118755
+ stopReason = "max_tokens";
118756
+ else if (finishReason === "tool_calls")
118757
+ stopReason = "tool_use";
118758
+ else if (finishReason === "stop")
118759
+ stopReason = "end_turn";
118760
+ const usage = parsed.usage;
118761
+ const result = {
118762
+ content,
118763
+ id: parsed.id,
118764
+ model,
118765
+ role: "assistant",
118766
+ stop_reason: stopReason,
118767
+ stop_sequence: null,
118768
+ type: "message",
118769
+ usage: {
118770
+ input_tokens: usage?.prompt_tokens ?? 0,
118771
+ output_tokens: usage?.completion_tokens ?? 0
118772
+ }
118773
+ };
118774
+ return JSON.stringify(result);
118775
+ }
118776
+
118230
118777
  function isPlainObject$1(value) {
118231
118778
  return typeof value === "object" && value !== null && !Array.isArray(value);
118232
118779
  }
@@ -118297,6 +118844,106 @@ function extractAnthropicNonStreamUsage(body) {
118297
118844
  return null;
118298
118845
  }
118299
118846
  }
118847
+ function extractOpenAIUsageFromJSON(body) {
118848
+ try {
118849
+ const parsed = JSON.parse(body);
118850
+ if (!isPlainObject$1(parsed) || !isPlainObject$1(parsed.usage))
118851
+ return null;
118852
+ const usage = parsed.usage;
118853
+ return {
118854
+ inputTokens: typeof usage.prompt_tokens === "number" ? usage.prompt_tokens : null,
118855
+ outputTokens: typeof usage.completion_tokens === "number" ? usage.completion_tokens : null
118856
+ };
118857
+ }
118858
+ catch {
118859
+ return null;
118860
+ }
118861
+ }
118862
+ function assembleOpenAISSE(body) {
118863
+ const lines = body.split("\n");
118864
+ let content = "";
118865
+ let promptTokens = null;
118866
+ let completionTokens = null;
118867
+ for (const line of lines) {
118868
+ const trimmed = line.trim();
118869
+ if (!trimmed.startsWith("data:"))
118870
+ continue;
118871
+ const payload = trimmed.slice(5).trim();
118872
+ if (payload === "[DONE]")
118873
+ continue;
118874
+ try {
118875
+ const parsed = JSON.parse(payload);
118876
+ if (!isPlainObject$1(parsed))
118877
+ continue;
118878
+ const choices = parsed.choices;
118879
+ if (Array.isArray(choices) && choices.length > 0) {
118880
+ const choice = choices[0];
118881
+ const delta = choice.delta;
118882
+ if (delta && typeof delta.content === "string") {
118883
+ content += delta.content;
118884
+ }
118885
+ }
118886
+ const usage = parsed.usage;
118887
+ if (usage) {
118888
+ if (typeof usage.prompt_tokens === "number")
118889
+ promptTokens = usage.prompt_tokens;
118890
+ if (typeof usage.completion_tokens === "number")
118891
+ completionTokens = usage.completion_tokens;
118892
+ }
118893
+ }
118894
+ catch {
118895
+ // ignore
118896
+ }
118897
+ }
118898
+ return { completionTokens, content, promptTokens };
118899
+ }
118900
+ function translateOpenAIStreamToAnthropicSSE(body, model) {
118901
+ const assembled = assembleOpenAISSE(body);
118902
+ const content = assembled?.content ?? "";
118903
+ const inputTokens = assembled?.promptTokens ?? 0;
118904
+ const outputTokens = assembled?.completionTokens ?? 0;
118905
+ const events = [];
118906
+ const messageId = `msg_${Date.now()}`;
118907
+ events.push(`event: message_start\ndata: ${JSON.stringify({
118908
+ message: {
118909
+ content: [],
118910
+ id: messageId,
118911
+ input_tokens: inputTokens,
118912
+ model,
118913
+ output_tokens: outputTokens,
118914
+ role: "assistant",
118915
+ stop_reason: null,
118916
+ type: "message",
118917
+ usage: { input_tokens: inputTokens, output_tokens: 0 }
118918
+ },
118919
+ type: "message_start"
118920
+ })}`);
118921
+ events.push(`event: content_block_start\ndata: ${JSON.stringify({
118922
+ content_block: { text: "", type: "text" },
118923
+ index: 0,
118924
+ type: "content_block_start"
118925
+ })}`);
118926
+ const chunkSize = 20;
118927
+ for (let i = 0; i < content.length; i += chunkSize) {
118928
+ const textChunk = content.slice(i, i + chunkSize);
118929
+ events.push(`event: content_block_delta\ndata: ${JSON.stringify({
118930
+ delta: { text: textChunk, type: "text_delta" },
118931
+ index: 0,
118932
+ type: "content_block_delta"
118933
+ })}`);
118934
+ }
118935
+ events.push(`event: content_block_stop\ndata: ${JSON.stringify({
118936
+ index: 0,
118937
+ type: "content_block_stop"
118938
+ })}`);
118939
+ events.push(`event: message_delta\ndata: ${JSON.stringify({
118940
+ delta: { stop_reason: "end_turn", stop_sequence: null },
118941
+ type: "message_delta",
118942
+ usage: { output_tokens: outputTokens }
118943
+ })}`);
118944
+ events.push('event: message_stop\ndata: {"type":"message_stop"}');
118945
+ return events.join("\n\n") + "\n\n";
118946
+ }
118300
118947
  async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }) {
118301
118948
  function reportMetricsSafe(payload) {
118302
118949
  reportMetrics(payload).catch(error => {
@@ -118307,10 +118954,17 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118307
118954
  });
118308
118955
  }
118309
118956
  const engineType = conduitConfiguration.engineConfig?.type ?? null;
118957
+ const needsTranslation = !engineSupportsNativeAnthropic(engineType);
118310
118958
  const { bytes: requestBodyBytes, payload: serializedBody } = serializeRequestBody(body);
118311
118959
  const requestStartedAt = Date.now();
118312
118960
  const requestBody = JSON.parse(serializedBody);
118313
118961
  const streamRequested = requestBody.stream === true;
118962
+ const targetPath = needsTranslation
118963
+ ? translateAnthropicRequestToOpenAI(serializedBody).path
118964
+ : "/v1/messages";
118965
+ const targetBody = needsTranslation
118966
+ ? translateAnthropicRequestToOpenAI(serializedBody).body
118967
+ : serializedBody;
118314
118968
  const onMonitoringComplete = ({ durationMs, error, responseBytes, usage }) => {
118315
118969
  const promptTokens = normalizeTokenCount(usage?.inputTokens);
118316
118970
  const completionTokens = normalizeTokenCount(usage?.outputTokens);
@@ -118339,8 +118993,8 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118339
118993
  });
118340
118994
  };
118341
118995
  const response = await modelManager
118342
- .fetchOpenAI("/v1/messages", {
118343
- body: serializedBody,
118996
+ .fetchOpenAI(targetPath, {
118997
+ body: targetBody,
118344
118998
  headers: {
118345
118999
  "Content-Type": "application/json"
118346
119000
  },
@@ -118447,73 +119101,157 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118447
119101
  }
118448
119102
  const rawBody = Readable.fromWeb(response.body);
118449
119103
  if (streamRequested) {
118450
- let buffer = "";
118451
- rawBody.on("data", (chunk) => {
118452
- const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
118453
- responseBytes += chunkBuffer.length;
118454
- buffer += chunkBuffer.toString("utf8");
118455
- const lines = buffer.split("\n");
118456
- buffer = lines.pop() ?? "";
118457
- for (const line of lines) {
118458
- const extracted = extractAnthropicStreamUsage(line.trim());
118459
- if (extracted?.inputTokens !== undefined && extracted.inputTokens !== null) {
118460
- usage.inputTokens = extracted.inputTokens;
119104
+ if (needsTranslation) {
119105
+ const chunks = [];
119106
+ rawBody.on("data", (chunk) => {
119107
+ const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
119108
+ responseBytes += chunkBuffer.length;
119109
+ chunks.push(chunkBuffer);
119110
+ });
119111
+ rawBody.once("end", () => {
119112
+ const fullBody = Buffer.concat(chunks).toString("utf8");
119113
+ if (!response.ok) {
119114
+ passThrough.write(fullBody);
119115
+ responseBytes = Buffer.byteLength(fullBody, "utf8");
119116
+ logEngineMetrics({
119117
+ agentEngineType: engineType ?? "unknown",
119118
+ level: "error",
119119
+ logger,
119120
+ requestBodyBytes,
119121
+ requestPath: "/v1/messages",
119122
+ responseBytes,
119123
+ usage: null
119124
+ });
119125
+ finalize(upstreamError);
119126
+ passThrough.end();
119127
+ return;
118461
119128
  }
118462
- if (extracted?.outputTokens !== undefined && extracted.outputTokens !== null) {
118463
- usage.outputTokens = extracted.outputTokens;
119129
+ const modelId = requestBody.model ?? "unknown";
119130
+ const assembled = assembleOpenAISSE(fullBody);
119131
+ if (assembled) {
119132
+ usage.inputTokens = assembled.promptTokens;
119133
+ usage.outputTokens = assembled.completionTokens;
118464
119134
  }
118465
- }
118466
- passThrough.write(chunkBuffer);
118467
- });
118468
- rawBody.once("error", err => {
118469
- const normalizedError = asError(err);
118470
- logEngineMetrics({
118471
- agentEngineType: engineType ?? "unknown",
118472
- error: normalizedError,
118473
- level: "error",
118474
- logger,
118475
- requestBodyBytes,
118476
- requestPath: "/v1/messages",
118477
- responseBytes,
118478
- usage: null
119135
+ try {
119136
+ const anthropicSSE = translateOpenAIStreamToAnthropicSSE(fullBody, modelId);
119137
+ const output = Buffer.from(anthropicSSE, "utf8");
119138
+ responseBytes = output.length;
119139
+ passThrough.write(output);
119140
+ }
119141
+ catch (translateError) {
119142
+ const normalizedTranslateError = asError(translateError);
119143
+ logEngineMetrics({
119144
+ agentEngineType: engineType ?? "unknown",
119145
+ error: normalizedTranslateError,
119146
+ level: "error",
119147
+ logger,
119148
+ requestBodyBytes,
119149
+ requestPath: "/v1/messages",
119150
+ responseBytes,
119151
+ usage: null
119152
+ });
119153
+ finalize(normalizedTranslateError);
119154
+ passThrough.destroy(normalizedTranslateError);
119155
+ return;
119156
+ }
119157
+ logEngineMetrics({
119158
+ agentEngineType: engineType ?? "unknown",
119159
+ level: upstreamError ? "error" : "info",
119160
+ logger,
119161
+ requestBodyBytes,
119162
+ requestPath: "/v1/messages",
119163
+ responseBytes,
119164
+ usage: null
119165
+ });
119166
+ finalize(upstreamError);
119167
+ passThrough.end();
118479
119168
  });
118480
- finalize(normalizedError);
118481
- passThrough.destroy(normalizedError);
118482
- });
118483
- rawBody.once("end", () => {
118484
- logEngineMetrics({
118485
- agentEngineType: engineType ?? "unknown",
118486
- level: upstreamError ? "error" : "info",
118487
- logger,
118488
- requestBodyBytes,
118489
- requestPath: "/v1/messages",
118490
- responseBytes,
118491
- usage: null
119169
+ rawBody.once("error", err => {
119170
+ const normalizedError = asError(err);
119171
+ finalize(normalizedError);
119172
+ passThrough.destroy(normalizedError);
118492
119173
  });
118493
- finalize(upstreamError);
118494
- passThrough.end();
118495
- });
118496
- rawBody.once("close", () => {
118497
- if (completed) {
119174
+ rawBody.once("close", () => {
119175
+ if (completed) {
119176
+ if (!passThrough.writableEnded)
119177
+ passThrough.end();
119178
+ return;
119179
+ }
119180
+ const closeError = new Error("Engine response stream closed before completion");
119181
+ finalize(closeError);
118498
119182
  if (!passThrough.writableEnded)
118499
119183
  passThrough.end();
118500
- return;
118501
- }
118502
- const closeError = new Error("Engine response stream closed before completion");
118503
- logEngineMetrics({
118504
- agentEngineType: engineType ?? "unknown",
118505
- error: closeError,
118506
- level: "error",
118507
- logger,
118508
- requestBodyBytes,
118509
- requestPath: "/v1/messages",
118510
- responseBytes,
118511
- usage: null
118512
119184
  });
118513
- finalize(closeError);
118514
- if (!passThrough.writableEnded)
119185
+ }
119186
+ else {
119187
+ let buffer = "";
119188
+ rawBody.on("data", (chunk) => {
119189
+ const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
119190
+ responseBytes += chunkBuffer.length;
119191
+ buffer += chunkBuffer.toString("utf8");
119192
+ const lines = buffer.split("\n");
119193
+ buffer = lines.pop() ?? "";
119194
+ for (const line of lines) {
119195
+ const extracted = extractAnthropicStreamUsage(line.trim());
119196
+ if (extracted?.inputTokens !== undefined && extracted.inputTokens !== null) {
119197
+ usage.inputTokens = extracted.inputTokens;
119198
+ }
119199
+ if (extracted?.outputTokens !== undefined && extracted.outputTokens !== null) {
119200
+ usage.outputTokens = extracted.outputTokens;
119201
+ }
119202
+ }
119203
+ passThrough.write(chunkBuffer);
119204
+ });
119205
+ rawBody.once("error", err => {
119206
+ const normalizedError = asError(err);
119207
+ logEngineMetrics({
119208
+ agentEngineType: engineType ?? "unknown",
119209
+ error: normalizedError,
119210
+ level: "error",
119211
+ logger,
119212
+ requestBodyBytes,
119213
+ requestPath: "/v1/messages",
119214
+ responseBytes,
119215
+ usage: null
119216
+ });
119217
+ finalize(normalizedError);
119218
+ passThrough.destroy(normalizedError);
119219
+ });
119220
+ rawBody.once("end", () => {
119221
+ logEngineMetrics({
119222
+ agentEngineType: engineType ?? "unknown",
119223
+ level: upstreamError ? "error" : "info",
119224
+ logger,
119225
+ requestBodyBytes,
119226
+ requestPath: "/v1/messages",
119227
+ responseBytes,
119228
+ usage: null
119229
+ });
119230
+ finalize(upstreamError);
118515
119231
  passThrough.end();
118516
- });
119232
+ });
119233
+ rawBody.once("close", () => {
119234
+ if (completed) {
119235
+ if (!passThrough.writableEnded)
119236
+ passThrough.end();
119237
+ return;
119238
+ }
119239
+ const closeError = new Error("Engine response stream closed before completion");
119240
+ logEngineMetrics({
119241
+ agentEngineType: engineType ?? "unknown",
119242
+ error: closeError,
119243
+ level: "error",
119244
+ logger,
119245
+ requestBodyBytes,
119246
+ requestPath: "/v1/messages",
119247
+ responseBytes,
119248
+ usage: null
119249
+ });
119250
+ finalize(closeError);
119251
+ if (!passThrough.writableEnded)
119252
+ passThrough.end();
119253
+ });
119254
+ }
118517
119255
  }
118518
119256
  else {
118519
119257
  const chunks = [];
@@ -118521,7 +119259,6 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118521
119259
  const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
118522
119260
  responseBytes += chunkBuffer.length;
118523
119261
  chunks.push(chunkBuffer);
118524
- passThrough.write(chunkBuffer);
118525
119262
  });
118526
119263
  rawBody.once("error", err => {
118527
119264
  const normalizedError = asError(err);
@@ -118540,11 +119277,48 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118540
119277
  });
118541
119278
  rawBody.once("end", () => {
118542
119279
  const fullBody = Buffer.concat(chunks).toString("utf8");
118543
- const extractedUsage = extractAnthropicNonStreamUsage(fullBody);
118544
- if (extractedUsage) {
118545
- usage.inputTokens = extractedUsage.inputTokens;
118546
- usage.outputTokens = extractedUsage.outputTokens;
119280
+ let outputBody = fullBody;
119281
+ if (needsTranslation && response.ok) {
119282
+ const modelId = requestBody.model ?? "unknown";
119283
+ try {
119284
+ outputBody = translateOpenAIResponseToAnthropic({
119285
+ body: fullBody,
119286
+ model: modelId
119287
+ });
119288
+ }
119289
+ catch (translateError) {
119290
+ const normalizedTranslateError = asError(translateError);
119291
+ logEngineMetrics({
119292
+ agentEngineType: engineType ?? "unknown",
119293
+ error: normalizedTranslateError,
119294
+ level: "error",
119295
+ logger,
119296
+ requestBodyBytes,
119297
+ requestPath: "/v1/messages",
119298
+ responseBytes,
119299
+ usage: null
119300
+ });
119301
+ finalize(normalizedTranslateError);
119302
+ passThrough.destroy(normalizedTranslateError);
119303
+ return;
119304
+ }
119305
+ }
119306
+ if (needsTranslation) {
119307
+ const extractedUsage = extractOpenAIUsageFromJSON(fullBody);
119308
+ if (extractedUsage) {
119309
+ usage.inputTokens = extractedUsage.inputTokens;
119310
+ usage.outputTokens = extractedUsage.outputTokens;
119311
+ }
119312
+ }
119313
+ else {
119314
+ const extractedUsage = extractAnthropicNonStreamUsage(fullBody);
119315
+ if (extractedUsage) {
119316
+ usage.inputTokens = extractedUsage.inputTokens;
119317
+ usage.outputTokens = extractedUsage.outputTokens;
119318
+ }
118547
119319
  }
119320
+ passThrough.write(outputBody);
119321
+ responseBytes = Buffer.byteLength(outputBody, "utf8");
118548
119322
  logEngineMetrics({
118549
119323
  agentEngineType: engineType ?? "unknown",
118550
119324
  level: upstreamError ? "error" : "info",
@@ -118579,9 +119353,17 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
118579
119353
  passThrough.end();
118580
119354
  });
118581
119355
  }
119356
+ const responseHeaders = Object.fromEntries(response.headers.entries());
119357
+ if (needsTranslation) {
119358
+ delete responseHeaders["content-length"];
119359
+ delete responseHeaders["content-encoding"];
119360
+ responseHeaders["content-type"] = streamRequested
119361
+ ? "text/event-stream"
119362
+ : "application/json";
119363
+ }
118582
119364
  return {
118583
119365
  body: passThrough,
118584
- headers: Object.fromEntries(response.headers.entries()),
119366
+ headers: responseHeaders,
118585
119367
  status: response.status
118586
119368
  };
118587
119369
  }
@@ -128382,6 +129164,58 @@ async function detectVLLMVersion() {
128382
129164
  return null;
128383
129165
  }
128384
129166
  }
129167
+ async function detectSGLangVersion() {
129168
+ try {
129169
+ const { stdout } = await execa(process.env.SGLANG_EXECUTABLE ?? "python3", [
129170
+ "-c",
129171
+ "import importlib.metadata as md; print(md.version('sglang'))"
129172
+ ]);
129173
+ const version = stdout.trim();
129174
+ return version.length > 0 ? version : null;
129175
+ }
129176
+ catch {
129177
+ return null;
129178
+ }
129179
+ }
129180
+ async function detectTensorRTLLMVersion() {
129181
+ try {
129182
+ const { stdout } = await execa(process.env.TRTLLM_EXECUTABLE ?? "python3", [
129183
+ "-c",
129184
+ "import importlib.metadata as md; print(md.version('tensorrt_llm'))"
129185
+ ]);
129186
+ const version = stdout.trim();
129187
+ return version.length > 0 ? version : null;
129188
+ }
129189
+ catch {
129190
+ return null;
129191
+ }
129192
+ }
129193
+ async function detectExllamav3Version() {
129194
+ try {
129195
+ const { stdout } = await execa(process.env.EXLLAMAV3_EXECUTABLE ?? "python3", [
129196
+ "-c",
129197
+ "import importlib.metadata as md; print(md.version('exllamav3'))"
129198
+ ]);
129199
+ const version = stdout.trim();
129200
+ return version.length > 0 ? version : null;
129201
+ }
129202
+ catch {
129203
+ return null;
129204
+ }
129205
+ }
129206
+ async function detectMLXLMVersion() {
129207
+ try {
129208
+ const { stdout } = await execa(process.env.MLXLM_EXECUTABLE ?? "python3", [
129209
+ "-c",
129210
+ "import importlib.metadata as md; print(md.version('mlx-lm'))"
129211
+ ]);
129212
+ const version = stdout.trim();
129213
+ return version.length > 0 ? version : null;
129214
+ }
129215
+ catch {
129216
+ return null;
129217
+ }
129218
+ }
128385
129219
  function normalizeMegabytes(value) {
128386
129220
  if (typeof value !== "number" || Number.isNaN(value)) {
128387
129221
  return null;
@@ -128668,6 +129502,7 @@ async function collectMachineMetadata() {
128668
129502
  model: cpuInfo?.brand ?? null,
128669
129503
  physicalCores: cpuInfo?.physicalCores ?? null
128670
129504
  },
129505
+ exllamav3Version: await detectExllamav3Version(),
128671
129506
  gpus,
128672
129507
  hostname: os.hostname(),
128673
129508
  llamaCppVersion: await detectLlamaCppVersion(),
@@ -128676,6 +129511,7 @@ async function collectMachineMetadata() {
128676
129511
  availableBytes: memInfo?.available ?? null,
128677
129512
  totalBytes: memInfo?.total ?? null
128678
129513
  },
129514
+ mlxlmVersion: await detectMLXLMVersion(),
128679
129515
  os: {
128680
129516
  arch: osInfo?.arch ?? os.arch(),
128681
129517
  platform: osInfo?.platform ?? os.platform(),
@@ -128683,6 +129519,8 @@ async function collectMachineMetadata() {
128683
129519
  type: osInfo?.kernel ?? null,
128684
129520
  version: osInfo?.build ?? null
128685
129521
  },
129522
+ sglangVersion: await detectSGLangVersion(),
129523
+ tensorrtLlmVersion: await detectTensorRTLLMVersion(),
128686
129524
  vllmVersion: await detectVLLMVersion()
128687
129525
  };
128688
129526
  return machineMetadata;
@@ -42,6 +42,7 @@ export declare class ModelManager extends EventEmitter<ModelManagerEvents> {
42
42
  get canStop(): boolean;
43
43
  get state(): EngineLifecycleState;
44
44
  private checkEngineReadiness;
45
+ private checkGenericHealthReadiness;
45
46
  private checkLlamacppReadiness;
46
47
  private checkVLLMReadiness;
47
48
  private waitForEngineReady;
@@ -0,0 +1,6 @@
1
+ import { ProcessManager } from "@infersec/utils";
2
+ import type { ModelManager } from "../ModelManager.js";
3
+ export declare function startExllamav3(this: ModelManager, { enginePort, targetDirectory }: {
4
+ enginePort: number;
5
+ targetDirectory: string;
6
+ }): Promise<ProcessManager>;
@@ -0,0 +1,6 @@
1
+ import { ProcessManager } from "@infersec/utils";
2
+ import type { ModelManager } from "./ModelManager.js";
3
+ export declare function startMLXLM(this: ModelManager, { enginePort, targetDirectory }: {
4
+ enginePort: number;
5
+ targetDirectory: string;
6
+ }): Promise<ProcessManager>;
@@ -0,0 +1,6 @@
1
+ import { ProcessManager } from "@infersec/utils";
2
+ import type { ModelManager } from "./ModelManager.js";
3
+ export declare function startSGLang(this: ModelManager, { enginePort, targetDirectory }: {
4
+ enginePort: number;
5
+ targetDirectory: string;
6
+ }): Promise<ProcessManager>;
@@ -0,0 +1,6 @@
1
+ import { ProcessManager } from "@infersec/utils";
2
+ import type { ModelManager } from "./ModelManager.js";
3
+ export declare function startTensorRTLLM(this: ModelManager, { enginePort, targetDirectory }: {
4
+ enginePort: number;
5
+ targetDirectory: string;
6
+ }): Promise<ProcessManager>;
@@ -0,0 +1,13 @@
1
+ import type { EngineConfig, LLMEngine } from "@infersec/definitions";
2
+ import { ENGINE_API_COMPATIBILITY } from "@infersec/definitions";
3
+ export declare function engineSupportsNativeAnthropic(engineType: LLMEngine | null): boolean;
4
+ export declare function translateAnthropicRequestToOpenAI(body: string): {
5
+ body: string;
6
+ path: string;
7
+ };
8
+ export declare function translateOpenAIResponseToAnthropic({ body, model }: {
9
+ body: string;
10
+ model: string;
11
+ }): string;
12
+ export { ENGINE_API_COMPATIBILITY };
13
+ export type { EngineConfig };
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@infersec/conduit",
3
3
  "description": "End user conduit agent for connecting local LLMs to the cloud.",
4
- "version": "1.75.0",
4
+ "version": "1.76.1",
5
5
  "bin": {
6
6
  "infersec-conduit": "./dist/cli.js"
7
7
  },