@infersec/conduit 1.74.1 → 1.76.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/benchmark/provision.d.ts +1 -1
- package/dist/benchmark/types.d.ts +1 -1
- package/dist/cli.js +915 -77
- package/dist/modelManagement/ModelManager.d.ts +1 -0
- package/dist/modelManagement/exllamav3/index.d.ts +6 -0
- package/dist/modelManagement/mlxlm.d.ts +6 -0
- package/dist/modelManagement/sglang.d.ts +6 -0
- package/dist/modelManagement/tensorrtllm.d.ts +6 -0
- package/dist/utils/anthropicTranslator/index.d.ts +13 -0
- package/package.json +1 -1
|
@@ -9,7 +9,7 @@ export declare function provisionResources(options: {
|
|
|
9
9
|
apiUrl: string;
|
|
10
10
|
apiKey: string;
|
|
11
11
|
contextLength: number;
|
|
12
|
-
engine: "llama.cpp" | "vllm";
|
|
12
|
+
engine: "exllamav3" | "llama.cpp" | "mlx-lm" | "sglang" | "tensorrt-llm" | "vllm";
|
|
13
13
|
format: string;
|
|
14
14
|
parallelism: number;
|
|
15
15
|
quantization: string | null;
|
|
@@ -17,7 +17,7 @@ export interface TestEntry {
|
|
|
17
17
|
};
|
|
18
18
|
concurrency: number;
|
|
19
19
|
contextLength: number;
|
|
20
|
-
engine: "llama.cpp" | "vllm";
|
|
20
|
+
engine: "exllamav3" | "llama.cpp" | "mlx-lm" | "sglang" | "tensorrt-llm" | "vllm";
|
|
21
21
|
format: string;
|
|
22
22
|
generate: boolean;
|
|
23
23
|
parallelism: number;
|
package/dist/cli.js
CHANGED
|
@@ -4348,13 +4348,29 @@ function getEffectiveContextLength({ contextLength, engineConfig, engineType })
|
|
|
4348
4348
|
if (contextLength === null || contextLength <= 0) {
|
|
4349
4349
|
return null;
|
|
4350
4350
|
}
|
|
4351
|
-
if (
|
|
4352
|
-
|
|
4353
|
-
|
|
4354
|
-
|
|
4351
|
+
if (!engineConfig) {
|
|
4352
|
+
return contextLength;
|
|
4353
|
+
}
|
|
4354
|
+
switch (engineType) {
|
|
4355
|
+
case "llama.cpp": {
|
|
4356
|
+
const parallelism = engineConfig.parallelism;
|
|
4357
|
+
if (typeof parallelism === "number" && parallelism > 0) {
|
|
4358
|
+
return contextLength / parallelism;
|
|
4359
|
+
}
|
|
4360
|
+
return contextLength;
|
|
4361
|
+
}
|
|
4362
|
+
case "sglang":
|
|
4363
|
+
case "tensorrt-llm":
|
|
4364
|
+
case "vllm": {
|
|
4365
|
+
const tensorParallelSize = engineConfig.tensorParallelSize;
|
|
4366
|
+
if (typeof tensorParallelSize === "number" && tensorParallelSize > 0) {
|
|
4367
|
+
return contextLength / tensorParallelSize;
|
|
4368
|
+
}
|
|
4369
|
+
return contextLength;
|
|
4355
4370
|
}
|
|
4371
|
+
default:
|
|
4372
|
+
return contextLength;
|
|
4356
4373
|
}
|
|
4357
|
-
return contextLength;
|
|
4358
4374
|
}
|
|
4359
4375
|
|
|
4360
4376
|
function asError(error) {
|
|
@@ -19889,7 +19905,14 @@ object({
|
|
|
19889
19905
|
PATCH: APIEndpointSchema.optional()
|
|
19890
19906
|
});
|
|
19891
19907
|
|
|
19892
|
-
const LLMEngineSchema = _enum([
|
|
19908
|
+
const LLMEngineSchema = _enum([
|
|
19909
|
+
"exllamav3",
|
|
19910
|
+
"llama.cpp",
|
|
19911
|
+
"mlx-lm",
|
|
19912
|
+
"sglang",
|
|
19913
|
+
"tensorrt-llm",
|
|
19914
|
+
"vllm"
|
|
19915
|
+
]);
|
|
19893
19916
|
const LlamacppEngineConfigSchema = object({
|
|
19894
19917
|
batchSize: number$1().int().positive().nullable().optional(),
|
|
19895
19918
|
cacheTypeK: string$1().nullable().optional(),
|
|
@@ -19908,19 +19931,59 @@ const VLLMEngineConfigSchema = object({
|
|
|
19908
19931
|
extraArgs: array(string$1()).optional(),
|
|
19909
19932
|
tensorParallelSize: number$1().int().positive().optional()
|
|
19910
19933
|
});
|
|
19934
|
+
const SGLangEngineConfigSchema = object({
|
|
19935
|
+
device: string$1().optional(),
|
|
19936
|
+
dtype: string$1().optional(),
|
|
19937
|
+
extraArgs: array(string$1()).optional(),
|
|
19938
|
+
tensorParallelSize: number$1().int().positive().optional()
|
|
19939
|
+
});
|
|
19940
|
+
const TensorRTLLMEngineConfigSchema = object({
|
|
19941
|
+
backend: _enum(["_autodeploy", "pytorch", "tensorrt"]).optional(),
|
|
19942
|
+
dtype: string$1().optional(),
|
|
19943
|
+
extraArgs: array(string$1()).optional(),
|
|
19944
|
+
tensorParallelSize: number$1().int().positive().optional()
|
|
19945
|
+
});
|
|
19946
|
+
const Exllamav3EngineConfigSchema = object({
|
|
19947
|
+
cacheMode: _enum(["fp16", "q4", "q6", "q8"]).optional(),
|
|
19948
|
+
extraArgs: array(string$1()).optional(),
|
|
19949
|
+
gpuSplit: string$1().optional(),
|
|
19950
|
+
maxSeqLen: number$1().int().positive().optional()
|
|
19951
|
+
});
|
|
19952
|
+
const MLXLMEngineConfigSchema = object({
|
|
19953
|
+
extraArgs: array(string$1()).optional(),
|
|
19954
|
+
maxKvSize: number$1().int().positive().optional(),
|
|
19955
|
+
trustRemoteCode: boolean$1().optional()
|
|
19956
|
+
});
|
|
19911
19957
|
const EngineConfigSchema = discriminatedUnion("type", [
|
|
19958
|
+
object({ config: Exllamav3EngineConfigSchema, type: literal("exllamav3") }),
|
|
19912
19959
|
object({ config: LlamacppEngineConfigSchema, type: literal("llama.cpp") }),
|
|
19960
|
+
object({ config: MLXLMEngineConfigSchema, type: literal("mlx-lm") }),
|
|
19961
|
+
object({ config: SGLangEngineConfigSchema, type: literal("sglang") }),
|
|
19962
|
+
object({
|
|
19963
|
+
config: TensorRTLLMEngineConfigSchema,
|
|
19964
|
+
type: literal("tensorrt-llm")
|
|
19965
|
+
}),
|
|
19913
19966
|
object({ config: VLLMEngineConfigSchema, type: literal("vllm") })
|
|
19914
19967
|
]);
|
|
19915
19968
|
const LLMModelFormatSchema = _enum([
|
|
19916
|
-
// VLLM
|
|
19969
|
+
// VLLM / SGLang / TensorRT-LLM
|
|
19917
19970
|
"safetensors",
|
|
19918
19971
|
"pytorch",
|
|
19919
19972
|
"awq",
|
|
19920
19973
|
"gptq",
|
|
19921
19974
|
// Llama.cpp
|
|
19922
|
-
"gguf"
|
|
19975
|
+
"gguf",
|
|
19976
|
+
// ExLlamaV3
|
|
19977
|
+
"exl3",
|
|
19978
|
+
"exl2",
|
|
19979
|
+
// MLX-LM
|
|
19980
|
+
"mlx"
|
|
19923
19981
|
]);
|
|
19982
|
+
object({
|
|
19983
|
+
nativeAnthropicMessages: boolean$1(),
|
|
19984
|
+
supportsEmbeddings: boolean$1(),
|
|
19985
|
+
supportsVision: boolean$1()
|
|
19986
|
+
});
|
|
19924
19987
|
const LLMModelTaskTypeSchema = _enum(["text-generation", "embeddings"]);
|
|
19925
19988
|
const LLMModelSchema = object({
|
|
19926
19989
|
format: LLMModelFormatSchema,
|
|
@@ -19980,6 +20043,7 @@ const InferenceAgentMachineMetadataSchema = object({
|
|
|
19980
20043
|
model: string$1().nullable(),
|
|
19981
20044
|
physicalCores: number$1().int().positive().nullable()
|
|
19982
20045
|
}),
|
|
20046
|
+
exllamav3Version: string$1().nullable(),
|
|
19983
20047
|
gpus: array(InferenceAgentMachineGPUSchema),
|
|
19984
20048
|
hostname: string$1(),
|
|
19985
20049
|
llamaCppVersion: string$1().nullable(),
|
|
@@ -19988,6 +20052,7 @@ const InferenceAgentMachineMetadataSchema = object({
|
|
|
19988
20052
|
availableBytes: number$1().int().nonnegative().nullable(),
|
|
19989
20053
|
totalBytes: number$1().int().nonnegative().nullable()
|
|
19990
20054
|
}),
|
|
20055
|
+
mlxlmVersion: string$1().nullable(),
|
|
19991
20056
|
os: object({
|
|
19992
20057
|
arch: string$1(),
|
|
19993
20058
|
platform: string$1(),
|
|
@@ -19995,6 +20060,8 @@ const InferenceAgentMachineMetadataSchema = object({
|
|
|
19995
20060
|
type: string$1().nullable(),
|
|
19996
20061
|
version: string$1().nullable()
|
|
19997
20062
|
}),
|
|
20063
|
+
sglangVersion: string$1().nullable(),
|
|
20064
|
+
tensorrtLlmVersion: string$1().nullable(),
|
|
19998
20065
|
vllmVersion: string$1().nullable()
|
|
19999
20066
|
});
|
|
20000
20067
|
const InferenceAgentMachineReportPayloadSchema = object({
|
|
@@ -20969,6 +21036,10 @@ object({
|
|
|
20969
21036
|
});
|
|
20970
21037
|
const EngineOutputSchema = object({
|
|
20971
21038
|
created: string$1(),
|
|
21039
|
+
exllamav3CacheMode: string$1().nullable(),
|
|
21040
|
+
exllamav3ExtraArgs: array(string$1()),
|
|
21041
|
+
exllamav3GpuSplit: string$1().nullable(),
|
|
21042
|
+
exllamav3MaxSeqLen: number$1().nullable(),
|
|
20972
21043
|
id: ULIDSchema,
|
|
20973
21044
|
llamacppBatchSize: number$1().nullable(),
|
|
20974
21045
|
llamacppCacheTypeK: string$1().nullable(),
|
|
@@ -20980,7 +21051,18 @@ const EngineOutputSchema = object({
|
|
|
20980
21051
|
llamacppParallelism: number$1(),
|
|
20981
21052
|
llamacppTensorSplit: string$1().nullable(),
|
|
20982
21053
|
llamacppUbatchSize: number$1().nullable(),
|
|
21054
|
+
mlxlmExtraArgs: array(string$1()),
|
|
21055
|
+
mlxlmMaxKvSize: number$1().nullable(),
|
|
21056
|
+
mlxlmTrustRemoteCode: boolean$1(),
|
|
20983
21057
|
name: string$1(),
|
|
21058
|
+
sglangDevice: string$1().nullable(),
|
|
21059
|
+
sglangDtype: string$1().nullable(),
|
|
21060
|
+
sglangExtraArgs: array(string$1()),
|
|
21061
|
+
sglangTensorParallelSize: number$1(),
|
|
21062
|
+
trtllmBackend: string$1().nullable(),
|
|
21063
|
+
trtllmDtype: string$1().nullable(),
|
|
21064
|
+
trtllmExtraArgs: array(string$1()),
|
|
21065
|
+
trtllmTensorParallelSize: number$1(),
|
|
20984
21066
|
type: LLMEngineSchema,
|
|
20985
21067
|
updated: string$1(),
|
|
20986
21068
|
vllmDevice: string$1().nullable(),
|
|
@@ -20989,6 +21071,10 @@ const EngineOutputSchema = object({
|
|
|
20989
21071
|
vllmTensorParallelSize: number$1()
|
|
20990
21072
|
});
|
|
20991
21073
|
object({
|
|
21074
|
+
exllamav3CacheMode: string$1().nullable().optional(),
|
|
21075
|
+
exllamav3ExtraArgs: array(string$1()).optional(),
|
|
21076
|
+
exllamav3GpuSplit: string$1().nullable().optional(),
|
|
21077
|
+
exllamav3MaxSeqLen: number$1().int().positive().nullable().optional(),
|
|
20992
21078
|
llamacppBatchSize: number$1().int().positive().nullable().optional(),
|
|
20993
21079
|
llamacppCacheTypeK: string$1().nullable().optional(),
|
|
20994
21080
|
llamacppCacheTypeV: string$1().nullable().optional(),
|
|
@@ -20999,7 +21085,18 @@ object({
|
|
|
20999
21085
|
llamacppParallelism: number$1().int().positive().optional(),
|
|
21000
21086
|
llamacppTensorSplit: string$1().nullable().optional(),
|
|
21001
21087
|
llamacppUbatchSize: number$1().int().positive().nullable().optional(),
|
|
21088
|
+
mlxlmExtraArgs: array(string$1()).optional(),
|
|
21089
|
+
mlxlmMaxKvSize: number$1().int().positive().nullable().optional(),
|
|
21090
|
+
mlxlmTrustRemoteCode: boolean$1().optional(),
|
|
21002
21091
|
name: ResourceNameSchema,
|
|
21092
|
+
sglangDevice: string$1().nullable().optional(),
|
|
21093
|
+
sglangDtype: string$1().nullable().optional(),
|
|
21094
|
+
sglangExtraArgs: array(string$1()).optional(),
|
|
21095
|
+
sglangTensorParallelSize: number$1().int().positive().optional(),
|
|
21096
|
+
trtllmBackend: string$1().nullable().optional(),
|
|
21097
|
+
trtllmDtype: string$1().nullable().optional(),
|
|
21098
|
+
trtllmExtraArgs: array(string$1()).optional(),
|
|
21099
|
+
trtllmTensorParallelSize: number$1().int().positive().optional(),
|
|
21003
21100
|
type: LLMEngineSchema,
|
|
21004
21101
|
vllmDevice: string$1().nullable().optional(),
|
|
21005
21102
|
vllmDtype: string$1().nullable().optional(),
|
|
@@ -21007,6 +21104,10 @@ object({
|
|
|
21007
21104
|
vllmTensorParallelSize: number$1().int().positive().optional()
|
|
21008
21105
|
});
|
|
21009
21106
|
object({
|
|
21107
|
+
exllamav3CacheMode: string$1().nullable().optional(),
|
|
21108
|
+
exllamav3ExtraArgs: array(string$1()).optional(),
|
|
21109
|
+
exllamav3GpuSplit: string$1().nullable().optional(),
|
|
21110
|
+
exllamav3MaxSeqLen: number$1().int().positive().nullable().optional(),
|
|
21010
21111
|
llamacppBatchSize: number$1().int().positive().nullable().optional(),
|
|
21011
21112
|
llamacppCacheTypeK: string$1().nullable().optional(),
|
|
21012
21113
|
llamacppCacheTypeV: string$1().nullable().optional(),
|
|
@@ -21017,7 +21118,18 @@ object({
|
|
|
21017
21118
|
llamacppParallelism: number$1().int().positive().optional(),
|
|
21018
21119
|
llamacppTensorSplit: string$1().nullable().optional(),
|
|
21019
21120
|
llamacppUbatchSize: number$1().int().positive().nullable().optional(),
|
|
21121
|
+
mlxlmExtraArgs: array(string$1()).optional(),
|
|
21122
|
+
mlxlmMaxKvSize: number$1().int().positive().nullable().optional(),
|
|
21123
|
+
mlxlmTrustRemoteCode: boolean$1().optional(),
|
|
21020
21124
|
name: ResourceNameSchema.optional(),
|
|
21125
|
+
sglangDevice: string$1().nullable().optional(),
|
|
21126
|
+
sglangDtype: string$1().nullable().optional(),
|
|
21127
|
+
sglangExtraArgs: array(string$1()).optional(),
|
|
21128
|
+
sglangTensorParallelSize: number$1().int().positive().optional(),
|
|
21129
|
+
trtllmBackend: string$1().nullable().optional(),
|
|
21130
|
+
trtllmDtype: string$1().nullable().optional(),
|
|
21131
|
+
trtllmExtraArgs: array(string$1()).optional(),
|
|
21132
|
+
trtllmTensorParallelSize: number$1().int().positive().optional(),
|
|
21021
21133
|
type: LLMEngineSchema.optional(),
|
|
21022
21134
|
vllmDevice: string$1().nullable().optional(),
|
|
21023
21135
|
vllmDtype: string$1().nullable().optional(),
|
|
@@ -21090,6 +21202,39 @@ object({
|
|
|
21090
21202
|
}
|
|
21091
21203
|
});
|
|
21092
21204
|
|
|
21205
|
+
const ENGINE_API_COMPATIBILITY = {
|
|
21206
|
+
exllamav3: {
|
|
21207
|
+
nativeAnthropicMessages: false,
|
|
21208
|
+
supportsEmbeddings: false,
|
|
21209
|
+
supportsVision: true
|
|
21210
|
+
},
|
|
21211
|
+
"llama.cpp": {
|
|
21212
|
+
nativeAnthropicMessages: true,
|
|
21213
|
+
supportsEmbeddings: true,
|
|
21214
|
+
supportsVision: true
|
|
21215
|
+
},
|
|
21216
|
+
"mlx-lm": {
|
|
21217
|
+
nativeAnthropicMessages: false,
|
|
21218
|
+
supportsEmbeddings: false,
|
|
21219
|
+
supportsVision: false
|
|
21220
|
+
},
|
|
21221
|
+
sglang: {
|
|
21222
|
+
nativeAnthropicMessages: false,
|
|
21223
|
+
supportsEmbeddings: true,
|
|
21224
|
+
supportsVision: true
|
|
21225
|
+
},
|
|
21226
|
+
"tensorrt-llm": {
|
|
21227
|
+
nativeAnthropicMessages: false,
|
|
21228
|
+
supportsEmbeddings: true,
|
|
21229
|
+
supportsVision: true
|
|
21230
|
+
},
|
|
21231
|
+
vllm: {
|
|
21232
|
+
nativeAnthropicMessages: true,
|
|
21233
|
+
supportsEmbeddings: true,
|
|
21234
|
+
supportsVision: true
|
|
21235
|
+
}
|
|
21236
|
+
};
|
|
21237
|
+
|
|
21093
21238
|
object({
|
|
21094
21239
|
accountID: ULIDSchema.optional(),
|
|
21095
21240
|
email: string$1().email(),
|
|
@@ -116639,6 +116784,43 @@ async function downloadFileWithRange({ accessToken, filePath, fileSize, modelSlu
|
|
|
116639
116784
|
throw new Error(errorMessage);
|
|
116640
116785
|
}
|
|
116641
116786
|
|
|
116787
|
+
const EXLLAMAV3_EXECUTABLE = process.env.EXLLAMAV3_EXECUTABLE ?? "python3";
|
|
116788
|
+
const SERVER_SCRIPT = join(import.meta.dirname, "server.py");
|
|
116789
|
+
const DEFAULT_EXLLAMAV3_CONTEXT_LENGTH = 4096;
|
|
116790
|
+
async function startExllamav3({ enginePort, targetDirectory }) {
|
|
116791
|
+
const contextLength = Math.max(1, this.contextLength ?? DEFAULT_EXLLAMAV3_CONTEXT_LENGTH);
|
|
116792
|
+
const engineConfig = this.engineConfig;
|
|
116793
|
+
const cacheMode = typeof engineConfig?.cacheMode === "string" ? engineConfig.cacheMode : "q4";
|
|
116794
|
+
const gpuSplit = typeof engineConfig?.gpuSplit === "string" ? engineConfig.gpuSplit : null;
|
|
116795
|
+
const maxSeqLen = typeof engineConfig?.maxSeqLen === "number" ? engineConfig.maxSeqLen : contextLength;
|
|
116796
|
+
const args = [
|
|
116797
|
+
SERVER_SCRIPT,
|
|
116798
|
+
"--model",
|
|
116799
|
+
targetDirectory,
|
|
116800
|
+
"--host",
|
|
116801
|
+
"127.0.0.1",
|
|
116802
|
+
"--port",
|
|
116803
|
+
String(enginePort),
|
|
116804
|
+
"--cache-mode",
|
|
116805
|
+
cacheMode,
|
|
116806
|
+
"--max-seq-len",
|
|
116807
|
+
String(maxSeqLen)
|
|
116808
|
+
];
|
|
116809
|
+
if (gpuSplit) {
|
|
116810
|
+
args.push("--gpu-split", gpuSplit);
|
|
116811
|
+
}
|
|
116812
|
+
const extraArgs = engineConfig?.extraArgs;
|
|
116813
|
+
if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
|
|
116814
|
+
args.push(...extraArgs);
|
|
116815
|
+
}
|
|
116816
|
+
const processManager = new ProcessManager({
|
|
116817
|
+
command: EXLLAMAV3_EXECUTABLE,
|
|
116818
|
+
args
|
|
116819
|
+
});
|
|
116820
|
+
await processManager.start();
|
|
116821
|
+
return processManager;
|
|
116822
|
+
}
|
|
116823
|
+
|
|
116642
116824
|
const DEFAULT_LLAMACPP_GPU_LAYERS = 999;
|
|
116643
116825
|
const LLAMACPP_START_ARGS = ["--host", "0.0.0.0", "--jinja"];
|
|
116644
116826
|
const LLAMACPP_EXECUTABLE = process.env.LLAMACPP_EXECUTABLE ?? "llama-server";
|
|
@@ -116738,6 +116920,46 @@ async function startLlamacpp({ enginePort, targetDirectory }) {
|
|
|
116738
116920
|
return processManager;
|
|
116739
116921
|
}
|
|
116740
116922
|
|
|
116923
|
+
const MLXLM_EXECUTABLE = process.env.MLXLM_EXECUTABLE ?? "python3";
|
|
116924
|
+
const DEFAULT_MLXLM_CONTEXT_LENGTH = 4096;
|
|
116925
|
+
async function startMLXLM({ enginePort, targetDirectory }) {
|
|
116926
|
+
if (this.model.taskType === "embeddings") {
|
|
116927
|
+
throw new Error("MLX-LM engine does not support embeddings task type");
|
|
116928
|
+
}
|
|
116929
|
+
const contextLength = Math.max(1, this.contextLength ?? DEFAULT_MLXLM_CONTEXT_LENGTH);
|
|
116930
|
+
const engineConfig = this.engineConfig;
|
|
116931
|
+
const args = [
|
|
116932
|
+
"-m",
|
|
116933
|
+
"mlx_lm.server",
|
|
116934
|
+
"--model",
|
|
116935
|
+
targetDirectory,
|
|
116936
|
+
"--host",
|
|
116937
|
+
"127.0.0.1",
|
|
116938
|
+
"--port",
|
|
116939
|
+
String(enginePort),
|
|
116940
|
+
"--context-length",
|
|
116941
|
+
String(contextLength)
|
|
116942
|
+
];
|
|
116943
|
+
const maxKvSize = typeof engineConfig?.maxKvSize === "number" ? engineConfig.maxKvSize : null;
|
|
116944
|
+
if (maxKvSize !== null) {
|
|
116945
|
+
args.push("--max-kv-size", String(maxKvSize));
|
|
116946
|
+
}
|
|
116947
|
+
const trustRemoteCode = engineConfig?.trustRemoteCode === true;
|
|
116948
|
+
if (trustRemoteCode || process.env.MLXLM_TRUST_REMOTE_CODE === "true") {
|
|
116949
|
+
args.push("--trust-remote-code");
|
|
116950
|
+
}
|
|
116951
|
+
const extraArgs = engineConfig?.extraArgs;
|
|
116952
|
+
if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
|
|
116953
|
+
args.push(...extraArgs);
|
|
116954
|
+
}
|
|
116955
|
+
const processManager = new ProcessManager({
|
|
116956
|
+
command: MLXLM_EXECUTABLE,
|
|
116957
|
+
args
|
|
116958
|
+
});
|
|
116959
|
+
await processManager.start();
|
|
116960
|
+
return processManager;
|
|
116961
|
+
}
|
|
116962
|
+
|
|
116741
116963
|
const SAFE_CHARS = /[a-zA-Z0-9\-_.]/;
|
|
116742
116964
|
const SEPARATOR = "__";
|
|
116743
116965
|
function sanitizeSegment(value) {
|
|
@@ -116762,6 +116984,95 @@ function createModelStorageKey(model) {
|
|
|
116762
116984
|
return `${model.source.type}${SEPARATOR}${sanitizeSegment(identifier)}`;
|
|
116763
116985
|
}
|
|
116764
116986
|
|
|
116987
|
+
const SGLANG_START_ARGS = ["-m", "sglang.launch_server", "--host", "127.0.0.1"];
|
|
116988
|
+
const SGLANG_EXECUTABLE = process.env.SGLANG_EXECUTABLE ?? "python3";
|
|
116989
|
+
const DEFAULT_SGLANG_CONTEXT_LENGTH = 2048;
|
|
116990
|
+
async function startSGLang({ enginePort, targetDirectory }) {
|
|
116991
|
+
const contextLength = Math.max(1, this.contextLength ?? DEFAULT_SGLANG_CONTEXT_LENGTH);
|
|
116992
|
+
const engineConfig = this.engineConfig;
|
|
116993
|
+
const device = typeof engineConfig?.device === "string" ? engineConfig.device : undefined;
|
|
116994
|
+
const dtype = typeof engineConfig?.dtype === "string" ? engineConfig.dtype : undefined;
|
|
116995
|
+
const tensorParallelSize = typeof engineConfig?.tensorParallelSize === "number" ? engineConfig.tensorParallelSize : 1;
|
|
116996
|
+
const args = [
|
|
116997
|
+
...SGLANG_START_ARGS,
|
|
116998
|
+
"--port",
|
|
116999
|
+
String(enginePort),
|
|
117000
|
+
"--model-path",
|
|
117001
|
+
targetDirectory,
|
|
117002
|
+
"--served-model-name",
|
|
117003
|
+
this.model.id,
|
|
117004
|
+
"--context-length",
|
|
117005
|
+
String(contextLength),
|
|
117006
|
+
"--tp-size",
|
|
117007
|
+
String(tensorParallelSize)
|
|
117008
|
+
];
|
|
117009
|
+
if (this.model.taskType === "embeddings") {
|
|
117010
|
+
args.push("--task", "embed");
|
|
117011
|
+
}
|
|
117012
|
+
if (device) {
|
|
117013
|
+
args.push("--device", device);
|
|
117014
|
+
}
|
|
117015
|
+
if (dtype) {
|
|
117016
|
+
args.push("--dtype", dtype);
|
|
117017
|
+
}
|
|
117018
|
+
const extraArgs = engineConfig?.extraArgs;
|
|
117019
|
+
if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
|
|
117020
|
+
args.push(...extraArgs);
|
|
117021
|
+
}
|
|
117022
|
+
if (this.model.multimodalEnabled) {
|
|
117023
|
+
args.push("--limit-mm-per-prompt", process.env.SGLANG_MM_LIMIT ?? '{"image":5}');
|
|
117024
|
+
}
|
|
117025
|
+
if (process.env.SGLANG_TRUST_REMOTE_CODE === "true") {
|
|
117026
|
+
args.push("--trust-remote-code");
|
|
117027
|
+
}
|
|
117028
|
+
const processManager = new ProcessManager({
|
|
117029
|
+
command: SGLANG_EXECUTABLE,
|
|
117030
|
+
args
|
|
117031
|
+
});
|
|
117032
|
+
await processManager.start();
|
|
117033
|
+
return processManager;
|
|
117034
|
+
}
|
|
117035
|
+
|
|
117036
|
+
const TRTLLM_EXECUTABLE = process.env.TRTLLM_EXECUTABLE ?? "trtllm-serve";
|
|
117037
|
+
const DEFAULT_TRTLLM_CONTEXT_LENGTH = 2048;
|
|
117038
|
+
async function startTensorRTLLM({ enginePort, targetDirectory }) {
|
|
117039
|
+
const contextLength = Math.max(1, this.contextLength ?? DEFAULT_TRTLLM_CONTEXT_LENGTH);
|
|
117040
|
+
const engineConfig = this.engineConfig;
|
|
117041
|
+
const backend = typeof engineConfig?.backend === "string" ? engineConfig.backend : "pytorch";
|
|
117042
|
+
const dtype = typeof engineConfig?.dtype === "string" ? engineConfig.dtype : undefined;
|
|
117043
|
+
const tensorParallelSize = typeof engineConfig?.tensorParallelSize === "number" ? engineConfig.tensorParallelSize : 1;
|
|
117044
|
+
const args = [
|
|
117045
|
+
"serve",
|
|
117046
|
+
targetDirectory,
|
|
117047
|
+
"--host",
|
|
117048
|
+
"127.0.0.1",
|
|
117049
|
+
"--port",
|
|
117050
|
+
String(enginePort),
|
|
117051
|
+
"--backend",
|
|
117052
|
+
backend,
|
|
117053
|
+
"--max-seq-len",
|
|
117054
|
+
String(contextLength),
|
|
117055
|
+
"--tp-size",
|
|
117056
|
+
String(tensorParallelSize)
|
|
117057
|
+
];
|
|
117058
|
+
if (this.model.taskType === "embeddings") {
|
|
117059
|
+
args.push("--task", "embed");
|
|
117060
|
+
}
|
|
117061
|
+
if (dtype) {
|
|
117062
|
+
args.push("--dtype", dtype);
|
|
117063
|
+
}
|
|
117064
|
+
const extraArgs = engineConfig?.extraArgs;
|
|
117065
|
+
if (Array.isArray(extraArgs) && extraArgs.every((v) => typeof v === "string")) {
|
|
117066
|
+
args.push(...extraArgs);
|
|
117067
|
+
}
|
|
117068
|
+
const processManager = new ProcessManager({
|
|
117069
|
+
command: TRTLLM_EXECUTABLE,
|
|
117070
|
+
args
|
|
117071
|
+
});
|
|
117072
|
+
await processManager.start();
|
|
117073
|
+
return processManager;
|
|
117074
|
+
}
|
|
117075
|
+
|
|
116765
117076
|
const ENGINE_FETCH_TIMEOUT_MS$1 = 7200000;
|
|
116766
117077
|
const DOWNLOAD_LOCK_TIMEOUT_MS = 20 * 60 * 1000;
|
|
116767
117078
|
const DOWNLOAD_LOCK_POLL_INTERVAL_MS = 5000;
|
|
@@ -116795,7 +117106,11 @@ class ModelManager extends EventEmitter {
|
|
|
116795
117106
|
}
|
|
116796
117107
|
async fetchOpenAI(path, opts) {
|
|
116797
117108
|
switch (this.engine) {
|
|
117109
|
+
case "exllamav3":
|
|
116798
117110
|
case "llama.cpp":
|
|
117111
|
+
case "mlx-lm":
|
|
117112
|
+
case "sglang":
|
|
117113
|
+
case "tensorrt-llm":
|
|
116799
117114
|
case "vllm": {
|
|
116800
117115
|
this.logger.debug(`Fetching from engine: ${path}`);
|
|
116801
117116
|
const callerSignal = opts?.signal;
|
|
@@ -116843,7 +117158,11 @@ class ModelManager extends EventEmitter {
|
|
|
116843
117158
|
modelID: this.model.id
|
|
116844
117159
|
});
|
|
116845
117160
|
switch (this.engine) {
|
|
117161
|
+
case "exllamav3":
|
|
116846
117162
|
case "llama.cpp":
|
|
117163
|
+
case "mlx-lm":
|
|
117164
|
+
case "sglang":
|
|
117165
|
+
case "tensorrt-llm":
|
|
116847
117166
|
case "vllm":
|
|
116848
117167
|
if (this.model.source.type !== "huggingface") {
|
|
116849
117168
|
throw new Error(`Model source not implemented: ${this.model.source.type}`);
|
|
@@ -116950,10 +117269,34 @@ class ModelManager extends EventEmitter {
|
|
|
116950
117269
|
case "vllm": {
|
|
116951
117270
|
return this.checkVLLMReadiness();
|
|
116952
117271
|
}
|
|
117272
|
+
case "exllamav3":
|
|
117273
|
+
case "mlx-lm":
|
|
117274
|
+
case "sglang":
|
|
117275
|
+
case "tensorrt-llm": {
|
|
117276
|
+
return this.checkGenericHealthReadiness();
|
|
117277
|
+
}
|
|
116953
117278
|
default:
|
|
116954
117279
|
return "ready";
|
|
116955
117280
|
}
|
|
116956
117281
|
}
|
|
117282
|
+
async checkGenericHealthReadiness() {
|
|
117283
|
+
try {
|
|
117284
|
+
const response = await undiciExports.fetch(joinURL(`http://localhost:${this.enginePort}`, "/health"), {
|
|
117285
|
+
method: "GET",
|
|
117286
|
+
signal: AbortSignal.timeout(5000)
|
|
117287
|
+
});
|
|
117288
|
+
if (response.status === 503) {
|
|
117289
|
+
return "loading";
|
|
117290
|
+
}
|
|
117291
|
+
if (response.ok) {
|
|
117292
|
+
return "ready";
|
|
117293
|
+
}
|
|
117294
|
+
return "loading";
|
|
117295
|
+
}
|
|
117296
|
+
catch (_error) {
|
|
117297
|
+
return "unreachable";
|
|
117298
|
+
}
|
|
117299
|
+
}
|
|
116957
117300
|
async checkLlamacppReadiness() {
|
|
116958
117301
|
try {
|
|
116959
117302
|
const response = await undiciExports.fetch(joinURL(`http://localhost:${this.enginePort}`, "/health"), {
|
|
@@ -117150,16 +117493,37 @@ class ModelManager extends EventEmitter {
|
|
|
117150
117493
|
});
|
|
117151
117494
|
}
|
|
117152
117495
|
async startEngineProcess() {
|
|
117496
|
+
const targetDir = join(this.modelsDirectory, this.uniqueName);
|
|
117153
117497
|
switch (this.engine) {
|
|
117498
|
+
case "exllamav3":
|
|
117499
|
+
return startExllamav3.call(this, {
|
|
117500
|
+
enginePort: this.enginePort,
|
|
117501
|
+
targetDirectory: targetDir
|
|
117502
|
+
});
|
|
117154
117503
|
case "llama.cpp":
|
|
117155
117504
|
return startLlamacpp.call(this, {
|
|
117156
117505
|
enginePort: this.enginePort,
|
|
117157
|
-
targetDirectory:
|
|
117506
|
+
targetDirectory: targetDir
|
|
117507
|
+
});
|
|
117508
|
+
case "mlx-lm":
|
|
117509
|
+
return startMLXLM.call(this, {
|
|
117510
|
+
enginePort: this.enginePort,
|
|
117511
|
+
targetDirectory: targetDir
|
|
117512
|
+
});
|
|
117513
|
+
case "sglang":
|
|
117514
|
+
return startSGLang.call(this, {
|
|
117515
|
+
enginePort: this.enginePort,
|
|
117516
|
+
targetDirectory: targetDir
|
|
117517
|
+
});
|
|
117518
|
+
case "tensorrt-llm":
|
|
117519
|
+
return startTensorRTLLM.call(this, {
|
|
117520
|
+
enginePort: this.enginePort,
|
|
117521
|
+
targetDirectory: targetDir
|
|
117158
117522
|
});
|
|
117159
117523
|
case "vllm":
|
|
117160
117524
|
return startVLLM.call(this, {
|
|
117161
117525
|
enginePort: this.enginePort,
|
|
117162
|
-
targetDirectory:
|
|
117526
|
+
targetDirectory: targetDir
|
|
117163
117527
|
});
|
|
117164
117528
|
default: {
|
|
117165
117529
|
const engineType = this.engine;
|
|
@@ -118227,6 +118591,189 @@ function createPostEmbeddingsHandler(options) {
|
|
|
118227
118591
|
return createConduitOpenAIAPIReferenceHandlers(options)["/v1/embeddings"].POST;
|
|
118228
118592
|
}
|
|
118229
118593
|
|
|
118594
|
+
function engineSupportsNativeAnthropic(engineType) {
|
|
118595
|
+
if (!engineType)
|
|
118596
|
+
return false;
|
|
118597
|
+
return ENGINE_API_COMPATIBILITY[engineType]?.nativeAnthropicMessages ?? false;
|
|
118598
|
+
}
|
|
118599
|
+
function translateAnthropicRequestToOpenAI(body) {
|
|
118600
|
+
const parsed = JSON.parse(body);
|
|
118601
|
+
const messages = Array.isArray(parsed.messages) ? parsed.messages : [];
|
|
118602
|
+
const openaiMessages = [];
|
|
118603
|
+
const system = parsed.system;
|
|
118604
|
+
if (typeof system === "string" && system.length > 0) {
|
|
118605
|
+
openaiMessages.push({ content: system, role: "system" });
|
|
118606
|
+
}
|
|
118607
|
+
else if (Array.isArray(system)) {
|
|
118608
|
+
const textParts = system
|
|
118609
|
+
.filter((b) => {
|
|
118610
|
+
const block = b;
|
|
118611
|
+
return block?.type === "text" && typeof block.text === "string";
|
|
118612
|
+
})
|
|
118613
|
+
.map((b) => b.text);
|
|
118614
|
+
if (textParts.length > 0) {
|
|
118615
|
+
openaiMessages.push({ content: textParts.join("\n"), role: "system" });
|
|
118616
|
+
}
|
|
118617
|
+
}
|
|
118618
|
+
for (const msg of messages) {
|
|
118619
|
+
const m = msg;
|
|
118620
|
+
const role = m.role;
|
|
118621
|
+
const content = m.content;
|
|
118622
|
+
if (typeof content === "string") {
|
|
118623
|
+
openaiMessages.push({ content, role });
|
|
118624
|
+
continue;
|
|
118625
|
+
}
|
|
118626
|
+
if (!Array.isArray(content)) {
|
|
118627
|
+
openaiMessages.push({ content: content ?? "", role });
|
|
118628
|
+
continue;
|
|
118629
|
+
}
|
|
118630
|
+
if (role === "assistant") {
|
|
118631
|
+
const toolCalls = [];
|
|
118632
|
+
const textParts = [];
|
|
118633
|
+
for (const block of content) {
|
|
118634
|
+
const b = block;
|
|
118635
|
+
if (b.type === "text" && typeof b.text === "string") {
|
|
118636
|
+
textParts.push(b.text);
|
|
118637
|
+
}
|
|
118638
|
+
else if (b.type === "tool_use") {
|
|
118639
|
+
toolCalls.push({
|
|
118640
|
+
function: {
|
|
118641
|
+
arguments: JSON.stringify(b.input ?? {}),
|
|
118642
|
+
name: b.name
|
|
118643
|
+
},
|
|
118644
|
+
id: b.id,
|
|
118645
|
+
type: "function"
|
|
118646
|
+
});
|
|
118647
|
+
}
|
|
118648
|
+
}
|
|
118649
|
+
openaiMessages.push({
|
|
118650
|
+
content: textParts.join("") || null,
|
|
118651
|
+
role,
|
|
118652
|
+
...(toolCalls.length > 0 ? { tool_calls: toolCalls } : {})
|
|
118653
|
+
});
|
|
118654
|
+
}
|
|
118655
|
+
else if (role === "user") {
|
|
118656
|
+
const textParts = [];
|
|
118657
|
+
const toolResults = [];
|
|
118658
|
+
for (const block of content) {
|
|
118659
|
+
const b = block;
|
|
118660
|
+
if (b.type === "text" && typeof b.text === "string") {
|
|
118661
|
+
textParts.push(b.text);
|
|
118662
|
+
}
|
|
118663
|
+
else if (b.type === "tool_result") {
|
|
118664
|
+
toolResults.push({
|
|
118665
|
+
content: typeof b.content === "string"
|
|
118666
|
+
? b.content
|
|
118667
|
+
: JSON.stringify(b.content ?? ""),
|
|
118668
|
+
role: "tool",
|
|
118669
|
+
tool_call_id: b.tool_use_id
|
|
118670
|
+
});
|
|
118671
|
+
}
|
|
118672
|
+
}
|
|
118673
|
+
if (textParts.length > 0) {
|
|
118674
|
+
openaiMessages.push({ content: textParts.join("\n"), role });
|
|
118675
|
+
}
|
|
118676
|
+
for (const tr of toolResults) {
|
|
118677
|
+
openaiMessages.push(tr);
|
|
118678
|
+
}
|
|
118679
|
+
}
|
|
118680
|
+
else {
|
|
118681
|
+
openaiMessages.push({ content: JSON.stringify(content), role });
|
|
118682
|
+
}
|
|
118683
|
+
}
|
|
118684
|
+
const result = {
|
|
118685
|
+
max_tokens: parsed.max_tokens ?? 4096,
|
|
118686
|
+
messages: openaiMessages,
|
|
118687
|
+
model: parsed.model,
|
|
118688
|
+
stream: parsed.stream ?? false
|
|
118689
|
+
};
|
|
118690
|
+
if (typeof parsed.temperature === "number")
|
|
118691
|
+
result.temperature = parsed.temperature;
|
|
118692
|
+
if (typeof parsed.top_p === "number")
|
|
118693
|
+
result.top_p = parsed.top_p;
|
|
118694
|
+
if (Array.isArray(parsed.stop_sequences))
|
|
118695
|
+
result.stop = parsed.stop_sequences;
|
|
118696
|
+
if (Array.isArray(parsed.tools)) {
|
|
118697
|
+
result.tools = parsed.tools.map((tool) => {
|
|
118698
|
+
const t = tool;
|
|
118699
|
+
return {
|
|
118700
|
+
function: {
|
|
118701
|
+
...(typeof t.description === "string" ? { description: t.description } : {}),
|
|
118702
|
+
name: t.name,
|
|
118703
|
+
parameters: t.input_schema ?? {}
|
|
118704
|
+
},
|
|
118705
|
+
type: "function"
|
|
118706
|
+
};
|
|
118707
|
+
});
|
|
118708
|
+
}
|
|
118709
|
+
if (parsed.tool_choice && typeof parsed.tool_choice === "object") {
|
|
118710
|
+
const tc = parsed.tool_choice;
|
|
118711
|
+
if (tc.type === "auto")
|
|
118712
|
+
result.tool_choice = "auto";
|
|
118713
|
+
else if (tc.type === "any")
|
|
118714
|
+
result.tool_choice = "required";
|
|
118715
|
+
else if (tc.type === "tool" && typeof tc.name === "string") {
|
|
118716
|
+
result.tool_choice = { function: { name: tc.name }, type: "function" };
|
|
118717
|
+
}
|
|
118718
|
+
}
|
|
118719
|
+
return { body: JSON.stringify(result), path: "/v1/chat/completions" };
|
|
118720
|
+
}
|
|
118721
|
+
function translateOpenAIResponseToAnthropic({ body, model }) {
|
|
118722
|
+
const parsed = JSON.parse(body);
|
|
118723
|
+
const choices = Array.isArray(parsed.choices) ? parsed.choices : [];
|
|
118724
|
+
const choice = choices[0];
|
|
118725
|
+
const message = choice?.message;
|
|
118726
|
+
const content = [];
|
|
118727
|
+
if (message) {
|
|
118728
|
+
if (typeof message.content === "string" && message.content.length > 0) {
|
|
118729
|
+
content.push({ text: message.content, type: "text" });
|
|
118730
|
+
}
|
|
118731
|
+
if (Array.isArray(message.tool_calls)) {
|
|
118732
|
+
for (const tc of message.tool_calls) {
|
|
118733
|
+
const call = tc;
|
|
118734
|
+
const fn = call.function;
|
|
118735
|
+
const argsRaw = typeof fn?.arguments === "string" ? fn.arguments : "{}";
|
|
118736
|
+
let parsedInput = {};
|
|
118737
|
+
try {
|
|
118738
|
+
parsedInput = JSON.parse(argsRaw);
|
|
118739
|
+
}
|
|
118740
|
+
catch {
|
|
118741
|
+
parsedInput = {};
|
|
118742
|
+
}
|
|
118743
|
+
content.push({
|
|
118744
|
+
id: call.id,
|
|
118745
|
+
input: parsedInput,
|
|
118746
|
+
name: fn?.name,
|
|
118747
|
+
type: "tool_use"
|
|
118748
|
+
});
|
|
118749
|
+
}
|
|
118750
|
+
}
|
|
118751
|
+
}
|
|
118752
|
+
const finishReason = choice?.finish_reason;
|
|
118753
|
+
let stopReason = "end_turn";
|
|
118754
|
+
if (finishReason === "length")
|
|
118755
|
+
stopReason = "max_tokens";
|
|
118756
|
+
else if (finishReason === "tool_calls")
|
|
118757
|
+
stopReason = "tool_use";
|
|
118758
|
+
else if (finishReason === "stop")
|
|
118759
|
+
stopReason = "end_turn";
|
|
118760
|
+
const usage = parsed.usage;
|
|
118761
|
+
const result = {
|
|
118762
|
+
content,
|
|
118763
|
+
id: parsed.id,
|
|
118764
|
+
model,
|
|
118765
|
+
role: "assistant",
|
|
118766
|
+
stop_reason: stopReason,
|
|
118767
|
+
stop_sequence: null,
|
|
118768
|
+
type: "message",
|
|
118769
|
+
usage: {
|
|
118770
|
+
input_tokens: usage?.prompt_tokens ?? 0,
|
|
118771
|
+
output_tokens: usage?.completion_tokens ?? 0
|
|
118772
|
+
}
|
|
118773
|
+
};
|
|
118774
|
+
return JSON.stringify(result);
|
|
118775
|
+
}
|
|
118776
|
+
|
|
118230
118777
|
function isPlainObject$1(value) {
|
|
118231
118778
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
118232
118779
|
}
|
|
@@ -118297,6 +118844,106 @@ function extractAnthropicNonStreamUsage(body) {
|
|
|
118297
118844
|
return null;
|
|
118298
118845
|
}
|
|
118299
118846
|
}
|
|
118847
|
+
function extractOpenAIUsageFromJSON(body) {
|
|
118848
|
+
try {
|
|
118849
|
+
const parsed = JSON.parse(body);
|
|
118850
|
+
if (!isPlainObject$1(parsed) || !isPlainObject$1(parsed.usage))
|
|
118851
|
+
return null;
|
|
118852
|
+
const usage = parsed.usage;
|
|
118853
|
+
return {
|
|
118854
|
+
inputTokens: typeof usage.prompt_tokens === "number" ? usage.prompt_tokens : null,
|
|
118855
|
+
outputTokens: typeof usage.completion_tokens === "number" ? usage.completion_tokens : null
|
|
118856
|
+
};
|
|
118857
|
+
}
|
|
118858
|
+
catch {
|
|
118859
|
+
return null;
|
|
118860
|
+
}
|
|
118861
|
+
}
|
|
118862
|
+
function assembleOpenAISSE(body) {
|
|
118863
|
+
const lines = body.split("\n");
|
|
118864
|
+
let content = "";
|
|
118865
|
+
let promptTokens = null;
|
|
118866
|
+
let completionTokens = null;
|
|
118867
|
+
for (const line of lines) {
|
|
118868
|
+
const trimmed = line.trim();
|
|
118869
|
+
if (!trimmed.startsWith("data:"))
|
|
118870
|
+
continue;
|
|
118871
|
+
const payload = trimmed.slice(5).trim();
|
|
118872
|
+
if (payload === "[DONE]")
|
|
118873
|
+
continue;
|
|
118874
|
+
try {
|
|
118875
|
+
const parsed = JSON.parse(payload);
|
|
118876
|
+
if (!isPlainObject$1(parsed))
|
|
118877
|
+
continue;
|
|
118878
|
+
const choices = parsed.choices;
|
|
118879
|
+
if (Array.isArray(choices) && choices.length > 0) {
|
|
118880
|
+
const choice = choices[0];
|
|
118881
|
+
const delta = choice.delta;
|
|
118882
|
+
if (delta && typeof delta.content === "string") {
|
|
118883
|
+
content += delta.content;
|
|
118884
|
+
}
|
|
118885
|
+
}
|
|
118886
|
+
const usage = parsed.usage;
|
|
118887
|
+
if (usage) {
|
|
118888
|
+
if (typeof usage.prompt_tokens === "number")
|
|
118889
|
+
promptTokens = usage.prompt_tokens;
|
|
118890
|
+
if (typeof usage.completion_tokens === "number")
|
|
118891
|
+
completionTokens = usage.completion_tokens;
|
|
118892
|
+
}
|
|
118893
|
+
}
|
|
118894
|
+
catch {
|
|
118895
|
+
// ignore
|
|
118896
|
+
}
|
|
118897
|
+
}
|
|
118898
|
+
return { completionTokens, content, promptTokens };
|
|
118899
|
+
}
|
|
118900
|
+
function translateOpenAIStreamToAnthropicSSE(body, model) {
|
|
118901
|
+
const assembled = assembleOpenAISSE(body);
|
|
118902
|
+
const content = assembled?.content ?? "";
|
|
118903
|
+
const inputTokens = assembled?.promptTokens ?? 0;
|
|
118904
|
+
const outputTokens = assembled?.completionTokens ?? 0;
|
|
118905
|
+
const events = [];
|
|
118906
|
+
const messageId = `msg_${Date.now()}`;
|
|
118907
|
+
events.push(`event: message_start\ndata: ${JSON.stringify({
|
|
118908
|
+
message: {
|
|
118909
|
+
content: [],
|
|
118910
|
+
id: messageId,
|
|
118911
|
+
input_tokens: inputTokens,
|
|
118912
|
+
model,
|
|
118913
|
+
output_tokens: outputTokens,
|
|
118914
|
+
role: "assistant",
|
|
118915
|
+
stop_reason: null,
|
|
118916
|
+
type: "message",
|
|
118917
|
+
usage: { input_tokens: inputTokens, output_tokens: 0 }
|
|
118918
|
+
},
|
|
118919
|
+
type: "message_start"
|
|
118920
|
+
})}`);
|
|
118921
|
+
events.push(`event: content_block_start\ndata: ${JSON.stringify({
|
|
118922
|
+
content_block: { text: "", type: "text" },
|
|
118923
|
+
index: 0,
|
|
118924
|
+
type: "content_block_start"
|
|
118925
|
+
})}`);
|
|
118926
|
+
const chunkSize = 20;
|
|
118927
|
+
for (let i = 0; i < content.length; i += chunkSize) {
|
|
118928
|
+
const textChunk = content.slice(i, i + chunkSize);
|
|
118929
|
+
events.push(`event: content_block_delta\ndata: ${JSON.stringify({
|
|
118930
|
+
delta: { text: textChunk, type: "text_delta" },
|
|
118931
|
+
index: 0,
|
|
118932
|
+
type: "content_block_delta"
|
|
118933
|
+
})}`);
|
|
118934
|
+
}
|
|
118935
|
+
events.push(`event: content_block_stop\ndata: ${JSON.stringify({
|
|
118936
|
+
index: 0,
|
|
118937
|
+
type: "content_block_stop"
|
|
118938
|
+
})}`);
|
|
118939
|
+
events.push(`event: message_delta\ndata: ${JSON.stringify({
|
|
118940
|
+
delta: { stop_reason: "end_turn", stop_sequence: null },
|
|
118941
|
+
type: "message_delta",
|
|
118942
|
+
usage: { output_tokens: outputTokens }
|
|
118943
|
+
})}`);
|
|
118944
|
+
events.push('event: message_stop\ndata: {"type":"message_stop"}');
|
|
118945
|
+
return events.join("\n\n") + "\n\n";
|
|
118946
|
+
}
|
|
118300
118947
|
async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }) {
|
|
118301
118948
|
function reportMetricsSafe(payload) {
|
|
118302
118949
|
reportMetrics(payload).catch(error => {
|
|
@@ -118307,10 +118954,17 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118307
118954
|
});
|
|
118308
118955
|
}
|
|
118309
118956
|
const engineType = conduitConfiguration.engineConfig?.type ?? null;
|
|
118957
|
+
const needsTranslation = !engineSupportsNativeAnthropic(engineType);
|
|
118310
118958
|
const { bytes: requestBodyBytes, payload: serializedBody } = serializeRequestBody(body);
|
|
118311
118959
|
const requestStartedAt = Date.now();
|
|
118312
118960
|
const requestBody = JSON.parse(serializedBody);
|
|
118313
118961
|
const streamRequested = requestBody.stream === true;
|
|
118962
|
+
const targetPath = needsTranslation
|
|
118963
|
+
? translateAnthropicRequestToOpenAI(serializedBody).path
|
|
118964
|
+
: "/v1/messages";
|
|
118965
|
+
const targetBody = needsTranslation
|
|
118966
|
+
? translateAnthropicRequestToOpenAI(serializedBody).body
|
|
118967
|
+
: serializedBody;
|
|
118314
118968
|
const onMonitoringComplete = ({ durationMs, error, responseBytes, usage }) => {
|
|
118315
118969
|
const promptTokens = normalizeTokenCount(usage?.inputTokens);
|
|
118316
118970
|
const completionTokens = normalizeTokenCount(usage?.outputTokens);
|
|
@@ -118339,8 +118993,8 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118339
118993
|
});
|
|
118340
118994
|
};
|
|
118341
118995
|
const response = await modelManager
|
|
118342
|
-
.fetchOpenAI(
|
|
118343
|
-
body:
|
|
118996
|
+
.fetchOpenAI(targetPath, {
|
|
118997
|
+
body: targetBody,
|
|
118344
118998
|
headers: {
|
|
118345
118999
|
"Content-Type": "application/json"
|
|
118346
119000
|
},
|
|
@@ -118447,73 +119101,157 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118447
119101
|
}
|
|
118448
119102
|
const rawBody = Readable.fromWeb(response.body);
|
|
118449
119103
|
if (streamRequested) {
|
|
118450
|
-
|
|
118451
|
-
|
|
118452
|
-
|
|
118453
|
-
|
|
118454
|
-
|
|
118455
|
-
|
|
118456
|
-
|
|
118457
|
-
|
|
118458
|
-
const
|
|
118459
|
-
if (
|
|
118460
|
-
|
|
119104
|
+
if (needsTranslation) {
|
|
119105
|
+
const chunks = [];
|
|
119106
|
+
rawBody.on("data", (chunk) => {
|
|
119107
|
+
const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
|
|
119108
|
+
responseBytes += chunkBuffer.length;
|
|
119109
|
+
chunks.push(chunkBuffer);
|
|
119110
|
+
});
|
|
119111
|
+
rawBody.once("end", () => {
|
|
119112
|
+
const fullBody = Buffer.concat(chunks).toString("utf8");
|
|
119113
|
+
if (!response.ok) {
|
|
119114
|
+
passThrough.write(fullBody);
|
|
119115
|
+
responseBytes = Buffer.byteLength(fullBody, "utf8");
|
|
119116
|
+
logEngineMetrics({
|
|
119117
|
+
agentEngineType: engineType ?? "unknown",
|
|
119118
|
+
level: "error",
|
|
119119
|
+
logger,
|
|
119120
|
+
requestBodyBytes,
|
|
119121
|
+
requestPath: "/v1/messages",
|
|
119122
|
+
responseBytes,
|
|
119123
|
+
usage: null
|
|
119124
|
+
});
|
|
119125
|
+
finalize(upstreamError);
|
|
119126
|
+
passThrough.end();
|
|
119127
|
+
return;
|
|
118461
119128
|
}
|
|
118462
|
-
|
|
118463
|
-
|
|
119129
|
+
const modelId = requestBody.model ?? "unknown";
|
|
119130
|
+
const assembled = assembleOpenAISSE(fullBody);
|
|
119131
|
+
if (assembled) {
|
|
119132
|
+
usage.inputTokens = assembled.promptTokens;
|
|
119133
|
+
usage.outputTokens = assembled.completionTokens;
|
|
118464
119134
|
}
|
|
118465
|
-
|
|
118466
|
-
|
|
118467
|
-
|
|
118468
|
-
|
|
118469
|
-
|
|
118470
|
-
|
|
118471
|
-
|
|
118472
|
-
|
|
118473
|
-
|
|
118474
|
-
|
|
118475
|
-
|
|
118476
|
-
|
|
118477
|
-
|
|
118478
|
-
|
|
119135
|
+
try {
|
|
119136
|
+
const anthropicSSE = translateOpenAIStreamToAnthropicSSE(fullBody, modelId);
|
|
119137
|
+
const output = Buffer.from(anthropicSSE, "utf8");
|
|
119138
|
+
responseBytes = output.length;
|
|
119139
|
+
passThrough.write(output);
|
|
119140
|
+
}
|
|
119141
|
+
catch (translateError) {
|
|
119142
|
+
const normalizedTranslateError = asError(translateError);
|
|
119143
|
+
logEngineMetrics({
|
|
119144
|
+
agentEngineType: engineType ?? "unknown",
|
|
119145
|
+
error: normalizedTranslateError,
|
|
119146
|
+
level: "error",
|
|
119147
|
+
logger,
|
|
119148
|
+
requestBodyBytes,
|
|
119149
|
+
requestPath: "/v1/messages",
|
|
119150
|
+
responseBytes,
|
|
119151
|
+
usage: null
|
|
119152
|
+
});
|
|
119153
|
+
finalize(normalizedTranslateError);
|
|
119154
|
+
passThrough.destroy(normalizedTranslateError);
|
|
119155
|
+
return;
|
|
119156
|
+
}
|
|
119157
|
+
logEngineMetrics({
|
|
119158
|
+
agentEngineType: engineType ?? "unknown",
|
|
119159
|
+
level: upstreamError ? "error" : "info",
|
|
119160
|
+
logger,
|
|
119161
|
+
requestBodyBytes,
|
|
119162
|
+
requestPath: "/v1/messages",
|
|
119163
|
+
responseBytes,
|
|
119164
|
+
usage: null
|
|
119165
|
+
});
|
|
119166
|
+
finalize(upstreamError);
|
|
119167
|
+
passThrough.end();
|
|
118479
119168
|
});
|
|
118480
|
-
|
|
118481
|
-
|
|
118482
|
-
|
|
118483
|
-
|
|
118484
|
-
logEngineMetrics({
|
|
118485
|
-
agentEngineType: engineType ?? "unknown",
|
|
118486
|
-
level: upstreamError ? "error" : "info",
|
|
118487
|
-
logger,
|
|
118488
|
-
requestBodyBytes,
|
|
118489
|
-
requestPath: "/v1/messages",
|
|
118490
|
-
responseBytes,
|
|
118491
|
-
usage: null
|
|
119169
|
+
rawBody.once("error", err => {
|
|
119170
|
+
const normalizedError = asError(err);
|
|
119171
|
+
finalize(normalizedError);
|
|
119172
|
+
passThrough.destroy(normalizedError);
|
|
118492
119173
|
});
|
|
118493
|
-
|
|
118494
|
-
|
|
118495
|
-
|
|
118496
|
-
|
|
118497
|
-
|
|
119174
|
+
rawBody.once("close", () => {
|
|
119175
|
+
if (completed) {
|
|
119176
|
+
if (!passThrough.writableEnded)
|
|
119177
|
+
passThrough.end();
|
|
119178
|
+
return;
|
|
119179
|
+
}
|
|
119180
|
+
const closeError = new Error("Engine response stream closed before completion");
|
|
119181
|
+
finalize(closeError);
|
|
118498
119182
|
if (!passThrough.writableEnded)
|
|
118499
119183
|
passThrough.end();
|
|
118500
|
-
return;
|
|
118501
|
-
}
|
|
118502
|
-
const closeError = new Error("Engine response stream closed before completion");
|
|
118503
|
-
logEngineMetrics({
|
|
118504
|
-
agentEngineType: engineType ?? "unknown",
|
|
118505
|
-
error: closeError,
|
|
118506
|
-
level: "error",
|
|
118507
|
-
logger,
|
|
118508
|
-
requestBodyBytes,
|
|
118509
|
-
requestPath: "/v1/messages",
|
|
118510
|
-
responseBytes,
|
|
118511
|
-
usage: null
|
|
118512
119184
|
});
|
|
118513
|
-
|
|
118514
|
-
|
|
119185
|
+
}
|
|
119186
|
+
else {
|
|
119187
|
+
let buffer = "";
|
|
119188
|
+
rawBody.on("data", (chunk) => {
|
|
119189
|
+
const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
|
|
119190
|
+
responseBytes += chunkBuffer.length;
|
|
119191
|
+
buffer += chunkBuffer.toString("utf8");
|
|
119192
|
+
const lines = buffer.split("\n");
|
|
119193
|
+
buffer = lines.pop() ?? "";
|
|
119194
|
+
for (const line of lines) {
|
|
119195
|
+
const extracted = extractAnthropicStreamUsage(line.trim());
|
|
119196
|
+
if (extracted?.inputTokens !== undefined && extracted.inputTokens !== null) {
|
|
119197
|
+
usage.inputTokens = extracted.inputTokens;
|
|
119198
|
+
}
|
|
119199
|
+
if (extracted?.outputTokens !== undefined && extracted.outputTokens !== null) {
|
|
119200
|
+
usage.outputTokens = extracted.outputTokens;
|
|
119201
|
+
}
|
|
119202
|
+
}
|
|
119203
|
+
passThrough.write(chunkBuffer);
|
|
119204
|
+
});
|
|
119205
|
+
rawBody.once("error", err => {
|
|
119206
|
+
const normalizedError = asError(err);
|
|
119207
|
+
logEngineMetrics({
|
|
119208
|
+
agentEngineType: engineType ?? "unknown",
|
|
119209
|
+
error: normalizedError,
|
|
119210
|
+
level: "error",
|
|
119211
|
+
logger,
|
|
119212
|
+
requestBodyBytes,
|
|
119213
|
+
requestPath: "/v1/messages",
|
|
119214
|
+
responseBytes,
|
|
119215
|
+
usage: null
|
|
119216
|
+
});
|
|
119217
|
+
finalize(normalizedError);
|
|
119218
|
+
passThrough.destroy(normalizedError);
|
|
119219
|
+
});
|
|
119220
|
+
rawBody.once("end", () => {
|
|
119221
|
+
logEngineMetrics({
|
|
119222
|
+
agentEngineType: engineType ?? "unknown",
|
|
119223
|
+
level: upstreamError ? "error" : "info",
|
|
119224
|
+
logger,
|
|
119225
|
+
requestBodyBytes,
|
|
119226
|
+
requestPath: "/v1/messages",
|
|
119227
|
+
responseBytes,
|
|
119228
|
+
usage: null
|
|
119229
|
+
});
|
|
119230
|
+
finalize(upstreamError);
|
|
118515
119231
|
passThrough.end();
|
|
118516
|
-
|
|
119232
|
+
});
|
|
119233
|
+
rawBody.once("close", () => {
|
|
119234
|
+
if (completed) {
|
|
119235
|
+
if (!passThrough.writableEnded)
|
|
119236
|
+
passThrough.end();
|
|
119237
|
+
return;
|
|
119238
|
+
}
|
|
119239
|
+
const closeError = new Error("Engine response stream closed before completion");
|
|
119240
|
+
logEngineMetrics({
|
|
119241
|
+
agentEngineType: engineType ?? "unknown",
|
|
119242
|
+
error: closeError,
|
|
119243
|
+
level: "error",
|
|
119244
|
+
logger,
|
|
119245
|
+
requestBodyBytes,
|
|
119246
|
+
requestPath: "/v1/messages",
|
|
119247
|
+
responseBytes,
|
|
119248
|
+
usage: null
|
|
119249
|
+
});
|
|
119250
|
+
finalize(closeError);
|
|
119251
|
+
if (!passThrough.writableEnded)
|
|
119252
|
+
passThrough.end();
|
|
119253
|
+
});
|
|
119254
|
+
}
|
|
118517
119255
|
}
|
|
118518
119256
|
else {
|
|
118519
119257
|
const chunks = [];
|
|
@@ -118521,7 +119259,6 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118521
119259
|
const chunkBuffer = Buffer.isBuffer(chunk) ? chunk : Buffer.from(chunk);
|
|
118522
119260
|
responseBytes += chunkBuffer.length;
|
|
118523
119261
|
chunks.push(chunkBuffer);
|
|
118524
|
-
passThrough.write(chunkBuffer);
|
|
118525
119262
|
});
|
|
118526
119263
|
rawBody.once("error", err => {
|
|
118527
119264
|
const normalizedError = asError(err);
|
|
@@ -118540,11 +119277,48 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118540
119277
|
});
|
|
118541
119278
|
rawBody.once("end", () => {
|
|
118542
119279
|
const fullBody = Buffer.concat(chunks).toString("utf8");
|
|
118543
|
-
|
|
118544
|
-
if (
|
|
118545
|
-
|
|
118546
|
-
|
|
119280
|
+
let outputBody = fullBody;
|
|
119281
|
+
if (needsTranslation && response.ok) {
|
|
119282
|
+
const modelId = requestBody.model ?? "unknown";
|
|
119283
|
+
try {
|
|
119284
|
+
outputBody = translateOpenAIResponseToAnthropic({
|
|
119285
|
+
body: fullBody,
|
|
119286
|
+
model: modelId
|
|
119287
|
+
});
|
|
119288
|
+
}
|
|
119289
|
+
catch (translateError) {
|
|
119290
|
+
const normalizedTranslateError = asError(translateError);
|
|
119291
|
+
logEngineMetrics({
|
|
119292
|
+
agentEngineType: engineType ?? "unknown",
|
|
119293
|
+
error: normalizedTranslateError,
|
|
119294
|
+
level: "error",
|
|
119295
|
+
logger,
|
|
119296
|
+
requestBodyBytes,
|
|
119297
|
+
requestPath: "/v1/messages",
|
|
119298
|
+
responseBytes,
|
|
119299
|
+
usage: null
|
|
119300
|
+
});
|
|
119301
|
+
finalize(normalizedTranslateError);
|
|
119302
|
+
passThrough.destroy(normalizedTranslateError);
|
|
119303
|
+
return;
|
|
119304
|
+
}
|
|
119305
|
+
}
|
|
119306
|
+
if (needsTranslation) {
|
|
119307
|
+
const extractedUsage = extractOpenAIUsageFromJSON(fullBody);
|
|
119308
|
+
if (extractedUsage) {
|
|
119309
|
+
usage.inputTokens = extractedUsage.inputTokens;
|
|
119310
|
+
usage.outputTokens = extractedUsage.outputTokens;
|
|
119311
|
+
}
|
|
119312
|
+
}
|
|
119313
|
+
else {
|
|
119314
|
+
const extractedUsage = extractAnthropicNonStreamUsage(fullBody);
|
|
119315
|
+
if (extractedUsage) {
|
|
119316
|
+
usage.inputTokens = extractedUsage.inputTokens;
|
|
119317
|
+
usage.outputTokens = extractedUsage.outputTokens;
|
|
119318
|
+
}
|
|
118547
119319
|
}
|
|
119320
|
+
passThrough.write(outputBody);
|
|
119321
|
+
responseBytes = Buffer.byteLength(outputBody, "utf8");
|
|
118548
119322
|
logEngineMetrics({
|
|
118549
119323
|
agentEngineType: engineType ?? "unknown",
|
|
118550
119324
|
level: upstreamError ? "error" : "info",
|
|
@@ -118579,9 +119353,17 @@ async function proxyAnthropicStreamingRoute({ body, conduitConfiguration, endpoi
|
|
|
118579
119353
|
passThrough.end();
|
|
118580
119354
|
});
|
|
118581
119355
|
}
|
|
119356
|
+
const responseHeaders = Object.fromEntries(response.headers.entries());
|
|
119357
|
+
if (needsTranslation) {
|
|
119358
|
+
delete responseHeaders["content-length"];
|
|
119359
|
+
delete responseHeaders["content-encoding"];
|
|
119360
|
+
responseHeaders["content-type"] = streamRequested
|
|
119361
|
+
? "text/event-stream"
|
|
119362
|
+
: "application/json";
|
|
119363
|
+
}
|
|
118582
119364
|
return {
|
|
118583
119365
|
body: passThrough,
|
|
118584
|
-
headers:
|
|
119366
|
+
headers: responseHeaders,
|
|
118585
119367
|
status: response.status
|
|
118586
119368
|
};
|
|
118587
119369
|
}
|
|
@@ -128382,6 +129164,58 @@ async function detectVLLMVersion() {
|
|
|
128382
129164
|
return null;
|
|
128383
129165
|
}
|
|
128384
129166
|
}
|
|
129167
|
+
async function detectSGLangVersion() {
|
|
129168
|
+
try {
|
|
129169
|
+
const { stdout } = await execa(process.env.SGLANG_EXECUTABLE ?? "python3", [
|
|
129170
|
+
"-c",
|
|
129171
|
+
"import importlib.metadata as md; print(md.version('sglang'))"
|
|
129172
|
+
]);
|
|
129173
|
+
const version = stdout.trim();
|
|
129174
|
+
return version.length > 0 ? version : null;
|
|
129175
|
+
}
|
|
129176
|
+
catch {
|
|
129177
|
+
return null;
|
|
129178
|
+
}
|
|
129179
|
+
}
|
|
129180
|
+
async function detectTensorRTLLMVersion() {
|
|
129181
|
+
try {
|
|
129182
|
+
const { stdout } = await execa(process.env.TRTLLM_EXECUTABLE ?? "python3", [
|
|
129183
|
+
"-c",
|
|
129184
|
+
"import importlib.metadata as md; print(md.version('tensorrt_llm'))"
|
|
129185
|
+
]);
|
|
129186
|
+
const version = stdout.trim();
|
|
129187
|
+
return version.length > 0 ? version : null;
|
|
129188
|
+
}
|
|
129189
|
+
catch {
|
|
129190
|
+
return null;
|
|
129191
|
+
}
|
|
129192
|
+
}
|
|
129193
|
+
async function detectExllamav3Version() {
|
|
129194
|
+
try {
|
|
129195
|
+
const { stdout } = await execa(process.env.EXLLAMAV3_EXECUTABLE ?? "python3", [
|
|
129196
|
+
"-c",
|
|
129197
|
+
"import importlib.metadata as md; print(md.version('exllamav3'))"
|
|
129198
|
+
]);
|
|
129199
|
+
const version = stdout.trim();
|
|
129200
|
+
return version.length > 0 ? version : null;
|
|
129201
|
+
}
|
|
129202
|
+
catch {
|
|
129203
|
+
return null;
|
|
129204
|
+
}
|
|
129205
|
+
}
|
|
129206
|
+
async function detectMLXLMVersion() {
|
|
129207
|
+
try {
|
|
129208
|
+
const { stdout } = await execa(process.env.MLXLM_EXECUTABLE ?? "python3", [
|
|
129209
|
+
"-c",
|
|
129210
|
+
"import importlib.metadata as md; print(md.version('mlx-lm'))"
|
|
129211
|
+
]);
|
|
129212
|
+
const version = stdout.trim();
|
|
129213
|
+
return version.length > 0 ? version : null;
|
|
129214
|
+
}
|
|
129215
|
+
catch {
|
|
129216
|
+
return null;
|
|
129217
|
+
}
|
|
129218
|
+
}
|
|
128385
129219
|
function normalizeMegabytes(value) {
|
|
128386
129220
|
if (typeof value !== "number" || Number.isNaN(value)) {
|
|
128387
129221
|
return null;
|
|
@@ -128668,6 +129502,7 @@ async function collectMachineMetadata() {
|
|
|
128668
129502
|
model: cpuInfo?.brand ?? null,
|
|
128669
129503
|
physicalCores: cpuInfo?.physicalCores ?? null
|
|
128670
129504
|
},
|
|
129505
|
+
exllamav3Version: await detectExllamav3Version(),
|
|
128671
129506
|
gpus,
|
|
128672
129507
|
hostname: os.hostname(),
|
|
128673
129508
|
llamaCppVersion: await detectLlamaCppVersion(),
|
|
@@ -128676,6 +129511,7 @@ async function collectMachineMetadata() {
|
|
|
128676
129511
|
availableBytes: memInfo?.available ?? null,
|
|
128677
129512
|
totalBytes: memInfo?.total ?? null
|
|
128678
129513
|
},
|
|
129514
|
+
mlxlmVersion: await detectMLXLMVersion(),
|
|
128679
129515
|
os: {
|
|
128680
129516
|
arch: osInfo?.arch ?? os.arch(),
|
|
128681
129517
|
platform: osInfo?.platform ?? os.platform(),
|
|
@@ -128683,6 +129519,8 @@ async function collectMachineMetadata() {
|
|
|
128683
129519
|
type: osInfo?.kernel ?? null,
|
|
128684
129520
|
version: osInfo?.build ?? null
|
|
128685
129521
|
},
|
|
129522
|
+
sglangVersion: await detectSGLangVersion(),
|
|
129523
|
+
tensorrtLlmVersion: await detectTensorRTLLMVersion(),
|
|
128686
129524
|
vllmVersion: await detectVLLMVersion()
|
|
128687
129525
|
};
|
|
128688
129526
|
return machineMetadata;
|
|
@@ -42,6 +42,7 @@ export declare class ModelManager extends EventEmitter<ModelManagerEvents> {
|
|
|
42
42
|
get canStop(): boolean;
|
|
43
43
|
get state(): EngineLifecycleState;
|
|
44
44
|
private checkEngineReadiness;
|
|
45
|
+
private checkGenericHealthReadiness;
|
|
45
46
|
private checkLlamacppReadiness;
|
|
46
47
|
private checkVLLMReadiness;
|
|
47
48
|
private waitForEngineReady;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { ProcessManager } from "@infersec/utils";
|
|
2
|
+
import type { ModelManager } from "../ModelManager.js";
|
|
3
|
+
export declare function startExllamav3(this: ModelManager, { enginePort, targetDirectory }: {
|
|
4
|
+
enginePort: number;
|
|
5
|
+
targetDirectory: string;
|
|
6
|
+
}): Promise<ProcessManager>;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { ProcessManager } from "@infersec/utils";
|
|
2
|
+
import type { ModelManager } from "./ModelManager.js";
|
|
3
|
+
export declare function startMLXLM(this: ModelManager, { enginePort, targetDirectory }: {
|
|
4
|
+
enginePort: number;
|
|
5
|
+
targetDirectory: string;
|
|
6
|
+
}): Promise<ProcessManager>;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { ProcessManager } from "@infersec/utils";
|
|
2
|
+
import type { ModelManager } from "./ModelManager.js";
|
|
3
|
+
export declare function startSGLang(this: ModelManager, { enginePort, targetDirectory }: {
|
|
4
|
+
enginePort: number;
|
|
5
|
+
targetDirectory: string;
|
|
6
|
+
}): Promise<ProcessManager>;
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
import { ProcessManager } from "@infersec/utils";
|
|
2
|
+
import type { ModelManager } from "./ModelManager.js";
|
|
3
|
+
export declare function startTensorRTLLM(this: ModelManager, { enginePort, targetDirectory }: {
|
|
4
|
+
enginePort: number;
|
|
5
|
+
targetDirectory: string;
|
|
6
|
+
}): Promise<ProcessManager>;
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
import type { EngineConfig, LLMEngine } from "@infersec/definitions";
|
|
2
|
+
import { ENGINE_API_COMPATIBILITY } from "@infersec/definitions";
|
|
3
|
+
export declare function engineSupportsNativeAnthropic(engineType: LLMEngine | null): boolean;
|
|
4
|
+
export declare function translateAnthropicRequestToOpenAI(body: string): {
|
|
5
|
+
body: string;
|
|
6
|
+
path: string;
|
|
7
|
+
};
|
|
8
|
+
export declare function translateOpenAIResponseToAnthropic({ body, model }: {
|
|
9
|
+
body: string;
|
|
10
|
+
model: string;
|
|
11
|
+
}): string;
|
|
12
|
+
export { ENGINE_API_COMPATIBILITY };
|
|
13
|
+
export type { EngineConfig };
|