@infersec/conduit 1.73.0 → 1.74.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js
CHANGED
|
@@ -19921,6 +19921,7 @@ const LLMModelFormatSchema = _enum([
|
|
|
19921
19921
|
// Llama.cpp
|
|
19922
19922
|
"gguf"
|
|
19923
19923
|
]);
|
|
19924
|
+
const LLMModelTaskTypeSchema = _enum(["text-generation", "embeddings"]);
|
|
19924
19925
|
const LLMModelSchema = object({
|
|
19925
19926
|
format: LLMModelFormatSchema,
|
|
19926
19927
|
id: string$1().min(1),
|
|
@@ -19935,7 +19936,8 @@ const LLMModelSchema = object({
|
|
|
19935
19936
|
slug: string$1().min(1),
|
|
19936
19937
|
type: literal("huggingface")
|
|
19937
19938
|
})
|
|
19938
|
-
])
|
|
19939
|
+
]),
|
|
19940
|
+
taskType: LLMModelTaskTypeSchema
|
|
19939
19941
|
});
|
|
19940
19942
|
object({
|
|
19941
19943
|
filePath: string$1().min(1),
|
|
@@ -20643,6 +20645,34 @@ const CompletionCreateParamsSchema = object({
|
|
|
20643
20645
|
top_p: number$1().min(0).max(1).nullable().optional(),
|
|
20644
20646
|
user: string$1().optional()
|
|
20645
20647
|
});
|
|
20648
|
+
// ==================== EMBEDDINGS ====================
|
|
20649
|
+
const EmbeddingCreateParamsSchema = object({
|
|
20650
|
+
dimensions: number$1().int().positive().nullable().optional(),
|
|
20651
|
+
encoding_format: _enum(["float", "base64"]).nullable().optional(),
|
|
20652
|
+
input: union([
|
|
20653
|
+
string$1(),
|
|
20654
|
+
array(string$1()),
|
|
20655
|
+
array(number$1()),
|
|
20656
|
+
array(array(number$1()))
|
|
20657
|
+
]),
|
|
20658
|
+
model: string$1(),
|
|
20659
|
+
user: string$1().optional()
|
|
20660
|
+
});
|
|
20661
|
+
const EmbeddingUsageSchema = object({
|
|
20662
|
+
prompt_tokens: number$1(),
|
|
20663
|
+
total_tokens: number$1()
|
|
20664
|
+
});
|
|
20665
|
+
const EmbeddingDataSchema = object({
|
|
20666
|
+
embedding: array(number$1()),
|
|
20667
|
+
index: number$1(),
|
|
20668
|
+
object: literal("embedding")
|
|
20669
|
+
});
|
|
20670
|
+
object({
|
|
20671
|
+
data: array(EmbeddingDataSchema),
|
|
20672
|
+
model: string$1(),
|
|
20673
|
+
object: literal("list"),
|
|
20674
|
+
usage: EmbeddingUsageSchema
|
|
20675
|
+
});
|
|
20646
20676
|
|
|
20647
20677
|
const API_CLIENT_CONDUIT_GENERAL_REFERENCE = {
|
|
20648
20678
|
"/conduit/engine/start": {
|
|
@@ -20708,6 +20738,17 @@ const API_CLIENT_CONDUIT_OPENAI_REFERENCE = {
|
|
|
20708
20738
|
}
|
|
20709
20739
|
}
|
|
20710
20740
|
},
|
|
20741
|
+
"/v1/embeddings": {
|
|
20742
|
+
POST: {
|
|
20743
|
+
auth: {
|
|
20744
|
+
type: "shared-secret"
|
|
20745
|
+
},
|
|
20746
|
+
body: EmbeddingCreateParamsSchema,
|
|
20747
|
+
response: {
|
|
20748
|
+
type: "text-stream"
|
|
20749
|
+
}
|
|
20750
|
+
}
|
|
20751
|
+
},
|
|
20711
20752
|
"/v1/models": {
|
|
20712
20753
|
GET: {
|
|
20713
20754
|
auth: {
|
|
@@ -20743,6 +20784,12 @@ const API_CLIENT_CONDUIT_OPENAI_REFERENCE = {
|
|
|
20743
20784
|
endpointID: ULIDSchema.describe("Endpoint identifier")
|
|
20744
20785
|
}}
|
|
20745
20786
|
},
|
|
20787
|
+
"/api/inferencing/:endpointID/oai/v1/embeddings": {
|
|
20788
|
+
POST: {
|
|
20789
|
+
parameters: {
|
|
20790
|
+
endpointID: ULIDSchema.describe("Endpoint identifier")
|
|
20791
|
+
}}
|
|
20792
|
+
},
|
|
20746
20793
|
"/api/inferencing/:endpointID/oai/v1/models": {
|
|
20747
20794
|
GET: {
|
|
20748
20795
|
parameters: {
|
|
@@ -20771,7 +20818,8 @@ object({
|
|
|
20771
20818
|
.min(3)
|
|
20772
20819
|
.refine(value => value.includes("/"), {
|
|
20773
20820
|
message: "Slug must be fully qualified (owner/repo)"
|
|
20774
|
-
})
|
|
20821
|
+
}),
|
|
20822
|
+
taskType: LLMModelTaskTypeSchema.optional()
|
|
20775
20823
|
});
|
|
20776
20824
|
object({
|
|
20777
20825
|
results: array(object({
|
|
@@ -20782,6 +20830,7 @@ object({
|
|
|
20782
20830
|
name: string$1(),
|
|
20783
20831
|
provider: _enum(["storage", "huggingface"]),
|
|
20784
20832
|
providerSlug: string$1(),
|
|
20833
|
+
taskType: LLMModelTaskTypeSchema,
|
|
20785
20834
|
updated: string$1()
|
|
20786
20835
|
}))
|
|
20787
20836
|
});
|
|
@@ -20802,11 +20851,13 @@ object({
|
|
|
20802
20851
|
name: string$1(),
|
|
20803
20852
|
updated: string$1()
|
|
20804
20853
|
})),
|
|
20854
|
+
taskType: LLMModelTaskTypeSchema,
|
|
20805
20855
|
updated: string$1()
|
|
20806
20856
|
});
|
|
20807
20857
|
object({
|
|
20858
|
+
multimodalEnabled: boolean$1().optional(),
|
|
20808
20859
|
name: ResourceNameSchema.optional(),
|
|
20809
|
-
|
|
20860
|
+
taskType: LLMModelTaskTypeSchema.optional()
|
|
20810
20861
|
});
|
|
20811
20862
|
object({
|
|
20812
20863
|
success: literal(true)
|
|
@@ -20851,7 +20902,8 @@ object({
|
|
|
20851
20902
|
modelFormat: LLMModelFormatSchema,
|
|
20852
20903
|
name: string$1(),
|
|
20853
20904
|
provider: _enum(["storage", "huggingface"]),
|
|
20854
|
-
providerSlug: string$1()
|
|
20905
|
+
providerSlug: string$1(),
|
|
20906
|
+
taskType: LLMModelTaskTypeSchema
|
|
20855
20907
|
})
|
|
20856
20908
|
.nullable(),
|
|
20857
20909
|
modelQuantizationLabel: string$1().nullable(),
|
|
@@ -21847,6 +21899,40 @@ function parseSSEEvent(rawEvent) {
|
|
|
21847
21899
|
};
|
|
21848
21900
|
}
|
|
21849
21901
|
|
|
21902
|
+
function buildConfigurationOverrides(options) {
|
|
21903
|
+
const configurationOverrides = {};
|
|
21904
|
+
if (options.apiUrl) {
|
|
21905
|
+
configurationOverrides.apiURL = options.apiUrl;
|
|
21906
|
+
}
|
|
21907
|
+
if (options.enginePort) {
|
|
21908
|
+
const enginePort = Number.parseInt(options.enginePort, 10);
|
|
21909
|
+
if (Number.isNaN(enginePort) || enginePort < 1 || enginePort > 65535) {
|
|
21910
|
+
throw new Error(`Invalid engine port: ${options.enginePort}`);
|
|
21911
|
+
}
|
|
21912
|
+
configurationOverrides.enginePort = enginePort;
|
|
21913
|
+
}
|
|
21914
|
+
if (options.key) {
|
|
21915
|
+
configurationOverrides.apiKey = options.key;
|
|
21916
|
+
}
|
|
21917
|
+
if (options.port) {
|
|
21918
|
+
const port = Number.parseInt(options.port, 10);
|
|
21919
|
+
if (Number.isNaN(port) || port < 1 || port > 65535) {
|
|
21920
|
+
throw new Error(`Invalid port: ${options.port}`);
|
|
21921
|
+
}
|
|
21922
|
+
configurationOverrides.port = port;
|
|
21923
|
+
}
|
|
21924
|
+
if (options.root) {
|
|
21925
|
+
configurationOverrides.rootDirectory = options.root;
|
|
21926
|
+
}
|
|
21927
|
+
if (options.startMode) {
|
|
21928
|
+
configurationOverrides.startMode = options.startMode;
|
|
21929
|
+
}
|
|
21930
|
+
if (options.source) {
|
|
21931
|
+
configurationOverrides.inferenceSourceID = options.source;
|
|
21932
|
+
}
|
|
21933
|
+
return configurationOverrides;
|
|
21934
|
+
}
|
|
21935
|
+
|
|
21850
21936
|
function commonjsRequire(path) {
|
|
21851
21937
|
throw new Error('Could not dynamically require "' + path + '". Please configure the dynamicRequireTargets or/and ignoreDynamicRequires option of @rollup/plugin-commonjs appropriately for this require call to work.');
|
|
21852
21938
|
}
|
|
@@ -114830,6 +114916,9 @@ async function startVLLM({ enginePort, targetDirectory }) {
|
|
|
114830
114916
|
"--tensor-parallel-size",
|
|
114831
114917
|
String(tensorParallelSize)
|
|
114832
114918
|
];
|
|
114919
|
+
if (this.model.taskType === "embeddings") {
|
|
114920
|
+
args.push("--task", "embed");
|
|
114921
|
+
}
|
|
114833
114922
|
if (device) {
|
|
114834
114923
|
args.push("--device", device);
|
|
114835
114924
|
}
|
|
@@ -116583,6 +116672,9 @@ async function startLlamacpp({ enginePort, targetDirectory }) {
|
|
|
116583
116672
|
"--ctx-size",
|
|
116584
116673
|
String(contextLength)
|
|
116585
116674
|
];
|
|
116675
|
+
if (this.model.taskType === "embeddings") {
|
|
116676
|
+
args.push("--embedding");
|
|
116677
|
+
}
|
|
116586
116678
|
const gpuLayers = typeof engineConfig?.gpuLayers === "number"
|
|
116587
116679
|
? engineConfig.gpuLayers
|
|
116588
116680
|
: Number.parseInt(process.env.LLAMACPP_GPU_LAYERS ?? String(DEFAULT_LLAMACPP_GPU_LAYERS), 10);
|
|
@@ -117688,6 +117780,153 @@ function calculateTokensPerSecond$2({ durationMs, totalTokens }) {
|
|
|
117688
117780
|
}
|
|
117689
117781
|
return Math.round(tokensPerSecond);
|
|
117690
117782
|
}
|
|
117783
|
+
async function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }) {
|
|
117784
|
+
function normalizeTokenCount(value) {
|
|
117785
|
+
if (typeof value === "number" && Number.isFinite(value) && value >= 0) {
|
|
117786
|
+
return value;
|
|
117787
|
+
}
|
|
117788
|
+
return 0;
|
|
117789
|
+
}
|
|
117790
|
+
function reportMetricsSafe(payload) {
|
|
117791
|
+
reportMetrics(payload).catch(error => {
|
|
117792
|
+
logger.warn("Failed to upload LLM prompt metrics", {
|
|
117793
|
+
error: asError(error),
|
|
117794
|
+
requestUrl: "/v1/embeddings"
|
|
117795
|
+
});
|
|
117796
|
+
});
|
|
117797
|
+
}
|
|
117798
|
+
const engineType = conduitConfiguration.engineConfig?.type ?? null;
|
|
117799
|
+
const engineConfig = conduitConfiguration.engineConfig?.config ?? null;
|
|
117800
|
+
const serializedBody = isPlainObject$2(body)
|
|
117801
|
+
? JSON.stringify(body)
|
|
117802
|
+
: typeof body === "string"
|
|
117803
|
+
? body
|
|
117804
|
+
: JSON.stringify(body);
|
|
117805
|
+
const requestBodyBytes = Buffer.byteLength(serializedBody, "utf8");
|
|
117806
|
+
const requestStartedAt = Date.now();
|
|
117807
|
+
let upstreamResponseOk = true;
|
|
117808
|
+
const onMonitoringComplete = ({ durationMs, error, responseBytes, usage }) => {
|
|
117809
|
+
const promptTokens = normalizeTokenCount(usage?.promptTokens);
|
|
117810
|
+
const totalTokens = normalizeTokenCount(usage?.totalTokens ?? promptTokens);
|
|
117811
|
+
const latencyMs = Math.max(0, durationMs);
|
|
117812
|
+
reportMetricsSafe({
|
|
117813
|
+
bytes: requestBodyBytes + responseBytes,
|
|
117814
|
+
completionTokens: 0,
|
|
117815
|
+
engine: engineType,
|
|
117816
|
+
endpointId: endpointId ?? null,
|
|
117817
|
+
latencyMs,
|
|
117818
|
+
modelId: modelID,
|
|
117819
|
+
promptTokens,
|
|
117820
|
+
requestBytes: requestBodyBytes,
|
|
117821
|
+
requestId: null,
|
|
117822
|
+
requestMethod: "POST",
|
|
117823
|
+
requestPath: "/v1/embeddings",
|
|
117824
|
+
responseBytes,
|
|
117825
|
+
successful: upstreamResponseOk && !error,
|
|
117826
|
+
timeToFirstTokenMs: null,
|
|
117827
|
+
tokensPerSecond: calculateTokensPerSecond$2({
|
|
117828
|
+
durationMs: latencyMs,
|
|
117829
|
+
totalTokens
|
|
117830
|
+
}),
|
|
117831
|
+
totalTokens
|
|
117832
|
+
});
|
|
117833
|
+
};
|
|
117834
|
+
const response = await modelManager
|
|
117835
|
+
.fetchOpenAI("/v1/embeddings", {
|
|
117836
|
+
body: serializedBody,
|
|
117837
|
+
headers: {
|
|
117838
|
+
"Content-Type": "application/json"
|
|
117839
|
+
},
|
|
117840
|
+
method: "POST",
|
|
117841
|
+
signal
|
|
117842
|
+
})
|
|
117843
|
+
.catch(error => {
|
|
117844
|
+
const err = asError(error);
|
|
117845
|
+
logEngineMetrics({
|
|
117846
|
+
agentEngineType: engineType ?? "unknown",
|
|
117847
|
+
error: err,
|
|
117848
|
+
level: "error",
|
|
117849
|
+
logger,
|
|
117850
|
+
requestBodyBytes,
|
|
117851
|
+
requestPath: "/v1/embeddings",
|
|
117852
|
+
responseBytes: 0,
|
|
117853
|
+
usage: null
|
|
117854
|
+
});
|
|
117855
|
+
const latencyMs = Math.max(0, Date.now() - requestStartedAt);
|
|
117856
|
+
reportMetricsSafe({
|
|
117857
|
+
bytes: requestBodyBytes,
|
|
117858
|
+
completionTokens: 0,
|
|
117859
|
+
engine: engineType,
|
|
117860
|
+
endpointId: endpointId ?? null,
|
|
117861
|
+
latencyMs,
|
|
117862
|
+
modelId: modelID,
|
|
117863
|
+
promptTokens: 0,
|
|
117864
|
+
requestBytes: requestBodyBytes,
|
|
117865
|
+
requestId: null,
|
|
117866
|
+
requestMethod: "POST",
|
|
117867
|
+
requestPath: "/v1/embeddings",
|
|
117868
|
+
responseBytes: 0,
|
|
117869
|
+
successful: false,
|
|
117870
|
+
timeToFirstTokenMs: null,
|
|
117871
|
+
tokensPerSecond: 0,
|
|
117872
|
+
totalTokens: 0
|
|
117873
|
+
});
|
|
117874
|
+
throw err;
|
|
117875
|
+
});
|
|
117876
|
+
upstreamResponseOk = response.ok;
|
|
117877
|
+
const responseStatusText = response.statusText ?? "Upstream request failed";
|
|
117878
|
+
if (!response.body) {
|
|
117879
|
+
logEngineMetrics({
|
|
117880
|
+
agentEngineType: engineType ?? "unknown",
|
|
117881
|
+
level: response.ok ? "info" : "error",
|
|
117882
|
+
logger,
|
|
117883
|
+
requestBodyBytes,
|
|
117884
|
+
requestPath: "/v1/embeddings",
|
|
117885
|
+
responseBytes: 0,
|
|
117886
|
+
usage: null
|
|
117887
|
+
});
|
|
117888
|
+
const latencyMs = Math.max(0, Date.now() - requestStartedAt);
|
|
117889
|
+
reportMetricsSafe({
|
|
117890
|
+
bytes: requestBodyBytes,
|
|
117891
|
+
completionTokens: 0,
|
|
117892
|
+
engine: engineType,
|
|
117893
|
+
endpointId: endpointId ?? null,
|
|
117894
|
+
latencyMs,
|
|
117895
|
+
modelId: modelID,
|
|
117896
|
+
promptTokens: 0,
|
|
117897
|
+
requestBytes: requestBodyBytes,
|
|
117898
|
+
requestId: null,
|
|
117899
|
+
requestMethod: "POST",
|
|
117900
|
+
requestPath: "/v1/embeddings",
|
|
117901
|
+
responseBytes: 0,
|
|
117902
|
+
successful: false,
|
|
117903
|
+
timeToFirstTokenMs: null,
|
|
117904
|
+
tokensPerSecond: 0,
|
|
117905
|
+
totalTokens: 0
|
|
117906
|
+
});
|
|
117907
|
+
return {
|
|
117908
|
+
status: response.status,
|
|
117909
|
+
statusText: responseStatusText
|
|
117910
|
+
};
|
|
117911
|
+
}
|
|
117912
|
+
const monitoredResponse = monitorEngineResponseSingle({
|
|
117913
|
+
agentEngineType: engineType ?? "unknown",
|
|
117914
|
+
body: Readable.fromWeb(response.body),
|
|
117915
|
+
contextLength: modelManager.contextLength,
|
|
117916
|
+
engineConfig,
|
|
117917
|
+
engineType: engineType ?? "unknown",
|
|
117918
|
+
logger,
|
|
117919
|
+
onComplete: onMonitoringComplete,
|
|
117920
|
+
requestBodyBytes,
|
|
117921
|
+
requestPath: "/v1/embeddings",
|
|
117922
|
+
requestStartedAt
|
|
117923
|
+
});
|
|
117924
|
+
return {
|
|
117925
|
+
body: monitoredResponse.stream,
|
|
117926
|
+
headers: Object.fromEntries(response.headers.entries()),
|
|
117927
|
+
status: response.status
|
|
117928
|
+
};
|
|
117929
|
+
}
|
|
117691
117930
|
async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, path, reportMetrics, signal }) {
|
|
117692
117931
|
function normalizeTokenCount(value) {
|
|
117693
117932
|
if (typeof value === "number" && Number.isFinite(value) && value >= 0) {
|
|
@@ -117710,6 +117949,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
|
|
|
117710
117949
|
const requestStartedAt = Date.now();
|
|
117711
117950
|
const requestBody = JSON.parse(serializedBody);
|
|
117712
117951
|
const streamRequested = requestBody.stream === true;
|
|
117952
|
+
let upstreamResponseOk = true;
|
|
117713
117953
|
const onMonitoringComplete = ({ durationMs, error, responseBytes, timeToFirstTokenMs, usage }) => {
|
|
117714
117954
|
const completionTokens = normalizeTokenCount(usage?.completionTokens);
|
|
117715
117955
|
const promptTokens = normalizeTokenCount(usage?.promptTokens);
|
|
@@ -117728,7 +117968,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
|
|
|
117728
117968
|
requestMethod: "POST",
|
|
117729
117969
|
requestPath: path,
|
|
117730
117970
|
responseBytes,
|
|
117731
|
-
successful: !error,
|
|
117971
|
+
successful: upstreamResponseOk && !error,
|
|
117732
117972
|
timeToFirstTokenMs,
|
|
117733
117973
|
tokensPerSecond: calculateTokensPerSecond$2({
|
|
117734
117974
|
durationMs: latencyMs,
|
|
@@ -117779,6 +118019,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
|
|
|
117779
118019
|
});
|
|
117780
118020
|
throw err;
|
|
117781
118021
|
});
|
|
118022
|
+
upstreamResponseOk = response.ok;
|
|
117782
118023
|
const responseStatusText = response.statusText ?? "Upstream request failed";
|
|
117783
118024
|
if (!response.ok) {
|
|
117784
118025
|
if (!response.body) {
|
|
@@ -117923,6 +118164,26 @@ function createConduitOpenAIAPIReferenceHandlers({ apiClient, conduitConfigurati
|
|
|
117923
118164
|
});
|
|
117924
118165
|
}
|
|
117925
118166
|
},
|
|
118167
|
+
"/v1/embeddings": {
|
|
118168
|
+
POST: async ({ body, req, res }) => {
|
|
118169
|
+
const modelID = getModelID();
|
|
118170
|
+
const modelManager = getModelManager();
|
|
118171
|
+
const abortController = new AbortController();
|
|
118172
|
+
res.on("close", () => {
|
|
118173
|
+
abortController.abort();
|
|
118174
|
+
});
|
|
118175
|
+
return proxyEmbeddingsRoute({
|
|
118176
|
+
body,
|
|
118177
|
+
conduitConfiguration: conduitConfiguration(),
|
|
118178
|
+
endpointId: extractEndpointId$1(req),
|
|
118179
|
+
logger,
|
|
118180
|
+
modelID,
|
|
118181
|
+
modelManager,
|
|
118182
|
+
reportMetrics: apiClient.reportPromptMetrics,
|
|
118183
|
+
signal: abortController.signal
|
|
118184
|
+
});
|
|
118185
|
+
}
|
|
118186
|
+
},
|
|
117926
118187
|
"/v1/models": {
|
|
117927
118188
|
GET: async () => {
|
|
117928
118189
|
const modelManager = getModelManager();
|
|
@@ -117962,6 +118223,9 @@ function createPostChatCompletionsHandler(options) {
|
|
|
117962
118223
|
function createPostCompletionsHandler(options) {
|
|
117963
118224
|
return createConduitOpenAIAPIReferenceHandlers(options)["/v1/completions"].POST;
|
|
117964
118225
|
}
|
|
118226
|
+
function createPostEmbeddingsHandler(options) {
|
|
118227
|
+
return createConduitOpenAIAPIReferenceHandlers(options)["/v1/embeddings"].POST;
|
|
118228
|
+
}
|
|
117965
118229
|
|
|
117966
118230
|
function isPlainObject$1(value) {
|
|
117967
118231
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
@@ -128707,6 +128971,17 @@ async function createApplication({ abortController, apiClient, configuration, lo
|
|
|
128707
128971
|
startup
|
|
128708
128972
|
})
|
|
128709
128973
|
},
|
|
128974
|
+
"/v1/embeddings": {
|
|
128975
|
+
POST: createPostEmbeddingsHandler({
|
|
128976
|
+
apiClient,
|
|
128977
|
+
conduitConfiguration: () => conduitConfiguration,
|
|
128978
|
+
configuration,
|
|
128979
|
+
getModelID: () => conduitConfiguration.targetModel.id,
|
|
128980
|
+
getModelManager: () => modelManager,
|
|
128981
|
+
logger,
|
|
128982
|
+
startup
|
|
128983
|
+
})
|
|
128984
|
+
},
|
|
128710
128985
|
"/v1/models": {
|
|
128711
128986
|
GET: createGetModelsHandler({
|
|
128712
128987
|
apiClient,
|
|
@@ -129037,36 +129312,7 @@ function registerInferenceCommands({ program }) {
|
|
|
129037
129312
|
.option("--start-mode <mode>", "Startup mode: auto|idle (or START_MODE env)")
|
|
129038
129313
|
.option("--source <id>", "Inference source ID (or SOURCE env)")
|
|
129039
129314
|
.action(async (options) => {
|
|
129040
|
-
const configurationOverrides =
|
|
129041
|
-
if (options["api-url"]) {
|
|
129042
|
-
configurationOverrides.apiURL = options["api-url"];
|
|
129043
|
-
}
|
|
129044
|
-
if (options["engine-port"]) {
|
|
129045
|
-
const enginePort = Number.parseInt(options["engine-port"], 10);
|
|
129046
|
-
if (Number.isNaN(enginePort) || enginePort < 1 || enginePort > 65535) {
|
|
129047
|
-
throw new Error(`Invalid engine port: ${options["engine-port"]}`);
|
|
129048
|
-
}
|
|
129049
|
-
configurationOverrides.enginePort = enginePort;
|
|
129050
|
-
}
|
|
129051
|
-
if (options.key) {
|
|
129052
|
-
configurationOverrides.apiKey = options.key;
|
|
129053
|
-
}
|
|
129054
|
-
if (options.port) {
|
|
129055
|
-
const port = Number.parseInt(options.port, 10);
|
|
129056
|
-
if (Number.isNaN(port)) {
|
|
129057
|
-
throw new Error(`Invalid port: ${options.port}`);
|
|
129058
|
-
}
|
|
129059
|
-
configurationOverrides.port = port;
|
|
129060
|
-
}
|
|
129061
|
-
if (options.root) {
|
|
129062
|
-
configurationOverrides.rootDirectory = options.root;
|
|
129063
|
-
}
|
|
129064
|
-
if (options["start-mode"]) {
|
|
129065
|
-
configurationOverrides.startMode = options["start-mode"];
|
|
129066
|
-
}
|
|
129067
|
-
if (options.source) {
|
|
129068
|
-
configurationOverrides.inferenceSourceID = options.source;
|
|
129069
|
-
}
|
|
129315
|
+
const configurationOverrides = buildConfigurationOverrides(options);
|
|
129070
129316
|
await startInferenceAgent({ configurationOverrides });
|
|
129071
129317
|
});
|
|
129072
129318
|
}
|
|
@@ -129711,8 +129957,9 @@ class HuggingFaceClient {
|
|
|
129711
129957
|
}
|
|
129712
129958
|
}
|
|
129713
129959
|
}
|
|
129714
|
-
const
|
|
129715
|
-
|
|
129960
|
+
const taskPriority = new Map();
|
|
129961
|
+
pipelineTasks.forEach((task, index) => taskPriority.set(task, index));
|
|
129962
|
+
const modelsById = new Map();
|
|
129716
129963
|
await Promise.all(queries.map(async ({ task, tag }) => {
|
|
129717
129964
|
const searchParams = {
|
|
129718
129965
|
accessToken: this.apiKey ?? undefined,
|
|
@@ -129724,9 +129971,6 @@ class HuggingFaceClient {
|
|
|
129724
129971
|
}
|
|
129725
129972
|
};
|
|
129726
129973
|
for await (const entry of executeListWithRetry(searchParams)) {
|
|
129727
|
-
if (seenIds.has(entry.id)) {
|
|
129728
|
-
continue;
|
|
129729
|
-
}
|
|
129730
129974
|
const entryForUtils = {
|
|
129731
129975
|
config: entry.config,
|
|
129732
129976
|
gated: entry.gated,
|
|
@@ -129742,10 +129986,15 @@ class HuggingFaceClient {
|
|
|
129742
129986
|
if (targetFormats.length > 0 && !targetFormats.includes(format)) {
|
|
129743
129987
|
continue;
|
|
129744
129988
|
}
|
|
129745
|
-
|
|
129989
|
+
const existing = modelsById.get(entry.id);
|
|
129990
|
+
if (existing &&
|
|
129991
|
+
(taskPriority.get(task) ?? Number.MAX_SAFE_INTEGER) >=
|
|
129992
|
+
(taskPriority.get(existing.pipelineTask) ?? Number.MAX_SAFE_INTEGER)) {
|
|
129993
|
+
continue;
|
|
129994
|
+
}
|
|
129746
129995
|
const parameterCount = parseParameterCount(entry.id, entry.safetensors?.parameters);
|
|
129747
129996
|
const slug = entry.name?.trim() || entry.id;
|
|
129748
|
-
|
|
129997
|
+
modelsById.set(entry.id, {
|
|
129749
129998
|
downloads: entry.downloads,
|
|
129750
129999
|
format,
|
|
129751
130000
|
gated: entry.gated || false,
|
|
@@ -129753,13 +130002,14 @@ class HuggingFaceClient {
|
|
|
129753
130002
|
likes: entry.likes,
|
|
129754
130003
|
name: entry.name || entry.id,
|
|
129755
130004
|
parameterCount,
|
|
130005
|
+
pipelineTask: task,
|
|
129756
130006
|
quantization: extractQuantization(entryForUtils),
|
|
129757
130007
|
slug,
|
|
129758
130008
|
updatedAt: entry.updatedAt
|
|
129759
130009
|
});
|
|
129760
130010
|
}
|
|
129761
130011
|
}));
|
|
129762
|
-
return
|
|
130012
|
+
return Array.from(modelsById.values());
|
|
129763
130013
|
}
|
|
129764
130014
|
}
|
|
129765
130015
|
|
|
@@ -156128,13 +156378,13 @@ function registerBenchmarkCommands({ program }) {
|
|
|
156128
156378
|
.option("--account-id <id>", "Account ID (or ACCOUNT_ID env)")
|
|
156129
156379
|
.option("--output-dir <path>", "Override output directory from config")
|
|
156130
156380
|
.action(async (options) => {
|
|
156131
|
-
const apiUrl = options
|
|
156381
|
+
const apiUrl = options.apiUrl || process.env.API_URL;
|
|
156132
156382
|
if (!apiUrl)
|
|
156133
156383
|
throw new Error("API URL is required (--api-url or API_URL env)");
|
|
156134
|
-
const apiKey = options
|
|
156384
|
+
const apiKey = options.apiKey || process.env.API_KEY;
|
|
156135
156385
|
if (!apiKey)
|
|
156136
156386
|
throw new Error("API key is required (--api-key or API_KEY env)");
|
|
156137
|
-
const accountId = options
|
|
156387
|
+
const accountId = options.accountId || process.env.ACCOUNT_ID;
|
|
156138
156388
|
if (!accountId)
|
|
156139
156389
|
throw new Error("Account ID is required (--account-id or ACCOUNT_ID env)");
|
|
156140
156390
|
const configPath = options.config;
|
|
@@ -156145,7 +156395,7 @@ function registerBenchmarkCommands({ program }) {
|
|
|
156145
156395
|
apiKey,
|
|
156146
156396
|
apiUrl,
|
|
156147
156397
|
configPath,
|
|
156148
|
-
outputDir: options
|
|
156398
|
+
outputDir: options.outputDir
|
|
156149
156399
|
});
|
|
156150
156400
|
});
|
|
156151
156401
|
}
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { ConfigurationOverrides } from "../configuration.js";
|
|
2
|
+
export interface StartCommandOptions {
|
|
3
|
+
apiUrl?: string;
|
|
4
|
+
enginePort?: string;
|
|
5
|
+
key?: string;
|
|
6
|
+
port?: string;
|
|
7
|
+
root?: string;
|
|
8
|
+
source?: string;
|
|
9
|
+
startMode?: string;
|
|
10
|
+
}
|
|
11
|
+
export declare function buildConfigurationOverrides(options: StartCommandOptions): ConfigurationOverrides;
|
|
@@ -209,4 +209,34 @@ export declare function createPostCompletionsHandler(options: {
|
|
|
209
209
|
status: number;
|
|
210
210
|
statusText: string;
|
|
211
211
|
}>;
|
|
212
|
+
export declare function createPostEmbeddingsHandler(options: {
|
|
213
|
+
apiClient: APIClient;
|
|
214
|
+
conduitConfiguration: () => InferenceAgentConfiguration;
|
|
215
|
+
configuration: Configuration;
|
|
216
|
+
getModelID: () => string;
|
|
217
|
+
getModelManager: () => ModelManager;
|
|
218
|
+
logger: Logger;
|
|
219
|
+
startup: number;
|
|
220
|
+
}): (params: {
|
|
221
|
+
req: APIRequest;
|
|
222
|
+
res: import("@infersec/fetch").APIResponse;
|
|
223
|
+
parameters: Record<string, never>;
|
|
224
|
+
query: Record<string, never>;
|
|
225
|
+
body: {
|
|
226
|
+
input: string | number[] | string[] | number[][];
|
|
227
|
+
model: string;
|
|
228
|
+
dimensions?: number | null | undefined;
|
|
229
|
+
encoding_format?: "base64" | "float" | null | undefined;
|
|
230
|
+
user?: string | undefined;
|
|
231
|
+
};
|
|
232
|
+
responseSchema: undefined;
|
|
233
|
+
}) => Promise<{
|
|
234
|
+
body: import("stream").Readable;
|
|
235
|
+
headers?: Record<string, string>;
|
|
236
|
+
status: number;
|
|
237
|
+
} | {
|
|
238
|
+
headers?: Record<string, string>;
|
|
239
|
+
status: number;
|
|
240
|
+
statusText: string;
|
|
241
|
+
}>;
|
|
212
242
|
export {};
|
package/dist/utils/openai.d.ts
CHANGED
|
@@ -3,6 +3,23 @@ import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type ULID
|
|
|
3
3
|
import { Logger } from "@infersec/logger";
|
|
4
4
|
import { Configuration } from "../configuration.js";
|
|
5
5
|
import { ModelManager } from "../modelManagement/ModelManager.js";
|
|
6
|
+
export declare function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }: {
|
|
7
|
+
body: unknown;
|
|
8
|
+
conduitConfiguration: InferenceAgentConfiguration;
|
|
9
|
+
endpointId?: ULID | null;
|
|
10
|
+
logger: Logger;
|
|
11
|
+
modelID: ULID;
|
|
12
|
+
modelManager: ModelManager;
|
|
13
|
+
reportMetrics: (payload: InferenceAgentLLMMetricsPayload) => Promise<void>;
|
|
14
|
+
signal?: AbortSignal;
|
|
15
|
+
}): Promise<{
|
|
16
|
+
body: Readable;
|
|
17
|
+
headers: Record<string, string>;
|
|
18
|
+
status: number;
|
|
19
|
+
} | {
|
|
20
|
+
status: number;
|
|
21
|
+
statusText: string;
|
|
22
|
+
}>;
|
|
6
23
|
export declare function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, path, reportMetrics, signal }: {
|
|
7
24
|
body: unknown;
|
|
8
25
|
conduitConfiguration: InferenceAgentConfiguration;
|