@infersec/conduit 1.73.0 → 1.74.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -19921,6 +19921,7 @@ const LLMModelFormatSchema = _enum([
19921
19921
  // Llama.cpp
19922
19922
  "gguf"
19923
19923
  ]);
19924
+ const LLMModelTaskTypeSchema = _enum(["text-generation", "embeddings"]);
19924
19925
  const LLMModelSchema = object({
19925
19926
  format: LLMModelFormatSchema,
19926
19927
  id: string$1().min(1),
@@ -19935,7 +19936,8 @@ const LLMModelSchema = object({
19935
19936
  slug: string$1().min(1),
19936
19937
  type: literal("huggingface")
19937
19938
  })
19938
- ])
19939
+ ]),
19940
+ taskType: LLMModelTaskTypeSchema
19939
19941
  });
19940
19942
  object({
19941
19943
  filePath: string$1().min(1),
@@ -20643,6 +20645,34 @@ const CompletionCreateParamsSchema = object({
20643
20645
  top_p: number$1().min(0).max(1).nullable().optional(),
20644
20646
  user: string$1().optional()
20645
20647
  });
20648
+ // ==================== EMBEDDINGS ====================
20649
+ const EmbeddingCreateParamsSchema = object({
20650
+ dimensions: number$1().int().positive().nullable().optional(),
20651
+ encoding_format: _enum(["float", "base64"]).nullable().optional(),
20652
+ input: union([
20653
+ string$1(),
20654
+ array(string$1()),
20655
+ array(number$1()),
20656
+ array(array(number$1()))
20657
+ ]),
20658
+ model: string$1(),
20659
+ user: string$1().optional()
20660
+ });
20661
+ const EmbeddingUsageSchema = object({
20662
+ prompt_tokens: number$1(),
20663
+ total_tokens: number$1()
20664
+ });
20665
+ const EmbeddingDataSchema = object({
20666
+ embedding: array(number$1()),
20667
+ index: number$1(),
20668
+ object: literal("embedding")
20669
+ });
20670
+ object({
20671
+ data: array(EmbeddingDataSchema),
20672
+ model: string$1(),
20673
+ object: literal("list"),
20674
+ usage: EmbeddingUsageSchema
20675
+ });
20646
20676
 
20647
20677
  const API_CLIENT_CONDUIT_GENERAL_REFERENCE = {
20648
20678
  "/conduit/engine/start": {
@@ -20708,6 +20738,17 @@ const API_CLIENT_CONDUIT_OPENAI_REFERENCE = {
20708
20738
  }
20709
20739
  }
20710
20740
  },
20741
+ "/v1/embeddings": {
20742
+ POST: {
20743
+ auth: {
20744
+ type: "shared-secret"
20745
+ },
20746
+ body: EmbeddingCreateParamsSchema,
20747
+ response: {
20748
+ type: "text-stream"
20749
+ }
20750
+ }
20751
+ },
20711
20752
  "/v1/models": {
20712
20753
  GET: {
20713
20754
  auth: {
@@ -20743,6 +20784,12 @@ const API_CLIENT_CONDUIT_OPENAI_REFERENCE = {
20743
20784
  endpointID: ULIDSchema.describe("Endpoint identifier")
20744
20785
  }}
20745
20786
  },
20787
+ "/api/inferencing/:endpointID/oai/v1/embeddings": {
20788
+ POST: {
20789
+ parameters: {
20790
+ endpointID: ULIDSchema.describe("Endpoint identifier")
20791
+ }}
20792
+ },
20746
20793
  "/api/inferencing/:endpointID/oai/v1/models": {
20747
20794
  GET: {
20748
20795
  parameters: {
@@ -20771,7 +20818,8 @@ object({
20771
20818
  .min(3)
20772
20819
  .refine(value => value.includes("/"), {
20773
20820
  message: "Slug must be fully qualified (owner/repo)"
20774
- })
20821
+ }),
20822
+ taskType: LLMModelTaskTypeSchema.optional()
20775
20823
  });
20776
20824
  object({
20777
20825
  results: array(object({
@@ -20782,6 +20830,7 @@ object({
20782
20830
  name: string$1(),
20783
20831
  provider: _enum(["storage", "huggingface"]),
20784
20832
  providerSlug: string$1(),
20833
+ taskType: LLMModelTaskTypeSchema,
20785
20834
  updated: string$1()
20786
20835
  }))
20787
20836
  });
@@ -20802,11 +20851,13 @@ object({
20802
20851
  name: string$1(),
20803
20852
  updated: string$1()
20804
20853
  })),
20854
+ taskType: LLMModelTaskTypeSchema,
20805
20855
  updated: string$1()
20806
20856
  });
20807
20857
  object({
20858
+ multimodalEnabled: boolean$1().optional(),
20808
20859
  name: ResourceNameSchema.optional(),
20809
- multimodalEnabled: boolean$1().optional()
20860
+ taskType: LLMModelTaskTypeSchema.optional()
20810
20861
  });
20811
20862
  object({
20812
20863
  success: literal(true)
@@ -20851,7 +20902,8 @@ object({
20851
20902
  modelFormat: LLMModelFormatSchema,
20852
20903
  name: string$1(),
20853
20904
  provider: _enum(["storage", "huggingface"]),
20854
- providerSlug: string$1()
20905
+ providerSlug: string$1(),
20906
+ taskType: LLMModelTaskTypeSchema
20855
20907
  })
20856
20908
  .nullable(),
20857
20909
  modelQuantizationLabel: string$1().nullable(),
@@ -21847,6 +21899,40 @@ function parseSSEEvent(rawEvent) {
21847
21899
  };
21848
21900
  }
21849
21901
 
21902
+ function buildConfigurationOverrides(options) {
21903
+ const configurationOverrides = {};
21904
+ if (options.apiUrl) {
21905
+ configurationOverrides.apiURL = options.apiUrl;
21906
+ }
21907
+ if (options.enginePort) {
21908
+ const enginePort = Number.parseInt(options.enginePort, 10);
21909
+ if (Number.isNaN(enginePort) || enginePort < 1 || enginePort > 65535) {
21910
+ throw new Error(`Invalid engine port: ${options.enginePort}`);
21911
+ }
21912
+ configurationOverrides.enginePort = enginePort;
21913
+ }
21914
+ if (options.key) {
21915
+ configurationOverrides.apiKey = options.key;
21916
+ }
21917
+ if (options.port) {
21918
+ const port = Number.parseInt(options.port, 10);
21919
+ if (Number.isNaN(port) || port < 1 || port > 65535) {
21920
+ throw new Error(`Invalid port: ${options.port}`);
21921
+ }
21922
+ configurationOverrides.port = port;
21923
+ }
21924
+ if (options.root) {
21925
+ configurationOverrides.rootDirectory = options.root;
21926
+ }
21927
+ if (options.startMode) {
21928
+ configurationOverrides.startMode = options.startMode;
21929
+ }
21930
+ if (options.source) {
21931
+ configurationOverrides.inferenceSourceID = options.source;
21932
+ }
21933
+ return configurationOverrides;
21934
+ }
21935
+
21850
21936
  function commonjsRequire(path) {
21851
21937
  throw new Error('Could not dynamically require "' + path + '". Please configure the dynamicRequireTargets or/and ignoreDynamicRequires option of @rollup/plugin-commonjs appropriately for this require call to work.');
21852
21938
  }
@@ -114830,6 +114916,9 @@ async function startVLLM({ enginePort, targetDirectory }) {
114830
114916
  "--tensor-parallel-size",
114831
114917
  String(tensorParallelSize)
114832
114918
  ];
114919
+ if (this.model.taskType === "embeddings") {
114920
+ args.push("--task", "embed");
114921
+ }
114833
114922
  if (device) {
114834
114923
  args.push("--device", device);
114835
114924
  }
@@ -116583,6 +116672,9 @@ async function startLlamacpp({ enginePort, targetDirectory }) {
116583
116672
  "--ctx-size",
116584
116673
  String(contextLength)
116585
116674
  ];
116675
+ if (this.model.taskType === "embeddings") {
116676
+ args.push("--embedding");
116677
+ }
116586
116678
  const gpuLayers = typeof engineConfig?.gpuLayers === "number"
116587
116679
  ? engineConfig.gpuLayers
116588
116680
  : Number.parseInt(process.env.LLAMACPP_GPU_LAYERS ?? String(DEFAULT_LLAMACPP_GPU_LAYERS), 10);
@@ -117688,6 +117780,153 @@ function calculateTokensPerSecond$2({ durationMs, totalTokens }) {
117688
117780
  }
117689
117781
  return Math.round(tokensPerSecond);
117690
117782
  }
117783
+ async function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }) {
117784
+ function normalizeTokenCount(value) {
117785
+ if (typeof value === "number" && Number.isFinite(value) && value >= 0) {
117786
+ return value;
117787
+ }
117788
+ return 0;
117789
+ }
117790
+ function reportMetricsSafe(payload) {
117791
+ reportMetrics(payload).catch(error => {
117792
+ logger.warn("Failed to upload LLM prompt metrics", {
117793
+ error: asError(error),
117794
+ requestUrl: "/v1/embeddings"
117795
+ });
117796
+ });
117797
+ }
117798
+ const engineType = conduitConfiguration.engineConfig?.type ?? null;
117799
+ const engineConfig = conduitConfiguration.engineConfig?.config ?? null;
117800
+ const serializedBody = isPlainObject$2(body)
117801
+ ? JSON.stringify(body)
117802
+ : typeof body === "string"
117803
+ ? body
117804
+ : JSON.stringify(body);
117805
+ const requestBodyBytes = Buffer.byteLength(serializedBody, "utf8");
117806
+ const requestStartedAt = Date.now();
117807
+ let upstreamResponseOk = true;
117808
+ const onMonitoringComplete = ({ durationMs, error, responseBytes, usage }) => {
117809
+ const promptTokens = normalizeTokenCount(usage?.promptTokens);
117810
+ const totalTokens = normalizeTokenCount(usage?.totalTokens ?? promptTokens);
117811
+ const latencyMs = Math.max(0, durationMs);
117812
+ reportMetricsSafe({
117813
+ bytes: requestBodyBytes + responseBytes,
117814
+ completionTokens: 0,
117815
+ engine: engineType,
117816
+ endpointId: endpointId ?? null,
117817
+ latencyMs,
117818
+ modelId: modelID,
117819
+ promptTokens,
117820
+ requestBytes: requestBodyBytes,
117821
+ requestId: null,
117822
+ requestMethod: "POST",
117823
+ requestPath: "/v1/embeddings",
117824
+ responseBytes,
117825
+ successful: upstreamResponseOk && !error,
117826
+ timeToFirstTokenMs: null,
117827
+ tokensPerSecond: calculateTokensPerSecond$2({
117828
+ durationMs: latencyMs,
117829
+ totalTokens
117830
+ }),
117831
+ totalTokens
117832
+ });
117833
+ };
117834
+ const response = await modelManager
117835
+ .fetchOpenAI("/v1/embeddings", {
117836
+ body: serializedBody,
117837
+ headers: {
117838
+ "Content-Type": "application/json"
117839
+ },
117840
+ method: "POST",
117841
+ signal
117842
+ })
117843
+ .catch(error => {
117844
+ const err = asError(error);
117845
+ logEngineMetrics({
117846
+ agentEngineType: engineType ?? "unknown",
117847
+ error: err,
117848
+ level: "error",
117849
+ logger,
117850
+ requestBodyBytes,
117851
+ requestPath: "/v1/embeddings",
117852
+ responseBytes: 0,
117853
+ usage: null
117854
+ });
117855
+ const latencyMs = Math.max(0, Date.now() - requestStartedAt);
117856
+ reportMetricsSafe({
117857
+ bytes: requestBodyBytes,
117858
+ completionTokens: 0,
117859
+ engine: engineType,
117860
+ endpointId: endpointId ?? null,
117861
+ latencyMs,
117862
+ modelId: modelID,
117863
+ promptTokens: 0,
117864
+ requestBytes: requestBodyBytes,
117865
+ requestId: null,
117866
+ requestMethod: "POST",
117867
+ requestPath: "/v1/embeddings",
117868
+ responseBytes: 0,
117869
+ successful: false,
117870
+ timeToFirstTokenMs: null,
117871
+ tokensPerSecond: 0,
117872
+ totalTokens: 0
117873
+ });
117874
+ throw err;
117875
+ });
117876
+ upstreamResponseOk = response.ok;
117877
+ const responseStatusText = response.statusText ?? "Upstream request failed";
117878
+ if (!response.body) {
117879
+ logEngineMetrics({
117880
+ agentEngineType: engineType ?? "unknown",
117881
+ level: response.ok ? "info" : "error",
117882
+ logger,
117883
+ requestBodyBytes,
117884
+ requestPath: "/v1/embeddings",
117885
+ responseBytes: 0,
117886
+ usage: null
117887
+ });
117888
+ const latencyMs = Math.max(0, Date.now() - requestStartedAt);
117889
+ reportMetricsSafe({
117890
+ bytes: requestBodyBytes,
117891
+ completionTokens: 0,
117892
+ engine: engineType,
117893
+ endpointId: endpointId ?? null,
117894
+ latencyMs,
117895
+ modelId: modelID,
117896
+ promptTokens: 0,
117897
+ requestBytes: requestBodyBytes,
117898
+ requestId: null,
117899
+ requestMethod: "POST",
117900
+ requestPath: "/v1/embeddings",
117901
+ responseBytes: 0,
117902
+ successful: false,
117903
+ timeToFirstTokenMs: null,
117904
+ tokensPerSecond: 0,
117905
+ totalTokens: 0
117906
+ });
117907
+ return {
117908
+ status: response.status,
117909
+ statusText: responseStatusText
117910
+ };
117911
+ }
117912
+ const monitoredResponse = monitorEngineResponseSingle({
117913
+ agentEngineType: engineType ?? "unknown",
117914
+ body: Readable.fromWeb(response.body),
117915
+ contextLength: modelManager.contextLength,
117916
+ engineConfig,
117917
+ engineType: engineType ?? "unknown",
117918
+ logger,
117919
+ onComplete: onMonitoringComplete,
117920
+ requestBodyBytes,
117921
+ requestPath: "/v1/embeddings",
117922
+ requestStartedAt
117923
+ });
117924
+ return {
117925
+ body: monitoredResponse.stream,
117926
+ headers: Object.fromEntries(response.headers.entries()),
117927
+ status: response.status
117928
+ };
117929
+ }
117691
117930
  async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, path, reportMetrics, signal }) {
117692
117931
  function normalizeTokenCount(value) {
117693
117932
  if (typeof value === "number" && Number.isFinite(value) && value >= 0) {
@@ -117710,6 +117949,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
117710
117949
  const requestStartedAt = Date.now();
117711
117950
  const requestBody = JSON.parse(serializedBody);
117712
117951
  const streamRequested = requestBody.stream === true;
117952
+ let upstreamResponseOk = true;
117713
117953
  const onMonitoringComplete = ({ durationMs, error, responseBytes, timeToFirstTokenMs, usage }) => {
117714
117954
  const completionTokens = normalizeTokenCount(usage?.completionTokens);
117715
117955
  const promptTokens = normalizeTokenCount(usage?.promptTokens);
@@ -117728,7 +117968,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
117728
117968
  requestMethod: "POST",
117729
117969
  requestPath: path,
117730
117970
  responseBytes,
117731
- successful: !error,
117971
+ successful: upstreamResponseOk && !error,
117732
117972
  timeToFirstTokenMs,
117733
117973
  tokensPerSecond: calculateTokensPerSecond$2({
117734
117974
  durationMs: latencyMs,
@@ -117779,6 +118019,7 @@ async function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointI
117779
118019
  });
117780
118020
  throw err;
117781
118021
  });
118022
+ upstreamResponseOk = response.ok;
117782
118023
  const responseStatusText = response.statusText ?? "Upstream request failed";
117783
118024
  if (!response.ok) {
117784
118025
  if (!response.body) {
@@ -117923,6 +118164,26 @@ function createConduitOpenAIAPIReferenceHandlers({ apiClient, conduitConfigurati
117923
118164
  });
117924
118165
  }
117925
118166
  },
118167
+ "/v1/embeddings": {
118168
+ POST: async ({ body, req, res }) => {
118169
+ const modelID = getModelID();
118170
+ const modelManager = getModelManager();
118171
+ const abortController = new AbortController();
118172
+ res.on("close", () => {
118173
+ abortController.abort();
118174
+ });
118175
+ return proxyEmbeddingsRoute({
118176
+ body,
118177
+ conduitConfiguration: conduitConfiguration(),
118178
+ endpointId: extractEndpointId$1(req),
118179
+ logger,
118180
+ modelID,
118181
+ modelManager,
118182
+ reportMetrics: apiClient.reportPromptMetrics,
118183
+ signal: abortController.signal
118184
+ });
118185
+ }
118186
+ },
117926
118187
  "/v1/models": {
117927
118188
  GET: async () => {
117928
118189
  const modelManager = getModelManager();
@@ -117962,6 +118223,9 @@ function createPostChatCompletionsHandler(options) {
117962
118223
  function createPostCompletionsHandler(options) {
117963
118224
  return createConduitOpenAIAPIReferenceHandlers(options)["/v1/completions"].POST;
117964
118225
  }
118226
+ function createPostEmbeddingsHandler(options) {
118227
+ return createConduitOpenAIAPIReferenceHandlers(options)["/v1/embeddings"].POST;
118228
+ }
117965
118229
 
117966
118230
  function isPlainObject$1(value) {
117967
118231
  return typeof value === "object" && value !== null && !Array.isArray(value);
@@ -128707,6 +128971,17 @@ async function createApplication({ abortController, apiClient, configuration, lo
128707
128971
  startup
128708
128972
  })
128709
128973
  },
128974
+ "/v1/embeddings": {
128975
+ POST: createPostEmbeddingsHandler({
128976
+ apiClient,
128977
+ conduitConfiguration: () => conduitConfiguration,
128978
+ configuration,
128979
+ getModelID: () => conduitConfiguration.targetModel.id,
128980
+ getModelManager: () => modelManager,
128981
+ logger,
128982
+ startup
128983
+ })
128984
+ },
128710
128985
  "/v1/models": {
128711
128986
  GET: createGetModelsHandler({
128712
128987
  apiClient,
@@ -129037,36 +129312,7 @@ function registerInferenceCommands({ program }) {
129037
129312
  .option("--start-mode <mode>", "Startup mode: auto|idle (or START_MODE env)")
129038
129313
  .option("--source <id>", "Inference source ID (or SOURCE env)")
129039
129314
  .action(async (options) => {
129040
- const configurationOverrides = {};
129041
- if (options["api-url"]) {
129042
- configurationOverrides.apiURL = options["api-url"];
129043
- }
129044
- if (options["engine-port"]) {
129045
- const enginePort = Number.parseInt(options["engine-port"], 10);
129046
- if (Number.isNaN(enginePort) || enginePort < 1 || enginePort > 65535) {
129047
- throw new Error(`Invalid engine port: ${options["engine-port"]}`);
129048
- }
129049
- configurationOverrides.enginePort = enginePort;
129050
- }
129051
- if (options.key) {
129052
- configurationOverrides.apiKey = options.key;
129053
- }
129054
- if (options.port) {
129055
- const port = Number.parseInt(options.port, 10);
129056
- if (Number.isNaN(port)) {
129057
- throw new Error(`Invalid port: ${options.port}`);
129058
- }
129059
- configurationOverrides.port = port;
129060
- }
129061
- if (options.root) {
129062
- configurationOverrides.rootDirectory = options.root;
129063
- }
129064
- if (options["start-mode"]) {
129065
- configurationOverrides.startMode = options["start-mode"];
129066
- }
129067
- if (options.source) {
129068
- configurationOverrides.inferenceSourceID = options.source;
129069
- }
129315
+ const configurationOverrides = buildConfigurationOverrides(options);
129070
129316
  await startInferenceAgent({ configurationOverrides });
129071
129317
  });
129072
129318
  }
@@ -129711,8 +129957,9 @@ class HuggingFaceClient {
129711
129957
  }
129712
129958
  }
129713
129959
  }
129714
- const seenIds = new Set();
129715
- const models = [];
129960
+ const taskPriority = new Map();
129961
+ pipelineTasks.forEach((task, index) => taskPriority.set(task, index));
129962
+ const modelsById = new Map();
129716
129963
  await Promise.all(queries.map(async ({ task, tag }) => {
129717
129964
  const searchParams = {
129718
129965
  accessToken: this.apiKey ?? undefined,
@@ -129724,9 +129971,6 @@ class HuggingFaceClient {
129724
129971
  }
129725
129972
  };
129726
129973
  for await (const entry of executeListWithRetry(searchParams)) {
129727
- if (seenIds.has(entry.id)) {
129728
- continue;
129729
- }
129730
129974
  const entryForUtils = {
129731
129975
  config: entry.config,
129732
129976
  gated: entry.gated,
@@ -129742,10 +129986,15 @@ class HuggingFaceClient {
129742
129986
  if (targetFormats.length > 0 && !targetFormats.includes(format)) {
129743
129987
  continue;
129744
129988
  }
129745
- seenIds.add(entry.id);
129989
+ const existing = modelsById.get(entry.id);
129990
+ if (existing &&
129991
+ (taskPriority.get(task) ?? Number.MAX_SAFE_INTEGER) >=
129992
+ (taskPriority.get(existing.pipelineTask) ?? Number.MAX_SAFE_INTEGER)) {
129993
+ continue;
129994
+ }
129746
129995
  const parameterCount = parseParameterCount(entry.id, entry.safetensors?.parameters);
129747
129996
  const slug = entry.name?.trim() || entry.id;
129748
- models.push({
129997
+ modelsById.set(entry.id, {
129749
129998
  downloads: entry.downloads,
129750
129999
  format,
129751
130000
  gated: entry.gated || false,
@@ -129753,13 +130002,14 @@ class HuggingFaceClient {
129753
130002
  likes: entry.likes,
129754
130003
  name: entry.name || entry.id,
129755
130004
  parameterCount,
130005
+ pipelineTask: task,
129756
130006
  quantization: extractQuantization(entryForUtils),
129757
130007
  slug,
129758
130008
  updatedAt: entry.updatedAt
129759
130009
  });
129760
130010
  }
129761
130011
  }));
129762
- return models;
130012
+ return Array.from(modelsById.values());
129763
130013
  }
129764
130014
  }
129765
130015
 
@@ -156128,13 +156378,13 @@ function registerBenchmarkCommands({ program }) {
156128
156378
  .option("--account-id <id>", "Account ID (or ACCOUNT_ID env)")
156129
156379
  .option("--output-dir <path>", "Override output directory from config")
156130
156380
  .action(async (options) => {
156131
- const apiUrl = options["api-url"] || process.env.API_URL;
156381
+ const apiUrl = options.apiUrl || process.env.API_URL;
156132
156382
  if (!apiUrl)
156133
156383
  throw new Error("API URL is required (--api-url or API_URL env)");
156134
- const apiKey = options["api-key"] || process.env.API_KEY;
156384
+ const apiKey = options.apiKey || process.env.API_KEY;
156135
156385
  if (!apiKey)
156136
156386
  throw new Error("API key is required (--api-key or API_KEY env)");
156137
- const accountId = options["account-id"] || process.env.ACCOUNT_ID;
156387
+ const accountId = options.accountId || process.env.ACCOUNT_ID;
156138
156388
  if (!accountId)
156139
156389
  throw new Error("Account ID is required (--account-id or ACCOUNT_ID env)");
156140
156390
  const configPath = options.config;
@@ -156145,7 +156395,7 @@ function registerBenchmarkCommands({ program }) {
156145
156395
  apiKey,
156146
156396
  apiUrl,
156147
156397
  configPath,
156148
- outputDir: options["output-dir"]
156398
+ outputDir: options.outputDir
156149
156399
  });
156150
156400
  });
156151
156401
  }
@@ -0,0 +1,11 @@
1
+ import type { ConfigurationOverrides } from "../configuration.js";
2
+ export interface StartCommandOptions {
3
+ apiUrl?: string;
4
+ enginePort?: string;
5
+ key?: string;
6
+ port?: string;
7
+ root?: string;
8
+ source?: string;
9
+ startMode?: string;
10
+ }
11
+ export declare function buildConfigurationOverrides(options: StartCommandOptions): ConfigurationOverrides;
@@ -209,4 +209,34 @@ export declare function createPostCompletionsHandler(options: {
209
209
  status: number;
210
210
  statusText: string;
211
211
  }>;
212
+ export declare function createPostEmbeddingsHandler(options: {
213
+ apiClient: APIClient;
214
+ conduitConfiguration: () => InferenceAgentConfiguration;
215
+ configuration: Configuration;
216
+ getModelID: () => string;
217
+ getModelManager: () => ModelManager;
218
+ logger: Logger;
219
+ startup: number;
220
+ }): (params: {
221
+ req: APIRequest;
222
+ res: import("@infersec/fetch").APIResponse;
223
+ parameters: Record<string, never>;
224
+ query: Record<string, never>;
225
+ body: {
226
+ input: string | number[] | string[] | number[][];
227
+ model: string;
228
+ dimensions?: number | null | undefined;
229
+ encoding_format?: "base64" | "float" | null | undefined;
230
+ user?: string | undefined;
231
+ };
232
+ responseSchema: undefined;
233
+ }) => Promise<{
234
+ body: import("stream").Readable;
235
+ headers?: Record<string, string>;
236
+ status: number;
237
+ } | {
238
+ headers?: Record<string, string>;
239
+ status: number;
240
+ statusText: string;
241
+ }>;
212
242
  export {};
@@ -3,6 +3,23 @@ import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type ULID
3
3
  import { Logger } from "@infersec/logger";
4
4
  import { Configuration } from "../configuration.js";
5
5
  import { ModelManager } from "../modelManagement/ModelManager.js";
6
+ export declare function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }: {
7
+ body: unknown;
8
+ conduitConfiguration: InferenceAgentConfiguration;
9
+ endpointId?: ULID | null;
10
+ logger: Logger;
11
+ modelID: ULID;
12
+ modelManager: ModelManager;
13
+ reportMetrics: (payload: InferenceAgentLLMMetricsPayload) => Promise<void>;
14
+ signal?: AbortSignal;
15
+ }): Promise<{
16
+ body: Readable;
17
+ headers: Record<string, string>;
18
+ status: number;
19
+ } | {
20
+ status: number;
21
+ statusText: string;
22
+ }>;
6
23
  export declare function proxyOpenAIStreamingRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, path, reportMetrics, signal }: {
7
24
  body: unknown;
8
25
  conduitConfiguration: InferenceAgentConfiguration;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@infersec/conduit",
3
3
  "description": "End user conduit agent for connecting local LLMs to the cloud.",
4
- "version": "1.73.0",
4
+ "version": "1.74.1",
5
5
  "bin": {
6
6
  "infersec-conduit": "./dist/cli.js"
7
7
  },