@infersec/conduit 1.100.0 → 1.102.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -33,6 +33,7 @@ export declare class ModelManager extends EventEmitter<ModelManagerEvents> {
33
33
  model: LLMModel;
34
34
  root: string;
35
35
  });
36
+ verifyChatTemplate(): Promise<string | null>;
36
37
  fetchOpenAI(path: string, opts?: RequestInit): Promise<Response>;
37
38
  prepare({ onDownloadProgress }?: {
38
39
  onDownloadProgress?: (update: ModelDownloadProgressUpdate) => void;
@@ -0,0 +1,41 @@
1
+ import { type LLMModel } from "@infersec/definitions";
2
+ export declare function getChatTemplateLocalPath(targetDirectory: string): string;
3
+ /**
4
+ * Materializes the model's chat template override into the model directory:
5
+ * - Always writes the canonical copy for flag-based engines.
6
+ * - For engines that read the model directory (exllamav3, mlx-lm), also writes
7
+ * the transformers-style `chat_template.jinja` drop-in.
8
+ * - When no override is configured, removes any previously materialized files,
9
+ * restoring the model repo's own `chat_template.jinja` if we replaced it.
10
+ */
11
+ export declare function materializeChatTemplate({ engine, huggingFaceToken, model, targetDirectory }: {
12
+ engine: string;
13
+ huggingFaceToken?: string | null;
14
+ model: LLMModel;
15
+ targetDirectory: string;
16
+ }): Promise<void>;
17
+ /**
18
+ * CLI arguments applying the materialized chat template for flag-based engines.
19
+ * Returns an empty array when no override exists, the engine applies templates
20
+ * from the model directory, or the engine does not support overrides.
21
+ */
22
+ export declare function getChatTemplateEngineArgs({ engine, model, targetDirectory }: {
23
+ engine: string;
24
+ model: LLMModel;
25
+ targetDirectory: string;
26
+ }): Promise<string[]>;
27
+ /**
28
+ * Post-start verification for llama.cpp: compares the template the server
29
+ * reports via `/props` against the materialized override file. Returns a
30
+ * warning message on mismatch, or null when the template applied cleanly
31
+ * (or verification is not applicable).
32
+ */
33
+ export declare function verifyEngineChatTemplate({ engine, enginePort, logger, model, targetDirectory }: {
34
+ engine: string;
35
+ enginePort: number;
36
+ logger: {
37
+ warn: (message: string, meta?: Record<string, unknown>) => void;
38
+ };
39
+ model: LLMModel;
40
+ targetDirectory: string;
41
+ }): Promise<string | null>;
@@ -87,7 +87,7 @@ export declare function createPostChatCompletionsHandler(options: {
87
87
  type: "image_url";
88
88
  image_url: {
89
89
  url: string;
90
- detail?: "auto" | "low" | "high" | undefined;
90
+ detail?: "low" | "high" | "auto" | undefined;
91
91
  };
92
92
  })[];
93
93
  name?: string | undefined;
@@ -118,6 +118,7 @@ export declare function createPostChatCompletionsHandler(options: {
118
118
  name: string;
119
119
  })[];
120
120
  model: string;
121
+ chat_template_kwargs?: Record<string, unknown> | null | undefined;
121
122
  frequency_penalty?: number | null | undefined;
122
123
  function_call?: "none" | "auto" | {
123
124
  name: string;
@@ -133,6 +134,7 @@ export declare function createPostChatCompletionsHandler(options: {
133
134
  max_tokens?: number | null | undefined;
134
135
  n?: number | null | undefined;
135
136
  presence_penalty?: number | null | undefined;
137
+ reasoning_effort?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null | undefined;
136
138
  response_format?: {
137
139
  type: "text" | "json_object";
138
140
  } | undefined;
@@ -1,8 +1,24 @@
1
1
  import { Readable } from "node:stream";
2
- import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type ULID } from "@infersec/definitions";
2
+ import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type LLMModel, type ULID } from "@infersec/definitions";
3
3
  import { Logger } from "@infersec/logger";
4
4
  import { Configuration } from "../configuration.js";
5
5
  import { ModelManager } from "../modelManagement/ModelManager.js";
6
+ type PlainObject = Record<string, unknown>;
7
+ /**
8
+ * Builds `chat_template_kwargs` for engines that read template variables at
9
+ * render time. Only active when the model carries a template override or
10
+ * thinking config; otherwise the body forwards untouched (engines like vLLM
11
+ * consume top-level `reasoning_effort` natively for supported models).
12
+ *
13
+ * Precedence (highest wins):
14
+ * 1. Request `chat_template_kwargs` (per key)
15
+ * 2. Request top-level `reasoning_effort`
16
+ * 3. Model thinking config defaults
17
+ */
18
+ export declare function applyChatTemplateKwargs({ body, model }: {
19
+ body: PlainObject;
20
+ model: LLMModel;
21
+ }): PlainObject;
6
22
  export declare function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }: {
7
23
  body: unknown;
8
24
  conduitConfiguration: InferenceAgentConfiguration;
@@ -39,3 +55,4 @@ export declare function proxyOpenAIStreamingRoute({ body, conduitConfiguration,
39
55
  status: number;
40
56
  statusText: string;
41
57
  }>;
58
+ export {};
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "@infersec/conduit",
3
3
  "description": "End user conduit agent for connecting local LLMs to the cloud.",
4
- "version": "1.100.0",
4
+ "version": "1.102.0",
5
5
  "bin": {
6
6
  "infersec-conduit": "./dist/cli.js"
7
7
  },
@@ -1 +0,0 @@
1
- export declare function parseExtraArgs(extraArgs: unknown): Array<string>;