@infersec/conduit 1.100.0 → 1.102.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +2357 -2230
- package/dist/cli.sea.cjs +2355 -2228
- package/dist/modelManagement/ModelManager.d.ts +1 -0
- package/dist/modelManagement/chatTemplate.d.ts +41 -0
- package/dist/requestHandlers/createConduitOpenAIAPIReferenceHandlers.d.ts +3 -1
- package/dist/utils/openai.d.ts +18 -1
- package/package.json +1 -1
- package/dist/modelManagement/extraArgs.d.ts +0 -1
|
@@ -33,6 +33,7 @@ export declare class ModelManager extends EventEmitter<ModelManagerEvents> {
|
|
|
33
33
|
model: LLMModel;
|
|
34
34
|
root: string;
|
|
35
35
|
});
|
|
36
|
+
verifyChatTemplate(): Promise<string | null>;
|
|
36
37
|
fetchOpenAI(path: string, opts?: RequestInit): Promise<Response>;
|
|
37
38
|
prepare({ onDownloadProgress }?: {
|
|
38
39
|
onDownloadProgress?: (update: ModelDownloadProgressUpdate) => void;
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
import { type LLMModel } from "@infersec/definitions";
|
|
2
|
+
export declare function getChatTemplateLocalPath(targetDirectory: string): string;
|
|
3
|
+
/**
|
|
4
|
+
* Materializes the model's chat template override into the model directory:
|
|
5
|
+
* - Always writes the canonical copy for flag-based engines.
|
|
6
|
+
* - For engines that read the model directory (exllamav3, mlx-lm), also writes
|
|
7
|
+
* the transformers-style `chat_template.jinja` drop-in.
|
|
8
|
+
* - When no override is configured, removes any previously materialized files,
|
|
9
|
+
* restoring the model repo's own `chat_template.jinja` if we replaced it.
|
|
10
|
+
*/
|
|
11
|
+
export declare function materializeChatTemplate({ engine, huggingFaceToken, model, targetDirectory }: {
|
|
12
|
+
engine: string;
|
|
13
|
+
huggingFaceToken?: string | null;
|
|
14
|
+
model: LLMModel;
|
|
15
|
+
targetDirectory: string;
|
|
16
|
+
}): Promise<void>;
|
|
17
|
+
/**
|
|
18
|
+
* CLI arguments applying the materialized chat template for flag-based engines.
|
|
19
|
+
* Returns an empty array when no override exists, the engine applies templates
|
|
20
|
+
* from the model directory, or the engine does not support overrides.
|
|
21
|
+
*/
|
|
22
|
+
export declare function getChatTemplateEngineArgs({ engine, model, targetDirectory }: {
|
|
23
|
+
engine: string;
|
|
24
|
+
model: LLMModel;
|
|
25
|
+
targetDirectory: string;
|
|
26
|
+
}): Promise<string[]>;
|
|
27
|
+
/**
|
|
28
|
+
* Post-start verification for llama.cpp: compares the template the server
|
|
29
|
+
* reports via `/props` against the materialized override file. Returns a
|
|
30
|
+
* warning message on mismatch, or null when the template applied cleanly
|
|
31
|
+
* (or verification is not applicable).
|
|
32
|
+
*/
|
|
33
|
+
export declare function verifyEngineChatTemplate({ engine, enginePort, logger, model, targetDirectory }: {
|
|
34
|
+
engine: string;
|
|
35
|
+
enginePort: number;
|
|
36
|
+
logger: {
|
|
37
|
+
warn: (message: string, meta?: Record<string, unknown>) => void;
|
|
38
|
+
};
|
|
39
|
+
model: LLMModel;
|
|
40
|
+
targetDirectory: string;
|
|
41
|
+
}): Promise<string | null>;
|
|
@@ -87,7 +87,7 @@ export declare function createPostChatCompletionsHandler(options: {
|
|
|
87
87
|
type: "image_url";
|
|
88
88
|
image_url: {
|
|
89
89
|
url: string;
|
|
90
|
-
detail?: "
|
|
90
|
+
detail?: "low" | "high" | "auto" | undefined;
|
|
91
91
|
};
|
|
92
92
|
})[];
|
|
93
93
|
name?: string | undefined;
|
|
@@ -118,6 +118,7 @@ export declare function createPostChatCompletionsHandler(options: {
|
|
|
118
118
|
name: string;
|
|
119
119
|
})[];
|
|
120
120
|
model: string;
|
|
121
|
+
chat_template_kwargs?: Record<string, unknown> | null | undefined;
|
|
121
122
|
frequency_penalty?: number | null | undefined;
|
|
122
123
|
function_call?: "none" | "auto" | {
|
|
123
124
|
name: string;
|
|
@@ -133,6 +134,7 @@ export declare function createPostChatCompletionsHandler(options: {
|
|
|
133
134
|
max_tokens?: number | null | undefined;
|
|
134
135
|
n?: number | null | undefined;
|
|
135
136
|
presence_penalty?: number | null | undefined;
|
|
137
|
+
reasoning_effort?: "none" | "minimal" | "low" | "medium" | "high" | "xhigh" | null | undefined;
|
|
136
138
|
response_format?: {
|
|
137
139
|
type: "text" | "json_object";
|
|
138
140
|
} | undefined;
|
package/dist/utils/openai.d.ts
CHANGED
|
@@ -1,8 +1,24 @@
|
|
|
1
1
|
import { Readable } from "node:stream";
|
|
2
|
-
import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type ULID } from "@infersec/definitions";
|
|
2
|
+
import { InferenceAgentConfiguration, InferenceAgentLLMMetricsPayload, type LLMModel, type ULID } from "@infersec/definitions";
|
|
3
3
|
import { Logger } from "@infersec/logger";
|
|
4
4
|
import { Configuration } from "../configuration.js";
|
|
5
5
|
import { ModelManager } from "../modelManagement/ModelManager.js";
|
|
6
|
+
type PlainObject = Record<string, unknown>;
|
|
7
|
+
/**
|
|
8
|
+
* Builds `chat_template_kwargs` for engines that read template variables at
|
|
9
|
+
* render time. Only active when the model carries a template override or
|
|
10
|
+
* thinking config; otherwise the body forwards untouched (engines like vLLM
|
|
11
|
+
* consume top-level `reasoning_effort` natively for supported models).
|
|
12
|
+
*
|
|
13
|
+
* Precedence (highest wins):
|
|
14
|
+
* 1. Request `chat_template_kwargs` (per key)
|
|
15
|
+
* 2. Request top-level `reasoning_effort`
|
|
16
|
+
* 3. Model thinking config defaults
|
|
17
|
+
*/
|
|
18
|
+
export declare function applyChatTemplateKwargs({ body, model }: {
|
|
19
|
+
body: PlainObject;
|
|
20
|
+
model: LLMModel;
|
|
21
|
+
}): PlainObject;
|
|
6
22
|
export declare function proxyEmbeddingsRoute({ body, conduitConfiguration, endpointId, logger, modelID, modelManager, reportMetrics, signal }: {
|
|
7
23
|
body: unknown;
|
|
8
24
|
conduitConfiguration: InferenceAgentConfiguration;
|
|
@@ -39,3 +55,4 @@ export declare function proxyOpenAIStreamingRoute({ body, conduitConfiguration,
|
|
|
39
55
|
status: number;
|
|
40
56
|
statusText: string;
|
|
41
57
|
}>;
|
|
58
|
+
export {};
|
package/package.json
CHANGED
|
@@ -1 +0,0 @@
|
|
|
1
|
-
export declare function parseExtraArgs(extraArgs: unknown): Array<string>;
|