@modular-prompt/driver 0.13.5 → 0.15.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +117 -4
- package/dist/driver-registry/ai-service.d.ts +23 -1
- package/dist/driver-registry/ai-service.d.ts.map +1 -1
- package/dist/driver-registry/ai-service.js +44 -10
- package/dist/driver-registry/ai-service.js.map +1 -1
- package/dist/driver-registry/config-based-factory.d.ts.map +1 -1
- package/dist/driver-registry/config-based-factory.js +16 -0
- package/dist/driver-registry/config-based-factory.js.map +1 -1
- package/dist/driver-registry/factory-helper.d.ts +2 -0
- package/dist/driver-registry/factory-helper.d.ts.map +1 -1
- package/dist/driver-registry/factory-helper.js +21 -3
- package/dist/driver-registry/factory-helper.js.map +1 -1
- package/dist/driver-registry/index.d.ts +2 -2
- package/dist/driver-registry/index.d.ts.map +1 -1
- package/dist/driver-registry/index.js +1 -1
- package/dist/driver-registry/index.js.map +1 -1
- package/dist/driver-registry/registry.d.ts.map +1 -1
- package/dist/driver-registry/registry.js +3 -1
- package/dist/driver-registry/registry.js.map +1 -1
- package/dist/driver-registry/types.d.ts +8 -2
- package/dist/driver-registry/types.d.ts.map +1 -1
- package/dist/index.d.ts +11 -2
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +11 -1
- package/dist/index.js.map +1 -1
- package/dist/local-inference/adapters.d.ts +66 -0
- package/dist/local-inference/adapters.d.ts.map +1 -0
- package/dist/local-inference/adapters.js +2 -0
- package/dist/local-inference/adapters.js.map +1 -0
- package/dist/local-inference/driver.d.ts +51 -0
- package/dist/local-inference/driver.d.ts.map +1 -0
- package/dist/local-inference/driver.js +309 -0
- package/dist/local-inference/driver.js.map +1 -0
- package/dist/local-inference/index.d.ts +22 -0
- package/dist/local-inference/index.d.ts.map +1 -0
- package/dist/local-inference/index.js +17 -0
- package/dist/local-inference/index.js.map +1 -0
- package/dist/local-inference/process-client.d.ts +50 -0
- package/dist/local-inference/process-client.d.ts.map +1 -0
- package/dist/local-inference/process-client.js +92 -0
- package/dist/local-inference/process-client.js.map +1 -0
- package/dist/local-inference/process-communication.d.ts +41 -0
- package/dist/local-inference/process-communication.d.ts.map +1 -0
- package/dist/{mlx-ml/process → local-inference}/process-communication.js +40 -47
- package/dist/local-inference/process-communication.js.map +1 -0
- package/dist/local-inference/process-port.d.ts +12 -0
- package/dist/local-inference/process-port.d.ts.map +1 -0
- package/dist/local-inference/process-port.js +2 -0
- package/dist/local-inference/process-port.js.map +1 -0
- package/dist/local-inference/prompt-utils.d.ts +6 -0
- package/dist/local-inference/prompt-utils.d.ts.map +1 -0
- package/dist/local-inference/prompt-utils.js +17 -0
- package/dist/local-inference/prompt-utils.js.map +1 -0
- package/dist/local-inference/protocol.d.ts +192 -0
- package/dist/local-inference/protocol.d.ts.map +1 -0
- package/dist/local-inference/protocol.js +2 -0
- package/dist/local-inference/protocol.js.map +1 -0
- package/dist/local-inference/queue-types.d.ts +54 -0
- package/dist/local-inference/queue-types.d.ts.map +1 -0
- package/dist/local-inference/queue-types.js +2 -0
- package/dist/local-inference/queue-types.js.map +1 -0
- package/dist/local-inference/request-queue.d.ts +36 -0
- package/dist/local-inference/request-queue.d.ts.map +1 -0
- package/dist/{mlx-ml/process/queue.js → local-inference/request-queue.js} +89 -56
- package/dist/local-inference/request-queue.js.map +1 -0
- package/dist/local-inference/stream-utils.d.ts +19 -0
- package/dist/local-inference/stream-utils.d.ts.map +1 -0
- package/dist/local-inference/stream-utils.js +76 -0
- package/dist/local-inference/stream-utils.js.map +1 -0
- package/dist/mlx-ml/mlx-cache-support.d.ts +23 -0
- package/dist/mlx-ml/mlx-cache-support.d.ts.map +1 -0
- package/dist/mlx-ml/mlx-cache-support.js +45 -0
- package/dist/mlx-ml/mlx-cache-support.js.map +1 -0
- package/dist/mlx-ml/mlx-driver.d.ts +20 -59
- package/dist/mlx-ml/mlx-driver.d.ts.map +1 -1
- package/dist/mlx-ml/mlx-driver.js +86 -415
- package/dist/mlx-ml/mlx-driver.js.map +1 -1
- package/dist/mlx-ml/mlx-local-inference-adapters.d.ts +3 -0
- package/dist/mlx-ml/mlx-local-inference-adapters.d.ts.map +1 -0
- package/dist/mlx-ml/mlx-local-inference-adapters.js +19 -0
- package/dist/mlx-ml/mlx-local-inference-adapters.js.map +1 -0
- package/dist/mlx-ml/mlx-options.d.ts +19 -0
- package/dist/mlx-ml/mlx-options.d.ts.map +1 -0
- package/dist/mlx-ml/mlx-options.js +30 -0
- package/dist/mlx-ml/mlx-options.js.map +1 -0
- package/dist/mlx-ml/process/index.d.ts +11 -8
- package/dist/mlx-ml/process/index.d.ts.map +1 -1
- package/dist/mlx-ml/process/index.js +75 -52
- package/dist/mlx-ml/process/index.js.map +1 -1
- package/dist/mlx-ml/process/model-specific.d.ts +2 -1
- package/dist/mlx-ml/process/model-specific.d.ts.map +1 -1
- package/dist/mlx-ml/process/model-specific.js.map +1 -1
- package/dist/mlx-ml/process/prompt-builder.d.ts +11 -0
- package/dist/mlx-ml/process/prompt-builder.d.ts.map +1 -0
- package/dist/mlx-ml/process/prompt-builder.js +51 -0
- package/dist/mlx-ml/process/prompt-builder.js.map +1 -0
- package/dist/mlx-ml/process/types.d.ts +15 -183
- package/dist/mlx-ml/process/types.d.ts.map +1 -1
- package/dist/mlx-ml/tool-call-parser/tool-formatter.js +2 -2
- package/dist/mlx-ml/tool-call-parser/tool-formatter.js.map +1 -1
- package/dist/mlx-ml/types.d.ts +2 -45
- package/dist/mlx-ml/types.d.ts.map +1 -1
- package/dist/models-config/index.d.ts +8 -0
- package/dist/models-config/index.d.ts.map +1 -0
- package/dist/models-config/index.js +7 -0
- package/dist/models-config/index.js.map +1 -0
- package/dist/models-config/loader.d.ts +20 -0
- package/dist/models-config/loader.d.ts.map +1 -0
- package/dist/models-config/loader.js +85 -0
- package/dist/models-config/loader.js.map +1 -0
- package/dist/models-config/paths.d.ts +7 -0
- package/dist/models-config/paths.d.ts.map +1 -0
- package/dist/models-config/paths.js +11 -0
- package/dist/models-config/paths.js.map +1 -0
- package/dist/models-config/resolve.d.ts +57 -0
- package/dist/models-config/resolve.d.ts.map +1 -0
- package/dist/models-config/resolve.js +187 -0
- package/dist/models-config/resolve.js.map +1 -0
- package/dist/models-config/types.d.ts +60 -0
- package/dist/models-config/types.d.ts.map +1 -0
- package/dist/models-config/types.js +5 -0
- package/dist/models-config/types.js.map +1 -0
- package/dist/pytorch/process/index.d.ts +35 -0
- package/dist/pytorch/process/index.d.ts.map +1 -0
- package/dist/pytorch/process/index.js +69 -0
- package/dist/pytorch/process/index.js.map +1 -0
- package/dist/pytorch/pytorch-driver.d.ts +35 -0
- package/dist/pytorch/pytorch-driver.d.ts.map +1 -0
- package/dist/pytorch/pytorch-driver.js +48 -0
- package/dist/pytorch/pytorch-driver.js.map +1 -0
- package/dist/pytorch/pytorch-local-inference-adapters.d.ts +3 -0
- package/dist/pytorch/pytorch-local-inference-adapters.d.ts.map +1 -0
- package/dist/pytorch/pytorch-local-inference-adapters.js +19 -0
- package/dist/pytorch/pytorch-local-inference-adapters.js.map +1 -0
- package/dist/pytorch/pytorch-options.d.ts +8 -0
- package/dist/pytorch/pytorch-options.d.ts.map +1 -0
- package/dist/pytorch/pytorch-options.js +21 -0
- package/dist/pytorch/pytorch-options.js.map +1 -0
- package/dist/query-logger.js +1 -1
- package/dist/query-logger.js.map +1 -1
- package/dist/query-utils.d.ts +29 -0
- package/dist/query-utils.d.ts.map +1 -0
- package/dist/query-utils.js +61 -0
- package/dist/query-utils.js.map +1 -0
- package/dist/runtime/check.d.ts +10 -0
- package/dist/runtime/check.d.ts.map +1 -0
- package/dist/runtime/check.js +24 -0
- package/dist/runtime/check.js.map +1 -0
- package/dist/runtime/index.d.ts +4 -0
- package/dist/runtime/index.d.ts.map +1 -0
- package/dist/runtime/index.js +4 -0
- package/dist/runtime/index.js.map +1 -0
- package/dist/runtime/manifest-core.d.mts +27 -0
- package/dist/runtime/manifest-core.d.mts.map +1 -0
- package/dist/runtime/manifest-core.mjs +68 -0
- package/dist/runtime/manifest-core.mjs.map +1 -0
- package/dist/runtime/manifest.d.ts +18 -0
- package/dist/runtime/manifest.d.ts.map +1 -0
- package/dist/runtime/manifest.js +9 -0
- package/dist/runtime/manifest.js.map +1 -0
- package/dist/runtime/paths-core.d.mts +17 -0
- package/dist/runtime/paths-core.d.mts.map +1 -0
- package/dist/runtime/paths-core.mjs +67 -0
- package/dist/runtime/paths-core.mjs.map +1 -0
- package/dist/runtime/paths.d.ts +12 -0
- package/dist/runtime/paths.d.ts.map +1 -0
- package/dist/runtime/paths.js +17 -0
- package/dist/runtime/paths.js.map +1 -0
- package/dist/types.d.ts +20 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/package.json +8 -5
- package/scripts/download-model.js +25 -9
- package/scripts/runtime-cli.js +315 -0
- package/skills/driver-usage/SKILL.md +56 -1
- package/src/mlx-ml/python/__main__.py +43 -4
- package/src/mlx-ml/python/handlers/__init__.py +2 -1
- package/src/mlx-ml/python/handlers/cancel.py +53 -0
- package/src/mlx-ml/python/handlers/completion.py +3 -24
- package/src/mlx-ml/python/handlers/{chat.py → generate.py} +36 -106
- package/src/mlx-ml/python/handlers/render.py +40 -0
- package/src/mlx-ml/python/pyproject.toml +3 -2
- package/src/mlx-ml/python/server.py +35 -8
- package/src/mlx-ml/python/utils/template_render.py +80 -0
- package/src/mlx-ml/python/utils/token_utils.py +2 -2
- package/src/mlx-ml/python/uv.lock +549 -454
- package/src/pytorch/python/__main__.py +19 -0
- package/src/pytorch/python/backends/__init__.py +3 -0
- package/src/pytorch/python/backends/base.py +84 -0
- package/src/pytorch/python/backends/transformers_lm.py +127 -0
- package/src/pytorch/python/handlers/__init__.py +6 -0
- package/src/pytorch/python/handlers/cancel.py +53 -0
- package/src/pytorch/python/handlers/capabilities.py +6 -0
- package/src/pytorch/python/handlers/completion.py +15 -0
- package/src/pytorch/python/handlers/format_test.py +70 -0
- package/src/pytorch/python/handlers/generate.py +68 -0
- package/src/pytorch/python/handlers/render.py +40 -0
- package/src/pytorch/python/handlers/tokenize.py +63 -0
- package/src/pytorch/python/pyproject.toml +36 -0
- package/src/pytorch/python/server.py +140 -0
- package/src/pytorch/python/utils/__init__.py +0 -0
- package/src/pytorch/python/utils/chat_template_constraints.py +164 -0
- package/src/pytorch/python/utils/prompt_builder.py +54 -0
- package/src/pytorch/python/utils/template_render.py +80 -0
- package/src/pytorch/python/utils/token_utils.py +376 -0
- package/src/pytorch/python/uv.lock +694 -0
- package/dist/mlx-ml/process/process-communication.d.ts +0 -37
- package/dist/mlx-ml/process/process-communication.d.ts.map +0 -1
- package/dist/mlx-ml/process/process-communication.js.map +0 -1
- package/dist/mlx-ml/process/queue.d.ts +0 -33
- package/dist/mlx-ml/process/queue.d.ts.map +0 -1
- package/dist/mlx-ml/process/queue.js.map +0 -1
- package/scripts/setup-mlx.js +0 -53
|
@@ -1,22 +1,23 @@
|
|
|
1
|
-
import type {
|
|
2
|
-
import
|
|
3
|
-
import
|
|
4
|
-
import type { CompiledPrompt } from '@modular-prompt/core';
|
|
1
|
+
import type { MlxBackendMode } from '../driver-registry/types.js';
|
|
2
|
+
import { LocalInferenceDriver } from '../local-inference/driver.js';
|
|
3
|
+
import { hasMessageElement } from '../local-inference/prompt-utils.js';
|
|
5
4
|
import type { PromptCacheController } from '../cache-controller.js';
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
export
|
|
5
|
+
import type { MlxModelCapabilities } from './types.js';
|
|
6
|
+
import type { MlxQueryOptions } from './mlx-options.js';
|
|
7
|
+
import type { FormatterOptions } from '../formatter/types.js';
|
|
8
|
+
export { hasMessageElement };
|
|
10
9
|
/**
|
|
11
10
|
* MLX ML driver configuration
|
|
12
11
|
*/
|
|
13
12
|
export interface MlxDriverConfig {
|
|
14
13
|
model: string;
|
|
15
|
-
defaultOptions?: Partial<
|
|
14
|
+
defaultOptions?: Partial<MlxQueryOptions>;
|
|
16
15
|
formatterOptions?: FormatterOptions;
|
|
17
16
|
/** VLM画像の最大辺ピクセル数(デフォルト: 768) */
|
|
18
17
|
maxImageSize?: number;
|
|
19
|
-
/**
|
|
18
|
+
/** MLX Python バックエンド(デフォルト: auto) */
|
|
19
|
+
backend?: MlxBackendMode;
|
|
20
|
+
/** VLMモデルをtext-onlyモードで使用する(backend: lm と同等。後方互換) */
|
|
20
21
|
textOnly?: boolean;
|
|
21
22
|
/** Speculative decoding用のdrafter model名 */
|
|
22
23
|
drafterModel?: string;
|
|
@@ -26,63 +27,23 @@ export interface MlxDriverConfig {
|
|
|
26
27
|
cacheController?: PromptCacheController;
|
|
27
28
|
}
|
|
28
29
|
/**
|
|
29
|
-
* MLX ML driver using Python subprocess
|
|
30
|
+
* MLX ML driver using Python subprocess.
|
|
31
|
+
* 共通ロジックは LocalInferenceDriver に委譲する。
|
|
30
32
|
*/
|
|
31
|
-
export declare class MlxDriver
|
|
32
|
-
private
|
|
33
|
-
private
|
|
34
|
-
private
|
|
35
|
-
private runtimeInfo;
|
|
36
|
-
private modelProcessor;
|
|
37
|
-
private formatterOptions;
|
|
38
|
-
private maxImageSize;
|
|
39
|
-
private queryLogger;
|
|
40
|
-
private cacheController?;
|
|
41
|
-
private cacheControllerBound;
|
|
42
|
-
get defaultOptions(): Partial<MlxMlModelOptions>;
|
|
43
|
-
set defaultOptions(value: Partial<MlxMlModelOptions>);
|
|
33
|
+
export declare class MlxDriver extends LocalInferenceDriver {
|
|
34
|
+
private cacheControllerRaw?;
|
|
35
|
+
private cacheDisabledForVlm;
|
|
36
|
+
private readonly cacheBindingState;
|
|
44
37
|
constructor(config: MlxDriverConfig);
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
*/
|
|
48
|
-
private ensureInitialized;
|
|
49
|
-
/**
|
|
50
|
-
* VLMモデルかどうかを判定
|
|
51
|
-
*/
|
|
52
|
-
private isVLM;
|
|
53
|
-
/**
|
|
54
|
-
* Determine which API to use (chat or completion)
|
|
55
|
-
* Simple logic based on runtime info only
|
|
56
|
-
*/
|
|
57
|
-
private determineApi;
|
|
58
|
-
/**
|
|
59
|
-
* モデルがnativeツール対応かを判定
|
|
60
|
-
* tool_call_format(Python側検出結果)を唯一の判断基準とする
|
|
61
|
-
*/
|
|
62
|
-
private hasNativeToolSupport;
|
|
63
|
-
/**
|
|
64
|
-
* Execute query and return stream
|
|
65
|
-
* Common logic for query and streamQuery
|
|
66
|
-
*/
|
|
67
|
-
private executeQuery;
|
|
68
|
-
/**
|
|
69
|
-
* Query the AI model with a compiled prompt
|
|
70
|
-
*/
|
|
71
|
-
query(prompt: CompiledPrompt, options?: QueryOptions): Promise<QueryResult>;
|
|
72
|
-
/**
|
|
73
|
-
* Stream query implementation
|
|
74
|
-
*/
|
|
75
|
-
streamQuery(prompt: CompiledPrompt, options?: QueryOptions): Promise<StreamResult>;
|
|
38
|
+
get defaultOptions(): Partial<MlxQueryOptions>;
|
|
39
|
+
set defaultOptions(value: Partial<MlxQueryOptions>);
|
|
76
40
|
/**
|
|
77
41
|
* Get model capabilities (public API)
|
|
78
42
|
*
|
|
79
43
|
* Returns runtime information converted to camelCase
|
|
80
44
|
*/
|
|
81
45
|
getCapabilities(): Promise<MlxModelCapabilities>;
|
|
82
|
-
private logCacheStats;
|
|
83
|
-
/**
|
|
84
|
-
* Close the process
|
|
85
|
-
*/
|
|
86
46
|
close(): Promise<void>;
|
|
47
|
+
private logCacheStats;
|
|
87
48
|
}
|
|
88
49
|
//# sourceMappingURL=mlx-driver.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"mlx-driver.d.ts","sourceRoot":"","sources":["../../src/mlx-ml/mlx-driver.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"mlx-driver.d.ts","sourceRoot":"","sources":["../../src/mlx-ml/mlx-driver.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,6BAA6B,CAAC;AAClE,OAAO,EAAE,oBAAoB,EAAE,MAAM,8BAA8B,CAAC;AACpE,OAAO,EAAE,iBAAiB,EAAE,MAAM,oCAAoC,CAAC;AACvE,OAAO,KAAK,EAAE,qBAAqB,EAAE,MAAM,wBAAwB,CAAC;AACpE,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,YAAY,CAAC;AACvD,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,kBAAkB,CAAC;AACxD,OAAO,KAAK,EAAE,gBAAgB,EAAE,MAAM,uBAAuB,CAAC;AAS9D,OAAO,EAAE,iBAAiB,EAAE,CAAC;AAE7B;;GAEG;AACH,MAAM,WAAW,eAAe;IAC9B,KAAK,EAAE,MAAM,CAAC;IACd,cAAc,CAAC,EAAE,OAAO,CAAC,eAAe,CAAC,CAAC;IAC1C,gBAAgB,CAAC,EAAE,gBAAgB,CAAC;IACpC,iCAAiC;IACjC,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,qCAAqC;IACrC,OAAO,CAAC,EAAE,cAAc,CAAC;IACzB,qDAAqD;IACrD,QAAQ,CAAC,EAAE,OAAO,CAAC;IACnB,2CAA2C;IAC3C,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,2DAA2D;IAC3D,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,0BAA0B;IAC1B,eAAe,CAAC,EAAE,qBAAqB,CAAC;CACzC;AAED;;;GAGG;AACH,qBAAa,SAAU,SAAQ,oBAAoB;IACjD,OAAO,CAAC,kBAAkB,CAAC,CAAwB;IACnD,OAAO,CAAC,mBAAmB,CAAS;IACpC,OAAO,CAAC,QAAQ,CAAC,iBAAiB,CAAoB;gBAE1C,MAAM,EAAE,eAAe;IA6CnC,IAAI,cAAc,IAAI,OAAO,CAAC,eAAe,CAAC,CAE7C;IAED,IAAI,cAAc,CAAC,KAAK,EAAE,OAAO,CAAC,eAAe,CAAC,EAEjD;IAED;;;;OAIG;IACG,eAAe,IAAI,OAAO,CAAC,oBAAoB,CAAC;IA8CvC,KAAK,IAAI,OAAO,CAAC,IAAI,CAAC;IAQrC,OAAO,CAAC,aAAa;CAoBtB"}
|
|
@@ -1,389 +1,57 @@
|
|
|
1
|
-
import {
|
|
2
|
-
import {
|
|
3
|
-
import { formatCompletionPrompt } from '../formatter/completion-formatter.js';
|
|
1
|
+
import { LocalInferenceDriver } from '../local-inference/driver.js';
|
|
2
|
+
import { hasMessageElement } from '../local-inference/prompt-utils.js';
|
|
4
3
|
import { MlxProcess } from './process/index.js';
|
|
5
|
-
import {
|
|
6
|
-
import {
|
|
7
|
-
import { extractJSON } from '@modular-prompt/utils';
|
|
8
|
-
import { formatToolDefinitionsAsText } from './tool-call-parser/index.js';
|
|
9
|
-
import { convertMessages, convertToolDefinitions, extractImagePaths } from './mlx-message-utils.js';
|
|
10
|
-
import { QueryLogger } from '../query-logger.js';
|
|
11
|
-
import { extractCacheablePrefix } from '../cache-utils.js';
|
|
4
|
+
import { mlxLocalInferenceAdapters } from './mlx-local-inference-adapters.js';
|
|
5
|
+
import { bindMlxCacheOnCapabilitiesLoaded, createMlxCacheSupport, } from './mlx-cache-support.js';
|
|
12
6
|
import { MlxCacheController } from './mlx-cache-controller.js';
|
|
13
|
-
|
|
14
|
-
// Utility Functions (exported for testing)
|
|
15
|
-
// ========================================================================
|
|
7
|
+
export { hasMessageElement };
|
|
16
8
|
/**
|
|
17
|
-
*
|
|
9
|
+
* MLX ML driver using Python subprocess.
|
|
10
|
+
* 共通ロジックは LocalInferenceDriver に委譲する。
|
|
18
11
|
*/
|
|
19
|
-
export
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
return elements.some(element => {
|
|
24
|
-
const el = element;
|
|
25
|
-
return el?.type === 'message';
|
|
26
|
-
});
|
|
27
|
-
};
|
|
28
|
-
return checkElements(prompt.instructions) ||
|
|
29
|
-
checkElements(prompt.data) ||
|
|
30
|
-
checkElements(prompt.output);
|
|
31
|
-
}
|
|
32
|
-
const META_MARKER = '\x1e__META__:';
|
|
33
|
-
function extractStreamMeta(content) {
|
|
34
|
-
const idx = content.lastIndexOf(META_MARKER);
|
|
35
|
-
if (idx === -1)
|
|
36
|
-
return { content, meta: {} };
|
|
37
|
-
const jsonStr = content.slice(idx + META_MARKER.length);
|
|
38
|
-
try {
|
|
39
|
-
return { content: content.slice(0, idx), meta: JSON.parse(jsonStr) };
|
|
40
|
-
}
|
|
41
|
-
catch {
|
|
42
|
-
return { content: content.slice(0, idx), meta: {} };
|
|
43
|
-
}
|
|
44
|
-
}
|
|
45
|
-
function createStreamIterable(stream) {
|
|
46
|
-
const chunks = [];
|
|
47
|
-
let resolveCompletion;
|
|
48
|
-
const completion = new Promise((resolve) => {
|
|
49
|
-
resolveCompletion = resolve;
|
|
50
|
-
});
|
|
51
|
-
const iterable = {
|
|
52
|
-
async *[Symbol.asyncIterator]() {
|
|
53
|
-
try {
|
|
54
|
-
let buffer = '';
|
|
55
|
-
let markerFound = false;
|
|
56
|
-
for await (const chunk of stream) {
|
|
57
|
-
const str = chunk.toString();
|
|
58
|
-
chunks.push(str);
|
|
59
|
-
if (markerFound)
|
|
60
|
-
continue;
|
|
61
|
-
buffer += str;
|
|
62
|
-
const markerIdx = buffer.indexOf(META_MARKER);
|
|
63
|
-
if (markerIdx !== -1) {
|
|
64
|
-
const text = buffer.slice(0, markerIdx);
|
|
65
|
-
if (text)
|
|
66
|
-
yield text;
|
|
67
|
-
markerFound = true;
|
|
68
|
-
}
|
|
69
|
-
else {
|
|
70
|
-
const safeLen = buffer.length - (META_MARKER.length - 1);
|
|
71
|
-
if (safeLen > 0) {
|
|
72
|
-
yield buffer.slice(0, safeLen);
|
|
73
|
-
buffer = buffer.slice(safeLen);
|
|
74
|
-
}
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
|
-
if (!markerFound && buffer)
|
|
78
|
-
yield buffer;
|
|
79
|
-
const raw = chunks.join('');
|
|
80
|
-
const { content, meta } = extractStreamMeta(raw);
|
|
81
|
-
resolveCompletion({ content, meta, error: null });
|
|
82
|
-
}
|
|
83
|
-
catch (error) {
|
|
84
|
-
const raw = chunks.join('');
|
|
85
|
-
const { content, meta } = extractStreamMeta(raw);
|
|
86
|
-
resolveCompletion({ content, meta, error: error });
|
|
87
|
-
throw error;
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
};
|
|
91
|
-
return { iterable, completion };
|
|
92
|
-
}
|
|
93
|
-
/**
|
|
94
|
-
* MLX ML driver using Python subprocess
|
|
95
|
-
*/
|
|
96
|
-
export class MlxDriver {
|
|
97
|
-
process;
|
|
98
|
-
model;
|
|
99
|
-
_defaultOptions;
|
|
100
|
-
runtimeInfo = null;
|
|
101
|
-
modelProcessor;
|
|
102
|
-
formatterOptions;
|
|
103
|
-
maxImageSize;
|
|
104
|
-
queryLogger = new QueryLogger('MLX');
|
|
105
|
-
cacheController;
|
|
106
|
-
cacheControllerBound = false;
|
|
107
|
-
get defaultOptions() {
|
|
108
|
-
return this._defaultOptions;
|
|
109
|
-
}
|
|
110
|
-
set defaultOptions(value) {
|
|
111
|
-
this._defaultOptions = value;
|
|
112
|
-
}
|
|
12
|
+
export class MlxDriver extends LocalInferenceDriver {
|
|
13
|
+
cacheControllerRaw;
|
|
14
|
+
cacheDisabledForVlm = false;
|
|
15
|
+
cacheBindingState = { bound: false };
|
|
113
16
|
constructor(config) {
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
this.formatterOptions = config.formatterOptions || {};
|
|
117
|
-
this.maxImageSize = config.maxImageSize ?? 768;
|
|
118
|
-
this.process = new MlxProcess(config.model, {
|
|
17
|
+
const mlxProcess = new MlxProcess(config.model, {
|
|
18
|
+
backend: config.backend,
|
|
119
19
|
textOnly: config.textOnly,
|
|
120
20
|
drafterModel: config.drafterModel,
|
|
121
|
-
draftBlockSize: config.draftBlockSize
|
|
21
|
+
draftBlockSize: config.draftBlockSize,
|
|
22
|
+
});
|
|
23
|
+
const cacheSupport = config.cacheController
|
|
24
|
+
? createMlxCacheSupport(config.cacheController)
|
|
25
|
+
: undefined;
|
|
26
|
+
super({
|
|
27
|
+
model: config.model,
|
|
28
|
+
process: mlxProcess,
|
|
29
|
+
adapters: mlxLocalInferenceAdapters,
|
|
30
|
+
formatterOptions: config.formatterOptions,
|
|
31
|
+
maxImageSize: config.maxImageSize,
|
|
32
|
+
defaultOptions: config.defaultOptions,
|
|
33
|
+
loggerPrefix: 'MLX',
|
|
34
|
+
cache: cacheSupport,
|
|
35
|
+
onCapabilitiesLoaded: async (runtimeInfo, ctx) => {
|
|
36
|
+
if (!cacheSupport)
|
|
37
|
+
return;
|
|
38
|
+
await bindMlxCacheOnCapabilitiesLoaded(cacheSupport, this.cacheBindingState, runtimeInfo, ctx, () => {
|
|
39
|
+
this.queryLogger.log.info('VLM models do not support prompt caching — cacheController disabled');
|
|
40
|
+
this.cacheDisabledForVlm = true;
|
|
41
|
+
this.disableCacheSupport();
|
|
42
|
+
});
|
|
43
|
+
},
|
|
122
44
|
});
|
|
123
|
-
this.
|
|
124
|
-
this.cacheController = config.cacheController;
|
|
45
|
+
this.cacheControllerRaw = config.cacheController;
|
|
125
46
|
if (config.drafterModel) {
|
|
126
47
|
this.queryLogger.log.info(`Drafter model: ${config.drafterModel}`);
|
|
127
48
|
}
|
|
128
49
|
}
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
*/
|
|
132
|
-
async ensureInitialized() {
|
|
133
|
-
// Ensure process is initialized
|
|
134
|
-
await this.process.ensureInitialized();
|
|
135
|
-
// Cache runtime info if not already cached
|
|
136
|
-
if (!this.runtimeInfo) {
|
|
137
|
-
try {
|
|
138
|
-
this.runtimeInfo = await this.process.getCapabilities();
|
|
139
|
-
// Update formatterOptions with special tokens from runtime info
|
|
140
|
-
if (this.runtimeInfo.special_tokens) {
|
|
141
|
-
this.formatterOptions.specialTokens = this.runtimeInfo.special_tokens;
|
|
142
|
-
}
|
|
143
|
-
// Update model processor with runtime context
|
|
144
|
-
this.modelProcessor.setRuntimeContext({
|
|
145
|
-
chatRestrictions: this.runtimeInfo.chat_restrictions,
|
|
146
|
-
modelKind: this.runtimeInfo.model_kind,
|
|
147
|
-
});
|
|
148
|
-
// Bind cache controller if provided and not yet bound
|
|
149
|
-
// NOTE: instanceof guard means VLM check only covers MlxCacheController.
|
|
150
|
-
// A custom PromptCacheController on a VLM model would bypass this — add a
|
|
151
|
-
// model-kind guard here if another implementation is introduced.
|
|
152
|
-
if (this.cacheController instanceof MlxCacheController && !this.cacheControllerBound) {
|
|
153
|
-
if (this.runtimeInfo.model_kind === 'vlm') {
|
|
154
|
-
this.queryLogger.log.info('VLM models do not support prompt caching — cacheController disabled');
|
|
155
|
-
this.cacheController = undefined;
|
|
156
|
-
}
|
|
157
|
-
else {
|
|
158
|
-
await this.cacheController.bind(this.process, this.formatterOptions, (msgs) => this.modelProcessor.applyChatSpecificProcessing(msgs));
|
|
159
|
-
this.cacheControllerBound = true;
|
|
160
|
-
}
|
|
161
|
-
}
|
|
162
|
-
}
|
|
163
|
-
catch (error) {
|
|
164
|
-
this.queryLogger.log.error('Failed to get MLX runtime info:', error instanceof Error ? error.message : String(error));
|
|
165
|
-
}
|
|
166
|
-
}
|
|
167
|
-
}
|
|
168
|
-
/**
|
|
169
|
-
* VLMモデルかどうかを判定
|
|
170
|
-
*/
|
|
171
|
-
isVLM() {
|
|
172
|
-
return this.runtimeInfo?.model_kind === 'vlm';
|
|
173
|
-
}
|
|
174
|
-
/**
|
|
175
|
-
* Determine which API to use (chat or completion)
|
|
176
|
-
* Simple logic based on runtime info only
|
|
177
|
-
*/
|
|
178
|
-
determineApi(options) {
|
|
179
|
-
return selectApi(options?.apiStrategy || 'auto', options?.mode, !!this.runtimeInfo?.features.apply_chat_template, this.modelProcessor.hasCompletionProcessor());
|
|
180
|
-
}
|
|
181
|
-
/**
|
|
182
|
-
* モデルがnativeツール対応かを判定
|
|
183
|
-
* tool_call_format(Python側検出結果)を唯一の判断基準とする
|
|
184
|
-
*/
|
|
185
|
-
hasNativeToolSupport() {
|
|
186
|
-
return !!this.runtimeInfo?.features?.chat_template?.tool_call_format?.call_start;
|
|
187
|
-
}
|
|
188
|
-
/**
|
|
189
|
-
* Execute query and return stream
|
|
190
|
-
* Common logic for query and streamQuery
|
|
191
|
-
*/
|
|
192
|
-
async executeQuery(prompt, mlxOptions, options) {
|
|
193
|
-
// APIを選択
|
|
194
|
-
const api = this.determineApi(options);
|
|
195
|
-
// tools変換
|
|
196
|
-
const tools = options?.tools ? convertToolDefinitions(options.tools) : undefined;
|
|
197
|
-
// completion API または nativeツール非対応の場合、tool定義をテキストとしてプロンプトに注入
|
|
198
|
-
let augmentedPrompt = prompt;
|
|
199
|
-
if (options?.tools && options.tools.length > 0 && (api === 'completion' || !this.hasNativeToolSupport())) {
|
|
200
|
-
const toolsText = formatToolDefinitionsAsText(options.tools, this.runtimeInfo?.special_tokens, this.runtimeInfo?.features?.chat_template?.tool_call_format);
|
|
201
|
-
augmentedPrompt = {
|
|
202
|
-
...prompt,
|
|
203
|
-
instructions: [
|
|
204
|
-
...prompt.instructions,
|
|
205
|
-
{ type: 'text', content: toolsText }
|
|
206
|
-
]
|
|
207
|
-
};
|
|
208
|
-
}
|
|
209
|
-
// Record all queries, regardless of API mode or cache usage
|
|
210
|
-
this.cacheController?.recordQuery?.();
|
|
211
|
-
let stream;
|
|
212
|
-
if (api === 'completion') {
|
|
213
|
-
let formattedPrompt = formatCompletionPrompt(augmentedPrompt, this.formatterOptions);
|
|
214
|
-
formattedPrompt = this.modelProcessor.applyCompletionSpecificProcessing(formattedPrompt);
|
|
215
|
-
stream = await this.process.completion(formattedPrompt, mlxOptions);
|
|
216
|
-
}
|
|
217
|
-
else {
|
|
218
|
-
const messages = formatPromptAsMessages(augmentedPrompt, this.formatterOptions);
|
|
219
|
-
const vlm = this.isVLM();
|
|
220
|
-
let mlxMessages = convertMessages(messages, vlm);
|
|
221
|
-
mlxMessages = this.modelProcessor.applyChatSpecificProcessing(mlxMessages);
|
|
222
|
-
const nativeTools = this.hasNativeToolSupport() && tools?.length ? tools : undefined;
|
|
223
|
-
const images = vlm
|
|
224
|
-
? messages.flatMap(m => 'content' in m && !isToolResult(m) ? extractImagePaths(m.content) : [])
|
|
225
|
-
: [];
|
|
226
|
-
// Cache: chat APIのみ、以下の条件を全て満たす場合にキャッシュを使用
|
|
227
|
-
// - options.cache !== false(呼び出し側が明示的に無効化していない)
|
|
228
|
-
// - trustRemoteCode未指定(明示的なtrue/falseどちらもapply_chat_template kwargsに影響)
|
|
229
|
-
let cachePath;
|
|
230
|
-
let cacheTrimTokens;
|
|
231
|
-
const trustRemoteCode = mlxOptions.trustRemoteCode;
|
|
232
|
-
if (this.cacheController && options?.cache !== false && trustRemoteCode === undefined) {
|
|
233
|
-
const prefix = extractCacheablePrefix(augmentedPrompt);
|
|
234
|
-
const hasCacheableContent = prefix.instructions.length > 0 ||
|
|
235
|
-
prefix.data.length > 0;
|
|
236
|
-
if (hasCacheableContent) {
|
|
237
|
-
const cacheStart = performance.now();
|
|
238
|
-
const handle = await this.cacheController.prepare({
|
|
239
|
-
model: this.model,
|
|
240
|
-
instructions: prefix.instructions,
|
|
241
|
-
data: prefix.data,
|
|
242
|
-
tools: nativeTools ? options.tools : undefined,
|
|
243
|
-
reasoningEffort: options?.reasoningEffort,
|
|
244
|
-
readOnly: options?.cache === 'read-only',
|
|
245
|
-
});
|
|
246
|
-
cachePath = handle.ref || undefined;
|
|
247
|
-
cacheTrimTokens = handle.trimTokens;
|
|
248
|
-
if (cachePath) {
|
|
249
|
-
this.queryLogger.log.debug(`cache prepare ${(performance.now() - cacheStart).toFixed(0)}ms`, `(${prefix.instructions.length}i+${prefix.data.length}d)`, cacheTrimTokens != null ? `trim=${cacheTrimTokens}` : '');
|
|
250
|
-
}
|
|
251
|
-
}
|
|
252
|
-
}
|
|
253
|
-
stream = await this.process.chat(mlxMessages, undefined, mlxOptions, nativeTools, images.length > 0 ? images : undefined, images.length > 0 ? this.maxImageSize : undefined, options?.reasoningEffort, cachePath, cacheTrimTokens);
|
|
254
|
-
const cacheTokensUsed = cachePath
|
|
255
|
-
? (cacheTrimTokens ?? (this.cacheController instanceof MlxCacheController
|
|
256
|
-
? this.cacheController.readCacheTokenCount(cachePath) : 0))
|
|
257
|
-
: 0;
|
|
258
|
-
return { stream, cacheTokensUsed };
|
|
259
|
-
}
|
|
260
|
-
return { stream, cacheTokensUsed: 0 };
|
|
261
|
-
}
|
|
262
|
-
/**
|
|
263
|
-
* Query the AI model with a compiled prompt
|
|
264
|
-
*/
|
|
265
|
-
async query(prompt, options) {
|
|
266
|
-
// Use streamQuery for consistency with other drivers
|
|
267
|
-
const { stream, result } = await this.streamQuery(prompt, options);
|
|
268
|
-
// Consume the stream to trigger completion
|
|
269
|
-
// This is necessary because the result promise only resolves when the stream is fully consumed
|
|
270
|
-
// eslint-disable-next-line @typescript-eslint/no-unused-vars
|
|
271
|
-
for await (const _chunk of stream) {
|
|
272
|
-
// Just consume the stream, don't need to do anything with the chunks
|
|
273
|
-
}
|
|
274
|
-
return result;
|
|
50
|
+
get defaultOptions() {
|
|
51
|
+
return super.defaultOptions;
|
|
275
52
|
}
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
*/
|
|
279
|
-
async streamQuery(prompt, options) {
|
|
280
|
-
await this.ensureInitialized();
|
|
281
|
-
// Merge options (only override if explicitly provided)
|
|
282
|
-
const mlxOptions = {
|
|
283
|
-
...this.defaultOptions,
|
|
284
|
-
...(options?.maxTokens !== undefined && { maxTokens: options.maxTokens }),
|
|
285
|
-
...(options?.temperature !== undefined && { temperature: options.temperature }),
|
|
286
|
-
...(options?.topP !== undefined && { topP: options.topP }),
|
|
287
|
-
...(options?.topK !== undefined && { topK: options.topK }),
|
|
288
|
-
};
|
|
289
|
-
this.queryLogger.mark(mlxOptions);
|
|
290
|
-
// Use executeQuery for the actual stream generation
|
|
291
|
-
const queryStart = performance.now();
|
|
292
|
-
const { stream, cacheTokensUsed } = await this.executeQuery(prompt, mlxOptions, options);
|
|
293
|
-
const streamStart = performance.now();
|
|
294
|
-
this.queryLogger.log.debug(`setup ${(streamStart - queryStart).toFixed(0)}ms`);
|
|
295
|
-
// Convert stream to async iterable with collection
|
|
296
|
-
const { iterable, completion } = createStreamIterable(stream);
|
|
297
|
-
// Wrap iterable with phase-separated timing:
|
|
298
|
-
// TTFT: streamStart → first chunk (inference latency, shows cache benefit)
|
|
299
|
-
// generation: first chunk → last chunk (output speed, tok/s)
|
|
300
|
-
// query total: queryStart → last chunk (end-to-end)
|
|
301
|
-
const queryLogger = this.queryLogger;
|
|
302
|
-
let firstChunkTime = 0;
|
|
303
|
-
const wrappedIterable = {
|
|
304
|
-
[Symbol.asyncIterator]() {
|
|
305
|
-
const inner = iterable[Symbol.asyncIterator]();
|
|
306
|
-
let firstChunk = true;
|
|
307
|
-
return {
|
|
308
|
-
async next() {
|
|
309
|
-
const result = await inner.next();
|
|
310
|
-
if (!result.done) {
|
|
311
|
-
if (firstChunk) {
|
|
312
|
-
firstChunk = false;
|
|
313
|
-
firstChunkTime = performance.now();
|
|
314
|
-
queryLogger.log.debug(`TTFT ${(firstChunkTime - streamStart).toFixed(0)}ms`);
|
|
315
|
-
}
|
|
316
|
-
}
|
|
317
|
-
if (result.done) {
|
|
318
|
-
const now = performance.now();
|
|
319
|
-
if (firstChunkTime > 0) {
|
|
320
|
-
const genMs = now - firstChunkTime;
|
|
321
|
-
queryLogger.log.debug(`generation ${genMs.toFixed(0)}ms (query total ${(now - queryStart).toFixed(0)}ms)`);
|
|
322
|
-
}
|
|
323
|
-
else {
|
|
324
|
-
queryLogger.log.debug(`query total ${(performance.now() - queryStart).toFixed(0)}ms`);
|
|
325
|
-
}
|
|
326
|
-
}
|
|
327
|
-
return result;
|
|
328
|
-
},
|
|
329
|
-
async return(value) {
|
|
330
|
-
return inner.return?.(value) ?? { done: true, value: undefined };
|
|
331
|
-
},
|
|
332
|
-
async throw(e) {
|
|
333
|
-
return inner.throw?.(e) ?? { done: true, value: undefined };
|
|
334
|
-
},
|
|
335
|
-
};
|
|
336
|
-
},
|
|
337
|
-
};
|
|
338
|
-
// Create result promise that waits for stream completion
|
|
339
|
-
const cacheController = this.cacheController;
|
|
340
|
-
const resultPromise = completion.then(({ content, meta, error }) => {
|
|
341
|
-
// If there was an error, log and throw it
|
|
342
|
-
if (error) {
|
|
343
|
-
this.queryLogger.log.error('Stream error:', error.message);
|
|
344
|
-
throw error;
|
|
345
|
-
}
|
|
346
|
-
if (cacheController instanceof MlxCacheController && meta.prompt_tokens != null) {
|
|
347
|
-
cacheController.recordPromptTokens(meta.prompt_tokens, cacheTokensUsed);
|
|
348
|
-
}
|
|
349
|
-
// Log accurate generation stats if available
|
|
350
|
-
if (meta.generation_tokens != null && firstChunkTime > 0) {
|
|
351
|
-
const genMs = performance.now() - firstChunkTime;
|
|
352
|
-
const actualTps = (meta.generation_tokens / genMs * 1000).toFixed(1);
|
|
353
|
-
this.queryLogger.log.debug(`${meta.generation_tokens} tokens, ${actualTps} tok/s`);
|
|
354
|
-
}
|
|
355
|
-
// Response post-processing: thinking抽出 + tool call解析
|
|
356
|
-
const hasTools = options?.tools && options.tools.length > 0;
|
|
357
|
-
const responseProcessor = selectResponseProcessor(this.model, this.runtimeInfo, { enableToolParsing: !!hasTools });
|
|
358
|
-
const parsed = responseProcessor(content);
|
|
359
|
-
let finalContent = parsed.content;
|
|
360
|
-
const thinkingContent = parsed.thinkingContent;
|
|
361
|
-
const toolCalls = parsed.toolCalls;
|
|
362
|
-
if (thinkingContent) {
|
|
363
|
-
this.queryLogger.log.verbose('Thinking content:', thinkingContent);
|
|
364
|
-
}
|
|
365
|
-
// Handle structured output if schema is provided
|
|
366
|
-
let structuredOutput;
|
|
367
|
-
if (prompt.metadata?.outputSchema && finalContent) {
|
|
368
|
-
const extracted = extractJSON(finalContent, { multiple: false });
|
|
369
|
-
if (extracted.source !== 'none' && extracted.data !== null) {
|
|
370
|
-
structuredOutput = extracted.data;
|
|
371
|
-
}
|
|
372
|
-
}
|
|
373
|
-
const finishReason = toolCalls ? 'tool_calls' : 'stop';
|
|
374
|
-
return {
|
|
375
|
-
content: finalContent,
|
|
376
|
-
thinkingContent,
|
|
377
|
-
structuredOutput,
|
|
378
|
-
toolCalls,
|
|
379
|
-
finishReason,
|
|
380
|
-
...this.queryLogger.collect()
|
|
381
|
-
};
|
|
382
|
-
});
|
|
383
|
-
return {
|
|
384
|
-
stream: wrappedIterable,
|
|
385
|
-
result: resultPromise
|
|
386
|
-
};
|
|
53
|
+
set defaultOptions(value) {
|
|
54
|
+
super.defaultOptions = value ?? {};
|
|
387
55
|
}
|
|
388
56
|
/**
|
|
389
57
|
* Get model capabilities (public API)
|
|
@@ -392,51 +60,62 @@ export class MlxDriver {
|
|
|
392
60
|
*/
|
|
393
61
|
async getCapabilities() {
|
|
394
62
|
await this.ensureInitialized();
|
|
395
|
-
|
|
63
|
+
const runtimeInfo = this.getRuntimeInfo();
|
|
64
|
+
if (!runtimeInfo) {
|
|
396
65
|
throw new Error('Failed to retrieve model capabilities');
|
|
397
66
|
}
|
|
398
|
-
// Convert snake_case to camelCase
|
|
399
67
|
return {
|
|
400
|
-
methods:
|
|
401
|
-
specialTokens:
|
|
68
|
+
methods: runtimeInfo.methods,
|
|
69
|
+
specialTokens: runtimeInfo.special_tokens,
|
|
402
70
|
features: {
|
|
403
|
-
hasChatTemplate:
|
|
404
|
-
vocabSize:
|
|
405
|
-
modelMaxLength:
|
|
406
|
-
chatTemplate:
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
415
|
-
|
|
416
|
-
|
|
417
|
-
|
|
71
|
+
hasChatTemplate: runtimeInfo.features.apply_chat_template,
|
|
72
|
+
vocabSize: runtimeInfo.features.vocab_size,
|
|
73
|
+
modelMaxLength: runtimeInfo.features.model_max_length,
|
|
74
|
+
chatTemplate: runtimeInfo.features.chat_template
|
|
75
|
+
? {
|
|
76
|
+
supportedRoles: runtimeInfo.features.chat_template.supported_roles,
|
|
77
|
+
preview: runtimeInfo.features.chat_template.preview,
|
|
78
|
+
constraints: runtimeInfo.features.chat_template.constraints,
|
|
79
|
+
toolCallFormat: runtimeInfo.features.chat_template.tool_call_format
|
|
80
|
+
? {
|
|
81
|
+
toolParserType: runtimeInfo.features.chat_template.tool_call_format.tool_parser_type,
|
|
82
|
+
callStart: runtimeInfo.features.chat_template.tool_call_format.call_start,
|
|
83
|
+
callEnd: runtimeInfo.features.chat_template.tool_call_format.call_end,
|
|
84
|
+
responseStart: runtimeInfo.features.chat_template.tool_call_format.response_start,
|
|
85
|
+
responseEnd: runtimeInfo.features.chat_template.tool_call_format.response_end,
|
|
86
|
+
}
|
|
87
|
+
: undefined,
|
|
88
|
+
}
|
|
89
|
+
: undefined,
|
|
418
90
|
},
|
|
419
|
-
chatRestrictions:
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
|
|
91
|
+
chatRestrictions: runtimeInfo.chat_restrictions
|
|
92
|
+
? {
|
|
93
|
+
singleSystemAtStart: runtimeInfo.chat_restrictions.single_system_at_start,
|
|
94
|
+
maxSystemMessages: runtimeInfo.chat_restrictions.max_system_messages,
|
|
95
|
+
alternatingTurns: runtimeInfo.chat_restrictions.alternating_turns,
|
|
96
|
+
requiresUserLast: runtimeInfo.chat_restrictions.requires_user_last,
|
|
97
|
+
allowEmptyMessages: runtimeInfo.chat_restrictions.allow_empty_messages,
|
|
98
|
+
}
|
|
99
|
+
: undefined,
|
|
426
100
|
};
|
|
427
101
|
}
|
|
102
|
+
async close() {
|
|
103
|
+
this.logCacheStats();
|
|
104
|
+
if (this.cacheDisabledForVlm) {
|
|
105
|
+
await this.cacheControllerRaw?.close();
|
|
106
|
+
}
|
|
107
|
+
await super.close();
|
|
108
|
+
}
|
|
428
109
|
logCacheStats() {
|
|
429
|
-
if (
|
|
110
|
+
if (this.cacheDisabledForVlm)
|
|
430
111
|
return;
|
|
431
|
-
|
|
112
|
+
if (!(this.cacheControllerRaw instanceof MlxCacheController))
|
|
113
|
+
return;
|
|
114
|
+
const s = this.cacheControllerRaw.getStats();
|
|
432
115
|
if (s.totalQueries === 0)
|
|
433
116
|
return;
|
|
434
|
-
const queryBreakdown = s.incremental + s.fresh > 0
|
|
435
|
-
|
|
436
|
-
: '';
|
|
437
|
-
const parts = [
|
|
438
|
-
`cache stats: ${s.totalQueries} queries${queryBreakdown}`,
|
|
439
|
-
];
|
|
117
|
+
const queryBreakdown = s.incremental + s.fresh > 0 ? ` (incremental ${s.incremental}, fresh ${s.fresh})` : '';
|
|
118
|
+
const parts = [`cache stats: ${s.totalQueries} queries${queryBreakdown}`];
|
|
440
119
|
if (s.totalPromptTokens > 0) {
|
|
441
120
|
const reusedRate = ((s.prefillReusedTokens / s.totalPromptTokens) * 100).toFixed(0);
|
|
442
121
|
parts.push(`prompt ${s.totalPromptTokens} tokens, ${s.prefillReusedTokens} reused (${reusedRate}%)`);
|
|
@@ -446,13 +125,5 @@ export class MlxDriver {
|
|
|
446
125
|
}
|
|
447
126
|
this.queryLogger.log.verbose(parts.join(' | '));
|
|
448
127
|
}
|
|
449
|
-
/**
|
|
450
|
-
* Close the process
|
|
451
|
-
*/
|
|
452
|
-
async close() {
|
|
453
|
-
this.logCacheStats();
|
|
454
|
-
await this.cacheController?.close();
|
|
455
|
-
await this.process.exit();
|
|
456
|
-
}
|
|
457
128
|
}
|
|
458
129
|
//# sourceMappingURL=mlx-driver.js.map
|