@modular-prompt/driver 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (198) hide show
  1. package/README.md +124 -9
  2. package/dist/cache-controller.d.ts +4 -0
  3. package/dist/cache-controller.d.ts.map +1 -1
  4. package/dist/driver-registry/config-based-factory.d.ts.map +1 -1
  5. package/dist/driver-registry/config-based-factory.js +10 -3
  6. package/dist/driver-registry/config-based-factory.js.map +1 -1
  7. package/dist/driver-registry/factory-helper.d.ts.map +1 -1
  8. package/dist/driver-registry/factory-helper.js +10 -2
  9. package/dist/driver-registry/factory-helper.js.map +1 -1
  10. package/dist/driver-registry/index.d.ts +1 -1
  11. package/dist/driver-registry/index.d.ts.map +1 -1
  12. package/dist/driver-registry/types.d.ts +18 -1
  13. package/dist/driver-registry/types.d.ts.map +1 -1
  14. package/dist/formatter/converter.d.ts.map +1 -1
  15. package/dist/formatter/converter.js +31 -2
  16. package/dist/formatter/converter.js.map +1 -1
  17. package/dist/index.d.ts +5 -3
  18. package/dist/index.d.ts.map +1 -1
  19. package/dist/index.js +5 -3
  20. package/dist/index.js.map +1 -1
  21. package/dist/local-inference/adapters.d.ts +6 -0
  22. package/dist/local-inference/adapters.d.ts.map +1 -1
  23. package/dist/local-inference/driver.d.ts.map +1 -1
  24. package/dist/local-inference/driver.js +45 -24
  25. package/dist/local-inference/driver.js.map +1 -1
  26. package/dist/local-inference/process-client.d.ts +4 -2
  27. package/dist/local-inference/process-client.d.ts.map +1 -1
  28. package/dist/local-inference/process-client.js +24 -8
  29. package/dist/local-inference/process-client.js.map +1 -1
  30. package/dist/local-inference/process-communication.d.ts +9 -2
  31. package/dist/local-inference/process-communication.d.ts.map +1 -1
  32. package/dist/local-inference/process-communication.js +37 -5
  33. package/dist/local-inference/process-communication.js.map +1 -1
  34. package/dist/local-inference/protocol.d.ts +4 -0
  35. package/dist/local-inference/protocol.d.ts.map +1 -1
  36. package/dist/local-inference/request-queue.d.ts +1 -1
  37. package/dist/local-inference/request-queue.d.ts.map +1 -1
  38. package/dist/local-inference/request-queue.js +26 -7
  39. package/dist/local-inference/request-queue.js.map +1 -1
  40. package/dist/local-inference/stream-utils.d.ts +6 -0
  41. package/dist/local-inference/stream-utils.d.ts.map +1 -1
  42. package/dist/local-inference/stream-utils.js.map +1 -1
  43. package/dist/mlx-ml/mlx-cache-controller.d.ts +9 -0
  44. package/dist/mlx-ml/mlx-cache-controller.d.ts.map +1 -1
  45. package/dist/mlx-ml/mlx-cache-controller.js +158 -32
  46. package/dist/mlx-ml/mlx-cache-controller.js.map +1 -1
  47. package/dist/mlx-ml/mlx-cache-support.d.ts +2 -2
  48. package/dist/mlx-ml/mlx-cache-support.d.ts.map +1 -1
  49. package/dist/mlx-ml/mlx-cache-support.js +8 -3
  50. package/dist/mlx-ml/mlx-cache-support.js.map +1 -1
  51. package/dist/mlx-ml/mlx-driver.d.ts +0 -1
  52. package/dist/mlx-ml/mlx-driver.d.ts.map +1 -1
  53. package/dist/mlx-ml/mlx-driver.js +1 -8
  54. package/dist/mlx-ml/mlx-driver.js.map +1 -1
  55. package/dist/mlx-ml/process/index.d.ts +1 -1
  56. package/dist/mlx-ml/process/index.d.ts.map +1 -1
  57. package/dist/mlx-ml/process/index.js +2 -2
  58. package/dist/mlx-ml/process/index.js.map +1 -1
  59. package/dist/models-config/index.d.ts +2 -2
  60. package/dist/models-config/index.d.ts.map +1 -1
  61. package/dist/models-config/index.js +2 -2
  62. package/dist/models-config/index.js.map +1 -1
  63. package/dist/models-config/paths.d.ts +8 -0
  64. package/dist/models-config/paths.d.ts.map +1 -1
  65. package/dist/models-config/paths.js +16 -1
  66. package/dist/models-config/paths.js.map +1 -1
  67. package/dist/models-config/resolve.d.ts +10 -2
  68. package/dist/models-config/resolve.d.ts.map +1 -1
  69. package/dist/models-config/resolve.js +119 -6
  70. package/dist/models-config/resolve.js.map +1 -1
  71. package/dist/models-config/types.d.ts +5 -1
  72. package/dist/models-config/types.d.ts.map +1 -1
  73. package/dist/pytorch/process/index.d.ts +4 -2
  74. package/dist/pytorch/process/index.d.ts.map +1 -1
  75. package/dist/pytorch/process/index.js +24 -7
  76. package/dist/pytorch/process/index.js.map +1 -1
  77. package/dist/pytorch/pytorch-cache-controller.d.ts +84 -0
  78. package/dist/pytorch/pytorch-cache-controller.d.ts.map +1 -0
  79. package/dist/pytorch/pytorch-cache-controller.js +742 -0
  80. package/dist/pytorch/pytorch-cache-controller.js.map +1 -0
  81. package/dist/pytorch/pytorch-cache-support.d.ts +23 -0
  82. package/dist/pytorch/pytorch-cache-support.d.ts.map +1 -0
  83. package/dist/pytorch/pytorch-cache-support.js +47 -0
  84. package/dist/pytorch/pytorch-cache-support.js.map +1 -0
  85. package/dist/pytorch/pytorch-driver.d.ts +8 -1
  86. package/dist/pytorch/pytorch-driver.d.ts.map +1 -1
  87. package/dist/pytorch/pytorch-driver.js +40 -0
  88. package/dist/pytorch/pytorch-driver.js.map +1 -1
  89. package/dist/runtime/check.d.ts.map +1 -1
  90. package/dist/runtime/check.js +9 -6
  91. package/dist/runtime/check.js.map +1 -1
  92. package/dist/runtime/index.d.ts +2 -1
  93. package/dist/runtime/index.d.ts.map +1 -1
  94. package/dist/runtime/index.js +2 -1
  95. package/dist/runtime/index.js.map +1 -1
  96. package/dist/runtime/manifest-core.d.mts +1 -0
  97. package/dist/runtime/manifest-core.mjs +1 -0
  98. package/dist/runtime/manifest-core.mjs.map +1 -1
  99. package/dist/runtime/manifest.d.ts +2 -0
  100. package/dist/runtime/manifest.d.ts.map +1 -1
  101. package/dist/runtime/manifest.js.map +1 -1
  102. package/dist/runtime/paths-core.d.mts +15 -1
  103. package/dist/runtime/paths-core.d.mts.map +1 -1
  104. package/dist/runtime/paths-core.mjs +50 -5
  105. package/dist/runtime/paths-core.mjs.map +1 -1
  106. package/dist/runtime/paths.d.ts +2 -2
  107. package/dist/runtime/paths.d.ts.map +1 -1
  108. package/dist/runtime/paths.js +2 -2
  109. package/dist/runtime/paths.js.map +1 -1
  110. package/dist/runtime/pytorch-template-core.d.mts +11 -0
  111. package/dist/runtime/pytorch-template-core.d.mts.map +1 -0
  112. package/dist/runtime/pytorch-template-core.mjs +54 -0
  113. package/dist/runtime/pytorch-template-core.mjs.map +1 -0
  114. package/dist/runtime/setup-commands-core.d.mts +16 -0
  115. package/dist/runtime/setup-commands-core.d.mts.map +1 -0
  116. package/dist/runtime/setup-commands-core.mjs +18 -0
  117. package/dist/runtime/setup-commands-core.mjs.map +1 -0
  118. package/dist/runtime/setup-commands.d.ts +2 -0
  119. package/dist/runtime/setup-commands.d.ts.map +1 -0
  120. package/dist/runtime/setup-commands.js +2 -0
  121. package/dist/runtime/setup-commands.js.map +1 -0
  122. package/docs/DRIVER_API.md +455 -0
  123. package/docs/LOCAL_MODEL_SETUP.md +765 -0
  124. package/docs/mlx-api-selection.md +301 -0
  125. package/package.json +12 -5
  126. package/scripts/download-model.js +3 -2
  127. package/scripts/runtime-cli.bin.test.ts +142 -0
  128. package/scripts/runtime-cli.js +322 -47
  129. package/scripts/runtime-cli.test.ts +163 -0
  130. package/src/mlx-ml/python/__main__.py +1 -1
  131. package/src/mlx-ml/python/backends/base.py +88 -18
  132. package/src/mlx-ml/python/backends/cache_archive.py +41 -0
  133. package/src/mlx-ml/python/backends/mlx_lm.py +45 -4
  134. package/src/mlx-ml/python/backends/mlx_vlm.py +679 -2
  135. package/src/mlx-ml/python/handlers/cache.py +4 -0
  136. package/src/mlx-ml/python/handlers/generate.py +33 -10
  137. package/src/mlx-ml/python/handlers/tokenize.py +1 -4
  138. package/src/mlx-ml/python/pyproject.toml +9 -3
  139. package/src/mlx-ml/python/server.py +2 -0
  140. package/src/mlx-ml/python/uv.lock +193 -433
  141. package/src/pytorch/templates/cpu-minimal/backends/base.py +139 -0
  142. package/src/pytorch/templates/cpu-minimal/backends/transformers_lm.py +1167 -0
  143. package/src/pytorch/{python → templates/cpu-minimal}/handlers/__init__.py +1 -0
  144. package/src/pytorch/templates/cpu-minimal/handlers/cache.py +88 -0
  145. package/src/pytorch/templates/cpu-minimal/handlers/generate.py +157 -0
  146. package/src/pytorch/{python → templates/cpu-minimal}/pyproject.toml +2 -2
  147. package/src/pytorch/{python → templates/cpu-minimal}/server.py +20 -2
  148. package/src/pytorch/templates/cpu-minimal/tests/test_cache_handler.py +284 -0
  149. package/src/pytorch/templates/cpu-minimal/tests/test_capabilities.py +14 -0
  150. package/src/pytorch/templates/cpu-minimal/tests/test_server.py +141 -0
  151. package/src/pytorch/templates/cpu-minimal/tests/test_transformers_errors.py +89 -0
  152. package/src/pytorch/templates/cpu-minimal/tests/test_transformers_lm_cache.py +554 -0
  153. package/src/pytorch/templates/cpu-minimal/utils/__init__.py +0 -0
  154. package/src/pytorch/{python → templates/cpu-minimal}/utils/token_utils.py +2 -2
  155. package/src/pytorch/templates/cpu-minimal/utils/transformers_errors.py +54 -0
  156. package/src/pytorch/{python → templates/cpu-minimal}/uv.lock +149 -109
  157. package/src/pytorch/templates/cuda/__main__.py +19 -0
  158. package/src/pytorch/templates/cuda/backends/__init__.py +3 -0
  159. package/src/pytorch/{python → templates/cuda}/backends/base.py +54 -6
  160. package/src/pytorch/templates/cuda/backends/transformers_lm.py +379 -0
  161. package/src/pytorch/templates/cuda/handlers/__init__.py +7 -0
  162. package/src/pytorch/templates/cuda/handlers/cache.py +93 -0
  163. package/src/pytorch/templates/cuda/handlers/cancel.py +53 -0
  164. package/src/pytorch/templates/cuda/handlers/capabilities.py +6 -0
  165. package/src/pytorch/templates/cuda/handlers/completion.py +15 -0
  166. package/src/pytorch/templates/cuda/handlers/format_test.py +70 -0
  167. package/src/pytorch/templates/cuda/handlers/generate.py +152 -0
  168. package/src/pytorch/templates/cuda/handlers/render.py +40 -0
  169. package/src/pytorch/templates/cuda/handlers/tokenize.py +63 -0
  170. package/src/pytorch/templates/cuda/pyproject.toml +37 -0
  171. package/src/pytorch/templates/cuda/server.py +158 -0
  172. package/src/pytorch/templates/cuda/tests/__init__.py +0 -0
  173. package/src/pytorch/templates/cuda/tests/test_cache_handler.py +207 -0
  174. package/src/pytorch/templates/cuda/tests/test_capabilities.py +14 -0
  175. package/src/pytorch/templates/cuda/tests/test_server.py +145 -0
  176. package/src/pytorch/templates/cuda/tests/test_transformers_errors.py +89 -0
  177. package/src/pytorch/templates/cuda/tests/test_transformers_lm_cache.py +288 -0
  178. package/src/pytorch/templates/cuda/utils/__init__.py +0 -0
  179. package/src/pytorch/templates/cuda/utils/chat_template_constraints.py +164 -0
  180. package/src/pytorch/templates/cuda/utils/prompt_builder.py +54 -0
  181. package/src/pytorch/templates/cuda/utils/template_render.py +80 -0
  182. package/src/pytorch/templates/cuda/utils/token_utils.py +376 -0
  183. package/src/pytorch/templates/cuda/utils/transformers_errors.py +54 -0
  184. package/src/pytorch/templates/cuda/uv.lock +734 -0
  185. package/src/pytorch/python/backends/transformers_lm.py +0 -127
  186. package/src/pytorch/python/handlers/generate.py +0 -68
  187. /package/src/pytorch/{python → templates/cpu-minimal}/__main__.py +0 -0
  188. /package/src/pytorch/{python → templates/cpu-minimal}/backends/__init__.py +0 -0
  189. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/cancel.py +0 -0
  190. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/capabilities.py +0 -0
  191. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/completion.py +0 -0
  192. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/format_test.py +0 -0
  193. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/render.py +0 -0
  194. /package/src/pytorch/{python → templates/cpu-minimal}/handlers/tokenize.py +0 -0
  195. /package/src/pytorch/{python/utils → templates/cpu-minimal/tests}/__init__.py +0 -0
  196. /package/src/pytorch/{python → templates/cpu-minimal}/utils/chat_template_constraints.py +0 -0
  197. /package/src/pytorch/{python → templates/cpu-minimal}/utils/prompt_builder.py +0 -0
  198. /package/src/pytorch/{python → templates/cpu-minimal}/utils/template_render.py +0 -0
@@ -0,0 +1,765 @@
1
+ # ローカルモデルセットアップガイド
2
+
3
+ ローカル環境でAIモデルを実行するための完全ガイド。
4
+
5
+ ## 目次
6
+
7
+ - [MLX (Apple Silicon)](#mlx-apple-silicon)
8
+ - [環境要件](#環境要件)
9
+ - [初回セットアップ](#初回セットアップ)
10
+ - [モデル設定ファイル](#モデル設定ファイル)
11
+ - [テスト用モデルのダウンロード](#テスト用モデルのダウンロード)
12
+ - [任意のモデルのダウンロード](#任意のモデルのダウンロード)
13
+ - [トラブルシューティング](#トラブルシューティング-mlx)
14
+ - [PyTorch (Transformers)](#pytorch-transformers)
15
+ - [環境要件](#環境要件-pytorch)
16
+ - [初回セットアップ](#初回セットアップ-pytorch)
17
+ - [サポートモデルと Transformers バージョン](#サポートモデルと-transformers-バージョン)
18
+ - [既存ユーザーからの移行](#既存ユーザーからの移行-pytorch)
19
+ - [依存・runtime のカスタマイズ](#依存runtime-のカスタマイズ)
20
+ - [カスタム index / 手動カスタマイズ](#カスタム-index--手動カスタマイズ-pytorch)
21
+ - [トラブルシューティング](#トラブルシューティング-pytorch)
22
+ - [Ollama](#ollama)
23
+ - [インストール](#インストール)
24
+ - [サービスの起動](#サービスの起動)
25
+ - [モデルのダウンロード](#モデルのダウンロード-1)
26
+ - [トラブルシューティング](#トラブルシューティング-ollama)
27
+ - [vLLM (CUDA GPU)](#vllm-cuda-gpu)
28
+ - [環境要件](#環境要件-1)
29
+ - [初回セットアップ](#初回セットアップ-1)
30
+ - [エンジンの起動](#エンジンの起動)
31
+ - [トラブルシューティング](#トラブルシューティング-vllm)
32
+
33
+ ## MLX (Apple Silicon)
34
+
35
+ Apple Silicon Mac専用の高速ローカルLLM実行環境。
36
+
37
+ ### 環境要件
38
+
39
+ - **ハードウェア**: Apple Silicon Mac (M1/M2/M3/M4)
40
+ - **OS**: macOS
41
+ - **Python**: 3.13(`modular-prompt-runtime setup mlx` が venv を作成。手動構成では 3.11 以上でも可)
42
+ - **uv**: Pythonパッケージマネージャー(自動インストールされます)
43
+
44
+ ### 初回セットアップ
45
+
46
+ MLX ドライバーを使うには、Python ランタイムを **明示的にセットアップ** します(`npm install` では自動セットアップされません)。
47
+
48
+ ```bash
49
+ # monorepo ルートから
50
+ pnpm run setup-mlx
51
+
52
+ # @modular-prompt/driver を npm インストールした場合
53
+ modular-prompt-runtime setup mlx
54
+
55
+ # driver パッケージディレクトリから
56
+ cd node_modules/@modular-prompt/driver
57
+ pnpm run setup-mlx
58
+ ```
59
+
60
+ `@modular-prompt/driver` を更新したあと(MLX Python 依存の変更を含む)は、同じコマンドで `~/.modular-prompt/runtimes/mlx/.venv` を再同期してください。
61
+
62
+ Python 環境は `~/.modular-prompt/runtimes/mlx/` に作成されます(プロジェクトや `node_modules` 内には作られません)。
63
+
64
+ **状態確認・掃除:**
65
+
66
+ ```bash
67
+ pnpm --filter @modular-prompt/driver run runtime:status
68
+ pnpm --filter @modular-prompt/driver run runtime:cleanup mlx -- --yes
69
+ ```
70
+
71
+ **セットアップ内容:**
72
+
73
+ 1. uv パッケージマネージャーのインストール(未インストールの場合)
74
+ 2. `~/.modular-prompt/runtimes/mlx/.venv` に Python 仮想環境を作成
75
+ 3. MLX 関連パッケージのインストール
76
+
77
+ ### モデル設定ファイル
78
+
79
+ 通常利用のモデル alias は `~/.modular-prompt/models.yaml`、ローカル統合テスト用の alias は `~/.modular-prompt/models.testing.yaml` に分けて管理できます。別のディレクトリを使う場合は `MODULAR_PROMPT_HOME` を指定します。
80
+
81
+ #### 通常利用の設定
82
+
83
+ simple-chat や extract を `-m` なしで実行するには、通常利用用の `models.yaml` に `models.default` を明示します。次の最小設定を保存したあと、`simple-chat "こんにちは"` などの CLI 例を実行できます。
84
+
85
+ ```bash
86
+ mkdir -p ~/.modular-prompt
87
+ cat > ~/.modular-prompt/models.yaml <<'YAML'
88
+ models:
89
+ default:
90
+ provider: mlx
91
+ model: mlx-community/gemma-3-270m-it-4bit
92
+ YAML
93
+ ```
94
+
95
+ `MODULAR_PROMPT_HOME` を設定している場合は、上記ファイルをそのディレクトリの `models.yaml` として作成してください。モデルを設定しない場合は、CLI の `-m <model-id-or-alias>` で明示的にモデルを指定します。
96
+
97
+ #### 統合テスト用の設定
98
+
99
+ `models.testing.yaml` は通常利用の設定とは別に、統合テストや testing profile で使うモデルを定義するためのファイルです。通常利用用の `models.yaml` の代わりにはなりません。
100
+
101
+ ```bash
102
+ cp packages/driver/test/integration/models.testing.yaml.example \
103
+ ~/.modular-prompt/models.testing.yaml
104
+ ```
105
+
106
+ テスト実行時(Vitest または `NODE_ENV=test`)は `models.testing.yaml` が自動的にマージされます。extract や simple-chat を手元で testing モデルで実行する場合は、profile を明示します。
107
+
108
+ ```bash
109
+ MODULAR_PROMPT_MODELS_PROFILE=testing modular-prompt-extract create meeting -m default docs/notes.txt
110
+ MODULAR_PROMPT_MODELS_PROFILE=testing simple-chat -m default "こんにちは"
111
+ ```
112
+
113
+ マージ順は **base → `models.yaml` → `models.testing.yaml` → overlay** で、testing 側の同名 alias が通常設定を上書きします。認証情報は example に記載せず、環境変数またはローカルの `drivers` 設定で管理してください。
114
+
115
+ `models.testing.yaml.example` の `models.default` は MLX cache 統合テスト向けの text-only LM で、`driverOptions.backend: lm` を明示しています。MLX VLM は `driverOptions.backend: vlm`(または `auto`)でテキストのみの `exact_cache_v1` prompt cache と画像付きの `vision_cache_v1` prompt cache を別 namespace にディスク永続化できます。画像 cache と LM cache との相互利用、VLM incremental prefill は対応していません。
116
+
117
+ ### テスト用モデルのダウンロード
118
+
119
+ 開発・テスト・動作確認用の小型モデルをダウンロードできます:
120
+
121
+ ```bash
122
+ cd node_modules/@modular-prompt/driver
123
+ npm run download-model
124
+ ```
125
+
126
+ **モデル情報:**
127
+ - **モデル名**: `mlx-community/gemma-3-270m-it-4bit`
128
+ - **サイズ**: 約270MB
129
+ - **用途**: 動作確認、開発、ユニットテスト
130
+
131
+ このモデルは軽量で、MLX環境が正しく動作しているかを確認するのに最適です。
132
+
133
+ ### 任意のモデルのダウンロード
134
+
135
+ Hugging Face上の任意のMLXモデルをダウンロードできます。
136
+
137
+ **推奨(テスト用モデル):**
138
+
139
+ ```bash
140
+ pnpm --filter @modular-prompt/driver run download-model
141
+ ```
142
+
143
+ **手動で任意モデルを取得する場合**(`UV_PROJECT_ENVIRONMENT` でホーム venv を指定):
144
+
145
+ ```bash
146
+ cd node_modules/@modular-prompt/driver/src/mlx-ml/python
147
+ UV_PROJECT_ENVIRONMENT=~/.modular-prompt/runtimes/mlx/.venv \
148
+ uv run mlx_lm.generate --model <model-name> --prompt "test" --max-tokens 1
149
+ ```
150
+
151
+ **例:**
152
+
153
+ ```bash
154
+ # Gemma 2B
155
+ UV_PROJECT_ENVIRONMENT=~/.modular-prompt/runtimes/mlx/.venv \
156
+ uv run mlx_lm.generate --model mlx-community/gemma-2-2b-it-4bit --prompt "test" --max-tokens 1
157
+
158
+ # Llama 3.2 3B
159
+ UV_PROJECT_ENVIRONMENT=~/.modular-prompt/runtimes/mlx/.venv \
160
+ uv run mlx_lm.generate --model mlx-community/Llama-3.2-3B-Instruct-4bit --prompt "test" --max-tokens 1
161
+ ```
162
+
163
+ **モデルの保存場所:**
164
+
165
+ ```
166
+ ~/.cache/huggingface/hub/
167
+ ```
168
+
169
+ **注意:**
170
+ - 初回実行時にモデルが自動ダウンロードされるため、事前ダウンロードは必須ではありません
171
+ - モデルサイズに応じて、ダウンロードに時間がかかる場合があります
172
+
173
+ ### トラブルシューティング (MLX)
174
+
175
+ #### Python環境が見つからない
176
+
177
+ ```bash
178
+ # uvの再インストール
179
+ curl -LsSf https://astral.sh/uv/install.sh | sh
180
+
181
+ # MLX環境の再セットアップ(monorepo ルートから)
182
+ pnpm run setup-mlx
183
+ ```
184
+
185
+ #### モデルのダウンロードが失敗する
186
+
187
+ ```bash
188
+ # キャッシュをクリア
189
+ rm -rf ~/.cache/huggingface/hub/
190
+
191
+ # 再度ダウンロード
192
+ npm run download-model
193
+ ```
194
+
195
+ #### メモリ不足エラー
196
+
197
+ より小さいモデル(テスト用の270MBモデルなど)を使用するか、他のアプリケーションを終了してメモリを確保してください。
198
+
199
+ ## PyTorch (Transformers)
200
+
201
+ Windows / Linux など **MLX が使えない環境**向けの Thin Python 推論ドライバ(Local Inference Protocol)。
202
+
203
+ - `cpu-minimal`: `torch` CPU wheel + `transformers` の最小構成
204
+ - `cuda`: CUDA 対応 `torch` wheel + `transformers`(デフォルトは CUDA 12.4 / `cu124`)
205
+ - 量子化や追加依存は下記「カスタム index / 手動カスタマイズ」で調整
206
+ - Linux + NVIDIA で本番寄りの推論が必要な場合は [vLLM](#vllm-cuda-gpu) を検討
207
+
208
+ ### 環境要件 (PyTorch)
209
+
210
+ - **OS**: Windows / Linux / macOS(CUDA variant は NVIDIA ドライバーが使える Linux / Windows 向け。macOS では MLX を推奨)
211
+ - **Python**: 3.12(`setup-pytorch` が venv に使用)
212
+ - **uv**: パッケージマネージャー(未インストール時は自動インストール)
213
+
214
+ ### 初回セットアップ (PyTorch)
215
+
216
+ ```bash
217
+ # monorepo ルートから
218
+ pnpm run setup-pytorch
219
+
220
+ # 状態確認
221
+ pnpm --filter @modular-prompt/driver run runtime:status
222
+ ```
223
+
224
+ Python プロジェクトは `~/.modular-prompt/runtimes/pytorch/python/` に、仮想環境は
225
+ `~/.modular-prompt/runtimes/pytorch/.venv` に作成されます。パッケージ内の
226
+ `src/pytorch/templates/cpu-minimal/` と `src/pytorch/templates/cuda/` は初回 seed 用の template であり、実行時には参照されません。
227
+
228
+ #### CUDA variant
229
+
230
+ NVIDIA GPU を使う場合は `cuda` variant を選択します。CUDA index のデフォルトは `cu124` です。
231
+
232
+ ```bash
233
+ # monorepo ルートから
234
+ pnpm --filter @modular-prompt/driver run setup-pytorch -- --variant cuda
235
+
236
+ # CUDA 12.1 の wheel を選択する例
237
+ pnpm --filter @modular-prompt/driver run setup-pytorch -- --variant cuda --cuda 12.1
238
+
239
+ # @modular-prompt/driver を npm インストールした場合
240
+ modular-prompt-runtime setup pytorch --variant cuda --cuda 12.4
241
+ ```
242
+
243
+ `--cuda 12.4` は PyTorch の `cu124` index に解決されます。`cu124` のような index 名も指定できます。
244
+ セットアップ時に NVIDIA GPU / ドライバーを検出できない場合も、警告を表示して続行します。実行前に
245
+ `runtime:status` の CUDA 状態を確認してください。
246
+
247
+ **セットアップ内容:**
248
+
249
+ 1. `uv venv --python 3.12`
250
+ 2. `torch==2.9.1` を variant に対応する index からインストール(CPU は CPU index、CUDA は `cu124` など)
251
+ 3. `transformers` 等の依存を runtime 側プロジェクトからインストール
252
+
253
+ `runtime:status` は、インストール済み manifest の `variant` / `cudaVersion` / `torchVersion` と、CUDA variant の
254
+ `torch.cuda.is_available()` の結果を表示します。CUDA variant の既定 device は `cuda` です。CUDA が利用できない場合は、
255
+ 推論開始時に NVIDIA ドライバーと CUDA 対応 torch wheel の確認を促すエラーになります。
256
+
257
+ ### サポートモデルと Transformers バージョン
258
+
259
+ cpu-minimal template は `transformers>=5.14.0` を使用します。template の
260
+ `uv.lock` では現在 `transformers==5.15.1` に解決されています。
261
+ `qwen3_5` を使う Qwen 3.5 / 3.6 / 3.8 系のテキスト生成モデル(例:
262
+ `Qwen/Qwen3.5-0.8B`、`Qwen/Qwen3.8-27B-FP8`)は
263
+ このバージョン要件を満たす PyTorch runtime で読み込めます。
264
+
265
+ PyTorch の cpu-minimal backend はテキスト専用で、画像入力には対応していません。
266
+ また、FP8 などの量子化モデルで必要になる CUDA / `accelerate` 等の追加依存は
267
+ `setup-pytorch` に含めていないため、モデルと実行環境に合わせて「手動カスタマイズ」
268
+ を行ってください。
269
+
270
+ `@modular-prompt/driver` を更新して Transformers の依存が変わった場合は、必ず
271
+ `setup-pytorch` を再実行して runtime を更新してください。既存 runtime ではユーザーの
272
+ `pyproject.toml` / `uv.lock` が保持されるため、古い `transformers` の pin が残っている
273
+ 場合は runtime 側の制約を `transformers>=5.14.0` と `safetensors==0.8.0` に更新してから
274
+ sync します。
275
+
276
+ ### 既存ユーザーからの移行 (PyTorch)
277
+
278
+ 既存の `~/.modular-prompt/runtimes/pytorch/`(venv のみ)を利用している場合は、
279
+ `setup-pytorch` を再実行してください。パッケージ内 template から runtime 側の
280
+ `python/` が seed され、以後は runtime 側の Python プロジェクトが実行に使われます。
281
+
282
+ monorepo では:
283
+
284
+ ```bash
285
+ pnpm run setup-pytorch
286
+ ```
287
+
288
+ npm パッケージ利用時は:
289
+
290
+ ```bash
291
+ modular-prompt-runtime setup pytorch
292
+ # または、都度実行する場合
293
+ npx --package @modular-prompt/driver modular-prompt-runtime setup pytorch
294
+ ```
295
+
296
+ パッケージを更新したあとに Python コードを反映する場合は、次の sync を実行します。
297
+ `setup --status` で driver バージョンの差分が表示された場合も同じコマンドを利用できます。
298
+
299
+ ```bash
300
+ modular-prompt-runtime sync pytorch
301
+ ```
302
+
303
+ monorepo では `pnpm --filter @modular-prompt/driver run runtime:sync-pytorch` を使えます。
304
+
305
+ ### 依存・runtime のカスタマイズ
306
+
307
+ runtime 側の `pyproject.toml` はユーザーが編集できます。編集後に sync すると、
308
+ `pyproject.toml` を保持したまま Python コードを更新し、依存を再解決します。
309
+
310
+ ```bash
311
+ vi ~/.modular-prompt/runtimes/pytorch/python/pyproject.toml
312
+ modular-prompt-runtime sync pytorch
313
+ ```
314
+
315
+ `sync pytorch` は package 内 template のコード(`backends/`、`handlers/`、`__main__.py` など)を
316
+ runtime 側へ同期します。runtime 側の `pyproject.toml` と `uv.lock` は上書きされません。
317
+ template は package 更新で置き換わるため、依存設定や永続化したい変更は runtime 側を編集してください。
318
+
319
+ ### カスタム index / 手動カスタマイズ (PyTorch)
320
+
321
+ 自動セットアップが対応していない PyTorch index や torch バージョンを使う場合は、runtime 側の venv に手動で差し替えます。
322
+ 自動セットアップ済みの CUDA variant を別の index に変更する場合にも利用できます。
323
+
324
+ #### torch の手動差し替え
325
+
326
+ ```bash
327
+ PYTORCH_DIR=~/.modular-prompt/runtimes/pytorch
328
+ cd "$PYTORCH_DIR/python"
329
+
330
+ # 例: カスタム CUDA index(環境に合わせて index を選ぶ)
331
+ UV_PROJECT_ENVIRONMENT=$PYTORCH_DIR/.venv \
332
+ uv pip install --upgrade torch --index-url https://download.pytorch.org/whl/cu124
333
+
334
+ UV_PROJECT_ENVIRONMENT=$PYTORCH_DIR/.venv \
335
+ uv run python -c "import torch; print(torch.__version__, torch.cuda.is_available())"
336
+ ```
337
+
338
+ [CUDA 対応表は PyTorch 公式](https://pytorch.org/get-started/locally/)を参照してください。
339
+
340
+ #### 外部 venv / conda の利用
341
+
342
+ 外部 venv / conda を指定する場合も、先に `setup-pytorch` を一度実行して
343
+ `~/.modular-prompt/runtimes/pytorch/python/` を seed してください。実行時の Python
344
+ プロジェクトは常にこの runtime 側を使い、`venvPath`(または環境変数)だけを外部環境へ変更します。
345
+
346
+ ```typescript
347
+ import { PyTorchDriver } from '@modular-prompt/driver';
348
+
349
+ const driver = new PyTorchDriver({
350
+ model: 'gpt2',
351
+ venvPath: '/path/to/existing/.venv',
352
+ device: 'cuda',
353
+ });
354
+ ```
355
+
356
+ または環境変数 `MODULAR_PROMPT_PYTORCH_VENV` で venv パスを指定できます。
357
+
358
+ #### 追加依存(accelerate / 量子化など)
359
+
360
+ 依存を永続化する場合は、runtime 側の `pyproject.toml` に追加してから sync します。
361
+
362
+ ```bash
363
+ vi ~/.modular-prompt/runtimes/pytorch/python/pyproject.toml
364
+ modular-prompt-runtime sync pytorch
365
+ ```
366
+
367
+ モデル要件に応じてユーザーが選択する想定です。現行の CUDA template は `device_map` を使わず model を指定 device に移すため、
368
+ `accelerate` は標準依存に含めていません。必要なモデルで使う場合は runtime 側へ追加してください。
369
+
370
+ ### トラブルシューティング (PyTorch)
371
+
372
+ #### runtime が見つからない
373
+
374
+ ```bash
375
+ pnpm run setup-pytorch
376
+ ```
377
+
378
+ #### CUDA が有効にならない
379
+
380
+ まず状態を確認します。
381
+
382
+ ```bash
383
+ modular-prompt-runtime setup --status
384
+ ```
385
+
386
+ `CUDA: unavailable` または `CUDA: unknown` の場合は、NVIDIA ドライバー、GPU の可視性、選択した
387
+ CUDA index の torch wheel を確認してください。CUDA 版 torch を別の index に差し替える必要がある場合は、
388
+ 上記「カスタム index / 手動カスタマイズ」の手順を実施します。CPU に戻す場合:
389
+
390
+ ```bash
391
+ pnpm --filter @modular-prompt/driver run runtime:cleanup pytorch -- --yes
392
+ pnpm run setup-pytorch
393
+ ```
394
+
395
+ ## Ollama
396
+
397
+ クロスプラットフォーム対応のローカルLLM実行環境。
398
+
399
+ ### インストール
400
+
401
+ #### macOS / Linux
402
+
403
+ ```bash
404
+ curl -fsSL https://ollama.com/install.sh | sh
405
+ ```
406
+
407
+ #### macOS (Homebrew)
408
+
409
+ ```bash
410
+ brew install ollama
411
+ ```
412
+
413
+ #### Windows
414
+
415
+ [ollama.com](https://ollama.com)から Windows版をダウンロードしてインストール。
416
+
417
+ ### サービスの起動
418
+
419
+ #### macOS (Homebrewでインストールした場合)
420
+
421
+ ```bash
422
+ # サービス起動
423
+ brew services start ollama
424
+ ```
425
+
426
+ #### その他
427
+
428
+ ```bash
429
+ # フォアグラウンドで起動
430
+ ollama serve
431
+ ```
432
+
433
+ #### 起動確認
434
+
435
+ ```bash
436
+ # APIが応答するか確認
437
+ curl http://localhost:11434/api/tags
438
+
439
+ # または
440
+ ollama list
441
+ ```
442
+
443
+ ### モデルのダウンロード
444
+
445
+ Ollamaでモデルを使用するには、事前にダウンロードが必要です:
446
+
447
+ ```bash
448
+ # モデルのダウンロード
449
+ ollama pull <model-name>
450
+
451
+ # 例: Llama 3.2のダウンロード
452
+ ollama pull llama3.2
453
+ ```
454
+
455
+ #### ダウンロード状況の確認
456
+
457
+ ```bash
458
+ # ダウンロード済みモデル一覧
459
+ ollama list
460
+ ```
461
+
462
+ **出力例:**
463
+
464
+ ```
465
+ NAME ID SIZE MODIFIED
466
+ llama3.2:latest a80c4f17acd5 2.0 GB 2 hours ago
467
+ gemma2:2b 8ccf136fdd52 1.6 GB 1 day ago
468
+ ```
469
+
470
+ 利用可能なモデルの完全なリストは [ollama.com/library](https://ollama.com/library) を参照してください。
471
+
472
+ ### トラブルシューティング (Ollama)
473
+
474
+ #### サービスが起動しない
475
+
476
+ ```bash
477
+ # プロセスを確認
478
+ ps aux | grep ollama
479
+
480
+ # ポート11434が使用中か確認
481
+ lsof -i :11434
482
+
483
+ # 既存のプロセスを終了して再起動
484
+ pkill ollama
485
+ ollama serve
486
+ ```
487
+
488
+ #### モデルのダウンロードが遅い
489
+
490
+ ネットワーク接続を確認してください。モデルサイズに応じて、数分から数十分かかる場合があります。
491
+
492
+ #### メモリ不足
493
+
494
+ Ollamaはモデルをメモリに読み込むため、モデルサイズの1.5〜2倍のRAMが推奨されます。
495
+
496
+ ## vLLM (CUDA GPU)
497
+
498
+ CUDA GPU環境(Linux)専用の高速LLM推論エンジン。
499
+
500
+ ### 環境要件
501
+
502
+ - **ハードウェア**: NVIDIA CUDA対応GPU
503
+ - **OS**: Linux(CUDA環境)
504
+ - **Python**: 3.10以上(3.14未満)
505
+ - **uv**: Pythonパッケージマネージャー
506
+
507
+ ### 初回セットアップ
508
+
509
+ vLLMドライバーのPython環境をセットアップします:
510
+
511
+ ```bash
512
+ cd node_modules/@modular-prompt/driver/src/vllm/python
513
+ uv sync
514
+ ```
515
+
516
+ **セットアップ内容:**
517
+
518
+ 1. Python仮想環境の作成
519
+ 2. vLLM関連パッケージのインストール(vLLM >= 0.8.0、transformers >= 4.45)
520
+
521
+ **注意:**
522
+ - vLLMはCUDA GPU環境(Linux)でのみ動作します
523
+ - Apple SiliconやWindowsでは使用できません
524
+
525
+ ### エンジンの起動
526
+
527
+ vLLMエンジンはTypeScriptドライバーとは独立して起動します。Unix ドメインソケットを通じて通信します。
528
+
529
+ #### 基本的な起動
530
+
531
+ ```bash
532
+ uv --project node_modules/@modular-prompt/driver/src/vllm/python run python __main__.py \
533
+ --model Qwen/Qwen2.5-7B-Instruct \
534
+ --socket /tmp/vllm.sock
535
+ ```
536
+
537
+ #### ツールコール対応モデルの起動
538
+
539
+ ```bash
540
+ uv --project node_modules/@modular-prompt/driver/src/vllm/python run python __main__.py \
541
+ --model Qwen/Qwen2.5-7B-Instruct \
542
+ --socket /tmp/vllm.sock \
543
+ --tool-call-parser hermes
544
+ ```
545
+
546
+ **利用可能なツールパーサー:**
547
+ - `hermes` - Hermes形式のツールコール
548
+ - `mistral` - Mistral形式のツールコール
549
+ - その他、vLLMのToolParserManagerがサポートするパーサー
550
+
551
+ #### オプション設定
552
+
553
+ ```bash
554
+ uv --project ... run python __main__.py \
555
+ --model <model-name> \
556
+ --socket <socket-path> \
557
+ --tool-call-parser <parser-name> \
558
+ --gpu-memory-utilization 0.9 \
559
+ --tensor-parallel-size 2 \
560
+ --max-model-len 8192
561
+ ```
562
+
563
+ **主要オプション:**
564
+ - `--model`: HuggingFace モデルID(必須)
565
+ - `--socket`: Unix ソケットパス(必須)
566
+ - `--tool-call-parser`: ツールコールパーサー名(オプション)
567
+ - `--gpu-memory-utilization`: GPU メモリ使用率(0.0-1.0)
568
+ - `--tensor-parallel-size`: テンソル並列サイズ
569
+ - `--max-model-len`: 最大モデル長(トークン数)
570
+
571
+ #### エンジンの動作確認
572
+
573
+ エンジンが正常に起動すると、次のメッセージが表示されます:
574
+
575
+ ```
576
+ Loading model: Qwen/Qwen2.5-7B-Instruct
577
+ Model loaded: Qwen/Qwen2.5-7B-Instruct
578
+ Tool parser initialized: hermes
579
+ vLLM engine listening on /tmp/vllm.sock
580
+ ```
581
+
582
+ ### トラブルシューティング (vLLM)
583
+
584
+ #### CUDA環境が見つからない
585
+
586
+ ```bash
587
+ # CUDA バージョン確認
588
+ nvidia-smi
589
+
590
+ # vLLM が CUDA を認識しているか確認
591
+ uv --project ... run python -c "import torch; print(torch.cuda.is_available())"
592
+ ```
593
+
594
+ #### メモリ不足エラー
595
+
596
+ GPU メモリが不足している場合は、以下のオプションを調整してください:
597
+
598
+ ```bash
599
+ # GPU メモリ使用率を下げる
600
+ --gpu-memory-utilization 0.7
601
+
602
+ # より小さいモデルを使用
603
+ --model mlx-community/gemma-2-2b-it-4bit
604
+ ```
605
+
606
+ #### ソケット接続エラー
607
+
608
+ ```bash
609
+ # ソケットファイルが残っている場合は削除
610
+ rm /tmp/vllm.sock
611
+
612
+ # エンジンを再起動
613
+ uv --project ... run python __main__.py ...
614
+ ```
615
+
616
+ #### モデルのダウンロードが失敗する
617
+
618
+ 初回起動時、HuggingFace Hubからモデルが自動的にダウンロードされます。ネットワーク接続を確認してください。
619
+
620
+ ```bash
621
+ # キャッシュをクリア
622
+ rm -rf ~/.cache/huggingface/hub/
623
+
624
+ # 再度起動
625
+ uv --project ... run python __main__.py ...
626
+ ```
627
+
628
+ ## 使用例
629
+
630
+ ### MLX
631
+
632
+ ```typescript
633
+ import { MlxDriver } from '@modular-prompt/driver';
634
+
635
+ const driver = new MlxDriver({
636
+ model: 'mlx-community/gemma-2-2b-it-4bit',
637
+ defaultOptions: {
638
+ max_tokens: 500,
639
+ temperature: 0.7
640
+ }
641
+ });
642
+
643
+ const result = await driver.query(prompt);
644
+ console.log(result.content);
645
+
646
+ await driver.close();
647
+ ```
648
+
649
+ #### VLMモデルをtext-onlyモードで使用
650
+
651
+ VLM(Vision Language Model)対応モデルを画像なしのテキストのみで使用する場合は、`textOnly`フラグを使用します。
652
+
653
+ ```typescript
654
+ const driver = new MlxDriver({
655
+ model: 'mlx-community/Qwen2-VL-2B-Instruct-4bit',
656
+ textOnly: true, // VLMモデルをtext-onlyモードで起動
657
+ defaultOptions: {
658
+ max_tokens: 500,
659
+ temperature: 0.7
660
+ }
661
+ });
662
+
663
+ const result = await driver.query(prompt);
664
+ console.log(result.content);
665
+
666
+ await driver.close();
667
+ ```
668
+
669
+ **`textOnly`フラグの用途:**
670
+ - VLM対応モデルを画像なしで使用したい場合
671
+ - VLMモデルの起動を高速化したい場合(`mlx-vlm`の代わりに`mlx-lm`で起動)
672
+ - VLMモデルでテキストのみのベンチマークを行う場合
673
+
674
+ ### PyTorch
675
+
676
+ ```typescript
677
+ import { PyTorchDriver } from '@modular-prompt/driver';
678
+
679
+ const driver = new PyTorchDriver({
680
+ model: 'gpt2',
681
+ defaultOptions: {
682
+ maxTokens: 128,
683
+ temperature: 0.7,
684
+ },
685
+ });
686
+
687
+ const result = await driver.query(prompt);
688
+ console.log(result.content);
689
+
690
+ await driver.close();
691
+ ```
692
+
693
+ 事前に `pnpm run setup-pytorch` が必要です。
694
+
695
+ ### Ollama
696
+
697
+ ```typescript
698
+ import { OllamaDriver } from '@modular-prompt/driver';
699
+
700
+ const driver = new OllamaDriver({
701
+ model: 'llama3.2',
702
+ defaultOptions: {
703
+ temperature: 0.7,
704
+ maxTokens: 500
705
+ }
706
+ });
707
+
708
+ const result = await driver.query(prompt);
709
+ console.log(result.content);
710
+ ```
711
+
712
+ ### vLLM
713
+
714
+ ```typescript
715
+ import { VllmDriver } from '@modular-prompt/driver';
716
+
717
+ // エンジンを事前に起動しておく必要があります
718
+ // uv --project ... run python __main__.py --model Qwen/Qwen2.5-7B-Instruct --socket /tmp/vllm.sock
719
+
720
+ const driver = new VllmDriver({
721
+ socketPath: '/tmp/vllm.sock',
722
+ defaultOptions: {
723
+ maxTokens: 500,
724
+ temperature: 0.7
725
+ }
726
+ });
727
+
728
+ const result = await driver.query(prompt);
729
+ console.log(result.content);
730
+
731
+ await driver.close();
732
+ ```
733
+
734
+ ### vLLM - ツールコール付き
735
+
736
+ ```typescript
737
+ const driver = new VllmDriver({
738
+ socketPath: '/tmp/vllm.sock'
739
+ });
740
+
741
+ const result = await driver.query(prompt, {
742
+ tools: [
743
+ {
744
+ name: 'get_weather',
745
+ description: 'Get weather information',
746
+ parameters: {
747
+ type: 'object',
748
+ properties: {
749
+ location: { type: 'string' }
750
+ }
751
+ }
752
+ }
753
+ ]
754
+ });
755
+
756
+ if (result.toolCalls) {
757
+ console.log('Tool calls:', result.toolCalls);
758
+ }
759
+ ```
760
+
761
+ ## 関連ドキュメント
762
+
763
+ - [Driver APIリファレンス](./DRIVER_API.md)
764
+ - [packages/driver/README.md](../packages/driver/README.md)
765
+ - [Structured Outputs](./STRUCTURED_OUTPUTS.md)