whisper-windows-mcp 1.4.2 → 1.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -10,7 +10,7 @@ import { CallToolRequestSchema, ListToolsRequestSchema, } from "@modelcontextpro
10
10
  import { execFile } from "child_process";
11
11
  import { existsSync, unlinkSync, readdirSync, writeFileSync, readFileSync } from "fs";
12
12
  import { cpus, tmpdir } from "os";
13
- import { join, extname, basename } from "path";
13
+ import { join, extname, basename, dirname } from "path";
14
14
  import { promisify } from "util";
15
15
  const execFileAsync = promisify(execFile);
16
16
  // ---------------------------------------------------------------------------
@@ -53,6 +53,48 @@ async function isWhisperRunning() {
53
53
  return false;
54
54
  }
55
55
  }
56
+ async function detectGpus() {
57
+ try {
58
+ const { stdout } = await execFileAsync("wmic", ["path", "win32_VideoController", "get", "name,AdapterRAM", "/format:csv"], { windowsHide: true });
59
+ const gpus = [];
60
+ for (const line of stdout.split(/\r?\n/)) {
61
+ const trimmed = line.trim();
62
+ if (!trimmed || trimmed.startsWith("Node") || trimmed.startsWith(",AdapterRAM"))
63
+ continue;
64
+ const parts = trimmed.split(",");
65
+ if (parts.length < 3)
66
+ continue;
67
+ const vramBytes = parseInt(parts[1] ?? "0", 10) || 0;
68
+ const name = (parts[2] ?? "").trim();
69
+ if (name && name !== "Name")
70
+ gpus.push({ name, vramBytes });
71
+ }
72
+ return gpus;
73
+ }
74
+ catch {
75
+ return [];
76
+ }
77
+ }
78
+ function formatVram(bytes) {
79
+ if (!bytes || bytes < 1024 * 1024)
80
+ return "Unknown";
81
+ const gb = bytes / (1024 * 1024 * 1024);
82
+ return gb >= 1 ? `${gb.toFixed(1)} GB` : `${Math.round(bytes / (1024 * 1024))} MB`;
83
+ }
84
+ function recommendedModel(vramBytes) {
85
+ const gb = vramBytes / (1024 * 1024 * 1024);
86
+ if (gb >= 6)
87
+ return "large-v3 (ggml-large-v3.bin) — fits comfortably in your VRAM";
88
+ if (gb >= 4)
89
+ return "medium.en (ggml-medium.en.bin) — good fit for your VRAM";
90
+ if (gb >= 2)
91
+ return "small.en (ggml-small.en.bin) — safe choice for your VRAM";
92
+ return "base.en (ggml-base.en.bin) — recommended for limited VRAM";
93
+ }
94
+ function hasVulkanDll() {
95
+ const whisperDir = dirname(WHISPER_CLI_PATH);
96
+ return existsSync(join(whisperDir, "ggml-vulkan.dll"));
97
+ }
56
98
  function needsConversion(filePath) {
57
99
  return !NATIVE_EXTENSIONS.includes(extname(filePath).toLowerCase());
58
100
  }
@@ -144,7 +186,7 @@ function getFiles(dir, recursive) {
144
186
  // ---------------------------------------------------------------------------
145
187
  // MCP Server
146
188
  // ---------------------------------------------------------------------------
147
- const server = new Server({ name: "whisper-windows-mcp", version: "1.4.0" }, { capabilities: { tools: {} } });
189
+ const server = new Server({ name: "whisper-windows-mcp", version: "1.5.0" }, { capabilities: { tools: {} } });
148
190
  server.setRequestHandler(ListToolsRequestSchema, async () => ({
149
191
  tools: [
150
192
  {
@@ -216,6 +258,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
216
258
  description: "Verify whisper-cli.exe, model, and FFmpeg are all available. Run this first if anything fails.",
217
259
  inputSchema: { type: "object", properties: {} },
218
260
  },
261
+ {
262
+ name: "check_system",
263
+ description: "Detect GPU hardware and verify Vulkan acceleration is available. " +
264
+ "Reports GPU name, VRAM, whether the Vulkan binary is installed, " +
265
+ "and recommends the best Whisper model for your hardware. " +
266
+ "Run this if you want to confirm GPU acceleration is working or diagnose why it isn't.",
267
+ inputSchema: { type: "object", properties: {} },
268
+ },
219
269
  ],
220
270
  }));
221
271
  server.setRequestHandler(CallToolRequestSchema, async (request) => {
@@ -247,6 +297,41 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
247
297
  };
248
298
  }
249
299
  // -------------------------------------------------------------------------
300
+ // check_system
301
+ // -------------------------------------------------------------------------
302
+ if (name === "check_system") {
303
+ const vulkan = hasVulkanDll();
304
+ const gpus = await detectGpus();
305
+ let gpuLines = "";
306
+ if (gpus.length === 0) {
307
+ gpuLines = "⚠️ No GPU detected via wmic — this may indicate a driver issue.\n";
308
+ }
309
+ else {
310
+ for (const gpu of gpus) {
311
+ const vramStr = formatVram(gpu.vramBytes);
312
+ gpuLines += `🖥️ GPU: ${gpu.name}\n`;
313
+ gpuLines += `💾 VRAM: ${vramStr} (reported by Windows — may be half of actual on AMD cards)\n`;
314
+ if (gpu.vramBytes > 0) {
315
+ gpuLines += `📦 Recommended model: ${recommendedModel(gpu.vramBytes)}\n`;
316
+ }
317
+ gpuLines += "\n";
318
+ }
319
+ }
320
+ const vulkanLine = vulkan
321
+ ? `✅ Vulkan binary: ggml-vulkan.dll found — GPU acceleration is active`
322
+ : `❌ Vulkan binary: ggml-vulkan.dll NOT found — whisper is running CPU-only\n\n` +
323
+ ` To enable GPU acceleration:\n` +
324
+ ` Download whisper-vulkan-win-x64.zip from:\n` +
325
+ ` https://github.com/eviscerations/whisper-windows-mcp/releases\n` +
326
+ ` Extract to: ${dirname(WHISPER_CLI_PATH)}`;
327
+ return {
328
+ content: [{
329
+ type: "text",
330
+ text: `System check\n${"─".repeat(40)}\n\n${gpuLines}${vulkanLine}`,
331
+ }],
332
+ };
333
+ }
334
+ // -------------------------------------------------------------------------
250
335
  // transcribe_audio
251
336
  // -------------------------------------------------------------------------
252
337
  if (name === "transcribe_audio") {
@@ -391,7 +476,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
391
476
  async function main() {
392
477
  const transport = new StdioServerTransport();
393
478
  await server.connect(transport);
394
- console.error(`whisper-windows-mcp v1.4.0 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS}`);
479
+ console.error(`whisper-windows-mcp v1.5.0 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS}`);
395
480
  }
396
481
  main().catch((err) => {
397
482
  console.error("Fatal error:", err);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "whisper-windows-mcp",
3
- "version": "1.4.2",
3
+ "version": "1.5.1",
4
4
  "description": "Windows-native MCP server for local audio transcription using whisper.cpp with Vulkan GPU acceleration",
5
5
  "main": "dist/index.js",
6
6
  "bin": {
@@ -1,193 +0,0 @@
1
- # whisper-windows-mcp — Roadmap
2
-
3
- Current version is a functional early release. It works, but it has known limitations that make it unsuitable for production use with long files or batch workflows. This document tracks what needs to be fixed and what's planned.
4
-
5
- ---
6
-
7
- ## Known Issues (Current Version)
8
-
9
- ### 1. No Process Lock
10
- Multiple `whisper-cli.exe` instances can be spawned concurrently if Claude retries a timed-out request. This maxes CPU and degrades or breaks both transcriptions. **This is the most critical bug.**
11
-
12
- ### 2. No Progress Visibility
13
- The user has no indicator that transcription is happening or how far along it is, short of watching Task Manager or listening to their CPU fan. This is not acceptable UX for a public tool.
14
-
15
- ### 3. 4-Minute Claude MCP Timeout
16
- The Claude web client cuts MCP connections after ~4 minutes. Whisper continues running in the background after the timeout fires, but Claude can't confirm completion. For files longer than ~2-3 minutes of audio on CPU, this timeout is frequently hit.
17
-
18
- ### 4. No GPU Autodetection
19
- The bundled `whisper-cli.exe` binary is CPU-only. AMD GPUs are not detected or used. NVIDIA GPUs may work depending on the binary source, but there's no verification at startup.
20
-
21
- ### 5. High CPU Load
22
- Sustained 50-55% CPU utilization during transcription over multi-hour sessions puts significant strain on hardware. GPU acceleration is the primary fix, but binary selection matters too.
23
-
24
- ### 6. Background Batch Non-Functional
25
- The background batch mode (intended to open a CMD window with progress) does not work as designed and was cut from the current release.
26
-
27
- ### 7. No File Pre-Analysis
28
- No way to know file duration or size before processing starts. Can't sort a queue by length or estimate how long a job will take.
29
-
30
- ---
31
-
32
- ## Roadmap
33
-
34
- ### Priority 1 — GPU Acceleration
35
-
36
- **Problem:** The current `whisper-cli.exe` binary is CPU-only.
37
-
38
- **Solution options (in order of recommendation):**
39
-
40
- - **Vulkan backend** — Compile whisper.cpp with Vulkan support. Works with AMD, NVIDIA, and Intel GPUs on Windows without vendor-specific SDKs. Best cross-compatibility.
41
- - **ROCm backend** — AMD-native but Windows ROCm support is limited and complex to install.
42
- - **Switch to faster-whisper** — Python-based implementation with GPU support via CuBLAS/ROCm. Often faster than whisper.cpp for equivalent accuracy. Requires Python environment.
43
-
44
- **GPU autodetection:** On startup, check for available GPU via `dxdiag` or `wmic path win32_VideoController` and report to the user. If a supported GPU is found but the binary doesn't use it, surface a warning with instructions.
45
-
46
- ---
47
-
48
- ### Priority 2 — Process Lock
49
-
50
- **Problem:** Retrying a timed-out transcription spawns a second `whisper-cli.exe` while the first is still running.
51
-
52
- **Solution:** Before spawning any process, check for existing instances:
53
-
54
- ```batch
55
- tasklist /FI "IMAGENAME eq whisper-cli.exe" /NH
56
- ```
57
-
58
- If found, return an error: `"Transcription already in progress. Wait for the current job to complete before starting another."` Do not spawn. Period.
59
-
60
- This is a small code change with major impact.
61
-
62
- ---
63
-
64
- ### Priority 3 — File Pre-Analysis Tool
65
-
66
- **Problem:** No duration/size info before processing. Can't sort intelligently or estimate time.
67
-
68
- **Solution:** New tool `analyze_media` using FFprobe (already bundled with FFmpeg):
69
-
70
- ```
71
- ffprobe -v quiet -print_format json -show_format -show_streams <file>
72
- ```
73
-
74
- Returns: duration in seconds, file size, codec, bitrate.
75
-
76
- For a folder scan, returns a sorted table:
77
-
78
- ```
79
- filename | duration | size | est. time (CPU) | est. time (GPU)
80
- 2026-01-11 10-11-42.mp4 | 0:35 | 42 MB | ~1 min | ~10 sec
81
- 2026-03-16 06-56-43.mp4 | 1:17 | 98 MB | ~3 min | ~20 sec
82
- ...
83
- ```
84
-
85
- Throughput estimates are calibrated from actual completed jobs (logged internally) and refined over time.
86
-
87
- ---
88
-
89
- ### Priority 4 — Progress Visibility
90
-
91
- **Problem:** No way to know how far along transcription is without Task Manager.
92
-
93
- **Solution:** `whisper-cli.exe` outputs segment timestamps to stderr as it processes (e.g. `[00:01:30 --> 00:01:35]`). The MCP should:
94
-
95
- 1. Pipe stderr from the child process to a log file during processing
96
- 2. Expose a `check_progress` tool that reads the log and returns:
97
- - Last completed timestamp
98
- - Percentage complete (last timestamp / total duration)
99
- - Estimated time remaining
100
- - Whether the process is still running (check PID)
101
-
102
- No changes needed to `whisper-cli` itself — only changes to how the MCP monitors it.
103
-
104
- ---
105
-
106
- ### Priority 5 — Timeout Workaround (Detached Process Architecture)
107
-
108
- **Problem:** The 4-minute Claude MCP timeout kills the connection before long transcriptions finish.
109
-
110
- **Solution:** Rearchitect transcription to use fully detached background processes instead of blocking child processes.
111
-
112
- Flow:
113
- 1. `transcribe_audio` called → spawn whisper as **detached process** → write `job.json` (PID, source file, start time, expected duration, output path, log path) → **return immediately** with job ID
114
- 2. `check_progress` called → read `job.json` → check if PID is still alive → read last N lines of log → return status + percentage
115
- 3. When `check_progress` returns "complete" → read and return transcript
116
-
117
- The transcribe call **never blocks**. It always returns in under a second. The timeout problem disappears entirely.
118
-
119
- This is the correct architecture for any long-running MCP tool.
120
-
121
- ---
122
-
123
- ### Priority 6 — Sequential Batch with Validation
124
-
125
- **Problem:** Background batch mode is non-functional. No post-processing validation.
126
-
127
- **Solution:** Rebuild batch mode using the detached process architecture (Priority 5):
128
-
129
- 1. `transcribe_batch` → runs `analyze_media` on folder → sorts by duration ascending → processes one file at a time using detached process
130
- 2. After each file completes, validate: .txt exists + is non-empty + line count is proportional to duration (blank = likely failed)
131
- 3. Flag any suspect outputs for re-run
132
- 4. Write progress to `batch_progress.log`
133
- 5. `check_batch_progress` returns: files done, files remaining, current file, overall ETA, any failed files
134
-
135
- ---
136
-
137
- ### Priority 7 — Multi-Language Support and Translation
138
-
139
- `whisper-cli` already supports `--language` and `--translate` natively. Expose these properly in the MCP:
140
-
141
- - `language` parameter: auto-detect (default) or specify (e.g. `ja`, `es`, `de`)
142
- - `translate_to_english`: boolean flag — uses whisper's built-in translation model
143
- - For dual output (native + translated): two whisper passes, two output files
144
-
145
- **Example use case:** Japanese DVD rip → Japanese SRT + English SRT simultaneously. Whisper handles abbreviated subject-drop Japanese better than literal translators because it was trained on natural speech.
146
-
147
- ---
148
-
149
- ### Priority 8 — Filename-Based References Throughout
150
-
151
- **Problem:** When Claude references transcripts by positional index ("file 2", "file 11"), the user has no frame of reference. The index means nothing outside the tool call context.
152
-
153
- **Solution:** All tool outputs — status messages, batch listings, progress reports, analysis results — must reference the **full source filename** at all times. Never use positional indices as the primary identifier.
154
-
155
- Bad: `"File 11 is complete."`
156
- Good: `"2025-09-01 03-15-13.mp4 — complete. Transcript saved to 2025-09-01 03-15-13.txt"`
157
-
158
- When Claude ingests transcripts for analysis, each content block should be tagged with its source filename so quotes can always be traced back to their origin without ambiguity.
159
-
160
- ---
161
-
162
- ### Priority 9 — HWID / System Diagnostics Tool
163
-
164
- New tool: `check_system`
165
-
166
- Returns:
167
- - GPU vendor and model
168
- - VRAM available
169
- - Whether a GPU-accelerated whisper binary is available and configured
170
- - Recommended whisper model size for available hardware
171
- - Estimated throughput (tokens/sec) based on hardware profile
172
-
173
- This makes setup easier for new users and helps diagnose configuration problems without needing to open Task Manager or Device Manager.
174
-
175
- ---
176
-
177
- ## Design Principles
178
-
179
- **Minimize Claude API usage.** Every MCP tool call consumes from the user's usage limit. Free-tier users have a hard cap. The entire transcription workflow — scan, analyze, queue, run, validate — should require fewer than 20 Claude interactions for a 60-file batch.
180
-
181
- **One whisper instance at all times.** Never spawn a second process while one is running. Enforce this unconditionally.
182
-
183
- **Local-first, private by default.** No audio leaves the machine. No cloud APIs required. This is a feature, not a limitation.
184
-
185
- **Works for free-tier users.** Courtroom transcription, documentary research, foreign film subtitling — these are real use cases for people who can't afford cloud transcription services. The tool should serve them.
186
-
187
- ---
188
-
189
- ## Contributing
190
-
191
- If you've worked out GPU acceleration for AMD (ROCm or Vulkan) or NVIDIA on Windows, please open an issue or PR — it's the most wanted feature and the implementation details for Windows are non-trivial.
192
-
193
- Pull requests welcome for any of the above priorities. Check existing issues before starting work.