whisper-windows-mcp 1.4.0 → 1.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/npmignore +7 -0
  2. package/package.json +38 -38
  3. package/src/index.ts.bak +0 -446
package/npmignore ADDED
@@ -0,0 +1,7 @@
1
+ src/
2
+ *.bak
3
+ FUNDING.yml
4
+ ROADMAP.md
5
+ TROUBLESHOOTING.md
6
+ .gitignore
7
+ tsconfig.json
package/package.json CHANGED
@@ -1,38 +1,38 @@
1
- {
2
- "name": "whisper-windows-mcp",
3
- "version": "1.4.0",
4
- "description": "Windows-native MCP server for local audio transcription using whisper.cpp with Vulkan GPU acceleration",
5
- "main": "dist/index.js",
6
- "bin": {
7
- "whisper-windows-mcp": "dist/index.js"
8
- },
9
- "scripts": {
10
- "build": "tsc",
11
- "start": "node dist/index.js"
12
- },
13
- "keywords": [
14
- "mcp",
15
- "whisper",
16
- "windows",
17
- "transcription",
18
- "speech-to-text",
19
- "audio",
20
- "claude",
21
- "model-context-protocol",
22
- "whisper-cpp",
23
- "vulkan",
24
- "gpu"
25
- ],
26
- "author": "eviscerations",
27
- "license": "MIT",
28
- "dependencies": {
29
- "@modelcontextprotocol/sdk": "^1.0.0"
30
- },
31
- "devDependencies": {
32
- "typescript": "^5.0.0",
33
- "@types/node": "^20.0.0"
34
- },
35
- "engines": {
36
- "node": ">=18"
37
- }
38
- }
1
+ {
2
+ "name": "whisper-windows-mcp",
3
+ "version": "1.4.1",
4
+ "description": "Windows-native MCP server for local audio transcription using whisper.cpp with Vulkan GPU acceleration",
5
+ "main": "dist/index.js",
6
+ "bin": {
7
+ "whisper-windows-mcp": "dist/index.js"
8
+ },
9
+ "scripts": {
10
+ "build": "tsc",
11
+ "start": "node dist/index.js"
12
+ },
13
+ "keywords": [
14
+ "mcp",
15
+ "whisper",
16
+ "windows",
17
+ "transcription",
18
+ "speech-to-text",
19
+ "audio",
20
+ "claude",
21
+ "model-context-protocol",
22
+ "whisper-cpp",
23
+ "vulkan",
24
+ "gpu"
25
+ ],
26
+ "author": "eviscerations",
27
+ "license": "MIT",
28
+ "dependencies": {
29
+ "@modelcontextprotocol/sdk": "^1.0.0"
30
+ },
31
+ "devDependencies": {
32
+ "typescript": "^5.0.0",
33
+ "@types/node": "^20.0.0"
34
+ },
35
+ "engines": {
36
+ "node": ">=18"
37
+ }
38
+ }
package/src/index.ts.bak DELETED
@@ -1,446 +0,0 @@
1
- #!/usr/bin/env node
2
- /**
3
- * whisper-windows-mcp
4
- * A Windows-native MCP server for local audio transcription using whisper.cpp
5
- * https://github.com/eviscerations/whisper-windows-mcp
6
- */
7
-
8
- import { Server } from "@modelcontextprotocol/sdk/server/index.js";
9
- import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
10
- import {
11
- CallToolRequestSchema,
12
- ListToolsRequestSchema,
13
- } from "@modelcontextprotocol/sdk/types.js";
14
- import { execFile } from "child_process";
15
- import {
16
- existsSync, unlinkSync, readdirSync,
17
- writeFileSync, readFileSync
18
- } from "fs";
19
- import { cpus, tmpdir } from "os";
20
- import { join, extname, basename } from "path";
21
- import { promisify } from "util";
22
-
23
- const execFileAsync = promisify(execFile);
24
-
25
- // ---------------------------------------------------------------------------
26
- // Configuration
27
- // ---------------------------------------------------------------------------
28
- const WHISPER_CLI_PATH =
29
- process.env.WHISPER_CLI_PATH ?? "C:\\whisper\\Release\\whisper-cli.exe";
30
- const WHISPER_MODEL =
31
- process.env.WHISPER_MODEL ?? "C:\\whisper\\models\\ggml-base.en.bin";
32
- const FFMPEG_PATH =
33
- process.env.FFMPEG_PATH ?? "ffmpeg";
34
-
35
- const SYSTEM_THREADS = cpus().length;
36
- const DEFAULT_THREADS = Math.max(2, Math.floor(SYSTEM_THREADS / 2));
37
- const WHISPER_THREADS = parseInt(process.env.WHISPER_THREADS ?? String(DEFAULT_THREADS), 10);
38
-
39
- const SUPPORTED_EXTENSIONS = [
40
- ".mp3", ".wav",
41
- ".mp4", ".mkv", ".avi", ".mov", ".webm", ".flv", ".wmv", ".m4v",
42
- ".m4a", ".ogg", ".flac",
43
- ];
44
- const NATIVE_EXTENSIONS = [".mp3", ".wav"];
45
-
46
- // ---------------------------------------------------------------------------
47
- // Helpers
48
- // ---------------------------------------------------------------------------
49
- function validatePaths(): string | null {
50
- if (!existsSync(WHISPER_CLI_PATH))
51
- return `whisper-cli.exe not found at: ${WHISPER_CLI_PATH}\nCheck WHISPER_CLI_PATH in claude_desktop_config.json`;
52
- if (!existsSync(WHISPER_MODEL))
53
- return `Whisper model not found at: ${WHISPER_MODEL}\nCheck WHISPER_MODEL in claude_desktop_config.json`;
54
- return null;
55
- }
56
-
57
- /**
58
- * Check whether a whisper-cli.exe process is already running.
59
- * Uses tasklist /FI which is available on all Windows versions.
60
- * Returns true if found, false if not (or if tasklist itself fails).
61
- */
62
- async function isWhisperRunning(): Promise<boolean> {
63
- try {
64
- const { stdout } = await execFileAsync(
65
- "tasklist",
66
- ["/FI", "IMAGENAME eq whisper-cli.exe", "/NH"],
67
- { windowsHide: true }
68
- );
69
- return stdout.toLowerCase().includes("whisper-cli.exe");
70
- } catch {
71
- // If tasklist fails for any reason, assume safe to proceed
72
- return false;
73
- }
74
- }
75
-
76
- function needsConversion(filePath: string): boolean {
77
- return !NATIVE_EXTENSIONS.includes(extname(filePath).toLowerCase());
78
- }
79
-
80
- function isSupportedFile(filePath: string): boolean {
81
- return SUPPORTED_EXTENSIONS.includes(extname(filePath).toLowerCase());
82
- }
83
-
84
- async function convertToWav(inputPath: string): Promise<string> {
85
- const tmpFile = join(tmpdir(), `whisper_tmp_${Date.now()}.wav`);
86
- await execFileAsync(FFMPEG_PATH, [
87
- "-y", "-i", inputPath,
88
- "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le", tmpFile,
89
- ], { windowsHide: true });
90
- return tmpFile;
91
- }
92
-
93
- type OutputFormat = "text" | "timestamps" | "json" | "srt";
94
-
95
- function buildArgs(
96
- filePath: string, model: string, language: string,
97
- outputFormat: OutputFormat, threads: number
98
- ): string[] {
99
- const args = ["-m", model, "-f", filePath, "-l", language, "-t", String(threads)];
100
- if (outputFormat === "srt") {
101
- args.push("-osrt", "-of", filePath.replace(/\.[^.]+$/, ""));
102
- } else if (outputFormat === "json") {
103
- args.push("-oj");
104
- } else if (outputFormat === "text") {
105
- args.push("--no-timestamps");
106
- }
107
- return args;
108
- }
109
-
110
- async function transcribeSingle(
111
- filePath: string, model: string, language: string,
112
- outputFormat: OutputFormat, threads: number, saveToFile = false
113
- ): Promise<{ text: string; srtPath?: string; savedTo?: string }> {
114
-
115
- // ---- Priority 2: Process lock ----
116
- // Never spawn a second whisper-cli.exe while one is already running.
117
- if (await isWhisperRunning()) {
118
- throw new Error(
119
- "Transcription already in progress.\n\n" +
120
- "whisper-cli.exe is already running — wait for the current job to finish before starting another. " +
121
- "If you believe this is wrong (e.g. a previous job crashed and left a stale process), " +
122
- "open Task Manager, find whisper-cli.exe under Details, and end the task."
123
- );
124
- }
125
-
126
- let transcribeFrom = filePath;
127
- let tmpFile: string | null = null;
128
-
129
- if (needsConversion(filePath)) {
130
- tmpFile = await convertToWav(filePath);
131
- transcribeFrom = tmpFile;
132
- }
133
-
134
- try {
135
- const cliArgs = buildArgs(transcribeFrom, model, language, outputFormat, threads);
136
- const { stdout, stderr } = await execFileAsync(WHISPER_CLI_PATH, cliArgs, {
137
- maxBuffer: 100 * 1024 * 1024,
138
- windowsHide: true,
139
- });
140
-
141
- const output = (stdout || stderr || "").trim();
142
-
143
- if (outputFormat === "srt") {
144
- const tmpSrt = transcribeFrom.replace(/\.[^.]+$/, ".srt");
145
- const destSrt = filePath.replace(/\.[^.]+$/, ".srt");
146
- if (tmpFile && existsSync(tmpSrt)) {
147
- writeFileSync(destSrt, readFileSync(tmpSrt, "utf8"));
148
- try { unlinkSync(tmpSrt); } catch { }
149
- }
150
- return { text: output, srtPath: destSrt };
151
- }
152
-
153
- if (saveToFile) {
154
- const txtPath = filePath.replace(/\.[^.]+$/, ".txt");
155
- writeFileSync(txtPath, output, "utf8");
156
- return { text: output, savedTo: txtPath };
157
- }
158
-
159
- return { text: output };
160
- } finally {
161
- if (tmpFile && existsSync(tmpFile)) try { unlinkSync(tmpFile); } catch { }
162
- }
163
- }
164
-
165
- function getFiles(dir: string, recursive: boolean): string[] {
166
- const entries = readdirSync(dir, { withFileTypes: true });
167
- const files: string[] = [];
168
- for (const entry of entries) {
169
- const fullPath = join(dir, entry.name);
170
- if (entry.isDirectory() && recursive) files.push(...getFiles(fullPath, true));
171
- else if (entry.isFile() && isSupportedFile(entry.name)) files.push(fullPath);
172
- }
173
- return files;
174
- }
175
-
176
- // ---------------------------------------------------------------------------
177
- // MCP Server
178
- // ---------------------------------------------------------------------------
179
- const server = new Server(
180
- { name: "whisper-windows-mcp", version: "1.3.1" },
181
- { capabilities: { tools: {} } }
182
- );
183
-
184
- server.setRequestHandler(ListToolsRequestSchema, async () => ({
185
- tools: [
186
- {
187
- name: "transcribe_audio",
188
- description:
189
- "Transcribe a single audio or video file using whisper.cpp on Windows. " +
190
- "Natively supports mp3 and wav. Automatically converts mp4, mkv, avi, mov, " +
191
- "webm, m4a, flac, ogg etc. via FFmpeg — no manual conversion needed. " +
192
- "Can output plain text, timestamps, JSON, or SRT subtitle files. " +
193
- "WARNING: files over 30 minutes cannot be cancelled cleanly from Claude — " +
194
- "use Task Manager to kill the process if needed.",
195
- inputSchema: {
196
- type: "object",
197
- properties: {
198
- file_path: { type: "string", description: "Absolute Windows path, e.g. C:\\Users\\You\\Downloads\\recording.mp4" },
199
- model: { type: "string", description: "Override model path. Leave blank to use WHISPER_MODEL." },
200
- language: { type: "string", description: "Language code, e.g. en, es, fr. Defaults to en.", default: "en" },
201
- output_format: {
202
- type: "string", enum: ["text", "timestamps", "json", "srt"],
203
- description: "text = plain (default), timestamps = with time codes, json = structured, srt = subtitle file saved next to source.",
204
- default: "text",
205
- },
206
- threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
207
- save_to_file: { type: "boolean", description: "Save transcript as .txt next to the source file.", default: false },
208
- },
209
- required: ["file_path"],
210
- },
211
- },
212
- {
213
- name: "transcribe_batch",
214
- description:
215
- "Transcribe multiple audio/video files in a folder interactively, one file at a time. " +
216
- "Shows a preview of each transcript and waits for confirmation before continuing. " +
217
- "Saves each transcript as a .txt file next to its source. " +
218
- "Files already transcribed (with matching .txt) are shown as done and skipped. " +
219
- "Supported formats: mp3, wav, mp4, mkv, avi, mov, webm, m4a, flac, ogg. " +
220
- "NOTE: For large unattended batch jobs, use whisper-cli.exe directly from the command line " +
221
- "— see TROUBLESHOOTING.md for the command syntax.",
222
- inputSchema: {
223
- type: "object",
224
- properties: {
225
- folder_path: { type: "string", description: "Absolute Windows path to the folder." },
226
- file_index: {
227
- type: "number",
228
- description: "Which file to process (1-based). Omit to list files first.",
229
- },
230
- language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
231
- threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
232
- recursive: { type: "boolean", description: "Include subfolders. Defaults to false.", default: false },
233
- },
234
- required: ["folder_path"],
235
- },
236
- },
237
- {
238
- name: "generate_subtitles",
239
- description:
240
- "Generate an SRT subtitle file for an audio or video file. " +
241
- "Saved next to the source file. Load in VLC via Subtitle → Add Subtitle File. " +
242
- "Supports all the same formats as transcribe_audio.",
243
- inputSchema: {
244
- type: "object",
245
- properties: {
246
- file_path: { type: "string", description: "Absolute Windows path to the file." },
247
- language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
248
- threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
249
- },
250
- required: ["file_path"],
251
- },
252
- },
253
- {
254
- name: "check_config",
255
- description: "Verify whisper-cli.exe, model, and FFmpeg are all available. Run this first if anything fails.",
256
- inputSchema: { type: "object", properties: {} },
257
- },
258
- ],
259
- }));
260
-
261
- server.setRequestHandler(CallToolRequestSchema, async (request) => {
262
- const { name, arguments: args } = request.params;
263
-
264
- // -------------------------------------------------------------------------
265
- // check_config
266
- // -------------------------------------------------------------------------
267
- if (name === "check_config") {
268
- const error = validatePaths();
269
- if (error) return { content: [{ type: "text", text: `❌ Configuration error:\n\n${error}` }], isError: true };
270
-
271
- let ffmpegStatus = "✅ Found";
272
- try { await execFileAsync(FFMPEG_PATH, ["-version"], { windowsHide: true }); }
273
- catch { ffmpegStatus = "⚠️ Not found — video/non-MP3 formats require FFmpeg in PATH"; }
274
-
275
- return {
276
- content: [{
277
- type: "text",
278
- text:
279
- `✅ Configuration looks good!\n\n` +
280
- `whisper-cli: ${WHISPER_CLI_PATH}\n` +
281
- `Model: ${WHISPER_MODEL}\n` +
282
- `Threads: ${WHISPER_THREADS} of ${SYSTEM_THREADS} logical cores\n` +
283
- `FFmpeg: ${ffmpegStatus}\n\n` +
284
- `Optional env vars: WHISPER_THREADS, FFMPEG_PATH`,
285
- }],
286
- };
287
- }
288
-
289
- // -------------------------------------------------------------------------
290
- // transcribe_audio
291
- // -------------------------------------------------------------------------
292
- if (name === "transcribe_audio") {
293
- const filePath = args?.file_path as string;
294
- const model = (args?.model as string) || WHISPER_MODEL;
295
- const language = (args?.language as string) || "en";
296
- const outputFormat = ((args?.output_format as string) || "text") as OutputFormat;
297
- const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round((args?.threads as number) || WHISPER_THREADS)));
298
- const saveToFile = (args?.save_to_file as boolean) || false;
299
-
300
- if (!filePath) return { content: [{ type: "text", text: "file_path is required." }], isError: true };
301
- if (!existsSync(filePath)) return { content: [{ type: "text", text: `File not found: ${filePath}` }], isError: true };
302
- const configError = validatePaths();
303
- if (configError) return { content: [{ type: "text", text: configError }], isError: true };
304
-
305
- try {
306
- const result = await transcribeSingle(filePath, model, language, outputFormat, threads, saveToFile);
307
- let response = result.text;
308
- if (result.savedTo) response += `\n\n[Transcript saved to: ${result.savedTo}]`;
309
- if (result.srtPath) response += `\n\n[SRT subtitle file saved to: ${result.srtPath}]`;
310
- return { content: [{ type: "text", text: response }] };
311
- } catch (err: any) {
312
- return { content: [{ type: "text", text: `Transcription failed:\n\n${err?.stderr || err?.message || String(err)}` }], isError: true };
313
- }
314
- }
315
-
316
- // -------------------------------------------------------------------------
317
- // generate_subtitles
318
- // -------------------------------------------------------------------------
319
- if (name === "generate_subtitles") {
320
- const filePath = args?.file_path as string;
321
- const language = (args?.language as string) || "en";
322
- const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round((args?.threads as number) || WHISPER_THREADS)));
323
-
324
- if (!filePath) return { content: [{ type: "text", text: "file_path is required." }], isError: true };
325
- if (!existsSync(filePath)) return { content: [{ type: "text", text: `File not found: ${filePath}` }], isError: true };
326
- const configError = validatePaths();
327
- if (configError) return { content: [{ type: "text", text: configError }], isError: true };
328
-
329
- try {
330
- const result = await transcribeSingle(filePath, WHISPER_MODEL, language, "srt", threads, false);
331
- return {
332
- content: [{
333
- type: "text",
334
- text:
335
- `✅ Subtitle file generated!\n\n` +
336
- `Saved to: ${result.srtPath}\n\n` +
337
- `To use in VLC: Subtitle → Add Subtitle File → select the .srt file.\n` +
338
- `Works in any video player that supports external subtitles.`,
339
- }],
340
- };
341
- } catch (err: any) {
342
- return { content: [{ type: "text", text: `Subtitle generation failed:\n\n${err?.stderr || err?.message || String(err)}` }], isError: true };
343
- }
344
- }
345
-
346
- // -------------------------------------------------------------------------
347
- // transcribe_batch (interactive only)
348
- // -------------------------------------------------------------------------
349
- if (name === "transcribe_batch") {
350
- const folderPath = args?.folder_path as string;
351
- const language = (args?.language as string) || "en";
352
- const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round((args?.threads as number) || WHISPER_THREADS)));
353
- const recursive = (args?.recursive as boolean) || false;
354
- const fileIndex = args?.file_index as number | undefined;
355
-
356
- if (!folderPath) return { content: [{ type: "text", text: "folder_path is required." }], isError: true };
357
- if (!existsSync(folderPath)) return { content: [{ type: "text", text: `Folder not found: ${folderPath}` }], isError: true };
358
- const configError = validatePaths();
359
- if (configError) return { content: [{ type: "text", text: configError }], isError: true };
360
-
361
- const files = getFiles(folderPath, recursive);
362
-
363
- if (files.length === 0) {
364
- return {
365
- content: [{
366
- type: "text",
367
- text: `No supported files found in: ${folderPath}\nSupported formats: ${SUPPORTED_EXTENSIONS.join(", ")}`,
368
- }],
369
- };
370
- }
371
-
372
- // No file_index: return file list
373
- if (fileIndex === undefined) {
374
- return {
375
- content: [{
376
- type: "text",
377
- text:
378
- `Found ${files.length} file(s) in: ${folderPath}\n\n` +
379
- files.map((f, i) => {
380
- const txtPath = f.replace(/\.[^.]+$/, ".txt");
381
- const done = existsSync(txtPath) ? " ✅ already done" : "";
382
- return ` ${i + 1}. ${basename(f)}${done}`;
383
- }).join("\n") +
384
- `\n\nTo start, say "transcribe file 1" (or any number). I'll process one file at a time and wait for your go-ahead before continuing.\n` +
385
- `\nFor large unattended batches, see the command line approach in TROUBLESHOOTING.md.`,
386
- }],
387
- };
388
- }
389
-
390
- // Process the requested file
391
- const idx = fileIndex - 1;
392
- if (idx < 0 || idx >= files.length) {
393
- return { content: [{ type: "text", text: `Invalid file number. Choose between 1 and ${files.length}.` }], isError: true };
394
- }
395
-
396
- const filePath = files[idx];
397
- const fileName = basename(filePath);
398
- const txtPath = filePath.replace(/\.[^.]+$/, ".txt");
399
-
400
- try {
401
- const result = await transcribeSingle(filePath, WHISPER_MODEL, language, "text", threads, true);
402
- const remaining = files.length - fileIndex;
403
- const nextMsg = remaining > 0
404
- ? `\n\n${remaining} file(s) remaining. Say "continue" or "transcribe file ${fileIndex + 1}" to proceed, or "stop" to finish.`
405
- : `\n\n✅ That was the last file. Batch complete!`;
406
-
407
- return {
408
- content: [{
409
- type: "text",
410
- text:
411
- `[${fileIndex}/${files.length}] ✅ ${fileName}\n\n` +
412
- `Saved to: ${txtPath}\n\n` +
413
- `Preview:\n${result.text.slice(0, 500)}${result.text.length > 500 ? "..." : ""}` +
414
- nextMsg,
415
- }],
416
- };
417
- } catch (err: any) {
418
- return {
419
- content: [{
420
- type: "text",
421
- text:
422
- `[${fileIndex}/${files.length}] ❌ Failed: ${fileName}\n\n` +
423
- `Error: ${err?.stderr || err?.message || String(err)}\n\n` +
424
- `Say "transcribe file ${fileIndex + 1}" to skip and continue.`,
425
- }],
426
- isError: true,
427
- };
428
- }
429
- }
430
-
431
- return { content: [{ type: "text", text: `Unknown tool: ${name}` }], isError: true };
432
- });
433
-
434
- // ---------------------------------------------------------------------------
435
- // Start
436
- // ---------------------------------------------------------------------------
437
- async function main() {
438
- const transport = new StdioServerTransport();
439
- await server.connect(transport);
440
- console.error(`whisper-windows-mcp v1.3.1 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS}`);
441
- }
442
-
443
- main().catch((err) => {
444
- console.error("Fatal error:", err);
445
- process.exit(1);
446
- });