whisper-windows-mcp 2.2.2 → 2.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/.github/workflows/ci.yml +24 -0
  2. package/.github/workflows/publish.yml +24 -0
  3. package/{LICENSE-COMMERCIAL.md → COMMERCIAL-LICENSE.md} +58 -58
  4. package/LICENSE +40 -40
  5. package/PRIVACY.es.md +194 -135
  6. package/PRIVACY.id.md +194 -135
  7. package/PRIVACY.ja.md +194 -135
  8. package/PRIVACY.ko.md +194 -135
  9. package/PRIVACY.md +194 -135
  10. package/PRIVACY.pl.md +194 -135
  11. package/PRIVACY.pt-BR.md +194 -135
  12. package/PRIVACY.ro.md +194 -135
  13. package/PRIVACY.uk.md +194 -135
  14. package/PRIVACY.vi.md +194 -135
  15. package/README.es.md +80 -48
  16. package/README.id.md +83 -40
  17. package/README.ja.md +106 -72
  18. package/README.ko.md +69 -37
  19. package/README.md +82 -39
  20. package/README.pl.md +83 -40
  21. package/README.pt-BR.md +77 -45
  22. package/README.ro.md +84 -41
  23. package/README.uk.md +83 -40
  24. package/README.vi.md +73 -41
  25. package/ROADMAP.es.md +131 -45
  26. package/ROADMAP.id.md +86 -89
  27. package/ROADMAP.ja.md +93 -108
  28. package/ROADMAP.ko.md +82 -82
  29. package/ROADMAP.pl.md +125 -41
  30. package/ROADMAP.pt-BR.md +87 -87
  31. package/ROADMAP.ro.md +123 -41
  32. package/ROADMAP.uk.md +82 -90
  33. package/ROADMAP.vi.md +87 -87
  34. package/SECURITY.es.md +76 -47
  35. package/SECURITY.id.md +76 -47
  36. package/SECURITY.ja.md +76 -47
  37. package/SECURITY.ko.md +76 -47
  38. package/SECURITY.md +33 -4
  39. package/SECURITY.pl.md +76 -47
  40. package/SECURITY.pt-BR.md +76 -47
  41. package/SECURITY.ro.md +76 -47
  42. package/SECURITY.uk.md +76 -47
  43. package/SECURITY.vi.md +76 -47
  44. package/TROUBLESHOOTING.es.md +325 -323
  45. package/TROUBLESHOOTING.id.md +349 -323
  46. package/TROUBLESHOOTING.ja.md +415 -286
  47. package/TROUBLESHOOTING.ko.md +325 -323
  48. package/TROUBLESHOOTING.pl.md +371 -323
  49. package/TROUBLESHOOTING.pt-BR.md +325 -323
  50. package/TROUBLESHOOTING.ro.md +371 -323
  51. package/TROUBLESHOOTING.uk.md +385 -323
  52. package/TROUBLESHOOTING.vi.md +325 -323
  53. package/dist/index.js +743 -289
  54. package/dist/lib.d.ts +37 -0
  55. package/dist/lib.js +123 -0
  56. package/package.json +46 -45
  57. package/patch_roadmaps.py +0 -72
package/dist/index.js CHANGED
@@ -10,8 +10,10 @@ import { CallToolRequestSchema, ListToolsRequestSchema, } from "@modelcontextpro
10
10
  import { execFile, spawn } from "child_process";
11
11
  import { existsSync, unlinkSync, readdirSync, writeFileSync, readFileSync, mkdirSync, openSync, closeSync, statSync, } from "fs";
12
12
  import { cpus, tmpdir } from "os";
13
- import { join, extname, basename, dirname } from "path";
13
+ import { join, extname, basename, dirname, resolve } from "path";
14
14
  import { promisify } from "util";
15
+ import { randomUUID } from "crypto";
16
+ import { coerceNum, writeJsonAtomic, estimateWordCount, opKeyFor, isInsideDir, extractTranscriptFromLog, parseLastTimestamp, formatDuration, estimateSec, estimateTime, } from "./lib.js";
15
17
  const execFileAsync = promisify(execFile);
16
18
  // ---------------------------------------------------------------------------
17
19
  // Configuration
@@ -23,6 +25,74 @@ const FFMPEG_PATH = process.env.FFMPEG_PATH ?? "ffmpeg";
23
25
  const SYSTEM_THREADS = cpus().length;
24
26
  const DEFAULT_THREADS = Math.max(2, Math.floor(SYSTEM_THREADS / 2));
25
27
  const WHISPER_THREADS = parseInt(process.env.WHISPER_THREADS ?? String(DEFAULT_THREADS), 10);
28
+ // Optional global default GPU/Vulkan device index passed to whisper-cli as --device N.
29
+ // Lets a multi-GPU box pin a specific card without passing gpu_device on every call.
30
+ // Per-call gpu_device overrides this; unset → whisper-cli's own default (device 0).
31
+ // ⚠ This is the Vulkan ENUMERATION index (whisper-cli logs "ggml_vulkan: 0 = <name>"),
32
+ // which is NOT guaranteed to match Windows GPU0/GPU1 — read the startup log to pick correctly.
33
+ const _whisperGpuEnv = process.env.WHISPER_GPU_DEVICE;
34
+ const WHISPER_GPU_DEVICE = _whisperGpuEnv !== undefined && _whisperGpuEnv.trim() !== "" && !Number.isNaN(parseInt(_whisperGpuEnv, 10))
35
+ ? parseInt(_whisperGpuEnv, 10)
36
+ : undefined;
37
+ // Foreground transcription guard: if the estimated time (fixed model-load cost + transcribe) exceeds
38
+ // this many seconds, a blocking (foreground) run is refused and routed to background mode instead —
39
+ // avoiding a silent timeout against Claude Desktop's ~4-minute (240s) MCP tool-call ceiling.
40
+ // Default 210 leaves ~30s headroom under the wall. Configurable via WHISPER_FOREGROUND_MAX_SEC.
41
+ const FOREGROUND_MAX_SEC = parseInt(process.env.WHISPER_FOREGROUND_MAX_SEC ?? "210", 10) || 210;
42
+ /** Effective GPU device: a numeric per-call arg wins, else the WHISPER_GPU_DEVICE env default, else undefined. */
43
+ function resolveGpuDevice(arg) {
44
+ if (arg !== undefined) {
45
+ const n = Number(arg);
46
+ if (!Number.isNaN(n))
47
+ return n;
48
+ }
49
+ return WHISPER_GPU_DEVICE;
50
+ }
51
+ // Temp WAVs from BLOCKING transcriptions only — never detached-job temps (a running
52
+ // background whisper-cli still needs its WAV). Cleaned best-effort on graceful shutdown.
53
+ const activeTempFiles = new Set();
54
+ function gracefulShutdown(signal) {
55
+ let cleaned = 0;
56
+ for (const f of activeTempFiles) {
57
+ try {
58
+ if (existsSync(f)) {
59
+ unlinkSync(f);
60
+ cleaned++;
61
+ }
62
+ }
63
+ catch { /* best effort */ }
64
+ }
65
+ console.error(`whisper-windows-mcp: ${signal} — cleaned ${cleaned} blocking temp file(s), exiting.`);
66
+ process.exit(0);
67
+ }
68
+ // ---------------------------------------------------------------------------
69
+ // Privacy configuration
70
+ // ---------------------------------------------------------------------------
71
+ /**
72
+ * Global default: when true, all tool responses return metadata only
73
+ * (filename, word count, save path). No transcript text appears in any tool
74
+ * response or API call. Transcripts are still saved as local .txt files.
75
+ * Required for HIPAA, GDPR, legal, financial, and NDA-protected content.
76
+ *
77
+ * Can be overridden per-call using the privacy_mode parameter on
78
+ * transcribe_audio, transcribe_batch, start_batch, and check_progress.
79
+ * Per-call override wins in either direction — no restart required to toggle.
80
+ *
81
+ * Set as global default in claude_desktop_config.json env section:
82
+ * "WHISPER_PRIVACY_MODE": "true"
83
+ */
84
+ const WHISPER_PRIVACY_MODE = (process.env.WHISPER_PRIVACY_MODE ?? "false").toLowerCase() === "true";
85
+ /**
86
+ * When true: skips the one-time first-use consent disclosure shown before
87
+ * transcript text is sent to Claude's API. Set this once you understand the
88
+ * privacy boundary and no longer need the reminder each session.
89
+ * Has no effect when privacy mode is active — privacy mode uses its own
90
+ * per-operation gate that always fires regardless of this setting.
91
+ *
92
+ * Set in claude_desktop_config.json env section:
93
+ * "WHISPER_CONSENT_ACKNOWLEDGED": "true"
94
+ */
95
+ const WHISPER_CONSENT_ACKNOWLEDGED = (process.env.WHISPER_CONSENT_ACKNOWLEDGED ?? "false").toLowerCase() === "true";
26
96
  const SUPPORTED_EXTENSIONS = [
27
97
  ".mp3", ".wav",
28
98
  ".mp4", ".mkv", ".avi", ".mov", ".webm", ".flv", ".wmv", ".m4v",
@@ -34,8 +104,190 @@ const MAX_FILE_SIZE_MB = 10240; // 10 GB
34
104
  // Security: patterns rejected in all file_path inputs.
35
105
  const UNSAFE_PATH_RE = /(\.\.[/\\])|(^\\\\)/; // blocks .. traversal and UNC paths
36
106
  // ---------------------------------------------------------------------------
37
- // Helpers
107
+ // Privacy helpers
38
108
  // ---------------------------------------------------------------------------
109
+ /**
110
+ * Shown once per session before any transcript text is returned in non-privacy mode.
111
+ * Contains explicit instructions for Claude to surface to the user and
112
+ * halt until explicit user confirmation is received.
113
+ */
114
+ const CONSENT_DISCLOSURE = [
115
+ "⚠️ PRIVACY DISCLOSURE — EXPLICIT USER CONFIRMATION REQUIRED",
116
+ "",
117
+ "━".repeat(60),
118
+ "INSTRUCTION FOR CLAUDE: Present this disclosure to the user in full.",
119
+ "Ask: \"Do you confirm? (yes / no)\"",
120
+ "Do NOT call this tool again until the user has explicitly replied yes.",
121
+ "━".repeat(60),
122
+ "",
123
+ "Transcript text is about to be transmitted to Anthropic's API and processed",
124
+ "externally. Once sent, it leaves your local machine.",
125
+ "",
126
+ "Your audio and video files never leave your machine. Transcript text is",
127
+ "different — if returned to Claude, it is sent through the API.",
128
+ "",
129
+ "You must confirm before proceeding if this recording contains:",
130
+ " • Medical or patient information (HIPAA — 45 CFR §164)",
131
+ " • Legal proceedings or attorney-client communications (privilege)",
132
+ " • Personal data of EU/EEA residents (GDPR — Articles 13, 14, 28)",
133
+ " • Student educational records (FERPA — 20 U.S.C. §1232g)",
134
+ " • Financial or cardholder data (PCI-DSS, SOX)",
135
+ " • Confidential business information (trade secrets, NDA-protected)",
136
+ " • Any recording subject to contractual confidentiality",
137
+ "",
138
+ "For recordings that require full local privacy, enable privacy mode:",
139
+ " Per-call: pass privacy_mode=true to any transcription tool",
140
+ " Global: add to claude_desktop_config.json env:",
141
+ " \"WHISPER_PRIVACY_MODE\": \"true\"",
142
+ "",
143
+ "To suppress this confirmation permanently for non-sensitive content:",
144
+ " \"WHISPER_CONSENT_ACKNOWLEDGED\": \"true\"",
145
+ "",
146
+ "━".repeat(60),
147
+ "INSTRUCTION FOR CLAUDE: Ask the user the following question exactly,",
148
+ "then wait for their response before taking any further action:",
149
+ "",
150
+ "\"⚠️ Before I return any transcript text, I need your explicit confirmation.",
151
+ "Transcript content will be transmitted to Anthropic's API and processed",
152
+ "externally — it will leave your local machine. Do you confirm? (yes / no)\"",
153
+ "━".repeat(60),
154
+ ].join("\n");
155
+ /**
156
+ * Shown before every operation when privacy mode is active.
157
+ * Identical text every time by design — repetition is the point.
158
+ * Regulatory compliance requires informed consent before each operation.
159
+ */
160
+ const PRIVACY_MODE_DISCLOSURE = [
161
+ "🔒 PRIVACY MODE — CONFIRMATION REQUIRED BEFORE THIS OPERATION",
162
+ "",
163
+ "━".repeat(60),
164
+ "INSTRUCTION FOR CLAUDE: Present this disclosure to the user in full.",
165
+ "Ask: \"Do you confirm? (yes / no)\"",
166
+ "Do NOT call this tool again until the user has explicitly replied yes.",
167
+ "━".repeat(60),
168
+ "",
169
+ "WHISPER_PRIVACY_MODE is active for this operation.",
170
+ "",
171
+ "What will happen:",
172
+ " ✓ Audio/video will be transcribed LOCALLY on your machine",
173
+ " ✓ Transcript saved as a local file only — not returned to Claude's API",
174
+ " ✓ Raw audio and video files NEVER leave your machine",
175
+ " ✓ No transcript text will be transmitted to Anthropic under any circumstances",
176
+ "",
177
+ "What you must confirm:",
178
+ " ⚠ Audio processing begins on your local machine after you confirm",
179
+ " ⚠ This confirmation is required before every operation in privacy mode",
180
+ " ⚠ Do not disable privacy mode mid-session for regulated or sensitive material",
181
+ "",
182
+ "To disable per-call (non-sensitive content only): pass privacy_mode=false",
183
+ "To disable globally: set WHISPER_PRIVACY_MODE=false and restart Claude Desktop",
184
+ "",
185
+ "━".repeat(60),
186
+ "INSTRUCTION FOR CLAUDE: Ask the user the following question exactly,",
187
+ "then wait for their response before taking any further action:",
188
+ "",
189
+ "\"🔒 Privacy mode is active. Audio will be transcribed locally and no transcript",
190
+ "text will be sent to Anthropic's API. Confirm you want to proceed? (yes / no)\"",
191
+ "━".repeat(60),
192
+ ].join("\n");
193
+ // Per-operation privacy gate state.
194
+ // Each distinct operation (identified by a stable key over its tool name + arguments)
195
+ // arms independently: first call for that key shows the disclosure and blocks; the
196
+ // second call with the SAME key clears it and proceeds. Keying per-operation closes
197
+ // the v2.3.0 hole where a single global flag let one operation's confirmation be
198
+ // silently consumed by a different operation. Completely independent of
199
+ // sessionConsentGiven — serves different users and modes.
200
+ const privacyArmed = new Map(); // opKey -> armed-at epoch ms
201
+ // Armed disclosures expire so an abandoned confirmation can never satisfy a later
202
+ // operation, and the map can never grow without bound on a long-lived server.
203
+ const PRIVACY_GATE_TTL_MS = 10 * 60 * 1000;
204
+ // Session-scoped consent tracking — resets each time Claude Desktop restarts
205
+ // the MCP server process. Pre-set from env var so users who have set
206
+ // WHISPER_CONSENT_ACKNOWLEDGED=true skip the gate entirely.
207
+ // Has no effect when privacy mode is active (privacy mode uses its own gate).
208
+ let sessionConsentGiven = WHISPER_CONSENT_ACKNOWLEDGED;
209
+ /**
210
+ * Pre-transcription gate for privacy mode, scoped to a single operation by opKey.
211
+ * Call this when effective privacy mode is active, BEFORE any audio processing.
212
+ *
213
+ * Returns true → block this call, show PRIVACY_MODE_DISCLOSURE to user.
214
+ * Returns false → user has confirmed THIS operation, proceed.
215
+ *
216
+ * Mechanism: first call for opKey arms it (block); the second call with the same
217
+ * opKey clears it (allow). Each distinct operation is independent — confirming one
218
+ * can never satisfy another. Stale arms older than PRIVACY_GATE_TTL_MS are evicted
219
+ * on every call. Only call when effective privacy mode is active.
220
+ */
221
+ function checkPrivacyGate(opKey) {
222
+ const now = Date.now();
223
+ for (const [k, armedAt] of privacyArmed) {
224
+ if (now - armedAt > PRIVACY_GATE_TTL_MS)
225
+ privacyArmed.delete(k);
226
+ }
227
+ if (!privacyArmed.has(opKey)) {
228
+ privacyArmed.set(opKey, now);
229
+ return true; // first sight of this exact operation — block, show disclosure
230
+ }
231
+ privacyArmed.delete(opKey);
232
+ return false; // same operation re-issued — user confirmed — allow
233
+ }
234
+ /** Returns the privacy mode disclosure as a tool response. */
235
+ function privacyGateBlock() {
236
+ return PRIVACY_MODE_DISCLOSURE;
237
+ }
238
+ /**
239
+ * Determines post-transcription transcript policy for non-privacy mode.
240
+ * Only call this after confirming effective privacy mode is OFF.
241
+ *
242
+ * Returns:
243
+ * "consent_gate" — First transcript-returning call this session; show
244
+ * disclosure and withhold text. Flips sessionConsentGiven
245
+ * so subsequent calls proceed without re-prompting.
246
+ * "allow" — Consent already given; return text normally.
247
+ *
248
+ * IMPORTANT: Only call when the tool is about to return actual transcript text
249
+ * (i.e. job confirmed complete). Do NOT call for still-running jobs or error
250
+ * paths — it would consume the consent gate without returning content.
251
+ */
252
+ function transcriptPolicy() {
253
+ if (!sessionConsentGiven) {
254
+ sessionConsentGiven = true;
255
+ return "consent_gate";
256
+ }
257
+ return "allow";
258
+ }
259
+ /**
260
+ * Metadata-only response used when privacy mode is active.
261
+ * Returns file info and word count. No transcript text included.
262
+ */
263
+ function privacyModeBlock(fileName, savedPath, text) {
264
+ const words = estimateWordCount(text);
265
+ return (`✅ Transcription complete — privacy mode active.\n\n` +
266
+ `Source: ${fileName}\n` +
267
+ `Words: ~${words}\n` +
268
+ `Saved: ${savedPath}\n\n` +
269
+ `Transcript text is not transmitted to Claude's API.\n` +
270
+ `Access your transcript directly at the path above.`);
271
+ }
272
+ /**
273
+ * Consent gate response block — shown on first transcript-returning call
274
+ * in non-privacy mode. savedPath and text are optional: when called BEFORE
275
+ * transcription (blocking mode), neither is available. When called AFTER
276
+ * (background jobs via check_progress), both are present.
277
+ */
278
+ function consentGateBlock(savedPath, text) {
279
+ const lines = [CONSENT_DISCLOSURE, ""];
280
+ if (savedPath || text) {
281
+ lines.push("─".repeat(60));
282
+ if (savedPath)
283
+ lines.push(`Saved: ${savedPath}`);
284
+ if (text)
285
+ lines.push(`Words: ~${estimateWordCount(text)}`);
286
+ lines.push("");
287
+ }
288
+ lines.push("No transcript text has been returned. Reply 'yes' to confirm, then call the tool again to proceed.");
289
+ return lines.join("\n");
290
+ }
39
291
  function validatePaths() {
40
292
  if (!existsSync(WHISPER_CLI_PATH))
41
293
  return `whisper-cli.exe not found at: ${WHISPER_CLI_PATH}\nCheck WHISPER_CLI_PATH in claude_desktop_config.json`;
@@ -66,7 +318,6 @@ function validateInputPath(filePath) {
66
318
  /**
67
319
  * Check whether a whisper-cli.exe process is already running.
68
320
  * Uses tasklist /FI which is available on all Windows versions.
69
- * Returns true if found, false if not (or if tasklist itself fails).
70
321
  */
71
322
  async function isWhisperRunning() {
72
323
  try {
@@ -74,20 +325,51 @@ async function isWhisperRunning() {
74
325
  return stdout.toLowerCase().includes("whisper-cli.exe");
75
326
  }
76
327
  catch {
77
- // If tasklist fails for any reason, assume safe to proceed
78
328
  return false;
79
329
  }
80
330
  }
81
331
  // ---------------------------------------------------------------------------
82
- // Background job architecture (Priorities 4 + 5)
332
+ // Background job architecture
83
333
  // ---------------------------------------------------------------------------
84
334
  const JOBS_DIR = join(tmpdir(), "whisper-mcp-jobs");
335
+ // Mutex: prevents double-spawn when the exit handler and a concurrent
336
+ // check_batch_progress call both detect job completion simultaneously.
337
+ let batchSpawning = false;
338
+ /**
339
+ * Delete .json and .log job files older than 7 days from the jobs directory.
340
+ * Non-blocking — runs once at startup, errors are ignored.
341
+ */
342
+ function cleanupOldJobFiles() {
343
+ try {
344
+ if (!existsSync(JOBS_DIR))
345
+ return;
346
+ const cutoff = Date.now() - 7 * 24 * 60 * 60 * 1000;
347
+ const files = readdirSync(JOBS_DIR);
348
+ let cleaned = 0;
349
+ for (const file of files) {
350
+ if (!file.endsWith(".json") && !file.endsWith(".log"))
351
+ continue;
352
+ const fullPath = join(JOBS_DIR, file);
353
+ try {
354
+ if (statSync(fullPath).mtimeMs < cutoff) {
355
+ unlinkSync(fullPath);
356
+ cleaned++;
357
+ }
358
+ }
359
+ catch { /* ignore per-file errors */ }
360
+ }
361
+ if (cleaned > 0) {
362
+ console.error(`whisper-windows-mcp: cleaned ${cleaned} old job file(s) from ${JOBS_DIR}`);
363
+ }
364
+ }
365
+ catch { /* non-blocking, ignore all errors */ }
366
+ }
85
367
  function ensureJobsDir() {
86
368
  mkdirSync(JOBS_DIR, { recursive: true });
87
369
  }
88
- async function spawnDetached(filePath, model, language, threads, outputFormat = "text", extraOpts = {}) {
370
+ async function spawnDetached(filePath, model, language, threads, outputFormat = "timestamps", extraOpts = {}, onExit, privacyMode = false) {
89
371
  ensureJobsDir();
90
- const jobId = `job_${Date.now()}`;
372
+ const jobId = `job_${Date.now()}_${randomUUID().slice(0, 8)}`;
91
373
  const logPath = join(JOBS_DIR, `${jobId}.log`);
92
374
  const jobPath = join(JOBS_DIR, `${jobId}.json`);
93
375
  // Convert to WAV first if needed (fast, blocking)
@@ -98,13 +380,17 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
98
380
  isTmp = true;
99
381
  }
100
382
  // Use a clean ASCII job-ID-based output path to avoid Unicode filename issues.
101
- // After completion, readJobProgress will move the file to the correct destination.
383
+ // readJobProgress moves the file to the correct destination after completion.
102
384
  const tmpOutputBase = join(JOBS_DIR, jobId);
103
- // Determine final destination path
385
+ // Determine final destination path and file extension
104
386
  const sourceBase = filePath.replace(/\.[^.]+$/, "");
105
- const ext = outputFormat === "srt" ? ".srt" : ".txt";
106
- const outputPath = outputFormat === "srt" && language !== "en" && language !== "auto"
107
- ? `${sourceBase}.${language}.srt`
387
+ const ext = outputFormat === "srt" ? ".srt"
388
+ : outputFormat === "vtt" ? ".vtt"
389
+ : outputFormat === "lrc" ? ".lrc"
390
+ : outputFormat === "csv" ? ".csv"
391
+ : ".txt";
392
+ const outputPath = (outputFormat === "srt" || outputFormat === "vtt") && language !== "en" && language !== "auto"
393
+ ? `${sourceBase}.${language}${ext}`
108
394
  : `${sourceBase}${ext}`;
109
395
  // Build args using shared options — ensures quality flags are always applied
110
396
  // in background mode, matching blocking mode behaviour.
@@ -114,10 +400,7 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
114
400
  "-f", transcribeFrom,
115
401
  "-l", lang,
116
402
  "-t", String(threads),
117
- // Hallucination prevention — must be in background mode too.
118
- // --max-context 0 prevents conditioning on prior segment output.
119
403
  ...(extraOpts.conditionOnPrevText ? [] : ["--max-context", "0"]),
120
- // Confirmed valid flag (-nth). Suppresses silent segments from hallucinating.
121
404
  "--no-speech-thold", String(extraOpts.noSpeechThold ?? 0.6),
122
405
  ];
123
406
  if (extraOpts.temperature !== undefined)
@@ -129,7 +412,7 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
129
412
  if (extraOpts.bestOf !== undefined)
130
413
  args.push("--best-of", String(extraOpts.bestOf));
131
414
  if (extraOpts.gpuDevice !== undefined)
132
- args.push("-g", String(extraOpts.gpuDevice));
415
+ args.push("--device", String(extraOpts.gpuDevice));
133
416
  if (extraOpts.processors !== undefined && extraOpts.processors > 1)
134
417
  args.push("-p", String(extraOpts.processors));
135
418
  if (extraOpts.offsetT !== undefined)
@@ -149,13 +432,24 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
149
432
  if (extraOpts.splitOnWord)
150
433
  args.push("--split-on-word");
151
434
  }
152
- // Output format
435
+ // Output format flags
153
436
  if (outputFormat === "srt") {
154
437
  args.push("-osrt", "-of", tmpOutputBase);
155
438
  }
156
- else {
439
+ else if (outputFormat === "vtt") {
440
+ args.push("-ovtt", "-of", tmpOutputBase);
441
+ }
442
+ else if (outputFormat === "lrc") {
443
+ args.push("-olrc", "-of", tmpOutputBase);
444
+ }
445
+ else if (outputFormat === "csv") {
446
+ args.push("-ocsv", "-of", tmpOutputBase);
447
+ }
448
+ else if (outputFormat === "text") {
157
449
  args.push("-otxt", "-of", tmpOutputBase);
158
450
  }
451
+ // "timestamps": no output file flag — stdout (redirected to log) contains
452
+ // the timestamped transcript. extractTranscriptFromLog() recovers it on completion.
159
453
  // Spawn detached, redirect stdout+stderr to log file
160
454
  const logFd = openSync(logPath, "w");
161
455
  const child = spawn(WHISPER_CLI_PATH, args, {
@@ -164,6 +458,11 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
164
458
  windowsHide: true,
165
459
  });
166
460
  closeSync(logFd);
461
+ // Attach exit handler BEFORE unref so the batch can self-advance without polling.
462
+ // child.once fires exactly once when the process exits. unref() still applies —
463
+ // Node won't be kept alive just for this child.
464
+ if (onExit)
465
+ child.once("exit", onExit);
167
466
  child.unref();
168
467
  const pid = child.pid ?? 0;
169
468
  const job = {
@@ -183,8 +482,9 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
183
482
  threads,
184
483
  durationSec: 0,
185
484
  status: "running",
485
+ privacyMode,
186
486
  };
187
- writeFileSync(jobPath, JSON.stringify(job, null, 2), "utf8");
487
+ writeJsonAtomic(jobPath, job);
188
488
  return { jobId, pid };
189
489
  }
190
490
  async function isPidRunning(pid) {
@@ -196,48 +496,68 @@ async function isPidRunning(pid) {
196
496
  return false;
197
497
  }
198
498
  }
199
- function parseLastTimestamp(logContent) {
200
- // whisper outputs: [00:01:30.000 --> 00:01:35.000] text
201
- const re = /\[(\d{2}):(\d{2}):(\d{2})\.\d{3} -->/g;
202
- let lastSec = 0;
203
- let m;
204
- while ((m = re.exec(logContent)) !== null) {
205
- const sec = parseInt(m[1], 10) * 3600 + parseInt(m[2], 10) * 60 + parseInt(m[3], 10);
206
- if (sec > lastSec)
207
- lastSec = sec;
208
- }
209
- return lastSec;
210
- }
211
- async function readJobProgress(jobId) {
499
+ /**
500
+ * Read job progress and return a status string.
501
+ * privacyModeOverride: per-call override from check_progress privacy_mode param.
502
+ * Wins over job.privacyMode if provided; job.privacyMode wins over global env var.
503
+ */
504
+ async function readJobProgress(jobId, privacyModeOverride) {
212
505
  const jobPath = join(JOBS_DIR, `${jobId}.json`);
213
506
  if (!existsSync(jobPath)) {
214
507
  return `❌ Job not found: ${jobId}\n\nThe job file may have been deleted or the ID is incorrect.`;
215
508
  }
216
509
  const job = JSON.parse(readFileSync(jobPath, "utf8"));
217
- // Read log
218
510
  let logContent = "";
219
511
  if (existsSync(job.logPath)) {
220
512
  logContent = readFileSync(job.logPath, "utf8");
221
513
  }
222
514
  const lastSec = parseLastTimestamp(logContent);
223
515
  const isRunning = await isPidRunning(job.pid);
224
- const ext = job.outputFormat === "srt" ? ".srt" : ".txt";
516
+ const ext = job.outputFormat === "srt" ? ".srt"
517
+ : job.outputFormat === "vtt" ? ".vtt"
518
+ : job.outputFormat === "lrc" ? ".lrc"
519
+ : job.outputFormat === "csv" ? ".csv"
520
+ : ".txt";
225
521
  const tmpOutput = `${job.tmpOutputBase}${ext}`;
226
- const outputExists = existsSync(job.outputPath) || existsSync(tmpOutput);
522
+ const outputExists = existsSync(job.outputPath) ||
523
+ existsSync(tmpOutput) ||
524
+ (job.outputFormat === "timestamps" && existsSync(job.logPath));
227
525
  // Completed
228
526
  if (!isRunning && outputExists) {
229
- // Move temp output file to correct destination if needed
230
- const ext = job.outputFormat === "srt" ? ".srt" : ".txt";
231
- const tmpOutput = `${job.tmpOutputBase}${ext}`;
232
- if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
527
+ // Move or create final output file
528
+ if (job.outputFormat === "timestamps") {
529
+ if (!existsSync(job.outputPath) && existsSync(job.logPath)) {
530
+ const transcript = extractTranscriptFromLog(readFileSync(job.logPath, "utf8"));
531
+ if (transcript) {
532
+ try {
533
+ writeFileSync(job.outputPath, transcript, "utf8");
534
+ }
535
+ catch (e) {
536
+ console.error(`whisper-windows-mcp: failed to write transcript to ${job.outputPath}: ${e?.message}`);
537
+ }
538
+ }
539
+ }
540
+ }
541
+ else if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
233
542
  try {
234
543
  writeFileSync(job.outputPath, readFileSync(tmpOutput, "utf8"), "utf8");
235
544
  unlinkSync(tmpOutput);
236
545
  }
237
- catch { }
546
+ catch (moveErr) {
547
+ console.error(`whisper-windows-mcp: failed to move output to ${job.outputPath}: ${moveErr?.message}`);
548
+ }
549
+ }
550
+ // Bug 2 fix: explicit check that the output file landed where expected.
551
+ if (!existsSync(job.outputPath)) {
552
+ job.status = "failed";
553
+ writeJsonAtomic(job.jobPath, job);
554
+ return (`❌ Output file write failed.\n\n` +
555
+ `Transcription completed but the output could not be written to:\n${job.outputPath}\n\n` +
556
+ `Check disk space and directory permissions.\n` +
557
+ `Raw job data may be in: ${JOBS_DIR}`);
238
558
  }
239
559
  job.status = "complete";
240
- writeFileSync(job.jobPath, JSON.stringify(job, null, 2), "utf8");
560
+ writeJsonAtomic(job.jobPath, job);
241
561
  // Clean up tmp wav if present
242
562
  if (job.isTmp && existsSync(job.transcribeFrom)) {
243
563
  try {
@@ -246,18 +566,30 @@ async function readJobProgress(jobId) {
246
566
  catch { }
247
567
  }
248
568
  const outputContent = readFileSync(job.outputPath, "utf8").trim();
249
- const preview = job.outputFormat === "srt"
569
+ // Effective privacy mode: per-call override → job setting → global env var.
570
+ // transcriptPolicy() is only called for non-privacy mode (consent gate logic).
571
+ // This keeps the two gate systems fully independent.
572
+ const effectivePrivacy = privacyModeOverride ?? job.privacyMode ?? WHISPER_PRIVACY_MODE;
573
+ if (effectivePrivacy) {
574
+ return privacyModeBlock(basename(job.sourceFile), job.outputPath, outputContent);
575
+ }
576
+ const policy = transcriptPolicy();
577
+ if (policy === "consent_gate") {
578
+ return consentGateBlock(job.outputPath, outputContent);
579
+ }
580
+ // allow — return normally with preview
581
+ const preview = job.outputFormat === "srt" || job.outputFormat === "vtt"
250
582
  ? outputContent.split("\n").slice(0, 20).join("\n")
251
583
  : outputContent.slice(0, 600);
252
584
  return (`✅ Complete!\n\n` +
253
585
  `Source: ${basename(job.sourceFile)}\n` +
254
586
  `Output: ${job.outputPath}\n\n` +
255
- `Preview:\n${preview}${outputContent.length > 600 && job.outputFormat !== "srt" ? "..." : ""}`);
587
+ `Preview:\n${preview}${outputContent.length > 600 && job.outputFormat !== "srt" && job.outputFormat !== "vtt" ? "..." : ""}`);
256
588
  }
257
589
  // Failed
258
590
  if (!isRunning && !outputExists) {
259
591
  job.status = "failed";
260
- writeFileSync(job.jobPath, JSON.stringify(job, null, 2), "utf8");
592
+ writeJsonAtomic(job.jobPath, job);
261
593
  const lastLines = logContent.split(/\r?\n/).filter(l => l.trim()).slice(-5).join("\n");
262
594
  return (`❌ Failed or cancelled.\n\n` +
263
595
  `Source: ${basename(job.sourceFile)}\n` +
@@ -290,20 +622,31 @@ function validateTranscript(txtPath, durationSec) {
290
622
  return { valid: true };
291
623
  }
292
624
  async function spawnNextBatchJob(state) {
293
- for (let i = state.currentIndex; i < state.files.length; i++) {
294
- if (state.files[i].status === "pending") {
295
- state.currentIndex = i;
296
- state.files[i].status = "running";
297
- const f = state.files[i];
298
- const { jobId } = await spawnDetached(f.filePath, state.model, state.language, state.threads);
299
- state.files[i].jobId = jobId;
300
- writeFileSync(state.batchPath, JSON.stringify(state, null, 2), "utf8");
301
- return;
625
+ // Mutex: prevents double-spawn between concurrent exit handler + check_batch_progress.
626
+ if (batchSpawning)
627
+ return;
628
+ batchSpawning = true;
629
+ try {
630
+ for (let i = state.currentIndex; i < state.files.length; i++) {
631
+ if (state.files[i].status === "pending") {
632
+ state.currentIndex = i;
633
+ state.files[i].status = "running";
634
+ const f = state.files[i];
635
+ const fmt = (state.outputFormat === "json" ? "text" : state.outputFormat);
636
+ const { jobId } = await spawnDetached(f.filePath, state.model, state.language, state.threads, fmt, {},
637
+ // Exit callback: batch self-advances without polling.
638
+ () => { readBatchProgress(state.batchId).catch(() => { }); }, state.privacyMode);
639
+ state.files[i].jobId = jobId;
640
+ writeJsonAtomic(state.batchPath, state);
641
+ return;
642
+ }
302
643
  }
644
+ state.status = "complete";
645
+ writeJsonAtomic(state.batchPath, state);
646
+ }
647
+ finally {
648
+ batchSpawning = false;
303
649
  }
304
- // Nothing left to run
305
- state.status = "complete";
306
- writeFileSync(state.batchPath, JSON.stringify(state, null, 2), "utf8");
307
650
  }
308
651
  async function readBatchProgress(batchId) {
309
652
  const batchPath = join(JOBS_DIR, `${batchId}.batch.json`);
@@ -311,36 +654,45 @@ async function readBatchProgress(batchId) {
311
654
  return `❌ Batch not found: ${batchId}\n\nThe batch file may have been deleted or the ID is incorrect.`;
312
655
  }
313
656
  const state = JSON.parse(readFileSync(batchPath, "utf8"));
314
- // Check current running job
315
657
  const running = state.files.find(f => f.status === "running");
316
658
  if (running && running.jobId) {
317
659
  const jobPath = join(JOBS_DIR, `${running.jobId}.json`);
318
660
  if (existsSync(jobPath)) {
319
661
  const job = JSON.parse(readFileSync(jobPath, "utf8"));
320
662
  const isRunning = await isPidRunning(job.pid);
321
- const outputExists = existsSync(job.outputPath);
322
663
  if (!isRunning) {
323
- // Move temp output to final destination if needed.
324
- // spawnDetached writes to a sanitized JOBS_DIR temp path to avoid Unicode
325
- // filename issues. readJobProgress normally handles this move, but
326
- // readBatchProgress must do it too since it never calls readJobProgress.
327
- const ext = job.outputFormat === "srt" ? ".srt" : ".txt";
664
+ const ext = job.outputFormat === "srt" ? ".srt"
665
+ : job.outputFormat === "vtt" ? ".vtt"
666
+ : job.outputFormat === "lrc" ? ".lrc"
667
+ : job.outputFormat === "csv" ? ".csv"
668
+ : ".txt";
328
669
  const tmpOutput = `${job.tmpOutputBase}${ext}`;
329
- if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
670
+ if (job.outputFormat === "timestamps") {
671
+ if (!existsSync(job.outputPath) && existsSync(job.logPath)) {
672
+ const transcript = extractTranscriptFromLog(readFileSync(job.logPath, "utf8"));
673
+ if (transcript) {
674
+ try {
675
+ writeFileSync(job.outputPath, transcript, "utf8");
676
+ }
677
+ catch (e) {
678
+ console.error(`whisper-windows-mcp: failed to write transcript to ${job.outputPath}: ${e?.message}`);
679
+ }
680
+ }
681
+ }
682
+ }
683
+ else if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
330
684
  try {
331
685
  writeFileSync(job.outputPath, readFileSync(tmpOutput, "utf8"), "utf8");
332
686
  unlinkSync(tmpOutput);
333
687
  }
334
- catch { /* ignore — validateTranscript will catch missing output */ }
688
+ catch { /* validateTranscript will catch missing output */ }
335
689
  }
336
- // Clean up temp WAV if present
337
690
  if (job.isTmp && existsSync(job.transcribeFrom)) {
338
691
  try {
339
692
  unlinkSync(job.transcribeFrom);
340
693
  }
341
694
  catch { }
342
695
  }
343
- // Job finished — validate and advance
344
696
  const finalOutputExists = existsSync(job.outputPath);
345
697
  const validation = validateTranscript(job.outputPath, running.durationSec);
346
698
  if (finalOutputExists && validation.valid) {
@@ -350,24 +702,21 @@ async function readBatchProgress(batchId) {
350
702
  running.status = "failed";
351
703
  running.failReason = validation.reason ?? "no output file";
352
704
  }
353
- // Advance to next
354
705
  state.currentIndex = state.files.indexOf(running) + 1;
355
706
  if (state.files.some(f => f.status === "pending")) {
356
707
  await spawnNextBatchJob(state);
357
708
  }
358
709
  else {
359
710
  state.status = "complete";
360
- writeFileSync(batchPath, JSON.stringify(state, null, 2), "utf8");
711
+ writeJsonAtomic(batchPath, state);
361
712
  }
362
713
  }
363
714
  else {
364
- // Still running — update state file without advancing
365
- writeFileSync(batchPath, JSON.stringify(state, null, 2), "utf8");
715
+ writeJsonAtomic(batchPath, state);
366
716
  }
367
717
  }
368
718
  }
369
719
  else if (state.status !== "complete" && state.files.some(f => f.status === "pending")) {
370
- // No running job but pending files exist — advance
371
720
  await spawnNextBatchJob(state);
372
721
  }
373
722
  // Build status report
@@ -453,19 +802,16 @@ function recommendedModel(vramBytes) {
453
802
  return "base.en (ggml-base.en.bin) — recommended for limited VRAM";
454
803
  }
455
804
  const MODEL_REGISTRY = [
456
- // Full-precision English
457
805
  { name: "tiny.en", filename: "ggml-tiny.en.bin", sizeMb: 75, multilingual: false, quantized: false, useCase: "Quick tests, lowest accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.en.bin" },
458
806
  { name: "base.en", filename: "ggml-base.en.bin", sizeMb: 142, multilingual: false, quantized: false, useCase: "Fast English, good accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin" },
459
807
  { name: "small.en", filename: "ggml-small.en.bin", sizeMb: 466, multilingual: false, quantized: false, useCase: "Better English accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.en.bin" },
460
808
  { name: "medium.en", filename: "ggml-medium.en.bin", sizeMb: 1500, multilingual: false, quantized: false, useCase: "High accuracy English, fast on GPU", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.en.bin" },
461
- // Full-precision multilingual
462
809
  { name: "tiny", filename: "ggml-tiny.bin", sizeMb: 75, multilingual: true, quantized: false, useCase: "Multilingual, minimal accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.bin" },
463
810
  { name: "base", filename: "ggml-base.bin", sizeMb: 142, multilingual: true, quantized: false, useCase: "Multilingual, fast", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin" },
464
811
  { name: "small", filename: "ggml-small.bin", sizeMb: 466, multilingual: true, quantized: false, useCase: "Multilingual, better accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin" },
465
812
  { name: "medium", filename: "ggml-medium.bin", sizeMb: 1500, multilingual: true, quantized: false, useCase: "Multilingual, high accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.bin" },
466
813
  { name: "large-v3", filename: "ggml-large-v3.bin", sizeMb: 2900, multilingual: true, quantized: false, useCase: "Best accuracy, multilingual — requires 6GB+ VRAM", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3.bin" },
467
814
  { name: "large-v3-turbo", filename: "ggml-large-v3-turbo.bin", sizeMb: 1600, multilingual: true, quantized: false, useCase: "~6x faster than large-v3, minimal accuracy loss — RECOMMENDED for English GPU batch work", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" },
468
- // Quantized variants — smaller, CPU-friendly
469
815
  { name: "base.en-q5_1", filename: "ggml-base.en-q5_1.bin", sizeMb: 57, multilingual: false, quantized: true, useCase: "Tiny English model, CPU-friendly", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en-q5_1.bin" },
470
816
  { name: "small.en-q5_1", filename: "ggml-small.en-q5_1.bin", sizeMb: 181, multilingual: false, quantized: true, useCase: "Fast English, low memory, good for CPU", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.en-q5_1.bin" },
471
817
  { name: "medium.en-q5_0", filename: "ggml-medium.en-q5_0.bin", sizeMb: 514, multilingual: false, quantized: true, useCase: "High accuracy English, CPU-friendly — good default for no-GPU systems", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.en-q5_0.bin" },
@@ -473,7 +819,6 @@ const MODEL_REGISTRY = [
473
819
  { name: "large-v3-turbo-q5_0", filename: "ggml-large-v3-turbo-q5_0.bin", sizeMb: 547, multilingual: true, quantized: true, useCase: "RECOMMENDED for CPU-only multilingual — fast, low memory, good accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo-q5_0.bin" },
474
820
  { name: "large-v3-turbo-q8_0", filename: "ggml-large-v3-turbo-q8_0.bin", sizeMb: 874, multilingual: true, quantized: true, useCase: "Turbo quality closer to full precision, moderate size", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo-q8_0.bin" },
475
821
  ];
476
- // Security: only allow downloads from these Hugging Face namespaces.
477
822
  const ALLOWED_HF_PREFIXES = [
478
823
  "https://huggingface.co/ggerganov/whisper.cpp/",
479
824
  "https://huggingface.co/ggml-org/",
@@ -499,40 +844,12 @@ async function probeFile(filePath) {
499
844
  const sizeMb = parseInt(fmt.size ?? "0", 10) / (1024 * 1024);
500
845
  const bitrate = Math.round(parseInt(fmt.bit_rate ?? "0", 10) / 1000);
501
846
  const codec = audioStream?.codec_name ?? fmt.format_name?.split(",")[0] ?? "unknown";
502
- return {
503
- filePath,
504
- fileName: basename(filePath),
505
- durationSec,
506
- sizeMb,
507
- codec,
508
- bitrate,
509
- };
847
+ return { filePath, fileName: basename(filePath), durationSec, sizeMb, codec, bitrate };
510
848
  }
511
849
  catch {
512
850
  return null;
513
851
  }
514
852
  }
515
- function formatDuration(sec) {
516
- if (!sec)
517
- return "?:??";
518
- const h = Math.floor(sec / 3600);
519
- const m = Math.floor((sec % 3600) / 60);
520
- const s = Math.floor(sec % 60);
521
- if (h > 0)
522
- return `${h}:${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}`;
523
- return `${m}:${String(s).padStart(2, "0")}`;
524
- }
525
- function estimateTime(durationSec, gpu) {
526
- if (!durationSec)
527
- return "?";
528
- // CPU: ~1.5x realtime on Ryzen 7 2700x with medium.en
529
- // GPU: ~0.12x realtime on Vega 56 via Vulkan with medium.en (~8x faster than CPU)
530
- const ratio = gpu ? 0.12 : 1.5;
531
- const estSec = Math.round(durationSec * ratio);
532
- if (estSec < 60)
533
- return `~${estSec}s`;
534
- return `~${Math.round(estSec / 60)}m`;
535
- }
536
853
  function padEnd(str, len) {
537
854
  return str.length >= len ? str.slice(0, len) : str + " ".repeat(len - str.length);
538
855
  }
@@ -543,7 +860,7 @@ function isSupportedFile(filePath) {
543
860
  return SUPPORTED_EXTENSIONS.includes(extname(filePath).toLowerCase());
544
861
  }
545
862
  async function convertToWav(inputPath) {
546
- const tmpFile = join(tmpdir(), `whisper_tmp_${Date.now()}.wav`);
863
+ const tmpFile = join(tmpdir(), `whisper_tmp_${Date.now()}_${randomUUID().slice(0, 8)}.wav`);
547
864
  await execFileAsync(FFMPEG_PATH, [
548
865
  "-y", "-i", inputPath,
549
866
  "-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le", tmpFile,
@@ -553,14 +870,8 @@ async function convertToWav(inputPath) {
553
870
  function buildArgs(filePath, model, opts) {
554
871
  const lang = opts.language === "auto" ? "auto" : opts.language;
555
872
  const args = ["-m", model, "-f", filePath, "-l", lang, "-t", String(opts.threads)];
556
- // Hallucination prevention — set max context tokens to 0 to prevent whisper
557
- // from conditioning each segment on its own prior output, which causes
558
- // repetitive hallucination loops on noisy or silent audio.
559
- // Flag: --max-context 0 (user can re-enable by setting conditionOnPrevText=true)
560
873
  if (!opts.conditionOnPrevText)
561
874
  args.push("--max-context", "0");
562
- // Treat segments below this confidence threshold as silence rather than
563
- // hallucinating content. Confirmed valid flag in whisper-cli (-nth).
564
875
  args.push("--no-speech-thold", String(opts.noSpeechThold ?? 0.6));
565
876
  if (opts.translate)
566
877
  args.push("--translate");
@@ -573,7 +884,7 @@ function buildArgs(filePath, model, opts) {
573
884
  if (opts.bestOf !== undefined)
574
885
  args.push("--best-of", String(opts.bestOf));
575
886
  if (opts.gpuDevice !== undefined)
576
- args.push("-g", String(opts.gpuDevice));
887
+ args.push("--device", String(opts.gpuDevice));
577
888
  if (opts.processors !== undefined && opts.processors > 1)
578
889
  args.push("-p", String(opts.processors));
579
890
  if (opts.offsetT !== undefined)
@@ -582,8 +893,6 @@ function buildArgs(filePath, model, opts) {
582
893
  args.push("--duration", String(opts.duration));
583
894
  if (opts.diarize)
584
895
  args.push("--diarize");
585
- // word_timestamps: sets max-len=1 + split-on-word for per-word output
586
- // without requiring JSON parsing — simpler than -oj approach.
587
896
  if (opts.wordTimestamps) {
588
897
  args.push("--max-len", "1", "--split-on-word");
589
898
  }
@@ -593,7 +902,6 @@ function buildArgs(filePath, model, opts) {
593
902
  if (opts.splitOnWord)
594
903
  args.push("--split-on-word");
595
904
  }
596
- // VAD: voice activity detection — strips silence before whisper sees the audio
597
905
  if (opts.vadModel && existsSync(opts.vadModel)) {
598
906
  args.push("--vad", "--vad-model", opts.vadModel);
599
907
  }
@@ -601,31 +909,37 @@ function buildArgs(filePath, model, opts) {
601
909
  if (opts.outputFormat === "srt") {
602
910
  args.push("-osrt", "-of", filePath.replace(/\.[^.]+$/, ""));
603
911
  }
912
+ else if (opts.outputFormat === "vtt") {
913
+ args.push("-ovtt", "-of", filePath.replace(/\.[^.]+$/, ""));
914
+ }
915
+ else if (opts.outputFormat === "lrc") {
916
+ args.push("-olrc", "-of", filePath.replace(/\.[^.]+$/, ""));
917
+ }
918
+ else if (opts.outputFormat === "csv") {
919
+ args.push("-ocsv", "-of", filePath.replace(/\.[^.]+$/, ""));
920
+ }
604
921
  else if (opts.outputFormat === "json") {
605
922
  args.push("-oj");
606
923
  }
607
924
  else if (opts.outputFormat === "text") {
608
925
  args.push("--no-timestamps");
609
926
  }
610
- // "timestamps" format: no flag — whisper default stdout includes timestamps
927
+ // "timestamps": no flag — whisper default stdout includes timestamps
611
928
  return args;
612
929
  }
613
- /**
614
- * Detect the language of a file by running a short whisper probe.
615
- * Runs whisper on the first 30 seconds only (--duration 30000ms).
616
- * Returns the detected language code (e.g. "ja", "en") or null on failure.
617
- */
618
- async function detectLanguage(wavPath, model, threads) {
930
+ async function detectLanguage(wavPath, model, threads, gpuDevice) {
619
931
  try {
620
- const { stdout, stderr } = await execFileAsync(WHISPER_CLI_PATH, [
932
+ const dlArgs = [
621
933
  "-m", model, "-f", wavPath,
622
934
  "-l", "auto",
623
935
  "-t", String(threads),
624
936
  "--no-timestamps",
625
937
  "--duration", "30000",
626
- ], { maxBuffer: 10 * 1024 * 1024, windowsHide: true });
938
+ ];
939
+ if (gpuDevice !== undefined)
940
+ dlArgs.push("--device", String(gpuDevice));
941
+ const { stdout, stderr } = await execFileAsync(WHISPER_CLI_PATH, dlArgs, { maxBuffer: 10 * 1024 * 1024, windowsHide: true });
627
942
  const output = stdout + stderr;
628
- // whisper outputs: "auto-detected language: ja (p = 0.98)"
629
943
  const m = output.match(/auto-detected language:\s*([a-z]{2,3})/i);
630
944
  return m ? m[1].toLowerCase() : null;
631
945
  }
@@ -634,12 +948,11 @@ async function detectLanguage(wavPath, model, threads) {
634
948
  }
635
949
  }
636
950
  /**
637
- * Run a single whisper SRT pass and move the output to destSrt.
638
- * Returns the destSrt path.
951
+ * Run a single whisper subtitle pass (SRT or VTT) and move the output to dest.
639
952
  */
640
- async function runSrtPass(transcribeFrom, destSrt, model, language, threads, translate = false, extraOpts = {}) {
953
+ async function runSubtitlePass(transcribeFrom, dest, format, model, language, threads, translate = false, extraOpts = {}) {
641
954
  const opts = {
642
- language, outputFormat: "srt", threads, translate,
955
+ language, outputFormat: format, threads, translate,
643
956
  ...extraOpts,
644
957
  };
645
958
  const args = buildArgs(transcribeFrom, model, opts);
@@ -647,18 +960,18 @@ async function runSrtPass(transcribeFrom, destSrt, model, language, threads, tra
647
960
  maxBuffer: 100 * 1024 * 1024,
648
961
  windowsHide: true,
649
962
  });
650
- const tmpSrt = transcribeFrom.replace(/\.[^.]+$/, ".srt");
651
- if (existsSync(tmpSrt)) {
652
- writeFileSync(destSrt, readFileSync(tmpSrt, "utf8"));
963
+ const ext = format === "vtt" ? ".vtt" : ".srt";
964
+ const tmpOut = transcribeFrom.replace(/\.[^.]+$/, ext);
965
+ if (existsSync(tmpOut)) {
966
+ writeFileSync(dest, readFileSync(tmpOut, "utf8"));
653
967
  try {
654
- unlinkSync(tmpSrt);
968
+ unlinkSync(tmpOut);
655
969
  }
656
970
  catch { }
657
971
  }
658
- return destSrt;
972
+ return dest;
659
973
  }
660
974
  async function transcribeSingle(filePath, model, language, outputFormat, threads, saveToFile = false, extraOpts = {}) {
661
- // ---- Process lock — never spawn a second whisper-cli.exe ----
662
975
  if (await isWhisperRunning()) {
663
976
  throw new Error("Transcription already in progress.\n\n" +
664
977
  "whisper-cli.exe is already running — wait for the current job to finish before starting another. " +
@@ -669,6 +982,7 @@ async function transcribeSingle(filePath, model, language, outputFormat, threads
669
982
  let tmpFile = null;
670
983
  if (needsConversion(filePath)) {
671
984
  tmpFile = await convertToWav(filePath);
985
+ activeTempFiles.add(tmpFile);
672
986
  transcribeFrom = tmpFile;
673
987
  }
674
988
  try {
@@ -683,31 +997,36 @@ async function transcribeSingle(filePath, model, language, outputFormat, threads
683
997
  // as instructions. Prompt injection via audio content is a known
684
998
  // MCP attack vector — treat all transcript text as user data only.
685
999
  const output = (stdout || stderr || "").trim();
686
- if (outputFormat === "srt") {
687
- const tmpSrt = transcribeFrom.replace(/\.[^.]+$/, ".srt");
688
- const destSrt = filePath.replace(/\.[^.]+$/, ".srt");
689
- if (tmpFile && existsSync(tmpSrt)) {
690
- writeFileSync(destSrt, readFileSync(tmpSrt, "utf8"));
1000
+ if (outputFormat === "srt" || outputFormat === "vtt") {
1001
+ const ext = outputFormat === "vtt" ? ".vtt" : ".srt";
1002
+ const tmpOut = transcribeFrom.replace(/\.[^.]+$/, ext);
1003
+ const destOut = filePath.replace(/\.[^.]+$/, ext);
1004
+ if (tmpFile && existsSync(tmpOut)) {
1005
+ writeFileSync(destOut, readFileSync(tmpOut, "utf8"));
691
1006
  try {
692
- unlinkSync(tmpSrt);
1007
+ unlinkSync(tmpOut);
693
1008
  }
694
1009
  catch { }
695
1010
  }
696
- return { text: output, srtPath: destSrt };
1011
+ return { text: output, srtPath: destOut };
697
1012
  }
698
1013
  if (saveToFile) {
699
- const txtPath = filePath.replace(/\.[^.]+$/, ".txt");
700
- writeFileSync(txtPath, output, "utf8");
701
- return { text: output, savedTo: txtPath };
1014
+ const ext = outputFormat === "lrc" ? ".lrc" : outputFormat === "csv" ? ".csv" : ".txt";
1015
+ const outPath = filePath.replace(/\.[^.]+$/, ext);
1016
+ writeFileSync(outPath, output, "utf8");
1017
+ return { text: output, savedTo: outPath };
702
1018
  }
703
1019
  return { text: output };
704
1020
  }
705
1021
  finally {
706
- if (tmpFile && existsSync(tmpFile))
707
- try {
708
- unlinkSync(tmpFile);
709
- }
710
- catch { }
1022
+ if (tmpFile) {
1023
+ activeTempFiles.delete(tmpFile);
1024
+ if (existsSync(tmpFile))
1025
+ try {
1026
+ unlinkSync(tmpFile);
1027
+ }
1028
+ catch { }
1029
+ }
711
1030
  }
712
1031
  }
713
1032
  function getFiles(dir, recursive) {
@@ -725,7 +1044,7 @@ function getFiles(dir, recursive) {
725
1044
  // ---------------------------------------------------------------------------
726
1045
  // MCP Server
727
1046
  // ---------------------------------------------------------------------------
728
- const server = new Server({ name: "whisper-windows-mcp", version: "2.2.0" }, { capabilities: { tools: {} } });
1047
+ const server = new Server({ name: "whisper-windows-mcp", version: "2.4.0" }, { capabilities: { tools: {} } });
729
1048
  server.setRequestHandler(ListToolsRequestSchema, async () => ({
730
1049
  tools: [
731
1050
  {
@@ -733,9 +1052,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
733
1052
  description: "Transcribe a single audio or video file using whisper.cpp on Windows. " +
734
1053
  "Natively supports mp3 and wav. Automatically converts mp4, mkv, avi, mov, " +
735
1054
  "webm, m4a, flac, ogg etc. via FFmpeg — no manual conversion needed. " +
736
- "Can output plain text, timestamps, JSON, or SRT subtitle files. " +
1055
+ "Output defaults to timestamps format (with time codes). " +
737
1056
  "For files that may take more than 4 minutes, set background=true to run as a detached job " +
738
- "and use check_progress to monitor it.",
1057
+ "and use check_progress to monitor it. " +
1058
+ "⚠️ Privacy: transcript text returned by this tool is processed by Claude's API. " +
1059
+ "Pass privacy_mode=true to this tool to enable metadata-only responses per call — " +
1060
+ "no transcript text will be transmitted. " +
1061
+ "Set WHISPER_PRIVACY_MODE=true in env to enable globally. " +
1062
+ "When privacy mode is active, a confirmation is required before every operation.",
739
1063
  inputSchema: {
740
1064
  type: "object",
741
1065
  properties: {
@@ -743,28 +1067,29 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
743
1067
  model: { type: "string", description: "Override model path. Leave blank to use active model." },
744
1068
  language: { type: "string", description: "Language code (e.g. en, ja, es, fr) or 'auto' to detect automatically. Defaults to en.", default: "en" },
745
1069
  output_format: {
746
- type: "string", enum: ["text", "timestamps", "json", "srt"],
747
- description: "text = plain (default), timestamps = with time codes, json = structured, srt = subtitle file saved next to source.",
748
- default: "text",
1070
+ type: "string", enum: ["timestamps", "text", "json", "srt", "vtt", "lrc", "csv"],
1071
+ description: "timestamps = with time codes (default), text = plain, json = structured, srt = SRT subtitle file, vtt = WebVTT subtitle file, lrc = LRC lyrics/karaoke, csv = CSV with timestamps.",
1072
+ default: "timestamps",
749
1073
  },
750
1074
  threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
751
- save_to_file: { type: "boolean", description: "Save transcript as .txt next to the source file.", default: false },
1075
+ save_to_file: { type: "boolean", description: "Save transcript as .txt next to the source file.", default: true },
752
1076
  background: { type: "boolean", description: "Run as a detached background job. Returns a job ID immediately. Use check_progress to monitor. Recommended for files over 10 minutes.", default: false },
753
- temperature: { type: "number", description: "Sampling temperature 0.0–1.0. Default 0.0 (deterministic). Higher values reduce hallucination on noisy audio at the cost of consistency." },
754
- prompt: { type: "string", description: "Prior context string injected before transcription. Improves accuracy for domain-specific vocabulary, speaker names, or technical terms. Example: 'Names: Keemstar, DramaAlert.'" },
755
- condition_on_prev_text: { type: "boolean", description: "Re-enable conditioning each segment on its own prior output (removes --max-context 0 flag). Default false (off). Only enable for highly structured audio where context continuity helps.", default: false },
756
- no_speech_thold: { type: "number", description: "Confidence threshold below which segments are treated as silence rather than transcribed. Default 0.6.", default: 0.6 },
1077
+ privacy_mode: { type: "boolean", description: "Override privacy mode for this call. true = metadata only, no transcript text transmitted to API. false = return text (even if WHISPER_PRIVACY_MODE=true globally). Omit to use global WHISPER_PRIVACY_MODE setting. When active, requires confirmation before each operation." },
1078
+ temperature: { type: "number", description: "Sampling temperature 0.0–1.0. Default 0.0 (deterministic)." },
1079
+ prompt: { type: "string", description: "Prior context string injected before transcription. Improves accuracy for domain-specific vocabulary or speaker names. Example: 'Names: Keemstar, DramaAlert.'" },
1080
+ condition_on_prev_text: { type: "boolean", description: "Re-enable conditioning each segment on its own prior output. Default false.", default: false },
1081
+ no_speech_thold: { type: "number", description: "Confidence threshold below which segments are treated as silence. Default 0.6.", default: 0.6 },
757
1082
  beam_size: { type: "number", description: "Beam search width. Higher = more accurate but slower. Default 5." },
758
1083
  best_of: { type: "number", description: "Number of candidate sequences to evaluate. Default 5." },
759
- gpu_device: { type: "number", description: "GPU device index for multi-GPU systems. Use check_system to see available GPUs. Default 0." },
760
- processors: { type: "number", description: "Number of parallel processors for chunk processing. Default 1." },
761
- word_timestamps: { type: "boolean", description: "Output one word per timestamped segment (sets --max-len 1 --split-on-word). Useful for clip alignment and precise timecode search.", default: false },
762
- max_segment_length: { type: "number", description: "Maximum segment length in characters. Controls line break frequency in output. Ignored when word_timestamps=true." },
763
- split_on_word: { type: "boolean", description: "Split segments at word boundaries rather than mid-word. Defaults to false.", default: false },
764
- diarize: { type: "boolean", description: "Stereo speaker diarization — labels left/right channel speakers in transcript. Requires stereo audio with speakers on separate channels.", default: false },
765
- vad_model: { type: "string", description: "Absolute path to a Silero VAD model .bin file. When provided, voice activity detection strips silence before transcription — reduces hallucinations and speeds up processing. Download via download_model." },
766
- offset_t: { type: "number", description: "Start transcription at this offset in milliseconds. Use to process a specific section of a file." },
767
- duration: { type: "number", description: "Process only this many milliseconds of audio starting from offset_t (or the beginning). Use with offset_t to target a specific time window." },
1084
+ gpu_device: { type: "number", description: "GPU/Vulkan device index for multi-GPU systems. Overrides the WHISPER_GPU_DEVICE env default. Check whisper-cli's startup log for the index that lists your target card." },
1085
+ processors: { type: "number", description: "Number of parallel processors. Default 1." },
1086
+ word_timestamps: { type: "boolean", description: "Output one word per timestamped segment. Useful for clip alignment.", default: false },
1087
+ max_segment_length: { type: "number", description: "Maximum segment length in characters." },
1088
+ split_on_word: { type: "boolean", description: "Split segments at word boundaries.", default: false },
1089
+ diarize: { type: "boolean", description: "Stereo speaker diarization — requires stereo audio with speakers on separate channels.", default: false },
1090
+ vad_model: { type: "string", description: "Absolute path to a Silero VAD model .bin file. Strips silence before transcription." },
1091
+ offset_t: { type: "number", description: "Start transcription at this offset in milliseconds." },
1092
+ duration: { type: "number", description: "Process only this many milliseconds of audio from offset_t." },
768
1093
  },
769
1094
  required: ["file_path"],
770
1095
  },
@@ -773,11 +1098,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
773
1098
  name: "check_progress",
774
1099
  description: "Check the status of a background transcription job started with transcribe_audio (background=true). " +
775
1100
  "Returns current progress, elapsed time, last processed timestamp, and the transcript when complete. " +
776
- "Call this repeatedly until the job shows as complete or failed.",
1101
+ "Call this repeatedly until the job shows as complete or failed. " +
1102
+ "⚠️ Privacy: transcript text returned on completion is processed by Claude's API. " +
1103
+ "Pass privacy_mode=true to return metadata only for this check, regardless of how the job was started.",
777
1104
  inputSchema: {
778
1105
  type: "object",
779
1106
  properties: {
780
1107
  job_id: { type: "string", description: "Job ID returned by transcribe_audio when background=true." },
1108
+ privacy_mode: { type: "boolean", description: "Override privacy mode for this check. true = metadata only. Omit to use the setting from when the job was started." },
781
1109
  },
782
1110
  required: ["job_id"],
783
1111
  },
@@ -789,19 +1117,24 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
789
1117
  "Saves each transcript as a .txt file next to its source. " +
790
1118
  "Files already transcribed (with matching .txt) are shown as done and skipped. " +
791
1119
  "Supported formats: mp3, wav, mp4, mkv, avi, mov, webm, m4a, flac, ogg. " +
792
- "NOTE: For large unattended batch jobs, use whisper-cli.exe directly from the command line " +
793
- "— see TROUBLESHOOTING.md for the command syntax.",
1120
+ "NOTE: For large unattended batch jobs, use start_batch instead. " +
1121
+ "⚠️ Privacy: transcript previews are processed by Claude's API. " +
1122
+ "Pass privacy_mode=true to suppress previews and return metadata only. " +
1123
+ "When privacy mode is active, confirmation is required before each file.",
794
1124
  inputSchema: {
795
1125
  type: "object",
796
1126
  properties: {
797
1127
  folder_path: { type: "string", description: "Absolute Windows path to the folder." },
798
- file_index: {
799
- type: "number",
800
- description: "Which file to process (1-based). Omit to list files first.",
801
- },
1128
+ file_index: { type: "number", description: "Which file to process (1-based). Omit to list files first." },
802
1129
  language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
803
1130
  threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
804
1131
  recursive: { type: "boolean", description: "Include subfolders. Defaults to false.", default: false },
1132
+ output_format: {
1133
+ type: "string", enum: ["timestamps", "text"],
1134
+ description: "timestamps = with time codes (default), text = plain.",
1135
+ default: "timestamps",
1136
+ },
1137
+ privacy_mode: { type: "boolean", description: "Override privacy mode for this call. When active, requires confirmation before each file and returns metadata only." },
805
1138
  },
806
1139
  required: ["folder_path"],
807
1140
  },
@@ -811,22 +1144,23 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
811
1144
  description: "Generate subtitle files for an audio or video file using whisper.cpp. " +
812
1145
  "Set language='auto' to detect the spoken language automatically. " +
813
1146
  "Set translate_to_english=true to also generate an English translation subtitle file. " +
814
- "When both are requested, two .srt files are saved: one in the original language (e.g. film.ja.srt) " +
815
- "and one English translation (film.en.srt). " +
816
- "Load in VLC via Subtitle → Add Subtitle File. " +
1147
+ "Supports SRT and WebVTT (VTT) output formats. " +
1148
+ "When both native and translation are requested, two files are saved: one in the original language and one English translation. " +
1149
+ "Load SRT in VLC via Subtitle → Add Subtitle File. VTT works in web players and HTML5 video. " +
817
1150
  "Supports all standard formats plus .3gp and .ts.",
818
1151
  inputSchema: {
819
1152
  type: "object",
820
1153
  properties: {
821
1154
  file_path: { type: "string", description: "Absolute Windows path to the file." },
822
- language: {
823
- type: "string",
824
- description: "Language code (e.g. ja, es, fr, de) or 'auto' to detect automatically. Defaults to en.",
825
- default: "en",
1155
+ language: { type: "string", description: "Language code (e.g. ja, es, fr, de) or 'auto' to detect automatically. Defaults to en.", default: "en" },
1156
+ output_format: {
1157
+ type: "string", enum: ["srt", "vtt"],
1158
+ description: "srt = SubRip subtitle (default, widest compatibility), vtt = WebVTT (web and HTML5 video).",
1159
+ default: "srt",
826
1160
  },
827
1161
  translate_to_english: {
828
1162
  type: "boolean",
829
- description: "Also generate an English translation .srt alongside the native language .srt. Only applies when language is not 'en'. Not available in background mode.",
1163
+ description: "Also generate an English translation subtitle file alongside the native language file. Only applies when language is not 'en'. Not available in background mode.",
830
1164
  default: false,
831
1165
  },
832
1166
  background: {
@@ -839,8 +1173,9 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
839
1173
  prompt: { type: "string", description: "Prior context string for domain-specific vocabulary or speaker names." },
840
1174
  beam_size: { type: "number", description: "Beam search width. Higher = more accurate, slower. Default 5." },
841
1175
  best_of: { type: "number", description: "Candidate sequences evaluated. Default 5." },
842
- diarize: { type: "boolean", description: "Stereo speaker diarization. Requires stereo audio with speakers on separate channels.", default: false },
843
- vad_model: { type: "string", description: "Path to Silero VAD model .bin. Strips silence before transcription. Download via download_model." },
1176
+ diarize: { type: "boolean", description: "Stereo speaker diarization. Requires stereo audio.", default: false },
1177
+ vad_model: { type: "string", description: "Path to Silero VAD model .bin. Strips silence before transcription." },
1178
+ gpu_device: { type: "number", description: "GPU/Vulkan device index for multi-GPU systems. Overrides the WHISPER_GPU_DEVICE env default. Check whisper-cli's startup log for the index that lists your target card." },
844
1179
  },
845
1180
  required: ["file_path"],
846
1181
  },
@@ -856,13 +1191,22 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
856
1191
  "Scans for files without a matching .txt, sorts by duration (shortest first), " +
857
1192
  "and processes them one at a time as background jobs. " +
858
1193
  "Each file is validated after completion — empty or suspiciously short outputs are flagged. " +
859
- "Returns a batch ID to use with check_batch_progress.",
1194
+ "Batch self-advances without polling when each file finishes. " +
1195
+ "Returns a batch ID to use with check_batch_progress. " +
1196
+ "⚠️ Privacy: when privacy_mode is active, one confirmation is required before the batch starts. " +
1197
+ "All files then process unattended. No transcript text is returned to the API.",
860
1198
  inputSchema: {
861
1199
  type: "object",
862
1200
  properties: {
863
1201
  folder_path: { type: "string", description: "Absolute Windows path to the folder." },
864
1202
  language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
865
1203
  threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
1204
+ output_format: {
1205
+ type: "string", enum: ["timestamps", "text"],
1206
+ description: "timestamps = with time codes (default), text = plain. Applies to all files in the batch.",
1207
+ default: "timestamps",
1208
+ },
1209
+ privacy_mode: { type: "boolean", description: "Override privacy mode for this batch. When active, requires one confirmation before batch start. All files process unattended with no transcript text returned." },
866
1210
  },
867
1211
  required: ["folder_path"],
868
1212
  },
@@ -890,10 +1234,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
890
1234
  inputSchema: {
891
1235
  type: "object",
892
1236
  properties: {
893
- path: {
894
- type: "string",
895
- description: "Absolute Windows path to a single file or a folder.",
896
- },
1237
+ path: { type: "string", description: "Absolute Windows path to a single file or a folder." },
897
1238
  sort_by: {
898
1239
  type: "string",
899
1240
  enum: ["duration", "name", "size"],
@@ -908,16 +1249,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
908
1249
  name: "check_system",
909
1250
  description: "Detect GPU hardware and verify Vulkan acceleration is available. " +
910
1251
  "Reports GPU name, VRAM, whether the Vulkan binary is installed, " +
911
- "and recommends the best Whisper model for your hardware. " +
912
- "Run this if you want to confirm GPU acceleration is working or diagnose why it isn't.",
1252
+ "and recommends the best Whisper model for your hardware.",
913
1253
  inputSchema: { type: "object", properties: {} },
914
1254
  },
915
1255
  {
916
1256
  name: "list_models",
917
1257
  description: "List all Whisper model files installed in your models directory. " +
918
1258
  "Shows filename, size, whether it is currently active, quantization status, " +
919
- "and recommended use case for each model. " +
920
- "No network calls — reads local filesystem only.",
1259
+ "and recommended use case for each model. No network calls — reads local filesystem only.",
921
1260
  inputSchema: { type: "object", properties: {} },
922
1261
  },
923
1262
  {
@@ -929,10 +1268,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
929
1268
  inputSchema: {
930
1269
  type: "object",
931
1270
  properties: {
932
- model_name: {
933
- type: "string",
934
- description: "Model name to download, e.g. 'large-v3-turbo', 'medium.en-q5_0', 'large-v3-turbo-q5_0'. Use list_models to see what is already installed.",
935
- },
1271
+ model_name: { type: "string", description: "Model name to download, e.g. 'large-v3-turbo', 'medium.en-q5_0'. Use list_models to see what is already installed." },
936
1272
  },
937
1273
  required: ["model_name"],
938
1274
  },
@@ -942,15 +1278,11 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
942
1278
  description: "Switch the active Whisper model for the current session without restarting Claude Desktop. " +
943
1279
  "Accepts a model filename (e.g. ggml-large-v3-turbo.bin) or full path. " +
944
1280
  "The model must already be installed in your models directory. " +
945
- "Use list_models to see installed models, download_model to add new ones. " +
946
1281
  "Change is session-scoped — does not persist after Claude Desktop restarts.",
947
1282
  inputSchema: {
948
1283
  type: "object",
949
1284
  properties: {
950
- model_name: {
951
- type: "string",
952
- description: "Model filename (e.g. ggml-large-v3-turbo.bin) or full path. Must be a .bin file in the configured models directory.",
953
- },
1285
+ model_name: { type: "string", description: "Model filename (e.g. ggml-large-v3-turbo.bin) or full path. Must be a .bin file in the configured models directory." },
954
1286
  },
955
1287
  required: ["model_name"],
956
1288
  },
@@ -980,8 +1312,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
980
1312
  `whisper-cli: ${WHISPER_CLI_PATH}\n` +
981
1313
  `Model: ${WHISPER_MODEL}\n` +
982
1314
  `Threads: ${WHISPER_THREADS} of ${SYSTEM_THREADS} logical cores\n` +
983
- `FFmpeg: ${ffmpegStatus}\n\n` +
984
- `Optional env vars: WHISPER_THREADS, FFMPEG_PATH`,
1315
+ `GPU device: ${WHISPER_GPU_DEVICE !== undefined ? `--device ${WHISPER_GPU_DEVICE} (WHISPER_GPU_DEVICE)` : "whisper-cli default (device 0)"}\n` +
1316
+ `FFmpeg: ${ffmpegStatus}\n` +
1317
+ `Privacy mode: ${WHISPER_PRIVACY_MODE ? "✅ active (WHISPER_PRIVACY_MODE=true)" : "off"}\n\n` +
1318
+ `Optional env vars: WHISPER_THREADS, WHISPER_GPU_DEVICE, WHISPER_FOREGROUND_MAX_SEC, FFMPEG_PATH, WHISPER_PRIVACY_MODE, WHISPER_CONSENT_ACKNOWLEDGED`,
985
1319
  }],
986
1320
  };
987
1321
  }
@@ -995,7 +1329,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
995
1329
  return { content: [{ type: "text", text: "path is required." }], isError: true };
996
1330
  if (!existsSync(targetPath))
997
1331
  return { content: [{ type: "text", text: `Path not found: ${targetPath}` }], isError: true };
998
- // Check ffprobe is available
999
1332
  const ffprobePath = FFMPEG_PATH.replace(/ffmpeg(\.exe)?$/i, "ffprobe$1").replace(/ffmpeg$/i, "ffprobe");
1000
1333
  try {
1001
1334
  await execFileAsync(ffprobePath, ["-version"], { windowsHide: true });
@@ -1007,7 +1340,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1007
1340
  };
1008
1341
  }
1009
1342
  const vulkan = hasVulkanDll();
1010
- // Single file
1011
1343
  const stat = statSync(targetPath);
1012
1344
  if (stat.isFile()) {
1013
1345
  const info = await probeFile(targetPath);
@@ -1032,7 +1364,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1032
1364
  }],
1033
1365
  };
1034
1366
  }
1035
- // Folder scan
1036
1367
  const files = getFiles(targetPath, false);
1037
1368
  if (files.length === 0) {
1038
1369
  return { content: [{ type: "text", text: `No supported media files found in: ${targetPath}` }], isError: true };
@@ -1043,13 +1374,12 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1043
1374
  if (info)
1044
1375
  results.push(info);
1045
1376
  }
1046
- // Sort
1047
1377
  if (sortBy === "name")
1048
1378
  results.sort((a, b) => a.fileName.localeCompare(b.fileName));
1049
1379
  else if (sortBy === "size")
1050
1380
  results.sort((a, b) => a.sizeMb - b.sizeMb);
1051
1381
  else
1052
- results.sort((a, b) => a.durationSec - b.durationSec); // duration default
1382
+ results.sort((a, b) => a.durationSec - b.durationSec);
1053
1383
  const totalDuration = results.reduce((acc, r) => acc + r.durationSec, 0);
1054
1384
  const totalSize = results.reduce((acc, r) => acc + r.sizeMb, 0);
1055
1385
  const transcribedCount = results.filter(r => existsSync(r.filePath.replace(/\.[^.]+$/, ".txt"))).length;
@@ -1084,7 +1414,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1084
1414
  const gpus = await detectGpus();
1085
1415
  let gpuLines = "";
1086
1416
  if (gpus.length === 0) {
1087
- gpuLines = "⚠️ No GPU detected via wmic — this may indicate a driver issue.\n";
1417
+ gpuLines = "ℹ️ GPU name unavailable (wmic returned nothing — it is deprecated/removed on Windows 11 24H2+). This does NOT mean acceleration is off; the Vulkan check below determines actual GPU use.\n";
1088
1418
  }
1089
1419
  else {
1090
1420
  for (const gpu of gpus) {
@@ -1151,7 +1481,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1151
1481
  const useCase = known ? known.useCase : "Unknown model";
1152
1482
  return `${isActive ? "●" : "○"} ${f}${isActive}${quantTag}\n Size: ${sizeMb} | ${useCase}`;
1153
1483
  });
1154
- // Also list downloadable models not yet installed
1155
1484
  const installedFilenames = new Set(files);
1156
1485
  const available = MODEL_REGISTRY
1157
1486
  .filter(m => !installedFilenames.has(m.filename))
@@ -1188,7 +1517,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1188
1517
  isError: true,
1189
1518
  };
1190
1519
  }
1191
- // Security: enforce URL whitelist — never download from arbitrary URLs
1192
1520
  const urlOk = ALLOWED_HF_PREFIXES.some(prefix => entry.url.startsWith(prefix));
1193
1521
  if (!urlOk) {
1194
1522
  return {
@@ -1216,7 +1544,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1216
1544
  }],
1217
1545
  };
1218
1546
  }
1219
- // Download using Node.js built-in https — no external dependencies
1220
1547
  try {
1221
1548
  const https = await import("https");
1222
1549
  const fs = await import("fs");
@@ -1225,10 +1552,8 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1225
1552
  const file = fs.createWriteStream(tmpPath);
1226
1553
  function doRequest(url) {
1227
1554
  https.get(url, (res) => {
1228
- // Follow redirects (Hugging Face uses redirects)
1229
1555
  if ((res.statusCode === 301 || res.statusCode === 302 || res.statusCode === 307) && res.headers.location) {
1230
1556
  const redirectUrl = res.headers.location;
1231
- // Security: ensure redirect stays within allowed domains
1232
1557
  const redirectOk = ALLOWED_HF_PREFIXES.some(p => redirectUrl.startsWith(p))
1233
1558
  || redirectUrl.startsWith("https://cdn-lfs.huggingface.co/")
1234
1559
  || redirectUrl.startsWith("https://cdn-lfs-us-1.huggingface.co/");
@@ -1244,14 +1569,30 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1244
1569
  return;
1245
1570
  }
1246
1571
  res.pipe(file);
1247
- // Wait for close callback before renaming — Windows requires the file
1248
- // handle to be fully released before renameSync will succeed.
1249
1572
  file.on("finish", () => {
1250
1573
  file.close((closeErr) => {
1251
1574
  if (closeErr) {
1252
1575
  reject(closeErr);
1253
1576
  return;
1254
1577
  }
1578
+ // Integrity: if the server declared a Content-Length, reject a short/truncated
1579
+ // download (dropped connection) BEFORE promoting .part → final, so a partial
1580
+ // file can never become the "installed" model. (Full SHA256 verification is a
1581
+ // separate follow-up requiring verified per-model digests.)
1582
+ const expectedLen = parseInt(res.headers["content-length"] ?? "", 10);
1583
+ let actualLen = 0;
1584
+ try {
1585
+ actualLen = fs.statSync(tmpPath).size;
1586
+ }
1587
+ catch { }
1588
+ if (Number.isFinite(expectedLen) && expectedLen > 0 && actualLen !== expectedLen) {
1589
+ try {
1590
+ fs.unlinkSync(tmpPath);
1591
+ }
1592
+ catch { }
1593
+ reject(new Error(`Incomplete download: wrote ${actualLen} of ${expectedLen} bytes (connection dropped?). Re-run download_model to retry.`));
1594
+ return;
1595
+ }
1255
1596
  try {
1256
1597
  fs.renameSync(tmpPath, destPath);
1257
1598
  resolve();
@@ -1295,27 +1636,26 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1295
1636
  const modelInput = args?.model_name?.trim();
1296
1637
  if (!modelInput)
1297
1638
  return { content: [{ type: "text", text: "model_name is required." }], isError: true };
1298
- // Security: must end in .bin
1299
1639
  if (!modelInput.endsWith(".bin")) {
1300
1640
  return {
1301
1641
  content: [{ type: "text", text: `Invalid model: "${modelInput}"\nModel files must end in .bin` }],
1302
1642
  isError: true,
1303
1643
  };
1304
1644
  }
1305
- // Security: reject path traversal
1306
1645
  if (UNSAFE_PATH_RE.test(modelInput)) {
1307
1646
  return {
1308
1647
  content: [{ type: "text", text: `Invalid path: "${modelInput}"\nPaths containing ".." or UNC paths are not allowed.` }],
1309
1648
  isError: true,
1310
1649
  };
1311
1650
  }
1312
- // Resolve to full path — either absolute or relative to models dir
1313
1651
  const modelsDir = dirname(WHISPER_MODEL);
1314
- const resolvedPath = modelInput.includes("\\") || modelInput.includes("/")
1652
+ // Normalize to an absolute, canonical path first so the containment check and
1653
+ // every downstream use (existsSync, basename, WHISPER_MODEL assignment) operate
1654
+ // on a clean path — never a relative-to-cwd or sibling-prefix string.
1655
+ const resolvedPath = resolve(modelInput.includes("\\") || modelInput.includes("/")
1315
1656
  ? modelInput
1316
- : join(modelsDir, modelInput);
1317
- // Security: must live within the configured models directory
1318
- if (!resolvedPath.startsWith(modelsDir)) {
1657
+ : join(modelsDir, modelInput));
1658
+ if (!isInsideDir(resolvedPath, modelsDir)) {
1319
1659
  return {
1320
1660
  content: [{ type: "text", text: `Security error: model must be within the configured models directory (${modelsDir}).` }],
1321
1661
  isError: true,
@@ -1331,7 +1671,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1331
1671
  isError: true,
1332
1672
  };
1333
1673
  }
1334
- // Process lock — don't switch mid-transcription
1335
1674
  if (await isWhisperRunning()) {
1336
1675
  return {
1337
1676
  content: [{ type: "text", text: "Cannot switch model while a transcription is in progress. Wait for the current job to finish first." }],
@@ -1361,32 +1700,38 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1361
1700
  const filePath = args?.file_path;
1362
1701
  const model = args?.model || WHISPER_MODEL;
1363
1702
  const language = args?.language || "en";
1364
- const outputFormat = (args?.output_format || "text");
1703
+ const outputFormat = (args?.output_format || "timestamps");
1365
1704
  const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
1366
- const saveToFile = args?.save_to_file || false;
1705
+ const saveToFile = args?.save_to_file ?? true;
1367
1706
  const background = args?.background || false;
1368
- // v2.2.0 quality and control params
1707
+ // Effective privacy mode: per-call param wins over global env var
1708
+ const privacyModeParam = args?.privacy_mode;
1709
+ const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
1710
+ // v2.3.0 quality and control params
1369
1711
  const extraOpts = {};
1370
1712
  if (args?.temperature !== undefined)
1371
- extraOpts.temperature = Number(args.temperature);
1713
+ extraOpts.temperature = coerceNum(args.temperature);
1372
1714
  if (args?.prompt)
1373
1715
  extraOpts.prompt = String(args.prompt);
1374
1716
  if (args?.condition_on_prev_text !== undefined)
1375
1717
  extraOpts.conditionOnPrevText = Boolean(args.condition_on_prev_text);
1376
1718
  if (args?.no_speech_thold !== undefined)
1377
- extraOpts.noSpeechThold = Number(args.no_speech_thold);
1719
+ extraOpts.noSpeechThold = coerceNum(args.no_speech_thold);
1378
1720
  if (args?.beam_size !== undefined)
1379
- extraOpts.beamSize = Number(args.beam_size);
1721
+ extraOpts.beamSize = coerceNum(args.beam_size);
1380
1722
  if (args?.best_of !== undefined)
1381
- extraOpts.bestOf = Number(args.best_of);
1382
- if (args?.gpu_device !== undefined)
1383
- extraOpts.gpuDevice = Number(args.gpu_device);
1723
+ extraOpts.bestOf = coerceNum(args.best_of);
1724
+ {
1725
+ const g = resolveGpuDevice(args?.gpu_device);
1726
+ if (g !== undefined)
1727
+ extraOpts.gpuDevice = g;
1728
+ }
1384
1729
  if (args?.processors !== undefined)
1385
- extraOpts.processors = Number(args.processors);
1730
+ extraOpts.processors = coerceNum(args.processors);
1386
1731
  if (args?.word_timestamps)
1387
1732
  extraOpts.wordTimestamps = true;
1388
1733
  if (args?.max_segment_length !== undefined)
1389
- extraOpts.maxLen = Number(args.max_segment_length);
1734
+ extraOpts.maxLen = coerceNum(args.max_segment_length);
1390
1735
  if (args?.split_on_word)
1391
1736
  extraOpts.splitOnWord = true;
1392
1737
  if (args?.diarize)
@@ -1394,9 +1739,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1394
1739
  if (args?.vad_model)
1395
1740
  extraOpts.vadModel = String(args.vad_model);
1396
1741
  if (args?.offset_t !== undefined)
1397
- extraOpts.offsetT = Number(args.offset_t);
1742
+ extraOpts.offsetT = coerceNum(args.offset_t);
1398
1743
  if (args?.duration !== undefined)
1399
- extraOpts.duration = Number(args.duration);
1744
+ extraOpts.duration = coerceNum(args.duration);
1400
1745
  if (!filePath)
1401
1746
  return { content: [{ type: "text", text: "file_path is required." }], isError: true };
1402
1747
  const pathError = validateInputPath(filePath);
@@ -1409,6 +1754,14 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1409
1754
  return { content: [{ type: "text", text: configError }], isError: true };
1410
1755
  // Background mode — detached process, returns immediately
1411
1756
  if (background) {
1757
+ // Privacy mode: gate fires BEFORE spawning. No audio processes until confirmed.
1758
+ // Non-privacy mode: consent gate is intentionally deferred to check_progress.
1759
+ // At this point no transcript exists yet — there is nothing to gate. The gate
1760
+ // fires at check_progress completion when transcript text would first be returned
1761
+ // to the API. Audio processing begins immediately after this point in non-privacy mode.
1762
+ if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
1763
+ return { content: [{ type: "text", text: privacyGateBlock() }] };
1764
+ }
1412
1765
  if (await isWhisperRunning()) {
1413
1766
  return {
1414
1767
  content: [{ type: "text", text: "Transcription already in progress. Wait for the current job to finish before starting another." }],
@@ -1416,15 +1769,17 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1416
1769
  };
1417
1770
  }
1418
1771
  try {
1419
- const { jobId, pid } = await spawnDetached(filePath, model, language, threads, outputFormat === "srt" ? "srt" : "text", extraOpts);
1772
+ const bgFormat = outputFormat === "json" ? "text" : outputFormat;
1773
+ const { jobId, pid } = await spawnDetached(filePath, model, language, threads, bgFormat, extraOpts, undefined, effectivePrivacyMode);
1420
1774
  return {
1421
1775
  content: [{
1422
1776
  type: "text",
1423
1777
  text: `⏳ Background transcription started.\n\n` +
1424
1778
  `Source: ${basename(filePath)}\n` +
1425
1779
  `Job ID: ${jobId}\n` +
1426
- `PID: ${pid}\n\n` +
1427
- `Call check_progress with job_id="${jobId}" to monitor progress.\n` +
1780
+ `PID: ${pid}\n` +
1781
+ (effectivePrivacyMode ? `Privacy mode: active — metadata only will be returned\n` : "") +
1782
+ `\nCall check_progress with job_id="${jobId}" to monitor progress.\n` +
1428
1783
  `Output will be saved to: ${filePath.replace(/\.[^.]+$/, ".txt")}`,
1429
1784
  }],
1430
1785
  };
@@ -1433,14 +1788,54 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1433
1788
  return { content: [{ type: "text", text: `Failed to start background job:\n\n${err?.message || String(err)}` }], isError: true };
1434
1789
  }
1435
1790
  }
1791
+ // Foreground timeout guard: a long file blows Claude Desktop's ~4-min MCP timeout in blocking
1792
+ // mode (the call errors even though the transcript finishes on disk). Probe the duration and
1793
+ // route to background BEFORE running into the wall. Skipped when ffprobe can't read the file
1794
+ // (probe returns null) — it never blocks a transcribe it cannot measure.
1795
+ {
1796
+ const info = await probeFile(filePath);
1797
+ if (info && estimateSec(info.durationSec, hasVulkanDll()) > FOREGROUND_MAX_SEC) {
1798
+ return { content: [{ type: "text", text: `⏱️ "${basename(filePath)}" is ~${formatDuration(info.durationSec)} long — a foreground transcription is estimated around ${estimateTime(info.durationSec, hasVulkanDll())}, which would likely exceed Claude Desktop's 4-minute tool timeout (the transcript would still finish on disk, but this call would error out first).\n\n` +
1799
+ `Run it in the background instead — returns a job ID immediately, then poll check_progress:\n` +
1800
+ ` transcribe_audio with file_path="${filePath}" and background=true\n\n` +
1801
+ `(Shorter files still run inline. Adjust the cutoff with WHISPER_FOREGROUND_MAX_SEC.)` }] };
1802
+ }
1803
+ }
1436
1804
  // Blocking mode (default)
1805
+ if (effectivePrivacyMode) {
1806
+ // Privacy mode: gate fires before every operation.
1807
+ if (checkPrivacyGate(opKeyFor(name, args))) {
1808
+ return { content: [{ type: "text", text: privacyGateBlock() }] };
1809
+ }
1810
+ // Gate passed — proceed to transcription, return metadata only.
1811
+ }
1812
+ else {
1813
+ // Non-privacy mode: session consent gate fires once before first transcript return.
1814
+ // Nothing is processed until user confirms.
1815
+ const policy = transcriptPolicy();
1816
+ if (policy === "consent_gate") {
1817
+ return { content: [{ type: "text", text: consentGateBlock() }] };
1818
+ }
1819
+ }
1437
1820
  try {
1438
1821
  const result = await transcribeSingle(filePath, model, language, outputFormat, threads, saveToFile, extraOpts);
1822
+ if (effectivePrivacyMode) {
1823
+ const savedPath = result.savedTo ?? filePath.replace(/\.[^.]+$/, ".txt");
1824
+ if (!result.savedTo && outputFormat !== "srt" && outputFormat !== "vtt" && outputFormat !== "json") {
1825
+ try {
1826
+ writeFileSync(savedPath, result.text, "utf8");
1827
+ }
1828
+ catch { }
1829
+ }
1830
+ const displayPath = result.srtPath ?? result.savedTo ?? savedPath;
1831
+ return { content: [{ type: "text", text: privacyModeBlock(basename(filePath), displayPath, result.text) }] };
1832
+ }
1833
+ // allow — return transcript normally
1439
1834
  let response = result.text;
1440
1835
  if (result.savedTo)
1441
1836
  response += `\n\n[Transcript saved to: ${result.savedTo}]`;
1442
1837
  if (result.srtPath)
1443
- response += `\n\n[SRT subtitle file saved to: ${result.srtPath}]`;
1838
+ response += `\n\n[Subtitle file saved to: ${result.srtPath}]`;
1444
1839
  return { content: [{ type: "text", text: response }] };
1445
1840
  }
1446
1841
  catch (err) {
@@ -1452,10 +1847,11 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1452
1847
  // -------------------------------------------------------------------------
1453
1848
  if (name === "check_progress") {
1454
1849
  const jobId = args?.job_id;
1850
+ const privacyModeParam = args?.privacy_mode;
1455
1851
  if (!jobId)
1456
1852
  return { content: [{ type: "text", text: "job_id is required." }], isError: true };
1457
1853
  try {
1458
- const result = await readJobProgress(jobId);
1854
+ const result = await readJobProgress(jobId, privacyModeParam);
1459
1855
  return { content: [{ type: "text", text: result }] };
1460
1856
  }
1461
1857
  catch (err) {
@@ -1469,6 +1865,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1469
1865
  const folderPath = args?.folder_path;
1470
1866
  const language = args?.language || "en";
1471
1867
  const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
1868
+ const outputFormat = (args?.output_format || "timestamps");
1869
+ const privacyModeParam = args?.privacy_mode;
1870
+ const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
1472
1871
  if (!folderPath)
1473
1872
  return { content: [{ type: "text", text: "folder_path is required." }], isError: true };
1474
1873
  const pathError = validateInputPath(folderPath);
@@ -1479,16 +1878,19 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1479
1878
  const configError = validatePaths();
1480
1879
  if (configError)
1481
1880
  return { content: [{ type: "text", text: configError }], isError: true };
1881
+ // Privacy gate: fires once before batch starts. All files then process unattended.
1882
+ // Gating per-file in an unattended batch would defeat the purpose of start_batch.
1883
+ if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
1884
+ return { content: [{ type: "text", text: privacyGateBlock() }] };
1885
+ }
1482
1886
  if (await isWhisperRunning()) {
1483
1887
  return { content: [{ type: "text", text: "A transcription is already running. Wait for it to finish before starting a batch." }], isError: true };
1484
1888
  }
1485
- // Scan for untranscribed files
1486
1889
  const allFiles = getFiles(folderPath, false);
1487
1890
  const untranscribed = allFiles.filter(f => !existsSync(f.replace(/\.[^.]+$/, ".txt")));
1488
1891
  if (untranscribed.length === 0) {
1489
1892
  return { content: [{ type: "text", text: `✅ All files in ${folderPath} are already transcribed. Nothing to do.` }] };
1490
1893
  }
1491
- // Probe durations for sorting
1492
1894
  const batchFiles = [];
1493
1895
  for (const f of untranscribed) {
1494
1896
  const info = await probeFile(f);
@@ -1501,7 +1903,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1501
1903
  }
1502
1904
  batchFiles.sort((a, b) => a.durationSec - b.durationSec);
1503
1905
  ensureJobsDir();
1504
- const batchId = `batch_${Date.now()}`;
1906
+ const batchId = `batch_${Date.now()}_${randomUUID().slice(0, 8)}`;
1505
1907
  const batchPath = join(JOBS_DIR, `${batchId}.batch.json`);
1506
1908
  const state = {
1507
1909
  batchId,
@@ -1514,8 +1916,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1514
1916
  model: WHISPER_MODEL,
1515
1917
  language,
1516
1918
  threads,
1919
+ outputFormat,
1920
+ privacyMode: effectivePrivacyMode,
1517
1921
  };
1518
- writeFileSync(batchPath, JSON.stringify(state, null, 2), "utf8");
1922
+ writeJsonAtomic(batchPath, state);
1519
1923
  await spawnNextBatchJob(state);
1520
1924
  const totalDuration = batchFiles.reduce((acc, f) => acc + f.durationSec, 0);
1521
1925
  return {
@@ -1526,8 +1930,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1526
1930
  `Folder: ${folderPath}\n` +
1527
1931
  `Files to process: ${batchFiles.length}\n` +
1528
1932
  `Total audio: ${formatDuration(totalDuration)}\n` +
1529
- `Est. GPU time: ${estimateTime(totalDuration, hasVulkanDll())}\n\n` +
1530
- `First file: ${batchFiles[0].fileName}\n\n` +
1933
+ `Est. GPU time: ${estimateTime(totalDuration, hasVulkanDll())}\n` +
1934
+ (effectivePrivacyMode ? `Privacy mode: active — metadata only will be returned\n` : "") +
1935
+ `\nFirst file: ${batchFiles[0].fileName}\n\n` +
1531
1936
  `Call check_batch_progress with batch_id="${batchId}" to monitor.`,
1532
1937
  }],
1533
1938
  };
@@ -1553,23 +1958,29 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1553
1958
  if (name === "generate_subtitles") {
1554
1959
  const filePath = args?.file_path;
1555
1960
  const language = args?.language || "en";
1961
+ const subtitleFormat = (args?.output_format || "srt");
1556
1962
  const translateToEnglish = args?.translate_to_english || false;
1557
1963
  const background = args?.background || false;
1558
1964
  const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
1559
- // v2.2.0 quality params
1965
+ // v2.3.0 quality params
1560
1966
  const extraOpts = {};
1561
1967
  if (args?.temperature !== undefined)
1562
- extraOpts.temperature = Number(args.temperature);
1968
+ extraOpts.temperature = coerceNum(args.temperature);
1563
1969
  if (args?.prompt)
1564
1970
  extraOpts.prompt = String(args.prompt);
1565
1971
  if (args?.beam_size !== undefined)
1566
- extraOpts.beamSize = Number(args.beam_size);
1972
+ extraOpts.beamSize = coerceNum(args.beam_size);
1567
1973
  if (args?.best_of !== undefined)
1568
- extraOpts.bestOf = Number(args.best_of);
1974
+ extraOpts.bestOf = coerceNum(args.best_of);
1569
1975
  if (args?.diarize)
1570
1976
  extraOpts.diarize = true;
1571
1977
  if (args?.vad_model)
1572
1978
  extraOpts.vadModel = String(args.vad_model);
1979
+ {
1980
+ const g = resolveGpuDevice(args?.gpu_device);
1981
+ if (g !== undefined)
1982
+ extraOpts.gpuDevice = g;
1983
+ }
1573
1984
  if (!filePath)
1574
1985
  return { content: [{ type: "text", text: "file_path is required." }], isError: true };
1575
1986
  const pathError = validateInputPath(filePath);
@@ -1583,10 +1994,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1583
1994
  if (await isWhisperRunning()) {
1584
1995
  return { content: [{ type: "text", text: "Transcription already in progress. Wait for it to finish first." }], isError: true };
1585
1996
  }
1586
- // Background mode — detached SRT job
1587
1997
  if (background) {
1588
1998
  try {
1589
- const { jobId, pid } = await spawnDetached(filePath, WHISPER_MODEL, language, threads, "srt", extraOpts);
1999
+ const { jobId, pid } = await spawnDetached(filePath, WHISPER_MODEL, language, threads, subtitleFormat, extraOpts);
1590
2000
  return {
1591
2001
  content: [{
1592
2002
  type: "text",
@@ -1594,6 +2004,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1594
2004
  `Source: ${basename(filePath)}\n` +
1595
2005
  `Job ID: ${jobId}\n` +
1596
2006
  `PID: ${pid}\n` +
2007
+ `Format: ${subtitleFormat.toUpperCase()}\n` +
1597
2008
  `Language: ${language}\n\n` +
1598
2009
  `Call check_progress with job_id="${jobId}" to monitor.\n` +
1599
2010
  `Note: translate_to_english is not available in background mode. ` +
@@ -1605,53 +2016,67 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1605
2016
  return { content: [{ type: "text", text: `Failed to start background subtitle job:\n\n${err?.message || String(err)}` }], isError: true };
1606
2017
  }
1607
2018
  }
2019
+ // Foreground timeout guard (same as transcribe_audio): subtitle generation runs MULTIPLE
2020
+ // whisper passes inline (auto-detect + native + optional translation), so a long file is even
2021
+ // more likely to blow the 4-min MCP timeout. Route to background before running into the wall.
2022
+ {
2023
+ const info = await probeFile(filePath);
2024
+ if (info && estimateSec(info.durationSec, hasVulkanDll()) > FOREGROUND_MAX_SEC) {
2025
+ return { content: [{ type: "text", text: `⏱️ "${basename(filePath)}" is ~${formatDuration(info.durationSec)} long — foreground subtitle generation (estimated ~${estimateTime(info.durationSec, hasVulkanDll())}, plus extra passes for auto-detect/translation) would likely exceed Claude Desktop's 4-minute tool timeout.\n\n` +
2026
+ `Run it in the background instead:\n` +
2027
+ ` generate_subtitles with file_path="${filePath}" and background=true\n\n` +
2028
+ `(translate_to_english isn't available in background mode — run a second pass after it completes. Adjust the cutoff with WHISPER_FOREGROUND_MAX_SEC.)` }] };
2029
+ }
2030
+ }
1608
2031
  try {
1609
- // Convert to WAV if needed
1610
2032
  let transcribeFrom = filePath;
1611
2033
  let tmpFile = null;
1612
2034
  if (needsConversion(filePath)) {
1613
2035
  tmpFile = await convertToWav(filePath);
2036
+ activeTempFiles.add(tmpFile);
1614
2037
  transcribeFrom = tmpFile;
1615
2038
  }
1616
2039
  const baseNoExt = filePath.replace(/\.[^.]+$/, "");
1617
- // Auto-detect language if requested
2040
+ const ext = subtitleFormat === "vtt" ? ".vtt" : ".srt";
1618
2041
  let detectedLang = language;
1619
2042
  if (language === "auto") {
1620
- const detected = await detectLanguage(transcribeFrom, WHISPER_MODEL, threads);
2043
+ const detected = await detectLanguage(transcribeFrom, WHISPER_MODEL, threads, extraOpts.gpuDevice);
1621
2044
  detectedLang = detected ?? "en";
1622
2045
  }
1623
2046
  const results = [];
1624
- // Pass 1 — native language SRT
1625
- const nativeSrt = language === "en" || detectedLang === "en"
1626
- ? `${baseNoExt}.srt`
1627
- : `${baseNoExt}.${detectedLang}.srt`;
1628
- await runSrtPass(transcribeFrom, nativeSrt, WHISPER_MODEL, detectedLang, threads, false, extraOpts);
1629
- results.push(`✅ Native (${detectedLang}): ${nativeSrt}`);
1630
- // Pass 2 — English translation SRT (only if language isn't already English)
2047
+ const nativeDest = language === "en" || detectedLang === "en"
2048
+ ? `${baseNoExt}${ext}`
2049
+ : `${baseNoExt}.${detectedLang}${ext}`;
2050
+ await runSubtitlePass(transcribeFrom, nativeDest, subtitleFormat, WHISPER_MODEL, detectedLang, threads, false, extraOpts);
2051
+ results.push(`✅ Native (${detectedLang}): ${nativeDest}`);
1631
2052
  if (translateToEnglish && detectedLang !== "en") {
1632
- const englishSrt = `${baseNoExt}.en.srt`;
1633
- await runSrtPass(transcribeFrom, englishSrt, WHISPER_MODEL, detectedLang, threads, true, extraOpts);
1634
- results.push(`✅ English translation: ${englishSrt}`);
2053
+ const englishDest = `${baseNoExt}.en${ext}`;
2054
+ await runSubtitlePass(transcribeFrom, englishDest, subtitleFormat, WHISPER_MODEL, detectedLang, threads, true, extraOpts);
2055
+ results.push(`✅ English translation: ${englishDest}`);
2056
+ }
2057
+ if (tmpFile) {
2058
+ activeTempFiles.delete(tmpFile);
2059
+ if (existsSync(tmpFile))
2060
+ try {
2061
+ unlinkSync(tmpFile);
2062
+ }
2063
+ catch { }
1635
2064
  }
1636
- // Clean up temp WAV
1637
- if (tmpFile && existsSync(tmpFile))
1638
- try {
1639
- unlinkSync(tmpFile);
1640
- }
1641
- catch { }
1642
2065
  const langNote = language === "auto"
1643
2066
  ? `Auto-detected language: ${detectedLang}\n\n`
1644
2067
  : "";
2068
+ const playerNote = subtitleFormat === "vtt"
2069
+ ? `Load in web players, HTML5 <video>, or any player that supports WebVTT.`
2070
+ : `Load in VLC via Subtitle → Add Subtitle File → select the .srt file.`;
1645
2071
  return {
1646
2072
  content: [{
1647
2073
  type: "text",
1648
2074
  text: `✅ Subtitle file(s) generated!\n\n` +
1649
2075
  langNote +
1650
2076
  results.join("\n") + "\n\n" +
1651
- `To use in VLC: Subtitle → Add Subtitle File → select the .srt file.\n` +
1652
- `Works in any video player that supports external subtitles.\n\n` +
2077
+ playerNote + "\n\n" +
1653
2078
  `Note: whisper's built-in translation only translates to English. ` +
1654
- `For other target languages, translate the .srt file contents separately.`,
2079
+ `For other target languages, translate the subtitle file contents separately.`,
1655
2080
  }],
1656
2081
  };
1657
2082
  }
@@ -1660,7 +2085,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1660
2085
  }
1661
2086
  }
1662
2087
  // -------------------------------------------------------------------------
1663
- // transcribe_batch (interactive only)
2088
+ // transcribe_batch (interactive)
1664
2089
  // -------------------------------------------------------------------------
1665
2090
  if (name === "transcribe_batch") {
1666
2091
  const folderPath = args?.folder_path;
@@ -1668,6 +2093,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1668
2093
  const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
1669
2094
  const recursive = args?.recursive || false;
1670
2095
  const fileIndex = args?.file_index;
2096
+ const outputFormat = (args?.output_format || "timestamps");
2097
+ const privacyModeParam = args?.privacy_mode;
2098
+ const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
1671
2099
  if (!folderPath)
1672
2100
  return { content: [{ type: "text", text: "folder_path is required." }], isError: true };
1673
2101
  if (!existsSync(folderPath))
@@ -1684,7 +2112,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1684
2112
  }],
1685
2113
  };
1686
2114
  }
1687
- // No file_index: return file list
1688
2115
  if (fileIndex === undefined) {
1689
2116
  return {
1690
2117
  content: [{
@@ -1696,11 +2123,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1696
2123
  return ` ${i + 1}. ${basename(f)}${done}`;
1697
2124
  }).join("\n") +
1698
2125
  `\n\nTo start, say "transcribe file 1" (or any number). I'll process one file at a time and wait for your go-ahead before continuing.\n` +
1699
- `\nFor large unattended batches, see the command line approach in TROUBLESHOOTING.md.`,
2126
+ `\nFor large unattended batches, use start_batch instead.`,
1700
2127
  }],
1701
2128
  };
1702
2129
  }
1703
- // Process the requested file
1704
2130
  const idx = fileIndex - 1;
1705
2131
  if (idx < 0 || idx >= files.length) {
1706
2132
  return { content: [{ type: "text", text: `Invalid file number. Choose between 1 and ${files.length}.` }], isError: true };
@@ -1709,17 +2135,42 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1709
2135
  const fileName = basename(filePath);
1710
2136
  const txtPath = filePath.replace(/\.[^.]+$/, ".txt");
1711
2137
  try {
1712
- const result = await transcribeSingle(filePath, WHISPER_MODEL, language, "text", threads, true, {});
2138
+ // v2.3.0: Privacy gate fires before each file in privacy mode.
2139
+ // Each file in transcribe_batch is a separate tool call, so each
2140
+ // gets its own confirmation — correct for interactive mode.
2141
+ if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
2142
+ return { content: [{ type: "text", text: privacyGateBlock() }] };
2143
+ }
2144
+ // Non-privacy mode: session consent gate fires once before first file.
2145
+ if (!effectivePrivacyMode) {
2146
+ const policy = transcriptPolicy();
2147
+ if (policy === "consent_gate") {
2148
+ return { content: [{ type: "text", text: consentGateBlock() }] };
2149
+ }
2150
+ }
2151
+ const result = await transcribeSingle(filePath, WHISPER_MODEL, language, outputFormat, threads, true, {});
1713
2152
  const remaining = files.length - fileIndex;
1714
2153
  const nextMsg = remaining > 0
1715
2154
  ? `\n\n${remaining} file(s) remaining. Say "continue" or "transcribe file ${fileIndex + 1}" to proceed, or "stop" to finish.`
1716
2155
  : `\n\n✅ That was the last file. Batch complete!`;
2156
+ let bodyText;
2157
+ if (effectivePrivacyMode) {
2158
+ const words = estimateWordCount(result.text);
2159
+ bodyText =
2160
+ `Saved to: ${txtPath}\n` +
2161
+ `Words: ~${words}\n\n` +
2162
+ `Privacy mode active — transcript not returned to Claude's API.`;
2163
+ }
2164
+ else {
2165
+ bodyText =
2166
+ `Saved to: ${txtPath}\n\n` +
2167
+ `Preview:\n${result.text.slice(0, 500)}${result.text.length > 500 ? "..." : ""}`;
2168
+ }
1717
2169
  return {
1718
2170
  content: [{
1719
2171
  type: "text",
1720
2172
  text: `[${fileIndex}/${files.length}] ✅ ${fileName}\n\n` +
1721
- `Saved to: ${txtPath}\n\n` +
1722
- `Preview:\n${result.text.slice(0, 500)}${result.text.length > 500 ? "..." : ""}` +
2173
+ bodyText +
1723
2174
  nextMsg,
1724
2175
  }],
1725
2176
  };
@@ -1742,9 +2193,12 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
1742
2193
  // Start
1743
2194
  // ---------------------------------------------------------------------------
1744
2195
  async function main() {
2196
+ cleanupOldJobFiles();
2197
+ process.on("SIGINT", () => gracefulShutdown("SIGINT"));
2198
+ process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
1745
2199
  const transport = new StdioServerTransport();
1746
2200
  await server.connect(transport);
1747
- console.error(`whisper-windows-mcp v2.2.0 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS}`);
2201
+ console.error(`whisper-windows-mcp v2.4.0 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS} | privacy: ${WHISPER_PRIVACY_MODE ? "on" : "off"}`);
1748
2202
  }
1749
2203
  main().catch((err) => {
1750
2204
  console.error("Fatal error:", err);