whisper-windows-mcp 2.2.2 → 2.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +24 -0
- package/.github/workflows/publish.yml +24 -0
- package/{LICENSE-COMMERCIAL.md → COMMERCIAL-LICENSE.md} +58 -58
- package/LICENSE +40 -40
- package/PRIVACY.es.md +194 -135
- package/PRIVACY.id.md +194 -135
- package/PRIVACY.ja.md +194 -135
- package/PRIVACY.ko.md +194 -135
- package/PRIVACY.md +194 -135
- package/PRIVACY.pl.md +194 -135
- package/PRIVACY.pt-BR.md +194 -135
- package/PRIVACY.ro.md +194 -135
- package/PRIVACY.uk.md +194 -135
- package/PRIVACY.vi.md +194 -135
- package/README.es.md +80 -48
- package/README.id.md +83 -40
- package/README.ja.md +106 -72
- package/README.ko.md +69 -37
- package/README.md +82 -39
- package/README.pl.md +83 -40
- package/README.pt-BR.md +77 -45
- package/README.ro.md +84 -41
- package/README.uk.md +83 -40
- package/README.vi.md +73 -41
- package/ROADMAP.es.md +131 -45
- package/ROADMAP.id.md +86 -89
- package/ROADMAP.ja.md +93 -108
- package/ROADMAP.ko.md +82 -82
- package/ROADMAP.pl.md +125 -41
- package/ROADMAP.pt-BR.md +87 -87
- package/ROADMAP.ro.md +123 -41
- package/ROADMAP.uk.md +82 -90
- package/ROADMAP.vi.md +87 -87
- package/SECURITY.es.md +76 -47
- package/SECURITY.id.md +76 -47
- package/SECURITY.ja.md +76 -47
- package/SECURITY.ko.md +76 -47
- package/SECURITY.md +33 -4
- package/SECURITY.pl.md +76 -47
- package/SECURITY.pt-BR.md +76 -47
- package/SECURITY.ro.md +76 -47
- package/SECURITY.uk.md +76 -47
- package/SECURITY.vi.md +76 -47
- package/TROUBLESHOOTING.es.md +325 -323
- package/TROUBLESHOOTING.id.md +349 -323
- package/TROUBLESHOOTING.ja.md +415 -286
- package/TROUBLESHOOTING.ko.md +325 -323
- package/TROUBLESHOOTING.pl.md +371 -323
- package/TROUBLESHOOTING.pt-BR.md +325 -323
- package/TROUBLESHOOTING.ro.md +371 -323
- package/TROUBLESHOOTING.uk.md +385 -323
- package/TROUBLESHOOTING.vi.md +325 -323
- package/dist/index.js +743 -289
- package/dist/lib.d.ts +37 -0
- package/dist/lib.js +123 -0
- package/package.json +46 -45
- package/patch_roadmaps.py +0 -72
package/dist/index.js
CHANGED
|
@@ -10,8 +10,10 @@ import { CallToolRequestSchema, ListToolsRequestSchema, } from "@modelcontextpro
|
|
|
10
10
|
import { execFile, spawn } from "child_process";
|
|
11
11
|
import { existsSync, unlinkSync, readdirSync, writeFileSync, readFileSync, mkdirSync, openSync, closeSync, statSync, } from "fs";
|
|
12
12
|
import { cpus, tmpdir } from "os";
|
|
13
|
-
import { join, extname, basename, dirname } from "path";
|
|
13
|
+
import { join, extname, basename, dirname, resolve } from "path";
|
|
14
14
|
import { promisify } from "util";
|
|
15
|
+
import { randomUUID } from "crypto";
|
|
16
|
+
import { coerceNum, writeJsonAtomic, estimateWordCount, opKeyFor, isInsideDir, extractTranscriptFromLog, parseLastTimestamp, formatDuration, estimateSec, estimateTime, } from "./lib.js";
|
|
15
17
|
const execFileAsync = promisify(execFile);
|
|
16
18
|
// ---------------------------------------------------------------------------
|
|
17
19
|
// Configuration
|
|
@@ -23,6 +25,74 @@ const FFMPEG_PATH = process.env.FFMPEG_PATH ?? "ffmpeg";
|
|
|
23
25
|
const SYSTEM_THREADS = cpus().length;
|
|
24
26
|
const DEFAULT_THREADS = Math.max(2, Math.floor(SYSTEM_THREADS / 2));
|
|
25
27
|
const WHISPER_THREADS = parseInt(process.env.WHISPER_THREADS ?? String(DEFAULT_THREADS), 10);
|
|
28
|
+
// Optional global default GPU/Vulkan device index passed to whisper-cli as --device N.
|
|
29
|
+
// Lets a multi-GPU box pin a specific card without passing gpu_device on every call.
|
|
30
|
+
// Per-call gpu_device overrides this; unset → whisper-cli's own default (device 0).
|
|
31
|
+
// ⚠ This is the Vulkan ENUMERATION index (whisper-cli logs "ggml_vulkan: 0 = <name>"),
|
|
32
|
+
// which is NOT guaranteed to match Windows GPU0/GPU1 — read the startup log to pick correctly.
|
|
33
|
+
const _whisperGpuEnv = process.env.WHISPER_GPU_DEVICE;
|
|
34
|
+
const WHISPER_GPU_DEVICE = _whisperGpuEnv !== undefined && _whisperGpuEnv.trim() !== "" && !Number.isNaN(parseInt(_whisperGpuEnv, 10))
|
|
35
|
+
? parseInt(_whisperGpuEnv, 10)
|
|
36
|
+
: undefined;
|
|
37
|
+
// Foreground transcription guard: if the estimated time (fixed model-load cost + transcribe) exceeds
|
|
38
|
+
// this many seconds, a blocking (foreground) run is refused and routed to background mode instead —
|
|
39
|
+
// avoiding a silent timeout against Claude Desktop's ~4-minute (240s) MCP tool-call ceiling.
|
|
40
|
+
// Default 210 leaves ~30s headroom under the wall. Configurable via WHISPER_FOREGROUND_MAX_SEC.
|
|
41
|
+
const FOREGROUND_MAX_SEC = parseInt(process.env.WHISPER_FOREGROUND_MAX_SEC ?? "210", 10) || 210;
|
|
42
|
+
/** Effective GPU device: a numeric per-call arg wins, else the WHISPER_GPU_DEVICE env default, else undefined. */
|
|
43
|
+
function resolveGpuDevice(arg) {
|
|
44
|
+
if (arg !== undefined) {
|
|
45
|
+
const n = Number(arg);
|
|
46
|
+
if (!Number.isNaN(n))
|
|
47
|
+
return n;
|
|
48
|
+
}
|
|
49
|
+
return WHISPER_GPU_DEVICE;
|
|
50
|
+
}
|
|
51
|
+
// Temp WAVs from BLOCKING transcriptions only — never detached-job temps (a running
|
|
52
|
+
// background whisper-cli still needs its WAV). Cleaned best-effort on graceful shutdown.
|
|
53
|
+
const activeTempFiles = new Set();
|
|
54
|
+
function gracefulShutdown(signal) {
|
|
55
|
+
let cleaned = 0;
|
|
56
|
+
for (const f of activeTempFiles) {
|
|
57
|
+
try {
|
|
58
|
+
if (existsSync(f)) {
|
|
59
|
+
unlinkSync(f);
|
|
60
|
+
cleaned++;
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
catch { /* best effort */ }
|
|
64
|
+
}
|
|
65
|
+
console.error(`whisper-windows-mcp: ${signal} — cleaned ${cleaned} blocking temp file(s), exiting.`);
|
|
66
|
+
process.exit(0);
|
|
67
|
+
}
|
|
68
|
+
// ---------------------------------------------------------------------------
|
|
69
|
+
// Privacy configuration
|
|
70
|
+
// ---------------------------------------------------------------------------
|
|
71
|
+
/**
|
|
72
|
+
* Global default: when true, all tool responses return metadata only
|
|
73
|
+
* (filename, word count, save path). No transcript text appears in any tool
|
|
74
|
+
* response or API call. Transcripts are still saved as local .txt files.
|
|
75
|
+
* Required for HIPAA, GDPR, legal, financial, and NDA-protected content.
|
|
76
|
+
*
|
|
77
|
+
* Can be overridden per-call using the privacy_mode parameter on
|
|
78
|
+
* transcribe_audio, transcribe_batch, start_batch, and check_progress.
|
|
79
|
+
* Per-call override wins in either direction — no restart required to toggle.
|
|
80
|
+
*
|
|
81
|
+
* Set as global default in claude_desktop_config.json env section:
|
|
82
|
+
* "WHISPER_PRIVACY_MODE": "true"
|
|
83
|
+
*/
|
|
84
|
+
const WHISPER_PRIVACY_MODE = (process.env.WHISPER_PRIVACY_MODE ?? "false").toLowerCase() === "true";
|
|
85
|
+
/**
|
|
86
|
+
* When true: skips the one-time first-use consent disclosure shown before
|
|
87
|
+
* transcript text is sent to Claude's API. Set this once you understand the
|
|
88
|
+
* privacy boundary and no longer need the reminder each session.
|
|
89
|
+
* Has no effect when privacy mode is active — privacy mode uses its own
|
|
90
|
+
* per-operation gate that always fires regardless of this setting.
|
|
91
|
+
*
|
|
92
|
+
* Set in claude_desktop_config.json env section:
|
|
93
|
+
* "WHISPER_CONSENT_ACKNOWLEDGED": "true"
|
|
94
|
+
*/
|
|
95
|
+
const WHISPER_CONSENT_ACKNOWLEDGED = (process.env.WHISPER_CONSENT_ACKNOWLEDGED ?? "false").toLowerCase() === "true";
|
|
26
96
|
const SUPPORTED_EXTENSIONS = [
|
|
27
97
|
".mp3", ".wav",
|
|
28
98
|
".mp4", ".mkv", ".avi", ".mov", ".webm", ".flv", ".wmv", ".m4v",
|
|
@@ -34,8 +104,190 @@ const MAX_FILE_SIZE_MB = 10240; // 10 GB
|
|
|
34
104
|
// Security: patterns rejected in all file_path inputs.
|
|
35
105
|
const UNSAFE_PATH_RE = /(\.\.[/\\])|(^\\\\)/; // blocks .. traversal and UNC paths
|
|
36
106
|
// ---------------------------------------------------------------------------
|
|
37
|
-
//
|
|
107
|
+
// Privacy helpers
|
|
38
108
|
// ---------------------------------------------------------------------------
|
|
109
|
+
/**
|
|
110
|
+
* Shown once per session before any transcript text is returned in non-privacy mode.
|
|
111
|
+
* Contains explicit instructions for Claude to surface to the user and
|
|
112
|
+
* halt until explicit user confirmation is received.
|
|
113
|
+
*/
|
|
114
|
+
const CONSENT_DISCLOSURE = [
|
|
115
|
+
"⚠️ PRIVACY DISCLOSURE — EXPLICIT USER CONFIRMATION REQUIRED",
|
|
116
|
+
"",
|
|
117
|
+
"━".repeat(60),
|
|
118
|
+
"INSTRUCTION FOR CLAUDE: Present this disclosure to the user in full.",
|
|
119
|
+
"Ask: \"Do you confirm? (yes / no)\"",
|
|
120
|
+
"Do NOT call this tool again until the user has explicitly replied yes.",
|
|
121
|
+
"━".repeat(60),
|
|
122
|
+
"",
|
|
123
|
+
"Transcript text is about to be transmitted to Anthropic's API and processed",
|
|
124
|
+
"externally. Once sent, it leaves your local machine.",
|
|
125
|
+
"",
|
|
126
|
+
"Your audio and video files never leave your machine. Transcript text is",
|
|
127
|
+
"different — if returned to Claude, it is sent through the API.",
|
|
128
|
+
"",
|
|
129
|
+
"You must confirm before proceeding if this recording contains:",
|
|
130
|
+
" • Medical or patient information (HIPAA — 45 CFR §164)",
|
|
131
|
+
" • Legal proceedings or attorney-client communications (privilege)",
|
|
132
|
+
" • Personal data of EU/EEA residents (GDPR — Articles 13, 14, 28)",
|
|
133
|
+
" • Student educational records (FERPA — 20 U.S.C. §1232g)",
|
|
134
|
+
" • Financial or cardholder data (PCI-DSS, SOX)",
|
|
135
|
+
" • Confidential business information (trade secrets, NDA-protected)",
|
|
136
|
+
" • Any recording subject to contractual confidentiality",
|
|
137
|
+
"",
|
|
138
|
+
"For recordings that require full local privacy, enable privacy mode:",
|
|
139
|
+
" Per-call: pass privacy_mode=true to any transcription tool",
|
|
140
|
+
" Global: add to claude_desktop_config.json env:",
|
|
141
|
+
" \"WHISPER_PRIVACY_MODE\": \"true\"",
|
|
142
|
+
"",
|
|
143
|
+
"To suppress this confirmation permanently for non-sensitive content:",
|
|
144
|
+
" \"WHISPER_CONSENT_ACKNOWLEDGED\": \"true\"",
|
|
145
|
+
"",
|
|
146
|
+
"━".repeat(60),
|
|
147
|
+
"INSTRUCTION FOR CLAUDE: Ask the user the following question exactly,",
|
|
148
|
+
"then wait for their response before taking any further action:",
|
|
149
|
+
"",
|
|
150
|
+
"\"⚠️ Before I return any transcript text, I need your explicit confirmation.",
|
|
151
|
+
"Transcript content will be transmitted to Anthropic's API and processed",
|
|
152
|
+
"externally — it will leave your local machine. Do you confirm? (yes / no)\"",
|
|
153
|
+
"━".repeat(60),
|
|
154
|
+
].join("\n");
|
|
155
|
+
/**
|
|
156
|
+
* Shown before every operation when privacy mode is active.
|
|
157
|
+
* Identical text every time by design — repetition is the point.
|
|
158
|
+
* Regulatory compliance requires informed consent before each operation.
|
|
159
|
+
*/
|
|
160
|
+
const PRIVACY_MODE_DISCLOSURE = [
|
|
161
|
+
"🔒 PRIVACY MODE — CONFIRMATION REQUIRED BEFORE THIS OPERATION",
|
|
162
|
+
"",
|
|
163
|
+
"━".repeat(60),
|
|
164
|
+
"INSTRUCTION FOR CLAUDE: Present this disclosure to the user in full.",
|
|
165
|
+
"Ask: \"Do you confirm? (yes / no)\"",
|
|
166
|
+
"Do NOT call this tool again until the user has explicitly replied yes.",
|
|
167
|
+
"━".repeat(60),
|
|
168
|
+
"",
|
|
169
|
+
"WHISPER_PRIVACY_MODE is active for this operation.",
|
|
170
|
+
"",
|
|
171
|
+
"What will happen:",
|
|
172
|
+
" ✓ Audio/video will be transcribed LOCALLY on your machine",
|
|
173
|
+
" ✓ Transcript saved as a local file only — not returned to Claude's API",
|
|
174
|
+
" ✓ Raw audio and video files NEVER leave your machine",
|
|
175
|
+
" ✓ No transcript text will be transmitted to Anthropic under any circumstances",
|
|
176
|
+
"",
|
|
177
|
+
"What you must confirm:",
|
|
178
|
+
" ⚠ Audio processing begins on your local machine after you confirm",
|
|
179
|
+
" ⚠ This confirmation is required before every operation in privacy mode",
|
|
180
|
+
" ⚠ Do not disable privacy mode mid-session for regulated or sensitive material",
|
|
181
|
+
"",
|
|
182
|
+
"To disable per-call (non-sensitive content only): pass privacy_mode=false",
|
|
183
|
+
"To disable globally: set WHISPER_PRIVACY_MODE=false and restart Claude Desktop",
|
|
184
|
+
"",
|
|
185
|
+
"━".repeat(60),
|
|
186
|
+
"INSTRUCTION FOR CLAUDE: Ask the user the following question exactly,",
|
|
187
|
+
"then wait for their response before taking any further action:",
|
|
188
|
+
"",
|
|
189
|
+
"\"🔒 Privacy mode is active. Audio will be transcribed locally and no transcript",
|
|
190
|
+
"text will be sent to Anthropic's API. Confirm you want to proceed? (yes / no)\"",
|
|
191
|
+
"━".repeat(60),
|
|
192
|
+
].join("\n");
|
|
193
|
+
// Per-operation privacy gate state.
|
|
194
|
+
// Each distinct operation (identified by a stable key over its tool name + arguments)
|
|
195
|
+
// arms independently: first call for that key shows the disclosure and blocks; the
|
|
196
|
+
// second call with the SAME key clears it and proceeds. Keying per-operation closes
|
|
197
|
+
// the v2.3.0 hole where a single global flag let one operation's confirmation be
|
|
198
|
+
// silently consumed by a different operation. Completely independent of
|
|
199
|
+
// sessionConsentGiven — serves different users and modes.
|
|
200
|
+
const privacyArmed = new Map(); // opKey -> armed-at epoch ms
|
|
201
|
+
// Armed disclosures expire so an abandoned confirmation can never satisfy a later
|
|
202
|
+
// operation, and the map can never grow without bound on a long-lived server.
|
|
203
|
+
const PRIVACY_GATE_TTL_MS = 10 * 60 * 1000;
|
|
204
|
+
// Session-scoped consent tracking — resets each time Claude Desktop restarts
|
|
205
|
+
// the MCP server process. Pre-set from env var so users who have set
|
|
206
|
+
// WHISPER_CONSENT_ACKNOWLEDGED=true skip the gate entirely.
|
|
207
|
+
// Has no effect when privacy mode is active (privacy mode uses its own gate).
|
|
208
|
+
let sessionConsentGiven = WHISPER_CONSENT_ACKNOWLEDGED;
|
|
209
|
+
/**
|
|
210
|
+
* Pre-transcription gate for privacy mode, scoped to a single operation by opKey.
|
|
211
|
+
* Call this when effective privacy mode is active, BEFORE any audio processing.
|
|
212
|
+
*
|
|
213
|
+
* Returns true → block this call, show PRIVACY_MODE_DISCLOSURE to user.
|
|
214
|
+
* Returns false → user has confirmed THIS operation, proceed.
|
|
215
|
+
*
|
|
216
|
+
* Mechanism: first call for opKey arms it (block); the second call with the same
|
|
217
|
+
* opKey clears it (allow). Each distinct operation is independent — confirming one
|
|
218
|
+
* can never satisfy another. Stale arms older than PRIVACY_GATE_TTL_MS are evicted
|
|
219
|
+
* on every call. Only call when effective privacy mode is active.
|
|
220
|
+
*/
|
|
221
|
+
function checkPrivacyGate(opKey) {
|
|
222
|
+
const now = Date.now();
|
|
223
|
+
for (const [k, armedAt] of privacyArmed) {
|
|
224
|
+
if (now - armedAt > PRIVACY_GATE_TTL_MS)
|
|
225
|
+
privacyArmed.delete(k);
|
|
226
|
+
}
|
|
227
|
+
if (!privacyArmed.has(opKey)) {
|
|
228
|
+
privacyArmed.set(opKey, now);
|
|
229
|
+
return true; // first sight of this exact operation — block, show disclosure
|
|
230
|
+
}
|
|
231
|
+
privacyArmed.delete(opKey);
|
|
232
|
+
return false; // same operation re-issued — user confirmed — allow
|
|
233
|
+
}
|
|
234
|
+
/** Returns the privacy mode disclosure as a tool response. */
|
|
235
|
+
function privacyGateBlock() {
|
|
236
|
+
return PRIVACY_MODE_DISCLOSURE;
|
|
237
|
+
}
|
|
238
|
+
/**
|
|
239
|
+
* Determines post-transcription transcript policy for non-privacy mode.
|
|
240
|
+
* Only call this after confirming effective privacy mode is OFF.
|
|
241
|
+
*
|
|
242
|
+
* Returns:
|
|
243
|
+
* "consent_gate" — First transcript-returning call this session; show
|
|
244
|
+
* disclosure and withhold text. Flips sessionConsentGiven
|
|
245
|
+
* so subsequent calls proceed without re-prompting.
|
|
246
|
+
* "allow" — Consent already given; return text normally.
|
|
247
|
+
*
|
|
248
|
+
* IMPORTANT: Only call when the tool is about to return actual transcript text
|
|
249
|
+
* (i.e. job confirmed complete). Do NOT call for still-running jobs or error
|
|
250
|
+
* paths — it would consume the consent gate without returning content.
|
|
251
|
+
*/
|
|
252
|
+
function transcriptPolicy() {
|
|
253
|
+
if (!sessionConsentGiven) {
|
|
254
|
+
sessionConsentGiven = true;
|
|
255
|
+
return "consent_gate";
|
|
256
|
+
}
|
|
257
|
+
return "allow";
|
|
258
|
+
}
|
|
259
|
+
/**
|
|
260
|
+
* Metadata-only response used when privacy mode is active.
|
|
261
|
+
* Returns file info and word count. No transcript text included.
|
|
262
|
+
*/
|
|
263
|
+
function privacyModeBlock(fileName, savedPath, text) {
|
|
264
|
+
const words = estimateWordCount(text);
|
|
265
|
+
return (`✅ Transcription complete — privacy mode active.\n\n` +
|
|
266
|
+
`Source: ${fileName}\n` +
|
|
267
|
+
`Words: ~${words}\n` +
|
|
268
|
+
`Saved: ${savedPath}\n\n` +
|
|
269
|
+
`Transcript text is not transmitted to Claude's API.\n` +
|
|
270
|
+
`Access your transcript directly at the path above.`);
|
|
271
|
+
}
|
|
272
|
+
/**
|
|
273
|
+
* Consent gate response block — shown on first transcript-returning call
|
|
274
|
+
* in non-privacy mode. savedPath and text are optional: when called BEFORE
|
|
275
|
+
* transcription (blocking mode), neither is available. When called AFTER
|
|
276
|
+
* (background jobs via check_progress), both are present.
|
|
277
|
+
*/
|
|
278
|
+
function consentGateBlock(savedPath, text) {
|
|
279
|
+
const lines = [CONSENT_DISCLOSURE, ""];
|
|
280
|
+
if (savedPath || text) {
|
|
281
|
+
lines.push("─".repeat(60));
|
|
282
|
+
if (savedPath)
|
|
283
|
+
lines.push(`Saved: ${savedPath}`);
|
|
284
|
+
if (text)
|
|
285
|
+
lines.push(`Words: ~${estimateWordCount(text)}`);
|
|
286
|
+
lines.push("");
|
|
287
|
+
}
|
|
288
|
+
lines.push("No transcript text has been returned. Reply 'yes' to confirm, then call the tool again to proceed.");
|
|
289
|
+
return lines.join("\n");
|
|
290
|
+
}
|
|
39
291
|
function validatePaths() {
|
|
40
292
|
if (!existsSync(WHISPER_CLI_PATH))
|
|
41
293
|
return `whisper-cli.exe not found at: ${WHISPER_CLI_PATH}\nCheck WHISPER_CLI_PATH in claude_desktop_config.json`;
|
|
@@ -66,7 +318,6 @@ function validateInputPath(filePath) {
|
|
|
66
318
|
/**
|
|
67
319
|
* Check whether a whisper-cli.exe process is already running.
|
|
68
320
|
* Uses tasklist /FI which is available on all Windows versions.
|
|
69
|
-
* Returns true if found, false if not (or if tasklist itself fails).
|
|
70
321
|
*/
|
|
71
322
|
async function isWhisperRunning() {
|
|
72
323
|
try {
|
|
@@ -74,20 +325,51 @@ async function isWhisperRunning() {
|
|
|
74
325
|
return stdout.toLowerCase().includes("whisper-cli.exe");
|
|
75
326
|
}
|
|
76
327
|
catch {
|
|
77
|
-
// If tasklist fails for any reason, assume safe to proceed
|
|
78
328
|
return false;
|
|
79
329
|
}
|
|
80
330
|
}
|
|
81
331
|
// ---------------------------------------------------------------------------
|
|
82
|
-
// Background job architecture
|
|
332
|
+
// Background job architecture
|
|
83
333
|
// ---------------------------------------------------------------------------
|
|
84
334
|
const JOBS_DIR = join(tmpdir(), "whisper-mcp-jobs");
|
|
335
|
+
// Mutex: prevents double-spawn when the exit handler and a concurrent
|
|
336
|
+
// check_batch_progress call both detect job completion simultaneously.
|
|
337
|
+
let batchSpawning = false;
|
|
338
|
+
/**
|
|
339
|
+
* Delete .json and .log job files older than 7 days from the jobs directory.
|
|
340
|
+
* Non-blocking — runs once at startup, errors are ignored.
|
|
341
|
+
*/
|
|
342
|
+
function cleanupOldJobFiles() {
|
|
343
|
+
try {
|
|
344
|
+
if (!existsSync(JOBS_DIR))
|
|
345
|
+
return;
|
|
346
|
+
const cutoff = Date.now() - 7 * 24 * 60 * 60 * 1000;
|
|
347
|
+
const files = readdirSync(JOBS_DIR);
|
|
348
|
+
let cleaned = 0;
|
|
349
|
+
for (const file of files) {
|
|
350
|
+
if (!file.endsWith(".json") && !file.endsWith(".log"))
|
|
351
|
+
continue;
|
|
352
|
+
const fullPath = join(JOBS_DIR, file);
|
|
353
|
+
try {
|
|
354
|
+
if (statSync(fullPath).mtimeMs < cutoff) {
|
|
355
|
+
unlinkSync(fullPath);
|
|
356
|
+
cleaned++;
|
|
357
|
+
}
|
|
358
|
+
}
|
|
359
|
+
catch { /* ignore per-file errors */ }
|
|
360
|
+
}
|
|
361
|
+
if (cleaned > 0) {
|
|
362
|
+
console.error(`whisper-windows-mcp: cleaned ${cleaned} old job file(s) from ${JOBS_DIR}`);
|
|
363
|
+
}
|
|
364
|
+
}
|
|
365
|
+
catch { /* non-blocking, ignore all errors */ }
|
|
366
|
+
}
|
|
85
367
|
function ensureJobsDir() {
|
|
86
368
|
mkdirSync(JOBS_DIR, { recursive: true });
|
|
87
369
|
}
|
|
88
|
-
async function spawnDetached(filePath, model, language, threads, outputFormat = "
|
|
370
|
+
async function spawnDetached(filePath, model, language, threads, outputFormat = "timestamps", extraOpts = {}, onExit, privacyMode = false) {
|
|
89
371
|
ensureJobsDir();
|
|
90
|
-
const jobId = `job_${Date.now()}`;
|
|
372
|
+
const jobId = `job_${Date.now()}_${randomUUID().slice(0, 8)}`;
|
|
91
373
|
const logPath = join(JOBS_DIR, `${jobId}.log`);
|
|
92
374
|
const jobPath = join(JOBS_DIR, `${jobId}.json`);
|
|
93
375
|
// Convert to WAV first if needed (fast, blocking)
|
|
@@ -98,13 +380,17 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
98
380
|
isTmp = true;
|
|
99
381
|
}
|
|
100
382
|
// Use a clean ASCII job-ID-based output path to avoid Unicode filename issues.
|
|
101
|
-
//
|
|
383
|
+
// readJobProgress moves the file to the correct destination after completion.
|
|
102
384
|
const tmpOutputBase = join(JOBS_DIR, jobId);
|
|
103
|
-
// Determine final destination path
|
|
385
|
+
// Determine final destination path and file extension
|
|
104
386
|
const sourceBase = filePath.replace(/\.[^.]+$/, "");
|
|
105
|
-
const ext = outputFormat === "srt" ? ".srt"
|
|
106
|
-
|
|
107
|
-
|
|
387
|
+
const ext = outputFormat === "srt" ? ".srt"
|
|
388
|
+
: outputFormat === "vtt" ? ".vtt"
|
|
389
|
+
: outputFormat === "lrc" ? ".lrc"
|
|
390
|
+
: outputFormat === "csv" ? ".csv"
|
|
391
|
+
: ".txt";
|
|
392
|
+
const outputPath = (outputFormat === "srt" || outputFormat === "vtt") && language !== "en" && language !== "auto"
|
|
393
|
+
? `${sourceBase}.${language}${ext}`
|
|
108
394
|
: `${sourceBase}${ext}`;
|
|
109
395
|
// Build args using shared options — ensures quality flags are always applied
|
|
110
396
|
// in background mode, matching blocking mode behaviour.
|
|
@@ -114,10 +400,7 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
114
400
|
"-f", transcribeFrom,
|
|
115
401
|
"-l", lang,
|
|
116
402
|
"-t", String(threads),
|
|
117
|
-
// Hallucination prevention — must be in background mode too.
|
|
118
|
-
// --max-context 0 prevents conditioning on prior segment output.
|
|
119
403
|
...(extraOpts.conditionOnPrevText ? [] : ["--max-context", "0"]),
|
|
120
|
-
// Confirmed valid flag (-nth). Suppresses silent segments from hallucinating.
|
|
121
404
|
"--no-speech-thold", String(extraOpts.noSpeechThold ?? 0.6),
|
|
122
405
|
];
|
|
123
406
|
if (extraOpts.temperature !== undefined)
|
|
@@ -129,7 +412,7 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
129
412
|
if (extraOpts.bestOf !== undefined)
|
|
130
413
|
args.push("--best-of", String(extraOpts.bestOf));
|
|
131
414
|
if (extraOpts.gpuDevice !== undefined)
|
|
132
|
-
args.push("
|
|
415
|
+
args.push("--device", String(extraOpts.gpuDevice));
|
|
133
416
|
if (extraOpts.processors !== undefined && extraOpts.processors > 1)
|
|
134
417
|
args.push("-p", String(extraOpts.processors));
|
|
135
418
|
if (extraOpts.offsetT !== undefined)
|
|
@@ -149,13 +432,24 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
149
432
|
if (extraOpts.splitOnWord)
|
|
150
433
|
args.push("--split-on-word");
|
|
151
434
|
}
|
|
152
|
-
// Output format
|
|
435
|
+
// Output format flags
|
|
153
436
|
if (outputFormat === "srt") {
|
|
154
437
|
args.push("-osrt", "-of", tmpOutputBase);
|
|
155
438
|
}
|
|
156
|
-
else {
|
|
439
|
+
else if (outputFormat === "vtt") {
|
|
440
|
+
args.push("-ovtt", "-of", tmpOutputBase);
|
|
441
|
+
}
|
|
442
|
+
else if (outputFormat === "lrc") {
|
|
443
|
+
args.push("-olrc", "-of", tmpOutputBase);
|
|
444
|
+
}
|
|
445
|
+
else if (outputFormat === "csv") {
|
|
446
|
+
args.push("-ocsv", "-of", tmpOutputBase);
|
|
447
|
+
}
|
|
448
|
+
else if (outputFormat === "text") {
|
|
157
449
|
args.push("-otxt", "-of", tmpOutputBase);
|
|
158
450
|
}
|
|
451
|
+
// "timestamps": no output file flag — stdout (redirected to log) contains
|
|
452
|
+
// the timestamped transcript. extractTranscriptFromLog() recovers it on completion.
|
|
159
453
|
// Spawn detached, redirect stdout+stderr to log file
|
|
160
454
|
const logFd = openSync(logPath, "w");
|
|
161
455
|
const child = spawn(WHISPER_CLI_PATH, args, {
|
|
@@ -164,6 +458,11 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
164
458
|
windowsHide: true,
|
|
165
459
|
});
|
|
166
460
|
closeSync(logFd);
|
|
461
|
+
// Attach exit handler BEFORE unref so the batch can self-advance without polling.
|
|
462
|
+
// child.once fires exactly once when the process exits. unref() still applies —
|
|
463
|
+
// Node won't be kept alive just for this child.
|
|
464
|
+
if (onExit)
|
|
465
|
+
child.once("exit", onExit);
|
|
167
466
|
child.unref();
|
|
168
467
|
const pid = child.pid ?? 0;
|
|
169
468
|
const job = {
|
|
@@ -183,8 +482,9 @@ async function spawnDetached(filePath, model, language, threads, outputFormat =
|
|
|
183
482
|
threads,
|
|
184
483
|
durationSec: 0,
|
|
185
484
|
status: "running",
|
|
485
|
+
privacyMode,
|
|
186
486
|
};
|
|
187
|
-
|
|
487
|
+
writeJsonAtomic(jobPath, job);
|
|
188
488
|
return { jobId, pid };
|
|
189
489
|
}
|
|
190
490
|
async function isPidRunning(pid) {
|
|
@@ -196,48 +496,68 @@ async function isPidRunning(pid) {
|
|
|
196
496
|
return false;
|
|
197
497
|
}
|
|
198
498
|
}
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
const sec = parseInt(m[1], 10) * 3600 + parseInt(m[2], 10) * 60 + parseInt(m[3], 10);
|
|
206
|
-
if (sec > lastSec)
|
|
207
|
-
lastSec = sec;
|
|
208
|
-
}
|
|
209
|
-
return lastSec;
|
|
210
|
-
}
|
|
211
|
-
async function readJobProgress(jobId) {
|
|
499
|
+
/**
|
|
500
|
+
* Read job progress and return a status string.
|
|
501
|
+
* privacyModeOverride: per-call override from check_progress privacy_mode param.
|
|
502
|
+
* Wins over job.privacyMode if provided; job.privacyMode wins over global env var.
|
|
503
|
+
*/
|
|
504
|
+
async function readJobProgress(jobId, privacyModeOverride) {
|
|
212
505
|
const jobPath = join(JOBS_DIR, `${jobId}.json`);
|
|
213
506
|
if (!existsSync(jobPath)) {
|
|
214
507
|
return `❌ Job not found: ${jobId}\n\nThe job file may have been deleted or the ID is incorrect.`;
|
|
215
508
|
}
|
|
216
509
|
const job = JSON.parse(readFileSync(jobPath, "utf8"));
|
|
217
|
-
// Read log
|
|
218
510
|
let logContent = "";
|
|
219
511
|
if (existsSync(job.logPath)) {
|
|
220
512
|
logContent = readFileSync(job.logPath, "utf8");
|
|
221
513
|
}
|
|
222
514
|
const lastSec = parseLastTimestamp(logContent);
|
|
223
515
|
const isRunning = await isPidRunning(job.pid);
|
|
224
|
-
const ext = job.outputFormat === "srt" ? ".srt"
|
|
516
|
+
const ext = job.outputFormat === "srt" ? ".srt"
|
|
517
|
+
: job.outputFormat === "vtt" ? ".vtt"
|
|
518
|
+
: job.outputFormat === "lrc" ? ".lrc"
|
|
519
|
+
: job.outputFormat === "csv" ? ".csv"
|
|
520
|
+
: ".txt";
|
|
225
521
|
const tmpOutput = `${job.tmpOutputBase}${ext}`;
|
|
226
|
-
const outputExists = existsSync(job.outputPath) ||
|
|
522
|
+
const outputExists = existsSync(job.outputPath) ||
|
|
523
|
+
existsSync(tmpOutput) ||
|
|
524
|
+
(job.outputFormat === "timestamps" && existsSync(job.logPath));
|
|
227
525
|
// Completed
|
|
228
526
|
if (!isRunning && outputExists) {
|
|
229
|
-
// Move
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
527
|
+
// Move or create final output file
|
|
528
|
+
if (job.outputFormat === "timestamps") {
|
|
529
|
+
if (!existsSync(job.outputPath) && existsSync(job.logPath)) {
|
|
530
|
+
const transcript = extractTranscriptFromLog(readFileSync(job.logPath, "utf8"));
|
|
531
|
+
if (transcript) {
|
|
532
|
+
try {
|
|
533
|
+
writeFileSync(job.outputPath, transcript, "utf8");
|
|
534
|
+
}
|
|
535
|
+
catch (e) {
|
|
536
|
+
console.error(`whisper-windows-mcp: failed to write transcript to ${job.outputPath}: ${e?.message}`);
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
}
|
|
540
|
+
}
|
|
541
|
+
else if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
|
|
233
542
|
try {
|
|
234
543
|
writeFileSync(job.outputPath, readFileSync(tmpOutput, "utf8"), "utf8");
|
|
235
544
|
unlinkSync(tmpOutput);
|
|
236
545
|
}
|
|
237
|
-
catch {
|
|
546
|
+
catch (moveErr) {
|
|
547
|
+
console.error(`whisper-windows-mcp: failed to move output to ${job.outputPath}: ${moveErr?.message}`);
|
|
548
|
+
}
|
|
549
|
+
}
|
|
550
|
+
// Bug 2 fix: explicit check that the output file landed where expected.
|
|
551
|
+
if (!existsSync(job.outputPath)) {
|
|
552
|
+
job.status = "failed";
|
|
553
|
+
writeJsonAtomic(job.jobPath, job);
|
|
554
|
+
return (`❌ Output file write failed.\n\n` +
|
|
555
|
+
`Transcription completed but the output could not be written to:\n${job.outputPath}\n\n` +
|
|
556
|
+
`Check disk space and directory permissions.\n` +
|
|
557
|
+
`Raw job data may be in: ${JOBS_DIR}`);
|
|
238
558
|
}
|
|
239
559
|
job.status = "complete";
|
|
240
|
-
|
|
560
|
+
writeJsonAtomic(job.jobPath, job);
|
|
241
561
|
// Clean up tmp wav if present
|
|
242
562
|
if (job.isTmp && existsSync(job.transcribeFrom)) {
|
|
243
563
|
try {
|
|
@@ -246,18 +566,30 @@ async function readJobProgress(jobId) {
|
|
|
246
566
|
catch { }
|
|
247
567
|
}
|
|
248
568
|
const outputContent = readFileSync(job.outputPath, "utf8").trim();
|
|
249
|
-
|
|
569
|
+
// Effective privacy mode: per-call override → job setting → global env var.
|
|
570
|
+
// transcriptPolicy() is only called for non-privacy mode (consent gate logic).
|
|
571
|
+
// This keeps the two gate systems fully independent.
|
|
572
|
+
const effectivePrivacy = privacyModeOverride ?? job.privacyMode ?? WHISPER_PRIVACY_MODE;
|
|
573
|
+
if (effectivePrivacy) {
|
|
574
|
+
return privacyModeBlock(basename(job.sourceFile), job.outputPath, outputContent);
|
|
575
|
+
}
|
|
576
|
+
const policy = transcriptPolicy();
|
|
577
|
+
if (policy === "consent_gate") {
|
|
578
|
+
return consentGateBlock(job.outputPath, outputContent);
|
|
579
|
+
}
|
|
580
|
+
// allow — return normally with preview
|
|
581
|
+
const preview = job.outputFormat === "srt" || job.outputFormat === "vtt"
|
|
250
582
|
? outputContent.split("\n").slice(0, 20).join("\n")
|
|
251
583
|
: outputContent.slice(0, 600);
|
|
252
584
|
return (`✅ Complete!\n\n` +
|
|
253
585
|
`Source: ${basename(job.sourceFile)}\n` +
|
|
254
586
|
`Output: ${job.outputPath}\n\n` +
|
|
255
|
-
`Preview:\n${preview}${outputContent.length > 600 && job.outputFormat !== "srt" ? "..." : ""}`);
|
|
587
|
+
`Preview:\n${preview}${outputContent.length > 600 && job.outputFormat !== "srt" && job.outputFormat !== "vtt" ? "..." : ""}`);
|
|
256
588
|
}
|
|
257
589
|
// Failed
|
|
258
590
|
if (!isRunning && !outputExists) {
|
|
259
591
|
job.status = "failed";
|
|
260
|
-
|
|
592
|
+
writeJsonAtomic(job.jobPath, job);
|
|
261
593
|
const lastLines = logContent.split(/\r?\n/).filter(l => l.trim()).slice(-5).join("\n");
|
|
262
594
|
return (`❌ Failed or cancelled.\n\n` +
|
|
263
595
|
`Source: ${basename(job.sourceFile)}\n` +
|
|
@@ -290,20 +622,31 @@ function validateTranscript(txtPath, durationSec) {
|
|
|
290
622
|
return { valid: true };
|
|
291
623
|
}
|
|
292
624
|
async function spawnNextBatchJob(state) {
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
state.files[i].
|
|
300
|
-
|
|
301
|
-
|
|
625
|
+
// Mutex: prevents double-spawn between concurrent exit handler + check_batch_progress.
|
|
626
|
+
if (batchSpawning)
|
|
627
|
+
return;
|
|
628
|
+
batchSpawning = true;
|
|
629
|
+
try {
|
|
630
|
+
for (let i = state.currentIndex; i < state.files.length; i++) {
|
|
631
|
+
if (state.files[i].status === "pending") {
|
|
632
|
+
state.currentIndex = i;
|
|
633
|
+
state.files[i].status = "running";
|
|
634
|
+
const f = state.files[i];
|
|
635
|
+
const fmt = (state.outputFormat === "json" ? "text" : state.outputFormat);
|
|
636
|
+
const { jobId } = await spawnDetached(f.filePath, state.model, state.language, state.threads, fmt, {},
|
|
637
|
+
// Exit callback: batch self-advances without polling.
|
|
638
|
+
() => { readBatchProgress(state.batchId).catch(() => { }); }, state.privacyMode);
|
|
639
|
+
state.files[i].jobId = jobId;
|
|
640
|
+
writeJsonAtomic(state.batchPath, state);
|
|
641
|
+
return;
|
|
642
|
+
}
|
|
302
643
|
}
|
|
644
|
+
state.status = "complete";
|
|
645
|
+
writeJsonAtomic(state.batchPath, state);
|
|
646
|
+
}
|
|
647
|
+
finally {
|
|
648
|
+
batchSpawning = false;
|
|
303
649
|
}
|
|
304
|
-
// Nothing left to run
|
|
305
|
-
state.status = "complete";
|
|
306
|
-
writeFileSync(state.batchPath, JSON.stringify(state, null, 2), "utf8");
|
|
307
650
|
}
|
|
308
651
|
async function readBatchProgress(batchId) {
|
|
309
652
|
const batchPath = join(JOBS_DIR, `${batchId}.batch.json`);
|
|
@@ -311,36 +654,45 @@ async function readBatchProgress(batchId) {
|
|
|
311
654
|
return `❌ Batch not found: ${batchId}\n\nThe batch file may have been deleted or the ID is incorrect.`;
|
|
312
655
|
}
|
|
313
656
|
const state = JSON.parse(readFileSync(batchPath, "utf8"));
|
|
314
|
-
// Check current running job
|
|
315
657
|
const running = state.files.find(f => f.status === "running");
|
|
316
658
|
if (running && running.jobId) {
|
|
317
659
|
const jobPath = join(JOBS_DIR, `${running.jobId}.json`);
|
|
318
660
|
if (existsSync(jobPath)) {
|
|
319
661
|
const job = JSON.parse(readFileSync(jobPath, "utf8"));
|
|
320
662
|
const isRunning = await isPidRunning(job.pid);
|
|
321
|
-
const outputExists = existsSync(job.outputPath);
|
|
322
663
|
if (!isRunning) {
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
664
|
+
const ext = job.outputFormat === "srt" ? ".srt"
|
|
665
|
+
: job.outputFormat === "vtt" ? ".vtt"
|
|
666
|
+
: job.outputFormat === "lrc" ? ".lrc"
|
|
667
|
+
: job.outputFormat === "csv" ? ".csv"
|
|
668
|
+
: ".txt";
|
|
328
669
|
const tmpOutput = `${job.tmpOutputBase}${ext}`;
|
|
329
|
-
if (
|
|
670
|
+
if (job.outputFormat === "timestamps") {
|
|
671
|
+
if (!existsSync(job.outputPath) && existsSync(job.logPath)) {
|
|
672
|
+
const transcript = extractTranscriptFromLog(readFileSync(job.logPath, "utf8"));
|
|
673
|
+
if (transcript) {
|
|
674
|
+
try {
|
|
675
|
+
writeFileSync(job.outputPath, transcript, "utf8");
|
|
676
|
+
}
|
|
677
|
+
catch (e) {
|
|
678
|
+
console.error(`whisper-windows-mcp: failed to write transcript to ${job.outputPath}: ${e?.message}`);
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
}
|
|
682
|
+
}
|
|
683
|
+
else if (existsSync(tmpOutput) && tmpOutput !== job.outputPath) {
|
|
330
684
|
try {
|
|
331
685
|
writeFileSync(job.outputPath, readFileSync(tmpOutput, "utf8"), "utf8");
|
|
332
686
|
unlinkSync(tmpOutput);
|
|
333
687
|
}
|
|
334
|
-
catch { /*
|
|
688
|
+
catch { /* validateTranscript will catch missing output */ }
|
|
335
689
|
}
|
|
336
|
-
// Clean up temp WAV if present
|
|
337
690
|
if (job.isTmp && existsSync(job.transcribeFrom)) {
|
|
338
691
|
try {
|
|
339
692
|
unlinkSync(job.transcribeFrom);
|
|
340
693
|
}
|
|
341
694
|
catch { }
|
|
342
695
|
}
|
|
343
|
-
// Job finished — validate and advance
|
|
344
696
|
const finalOutputExists = existsSync(job.outputPath);
|
|
345
697
|
const validation = validateTranscript(job.outputPath, running.durationSec);
|
|
346
698
|
if (finalOutputExists && validation.valid) {
|
|
@@ -350,24 +702,21 @@ async function readBatchProgress(batchId) {
|
|
|
350
702
|
running.status = "failed";
|
|
351
703
|
running.failReason = validation.reason ?? "no output file";
|
|
352
704
|
}
|
|
353
|
-
// Advance to next
|
|
354
705
|
state.currentIndex = state.files.indexOf(running) + 1;
|
|
355
706
|
if (state.files.some(f => f.status === "pending")) {
|
|
356
707
|
await spawnNextBatchJob(state);
|
|
357
708
|
}
|
|
358
709
|
else {
|
|
359
710
|
state.status = "complete";
|
|
360
|
-
|
|
711
|
+
writeJsonAtomic(batchPath, state);
|
|
361
712
|
}
|
|
362
713
|
}
|
|
363
714
|
else {
|
|
364
|
-
|
|
365
|
-
writeFileSync(batchPath, JSON.stringify(state, null, 2), "utf8");
|
|
715
|
+
writeJsonAtomic(batchPath, state);
|
|
366
716
|
}
|
|
367
717
|
}
|
|
368
718
|
}
|
|
369
719
|
else if (state.status !== "complete" && state.files.some(f => f.status === "pending")) {
|
|
370
|
-
// No running job but pending files exist — advance
|
|
371
720
|
await spawnNextBatchJob(state);
|
|
372
721
|
}
|
|
373
722
|
// Build status report
|
|
@@ -453,19 +802,16 @@ function recommendedModel(vramBytes) {
|
|
|
453
802
|
return "base.en (ggml-base.en.bin) — recommended for limited VRAM";
|
|
454
803
|
}
|
|
455
804
|
const MODEL_REGISTRY = [
|
|
456
|
-
// Full-precision English
|
|
457
805
|
{ name: "tiny.en", filename: "ggml-tiny.en.bin", sizeMb: 75, multilingual: false, quantized: false, useCase: "Quick tests, lowest accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.en.bin" },
|
|
458
806
|
{ name: "base.en", filename: "ggml-base.en.bin", sizeMb: 142, multilingual: false, quantized: false, useCase: "Fast English, good accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en.bin" },
|
|
459
807
|
{ name: "small.en", filename: "ggml-small.en.bin", sizeMb: 466, multilingual: false, quantized: false, useCase: "Better English accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.en.bin" },
|
|
460
808
|
{ name: "medium.en", filename: "ggml-medium.en.bin", sizeMb: 1500, multilingual: false, quantized: false, useCase: "High accuracy English, fast on GPU", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.en.bin" },
|
|
461
|
-
// Full-precision multilingual
|
|
462
809
|
{ name: "tiny", filename: "ggml-tiny.bin", sizeMb: 75, multilingual: true, quantized: false, useCase: "Multilingual, minimal accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-tiny.bin" },
|
|
463
810
|
{ name: "base", filename: "ggml-base.bin", sizeMb: 142, multilingual: true, quantized: false, useCase: "Multilingual, fast", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.bin" },
|
|
464
811
|
{ name: "small", filename: "ggml-small.bin", sizeMb: 466, multilingual: true, quantized: false, useCase: "Multilingual, better accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.bin" },
|
|
465
812
|
{ name: "medium", filename: "ggml-medium.bin", sizeMb: 1500, multilingual: true, quantized: false, useCase: "Multilingual, high accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.bin" },
|
|
466
813
|
{ name: "large-v3", filename: "ggml-large-v3.bin", sizeMb: 2900, multilingual: true, quantized: false, useCase: "Best accuracy, multilingual — requires 6GB+ VRAM", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3.bin" },
|
|
467
814
|
{ name: "large-v3-turbo", filename: "ggml-large-v3-turbo.bin", sizeMb: 1600, multilingual: true, quantized: false, useCase: "~6x faster than large-v3, minimal accuracy loss — RECOMMENDED for English GPU batch work", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo.bin" },
|
|
468
|
-
// Quantized variants — smaller, CPU-friendly
|
|
469
815
|
{ name: "base.en-q5_1", filename: "ggml-base.en-q5_1.bin", sizeMb: 57, multilingual: false, quantized: true, useCase: "Tiny English model, CPU-friendly", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-base.en-q5_1.bin" },
|
|
470
816
|
{ name: "small.en-q5_1", filename: "ggml-small.en-q5_1.bin", sizeMb: 181, multilingual: false, quantized: true, useCase: "Fast English, low memory, good for CPU", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-small.en-q5_1.bin" },
|
|
471
817
|
{ name: "medium.en-q5_0", filename: "ggml-medium.en-q5_0.bin", sizeMb: 514, multilingual: false, quantized: true, useCase: "High accuracy English, CPU-friendly — good default for no-GPU systems", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-medium.en-q5_0.bin" },
|
|
@@ -473,7 +819,6 @@ const MODEL_REGISTRY = [
|
|
|
473
819
|
{ name: "large-v3-turbo-q5_0", filename: "ggml-large-v3-turbo-q5_0.bin", sizeMb: 547, multilingual: true, quantized: true, useCase: "RECOMMENDED for CPU-only multilingual — fast, low memory, good accuracy", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo-q5_0.bin" },
|
|
474
820
|
{ name: "large-v3-turbo-q8_0", filename: "ggml-large-v3-turbo-q8_0.bin", sizeMb: 874, multilingual: true, quantized: true, useCase: "Turbo quality closer to full precision, moderate size", url: "https://huggingface.co/ggerganov/whisper.cpp/resolve/main/ggml-large-v3-turbo-q8_0.bin" },
|
|
475
821
|
];
|
|
476
|
-
// Security: only allow downloads from these Hugging Face namespaces.
|
|
477
822
|
const ALLOWED_HF_PREFIXES = [
|
|
478
823
|
"https://huggingface.co/ggerganov/whisper.cpp/",
|
|
479
824
|
"https://huggingface.co/ggml-org/",
|
|
@@ -499,40 +844,12 @@ async function probeFile(filePath) {
|
|
|
499
844
|
const sizeMb = parseInt(fmt.size ?? "0", 10) / (1024 * 1024);
|
|
500
845
|
const bitrate = Math.round(parseInt(fmt.bit_rate ?? "0", 10) / 1000);
|
|
501
846
|
const codec = audioStream?.codec_name ?? fmt.format_name?.split(",")[0] ?? "unknown";
|
|
502
|
-
return {
|
|
503
|
-
filePath,
|
|
504
|
-
fileName: basename(filePath),
|
|
505
|
-
durationSec,
|
|
506
|
-
sizeMb,
|
|
507
|
-
codec,
|
|
508
|
-
bitrate,
|
|
509
|
-
};
|
|
847
|
+
return { filePath, fileName: basename(filePath), durationSec, sizeMb, codec, bitrate };
|
|
510
848
|
}
|
|
511
849
|
catch {
|
|
512
850
|
return null;
|
|
513
851
|
}
|
|
514
852
|
}
|
|
515
|
-
function formatDuration(sec) {
|
|
516
|
-
if (!sec)
|
|
517
|
-
return "?:??";
|
|
518
|
-
const h = Math.floor(sec / 3600);
|
|
519
|
-
const m = Math.floor((sec % 3600) / 60);
|
|
520
|
-
const s = Math.floor(sec % 60);
|
|
521
|
-
if (h > 0)
|
|
522
|
-
return `${h}:${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}`;
|
|
523
|
-
return `${m}:${String(s).padStart(2, "0")}`;
|
|
524
|
-
}
|
|
525
|
-
function estimateTime(durationSec, gpu) {
|
|
526
|
-
if (!durationSec)
|
|
527
|
-
return "?";
|
|
528
|
-
// CPU: ~1.5x realtime on Ryzen 7 2700x with medium.en
|
|
529
|
-
// GPU: ~0.12x realtime on Vega 56 via Vulkan with medium.en (~8x faster than CPU)
|
|
530
|
-
const ratio = gpu ? 0.12 : 1.5;
|
|
531
|
-
const estSec = Math.round(durationSec * ratio);
|
|
532
|
-
if (estSec < 60)
|
|
533
|
-
return `~${estSec}s`;
|
|
534
|
-
return `~${Math.round(estSec / 60)}m`;
|
|
535
|
-
}
|
|
536
853
|
function padEnd(str, len) {
|
|
537
854
|
return str.length >= len ? str.slice(0, len) : str + " ".repeat(len - str.length);
|
|
538
855
|
}
|
|
@@ -543,7 +860,7 @@ function isSupportedFile(filePath) {
|
|
|
543
860
|
return SUPPORTED_EXTENSIONS.includes(extname(filePath).toLowerCase());
|
|
544
861
|
}
|
|
545
862
|
async function convertToWav(inputPath) {
|
|
546
|
-
const tmpFile = join(tmpdir(), `whisper_tmp_${Date.now()}.wav`);
|
|
863
|
+
const tmpFile = join(tmpdir(), `whisper_tmp_${Date.now()}_${randomUUID().slice(0, 8)}.wav`);
|
|
547
864
|
await execFileAsync(FFMPEG_PATH, [
|
|
548
865
|
"-y", "-i", inputPath,
|
|
549
866
|
"-ar", "16000", "-ac", "1", "-c:a", "pcm_s16le", tmpFile,
|
|
@@ -553,14 +870,8 @@ async function convertToWav(inputPath) {
|
|
|
553
870
|
function buildArgs(filePath, model, opts) {
|
|
554
871
|
const lang = opts.language === "auto" ? "auto" : opts.language;
|
|
555
872
|
const args = ["-m", model, "-f", filePath, "-l", lang, "-t", String(opts.threads)];
|
|
556
|
-
// Hallucination prevention — set max context tokens to 0 to prevent whisper
|
|
557
|
-
// from conditioning each segment on its own prior output, which causes
|
|
558
|
-
// repetitive hallucination loops on noisy or silent audio.
|
|
559
|
-
// Flag: --max-context 0 (user can re-enable by setting conditionOnPrevText=true)
|
|
560
873
|
if (!opts.conditionOnPrevText)
|
|
561
874
|
args.push("--max-context", "0");
|
|
562
|
-
// Treat segments below this confidence threshold as silence rather than
|
|
563
|
-
// hallucinating content. Confirmed valid flag in whisper-cli (-nth).
|
|
564
875
|
args.push("--no-speech-thold", String(opts.noSpeechThold ?? 0.6));
|
|
565
876
|
if (opts.translate)
|
|
566
877
|
args.push("--translate");
|
|
@@ -573,7 +884,7 @@ function buildArgs(filePath, model, opts) {
|
|
|
573
884
|
if (opts.bestOf !== undefined)
|
|
574
885
|
args.push("--best-of", String(opts.bestOf));
|
|
575
886
|
if (opts.gpuDevice !== undefined)
|
|
576
|
-
args.push("
|
|
887
|
+
args.push("--device", String(opts.gpuDevice));
|
|
577
888
|
if (opts.processors !== undefined && opts.processors > 1)
|
|
578
889
|
args.push("-p", String(opts.processors));
|
|
579
890
|
if (opts.offsetT !== undefined)
|
|
@@ -582,8 +893,6 @@ function buildArgs(filePath, model, opts) {
|
|
|
582
893
|
args.push("--duration", String(opts.duration));
|
|
583
894
|
if (opts.diarize)
|
|
584
895
|
args.push("--diarize");
|
|
585
|
-
// word_timestamps: sets max-len=1 + split-on-word for per-word output
|
|
586
|
-
// without requiring JSON parsing — simpler than -oj approach.
|
|
587
896
|
if (opts.wordTimestamps) {
|
|
588
897
|
args.push("--max-len", "1", "--split-on-word");
|
|
589
898
|
}
|
|
@@ -593,7 +902,6 @@ function buildArgs(filePath, model, opts) {
|
|
|
593
902
|
if (opts.splitOnWord)
|
|
594
903
|
args.push("--split-on-word");
|
|
595
904
|
}
|
|
596
|
-
// VAD: voice activity detection — strips silence before whisper sees the audio
|
|
597
905
|
if (opts.vadModel && existsSync(opts.vadModel)) {
|
|
598
906
|
args.push("--vad", "--vad-model", opts.vadModel);
|
|
599
907
|
}
|
|
@@ -601,31 +909,37 @@ function buildArgs(filePath, model, opts) {
|
|
|
601
909
|
if (opts.outputFormat === "srt") {
|
|
602
910
|
args.push("-osrt", "-of", filePath.replace(/\.[^.]+$/, ""));
|
|
603
911
|
}
|
|
912
|
+
else if (opts.outputFormat === "vtt") {
|
|
913
|
+
args.push("-ovtt", "-of", filePath.replace(/\.[^.]+$/, ""));
|
|
914
|
+
}
|
|
915
|
+
else if (opts.outputFormat === "lrc") {
|
|
916
|
+
args.push("-olrc", "-of", filePath.replace(/\.[^.]+$/, ""));
|
|
917
|
+
}
|
|
918
|
+
else if (opts.outputFormat === "csv") {
|
|
919
|
+
args.push("-ocsv", "-of", filePath.replace(/\.[^.]+$/, ""));
|
|
920
|
+
}
|
|
604
921
|
else if (opts.outputFormat === "json") {
|
|
605
922
|
args.push("-oj");
|
|
606
923
|
}
|
|
607
924
|
else if (opts.outputFormat === "text") {
|
|
608
925
|
args.push("--no-timestamps");
|
|
609
926
|
}
|
|
610
|
-
// "timestamps"
|
|
927
|
+
// "timestamps": no flag — whisper default stdout includes timestamps
|
|
611
928
|
return args;
|
|
612
929
|
}
|
|
613
|
-
|
|
614
|
-
* Detect the language of a file by running a short whisper probe.
|
|
615
|
-
* Runs whisper on the first 30 seconds only (--duration 30000ms).
|
|
616
|
-
* Returns the detected language code (e.g. "ja", "en") or null on failure.
|
|
617
|
-
*/
|
|
618
|
-
async function detectLanguage(wavPath, model, threads) {
|
|
930
|
+
async function detectLanguage(wavPath, model, threads, gpuDevice) {
|
|
619
931
|
try {
|
|
620
|
-
const
|
|
932
|
+
const dlArgs = [
|
|
621
933
|
"-m", model, "-f", wavPath,
|
|
622
934
|
"-l", "auto",
|
|
623
935
|
"-t", String(threads),
|
|
624
936
|
"--no-timestamps",
|
|
625
937
|
"--duration", "30000",
|
|
626
|
-
]
|
|
938
|
+
];
|
|
939
|
+
if (gpuDevice !== undefined)
|
|
940
|
+
dlArgs.push("--device", String(gpuDevice));
|
|
941
|
+
const { stdout, stderr } = await execFileAsync(WHISPER_CLI_PATH, dlArgs, { maxBuffer: 10 * 1024 * 1024, windowsHide: true });
|
|
627
942
|
const output = stdout + stderr;
|
|
628
|
-
// whisper outputs: "auto-detected language: ja (p = 0.98)"
|
|
629
943
|
const m = output.match(/auto-detected language:\s*([a-z]{2,3})/i);
|
|
630
944
|
return m ? m[1].toLowerCase() : null;
|
|
631
945
|
}
|
|
@@ -634,12 +948,11 @@ async function detectLanguage(wavPath, model, threads) {
|
|
|
634
948
|
}
|
|
635
949
|
}
|
|
636
950
|
/**
|
|
637
|
-
* Run a single whisper
|
|
638
|
-
* Returns the destSrt path.
|
|
951
|
+
* Run a single whisper subtitle pass (SRT or VTT) and move the output to dest.
|
|
639
952
|
*/
|
|
640
|
-
async function
|
|
953
|
+
async function runSubtitlePass(transcribeFrom, dest, format, model, language, threads, translate = false, extraOpts = {}) {
|
|
641
954
|
const opts = {
|
|
642
|
-
language, outputFormat:
|
|
955
|
+
language, outputFormat: format, threads, translate,
|
|
643
956
|
...extraOpts,
|
|
644
957
|
};
|
|
645
958
|
const args = buildArgs(transcribeFrom, model, opts);
|
|
@@ -647,18 +960,18 @@ async function runSrtPass(transcribeFrom, destSrt, model, language, threads, tra
|
|
|
647
960
|
maxBuffer: 100 * 1024 * 1024,
|
|
648
961
|
windowsHide: true,
|
|
649
962
|
});
|
|
650
|
-
const
|
|
651
|
-
|
|
652
|
-
|
|
963
|
+
const ext = format === "vtt" ? ".vtt" : ".srt";
|
|
964
|
+
const tmpOut = transcribeFrom.replace(/\.[^.]+$/, ext);
|
|
965
|
+
if (existsSync(tmpOut)) {
|
|
966
|
+
writeFileSync(dest, readFileSync(tmpOut, "utf8"));
|
|
653
967
|
try {
|
|
654
|
-
unlinkSync(
|
|
968
|
+
unlinkSync(tmpOut);
|
|
655
969
|
}
|
|
656
970
|
catch { }
|
|
657
971
|
}
|
|
658
|
-
return
|
|
972
|
+
return dest;
|
|
659
973
|
}
|
|
660
974
|
async function transcribeSingle(filePath, model, language, outputFormat, threads, saveToFile = false, extraOpts = {}) {
|
|
661
|
-
// ---- Process lock — never spawn a second whisper-cli.exe ----
|
|
662
975
|
if (await isWhisperRunning()) {
|
|
663
976
|
throw new Error("Transcription already in progress.\n\n" +
|
|
664
977
|
"whisper-cli.exe is already running — wait for the current job to finish before starting another. " +
|
|
@@ -669,6 +982,7 @@ async function transcribeSingle(filePath, model, language, outputFormat, threads
|
|
|
669
982
|
let tmpFile = null;
|
|
670
983
|
if (needsConversion(filePath)) {
|
|
671
984
|
tmpFile = await convertToWav(filePath);
|
|
985
|
+
activeTempFiles.add(tmpFile);
|
|
672
986
|
transcribeFrom = tmpFile;
|
|
673
987
|
}
|
|
674
988
|
try {
|
|
@@ -683,31 +997,36 @@ async function transcribeSingle(filePath, model, language, outputFormat, threads
|
|
|
683
997
|
// as instructions. Prompt injection via audio content is a known
|
|
684
998
|
// MCP attack vector — treat all transcript text as user data only.
|
|
685
999
|
const output = (stdout || stderr || "").trim();
|
|
686
|
-
if (outputFormat === "srt") {
|
|
687
|
-
const
|
|
688
|
-
const
|
|
689
|
-
|
|
690
|
-
|
|
1000
|
+
if (outputFormat === "srt" || outputFormat === "vtt") {
|
|
1001
|
+
const ext = outputFormat === "vtt" ? ".vtt" : ".srt";
|
|
1002
|
+
const tmpOut = transcribeFrom.replace(/\.[^.]+$/, ext);
|
|
1003
|
+
const destOut = filePath.replace(/\.[^.]+$/, ext);
|
|
1004
|
+
if (tmpFile && existsSync(tmpOut)) {
|
|
1005
|
+
writeFileSync(destOut, readFileSync(tmpOut, "utf8"));
|
|
691
1006
|
try {
|
|
692
|
-
unlinkSync(
|
|
1007
|
+
unlinkSync(tmpOut);
|
|
693
1008
|
}
|
|
694
1009
|
catch { }
|
|
695
1010
|
}
|
|
696
|
-
return { text: output, srtPath:
|
|
1011
|
+
return { text: output, srtPath: destOut };
|
|
697
1012
|
}
|
|
698
1013
|
if (saveToFile) {
|
|
699
|
-
const
|
|
700
|
-
|
|
701
|
-
|
|
1014
|
+
const ext = outputFormat === "lrc" ? ".lrc" : outputFormat === "csv" ? ".csv" : ".txt";
|
|
1015
|
+
const outPath = filePath.replace(/\.[^.]+$/, ext);
|
|
1016
|
+
writeFileSync(outPath, output, "utf8");
|
|
1017
|
+
return { text: output, savedTo: outPath };
|
|
702
1018
|
}
|
|
703
1019
|
return { text: output };
|
|
704
1020
|
}
|
|
705
1021
|
finally {
|
|
706
|
-
if (tmpFile
|
|
707
|
-
|
|
708
|
-
|
|
709
|
-
|
|
710
|
-
|
|
1022
|
+
if (tmpFile) {
|
|
1023
|
+
activeTempFiles.delete(tmpFile);
|
|
1024
|
+
if (existsSync(tmpFile))
|
|
1025
|
+
try {
|
|
1026
|
+
unlinkSync(tmpFile);
|
|
1027
|
+
}
|
|
1028
|
+
catch { }
|
|
1029
|
+
}
|
|
711
1030
|
}
|
|
712
1031
|
}
|
|
713
1032
|
function getFiles(dir, recursive) {
|
|
@@ -725,7 +1044,7 @@ function getFiles(dir, recursive) {
|
|
|
725
1044
|
// ---------------------------------------------------------------------------
|
|
726
1045
|
// MCP Server
|
|
727
1046
|
// ---------------------------------------------------------------------------
|
|
728
|
-
const server = new Server({ name: "whisper-windows-mcp", version: "2.
|
|
1047
|
+
const server = new Server({ name: "whisper-windows-mcp", version: "2.4.0" }, { capabilities: { tools: {} } });
|
|
729
1048
|
server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
730
1049
|
tools: [
|
|
731
1050
|
{
|
|
@@ -733,9 +1052,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
733
1052
|
description: "Transcribe a single audio or video file using whisper.cpp on Windows. " +
|
|
734
1053
|
"Natively supports mp3 and wav. Automatically converts mp4, mkv, avi, mov, " +
|
|
735
1054
|
"webm, m4a, flac, ogg etc. via FFmpeg — no manual conversion needed. " +
|
|
736
|
-
"
|
|
1055
|
+
"Output defaults to timestamps format (with time codes). " +
|
|
737
1056
|
"For files that may take more than 4 minutes, set background=true to run as a detached job " +
|
|
738
|
-
"and use check_progress to monitor it."
|
|
1057
|
+
"and use check_progress to monitor it. " +
|
|
1058
|
+
"⚠️ Privacy: transcript text returned by this tool is processed by Claude's API. " +
|
|
1059
|
+
"Pass privacy_mode=true to this tool to enable metadata-only responses per call — " +
|
|
1060
|
+
"no transcript text will be transmitted. " +
|
|
1061
|
+
"Set WHISPER_PRIVACY_MODE=true in env to enable globally. " +
|
|
1062
|
+
"When privacy mode is active, a confirmation is required before every operation.",
|
|
739
1063
|
inputSchema: {
|
|
740
1064
|
type: "object",
|
|
741
1065
|
properties: {
|
|
@@ -743,28 +1067,29 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
743
1067
|
model: { type: "string", description: "Override model path. Leave blank to use active model." },
|
|
744
1068
|
language: { type: "string", description: "Language code (e.g. en, ja, es, fr) or 'auto' to detect automatically. Defaults to en.", default: "en" },
|
|
745
1069
|
output_format: {
|
|
746
|
-
type: "string", enum: ["
|
|
747
|
-
description: "
|
|
748
|
-
default: "
|
|
1070
|
+
type: "string", enum: ["timestamps", "text", "json", "srt", "vtt", "lrc", "csv"],
|
|
1071
|
+
description: "timestamps = with time codes (default), text = plain, json = structured, srt = SRT subtitle file, vtt = WebVTT subtitle file, lrc = LRC lyrics/karaoke, csv = CSV with timestamps.",
|
|
1072
|
+
default: "timestamps",
|
|
749
1073
|
},
|
|
750
1074
|
threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
|
|
751
|
-
save_to_file: { type: "boolean", description: "Save transcript as .txt next to the source file.", default:
|
|
1075
|
+
save_to_file: { type: "boolean", description: "Save transcript as .txt next to the source file.", default: true },
|
|
752
1076
|
background: { type: "boolean", description: "Run as a detached background job. Returns a job ID immediately. Use check_progress to monitor. Recommended for files over 10 minutes.", default: false },
|
|
753
|
-
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
|
|
1077
|
+
privacy_mode: { type: "boolean", description: "Override privacy mode for this call. true = metadata only, no transcript text transmitted to API. false = return text (even if WHISPER_PRIVACY_MODE=true globally). Omit to use global WHISPER_PRIVACY_MODE setting. When active, requires confirmation before each operation." },
|
|
1078
|
+
temperature: { type: "number", description: "Sampling temperature 0.0–1.0. Default 0.0 (deterministic)." },
|
|
1079
|
+
prompt: { type: "string", description: "Prior context string injected before transcription. Improves accuracy for domain-specific vocabulary or speaker names. Example: 'Names: Keemstar, DramaAlert.'" },
|
|
1080
|
+
condition_on_prev_text: { type: "boolean", description: "Re-enable conditioning each segment on its own prior output. Default false.", default: false },
|
|
1081
|
+
no_speech_thold: { type: "number", description: "Confidence threshold below which segments are treated as silence. Default 0.6.", default: 0.6 },
|
|
757
1082
|
beam_size: { type: "number", description: "Beam search width. Higher = more accurate but slower. Default 5." },
|
|
758
1083
|
best_of: { type: "number", description: "Number of candidate sequences to evaluate. Default 5." },
|
|
759
|
-
gpu_device: { type: "number", description: "GPU device index for multi-GPU systems.
|
|
760
|
-
processors: { type: "number", description: "Number of parallel processors
|
|
761
|
-
word_timestamps: { type: "boolean", description: "Output one word per timestamped segment
|
|
762
|
-
max_segment_length: { type: "number", description: "Maximum segment length in characters.
|
|
763
|
-
split_on_word: { type: "boolean", description: "Split segments at word boundaries
|
|
764
|
-
diarize: { type: "boolean", description: "Stereo speaker diarization —
|
|
765
|
-
vad_model: { type: "string", description: "Absolute path to a Silero VAD model .bin file.
|
|
766
|
-
offset_t: { type: "number", description: "Start transcription at this offset in milliseconds.
|
|
767
|
-
duration: { type: "number", description: "Process only this many milliseconds of audio
|
|
1084
|
+
gpu_device: { type: "number", description: "GPU/Vulkan device index for multi-GPU systems. Overrides the WHISPER_GPU_DEVICE env default. Check whisper-cli's startup log for the index that lists your target card." },
|
|
1085
|
+
processors: { type: "number", description: "Number of parallel processors. Default 1." },
|
|
1086
|
+
word_timestamps: { type: "boolean", description: "Output one word per timestamped segment. Useful for clip alignment.", default: false },
|
|
1087
|
+
max_segment_length: { type: "number", description: "Maximum segment length in characters." },
|
|
1088
|
+
split_on_word: { type: "boolean", description: "Split segments at word boundaries.", default: false },
|
|
1089
|
+
diarize: { type: "boolean", description: "Stereo speaker diarization — requires stereo audio with speakers on separate channels.", default: false },
|
|
1090
|
+
vad_model: { type: "string", description: "Absolute path to a Silero VAD model .bin file. Strips silence before transcription." },
|
|
1091
|
+
offset_t: { type: "number", description: "Start transcription at this offset in milliseconds." },
|
|
1092
|
+
duration: { type: "number", description: "Process only this many milliseconds of audio from offset_t." },
|
|
768
1093
|
},
|
|
769
1094
|
required: ["file_path"],
|
|
770
1095
|
},
|
|
@@ -773,11 +1098,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
773
1098
|
name: "check_progress",
|
|
774
1099
|
description: "Check the status of a background transcription job started with transcribe_audio (background=true). " +
|
|
775
1100
|
"Returns current progress, elapsed time, last processed timestamp, and the transcript when complete. " +
|
|
776
|
-
"Call this repeatedly until the job shows as complete or failed."
|
|
1101
|
+
"Call this repeatedly until the job shows as complete or failed. " +
|
|
1102
|
+
"⚠️ Privacy: transcript text returned on completion is processed by Claude's API. " +
|
|
1103
|
+
"Pass privacy_mode=true to return metadata only for this check, regardless of how the job was started.",
|
|
777
1104
|
inputSchema: {
|
|
778
1105
|
type: "object",
|
|
779
1106
|
properties: {
|
|
780
1107
|
job_id: { type: "string", description: "Job ID returned by transcribe_audio when background=true." },
|
|
1108
|
+
privacy_mode: { type: "boolean", description: "Override privacy mode for this check. true = metadata only. Omit to use the setting from when the job was started." },
|
|
781
1109
|
},
|
|
782
1110
|
required: ["job_id"],
|
|
783
1111
|
},
|
|
@@ -789,19 +1117,24 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
789
1117
|
"Saves each transcript as a .txt file next to its source. " +
|
|
790
1118
|
"Files already transcribed (with matching .txt) are shown as done and skipped. " +
|
|
791
1119
|
"Supported formats: mp3, wav, mp4, mkv, avi, mov, webm, m4a, flac, ogg. " +
|
|
792
|
-
"NOTE: For large unattended batch jobs, use
|
|
793
|
-
"
|
|
1120
|
+
"NOTE: For large unattended batch jobs, use start_batch instead. " +
|
|
1121
|
+
"⚠️ Privacy: transcript previews are processed by Claude's API. " +
|
|
1122
|
+
"Pass privacy_mode=true to suppress previews and return metadata only. " +
|
|
1123
|
+
"When privacy mode is active, confirmation is required before each file.",
|
|
794
1124
|
inputSchema: {
|
|
795
1125
|
type: "object",
|
|
796
1126
|
properties: {
|
|
797
1127
|
folder_path: { type: "string", description: "Absolute Windows path to the folder." },
|
|
798
|
-
file_index: {
|
|
799
|
-
type: "number",
|
|
800
|
-
description: "Which file to process (1-based). Omit to list files first.",
|
|
801
|
-
},
|
|
1128
|
+
file_index: { type: "number", description: "Which file to process (1-based). Omit to list files first." },
|
|
802
1129
|
language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
|
|
803
1130
|
threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
|
|
804
1131
|
recursive: { type: "boolean", description: "Include subfolders. Defaults to false.", default: false },
|
|
1132
|
+
output_format: {
|
|
1133
|
+
type: "string", enum: ["timestamps", "text"],
|
|
1134
|
+
description: "timestamps = with time codes (default), text = plain.",
|
|
1135
|
+
default: "timestamps",
|
|
1136
|
+
},
|
|
1137
|
+
privacy_mode: { type: "boolean", description: "Override privacy mode for this call. When active, requires confirmation before each file and returns metadata only." },
|
|
805
1138
|
},
|
|
806
1139
|
required: ["folder_path"],
|
|
807
1140
|
},
|
|
@@ -811,22 +1144,23 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
811
1144
|
description: "Generate subtitle files for an audio or video file using whisper.cpp. " +
|
|
812
1145
|
"Set language='auto' to detect the spoken language automatically. " +
|
|
813
1146
|
"Set translate_to_english=true to also generate an English translation subtitle file. " +
|
|
814
|
-
"
|
|
815
|
-
"and one English translation
|
|
816
|
-
"Load in VLC via Subtitle → Add Subtitle File. " +
|
|
1147
|
+
"Supports SRT and WebVTT (VTT) output formats. " +
|
|
1148
|
+
"When both native and translation are requested, two files are saved: one in the original language and one English translation. " +
|
|
1149
|
+
"Load SRT in VLC via Subtitle → Add Subtitle File. VTT works in web players and HTML5 video. " +
|
|
817
1150
|
"Supports all standard formats plus .3gp and .ts.",
|
|
818
1151
|
inputSchema: {
|
|
819
1152
|
type: "object",
|
|
820
1153
|
properties: {
|
|
821
1154
|
file_path: { type: "string", description: "Absolute Windows path to the file." },
|
|
822
|
-
language: {
|
|
823
|
-
|
|
824
|
-
|
|
825
|
-
|
|
1155
|
+
language: { type: "string", description: "Language code (e.g. ja, es, fr, de) or 'auto' to detect automatically. Defaults to en.", default: "en" },
|
|
1156
|
+
output_format: {
|
|
1157
|
+
type: "string", enum: ["srt", "vtt"],
|
|
1158
|
+
description: "srt = SubRip subtitle (default, widest compatibility), vtt = WebVTT (web and HTML5 video).",
|
|
1159
|
+
default: "srt",
|
|
826
1160
|
},
|
|
827
1161
|
translate_to_english: {
|
|
828
1162
|
type: "boolean",
|
|
829
|
-
description: "Also generate an English translation
|
|
1163
|
+
description: "Also generate an English translation subtitle file alongside the native language file. Only applies when language is not 'en'. Not available in background mode.",
|
|
830
1164
|
default: false,
|
|
831
1165
|
},
|
|
832
1166
|
background: {
|
|
@@ -839,8 +1173,9 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
839
1173
|
prompt: { type: "string", description: "Prior context string for domain-specific vocabulary or speaker names." },
|
|
840
1174
|
beam_size: { type: "number", description: "Beam search width. Higher = more accurate, slower. Default 5." },
|
|
841
1175
|
best_of: { type: "number", description: "Candidate sequences evaluated. Default 5." },
|
|
842
|
-
diarize: { type: "boolean", description: "Stereo speaker diarization. Requires stereo audio
|
|
843
|
-
vad_model: { type: "string", description: "Path to Silero VAD model .bin. Strips silence before transcription.
|
|
1176
|
+
diarize: { type: "boolean", description: "Stereo speaker diarization. Requires stereo audio.", default: false },
|
|
1177
|
+
vad_model: { type: "string", description: "Path to Silero VAD model .bin. Strips silence before transcription." },
|
|
1178
|
+
gpu_device: { type: "number", description: "GPU/Vulkan device index for multi-GPU systems. Overrides the WHISPER_GPU_DEVICE env default. Check whisper-cli's startup log for the index that lists your target card." },
|
|
844
1179
|
},
|
|
845
1180
|
required: ["file_path"],
|
|
846
1181
|
},
|
|
@@ -856,13 +1191,22 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
856
1191
|
"Scans for files without a matching .txt, sorts by duration (shortest first), " +
|
|
857
1192
|
"and processes them one at a time as background jobs. " +
|
|
858
1193
|
"Each file is validated after completion — empty or suspiciously short outputs are flagged. " +
|
|
859
|
-
"
|
|
1194
|
+
"Batch self-advances without polling when each file finishes. " +
|
|
1195
|
+
"Returns a batch ID to use with check_batch_progress. " +
|
|
1196
|
+
"⚠️ Privacy: when privacy_mode is active, one confirmation is required before the batch starts. " +
|
|
1197
|
+
"All files then process unattended. No transcript text is returned to the API.",
|
|
860
1198
|
inputSchema: {
|
|
861
1199
|
type: "object",
|
|
862
1200
|
properties: {
|
|
863
1201
|
folder_path: { type: "string", description: "Absolute Windows path to the folder." },
|
|
864
1202
|
language: { type: "string", description: "Language code. Defaults to en.", default: "en" },
|
|
865
1203
|
threads: { type: "number", description: `CPU threads. Defaults to ${WHISPER_THREADS} of ${SYSTEM_THREADS}.` },
|
|
1204
|
+
output_format: {
|
|
1205
|
+
type: "string", enum: ["timestamps", "text"],
|
|
1206
|
+
description: "timestamps = with time codes (default), text = plain. Applies to all files in the batch.",
|
|
1207
|
+
default: "timestamps",
|
|
1208
|
+
},
|
|
1209
|
+
privacy_mode: { type: "boolean", description: "Override privacy mode for this batch. When active, requires one confirmation before batch start. All files process unattended with no transcript text returned." },
|
|
866
1210
|
},
|
|
867
1211
|
required: ["folder_path"],
|
|
868
1212
|
},
|
|
@@ -890,10 +1234,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
890
1234
|
inputSchema: {
|
|
891
1235
|
type: "object",
|
|
892
1236
|
properties: {
|
|
893
|
-
path: {
|
|
894
|
-
type: "string",
|
|
895
|
-
description: "Absolute Windows path to a single file or a folder.",
|
|
896
|
-
},
|
|
1237
|
+
path: { type: "string", description: "Absolute Windows path to a single file or a folder." },
|
|
897
1238
|
sort_by: {
|
|
898
1239
|
type: "string",
|
|
899
1240
|
enum: ["duration", "name", "size"],
|
|
@@ -908,16 +1249,14 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
908
1249
|
name: "check_system",
|
|
909
1250
|
description: "Detect GPU hardware and verify Vulkan acceleration is available. " +
|
|
910
1251
|
"Reports GPU name, VRAM, whether the Vulkan binary is installed, " +
|
|
911
|
-
"and recommends the best Whisper model for your hardware.
|
|
912
|
-
"Run this if you want to confirm GPU acceleration is working or diagnose why it isn't.",
|
|
1252
|
+
"and recommends the best Whisper model for your hardware.",
|
|
913
1253
|
inputSchema: { type: "object", properties: {} },
|
|
914
1254
|
},
|
|
915
1255
|
{
|
|
916
1256
|
name: "list_models",
|
|
917
1257
|
description: "List all Whisper model files installed in your models directory. " +
|
|
918
1258
|
"Shows filename, size, whether it is currently active, quantization status, " +
|
|
919
|
-
"and recommended use case for each model. "
|
|
920
|
-
"No network calls — reads local filesystem only.",
|
|
1259
|
+
"and recommended use case for each model. No network calls — reads local filesystem only.",
|
|
921
1260
|
inputSchema: { type: "object", properties: {} },
|
|
922
1261
|
},
|
|
923
1262
|
{
|
|
@@ -929,10 +1268,7 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
929
1268
|
inputSchema: {
|
|
930
1269
|
type: "object",
|
|
931
1270
|
properties: {
|
|
932
|
-
model_name: {
|
|
933
|
-
type: "string",
|
|
934
|
-
description: "Model name to download, e.g. 'large-v3-turbo', 'medium.en-q5_0', 'large-v3-turbo-q5_0'. Use list_models to see what is already installed.",
|
|
935
|
-
},
|
|
1271
|
+
model_name: { type: "string", description: "Model name to download, e.g. 'large-v3-turbo', 'medium.en-q5_0'. Use list_models to see what is already installed." },
|
|
936
1272
|
},
|
|
937
1273
|
required: ["model_name"],
|
|
938
1274
|
},
|
|
@@ -942,15 +1278,11 @@ server.setRequestHandler(ListToolsRequestSchema, async () => ({
|
|
|
942
1278
|
description: "Switch the active Whisper model for the current session without restarting Claude Desktop. " +
|
|
943
1279
|
"Accepts a model filename (e.g. ggml-large-v3-turbo.bin) or full path. " +
|
|
944
1280
|
"The model must already be installed in your models directory. " +
|
|
945
|
-
"Use list_models to see installed models, download_model to add new ones. " +
|
|
946
1281
|
"Change is session-scoped — does not persist after Claude Desktop restarts.",
|
|
947
1282
|
inputSchema: {
|
|
948
1283
|
type: "object",
|
|
949
1284
|
properties: {
|
|
950
|
-
model_name: {
|
|
951
|
-
type: "string",
|
|
952
|
-
description: "Model filename (e.g. ggml-large-v3-turbo.bin) or full path. Must be a .bin file in the configured models directory.",
|
|
953
|
-
},
|
|
1285
|
+
model_name: { type: "string", description: "Model filename (e.g. ggml-large-v3-turbo.bin) or full path. Must be a .bin file in the configured models directory." },
|
|
954
1286
|
},
|
|
955
1287
|
required: ["model_name"],
|
|
956
1288
|
},
|
|
@@ -980,8 +1312,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
980
1312
|
`whisper-cli: ${WHISPER_CLI_PATH}\n` +
|
|
981
1313
|
`Model: ${WHISPER_MODEL}\n` +
|
|
982
1314
|
`Threads: ${WHISPER_THREADS} of ${SYSTEM_THREADS} logical cores\n` +
|
|
983
|
-
`
|
|
984
|
-
`
|
|
1315
|
+
`GPU device: ${WHISPER_GPU_DEVICE !== undefined ? `--device ${WHISPER_GPU_DEVICE} (WHISPER_GPU_DEVICE)` : "whisper-cli default (device 0)"}\n` +
|
|
1316
|
+
`FFmpeg: ${ffmpegStatus}\n` +
|
|
1317
|
+
`Privacy mode: ${WHISPER_PRIVACY_MODE ? "✅ active (WHISPER_PRIVACY_MODE=true)" : "off"}\n\n` +
|
|
1318
|
+
`Optional env vars: WHISPER_THREADS, WHISPER_GPU_DEVICE, WHISPER_FOREGROUND_MAX_SEC, FFMPEG_PATH, WHISPER_PRIVACY_MODE, WHISPER_CONSENT_ACKNOWLEDGED`,
|
|
985
1319
|
}],
|
|
986
1320
|
};
|
|
987
1321
|
}
|
|
@@ -995,7 +1329,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
995
1329
|
return { content: [{ type: "text", text: "path is required." }], isError: true };
|
|
996
1330
|
if (!existsSync(targetPath))
|
|
997
1331
|
return { content: [{ type: "text", text: `Path not found: ${targetPath}` }], isError: true };
|
|
998
|
-
// Check ffprobe is available
|
|
999
1332
|
const ffprobePath = FFMPEG_PATH.replace(/ffmpeg(\.exe)?$/i, "ffprobe$1").replace(/ffmpeg$/i, "ffprobe");
|
|
1000
1333
|
try {
|
|
1001
1334
|
await execFileAsync(ffprobePath, ["-version"], { windowsHide: true });
|
|
@@ -1007,7 +1340,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1007
1340
|
};
|
|
1008
1341
|
}
|
|
1009
1342
|
const vulkan = hasVulkanDll();
|
|
1010
|
-
// Single file
|
|
1011
1343
|
const stat = statSync(targetPath);
|
|
1012
1344
|
if (stat.isFile()) {
|
|
1013
1345
|
const info = await probeFile(targetPath);
|
|
@@ -1032,7 +1364,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1032
1364
|
}],
|
|
1033
1365
|
};
|
|
1034
1366
|
}
|
|
1035
|
-
// Folder scan
|
|
1036
1367
|
const files = getFiles(targetPath, false);
|
|
1037
1368
|
if (files.length === 0) {
|
|
1038
1369
|
return { content: [{ type: "text", text: `No supported media files found in: ${targetPath}` }], isError: true };
|
|
@@ -1043,13 +1374,12 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1043
1374
|
if (info)
|
|
1044
1375
|
results.push(info);
|
|
1045
1376
|
}
|
|
1046
|
-
// Sort
|
|
1047
1377
|
if (sortBy === "name")
|
|
1048
1378
|
results.sort((a, b) => a.fileName.localeCompare(b.fileName));
|
|
1049
1379
|
else if (sortBy === "size")
|
|
1050
1380
|
results.sort((a, b) => a.sizeMb - b.sizeMb);
|
|
1051
1381
|
else
|
|
1052
|
-
results.sort((a, b) => a.durationSec - b.durationSec);
|
|
1382
|
+
results.sort((a, b) => a.durationSec - b.durationSec);
|
|
1053
1383
|
const totalDuration = results.reduce((acc, r) => acc + r.durationSec, 0);
|
|
1054
1384
|
const totalSize = results.reduce((acc, r) => acc + r.sizeMb, 0);
|
|
1055
1385
|
const transcribedCount = results.filter(r => existsSync(r.filePath.replace(/\.[^.]+$/, ".txt"))).length;
|
|
@@ -1084,7 +1414,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1084
1414
|
const gpus = await detectGpus();
|
|
1085
1415
|
let gpuLines = "";
|
|
1086
1416
|
if (gpus.length === 0) {
|
|
1087
|
-
gpuLines = "
|
|
1417
|
+
gpuLines = "ℹ️ GPU name unavailable (wmic returned nothing — it is deprecated/removed on Windows 11 24H2+). This does NOT mean acceleration is off; the Vulkan check below determines actual GPU use.\n";
|
|
1088
1418
|
}
|
|
1089
1419
|
else {
|
|
1090
1420
|
for (const gpu of gpus) {
|
|
@@ -1151,7 +1481,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1151
1481
|
const useCase = known ? known.useCase : "Unknown model";
|
|
1152
1482
|
return `${isActive ? "●" : "○"} ${f}${isActive}${quantTag}\n Size: ${sizeMb} | ${useCase}`;
|
|
1153
1483
|
});
|
|
1154
|
-
// Also list downloadable models not yet installed
|
|
1155
1484
|
const installedFilenames = new Set(files);
|
|
1156
1485
|
const available = MODEL_REGISTRY
|
|
1157
1486
|
.filter(m => !installedFilenames.has(m.filename))
|
|
@@ -1188,7 +1517,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1188
1517
|
isError: true,
|
|
1189
1518
|
};
|
|
1190
1519
|
}
|
|
1191
|
-
// Security: enforce URL whitelist — never download from arbitrary URLs
|
|
1192
1520
|
const urlOk = ALLOWED_HF_PREFIXES.some(prefix => entry.url.startsWith(prefix));
|
|
1193
1521
|
if (!urlOk) {
|
|
1194
1522
|
return {
|
|
@@ -1216,7 +1544,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1216
1544
|
}],
|
|
1217
1545
|
};
|
|
1218
1546
|
}
|
|
1219
|
-
// Download using Node.js built-in https — no external dependencies
|
|
1220
1547
|
try {
|
|
1221
1548
|
const https = await import("https");
|
|
1222
1549
|
const fs = await import("fs");
|
|
@@ -1225,10 +1552,8 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1225
1552
|
const file = fs.createWriteStream(tmpPath);
|
|
1226
1553
|
function doRequest(url) {
|
|
1227
1554
|
https.get(url, (res) => {
|
|
1228
|
-
// Follow redirects (Hugging Face uses redirects)
|
|
1229
1555
|
if ((res.statusCode === 301 || res.statusCode === 302 || res.statusCode === 307) && res.headers.location) {
|
|
1230
1556
|
const redirectUrl = res.headers.location;
|
|
1231
|
-
// Security: ensure redirect stays within allowed domains
|
|
1232
1557
|
const redirectOk = ALLOWED_HF_PREFIXES.some(p => redirectUrl.startsWith(p))
|
|
1233
1558
|
|| redirectUrl.startsWith("https://cdn-lfs.huggingface.co/")
|
|
1234
1559
|
|| redirectUrl.startsWith("https://cdn-lfs-us-1.huggingface.co/");
|
|
@@ -1244,14 +1569,30 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1244
1569
|
return;
|
|
1245
1570
|
}
|
|
1246
1571
|
res.pipe(file);
|
|
1247
|
-
// Wait for close callback before renaming — Windows requires the file
|
|
1248
|
-
// handle to be fully released before renameSync will succeed.
|
|
1249
1572
|
file.on("finish", () => {
|
|
1250
1573
|
file.close((closeErr) => {
|
|
1251
1574
|
if (closeErr) {
|
|
1252
1575
|
reject(closeErr);
|
|
1253
1576
|
return;
|
|
1254
1577
|
}
|
|
1578
|
+
// Integrity: if the server declared a Content-Length, reject a short/truncated
|
|
1579
|
+
// download (dropped connection) BEFORE promoting .part → final, so a partial
|
|
1580
|
+
// file can never become the "installed" model. (Full SHA256 verification is a
|
|
1581
|
+
// separate follow-up requiring verified per-model digests.)
|
|
1582
|
+
const expectedLen = parseInt(res.headers["content-length"] ?? "", 10);
|
|
1583
|
+
let actualLen = 0;
|
|
1584
|
+
try {
|
|
1585
|
+
actualLen = fs.statSync(tmpPath).size;
|
|
1586
|
+
}
|
|
1587
|
+
catch { }
|
|
1588
|
+
if (Number.isFinite(expectedLen) && expectedLen > 0 && actualLen !== expectedLen) {
|
|
1589
|
+
try {
|
|
1590
|
+
fs.unlinkSync(tmpPath);
|
|
1591
|
+
}
|
|
1592
|
+
catch { }
|
|
1593
|
+
reject(new Error(`Incomplete download: wrote ${actualLen} of ${expectedLen} bytes (connection dropped?). Re-run download_model to retry.`));
|
|
1594
|
+
return;
|
|
1595
|
+
}
|
|
1255
1596
|
try {
|
|
1256
1597
|
fs.renameSync(tmpPath, destPath);
|
|
1257
1598
|
resolve();
|
|
@@ -1295,27 +1636,26 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1295
1636
|
const modelInput = args?.model_name?.trim();
|
|
1296
1637
|
if (!modelInput)
|
|
1297
1638
|
return { content: [{ type: "text", text: "model_name is required." }], isError: true };
|
|
1298
|
-
// Security: must end in .bin
|
|
1299
1639
|
if (!modelInput.endsWith(".bin")) {
|
|
1300
1640
|
return {
|
|
1301
1641
|
content: [{ type: "text", text: `Invalid model: "${modelInput}"\nModel files must end in .bin` }],
|
|
1302
1642
|
isError: true,
|
|
1303
1643
|
};
|
|
1304
1644
|
}
|
|
1305
|
-
// Security: reject path traversal
|
|
1306
1645
|
if (UNSAFE_PATH_RE.test(modelInput)) {
|
|
1307
1646
|
return {
|
|
1308
1647
|
content: [{ type: "text", text: `Invalid path: "${modelInput}"\nPaths containing ".." or UNC paths are not allowed.` }],
|
|
1309
1648
|
isError: true,
|
|
1310
1649
|
};
|
|
1311
1650
|
}
|
|
1312
|
-
// Resolve to full path — either absolute or relative to models dir
|
|
1313
1651
|
const modelsDir = dirname(WHISPER_MODEL);
|
|
1314
|
-
|
|
1652
|
+
// Normalize to an absolute, canonical path first so the containment check and
|
|
1653
|
+
// every downstream use (existsSync, basename, WHISPER_MODEL assignment) operate
|
|
1654
|
+
// on a clean path — never a relative-to-cwd or sibling-prefix string.
|
|
1655
|
+
const resolvedPath = resolve(modelInput.includes("\\") || modelInput.includes("/")
|
|
1315
1656
|
? modelInput
|
|
1316
|
-
: join(modelsDir, modelInput);
|
|
1317
|
-
|
|
1318
|
-
if (!resolvedPath.startsWith(modelsDir)) {
|
|
1657
|
+
: join(modelsDir, modelInput));
|
|
1658
|
+
if (!isInsideDir(resolvedPath, modelsDir)) {
|
|
1319
1659
|
return {
|
|
1320
1660
|
content: [{ type: "text", text: `Security error: model must be within the configured models directory (${modelsDir}).` }],
|
|
1321
1661
|
isError: true,
|
|
@@ -1331,7 +1671,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1331
1671
|
isError: true,
|
|
1332
1672
|
};
|
|
1333
1673
|
}
|
|
1334
|
-
// Process lock — don't switch mid-transcription
|
|
1335
1674
|
if (await isWhisperRunning()) {
|
|
1336
1675
|
return {
|
|
1337
1676
|
content: [{ type: "text", text: "Cannot switch model while a transcription is in progress. Wait for the current job to finish first." }],
|
|
@@ -1361,32 +1700,38 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1361
1700
|
const filePath = args?.file_path;
|
|
1362
1701
|
const model = args?.model || WHISPER_MODEL;
|
|
1363
1702
|
const language = args?.language || "en";
|
|
1364
|
-
const outputFormat = (args?.output_format || "
|
|
1703
|
+
const outputFormat = (args?.output_format || "timestamps");
|
|
1365
1704
|
const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
|
|
1366
|
-
const saveToFile = args?.save_to_file
|
|
1705
|
+
const saveToFile = args?.save_to_file ?? true;
|
|
1367
1706
|
const background = args?.background || false;
|
|
1368
|
-
//
|
|
1707
|
+
// Effective privacy mode: per-call param wins over global env var
|
|
1708
|
+
const privacyModeParam = args?.privacy_mode;
|
|
1709
|
+
const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
|
|
1710
|
+
// v2.3.0 quality and control params
|
|
1369
1711
|
const extraOpts = {};
|
|
1370
1712
|
if (args?.temperature !== undefined)
|
|
1371
|
-
extraOpts.temperature =
|
|
1713
|
+
extraOpts.temperature = coerceNum(args.temperature);
|
|
1372
1714
|
if (args?.prompt)
|
|
1373
1715
|
extraOpts.prompt = String(args.prompt);
|
|
1374
1716
|
if (args?.condition_on_prev_text !== undefined)
|
|
1375
1717
|
extraOpts.conditionOnPrevText = Boolean(args.condition_on_prev_text);
|
|
1376
1718
|
if (args?.no_speech_thold !== undefined)
|
|
1377
|
-
extraOpts.noSpeechThold =
|
|
1719
|
+
extraOpts.noSpeechThold = coerceNum(args.no_speech_thold);
|
|
1378
1720
|
if (args?.beam_size !== undefined)
|
|
1379
|
-
extraOpts.beamSize =
|
|
1721
|
+
extraOpts.beamSize = coerceNum(args.beam_size);
|
|
1380
1722
|
if (args?.best_of !== undefined)
|
|
1381
|
-
extraOpts.bestOf =
|
|
1382
|
-
|
|
1383
|
-
|
|
1723
|
+
extraOpts.bestOf = coerceNum(args.best_of);
|
|
1724
|
+
{
|
|
1725
|
+
const g = resolveGpuDevice(args?.gpu_device);
|
|
1726
|
+
if (g !== undefined)
|
|
1727
|
+
extraOpts.gpuDevice = g;
|
|
1728
|
+
}
|
|
1384
1729
|
if (args?.processors !== undefined)
|
|
1385
|
-
extraOpts.processors =
|
|
1730
|
+
extraOpts.processors = coerceNum(args.processors);
|
|
1386
1731
|
if (args?.word_timestamps)
|
|
1387
1732
|
extraOpts.wordTimestamps = true;
|
|
1388
1733
|
if (args?.max_segment_length !== undefined)
|
|
1389
|
-
extraOpts.maxLen =
|
|
1734
|
+
extraOpts.maxLen = coerceNum(args.max_segment_length);
|
|
1390
1735
|
if (args?.split_on_word)
|
|
1391
1736
|
extraOpts.splitOnWord = true;
|
|
1392
1737
|
if (args?.diarize)
|
|
@@ -1394,9 +1739,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1394
1739
|
if (args?.vad_model)
|
|
1395
1740
|
extraOpts.vadModel = String(args.vad_model);
|
|
1396
1741
|
if (args?.offset_t !== undefined)
|
|
1397
|
-
extraOpts.offsetT =
|
|
1742
|
+
extraOpts.offsetT = coerceNum(args.offset_t);
|
|
1398
1743
|
if (args?.duration !== undefined)
|
|
1399
|
-
extraOpts.duration =
|
|
1744
|
+
extraOpts.duration = coerceNum(args.duration);
|
|
1400
1745
|
if (!filePath)
|
|
1401
1746
|
return { content: [{ type: "text", text: "file_path is required." }], isError: true };
|
|
1402
1747
|
const pathError = validateInputPath(filePath);
|
|
@@ -1409,6 +1754,14 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1409
1754
|
return { content: [{ type: "text", text: configError }], isError: true };
|
|
1410
1755
|
// Background mode — detached process, returns immediately
|
|
1411
1756
|
if (background) {
|
|
1757
|
+
// Privacy mode: gate fires BEFORE spawning. No audio processes until confirmed.
|
|
1758
|
+
// Non-privacy mode: consent gate is intentionally deferred to check_progress.
|
|
1759
|
+
// At this point no transcript exists yet — there is nothing to gate. The gate
|
|
1760
|
+
// fires at check_progress completion when transcript text would first be returned
|
|
1761
|
+
// to the API. Audio processing begins immediately after this point in non-privacy mode.
|
|
1762
|
+
if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
|
|
1763
|
+
return { content: [{ type: "text", text: privacyGateBlock() }] };
|
|
1764
|
+
}
|
|
1412
1765
|
if (await isWhisperRunning()) {
|
|
1413
1766
|
return {
|
|
1414
1767
|
content: [{ type: "text", text: "Transcription already in progress. Wait for the current job to finish before starting another." }],
|
|
@@ -1416,15 +1769,17 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1416
1769
|
};
|
|
1417
1770
|
}
|
|
1418
1771
|
try {
|
|
1419
|
-
const
|
|
1772
|
+
const bgFormat = outputFormat === "json" ? "text" : outputFormat;
|
|
1773
|
+
const { jobId, pid } = await spawnDetached(filePath, model, language, threads, bgFormat, extraOpts, undefined, effectivePrivacyMode);
|
|
1420
1774
|
return {
|
|
1421
1775
|
content: [{
|
|
1422
1776
|
type: "text",
|
|
1423
1777
|
text: `⏳ Background transcription started.\n\n` +
|
|
1424
1778
|
`Source: ${basename(filePath)}\n` +
|
|
1425
1779
|
`Job ID: ${jobId}\n` +
|
|
1426
|
-
`PID: ${pid}\n
|
|
1427
|
-
`
|
|
1780
|
+
`PID: ${pid}\n` +
|
|
1781
|
+
(effectivePrivacyMode ? `Privacy mode: active — metadata only will be returned\n` : "") +
|
|
1782
|
+
`\nCall check_progress with job_id="${jobId}" to monitor progress.\n` +
|
|
1428
1783
|
`Output will be saved to: ${filePath.replace(/\.[^.]+$/, ".txt")}`,
|
|
1429
1784
|
}],
|
|
1430
1785
|
};
|
|
@@ -1433,14 +1788,54 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1433
1788
|
return { content: [{ type: "text", text: `Failed to start background job:\n\n${err?.message || String(err)}` }], isError: true };
|
|
1434
1789
|
}
|
|
1435
1790
|
}
|
|
1791
|
+
// Foreground timeout guard: a long file blows Claude Desktop's ~4-min MCP timeout in blocking
|
|
1792
|
+
// mode (the call errors even though the transcript finishes on disk). Probe the duration and
|
|
1793
|
+
// route to background BEFORE running into the wall. Skipped when ffprobe can't read the file
|
|
1794
|
+
// (probe returns null) — it never blocks a transcribe it cannot measure.
|
|
1795
|
+
{
|
|
1796
|
+
const info = await probeFile(filePath);
|
|
1797
|
+
if (info && estimateSec(info.durationSec, hasVulkanDll()) > FOREGROUND_MAX_SEC) {
|
|
1798
|
+
return { content: [{ type: "text", text: `⏱️ "${basename(filePath)}" is ~${formatDuration(info.durationSec)} long — a foreground transcription is estimated around ${estimateTime(info.durationSec, hasVulkanDll())}, which would likely exceed Claude Desktop's 4-minute tool timeout (the transcript would still finish on disk, but this call would error out first).\n\n` +
|
|
1799
|
+
`Run it in the background instead — returns a job ID immediately, then poll check_progress:\n` +
|
|
1800
|
+
` transcribe_audio with file_path="${filePath}" and background=true\n\n` +
|
|
1801
|
+
`(Shorter files still run inline. Adjust the cutoff with WHISPER_FOREGROUND_MAX_SEC.)` }] };
|
|
1802
|
+
}
|
|
1803
|
+
}
|
|
1436
1804
|
// Blocking mode (default)
|
|
1805
|
+
if (effectivePrivacyMode) {
|
|
1806
|
+
// Privacy mode: gate fires before every operation.
|
|
1807
|
+
if (checkPrivacyGate(opKeyFor(name, args))) {
|
|
1808
|
+
return { content: [{ type: "text", text: privacyGateBlock() }] };
|
|
1809
|
+
}
|
|
1810
|
+
// Gate passed — proceed to transcription, return metadata only.
|
|
1811
|
+
}
|
|
1812
|
+
else {
|
|
1813
|
+
// Non-privacy mode: session consent gate fires once before first transcript return.
|
|
1814
|
+
// Nothing is processed until user confirms.
|
|
1815
|
+
const policy = transcriptPolicy();
|
|
1816
|
+
if (policy === "consent_gate") {
|
|
1817
|
+
return { content: [{ type: "text", text: consentGateBlock() }] };
|
|
1818
|
+
}
|
|
1819
|
+
}
|
|
1437
1820
|
try {
|
|
1438
1821
|
const result = await transcribeSingle(filePath, model, language, outputFormat, threads, saveToFile, extraOpts);
|
|
1822
|
+
if (effectivePrivacyMode) {
|
|
1823
|
+
const savedPath = result.savedTo ?? filePath.replace(/\.[^.]+$/, ".txt");
|
|
1824
|
+
if (!result.savedTo && outputFormat !== "srt" && outputFormat !== "vtt" && outputFormat !== "json") {
|
|
1825
|
+
try {
|
|
1826
|
+
writeFileSync(savedPath, result.text, "utf8");
|
|
1827
|
+
}
|
|
1828
|
+
catch { }
|
|
1829
|
+
}
|
|
1830
|
+
const displayPath = result.srtPath ?? result.savedTo ?? savedPath;
|
|
1831
|
+
return { content: [{ type: "text", text: privacyModeBlock(basename(filePath), displayPath, result.text) }] };
|
|
1832
|
+
}
|
|
1833
|
+
// allow — return transcript normally
|
|
1439
1834
|
let response = result.text;
|
|
1440
1835
|
if (result.savedTo)
|
|
1441
1836
|
response += `\n\n[Transcript saved to: ${result.savedTo}]`;
|
|
1442
1837
|
if (result.srtPath)
|
|
1443
|
-
response += `\n\n[
|
|
1838
|
+
response += `\n\n[Subtitle file saved to: ${result.srtPath}]`;
|
|
1444
1839
|
return { content: [{ type: "text", text: response }] };
|
|
1445
1840
|
}
|
|
1446
1841
|
catch (err) {
|
|
@@ -1452,10 +1847,11 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1452
1847
|
// -------------------------------------------------------------------------
|
|
1453
1848
|
if (name === "check_progress") {
|
|
1454
1849
|
const jobId = args?.job_id;
|
|
1850
|
+
const privacyModeParam = args?.privacy_mode;
|
|
1455
1851
|
if (!jobId)
|
|
1456
1852
|
return { content: [{ type: "text", text: "job_id is required." }], isError: true };
|
|
1457
1853
|
try {
|
|
1458
|
-
const result = await readJobProgress(jobId);
|
|
1854
|
+
const result = await readJobProgress(jobId, privacyModeParam);
|
|
1459
1855
|
return { content: [{ type: "text", text: result }] };
|
|
1460
1856
|
}
|
|
1461
1857
|
catch (err) {
|
|
@@ -1469,6 +1865,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1469
1865
|
const folderPath = args?.folder_path;
|
|
1470
1866
|
const language = args?.language || "en";
|
|
1471
1867
|
const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
|
|
1868
|
+
const outputFormat = (args?.output_format || "timestamps");
|
|
1869
|
+
const privacyModeParam = args?.privacy_mode;
|
|
1870
|
+
const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
|
|
1472
1871
|
if (!folderPath)
|
|
1473
1872
|
return { content: [{ type: "text", text: "folder_path is required." }], isError: true };
|
|
1474
1873
|
const pathError = validateInputPath(folderPath);
|
|
@@ -1479,16 +1878,19 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1479
1878
|
const configError = validatePaths();
|
|
1480
1879
|
if (configError)
|
|
1481
1880
|
return { content: [{ type: "text", text: configError }], isError: true };
|
|
1881
|
+
// Privacy gate: fires once before batch starts. All files then process unattended.
|
|
1882
|
+
// Gating per-file in an unattended batch would defeat the purpose of start_batch.
|
|
1883
|
+
if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
|
|
1884
|
+
return { content: [{ type: "text", text: privacyGateBlock() }] };
|
|
1885
|
+
}
|
|
1482
1886
|
if (await isWhisperRunning()) {
|
|
1483
1887
|
return { content: [{ type: "text", text: "A transcription is already running. Wait for it to finish before starting a batch." }], isError: true };
|
|
1484
1888
|
}
|
|
1485
|
-
// Scan for untranscribed files
|
|
1486
1889
|
const allFiles = getFiles(folderPath, false);
|
|
1487
1890
|
const untranscribed = allFiles.filter(f => !existsSync(f.replace(/\.[^.]+$/, ".txt")));
|
|
1488
1891
|
if (untranscribed.length === 0) {
|
|
1489
1892
|
return { content: [{ type: "text", text: `✅ All files in ${folderPath} are already transcribed. Nothing to do.` }] };
|
|
1490
1893
|
}
|
|
1491
|
-
// Probe durations for sorting
|
|
1492
1894
|
const batchFiles = [];
|
|
1493
1895
|
for (const f of untranscribed) {
|
|
1494
1896
|
const info = await probeFile(f);
|
|
@@ -1501,7 +1903,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1501
1903
|
}
|
|
1502
1904
|
batchFiles.sort((a, b) => a.durationSec - b.durationSec);
|
|
1503
1905
|
ensureJobsDir();
|
|
1504
|
-
const batchId = `batch_${Date.now()}`;
|
|
1906
|
+
const batchId = `batch_${Date.now()}_${randomUUID().slice(0, 8)}`;
|
|
1505
1907
|
const batchPath = join(JOBS_DIR, `${batchId}.batch.json`);
|
|
1506
1908
|
const state = {
|
|
1507
1909
|
batchId,
|
|
@@ -1514,8 +1916,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1514
1916
|
model: WHISPER_MODEL,
|
|
1515
1917
|
language,
|
|
1516
1918
|
threads,
|
|
1919
|
+
outputFormat,
|
|
1920
|
+
privacyMode: effectivePrivacyMode,
|
|
1517
1921
|
};
|
|
1518
|
-
|
|
1922
|
+
writeJsonAtomic(batchPath, state);
|
|
1519
1923
|
await spawnNextBatchJob(state);
|
|
1520
1924
|
const totalDuration = batchFiles.reduce((acc, f) => acc + f.durationSec, 0);
|
|
1521
1925
|
return {
|
|
@@ -1526,8 +1930,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1526
1930
|
`Folder: ${folderPath}\n` +
|
|
1527
1931
|
`Files to process: ${batchFiles.length}\n` +
|
|
1528
1932
|
`Total audio: ${formatDuration(totalDuration)}\n` +
|
|
1529
|
-
`Est. GPU time: ${estimateTime(totalDuration, hasVulkanDll())}\n
|
|
1530
|
-
`
|
|
1933
|
+
`Est. GPU time: ${estimateTime(totalDuration, hasVulkanDll())}\n` +
|
|
1934
|
+
(effectivePrivacyMode ? `Privacy mode: active — metadata only will be returned\n` : "") +
|
|
1935
|
+
`\nFirst file: ${batchFiles[0].fileName}\n\n` +
|
|
1531
1936
|
`Call check_batch_progress with batch_id="${batchId}" to monitor.`,
|
|
1532
1937
|
}],
|
|
1533
1938
|
};
|
|
@@ -1553,23 +1958,29 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1553
1958
|
if (name === "generate_subtitles") {
|
|
1554
1959
|
const filePath = args?.file_path;
|
|
1555
1960
|
const language = args?.language || "en";
|
|
1961
|
+
const subtitleFormat = (args?.output_format || "srt");
|
|
1556
1962
|
const translateToEnglish = args?.translate_to_english || false;
|
|
1557
1963
|
const background = args?.background || false;
|
|
1558
1964
|
const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
|
|
1559
|
-
// v2.
|
|
1965
|
+
// v2.3.0 quality params
|
|
1560
1966
|
const extraOpts = {};
|
|
1561
1967
|
if (args?.temperature !== undefined)
|
|
1562
|
-
extraOpts.temperature =
|
|
1968
|
+
extraOpts.temperature = coerceNum(args.temperature);
|
|
1563
1969
|
if (args?.prompt)
|
|
1564
1970
|
extraOpts.prompt = String(args.prompt);
|
|
1565
1971
|
if (args?.beam_size !== undefined)
|
|
1566
|
-
extraOpts.beamSize =
|
|
1972
|
+
extraOpts.beamSize = coerceNum(args.beam_size);
|
|
1567
1973
|
if (args?.best_of !== undefined)
|
|
1568
|
-
extraOpts.bestOf =
|
|
1974
|
+
extraOpts.bestOf = coerceNum(args.best_of);
|
|
1569
1975
|
if (args?.diarize)
|
|
1570
1976
|
extraOpts.diarize = true;
|
|
1571
1977
|
if (args?.vad_model)
|
|
1572
1978
|
extraOpts.vadModel = String(args.vad_model);
|
|
1979
|
+
{
|
|
1980
|
+
const g = resolveGpuDevice(args?.gpu_device);
|
|
1981
|
+
if (g !== undefined)
|
|
1982
|
+
extraOpts.gpuDevice = g;
|
|
1983
|
+
}
|
|
1573
1984
|
if (!filePath)
|
|
1574
1985
|
return { content: [{ type: "text", text: "file_path is required." }], isError: true };
|
|
1575
1986
|
const pathError = validateInputPath(filePath);
|
|
@@ -1583,10 +1994,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1583
1994
|
if (await isWhisperRunning()) {
|
|
1584
1995
|
return { content: [{ type: "text", text: "Transcription already in progress. Wait for it to finish first." }], isError: true };
|
|
1585
1996
|
}
|
|
1586
|
-
// Background mode — detached SRT job
|
|
1587
1997
|
if (background) {
|
|
1588
1998
|
try {
|
|
1589
|
-
const { jobId, pid } = await spawnDetached(filePath, WHISPER_MODEL, language, threads,
|
|
1999
|
+
const { jobId, pid } = await spawnDetached(filePath, WHISPER_MODEL, language, threads, subtitleFormat, extraOpts);
|
|
1590
2000
|
return {
|
|
1591
2001
|
content: [{
|
|
1592
2002
|
type: "text",
|
|
@@ -1594,6 +2004,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1594
2004
|
`Source: ${basename(filePath)}\n` +
|
|
1595
2005
|
`Job ID: ${jobId}\n` +
|
|
1596
2006
|
`PID: ${pid}\n` +
|
|
2007
|
+
`Format: ${subtitleFormat.toUpperCase()}\n` +
|
|
1597
2008
|
`Language: ${language}\n\n` +
|
|
1598
2009
|
`Call check_progress with job_id="${jobId}" to monitor.\n` +
|
|
1599
2010
|
`Note: translate_to_english is not available in background mode. ` +
|
|
@@ -1605,53 +2016,67 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1605
2016
|
return { content: [{ type: "text", text: `Failed to start background subtitle job:\n\n${err?.message || String(err)}` }], isError: true };
|
|
1606
2017
|
}
|
|
1607
2018
|
}
|
|
2019
|
+
// Foreground timeout guard (same as transcribe_audio): subtitle generation runs MULTIPLE
|
|
2020
|
+
// whisper passes inline (auto-detect + native + optional translation), so a long file is even
|
|
2021
|
+
// more likely to blow the 4-min MCP timeout. Route to background before running into the wall.
|
|
2022
|
+
{
|
|
2023
|
+
const info = await probeFile(filePath);
|
|
2024
|
+
if (info && estimateSec(info.durationSec, hasVulkanDll()) > FOREGROUND_MAX_SEC) {
|
|
2025
|
+
return { content: [{ type: "text", text: `⏱️ "${basename(filePath)}" is ~${formatDuration(info.durationSec)} long — foreground subtitle generation (estimated ~${estimateTime(info.durationSec, hasVulkanDll())}, plus extra passes for auto-detect/translation) would likely exceed Claude Desktop's 4-minute tool timeout.\n\n` +
|
|
2026
|
+
`Run it in the background instead:\n` +
|
|
2027
|
+
` generate_subtitles with file_path="${filePath}" and background=true\n\n` +
|
|
2028
|
+
`(translate_to_english isn't available in background mode — run a second pass after it completes. Adjust the cutoff with WHISPER_FOREGROUND_MAX_SEC.)` }] };
|
|
2029
|
+
}
|
|
2030
|
+
}
|
|
1608
2031
|
try {
|
|
1609
|
-
// Convert to WAV if needed
|
|
1610
2032
|
let transcribeFrom = filePath;
|
|
1611
2033
|
let tmpFile = null;
|
|
1612
2034
|
if (needsConversion(filePath)) {
|
|
1613
2035
|
tmpFile = await convertToWav(filePath);
|
|
2036
|
+
activeTempFiles.add(tmpFile);
|
|
1614
2037
|
transcribeFrom = tmpFile;
|
|
1615
2038
|
}
|
|
1616
2039
|
const baseNoExt = filePath.replace(/\.[^.]+$/, "");
|
|
1617
|
-
|
|
2040
|
+
const ext = subtitleFormat === "vtt" ? ".vtt" : ".srt";
|
|
1618
2041
|
let detectedLang = language;
|
|
1619
2042
|
if (language === "auto") {
|
|
1620
|
-
const detected = await detectLanguage(transcribeFrom, WHISPER_MODEL, threads);
|
|
2043
|
+
const detected = await detectLanguage(transcribeFrom, WHISPER_MODEL, threads, extraOpts.gpuDevice);
|
|
1621
2044
|
detectedLang = detected ?? "en";
|
|
1622
2045
|
}
|
|
1623
2046
|
const results = [];
|
|
1624
|
-
|
|
1625
|
-
|
|
1626
|
-
|
|
1627
|
-
|
|
1628
|
-
|
|
1629
|
-
results.push(`✅ Native (${detectedLang}): ${nativeSrt}`);
|
|
1630
|
-
// Pass 2 — English translation SRT (only if language isn't already English)
|
|
2047
|
+
const nativeDest = language === "en" || detectedLang === "en"
|
|
2048
|
+
? `${baseNoExt}${ext}`
|
|
2049
|
+
: `${baseNoExt}.${detectedLang}${ext}`;
|
|
2050
|
+
await runSubtitlePass(transcribeFrom, nativeDest, subtitleFormat, WHISPER_MODEL, detectedLang, threads, false, extraOpts);
|
|
2051
|
+
results.push(`✅ Native (${detectedLang}): ${nativeDest}`);
|
|
1631
2052
|
if (translateToEnglish && detectedLang !== "en") {
|
|
1632
|
-
const
|
|
1633
|
-
await
|
|
1634
|
-
results.push(`✅ English translation: ${
|
|
2053
|
+
const englishDest = `${baseNoExt}.en${ext}`;
|
|
2054
|
+
await runSubtitlePass(transcribeFrom, englishDest, subtitleFormat, WHISPER_MODEL, detectedLang, threads, true, extraOpts);
|
|
2055
|
+
results.push(`✅ English translation: ${englishDest}`);
|
|
2056
|
+
}
|
|
2057
|
+
if (tmpFile) {
|
|
2058
|
+
activeTempFiles.delete(tmpFile);
|
|
2059
|
+
if (existsSync(tmpFile))
|
|
2060
|
+
try {
|
|
2061
|
+
unlinkSync(tmpFile);
|
|
2062
|
+
}
|
|
2063
|
+
catch { }
|
|
1635
2064
|
}
|
|
1636
|
-
// Clean up temp WAV
|
|
1637
|
-
if (tmpFile && existsSync(tmpFile))
|
|
1638
|
-
try {
|
|
1639
|
-
unlinkSync(tmpFile);
|
|
1640
|
-
}
|
|
1641
|
-
catch { }
|
|
1642
2065
|
const langNote = language === "auto"
|
|
1643
2066
|
? `Auto-detected language: ${detectedLang}\n\n`
|
|
1644
2067
|
: "";
|
|
2068
|
+
const playerNote = subtitleFormat === "vtt"
|
|
2069
|
+
? `Load in web players, HTML5 <video>, or any player that supports WebVTT.`
|
|
2070
|
+
: `Load in VLC via Subtitle → Add Subtitle File → select the .srt file.`;
|
|
1645
2071
|
return {
|
|
1646
2072
|
content: [{
|
|
1647
2073
|
type: "text",
|
|
1648
2074
|
text: `✅ Subtitle file(s) generated!\n\n` +
|
|
1649
2075
|
langNote +
|
|
1650
2076
|
results.join("\n") + "\n\n" +
|
|
1651
|
-
|
|
1652
|
-
`Works in any video player that supports external subtitles.\n\n` +
|
|
2077
|
+
playerNote + "\n\n" +
|
|
1653
2078
|
`Note: whisper's built-in translation only translates to English. ` +
|
|
1654
|
-
`For other target languages, translate the
|
|
2079
|
+
`For other target languages, translate the subtitle file contents separately.`,
|
|
1655
2080
|
}],
|
|
1656
2081
|
};
|
|
1657
2082
|
}
|
|
@@ -1660,7 +2085,7 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1660
2085
|
}
|
|
1661
2086
|
}
|
|
1662
2087
|
// -------------------------------------------------------------------------
|
|
1663
|
-
// transcribe_batch (interactive
|
|
2088
|
+
// transcribe_batch (interactive)
|
|
1664
2089
|
// -------------------------------------------------------------------------
|
|
1665
2090
|
if (name === "transcribe_batch") {
|
|
1666
2091
|
const folderPath = args?.folder_path;
|
|
@@ -1668,6 +2093,9 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1668
2093
|
const threads = Math.min(SYSTEM_THREADS, Math.max(1, Math.round(args?.threads || WHISPER_THREADS)));
|
|
1669
2094
|
const recursive = args?.recursive || false;
|
|
1670
2095
|
const fileIndex = args?.file_index;
|
|
2096
|
+
const outputFormat = (args?.output_format || "timestamps");
|
|
2097
|
+
const privacyModeParam = args?.privacy_mode;
|
|
2098
|
+
const effectivePrivacyMode = privacyModeParam ?? WHISPER_PRIVACY_MODE;
|
|
1671
2099
|
if (!folderPath)
|
|
1672
2100
|
return { content: [{ type: "text", text: "folder_path is required." }], isError: true };
|
|
1673
2101
|
if (!existsSync(folderPath))
|
|
@@ -1684,7 +2112,6 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1684
2112
|
}],
|
|
1685
2113
|
};
|
|
1686
2114
|
}
|
|
1687
|
-
// No file_index: return file list
|
|
1688
2115
|
if (fileIndex === undefined) {
|
|
1689
2116
|
return {
|
|
1690
2117
|
content: [{
|
|
@@ -1696,11 +2123,10 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1696
2123
|
return ` ${i + 1}. ${basename(f)}${done}`;
|
|
1697
2124
|
}).join("\n") +
|
|
1698
2125
|
`\n\nTo start, say "transcribe file 1" (or any number). I'll process one file at a time and wait for your go-ahead before continuing.\n` +
|
|
1699
|
-
`\nFor large unattended batches,
|
|
2126
|
+
`\nFor large unattended batches, use start_batch instead.`,
|
|
1700
2127
|
}],
|
|
1701
2128
|
};
|
|
1702
2129
|
}
|
|
1703
|
-
// Process the requested file
|
|
1704
2130
|
const idx = fileIndex - 1;
|
|
1705
2131
|
if (idx < 0 || idx >= files.length) {
|
|
1706
2132
|
return { content: [{ type: "text", text: `Invalid file number. Choose between 1 and ${files.length}.` }], isError: true };
|
|
@@ -1709,17 +2135,42 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1709
2135
|
const fileName = basename(filePath);
|
|
1710
2136
|
const txtPath = filePath.replace(/\.[^.]+$/, ".txt");
|
|
1711
2137
|
try {
|
|
1712
|
-
|
|
2138
|
+
// v2.3.0: Privacy gate fires before each file in privacy mode.
|
|
2139
|
+
// Each file in transcribe_batch is a separate tool call, so each
|
|
2140
|
+
// gets its own confirmation — correct for interactive mode.
|
|
2141
|
+
if (effectivePrivacyMode && checkPrivacyGate(opKeyFor(name, args))) {
|
|
2142
|
+
return { content: [{ type: "text", text: privacyGateBlock() }] };
|
|
2143
|
+
}
|
|
2144
|
+
// Non-privacy mode: session consent gate fires once before first file.
|
|
2145
|
+
if (!effectivePrivacyMode) {
|
|
2146
|
+
const policy = transcriptPolicy();
|
|
2147
|
+
if (policy === "consent_gate") {
|
|
2148
|
+
return { content: [{ type: "text", text: consentGateBlock() }] };
|
|
2149
|
+
}
|
|
2150
|
+
}
|
|
2151
|
+
const result = await transcribeSingle(filePath, WHISPER_MODEL, language, outputFormat, threads, true, {});
|
|
1713
2152
|
const remaining = files.length - fileIndex;
|
|
1714
2153
|
const nextMsg = remaining > 0
|
|
1715
2154
|
? `\n\n${remaining} file(s) remaining. Say "continue" or "transcribe file ${fileIndex + 1}" to proceed, or "stop" to finish.`
|
|
1716
2155
|
: `\n\n✅ That was the last file. Batch complete!`;
|
|
2156
|
+
let bodyText;
|
|
2157
|
+
if (effectivePrivacyMode) {
|
|
2158
|
+
const words = estimateWordCount(result.text);
|
|
2159
|
+
bodyText =
|
|
2160
|
+
`Saved to: ${txtPath}\n` +
|
|
2161
|
+
`Words: ~${words}\n\n` +
|
|
2162
|
+
`Privacy mode active — transcript not returned to Claude's API.`;
|
|
2163
|
+
}
|
|
2164
|
+
else {
|
|
2165
|
+
bodyText =
|
|
2166
|
+
`Saved to: ${txtPath}\n\n` +
|
|
2167
|
+
`Preview:\n${result.text.slice(0, 500)}${result.text.length > 500 ? "..." : ""}`;
|
|
2168
|
+
}
|
|
1717
2169
|
return {
|
|
1718
2170
|
content: [{
|
|
1719
2171
|
type: "text",
|
|
1720
2172
|
text: `[${fileIndex}/${files.length}] ✅ ${fileName}\n\n` +
|
|
1721
|
-
|
|
1722
|
-
`Preview:\n${result.text.slice(0, 500)}${result.text.length > 500 ? "..." : ""}` +
|
|
2173
|
+
bodyText +
|
|
1723
2174
|
nextMsg,
|
|
1724
2175
|
}],
|
|
1725
2176
|
};
|
|
@@ -1742,9 +2193,12 @@ server.setRequestHandler(CallToolRequestSchema, async (request) => {
|
|
|
1742
2193
|
// Start
|
|
1743
2194
|
// ---------------------------------------------------------------------------
|
|
1744
2195
|
async function main() {
|
|
2196
|
+
cleanupOldJobFiles();
|
|
2197
|
+
process.on("SIGINT", () => gracefulShutdown("SIGINT"));
|
|
2198
|
+
process.on("SIGTERM", () => gracefulShutdown("SIGTERM"));
|
|
1745
2199
|
const transport = new StdioServerTransport();
|
|
1746
2200
|
await server.connect(transport);
|
|
1747
|
-
console.error(`whisper-windows-mcp v2.
|
|
2201
|
+
console.error(`whisper-windows-mcp v2.4.0 running | threads: ${WHISPER_THREADS}/${SYSTEM_THREADS} | privacy: ${WHISPER_PRIVACY_MODE ? "on" : "off"}`);
|
|
1748
2202
|
}
|
|
1749
2203
|
main().catch((err) => {
|
|
1750
2204
|
console.error("Fatal error:", err);
|