mac-voice-mcp 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.js ADDED
@@ -0,0 +1,122 @@
1
+ /** MCP wiring: two tools (speak_and_listen, voice_setup) and two prompts (setup, voice_mode). */
2
+ import { McpServer } from "@modelcontextprotocol/sdk/server/mcp.js";
3
+ import { StdioServerTransport } from "@modelcontextprotocol/sdk/server/stdio.js";
4
+ import { z } from "zod";
5
+ import { CONFIG, DEFAULT_LISTEN_SECONDS, log, MAX_LISTEN_SECONDS, PKG } from "./config.js";
6
+ import { locateModel } from "./model.js";
7
+ import { activeChildren, SetupError } from "./proc.js";
8
+ import { runSetupFlow } from "./setup.js";
9
+ import { stopWhisperServer } from "./stt.js";
10
+ import { SERVER_INSTRUCTIONS, SETUP_PROMPT, SETUP_TOOL_DESCRIPTION, SPEAK_TOOL_DESCRIPTION, VOICE_MODE_PROMPT } from "./texts.js";
11
+ import { exclusive, speakAndListen } from "./voice.js";
12
+ /**
13
+ * MCP progress notifications, when the client asked for them (it sent a progressToken).
14
+ * They show what's happening and keep clients that reset their timeout on progress happy.
15
+ */
16
+ function progressReporter(extra) {
17
+ const token = extra._meta?.progressToken;
18
+ if (token === undefined)
19
+ return undefined;
20
+ let progress = 0;
21
+ return (message) => {
22
+ extra
23
+ .sendNotification({ method: "notifications/progress", params: { progressToken: token, progress: ++progress, message } })
24
+ .catch(() => { });
25
+ };
26
+ }
27
+ export function createServer() {
28
+ const server = new McpServer({ name: "voice-mcp", title: "Voice bridge", version: PKG.version }, { instructions: SERVER_INSTRUCTIONS });
29
+ server.registerTool("speak_and_listen", {
30
+ title: "Speak and listen",
31
+ description: SPEAK_TOOL_DESCRIPTION,
32
+ inputSchema: {
33
+ text_to_speak: z.string().min(1).describe("The summary or message to read aloud to the user. Plain, conversational text."),
34
+ listen_seconds: z
35
+ .number()
36
+ .positive()
37
+ .optional()
38
+ .describe(`Upper limit on how long to listen, in seconds (default ${DEFAULT_LISTEN_SECONDS}, max ${MAX_LISTEN_SECONDS}). ` +
39
+ "Listening already stops when the user finishes talking, so you rarely need this."),
40
+ },
41
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: false, openWorldHint: false },
42
+ }, async ({ text_to_speak, listen_seconds }, extra) => {
43
+ const progress = progressReporter(extra);
44
+ try {
45
+ const result = await exclusive(() => speakAndListen(text_to_speak, listen_seconds ?? DEFAULT_LISTEN_SECONDS, extra.signal, (phase) => progress?.(phase)));
46
+ return {
47
+ content: [{ type: "text", text: result.text }, ...result.notes.map((note) => ({ type: "text", text: `\n\n${note}` }))],
48
+ isError: !result.ok,
49
+ };
50
+ }
51
+ catch (err) {
52
+ const message = err instanceof Error ? err.message : String(err);
53
+ log("speak_and_listen failed:", message);
54
+ const prefix = err instanceof SetupError ? "voice-mcp is not set up yet: " : "voice-mcp error: ";
55
+ return { content: [{ type: "text", text: prefix + message }], isError: true };
56
+ }
57
+ });
58
+ server.registerTool("voice_setup", {
59
+ title: "Voice setup",
60
+ description: SETUP_TOOL_DESCRIPTION,
61
+ inputSchema: {
62
+ install: z
63
+ .boolean()
64
+ .optional()
65
+ .describe("false (default): only check. true: brew install missing packages and download the model. Ask the user first."),
66
+ },
67
+ annotations: { readOnlyHint: false, destructiveHint: false, idempotentHint: true, openWorldHint: true },
68
+ }, async ({ install }, extra) => {
69
+ try {
70
+ const outcome = await exclusive(() => runSetupFlow(install === true, progressReporter(extra)));
71
+ return { content: [{ type: "text", text: outcome.report }] };
72
+ }
73
+ catch (err) {
74
+ const message = err instanceof Error ? err.message : String(err);
75
+ log("voice_setup failed:", message);
76
+ return { content: [{ type: "text", text: `voice_setup error: ${message}` }], isError: true };
77
+ }
78
+ });
79
+ // Claude Code: /mcp__voice-mcp__setup
80
+ server.registerPrompt("setup", { title: "Set up voice", description: "Check what voice-mcp needs, install what's missing (with your OK), then test it." }, () => ({ messages: [{ role: "user", content: { type: "text", text: SETUP_PROMPT } }] }));
81
+ // Claude Code: /mcp__voice-mcp__voice_mode [task]; Claude Desktop and Cursor: the prompt picker.
82
+ server.registerPrompt("voice_mode", {
83
+ title: "Voice mode",
84
+ description: "Hands-free session: Claude checks in out loud via speak_and_listen, using speakable phrasing.",
85
+ argsSchema: {
86
+ task: z.string().optional().describe("Optional: what to work on. Claude will read its plan back to you first."),
87
+ },
88
+ }, ({ task }) => ({
89
+ messages: [{ role: "user", content: { type: "text", text: VOICE_MODE_PROMPT(task?.trim() || undefined) } }],
90
+ }));
91
+ return server;
92
+ }
93
+ export async function startServer() {
94
+ const server = createServer();
95
+ const transport = new StdioServerTransport();
96
+ let shuttingDown = false;
97
+ const shutdown = async (code = 0) => {
98
+ if (shuttingDown)
99
+ return;
100
+ shuttingDown = true;
101
+ stopWhisperServer();
102
+ for (const child of activeChildren)
103
+ child.kill("SIGTERM");
104
+ try {
105
+ await server.close();
106
+ }
107
+ catch {
108
+ /* ignore */
109
+ }
110
+ process.exit(code);
111
+ };
112
+ process.stdin.on("close", () => void shutdown(0));
113
+ process.on("SIGINT", () => void shutdown(0));
114
+ process.on("SIGTERM", () => void shutdown(0));
115
+ await server.connect(transport);
116
+ log(`${PKG.name} v${PKG.version} ready on stdio (model: ${CONFIG.modelPath ?? CONFIG.modelName}, language: ${CONFIG.language})`);
117
+ if (process.stdin.isTTY) {
118
+ log("This is an MCP server and expects an MCP client on stdin. Try `setup` or `test` instead, or see --help.");
119
+ }
120
+ // Cheap, install-free: link an existing model into the cache if there is one.
121
+ void locateModel().catch(() => null);
122
+ }
package/dist/setup.js ADDED
@@ -0,0 +1,183 @@
1
+ /**
2
+ * voice_setup: check what's already on this machine, install only what's missing,
3
+ * and only when asked (install=true, after the user agreed).
4
+ *
5
+ * Installs run in the background: if they take longer than a tool call should
6
+ * wait, the report says "installing…" and the next voice_setup call picks up
7
+ * the result. A slow `brew install` can never time the client out.
8
+ */
9
+ import { statSync } from "node:fs";
10
+ import { describeRecorder, findRecorder, findTts } from "./audio.js";
11
+ import { CONFIG, IS_MAC, IS_WIN, log } from "./config.js";
12
+ import { ensureModel, isModelDownloading, locateModel, modelSource, MODEL_SIZES_MB } from "./model.js";
13
+ import { isExecutable, resetWhichCache, run, sleep, tail, which } from "./proc.js";
14
+ import { describeStt, findWhisperCli, findWhisperServer } from "./stt.js";
15
+ /** How long one voice_setup call waits for installs before reporting "still installing". */
16
+ const INSTALL_WAIT_MS = 40_000;
17
+ export function findBrew() {
18
+ if (process.env.VOICE_MCP_EXTRA_PATH !== undefined)
19
+ return which("brew"); // isolated (tests): PATH only
20
+ return which("brew") ?? ["/opt/homebrew/bin/brew", "/usr/local/bin/brew"].find(isExecutable) ?? null;
21
+ }
22
+ let brewJob = null;
23
+ let modelJob = null;
24
+ /** Results of finished jobs not yet shown to the user. */
25
+ const finishedMessages = [];
26
+ function startJob(label, work) {
27
+ const job = { label, startedAt: Date.now(), done: false, promise: Promise.resolve() };
28
+ job.promise = work()
29
+ .then((msg) => {
30
+ job.result = msg;
31
+ })
32
+ .catch((e) => {
33
+ job.result = `${label} failed: ${e instanceof Error ? e.message : e}`;
34
+ })
35
+ .finally(() => {
36
+ job.done = true;
37
+ resetWhichCache();
38
+ finishedMessages.push(job.result);
39
+ log(job.result);
40
+ });
41
+ return job;
42
+ }
43
+ function startBrewInstall(brew, formulae) {
44
+ return startJob(`brew install ${formulae.join(" ")}`, async () => {
45
+ log(`Running: brew install ${formulae.join(" ")}`);
46
+ const r = await run(brew, ["install", ...formulae], {
47
+ timeoutMs: 30 * 60_000,
48
+ env: { HOMEBREW_NO_AUTO_UPDATE: "1", HOMEBREW_NO_ENV_HINTS: "1", HOMEBREW_NO_INSTALL_CLEANUP: "1", NONINTERACTIVE: "1" },
49
+ });
50
+ if (r.code !== 0)
51
+ throw new Error(`exit ${r.code}: ${tail(r.stderr, 4)}`);
52
+ return `Installed with Homebrew: ${formulae.join(", ")}.`;
53
+ });
54
+ }
55
+ function startModelDownload() {
56
+ return startJob(`Downloading the ${CONFIG.modelName} model`, async () => {
57
+ await ensureModel({ download: true });
58
+ return `Downloaded the ${CONFIG.modelName} speech model.`;
59
+ });
60
+ }
61
+ const running = (j) => !!j && !j.done;
62
+ const elapsed = (j) => `${Math.round((Date.now() - j.startedAt) / 1000)}s`;
63
+ // --- Checks -------------------------------------------------------------------------------
64
+ async function checkRequirements() {
65
+ const checks = [];
66
+ const brewing = running(brewJob);
67
+ const tts = findTts();
68
+ checks.push(tts
69
+ ? { label: "Text-to-speech", status: "ok", detail: IS_MAC ? "macOS `say` (built in)" : tts }
70
+ : { label: "Text-to-speech", status: "missing", detail: "no engine found — install espeak-ng (`sudo apt install espeak-ng`)" });
71
+ const rec = findRecorder();
72
+ checks.push(rec
73
+ ? { label: "Recorder", status: "ok", detail: describeRecorder(rec) }
74
+ : { label: "Recorder", status: brewing ? "installing" : "missing", detail: "SoX is not installed", brew: IS_WIN ? undefined : "sox" });
75
+ const cli = findWhisperCli();
76
+ const server = findWhisperServer();
77
+ if (cli || server) {
78
+ checks.push({ label: "Speech-to-text", status: "ok", detail: `whisper.cpp — ${describeStt()}` });
79
+ }
80
+ else {
81
+ checks.push({
82
+ label: "Speech-to-text",
83
+ status: brewing ? "installing" : "missing",
84
+ detail: "whisper.cpp is not installed",
85
+ brew: IS_WIN ? undefined : "whisper-cpp",
86
+ });
87
+ }
88
+ try {
89
+ const model = running(modelJob) || isModelDownloading() ? null : await locateModel();
90
+ if (model) {
91
+ const mb = (statSync(model).size / 1e6).toFixed(0);
92
+ checks.push({ label: "Speech model", status: "ok", detail: `${CONFIG.modelName} (${mb} MB) — ${modelSource}` });
93
+ }
94
+ else {
95
+ const size = MODEL_SIZES_MB[CONFIG.modelName];
96
+ checks.push({
97
+ label: "Speech model",
98
+ status: running(modelJob) ? "installing" : "missing",
99
+ detail: running(modelJob)
100
+ ? `downloading ${CONFIG.modelName} (${elapsed(modelJob)} so far)`
101
+ : `${CONFIG.modelName} is not on this computer — one-time download${size ? ` of ~${size} MB` : ""} to ${CONFIG.modelsDir}`,
102
+ model: true,
103
+ });
104
+ }
105
+ }
106
+ catch (e) {
107
+ checks.push({ label: "Speech model", status: "missing", detail: e instanceof Error ? e.message : String(e) });
108
+ }
109
+ if (checks.some((c) => c.brew && c.status === "missing")) {
110
+ const brew = findBrew();
111
+ checks.push(brew
112
+ ? { label: "Homebrew", status: "ok", detail: `found (${brew})` }
113
+ : {
114
+ label: "Homebrew",
115
+ status: "missing",
116
+ detail: "not installed, so packages can't be installed automatically — install it in Terminal from https://brew.sh (it needs your password), then run setup again",
117
+ });
118
+ }
119
+ return checks;
120
+ }
121
+ // --- The flow ------------------------------------------------------------------------------
122
+ /**
123
+ * @param install true only after the user agreed: brew-install missing formulae, download the model if absent.
124
+ * @param onProgress optional heartbeat while waiting (used for MCP progress notifications).
125
+ */
126
+ export async function runSetupFlow(install, onProgress) {
127
+ let checks = await checkRequirements();
128
+ if (install) {
129
+ const formulae = [...new Set(checks.filter((c) => c.status === "missing" && c.brew).map((c) => c.brew))];
130
+ const brew = findBrew();
131
+ if (formulae.length && brew && !running(brewJob))
132
+ brewJob = startBrewInstall(brew, formulae);
133
+ if (checks.some((c) => c.model && c.status === "missing") && !running(modelJob))
134
+ modelJob = startModelDownload();
135
+ }
136
+ // Wait (bounded) for anything in flight, with a heartbeat.
137
+ const jobs = [brewJob, modelJob].filter(running);
138
+ if (jobs.length) {
139
+ const deadline = Date.now() + INSTALL_WAIT_MS;
140
+ while (jobs.some((j) => !j.done) && Date.now() < deadline) {
141
+ onProgress?.(jobs.filter((j) => !j.done).map((j) => `${j.label} (${elapsed(j)})`).join("; "));
142
+ await Promise.race([Promise.all(jobs.map((j) => j.promise)), sleep(5000)]);
143
+ }
144
+ checks = await checkRequirements();
145
+ }
146
+ const installing = running(brewJob) || running(modelJob);
147
+ const ready = checks.every((c) => c.status === "ok" || c.status === "optional");
148
+ const icon = { ok: "✔", missing: "✘", optional: "•", installing: "…" };
149
+ const done = finishedMessages.splice(0);
150
+ const lines = [
151
+ `voice-mcp setup — ${ready ? "READY" : installing ? "INSTALLING" : "NOT READY"}`,
152
+ "",
153
+ ...checks.map((c) => `${icon[c.status]} ${c.label}: ${c.detail}${c.brew && c.status === "missing" ? ` → brew install ${c.brew}` : ""}`),
154
+ IS_MAC ? "• Microphone: macOS asks for permission the first time speak_and_listen listens — click Allow." : "",
155
+ ];
156
+ if (done.length)
157
+ lines.push("", "What was done:", ...done.map((d) => `- ${d}`));
158
+ if (install && !done.length && !installing && ready)
159
+ lines.push("", "Nothing to install — everything was already present.");
160
+ const toInstall = [...new Set(checks.filter((c) => c.status === "missing" && c.brew).map((c) => c.brew))];
161
+ const needsModel = checks.some((c) => c.model && c.status === "missing");
162
+ lines.push("");
163
+ if (ready) {
164
+ lines.push("Next: everything is in place — speak_and_listen is ready to use.");
165
+ }
166
+ else if (installing) {
167
+ lines.push("Next: installation is still running in the background. Tell the user, wait about a minute, then call voice_setup again (install=false) to check.");
168
+ }
169
+ else if (!install && (toInstall.length || needsModel) && (findBrew() || !toInstall.length)) {
170
+ const steps = [toInstall.length ? `brew install ${toInstall.join(" ")}` : "", needsModel ? `download the ${CONFIG.modelName} model` : ""]
171
+ .filter(Boolean)
172
+ .join(" and ");
173
+ lines.push(`Next: ask the user whether to ${steps}. Only if they agree, call voice_setup with install=true.`);
174
+ }
175
+ else {
176
+ lines.push("Next: the items marked ✘ need the user's attention (see details above); then call voice_setup again.");
177
+ }
178
+ return {
179
+ ready,
180
+ installing,
181
+ report: lines.filter((l, i, a) => !(l === "" && (i === 0 || a[i - 1] === ""))).join("\n").trim(),
182
+ };
183
+ }
@@ -0,0 +1,104 @@
1
+ /**
2
+ * Pure text helpers (no I/O, unit-tested):
3
+ * - prepareSpeech: turn "screen text" into "ear text" before it's spoken
4
+ * - cleanTranscript: strip whisper.cpp's non-speech markers from a transcript
5
+ */
6
+ /**
7
+ * Safety net for speakability. The model is asked (tool description, server
8
+ * instructions, voice_mode prompt) to send plain spoken sentences; this catches
9
+ * whatever slips through so the user never hears "backtick backtick backtick"
10
+ * or a 40-segment file path read aloud.
11
+ */
12
+ export function prepareSpeech(input, opts = {}) {
13
+ const maxWords = opts.maxWords ?? 120;
14
+ const maxChars = opts.maxChars ?? 4000;
15
+ const notes = new Set();
16
+ let s = input.replace(/\r\n?/g, "\n");
17
+ // Fenced code blocks and markdown tables can't be spoken meaningfully.
18
+ s = s.replace(/```[\s\S]*?(?:```|$)/g, () => {
19
+ notes.add("code blocks");
20
+ return "\n(I've left the code on screen.)\n";
21
+ });
22
+ s = s.replace(/(?:^[ \t]*\|.*\|[ \t]*$\n?){2,}/gm, () => {
23
+ notes.add("tables");
24
+ return "\n(There's a table on screen.)\n";
25
+ });
26
+ // Inline code: keep short, word-like snippets; drop long or symbol-heavy ones.
27
+ s = s.replace(/`([^`\n]+)`/g, (_m, code) => {
28
+ if (code.length <= 32 && !/[{}();=<>[\]$]/.test(code))
29
+ return code;
30
+ notes.add("inline code");
31
+ return "that code";
32
+ });
33
+ s = s
34
+ .replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1") // images → alt text
35
+ .replace(/\[([^\]]+)\]\([^)]*\)/g, "$1") // links → link text
36
+ .replace(/<\/?[a-z][^>]*>/gi, " "); // stray HTML tags
37
+ // Bare URLs → "a link to github.com".
38
+ s = s.replace(/\bhttps?:\/\/(?:www\.)?([^\s/?#)]+)[^\s)]*?(?=[.,;:!?)]*(?:\s|$))/g, (_m, host) => {
39
+ notes.add("URLs");
40
+ return `a link to ${host}`;
41
+ });
42
+ // Long file paths → just the file name.
43
+ s = s.replace(/(?<![\w/:])(?:(?:~|\.{1,2})?(?:\/[\w@.+-]+){2,}|[\w@.+-]+(?:\/[\w@.+-]+){2,})\/?/g, (m) => {
44
+ if (/^[\d/.-]+$/.test(m))
45
+ return m; // dates like 9/23/2026
46
+ notes.add("file paths");
47
+ const parts = m.split("/").filter(Boolean);
48
+ return parts[parts.length - 1] ?? m;
49
+ });
50
+ s = s
51
+ .replace(/\b([\w-]+\.[a-z]{1,5}):\d+(?::\d+)?\b/gi, "$1") // index.ts:120 → index.ts
52
+ .replace(/(?<![\d/])(\d+)\/(\d+)(?![\d/])/g, "$1 of $2"); // 42/42 → 42 of 42
53
+ // Markdown structure → sentences. Headings and bullets become their own sentence.
54
+ s = s
55
+ .split("\n")
56
+ .map((line) => line
57
+ .replace(/^\s{0,3}#{1,6}\s+/, "")
58
+ .replace(/^\s*>\s?/, "")
59
+ .replace(/^\s*(?:[-*+•]|\d+[.)])\s+/, "")
60
+ .replace(/^\s*[-*_]{3,}\s*$/, "")
61
+ .trim())
62
+ .filter(Boolean)
63
+ .map((line) => (/[.!?:;,…)]$/.test(line) ? line : `${line}.`))
64
+ .join(" ");
65
+ s = s
66
+ .replace(/(\*\*|__|\*|~~)(?=\S)([\s\S]*?\S)\1/g, "$2") // emphasis markers
67
+ .replace(/\p{Extended_Pictographic}️?/gu, "") // emoji
68
+ .replace(/\s*(?:->|=>|→)\s*/g, " to ")
69
+ .replace(/\s&\s/g, " and ")
70
+ .replace(/[*_#~|^\\]+/g, " ")
71
+ .replace(/\s+([.,;:!?])/g, "$1")
72
+ .replace(/([.!?])\.+/g, "$1")
73
+ .replace(/\s+/g, " ")
74
+ .trim();
75
+ // Keep it listenable: cut at a sentence boundary once it gets long.
76
+ const words = s.split(" ");
77
+ if (maxWords > 0 && words.length > maxWords) {
78
+ let cut = words.slice(0, maxWords).join(" ");
79
+ const lastStop = Math.max(cut.lastIndexOf(". "), cut.lastIndexOf("? "), cut.lastIndexOf("! "));
80
+ if (lastStop > cut.length * 0.4)
81
+ cut = cut.slice(0, lastStop + 1);
82
+ s = `${cut} The rest is on screen.`;
83
+ notes.add(`length (over ${maxWords} words)`);
84
+ }
85
+ s = s.slice(0, maxChars);
86
+ const noteList = notes.size
87
+ ? [
88
+ `voice-mcp note: text_to_speak was rewritten for speech (removed/shortened: ${[...notes].join(", ")}). ` +
89
+ "Next time send only short, plain spoken sentences — keep code, paths, links and tables in your on-screen reply.",
90
+ ]
91
+ : [];
92
+ return { text: s, notes: noteList };
93
+ }
94
+ /** Remove timestamps and non-speech markers like [BLANK_AUDIO] from whisper output. */
95
+ export function cleanTranscript(raw) {
96
+ return raw
97
+ .split("\n")
98
+ .map((l) => l.replace(/^\s*\[[\d:.\s\->]+\]\s*/, "")) // stray timestamps
99
+ .join(" ")
100
+ .replace(/\[(?:BLANK_AUDIO|MUSIC|NOISE|SILENCE|INAUDIBLE|_[A-Z_]+_)[^\]]*\]/gi, " ")
101
+ .replace(/\[\s*(?:silence|music|noise|inaudible)\s*\]/gi, " ")
102
+ .replace(/\s+/g, " ")
103
+ .trim();
104
+ }
package/dist/stt.js ADDED
@@ -0,0 +1,266 @@
1
+ /**
2
+ * Speech-to-text with whisper.cpp.
3
+ *
4
+ * Fast path: a warm `whisper-server` on 127.0.0.1 keeps the model loaded between
5
+ * turns, so each reply skips the model load (and Metal start-up on Apple Silicon).
6
+ * It's started in the background while the question is being spoken, and stopped
7
+ * after VOICE_MCP_SERVER_IDLE_MINUTES of inactivity to give the memory back.
8
+ *
9
+ * Fallback: `whisper-cli`, one process per transcription. Always correct, just slower.
10
+ */
11
+ import { spawn } from "node:child_process";
12
+ import { readFile } from "node:fs/promises";
13
+ import net from "node:net";
14
+ import os from "node:os";
15
+ import path from "node:path";
16
+ import { CONFIG, debug, IS_WIN, log } from "./config.js";
17
+ import { CancelledError, isExecutable, run, SetupError, sleep, tail, which } from "./proc.js";
18
+ import { cleanTranscript } from "./speech-text.js";
19
+ export const STT_MISSING = "whisper.cpp (speech-to-text) is not installed. Call the voice_setup tool to check and install what's missing " +
20
+ "(or run `brew install whisper-cpp`, or set VOICE_MCP_WHISPER_BIN to an existing `whisper-cli`).";
21
+ /** Where people usually clone and build whisper.cpp themselves. */
22
+ function sourceBuildDirs() {
23
+ const home = os.homedir();
24
+ return ["", "src", "code", "Code", "dev", "Developer", "projects", "Projects", "workspace", "git", "repos", "GitHub", "Downloads"].flatMap((d) => [path.join(home, d, "whisper.cpp", "build", "bin"), path.join(home, d, "whisper.cpp")]);
25
+ }
26
+ export function findWhisperCli() {
27
+ if (CONFIG.whisperBin)
28
+ return which(CONFIG.whisperBin);
29
+ // Homebrew's whisper-cpp formula ships `whisper-cli` (older releases: `whisper-cpp`).
30
+ for (const bin of ["whisper-cli", "whisper-cpp", "whisper-cpp-cli"]) {
31
+ const found = which(bin);
32
+ if (found)
33
+ return found;
34
+ }
35
+ // A whisper.cpp you built yourself and never put on PATH. Old builds call the CLI `main`.
36
+ for (const dir of sourceBuildDirs()) {
37
+ for (const bin of ["whisper-cli", "main"]) {
38
+ const candidate = path.join(dir, bin);
39
+ if (isExecutable(candidate))
40
+ return candidate;
41
+ }
42
+ }
43
+ return null;
44
+ }
45
+ export function findWhisperServer() {
46
+ if (CONFIG.whisperServerBin)
47
+ return which(CONFIG.whisperServerBin);
48
+ const onPath = which("whisper-server");
49
+ if (onPath)
50
+ return onPath;
51
+ const cli = findWhisperCli();
52
+ if (!cli)
53
+ return null;
54
+ // Next to the CLI; old source builds call the server `server` (and the CLI `main`).
55
+ const names = path.basename(cli) === "main" ? ["whisper-server", "server"] : ["whisper-server"];
56
+ for (const name of names) {
57
+ const sibling = path.join(path.dirname(cli), IS_WIN ? `${name}.exe` : name);
58
+ if (isExecutable(sibling))
59
+ return sibling;
60
+ }
61
+ return null;
62
+ }
63
+ function freePort() {
64
+ return new Promise((resolve, reject) => {
65
+ const srv = net.createServer();
66
+ srv.unref();
67
+ srv.on("error", reject);
68
+ srv.listen(0, "127.0.0.1", () => {
69
+ const { port } = srv.address();
70
+ srv.close(() => resolve(port));
71
+ });
72
+ });
73
+ }
74
+ /** fetch() with a timeout that also honours the caller's abort signal (Node 18 compatible). */
75
+ async function fetchWithin(url, init, timeoutMs, signal) {
76
+ const ctl = new AbortController();
77
+ const timer = setTimeout(() => ctl.abort(), timeoutMs);
78
+ const onAbort = () => ctl.abort();
79
+ signal?.addEventListener("abort", onAbort, { once: true });
80
+ try {
81
+ return await fetch(url, { ...init, signal: ctl.signal });
82
+ }
83
+ finally {
84
+ clearTimeout(timer);
85
+ signal?.removeEventListener("abort", onAbort);
86
+ }
87
+ }
88
+ class WarmWhisperServer {
89
+ child = null;
90
+ ready = null;
91
+ model = null;
92
+ idleTimer;
93
+ stderrTail = "";
94
+ failures = 0;
95
+ /** Too many start failures (e.g. an old whisper.cpp build) → stop trying for this session. */
96
+ get disabled() {
97
+ return this.failures >= 2;
98
+ }
99
+ get running() {
100
+ return !!this.child && this.child.exitCode === null;
101
+ }
102
+ /** Start (or reuse) a server for `model`. Resolves with its base URL once the model is loaded. */
103
+ warm(model) {
104
+ if (this.ready && this.model === model && this.running) {
105
+ this.touch();
106
+ return this.ready;
107
+ }
108
+ this.stop();
109
+ const bin = findWhisperServer();
110
+ if (!bin)
111
+ return Promise.reject(new Error("whisper-server is not installed"));
112
+ this.model = model;
113
+ const ready = this.start(bin, model);
114
+ this.ready = ready;
115
+ ready.then(() => {
116
+ this.failures = 0;
117
+ }, (e) => {
118
+ this.failures++;
119
+ debug("whisper-server failed to start:", e instanceof Error ? e.message : e);
120
+ if (this.ready === ready)
121
+ this.stop();
122
+ });
123
+ this.touch();
124
+ return ready;
125
+ }
126
+ async start(bin, model) {
127
+ const port = await freePort();
128
+ const args = ["-m", model, "--host", "127.0.0.1", "--port", String(port), "-t", String(CONFIG.threads), "-l", CONFIG.language, "-nt"];
129
+ if (CONFIG.prompt)
130
+ args.push("--prompt", CONFIG.prompt);
131
+ debug("starting whisper-server:", bin, args.join(" "));
132
+ // Orphan guard: the server runs under a tiny shell that waits on our stdin pipe
133
+ // and kills the server when that pipe closes — i.e. whenever this process exits,
134
+ // even if it's killed hard. `detached` gives it a process group we can stop as one.
135
+ const child = IS_WIN
136
+ ? spawn(bin, args, { stdio: ["ignore", "ignore", "pipe"], windowsHide: true })
137
+ : spawn("/bin/sh", ["-c", '"$@" </dev/null & pid=$!; read -r _; kill $pid 2>/dev/null; wait $pid 2>/dev/null', "voice-mcp-whisper", bin, ...args], { stdio: ["pipe", "ignore", "pipe"], detached: true });
138
+ this.child = child;
139
+ this.stderrTail = "";
140
+ child.stdin?.on("error", () => { });
141
+ child.stderr?.setEncoding("utf8").on("data", (d) => (this.stderrTail = (this.stderrTail + d).slice(-3000)));
142
+ child.on("exit", () => {
143
+ if (this.child === child) {
144
+ this.child = null;
145
+ this.ready = null;
146
+ }
147
+ });
148
+ const base = `http://127.0.0.1:${port}`;
149
+ const deadline = Date.now() + 180_000; // large models can take a while to load
150
+ while (Date.now() < deadline) {
151
+ if (this.child !== child)
152
+ throw new Error(`whisper-server exited during start-up: ${tail(this.stderrTail, 3)}`);
153
+ try {
154
+ const res = await fetchWithin(`${base}/health`, {}, 1000);
155
+ // 503 = still loading. 404 = an older server without /health, but it's up.
156
+ if (res.ok || res.status === 404) {
157
+ debug("whisper-server ready at", base);
158
+ return base;
159
+ }
160
+ }
161
+ catch {
162
+ /* not listening yet */
163
+ }
164
+ await sleep(100);
165
+ }
166
+ throw new Error("whisper-server did not become ready in time");
167
+ }
168
+ async transcribe(wavFile, signal) {
169
+ if (!this.ready)
170
+ throw new Error("whisper-server is not running");
171
+ const base = await this.ready;
172
+ const form = new FormData();
173
+ form.append("file", new Blob([await readFile(wavFile)], { type: "audio/wav" }), "reply.wav");
174
+ form.append("temperature", "0.0");
175
+ form.append("temperature_inc", "0.2");
176
+ form.append("response_format", "json");
177
+ const res = await fetchWithin(`${base}/inference`, { method: "POST", body: form }, 120_000, signal);
178
+ if (signal?.aborted)
179
+ throw new CancelledError();
180
+ if (!res.ok)
181
+ throw new Error(`whisper-server HTTP ${res.status}: ${(await res.text()).slice(0, 200)}`);
182
+ const json = (await res.json());
183
+ if (json.error)
184
+ throw new Error(`whisper-server: ${json.error}`);
185
+ this.touch();
186
+ return cleanTranscript(json.text ?? "");
187
+ }
188
+ touch() {
189
+ clearTimeout(this.idleTimer);
190
+ this.idleTimer = setTimeout(() => {
191
+ debug("whisper-server idle — stopping it to free memory");
192
+ this.stop();
193
+ }, CONFIG.serverIdleMinutes * 60_000);
194
+ this.idleTimer.unref();
195
+ }
196
+ stop() {
197
+ clearTimeout(this.idleTimer);
198
+ const child = this.child;
199
+ this.child = null;
200
+ this.ready = null;
201
+ if (!child)
202
+ return;
203
+ child.stdin?.end(); // wrapper kills the server on EOF
204
+ try {
205
+ if (!IS_WIN && child.pid)
206
+ process.kill(-child.pid, "SIGTERM");
207
+ else
208
+ child.kill("SIGTERM");
209
+ }
210
+ catch {
211
+ /* already gone */
212
+ }
213
+ }
214
+ }
215
+ const warmServer = new WarmWhisperServer();
216
+ const serverUsable = () => CONFIG.useWhisperServer && !warmServer.disabled && !!findWhisperServer();
217
+ /** Start loading the model in the background (called while the question is being spoken). */
218
+ export function prewarm(model) {
219
+ if (serverUsable())
220
+ warmServer.warm(model).catch(() => { });
221
+ }
222
+ export function stopWhisperServer() {
223
+ warmServer.stop();
224
+ }
225
+ /** Which engine the next transcription will use — for the setup report. */
226
+ export function describeStt() {
227
+ const cli = findWhisperCli();
228
+ const server = findWhisperServer();
229
+ const parts = [cli ? `whisper-cli (${cli})` : "", server ? "fast mode via whisper-server" : ""].filter(Boolean);
230
+ if (server && !CONFIG.useWhisperServer)
231
+ parts[parts.length - 1] = "whisper-server disabled by VOICE_MCP_WHISPER_SERVER=0";
232
+ return parts.join("; ");
233
+ }
234
+ async function transcribeCli(bin, wavFile, model, signal) {
235
+ const args = ["-m", model, "-f", wavFile, "-l", CONFIG.language, "-t", String(CONFIG.threads), "-nt", "-np"];
236
+ if (CONFIG.prompt)
237
+ args.push("--prompt", CONFIG.prompt);
238
+ const result = await run(bin, args, { signal, timeoutMs: 180_000 });
239
+ if (result.code !== 0) {
240
+ throw new Error(`whisper.cpp failed (exit ${result.code}): ${tail(result.stderr) || "no output"}`);
241
+ }
242
+ return cleanTranscript(result.stdout);
243
+ }
244
+ export async function transcribe(wavFile, model, signal) {
245
+ const started = Date.now();
246
+ if (serverUsable()) {
247
+ try {
248
+ await warmServer.warm(model);
249
+ const text = await warmServer.transcribe(wavFile, signal);
250
+ debug(`transcribed via whisper-server in ${Date.now() - started} ms`);
251
+ return text;
252
+ }
253
+ catch (e) {
254
+ if (signal?.aborted || e instanceof CancelledError)
255
+ throw new CancelledError();
256
+ log("whisper-server unavailable, using whisper-cli:", e instanceof Error ? e.message : e);
257
+ warmServer.stop();
258
+ }
259
+ }
260
+ const cli = findWhisperCli();
261
+ if (!cli)
262
+ throw new SetupError(STT_MISSING);
263
+ const text = await transcribeCli(cli, wavFile, model, signal);
264
+ debug(`transcribed via whisper-cli in ${Date.now() - started} ms`);
265
+ return text;
266
+ }