mac-voice-mcp 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +268 -0
- package/dist/audio.js +178 -0
- package/dist/config.js +103 -0
- package/dist/endpointer.js +171 -0
- package/dist/index.js +130 -0
- package/dist/model.js +189 -0
- package/dist/proc.js +112 -0
- package/dist/server.js +122 -0
- package/dist/setup.js +183 -0
- package/dist/speech-text.js +104 -0
- package/dist/stt.js +266 -0
- package/dist/texts.js +90 -0
- package/dist/voice.js +74 -0
- package/examples/CLAUDE.md +22 -0
- package/examples/voice-mcp.mdc +27 -0
- package/package.json +67 -0
package/dist/texts.js
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Everything the model reads: speaking rules, server instructions, tool
|
|
3
|
+
* descriptions and prompts. The rules are shared so every client sees the same
|
|
4
|
+
* guidance through whichever channel it actually surfaces (Claude Desktop, for
|
|
5
|
+
* instance, ignores server instructions but always shows tool descriptions).
|
|
6
|
+
*/
|
|
7
|
+
export const SPEECH_RULES = [
|
|
8
|
+
"Write text_to_speak for the ear, not the screen:",
|
|
9
|
+
"- 1–3 short sentences, under ~40 words; lead with the outcome, then one question.",
|
|
10
|
+
"- Plain words only: no markdown, bullets, emoji, code, file paths, URLs, stack traces or tables.",
|
|
11
|
+
'- Describe code instead of reading it ("I added a retry to the upload function"), say file names',
|
|
12
|
+
' not paths ("in index.ts"), round numbers ("about two hundred ms"), spell out symbols.',
|
|
13
|
+
'- Ask one question at a time, answerable in a few words ("Should I deploy it — yes or no?").',
|
|
14
|
+
"- Put the details (diffs, logs, links) in your normal on-screen reply, and say so out loud.",
|
|
15
|
+
].join("\n");
|
|
16
|
+
/** Claude Code reads these (truncated at 2 KB) — keep under that. */
|
|
17
|
+
export const SERVER_INSTRUCTIONS = [
|
|
18
|
+
"voice-mcp speaks to the user through their computer's speakers and returns a local transcript of",
|
|
19
|
+
"their spoken reply (tool: speak_and_listen). Use it when the user asked to be kept in the loop by",
|
|
20
|
+
"voice, is in a voice/hands-free session, or you need a quick decision while they may be away from",
|
|
21
|
+
"the keyboard. Do not use it for routine output the user can read on screen.",
|
|
22
|
+
"",
|
|
23
|
+
SPEECH_RULES,
|
|
24
|
+
"",
|
|
25
|
+
"Handling replies:",
|
|
26
|
+
"- The result is a speech-to-text transcript: expect homophones and mangled jargon; interpret",
|
|
27
|
+
" generously, and confirm by voice before anything destructive or irreversible.",
|
|
28
|
+
'- "(No speech detected …)" means no answer: never treat silence as consent. Ask once more or',
|
|
29
|
+
" continue with safe work and report on screen.",
|
|
30
|
+
"- Once the user is talking with you by voice, stay in voice: answer every turn with speak_and_listen,",
|
|
31
|
+
" not a text reply, until they say stop or start typing.",
|
|
32
|
+
"- Listening ends on its own when the user stops talking; no need to set listen_seconds.",
|
|
33
|
+
'- If the result includes a "voice-mcp note", follow it.',
|
|
34
|
+
"",
|
|
35
|
+
"Setup: if speak_and_listen says voice-mcp is not set up, call voice_setup (check only), tell the user",
|
|
36
|
+
"what is missing, and call voice_setup with install=true only after they agree.",
|
|
37
|
+
].join("\n");
|
|
38
|
+
export const SPEAK_TOOL_DESCRIPTION = [
|
|
39
|
+
"Say something out loud to the user and hear their spoken reply.",
|
|
40
|
+
"Speaks text_to_speak through the computer's speakers, then listens like a conversation turn — it waits for",
|
|
41
|
+
"the user to start talking and stops when they finish — and returns an on-device transcript of what they said.",
|
|
42
|
+
"",
|
|
43
|
+
SPEECH_RULES,
|
|
44
|
+
"",
|
|
45
|
+
'Good: "The build passed and all tests are green. Want me to open the pull request?"',
|
|
46
|
+
'Bad: "## Results\\n- `npm test` ✅ 42/42\\n- see /Users/x/repo/src/index.ts:120"',
|
|
47
|
+
"",
|
|
48
|
+
"Once the user is talking with you by voice, keep the conversation in voice: answer each transcript with",
|
|
49
|
+
"another speak_and_listen call (not a text reply) until they say stop or start typing.",
|
|
50
|
+
'A reply of "(No speech detected …)" means the user did not answer — never treat it as consent.',
|
|
51
|
+
"If it reports that voice-mcp is not set up, call voice_setup.",
|
|
52
|
+
].join("\n");
|
|
53
|
+
export const SETUP_TOOL_DESCRIPTION = [
|
|
54
|
+
"Check whether this computer has everything speak_and_listen needs — text-to-speech, a microphone recorder (SoX),",
|
|
55
|
+
"whisper.cpp speech-to-text and the speech model — and optionally install what's missing.",
|
|
56
|
+
"Anything already installed is reused (an existing model elsewhere on disk is symlinked, not re-downloaded).",
|
|
57
|
+
"Call it with install=false (the default) first and tell the user what is missing.",
|
|
58
|
+
"Only call it with install=true after the user agrees: it runs `brew install` for the missing packages",
|
|
59
|
+
"and downloads the speech model once. It never uninstalls or changes anything else.",
|
|
60
|
+
'If it reports INSTALLING, the install continues in the background — wait a minute and call it again to check.',
|
|
61
|
+
].join("\n");
|
|
62
|
+
export const SETUP_PROMPT = [
|
|
63
|
+
"Set up voice-mcp on this computer.",
|
|
64
|
+
"1. Call voice_setup (check only) and give me a short summary of what's already installed and what's missing.",
|
|
65
|
+
"2. If something is missing, ask me before installing. Only if I agree, call voice_setup with install=true.",
|
|
66
|
+
" If it reports INSTALLING, wait a minute and check again until it's READY.",
|
|
67
|
+
"3. When it reports READY, test it: call speak_and_listen with a one-sentence greeting that asks me to say something back,",
|
|
68
|
+
" then tell me what you heard. If the result says the microphone returned silence, walk me through",
|
|
69
|
+
" allowing microphone access in System Settings → Privacy & Security → Microphone.",
|
|
70
|
+
].join("\n");
|
|
71
|
+
export const VOICE_MODE_PROMPT = (task) => [
|
|
72
|
+
"Let's work in voice mode. I may be away from the screen, so keep me in the loop through the",
|
|
73
|
+
"speak_and_listen tool instead of waiting for me to type.",
|
|
74
|
+
"",
|
|
75
|
+
"- This is a spoken conversation. Every reply to me goes through speak_and_listen, and each transcript",
|
|
76
|
+
" is my next turn. Don't fall back to text replies until I say \"stop voice mode\" or \"I'm back\".",
|
|
77
|
+
"- While you work on something, say briefly what you're about to do, do it, then report back by voice.",
|
|
78
|
+
" Don't narrate every small action.",
|
|
79
|
+
"- If a transcript is unclear, ask again by voice.",
|
|
80
|
+
"- Confirm by voice before anything destructive, irreversible, or that costs money.",
|
|
81
|
+
"- If I don't answer, don't assume yes: carry on with safe work or pause, and summarize on screen.",
|
|
82
|
+
"- Keep writing full details (code, diffs, links) on screen as usual; the voice line is the headline.",
|
|
83
|
+
'- Stop using voice when I say "stop voice mode", "I\'m back", or start typing again.',
|
|
84
|
+
"",
|
|
85
|
+
SPEECH_RULES,
|
|
86
|
+
"",
|
|
87
|
+
task
|
|
88
|
+
? `Start by briefly telling me out loud what you're about to do for: ${task}. Then ask if I want to change anything.`
|
|
89
|
+
: "Start by calling speak_and_listen to say voice mode is on and ask what I'd like to work on.",
|
|
90
|
+
].join("\n");
|
package/dist/voice.js
ADDED
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/** The round trip: speak → listen for one turn → transcribe. */
|
|
2
|
+
import { mkdtemp, rm } from "node:fs/promises";
|
|
3
|
+
import os from "node:os";
|
|
4
|
+
import path from "node:path";
|
|
5
|
+
import { chime, findRecorder, listenForTurn, MIC_PERMISSION_HINT, RECORDER_MISSING, speak } from "./audio.js";
|
|
6
|
+
import { CONFIG, debug, DEFAULT_LISTEN_SECONDS, MAX_LISTEN_SECONDS, MAX_SPEAK_CHARS } from "./config.js";
|
|
7
|
+
import { ensureModel } from "./model.js";
|
|
8
|
+
import { SetupError } from "./proc.js";
|
|
9
|
+
import { prepareSpeech } from "./speech-text.js";
|
|
10
|
+
import { findWhisperCli, findWhisperServer, prewarm, STT_MISSING, transcribe } from "./stt.js";
|
|
11
|
+
/** Everything speak_and_listen needs, checked before saying a word. Never installs anything. */
|
|
12
|
+
async function preflight() {
|
|
13
|
+
if (!findRecorder())
|
|
14
|
+
throw new SetupError(RECORDER_MISSING);
|
|
15
|
+
if (!findWhisperCli() && !findWhisperServer())
|
|
16
|
+
throw new SetupError(STT_MISSING);
|
|
17
|
+
return { model: await ensureModel({ download: false }) };
|
|
18
|
+
}
|
|
19
|
+
export async function speakAndListen(textToSpeak, listenSeconds, signal, onPhase) {
|
|
20
|
+
const seconds = Math.min(MAX_LISTEN_SECONDS, Math.max(1, Number.isFinite(listenSeconds) ? listenSeconds : DEFAULT_LISTEN_SECONDS));
|
|
21
|
+
const { model } = await preflight();
|
|
22
|
+
// Load the model into the warm server while we talk, so it's ready when the user finishes.
|
|
23
|
+
prewarm(model);
|
|
24
|
+
const speech = prepareSpeech(textToSpeak, { maxWords: CONFIG.maxSpeakWords, maxChars: MAX_SPEAK_CHARS });
|
|
25
|
+
const notes = [...speech.notes];
|
|
26
|
+
const tmpDir = await mkdtemp(path.join(os.tmpdir(), "voice-mcp-"));
|
|
27
|
+
const wav = path.join(tmpDir, "reply.wav");
|
|
28
|
+
try {
|
|
29
|
+
const t0 = Date.now();
|
|
30
|
+
onPhase?.("speaking");
|
|
31
|
+
await speak(speech.text, signal);
|
|
32
|
+
await chime("start");
|
|
33
|
+
onPhase?.("listening");
|
|
34
|
+
const heard = await listenForTurn(seconds, wav, signal);
|
|
35
|
+
void chime("stop");
|
|
36
|
+
const t1 = Date.now();
|
|
37
|
+
debug("listen:", heard);
|
|
38
|
+
if (heard.digitalSilence) {
|
|
39
|
+
return {
|
|
40
|
+
ok: false,
|
|
41
|
+
text: `The microphone returned pure digital silence, which usually means microphone access is blocked. ${MIC_PERMISSION_HINT}`,
|
|
42
|
+
notes,
|
|
43
|
+
};
|
|
44
|
+
}
|
|
45
|
+
const waited = Math.min(CONFIG.startTimeoutSeconds, seconds);
|
|
46
|
+
const noSpeech = `(No speech detected — the user did not reply within ${waited} seconds.)`;
|
|
47
|
+
if (heard.reason === "no-speech")
|
|
48
|
+
return { ok: true, text: noSpeech, notes };
|
|
49
|
+
onPhase?.("transcribing");
|
|
50
|
+
const transcript = await transcribe(wav, model, signal);
|
|
51
|
+
debug(`timing: speak+listen ${t1 - t0} ms, transcribe ${Date.now() - t1} ms`);
|
|
52
|
+
if (!transcript)
|
|
53
|
+
return { ok: true, text: noSpeech, notes };
|
|
54
|
+
if (heard.reason === "max-duration") {
|
|
55
|
+
const l = heard.levels;
|
|
56
|
+
const f = (n) => (Number.isFinite(n) ? n.toFixed(0) : "?");
|
|
57
|
+
notes.push(`voice-mcp note: listening stopped at the ${seconds}-second limit while the user was still talking, ` +
|
|
58
|
+
"so the reply may be cut off. Ask them to continue if it seems unfinished. " +
|
|
59
|
+
`(Audio levels, for tuning: background ${f(l.floorDb)} dB, voice ${f(l.speechDb)} dB, ` +
|
|
60
|
+
`last second ${f(l.recentDb)} ± ${l.recentSpreadDb.toFixed(1)} dB.)`);
|
|
61
|
+
}
|
|
62
|
+
return { ok: true, text: transcript, notes };
|
|
63
|
+
}
|
|
64
|
+
finally {
|
|
65
|
+
await rm(tmpDir, { recursive: true, force: true });
|
|
66
|
+
}
|
|
67
|
+
}
|
|
68
|
+
/** Serialize calls: there is one speaker and one microphone. */
|
|
69
|
+
let queue = Promise.resolve();
|
|
70
|
+
export function exclusive(fn) {
|
|
71
|
+
const next = queue.then(fn, fn);
|
|
72
|
+
queue = next.catch(() => { });
|
|
73
|
+
return next;
|
|
74
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
## Voice check-ins (voice-mcp)
|
|
2
|
+
|
|
3
|
+
I have a `speak_and_listen` tool that talks to me through my Mac's speakers and returns a
|
|
4
|
+
transcript of my spoken reply. Use it when I ask to be kept in the loop by voice, when I start
|
|
5
|
+
voice mode, or when you need a quick decision and I may be away from the keyboard. Don't use it
|
|
6
|
+
for routine output I can read on screen.
|
|
7
|
+
|
|
8
|
+
When you call it, write `text_to_speak` for the ear, not the screen:
|
|
9
|
+
- 1–3 short sentences, under ~40 words. Lead with the outcome, then ask one question.
|
|
10
|
+
- Plain words only: no markdown, bullets, emoji, code, file paths, URLs, stack traces or tables.
|
|
11
|
+
- Describe code instead of reading it ("I added a retry to the upload function"). Say file names,
|
|
12
|
+
not paths ("in index.ts"). Round numbers ("about two hundred milliseconds").
|
|
13
|
+
- Ask questions I can answer in a few words ("Should I deploy it — yes or no?").
|
|
14
|
+
- Keep the details (diffs, logs, links) in your normal on-screen reply, and tell me they're there.
|
|
15
|
+
|
|
16
|
+
Treating my replies:
|
|
17
|
+
- Once I'm talking with you by voice, stay in voice: answer every turn with `speak_and_listen`,
|
|
18
|
+
not a text reply, until I say stop or start typing.
|
|
19
|
+
- They are speech-to-text transcripts. Expect misheard words and interpret them generously.
|
|
20
|
+
- Confirm by voice before anything destructive, irreversible, or that costs money.
|
|
21
|
+
- "(No speech detected …)" means I didn't answer. Never treat silence as a yes.
|
|
22
|
+
- Listening stops on its own when I finish talking, so leave `listen_seconds` at its default.
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: How to use the voice-mcp speak_and_listen tool with speakable phrasing
|
|
3
|
+
alwaysApply: true
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
## Voice check-ins (voice-mcp)
|
|
7
|
+
|
|
8
|
+
I have a `speak_and_listen` tool that talks to me through my Mac's speakers and returns a
|
|
9
|
+
transcript of my spoken reply. Use it when I ask to be kept in the loop by voice, when I start
|
|
10
|
+
voice mode, or when you need a quick decision and I may be away from the keyboard. Don't use it
|
|
11
|
+
for routine output I can read on screen.
|
|
12
|
+
|
|
13
|
+
When you call it, write `text_to_speak` for the ear, not the screen:
|
|
14
|
+
- 1–3 short sentences, under ~40 words. Lead with the outcome, then ask one question.
|
|
15
|
+
- Plain words only: no markdown, bullets, emoji, code, file paths, URLs, stack traces or tables.
|
|
16
|
+
- Describe code instead of reading it ("I added a retry to the upload function"). Say file names,
|
|
17
|
+
not paths ("in index.ts"). Round numbers ("about two hundred milliseconds").
|
|
18
|
+
- Ask questions I can answer in a few words ("Should I deploy it — yes or no?").
|
|
19
|
+
- Keep the details (diffs, logs, links) in your normal on-screen reply, and tell me they're there.
|
|
20
|
+
|
|
21
|
+
Treating my replies:
|
|
22
|
+
- Once I'm talking with you by voice, stay in voice: answer every turn with `speak_and_listen`,
|
|
23
|
+
not a text reply, until I say stop or start typing.
|
|
24
|
+
- They are speech-to-text transcripts. Expect misheard words and interpret them generously.
|
|
25
|
+
- Confirm by voice before anything destructive, irreversible, or that costs money.
|
|
26
|
+
- "(No speech detected …)" means I didn't answer. Never treat silence as a yes.
|
|
27
|
+
- Listening stops on its own when I finish talking, so leave `listen_seconds` at its default.
|
package/package.json
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "mac-voice-mcp",
|
|
3
|
+
"mcpName": "io.github.jeet0007/mac-voice-mcp",
|
|
4
|
+
"version": "0.1.0",
|
|
5
|
+
"description": "Talk with Claude out loud on your Mac: speaks with macOS `say`, listens for one natural conversational turn, and transcribes on-device with whisper.cpp. An MCP server.",
|
|
6
|
+
"type": "module",
|
|
7
|
+
"bin": {
|
|
8
|
+
"mac-voice-mcp": "dist/index.js",
|
|
9
|
+
"voice-mcp": "dist/index.js"
|
|
10
|
+
},
|
|
11
|
+
"main": "dist/index.js",
|
|
12
|
+
"files": [
|
|
13
|
+
"dist",
|
|
14
|
+
"examples",
|
|
15
|
+
"README.md",
|
|
16
|
+
"LICENSE"
|
|
17
|
+
],
|
|
18
|
+
"scripts": {
|
|
19
|
+
"build": "tsc && node -e \"require('fs').chmodSync('dist/index.js', 0o755)\"",
|
|
20
|
+
"dev": "tsc --watch",
|
|
21
|
+
"start": "node dist/index.js",
|
|
22
|
+
"test": "npm run build && node --test test/*.test.mjs",
|
|
23
|
+
"setup": "node dist/index.js setup",
|
|
24
|
+
"test:voice": "node dist/index.js test",
|
|
25
|
+
"inspect": "npx -y @modelcontextprotocol/inspector node dist/index.js",
|
|
26
|
+
"prepare": "npm run build",
|
|
27
|
+
"prepublishOnly": "npm test",
|
|
28
|
+
"audit": "npm audit --omit=dev --audit-level=high && npm audit signatures"
|
|
29
|
+
},
|
|
30
|
+
"dependencies": {
|
|
31
|
+
"@modelcontextprotocol/sdk": "^1.30.0",
|
|
32
|
+
"zod": "^4.1.0"
|
|
33
|
+
},
|
|
34
|
+
"devDependencies": {
|
|
35
|
+
"@types/node": "^22.0.0",
|
|
36
|
+
"typescript": "^7.0.2"
|
|
37
|
+
},
|
|
38
|
+
"engines": {
|
|
39
|
+
"node": ">=18.17"
|
|
40
|
+
},
|
|
41
|
+
"keywords": [
|
|
42
|
+
"mcp",
|
|
43
|
+
"mcp-server",
|
|
44
|
+
"model-context-protocol",
|
|
45
|
+
"voice",
|
|
46
|
+
"speech",
|
|
47
|
+
"tts",
|
|
48
|
+
"stt",
|
|
49
|
+
"whisper",
|
|
50
|
+
"whisper.cpp",
|
|
51
|
+
"macos",
|
|
52
|
+
"apple-silicon",
|
|
53
|
+
"claude",
|
|
54
|
+
"claude-code",
|
|
55
|
+
"vibe-coded"
|
|
56
|
+
],
|
|
57
|
+
"author": "Jeet",
|
|
58
|
+
"license": "MIT",
|
|
59
|
+
"repository": {
|
|
60
|
+
"type": "git",
|
|
61
|
+
"url": "git+https://github.com/jeet0007/mac-voice-mcp.git"
|
|
62
|
+
},
|
|
63
|
+
"homepage": "https://github.com/jeet0007/mac-voice-mcp#readme",
|
|
64
|
+
"bugs": {
|
|
65
|
+
"url": "https://github.com/jeet0007/mac-voice-mcp/issues"
|
|
66
|
+
}
|
|
67
|
+
}
|