mac-voice-mcp 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +268 -0
- package/dist/audio.js +178 -0
- package/dist/config.js +103 -0
- package/dist/endpointer.js +171 -0
- package/dist/index.js +130 -0
- package/dist/model.js +189 -0
- package/dist/proc.js +112 -0
- package/dist/server.js +122 -0
- package/dist/setup.js +183 -0
- package/dist/speech-text.js +104 -0
- package/dist/stt.js +266 -0
- package/dist/texts.js +90 -0
- package/dist/voice.js +74 -0
- package/examples/CLAUDE.md +22 -0
- package/examples/voice-mcp.mdc +27 -0
- package/package.json +67 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Jeet
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
<h1 align="center">ποΈ mac-voice-mcp</h1>
|
|
2
|
+
|
|
3
|
+
<p align="center">
|
|
4
|
+
<b>Talk with Claude out loud on your Mac.</b><br>
|
|
5
|
+
Natural turn-taking Β· on-device speech recognition Β· no audio leaves your computer
|
|
6
|
+
</p>
|
|
7
|
+
|
|
8
|
+
<p align="center">
|
|
9
|
+
<a href="https://www.npmjs.com/package/mac-voice-mcp"><img alt="npm version" src="https://img.shields.io/npm/v/mac-voice-mcp?logo=npm&color=cb3837"></a>
|
|
10
|
+
<a href="https://www.npmjs.com/package/mac-voice-mcp"><img alt="npm downloads" src="https://img.shields.io/npm/dm/mac-voice-mcp?color=cb3837"></a>
|
|
11
|
+
<a href="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/ci.yml"><img alt="Tests" src="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/ci.yml/badge.svg?branch=main"></a>
|
|
12
|
+
<a href="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/security.yml"><img alt="Security" src="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/security.yml/badge.svg?branch=main"></a>
|
|
13
|
+
<a href="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/codeql.yml"><img alt="CodeQL" src="https://github.com/jeet0007/mac-voice-mcp/actions/workflows/codeql.yml/badge.svg?branch=main"></a>
|
|
14
|
+
</p>
|
|
15
|
+
|
|
16
|
+
<p align="center">
|
|
17
|
+
<img alt="Platform: macOS, Apple Silicon" src="https://img.shields.io/badge/platform-macOS%20%C2%B7%20Apple%20Silicon-000000?logo=apple">
|
|
18
|
+
<a href="https://modelcontextprotocol.io"><img alt="MCP server" src="https://img.shields.io/badge/MCP-server-6f42c1"></a>
|
|
19
|
+
<img alt="Node.js 18+" src="https://img.shields.io/node/v/mac-voice-mcp?logo=node.js&color=339933">
|
|
20
|
+
<a href="LICENSE"><img alt="License: MIT" src="https://img.shields.io/badge/license-MIT-blue"></a>
|
|
21
|
+
<a href="SECURITY.md"><img alt="Dependabot enabled" src="https://img.shields.io/badge/Dependabot-enabled-025e8c?logo=dependabot"></a>
|
|
22
|
+
<a href="#-vibe-coded"><img alt="Vibe-coded with Claude" src="https://img.shields.io/badge/vibe--coded-with%20Claude-d97757"></a>
|
|
23
|
+
</p>
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
Claude says something through your Mac's speakers, listens to your answer the way a person would, and gets back what you said as text. Speech recognition runs on your Mac, so no audio leaves your computer.
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
Claude ββspeak_and_listen("Tests pass. Open the PR?")βββΆ π "Tests pass. Open the PR?"
|
|
31
|
+
π you: "yes, and tag Priya"
|
|
32
|
+
Claude βββββββββββββββββ "Yes, and tag Priya." βββββββββ whisper.cpp on your Mac
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
It's built for Apple Silicon Macs (M1βM4). Linux works too, with SoX and espeak-ng installed.
|
|
36
|
+
|
|
37
|
+
### π€ Vibe-coded
|
|
38
|
+
|
|
39
|
+
> This project was designed and written with Claude, in conversation. A human (me) steered it, tried it on a real Mac and checked the test suite, but most of the code was written by AI. It's a **proof of concept**: it works and has tests, but expect rough edges, and read the code before relying on it for anything important. Issues and pull requests are welcome.
|
|
40
|
+
>
|
|
41
|
+
> It's an independent project, not made or endorsed by Anthropic. It works with any MCP client, including Claude Desktop, Claude Code and Cursor.
|
|
42
|
+
|
|
43
|
+
## How it works
|
|
44
|
+
|
|
45
|
+
The server has **two tools and two prompts**:
|
|
46
|
+
|
|
47
|
+
| | What it does |
|
|
48
|
+
|---|---|
|
|
49
|
+
| `speak_and_listen` | Speaks a short message, listens for one conversational turn and returns the transcript. |
|
|
50
|
+
| `voice_setup` | Checks what's installed. After you agree, it installs only what's missing. |
|
|
51
|
+
| `/mcp__voice-mcp__setup` | A guided setup: it checks, asks you, installs, then runs a spoken test. |
|
|
52
|
+
| `/mcp__voice-mcp__voice_mode` | A hands-free session where Claude checks in by voice at natural points. |
|
|
53
|
+
|
|
54
|
+
**Listening works like a conversation.** A soft chime plays when the mic opens. The server waits for you to start talking and hands back to Claude about a second after you stop. It adjusts to background noise, doesn't cut you off at pauses mid-sentence, and ignores coughs and clicks. If you say nothing for 8 seconds, Claude gets "no speech", which it is told never to treat as a yes.
|
|
55
|
+
|
|
56
|
+
**Replies come back fast.** whisper.cpp's server keeps the speech model loaded between turns, and the model starts loading while Claude is still talking. You don't wait for a model load on each reply. After 15 idle minutes the server shuts down to free memory. It is stopped automatically even if the MCP server crashes.
|
|
57
|
+
|
|
58
|
+
**Claude sends text meant to be heard.** The tool description gives Claude rules for writing short spoken sentences. If code, paths, links or markdown still get through, the server rewrites them before speaking and tells Claude, so the next message is cleaner. See [below](#getting-claude-to-sound-natural).
|
|
59
|
+
|
|
60
|
+
## Install
|
|
61
|
+
|
|
62
|
+
**1. Add the server to your client.** You need Node.js 18.17 or newer.
|
|
63
|
+
|
|
64
|
+
**Claude Code**
|
|
65
|
+
|
|
66
|
+
```bash
|
|
67
|
+
claude mcp add voice-mcp -s user -- npx -y mac-voice-mcp
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
Or install it as a Claude Code plugin:
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
/plugin marketplace add jeet0007/mac-voice-mcp
|
|
74
|
+
/plugin install mac-voice-mcp@mac-voice-mcp
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
**Claude Desktop.** Add this to `~/Library/Application Support/Claude/claude_desktop_config.json`, then quit (βQ) and reopen the app:
|
|
78
|
+
|
|
79
|
+
```json
|
|
80
|
+
{
|
|
81
|
+
"mcpServers": {
|
|
82
|
+
"voice-mcp": {
|
|
83
|
+
"command": "npx",
|
|
84
|
+
"args": ["-y", "mac-voice-mcp"]
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
}
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
If you get `spawn npx ENOENT`, use the full path from `which npx`, e.g. `"command": "/opt/homebrew/bin/npx"`.
|
|
91
|
+
|
|
92
|
+
**Cursor.** Add the same `mcpServers` block to `~/.cursor/mcp.json`, or use the one-click link:
|
|
93
|
+
|
|
94
|
+
```
|
|
95
|
+
cursor://anysphere.cursor-deeplink/mcp/install?name=voice-mcp&config=eyJjb21tYW5kIjoibnB4IiwiYXJncyI6WyIteSIsIm1hYy12b2ljZS1tY3AiXX0=
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
**2. Run setup once.** Ask Claude to *"set up voice"*. In Claude Code you can also run `/mcp__voice-mcp__setup`, or from a terminal run `npx -y mac-voice-mcp setup`.
|
|
99
|
+
|
|
100
|
+
Setup checks what's already there before it changes anything:
|
|
101
|
+
|
|
102
|
+
| Needed | Provided by | If it's missing |
|
|
103
|
+
|---|---|---|
|
|
104
|
+
| Voice | macOS `say` | Nothing to do. It's part of macOS. |
|
|
105
|
+
| Microphone capture | SoX (`rec`) | `brew install sox` |
|
|
106
|
+
| Speech-to-text | whisper.cpp (`whisper-cli` and `whisper-server`, Metal-accelerated) | `brew install whisper-cpp` |
|
|
107
|
+
| Speech model | `base.en`, ~140 MB | Downloaded once to `~/.cache/mac-voice-mcp/models/` |
|
|
108
|
+
|
|
109
|
+
- **Nothing is redone.** Tools already on your PATH are used as they are. If the model is already somewhere on disk (a whisper.cpp checkout, Homebrew's share folder, another tool's cache, or anything Spotlight can find), it's **symlinked**, not downloaded again. `brew install` runs only for the missing formulae.
|
|
110
|
+
- **Nothing happens without your OK.** Claude calls `voice_setup` to check first, shows you the checklist, and asks before calling it with `install=true`.
|
|
111
|
+
- **Slow installs don't time out.** If `brew install whisper-cpp` takes a while, setup reports INSTALLING. The install carries on in the background, and the next check picks up the result.
|
|
112
|
+
|
|
113
|
+
**3. Allow the microphone.** The first time Claude listens, macOS asks whether Claude (or Cursor, or your terminal) can use the microphone. Click Allow.
|
|
114
|
+
|
|
115
|
+
### Installing from a clone
|
|
116
|
+
|
|
117
|
+
One command does everything above: it builds the project, runs setup (asking before installing anything), adds voice-mcp to Claude Desktop (backing up your config first) and to Claude Code, and offers a spoken test. It's safe to re-run, because each step checks first and skips anything already done.
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
git clone https://github.com/jeet0007/mac-voice-mcp && bash mac-voice-mcp/install.sh
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
## Using it
|
|
124
|
+
|
|
125
|
+
- **"Work on X and check in with me by voice when you need a decision."** Claude works quietly and only speaks at decision points.
|
|
126
|
+
- **`/mcp__voice-mcp__voice_mode fix the flaky login test`.** Claude reads its plan back to you, then checks in at each checkpoint. Say "stop voice mode" or "I'm back" to end it.
|
|
127
|
+
- **"Read me a 20-second summary of this PR and ask if I should approve it."** Use this for one-off briefings.
|
|
128
|
+
- **Just talk after the chime.** You don't need to hurry or fill silence. If you're still talking at the 30-second safety cap (`listen_seconds`), Claude is told your reply may be cut off and asks you to continue.
|
|
129
|
+
|
|
130
|
+
## Getting Claude to sound natural
|
|
131
|
+
|
|
132
|
+
Guidance reaches Claude through several channels, because each client shows different ones:
|
|
133
|
+
|
|
134
|
+
| Channel | Who sees it |
|
|
135
|
+
|---|---|
|
|
136
|
+
| Tool description (rules plus a good and a bad example) | Every client |
|
|
137
|
+
| Server instructions (when to use voice, how to handle replies, setup) | Claude Code (it reads up to 2 KB) |
|
|
138
|
+
| The `voice_mode` and `setup` prompts | Claude Code (as slash commands), Claude Desktop, Cursor |
|
|
139
|
+
| Server-side rewrite plus a `voice-mcp note` back to Claude | Always on |
|
|
140
|
+
|
|
141
|
+
The rules Claude is given:
|
|
142
|
+
|
|
143
|
+
```
|
|
144
|
+
- 1β3 short sentences, under ~40 words; lead with the outcome, then one question.
|
|
145
|
+
- Plain words only: no markdown, bullets, emoji, code, file paths, URLs, stack traces or tables.
|
|
146
|
+
- Describe code instead of reading it, say file names not paths, round numbers, spell out symbols.
|
|
147
|
+
- Ask one question at a time, answerable in a few words.
|
|
148
|
+
- Put the details (diffs, logs, links) in the on-screen reply, and say so out loud.
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
The safety net turns `## Results\n- \`npm test\` β
42/42\n- see /Users/x/repo/src/index.ts:120` into *"Results. npm test 42 of 42. see index.ts."*
|
|
152
|
+
|
|
153
|
+
Claude Desktop ignores server instructions. To make the rules stick there, paste [`examples/CLAUDE.md`](examples/CLAUDE.md) into *Settings β Profile β personal preferences* or into a Project's instructions. For Claude Code, add it to `CLAUDE.md`. For Cursor, copy [`examples/voice-mcp.mdc`](examples/voice-mcp.mdc) to `.cursor/rules/`.
|
|
154
|
+
|
|
155
|
+
## Configuration
|
|
156
|
+
|
|
157
|
+
Everything is optional. Set these in your client config's `"env": { β¦ }` block, or with `-e NAME=value` in `claude mcp add`.
|
|
158
|
+
|
|
159
|
+
**Speaking**
|
|
160
|
+
|
|
161
|
+
| Variable | Default | |
|
|
162
|
+
|---|---|---|
|
|
163
|
+
| `VOICE_MCP_VOICE` | system voice | macOS voice, e.g. `Samantha`, `Daniel`, `Kanya`. List them with `say -v '?'`. |
|
|
164
|
+
| `VOICE_MCP_RATE` | system rate | Words per minute, e.g. `200`. |
|
|
165
|
+
| `VOICE_MCP_MAX_SPEAK_WORDS` | `120` | Longer text is cut at a sentence boundary ("the rest is on screen"). |
|
|
166
|
+
| `VOICE_MCP_CHIME` | `1` | Set to `0` to turn off the mic open/close sounds. |
|
|
167
|
+
|
|
168
|
+
**Listening**
|
|
169
|
+
|
|
170
|
+
| Variable | Default | |
|
|
171
|
+
|---|---|---|
|
|
172
|
+
| `VOICE_MCP_END_SILENCE_MS` | `1200` | How long a pause ends your turn. Use `1800` if it cuts you off while you think, `800` for snappier replies. |
|
|
173
|
+
| `VOICE_MCP_START_TIMEOUT_SECONDS` | `8` | How long to wait for you to start talking. |
|
|
174
|
+
| `VOICE_MCP_SPEECH_MARGIN_DB` | `12` | How much louder than room noise counts as speech. Raise it in noisy rooms. |
|
|
175
|
+
| `VOICE_MCP_MIN_SPEECH_DB` | `-48` | The quietest level that ever counts as speech (dBFS). |
|
|
176
|
+
| `VOICE_MCP_RECORDER` | `auto` | `sox` or `ffmpeg` (`ffmpeg` is macOS only). |
|
|
177
|
+
|
|
178
|
+
**Speech-to-text**
|
|
179
|
+
|
|
180
|
+
| Variable | Default | |
|
|
181
|
+
|---|---|---|
|
|
182
|
+
| `VOICE_MCP_WHISPER_MODEL` | `base.en` | Which model to use (see the table below). |
|
|
183
|
+
| `VOICE_MCP_LANGUAGE` | `en` for `*.en` models, otherwise `auto` | `en`, `th`, `ja`, `de`, β¦ |
|
|
184
|
+
| `VOICE_MCP_WHISPER_PROMPT` | β | Words to bias toward: names, product terms, jargon. |
|
|
185
|
+
| `VOICE_MCP_WHISPER_MODEL_PATH` | β | Use this exact `ggml-*.bin` file. |
|
|
186
|
+
| `VOICE_MCP_MODEL_SEARCH_PATHS` | β | Extra folders to check for an existing model (`:`-separated). |
|
|
187
|
+
| `VOICE_MCP_WHISPER_SERVER` | `1` | Set to `0` to always use `whisper-cli`, with no warm server. |
|
|
188
|
+
| `VOICE_MCP_SERVER_IDLE_MINUTES` | `15` | How long the warm server stays up without use. |
|
|
189
|
+
| `VOICE_MCP_THREADS` | min(8, cores) | Number of whisper.cpp threads. |
|
|
190
|
+
| `VOICE_MCP_CACHE_DIR` | `~/.cache/mac-voice-mcp` | Where models are downloaded or symlinked. |
|
|
191
|
+
| `VOICE_MCP_DEBUG` | `0` | Verbose logs with per-turn timings, written to stderr. |
|
|
192
|
+
|
|
193
|
+
**Models** (whisper.cpp names):
|
|
194
|
+
|
|
195
|
+
| Model | Size | Good for |
|
|
196
|
+
|---|---|---|
|
|
197
|
+
| `tiny.en` | 75 MB | Yes/no answers, the lowest latency |
|
|
198
|
+
| `base.en` | 142 MB | **Default.** English conversation. |
|
|
199
|
+
| `small.en` | 466 MB | Noticeably more accurate English |
|
|
200
|
+
| `large-v3-turbo-q5_0` | 547 MB | Other languages, e.g. Thai with `VOICE_MCP_LANGUAGE=th` |
|
|
201
|
+
|
|
202
|
+
## Troubleshooting
|
|
203
|
+
|
|
204
|
+
| Symptom | Fix |
|
|
205
|
+
|---|---|
|
|
206
|
+
| "voice-mcp is not set up yet" | Ask Claude to *set up voice*, or run `npx -y mac-voice-mcp setup`. |
|
|
207
|
+
| "microphone returned pure digital silence" | macOS is blocking the mic for the host app. Go to **System Settings β Privacy & Security β Microphone**, enable Claude / Cursor / your terminal, then restart that app. |
|
|
208
|
+
| No permission prompt ever appears | Run `tccutil reset Microphone <bundle id>` and restart the app. Running `test` in Terminal only gives permission to Terminal, not to Claude Desktop. |
|
|
209
|
+
| It cuts me off while I'm thinking | Set `VOICE_MCP_END_SILENCE_MS=1800` (or up to `2500`). |
|
|
210
|
+
| It never stops listening | The room is too noisy for the defaults. Set `VOICE_MCP_SPEECH_MARGIN_DB=18`, or use a headset. |
|
|
211
|
+
| It hears its own voice | Use headphones, or turn the speaker volume down. It only listens after it finishes speaking, but echo can linger. |
|
|
212
|
+
| "Homebrew: not installed" | Install it from [brew.sh](https://brew.sh). It needs your password, so it can't run from Claude. Then run setup again. |
|
|
213
|
+
| It garbles names or jargon | Set `VOICE_MCP_WHISPER_PROMPT="Priya, Postgres, Kubernetes"`, or switch to `small.en`. |
|
|
214
|
+
|
|
215
|
+
## Known limitations
|
|
216
|
+
|
|
217
|
+
- **You can't interrupt it.** It finishes speaking, then listens. Barge-in would mean listening while the speakers play, which needs headphones or echo cancellation.
|
|
218
|
+
- **It's macOS-first.** Linux works with SoX and espeak-ng. Windows is untested.
|
|
219
|
+
- **Turn-taking is based on loudness, not a speech model.** It adapts to background noise, but very noisy rooms, music or TV can confuse it. A headset helps, and so do the listening settings above.
|
|
220
|
+
- **One conversation at a time.** There's one speaker and one microphone, so calls are queued.
|
|
221
|
+
|
|
222
|
+
## Privacy and safety
|
|
223
|
+
|
|
224
|
+
- **Audio stays on your machine.** Recordings go to a temporary file that's deleted after each turn. The only network use is the one-time model download from Hugging Face.
|
|
225
|
+
- **The warm whisper server is local only.** It listens on `127.0.0.1` on a random port, and stops when idle or when this server exits.
|
|
226
|
+
- **Setup can only install known packages.** Its install list is fixed in the code (`sox`, `whisper-cpp`), so nothing Claude says can make it install anything else. It never uninstalls or modifies other software.
|
|
227
|
+
|
|
228
|
+
## Security
|
|
229
|
+
|
|
230
|
+
- **Secrets:** every push and pull request is scanned for leaked secrets with [TruffleHog](https://github.com/trufflesecurity/trufflehog), and the whole history is scanned before the first push. GitHub secret scanning with push protection is also on.
|
|
231
|
+
- **Dependencies:** [Dependabot](https://docs.github.com/code-security/dependabot) opens weekly update pull requests. CI fails on high-severity advisories (`npm audit`), and dependency review blocks pull requests that add vulnerable packages.
|
|
232
|
+
- **Code:** [CodeQL](https://codeql.github.com) runs with the `security-extended` queries.
|
|
233
|
+
- **Releases:** releases publish through [npm trusted publishing](https://docs.npmjs.com/trusted-publishers/), with no long-lived npm token and a signed provenance attestation for every version.
|
|
234
|
+
|
|
235
|
+
To report a vulnerability, see [SECURITY.md](SECURITY.md).
|
|
236
|
+
|
|
237
|
+
## Development
|
|
238
|
+
|
|
239
|
+
```bash
|
|
240
|
+
npm install
|
|
241
|
+
npm test # build + 27 tests: unit tests and end-to-end tests over MCP with stub binaries
|
|
242
|
+
npm run audit # known-vulnerability and signature checks on dependencies
|
|
243
|
+
npm run setup # check what's installed; offers to install what's missing
|
|
244
|
+
npm run test:voice # one real speak β listen β transcribe turn
|
|
245
|
+
npm run inspect # MCP Inspector
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
| Module | Responsibility |
|
|
249
|
+
|---|---|
|
|
250
|
+
| `config.ts` | Environment settings, logging, the PATH fix-up for GUI apps |
|
|
251
|
+
| `speech-text.ts` | Rewriting screen text for speech, cleaning up transcripts (pure, unit-tested) |
|
|
252
|
+
| `endpointer.ts` | Turn-taking voice-activity detection (pure, unit-tested) |
|
|
253
|
+
| `audio.ts` | Text-to-speech, chimes, streaming mic capture |
|
|
254
|
+
| `model.ts` | Finding, symlinking or downloading the model |
|
|
255
|
+
| `stt.ts` | The warm `whisper-server` with orphan guard, and the `whisper-cli` fallback |
|
|
256
|
+
| `setup.ts` | Requirement checks and consent-based background installs |
|
|
257
|
+
| `voice.ts`, `server.ts`, `index.ts` | The round trip, the MCP tools and prompts, and the CLI |
|
|
258
|
+
|
|
259
|
+
The package installs two commands: `mac-voice-mcp` (the one `npx -y mac-voice-mcp` runs), and `voice-mcp`.
|
|
260
|
+
|
|
261
|
+
### Releasing
|
|
262
|
+
|
|
263
|
+
- **First release:** `bash publish.sh`. It asks before each public step and uses your own GitHub and npm logins. It creates the GitHub repo, publishes to npm, and lists the server in the [official MCP Registry](https://registry.modelcontextprotocol.io), which Smithery, Glama, PulseMCP and mcp.so pick up from.
|
|
264
|
+
- **Later releases:** bump `version` in `package.json` and `server.json`, then push a `v<version>` tag. The Publish workflow tests the build, checks that the tag matches both versions, and publishes to npm and the MCP Registry. It needs no secrets: both logins use GitHub's OIDC identity, once you've set the package's *Trusted Publisher* on npmjs.com (`publish.sh` prints the steps).
|
|
265
|
+
|
|
266
|
+
## License
|
|
267
|
+
|
|
268
|
+
MIT
|
package/dist/audio.js
ADDED
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/** Speaking (native TTS), the mic chimes, and listening for one conversational turn. */
|
|
2
|
+
import { spawn } from "node:child_process";
|
|
3
|
+
import { existsSync } from "node:fs";
|
|
4
|
+
import { writeFile } from "node:fs/promises";
|
|
5
|
+
import { CONFIG, debug, IS_MAC, IS_WIN } from "./config.js";
|
|
6
|
+
import { Endpointer, FRAME_BYTES, wavHeader } from "./endpointer.js";
|
|
7
|
+
import { activeChildren, CancelledError, run, SetupError, tail, which } from "./proc.js";
|
|
8
|
+
// ---------------------------------------------------------------------------
|
|
9
|
+
// Text-to-speech
|
|
10
|
+
// ---------------------------------------------------------------------------
|
|
11
|
+
export function findTts() {
|
|
12
|
+
if (IS_MAC)
|
|
13
|
+
return which("say") ?? (existsSync("/usr/bin/say") ? "/usr/bin/say" : null);
|
|
14
|
+
if (IS_WIN)
|
|
15
|
+
return "powershell.exe";
|
|
16
|
+
return which("espeak-ng") ?? which("espeak") ?? which("spd-say");
|
|
17
|
+
}
|
|
18
|
+
export async function speak(text, signal) {
|
|
19
|
+
if (!text)
|
|
20
|
+
return;
|
|
21
|
+
const bin = findTts();
|
|
22
|
+
if (!bin)
|
|
23
|
+
throw new SetupError("No text-to-speech engine found. Install espeak-ng (e.g. `sudo apt install espeak-ng`).");
|
|
24
|
+
let result;
|
|
25
|
+
if (IS_MAC) {
|
|
26
|
+
const args = [];
|
|
27
|
+
if (CONFIG.voice)
|
|
28
|
+
args.push("-v", CONFIG.voice);
|
|
29
|
+
if (CONFIG.rate)
|
|
30
|
+
args.push("-r", CONFIG.rate);
|
|
31
|
+
args.push("-f", "-"); // read from stdin: no argv length limits, no flag injection
|
|
32
|
+
result = await run(bin, args, { input: text, signal, timeoutMs: 180_000 });
|
|
33
|
+
}
|
|
34
|
+
else if (IS_WIN) {
|
|
35
|
+
const ps = "Add-Type -AssemblyName System.Speech; $s = New-Object System.Speech.Synthesis.SpeechSynthesizer; " +
|
|
36
|
+
"$s.Speak([Console]::In.ReadToEnd())";
|
|
37
|
+
result = await run(bin, ["-NoProfile", "-NonInteractive", "-Command", ps], { input: text, signal, timeoutMs: 180_000 });
|
|
38
|
+
}
|
|
39
|
+
else if (bin.endsWith("spd-say")) {
|
|
40
|
+
result = await run(bin, ["-w", ` ${text}`], { signal, timeoutMs: 180_000 });
|
|
41
|
+
}
|
|
42
|
+
else {
|
|
43
|
+
result = await run(bin, ["--stdin"], { input: text, signal, timeoutMs: 180_000 });
|
|
44
|
+
}
|
|
45
|
+
if (result.code !== 0) {
|
|
46
|
+
throw new Error(`Text-to-speech failed (exit ${result.code}): ${tail(result.stderr) || "no output"}`);
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
/** A short, quiet cue that the mic just opened ("start") or closed ("stop"). macOS only. */
|
|
50
|
+
export async function chime(kind) {
|
|
51
|
+
if (!CONFIG.chime || !IS_MAC)
|
|
52
|
+
return;
|
|
53
|
+
const sound = kind === "start" ? "/System/Library/Sounds/Tink.aiff" : "/System/Library/Sounds/Pop.aiff";
|
|
54
|
+
if (!existsSync(sound))
|
|
55
|
+
return;
|
|
56
|
+
try {
|
|
57
|
+
await run("/usr/bin/afplay", ["-v", "0.6", sound], { timeoutMs: 3000 });
|
|
58
|
+
}
|
|
59
|
+
catch {
|
|
60
|
+
/* a missing chime is never fatal */
|
|
61
|
+
}
|
|
62
|
+
}
|
|
63
|
+
export function findRecorder() {
|
|
64
|
+
const pref = CONFIG.recorder;
|
|
65
|
+
if (pref !== "ffmpeg") {
|
|
66
|
+
const rec = which("rec");
|
|
67
|
+
if (rec)
|
|
68
|
+
return { kind: "sox", bin: rec, viaSox: false };
|
|
69
|
+
const sox = which("sox");
|
|
70
|
+
if (sox)
|
|
71
|
+
return { kind: "sox", bin: sox, viaSox: true };
|
|
72
|
+
}
|
|
73
|
+
if (pref !== "sox" && IS_MAC) {
|
|
74
|
+
const ffmpeg = which("ffmpeg");
|
|
75
|
+
if (ffmpeg)
|
|
76
|
+
return { kind: "ffmpeg", bin: ffmpeg };
|
|
77
|
+
}
|
|
78
|
+
return null;
|
|
79
|
+
}
|
|
80
|
+
export function describeRecorder(r) {
|
|
81
|
+
return `${r.kind === "sox" ? "SoX" : "ffmpeg"} (${r.bin})`;
|
|
82
|
+
}
|
|
83
|
+
export const RECORDER_MISSING = "No audio recorder found. Call the voice_setup tool to check and install what's missing " +
|
|
84
|
+
(IS_WIN ? "(or install SoX from https://sourceforge.net/projects/sox/)." : "(or run `brew install sox`).");
|
|
85
|
+
export const MIC_PERMISSION_HINT = "Allow microphone access for the app running this server (Claude, Cursor, Terminal, iTerm, VS Code β¦) " +
|
|
86
|
+
"in System Settings β Privacy & Security β Microphone, then restart that app.";
|
|
87
|
+
function recorderArgs(r) {
|
|
88
|
+
if (r.kind === "sox") {
|
|
89
|
+
// Raw 16 kHz mono s16le to stdout; SoX resamples from the device rate.
|
|
90
|
+
return ["-q", "-V1", ...(r.viaSox ? ["-d"] : []), "-t", "raw", "-r", "16000", "-c", "1", "-b", "16", "-e", "signed-integer", "-"];
|
|
91
|
+
}
|
|
92
|
+
return [
|
|
93
|
+
"-hide_banner", "-loglevel", "error", "-nostdin",
|
|
94
|
+
"-f", "avfoundation", "-i", CONFIG.ffmpegDevice,
|
|
95
|
+
"-ac", "1", "-ar", "16000", "-f", "s16le", "-",
|
|
96
|
+
];
|
|
97
|
+
}
|
|
98
|
+
/**
|
|
99
|
+
* Listen for one conversational turn and write it to `outFile` (16 kHz mono WAV).
|
|
100
|
+
* Streams raw PCM from the recorder, runs the endpointer on it live, and stops the
|
|
101
|
+
* recorder the moment the user finishes. Leading and trailing silence are trimmed
|
|
102
|
+
* (keeping a little padding) so whisper gets just the utterance.
|
|
103
|
+
*/
|
|
104
|
+
export async function listenForTurn(maxSeconds, outFile, signal) {
|
|
105
|
+
const recorder = findRecorder();
|
|
106
|
+
if (!recorder)
|
|
107
|
+
throw new SetupError(RECORDER_MISSING);
|
|
108
|
+
const endpointer = new Endpointer({
|
|
109
|
+
maxMs: maxSeconds * 1000,
|
|
110
|
+
startTimeoutMs: CONFIG.startTimeoutSeconds * 1000,
|
|
111
|
+
endSilenceMs: CONFIG.endSilenceMs,
|
|
112
|
+
marginDb: CONFIG.speechMarginDb,
|
|
113
|
+
minSpeechDb: CONFIG.minSpeechDb,
|
|
114
|
+
});
|
|
115
|
+
const frames = [];
|
|
116
|
+
let pending = Buffer.alloc(0);
|
|
117
|
+
let reason = null;
|
|
118
|
+
let stderr = "";
|
|
119
|
+
const args = recorderArgs(recorder);
|
|
120
|
+
debug("exec:", recorder.bin, args.join(" "));
|
|
121
|
+
const child = spawn(recorder.bin, args, { stdio: ["ignore", "pipe", "pipe"], windowsHide: true });
|
|
122
|
+
activeChildren.add(child);
|
|
123
|
+
const exitCode = await new Promise((resolve, reject) => {
|
|
124
|
+
const stop = () => {
|
|
125
|
+
if (child.exitCode === null)
|
|
126
|
+
child.kill("SIGTERM");
|
|
127
|
+
setTimeout(() => child.exitCode === null && child.kill("SIGKILL"), 1500).unref();
|
|
128
|
+
};
|
|
129
|
+
const safety = setTimeout(() => {
|
|
130
|
+
reason ??= "max-duration";
|
|
131
|
+
stop();
|
|
132
|
+
}, (maxSeconds + 5) * 1000);
|
|
133
|
+
const onAbort = () => stop();
|
|
134
|
+
signal?.addEventListener("abort", onAbort, { once: true });
|
|
135
|
+
child.stderr.setEncoding("utf8").on("data", (d) => (stderr = (stderr + d).slice(-4000)));
|
|
136
|
+
child.stdout.on("data", (chunk) => {
|
|
137
|
+
if (reason)
|
|
138
|
+
return;
|
|
139
|
+
pending = pending.length ? Buffer.concat([pending, chunk]) : chunk;
|
|
140
|
+
let off = 0;
|
|
141
|
+
while (!reason && pending.length - off >= FRAME_BYTES) {
|
|
142
|
+
const frame = Buffer.from(pending.subarray(off, off + FRAME_BYTES));
|
|
143
|
+
off += FRAME_BYTES;
|
|
144
|
+
frames.push(frame);
|
|
145
|
+
reason = endpointer.push(frame);
|
|
146
|
+
}
|
|
147
|
+
pending = pending.subarray(off);
|
|
148
|
+
if (reason)
|
|
149
|
+
stop();
|
|
150
|
+
});
|
|
151
|
+
child.on("error", (e) => {
|
|
152
|
+
clearTimeout(safety);
|
|
153
|
+
activeChildren.delete(child);
|
|
154
|
+
reject(e);
|
|
155
|
+
});
|
|
156
|
+
child.on("close", (code) => {
|
|
157
|
+
clearTimeout(safety);
|
|
158
|
+
signal?.removeEventListener("abort", onAbort);
|
|
159
|
+
activeChildren.delete(child);
|
|
160
|
+
resolve(code);
|
|
161
|
+
});
|
|
162
|
+
});
|
|
163
|
+
if (signal?.aborted)
|
|
164
|
+
throw new CancelledError();
|
|
165
|
+
if (!frames.length) {
|
|
166
|
+
throw new Error(`Recording failed (${describeRecorder(recorder)}, exit ${exitCode}). ${tail(stderr)}`.trim() +
|
|
167
|
+
(IS_MAC ? `\n${MIC_PERMISSION_HINT}` : ""));
|
|
168
|
+
}
|
|
169
|
+
const [first, last] = endpointer.trimRange(frames.length);
|
|
170
|
+
const pcm = Buffer.concat(frames.slice(first, last));
|
|
171
|
+
await writeFile(outFile, Buffer.concat([wavHeader(pcm.length), pcm]));
|
|
172
|
+
return {
|
|
173
|
+
reason: reason ?? "max-duration",
|
|
174
|
+
digitalSilence: !endpointer.anyNonZero,
|
|
175
|
+
speechSeconds: endpointer.speechSeconds,
|
|
176
|
+
levels: endpointer.diagnostics(),
|
|
177
|
+
};
|
|
178
|
+
}
|
package/dist/config.js
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Configuration (all optional environment variables) and logging.
|
|
3
|
+
* stdout belongs to the MCP protocol, so every log line goes to stderr.
|
|
4
|
+
*/
|
|
5
|
+
import { createRequire } from "node:module";
|
|
6
|
+
import os from "node:os";
|
|
7
|
+
import path from "node:path";
|
|
8
|
+
const require = createRequire(import.meta.url);
|
|
9
|
+
export const PKG = require("../package.json");
|
|
10
|
+
export const IS_MAC = process.platform === "darwin";
|
|
11
|
+
export const IS_WIN = process.platform === "win32";
|
|
12
|
+
/** listen_seconds is an upper bound β listening normally ends when the user stops talking. */
|
|
13
|
+
export const DEFAULT_LISTEN_SECONDS = 30;
|
|
14
|
+
export const MAX_LISTEN_SECONDS = 120;
|
|
15
|
+
export const MAX_SPEAK_CHARS = 4000;
|
|
16
|
+
export function envNum(name, fallback) {
|
|
17
|
+
const raw = process.env[name];
|
|
18
|
+
if (raw === undefined || raw.trim() === "")
|
|
19
|
+
return fallback;
|
|
20
|
+
const n = Number(raw);
|
|
21
|
+
return Number.isFinite(n) ? n : fallback;
|
|
22
|
+
}
|
|
23
|
+
export function envBool(name, fallback) {
|
|
24
|
+
const raw = process.env[name]?.trim().toLowerCase();
|
|
25
|
+
if (!raw)
|
|
26
|
+
return fallback;
|
|
27
|
+
return !["0", "false", "no", "off"].includes(raw);
|
|
28
|
+
}
|
|
29
|
+
const modelName = (process.env.VOICE_MCP_WHISPER_MODEL ?? "base.en").trim();
|
|
30
|
+
/** Everything this server ever downloads or links lives here, shared by every version and every client. */
|
|
31
|
+
const CACHE_ROOT = process.env.VOICE_MCP_CACHE_DIR?.trim() ||
|
|
32
|
+
path.join(process.env.XDG_CACHE_HOME || path.join(os.homedir(), ".cache"), "mac-voice-mcp");
|
|
33
|
+
export const CONFIG = {
|
|
34
|
+
// --- Speaking
|
|
35
|
+
/** macOS voice name, e.g. "Samantha" (`say -v '?'` lists them). */
|
|
36
|
+
voice: process.env.VOICE_MCP_VOICE?.trim() || undefined,
|
|
37
|
+
/** Speech rate in words per minute (macOS `say -r`). */
|
|
38
|
+
rate: process.env.VOICE_MCP_RATE?.trim() || undefined,
|
|
39
|
+
/** Spoken text longer than this is cut at a sentence boundary (0 = no limit). */
|
|
40
|
+
maxSpeakWords: Math.max(0, Math.floor(envNum("VOICE_MCP_MAX_SPEAK_WORDS", 120))),
|
|
41
|
+
/** Play a short sound when the mic opens/closes (macOS). */
|
|
42
|
+
chime: envBool("VOICE_MCP_CHIME", true),
|
|
43
|
+
// --- Listening (turn-taking)
|
|
44
|
+
/** End of turn: stop listening after this much silence once the user has spoken. */
|
|
45
|
+
endSilenceMs: Math.max(300, envNum("VOICE_MCP_END_SILENCE_MS", 1200)),
|
|
46
|
+
/** Give up if the user hasn't started talking within this many seconds. */
|
|
47
|
+
startTimeoutSeconds: Math.max(1, envNum("VOICE_MCP_START_TIMEOUT_SECONDS", 8)),
|
|
48
|
+
/** Speech must be this many dB above the room's background noise. Lower = more sensitive. */
|
|
49
|
+
speechMarginDb: envNum("VOICE_MCP_SPEECH_MARGIN_DB", 12),
|
|
50
|
+
/** β¦and never quieter than this absolute level (dBFS). */
|
|
51
|
+
minSpeechDb: envNum("VOICE_MCP_MIN_SPEECH_DB", -48),
|
|
52
|
+
/** Recorder: auto (SoX, else ffmpeg) | sox | ffmpeg. */
|
|
53
|
+
recorder: (process.env.VOICE_MCP_RECORDER?.trim().toLowerCase() || "auto"),
|
|
54
|
+
/** ffmpeg avfoundation audio input (macOS fallback recorder). */
|
|
55
|
+
ffmpegDevice: process.env.VOICE_MCP_FFMPEG_DEVICE?.trim() || ":0",
|
|
56
|
+
// --- Speech-to-text
|
|
57
|
+
/** whisper.cpp model name (tiny.en, base.en, small.en, large-v3-turbo-q5_0, ...). */
|
|
58
|
+
modelName,
|
|
59
|
+
/** Absolute path to an existing ggml model β skips lookup and download entirely. */
|
|
60
|
+
modelPath: process.env.VOICE_MCP_WHISPER_MODEL_PATH?.trim() || undefined,
|
|
61
|
+
modelsDir: process.env.VOICE_MCP_MODELS_DIR?.trim() || path.join(CACHE_ROOT, "models"),
|
|
62
|
+
/** Extra folders to look in for an existing ggml model before downloading (":"-separated). */
|
|
63
|
+
modelSearchPaths: (process.env.VOICE_MCP_MODEL_SEARCH_PATHS ?? "")
|
|
64
|
+
.split(path.delimiter)
|
|
65
|
+
.map((p) => p.trim())
|
|
66
|
+
.filter(Boolean),
|
|
67
|
+
modelBaseUrl: (process.env.VOICE_MCP_MODEL_BASE_URL?.trim() || "https://huggingface.co/ggerganov/whisper.cpp/resolve/main").replace(/\/+$/, ""),
|
|
68
|
+
/** Explicit path to whisper.cpp's `whisper-cli`. */
|
|
69
|
+
whisperBin: process.env.VOICE_MCP_WHISPER_BIN?.trim() || undefined,
|
|
70
|
+
/** Explicit path to whisper.cpp's `whisper-server`. */
|
|
71
|
+
whisperServerBin: process.env.VOICE_MCP_WHISPER_SERVER_BIN?.trim() || undefined,
|
|
72
|
+
/** Keep the model loaded in a local whisper-server between turns (much lower latency). */
|
|
73
|
+
useWhisperServer: envBool("VOICE_MCP_WHISPER_SERVER", true),
|
|
74
|
+
/** Stop the warm whisper-server after this many idle minutes to free memory. */
|
|
75
|
+
serverIdleMinutes: Math.max(1, envNum("VOICE_MCP_SERVER_IDLE_MINUTES", 15)),
|
|
76
|
+
/** Spoken language code ("en", "th", "de", ...) or "auto". Defaults to "en" for *.en models, else "auto". */
|
|
77
|
+
language: process.env.VOICE_MCP_LANGUAGE?.trim() || (modelName.endsWith(".en") ? "en" : "auto"),
|
|
78
|
+
/** Optional initial prompt to bias vocabulary (names, jargon). */
|
|
79
|
+
prompt: process.env.VOICE_MCP_WHISPER_PROMPT?.trim() || undefined,
|
|
80
|
+
threads: Math.max(1, Math.floor(envNum("VOICE_MCP_THREADS", Math.min(8, os.cpus().length || 4)))),
|
|
81
|
+
debug: envBool("VOICE_MCP_DEBUG", false),
|
|
82
|
+
};
|
|
83
|
+
export function log(...args) {
|
|
84
|
+
console.error("[voice-mcp]", ...args);
|
|
85
|
+
}
|
|
86
|
+
export function debug(...args) {
|
|
87
|
+
if (CONFIG.debug)
|
|
88
|
+
log(...args);
|
|
89
|
+
}
|
|
90
|
+
// GUI apps on macOS (Claude Desktop, Cursor) launch MCP servers with a minimal
|
|
91
|
+
// PATH that misses Homebrew. Append the usual locations so `rec`, `ffmpeg`,
|
|
92
|
+
// `brew` and the whisper.cpp tools are found without extra config.
|
|
93
|
+
// VOICE_MCP_EXTRA_PATH overrides the list (":"-separated; empty = add nothing β the tests use this).
|
|
94
|
+
if (!IS_WIN) {
|
|
95
|
+
const extra = process.env.VOICE_MCP_EXTRA_PATH !== undefined
|
|
96
|
+
? process.env.VOICE_MCP_EXTRA_PATH.split(path.delimiter).filter(Boolean)
|
|
97
|
+
: ["/opt/homebrew/bin", "/usr/local/bin", "/opt/local/bin", path.join(os.homedir(), ".local", "bin")];
|
|
98
|
+
const parts = (process.env.PATH ?? "").split(path.delimiter).filter(Boolean);
|
|
99
|
+
for (const p of extra)
|
|
100
|
+
if (!parts.includes(p))
|
|
101
|
+
parts.push(p);
|
|
102
|
+
process.env.PATH = parts.join(path.delimiter);
|
|
103
|
+
}
|