voice2text 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: voice2text
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Local voice-to-text with Whisper + LLM cleanup
|
|
5
|
+
Requires-Python: >=3.11
|
|
6
|
+
Requires-Dist: loguru
|
|
7
|
+
Requires-Dist: mlx-whisper
|
|
8
|
+
Requires-Dist: numpy
|
|
9
|
+
Requires-Dist: pynput
|
|
10
|
+
Requires-Dist: scipy
|
|
11
|
+
Requires-Dist: sounddevice
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
|
|
14
|
+
# voice2text
|
|
15
|
+
|
|
16
|
+
Local voice-to-text with Whisper + LLM cleanup. Push-to-talk, pastes at cursor.
|
|
17
|
+
|
|
18
|
+
> **Note:** Before anyone suggests splitting this into modules and submodules — this is an intentional design choice. I want to demonstrate that in December 2025, you can have a fully local voice-to-text system with automatic cleanup and correction, running almost instantly on consumer hardware, all in ~270 lines of Python.
|
|
19
|
+
|
|
20
|
+
## Install
|
|
21
|
+
|
|
22
|
+
### Option 1: Pixi (recommended)
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pixi run ollama pull qwen2.5:3b
|
|
26
|
+
pixi run v2t
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
### Option 2: UV
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
brew install ollama
|
|
33
|
+
ollama pull qwen2.5:3b
|
|
34
|
+
uv sync
|
|
35
|
+
uv run v2t
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
## Usage
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
v2t # strict mode (restructures sentences)
|
|
42
|
+
v2t --casual # light cleanup (punctuation only)
|
|
43
|
+
v2t --pause-music # pause media while recording (macOS only, requires nowplaying-cli)
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Hold **Right Command** to record, release to transcribe and paste.
|
|
47
|
+
|
|
48
|
+
### `--pause-music` (macOS only)
|
|
49
|
+
|
|
50
|
+
Pauses any playing media while recording and resumes after. Requires:
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
brew install nowplaying-cli
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
Not available via pixi/conda-forge.
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
voice2text.py,sha256=Q7mEMbjYKQorC15mdxjh5C26m1Gr9kxJshUKuyrLd3Y,9264
|
|
2
|
+
voice2text-0.1.0.dist-info/METADATA,sha256=j0Q0rFTOfn4EC56SgekLj2pm8ULUZrBVzRz2tWwwC3s,1426
|
|
3
|
+
voice2text-0.1.0.dist-info/WHEEL,sha256=WLgqFyCfm_KASv4WHyYy0P3pM_m7J5L9k2skdKLirC8,87
|
|
4
|
+
voice2text-0.1.0.dist-info/entry_points.txt,sha256=iaFj-z8oAtCyZo2Murt_z0PjynofUXSmUE9QpXbRNzo,40
|
|
5
|
+
voice2text-0.1.0.dist-info/RECORD,,
|
voice2text.py
ADDED
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Voice-to-text with push-to-talk hotkey.
|
|
3
|
+
|
|
4
|
+
Hold Right Command to record, release to transcribe and paste.
|
|
5
|
+
|
|
6
|
+
Prerequisites:
|
|
7
|
+
brew install ollama
|
|
8
|
+
ollama pull qwen2.5:3b
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
import subprocess
|
|
12
|
+
import tempfile
|
|
13
|
+
import threading
|
|
14
|
+
import sys
|
|
15
|
+
import time
|
|
16
|
+
import argparse
|
|
17
|
+
import numpy as np
|
|
18
|
+
import sounddevice as sd
|
|
19
|
+
import mlx_whisper
|
|
20
|
+
from scipy.io import wavfile
|
|
21
|
+
from pynput import keyboard
|
|
22
|
+
from loguru import logger
|
|
23
|
+
|
|
24
|
+
# Config
|
|
25
|
+
SAMPLE_RATE = 16000
|
|
26
|
+
WHISPER_MODEL = "mlx-community/whisper-large-v3-turbo"
|
|
27
|
+
OLLAMA_MODEL = "qwen2.5:3b"
|
|
28
|
+
PUSH_TO_TALK_KEY = keyboard.Key.cmd_r
|
|
29
|
+
|
|
30
|
+
CLEANUP_PROMPT_STRICT = """Clean up this transcription. Fix punctuation, remove filler words (um, uh, like, you know), fix obvious mishearings, keep the meaning intact. Output ONLY the cleaned text, nothing else:
|
|
31
|
+
|
|
32
|
+
{text}"""
|
|
33
|
+
|
|
34
|
+
CLEANUP_PROMPT_CASUAL = """Lightly clean up this transcription. Only fix punctuation and remove filler words (um, uh, like, you know). Do NOT restructure sentences or change word order. Keep the original phrasing. Output ONLY the cleaned text, nothing else:
|
|
35
|
+
|
|
36
|
+
{text}"""
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def check_and_request_permissions():
|
|
40
|
+
"""Check for required permissions and open System Settings if needed."""
|
|
41
|
+
logger.info("Checking permissions...")
|
|
42
|
+
|
|
43
|
+
test_result = subprocess.run(
|
|
44
|
+
["osascript", "-e", 'tell application "System Events" to return "ok"'],
|
|
45
|
+
capture_output=True,
|
|
46
|
+
text=True
|
|
47
|
+
)
|
|
48
|
+
|
|
49
|
+
if "not allowed" in test_result.stderr.lower() or test_result.returncode != 0:
|
|
50
|
+
logger.warning("Permissions needed!")
|
|
51
|
+
logger.info("Grant permissions to your TERMINAL APP (Terminal, iTerm, Ghostty, VS Code, etc.)")
|
|
52
|
+
|
|
53
|
+
logger.info("Opening Accessibility settings...")
|
|
54
|
+
subprocess.run([
|
|
55
|
+
"open",
|
|
56
|
+
"x-apple.systempreferences:com.apple.preference.security?Privacy_Accessibility"
|
|
57
|
+
])
|
|
58
|
+
|
|
59
|
+
input("Press Enter after granting Accessibility permission...")
|
|
60
|
+
|
|
61
|
+
logger.info("Opening Input Monitoring settings...")
|
|
62
|
+
subprocess.run([
|
|
63
|
+
"open",
|
|
64
|
+
"x-apple.systempreferences:com.apple.preference.security?Privacy_ListenEvent"
|
|
65
|
+
])
|
|
66
|
+
|
|
67
|
+
input("Press Enter after granting Input Monitoring permission...")
|
|
68
|
+
|
|
69
|
+
logger.success("Permissions granted. You may need to restart your terminal, then run this script again.")
|
|
70
|
+
sys.exit(0)
|
|
71
|
+
|
|
72
|
+
logger.success("Permissions OK")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class VoiceToText:
|
|
76
|
+
def __init__(self, pause_music: bool = False, casual: bool = False):
|
|
77
|
+
self.recording = False
|
|
78
|
+
self.frames: list[np.ndarray] = []
|
|
79
|
+
self.stream = None
|
|
80
|
+
self.processing = False
|
|
81
|
+
self.record_start = 0.0
|
|
82
|
+
self.pause_music = pause_music
|
|
83
|
+
self.casual = casual
|
|
84
|
+
|
|
85
|
+
def audio_callback(self, indata, frame_count, time_info, status):
|
|
86
|
+
if self.recording:
|
|
87
|
+
self.frames.append(indata.copy())
|
|
88
|
+
|
|
89
|
+
def start_recording(self):
|
|
90
|
+
if self.recording or self.processing:
|
|
91
|
+
return
|
|
92
|
+
|
|
93
|
+
self.recording = True
|
|
94
|
+
self.frames = []
|
|
95
|
+
self.record_start = time.perf_counter()
|
|
96
|
+
|
|
97
|
+
if self.pause_music:
|
|
98
|
+
subprocess.run(["nowplaying-cli", "pause"])
|
|
99
|
+
|
|
100
|
+
logger.info("Recording...")
|
|
101
|
+
|
|
102
|
+
self.stream = sd.InputStream(
|
|
103
|
+
samplerate=SAMPLE_RATE,
|
|
104
|
+
channels=1,
|
|
105
|
+
dtype="float32",
|
|
106
|
+
callback=self.audio_callback
|
|
107
|
+
)
|
|
108
|
+
self.stream.start()
|
|
109
|
+
|
|
110
|
+
def stop_recording(self):
|
|
111
|
+
if not self.recording:
|
|
112
|
+
return
|
|
113
|
+
|
|
114
|
+
self.recording = False
|
|
115
|
+
duration = time.perf_counter() - self.record_start
|
|
116
|
+
if self.stream:
|
|
117
|
+
self.stream.stop()
|
|
118
|
+
self.stream.close()
|
|
119
|
+
self.stream = None
|
|
120
|
+
|
|
121
|
+
logger.info(f"Stopped ({duration:.1f}s)")
|
|
122
|
+
|
|
123
|
+
if self.frames:
|
|
124
|
+
threading.Thread(target=self.process_audio, daemon=True).start()
|
|
125
|
+
|
|
126
|
+
def process_audio(self):
|
|
127
|
+
self.processing = True
|
|
128
|
+
|
|
129
|
+
try:
|
|
130
|
+
audio = np.concatenate(self.frames, axis=0)
|
|
131
|
+
|
|
132
|
+
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
|
|
133
|
+
wavfile.write(f.name, SAMPLE_RATE, (audio * 32767).astype(np.int16))
|
|
134
|
+
temp_path = f.name
|
|
135
|
+
|
|
136
|
+
logger.info("Transcribing...")
|
|
137
|
+
t0 = time.perf_counter()
|
|
138
|
+
result = mlx_whisper.transcribe(temp_path, path_or_hf_repo=WHISPER_MODEL)
|
|
139
|
+
raw_text = result["text"].strip()
|
|
140
|
+
t1 = time.perf_counter()
|
|
141
|
+
logger.info(f"Raw: {raw_text} ({t1-t0:.2f}s)")
|
|
142
|
+
|
|
143
|
+
if not raw_text:
|
|
144
|
+
logger.warning("No speech detected")
|
|
145
|
+
return
|
|
146
|
+
|
|
147
|
+
logger.info("Cleaning up...")
|
|
148
|
+
t0 = time.perf_counter()
|
|
149
|
+
prompt_template = CLEANUP_PROMPT_CASUAL if self.casual else CLEANUP_PROMPT_STRICT
|
|
150
|
+
prompt = prompt_template.format(text=raw_text)
|
|
151
|
+
|
|
152
|
+
try:
|
|
153
|
+
result = subprocess.run(
|
|
154
|
+
["ollama", "run", OLLAMA_MODEL, prompt],
|
|
155
|
+
capture_output=True,
|
|
156
|
+
text=True,
|
|
157
|
+
timeout=30,
|
|
158
|
+
)
|
|
159
|
+
if result.returncode != 0:
|
|
160
|
+
raise Exception(f"Ollama exited with code {result.returncode}: {result.stderr}")
|
|
161
|
+
cleaned_text = result.stdout.strip()
|
|
162
|
+
if not cleaned_text:
|
|
163
|
+
raise Exception("Ollama returned empty response")
|
|
164
|
+
t1 = time.perf_counter()
|
|
165
|
+
logger.info(f"Clean: {cleaned_text} ({t1-t0:.2f}s)")
|
|
166
|
+
except Exception as e:
|
|
167
|
+
logger.error(f"LLM cleanup failed: {e}")
|
|
168
|
+
logger.warning("Falling back to raw transcription")
|
|
169
|
+
cleaned_text = raw_text
|
|
170
|
+
|
|
171
|
+
self.paste_to_cursor(cleaned_text)
|
|
172
|
+
logger.success("Pasted!")
|
|
173
|
+
|
|
174
|
+
finally:
|
|
175
|
+
if self.pause_music:
|
|
176
|
+
subprocess.run(["nowplaying-cli", "play"])
|
|
177
|
+
|
|
178
|
+
self.processing = False
|
|
179
|
+
|
|
180
|
+
def paste_to_cursor(self, text: str) -> None:
|
|
181
|
+
"""Copy to clipboard, paste at cursor, then restore original clipboard."""
|
|
182
|
+
original = subprocess.run(["pbpaste"], capture_output=True, text=True).stdout
|
|
183
|
+
|
|
184
|
+
subprocess.run(["pbcopy"], input=text, text=True)
|
|
185
|
+
subprocess.run([
|
|
186
|
+
"osascript", "-e",
|
|
187
|
+
'tell application "System Events" to keystroke "v" using command down'
|
|
188
|
+
])
|
|
189
|
+
|
|
190
|
+
time.sleep(0.15)
|
|
191
|
+
subprocess.run(["pbcopy"], input=original, text=True)
|
|
192
|
+
|
|
193
|
+
def on_press(self, key):
|
|
194
|
+
if key == PUSH_TO_TALK_KEY:
|
|
195
|
+
self.start_recording()
|
|
196
|
+
|
|
197
|
+
def on_release(self, key):
|
|
198
|
+
if key == PUSH_TO_TALK_KEY:
|
|
199
|
+
self.stop_recording()
|
|
200
|
+
|
|
201
|
+
def run(self):
|
|
202
|
+
mode = "Casual" if self.casual else "Strict"
|
|
203
|
+
logger.info(f"Voice-to-Text Started - {mode} Mode")
|
|
204
|
+
if self.pause_music:
|
|
205
|
+
logger.info("Pause Music - Turned On")
|
|
206
|
+
logger.info("Hold Right Command to record, release to transcribe and paste. Ctrl+C to quit.")
|
|
207
|
+
logger.warning("Note: Right Command is a modifier key - you cannot type while holding it")
|
|
208
|
+
|
|
209
|
+
try:
|
|
210
|
+
with keyboard.Listener(
|
|
211
|
+
on_press=self.on_press,
|
|
212
|
+
on_release=self.on_release
|
|
213
|
+
) as listener:
|
|
214
|
+
listener.join()
|
|
215
|
+
except KeyboardInterrupt:
|
|
216
|
+
logger.info("Shutting down...")
|
|
217
|
+
sys.exit(0)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def main():
|
|
221
|
+
parser = argparse.ArgumentParser(description="voice2text: push-to-talk transcription")
|
|
222
|
+
parser.add_argument(
|
|
223
|
+
"--pause-music",
|
|
224
|
+
action="store_true",
|
|
225
|
+
help="Pause Music/Spotify while recording, resume after paste"
|
|
226
|
+
)
|
|
227
|
+
parser.add_argument(
|
|
228
|
+
"--casual",
|
|
229
|
+
action="store_true",
|
|
230
|
+
help="Light cleanup only (punctuation + filler words). Won't restructure sentences."
|
|
231
|
+
)
|
|
232
|
+
args = parser.parse_args()
|
|
233
|
+
|
|
234
|
+
if args.pause_music:
|
|
235
|
+
nowplaying_check = subprocess.run(["which", "nowplaying-cli"], capture_output=True)
|
|
236
|
+
if nowplaying_check.returncode != 0:
|
|
237
|
+
logger.warning("nowplaying-cli not found. Install with: brew install nowplaying-cli")
|
|
238
|
+
logger.warning("Music pause feature disabled.")
|
|
239
|
+
args.pause_music = False
|
|
240
|
+
|
|
241
|
+
ollama_check = subprocess.run(["which", "ollama"], capture_output=True)
|
|
242
|
+
if ollama_check.returncode != 0:
|
|
243
|
+
logger.error("Ollama not found. Install with: brew install ollama")
|
|
244
|
+
sys.exit(1)
|
|
245
|
+
|
|
246
|
+
model_check = subprocess.run(
|
|
247
|
+
["ollama", "list"],
|
|
248
|
+
capture_output=True,
|
|
249
|
+
text=True
|
|
250
|
+
)
|
|
251
|
+
if OLLAMA_MODEL.split(":")[0] not in model_check.stdout:
|
|
252
|
+
logger.error(f"Model not found. Pull with: ollama pull {OLLAMA_MODEL}")
|
|
253
|
+
sys.exit(1)
|
|
254
|
+
|
|
255
|
+
check_and_request_permissions()
|
|
256
|
+
|
|
257
|
+
try:
|
|
258
|
+
logger.info("Loading Whisper model (first run downloads ~1.6GB)...")
|
|
259
|
+
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
|
|
260
|
+
silence = np.zeros(SAMPLE_RATE, dtype=np.int16)
|
|
261
|
+
wavfile.write(f.name, SAMPLE_RATE, silence)
|
|
262
|
+
mlx_whisper.transcribe(f.name, path_or_hf_repo=WHISPER_MODEL)
|
|
263
|
+
logger.success("Model loaded")
|
|
264
|
+
|
|
265
|
+
app = VoiceToText(pause_music=args.pause_music, casual=args.casual)
|
|
266
|
+
app.run()
|
|
267
|
+
except KeyboardInterrupt:
|
|
268
|
+
logger.info("Shutting down...")
|
|
269
|
+
sys.exit(0)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
if __name__ == "__main__":
|
|
273
|
+
main()
|