voice2text 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,56 @@
1
+ Metadata-Version: 2.4
2
+ Name: voice2text
3
+ Version: 0.1.0
4
+ Summary: Local voice-to-text with Whisper + LLM cleanup
5
+ Requires-Python: >=3.11
6
+ Requires-Dist: loguru
7
+ Requires-Dist: mlx-whisper
8
+ Requires-Dist: numpy
9
+ Requires-Dist: pynput
10
+ Requires-Dist: scipy
11
+ Requires-Dist: sounddevice
12
+ Description-Content-Type: text/markdown
13
+
14
+ # voice2text
15
+
16
+ Local voice-to-text with Whisper + LLM cleanup. Push-to-talk, pastes at cursor.
17
+
18
+ > **Note:** Before anyone suggests splitting this into modules and submodules — this is an intentional design choice. I want to demonstrate that in December 2025, you can have a fully local voice-to-text system with automatic cleanup and correction, running almost instantly on consumer hardware, all in ~270 lines of Python.
19
+
20
+ ## Install
21
+
22
+ ### Option 1: Pixi (recommended)
23
+
24
+ ```bash
25
+ pixi run ollama pull qwen2.5:3b
26
+ pixi run v2t
27
+ ```
28
+
29
+ ### Option 2: UV
30
+
31
+ ```bash
32
+ brew install ollama
33
+ ollama pull qwen2.5:3b
34
+ uv sync
35
+ uv run v2t
36
+ ```
37
+
38
+ ## Usage
39
+
40
+ ```bash
41
+ v2t # strict mode (restructures sentences)
42
+ v2t --casual # light cleanup (punctuation only)
43
+ v2t --pause-music # pause media while recording (macOS only, requires nowplaying-cli)
44
+ ```
45
+
46
+ Hold **Right Command** to record, release to transcribe and paste.
47
+
48
+ ### `--pause-music` (macOS only)
49
+
50
+ Pauses any playing media while recording and resumes after. Requires:
51
+
52
+ ```bash
53
+ brew install nowplaying-cli
54
+ ```
55
+
56
+ Not available via pixi/conda-forge.
@@ -0,0 +1,5 @@
1
+ voice2text.py,sha256=Q7mEMbjYKQorC15mdxjh5C26m1Gr9kxJshUKuyrLd3Y,9264
2
+ voice2text-0.1.0.dist-info/METADATA,sha256=j0Q0rFTOfn4EC56SgekLj2pm8ULUZrBVzRz2tWwwC3s,1426
3
+ voice2text-0.1.0.dist-info/WHEEL,sha256=WLgqFyCfm_KASv4WHyYy0P3pM_m7J5L9k2skdKLirC8,87
4
+ voice2text-0.1.0.dist-info/entry_points.txt,sha256=iaFj-z8oAtCyZo2Murt_z0PjynofUXSmUE9QpXbRNzo,40
5
+ voice2text-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.28.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ v2t = voice2text:main
voice2text.py ADDED
@@ -0,0 +1,273 @@
1
+ """
2
+ Voice-to-text with push-to-talk hotkey.
3
+
4
+ Hold Right Command to record, release to transcribe and paste.
5
+
6
+ Prerequisites:
7
+ brew install ollama
8
+ ollama pull qwen2.5:3b
9
+ """
10
+
11
+ import subprocess
12
+ import tempfile
13
+ import threading
14
+ import sys
15
+ import time
16
+ import argparse
17
+ import numpy as np
18
+ import sounddevice as sd
19
+ import mlx_whisper
20
+ from scipy.io import wavfile
21
+ from pynput import keyboard
22
+ from loguru import logger
23
+
24
+ # Config
25
+ SAMPLE_RATE = 16000
26
+ WHISPER_MODEL = "mlx-community/whisper-large-v3-turbo"
27
+ OLLAMA_MODEL = "qwen2.5:3b"
28
+ PUSH_TO_TALK_KEY = keyboard.Key.cmd_r
29
+
30
+ CLEANUP_PROMPT_STRICT = """Clean up this transcription. Fix punctuation, remove filler words (um, uh, like, you know), fix obvious mishearings, keep the meaning intact. Output ONLY the cleaned text, nothing else:
31
+
32
+ {text}"""
33
+
34
+ CLEANUP_PROMPT_CASUAL = """Lightly clean up this transcription. Only fix punctuation and remove filler words (um, uh, like, you know). Do NOT restructure sentences or change word order. Keep the original phrasing. Output ONLY the cleaned text, nothing else:
35
+
36
+ {text}"""
37
+
38
+
39
+ def check_and_request_permissions():
40
+ """Check for required permissions and open System Settings if needed."""
41
+ logger.info("Checking permissions...")
42
+
43
+ test_result = subprocess.run(
44
+ ["osascript", "-e", 'tell application "System Events" to return "ok"'],
45
+ capture_output=True,
46
+ text=True
47
+ )
48
+
49
+ if "not allowed" in test_result.stderr.lower() or test_result.returncode != 0:
50
+ logger.warning("Permissions needed!")
51
+ logger.info("Grant permissions to your TERMINAL APP (Terminal, iTerm, Ghostty, VS Code, etc.)")
52
+
53
+ logger.info("Opening Accessibility settings...")
54
+ subprocess.run([
55
+ "open",
56
+ "x-apple.systempreferences:com.apple.preference.security?Privacy_Accessibility"
57
+ ])
58
+
59
+ input("Press Enter after granting Accessibility permission...")
60
+
61
+ logger.info("Opening Input Monitoring settings...")
62
+ subprocess.run([
63
+ "open",
64
+ "x-apple.systempreferences:com.apple.preference.security?Privacy_ListenEvent"
65
+ ])
66
+
67
+ input("Press Enter after granting Input Monitoring permission...")
68
+
69
+ logger.success("Permissions granted. You may need to restart your terminal, then run this script again.")
70
+ sys.exit(0)
71
+
72
+ logger.success("Permissions OK")
73
+
74
+
75
+ class VoiceToText:
76
+ def __init__(self, pause_music: bool = False, casual: bool = False):
77
+ self.recording = False
78
+ self.frames: list[np.ndarray] = []
79
+ self.stream = None
80
+ self.processing = False
81
+ self.record_start = 0.0
82
+ self.pause_music = pause_music
83
+ self.casual = casual
84
+
85
+ def audio_callback(self, indata, frame_count, time_info, status):
86
+ if self.recording:
87
+ self.frames.append(indata.copy())
88
+
89
+ def start_recording(self):
90
+ if self.recording or self.processing:
91
+ return
92
+
93
+ self.recording = True
94
+ self.frames = []
95
+ self.record_start = time.perf_counter()
96
+
97
+ if self.pause_music:
98
+ subprocess.run(["nowplaying-cli", "pause"])
99
+
100
+ logger.info("Recording...")
101
+
102
+ self.stream = sd.InputStream(
103
+ samplerate=SAMPLE_RATE,
104
+ channels=1,
105
+ dtype="float32",
106
+ callback=self.audio_callback
107
+ )
108
+ self.stream.start()
109
+
110
+ def stop_recording(self):
111
+ if not self.recording:
112
+ return
113
+
114
+ self.recording = False
115
+ duration = time.perf_counter() - self.record_start
116
+ if self.stream:
117
+ self.stream.stop()
118
+ self.stream.close()
119
+ self.stream = None
120
+
121
+ logger.info(f"Stopped ({duration:.1f}s)")
122
+
123
+ if self.frames:
124
+ threading.Thread(target=self.process_audio, daemon=True).start()
125
+
126
+ def process_audio(self):
127
+ self.processing = True
128
+
129
+ try:
130
+ audio = np.concatenate(self.frames, axis=0)
131
+
132
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
133
+ wavfile.write(f.name, SAMPLE_RATE, (audio * 32767).astype(np.int16))
134
+ temp_path = f.name
135
+
136
+ logger.info("Transcribing...")
137
+ t0 = time.perf_counter()
138
+ result = mlx_whisper.transcribe(temp_path, path_or_hf_repo=WHISPER_MODEL)
139
+ raw_text = result["text"].strip()
140
+ t1 = time.perf_counter()
141
+ logger.info(f"Raw: {raw_text} ({t1-t0:.2f}s)")
142
+
143
+ if not raw_text:
144
+ logger.warning("No speech detected")
145
+ return
146
+
147
+ logger.info("Cleaning up...")
148
+ t0 = time.perf_counter()
149
+ prompt_template = CLEANUP_PROMPT_CASUAL if self.casual else CLEANUP_PROMPT_STRICT
150
+ prompt = prompt_template.format(text=raw_text)
151
+
152
+ try:
153
+ result = subprocess.run(
154
+ ["ollama", "run", OLLAMA_MODEL, prompt],
155
+ capture_output=True,
156
+ text=True,
157
+ timeout=30,
158
+ )
159
+ if result.returncode != 0:
160
+ raise Exception(f"Ollama exited with code {result.returncode}: {result.stderr}")
161
+ cleaned_text = result.stdout.strip()
162
+ if not cleaned_text:
163
+ raise Exception("Ollama returned empty response")
164
+ t1 = time.perf_counter()
165
+ logger.info(f"Clean: {cleaned_text} ({t1-t0:.2f}s)")
166
+ except Exception as e:
167
+ logger.error(f"LLM cleanup failed: {e}")
168
+ logger.warning("Falling back to raw transcription")
169
+ cleaned_text = raw_text
170
+
171
+ self.paste_to_cursor(cleaned_text)
172
+ logger.success("Pasted!")
173
+
174
+ finally:
175
+ if self.pause_music:
176
+ subprocess.run(["nowplaying-cli", "play"])
177
+
178
+ self.processing = False
179
+
180
+ def paste_to_cursor(self, text: str) -> None:
181
+ """Copy to clipboard, paste at cursor, then restore original clipboard."""
182
+ original = subprocess.run(["pbpaste"], capture_output=True, text=True).stdout
183
+
184
+ subprocess.run(["pbcopy"], input=text, text=True)
185
+ subprocess.run([
186
+ "osascript", "-e",
187
+ 'tell application "System Events" to keystroke "v" using command down'
188
+ ])
189
+
190
+ time.sleep(0.15)
191
+ subprocess.run(["pbcopy"], input=original, text=True)
192
+
193
+ def on_press(self, key):
194
+ if key == PUSH_TO_TALK_KEY:
195
+ self.start_recording()
196
+
197
+ def on_release(self, key):
198
+ if key == PUSH_TO_TALK_KEY:
199
+ self.stop_recording()
200
+
201
+ def run(self):
202
+ mode = "Casual" if self.casual else "Strict"
203
+ logger.info(f"Voice-to-Text Started - {mode} Mode")
204
+ if self.pause_music:
205
+ logger.info("Pause Music - Turned On")
206
+ logger.info("Hold Right Command to record, release to transcribe and paste. Ctrl+C to quit.")
207
+ logger.warning("Note: Right Command is a modifier key - you cannot type while holding it")
208
+
209
+ try:
210
+ with keyboard.Listener(
211
+ on_press=self.on_press,
212
+ on_release=self.on_release
213
+ ) as listener:
214
+ listener.join()
215
+ except KeyboardInterrupt:
216
+ logger.info("Shutting down...")
217
+ sys.exit(0)
218
+
219
+
220
+ def main():
221
+ parser = argparse.ArgumentParser(description="voice2text: push-to-talk transcription")
222
+ parser.add_argument(
223
+ "--pause-music",
224
+ action="store_true",
225
+ help="Pause Music/Spotify while recording, resume after paste"
226
+ )
227
+ parser.add_argument(
228
+ "--casual",
229
+ action="store_true",
230
+ help="Light cleanup only (punctuation + filler words). Won't restructure sentences."
231
+ )
232
+ args = parser.parse_args()
233
+
234
+ if args.pause_music:
235
+ nowplaying_check = subprocess.run(["which", "nowplaying-cli"], capture_output=True)
236
+ if nowplaying_check.returncode != 0:
237
+ logger.warning("nowplaying-cli not found. Install with: brew install nowplaying-cli")
238
+ logger.warning("Music pause feature disabled.")
239
+ args.pause_music = False
240
+
241
+ ollama_check = subprocess.run(["which", "ollama"], capture_output=True)
242
+ if ollama_check.returncode != 0:
243
+ logger.error("Ollama not found. Install with: brew install ollama")
244
+ sys.exit(1)
245
+
246
+ model_check = subprocess.run(
247
+ ["ollama", "list"],
248
+ capture_output=True,
249
+ text=True
250
+ )
251
+ if OLLAMA_MODEL.split(":")[0] not in model_check.stdout:
252
+ logger.error(f"Model not found. Pull with: ollama pull {OLLAMA_MODEL}")
253
+ sys.exit(1)
254
+
255
+ check_and_request_permissions()
256
+
257
+ try:
258
+ logger.info("Loading Whisper model (first run downloads ~1.6GB)...")
259
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
260
+ silence = np.zeros(SAMPLE_RATE, dtype=np.int16)
261
+ wavfile.write(f.name, SAMPLE_RATE, silence)
262
+ mlx_whisper.transcribe(f.name, path_or_hf_repo=WHISPER_MODEL)
263
+ logger.success("Model loaded")
264
+
265
+ app = VoiceToText(pause_music=args.pause_music, casual=args.casual)
266
+ app.run()
267
+ except KeyboardInterrupt:
268
+ logger.info("Shutting down...")
269
+ sys.exit(0)
270
+
271
+
272
+ if __name__ == "__main__":
273
+ main()