scribe-cli 0.18.0__tar.gz → 0.18.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.gitignore +1 -0
  2. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/PKG-INFO +102 -34
  3. scribe_cli-0.18.1/README.md +172 -0
  4. scribe_cli-0.18.1/docs/app-tray-menu.png +0 -0
  5. scribe_cli-0.18.1/docs/backends.md +400 -0
  6. scribe_cli-0.18.1/docs/cli.md +208 -0
  7. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/desktop-install.md +1 -1
  8. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/installation.md +46 -5
  9. scribe_cli-0.18.0/docs/keyboard.md → scribe_cli-0.18.1/docs/output.md +98 -36
  10. scribe_cli-0.18.1/docs/tray.md +173 -0
  11. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/pyproject.toml +43 -11
  12. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/_version.py +3 -3
  13. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/app.py +493 -148
  14. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/groq.py +4 -3
  15. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/openai_api.py +12 -3
  16. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/openai_realtime.py +59 -5
  17. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/vosk.py +20 -4
  18. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/whisper.py +17 -4
  19. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/whisper_futo.py +15 -3
  20. scribe_cli-0.18.1/scribe/dialog.py +82 -0
  21. scribe_cli-0.18.1/scribe/menu.py +1678 -0
  22. scribe_cli-0.18.1/scribe/models.py +466 -0
  23. scribe_cli-0.18.1/scribe/output.py +237 -0
  24. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/session.py +29 -4
  25. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/PKG-INFO +102 -34
  26. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/SOURCES.txt +9 -1
  27. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/requires.txt +8 -5
  28. scribe_cli-0.18.1/tests/test_backend_matrix.py +295 -0
  29. scribe_cli-0.18.1/tests/test_compose_prompt.py +153 -0
  30. scribe_cli-0.18.1/tests/test_debug_logging.py +191 -0
  31. scribe_cli-0.18.1/tests/test_output.py +165 -0
  32. scribe_cli-0.18.1/tests/test_output_file_picker.py +57 -0
  33. scribe_cli-0.18.1/tests/test_prompt_file_picker.py +165 -0
  34. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_pseudo_streaming.py +218 -43
  35. scribe_cli-0.18.0/README.md +0 -124
  36. scribe_cli-0.18.0/docs/app-tray-menu.png +0 -0
  37. scribe_cli-0.18.0/docs/backends.md +0 -238
  38. scribe_cli-0.18.0/docs/cli.md +0 -156
  39. scribe_cli-0.18.0/docs/tray.md +0 -97
  40. scribe_cli-0.18.0/scribe/menu.py +0 -1019
  41. scribe_cli-0.18.0/scribe/models.py +0 -333
  42. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.github/FUNDING.yml +0 -0
  43. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.github/workflows/pypi.yml +0 -0
  44. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/LICENSE +0 -0
  45. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/roadmap-libei.md +0 -0
  46. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/icon.xcf +0 -0
  47. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/__init__.py +0 -0
  48. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/audio.py +0 -0
  49. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/__init__.py +0 -0
  50. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/install_desktop.py +0 -0
  51. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/keyboard.py +0 -0
  52. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/models.toml +0 -0
  53. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/saverecording.py +0 -0
  54. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/testpynput.py +0 -0
  55. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/__init__.py +0 -0
  56. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/base.py +0 -0
  57. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/eitype.py +0 -0
  58. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/pynput.py +0 -0
  59. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/wtype.py +0 -0
  60. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/ydotool.py +0 -0
  61. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/util.py +0 -0
  62. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/dependency_links.txt +0 -0
  63. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/entry_points.txt +0 -0
  64. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/top_level.txt +0 -0
  65. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/__init__.py +0 -0
  66. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon.png +0 -0
  67. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon_recording.png +0 -0
  68. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon_writing.png +0 -0
  69. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/silero_vad.LICENSE +0 -0
  70. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/silero_vad.onnx +0 -0
  71. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/templates/scribe.desktop +0 -0
  72. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scripts/bench_whisper_local.py +0 -0
  73. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scripts/test_python_versions_install.sh +0 -0
  74. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/setup.cfg +0 -0
  75. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_openai_realtime_coalesce.py +0 -0
  76. {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_whisper_futo.py +0 -0
@@ -7,3 +7,4 @@ scribe/_version.py
7
7
 
8
8
  # Autonomous roadmap workflows (local coordination artifacts; never committed)
9
9
  workflows/
10
+ .worktrees/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scribe-cli
3
- Version: 0.18.0
3
+ Version: 0.18.1
4
4
  Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
5
5
  Author-email: Mahé Perrette <mahe.perrette@gmail.com>
6
6
  License: MIT License
@@ -33,13 +33,34 @@ License: MIT License
33
33
  licenses of all dependencies before using or distributing this software to
34
34
  ensure compliance with their respective terms.
35
35
  Project-URL: Homepage, https://github.com/perrette/scribe
36
- Keywords: speech-to-text,speech recognition,transcription,dictation,voice-typing,voice-to-text,realtime,streaming,language,AI,local,API,cli,tray,vosk,whisper,openai,groq,gpt-4o,linux,wayland,keyboard,clipboard
36
+ Project-URL: Source, https://github.com/perrette/scribe
37
+ Project-URL: Issues, https://github.com/perrette/scribe/issues
38
+ Project-URL: Changelog, https://github.com/perrette/scribe/releases
39
+ Project-URL: Funding, https://github.com/sponsors/perrette
40
+ Keywords: speech-to-text,stt,transcription,dictation,voice-typing,voice-recognition,multilingual,realtime,streaming,cli,tray,vosk,whisper,faster-whisper,openai,groq,gpt-4o,linux,wayland,keyboard,clipboard,microphone,audio
41
+ Classifier: Development Status :: 5 - Production/Stable
42
+ Classifier: Intended Audience :: End Users/Desktop
43
+ Classifier: Intended Audience :: Developers
44
+ Classifier: License :: OSI Approved :: MIT License
37
45
  Classifier: Programming Language :: Python :: 3.9
38
46
  Classifier: Programming Language :: Python :: 3.10
39
47
  Classifier: Programming Language :: Python :: 3.11
40
48
  Classifier: Programming Language :: Python :: 3.12
41
49
  Classifier: Programming Language :: Python :: 3.13
42
50
  Classifier: Operating System :: OS Independent
51
+ Classifier: Environment :: Console
52
+ Classifier: Environment :: X11 Applications
53
+ Classifier: Environment :: MacOS X
54
+ Classifier: Environment :: Win32 (MS Windows)
55
+ Classifier: Natural Language :: English
56
+ Classifier: Natural Language :: French
57
+ Classifier: Natural Language :: German
58
+ Classifier: Natural Language :: Italian
59
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
60
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
61
+ Classifier: Topic :: Office/Business
62
+ Classifier: Topic :: Text Processing :: Linguistic
63
+ Classifier: Topic :: Utilities
43
64
  Requires-Python: >=3.9
44
65
  Description-Content-Type: text/markdown
45
66
  License-File: LICENSE
@@ -51,8 +72,10 @@ Requires-Dist: pyperclip
51
72
  Requires-Dist: unidecode
52
73
  Requires-Dist: termcolor
53
74
  Requires-Dist: platformdirs
54
- Requires-Dist: desktop-ai-core>=0.2.0
75
+ Requires-Dist: desktop-ai-core>=0.3.1
55
76
  Requires-Dist: onnxruntime
77
+ Requires-Dist: pynput
78
+ Requires-Dist: pystray
56
79
  Provides-Extra: keyboard
57
80
  Requires-Dist: pynput; extra == "keyboard"
58
81
  Provides-Extra: whisper
@@ -62,8 +85,7 @@ Requires-Dist: pywhispercpp; extra == "whisper-futo"
62
85
  Provides-Extra: vosk
63
86
  Requires-Dist: vosk; extra == "vosk"
64
87
  Provides-Extra: app
65
- Requires-Dist: pystray; extra == "app"
66
- Requires-Dist: PyGObject; extra == "app"
88
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
67
89
  Provides-Extra: openai
68
90
  Requires-Dist: openai<3,>=2.37.0; extra == "openai"
69
91
  Requires-Dist: soundfile; extra == "openai"
@@ -72,13 +94,11 @@ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
72
94
  Requires-Dist: soundfile; extra == "groq"
73
95
  Provides-Extra: vad
74
96
  Provides-Extra: all
75
- Requires-Dist: pynput; extra == "all"
76
97
  Requires-Dist: faster-whisper; extra == "all"
77
- Requires-Dist: pywhispercpp; extra == "all"
78
98
  Requires-Dist: openai<3,>=2.37.0; extra == "all"
79
99
  Requires-Dist: soundfile; extra == "all"
80
100
  Requires-Dist: vosk; extra == "all"
81
- Requires-Dist: pystray; extra == "all"
101
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
82
102
  Dynamic: license-file
83
103
 
84
104
  [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
@@ -92,11 +112,13 @@ cloud-based APIs, batch and streaming workflows.
92
112
 
93
113
  ## What it does
94
114
 
95
- - Records from your mic and transcribes via one of four backends —
96
- **Vosk** (local, streaming), **Whisper** (local, batch), **OpenAI**
97
- (cloud, batch *or* streaming), **Groq** (cloud, batch).
98
- - Delivers the transcript three ways: paste into the focused window
99
- (default), copy to clipboard, or print to the terminal.
115
+ - Records from your mic and transcribes via one of five backends —
116
+ **Vosk** (local, streaming), **Whisper** (local, batch),
117
+ **Whisper FUTO** (local, batch — ACFT-tuned for short dictations),
118
+ **OpenAI** (cloud, batch *or* streaming), **Groq** (cloud, batch).
119
+ - Delivers the transcript four ways: paste into the focused window
120
+ (default), copy to clipboard, print to the terminal, or write to
121
+ a file.
100
122
  - Runs as a **system tray icon** with a single Record button, or as an
101
123
  interactive **terminal TUI** — same menu in both.
102
124
  - Hooks into your DE's keyboard shortcuts via `SIGUSR1` (toggle
@@ -106,13 +128,29 @@ cloud-based APIs, batch and streaming workflows.
106
128
 
107
129
  ## Install
108
130
 
131
+ **Linux / macOS:**
132
+
109
133
  ```bash
110
134
  sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
111
135
  pip install scribe-cli[all]
112
136
  export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
113
137
  ```
114
138
 
115
- See documentation below for setting up keyboard input on Ubuntu Wayland.
139
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
140
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
141
+
142
+ ```powershell
143
+ py -m venv .venv
144
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
145
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
146
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
147
+ ```
148
+
149
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
150
+ and `pystray` are regular dependencies, and there is nothing to create by
151
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
152
+ for the full Windows walkthrough, and the documentation below for setting up
153
+ keyboard input on Ubuntu Wayland.
116
154
 
117
155
 
118
156
  ## Usage
@@ -126,8 +164,8 @@ scribe
126
164
  This launches the system tray icon. Press Record, speak, press Stop —
127
165
  the transcription lands in the focused window. Scribe picks the first
128
166
  backend whose key / dependency is present, in order **`groq` →
129
- `openai` → `whisper` → `vosk`**, so with `GROQ_API_KEY` set the
130
- command above is equivalent to:
167
+ `openai` → `whisper-futo` → `whisper` → `vosk`**, so with `GROQ_API_KEY`
168
+ set the command above is equivalent to:
131
169
 
132
170
  ```bash
133
171
  scribe --backend groq --model whisper-large-v3-turbo
@@ -142,15 +180,17 @@ scribe --backend openai --model gpt-4o-mini-transcribe # OpenAI sweet spot
142
180
  scribe --backend openai --model gpt-realtime-whisper # OpenAI streaming
143
181
  scribe --backend whisper --model small # local, no API key
144
182
  scribe --frontend terminal # interactive TUI menu
145
- scribe --frontend terminal --no-interactive # record immediately, no menu
183
+ scribe --record # start recording immediately on launch (works in tray or terminal)
184
+ scribe --record --frontend terminal --mode file # one-shot batched dictation → file
185
+ scribe --record --frontend terminal --mode file --stream # streamed: chunks appended live as you speak
146
186
  scribe --mode clipboard # copy to clipboard, no keystroke
147
187
  scribe --mode terminal # only print to stdout
148
- scribe -o transcript.txt # also append to a file
188
+ scribe --mode file -o transcript.txt # append to a file (no keystroke / clipboard)
149
189
  ```
150
190
 
151
191
  With `--no-interactive` (terminal frontend only), scribe skips the
152
192
  interactive menu and starts recording right away — handy for scripted,
153
- one-shot transcriptions. `--no-prompt` is kept as a deprecated alias.
193
+ one-shot transcriptions.
154
194
 
155
195
  Bias the recogniser toward names, jargon, or a domain glossary with
156
196
  `--prompt "free text hint"` and `--words word1 word2 ...` (each also
@@ -161,12 +201,13 @@ for what each backend does with them.
161
201
 
162
202
  ## Backends at a glance
163
203
 
164
- | Backend | `--backend` | Default model | Streaming model(s) | Requires |
165
- |-----------------|-------------|----------------------------|---------------------------|-------------------------------------|
166
- | Groq (cloud) | `groq` | `whisper-large-v3-turbo` | — | `GROQ_API_KEY` |
167
- | OpenAI (cloud) | `openai` | `gpt-4o-mini-transcribe` | `gpt-realtime-whisper` | `OPENAI_API_KEY` |
168
- | Whisper (local) | `whisper` | `small` | — | `pip install scribe-cli[whisper]` |
169
- | Vosk (local) | `vosk` | language-dependent | all Vosk models | `pip install scribe-cli[vosk]` |
204
+ | Backend | `--backend` | Default model | Streaming model(s) | Requires |
205
+ |----------------------|-----------------|----------------------------|---------------------------|----------------------------------------|
206
+ | Groq (cloud) | `groq` | `whisper-large-v3-turbo` | — | `GROQ_API_KEY` |
207
+ | OpenAI (cloud) | `openai` | `gpt-4o-mini-transcribe` | `gpt-realtime-whisper` | `OPENAI_API_KEY` |
208
+ | Whisper FUTO (local) | `whisper-futo` | `small` | — | `pip install scribe-cli[whisper-futo]` |
209
+ | Whisper (local) | `whisper` | `small` | — | `pip install scribe-cli[whisper]` |
210
+ | Vosk (local) | `vosk` | language-dependent | all Vosk models | `pip install scribe-cli[vosk]` |
170
211
 
171
212
  Whether a transcription appears live as you speak or all at once when
172
213
  you stop depends on the **model** picked — see
@@ -175,8 +216,11 @@ you stop depends on the **model** picked — see
175
216
 
176
217
  ### Getting an API key
177
218
 
178
- Groq is a good cloud backend to start with — very fast, quite accurate, and the
179
- **free tier** is generous enough for everyday dictation. Sign up at
219
+ Groq is the **recommended cloud backend by default** — extremely fast
220
+ (by a wide margin compared to other cloud STT options, especially in
221
+ **Stream** mode where the per-chunk roundtrip latency dominates the
222
+ perceived speed), quite accurate, and the **free tier** is generous
223
+ enough for everyday dictation. Sign up at
180
224
  [console.groq.com](https://console.groq.com/), create an API key
181
225
  under **Settings → API Keys**, and export it as `GROQ_API_KEY`.
182
226
 
@@ -188,8 +232,9 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
188
232
  - [Installation & dependencies](docs/installation.md) — PortAudio,
189
233
  extras, Ubuntu / GNOME tray libs.
190
234
  - [Backends in detail](docs/backends.md) — model lists, when to pick
191
- which, the realtime model.
192
- - [Keyboard modes & typer backends](docs/keyboard.md) — keystroke vs
235
+ which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
236
+ (Balanced / Patient profiles).
237
+ - [Output modes & typer backends](docs/output.md) — keystroke vs
193
238
  clipboard, Wayland / `eitype`, `--type-direct`.
194
239
  - [System tray & global hotkeys](docs/tray.md) — menu tree, icon
195
240
  states, `SIGUSR1`/`SIGUSR2`.
@@ -198,10 +243,33 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
198
243
  - [Fine tuning & CLI reference](docs/cli.md) — every `scribe --help`
199
244
  flag with examples.
200
245
 
246
+ ## Related projects
247
+
248
+ - **[bard](https://github.com/perrette/bard)** — TTS sibling of scribe,
249
+ same tray/CLI architecture in reverse: highlight text, hear it
250
+ spoken. Shares the [`desktop-ai-core`](https://github.com/perrette/desktop-ai-core)
251
+ backbone (frontends, providers, dialog helpers).
252
+
201
253
  ## Compatibility
202
254
 
203
- Initially developed for Python 3 on Ubuntu 24.04 (GNOME + Wayland);
204
- works on macOS and Windows too. Wayland keystroke injection is
205
- convoluted but [solved](docs/keyboard.md). For dependencies of
206
- individual subsystems, check `pynput` (keyboard) and `pystray` (tray
207
- icon).
255
+ | OS | Status |
256
+ |--------------------|---------------------------------------------------------------------|
257
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
258
+ | macOS | Works. |
259
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
260
+
261
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
262
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
263
+ `pystray` (tray icon).
264
+
265
+ **Windows notes:**
266
+
267
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
268
+ default. Pin it via *Settings → Personalization → Taskbar → Other
269
+ system tray icons*.
270
+ - A **single click** on the tray icon fires the default action (Record).
271
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
272
+ AppIndicator backend only opens the menu (a backend limitation, not a
273
+ bug).
274
+ - If recording fails, allow mic access under *Settings → Privacy &
275
+ security → Microphone → "Let desktop apps access your microphone"*.
@@ -0,0 +1,172 @@
1
+ [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
2
+ ![](https://img.shields.io/python/required-version-toml?tomlFilePath=https%3A%2F%2Fraw.githubusercontent.com%2Fperrette%2Fscribe%2Frefs%2Fheads%2Fmain%2Fpyproject.toml)
3
+
4
+ # Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
5
+
6
+ **Talk. It types.** Scribe is a speech-to-text CLI and tray app that
7
+ pipes transcribed text straight into the focused window. It supports local and
8
+ cloud-based APIs, batch and streaming workflows.
9
+
10
+ ## What it does
11
+
12
+ - Records from your mic and transcribes via one of five backends —
13
+ **Vosk** (local, streaming), **Whisper** (local, batch),
14
+ **Whisper FUTO** (local, batch — ACFT-tuned for short dictations),
15
+ **OpenAI** (cloud, batch *or* streaming), **Groq** (cloud, batch).
16
+ - Delivers the transcript four ways: paste into the focused window
17
+ (default), copy to clipboard, print to the terminal, or write to
18
+ a file.
19
+ - Runs as a **system tray icon** with a single Record button, or as an
20
+ interactive **terminal TUI** — same menu in both.
21
+ - Hooks into your DE's keyboard shortcuts via `SIGUSR1` (toggle
22
+ recording) and `SIGUSR2` (cancel).
23
+ - Cross-platform: tested on Ubuntu (X11 and Wayland), macOS, Windows;
24
+ works under Termux for clipboard / terminal output.
25
+
26
+ ## Install
27
+
28
+ **Linux / macOS:**
29
+
30
+ ```bash
31
+ sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
32
+ pip install scribe-cli[all]
33
+ export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
34
+ ```
35
+
36
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
37
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
38
+
39
+ ```powershell
40
+ py -m venv .venv
41
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
42
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
43
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
44
+ ```
45
+
46
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
47
+ and `pystray` are regular dependencies, and there is nothing to create by
48
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
49
+ for the full Windows walkthrough, and the documentation below for setting up
50
+ keyboard input on Ubuntu Wayland.
51
+
52
+
53
+ ## Usage
54
+
55
+ In a terminal:
56
+
57
+ ```bash
58
+ scribe
59
+ ```
60
+
61
+ This launches the system tray icon. Press Record, speak, press Stop —
62
+ the transcription lands in the focused window. Scribe picks the first
63
+ backend whose key / dependency is present, in order **`groq` →
64
+ `openai` → `whisper-futo` → `whisper` → `vosk`**, so with `GROQ_API_KEY`
65
+ set the command above is equivalent to:
66
+
67
+ ```bash
68
+ scribe --backend groq --model whisper-large-v3-turbo
69
+ ```
70
+
71
+ <img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
72
+
73
+ You can override the defaults or drop the tray entirely:
74
+
75
+ ```bash
76
+ scribe --backend openai --model gpt-4o-mini-transcribe # OpenAI sweet spot
77
+ scribe --backend openai --model gpt-realtime-whisper # OpenAI streaming
78
+ scribe --backend whisper --model small # local, no API key
79
+ scribe --frontend terminal # interactive TUI menu
80
+ scribe --record # start recording immediately on launch (works in tray or terminal)
81
+ scribe --record --frontend terminal --mode file # one-shot batched dictation → file
82
+ scribe --record --frontend terminal --mode file --stream # streamed: chunks appended live as you speak
83
+ scribe --mode clipboard # copy to clipboard, no keystroke
84
+ scribe --mode terminal # only print to stdout
85
+ scribe --mode file -o transcript.txt # append to a file (no keystroke / clipboard)
86
+ ```
87
+
88
+ With `--no-interactive` (terminal frontend only), scribe skips the
89
+ interactive menu and starts recording right away — handy for scripted,
90
+ one-shot transcriptions.
91
+
92
+ Bias the recogniser toward names, jargon, or a domain glossary with
93
+ `--prompt "free text hint"` and `--words word1 word2 ...` (each also
94
+ accepts a `--prompt-file` / `--words-file` companion). See
95
+ [docs/backends.md › Vocabulary biasing](docs/backends.md#vocabulary-biasing)
96
+ for what each backend does with them.
97
+
98
+
99
+ ## Backends at a glance
100
+
101
+ | Backend | `--backend` | Default model | Streaming model(s) | Requires |
102
+ |----------------------|-----------------|----------------------------|---------------------------|----------------------------------------|
103
+ | Groq (cloud) | `groq` | `whisper-large-v3-turbo` | — | `GROQ_API_KEY` |
104
+ | OpenAI (cloud) | `openai` | `gpt-4o-mini-transcribe` | `gpt-realtime-whisper` | `OPENAI_API_KEY` |
105
+ | Whisper FUTO (local) | `whisper-futo` | `small` | — | `pip install scribe-cli[whisper-futo]` |
106
+ | Whisper (local) | `whisper` | `small` | — | `pip install scribe-cli[whisper]` |
107
+ | Vosk (local) | `vosk` | language-dependent | all Vosk models | `pip install scribe-cli[vosk]` |
108
+
109
+ Whether a transcription appears live as you speak or all at once when
110
+ you stop depends on the **model** picked — see
111
+ [docs/backends.md](docs/backends.md).
112
+
113
+
114
+ ### Getting an API key
115
+
116
+ Groq is the **recommended cloud backend by default** — extremely fast
117
+ (by a wide margin compared to other cloud STT options, especially in
118
+ **Stream** mode where the per-chunk roundtrip latency dominates the
119
+ perceived speed), quite accurate, and the **free tier** is generous
120
+ enough for everyday dictation. Sign up at
121
+ [console.groq.com](https://console.groq.com/), create an API key
122
+ under **Settings → API Keys**, and export it as `GROQ_API_KEY`.
123
+
124
+ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe` as it is also fast and perhaps more accurate for my accent-tainted English.
125
+
126
+
127
+ ## Documentation
128
+
129
+ - [Installation & dependencies](docs/installation.md) — PortAudio,
130
+ extras, Ubuntu / GNOME tray libs.
131
+ - [Backends in detail](docs/backends.md) — model lists, when to pick
132
+ which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
133
+ (Balanced / Patient profiles).
134
+ - [Output modes & typer backends](docs/output.md) — keystroke vs
135
+ clipboard, Wayland / `eitype`, `--type-direct`.
136
+ - [System tray & global hotkeys](docs/tray.md) — menu tree, icon
137
+ states, `SIGUSR1`/`SIGUSR2`.
138
+ - [Desktop entry & autostart (`scribe-install`)](docs/desktop-install.md)
139
+ — GNOME / KDE launcher integration.
140
+ - [Fine tuning & CLI reference](docs/cli.md) — every `scribe --help`
141
+ flag with examples.
142
+
143
+ ## Related projects
144
+
145
+ - **[bard](https://github.com/perrette/bard)** — TTS sibling of scribe,
146
+ same tray/CLI architecture in reverse: highlight text, hear it
147
+ spoken. Shares the [`desktop-ai-core`](https://github.com/perrette/desktop-ai-core)
148
+ backbone (frontends, providers, dialog helpers).
149
+
150
+ ## Compatibility
151
+
152
+ | OS | Status |
153
+ |--------------------|---------------------------------------------------------------------|
154
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
155
+ | macOS | Works. |
156
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
157
+
158
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
159
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
160
+ `pystray` (tray icon).
161
+
162
+ **Windows notes:**
163
+
164
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
165
+ default. Pin it via *Settings → Personalization → Taskbar → Other
166
+ system tray icons*.
167
+ - A **single click** on the tray icon fires the default action (Record).
168
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
169
+ AppIndicator backend only opens the menu (a backend limitation, not a
170
+ bug).
171
+ - If recording fails, allow mic access under *Settings → Privacy &
172
+ security → Microphone → "Let desktop apps access your microphone"*.
Binary file