scribe-cli 0.18.0__tar.gz → 0.18.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.gitignore +1 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/PKG-INFO +102 -34
- scribe_cli-0.18.1/README.md +172 -0
- scribe_cli-0.18.1/docs/app-tray-menu.png +0 -0
- scribe_cli-0.18.1/docs/backends.md +400 -0
- scribe_cli-0.18.1/docs/cli.md +208 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/desktop-install.md +1 -1
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/installation.md +46 -5
- scribe_cli-0.18.0/docs/keyboard.md → scribe_cli-0.18.1/docs/output.md +98 -36
- scribe_cli-0.18.1/docs/tray.md +173 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/pyproject.toml +43 -11
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/_version.py +3 -3
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/app.py +493 -148
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/groq.py +4 -3
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/openai_api.py +12 -3
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/openai_realtime.py +59 -5
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/vosk.py +20 -4
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/whisper.py +17 -4
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/whisper_futo.py +15 -3
- scribe_cli-0.18.1/scribe/dialog.py +82 -0
- scribe_cli-0.18.1/scribe/menu.py +1678 -0
- scribe_cli-0.18.1/scribe/models.py +466 -0
- scribe_cli-0.18.1/scribe/output.py +237 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/session.py +29 -4
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/PKG-INFO +102 -34
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/SOURCES.txt +9 -1
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/requires.txt +8 -5
- scribe_cli-0.18.1/tests/test_backend_matrix.py +295 -0
- scribe_cli-0.18.1/tests/test_compose_prompt.py +153 -0
- scribe_cli-0.18.1/tests/test_debug_logging.py +191 -0
- scribe_cli-0.18.1/tests/test_output.py +165 -0
- scribe_cli-0.18.1/tests/test_output_file_picker.py +57 -0
- scribe_cli-0.18.1/tests/test_prompt_file_picker.py +165 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_pseudo_streaming.py +218 -43
- scribe_cli-0.18.0/README.md +0 -124
- scribe_cli-0.18.0/docs/app-tray-menu.png +0 -0
- scribe_cli-0.18.0/docs/backends.md +0 -238
- scribe_cli-0.18.0/docs/cli.md +0 -156
- scribe_cli-0.18.0/docs/tray.md +0 -97
- scribe_cli-0.18.0/scribe/menu.py +0 -1019
- scribe_cli-0.18.0/scribe/models.py +0 -333
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.github/FUNDING.yml +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/.github/workflows/pypi.yml +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/LICENSE +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/docs/roadmap-libei.md +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/icon.xcf +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/__init__.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/audio.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/backends/__init__.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/install_desktop.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/keyboard.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/models.toml +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/saverecording.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/testpynput.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/__init__.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/base.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/eitype.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/pynput.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/wtype.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/typers/ydotool.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe/util.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/dependency_links.txt +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/entry_points.txt +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_cli.egg-info/top_level.txt +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/__init__.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon.png +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon_recording.png +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/share/icon_writing.png +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/silero_vad.LICENSE +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/silero_vad.onnx +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scribe_data/templates/scribe.desktop +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scripts/bench_whisper_local.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/scripts/test_python_versions_install.sh +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/setup.cfg +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_openai_realtime_coalesce.py +0 -0
- {scribe_cli-0.18.0 → scribe_cli-0.18.1}/tests/test_whisper_futo.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scribe-cli
|
|
3
|
-
Version: 0.18.
|
|
3
|
+
Version: 0.18.1
|
|
4
4
|
Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
|
|
5
5
|
Author-email: Mahé Perrette <mahe.perrette@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -33,13 +33,34 @@ License: MIT License
|
|
|
33
33
|
licenses of all dependencies before using or distributing this software to
|
|
34
34
|
ensure compliance with their respective terms.
|
|
35
35
|
Project-URL: Homepage, https://github.com/perrette/scribe
|
|
36
|
-
|
|
36
|
+
Project-URL: Source, https://github.com/perrette/scribe
|
|
37
|
+
Project-URL: Issues, https://github.com/perrette/scribe/issues
|
|
38
|
+
Project-URL: Changelog, https://github.com/perrette/scribe/releases
|
|
39
|
+
Project-URL: Funding, https://github.com/sponsors/perrette
|
|
40
|
+
Keywords: speech-to-text,stt,transcription,dictation,voice-typing,voice-recognition,multilingual,realtime,streaming,cli,tray,vosk,whisper,faster-whisper,openai,groq,gpt-4o,linux,wayland,keyboard,clipboard,microphone,audio
|
|
41
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
42
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
43
|
+
Classifier: Intended Audience :: Developers
|
|
44
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
37
45
|
Classifier: Programming Language :: Python :: 3.9
|
|
38
46
|
Classifier: Programming Language :: Python :: 3.10
|
|
39
47
|
Classifier: Programming Language :: Python :: 3.11
|
|
40
48
|
Classifier: Programming Language :: Python :: 3.12
|
|
41
49
|
Classifier: Programming Language :: Python :: 3.13
|
|
42
50
|
Classifier: Operating System :: OS Independent
|
|
51
|
+
Classifier: Environment :: Console
|
|
52
|
+
Classifier: Environment :: X11 Applications
|
|
53
|
+
Classifier: Environment :: MacOS X
|
|
54
|
+
Classifier: Environment :: Win32 (MS Windows)
|
|
55
|
+
Classifier: Natural Language :: English
|
|
56
|
+
Classifier: Natural Language :: French
|
|
57
|
+
Classifier: Natural Language :: German
|
|
58
|
+
Classifier: Natural Language :: Italian
|
|
59
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
60
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
61
|
+
Classifier: Topic :: Office/Business
|
|
62
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
63
|
+
Classifier: Topic :: Utilities
|
|
43
64
|
Requires-Python: >=3.9
|
|
44
65
|
Description-Content-Type: text/markdown
|
|
45
66
|
License-File: LICENSE
|
|
@@ -51,8 +72,10 @@ Requires-Dist: pyperclip
|
|
|
51
72
|
Requires-Dist: unidecode
|
|
52
73
|
Requires-Dist: termcolor
|
|
53
74
|
Requires-Dist: platformdirs
|
|
54
|
-
Requires-Dist: desktop-ai-core>=0.
|
|
75
|
+
Requires-Dist: desktop-ai-core>=0.3.1
|
|
55
76
|
Requires-Dist: onnxruntime
|
|
77
|
+
Requires-Dist: pynput
|
|
78
|
+
Requires-Dist: pystray
|
|
56
79
|
Provides-Extra: keyboard
|
|
57
80
|
Requires-Dist: pynput; extra == "keyboard"
|
|
58
81
|
Provides-Extra: whisper
|
|
@@ -62,8 +85,7 @@ Requires-Dist: pywhispercpp; extra == "whisper-futo"
|
|
|
62
85
|
Provides-Extra: vosk
|
|
63
86
|
Requires-Dist: vosk; extra == "vosk"
|
|
64
87
|
Provides-Extra: app
|
|
65
|
-
Requires-Dist:
|
|
66
|
-
Requires-Dist: PyGObject; extra == "app"
|
|
88
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
|
|
67
89
|
Provides-Extra: openai
|
|
68
90
|
Requires-Dist: openai<3,>=2.37.0; extra == "openai"
|
|
69
91
|
Requires-Dist: soundfile; extra == "openai"
|
|
@@ -72,13 +94,11 @@ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
|
|
|
72
94
|
Requires-Dist: soundfile; extra == "groq"
|
|
73
95
|
Provides-Extra: vad
|
|
74
96
|
Provides-Extra: all
|
|
75
|
-
Requires-Dist: pynput; extra == "all"
|
|
76
97
|
Requires-Dist: faster-whisper; extra == "all"
|
|
77
|
-
Requires-Dist: pywhispercpp; extra == "all"
|
|
78
98
|
Requires-Dist: openai<3,>=2.37.0; extra == "all"
|
|
79
99
|
Requires-Dist: soundfile; extra == "all"
|
|
80
100
|
Requires-Dist: vosk; extra == "all"
|
|
81
|
-
Requires-Dist:
|
|
101
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
|
|
82
102
|
Dynamic: license-file
|
|
83
103
|
|
|
84
104
|
[](https://pypi.org/project/scribe-cli)
|
|
@@ -92,11 +112,13 @@ cloud-based APIs, batch and streaming workflows.
|
|
|
92
112
|
|
|
93
113
|
## What it does
|
|
94
114
|
|
|
95
|
-
- Records from your mic and transcribes via one of
|
|
96
|
-
**Vosk** (local, streaming), **Whisper** (local, batch),
|
|
97
|
-
(
|
|
98
|
-
|
|
99
|
-
|
|
115
|
+
- Records from your mic and transcribes via one of five backends —
|
|
116
|
+
**Vosk** (local, streaming), **Whisper** (local, batch),
|
|
117
|
+
**Whisper FUTO** (local, batch — ACFT-tuned for short dictations),
|
|
118
|
+
**OpenAI** (cloud, batch *or* streaming), **Groq** (cloud, batch).
|
|
119
|
+
- Delivers the transcript four ways: paste into the focused window
|
|
120
|
+
(default), copy to clipboard, print to the terminal, or write to
|
|
121
|
+
a file.
|
|
100
122
|
- Runs as a **system tray icon** with a single Record button, or as an
|
|
101
123
|
interactive **terminal TUI** — same menu in both.
|
|
102
124
|
- Hooks into your DE's keyboard shortcuts via `SIGUSR1` (toggle
|
|
@@ -106,13 +128,29 @@ cloud-based APIs, batch and streaming workflows.
|
|
|
106
128
|
|
|
107
129
|
## Install
|
|
108
130
|
|
|
131
|
+
**Linux / macOS:**
|
|
132
|
+
|
|
109
133
|
```bash
|
|
110
134
|
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
111
135
|
pip install scribe-cli[all]
|
|
112
136
|
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
113
137
|
```
|
|
114
138
|
|
|
115
|
-
|
|
139
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
140
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
141
|
+
|
|
142
|
+
```powershell
|
|
143
|
+
py -m venv .venv
|
|
144
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
145
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
146
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
The tray app and keyboard typing work out of the box on Windows — `pynput`
|
|
150
|
+
and `pystray` are regular dependencies, and there is nothing to create by
|
|
151
|
+
hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
|
|
152
|
+
for the full Windows walkthrough, and the documentation below for setting up
|
|
153
|
+
keyboard input on Ubuntu Wayland.
|
|
116
154
|
|
|
117
155
|
|
|
118
156
|
## Usage
|
|
@@ -126,8 +164,8 @@ scribe
|
|
|
126
164
|
This launches the system tray icon. Press Record, speak, press Stop —
|
|
127
165
|
the transcription lands in the focused window. Scribe picks the first
|
|
128
166
|
backend whose key / dependency is present, in order **`groq` →
|
|
129
|
-
`openai` → `whisper` → `vosk`**, so with `GROQ_API_KEY`
|
|
130
|
-
command above is equivalent to:
|
|
167
|
+
`openai` → `whisper-futo` → `whisper` → `vosk`**, so with `GROQ_API_KEY`
|
|
168
|
+
set the command above is equivalent to:
|
|
131
169
|
|
|
132
170
|
```bash
|
|
133
171
|
scribe --backend groq --model whisper-large-v3-turbo
|
|
@@ -142,15 +180,17 @@ scribe --backend openai --model gpt-4o-mini-transcribe # OpenAI sweet spot
|
|
|
142
180
|
scribe --backend openai --model gpt-realtime-whisper # OpenAI streaming
|
|
143
181
|
scribe --backend whisper --model small # local, no API key
|
|
144
182
|
scribe --frontend terminal # interactive TUI menu
|
|
145
|
-
scribe --
|
|
183
|
+
scribe --record # start recording immediately on launch (works in tray or terminal)
|
|
184
|
+
scribe --record --frontend terminal --mode file # one-shot batched dictation → file
|
|
185
|
+
scribe --record --frontend terminal --mode file --stream # streamed: chunks appended live as you speak
|
|
146
186
|
scribe --mode clipboard # copy to clipboard, no keystroke
|
|
147
187
|
scribe --mode terminal # only print to stdout
|
|
148
|
-
scribe -o transcript.txt
|
|
188
|
+
scribe --mode file -o transcript.txt # append to a file (no keystroke / clipboard)
|
|
149
189
|
```
|
|
150
190
|
|
|
151
191
|
With `--no-interactive` (terminal frontend only), scribe skips the
|
|
152
192
|
interactive menu and starts recording right away — handy for scripted,
|
|
153
|
-
one-shot transcriptions.
|
|
193
|
+
one-shot transcriptions.
|
|
154
194
|
|
|
155
195
|
Bias the recogniser toward names, jargon, or a domain glossary with
|
|
156
196
|
`--prompt "free text hint"` and `--words word1 word2 ...` (each also
|
|
@@ -161,12 +201,13 @@ for what each backend does with them.
|
|
|
161
201
|
|
|
162
202
|
## Backends at a glance
|
|
163
203
|
|
|
164
|
-
| Backend
|
|
165
|
-
|
|
166
|
-
| Groq (cloud)
|
|
167
|
-
| OpenAI (cloud)
|
|
168
|
-
| Whisper (local) | `whisper`
|
|
169
|
-
|
|
|
204
|
+
| Backend | `--backend` | Default model | Streaming model(s) | Requires |
|
|
205
|
+
|----------------------|-----------------|----------------------------|---------------------------|----------------------------------------|
|
|
206
|
+
| Groq (cloud) | `groq` | `whisper-large-v3-turbo` | — | `GROQ_API_KEY` |
|
|
207
|
+
| OpenAI (cloud) | `openai` | `gpt-4o-mini-transcribe` | `gpt-realtime-whisper` | `OPENAI_API_KEY` |
|
|
208
|
+
| Whisper FUTO (local) | `whisper-futo` | `small` | — | `pip install scribe-cli[whisper-futo]` |
|
|
209
|
+
| Whisper (local) | `whisper` | `small` | — | `pip install scribe-cli[whisper]` |
|
|
210
|
+
| Vosk (local) | `vosk` | language-dependent | all Vosk models | `pip install scribe-cli[vosk]` |
|
|
170
211
|
|
|
171
212
|
Whether a transcription appears live as you speak or all at once when
|
|
172
213
|
you stop depends on the **model** picked — see
|
|
@@ -175,8 +216,11 @@ you stop depends on the **model** picked — see
|
|
|
175
216
|
|
|
176
217
|
### Getting an API key
|
|
177
218
|
|
|
178
|
-
Groq is
|
|
179
|
-
|
|
219
|
+
Groq is the **recommended cloud backend by default** — extremely fast
|
|
220
|
+
(by a wide margin compared to other cloud STT options, especially in
|
|
221
|
+
**Stream** mode where the per-chunk roundtrip latency dominates the
|
|
222
|
+
perceived speed), quite accurate, and the **free tier** is generous
|
|
223
|
+
enough for everyday dictation. Sign up at
|
|
180
224
|
[console.groq.com](https://console.groq.com/), create an API key
|
|
181
225
|
under **Settings → API Keys**, and export it as `GROQ_API_KEY`.
|
|
182
226
|
|
|
@@ -188,8 +232,9 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
188
232
|
- [Installation & dependencies](docs/installation.md) — PortAudio,
|
|
189
233
|
extras, Ubuntu / GNOME tray libs.
|
|
190
234
|
- [Backends in detail](docs/backends.md) — model lists, when to pick
|
|
191
|
-
which, the realtime model.
|
|
192
|
-
|
|
235
|
+
which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
|
|
236
|
+
(Balanced / Patient profiles).
|
|
237
|
+
- [Output modes & typer backends](docs/output.md) — keystroke vs
|
|
193
238
|
clipboard, Wayland / `eitype`, `--type-direct`.
|
|
194
239
|
- [System tray & global hotkeys](docs/tray.md) — menu tree, icon
|
|
195
240
|
states, `SIGUSR1`/`SIGUSR2`.
|
|
@@ -198,10 +243,33 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
198
243
|
- [Fine tuning & CLI reference](docs/cli.md) — every `scribe --help`
|
|
199
244
|
flag with examples.
|
|
200
245
|
|
|
246
|
+
## Related projects
|
|
247
|
+
|
|
248
|
+
- **[bard](https://github.com/perrette/bard)** — TTS sibling of scribe,
|
|
249
|
+
same tray/CLI architecture in reverse: highlight text, hear it
|
|
250
|
+
spoken. Shares the [`desktop-ai-core`](https://github.com/perrette/desktop-ai-core)
|
|
251
|
+
backbone (frontends, providers, dialog helpers).
|
|
252
|
+
|
|
201
253
|
## Compatibility
|
|
202
254
|
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
255
|
+
| OS | Status |
|
|
256
|
+
|--------------------|---------------------------------------------------------------------|
|
|
257
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
258
|
+
| macOS | Works. |
|
|
259
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
|
|
260
|
+
|
|
261
|
+
Wayland keystroke injection is convoluted but [solved](docs/output.md).
|
|
262
|
+
For dependencies of individual subsystems, check `pynput` (keyboard) and
|
|
263
|
+
`pystray` (tray icon).
|
|
264
|
+
|
|
265
|
+
**Windows notes:**
|
|
266
|
+
|
|
267
|
+
- The tray icon is hidden under the taskbar overflow arrow (`^`) by
|
|
268
|
+
default. Pin it via *Settings → Personalization → Taskbar → Other
|
|
269
|
+
system tray icons*.
|
|
270
|
+
- A **single click** on the tray icon fires the default action (Record).
|
|
271
|
+
This is a free bonus of pystray's Win32 backend; on Ubuntu the
|
|
272
|
+
AppIndicator backend only opens the menu (a backend limitation, not a
|
|
273
|
+
bug).
|
|
274
|
+
- If recording fails, allow mic access under *Settings → Privacy &
|
|
275
|
+
security → Microphone → "Let desktop apps access your microphone"*.
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
[](https://pypi.org/project/scribe-cli)
|
|
2
|
+

|
|
3
|
+
|
|
4
|
+
# Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
|
|
5
|
+
|
|
6
|
+
**Talk. It types.** Scribe is a speech-to-text CLI and tray app that
|
|
7
|
+
pipes transcribed text straight into the focused window. It supports local and
|
|
8
|
+
cloud-based APIs, batch and streaming workflows.
|
|
9
|
+
|
|
10
|
+
## What it does
|
|
11
|
+
|
|
12
|
+
- Records from your mic and transcribes via one of five backends —
|
|
13
|
+
**Vosk** (local, streaming), **Whisper** (local, batch),
|
|
14
|
+
**Whisper FUTO** (local, batch — ACFT-tuned for short dictations),
|
|
15
|
+
**OpenAI** (cloud, batch *or* streaming), **Groq** (cloud, batch).
|
|
16
|
+
- Delivers the transcript four ways: paste into the focused window
|
|
17
|
+
(default), copy to clipboard, print to the terminal, or write to
|
|
18
|
+
a file.
|
|
19
|
+
- Runs as a **system tray icon** with a single Record button, or as an
|
|
20
|
+
interactive **terminal TUI** — same menu in both.
|
|
21
|
+
- Hooks into your DE's keyboard shortcuts via `SIGUSR1` (toggle
|
|
22
|
+
recording) and `SIGUSR2` (cancel).
|
|
23
|
+
- Cross-platform: tested on Ubuntu (X11 and Wayland), macOS, Windows;
|
|
24
|
+
works under Termux for clipboard / terminal output.
|
|
25
|
+
|
|
26
|
+
## Install
|
|
27
|
+
|
|
28
|
+
**Linux / macOS:**
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
32
|
+
pip install scribe-cli[all]
|
|
33
|
+
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
37
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
38
|
+
|
|
39
|
+
```powershell
|
|
40
|
+
py -m venv .venv
|
|
41
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
42
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
43
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The tray app and keyboard typing work out of the box on Windows — `pynput`
|
|
47
|
+
and `pystray` are regular dependencies, and there is nothing to create by
|
|
48
|
+
hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
|
|
49
|
+
for the full Windows walkthrough, and the documentation below for setting up
|
|
50
|
+
keyboard input on Ubuntu Wayland.
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
## Usage
|
|
54
|
+
|
|
55
|
+
In a terminal:
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
scribe
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
This launches the system tray icon. Press Record, speak, press Stop —
|
|
62
|
+
the transcription lands in the focused window. Scribe picks the first
|
|
63
|
+
backend whose key / dependency is present, in order **`groq` →
|
|
64
|
+
`openai` → `whisper-futo` → `whisper` → `vosk`**, so with `GROQ_API_KEY`
|
|
65
|
+
set the command above is equivalent to:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
scribe --backend groq --model whisper-large-v3-turbo
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
<img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
|
|
72
|
+
|
|
73
|
+
You can override the defaults or drop the tray entirely:
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
scribe --backend openai --model gpt-4o-mini-transcribe # OpenAI sweet spot
|
|
77
|
+
scribe --backend openai --model gpt-realtime-whisper # OpenAI streaming
|
|
78
|
+
scribe --backend whisper --model small # local, no API key
|
|
79
|
+
scribe --frontend terminal # interactive TUI menu
|
|
80
|
+
scribe --record # start recording immediately on launch (works in tray or terminal)
|
|
81
|
+
scribe --record --frontend terminal --mode file # one-shot batched dictation → file
|
|
82
|
+
scribe --record --frontend terminal --mode file --stream # streamed: chunks appended live as you speak
|
|
83
|
+
scribe --mode clipboard # copy to clipboard, no keystroke
|
|
84
|
+
scribe --mode terminal # only print to stdout
|
|
85
|
+
scribe --mode file -o transcript.txt # append to a file (no keystroke / clipboard)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
With `--no-interactive` (terminal frontend only), scribe skips the
|
|
89
|
+
interactive menu and starts recording right away — handy for scripted,
|
|
90
|
+
one-shot transcriptions.
|
|
91
|
+
|
|
92
|
+
Bias the recogniser toward names, jargon, or a domain glossary with
|
|
93
|
+
`--prompt "free text hint"` and `--words word1 word2 ...` (each also
|
|
94
|
+
accepts a `--prompt-file` / `--words-file` companion). See
|
|
95
|
+
[docs/backends.md › Vocabulary biasing](docs/backends.md#vocabulary-biasing)
|
|
96
|
+
for what each backend does with them.
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
## Backends at a glance
|
|
100
|
+
|
|
101
|
+
| Backend | `--backend` | Default model | Streaming model(s) | Requires |
|
|
102
|
+
|----------------------|-----------------|----------------------------|---------------------------|----------------------------------------|
|
|
103
|
+
| Groq (cloud) | `groq` | `whisper-large-v3-turbo` | — | `GROQ_API_KEY` |
|
|
104
|
+
| OpenAI (cloud) | `openai` | `gpt-4o-mini-transcribe` | `gpt-realtime-whisper` | `OPENAI_API_KEY` |
|
|
105
|
+
| Whisper FUTO (local) | `whisper-futo` | `small` | — | `pip install scribe-cli[whisper-futo]` |
|
|
106
|
+
| Whisper (local) | `whisper` | `small` | — | `pip install scribe-cli[whisper]` |
|
|
107
|
+
| Vosk (local) | `vosk` | language-dependent | all Vosk models | `pip install scribe-cli[vosk]` |
|
|
108
|
+
|
|
109
|
+
Whether a transcription appears live as you speak or all at once when
|
|
110
|
+
you stop depends on the **model** picked — see
|
|
111
|
+
[docs/backends.md](docs/backends.md).
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
### Getting an API key
|
|
115
|
+
|
|
116
|
+
Groq is the **recommended cloud backend by default** — extremely fast
|
|
117
|
+
(by a wide margin compared to other cloud STT options, especially in
|
|
118
|
+
**Stream** mode where the per-chunk roundtrip latency dominates the
|
|
119
|
+
perceived speed), quite accurate, and the **free tier** is generous
|
|
120
|
+
enough for everyday dictation. Sign up at
|
|
121
|
+
[console.groq.com](https://console.groq.com/), create an API key
|
|
122
|
+
under **Settings → API Keys**, and export it as `GROQ_API_KEY`.
|
|
123
|
+
|
|
124
|
+
I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe` as it is also fast and perhaps more accurate for my accent-tainted English.
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
## Documentation
|
|
128
|
+
|
|
129
|
+
- [Installation & dependencies](docs/installation.md) — PortAudio,
|
|
130
|
+
extras, Ubuntu / GNOME tray libs.
|
|
131
|
+
- [Backends in detail](docs/backends.md) — model lists, when to pick
|
|
132
|
+
which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
|
|
133
|
+
(Balanced / Patient profiles).
|
|
134
|
+
- [Output modes & typer backends](docs/output.md) — keystroke vs
|
|
135
|
+
clipboard, Wayland / `eitype`, `--type-direct`.
|
|
136
|
+
- [System tray & global hotkeys](docs/tray.md) — menu tree, icon
|
|
137
|
+
states, `SIGUSR1`/`SIGUSR2`.
|
|
138
|
+
- [Desktop entry & autostart (`scribe-install`)](docs/desktop-install.md)
|
|
139
|
+
— GNOME / KDE launcher integration.
|
|
140
|
+
- [Fine tuning & CLI reference](docs/cli.md) — every `scribe --help`
|
|
141
|
+
flag with examples.
|
|
142
|
+
|
|
143
|
+
## Related projects
|
|
144
|
+
|
|
145
|
+
- **[bard](https://github.com/perrette/bard)** — TTS sibling of scribe,
|
|
146
|
+
same tray/CLI architecture in reverse: highlight text, hear it
|
|
147
|
+
spoken. Shares the [`desktop-ai-core`](https://github.com/perrette/desktop-ai-core)
|
|
148
|
+
backbone (frontends, providers, dialog helpers).
|
|
149
|
+
|
|
150
|
+
## Compatibility
|
|
151
|
+
|
|
152
|
+
| OS | Status |
|
|
153
|
+
|--------------------|---------------------------------------------------------------------|
|
|
154
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
155
|
+
| macOS | Works. |
|
|
156
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
|
|
157
|
+
|
|
158
|
+
Wayland keystroke injection is convoluted but [solved](docs/output.md).
|
|
159
|
+
For dependencies of individual subsystems, check `pynput` (keyboard) and
|
|
160
|
+
`pystray` (tray icon).
|
|
161
|
+
|
|
162
|
+
**Windows notes:**
|
|
163
|
+
|
|
164
|
+
- The tray icon is hidden under the taskbar overflow arrow (`^`) by
|
|
165
|
+
default. Pin it via *Settings → Personalization → Taskbar → Other
|
|
166
|
+
system tray icons*.
|
|
167
|
+
- A **single click** on the tray icon fires the default action (Record).
|
|
168
|
+
This is a free bonus of pystray's Win32 backend; on Ubuntu the
|
|
169
|
+
AppIndicator backend only opens the menu (a backend limitation, not a
|
|
170
|
+
bug).
|
|
171
|
+
- If recording fails, allow mic access under *Settings → Privacy &
|
|
172
|
+
security → Microphone → "Let desktop apps access your microphone"*.
|
|
Binary file
|