scribe-cli 1.0.1__tar.gz → 1.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/PKG-INFO +46 -14
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/README.md +40 -7
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/backends.md +70 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/cli.md +1 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/installation.md +45 -4
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/tray.md +51 -5
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/pyproject.toml +16 -5
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/_version.py +3 -3
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/app.py +195 -22
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/openai_api.py +1 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/whisper.py +5 -1
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/whisper_futo.py +5 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/dialog.py +26 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/menu.py +126 -2
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/models.py +94 -5
- scribe_cli-1.1.1/scribe/typers/base.py +63 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/eitype.py +13 -25
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/pynput.py +7 -11
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/wtype.py +13 -23
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/PKG-INFO +46 -14
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/SOURCES.txt +4 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/requires.txt +8 -5
- scribe_cli-1.1.1/tests/test_compose_prompt.py +153 -0
- scribe_cli-1.1.1/tests/test_debug_logging.py +191 -0
- scribe_cli-1.1.1/tests/test_prompt_file_picker.py +165 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_pseudo_streaming.py +161 -5
- scribe_cli-1.1.1/tests/test_typers_ascii_fallback.py +79 -0
- scribe_cli-1.0.1/scribe/typers/base.py +0 -18
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.github/FUNDING.yml +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.github/workflows/pypi.yml +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.gitignore +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/LICENSE +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/app-tray-menu.png +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/desktop-install.md +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/output.md +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/roadmap-libei.md +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/icon.xcf +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/__init__.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/audio.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/__init__.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/groq.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/openai_realtime.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/vosk.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/install_desktop.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/keyboard.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/models.toml +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/output.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/saverecording.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/session.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/testpynput.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/__init__.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/ydotool.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/util.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/dependency_links.txt +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/entry_points.txt +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/top_level.txt +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/__init__.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon.png +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon_recording.png +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon_writing.png +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/silero_vad.LICENSE +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/silero_vad.onnx +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/templates/scribe.desktop +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scripts/bench_whisper_local.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scripts/test_python_versions_install.sh +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/setup.cfg +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_backend_matrix.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_openai_realtime_coalesce.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_output.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_output_file_picker.py +0 -0
- {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_whisper_futo.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: scribe-cli
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.1
|
|
4
4
|
Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
|
|
5
5
|
Author-email: Mahé Perrette <mahe.perrette@gmail.com>
|
|
6
6
|
License: MIT License
|
|
@@ -72,8 +72,10 @@ Requires-Dist: pyperclip
|
|
|
72
72
|
Requires-Dist: unidecode
|
|
73
73
|
Requires-Dist: termcolor
|
|
74
74
|
Requires-Dist: platformdirs
|
|
75
|
-
Requires-Dist: desktop-ai-core>=0.
|
|
75
|
+
Requires-Dist: desktop-ai-core>=0.3.1
|
|
76
76
|
Requires-Dist: onnxruntime
|
|
77
|
+
Requires-Dist: pynput
|
|
78
|
+
Requires-Dist: pystray
|
|
77
79
|
Provides-Extra: keyboard
|
|
78
80
|
Requires-Dist: pynput; extra == "keyboard"
|
|
79
81
|
Provides-Extra: whisper
|
|
@@ -83,8 +85,7 @@ Requires-Dist: pywhispercpp; extra == "whisper-futo"
|
|
|
83
85
|
Provides-Extra: vosk
|
|
84
86
|
Requires-Dist: vosk; extra == "vosk"
|
|
85
87
|
Provides-Extra: app
|
|
86
|
-
Requires-Dist:
|
|
87
|
-
Requires-Dist: PyGObject; extra == "app"
|
|
88
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
|
|
88
89
|
Provides-Extra: openai
|
|
89
90
|
Requires-Dist: openai<3,>=2.37.0; extra == "openai"
|
|
90
91
|
Requires-Dist: soundfile; extra == "openai"
|
|
@@ -93,13 +94,11 @@ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
|
|
|
93
94
|
Requires-Dist: soundfile; extra == "groq"
|
|
94
95
|
Provides-Extra: vad
|
|
95
96
|
Provides-Extra: all
|
|
96
|
-
Requires-Dist: pynput; extra == "all"
|
|
97
97
|
Requires-Dist: faster-whisper; extra == "all"
|
|
98
|
-
Requires-Dist: pywhispercpp; extra == "all"
|
|
99
98
|
Requires-Dist: openai<3,>=2.37.0; extra == "all"
|
|
100
99
|
Requires-Dist: soundfile; extra == "all"
|
|
101
100
|
Requires-Dist: vosk; extra == "all"
|
|
102
|
-
Requires-Dist:
|
|
101
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
|
|
103
102
|
Dynamic: license-file
|
|
104
103
|
|
|
105
104
|
[](https://pypi.org/project/scribe-cli)
|
|
@@ -129,13 +128,29 @@ cloud-based APIs, batch and streaming workflows.
|
|
|
129
128
|
|
|
130
129
|
## Install
|
|
131
130
|
|
|
131
|
+
**Linux / macOS:**
|
|
132
|
+
|
|
132
133
|
```bash
|
|
133
134
|
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
134
135
|
pip install scribe-cli[all]
|
|
135
136
|
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
136
137
|
```
|
|
137
138
|
|
|
138
|
-
|
|
139
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
140
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
141
|
+
|
|
142
|
+
```powershell
|
|
143
|
+
py -m venv .venv
|
|
144
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
145
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
146
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
The tray app and keyboard typing work out of the box on Windows — `pynput`
|
|
150
|
+
and `pystray` are regular dependencies, and there is nothing to create by
|
|
151
|
+
hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
|
|
152
|
+
for the full Windows walkthrough, and the documentation below for setting up
|
|
153
|
+
keyboard input on Ubuntu Wayland.
|
|
139
154
|
|
|
140
155
|
|
|
141
156
|
## Usage
|
|
@@ -217,7 +232,8 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
217
232
|
- [Installation & dependencies](docs/installation.md) — PortAudio,
|
|
218
233
|
extras, Ubuntu / GNOME tray libs.
|
|
219
234
|
- [Backends in detail](docs/backends.md) — model lists, when to pick
|
|
220
|
-
which, the realtime model.
|
|
235
|
+
which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
|
|
236
|
+
(Balanced / Patient profiles).
|
|
221
237
|
- [Output modes & typer backends](docs/output.md) — keystroke vs
|
|
222
238
|
clipboard, Wayland / `eitype`, `--type-direct`.
|
|
223
239
|
- [System tray & global hotkeys](docs/tray.md) — menu tree, icon
|
|
@@ -236,8 +252,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
236
252
|
|
|
237
253
|
## Compatibility
|
|
238
254
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
255
|
+
| OS | Status |
|
|
256
|
+
|--------------------|---------------------------------------------------------------------|
|
|
257
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
258
|
+
| macOS | Works. |
|
|
259
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
|
|
260
|
+
|
|
261
|
+
Wayland keystroke injection is convoluted but [solved](docs/output.md).
|
|
262
|
+
For dependencies of individual subsystems, check `pynput` (keyboard) and
|
|
263
|
+
`pystray` (tray icon).
|
|
264
|
+
|
|
265
|
+
**Windows notes:**
|
|
266
|
+
|
|
267
|
+
- The tray icon is hidden under the taskbar overflow arrow (`^`) by
|
|
268
|
+
default. Pin it via *Settings → Personalization → Taskbar → Other
|
|
269
|
+
system tray icons*.
|
|
270
|
+
- A **single click** on the tray icon fires the default action (Record).
|
|
271
|
+
This is a free bonus of pystray's Win32 backend; on Ubuntu the
|
|
272
|
+
AppIndicator backend only opens the menu (a backend limitation, not a
|
|
273
|
+
bug).
|
|
274
|
+
- If recording fails, allow mic access under *Settings → Privacy &
|
|
275
|
+
security → Microphone → "Let desktop apps access your microphone"*.
|
|
@@ -25,13 +25,29 @@ cloud-based APIs, batch and streaming workflows.
|
|
|
25
25
|
|
|
26
26
|
## Install
|
|
27
27
|
|
|
28
|
+
**Linux / macOS:**
|
|
29
|
+
|
|
28
30
|
```bash
|
|
29
31
|
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
30
32
|
pip install scribe-cli[all]
|
|
31
33
|
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
32
34
|
```
|
|
33
35
|
|
|
34
|
-
|
|
36
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
37
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
38
|
+
|
|
39
|
+
```powershell
|
|
40
|
+
py -m venv .venv
|
|
41
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
42
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
43
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
The tray app and keyboard typing work out of the box on Windows — `pynput`
|
|
47
|
+
and `pystray` are regular dependencies, and there is nothing to create by
|
|
48
|
+
hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
|
|
49
|
+
for the full Windows walkthrough, and the documentation below for setting up
|
|
50
|
+
keyboard input on Ubuntu Wayland.
|
|
35
51
|
|
|
36
52
|
|
|
37
53
|
## Usage
|
|
@@ -113,7 +129,8 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
113
129
|
- [Installation & dependencies](docs/installation.md) — PortAudio,
|
|
114
130
|
extras, Ubuntu / GNOME tray libs.
|
|
115
131
|
- [Backends in detail](docs/backends.md) — model lists, when to pick
|
|
116
|
-
which, the realtime model.
|
|
132
|
+
which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
|
|
133
|
+
(Balanced / Patient profiles).
|
|
117
134
|
- [Output modes & typer backends](docs/output.md) — keystroke vs
|
|
118
135
|
clipboard, Wayland / `eitype`, `--type-direct`.
|
|
119
136
|
- [System tray & global hotkeys](docs/tray.md) — menu tree, icon
|
|
@@ -132,8 +149,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
|
|
|
132
149
|
|
|
133
150
|
## Compatibility
|
|
134
151
|
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
152
|
+
| OS | Status |
|
|
153
|
+
|--------------------|---------------------------------------------------------------------|
|
|
154
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
155
|
+
| macOS | Works. |
|
|
156
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
|
|
157
|
+
|
|
158
|
+
Wayland keystroke injection is convoluted but [solved](docs/output.md).
|
|
159
|
+
For dependencies of individual subsystems, check `pynput` (keyboard) and
|
|
160
|
+
`pystray` (tray icon).
|
|
161
|
+
|
|
162
|
+
**Windows notes:**
|
|
163
|
+
|
|
164
|
+
- The tray icon is hidden under the taskbar overflow arrow (`^`) by
|
|
165
|
+
default. Pin it via *Settings → Personalization → Taskbar → Other
|
|
166
|
+
system tray icons*.
|
|
167
|
+
- A **single click** on the tray icon fires the default action (Record).
|
|
168
|
+
This is a free bonus of pystray's Win32 backend; on Ubuntu the
|
|
169
|
+
AppIndicator backend only opens the menu (a backend limitation, not a
|
|
170
|
+
bug).
|
|
171
|
+
- If recording fails, allow mic access under *Settings → Privacy &
|
|
172
|
+
security → Microphone → "Let desktop apps access your microphone"*.
|
|
@@ -164,6 +164,31 @@ the one place a separate "dictionary" really exists — everywhere else
|
|
|
164
164
|
`--words` is just a convenience to keep your word list out of the
|
|
165
165
|
prompt string in the CLI.
|
|
166
166
|
|
|
167
|
+
### Prompt style biases output style
|
|
168
|
+
|
|
169
|
+
Whisper mirrors the *style* of whatever prompt it receives. A
|
|
170
|
+
prompt like `"Tierney Comet"` (a bare wordlist) biases the model
|
|
171
|
+
toward unpunctuated, list-style output — sentences come out without
|
|
172
|
+
periods. A prompt like `"Tierney, Comet."` (or any prose ending in a
|
|
173
|
+
period) biases it toward punctuated output. Two practical
|
|
174
|
+
consequences:
|
|
175
|
+
|
|
176
|
+
- **`--prompt` is yours to control.** If your `prompt.txt` ends with
|
|
177
|
+
a period and looks like a sentence, your transcripts will be
|
|
178
|
+
punctuated. If it ends with a bare keyword, they probably won't.
|
|
179
|
+
This effect is most visible in **Stream mode**, where Whisper sees
|
|
180
|
+
short audio chunks and leans more heavily on the prompt for style
|
|
181
|
+
cues.
|
|
182
|
+
- **`--words` is auto-formatted by scribe.** For backends that fold
|
|
183
|
+
words into the prompt (`whisper-futo`, `openai`, `groq`), scribe
|
|
184
|
+
renders the word list as `"word1, word2, …, wordN."` — comma-
|
|
185
|
+
separated with a single terminal period — so your `words.txt` can
|
|
186
|
+
stay a bare list with no special formatting and the bias still
|
|
187
|
+
comes out punctuated. Stray punctuation on individual entries is
|
|
188
|
+
stripped first, so `words.txt` content is normalised regardless of
|
|
189
|
+
layout. On `whisper` (faster-whisper, local), words go to the
|
|
190
|
+
dedicated `hotwords` channel and bypass the prompt entirely.
|
|
191
|
+
|
|
167
192
|
Both flags read from the corresponding `*-file` argument when present.
|
|
168
193
|
Inline + file inputs are combined.
|
|
169
194
|
|
|
@@ -250,6 +275,14 @@ Once the buffer has grown to at least `--stream-chunk-min` (default
|
|
|
250
275
|
(default 10 s) regardless of silence, to cap latency. The session
|
|
251
276
|
continues until you stop it manually.
|
|
252
277
|
|
|
278
|
+
The first chunk uses a higher floor (`--stream-first-chunk-min`,
|
|
279
|
+
default 3 s) so the bootstrap chunk has enough audio to seed the
|
|
280
|
+
rolling prompt for the rest. Auto-disabled when
|
|
281
|
+
`--stream-context-length 0` (Patient). If you stop talking before
|
|
282
|
+
the floor is reached, a pause past `--stream-context-reset-silence ×
|
|
283
|
+
--stream-chunk-silence-break` (default 1.8 s) flushes the buffer
|
|
284
|
+
anyway — your utterance is never stranded.
|
|
285
|
+
|
|
253
286
|
### Does pseudo-streaming change the API cost?
|
|
254
287
|
|
|
255
288
|
For cloud backends, going from one big transcription to N chunked
|
|
@@ -321,3 +354,40 @@ arbitrarily long pauses.
|
|
|
321
354
|
|
|
322
355
|
Short pauses (mid-sentence punctuation) keep the context; the cut at
|
|
323
356
|
the start of every new recording also clears it.
|
|
357
|
+
|
|
358
|
+
### Streaming recipes — two profiles
|
|
359
|
+
|
|
360
|
+
The defaults stream phrases in as you talk; the Patient profile waits
|
|
361
|
+
for natural pauses and transcribes one utterance at a time. They make
|
|
362
|
+
opposite trade-offs around the same fundamental tension: short audio
|
|
363
|
+
windows give Whisper less to work with, so cross-chunk *context*
|
|
364
|
+
matters more in Balanced, less in Patient.
|
|
365
|
+
|
|
366
|
+
#### Balanced (default)
|
|
367
|
+
|
|
368
|
+
```bash
|
|
369
|
+
scribe --stream
|
|
370
|
+
```
|
|
371
|
+
|
|
372
|
+
Phrases commit every ~10 s or on a 0.6 s pause, with a 200-char
|
|
373
|
+
rolling prompt carrying earlier text forward as context for each new
|
|
374
|
+
chunk. Whisper sees short audio windows in isolation; the rolling
|
|
375
|
+
context partially compensates by telling the model what was just
|
|
376
|
+
said. Good live-feel, small per-chunk accuracy hit vs. Patient.
|
|
377
|
+
|
|
378
|
+
#### Patient (auto-clip)
|
|
379
|
+
|
|
380
|
+
```bash
|
|
381
|
+
scribe --stream \
|
|
382
|
+
--stream-chunk-min 0.5 \
|
|
383
|
+
--stream-chunk-max 300 \
|
|
384
|
+
--stream-chunk-silence-break 2 \
|
|
385
|
+
--stream-context-length 0
|
|
386
|
+
```
|
|
387
|
+
|
|
388
|
+
Each utterance is a complete self-contained sentence. scribe waits
|
|
389
|
+
for a 2 s pause, transcribes the whole thing at once, then waits for
|
|
390
|
+
the next one. No rolling context (`context-length 0`) because each
|
|
391
|
+
chunk is already a full utterance — there's nothing short to
|
|
392
|
+
compensate for. Highest per-chunk accuracy; no text appears until
|
|
393
|
+
you finish talking.
|
|
@@ -121,6 +121,7 @@ silence-chunking knobs; they have their own end-of-utterance signal.
|
|
|
121
121
|
| `--clip` | default | Transcribe the whole recording at end. Same as the tray's **Mode: Clip**. |
|
|
122
122
|
| `--stream-chunk-max SECS` | `10` | Maximum chunk duration in seconds. Force-cut fires at this threshold when no silence pause has been detected (default `10`). |
|
|
123
123
|
| `--stream-chunk-min SECS` | `1.5` | Minimum chunk size before a silence-cut is allowed (default `1.5`). Prevents very short clips that cause Whisper hallucinations. |
|
|
124
|
+
| `--stream-first-chunk-min SECS` | `3.0` | Minimum chunk size for the *first* chunk of a streaming thread (default `3.0`). Higher than `--stream-chunk-min` so the bootstrap chunk has enough audio for Whisper to produce a punctuated transcript whose tail seeds the rolling prompt for the rest. Applies on recording start and right after a context-reset silence. Inactive when `--stream-context-length 0`. Clamped to `≤ --stream-chunk-max`. Set equal to `--stream-chunk-min` to disable. |
|
|
124
125
|
| `--stream-chunk-silence-break SECS` | `0.6` | Silence duration that triggers a chunk cut (default `0.6`). Special value `0` enables Auto mode (best-silence-in-window at force-cut time). |
|
|
125
126
|
| `--stream-context-reset-silence X` | `3.0` | Multiplier of `--stream-chunk-silence-break` above which the rolling cross-chunk prompt context is discarded (default `3.0`, i.e. 1.8 s at default silence-break). Use `inf` to never reset. |
|
|
126
127
|
| `--clip-timeout SECS` | `120` | Auto-stop after this many seconds in Clip mode (default `120`). |
|
|
@@ -21,7 +21,10 @@ On macOS use Homebrew:
|
|
|
21
21
|
brew install portaudio
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
On Windows there are **no system packages to install**: `sounddevice`
|
|
25
|
+
bundles PortAudio in its wheel and the clipboard uses the native Windows
|
|
26
|
+
API, so neither `portaudio19-dev` nor `xclip` apply. See the
|
|
27
|
+
[Windows quickstart](#windows) below.
|
|
25
28
|
|
|
26
29
|
## Python package
|
|
27
30
|
|
|
@@ -50,14 +53,52 @@ the four backends and the tray UI:
|
|
|
50
53
|
| `[vosk]` | `vosk` | local Vosk backend (streaming) |
|
|
51
54
|
| `[openai]` | `openai`, `soundfile` | OpenAI cloud backend (incl. realtime) |
|
|
52
55
|
| `[groq]` | `openai`, `soundfile` | Groq cloud backend |
|
|
53
|
-
| `[keyboard]` | `pynput` |
|
|
54
|
-
| `[app]` | `
|
|
55
|
-
| `[all]` |
|
|
56
|
+
| `[keyboard]` | `pynput` | back-compat only — `pynput` is a base dep now |
|
|
57
|
+
| `[app]` | `PyGObject` (Linux only) | the Linux AppIndicator tray binding |
|
|
58
|
+
| `[all]` | every backend + Linux tray binding | one-shot setup |
|
|
59
|
+
|
|
60
|
+
> **`pynput` and `pystray` are base dependencies.** The default run uses
|
|
61
|
+
> the keyboard typer and the system-tray app, so both ship with the plain
|
|
62
|
+
> `pip install scribe-cli` — you do **not** need `[keyboard]` or `[app]`
|
|
63
|
+
> for the standard experience. `[app]` now only adds the Linux-only
|
|
64
|
+
> `PyGObject` AppIndicator binding (skipped automatically on Windows/macOS
|
|
65
|
+
> via a `sys_platform == 'linux'` marker, since it needs GTK and won't
|
|
66
|
+
> pip-install elsewhere).
|
|
56
67
|
|
|
57
68
|
You need at least one backend extra (or none if you only plan to use
|
|
58
69
|
cloud backends *and* already have the `openai` package). The `groq`
|
|
59
70
|
backend reuses the `openai` client, so `[openai]` covers both.
|
|
60
71
|
|
|
72
|
+
## Windows
|
|
73
|
+
|
|
74
|
+
Windows 11 is tested and working on Python 3.14 (64-bit). Every
|
|
75
|
+
dependency — `onnxruntime`, `faster-whisper`/`ctranslate2`,
|
|
76
|
+
`pystray`/`Pillow`, `pynput` — resolves a ready-made `win_amd64` wheel,
|
|
77
|
+
so there is no build toolchain to install and no need to downgrade
|
|
78
|
+
Python.
|
|
79
|
+
|
|
80
|
+
From PowerShell:
|
|
81
|
+
|
|
82
|
+
```powershell
|
|
83
|
+
py -m venv .venv
|
|
84
|
+
.\.venv\Scripts\Activate.ps1
|
|
85
|
+
# If activation is blocked by the execution policy, run once:
|
|
86
|
+
# Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
87
|
+
pip install -e .[whisper] # or [all], or a cloud backend like [openai]
|
|
88
|
+
scribe
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
That's the whole setup. There are **no system packages** to install
|
|
92
|
+
(`apt`/`portaudio19-dev`/`xclip` are Linux-only) and **nothing to create
|
|
93
|
+
by hand** — earlier builds needed a manual `C:\tmp` folder, which is no
|
|
94
|
+
longer the case.
|
|
95
|
+
|
|
96
|
+
- **Tray icon:** appears under the taskbar overflow arrow (`^`) by
|
|
97
|
+
default; pin it via *Settings → Personalization → Taskbar → Other
|
|
98
|
+
system tray icons*. A single click on the icon starts recording.
|
|
99
|
+
- **Microphone:** if recording fails, enable *Settings → Privacy &
|
|
100
|
+
security → Microphone → "Let desktop apps access your microphone"*.
|
|
101
|
+
|
|
61
102
|
## Ubuntu / GNOME tray dependencies
|
|
62
103
|
|
|
63
104
|
The tray icon needs system libraries for the AppIndicator stack:
|
|
@@ -26,9 +26,17 @@ recording / waiting, transcribing, and idle.
|
|
|
26
26
|
Transcription and API errors are surfaced as a pop-up dialog instead
|
|
27
27
|
of just crashing the tray.
|
|
28
28
|
|
|
29
|
-
The tray
|
|
30
|
-
|
|
31
|
-
|
|
29
|
+
The tray uses `pystray`, which is a **base dependency** — it ships with
|
|
30
|
+
the plain `pip install scribe-cli`, so the default tray works on Windows
|
|
31
|
+
and macOS with no extras. On Linux the AppIndicator backend additionally
|
|
32
|
+
needs `PyGObject` plus the appindicator system libs; install those via
|
|
33
|
+
`[app]` or `[all]` (see [installation.md](installation.md)).
|
|
34
|
+
|
|
35
|
+
**Click behaviour differs by platform.** On Windows, a single click on
|
|
36
|
+
the tray icon fires the default action (Record), because pystray's Win32
|
|
37
|
+
backend can activate the default menu item. On Ubuntu the AppIndicator
|
|
38
|
+
backend doesn't support a click-to-default action, so a click only opens
|
|
39
|
+
the menu. This is a pystray backend difference, not a scribe bug.
|
|
32
40
|
|
|
33
41
|
You can predefine which models appear in the tray menu with
|
|
34
42
|
`--vosk-models`, `--whisper-models`, and `--whisper-futo-models`:
|
|
@@ -106,8 +114,46 @@ disabled "<vendor> — <reason>" row so you know what's missing.
|
|
|
106
114
|
|
|
107
115
|
## Global hotkey integration
|
|
108
116
|
|
|
109
|
-
In
|
|
110
|
-
|
|
117
|
+
### In-app global hotkeys
|
|
118
|
+
|
|
119
|
+
In tray / app mode scribe runs a built-in global-hotkey listener (via
|
|
120
|
+
`pynput`, a base dependency). Defaults are **OS-specific** so they land on
|
|
121
|
+
chords that are normally free on each platform:
|
|
122
|
+
|
|
123
|
+
| Action | Linux (X11) | Windows / macOS |
|
|
124
|
+
|-----------------|---------------|---------------------|
|
|
125
|
+
| toggle record | `Super`+`C` | `Ctrl`+`Alt`+`C` |
|
|
126
|
+
| cancel | `Super`+`Z` | `Ctrl`+`Alt`+`Z` |
|
|
127
|
+
|
|
128
|
+
On Windows the Win-key chords are avoided on purpose: `Win`+`C` (Copilot)
|
|
129
|
+
and `Win`+`Z` (Snap Layouts) are claimed by the shell, and pynput's hook
|
|
130
|
+
doesn't suppress the key, so *both* the OS action and scribe's callback
|
|
131
|
+
would fire. On macOS `Cmd` chords collide with copy/undo. The combos are
|
|
132
|
+
**configurable** either way:
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
scribe --hotkey-record "<cmd>+c" --hotkey-cancel "<cmd>+z" # Super/Win/Cmd + C/Z
|
|
136
|
+
scribe --no-hotkeys # turn the listener off
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
(`<cmd>` is the Super / Windows / Command key in pynput syntax.)
|
|
140
|
+
|
|
141
|
+
**Platform support is uneven** — this is why scribe also keeps the Unix
|
|
142
|
+
signal mechanism below:
|
|
143
|
+
|
|
144
|
+
| Platform | In-app global hotkeys |
|
|
145
|
+
|-----------------|---------------------------------------------------------------------|
|
|
146
|
+
| Windows | Works out of the box (default `Ctrl`+`Alt`+`C/Z`). |
|
|
147
|
+
| Linux (X11) | Works out of the box (default `Super`+`C/Z`). |
|
|
148
|
+
| Linux (Wayland) | **Doesn't work** — the compositor blocks global key capture. Use the SIGUSR1/2 + custom-shortcut path below instead. |
|
|
149
|
+
| macOS | Needs Accessibility / Input-Monitoring permission (System Settings → Privacy & Security). |
|
|
150
|
+
|
|
151
|
+
The listener is best-effort: if the OS won't grant a global hook, scribe
|
|
152
|
+
logs a line and keeps running (tray + signals still work).
|
|
153
|
+
|
|
154
|
+
### Unix signals (Linux / macOS)
|
|
155
|
+
|
|
156
|
+
scribe writes its PID to a pidfile and listens for two signals:
|
|
111
157
|
|
|
112
158
|
- `SIGUSR1` — toggle recording (same as clicking Record / Stop).
|
|
113
159
|
- `SIGUSR2` — cancel an in-flight recording.
|
|
@@ -21,12 +21,17 @@ dependencies = [
|
|
|
21
21
|
"unidecode",
|
|
22
22
|
"termcolor",
|
|
23
23
|
"platformdirs",
|
|
24
|
-
"desktop-ai-core>=0.
|
|
24
|
+
"desktop-ai-core>=0.3.1", # >=0.3.1 carries the portable-tempdir pidfile fix (no more /tmp on Windows)
|
|
25
25
|
# Runs the bundled silero VAD ONNX model (~2 MB shipped in scribe_data).
|
|
26
26
|
# In base deps so silero is available out of the box — see scribe/audio.py.
|
|
27
27
|
# `faster-whisper` already pulls it transitively, so installing with
|
|
28
28
|
# [whisper] is free; standalone adds ~57 MB which is trivial for an STT tool.
|
|
29
29
|
"onnxruntime",
|
|
30
|
+
# Default run uses the keyboard typer (needs pynput) and the system-tray app
|
|
31
|
+
# (needs pystray). Both are small, have wheels on all platforms, and back the
|
|
32
|
+
# README's headline zero-config experience, so they live in base deps.
|
|
33
|
+
"pynput",
|
|
34
|
+
"pystray",
|
|
30
35
|
]
|
|
31
36
|
|
|
32
37
|
classifiers = [
|
|
@@ -82,17 +87,23 @@ keywords = [
|
|
|
82
87
|
]
|
|
83
88
|
|
|
84
89
|
[project.optional-dependencies]
|
|
85
|
-
keyboard = ["pynput"]
|
|
90
|
+
keyboard = ["pynput"] # kept for back-compat; pynput is a base dep now
|
|
86
91
|
whisper = ["faster-whisper"]
|
|
87
|
-
whisper-futo = ["pywhispercpp"]
|
|
92
|
+
whisper-futo = ["pywhispercpp"] # CPU wheels exist but have known Windows DLL
|
|
93
|
+
# issues; keep strictly opt-in (not in [all])
|
|
88
94
|
vosk = ["vosk"]
|
|
89
|
-
|
|
95
|
+
# pystray is a base dep now; [app] only adds the Linux-only AppIndicator binding.
|
|
96
|
+
# PyGObject needs GTK system libraries and does not pip-install on Windows/macOS,
|
|
97
|
+
# so the PEP 508 marker skips it everywhere except Linux.
|
|
98
|
+
app = ["PyGObject; sys_platform == 'linux'"]
|
|
90
99
|
openai = ["openai>=2.37.0,<3", "soundfile"]
|
|
91
100
|
groq = ["openai>=2.37.0,<3", "soundfile"]
|
|
92
101
|
# [vad] is now a no-op alias kept for back-compat (`pip install scribe-cli[vad]`
|
|
93
102
|
# was the documented install before onnxruntime moved into base deps).
|
|
94
103
|
vad = []
|
|
95
|
-
|
|
104
|
+
# pynput + pystray are base deps; pywhispercpp intentionally omitted due to
|
|
105
|
+
# Windows build/DLL friction (install via scribe-cli[whisper-futo] if wanted).
|
|
106
|
+
all = ["faster-whisper", "openai>=2.37.0,<3", "soundfile", "vosk", "PyGObject; sys_platform == 'linux'"]
|
|
96
107
|
|
|
97
108
|
|
|
98
109
|
[tool.setuptools]
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '1.
|
|
22
|
-
__version_tuple__ = version_tuple = (1,
|
|
21
|
+
__version__ = version = '1.1.1'
|
|
22
|
+
__version_tuple__ = version_tuple = (1, 1, 1)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g7a2dcc7f5'
|