scribe-cli 1.0.1__tar.gz → 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/PKG-INFO +46 -14
  2. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/README.md +40 -7
  3. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/backends.md +70 -0
  4. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/cli.md +1 -0
  5. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/installation.md +45 -4
  6. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/tray.md +51 -5
  7. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/pyproject.toml +16 -5
  8. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/_version.py +3 -3
  9. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/app.py +195 -22
  10. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/openai_api.py +1 -0
  11. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/whisper.py +5 -1
  12. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/whisper_futo.py +5 -0
  13. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/dialog.py +26 -0
  14. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/menu.py +126 -2
  15. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/models.py +94 -5
  16. scribe_cli-1.1.1/scribe/typers/base.py +63 -0
  17. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/eitype.py +13 -25
  18. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/pynput.py +7 -11
  19. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/wtype.py +13 -23
  20. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/PKG-INFO +46 -14
  21. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/SOURCES.txt +4 -0
  22. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/requires.txt +8 -5
  23. scribe_cli-1.1.1/tests/test_compose_prompt.py +153 -0
  24. scribe_cli-1.1.1/tests/test_debug_logging.py +191 -0
  25. scribe_cli-1.1.1/tests/test_prompt_file_picker.py +165 -0
  26. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_pseudo_streaming.py +161 -5
  27. scribe_cli-1.1.1/tests/test_typers_ascii_fallback.py +79 -0
  28. scribe_cli-1.0.1/scribe/typers/base.py +0 -18
  29. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.github/FUNDING.yml +0 -0
  30. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.github/workflows/pypi.yml +0 -0
  31. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/.gitignore +0 -0
  32. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/LICENSE +0 -0
  33. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/app-tray-menu.png +0 -0
  34. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/desktop-install.md +0 -0
  35. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/output.md +0 -0
  36. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/docs/roadmap-libei.md +0 -0
  37. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/icon.xcf +0 -0
  38. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/__init__.py +0 -0
  39. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/audio.py +0 -0
  40. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/__init__.py +0 -0
  41. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/groq.py +0 -0
  42. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/openai_realtime.py +0 -0
  43. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/backends/vosk.py +0 -0
  44. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/install_desktop.py +0 -0
  45. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/keyboard.py +0 -0
  46. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/models.toml +0 -0
  47. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/output.py +0 -0
  48. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/saverecording.py +0 -0
  49. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/session.py +0 -0
  50. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/testpynput.py +0 -0
  51. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/__init__.py +0 -0
  52. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/typers/ydotool.py +0 -0
  53. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe/util.py +0 -0
  54. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/dependency_links.txt +0 -0
  55. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/entry_points.txt +0 -0
  56. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_cli.egg-info/top_level.txt +0 -0
  57. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/__init__.py +0 -0
  58. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon.png +0 -0
  59. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon_recording.png +0 -0
  60. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/share/icon_writing.png +0 -0
  61. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/silero_vad.LICENSE +0 -0
  62. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/silero_vad.onnx +0 -0
  63. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scribe_data/templates/scribe.desktop +0 -0
  64. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scripts/bench_whisper_local.py +0 -0
  65. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/scripts/test_python_versions_install.sh +0 -0
  66. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/setup.cfg +0 -0
  67. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_backend_matrix.py +0 -0
  68. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_openai_realtime_coalesce.py +0 -0
  69. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_output.py +0 -0
  70. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_output_file_picker.py +0 -0
  71. {scribe_cli-1.0.1 → scribe_cli-1.1.1}/tests/test_whisper_futo.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scribe-cli
3
- Version: 1.0.1
3
+ Version: 1.1.1
4
4
  Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
5
5
  Author-email: Mahé Perrette <mahe.perrette@gmail.com>
6
6
  License: MIT License
@@ -72,8 +72,10 @@ Requires-Dist: pyperclip
72
72
  Requires-Dist: unidecode
73
73
  Requires-Dist: termcolor
74
74
  Requires-Dist: platformdirs
75
- Requires-Dist: desktop-ai-core>=0.2.0
75
+ Requires-Dist: desktop-ai-core>=0.3.1
76
76
  Requires-Dist: onnxruntime
77
+ Requires-Dist: pynput
78
+ Requires-Dist: pystray
77
79
  Provides-Extra: keyboard
78
80
  Requires-Dist: pynput; extra == "keyboard"
79
81
  Provides-Extra: whisper
@@ -83,8 +85,7 @@ Requires-Dist: pywhispercpp; extra == "whisper-futo"
83
85
  Provides-Extra: vosk
84
86
  Requires-Dist: vosk; extra == "vosk"
85
87
  Provides-Extra: app
86
- Requires-Dist: pystray; extra == "app"
87
- Requires-Dist: PyGObject; extra == "app"
88
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
88
89
  Provides-Extra: openai
89
90
  Requires-Dist: openai<3,>=2.37.0; extra == "openai"
90
91
  Requires-Dist: soundfile; extra == "openai"
@@ -93,13 +94,11 @@ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
93
94
  Requires-Dist: soundfile; extra == "groq"
94
95
  Provides-Extra: vad
95
96
  Provides-Extra: all
96
- Requires-Dist: pynput; extra == "all"
97
97
  Requires-Dist: faster-whisper; extra == "all"
98
- Requires-Dist: pywhispercpp; extra == "all"
99
98
  Requires-Dist: openai<3,>=2.37.0; extra == "all"
100
99
  Requires-Dist: soundfile; extra == "all"
101
100
  Requires-Dist: vosk; extra == "all"
102
- Requires-Dist: pystray; extra == "all"
101
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
103
102
  Dynamic: license-file
104
103
 
105
104
  [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
@@ -129,13 +128,29 @@ cloud-based APIs, batch and streaming workflows.
129
128
 
130
129
  ## Install
131
130
 
131
+ **Linux / macOS:**
132
+
132
133
  ```bash
133
134
  sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
134
135
  pip install scribe-cli[all]
135
136
  export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
136
137
  ```
137
138
 
138
- See documentation below for setting up keyboard input on Ubuntu Wayland.
139
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
140
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
141
+
142
+ ```powershell
143
+ py -m venv .venv
144
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
145
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
146
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
147
+ ```
148
+
149
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
150
+ and `pystray` are regular dependencies, and there is nothing to create by
151
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
152
+ for the full Windows walkthrough, and the documentation below for setting up
153
+ keyboard input on Ubuntu Wayland.
139
154
 
140
155
 
141
156
  ## Usage
@@ -217,7 +232,8 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
217
232
  - [Installation & dependencies](docs/installation.md) — PortAudio,
218
233
  extras, Ubuntu / GNOME tray libs.
219
234
  - [Backends in detail](docs/backends.md) — model lists, when to pick
220
- which, the realtime model.
235
+ which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
236
+ (Balanced / Patient profiles).
221
237
  - [Output modes & typer backends](docs/output.md) — keystroke vs
222
238
  clipboard, Wayland / `eitype`, `--type-direct`.
223
239
  - [System tray & global hotkeys](docs/tray.md) — menu tree, icon
@@ -236,8 +252,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
236
252
 
237
253
  ## Compatibility
238
254
 
239
- Initially developed for Python 3 on Ubuntu 24.04 (GNOME + Wayland);
240
- works on macOS and Windows too. Wayland keystroke injection is
241
- convoluted but [solved](docs/output.md). For dependencies of
242
- individual subsystems, check `pynput` (keyboard) and `pystray` (tray
243
- icon).
255
+ | OS | Status |
256
+ |--------------------|---------------------------------------------------------------------|
257
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
258
+ | macOS | Works. |
259
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
260
+
261
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
262
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
263
+ `pystray` (tray icon).
264
+
265
+ **Windows notes:**
266
+
267
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
268
+ default. Pin it via *Settings → Personalization → Taskbar → Other
269
+ system tray icons*.
270
+ - A **single click** on the tray icon fires the default action (Record).
271
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
272
+ AppIndicator backend only opens the menu (a backend limitation, not a
273
+ bug).
274
+ - If recording fails, allow mic access under *Settings → Privacy &
275
+ security → Microphone → "Let desktop apps access your microphone"*.
@@ -25,13 +25,29 @@ cloud-based APIs, batch and streaming workflows.
25
25
 
26
26
  ## Install
27
27
 
28
+ **Linux / macOS:**
29
+
28
30
  ```bash
29
31
  sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
30
32
  pip install scribe-cli[all]
31
33
  export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
32
34
  ```
33
35
 
34
- See documentation below for setting up keyboard input on Ubuntu Wayland.
36
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
37
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
38
+
39
+ ```powershell
40
+ py -m venv .venv
41
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
42
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
43
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
44
+ ```
45
+
46
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
47
+ and `pystray` are regular dependencies, and there is nothing to create by
48
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
49
+ for the full Windows walkthrough, and the documentation below for setting up
50
+ keyboard input on Ubuntu Wayland.
35
51
 
36
52
 
37
53
  ## Usage
@@ -113,7 +129,8 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
113
129
  - [Installation & dependencies](docs/installation.md) — PortAudio,
114
130
  extras, Ubuntu / GNOME tray libs.
115
131
  - [Backends in detail](docs/backends.md) — model lists, when to pick
116
- which, the realtime model.
132
+ which, the realtime model, [Streaming recipes](docs/backends.md#streaming-recipes--two-profiles)
133
+ (Balanced / Patient profiles).
117
134
  - [Output modes & typer backends](docs/output.md) — keystroke vs
118
135
  clipboard, Wayland / `eitype`, `--type-direct`.
119
136
  - [System tray & global hotkeys](docs/tray.md) — menu tree, icon
@@ -132,8 +149,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
132
149
 
133
150
  ## Compatibility
134
151
 
135
- Initially developed for Python 3 on Ubuntu 24.04 (GNOME + Wayland);
136
- works on macOS and Windows too. Wayland keystroke injection is
137
- convoluted but [solved](docs/output.md). For dependencies of
138
- individual subsystems, check `pynput` (keyboard) and `pystray` (tray
139
- icon).
152
+ | OS | Status |
153
+ |--------------------|---------------------------------------------------------------------|
154
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
155
+ | macOS | Works. |
156
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
157
+
158
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
159
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
160
+ `pystray` (tray icon).
161
+
162
+ **Windows notes:**
163
+
164
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
165
+ default. Pin it via *Settings → Personalization → Taskbar → Other
166
+ system tray icons*.
167
+ - A **single click** on the tray icon fires the default action (Record).
168
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
169
+ AppIndicator backend only opens the menu (a backend limitation, not a
170
+ bug).
171
+ - If recording fails, allow mic access under *Settings → Privacy &
172
+ security → Microphone → "Let desktop apps access your microphone"*.
@@ -164,6 +164,31 @@ the one place a separate "dictionary" really exists — everywhere else
164
164
  `--words` is just a convenience to keep your word list out of the
165
165
  prompt string in the CLI.
166
166
 
167
+ ### Prompt style biases output style
168
+
169
+ Whisper mirrors the *style* of whatever prompt it receives. A
170
+ prompt like `"Tierney Comet"` (a bare wordlist) biases the model
171
+ toward unpunctuated, list-style output — sentences come out without
172
+ periods. A prompt like `"Tierney, Comet."` (or any prose ending in a
173
+ period) biases it toward punctuated output. Two practical
174
+ consequences:
175
+
176
+ - **`--prompt` is yours to control.** If your `prompt.txt` ends with
177
+ a period and looks like a sentence, your transcripts will be
178
+ punctuated. If it ends with a bare keyword, they probably won't.
179
+ This effect is most visible in **Stream mode**, where Whisper sees
180
+ short audio chunks and leans more heavily on the prompt for style
181
+ cues.
182
+ - **`--words` is auto-formatted by scribe.** For backends that fold
183
+ words into the prompt (`whisper-futo`, `openai`, `groq`), scribe
184
+ renders the word list as `"word1, word2, …, wordN."` — comma-
185
+ separated with a single terminal period — so your `words.txt` can
186
+ stay a bare list with no special formatting and the bias still
187
+ comes out punctuated. Stray punctuation on individual entries is
188
+ stripped first, so `words.txt` content is normalised regardless of
189
+ layout. On `whisper` (faster-whisper, local), words go to the
190
+ dedicated `hotwords` channel and bypass the prompt entirely.
191
+
167
192
  Both flags read from the corresponding `*-file` argument when present.
168
193
  Inline + file inputs are combined.
169
194
 
@@ -250,6 +275,14 @@ Once the buffer has grown to at least `--stream-chunk-min` (default
250
275
  (default 10 s) regardless of silence, to cap latency. The session
251
276
  continues until you stop it manually.
252
277
 
278
+ The first chunk uses a higher floor (`--stream-first-chunk-min`,
279
+ default 3 s) so the bootstrap chunk has enough audio to seed the
280
+ rolling prompt for the rest. Auto-disabled when
281
+ `--stream-context-length 0` (Patient). If you stop talking before
282
+ the floor is reached, a pause past `--stream-context-reset-silence ×
283
+ --stream-chunk-silence-break` (default 1.8 s) flushes the buffer
284
+ anyway — your utterance is never stranded.
285
+
253
286
  ### Does pseudo-streaming change the API cost?
254
287
 
255
288
  For cloud backends, going from one big transcription to N chunked
@@ -321,3 +354,40 @@ arbitrarily long pauses.
321
354
 
322
355
  Short pauses (mid-sentence punctuation) keep the context; the cut at
323
356
  the start of every new recording also clears it.
357
+
358
+ ### Streaming recipes — two profiles
359
+
360
+ The defaults stream phrases in as you talk; the Patient profile waits
361
+ for natural pauses and transcribes one utterance at a time. They make
362
+ opposite trade-offs around the same fundamental tension: short audio
363
+ windows give Whisper less to work with, so cross-chunk *context*
364
+ matters more in Balanced, less in Patient.
365
+
366
+ #### Balanced (default)
367
+
368
+ ```bash
369
+ scribe --stream
370
+ ```
371
+
372
+ Phrases commit every ~10 s or on a 0.6 s pause, with a 200-char
373
+ rolling prompt carrying earlier text forward as context for each new
374
+ chunk. Whisper sees short audio windows in isolation; the rolling
375
+ context partially compensates by telling the model what was just
376
+ said. Good live-feel, small per-chunk accuracy hit vs. Patient.
377
+
378
+ #### Patient (auto-clip)
379
+
380
+ ```bash
381
+ scribe --stream \
382
+ --stream-chunk-min 0.5 \
383
+ --stream-chunk-max 300 \
384
+ --stream-chunk-silence-break 2 \
385
+ --stream-context-length 0
386
+ ```
387
+
388
+ Each utterance is a complete self-contained sentence. scribe waits
389
+ for a 2 s pause, transcribes the whole thing at once, then waits for
390
+ the next one. No rolling context (`context-length 0`) because each
391
+ chunk is already a full utterance — there's nothing short to
392
+ compensate for. Highest per-chunk accuracy; no text appears until
393
+ you finish talking.
@@ -121,6 +121,7 @@ silence-chunking knobs; they have their own end-of-utterance signal.
121
121
  | `--clip` | default | Transcribe the whole recording at end. Same as the tray's **Mode: Clip**. |
122
122
  | `--stream-chunk-max SECS` | `10` | Maximum chunk duration in seconds. Force-cut fires at this threshold when no silence pause has been detected (default `10`). |
123
123
  | `--stream-chunk-min SECS` | `1.5` | Minimum chunk size before a silence-cut is allowed (default `1.5`). Prevents very short clips that cause Whisper hallucinations. |
124
+ | `--stream-first-chunk-min SECS` | `3.0` | Minimum chunk size for the *first* chunk of a streaming thread (default `3.0`). Higher than `--stream-chunk-min` so the bootstrap chunk has enough audio for Whisper to produce a punctuated transcript whose tail seeds the rolling prompt for the rest. Applies on recording start and right after a context-reset silence. Inactive when `--stream-context-length 0`. Clamped to `≤ --stream-chunk-max`. Set equal to `--stream-chunk-min` to disable. |
124
125
  | `--stream-chunk-silence-break SECS` | `0.6` | Silence duration that triggers a chunk cut (default `0.6`). Special value `0` enables Auto mode (best-silence-in-window at force-cut time). |
125
126
  | `--stream-context-reset-silence X` | `3.0` | Multiplier of `--stream-chunk-silence-break` above which the rolling cross-chunk prompt context is discarded (default `3.0`, i.e. 1.8 s at default silence-break). Use `inf` to never reset. |
126
127
  | `--clip-timeout SECS` | `120` | Auto-stop after this many seconds in Clip mode (default `120`). |
@@ -21,7 +21,10 @@ On macOS use Homebrew:
21
21
  brew install portaudio
22
22
  ```
23
23
 
24
- (Windows ships everything needed via the wheels.)
24
+ On Windows there are **no system packages to install**: `sounddevice`
25
+ bundles PortAudio in its wheel and the clipboard uses the native Windows
26
+ API, so neither `portaudio19-dev` nor `xclip` apply. See the
27
+ [Windows quickstart](#windows) below.
25
28
 
26
29
  ## Python package
27
30
 
@@ -50,14 +53,52 @@ the four backends and the tray UI:
50
53
  | `[vosk]` | `vosk` | local Vosk backend (streaming) |
51
54
  | `[openai]` | `openai`, `soundfile` | OpenAI cloud backend (incl. realtime) |
52
55
  | `[groq]` | `openai`, `soundfile` | Groq cloud backend |
53
- | `[keyboard]` | `pynput` | the `pynput` typer (XTest/Quartz/WinAPI)|
54
- | `[app]` | `pystray`, `PyGObject` | system tray icon |
55
- | `[all]` | all of the above | one-shot setup |
56
+ | `[keyboard]` | `pynput` | back-compat only — `pynput` is a base dep now |
57
+ | `[app]` | `PyGObject` (Linux only) | the Linux AppIndicator tray binding |
58
+ | `[all]` | every backend + Linux tray binding | one-shot setup |
59
+
60
+ > **`pynput` and `pystray` are base dependencies.** The default run uses
61
+ > the keyboard typer and the system-tray app, so both ship with the plain
62
+ > `pip install scribe-cli` — you do **not** need `[keyboard]` or `[app]`
63
+ > for the standard experience. `[app]` now only adds the Linux-only
64
+ > `PyGObject` AppIndicator binding (skipped automatically on Windows/macOS
65
+ > via a `sys_platform == 'linux'` marker, since it needs GTK and won't
66
+ > pip-install elsewhere).
56
67
 
57
68
  You need at least one backend extra (or none if you only plan to use
58
69
  cloud backends *and* already have the `openai` package). The `groq`
59
70
  backend reuses the `openai` client, so `[openai]` covers both.
60
71
 
72
+ ## Windows
73
+
74
+ Windows 11 is tested and working on Python 3.14 (64-bit). Every
75
+ dependency — `onnxruntime`, `faster-whisper`/`ctranslate2`,
76
+ `pystray`/`Pillow`, `pynput` — resolves a ready-made `win_amd64` wheel,
77
+ so there is no build toolchain to install and no need to downgrade
78
+ Python.
79
+
80
+ From PowerShell:
81
+
82
+ ```powershell
83
+ py -m venv .venv
84
+ .\.venv\Scripts\Activate.ps1
85
+ # If activation is blocked by the execution policy, run once:
86
+ # Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
87
+ pip install -e .[whisper] # or [all], or a cloud backend like [openai]
88
+ scribe
89
+ ```
90
+
91
+ That's the whole setup. There are **no system packages** to install
92
+ (`apt`/`portaudio19-dev`/`xclip` are Linux-only) and **nothing to create
93
+ by hand** — earlier builds needed a manual `C:\tmp` folder, which is no
94
+ longer the case.
95
+
96
+ - **Tray icon:** appears under the taskbar overflow arrow (`^`) by
97
+ default; pin it via *Settings → Personalization → Taskbar → Other
98
+ system tray icons*. A single click on the icon starts recording.
99
+ - **Microphone:** if recording fails, enable *Settings → Privacy &
100
+ security → Microphone → "Let desktop apps access your microphone"*.
101
+
61
102
  ## Ubuntu / GNOME tray dependencies
62
103
 
63
104
  The tray icon needs system libraries for the AppIndicator stack:
@@ -26,9 +26,17 @@ recording / waiting, transcribing, and idle.
26
26
  Transcription and API errors are surfaced as a pop-up dialog instead
27
27
  of just crashing the tray.
28
28
 
29
- The tray requires `pystray` (and on Linux, `PyGObject` plus the
30
- appindicator system libs — see [installation.md](installation.md)).
31
- This is included with `pip install scribe-cli[all]` or `[app]`.
29
+ The tray uses `pystray`, which is a **base dependency** — it ships with
30
+ the plain `pip install scribe-cli`, so the default tray works on Windows
31
+ and macOS with no extras. On Linux the AppIndicator backend additionally
32
+ needs `PyGObject` plus the appindicator system libs; install those via
33
+ `[app]` or `[all]` (see [installation.md](installation.md)).
34
+
35
+ **Click behaviour differs by platform.** On Windows, a single click on
36
+ the tray icon fires the default action (Record), because pystray's Win32
37
+ backend can activate the default menu item. On Ubuntu the AppIndicator
38
+ backend doesn't support a click-to-default action, so a click only opens
39
+ the menu. This is a pystray backend difference, not a scribe bug.
32
40
 
33
41
  You can predefine which models appear in the tray menu with
34
42
  `--vosk-models`, `--whisper-models`, and `--whisper-futo-models`:
@@ -106,8 +114,46 @@ disabled "<vendor> — <reason>" row so you know what's missing.
106
114
 
107
115
  ## Global hotkey integration
108
116
 
109
- In tray / app mode scribe writes its PID to a pidfile and listens for
110
- two signals:
117
+ ### In-app global hotkeys
118
+
119
+ In tray / app mode scribe runs a built-in global-hotkey listener (via
120
+ `pynput`, a base dependency). Defaults are **OS-specific** so they land on
121
+ chords that are normally free on each platform:
122
+
123
+ | Action | Linux (X11) | Windows / macOS |
124
+ |-----------------|---------------|---------------------|
125
+ | toggle record | `Super`+`C` | `Ctrl`+`Alt`+`C` |
126
+ | cancel | `Super`+`Z` | `Ctrl`+`Alt`+`Z` |
127
+
128
+ On Windows the Win-key chords are avoided on purpose: `Win`+`C` (Copilot)
129
+ and `Win`+`Z` (Snap Layouts) are claimed by the shell, and pynput's hook
130
+ doesn't suppress the key, so *both* the OS action and scribe's callback
131
+ would fire. On macOS `Cmd` chords collide with copy/undo. The combos are
132
+ **configurable** either way:
133
+
134
+ ```bash
135
+ scribe --hotkey-record "<cmd>+c" --hotkey-cancel "<cmd>+z" # Super/Win/Cmd + C/Z
136
+ scribe --no-hotkeys # turn the listener off
137
+ ```
138
+
139
+ (`<cmd>` is the Super / Windows / Command key in pynput syntax.)
140
+
141
+ **Platform support is uneven** — this is why scribe also keeps the Unix
142
+ signal mechanism below:
143
+
144
+ | Platform | In-app global hotkeys |
145
+ |-----------------|---------------------------------------------------------------------|
146
+ | Windows | Works out of the box (default `Ctrl`+`Alt`+`C/Z`). |
147
+ | Linux (X11) | Works out of the box (default `Super`+`C/Z`). |
148
+ | Linux (Wayland) | **Doesn't work** — the compositor blocks global key capture. Use the SIGUSR1/2 + custom-shortcut path below instead. |
149
+ | macOS | Needs Accessibility / Input-Monitoring permission (System Settings → Privacy & Security). |
150
+
151
+ The listener is best-effort: if the OS won't grant a global hook, scribe
152
+ logs a line and keeps running (tray + signals still work).
153
+
154
+ ### Unix signals (Linux / macOS)
155
+
156
+ scribe writes its PID to a pidfile and listens for two signals:
111
157
 
112
158
  - `SIGUSR1` — toggle recording (same as clicking Record / Stop).
113
159
  - `SIGUSR2` — cancel an in-flight recording.
@@ -21,12 +21,17 @@ dependencies = [
21
21
  "unidecode",
22
22
  "termcolor",
23
23
  "platformdirs",
24
- "desktop-ai-core>=0.2.0",
24
+ "desktop-ai-core>=0.3.1", # >=0.3.1 carries the portable-tempdir pidfile fix (no more /tmp on Windows)
25
25
  # Runs the bundled silero VAD ONNX model (~2 MB shipped in scribe_data).
26
26
  # In base deps so silero is available out of the box — see scribe/audio.py.
27
27
  # `faster-whisper` already pulls it transitively, so installing with
28
28
  # [whisper] is free; standalone adds ~57 MB which is trivial for an STT tool.
29
29
  "onnxruntime",
30
+ # Default run uses the keyboard typer (needs pynput) and the system-tray app
31
+ # (needs pystray). Both are small, have wheels on all platforms, and back the
32
+ # README's headline zero-config experience, so they live in base deps.
33
+ "pynput",
34
+ "pystray",
30
35
  ]
31
36
 
32
37
  classifiers = [
@@ -82,17 +87,23 @@ keywords = [
82
87
  ]
83
88
 
84
89
  [project.optional-dependencies]
85
- keyboard = ["pynput"]
90
+ keyboard = ["pynput"] # kept for back-compat; pynput is a base dep now
86
91
  whisper = ["faster-whisper"]
87
- whisper-futo = ["pywhispercpp"]
92
+ whisper-futo = ["pywhispercpp"] # CPU wheels exist but have known Windows DLL
93
+ # issues; keep strictly opt-in (not in [all])
88
94
  vosk = ["vosk"]
89
- app = ["pystray", "PyGObject"]
95
+ # pystray is a base dep now; [app] only adds the Linux-only AppIndicator binding.
96
+ # PyGObject needs GTK system libraries and does not pip-install on Windows/macOS,
97
+ # so the PEP 508 marker skips it everywhere except Linux.
98
+ app = ["PyGObject; sys_platform == 'linux'"]
90
99
  openai = ["openai>=2.37.0,<3", "soundfile"]
91
100
  groq = ["openai>=2.37.0,<3", "soundfile"]
92
101
  # [vad] is now a no-op alias kept for back-compat (`pip install scribe-cli[vad]`
93
102
  # was the documented install before onnxruntime moved into base deps).
94
103
  vad = []
95
- all = ["pynput", "faster-whisper", "pywhispercpp", "openai>=2.37.0,<3", "soundfile", "vosk", "pystray"]
104
+ # pynput + pystray are base deps; pywhispercpp intentionally omitted due to
105
+ # Windows build/DLL friction (install via scribe-cli[whisper-futo] if wanted).
106
+ all = ["faster-whisper", "openai>=2.37.0,<3", "soundfile", "vosk", "PyGObject; sys_platform == 'linux'"]
96
107
 
97
108
 
98
109
  [tool.setuptools]
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '1.0.1'
22
- __version_tuple__ = version_tuple = (1, 0, 1)
21
+ __version__ = version = '1.1.1'
22
+ __version_tuple__ = version_tuple = (1, 1, 1)
23
23
 
24
- __commit_id__ = commit_id = 'g768aa6b57'
24
+ __commit_id__ = commit_id = 'g7a2dcc7f5'