scribe-cli 1.1.0__tar.gz → 1.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/PKG-INFO +44 -13
  2. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/README.md +38 -6
  3. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/backends.md +7 -14
  4. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/installation.md +45 -4
  5. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/tray.md +51 -5
  6. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/pyproject.toml +16 -5
  7. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/_version.py +3 -3
  8. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/app.py +92 -0
  9. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/models.py +26 -0
  10. scribe_cli-1.1.1/scribe/typers/base.py +63 -0
  11. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/typers/eitype.py +13 -25
  12. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/typers/pynput.py +7 -11
  13. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/typers/wtype.py +13 -23
  14. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/PKG-INFO +44 -13
  15. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/SOURCES.txt +1 -0
  16. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/requires.txt +8 -5
  17. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_pseudo_streaming.py +58 -0
  18. scribe_cli-1.1.1/tests/test_typers_ascii_fallback.py +79 -0
  19. scribe_cli-1.1.0/scribe/typers/base.py +0 -18
  20. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/.github/FUNDING.yml +0 -0
  21. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/.github/workflows/pypi.yml +0 -0
  22. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/.gitignore +0 -0
  23. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/LICENSE +0 -0
  24. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/app-tray-menu.png +0 -0
  25. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/cli.md +0 -0
  26. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/desktop-install.md +0 -0
  27. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/output.md +0 -0
  28. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/docs/roadmap-libei.md +0 -0
  29. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/icon.xcf +0 -0
  30. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/__init__.py +0 -0
  31. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/audio.py +0 -0
  32. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/__init__.py +0 -0
  33. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/groq.py +0 -0
  34. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/openai_api.py +0 -0
  35. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/openai_realtime.py +0 -0
  36. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/vosk.py +0 -0
  37. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/whisper.py +0 -0
  38. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/backends/whisper_futo.py +0 -0
  39. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/dialog.py +0 -0
  40. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/install_desktop.py +0 -0
  41. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/keyboard.py +0 -0
  42. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/menu.py +0 -0
  43. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/models.toml +0 -0
  44. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/output.py +0 -0
  45. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/saverecording.py +0 -0
  46. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/session.py +0 -0
  47. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/testpynput.py +0 -0
  48. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/typers/__init__.py +0 -0
  49. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/typers/ydotool.py +0 -0
  50. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe/util.py +0 -0
  51. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/dependency_links.txt +0 -0
  52. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/entry_points.txt +0 -0
  53. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_cli.egg-info/top_level.txt +0 -0
  54. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/__init__.py +0 -0
  55. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/share/icon.png +0 -0
  56. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/share/icon_recording.png +0 -0
  57. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/share/icon_writing.png +0 -0
  58. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/silero_vad.LICENSE +0 -0
  59. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/silero_vad.onnx +0 -0
  60. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scribe_data/templates/scribe.desktop +0 -0
  61. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scripts/bench_whisper_local.py +0 -0
  62. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/scripts/test_python_versions_install.sh +0 -0
  63. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/setup.cfg +0 -0
  64. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_backend_matrix.py +0 -0
  65. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_compose_prompt.py +0 -0
  66. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_debug_logging.py +0 -0
  67. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_openai_realtime_coalesce.py +0 -0
  68. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_output.py +0 -0
  69. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_output_file_picker.py +0 -0
  70. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_prompt_file_picker.py +0 -0
  71. {scribe_cli-1.1.0 → scribe_cli-1.1.1}/tests/test_whisper_futo.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: scribe-cli
3
- Version: 1.1.0
3
+ Version: 1.1.1
4
4
  Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
5
5
  Author-email: Mahé Perrette <mahe.perrette@gmail.com>
6
6
  License: MIT License
@@ -72,8 +72,10 @@ Requires-Dist: pyperclip
72
72
  Requires-Dist: unidecode
73
73
  Requires-Dist: termcolor
74
74
  Requires-Dist: platformdirs
75
- Requires-Dist: desktop-ai-core>=0.2.0
75
+ Requires-Dist: desktop-ai-core>=0.3.1
76
76
  Requires-Dist: onnxruntime
77
+ Requires-Dist: pynput
78
+ Requires-Dist: pystray
77
79
  Provides-Extra: keyboard
78
80
  Requires-Dist: pynput; extra == "keyboard"
79
81
  Provides-Extra: whisper
@@ -83,8 +85,7 @@ Requires-Dist: pywhispercpp; extra == "whisper-futo"
83
85
  Provides-Extra: vosk
84
86
  Requires-Dist: vosk; extra == "vosk"
85
87
  Provides-Extra: app
86
- Requires-Dist: pystray; extra == "app"
87
- Requires-Dist: PyGObject; extra == "app"
88
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
88
89
  Provides-Extra: openai
89
90
  Requires-Dist: openai<3,>=2.37.0; extra == "openai"
90
91
  Requires-Dist: soundfile; extra == "openai"
@@ -93,13 +94,11 @@ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
93
94
  Requires-Dist: soundfile; extra == "groq"
94
95
  Provides-Extra: vad
95
96
  Provides-Extra: all
96
- Requires-Dist: pynput; extra == "all"
97
97
  Requires-Dist: faster-whisper; extra == "all"
98
- Requires-Dist: pywhispercpp; extra == "all"
99
98
  Requires-Dist: openai<3,>=2.37.0; extra == "all"
100
99
  Requires-Dist: soundfile; extra == "all"
101
100
  Requires-Dist: vosk; extra == "all"
102
- Requires-Dist: pystray; extra == "all"
101
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
103
102
  Dynamic: license-file
104
103
 
105
104
  [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
@@ -129,13 +128,29 @@ cloud-based APIs, batch and streaming workflows.
129
128
 
130
129
  ## Install
131
130
 
131
+ **Linux / macOS:**
132
+
132
133
  ```bash
133
134
  sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
134
135
  pip install scribe-cli[all]
135
136
  export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
136
137
  ```
137
138
 
138
- See documentation below for setting up keyboard input on Ubuntu Wayland.
139
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
140
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
141
+
142
+ ```powershell
143
+ py -m venv .venv
144
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
145
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
146
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
147
+ ```
148
+
149
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
150
+ and `pystray` are regular dependencies, and there is nothing to create by
151
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
152
+ for the full Windows walkthrough, and the documentation below for setting up
153
+ keyboard input on Ubuntu Wayland.
139
154
 
140
155
 
141
156
  ## Usage
@@ -237,8 +252,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
237
252
 
238
253
  ## Compatibility
239
254
 
240
- Initially developed for Python 3 on Ubuntu 24.04 (GNOME + Wayland);
241
- works on macOS and Windows too. Wayland keystroke injection is
242
- convoluted but [solved](docs/output.md). For dependencies of
243
- individual subsystems, check `pynput` (keyboard) and `pystray` (tray
244
- icon).
255
+ | OS | Status |
256
+ |--------------------|---------------------------------------------------------------------|
257
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
258
+ | macOS | Works. |
259
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
260
+
261
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
262
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
263
+ `pystray` (tray icon).
264
+
265
+ **Windows notes:**
266
+
267
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
268
+ default. Pin it via *Settings → Personalization → Taskbar → Other
269
+ system tray icons*.
270
+ - A **single click** on the tray icon fires the default action (Record).
271
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
272
+ AppIndicator backend only opens the menu (a backend limitation, not a
273
+ bug).
274
+ - If recording fails, allow mic access under *Settings → Privacy &
275
+ security → Microphone → "Let desktop apps access your microphone"*.
@@ -25,13 +25,29 @@ cloud-based APIs, batch and streaming workflows.
25
25
 
26
26
  ## Install
27
27
 
28
+ **Linux / macOS:**
29
+
28
30
  ```bash
29
31
  sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
30
32
  pip install scribe-cli[all]
31
33
  export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
32
34
  ```
33
35
 
34
- See documentation below for setting up keyboard input on Ubuntu Wayland.
36
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
37
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
38
+
39
+ ```powershell
40
+ py -m venv .venv
41
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
42
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
43
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
44
+ ```
45
+
46
+ The tray app and keyboard typing work out of the box on Windows — `pynput`
47
+ and `pystray` are regular dependencies, and there is nothing to create by
48
+ hand (no `C:\tmp`). See [docs/installation.md](docs/installation.md#windows)
49
+ for the full Windows walkthrough, and the documentation below for setting up
50
+ keyboard input on Ubuntu Wayland.
35
51
 
36
52
 
37
53
  ## Usage
@@ -133,8 +149,24 @@ I personally use [OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe`
133
149
 
134
150
  ## Compatibility
135
151
 
136
- Initially developed for Python 3 on Ubuntu 24.04 (GNOME + Wayland);
137
- works on macOS and Windows too. Wayland keystroke injection is
138
- convoluted but [solved](docs/output.md). For dependencies of
139
- individual subsystems, check `pynput` (keyboard) and `pystray` (tray
140
- icon).
152
+ | OS | Status |
153
+ |--------------------|---------------------------------------------------------------------|
154
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
155
+ | macOS | Works. |
156
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Every dependency resolves a ready-made wheel — no toolchain or Python downgrade needed. |
157
+
158
+ Wayland keystroke injection is convoluted but [solved](docs/output.md).
159
+ For dependencies of individual subsystems, check `pynput` (keyboard) and
160
+ `pystray` (tray icon).
161
+
162
+ **Windows notes:**
163
+
164
+ - The tray icon is hidden under the taskbar overflow arrow (`^`) by
165
+ default. Pin it via *Settings → Personalization → Taskbar → Other
166
+ system tray icons*.
167
+ - A **single click** on the tray icon fires the default action (Record).
168
+ This is a free bonus of pystray's Win32 backend; on Ubuntu the
169
+ AppIndicator backend only opens the menu (a backend limitation, not a
170
+ bug).
171
+ - If recording fails, allow mic access under *Settings → Privacy &
172
+ security → Microphone → "Let desktop apps access your microphone"*.
@@ -275,20 +275,13 @@ Once the buffer has grown to at least `--stream-chunk-min` (default
275
275
  (default 10 s) regardless of silence, to cap latency. The session
276
276
  continues until you stop it manually.
277
277
 
278
- The **first** chunk of a streaming thread uses a different floor:
279
- `--stream-first-chunk-min` (default 3 s). The bootstrap chunk has no
280
- prior text to bias Whisper's punctuation/casing, so a longer audio
281
- window lets the model produce a properly-punctuated transcript whose
282
- tail then seeds the rolling prompt for every chunk after it.
283
- Subsequent chunks fall back to `--stream-chunk-min`. The override
284
- also re-engages right after a context-reset silence (i.e. when a long
285
- pause cleared the rolling tail — see *Cross-chunk prompt context*
286
- below). Set `--stream-first-chunk-min` equal to `--stream-chunk-min`
287
- to disable the override. It's automatically inactive when
288
- `--stream-context-length 0` (Patient profile), where there is no
289
- rolling context to bootstrap. Internally clamped to `≤
290
- --stream-chunk-max` so a misconfigured pair can't deadlock the
291
- chunker.
278
+ The first chunk uses a higher floor (`--stream-first-chunk-min`,
279
+ default 3 s) so the bootstrap chunk has enough audio to seed the
280
+ rolling prompt for the rest. Auto-disabled when
281
+ `--stream-context-length 0` (Patient). If you stop talking before
282
+ the floor is reached, a pause past `--stream-context-reset-silence ×
283
+ --stream-chunk-silence-break` (default 1.8 s) flushes the buffer
284
+ anyway — your utterance is never stranded.
292
285
 
293
286
  ### Does pseudo-streaming change the API cost?
294
287
 
@@ -21,7 +21,10 @@ On macOS use Homebrew:
21
21
  brew install portaudio
22
22
  ```
23
23
 
24
- (Windows ships everything needed via the wheels.)
24
+ On Windows there are **no system packages to install**: `sounddevice`
25
+ bundles PortAudio in its wheel and the clipboard uses the native Windows
26
+ API, so neither `portaudio19-dev` nor `xclip` apply. See the
27
+ [Windows quickstart](#windows) below.
25
28
 
26
29
  ## Python package
27
30
 
@@ -50,14 +53,52 @@ the four backends and the tray UI:
50
53
  | `[vosk]` | `vosk` | local Vosk backend (streaming) |
51
54
  | `[openai]` | `openai`, `soundfile` | OpenAI cloud backend (incl. realtime) |
52
55
  | `[groq]` | `openai`, `soundfile` | Groq cloud backend |
53
- | `[keyboard]` | `pynput` | the `pynput` typer (XTest/Quartz/WinAPI)|
54
- | `[app]` | `pystray`, `PyGObject` | system tray icon |
55
- | `[all]` | all of the above | one-shot setup |
56
+ | `[keyboard]` | `pynput` | back-compat only — `pynput` is a base dep now |
57
+ | `[app]` | `PyGObject` (Linux only) | the Linux AppIndicator tray binding |
58
+ | `[all]` | every backend + Linux tray binding | one-shot setup |
59
+
60
+ > **`pynput` and `pystray` are base dependencies.** The default run uses
61
+ > the keyboard typer and the system-tray app, so both ship with the plain
62
+ > `pip install scribe-cli` — you do **not** need `[keyboard]` or `[app]`
63
+ > for the standard experience. `[app]` now only adds the Linux-only
64
+ > `PyGObject` AppIndicator binding (skipped automatically on Windows/macOS
65
+ > via a `sys_platform == 'linux'` marker, since it needs GTK and won't
66
+ > pip-install elsewhere).
56
67
 
57
68
  You need at least one backend extra (or none if you only plan to use
58
69
  cloud backends *and* already have the `openai` package). The `groq`
59
70
  backend reuses the `openai` client, so `[openai]` covers both.
60
71
 
72
+ ## Windows
73
+
74
+ Windows 11 is tested and working on Python 3.14 (64-bit). Every
75
+ dependency — `onnxruntime`, `faster-whisper`/`ctranslate2`,
76
+ `pystray`/`Pillow`, `pynput` — resolves a ready-made `win_amd64` wheel,
77
+ so there is no build toolchain to install and no need to downgrade
78
+ Python.
79
+
80
+ From PowerShell:
81
+
82
+ ```powershell
83
+ py -m venv .venv
84
+ .\.venv\Scripts\Activate.ps1
85
+ # If activation is blocked by the execution policy, run once:
86
+ # Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
87
+ pip install -e .[whisper] # or [all], or a cloud backend like [openai]
88
+ scribe
89
+ ```
90
+
91
+ That's the whole setup. There are **no system packages** to install
92
+ (`apt`/`portaudio19-dev`/`xclip` are Linux-only) and **nothing to create
93
+ by hand** — earlier builds needed a manual `C:\tmp` folder, which is no
94
+ longer the case.
95
+
96
+ - **Tray icon:** appears under the taskbar overflow arrow (`^`) by
97
+ default; pin it via *Settings → Personalization → Taskbar → Other
98
+ system tray icons*. A single click on the icon starts recording.
99
+ - **Microphone:** if recording fails, enable *Settings → Privacy &
100
+ security → Microphone → "Let desktop apps access your microphone"*.
101
+
61
102
  ## Ubuntu / GNOME tray dependencies
62
103
 
63
104
  The tray icon needs system libraries for the AppIndicator stack:
@@ -26,9 +26,17 @@ recording / waiting, transcribing, and idle.
26
26
  Transcription and API errors are surfaced as a pop-up dialog instead
27
27
  of just crashing the tray.
28
28
 
29
- The tray requires `pystray` (and on Linux, `PyGObject` plus the
30
- appindicator system libs — see [installation.md](installation.md)).
31
- This is included with `pip install scribe-cli[all]` or `[app]`.
29
+ The tray uses `pystray`, which is a **base dependency** — it ships with
30
+ the plain `pip install scribe-cli`, so the default tray works on Windows
31
+ and macOS with no extras. On Linux the AppIndicator backend additionally
32
+ needs `PyGObject` plus the appindicator system libs; install those via
33
+ `[app]` or `[all]` (see [installation.md](installation.md)).
34
+
35
+ **Click behaviour differs by platform.** On Windows, a single click on
36
+ the tray icon fires the default action (Record), because pystray's Win32
37
+ backend can activate the default menu item. On Ubuntu the AppIndicator
38
+ backend doesn't support a click-to-default action, so a click only opens
39
+ the menu. This is a pystray backend difference, not a scribe bug.
32
40
 
33
41
  You can predefine which models appear in the tray menu with
34
42
  `--vosk-models`, `--whisper-models`, and `--whisper-futo-models`:
@@ -106,8 +114,46 @@ disabled "<vendor> — <reason>" row so you know what's missing.
106
114
 
107
115
  ## Global hotkey integration
108
116
 
109
- In tray / app mode scribe writes its PID to a pidfile and listens for
110
- two signals:
117
+ ### In-app global hotkeys
118
+
119
+ In tray / app mode scribe runs a built-in global-hotkey listener (via
120
+ `pynput`, a base dependency). Defaults are **OS-specific** so they land on
121
+ chords that are normally free on each platform:
122
+
123
+ | Action | Linux (X11) | Windows / macOS |
124
+ |-----------------|---------------|---------------------|
125
+ | toggle record | `Super`+`C` | `Ctrl`+`Alt`+`C` |
126
+ | cancel | `Super`+`Z` | `Ctrl`+`Alt`+`Z` |
127
+
128
+ On Windows the Win-key chords are avoided on purpose: `Win`+`C` (Copilot)
129
+ and `Win`+`Z` (Snap Layouts) are claimed by the shell, and pynput's hook
130
+ doesn't suppress the key, so *both* the OS action and scribe's callback
131
+ would fire. On macOS `Cmd` chords collide with copy/undo. The combos are
132
+ **configurable** either way:
133
+
134
+ ```bash
135
+ scribe --hotkey-record "<cmd>+c" --hotkey-cancel "<cmd>+z" # Super/Win/Cmd + C/Z
136
+ scribe --no-hotkeys # turn the listener off
137
+ ```
138
+
139
+ (`<cmd>` is the Super / Windows / Command key in pynput syntax.)
140
+
141
+ **Platform support is uneven** — this is why scribe also keeps the Unix
142
+ signal mechanism below:
143
+
144
+ | Platform | In-app global hotkeys |
145
+ |-----------------|---------------------------------------------------------------------|
146
+ | Windows | Works out of the box (default `Ctrl`+`Alt`+`C/Z`). |
147
+ | Linux (X11) | Works out of the box (default `Super`+`C/Z`). |
148
+ | Linux (Wayland) | **Doesn't work** — the compositor blocks global key capture. Use the SIGUSR1/2 + custom-shortcut path below instead. |
149
+ | macOS | Needs Accessibility / Input-Monitoring permission (System Settings → Privacy & Security). |
150
+
151
+ The listener is best-effort: if the OS won't grant a global hook, scribe
152
+ logs a line and keeps running (tray + signals still work).
153
+
154
+ ### Unix signals (Linux / macOS)
155
+
156
+ scribe writes its PID to a pidfile and listens for two signals:
111
157
 
112
158
  - `SIGUSR1` — toggle recording (same as clicking Record / Stop).
113
159
  - `SIGUSR2` — cancel an in-flight recording.
@@ -21,12 +21,17 @@ dependencies = [
21
21
  "unidecode",
22
22
  "termcolor",
23
23
  "platformdirs",
24
- "desktop-ai-core>=0.2.0",
24
+ "desktop-ai-core>=0.3.1", # >=0.3.1 carries the portable-tempdir pidfile fix (no more /tmp on Windows)
25
25
  # Runs the bundled silero VAD ONNX model (~2 MB shipped in scribe_data).
26
26
  # In base deps so silero is available out of the box — see scribe/audio.py.
27
27
  # `faster-whisper` already pulls it transitively, so installing with
28
28
  # [whisper] is free; standalone adds ~57 MB which is trivial for an STT tool.
29
29
  "onnxruntime",
30
+ # Default run uses the keyboard typer (needs pynput) and the system-tray app
31
+ # (needs pystray). Both are small, have wheels on all platforms, and back the
32
+ # README's headline zero-config experience, so they live in base deps.
33
+ "pynput",
34
+ "pystray",
30
35
  ]
31
36
 
32
37
  classifiers = [
@@ -82,17 +87,23 @@ keywords = [
82
87
  ]
83
88
 
84
89
  [project.optional-dependencies]
85
- keyboard = ["pynput"]
90
+ keyboard = ["pynput"] # kept for back-compat; pynput is a base dep now
86
91
  whisper = ["faster-whisper"]
87
- whisper-futo = ["pywhispercpp"]
92
+ whisper-futo = ["pywhispercpp"] # CPU wheels exist but have known Windows DLL
93
+ # issues; keep strictly opt-in (not in [all])
88
94
  vosk = ["vosk"]
89
- app = ["pystray", "PyGObject"]
95
+ # pystray is a base dep now; [app] only adds the Linux-only AppIndicator binding.
96
+ # PyGObject needs GTK system libraries and does not pip-install on Windows/macOS,
97
+ # so the PEP 508 marker skips it everywhere except Linux.
98
+ app = ["PyGObject; sys_platform == 'linux'"]
90
99
  openai = ["openai>=2.37.0,<3", "soundfile"]
91
100
  groq = ["openai>=2.37.0,<3", "soundfile"]
92
101
  # [vad] is now a no-op alias kept for back-compat (`pip install scribe-cli[vad]`
93
102
  # was the documented install before onnxruntime moved into base deps).
94
103
  vad = []
95
- all = ["pynput", "faster-whisper", "pywhispercpp", "openai>=2.37.0,<3", "soundfile", "vosk", "pystray"]
104
+ # pynput + pystray are base deps; pywhispercpp intentionally omitted due to
105
+ # Windows build/DLL friction (install via scribe-cli[whisper-futo] if wanted).
106
+ all = ["faster-whisper", "openai>=2.37.0,<3", "soundfile", "vosk", "PyGObject; sys_platform == 'linux'"]
96
107
 
97
108
 
98
109
  [tool.setuptools]
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '1.1.0'
22
- __version_tuple__ = version_tuple = (1, 1, 0)
21
+ __version__ = version = '1.1.1'
22
+ __version_tuple__ = version_tuple = (1, 1, 1)
23
23
 
24
- __commit_id__ = commit_id = 'g9b8b835fd'
24
+ __commit_id__ = commit_id = 'g7a2dcc7f5'
@@ -2,6 +2,7 @@ import os
2
2
  from pathlib import Path
3
3
  import tomllib
4
4
  import signal
5
+ import sys
5
6
  import argparse
6
7
  import platformdirs
7
8
  from scribe.audio import Microphone
@@ -599,6 +600,17 @@ def get_parser():
599
600
  help="Whisper models offered in the tray menu.")
600
601
  group.add_argument("--whisper-futo-models", nargs="*", default=whisper_futo_models,
601
602
  help="FUTO ACFT Whisper models offered in the tray menu.")
603
+ _hk_record, _hk_cancel = _default_hotkeys()
604
+ group.add_argument("--hotkey-record", default=_hk_record,
605
+ help="Global hotkey to toggle recording in tray mode "
606
+ "(pynput combo syntax, e.g. '<cmd>+c', '<ctrl>+<alt>+c'; "
607
+ "<cmd> is the Super/Windows/Command key). "
608
+ "Default: %(default)s (OS-specific).")
609
+ group.add_argument("--hotkey-cancel", default=_hk_cancel,
610
+ help="Global hotkey to cancel an in-flight recording in tray mode. "
611
+ "Default: %(default)s (OS-specific).")
612
+ group.add_argument("--no-hotkeys", action="store_false", dest="hotkeys",
613
+ help="Disable the in-app global hotkey listener (tray mode).")
602
614
 
603
615
  return parser
604
616
 
@@ -743,6 +755,82 @@ def start_recording(micro, session, o, callback=None, **greetings):
743
755
 
744
756
 
745
757
 
758
+ def _default_hotkeys():
759
+ """Return OS-appropriate (record, cancel) global-hotkey combos.
760
+
761
+ Super+C / Super+Z are typically free on Linux/X11, but poor defaults
762
+ elsewhere: on Windows the Win key combos collide with the shell (Win+C →
763
+ Copilot, Win+Z → Snap Layouts) and pynput's non-suppressing hook lets *both*
764
+ the OS action and our callback fire; on macOS <cmd> combos clash with
765
+ copy/undo. So default to Ctrl+Alt+C / Ctrl+Alt+Z off Linux, where those
766
+ chords are normally unbound. All values stay overridable via the CLI.
767
+ """
768
+ if sys.platform == "linux":
769
+ return "<cmd>+c", "<cmd>+z"
770
+ return "<ctrl>+<alt>+c", "<ctrl>+<alt>+z"
771
+
772
+
773
+ def _start_global_hotkeys(icon, app_state):
774
+ """Start a pynput global-hotkey listener bound to the configured combos.
775
+
776
+ Mirrors the SIGUSR1/SIGUSR2 handlers (toggle record / cancel). Platform
777
+ support is uneven and intentionally best-effort — the Unix signal path
778
+ remains the supported route where global capture is restricted:
779
+
780
+ * Windows: works (Win32 hooks). Avoid OS-reserved Win+key combos.
781
+ * Linux/X11: works.
782
+ * Linux/Wayland: the compositor blocks global key grabs, so the listener
783
+ won't fire — use SIGUSR1/2 bound to a desktop shortcut instead.
784
+ * macOS: needs Accessibility/Input-Monitoring permission.
785
+
786
+ Runs as a daemon thread so it never blocks shutdown. Returns the listener
787
+ (or None when disabled/unavailable) and never raises — a failure here
788
+ must not take down the tray.
789
+ """
790
+ o = app_state.o
791
+ if not getattr(o, "hotkeys", True):
792
+ return None
793
+
794
+ _rec_default, _cancel_default = _default_hotkeys()
795
+ record_combo = getattr(o, "hotkey_record", _rec_default)
796
+ cancel_combo = getattr(o, "hotkey_cancel", _cancel_default)
797
+
798
+ try:
799
+ from pynput import keyboard
800
+ except Exception as exc: # pragma: no cover - pynput is a base dep
801
+ print(colored(f"Global hotkeys disabled (pynput unavailable): {exc}", "light_red"))
802
+ return None
803
+
804
+ def _on_record():
805
+ app_state.cb_record(icon, None)
806
+
807
+ def _on_cancel():
808
+ if icon._session.busy:
809
+ app_state.cb_cancel(icon, None)
810
+
811
+ mapping = {}
812
+ if record_combo:
813
+ mapping[record_combo] = _on_record
814
+ if cancel_combo:
815
+ mapping[cancel_combo] = _on_cancel
816
+ if not mapping:
817
+ return None
818
+
819
+ try:
820
+ listener = keyboard.GlobalHotKeys(mapping)
821
+ listener.daemon = True
822
+ listener.start()
823
+ except Exception as exc:
824
+ print(colored(f"Global hotkeys unavailable ({exc}); "
825
+ "use the tray menu or SIGUSR1/2.", "light_red"))
826
+ return None
827
+
828
+ shown = " / ".join(c for c in (record_combo, cancel_combo) if c)
829
+ print(f"Global hotkeys: {colored(shown, 'light_blue', attrs=['bold'])} "
830
+ "(record / cancel)")
831
+ return listener
832
+
833
+
746
834
  def create_app(micro, app_state):
747
835
  """Construct the system-tray pystray Icon from the unified menu spec.
748
836
 
@@ -831,6 +919,10 @@ def create_app(micro, app_state):
831
919
  register_signal_toggle(signal.SIGUSR2,
832
920
  lambda: icon._session.busy and app_state.cb_cancel(icon, None))
833
921
 
922
+ # In-app global hotkeys (cross-platform, incl. Windows where SIGUSR* don't
923
+ # exist). Kept on the icon so the reference lives as long as the tray does.
924
+ icon._hotkey_listener = _start_global_hotkeys(icon, app_state)
925
+
834
926
  return icon
835
927
 
836
928
 
@@ -307,6 +307,32 @@ class AbstractTranscriber(STTBackend):
307
307
  f"Cut at silence after {elapsed:.2f}s "
308
308
  f"(silent {sil_dur:.2f}s)"
309
309
  )
310
+ # Long-pause escape hatch. When silence reaches the
311
+ # context-reset threshold and we still haven't committed,
312
+ # flush whatever is in the buffer as long as it clears
313
+ # the lower stream_chunk_min floor (the Whisper-
314
+ # hallucination protection). This catches short
315
+ # utterances stranded below stream_first_chunk_min: the
316
+ # user clearly stopped talking — don't lose what they
317
+ # said waiting for a bootstrap window that's never
318
+ # going to fill. Also clears the rolling context, since
319
+ # a pause this long is the same topic-shift signal that
320
+ # the speech-resumption path uses for its own reset.
321
+ reset_threshold = (self.stream_context_reset_silence
322
+ * silence_break)
323
+ if (not math.isinf(reset_threshold)
324
+ and sil_dur >= reset_threshold
325
+ and buffer_ms >= self.stream_chunk_min * 1000):
326
+ if self._streaming_context:
327
+ self.log(
328
+ f"Clearing chunk context after {sil_dur:.2f}s pause"
329
+ )
330
+ self.clear_streaming_context()
331
+ raise SilenceDetected(
332
+ f"Cut at long silence after {elapsed:.2f}s "
333
+ f"(silent {sil_dur:.2f}s, "
334
+ f"below first-chunk floor)"
335
+ )
310
336
  else:
311
337
  # Auto and Max never silence-cut. Auto defers the cut
312
338
  # decision to the force-cut below (which picks the best
@@ -0,0 +1,63 @@
1
+ import re
2
+ from typing import Callable, Protocol, Tuple, Type, runtime_checkable
3
+
4
+ import unidecode
5
+
6
+ # Split a string into maximal runs of ASCII plus each individual non-ASCII
7
+ # character, preserving order: "café là" -> ["caf", "é", " l", "à", ""]-ish.
8
+ _ASCII_RUN_OR_CHAR = re.compile(r"[\x00-\x7f]+|[^\x00-\x7f]")
9
+
10
+
11
+ def type_ascii_safe(
12
+ emit: Callable[[str], None],
13
+ text: str,
14
+ errors: Tuple[Type[BaseException], ...],
15
+ ) -> None:
16
+ """Type ``text`` via ``emit`` while degrading untypable characters to ASCII
17
+ **without re-emitting already-typed text**.
18
+
19
+ Keystroke typers (wtype / eitype / pynput) emit left-to-right and abort
20
+ mid-string when the active xkb layout can't produce a character — but the
21
+ prefix is already out by then. The naive recovery (retry the whole string
22
+ transliterated to ASCII) re-types that prefix, so a chunk like
23
+ "le message dicté" lands as "le message dict" + "le message dicte" — the
24
+ duplicated-prefix bug.
25
+
26
+ Instead, emit maximal ASCII runs in one call each (always layout-typeable)
27
+ and, for each individual non-ASCII character, try it raw and fall back to
28
+ its ASCII transliteration on ``errors``. Unicode survives on layouts that
29
+ support it; the rest degrades per-character with no duplication. Failures
30
+ on ASCII content are genuine (compositor / daemon) and propagate.
31
+ """
32
+ for token in _ASCII_RUN_OR_CHAR.findall(text):
33
+ if token.isascii():
34
+ # Always layout-typeable; a failure here is genuine (compositor /
35
+ # daemon down) — let it propagate.
36
+ emit(token)
37
+ else:
38
+ try:
39
+ emit(token)
40
+ except errors:
41
+ try:
42
+ emit(unidecode.unidecode(token))
43
+ except errors:
44
+ # Even the transliteration is unrenderable — skip this
45
+ # one char rather than abort the whole transcript.
46
+ pass
47
+
48
+
49
+ @runtime_checkable
50
+ class Typer(Protocol):
51
+ name: str
52
+
53
+ def compatible(self) -> bool:
54
+ """True iff the host OS / session could *in principle* run this backend
55
+ (ignores setup). False means the backend is structurally impossible
56
+ here — e.g. ydotool on macOS, wtype on GNOME. Used by the menu to
57
+ hide incompatible rows entirely. Distinct from ``available()``, which
58
+ further requires binaries / daemons / sockets to be set up."""
59
+ ...
60
+
61
+ def available(self) -> bool: ...
62
+ def type(self, text: str) -> None: ...
63
+ def paste(self) -> None: ...