scribe-cli 1.1.0__tar.gz → 1.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. scribe_cli-1.2.0/.github/workflows/docs.yml +34 -0
  2. scribe_cli-1.2.0/PKG-INFO +222 -0
  3. scribe_cli-1.2.0/README.md +114 -0
  4. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/backends.md +15 -19
  5. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/cli.md +3 -2
  6. scribe_cli-1.2.0/docs/index.md +58 -0
  7. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/installation.md +50 -5
  8. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/output.md +21 -2
  9. scribe_cli-1.2.0/docs/quickstart.md +80 -0
  10. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/roadmap-libei.md +2 -2
  11. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/tray.md +51 -5
  12. scribe_cli-1.2.0/mkdocs.yml +76 -0
  13. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/pyproject.toml +25 -5
  14. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/_version.py +3 -3
  15. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/app.py +111 -4
  16. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/keyboard.py +34 -0
  17. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/models.py +79 -6
  18. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/output.py +4 -0
  19. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/session.py +4 -0
  20. scribe_cli-1.2.0/scribe/typers/base.py +68 -0
  21. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/typers/eitype.py +22 -18
  22. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/typers/pynput.py +7 -11
  23. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/typers/wtype.py +23 -18
  24. scribe_cli-1.2.0/scribe_cli.egg-info/PKG-INFO +222 -0
  25. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_cli.egg-info/SOURCES.txt +7 -0
  26. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_cli.egg-info/requires.txt +13 -5
  27. scribe_cli-1.2.0/tests/test_clip_silence_trim.py +121 -0
  28. scribe_cli-1.2.0/tests/test_clipboard_backend.py +64 -0
  29. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_pseudo_streaming.py +58 -0
  30. scribe_cli-1.2.0/tests/test_typers_ascii_fallback.py +80 -0
  31. scribe_cli-1.1.0/PKG-INFO +0 -244
  32. scribe_cli-1.1.0/README.md +0 -140
  33. scribe_cli-1.1.0/scribe/typers/base.py +0 -18
  34. scribe_cli-1.1.0/scribe_cli.egg-info/PKG-INFO +0 -244
  35. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/.github/FUNDING.yml +0 -0
  36. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/.github/workflows/pypi.yml +0 -0
  37. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/.gitignore +0 -0
  38. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/LICENSE +0 -0
  39. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/app-tray-menu.png +0 -0
  40. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/docs/desktop-install.md +0 -0
  41. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/icon.xcf +0 -0
  42. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/__init__.py +0 -0
  43. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/audio.py +0 -0
  44. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/__init__.py +0 -0
  45. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/groq.py +0 -0
  46. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/openai_api.py +0 -0
  47. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/openai_realtime.py +0 -0
  48. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/vosk.py +0 -0
  49. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/whisper.py +0 -0
  50. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/backends/whisper_futo.py +0 -0
  51. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/dialog.py +0 -0
  52. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/install_desktop.py +0 -0
  53. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/menu.py +0 -0
  54. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/models.toml +0 -0
  55. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/saverecording.py +0 -0
  56. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/testpynput.py +0 -0
  57. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/typers/__init__.py +0 -0
  58. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/typers/ydotool.py +0 -0
  59. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe/util.py +0 -0
  60. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_cli.egg-info/dependency_links.txt +0 -0
  61. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_cli.egg-info/entry_points.txt +0 -0
  62. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_cli.egg-info/top_level.txt +0 -0
  63. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/__init__.py +0 -0
  64. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/share/icon.png +0 -0
  65. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/share/icon_recording.png +0 -0
  66. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/share/icon_writing.png +0 -0
  67. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/silero_vad.LICENSE +0 -0
  68. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/silero_vad.onnx +0 -0
  69. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scribe_data/templates/scribe.desktop +0 -0
  70. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scripts/bench_whisper_local.py +0 -0
  71. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/scripts/test_python_versions_install.sh +0 -0
  72. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/setup.cfg +0 -0
  73. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_backend_matrix.py +0 -0
  74. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_compose_prompt.py +0 -0
  75. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_debug_logging.py +0 -0
  76. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_openai_realtime_coalesce.py +0 -0
  77. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_output.py +0 -0
  78. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_output_file_picker.py +0 -0
  79. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_prompt_file_picker.py +0 -0
  80. {scribe_cli-1.1.0 → scribe_cli-1.2.0}/tests/test_whisper_futo.py +0 -0
@@ -0,0 +1,34 @@
1
+ name: docs
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ paths:
7
+ - "docs/**"
8
+ - "mkdocs.yml"
9
+ - "README.md"
10
+ - ".github/workflows/docs.yml"
11
+ workflow_dispatch:
12
+
13
+ permissions:
14
+ contents: write
15
+
16
+ jobs:
17
+ deploy:
18
+ runs-on: ubuntu-latest
19
+ steps:
20
+ - uses: actions/checkout@v4
21
+ with:
22
+ fetch-depth: 0
23
+ - uses: actions/setup-python@v5
24
+ with:
25
+ python-version: "3.x"
26
+ - name: Cache mkdocs-material assets
27
+ uses: actions/cache@v4
28
+ with:
29
+ key: mkdocs-material-${{ github.sha }}
30
+ path: .cache
31
+ restore-keys: |
32
+ mkdocs-material-
33
+ - run: pip install .[docs]
34
+ - run: mkdocs gh-deploy --force
@@ -0,0 +1,222 @@
1
+ Metadata-Version: 2.4
2
+ Name: scribe-cli
3
+ Version: 1.2.0
4
+ Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
5
+ Author-email: Mahé Perrette <mahe.perrette@gmail.com>
6
+ License: MIT License
7
+
8
+ Copyright (c) 2024 Mahé Perrette
9
+
10
+ Permission is hereby granted, free of charge, to any person obtaining a copy
11
+ of this software and associated documentation files (the "Software"), to deal
12
+ in the Software without restriction, including without limitation the rights
13
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
14
+ copies of the Software, and to permit persons to whom the Software is
15
+ furnished to do so, subject to the following conditions:
16
+
17
+ The above copyright notice and this permission notice shall be included in all
18
+ copies or substantial portions of the Software.
19
+
20
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
21
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
22
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
23
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
24
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
25
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
26
+ SOFTWARE.
27
+
28
+ ---
29
+
30
+ Note: This project relies on external packages that may have more restrictive
31
+ licenses. For example, the `pynput` package is licensed under LGPLv3, which
32
+ has different requirements compared to the MIT License. Please review the
33
+ licenses of all dependencies before using or distributing this software to
34
+ ensure compliance with their respective terms.
35
+ Project-URL: Homepage, https://github.com/perrette/scribe
36
+ Project-URL: Source, https://github.com/perrette/scribe
37
+ Project-URL: Issues, https://github.com/perrette/scribe/issues
38
+ Project-URL: Documentation, https://perrette.github.io/scribe/
39
+ Project-URL: Changelog, https://github.com/perrette/scribe/releases
40
+ Project-URL: Funding, https://github.com/sponsors/perrette
41
+ Keywords: speech-to-text,stt,transcription,dictation,voice-typing,voice-recognition,multilingual,realtime,streaming,cli,tray,vosk,whisper,faster-whisper,openai,groq,gpt-4o,linux,wayland,keyboard,clipboard,microphone,audio
42
+ Classifier: Development Status :: 5 - Production/Stable
43
+ Classifier: Intended Audience :: End Users/Desktop
44
+ Classifier: Intended Audience :: Developers
45
+ Classifier: License :: OSI Approved :: MIT License
46
+ Classifier: Programming Language :: Python :: 3.9
47
+ Classifier: Programming Language :: Python :: 3.10
48
+ Classifier: Programming Language :: Python :: 3.11
49
+ Classifier: Programming Language :: Python :: 3.12
50
+ Classifier: Programming Language :: Python :: 3.13
51
+ Classifier: Operating System :: OS Independent
52
+ Classifier: Environment :: Console
53
+ Classifier: Environment :: X11 Applications
54
+ Classifier: Environment :: MacOS X
55
+ Classifier: Environment :: Win32 (MS Windows)
56
+ Classifier: Natural Language :: English
57
+ Classifier: Natural Language :: French
58
+ Classifier: Natural Language :: German
59
+ Classifier: Natural Language :: Italian
60
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
61
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
62
+ Classifier: Topic :: Office/Business
63
+ Classifier: Topic :: Text Processing :: Linguistic
64
+ Classifier: Topic :: Utilities
65
+ Requires-Python: >=3.9
66
+ Description-Content-Type: text/markdown
67
+ License-File: LICENSE
68
+ Requires-Dist: numpy
69
+ Requires-Dist: sounddevice
70
+ Requires-Dist: tqdm
71
+ Requires-Dist: requests
72
+ Requires-Dist: pyperclip
73
+ Requires-Dist: unidecode
74
+ Requires-Dist: termcolor
75
+ Requires-Dist: platformdirs
76
+ Requires-Dist: desktop-ai-core>=0.3.1
77
+ Requires-Dist: onnxruntime
78
+ Requires-Dist: pynput
79
+ Requires-Dist: pystray
80
+ Provides-Extra: keyboard
81
+ Requires-Dist: pynput; extra == "keyboard"
82
+ Provides-Extra: whisper
83
+ Requires-Dist: faster-whisper; extra == "whisper"
84
+ Provides-Extra: whisper-futo
85
+ Requires-Dist: pywhispercpp; extra == "whisper-futo"
86
+ Provides-Extra: vosk
87
+ Requires-Dist: vosk; extra == "vosk"
88
+ Provides-Extra: app
89
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
90
+ Provides-Extra: openai
91
+ Requires-Dist: openai<3,>=2.37.0; extra == "openai"
92
+ Requires-Dist: soundfile; extra == "openai"
93
+ Provides-Extra: groq
94
+ Requires-Dist: openai<3,>=2.37.0; extra == "groq"
95
+ Requires-Dist: soundfile; extra == "groq"
96
+ Provides-Extra: vad
97
+ Provides-Extra: all
98
+ Requires-Dist: faster-whisper; extra == "all"
99
+ Requires-Dist: openai<3,>=2.37.0; extra == "all"
100
+ Requires-Dist: soundfile; extra == "all"
101
+ Requires-Dist: vosk; extra == "all"
102
+ Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
103
+ Provides-Extra: docs
104
+ Requires-Dist: mkdocs<2,>=1.5; extra == "docs"
105
+ Requires-Dist: mkdocs-material>=9.5; extra == "docs"
106
+ Requires-Dist: mkdocs-include-markdown-plugin>=6.0; extra == "docs"
107
+ Dynamic: license-file
108
+
109
+ [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
110
+ ![](https://img.shields.io/python/required-version-toml?tomlFilePath=https%3A%2F%2Fraw.githubusercontent.com%2Fperrette%2Fscribe%2Frefs%2Fheads%2Fmain%2Fpyproject.toml)
111
+ [![docs](https://img.shields.io/badge/docs-perrette.github.io%2Fscribe-blue)](https://perrette.github.io/scribe/)
112
+
113
+ # Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
114
+
115
+ Scribe is a speech-to-text CLI and tray app that pipes transcribed text
116
+ into the focused window. It supports local and cloud-based APIs, batch and
117
+ streaming workflows.
118
+
119
+ <!-- intro-start -->
120
+ - **Five backends, one interface.** Records from your mic and transcribes via
121
+ **Vosk** (local, streaming), **Whisper** (local, batch), **Whisper FUTO**
122
+ (local, batch — ACFT-tuned for short dictations), **OpenAI** (cloud, batch
123
+ *or* streaming), or **Groq** (cloud, batch).
124
+ - **Four ways to deliver the transcript.** Paste into the focused window
125
+ (default), copy to the clipboard, print to the terminal, or append to a file.
126
+ - **Tray or terminal.** Runs as a **system tray icon** with a single Record
127
+ button, or as an interactive **terminal TUI** — same menu in both.
128
+ - **Hotkey-friendly.** Hooks into your desktop's keyboard shortcuts via
129
+ `SIGUSR1` (toggle recording) and `SIGUSR2` (cancel), plus built-in global
130
+ hotkeys on X11 / Windows.
131
+ - **Cross-platform.** Tested on Ubuntu (X11 and Wayland), macOS, and Windows;
132
+ works under Termux for clipboard / terminal output.
133
+ <!-- intro-end -->
134
+
135
+ <img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
136
+
137
+ ## Installation
138
+
139
+ **Linux / macOS:**
140
+
141
+ ```bash
142
+ sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
143
+ pip install scribe-cli[all]
144
+ export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
145
+ ```
146
+
147
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
148
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
149
+
150
+ ```powershell
151
+ py -m venv .venv
152
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
153
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
154
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
155
+ ```
156
+
157
+ See the [installation page](https://perrette.github.io/scribe/installation/)
158
+ for the full per-platform walkthrough (extras, Ubuntu / GNOME tray libs, the
159
+ Windows quickstart).
160
+
161
+ ## Quickstart
162
+
163
+ ```bash
164
+ scribe
165
+ ```
166
+
167
+ This launches the system tray icon. Press Record, speak, press Stop — the
168
+ transcription lands in the focused window. Scribe picks the first backend whose
169
+ key / dependency is present, in order **`groq` → `openai` → `whisper-futo` →
170
+ `whisper` → `vosk`**. Override the defaults or drop the tray entirely:
171
+
172
+ ```bash
173
+ scribe --backend whisper --model small # local, no API key
174
+ scribe --frontend terminal # interactive TUI menu
175
+ scribe --mode clipboard # copy to clipboard, no keystroke
176
+ ```
177
+
178
+ See the [quickstart](https://perrette.github.io/scribe/quickstart/) for more.
179
+
180
+ ## Documentation
181
+
182
+ Full documentation lives at **<https://perrette.github.io/scribe/>**:
183
+
184
+ - [Installation & dependencies](https://perrette.github.io/scribe/installation/)
185
+ — PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
186
+ - [Quickstart](https://perrette.github.io/scribe/quickstart/) — your first
187
+ dictation.
188
+ - [Backends in detail](https://perrette.github.io/scribe/backends/) — model
189
+ lists, streaming recipes, vocabulary biasing, the realtime model.
190
+ - [Output modes & typer backends](https://perrette.github.io/scribe/output/) —
191
+ keystroke vs clipboard, Wayland / `eitype`, `--type-direct`.
192
+ - [System tray & global hotkeys](https://perrette.github.io/scribe/tray/) —
193
+ menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
194
+ - [Desktop entry & autostart (`scribe-install`)](https://perrette.github.io/scribe/desktop-install/)
195
+ — GNOME / KDE launcher integration.
196
+ - [Fine tuning & CLI reference](https://perrette.github.io/scribe/cli/) — every
197
+ `scribe --help` flag with examples.
198
+
199
+ ## From the same author
200
+
201
+ A few other open-source tools I maintain.
202
+
203
+ **Scientific writing & data**
204
+
205
+ - [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
206
+ - [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
207
+ - [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
208
+
209
+ **Speech to Text (dictate) and Text to Speech (read-aloud) tools**
210
+
211
+ - [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
212
+
213
+ ## Compatibility
214
+
215
+ | OS | Status |
216
+ |--------------------|---------------------------------------------------------------------|
217
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
218
+ | macOS | Works. |
219
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Dependencies resolve pre-built wheels; no C toolchain or Python downgrade needed. |
220
+
221
+ Wayland keystroke injection is convoluted but
222
+ [solved](https://perrette.github.io/scribe/output/).
@@ -0,0 +1,114 @@
1
+ [![pypi](https://img.shields.io/pypi/v/scribe-cli)](https://pypi.org/project/scribe-cli)
2
+ ![](https://img.shields.io/python/required-version-toml?tomlFilePath=https%3A%2F%2Fraw.githubusercontent.com%2Fperrette%2Fscribe%2Frefs%2Fheads%2Fmain%2Fpyproject.toml)
3
+ [![docs](https://img.shields.io/badge/docs-perrette.github.io%2Fscribe-blue)](https://perrette.github.io/scribe/)
4
+
5
+ # Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
6
+
7
+ Scribe is a speech-to-text CLI and tray app that pipes transcribed text
8
+ into the focused window. It supports local and cloud-based APIs, batch and
9
+ streaming workflows.
10
+
11
+ <!-- intro-start -->
12
+ - **Five backends, one interface.** Records from your mic and transcribes via
13
+ **Vosk** (local, streaming), **Whisper** (local, batch), **Whisper FUTO**
14
+ (local, batch — ACFT-tuned for short dictations), **OpenAI** (cloud, batch
15
+ *or* streaming), or **Groq** (cloud, batch).
16
+ - **Four ways to deliver the transcript.** Paste into the focused window
17
+ (default), copy to the clipboard, print to the terminal, or append to a file.
18
+ - **Tray or terminal.** Runs as a **system tray icon** with a single Record
19
+ button, or as an interactive **terminal TUI** — same menu in both.
20
+ - **Hotkey-friendly.** Hooks into your desktop's keyboard shortcuts via
21
+ `SIGUSR1` (toggle recording) and `SIGUSR2` (cancel), plus built-in global
22
+ hotkeys on X11 / Windows.
23
+ - **Cross-platform.** Tested on Ubuntu (X11 and Wayland), macOS, and Windows;
24
+ works under Termux for clipboard / terminal output.
25
+ <!-- intro-end -->
26
+
27
+ <img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
28
+
29
+ ## Installation
30
+
31
+ **Linux / macOS:**
32
+
33
+ ```bash
34
+ sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
35
+ pip install scribe-cli[all]
36
+ export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
37
+ ```
38
+
39
+ **Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
40
+ PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
41
+
42
+ ```powershell
43
+ py -m venv .venv
44
+ .\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
45
+ pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
46
+ $env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
47
+ ```
48
+
49
+ See the [installation page](https://perrette.github.io/scribe/installation/)
50
+ for the full per-platform walkthrough (extras, Ubuntu / GNOME tray libs, the
51
+ Windows quickstart).
52
+
53
+ ## Quickstart
54
+
55
+ ```bash
56
+ scribe
57
+ ```
58
+
59
+ This launches the system tray icon. Press Record, speak, press Stop — the
60
+ transcription lands in the focused window. Scribe picks the first backend whose
61
+ key / dependency is present, in order **`groq` → `openai` → `whisper-futo` →
62
+ `whisper` → `vosk`**. Override the defaults or drop the tray entirely:
63
+
64
+ ```bash
65
+ scribe --backend whisper --model small # local, no API key
66
+ scribe --frontend terminal # interactive TUI menu
67
+ scribe --mode clipboard # copy to clipboard, no keystroke
68
+ ```
69
+
70
+ See the [quickstart](https://perrette.github.io/scribe/quickstart/) for more.
71
+
72
+ ## Documentation
73
+
74
+ Full documentation lives at **<https://perrette.github.io/scribe/>**:
75
+
76
+ - [Installation & dependencies](https://perrette.github.io/scribe/installation/)
77
+ — PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
78
+ - [Quickstart](https://perrette.github.io/scribe/quickstart/) — your first
79
+ dictation.
80
+ - [Backends in detail](https://perrette.github.io/scribe/backends/) — model
81
+ lists, streaming recipes, vocabulary biasing, the realtime model.
82
+ - [Output modes & typer backends](https://perrette.github.io/scribe/output/) —
83
+ keystroke vs clipboard, Wayland / `eitype`, `--type-direct`.
84
+ - [System tray & global hotkeys](https://perrette.github.io/scribe/tray/) —
85
+ menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
86
+ - [Desktop entry & autostart (`scribe-install`)](https://perrette.github.io/scribe/desktop-install/)
87
+ — GNOME / KDE launcher integration.
88
+ - [Fine tuning & CLI reference](https://perrette.github.io/scribe/cli/) — every
89
+ `scribe --help` flag with examples.
90
+
91
+ ## From the same author
92
+
93
+ A few other open-source tools I maintain.
94
+
95
+ **Scientific writing & data**
96
+
97
+ - [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
98
+ - [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
99
+ - [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
100
+
101
+ **Speech to Text (dictate) and Text to Speech (read-aloud) tools**
102
+
103
+ - [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
104
+
105
+ ## Compatibility
106
+
107
+ | OS | Status |
108
+ |--------------------|---------------------------------------------------------------------|
109
+ | Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
110
+ | macOS | Works. |
111
+ | Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Dependencies resolve pre-built wheels; no C toolchain or Python downgrade needed. |
112
+
113
+ Wayland keystroke injection is convoluted but
114
+ [solved](https://perrette.github.io/scribe/output/).
@@ -76,7 +76,7 @@ streaming models.
76
76
  There are many [Vosk models](https://alphacephei.com/vosk/models)
77
77
  available; a handful are pre-mapped to common languages (`en`, `fr`,
78
78
  `de`, `it`) in
79
- [`scribe/models.toml`](../scribe/models.toml). Pick one with
79
+ [`scribe/models.toml`](https://github.com/perrette/scribe/blob/main/scribe/models.toml). Pick one with
80
80
  `-l <lang>` or browse the full list interactively from the menu.
81
81
 
82
82
  ## `openai` (OpenAI cloud)
@@ -134,9 +134,12 @@ installing `[openai]` is enough for both.
134
134
  ## Stopping a recording
135
135
 
136
136
  For batch models (Whisper local, Whisper-via-API, Groq, `gpt-4o-*`) the
137
- recording continues for up to 2 minutes until you stop it manually
138
- (Stop in the tray, Ctrl+C in the terminal) — the transcription happens
139
- once when you stop.
137
+ recording continues until you stop it manually (Stop in the tray,
138
+ Ctrl+C in the terminal), with a safety stop after `--clip-timeout`
139
+ seconds (10 minutes by default) — the transcription happens once when
140
+ you stop. Silent pauses are capped at `--clip-max-silence` seconds
141
+ (2 by default) in the audio sent for transcription, so dead air does
142
+ not count toward what the cloud APIs bill by duration.
140
143
 
141
144
  Streaming models (Vosk, `gpt-realtime-whisper`) emit partials as you
142
145
  speak and stop on the same Stop / Ctrl+C action.
@@ -230,7 +233,7 @@ column in the table above):
230
233
  multilingual coverage for English accuracy.
231
234
  - **Vosk** — language isn't a runtime parameter; vosk ships a
232
235
  separate model per language. `-l fr` looks up the vosk model
233
- pre-mapped to French in [`scribe/models.toml`](../scribe/models.toml)
236
+ pre-mapped to French in [`scribe/models.toml`](https://github.com/perrette/scribe/blob/main/scribe/models.toml)
234
237
  and instantiates that one. Vosk has no auto-detect path, so the
235
238
  Language menu's `Auto` entry on vosk falls back to a sensible
236
239
  default — the tray shows `Auto (🇬🇧 en)` to make this explicit
@@ -275,20 +278,13 @@ Once the buffer has grown to at least `--stream-chunk-min` (default
275
278
  (default 10 s) regardless of silence, to cap latency. The session
276
279
  continues until you stop it manually.
277
280
 
278
- The **first** chunk of a streaming thread uses a different floor:
279
- `--stream-first-chunk-min` (default 3 s). The bootstrap chunk has no
280
- prior text to bias Whisper's punctuation/casing, so a longer audio
281
- window lets the model produce a properly-punctuated transcript whose
282
- tail then seeds the rolling prompt for every chunk after it.
283
- Subsequent chunks fall back to `--stream-chunk-min`. The override
284
- also re-engages right after a context-reset silence (i.e. when a long
285
- pause cleared the rolling tail — see *Cross-chunk prompt context*
286
- below). Set `--stream-first-chunk-min` equal to `--stream-chunk-min`
287
- to disable the override. It's automatically inactive when
288
- `--stream-context-length 0` (Patient profile), where there is no
289
- rolling context to bootstrap. Internally clamped to `≤
290
- --stream-chunk-max` so a misconfigured pair can't deadlock the
291
- chunker.
281
+ The first chunk uses a higher floor (`--stream-first-chunk-min`,
282
+ default 3 s) so the bootstrap chunk has enough audio to seed the
283
+ rolling prompt for the rest. Auto-disabled when
284
+ `--stream-context-length 0` (Patient). If you stop talking before
285
+ the floor is reached, a pause past `--stream-context-reset-silence ×
286
+ --stream-chunk-silence-break` (default 1.8 s) flushes the buffer
287
+ anyway — your utterance is never stranded.
292
288
 
293
289
  ### Does pseudo-streaming change the API cost?
294
290
 
@@ -7,7 +7,7 @@ scribe --help
7
7
  ```
8
8
 
9
9
  The flags are grouped to mirror the source-of-truth in
10
- [`scribe/app.py`](../scribe/app.py).
10
+ [`scribe/app.py`](https://github.com/perrette/scribe/blob/main/scribe/app.py).
11
11
 
12
12
  ## Backend
13
13
 
@@ -124,7 +124,8 @@ silence-chunking knobs; they have their own end-of-utterance signal.
124
124
  | `--stream-first-chunk-min SECS` | `3.0` | Minimum chunk size for the *first* chunk of a streaming thread (default `3.0`). Higher than `--stream-chunk-min` so the bootstrap chunk has enough audio for Whisper to produce a punctuated transcript whose tail seeds the rolling prompt for the rest. Applies on recording start and right after a context-reset silence. Inactive when `--stream-context-length 0`. Clamped to `≤ --stream-chunk-max`. Set equal to `--stream-chunk-min` to disable. |
125
125
  | `--stream-chunk-silence-break SECS` | `0.6` | Silence duration that triggers a chunk cut (default `0.6`). Special value `0` enables Auto mode (best-silence-in-window at force-cut time). |
126
126
  | `--stream-context-reset-silence X` | `3.0` | Multiplier of `--stream-chunk-silence-break` above which the rolling cross-chunk prompt context is discarded (default `3.0`, i.e. 1.8 s at default silence-break). Use `inf` to never reset. |
127
- | `--clip-timeout SECS` | `120` | Auto-stop after this many seconds in Clip mode (default `120`). |
127
+ | `--clip-timeout SECS` | `600` | Auto-stop after this many seconds in Clip mode (default `600`). |
128
+ | `--clip-max-silence SECS` | `2.0` | In Clip mode, cap each silent pause at this many seconds in the audio sent for transcription (default `2.0`). Remote APIs bill by audio duration, so trimmed silence is not paid for; pauses up to the cap are kept for punctuation cues. `0` disables trimming. |
128
129
  | `--stream-timeout SECS` | `None` | Auto-stop after this many seconds in Stream mode (`None` = Always On, no auto-stop). Tray equivalent: **Stream timeout** in the Stream (advanced) submenu. |
129
130
 
130
131
  Native streamers (vosk, `gpt-realtime-whisper`) are always streaming
@@ -0,0 +1,58 @@
1
+ <!--
2
+ Home page. The feature bullets are pulled straight from README.md (single
3
+ source of truth) via the include-markdown plugin; everything else links into
4
+ the guide.
5
+ -->
6
+ <p align="center">
7
+ <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" alt="Scribe" width="96">
8
+ </p>
9
+
10
+ # Scribe
11
+
12
+ Scribe is a speech-to-text CLI and tray app that pipes transcribed text
13
+ into the focused window. It supports local and cloud-based APIs, batch and
14
+ streaming workflows.
15
+
16
+ {%
17
+ include-markdown "../README.md"
18
+ start="<!-- intro-start -->"
19
+ end="<!-- intro-end -->"
20
+ %}
21
+
22
+ <p align="center">
23
+ <img src="https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png" alt="Scribe tray menu" width="300">
24
+ </p>
25
+
26
+ ## Get started
27
+
28
+ ```bash
29
+ pip install scribe-cli[all]
30
+ scribe
31
+ ```
32
+
33
+ - **[Installation](installation.md)** — PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
34
+ - **[Quickstart](quickstart.md)** — your first dictation in a couple of minutes.
35
+ - **[Backends](backends.md)** — Vosk, Whisper, Whisper FUTO, OpenAI, Groq; streaming vs batch.
36
+ - **[CLI reference](cli.md)** — every `scribe --help` flag with examples.
37
+
38
+ ## Guides
39
+
40
+ - [Backends in detail](backends.md) — model lists, streaming recipes, vocabulary biasing.
41
+ - [Output modes](output.md) — keystroke vs clipboard vs terminal vs file, Wayland / `eitype`, `--type-direct`.
42
+ - [System tray & global hotkeys](tray.md) — menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
43
+ - [Desktop entry & autostart](desktop-install.md) — `scribe-install` launcher integration.
44
+ - [CLI reference](cli.md) — full flag reference and fine tuning.
45
+
46
+ ## From the same author
47
+
48
+ A few other open-source tools I maintain.
49
+
50
+ **Scientific writing & data**
51
+
52
+ - [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
53
+ - [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
54
+ - [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
55
+
56
+ **Speech to Text (dictate) and Text to Speech (read-aloud) tools**
57
+
58
+ - [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
@@ -9,7 +9,11 @@ clipboard / terminal output.
9
9
  ## System dependencies
10
10
 
11
11
  Scribe records audio via PortAudio (through `sounddevice`) and reads /
12
- writes the clipboard via `xclip` on Linux. On Ubuntu:
12
+ writes the clipboard via `xclip` on Linux — preferred over `wl-copy`
13
+ on Wayland, where the latter briefly steals keyboard focus on GNOME
14
+ < 47 and breaks pasting into Electron apps (see
15
+ [Clipboard backend on Wayland](output.md#clipboard-backend-on-wayland)).
16
+ On Ubuntu:
13
17
 
14
18
  ```bash
15
19
  sudo apt-get install portaudio19-dev xclip
@@ -21,7 +25,10 @@ On macOS use Homebrew:
21
25
  brew install portaudio
22
26
  ```
23
27
 
24
- (Windows ships everything needed via the wheels.)
28
+ On Windows there are **no system packages to install**: `sounddevice`
29
+ bundles PortAudio in its wheel and the clipboard uses the native Windows
30
+ API, so neither `portaudio19-dev` nor `xclip` apply. See the
31
+ [Windows quickstart](#windows) below.
25
32
 
26
33
  ## Python package
27
34
 
@@ -50,14 +57,52 @@ the four backends and the tray UI:
50
57
  | `[vosk]` | `vosk` | local Vosk backend (streaming) |
51
58
  | `[openai]` | `openai`, `soundfile` | OpenAI cloud backend (incl. realtime) |
52
59
  | `[groq]` | `openai`, `soundfile` | Groq cloud backend |
53
- | `[keyboard]` | `pynput` | the `pynput` typer (XTest/Quartz/WinAPI)|
54
- | `[app]` | `pystray`, `PyGObject` | system tray icon |
55
- | `[all]` | all of the above | one-shot setup |
60
+ | `[keyboard]` | `pynput` | back-compat only — `pynput` is a base dep now |
61
+ | `[app]` | `PyGObject` (Linux only) | the Linux AppIndicator tray binding |
62
+ | `[all]` | every backend + Linux tray binding | one-shot setup |
63
+
64
+ > **`pynput` and `pystray` are base dependencies.** The default run uses
65
+ > the keyboard typer and the system-tray app, so both ship with the plain
66
+ > `pip install scribe-cli` — you do **not** need `[keyboard]` or `[app]`
67
+ > for the standard experience. `[app]` now only adds the Linux-only
68
+ > `PyGObject` AppIndicator binding (skipped automatically on Windows/macOS
69
+ > via a `sys_platform == 'linux'` marker, since it needs GTK and won't
70
+ > pip-install elsewhere).
56
71
 
57
72
  You need at least one backend extra (or none if you only plan to use
58
73
  cloud backends *and* already have the `openai` package). The `groq`
59
74
  backend reuses the `openai` client, so `[openai]` covers both.
60
75
 
76
+ ## Windows
77
+
78
+ Windows 11 is tested and working on Python 3.14 (64-bit). Every
79
+ dependency — `onnxruntime`, `faster-whisper`/`ctranslate2`,
80
+ `pystray`/`Pillow`, `pynput` — resolves a ready-made `win_amd64` wheel,
81
+ so there is no build toolchain to install and no need to downgrade
82
+ Python.
83
+
84
+ From PowerShell:
85
+
86
+ ```powershell
87
+ py -m venv .venv
88
+ .\.venv\Scripts\Activate.ps1
89
+ # If activation is blocked by the execution policy, run once:
90
+ # Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
91
+ pip install -e .[whisper] # or [all], or a cloud backend like [openai]
92
+ scribe
93
+ ```
94
+
95
+ That's the whole setup. There are **no system packages** to install
96
+ (`apt`/`portaudio19-dev`/`xclip` are Linux-only) and **nothing to create
97
+ by hand** — earlier builds needed a manual `C:\tmp` folder, which is no
98
+ longer the case.
99
+
100
+ - **Tray icon:** appears under the taskbar overflow arrow (`^`) by
101
+ default; pin it via *Settings → Personalization → Taskbar → Other
102
+ system tray icons*. A single click on the icon starts recording.
103
+ - **Microphone:** if recording fails, enable *Settings → Privacy &
104
+ security → Microphone → "Let desktop apps access your microphone"*.
105
+
61
106
  ## Ubuntu / GNOME tray dependencies
62
107
 
63
108
  The tray icon needs system libraries for the AppIndicator stack:
@@ -52,6 +52,24 @@ the moment scribe fires the paste: the terminal then sees
52
52
  remember Shift for terminal targets, nothing for GUI apps where plain
53
53
  `Ctrl+V` already works (including VS Code's editor pane).
54
54
 
55
+ ### Clipboard backend on Wayland
56
+
57
+ The paste path writes the clipboard through `pyperclip`, which on a
58
+ Wayland session would normally use `wl-copy`. On compositors without a
59
+ data-control protocol (GNOME < 47), `wl-copy` has to create a temporary
60
+ invisible window and briefly take keyboard focus in order to own the
61
+ selection. Most apps tolerate that focus blip, but Electron apps often
62
+ don't return focus to the input field inside the page afterwards, so
63
+ the synthesised Ctrl+V lands nowhere.
64
+
65
+ Scribe therefore prefers `xclip` whenever XWayland is available: the
66
+ clipboard is set through XWayland with no focus change, and the
67
+ compositor syncs the X and Wayland selections both ways, so every app
68
+ — X11-hosted or Wayland-native — sees the same text. The clipboard is
69
+ a single session-global resource owned by the compositor; which tool
70
+ set it is invisible to the app you paste into. When `xclip` or
71
+ XWayland is missing, scribe falls back to `wl-copy`.
72
+
55
73
  ### `--type-direct` — bypass the clipboard
56
74
 
57
75
  In keystroke mode `--type-direct` (or **Options → Keyboard → Input
@@ -214,8 +232,9 @@ scribe --mode clipboard
214
232
  The transcription is copied to the system clipboard at end of recording
215
233
  and you press Ctrl+V (or Ctrl+Shift+V in a terminal) yourself. No
216
234
  keystrokes are synthesised, so there is no typer-backend involvement
217
- and no Wayland portal prompt — `pyperclip` / `wl-copy` is all that is
218
- needed.
235
+ and no Wayland portal prompt — `pyperclip` (backed by `xclip` or
236
+ `wl-copy`, see [Clipboard backend on Wayland](#clipboard-backend-on-wayland))
237
+ is all that is needed.
219
238
 
220
239
  As with keystroke mode, the clipboard is left holding the
221
240
  transcription after scribe finishes; save any previous clipboard