scribe-cli 1.1.1__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- scribe_cli-1.2.0/.github/workflows/docs.yml +34 -0
- scribe_cli-1.2.0/PKG-INFO +222 -0
- scribe_cli-1.2.0/README.md +114 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/backends.md +8 -5
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/cli.md +3 -2
- scribe_cli-1.2.0/docs/index.md +58 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/installation.md +5 -1
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/output.md +21 -2
- scribe_cli-1.2.0/docs/quickstart.md +80 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/roadmap-libei.md +2 -2
- scribe_cli-1.2.0/mkdocs.yml +76 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/pyproject.toml +9 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/_version.py +3 -3
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/app.py +19 -4
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/keyboard.py +34 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/models.py +53 -6
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/output.py +4 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/session.py +4 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/base.py +12 -7
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/eitype.py +22 -6
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/wtype.py +22 -7
- scribe_cli-1.2.0/scribe_cli.egg-info/PKG-INFO +222 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_cli.egg-info/SOURCES.txt +6 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_cli.egg-info/requires.txt +5 -0
- scribe_cli-1.2.0/tests/test_clip_silence_trim.py +121 -0
- scribe_cli-1.2.0/tests/test_clipboard_backend.py +64 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_typers_ascii_fallback.py +10 -9
- scribe_cli-1.1.1/PKG-INFO +0 -275
- scribe_cli-1.1.1/README.md +0 -172
- scribe_cli-1.1.1/scribe_cli.egg-info/PKG-INFO +0 -275
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/.github/FUNDING.yml +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/.github/workflows/pypi.yml +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/.gitignore +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/LICENSE +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/app-tray-menu.png +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/desktop-install.md +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/docs/tray.md +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/icon.xcf +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/__init__.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/audio.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/__init__.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/groq.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/openai_api.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/openai_realtime.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/vosk.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/whisper.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/backends/whisper_futo.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/dialog.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/install_desktop.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/menu.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/models.toml +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/saverecording.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/testpynput.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/__init__.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/pynput.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/typers/ydotool.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe/util.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_cli.egg-info/dependency_links.txt +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_cli.egg-info/entry_points.txt +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_cli.egg-info/top_level.txt +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/__init__.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/share/icon.png +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/share/icon_recording.png +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/share/icon_writing.png +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/silero_vad.LICENSE +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/silero_vad.onnx +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scribe_data/templates/scribe.desktop +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scripts/bench_whisper_local.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/scripts/test_python_versions_install.sh +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/setup.cfg +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_backend_matrix.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_compose_prompt.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_debug_logging.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_openai_realtime_coalesce.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_output.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_output_file_picker.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_prompt_file_picker.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_pseudo_streaming.py +0 -0
- {scribe_cli-1.1.1 → scribe_cli-1.2.0}/tests/test_whisper_futo.py +0 -0
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: docs
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
paths:
|
|
7
|
+
- "docs/**"
|
|
8
|
+
- "mkdocs.yml"
|
|
9
|
+
- "README.md"
|
|
10
|
+
- ".github/workflows/docs.yml"
|
|
11
|
+
workflow_dispatch:
|
|
12
|
+
|
|
13
|
+
permissions:
|
|
14
|
+
contents: write
|
|
15
|
+
|
|
16
|
+
jobs:
|
|
17
|
+
deploy:
|
|
18
|
+
runs-on: ubuntu-latest
|
|
19
|
+
steps:
|
|
20
|
+
- uses: actions/checkout@v4
|
|
21
|
+
with:
|
|
22
|
+
fetch-depth: 0
|
|
23
|
+
- uses: actions/setup-python@v5
|
|
24
|
+
with:
|
|
25
|
+
python-version: "3.x"
|
|
26
|
+
- name: Cache mkdocs-material assets
|
|
27
|
+
uses: actions/cache@v4
|
|
28
|
+
with:
|
|
29
|
+
key: mkdocs-material-${{ github.sha }}
|
|
30
|
+
path: .cache
|
|
31
|
+
restore-keys: |
|
|
32
|
+
mkdocs-material-
|
|
33
|
+
- run: pip install .[docs]
|
|
34
|
+
- run: mkdocs gh-deploy --force
|
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: scribe-cli
|
|
3
|
+
Version: 1.2.0
|
|
4
|
+
Summary: Speech-to-text CLI and system-tray app for dictating into any focused window. Local (vosk, faster-whisper) or cloud (groq, openai) backends, batch or streaming.
|
|
5
|
+
Author-email: Mahé Perrette <mahe.perrette@gmail.com>
|
|
6
|
+
License: MIT License
|
|
7
|
+
|
|
8
|
+
Copyright (c) 2024 Mahé Perrette
|
|
9
|
+
|
|
10
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
11
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
12
|
+
in the Software without restriction, including without limitation the rights
|
|
13
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
14
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
15
|
+
furnished to do so, subject to the following conditions:
|
|
16
|
+
|
|
17
|
+
The above copyright notice and this permission notice shall be included in all
|
|
18
|
+
copies or substantial portions of the Software.
|
|
19
|
+
|
|
20
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
21
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
22
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
23
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
24
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
25
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
26
|
+
SOFTWARE.
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
Note: This project relies on external packages that may have more restrictive
|
|
31
|
+
licenses. For example, the `pynput` package is licensed under LGPLv3, which
|
|
32
|
+
has different requirements compared to the MIT License. Please review the
|
|
33
|
+
licenses of all dependencies before using or distributing this software to
|
|
34
|
+
ensure compliance with their respective terms.
|
|
35
|
+
Project-URL: Homepage, https://github.com/perrette/scribe
|
|
36
|
+
Project-URL: Source, https://github.com/perrette/scribe
|
|
37
|
+
Project-URL: Issues, https://github.com/perrette/scribe/issues
|
|
38
|
+
Project-URL: Documentation, https://perrette.github.io/scribe/
|
|
39
|
+
Project-URL: Changelog, https://github.com/perrette/scribe/releases
|
|
40
|
+
Project-URL: Funding, https://github.com/sponsors/perrette
|
|
41
|
+
Keywords: speech-to-text,stt,transcription,dictation,voice-typing,voice-recognition,multilingual,realtime,streaming,cli,tray,vosk,whisper,faster-whisper,openai,groq,gpt-4o,linux,wayland,keyboard,clipboard,microphone,audio
|
|
42
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
43
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
44
|
+
Classifier: Intended Audience :: Developers
|
|
45
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
46
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
47
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
48
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
49
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
50
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
51
|
+
Classifier: Operating System :: OS Independent
|
|
52
|
+
Classifier: Environment :: Console
|
|
53
|
+
Classifier: Environment :: X11 Applications
|
|
54
|
+
Classifier: Environment :: MacOS X
|
|
55
|
+
Classifier: Environment :: Win32 (MS Windows)
|
|
56
|
+
Classifier: Natural Language :: English
|
|
57
|
+
Classifier: Natural Language :: French
|
|
58
|
+
Classifier: Natural Language :: German
|
|
59
|
+
Classifier: Natural Language :: Italian
|
|
60
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
61
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
62
|
+
Classifier: Topic :: Office/Business
|
|
63
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
64
|
+
Classifier: Topic :: Utilities
|
|
65
|
+
Requires-Python: >=3.9
|
|
66
|
+
Description-Content-Type: text/markdown
|
|
67
|
+
License-File: LICENSE
|
|
68
|
+
Requires-Dist: numpy
|
|
69
|
+
Requires-Dist: sounddevice
|
|
70
|
+
Requires-Dist: tqdm
|
|
71
|
+
Requires-Dist: requests
|
|
72
|
+
Requires-Dist: pyperclip
|
|
73
|
+
Requires-Dist: unidecode
|
|
74
|
+
Requires-Dist: termcolor
|
|
75
|
+
Requires-Dist: platformdirs
|
|
76
|
+
Requires-Dist: desktop-ai-core>=0.3.1
|
|
77
|
+
Requires-Dist: onnxruntime
|
|
78
|
+
Requires-Dist: pynput
|
|
79
|
+
Requires-Dist: pystray
|
|
80
|
+
Provides-Extra: keyboard
|
|
81
|
+
Requires-Dist: pynput; extra == "keyboard"
|
|
82
|
+
Provides-Extra: whisper
|
|
83
|
+
Requires-Dist: faster-whisper; extra == "whisper"
|
|
84
|
+
Provides-Extra: whisper-futo
|
|
85
|
+
Requires-Dist: pywhispercpp; extra == "whisper-futo"
|
|
86
|
+
Provides-Extra: vosk
|
|
87
|
+
Requires-Dist: vosk; extra == "vosk"
|
|
88
|
+
Provides-Extra: app
|
|
89
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "app"
|
|
90
|
+
Provides-Extra: openai
|
|
91
|
+
Requires-Dist: openai<3,>=2.37.0; extra == "openai"
|
|
92
|
+
Requires-Dist: soundfile; extra == "openai"
|
|
93
|
+
Provides-Extra: groq
|
|
94
|
+
Requires-Dist: openai<3,>=2.37.0; extra == "groq"
|
|
95
|
+
Requires-Dist: soundfile; extra == "groq"
|
|
96
|
+
Provides-Extra: vad
|
|
97
|
+
Provides-Extra: all
|
|
98
|
+
Requires-Dist: faster-whisper; extra == "all"
|
|
99
|
+
Requires-Dist: openai<3,>=2.37.0; extra == "all"
|
|
100
|
+
Requires-Dist: soundfile; extra == "all"
|
|
101
|
+
Requires-Dist: vosk; extra == "all"
|
|
102
|
+
Requires-Dist: PyGObject; sys_platform == "linux" and extra == "all"
|
|
103
|
+
Provides-Extra: docs
|
|
104
|
+
Requires-Dist: mkdocs<2,>=1.5; extra == "docs"
|
|
105
|
+
Requires-Dist: mkdocs-material>=9.5; extra == "docs"
|
|
106
|
+
Requires-Dist: mkdocs-include-markdown-plugin>=6.0; extra == "docs"
|
|
107
|
+
Dynamic: license-file
|
|
108
|
+
|
|
109
|
+
[](https://pypi.org/project/scribe-cli)
|
|
110
|
+

|
|
111
|
+
[](https://perrette.github.io/scribe/)
|
|
112
|
+
|
|
113
|
+
# Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
|
|
114
|
+
|
|
115
|
+
Scribe is a speech-to-text CLI and tray app that pipes transcribed text
|
|
116
|
+
into the focused window. It supports local and cloud-based APIs, batch and
|
|
117
|
+
streaming workflows.
|
|
118
|
+
|
|
119
|
+
<!-- intro-start -->
|
|
120
|
+
- **Five backends, one interface.** Records from your mic and transcribes via
|
|
121
|
+
**Vosk** (local, streaming), **Whisper** (local, batch), **Whisper FUTO**
|
|
122
|
+
(local, batch — ACFT-tuned for short dictations), **OpenAI** (cloud, batch
|
|
123
|
+
*or* streaming), or **Groq** (cloud, batch).
|
|
124
|
+
- **Four ways to deliver the transcript.** Paste into the focused window
|
|
125
|
+
(default), copy to the clipboard, print to the terminal, or append to a file.
|
|
126
|
+
- **Tray or terminal.** Runs as a **system tray icon** with a single Record
|
|
127
|
+
button, or as an interactive **terminal TUI** — same menu in both.
|
|
128
|
+
- **Hotkey-friendly.** Hooks into your desktop's keyboard shortcuts via
|
|
129
|
+
`SIGUSR1` (toggle recording) and `SIGUSR2` (cancel), plus built-in global
|
|
130
|
+
hotkeys on X11 / Windows.
|
|
131
|
+
- **Cross-platform.** Tested on Ubuntu (X11 and Wayland), macOS, and Windows;
|
|
132
|
+
works under Termux for clipboard / terminal output.
|
|
133
|
+
<!-- intro-end -->
|
|
134
|
+
|
|
135
|
+
<img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
|
|
136
|
+
|
|
137
|
+
## Installation
|
|
138
|
+
|
|
139
|
+
**Linux / macOS:**
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
143
|
+
pip install scribe-cli[all]
|
|
144
|
+
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
148
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
149
|
+
|
|
150
|
+
```powershell
|
|
151
|
+
py -m venv .venv
|
|
152
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
153
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
154
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
See the [installation page](https://perrette.github.io/scribe/installation/)
|
|
158
|
+
for the full per-platform walkthrough (extras, Ubuntu / GNOME tray libs, the
|
|
159
|
+
Windows quickstart).
|
|
160
|
+
|
|
161
|
+
## Quickstart
|
|
162
|
+
|
|
163
|
+
```bash
|
|
164
|
+
scribe
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
This launches the system tray icon. Press Record, speak, press Stop — the
|
|
168
|
+
transcription lands in the focused window. Scribe picks the first backend whose
|
|
169
|
+
key / dependency is present, in order **`groq` → `openai` → `whisper-futo` →
|
|
170
|
+
`whisper` → `vosk`**. Override the defaults or drop the tray entirely:
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
scribe --backend whisper --model small # local, no API key
|
|
174
|
+
scribe --frontend terminal # interactive TUI menu
|
|
175
|
+
scribe --mode clipboard # copy to clipboard, no keystroke
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
See the [quickstart](https://perrette.github.io/scribe/quickstart/) for more.
|
|
179
|
+
|
|
180
|
+
## Documentation
|
|
181
|
+
|
|
182
|
+
Full documentation lives at **<https://perrette.github.io/scribe/>**:
|
|
183
|
+
|
|
184
|
+
- [Installation & dependencies](https://perrette.github.io/scribe/installation/)
|
|
185
|
+
— PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
|
|
186
|
+
- [Quickstart](https://perrette.github.io/scribe/quickstart/) — your first
|
|
187
|
+
dictation.
|
|
188
|
+
- [Backends in detail](https://perrette.github.io/scribe/backends/) — model
|
|
189
|
+
lists, streaming recipes, vocabulary biasing, the realtime model.
|
|
190
|
+
- [Output modes & typer backends](https://perrette.github.io/scribe/output/) —
|
|
191
|
+
keystroke vs clipboard, Wayland / `eitype`, `--type-direct`.
|
|
192
|
+
- [System tray & global hotkeys](https://perrette.github.io/scribe/tray/) —
|
|
193
|
+
menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
|
|
194
|
+
- [Desktop entry & autostart (`scribe-install`)](https://perrette.github.io/scribe/desktop-install/)
|
|
195
|
+
— GNOME / KDE launcher integration.
|
|
196
|
+
- [Fine tuning & CLI reference](https://perrette.github.io/scribe/cli/) — every
|
|
197
|
+
`scribe --help` flag with examples.
|
|
198
|
+
|
|
199
|
+
## From the same author
|
|
200
|
+
|
|
201
|
+
A few other open-source tools I maintain.
|
|
202
|
+
|
|
203
|
+
**Scientific writing & data**
|
|
204
|
+
|
|
205
|
+
- [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
|
|
206
|
+
- [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
|
|
207
|
+
- [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
|
|
208
|
+
|
|
209
|
+
**Speech to Text (dictate) and Text to Speech (read-aloud) tools**
|
|
210
|
+
|
|
211
|
+
- [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
|
|
212
|
+
|
|
213
|
+
## Compatibility
|
|
214
|
+
|
|
215
|
+
| OS | Status |
|
|
216
|
+
|--------------------|---------------------------------------------------------------------|
|
|
217
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
218
|
+
| macOS | Works. |
|
|
219
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Dependencies resolve pre-built wheels; no C toolchain or Python downgrade needed. |
|
|
220
|
+
|
|
221
|
+
Wayland keystroke injection is convoluted but
|
|
222
|
+
[solved](https://perrette.github.io/scribe/output/).
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
[](https://pypi.org/project/scribe-cli)
|
|
2
|
+

|
|
3
|
+
[](https://perrette.github.io/scribe/)
|
|
4
|
+
|
|
5
|
+
# Scribe <img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" width="48">
|
|
6
|
+
|
|
7
|
+
Scribe is a speech-to-text CLI and tray app that pipes transcribed text
|
|
8
|
+
into the focused window. It supports local and cloud-based APIs, batch and
|
|
9
|
+
streaming workflows.
|
|
10
|
+
|
|
11
|
+
<!-- intro-start -->
|
|
12
|
+
- **Five backends, one interface.** Records from your mic and transcribes via
|
|
13
|
+
**Vosk** (local, streaming), **Whisper** (local, batch), **Whisper FUTO**
|
|
14
|
+
(local, batch — ACFT-tuned for short dictations), **OpenAI** (cloud, batch
|
|
15
|
+
*or* streaming), or **Groq** (cloud, batch).
|
|
16
|
+
- **Four ways to deliver the transcript.** Paste into the focused window
|
|
17
|
+
(default), copy to the clipboard, print to the terminal, or append to a file.
|
|
18
|
+
- **Tray or terminal.** Runs as a **system tray icon** with a single Record
|
|
19
|
+
button, or as an interactive **terminal TUI** — same menu in both.
|
|
20
|
+
- **Hotkey-friendly.** Hooks into your desktop's keyboard shortcuts via
|
|
21
|
+
`SIGUSR1` (toggle recording) and `SIGUSR2` (cancel), plus built-in global
|
|
22
|
+
hotkeys on X11 / Windows.
|
|
23
|
+
- **Cross-platform.** Tested on Ubuntu (X11 and Wayland), macOS, and Windows;
|
|
24
|
+
works under Termux for clipboard / terminal output.
|
|
25
|
+
<!-- intro-end -->
|
|
26
|
+
|
|
27
|
+
<img src=https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png width=300px>
|
|
28
|
+
|
|
29
|
+
## Installation
|
|
30
|
+
|
|
31
|
+
**Linux / macOS:**
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
sudo apt-get install portaudio19-dev xclip # Ubuntu; macOS: brew install portaudio
|
|
35
|
+
pip install scribe-cli[all]
|
|
36
|
+
export GROQ_API_KEY=YOURAPIKEY # or OPENAI_API_KEY, or skip and run local
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
**Windows** (PowerShell) — no system packages needed; `sounddevice` bundles
|
|
40
|
+
PortAudio and the clipboard is native, so skip the `apt`/`brew` step:
|
|
41
|
+
|
|
42
|
+
```powershell
|
|
43
|
+
py -m venv .venv
|
|
44
|
+
.\.venv\Scripts\Activate.ps1 # if blocked: Set-ExecutionPolicy -Scope CurrentUser RemoteSigned
|
|
45
|
+
pip install scribe-cli[whisper] # local Whisper; or [all], or a cloud backend
|
|
46
|
+
$env:GROQ_API_KEY = "YOURAPIKEY" # or OPENAI_API_KEY, or skip and run local
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
See the [installation page](https://perrette.github.io/scribe/installation/)
|
|
50
|
+
for the full per-platform walkthrough (extras, Ubuntu / GNOME tray libs, the
|
|
51
|
+
Windows quickstart).
|
|
52
|
+
|
|
53
|
+
## Quickstart
|
|
54
|
+
|
|
55
|
+
```bash
|
|
56
|
+
scribe
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
This launches the system tray icon. Press Record, speak, press Stop — the
|
|
60
|
+
transcription lands in the focused window. Scribe picks the first backend whose
|
|
61
|
+
key / dependency is present, in order **`groq` → `openai` → `whisper-futo` →
|
|
62
|
+
`whisper` → `vosk`**. Override the defaults or drop the tray entirely:
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
scribe --backend whisper --model small # local, no API key
|
|
66
|
+
scribe --frontend terminal # interactive TUI menu
|
|
67
|
+
scribe --mode clipboard # copy to clipboard, no keystroke
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
See the [quickstart](https://perrette.github.io/scribe/quickstart/) for more.
|
|
71
|
+
|
|
72
|
+
## Documentation
|
|
73
|
+
|
|
74
|
+
Full documentation lives at **<https://perrette.github.io/scribe/>**:
|
|
75
|
+
|
|
76
|
+
- [Installation & dependencies](https://perrette.github.io/scribe/installation/)
|
|
77
|
+
— PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
|
|
78
|
+
- [Quickstart](https://perrette.github.io/scribe/quickstart/) — your first
|
|
79
|
+
dictation.
|
|
80
|
+
- [Backends in detail](https://perrette.github.io/scribe/backends/) — model
|
|
81
|
+
lists, streaming recipes, vocabulary biasing, the realtime model.
|
|
82
|
+
- [Output modes & typer backends](https://perrette.github.io/scribe/output/) —
|
|
83
|
+
keystroke vs clipboard, Wayland / `eitype`, `--type-direct`.
|
|
84
|
+
- [System tray & global hotkeys](https://perrette.github.io/scribe/tray/) —
|
|
85
|
+
menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
|
|
86
|
+
- [Desktop entry & autostart (`scribe-install`)](https://perrette.github.io/scribe/desktop-install/)
|
|
87
|
+
— GNOME / KDE launcher integration.
|
|
88
|
+
- [Fine tuning & CLI reference](https://perrette.github.io/scribe/cli/) — every
|
|
89
|
+
`scribe --help` flag with examples.
|
|
90
|
+
|
|
91
|
+
## From the same author
|
|
92
|
+
|
|
93
|
+
A few other open-source tools I maintain.
|
|
94
|
+
|
|
95
|
+
**Scientific writing & data**
|
|
96
|
+
|
|
97
|
+
- [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
|
|
98
|
+
- [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
|
|
99
|
+
- [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
|
|
100
|
+
|
|
101
|
+
**Speech to Text (dictate) and Text to Speech (read-aloud) tools**
|
|
102
|
+
|
|
103
|
+
- [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
|
|
104
|
+
|
|
105
|
+
## Compatibility
|
|
106
|
+
|
|
107
|
+
| OS | Status |
|
|
108
|
+
|--------------------|---------------------------------------------------------------------|
|
|
109
|
+
| Ubuntu 24.04 | Primary dev platform (GNOME, X11 and Wayland). |
|
|
110
|
+
| macOS | Works. |
|
|
111
|
+
| Windows 11 | Tested and working on Python 3.14 (64-bit / `win_amd64`). Dependencies resolve pre-built wheels; no C toolchain or Python downgrade needed. |
|
|
112
|
+
|
|
113
|
+
Wayland keystroke injection is convoluted but
|
|
114
|
+
[solved](https://perrette.github.io/scribe/output/).
|
|
@@ -76,7 +76,7 @@ streaming models.
|
|
|
76
76
|
There are many [Vosk models](https://alphacephei.com/vosk/models)
|
|
77
77
|
available; a handful are pre-mapped to common languages (`en`, `fr`,
|
|
78
78
|
`de`, `it`) in
|
|
79
|
-
[`scribe/models.toml`](
|
|
79
|
+
[`scribe/models.toml`](https://github.com/perrette/scribe/blob/main/scribe/models.toml). Pick one with
|
|
80
80
|
`-l <lang>` or browse the full list interactively from the menu.
|
|
81
81
|
|
|
82
82
|
## `openai` (OpenAI cloud)
|
|
@@ -134,9 +134,12 @@ installing `[openai]` is enough for both.
|
|
|
134
134
|
## Stopping a recording
|
|
135
135
|
|
|
136
136
|
For batch models (Whisper local, Whisper-via-API, Groq, `gpt-4o-*`) the
|
|
137
|
-
recording continues
|
|
138
|
-
|
|
139
|
-
once when
|
|
137
|
+
recording continues until you stop it manually (Stop in the tray,
|
|
138
|
+
Ctrl+C in the terminal), with a safety stop after `--clip-timeout`
|
|
139
|
+
seconds (10 minutes by default) — the transcription happens once when
|
|
140
|
+
you stop. Silent pauses are capped at `--clip-max-silence` seconds
|
|
141
|
+
(2 by default) in the audio sent for transcription, so dead air does
|
|
142
|
+
not count toward what the cloud APIs bill by duration.
|
|
140
143
|
|
|
141
144
|
Streaming models (Vosk, `gpt-realtime-whisper`) emit partials as you
|
|
142
145
|
speak and stop on the same Stop / Ctrl+C action.
|
|
@@ -230,7 +233,7 @@ column in the table above):
|
|
|
230
233
|
multilingual coverage for English accuracy.
|
|
231
234
|
- **Vosk** — language isn't a runtime parameter; vosk ships a
|
|
232
235
|
separate model per language. `-l fr` looks up the vosk model
|
|
233
|
-
pre-mapped to French in [`scribe/models.toml`](
|
|
236
|
+
pre-mapped to French in [`scribe/models.toml`](https://github.com/perrette/scribe/blob/main/scribe/models.toml)
|
|
234
237
|
and instantiates that one. Vosk has no auto-detect path, so the
|
|
235
238
|
Language menu's `Auto` entry on vosk falls back to a sensible
|
|
236
239
|
default — the tray shows `Auto (🇬🇧 en)` to make this explicit
|
|
@@ -7,7 +7,7 @@ scribe --help
|
|
|
7
7
|
```
|
|
8
8
|
|
|
9
9
|
The flags are grouped to mirror the source-of-truth in
|
|
10
|
-
[`scribe/app.py`](
|
|
10
|
+
[`scribe/app.py`](https://github.com/perrette/scribe/blob/main/scribe/app.py).
|
|
11
11
|
|
|
12
12
|
## Backend
|
|
13
13
|
|
|
@@ -124,7 +124,8 @@ silence-chunking knobs; they have their own end-of-utterance signal.
|
|
|
124
124
|
| `--stream-first-chunk-min SECS` | `3.0` | Minimum chunk size for the *first* chunk of a streaming thread (default `3.0`). Higher than `--stream-chunk-min` so the bootstrap chunk has enough audio for Whisper to produce a punctuated transcript whose tail seeds the rolling prompt for the rest. Applies on recording start and right after a context-reset silence. Inactive when `--stream-context-length 0`. Clamped to `≤ --stream-chunk-max`. Set equal to `--stream-chunk-min` to disable. |
|
|
125
125
|
| `--stream-chunk-silence-break SECS` | `0.6` | Silence duration that triggers a chunk cut (default `0.6`). Special value `0` enables Auto mode (best-silence-in-window at force-cut time). |
|
|
126
126
|
| `--stream-context-reset-silence X` | `3.0` | Multiplier of `--stream-chunk-silence-break` above which the rolling cross-chunk prompt context is discarded (default `3.0`, i.e. 1.8 s at default silence-break). Use `inf` to never reset. |
|
|
127
|
-
| `--clip-timeout SECS` | `
|
|
127
|
+
| `--clip-timeout SECS` | `600` | Auto-stop after this many seconds in Clip mode (default `600`). |
|
|
128
|
+
| `--clip-max-silence SECS` | `2.0` | In Clip mode, cap each silent pause at this many seconds in the audio sent for transcription (default `2.0`). Remote APIs bill by audio duration, so trimmed silence is not paid for; pauses up to the cap are kept for punctuation cues. `0` disables trimming. |
|
|
128
129
|
| `--stream-timeout SECS` | `None` | Auto-stop after this many seconds in Stream mode (`None` = Always On, no auto-stop). Tray equivalent: **Stream timeout** in the Stream (advanced) submenu. |
|
|
129
130
|
|
|
130
131
|
Native streamers (vosk, `gpt-realtime-whisper`) are always streaming
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
<!--
|
|
2
|
+
Home page. The feature bullets are pulled straight from README.md (single
|
|
3
|
+
source of truth) via the include-markdown plugin; everything else links into
|
|
4
|
+
the guide.
|
|
5
|
+
-->
|
|
6
|
+
<p align="center">
|
|
7
|
+
<img src="https://github.com/perrette/scribe/raw/main/scribe_data/share/icon.png" alt="Scribe" width="96">
|
|
8
|
+
</p>
|
|
9
|
+
|
|
10
|
+
# Scribe
|
|
11
|
+
|
|
12
|
+
Scribe is a speech-to-text CLI and tray app that pipes transcribed text
|
|
13
|
+
into the focused window. It supports local and cloud-based APIs, batch and
|
|
14
|
+
streaming workflows.
|
|
15
|
+
|
|
16
|
+
{%
|
|
17
|
+
include-markdown "../README.md"
|
|
18
|
+
start="<!-- intro-start -->"
|
|
19
|
+
end="<!-- intro-end -->"
|
|
20
|
+
%}
|
|
21
|
+
|
|
22
|
+
<p align="center">
|
|
23
|
+
<img src="https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png" alt="Scribe tray menu" width="300">
|
|
24
|
+
</p>
|
|
25
|
+
|
|
26
|
+
## Get started
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install scribe-cli[all]
|
|
30
|
+
scribe
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
- **[Installation](installation.md)** — PortAudio, extras, Ubuntu / GNOME tray libs, Windows.
|
|
34
|
+
- **[Quickstart](quickstart.md)** — your first dictation in a couple of minutes.
|
|
35
|
+
- **[Backends](backends.md)** — Vosk, Whisper, Whisper FUTO, OpenAI, Groq; streaming vs batch.
|
|
36
|
+
- **[CLI reference](cli.md)** — every `scribe --help` flag with examples.
|
|
37
|
+
|
|
38
|
+
## Guides
|
|
39
|
+
|
|
40
|
+
- [Backends in detail](backends.md) — model lists, streaming recipes, vocabulary biasing.
|
|
41
|
+
- [Output modes](output.md) — keystroke vs clipboard vs terminal vs file, Wayland / `eitype`, `--type-direct`.
|
|
42
|
+
- [System tray & global hotkeys](tray.md) — menu tree, icon states, `SIGUSR1`/`SIGUSR2`.
|
|
43
|
+
- [Desktop entry & autostart](desktop-install.md) — `scribe-install` launcher integration.
|
|
44
|
+
- [CLI reference](cli.md) — full flag reference and fine tuning.
|
|
45
|
+
|
|
46
|
+
## From the same author
|
|
47
|
+
|
|
48
|
+
A few other open-source tools I maintain.
|
|
49
|
+
|
|
50
|
+
**Scientific writing & data**
|
|
51
|
+
|
|
52
|
+
- [**texmark**](https://perrette.github.io/texmark/) — write scientific articles in Markdown and convert them to journal-ready LaTeX/PDF.
|
|
53
|
+
- [**papers**](https://perrette.github.io/papers/) — command-line BibTeX bibliography and PDF library manager.
|
|
54
|
+
- [**datamanifest**](https://perrette.github.io/datamanifest/) — declarative, reproducible dataset management. *(See also the [datamanifest.toml](https://perrette.github.io/datamanifest.toml/) format spec and the [DataManifest.jl](https://awi-esc.github.io/DataManifest.jl/) Julia port.)*
|
|
55
|
+
|
|
56
|
+
**Speech to Text (dictate) and Text to Speech (read-aloud) tools**
|
|
57
|
+
|
|
58
|
+
- [**bard**](https://perrette.github.io/bard/) — text-to-speech reader.
|
|
@@ -9,7 +9,11 @@ clipboard / terminal output.
|
|
|
9
9
|
## System dependencies
|
|
10
10
|
|
|
11
11
|
Scribe records audio via PortAudio (through `sounddevice`) and reads /
|
|
12
|
-
writes the clipboard via `xclip` on Linux
|
|
12
|
+
writes the clipboard via `xclip` on Linux — preferred over `wl-copy`
|
|
13
|
+
on Wayland, where the latter briefly steals keyboard focus on GNOME
|
|
14
|
+
< 47 and breaks pasting into Electron apps (see
|
|
15
|
+
[Clipboard backend on Wayland](output.md#clipboard-backend-on-wayland)).
|
|
16
|
+
On Ubuntu:
|
|
13
17
|
|
|
14
18
|
```bash
|
|
15
19
|
sudo apt-get install portaudio19-dev xclip
|
|
@@ -52,6 +52,24 @@ the moment scribe fires the paste: the terminal then sees
|
|
|
52
52
|
remember Shift for terminal targets, nothing for GUI apps where plain
|
|
53
53
|
`Ctrl+V` already works (including VS Code's editor pane).
|
|
54
54
|
|
|
55
|
+
### Clipboard backend on Wayland
|
|
56
|
+
|
|
57
|
+
The paste path writes the clipboard through `pyperclip`, which on a
|
|
58
|
+
Wayland session would normally use `wl-copy`. On compositors without a
|
|
59
|
+
data-control protocol (GNOME < 47), `wl-copy` has to create a temporary
|
|
60
|
+
invisible window and briefly take keyboard focus in order to own the
|
|
61
|
+
selection. Most apps tolerate that focus blip, but Electron apps often
|
|
62
|
+
don't return focus to the input field inside the page afterwards, so
|
|
63
|
+
the synthesised Ctrl+V lands nowhere.
|
|
64
|
+
|
|
65
|
+
Scribe therefore prefers `xclip` whenever XWayland is available: the
|
|
66
|
+
clipboard is set through XWayland with no focus change, and the
|
|
67
|
+
compositor syncs the X and Wayland selections both ways, so every app
|
|
68
|
+
— X11-hosted or Wayland-native — sees the same text. The clipboard is
|
|
69
|
+
a single session-global resource owned by the compositor; which tool
|
|
70
|
+
set it is invisible to the app you paste into. When `xclip` or
|
|
71
|
+
XWayland is missing, scribe falls back to `wl-copy`.
|
|
72
|
+
|
|
55
73
|
### `--type-direct` — bypass the clipboard
|
|
56
74
|
|
|
57
75
|
In keystroke mode `--type-direct` (or **Options → Keyboard → Input
|
|
@@ -214,8 +232,9 @@ scribe --mode clipboard
|
|
|
214
232
|
The transcription is copied to the system clipboard at end of recording
|
|
215
233
|
and you press Ctrl+V (or Ctrl+Shift+V in a terminal) yourself. No
|
|
216
234
|
keystrokes are synthesised, so there is no typer-backend involvement
|
|
217
|
-
and no Wayland portal prompt — `pyperclip`
|
|
218
|
-
|
|
235
|
+
and no Wayland portal prompt — `pyperclip` (backed by `xclip` or
|
|
236
|
+
`wl-copy`, see [Clipboard backend on Wayland](#clipboard-backend-on-wayland))
|
|
237
|
+
is all that is needed.
|
|
219
238
|
|
|
220
239
|
As with keystroke mode, the clipboard is left holding the
|
|
221
240
|
transcription after scribe finishes; save any previous clipboard
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Quickstart
|
|
2
|
+
|
|
3
|
+
Once Scribe is [installed](installation.md), launch it from a terminal:
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
scribe
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
This starts the **system tray icon**. Press Record, speak, press Stop — the
|
|
10
|
+
transcription lands in the focused window.
|
|
11
|
+
|
|
12
|
+
<p align="center">
|
|
13
|
+
<img src="https://raw.githubusercontent.com/perrette/scribe/main/docs/app-tray-menu.png" alt="Scribe tray menu" width="300">
|
|
14
|
+
</p>
|
|
15
|
+
|
|
16
|
+
## Backend auto-selection
|
|
17
|
+
|
|
18
|
+
Scribe picks the first backend whose key / dependency is present, in order
|
|
19
|
+
**`groq` → `openai` → `whisper-futo` → `whisper` → `vosk`**. So with
|
|
20
|
+
`GROQ_API_KEY` set, `scribe` is equivalent to:
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
scribe --backend groq --model whisper-large-v3-turbo
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
See [Backends](backends.md) for the full picture (streaming vs batch, model
|
|
27
|
+
lists, when to pick which).
|
|
28
|
+
|
|
29
|
+
## Overriding the defaults
|
|
30
|
+
|
|
31
|
+
You can override the backend / model or drop the tray entirely:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
scribe --backend openai --model gpt-4o-mini-transcribe # OpenAI sweet spot
|
|
35
|
+
scribe --backend openai --model gpt-realtime-whisper # OpenAI streaming
|
|
36
|
+
scribe --backend whisper --model small # local, no API key
|
|
37
|
+
scribe --frontend terminal # interactive TUI menu
|
|
38
|
+
scribe --record # start recording immediately on launch (tray or terminal)
|
|
39
|
+
scribe --record --frontend terminal --mode file # one-shot batched dictation → file
|
|
40
|
+
scribe --record --frontend terminal --mode file --stream # streamed: chunks appended live as you speak
|
|
41
|
+
scribe --mode clipboard # copy to clipboard, no keystroke
|
|
42
|
+
scribe --mode terminal # only print to stdout
|
|
43
|
+
scribe --mode file -o transcript.txt # append to a file (no keystroke / clipboard)
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
With `--no-interactive` (terminal frontend only), Scribe skips the interactive
|
|
47
|
+
menu and starts recording right away — handy for scripted, one-shot
|
|
48
|
+
transcriptions. See [Output modes](output.md) for where the transcript goes,
|
|
49
|
+
and the [CLI reference](cli.md) for every flag.
|
|
50
|
+
|
|
51
|
+
## Getting an API key
|
|
52
|
+
|
|
53
|
+
Groq is the **recommended cloud backend by default** — extremely fast (by a
|
|
54
|
+
wide margin compared to other cloud STT options, especially in **Stream** mode
|
|
55
|
+
where the per-chunk roundtrip latency dominates the perceived speed), quite
|
|
56
|
+
accurate, and the **free tier** is generous enough for everyday dictation.
|
|
57
|
+
Sign up at [console.groq.com](https://console.groq.com/), create an API key
|
|
58
|
+
under **Settings → API Keys**, and export it as `GROQ_API_KEY`:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
export GROQ_API_KEY=YOURAPIKEY
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
[OpenAI](https://openai.com/api/) with `gpt-4o-mini-transcribe` is another
|
|
65
|
+
fast option (`export OPENAI_API_KEY=...`), and the local Whisper / Vosk
|
|
66
|
+
backends need no key at all.
|
|
67
|
+
|
|
68
|
+
## Biasing the recogniser
|
|
69
|
+
|
|
70
|
+
Bias the recogniser toward names, jargon, or a domain glossary with
|
|
71
|
+
`--prompt "free text hint"` and `--words word1 word2 ...` (each also accepts a
|
|
72
|
+
`--prompt-file` / `--words-file` companion):
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
scribe --prompt "Patient notes from a cardiology consult." \
|
|
76
|
+
--words tachycardia bradycardia echocardiogram metoprolol
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
See [Backends › Vocabulary biasing](backends.md#vocabulary-biasing) for what
|
|
80
|
+
each backend does with them.
|
|
@@ -37,7 +37,7 @@ scribe has two output paths into the focused app:
|
|
|
37
37
|
added pain of layout-dependent character handling and 100+ keystrokes
|
|
38
38
|
per utterance.
|
|
39
39
|
|
|
40
|
-
Both paths live in [`scribe/keyboard.py`](
|
|
40
|
+
Both paths live in [`scribe/keyboard.py`](https://github.com/perrette/scribe/blob/main/scribe/keyboard.py).
|
|
41
41
|
|
|
42
42
|
## Target architecture
|
|
43
43
|
|
|
@@ -85,7 +85,7 @@ delegates `paste_text()` to `Typer.paste()`.
|
|
|
85
85
|
No new functionality; pure refactor so that subsequent phases plug in cleanly.
|
|
86
86
|
|
|
87
87
|
- Extract `Typer` protocol + `PynputTyper` from
|
|
88
|
-
[`scribe/keyboard.py`](
|
|
88
|
+
[`scribe/keyboard.py`](https://github.com/perrette/scribe/blob/main/scribe/keyboard.py).
|
|
89
89
|
- Move `paste_text()` and `safe_type_text()` into `PynputTyper`.
|
|
90
90
|
- `type_text(...)` becomes a thin facade that resolves a typer via
|
|
91
91
|
`pick_typer()` and delegates. Keep the public signature unchanged so
|