sttop 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sttop-0.1.0/.github/workflows/ci.yml +20 -0
- sttop-0.1.0/.github/workflows/release.yml +32 -0
- sttop-0.1.0/.gitignore +10 -0
- sttop-0.1.0/PKG-INFO +234 -0
- sttop-0.1.0/README.md +198 -0
- sttop-0.1.0/pyproject.toml +82 -0
- sttop-0.1.0/sttop/__init__.py +8 -0
- sttop-0.1.0/sttop/__main__.py +193 -0
- sttop-0.1.0/sttop/audio/__init__.py +1 -0
- sttop-0.1.0/sttop/audio/capture.py +228 -0
- sttop-0.1.0/sttop/audio/devices.py +90 -0
- sttop-0.1.0/sttop/audio/segmenter.py +170 -0
- sttop-0.1.0/sttop/config.py +215 -0
- sttop-0.1.0/sttop/diarize.py +166 -0
- sttop-0.1.0/sttop/engine.py +269 -0
- sttop-0.1.0/sttop/journal.py +149 -0
- sttop-0.1.0/sttop/stt/__init__.py +60 -0
- sttop-0.1.0/sttop/stt/base.py +30 -0
- sttop-0.1.0/sttop/stt/local.py +89 -0
- sttop-0.1.0/sttop/stt/parakeet.py +29 -0
- sttop-0.1.0/sttop/terminal.py +144 -0
- sttop-0.1.0/sttop/tui.py +187 -0
- sttop-0.1.0/tests/test_capture.py +46 -0
- sttop-0.1.0/tests/test_cli.py +82 -0
- sttop-0.1.0/tests/test_config.py +78 -0
- sttop-0.1.0/tests/test_devices.py +48 -0
- sttop-0.1.0/tests/test_engine.py +90 -0
- sttop-0.1.0/tests/test_journal.py +101 -0
- sttop-0.1.0/tests/test_pipeline.py +126 -0
- sttop-0.1.0/tests/test_segmenter.py +125 -0
- sttop-0.1.0/tests/test_stt.py +54 -0
- sttop-0.1.0/tests/test_terminal.py +77 -0
- sttop-0.1.0/uv.lock +1604 -0
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: ci
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main, master]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v5
|
|
13
|
+
- uses: astral-sh/setup-uv@v6
|
|
14
|
+
with:
|
|
15
|
+
enable-cache: true
|
|
16
|
+
- run: sudo apt-get update && sudo apt-get install -y ffmpeg
|
|
17
|
+
- run: uv sync --extra dev
|
|
18
|
+
# The audio-hardware tests stay opt-in; CI has no sound server.
|
|
19
|
+
- run: uv run --extra dev pytest -q
|
|
20
|
+
- run: uv run ruff check .
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
build:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v5
|
|
13
|
+
- uses: astral-sh/setup-uv@v6
|
|
14
|
+
- run: uv build
|
|
15
|
+
- uses: actions/upload-artifact@v4
|
|
16
|
+
with:
|
|
17
|
+
name: dist
|
|
18
|
+
path: dist/
|
|
19
|
+
|
|
20
|
+
publish:
|
|
21
|
+
needs: build
|
|
22
|
+
runs-on: ubuntu-latest
|
|
23
|
+
environment: pypi
|
|
24
|
+
# Trusted publishing: PyPI verifies this workflow's identity, no token needed.
|
|
25
|
+
permissions:
|
|
26
|
+
id-token: write
|
|
27
|
+
steps:
|
|
28
|
+
- uses: actions/download-artifact@v4
|
|
29
|
+
with:
|
|
30
|
+
name: dist
|
|
31
|
+
path: dist/
|
|
32
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
sttop-0.1.0/.gitignore
ADDED
sttop-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: sttop
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Live speech-to-text monitor for the terminal. Taps mic + system audio, transcribes and labels speakers in real time, stores everything locally.
|
|
5
|
+
Project-URL: Homepage, https://github.com/v4rgas/sttop
|
|
6
|
+
Project-URL: Repository, https://github.com/v4rgas/sttop
|
|
7
|
+
Project-URL: Issues, https://github.com/v4rgas/sttop/issues
|
|
8
|
+
Author: v4rgas
|
|
9
|
+
License: MIT
|
|
10
|
+
Keywords: diarization,speech-to-text,terminal,transcription,tui,whisper
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: End Users/Desktop
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
|
|
21
|
+
Classifier: Topic :: Utilities
|
|
22
|
+
Requires-Python: >=3.11
|
|
23
|
+
Requires-Dist: faster-whisper>=1.0.3
|
|
24
|
+
Requires-Dist: numpy>=1.24
|
|
25
|
+
Requires-Dist: onnx-asr[cpu,hub]>=0.12
|
|
26
|
+
Requires-Dist: platformdirs>=4.0
|
|
27
|
+
Requires-Dist: speechbrain>=1.0
|
|
28
|
+
Requires-Dist: textual>=0.80
|
|
29
|
+
Requires-Dist: torch>=2.2
|
|
30
|
+
Requires-Dist: torchaudio>=2.2
|
|
31
|
+
Requires-Dist: webrtcvad-wheels>=2.0.14
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
34
|
+
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# sttop
|
|
38
|
+
|
|
39
|
+
Live speech-to-text monitor for the terminal — `htop`, but for what is being said.
|
|
40
|
+
|
|
41
|
+
Taps your **microphone** and your **system audio output** as two independent streams,
|
|
42
|
+
transcribes both in real time, labels who is speaking, and appends every line to a
|
|
43
|
+
Markdown file as it happens. Fully local: no network, no API keys, nothing leaves
|
|
44
|
+
the machine.
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
sttop ● rec 04:12 mic ███······· sys ██████···· parakeet-tdt/cpu onnx · ecapa @0.50 · queue 0 · 37 lines · 3 voices
|
|
48
|
+
writing → ~/.local/share/sttop/sessions/2026-08-10-1432-standup.md
|
|
49
|
+
|
|
50
|
+
03:58 you so the migration lands friday?
|
|
51
|
+
04:02 spk1 friday is tight, monday is safer
|
|
52
|
+
04:09 spk2 +1 on monday
|
|
53
|
+
|
|
54
|
+
q quit space pause r rename speaker
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
## Why two streams
|
|
58
|
+
|
|
59
|
+
Capturing the mic and the speaker output separately means **you** are identified for
|
|
60
|
+
free — anything on the mic is you, no model required, never wrong. Voice embeddings
|
|
61
|
+
then only have to split the *remote* side into individual participants, which is a
|
|
62
|
+
much easier problem than diarizing a single mixed track.
|
|
63
|
+
|
|
64
|
+
## Install
|
|
65
|
+
|
|
66
|
+
Needs `ffmpeg` and PipeWire or PulseAudio (`pactl`).
|
|
67
|
+
|
|
68
|
+
```bash
|
|
69
|
+
uvx --index https://download.pytorch.org/whl/cpu sttop
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
The `--index` flag matters. Speaker labelling needs torch, and the stock PyPI torch
|
|
73
|
+
bundles CUDA — about 2.5GB of nvidia wheels that buy nothing here, since
|
|
74
|
+
CTranslate2 has no ROCm backend and CPU inference keeps up with live audio fine.
|
|
75
|
+
The flag points torch at the CPU builds and falls back to PyPI for everything else.
|
|
76
|
+
|
|
77
|
+
To install it permanently rather than running it ad hoc:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
uv tool install --index https://download.pytorch.org/whl/cpu sttop
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
From a checkout, `uv sync` reads the CPU index out of `pyproject.toml` already:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
git clone https://github.com/v4rgas/sttop && cd sttop
|
|
87
|
+
uv sync
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
## Use
|
|
91
|
+
|
|
92
|
+
```bash
|
|
93
|
+
uv run sttop # record with defaults
|
|
94
|
+
uv run sttop -t "standup" # title the session (used in the filename)
|
|
95
|
+
uv run sttop --backend whisper -m small
|
|
96
|
+
uv run sttop devices --test # list audio sources, record 1s from each
|
|
97
|
+
uv run sttop sessions # list past transcripts
|
|
98
|
+
uv run sttop config # write ~/.config/sttop/config.toml
|
|
99
|
+
uv run sttop theme # show the detected terminal colour scheme
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Keys: `q` quit · `space` pause · `r` rename a speaker (`spk1=Ana`, rewrites past lines too).
|
|
103
|
+
|
|
104
|
+
## Output
|
|
105
|
+
|
|
106
|
+
One Markdown file per session in `~/.local/share/sttop/sessions/`, flushed after every
|
|
107
|
+
line — kill it mid-meeting and the transcript so far is already on disk.
|
|
108
|
+
|
|
109
|
+
```markdown
|
|
110
|
+
# standup
|
|
111
|
+
|
|
112
|
+
- started: 2026-08-10 14:32:01 -04
|
|
113
|
+
- mic: `alsa_input.pci-0000_08_00.6.analog-stereo`
|
|
114
|
+
- system: `alsa_output.pci-0000_08_00.6.analog-stereo.monitor`
|
|
115
|
+
- backend: `parakeet-tdt/cpu onnx`
|
|
116
|
+
|
|
117
|
+
## Transcript
|
|
118
|
+
|
|
119
|
+
- `03:58` **you** — so the migration lands friday?
|
|
120
|
+
|
|
121
|
+
- `04:02` **spk1** — friday is tight, monday is safer
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
## How it works
|
|
125
|
+
|
|
126
|
+
```
|
|
127
|
+
ffmpeg -f pulse (mic) ─┐
|
|
128
|
+
├─ webrtcvad segmenter ─→ queue ─→ parakeet ─→ ecapa ─→ journal.md
|
|
129
|
+
ffmpeg -f pulse (monitor) ─┘
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Audio is cut into utterances by voice-activity detection (a segment closes after
|
|
133
|
+
700 ms of silence, or at 15 s for a monologue), and only speech reaches the model.
|
|
134
|
+
|
|
135
|
+
**One thread boundary, and it is the model.** The two capture readers and the
|
|
136
|
+
consumer are asyncio tasks — they are blocking pipe I/O, which is what an event
|
|
137
|
+
loop is for — while transcription and voice embedding run in a single-worker
|
|
138
|
+
`ThreadPoolExecutor`. So the UI needs no cross-thread marshalling, shutdown is
|
|
139
|
+
ordinary task cancellation, and utterances stay in the order they were spoken. The
|
|
140
|
+
executor is single-worker on purpose: transcription is CPU-bound and already
|
|
141
|
+
internally parallel, so a second worker would only thrash the cache. When it falls
|
|
142
|
+
behind, the queue absorbs the lag — visible as `queue N` in the status bar — rather
|
|
143
|
+
than dropping audio.
|
|
144
|
+
|
|
145
|
+
Speaker labels come from online clustering of ECAPA-TDNN voice embeddings: each
|
|
146
|
+
utterance is matched against running centroids by cosine similarity. A confident
|
|
147
|
+
match (≥ `threshold`) joins that speaker and updates its centroid; a near miss
|
|
148
|
+
(within `margin` below it) joins without touching the centroid; only a clearly
|
|
149
|
+
distant voice opens a new speaker. That hysteresis matters — without it a single
|
|
150
|
+
noisy embedding mints a phantom participant, and one person ends up spread across
|
|
151
|
+
`spk1`/`spk2`/`spk3`. Being online means labels are assigned as audio arrives and are
|
|
152
|
+
never revised, which is the price of real time. Segments under 1.5 s are too short to
|
|
153
|
+
embed reliably and inherit the previous speaker, or show as `spk?`.
|
|
154
|
+
|
|
155
|
+
## Backends
|
|
156
|
+
|
|
157
|
+
**parakeet** (default) — NVIDIA Parakeet TDT 0.6b v3 through onnxruntime. Multilingual
|
|
158
|
+
across 25 European languages with autodetection, punctuated output, and roughly 19×
|
|
159
|
+
real time on CPU. Needs neither torch nor the NeMo toolkit, since onnx-asr runs the
|
|
160
|
+
exported graph directly. On the same 11 s clip where `whisper tiny` produced a
|
|
161
|
+
hallucinated lead-in and lost its punctuation, Parakeet returned the sentence verbatim.
|
|
162
|
+
|
|
163
|
+
**whisper** — faster-whisper/CTranslate2, if you want Whisper's language coverage.
|
|
164
|
+
CTranslate2 ships **CUDA and CPU backends only — there is no ROCm build**, so on an AMD
|
|
165
|
+
GPU this runs on CPU no matter what torch reports. The device is detected at startup
|
|
166
|
+
(`cuda` if CTranslate2 sees one, else `cpu`) and shown in the status bar.
|
|
167
|
+
|
|
168
|
+
To push a Radeon card at the *diarization* half, resync torch against the ROCm index
|
|
169
|
+
(see the comment in `pyproject.toml`).
|
|
170
|
+
|
|
171
|
+
## Config
|
|
172
|
+
|
|
173
|
+
`~/.config/sttop/config.toml`. Run `sttop config` to write a default with every
|
|
174
|
+
knob and its documentation in it; the comments come from the source, so the file
|
|
175
|
+
never drifts from the code. Anything you leave out keeps its default, and blank
|
|
176
|
+
means "you decide" wherever a default is picked for you.
|
|
177
|
+
|
|
178
|
+
```toml
|
|
179
|
+
sessions_dir = "~/.local/share/sttop/sessions"
|
|
180
|
+
|
|
181
|
+
[audio]
|
|
182
|
+
mic_source = "" # substring match against pactl source names; blank = default
|
|
183
|
+
system_source = "" # blank = monitor of the default sink
|
|
184
|
+
save_wav = false
|
|
185
|
+
|
|
186
|
+
[vad]
|
|
187
|
+
aggressiveness = 2 # 0 permissive .. 3 strict
|
|
188
|
+
silence_ms = 700
|
|
189
|
+
max_segment_s = 15.0
|
|
190
|
+
|
|
191
|
+
[stt]
|
|
192
|
+
backend = "parakeet" # parakeet | whisper
|
|
193
|
+
model = "" # blank = the backend's default model
|
|
194
|
+
device = "auto" # whisper only
|
|
195
|
+
language = "" # blank = autodetect
|
|
196
|
+
|
|
197
|
+
[ui]
|
|
198
|
+
theme = "auto" # auto follows your terminal; or gruvbox, nord, ...
|
|
199
|
+
|
|
200
|
+
[diarize]
|
|
201
|
+
enabled = true
|
|
202
|
+
threshold = 0.50 # lower = fewer, broader speakers
|
|
203
|
+
margin = 0.15 # grey zone that attaches instead of opening a speaker
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## Theming
|
|
207
|
+
|
|
208
|
+
By default sttop paints with the `ansi-dark` / `ansi-light` Textual themes, which
|
|
209
|
+
use only the terminal's own 16 ANSI colours — so it inherits whatever palette you
|
|
210
|
+
already have rather than imposing its own. Which of the two is picked by reading
|
|
211
|
+
`COLORFGBG`, and failing that by asking the terminal for its background colour over
|
|
212
|
+
OSC 11 (supported by ghostty, kitty, alacritty, wezterm, foot, xterm). If nothing
|
|
213
|
+
answers, it assumes dark. Run `sttop theme` to see what was detected.
|
|
214
|
+
|
|
215
|
+
Set `ui.theme` to any Textual theme name (`gruvbox`, `nord`, `catppuccin-mocha`,
|
|
216
|
+
`solarized-light`, …) to override the terminal-following behaviour.
|
|
217
|
+
|
|
218
|
+
## Tests
|
|
219
|
+
|
|
220
|
+
```bash
|
|
221
|
+
uv run --extra dev pytest
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
The audio-dependent path is exercised by `tests/test_pipeline.py`, which plays a speech
|
|
225
|
+
sample into the default sink and reads it back off the monitor. It needs real audio
|
|
226
|
+
hardware and downloads a model, so it is opt-in:
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
STTOP_INTEGRATION=1 uv run --extra dev pytest
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
## License
|
|
233
|
+
|
|
234
|
+
MIT
|
sttop-0.1.0/README.md
ADDED
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
# sttop
|
|
2
|
+
|
|
3
|
+
Live speech-to-text monitor for the terminal — `htop`, but for what is being said.
|
|
4
|
+
|
|
5
|
+
Taps your **microphone** and your **system audio output** as two independent streams,
|
|
6
|
+
transcribes both in real time, labels who is speaking, and appends every line to a
|
|
7
|
+
Markdown file as it happens. Fully local: no network, no API keys, nothing leaves
|
|
8
|
+
the machine.
|
|
9
|
+
|
|
10
|
+
```
|
|
11
|
+
sttop ● rec 04:12 mic ███······· sys ██████···· parakeet-tdt/cpu onnx · ecapa @0.50 · queue 0 · 37 lines · 3 voices
|
|
12
|
+
writing → ~/.local/share/sttop/sessions/2026-08-10-1432-standup.md
|
|
13
|
+
|
|
14
|
+
03:58 you so the migration lands friday?
|
|
15
|
+
04:02 spk1 friday is tight, monday is safer
|
|
16
|
+
04:09 spk2 +1 on monday
|
|
17
|
+
|
|
18
|
+
q quit space pause r rename speaker
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
## Why two streams
|
|
22
|
+
|
|
23
|
+
Capturing the mic and the speaker output separately means **you** are identified for
|
|
24
|
+
free — anything on the mic is you, no model required, never wrong. Voice embeddings
|
|
25
|
+
then only have to split the *remote* side into individual participants, which is a
|
|
26
|
+
much easier problem than diarizing a single mixed track.
|
|
27
|
+
|
|
28
|
+
## Install
|
|
29
|
+
|
|
30
|
+
Needs `ffmpeg` and PipeWire or PulseAudio (`pactl`).
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
uvx --index https://download.pytorch.org/whl/cpu sttop
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
The `--index` flag matters. Speaker labelling needs torch, and the stock PyPI torch
|
|
37
|
+
bundles CUDA — about 2.5GB of nvidia wheels that buy nothing here, since
|
|
38
|
+
CTranslate2 has no ROCm backend and CPU inference keeps up with live audio fine.
|
|
39
|
+
The flag points torch at the CPU builds and falls back to PyPI for everything else.
|
|
40
|
+
|
|
41
|
+
To install it permanently rather than running it ad hoc:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
uv tool install --index https://download.pytorch.org/whl/cpu sttop
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
From a checkout, `uv sync` reads the CPU index out of `pyproject.toml` already:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
git clone https://github.com/v4rgas/sttop && cd sttop
|
|
51
|
+
uv sync
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
## Use
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
uv run sttop # record with defaults
|
|
58
|
+
uv run sttop -t "standup" # title the session (used in the filename)
|
|
59
|
+
uv run sttop --backend whisper -m small
|
|
60
|
+
uv run sttop devices --test # list audio sources, record 1s from each
|
|
61
|
+
uv run sttop sessions # list past transcripts
|
|
62
|
+
uv run sttop config # write ~/.config/sttop/config.toml
|
|
63
|
+
uv run sttop theme # show the detected terminal colour scheme
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Keys: `q` quit · `space` pause · `r` rename a speaker (`spk1=Ana`, rewrites past lines too).
|
|
67
|
+
|
|
68
|
+
## Output
|
|
69
|
+
|
|
70
|
+
One Markdown file per session in `~/.local/share/sttop/sessions/`, flushed after every
|
|
71
|
+
line — kill it mid-meeting and the transcript so far is already on disk.
|
|
72
|
+
|
|
73
|
+
```markdown
|
|
74
|
+
# standup
|
|
75
|
+
|
|
76
|
+
- started: 2026-08-10 14:32:01 -04
|
|
77
|
+
- mic: `alsa_input.pci-0000_08_00.6.analog-stereo`
|
|
78
|
+
- system: `alsa_output.pci-0000_08_00.6.analog-stereo.monitor`
|
|
79
|
+
- backend: `parakeet-tdt/cpu onnx`
|
|
80
|
+
|
|
81
|
+
## Transcript
|
|
82
|
+
|
|
83
|
+
- `03:58` **you** — so the migration lands friday?
|
|
84
|
+
|
|
85
|
+
- `04:02` **spk1** — friday is tight, monday is safer
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
## How it works
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
ffmpeg -f pulse (mic) ─┐
|
|
92
|
+
├─ webrtcvad segmenter ─→ queue ─→ parakeet ─→ ecapa ─→ journal.md
|
|
93
|
+
ffmpeg -f pulse (monitor) ─┘
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Audio is cut into utterances by voice-activity detection (a segment closes after
|
|
97
|
+
700 ms of silence, or at 15 s for a monologue), and only speech reaches the model.
|
|
98
|
+
|
|
99
|
+
**One thread boundary, and it is the model.** The two capture readers and the
|
|
100
|
+
consumer are asyncio tasks — they are blocking pipe I/O, which is what an event
|
|
101
|
+
loop is for — while transcription and voice embedding run in a single-worker
|
|
102
|
+
`ThreadPoolExecutor`. So the UI needs no cross-thread marshalling, shutdown is
|
|
103
|
+
ordinary task cancellation, and utterances stay in the order they were spoken. The
|
|
104
|
+
executor is single-worker on purpose: transcription is CPU-bound and already
|
|
105
|
+
internally parallel, so a second worker would only thrash the cache. When it falls
|
|
106
|
+
behind, the queue absorbs the lag — visible as `queue N` in the status bar — rather
|
|
107
|
+
than dropping audio.
|
|
108
|
+
|
|
109
|
+
Speaker labels come from online clustering of ECAPA-TDNN voice embeddings: each
|
|
110
|
+
utterance is matched against running centroids by cosine similarity. A confident
|
|
111
|
+
match (≥ `threshold`) joins that speaker and updates its centroid; a near miss
|
|
112
|
+
(within `margin` below it) joins without touching the centroid; only a clearly
|
|
113
|
+
distant voice opens a new speaker. That hysteresis matters — without it a single
|
|
114
|
+
noisy embedding mints a phantom participant, and one person ends up spread across
|
|
115
|
+
`spk1`/`spk2`/`spk3`. Being online means labels are assigned as audio arrives and are
|
|
116
|
+
never revised, which is the price of real time. Segments under 1.5 s are too short to
|
|
117
|
+
embed reliably and inherit the previous speaker, or show as `spk?`.
|
|
118
|
+
|
|
119
|
+
## Backends
|
|
120
|
+
|
|
121
|
+
**parakeet** (default) — NVIDIA Parakeet TDT 0.6b v3 through onnxruntime. Multilingual
|
|
122
|
+
across 25 European languages with autodetection, punctuated output, and roughly 19×
|
|
123
|
+
real time on CPU. Needs neither torch nor the NeMo toolkit, since onnx-asr runs the
|
|
124
|
+
exported graph directly. On the same 11 s clip where `whisper tiny` produced a
|
|
125
|
+
hallucinated lead-in and lost its punctuation, Parakeet returned the sentence verbatim.
|
|
126
|
+
|
|
127
|
+
**whisper** — faster-whisper/CTranslate2, if you want Whisper's language coverage.
|
|
128
|
+
CTranslate2 ships **CUDA and CPU backends only — there is no ROCm build**, so on an AMD
|
|
129
|
+
GPU this runs on CPU no matter what torch reports. The device is detected at startup
|
|
130
|
+
(`cuda` if CTranslate2 sees one, else `cpu`) and shown in the status bar.
|
|
131
|
+
|
|
132
|
+
To push a Radeon card at the *diarization* half, resync torch against the ROCm index
|
|
133
|
+
(see the comment in `pyproject.toml`).
|
|
134
|
+
|
|
135
|
+
## Config
|
|
136
|
+
|
|
137
|
+
`~/.config/sttop/config.toml`. Run `sttop config` to write a default with every
|
|
138
|
+
knob and its documentation in it; the comments come from the source, so the file
|
|
139
|
+
never drifts from the code. Anything you leave out keeps its default, and blank
|
|
140
|
+
means "you decide" wherever a default is picked for you.
|
|
141
|
+
|
|
142
|
+
```toml
|
|
143
|
+
sessions_dir = "~/.local/share/sttop/sessions"
|
|
144
|
+
|
|
145
|
+
[audio]
|
|
146
|
+
mic_source = "" # substring match against pactl source names; blank = default
|
|
147
|
+
system_source = "" # blank = monitor of the default sink
|
|
148
|
+
save_wav = false
|
|
149
|
+
|
|
150
|
+
[vad]
|
|
151
|
+
aggressiveness = 2 # 0 permissive .. 3 strict
|
|
152
|
+
silence_ms = 700
|
|
153
|
+
max_segment_s = 15.0
|
|
154
|
+
|
|
155
|
+
[stt]
|
|
156
|
+
backend = "parakeet" # parakeet | whisper
|
|
157
|
+
model = "" # blank = the backend's default model
|
|
158
|
+
device = "auto" # whisper only
|
|
159
|
+
language = "" # blank = autodetect
|
|
160
|
+
|
|
161
|
+
[ui]
|
|
162
|
+
theme = "auto" # auto follows your terminal; or gruvbox, nord, ...
|
|
163
|
+
|
|
164
|
+
[diarize]
|
|
165
|
+
enabled = true
|
|
166
|
+
threshold = 0.50 # lower = fewer, broader speakers
|
|
167
|
+
margin = 0.15 # grey zone that attaches instead of opening a speaker
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
## Theming
|
|
171
|
+
|
|
172
|
+
By default sttop paints with the `ansi-dark` / `ansi-light` Textual themes, which
|
|
173
|
+
use only the terminal's own 16 ANSI colours — so it inherits whatever palette you
|
|
174
|
+
already have rather than imposing its own. Which of the two is picked by reading
|
|
175
|
+
`COLORFGBG`, and failing that by asking the terminal for its background colour over
|
|
176
|
+
OSC 11 (supported by ghostty, kitty, alacritty, wezterm, foot, xterm). If nothing
|
|
177
|
+
answers, it assumes dark. Run `sttop theme` to see what was detected.
|
|
178
|
+
|
|
179
|
+
Set `ui.theme` to any Textual theme name (`gruvbox`, `nord`, `catppuccin-mocha`,
|
|
180
|
+
`solarized-light`, …) to override the terminal-following behaviour.
|
|
181
|
+
|
|
182
|
+
## Tests
|
|
183
|
+
|
|
184
|
+
```bash
|
|
185
|
+
uv run --extra dev pytest
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
The audio-dependent path is exercised by `tests/test_pipeline.py`, which plays a speech
|
|
189
|
+
sample into the default sink and reads it back off the monitor. It needs real audio
|
|
190
|
+
hardware and downloads a model, so it is opt-in:
|
|
191
|
+
|
|
192
|
+
```bash
|
|
193
|
+
STTOP_INTEGRATION=1 uv run --extra dev pytest
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
## License
|
|
197
|
+
|
|
198
|
+
MIT
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "sttop"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Live speech-to-text monitor for the terminal. Taps mic + system audio, transcribes and labels speakers in real time, stores everything locally."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
authors = [{ name = "v4rgas" }]
|
|
9
|
+
keywords = ["speech-to-text", "transcription", "whisper", "diarization", "tui", "terminal"]
|
|
10
|
+
classifiers = [
|
|
11
|
+
"Development Status :: 3 - Alpha",
|
|
12
|
+
"Environment :: Console",
|
|
13
|
+
"Intended Audience :: End Users/Desktop",
|
|
14
|
+
"License :: OSI Approved :: MIT License",
|
|
15
|
+
"Operating System :: POSIX :: Linux",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Programming Language :: Python :: 3.13",
|
|
20
|
+
"Topic :: Multimedia :: Sound/Audio :: Speech",
|
|
21
|
+
"Topic :: Utilities",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"textual>=0.80",
|
|
25
|
+
"numpy>=1.24",
|
|
26
|
+
"faster-whisper>=1.0.3",
|
|
27
|
+
"webrtcvad-wheels>=2.0.14",
|
|
28
|
+
"platformdirs>=4.0",
|
|
29
|
+
"onnx-asr[cpu,hub]>=0.12",
|
|
30
|
+
# Speaker identification. Pulls torch, but the CPU wheels keep it manageable
|
|
31
|
+
# and a transcript without speaker labels is not worth much.
|
|
32
|
+
"torch>=2.2",
|
|
33
|
+
"torchaudio>=2.2",
|
|
34
|
+
"speechbrain>=1.0",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[project.optional-dependencies]
|
|
38
|
+
dev = [
|
|
39
|
+
"pytest>=8.0",
|
|
40
|
+
"ruff>=0.6",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
[project.scripts]
|
|
44
|
+
sttop = "sttop.__main__:main"
|
|
45
|
+
|
|
46
|
+
[project.urls]
|
|
47
|
+
Homepage = "https://github.com/v4rgas/sttop"
|
|
48
|
+
Repository = "https://github.com/v4rgas/sttop"
|
|
49
|
+
Issues = "https://github.com/v4rgas/sttop/issues"
|
|
50
|
+
|
|
51
|
+
# Default torch to the CPU wheels. The stock PyPI torch bundles CUDA (~2.5GB) and
|
|
52
|
+
# CTranslate2 (faster-whisper) has no ROCm backend anyway, so a GPU build buys
|
|
53
|
+
# nothing here. To use an AMD GPU for diarization, resync against
|
|
54
|
+
# https://download.pytorch.org/whl/rocm6.2 instead.
|
|
55
|
+
[[tool.uv.index]]
|
|
56
|
+
name = "pytorch-cpu"
|
|
57
|
+
url = "https://download.pytorch.org/whl/cpu"
|
|
58
|
+
explicit = true
|
|
59
|
+
|
|
60
|
+
[tool.uv.sources]
|
|
61
|
+
torch = { index = "pytorch-cpu" }
|
|
62
|
+
torchaudio = { index = "pytorch-cpu" }
|
|
63
|
+
|
|
64
|
+
[build-system]
|
|
65
|
+
requires = ["hatchling"]
|
|
66
|
+
build-backend = "hatchling.build"
|
|
67
|
+
|
|
68
|
+
[tool.hatch.build.targets.wheel]
|
|
69
|
+
packages = ["sttop"]
|
|
70
|
+
|
|
71
|
+
[tool.ruff]
|
|
72
|
+
line-length = 90
|
|
73
|
+
target-version = "py311"
|
|
74
|
+
|
|
75
|
+
[tool.ruff.lint]
|
|
76
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
77
|
+
# The wav and journal handles stay open for the length of a session by design,
|
|
78
|
+
# so a context manager cannot express their lifetime.
|
|
79
|
+
ignore = ["SIM115"]
|
|
80
|
+
|
|
81
|
+
[tool.pytest.ini_options]
|
|
82
|
+
testpaths = ["tests"]
|