sttop 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ name: ci
2
+
3
+ on:
4
+ push:
5
+ branches: [main, master]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ steps:
12
+ - uses: actions/checkout@v5
13
+ - uses: astral-sh/setup-uv@v6
14
+ with:
15
+ enable-cache: true
16
+ - run: sudo apt-get update && sudo apt-get install -y ffmpeg
17
+ - run: uv sync --extra dev
18
+ # The audio-hardware tests stay opt-in; CI has no sound server.
19
+ - run: uv run --extra dev pytest -q
20
+ - run: uv run ruff check .
@@ -0,0 +1,32 @@
1
+ name: release
2
+
3
+ on:
4
+ push:
5
+ tags: ["v*"]
6
+ workflow_dispatch:
7
+
8
+ jobs:
9
+ build:
10
+ runs-on: ubuntu-latest
11
+ steps:
12
+ - uses: actions/checkout@v5
13
+ - uses: astral-sh/setup-uv@v6
14
+ - run: uv build
15
+ - uses: actions/upload-artifact@v4
16
+ with:
17
+ name: dist
18
+ path: dist/
19
+
20
+ publish:
21
+ needs: build
22
+ runs-on: ubuntu-latest
23
+ environment: pypi
24
+ # Trusted publishing: PyPI verifies this workflow's identity, no token needed.
25
+ permissions:
26
+ id-token: write
27
+ steps:
28
+ - uses: actions/download-artifact@v4
29
+ with:
30
+ name: dist
31
+ path: dist/
32
+ - uses: pypa/gh-action-pypi-publish@release/v1
sttop-0.1.0/.gitignore ADDED
@@ -0,0 +1,10 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ dist/
6
+ build/
7
+ .pytest_cache/
8
+ sessions/
9
+ *.wav
10
+ dist/
sttop-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,234 @@
1
+ Metadata-Version: 2.4
2
+ Name: sttop
3
+ Version: 0.1.0
4
+ Summary: Live speech-to-text monitor for the terminal. Taps mic + system audio, transcribes and labels speakers in real time, stores everything locally.
5
+ Project-URL: Homepage, https://github.com/v4rgas/sttop
6
+ Project-URL: Repository, https://github.com/v4rgas/sttop
7
+ Project-URL: Issues, https://github.com/v4rgas/sttop/issues
8
+ Author: v4rgas
9
+ License: MIT
10
+ Keywords: diarization,speech-to-text,terminal,transcription,tui,whisper
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Environment :: Console
13
+ Classifier: Intended Audience :: End Users/Desktop
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: POSIX :: Linux
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Multimedia :: Sound/Audio :: Speech
21
+ Classifier: Topic :: Utilities
22
+ Requires-Python: >=3.11
23
+ Requires-Dist: faster-whisper>=1.0.3
24
+ Requires-Dist: numpy>=1.24
25
+ Requires-Dist: onnx-asr[cpu,hub]>=0.12
26
+ Requires-Dist: platformdirs>=4.0
27
+ Requires-Dist: speechbrain>=1.0
28
+ Requires-Dist: textual>=0.80
29
+ Requires-Dist: torch>=2.2
30
+ Requires-Dist: torchaudio>=2.2
31
+ Requires-Dist: webrtcvad-wheels>=2.0.14
32
+ Provides-Extra: dev
33
+ Requires-Dist: pytest>=8.0; extra == 'dev'
34
+ Requires-Dist: ruff>=0.6; extra == 'dev'
35
+ Description-Content-Type: text/markdown
36
+
37
+ # sttop
38
+
39
+ Live speech-to-text monitor for the terminal — `htop`, but for what is being said.
40
+
41
+ Taps your **microphone** and your **system audio output** as two independent streams,
42
+ transcribes both in real time, labels who is speaking, and appends every line to a
43
+ Markdown file as it happens. Fully local: no network, no API keys, nothing leaves
44
+ the machine.
45
+
46
+ ```
47
+ sttop ● rec 04:12 mic ███······· sys ██████···· parakeet-tdt/cpu onnx · ecapa @0.50 · queue 0 · 37 lines · 3 voices
48
+ writing → ~/.local/share/sttop/sessions/2026-08-10-1432-standup.md
49
+
50
+ 03:58 you so the migration lands friday?
51
+ 04:02 spk1 friday is tight, monday is safer
52
+ 04:09 spk2 +1 on monday
53
+
54
+ q quit space pause r rename speaker
55
+ ```
56
+
57
+ ## Why two streams
58
+
59
+ Capturing the mic and the speaker output separately means **you** are identified for
60
+ free — anything on the mic is you, no model required, never wrong. Voice embeddings
61
+ then only have to split the *remote* side into individual participants, which is a
62
+ much easier problem than diarizing a single mixed track.
63
+
64
+ ## Install
65
+
66
+ Needs `ffmpeg` and PipeWire or PulseAudio (`pactl`).
67
+
68
+ ```bash
69
+ uvx --index https://download.pytorch.org/whl/cpu sttop
70
+ ```
71
+
72
+ The `--index` flag matters. Speaker labelling needs torch, and the stock PyPI torch
73
+ bundles CUDA — about 2.5GB of nvidia wheels that buy nothing here, since
74
+ CTranslate2 has no ROCm backend and CPU inference keeps up with live audio fine.
75
+ The flag points torch at the CPU builds and falls back to PyPI for everything else.
76
+
77
+ To install it permanently rather than running it ad hoc:
78
+
79
+ ```bash
80
+ uv tool install --index https://download.pytorch.org/whl/cpu sttop
81
+ ```
82
+
83
+ From a checkout, `uv sync` reads the CPU index out of `pyproject.toml` already:
84
+
85
+ ```bash
86
+ git clone https://github.com/v4rgas/sttop && cd sttop
87
+ uv sync
88
+ ```
89
+
90
+ ## Use
91
+
92
+ ```bash
93
+ uv run sttop # record with defaults
94
+ uv run sttop -t "standup" # title the session (used in the filename)
95
+ uv run sttop --backend whisper -m small
96
+ uv run sttop devices --test # list audio sources, record 1s from each
97
+ uv run sttop sessions # list past transcripts
98
+ uv run sttop config # write ~/.config/sttop/config.toml
99
+ uv run sttop theme # show the detected terminal colour scheme
100
+ ```
101
+
102
+ Keys: `q` quit · `space` pause · `r` rename a speaker (`spk1=Ana`, rewrites past lines too).
103
+
104
+ ## Output
105
+
106
+ One Markdown file per session in `~/.local/share/sttop/sessions/`, flushed after every
107
+ line — kill it mid-meeting and the transcript so far is already on disk.
108
+
109
+ ```markdown
110
+ # standup
111
+
112
+ - started: 2026-08-10 14:32:01 -04
113
+ - mic: `alsa_input.pci-0000_08_00.6.analog-stereo`
114
+ - system: `alsa_output.pci-0000_08_00.6.analog-stereo.monitor`
115
+ - backend: `parakeet-tdt/cpu onnx`
116
+
117
+ ## Transcript
118
+
119
+ - `03:58` **you** — so the migration lands friday?
120
+
121
+ - `04:02` **spk1** — friday is tight, monday is safer
122
+ ```
123
+
124
+ ## How it works
125
+
126
+ ```
127
+ ffmpeg -f pulse (mic) ─┐
128
+ ├─ webrtcvad segmenter ─→ queue ─→ parakeet ─→ ecapa ─→ journal.md
129
+ ffmpeg -f pulse (monitor) ─┘
130
+ ```
131
+
132
+ Audio is cut into utterances by voice-activity detection (a segment closes after
133
+ 700 ms of silence, or at 15 s for a monologue), and only speech reaches the model.
134
+
135
+ **One thread boundary, and it is the model.** The two capture readers and the
136
+ consumer are asyncio tasks — they are blocking pipe I/O, which is what an event
137
+ loop is for — while transcription and voice embedding run in a single-worker
138
+ `ThreadPoolExecutor`. So the UI needs no cross-thread marshalling, shutdown is
139
+ ordinary task cancellation, and utterances stay in the order they were spoken. The
140
+ executor is single-worker on purpose: transcription is CPU-bound and already
141
+ internally parallel, so a second worker would only thrash the cache. When it falls
142
+ behind, the queue absorbs the lag — visible as `queue N` in the status bar — rather
143
+ than dropping audio.
144
+
145
+ Speaker labels come from online clustering of ECAPA-TDNN voice embeddings: each
146
+ utterance is matched against running centroids by cosine similarity. A confident
147
+ match (≥ `threshold`) joins that speaker and updates its centroid; a near miss
148
+ (within `margin` below it) joins without touching the centroid; only a clearly
149
+ distant voice opens a new speaker. That hysteresis matters — without it a single
150
+ noisy embedding mints a phantom participant, and one person ends up spread across
151
+ `spk1`/`spk2`/`spk3`. Being online means labels are assigned as audio arrives and are
152
+ never revised, which is the price of real time. Segments under 1.5 s are too short to
153
+ embed reliably and inherit the previous speaker, or show as `spk?`.
154
+
155
+ ## Backends
156
+
157
+ **parakeet** (default) — NVIDIA Parakeet TDT 0.6b v3 through onnxruntime. Multilingual
158
+ across 25 European languages with autodetection, punctuated output, and roughly 19×
159
+ real time on CPU. Needs neither torch nor the NeMo toolkit, since onnx-asr runs the
160
+ exported graph directly. On the same 11 s clip where `whisper tiny` produced a
161
+ hallucinated lead-in and lost its punctuation, Parakeet returned the sentence verbatim.
162
+
163
+ **whisper** — faster-whisper/CTranslate2, if you want Whisper's language coverage.
164
+ CTranslate2 ships **CUDA and CPU backends only — there is no ROCm build**, so on an AMD
165
+ GPU this runs on CPU no matter what torch reports. The device is detected at startup
166
+ (`cuda` if CTranslate2 sees one, else `cpu`) and shown in the status bar.
167
+
168
+ To push a Radeon card at the *diarization* half, resync torch against the ROCm index
169
+ (see the comment in `pyproject.toml`).
170
+
171
+ ## Config
172
+
173
+ `~/.config/sttop/config.toml`. Run `sttop config` to write a default with every
174
+ knob and its documentation in it; the comments come from the source, so the file
175
+ never drifts from the code. Anything you leave out keeps its default, and blank
176
+ means "you decide" wherever a default is picked for you.
177
+
178
+ ```toml
179
+ sessions_dir = "~/.local/share/sttop/sessions"
180
+
181
+ [audio]
182
+ mic_source = "" # substring match against pactl source names; blank = default
183
+ system_source = "" # blank = monitor of the default sink
184
+ save_wav = false
185
+
186
+ [vad]
187
+ aggressiveness = 2 # 0 permissive .. 3 strict
188
+ silence_ms = 700
189
+ max_segment_s = 15.0
190
+
191
+ [stt]
192
+ backend = "parakeet" # parakeet | whisper
193
+ model = "" # blank = the backend's default model
194
+ device = "auto" # whisper only
195
+ language = "" # blank = autodetect
196
+
197
+ [ui]
198
+ theme = "auto" # auto follows your terminal; or gruvbox, nord, ...
199
+
200
+ [diarize]
201
+ enabled = true
202
+ threshold = 0.50 # lower = fewer, broader speakers
203
+ margin = 0.15 # grey zone that attaches instead of opening a speaker
204
+ ```
205
+
206
+ ## Theming
207
+
208
+ By default sttop paints with the `ansi-dark` / `ansi-light` Textual themes, which
209
+ use only the terminal's own 16 ANSI colours — so it inherits whatever palette you
210
+ already have rather than imposing its own. Which of the two is picked by reading
211
+ `COLORFGBG`, and failing that by asking the terminal for its background colour over
212
+ OSC 11 (supported by ghostty, kitty, alacritty, wezterm, foot, xterm). If nothing
213
+ answers, it assumes dark. Run `sttop theme` to see what was detected.
214
+
215
+ Set `ui.theme` to any Textual theme name (`gruvbox`, `nord`, `catppuccin-mocha`,
216
+ `solarized-light`, …) to override the terminal-following behaviour.
217
+
218
+ ## Tests
219
+
220
+ ```bash
221
+ uv run --extra dev pytest
222
+ ```
223
+
224
+ The audio-dependent path is exercised by `tests/test_pipeline.py`, which plays a speech
225
+ sample into the default sink and reads it back off the monitor. It needs real audio
226
+ hardware and downloads a model, so it is opt-in:
227
+
228
+ ```bash
229
+ STTOP_INTEGRATION=1 uv run --extra dev pytest
230
+ ```
231
+
232
+ ## License
233
+
234
+ MIT
sttop-0.1.0/README.md ADDED
@@ -0,0 +1,198 @@
1
+ # sttop
2
+
3
+ Live speech-to-text monitor for the terminal — `htop`, but for what is being said.
4
+
5
+ Taps your **microphone** and your **system audio output** as two independent streams,
6
+ transcribes both in real time, labels who is speaking, and appends every line to a
7
+ Markdown file as it happens. Fully local: no network, no API keys, nothing leaves
8
+ the machine.
9
+
10
+ ```
11
+ sttop ● rec 04:12 mic ███······· sys ██████···· parakeet-tdt/cpu onnx · ecapa @0.50 · queue 0 · 37 lines · 3 voices
12
+ writing → ~/.local/share/sttop/sessions/2026-08-10-1432-standup.md
13
+
14
+ 03:58 you so the migration lands friday?
15
+ 04:02 spk1 friday is tight, monday is safer
16
+ 04:09 spk2 +1 on monday
17
+
18
+ q quit space pause r rename speaker
19
+ ```
20
+
21
+ ## Why two streams
22
+
23
+ Capturing the mic and the speaker output separately means **you** are identified for
24
+ free — anything on the mic is you, no model required, never wrong. Voice embeddings
25
+ then only have to split the *remote* side into individual participants, which is a
26
+ much easier problem than diarizing a single mixed track.
27
+
28
+ ## Install
29
+
30
+ Needs `ffmpeg` and PipeWire or PulseAudio (`pactl`).
31
+
32
+ ```bash
33
+ uvx --index https://download.pytorch.org/whl/cpu sttop
34
+ ```
35
+
36
+ The `--index` flag matters. Speaker labelling needs torch, and the stock PyPI torch
37
+ bundles CUDA — about 2.5GB of nvidia wheels that buy nothing here, since
38
+ CTranslate2 has no ROCm backend and CPU inference keeps up with live audio fine.
39
+ The flag points torch at the CPU builds and falls back to PyPI for everything else.
40
+
41
+ To install it permanently rather than running it ad hoc:
42
+
43
+ ```bash
44
+ uv tool install --index https://download.pytorch.org/whl/cpu sttop
45
+ ```
46
+
47
+ From a checkout, `uv sync` reads the CPU index out of `pyproject.toml` already:
48
+
49
+ ```bash
50
+ git clone https://github.com/v4rgas/sttop && cd sttop
51
+ uv sync
52
+ ```
53
+
54
+ ## Use
55
+
56
+ ```bash
57
+ uv run sttop # record with defaults
58
+ uv run sttop -t "standup" # title the session (used in the filename)
59
+ uv run sttop --backend whisper -m small
60
+ uv run sttop devices --test # list audio sources, record 1s from each
61
+ uv run sttop sessions # list past transcripts
62
+ uv run sttop config # write ~/.config/sttop/config.toml
63
+ uv run sttop theme # show the detected terminal colour scheme
64
+ ```
65
+
66
+ Keys: `q` quit · `space` pause · `r` rename a speaker (`spk1=Ana`, rewrites past lines too).
67
+
68
+ ## Output
69
+
70
+ One Markdown file per session in `~/.local/share/sttop/sessions/`, flushed after every
71
+ line — kill it mid-meeting and the transcript so far is already on disk.
72
+
73
+ ```markdown
74
+ # standup
75
+
76
+ - started: 2026-08-10 14:32:01 -04
77
+ - mic: `alsa_input.pci-0000_08_00.6.analog-stereo`
78
+ - system: `alsa_output.pci-0000_08_00.6.analog-stereo.monitor`
79
+ - backend: `parakeet-tdt/cpu onnx`
80
+
81
+ ## Transcript
82
+
83
+ - `03:58` **you** — so the migration lands friday?
84
+
85
+ - `04:02` **spk1** — friday is tight, monday is safer
86
+ ```
87
+
88
+ ## How it works
89
+
90
+ ```
91
+ ffmpeg -f pulse (mic) ─┐
92
+ ├─ webrtcvad segmenter ─→ queue ─→ parakeet ─→ ecapa ─→ journal.md
93
+ ffmpeg -f pulse (monitor) ─┘
94
+ ```
95
+
96
+ Audio is cut into utterances by voice-activity detection (a segment closes after
97
+ 700 ms of silence, or at 15 s for a monologue), and only speech reaches the model.
98
+
99
+ **One thread boundary, and it is the model.** The two capture readers and the
100
+ consumer are asyncio tasks — they are blocking pipe I/O, which is what an event
101
+ loop is for — while transcription and voice embedding run in a single-worker
102
+ `ThreadPoolExecutor`. So the UI needs no cross-thread marshalling, shutdown is
103
+ ordinary task cancellation, and utterances stay in the order they were spoken. The
104
+ executor is single-worker on purpose: transcription is CPU-bound and already
105
+ internally parallel, so a second worker would only thrash the cache. When it falls
106
+ behind, the queue absorbs the lag — visible as `queue N` in the status bar — rather
107
+ than dropping audio.
108
+
109
+ Speaker labels come from online clustering of ECAPA-TDNN voice embeddings: each
110
+ utterance is matched against running centroids by cosine similarity. A confident
111
+ match (≥ `threshold`) joins that speaker and updates its centroid; a near miss
112
+ (within `margin` below it) joins without touching the centroid; only a clearly
113
+ distant voice opens a new speaker. That hysteresis matters — without it a single
114
+ noisy embedding mints a phantom participant, and one person ends up spread across
115
+ `spk1`/`spk2`/`spk3`. Being online means labels are assigned as audio arrives and are
116
+ never revised, which is the price of real time. Segments under 1.5 s are too short to
117
+ embed reliably and inherit the previous speaker, or show as `spk?`.
118
+
119
+ ## Backends
120
+
121
+ **parakeet** (default) — NVIDIA Parakeet TDT 0.6b v3 through onnxruntime. Multilingual
122
+ across 25 European languages with autodetection, punctuated output, and roughly 19×
123
+ real time on CPU. Needs neither torch nor the NeMo toolkit, since onnx-asr runs the
124
+ exported graph directly. On the same 11 s clip where `whisper tiny` produced a
125
+ hallucinated lead-in and lost its punctuation, Parakeet returned the sentence verbatim.
126
+
127
+ **whisper** — faster-whisper/CTranslate2, if you want Whisper's language coverage.
128
+ CTranslate2 ships **CUDA and CPU backends only — there is no ROCm build**, so on an AMD
129
+ GPU this runs on CPU no matter what torch reports. The device is detected at startup
130
+ (`cuda` if CTranslate2 sees one, else `cpu`) and shown in the status bar.
131
+
132
+ To push a Radeon card at the *diarization* half, resync torch against the ROCm index
133
+ (see the comment in `pyproject.toml`).
134
+
135
+ ## Config
136
+
137
+ `~/.config/sttop/config.toml`. Run `sttop config` to write a default with every
138
+ knob and its documentation in it; the comments come from the source, so the file
139
+ never drifts from the code. Anything you leave out keeps its default, and blank
140
+ means "you decide" wherever a default is picked for you.
141
+
142
+ ```toml
143
+ sessions_dir = "~/.local/share/sttop/sessions"
144
+
145
+ [audio]
146
+ mic_source = "" # substring match against pactl source names; blank = default
147
+ system_source = "" # blank = monitor of the default sink
148
+ save_wav = false
149
+
150
+ [vad]
151
+ aggressiveness = 2 # 0 permissive .. 3 strict
152
+ silence_ms = 700
153
+ max_segment_s = 15.0
154
+
155
+ [stt]
156
+ backend = "parakeet" # parakeet | whisper
157
+ model = "" # blank = the backend's default model
158
+ device = "auto" # whisper only
159
+ language = "" # blank = autodetect
160
+
161
+ [ui]
162
+ theme = "auto" # auto follows your terminal; or gruvbox, nord, ...
163
+
164
+ [diarize]
165
+ enabled = true
166
+ threshold = 0.50 # lower = fewer, broader speakers
167
+ margin = 0.15 # grey zone that attaches instead of opening a speaker
168
+ ```
169
+
170
+ ## Theming
171
+
172
+ By default sttop paints with the `ansi-dark` / `ansi-light` Textual themes, which
173
+ use only the terminal's own 16 ANSI colours — so it inherits whatever palette you
174
+ already have rather than imposing its own. Which of the two is picked by reading
175
+ `COLORFGBG`, and failing that by asking the terminal for its background colour over
176
+ OSC 11 (supported by ghostty, kitty, alacritty, wezterm, foot, xterm). If nothing
177
+ answers, it assumes dark. Run `sttop theme` to see what was detected.
178
+
179
+ Set `ui.theme` to any Textual theme name (`gruvbox`, `nord`, `catppuccin-mocha`,
180
+ `solarized-light`, …) to override the terminal-following behaviour.
181
+
182
+ ## Tests
183
+
184
+ ```bash
185
+ uv run --extra dev pytest
186
+ ```
187
+
188
+ The audio-dependent path is exercised by `tests/test_pipeline.py`, which plays a speech
189
+ sample into the default sink and reads it back off the monitor. It needs real audio
190
+ hardware and downloads a model, so it is opt-in:
191
+
192
+ ```bash
193
+ STTOP_INTEGRATION=1 uv run --extra dev pytest
194
+ ```
195
+
196
+ ## License
197
+
198
+ MIT
@@ -0,0 +1,82 @@
1
+ [project]
2
+ name = "sttop"
3
+ version = "0.1.0"
4
+ description = "Live speech-to-text monitor for the terminal. Taps mic + system audio, transcribes and labels speakers in real time, stores everything locally."
5
+ readme = "README.md"
6
+ requires-python = ">=3.11"
7
+ license = { text = "MIT" }
8
+ authors = [{ name = "v4rgas" }]
9
+ keywords = ["speech-to-text", "transcription", "whisper", "diarization", "tui", "terminal"]
10
+ classifiers = [
11
+ "Development Status :: 3 - Alpha",
12
+ "Environment :: Console",
13
+ "Intended Audience :: End Users/Desktop",
14
+ "License :: OSI Approved :: MIT License",
15
+ "Operating System :: POSIX :: Linux",
16
+ "Programming Language :: Python :: 3",
17
+ "Programming Language :: Python :: 3.11",
18
+ "Programming Language :: Python :: 3.12",
19
+ "Programming Language :: Python :: 3.13",
20
+ "Topic :: Multimedia :: Sound/Audio :: Speech",
21
+ "Topic :: Utilities",
22
+ ]
23
+ dependencies = [
24
+ "textual>=0.80",
25
+ "numpy>=1.24",
26
+ "faster-whisper>=1.0.3",
27
+ "webrtcvad-wheels>=2.0.14",
28
+ "platformdirs>=4.0",
29
+ "onnx-asr[cpu,hub]>=0.12",
30
+ # Speaker identification. Pulls torch, but the CPU wheels keep it manageable
31
+ # and a transcript without speaker labels is not worth much.
32
+ "torch>=2.2",
33
+ "torchaudio>=2.2",
34
+ "speechbrain>=1.0",
35
+ ]
36
+
37
+ [project.optional-dependencies]
38
+ dev = [
39
+ "pytest>=8.0",
40
+ "ruff>=0.6",
41
+ ]
42
+
43
+ [project.scripts]
44
+ sttop = "sttop.__main__:main"
45
+
46
+ [project.urls]
47
+ Homepage = "https://github.com/v4rgas/sttop"
48
+ Repository = "https://github.com/v4rgas/sttop"
49
+ Issues = "https://github.com/v4rgas/sttop/issues"
50
+
51
+ # Default torch to the CPU wheels. The stock PyPI torch bundles CUDA (~2.5GB) and
52
+ # CTranslate2 (faster-whisper) has no ROCm backend anyway, so a GPU build buys
53
+ # nothing here. To use an AMD GPU for diarization, resync against
54
+ # https://download.pytorch.org/whl/rocm6.2 instead.
55
+ [[tool.uv.index]]
56
+ name = "pytorch-cpu"
57
+ url = "https://download.pytorch.org/whl/cpu"
58
+ explicit = true
59
+
60
+ [tool.uv.sources]
61
+ torch = { index = "pytorch-cpu" }
62
+ torchaudio = { index = "pytorch-cpu" }
63
+
64
+ [build-system]
65
+ requires = ["hatchling"]
66
+ build-backend = "hatchling.build"
67
+
68
+ [tool.hatch.build.targets.wheel]
69
+ packages = ["sttop"]
70
+
71
+ [tool.ruff]
72
+ line-length = 90
73
+ target-version = "py311"
74
+
75
+ [tool.ruff.lint]
76
+ select = ["E", "F", "I", "UP", "B", "SIM"]
77
+ # The wav and journal handles stay open for the length of a session by design,
78
+ # so a context manager cannot express their lifetime.
79
+ ignore = ["SIM115"]
80
+
81
+ [tool.pytest.ini_options]
82
+ testpaths = ["tests"]
@@ -0,0 +1,8 @@
1
+ """sttop - live speech-to-text monitor for the terminal."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ SAMPLE_RATE = 16_000
6
+ FRAME_MS = 20
7
+ FRAME_SAMPLES = SAMPLE_RATE * FRAME_MS // 1000 # 320 samples
8
+ FRAME_BYTES = FRAME_SAMPLES * 2 # int16 mono