sttop 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sttop/__init__.py +8 -0
- sttop/__main__.py +193 -0
- sttop/audio/__init__.py +1 -0
- sttop/audio/capture.py +228 -0
- sttop/audio/devices.py +90 -0
- sttop/audio/segmenter.py +170 -0
- sttop/config.py +215 -0
- sttop/diarize.py +166 -0
- sttop/engine.py +269 -0
- sttop/journal.py +149 -0
- sttop/stt/__init__.py +60 -0
- sttop/stt/base.py +30 -0
- sttop/stt/local.py +89 -0
- sttop/stt/parakeet.py +29 -0
- sttop/terminal.py +144 -0
- sttop/tui.py +187 -0
- sttop-0.1.0.dist-info/METADATA +234 -0
- sttop-0.1.0.dist-info/RECORD +20 -0
- sttop-0.1.0.dist-info/WHEEL +4 -0
- sttop-0.1.0.dist-info/entry_points.txt +2 -0
sttop/__init__.py
ADDED
sttop/__main__.py
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Command line entry point."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import sys
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from . import __version__
|
|
10
|
+
from .config import CONFIG_PATH, Config, ConfigError, write_default_config
|
|
11
|
+
from .stt import BACKENDS
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
15
|
+
parser = argparse.ArgumentParser(
|
|
16
|
+
prog="sttop",
|
|
17
|
+
description="Live speech-to-text monitor. Taps mic + system audio, "
|
|
18
|
+
"transcribes and labels speakers in real time, writes Markdown.",
|
|
19
|
+
)
|
|
20
|
+
parser.add_argument("--version", action="version", version=f"sttop {__version__}")
|
|
21
|
+
parser.add_argument("-c", "--config", type=Path, help=f"default: {CONFIG_PATH}")
|
|
22
|
+
|
|
23
|
+
# `--config` reads as a global option, so accept it on either side of the
|
|
24
|
+
# subcommand. SUPPRESS is what makes that work: without it the subcommand's
|
|
25
|
+
# own default would overwrite a value already given before the subcommand.
|
|
26
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
27
|
+
common.add_argument(
|
|
28
|
+
"-c", "--config", type=Path, default=argparse.SUPPRESS, help=argparse.SUPPRESS
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
sub = parser.add_subparsers(dest="command")
|
|
32
|
+
|
|
33
|
+
def command(name: str, summary: str) -> argparse.ArgumentParser:
|
|
34
|
+
return sub.add_parser(name, help=summary, parents=[common])
|
|
35
|
+
|
|
36
|
+
record = command("record", "start a session (default)")
|
|
37
|
+
record.add_argument("-t", "--title", help="session title, used in the filename")
|
|
38
|
+
record.add_argument("--mic", help="mic source name or substring")
|
|
39
|
+
record.add_argument("--system", help="system/monitor source name or substring")
|
|
40
|
+
record.add_argument("-m", "--model", help="override the backend's default model")
|
|
41
|
+
record.add_argument("--backend", choices=list(BACKENDS))
|
|
42
|
+
record.add_argument(
|
|
43
|
+
"--language", help="force a language, e.g. es (default: autodetect)"
|
|
44
|
+
)
|
|
45
|
+
record.add_argument("--no-diarize", action="store_true", help="skip speaker id")
|
|
46
|
+
record.add_argument("--save-wav", action="store_true", help="keep the raw audio")
|
|
47
|
+
|
|
48
|
+
devices_cmd = command("devices", "list audio sources")
|
|
49
|
+
devices_cmd.add_argument(
|
|
50
|
+
"--test", action="store_true", help="record 1s from each and report levels"
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
command("sessions", "list recorded sessions")
|
|
54
|
+
command("theme", "show the detected terminal colour scheme")
|
|
55
|
+
command("config", "write a default config file")
|
|
56
|
+
|
|
57
|
+
return parser
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
#: Global options that may appear before the subcommand, and whether the option
|
|
61
|
+
#: swallows the token after it.
|
|
62
|
+
_GLOBAL_OPTIONS = {"-c": True, "--config": True, "-h": False, "--help": False,
|
|
63
|
+
"--version": False}
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def with_default_command(argv: list[str], commands: set[str]) -> list[str]:
|
|
67
|
+
"""Insert `record` when no subcommand was given.
|
|
68
|
+
|
|
69
|
+
`sttop -t standup` means `sttop record -t standup`. Rather than parse twice
|
|
70
|
+
and hope the first attempt fails cleanly - it does not, since `-t` is a
|
|
71
|
+
record option and argparse rejects it outright - the command is filled in
|
|
72
|
+
before parsing, so there is only ever one well-formed parse.
|
|
73
|
+
"""
|
|
74
|
+
index = 0
|
|
75
|
+
while index < len(argv):
|
|
76
|
+
token = argv[index]
|
|
77
|
+
if token in commands:
|
|
78
|
+
return argv
|
|
79
|
+
option, joined, _ = token.partition("=")
|
|
80
|
+
if option not in _GLOBAL_OPTIONS:
|
|
81
|
+
break # a record option, or a positional - record starts here
|
|
82
|
+
# Skip the global option, plus its value when given as a separate
|
|
83
|
+
# token (`-c path`) rather than joined on (`--config=path`).
|
|
84
|
+
index += 2 if _GLOBAL_OPTIONS[option] and not joined else 1
|
|
85
|
+
return [*argv[:index], "record", *argv[index:]]
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def cmd_devices(config: Config, args) -> int:
|
|
89
|
+
from .audio import capture, devices
|
|
90
|
+
|
|
91
|
+
try:
|
|
92
|
+
sources = devices.list_sources()
|
|
93
|
+
mic = devices.resolve(config.audio.mic_source, monitor=False)
|
|
94
|
+
system = devices.resolve(config.audio.system_source, monitor=True)
|
|
95
|
+
except devices.AudioError as exc:
|
|
96
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
97
|
+
return 1
|
|
98
|
+
|
|
99
|
+
print(f"mic -> {mic}")
|
|
100
|
+
print(f"system -> {system}\n")
|
|
101
|
+
for source in sources:
|
|
102
|
+
role = "mic" if source.name == mic else "sys" if source.name == system else " "
|
|
103
|
+
kind = "monitor" if source.is_monitor else "input "
|
|
104
|
+
line = f"{role} {kind} {source.state:<10} {source.name}"
|
|
105
|
+
if args.test:
|
|
106
|
+
try:
|
|
107
|
+
peak = capture.check_source(source.name, seconds=1.0)
|
|
108
|
+
line += f" peak {peak:.3f}" + ("" if peak > 0.001 else " (silent)")
|
|
109
|
+
except Exception as exc:
|
|
110
|
+
line += f" [failed: {str(exc).splitlines()[0][:40]}]"
|
|
111
|
+
print(line)
|
|
112
|
+
return 0
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def cmd_theme(config: Config) -> int:
|
|
116
|
+
from .terminal import detect_theme, theme_sources
|
|
117
|
+
|
|
118
|
+
for name, verdict in theme_sources():
|
|
119
|
+
print(f"{name:<12}{verdict or '<no answer>'}")
|
|
120
|
+
print(f"configured {config.ui.theme}")
|
|
121
|
+
print(f"\nusing {detect_theme(config.ui.theme)}")
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def cmd_sessions(config: Config) -> int:
|
|
126
|
+
from .journal import list_sessions
|
|
127
|
+
|
|
128
|
+
directory = Path(config.sessions_dir)
|
|
129
|
+
paths = list_sessions(directory)
|
|
130
|
+
if not paths:
|
|
131
|
+
print(f"no sessions yet in {directory}")
|
|
132
|
+
return 0
|
|
133
|
+
for path in paths:
|
|
134
|
+
size = path.stat().st_size
|
|
135
|
+
print(f"{path.name:<52} {size / 1024:6.1f} KiB")
|
|
136
|
+
print(f"\n{len(paths)} session(s) in {directory}")
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def cmd_record(config: Config, args) -> int:
|
|
141
|
+
from .tui import SttopApp
|
|
142
|
+
|
|
143
|
+
if args.mic:
|
|
144
|
+
config.audio.mic_source = args.mic
|
|
145
|
+
if args.system:
|
|
146
|
+
config.audio.system_source = args.system
|
|
147
|
+
if args.model:
|
|
148
|
+
config.stt.model = args.model
|
|
149
|
+
if args.backend:
|
|
150
|
+
config.stt.backend = args.backend
|
|
151
|
+
if args.language:
|
|
152
|
+
config.stt.language = args.language
|
|
153
|
+
if args.no_diarize:
|
|
154
|
+
config.diarize.enabled = False
|
|
155
|
+
if args.save_wav:
|
|
156
|
+
config.audio.save_wav = True
|
|
157
|
+
|
|
158
|
+
path = SttopApp(config, args.title).run()
|
|
159
|
+
if path:
|
|
160
|
+
print(f"transcript: {path}")
|
|
161
|
+
return 0
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
COMMANDS = {"record", "devices", "sessions", "theme", "config"}
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def main(argv: list[str] | None = None) -> int:
|
|
168
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
169
|
+
parser = build_parser()
|
|
170
|
+
args = parser.parse_args(with_default_command(argv, COMMANDS))
|
|
171
|
+
|
|
172
|
+
if args.command == "config": # writing a config must not require a valid one
|
|
173
|
+
path = write_default_config(args.config)
|
|
174
|
+
print(f"wrote {path}")
|
|
175
|
+
return 0
|
|
176
|
+
|
|
177
|
+
try:
|
|
178
|
+
config = Config.load(args.config)
|
|
179
|
+
except ConfigError as exc:
|
|
180
|
+
print(f"error: bad config: {exc}", file=sys.stderr)
|
|
181
|
+
return 1
|
|
182
|
+
|
|
183
|
+
if args.command == "devices":
|
|
184
|
+
return cmd_devices(config, args)
|
|
185
|
+
if args.command == "sessions":
|
|
186
|
+
return cmd_sessions(config)
|
|
187
|
+
if args.command == "theme":
|
|
188
|
+
return cmd_theme(config)
|
|
189
|
+
return cmd_record(config, args)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
if __name__ == "__main__":
|
|
193
|
+
sys.exit(main())
|
sttop/audio/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Audio capture and voice-activity segmentation."""
|
sttop/audio/capture.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""Capture a PulseAudio/PipeWire source as 16 kHz mono PCM via ffmpeg.
|
|
2
|
+
|
|
3
|
+
ffmpeg is used instead of a PortAudio binding because monitor sources (the
|
|
4
|
+
"what you hear" side of the capture) are exposed cleanly by the pulse backend,
|
|
5
|
+
whereas PortAudio device indices for monitors are inconsistent under PipeWire.
|
|
6
|
+
|
|
7
|
+
Reading is an asyncio task, not a thread: it is blocking I/O on a pipe, which
|
|
8
|
+
is exactly what an event loop handles well, and it makes cancellation and
|
|
9
|
+
shutdown structured rather than hand-rolled from events and joins.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import asyncio
|
|
15
|
+
import contextlib
|
|
16
|
+
import shutil
|
|
17
|
+
import subprocess
|
|
18
|
+
import wave
|
|
19
|
+
from collections import deque
|
|
20
|
+
from collections.abc import Callable
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
|
|
23
|
+
import numpy as np
|
|
24
|
+
|
|
25
|
+
from .. import FRAME_BYTES, SAMPLE_RATE
|
|
26
|
+
from .devices import AudioError
|
|
27
|
+
|
|
28
|
+
FrameCallback = Callable[[bytes], None]
|
|
29
|
+
ErrorCallback = Callable[[str], None]
|
|
30
|
+
|
|
31
|
+
#: Keep only the tail of ffmpeg's stderr - all we ever report is the last line,
|
|
32
|
+
#: and an unbounded buffer would grow for the whole session.
|
|
33
|
+
_STDERR_TAIL_LINES = 5
|
|
34
|
+
|
|
35
|
+
#: `level` is a meter reading, not a measurement: speech sits well below full
|
|
36
|
+
#: scale, so it is gained up to fill the bar, and smoothed asymmetrically -
|
|
37
|
+
#: rising fast, falling slow - so brief peaks stay visible for a frame or two.
|
|
38
|
+
_GAIN = 4.0
|
|
39
|
+
_ATTACK = 0.5
|
|
40
|
+
_RELEASE = 0.15
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _ffmpeg_command(pulse_source: str, *extra: str) -> list[str]:
|
|
44
|
+
"""The one true ffmpeg invocation: one pulse source in, 16 kHz mono s16le
|
|
45
|
+
out on stdout. `extra` is inserted between the input and output options."""
|
|
46
|
+
return [
|
|
47
|
+
"ffmpeg",
|
|
48
|
+
"-hide_banner",
|
|
49
|
+
"-loglevel", "error",
|
|
50
|
+
"-nostdin",
|
|
51
|
+
"-f", "pulse",
|
|
52
|
+
"-i", pulse_source,
|
|
53
|
+
*extra,
|
|
54
|
+
"-ac", "1",
|
|
55
|
+
"-ar", str(SAMPLE_RATE),
|
|
56
|
+
"-f", "s16le",
|
|
57
|
+
"-",
|
|
58
|
+
]
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class SourceCapture:
|
|
62
|
+
"""Streams one audio source, handing fixed-size PCM frames to a callback.
|
|
63
|
+
|
|
64
|
+
`on_frame` runs on the event loop, so it must stay cheap - VAD only. Any
|
|
65
|
+
real work belongs behind the queue the segmenter feeds.
|
|
66
|
+
"""
|
|
67
|
+
|
|
68
|
+
def __init__(
|
|
69
|
+
self,
|
|
70
|
+
label: str,
|
|
71
|
+
pulse_source: str,
|
|
72
|
+
on_frame: FrameCallback,
|
|
73
|
+
on_error: ErrorCallback | None = None,
|
|
74
|
+
wav_path: Path | None = None,
|
|
75
|
+
) -> None:
|
|
76
|
+
self.label = label
|
|
77
|
+
self.pulse_source = pulse_source
|
|
78
|
+
self._on_frame = on_frame
|
|
79
|
+
self._on_error = on_error
|
|
80
|
+
self._wav_path = wav_path
|
|
81
|
+
|
|
82
|
+
self._proc: asyncio.subprocess.Process | None = None
|
|
83
|
+
self._task: asyncio.Task | None = None
|
|
84
|
+
self._stderr_task: asyncio.Task | None = None
|
|
85
|
+
self._stderr_tail: deque[str] = deque(maxlen=_STDERR_TAIL_LINES)
|
|
86
|
+
self._wav: wave.Wave_write | None = None
|
|
87
|
+
self._stopping = False
|
|
88
|
+
|
|
89
|
+
#: Smoothed 0..1 loudness, for the level meters in the TUI.
|
|
90
|
+
self.level: float = 0.0
|
|
91
|
+
self.frames_seen: int = 0
|
|
92
|
+
|
|
93
|
+
async def start(self) -> None:
|
|
94
|
+
if not shutil.which("ffmpeg"):
|
|
95
|
+
raise AudioError("ffmpeg not found on PATH")
|
|
96
|
+
|
|
97
|
+
self._proc = await asyncio.create_subprocess_exec(
|
|
98
|
+
*_ffmpeg_command(self.pulse_source),
|
|
99
|
+
stdout=asyncio.subprocess.PIPE,
|
|
100
|
+
stderr=asyncio.subprocess.PIPE,
|
|
101
|
+
)
|
|
102
|
+
try:
|
|
103
|
+
self._open_wav()
|
|
104
|
+
except Exception:
|
|
105
|
+
# Half-started is not a state a caller can clean up: whoever gets
|
|
106
|
+
# the exception has no capture object to stop.
|
|
107
|
+
await self.stop()
|
|
108
|
+
raise
|
|
109
|
+
|
|
110
|
+
self._task = asyncio.create_task(self._pump(), name=f"capture-{self.label}")
|
|
111
|
+
self._stderr_task = asyncio.create_task(
|
|
112
|
+
self._drain_stderr(), name=f"capture-{self.label}-stderr"
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
def _open_wav(self) -> None:
|
|
116
|
+
if self._wav_path is None:
|
|
117
|
+
return
|
|
118
|
+
self._wav_path.parent.mkdir(parents=True, exist_ok=True)
|
|
119
|
+
self._wav = wave.open(str(self._wav_path), "wb")
|
|
120
|
+
self._wav.setnchannels(1)
|
|
121
|
+
self._wav.setsampwidth(2)
|
|
122
|
+
self._wav.setframerate(SAMPLE_RATE)
|
|
123
|
+
|
|
124
|
+
async def _pump(self) -> None:
|
|
125
|
+
assert self._proc is not None and self._proc.stdout is not None
|
|
126
|
+
try:
|
|
127
|
+
while True:
|
|
128
|
+
frame = await self._proc.stdout.readexactly(FRAME_BYTES)
|
|
129
|
+
self.frames_seen += 1
|
|
130
|
+
self._update_level(frame)
|
|
131
|
+
if self._wav is not None:
|
|
132
|
+
self._wav.writeframes(frame)
|
|
133
|
+
self._on_frame(frame)
|
|
134
|
+
except asyncio.CancelledError:
|
|
135
|
+
raise
|
|
136
|
+
except asyncio.IncompleteReadError:
|
|
137
|
+
# The pipe closed: either we are stopping, or the source went away.
|
|
138
|
+
if not self._stopping:
|
|
139
|
+
self._report_failure()
|
|
140
|
+
except Exception as exc:
|
|
141
|
+
# Anything else kills this task, and a dead pump is silent: frames
|
|
142
|
+
# stop arriving and the meter freezes at its last value, which
|
|
143
|
+
# reads as a working capture. Say so instead.
|
|
144
|
+
self._report_failure(f"{type(exc).__name__}: {exc}")
|
|
145
|
+
|
|
146
|
+
async def _drain_stderr(self) -> None:
|
|
147
|
+
"""Read stderr continuously. Not only for the message - an unread pipe
|
|
148
|
+
fills at 64 KiB and would block ffmpeg, stalling capture entirely."""
|
|
149
|
+
pipe = self._proc.stderr if self._proc is not None else None
|
|
150
|
+
if pipe is None:
|
|
151
|
+
return
|
|
152
|
+
async for raw in pipe:
|
|
153
|
+
line = raw.decode(errors="replace").strip()
|
|
154
|
+
if line:
|
|
155
|
+
self._stderr_tail.append(line)
|
|
156
|
+
|
|
157
|
+
def _update_level(self, frame: bytes) -> None:
|
|
158
|
+
samples = np.frombuffer(frame, dtype=np.int16).astype(np.float32) / 32768.0
|
|
159
|
+
rms = float(np.sqrt(np.mean(samples * samples)))
|
|
160
|
+
weight = _ATTACK if rms > self.level else _RELEASE
|
|
161
|
+
self.level = (1 - weight) * self.level + weight * min(rms * _GAIN, 1.0)
|
|
162
|
+
|
|
163
|
+
def _report_failure(self, detail: str | None = None) -> None:
|
|
164
|
+
# A source that has stopped producing must not keep a live-looking
|
|
165
|
+
# meter, or the UI contradicts the warning printed right below it.
|
|
166
|
+
self.level = 0.0
|
|
167
|
+
if self._on_error is None:
|
|
168
|
+
return
|
|
169
|
+
if detail is None:
|
|
170
|
+
code = self._proc.returncode if self._proc is not None else None
|
|
171
|
+
detail = (
|
|
172
|
+
self._stderr_tail[-1] if self._stderr_tail else f"ffmpeg exited {code}"
|
|
173
|
+
)
|
|
174
|
+
self._on_error(f"[{self.label}] {detail}")
|
|
175
|
+
|
|
176
|
+
async def stop(self) -> None:
|
|
177
|
+
"""Terminate ffmpeg, await the reader tasks, close the WAV. Idempotent.
|
|
178
|
+
|
|
179
|
+
Every step runs even if an earlier one failed: teardown that gives up
|
|
180
|
+
halfway leaves an ffmpeg running and a WAV missing its last seconds,
|
|
181
|
+
which is worse than whatever raised.
|
|
182
|
+
"""
|
|
183
|
+
self._stopping = True
|
|
184
|
+
|
|
185
|
+
if self._proc is not None and self._proc.returncode is None:
|
|
186
|
+
with contextlib.suppress(ProcessLookupError):
|
|
187
|
+
self._proc.terminate()
|
|
188
|
+
|
|
189
|
+
for attr in ("_task", "_stderr_task"):
|
|
190
|
+
task = getattr(self, attr)
|
|
191
|
+
if task is not None:
|
|
192
|
+
task.cancel()
|
|
193
|
+
# The task may already have died of something other than the
|
|
194
|
+
# cancellation; it was reported when it happened.
|
|
195
|
+
with contextlib.suppress(asyncio.CancelledError, Exception):
|
|
196
|
+
await task
|
|
197
|
+
setattr(self, attr, None)
|
|
198
|
+
|
|
199
|
+
if self._proc is not None:
|
|
200
|
+
with contextlib.suppress(TimeoutError):
|
|
201
|
+
await asyncio.wait_for(self._proc.wait(), timeout=3)
|
|
202
|
+
if self._proc.returncode is None:
|
|
203
|
+
self._proc.kill()
|
|
204
|
+
await self._proc.wait()
|
|
205
|
+
self._proc = None
|
|
206
|
+
|
|
207
|
+
if self._wav is not None:
|
|
208
|
+
with contextlib.suppress(Exception):
|
|
209
|
+
self._wav.close()
|
|
210
|
+
self._wav = None
|
|
211
|
+
self.level = 0.0
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def check_source(pulse_source: str, seconds: float = 1.0) -> float:
|
|
215
|
+
"""Record briefly and report peak level - used by `sttop devices --test`."""
|
|
216
|
+
proc = subprocess.run(
|
|
217
|
+
_ffmpeg_command(pulse_source, "-t", str(seconds)),
|
|
218
|
+
capture_output=True,
|
|
219
|
+
timeout=seconds + 10,
|
|
220
|
+
)
|
|
221
|
+
if proc.returncode != 0:
|
|
222
|
+
raise AudioError(
|
|
223
|
+
proc.stderr.decode(errors="replace").strip() or "capture failed"
|
|
224
|
+
)
|
|
225
|
+
if not proc.stdout:
|
|
226
|
+
return 0.0
|
|
227
|
+
samples = np.frombuffer(proc.stdout, dtype=np.int16).astype(np.float32) / 32768.0
|
|
228
|
+
return float(np.max(np.abs(samples))) if samples.size else 0.0
|
sttop/audio/devices.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""PulseAudio/PipeWire source discovery via pactl."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import shutil
|
|
6
|
+
import subprocess
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class AudioError(RuntimeError):
|
|
11
|
+
pass
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class Source:
|
|
16
|
+
index: int
|
|
17
|
+
name: str
|
|
18
|
+
driver: str
|
|
19
|
+
spec: str
|
|
20
|
+
state: str
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def is_monitor(self) -> bool:
|
|
24
|
+
return self.name.endswith(".monitor")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _pactl(*args: str) -> str:
|
|
28
|
+
if not shutil.which("pactl"):
|
|
29
|
+
raise AudioError("pactl not found - sttop needs PipeWire or PulseAudio")
|
|
30
|
+
try:
|
|
31
|
+
out = subprocess.run(
|
|
32
|
+
["pactl", *args], capture_output=True, text=True, timeout=5, check=True
|
|
33
|
+
)
|
|
34
|
+
except subprocess.CalledProcessError as exc:
|
|
35
|
+
raise AudioError(f"pactl {' '.join(args)} failed: {exc.stderr.strip()}") from exc
|
|
36
|
+
except subprocess.TimeoutExpired as exc:
|
|
37
|
+
raise AudioError("pactl timed out - is the audio server running?") from exc
|
|
38
|
+
return out.stdout.strip()
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def list_sources() -> list[Source]:
|
|
42
|
+
sources = []
|
|
43
|
+
for line in _pactl("list", "short", "sources").splitlines():
|
|
44
|
+
parts = line.split("\t")
|
|
45
|
+
if len(parts) < 5 or not parts[0].isdigit():
|
|
46
|
+
continue # a header or some other row shape we do not understand
|
|
47
|
+
index, name, driver, spec, state = parts[:5]
|
|
48
|
+
sources.append(Source(int(index), name, driver, spec, state))
|
|
49
|
+
return sources
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def default_sink_monitor() -> str:
|
|
53
|
+
"""The monitor source carrying whatever is currently playing on this machine."""
|
|
54
|
+
sink = _pactl("get-default-sink")
|
|
55
|
+
if not sink or sink == "@DEFAULT_SINK@":
|
|
56
|
+
raise AudioError("no default sink - cannot capture system audio")
|
|
57
|
+
return f"{sink}.monitor"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def default_source() -> str:
|
|
61
|
+
"""The default recording source, i.e. the active microphone."""
|
|
62
|
+
source = _pactl("get-default-source")
|
|
63
|
+
if not source or source == "@DEFAULT_SOURCE@":
|
|
64
|
+
raise AudioError("no default source - cannot capture the microphone")
|
|
65
|
+
return source
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def resolve(requested: str | None, *, monitor: bool) -> str:
|
|
69
|
+
"""Resolve a configured source name, falling back to the sensible default.
|
|
70
|
+
|
|
71
|
+
A requested name may be a full source name or a unique substring of one,
|
|
72
|
+
so you can write `system_source = "hdmi"` instead of the full alsa id.
|
|
73
|
+
|
|
74
|
+
Blank means "no preference", the same as unset - otherwise the empty string
|
|
75
|
+
goes on to match every source and resolution fails as ambiguous, which is a
|
|
76
|
+
baffling way to punish someone for writing `mic_source = ""`.
|
|
77
|
+
"""
|
|
78
|
+
if requested is None or not requested.strip():
|
|
79
|
+
return default_sink_monitor() if monitor else default_source()
|
|
80
|
+
|
|
81
|
+
names = [s.name for s in list_sources()]
|
|
82
|
+
if requested in names:
|
|
83
|
+
return requested
|
|
84
|
+
|
|
85
|
+
matches = [n for n in names if requested in n]
|
|
86
|
+
if len(matches) == 1:
|
|
87
|
+
return matches[0]
|
|
88
|
+
if not matches:
|
|
89
|
+
raise AudioError(f"no audio source matching {requested!r}")
|
|
90
|
+
raise AudioError(f"{requested!r} is ambiguous, matches: {', '.join(matches)}")
|
sttop/audio/segmenter.py
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""Turn a continuous PCM frame stream into utterance-sized speech segments."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import time
|
|
6
|
+
from collections import deque
|
|
7
|
+
from collections.abc import Callable
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
|
|
10
|
+
import webrtcvad
|
|
11
|
+
|
|
12
|
+
from .. import FRAME_MS, SAMPLE_RATE
|
|
13
|
+
from ..config import VadConfig
|
|
14
|
+
|
|
15
|
+
FRAME_S = FRAME_MS / 1000.0
|
|
16
|
+
|
|
17
|
+
#: Window the trigger decision is measured over, and the fraction of it that
|
|
18
|
+
#: must be speech before a segment opens. Measured against the *full* window,
|
|
19
|
+
#: not the frames seen so far, so a single stray voiced frame at stream start
|
|
20
|
+
#: cannot trigger. Kept separate from `pad_ms`: how much audio to keep from
|
|
21
|
+
#: before speech onset is a question about clipped syllables, while this is a
|
|
22
|
+
#: question about sensitivity, and tuning one must not silently move the other.
|
|
23
|
+
TRIGGER_WINDOW_MS = 300
|
|
24
|
+
TRIGGER_RATIO = 0.6
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class Segment:
|
|
29
|
+
"""A contiguous run of speech from one source."""
|
|
30
|
+
|
|
31
|
+
source: str
|
|
32
|
+
pcm: bytes
|
|
33
|
+
start: float # seconds since session start
|
|
34
|
+
end: float
|
|
35
|
+
|
|
36
|
+
@property
|
|
37
|
+
def duration(self) -> float:
|
|
38
|
+
return self.end - self.start
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class Segmenter:
|
|
42
|
+
"""Voice-activity state machine over 20 ms frames.
|
|
43
|
+
|
|
44
|
+
Collects frames while speech is present, and emits a Segment once the
|
|
45
|
+
speaker has been quiet for `silence_ms` (or the segment hits its ceiling).
|
|
46
|
+
A rolling pre-roll buffer is prepended so the first syllable isn't clipped.
|
|
47
|
+
"""
|
|
48
|
+
|
|
49
|
+
def __init__(
|
|
50
|
+
self,
|
|
51
|
+
source: str,
|
|
52
|
+
config: VadConfig,
|
|
53
|
+
on_segment: Callable[[Segment], None],
|
|
54
|
+
clock: Callable[[], float] = time.monotonic,
|
|
55
|
+
) -> None:
|
|
56
|
+
self.source = source
|
|
57
|
+
self._on_segment = on_segment
|
|
58
|
+
self._clock = clock
|
|
59
|
+
|
|
60
|
+
self._vad = webrtcvad.Vad(config.aggressiveness)
|
|
61
|
+
self._pad_frames = max(1, config.pad_ms // FRAME_MS)
|
|
62
|
+
self._silence_frames = max(1, config.silence_ms // FRAME_MS)
|
|
63
|
+
self._max_frames = max(1, int(config.max_segment_s / FRAME_S))
|
|
64
|
+
self._min_frames = max(1, config.min_segment_ms // FRAME_MS)
|
|
65
|
+
|
|
66
|
+
self._trigger_frames = max(1, TRIGGER_WINDOW_MS // FRAME_MS)
|
|
67
|
+
self._preroll: deque[bytes] = deque(maxlen=self._pad_frames)
|
|
68
|
+
self._voiced: deque[bool] = deque(maxlen=self._trigger_frames)
|
|
69
|
+
self._buffer: list[bytes] = []
|
|
70
|
+
self._triggered = False
|
|
71
|
+
self._silence_run = 0
|
|
72
|
+
self._segment_start_index = 0
|
|
73
|
+
self._voiced_count = 0 # speech frames in the open segment
|
|
74
|
+
|
|
75
|
+
self._index = 0 # frames consumed since the stream opened
|
|
76
|
+
self._origin: float | None = None # session time of frame 0
|
|
77
|
+
self._paused = False
|
|
78
|
+
|
|
79
|
+
@property
|
|
80
|
+
def paused(self) -> bool:
|
|
81
|
+
return self._paused
|
|
82
|
+
|
|
83
|
+
@paused.setter
|
|
84
|
+
def paused(self, value: bool) -> None:
|
|
85
|
+
"""Pausing closes any open utterance, so resuming never splices audio
|
|
86
|
+
from either side of the gap into a single segment."""
|
|
87
|
+
if value and not self._paused:
|
|
88
|
+
self.close()
|
|
89
|
+
self._paused = value
|
|
90
|
+
|
|
91
|
+
def feed(self, frame: bytes) -> None:
|
|
92
|
+
if self._origin is None:
|
|
93
|
+
self._origin = self._clock()
|
|
94
|
+
index = self._index
|
|
95
|
+
self._index += 1
|
|
96
|
+
|
|
97
|
+
if self._paused:
|
|
98
|
+
return
|
|
99
|
+
|
|
100
|
+
speech = self._vad.is_speech(frame, SAMPLE_RATE)
|
|
101
|
+
|
|
102
|
+
if not self._triggered:
|
|
103
|
+
self._preroll.append(frame)
|
|
104
|
+
self._voiced.append(speech)
|
|
105
|
+
# Enough of the recent window is speech - open a segment.
|
|
106
|
+
if sum(self._voiced) >= TRIGGER_RATIO * self._trigger_frames:
|
|
107
|
+
self._triggered = True
|
|
108
|
+
self._segment_start_index = index - len(self._preroll) + 1
|
|
109
|
+
self._buffer = list(self._preroll)
|
|
110
|
+
self._voiced_count = sum(self._voiced)
|
|
111
|
+
self._silence_run = 0
|
|
112
|
+
self._preroll.clear()
|
|
113
|
+
self._voiced.clear()
|
|
114
|
+
return
|
|
115
|
+
|
|
116
|
+
self._buffer.append(frame)
|
|
117
|
+
self._voiced_count += speech
|
|
118
|
+
self._silence_run = 0 if speech else self._silence_run + 1
|
|
119
|
+
|
|
120
|
+
if self._silence_run >= self._silence_frames:
|
|
121
|
+
self._flush(trailing_silence=self._silence_run)
|
|
122
|
+
elif len(self._buffer) >= self._max_frames:
|
|
123
|
+
# Carry the silence run across the split: the speaker may already
|
|
124
|
+
# be most of the way to a pause, and restarting the count would
|
|
125
|
+
# hold the next segment open for a further full silence_ms.
|
|
126
|
+
carried = self._silence_run
|
|
127
|
+
self._flush(trailing_silence=0)
|
|
128
|
+
# Long monologue: stay open so the next chunk continues immediately.
|
|
129
|
+
self._triggered = True
|
|
130
|
+
self._segment_start_index = index + 1
|
|
131
|
+
self._buffer = []
|
|
132
|
+
self._voiced_count = 0
|
|
133
|
+
self._silence_run = carried
|
|
134
|
+
|
|
135
|
+
def _flush(self, trailing_silence: int) -> None:
|
|
136
|
+
frames = self._buffer
|
|
137
|
+
# Drop the silence we used as an end-of-utterance signal, keeping one
|
|
138
|
+
# frame so the last word has a little air after it.
|
|
139
|
+
if trailing_silence:
|
|
140
|
+
keep = max(0, len(frames) - trailing_silence + 1)
|
|
141
|
+
frames = frames[:keep]
|
|
142
|
+
|
|
143
|
+
voiced_count = self._voiced_count
|
|
144
|
+
self._triggered = False
|
|
145
|
+
self._buffer = []
|
|
146
|
+
self._voiced_count = 0
|
|
147
|
+
self._silence_run = 0
|
|
148
|
+
self._preroll.clear()
|
|
149
|
+
self._voiced.clear()
|
|
150
|
+
|
|
151
|
+
# Measure against actual speech, not padded length - otherwise the
|
|
152
|
+
# pre-roll alone can push a blip over the minimum.
|
|
153
|
+
if voiced_count < self._min_frames:
|
|
154
|
+
return
|
|
155
|
+
|
|
156
|
+
origin = self._origin or 0.0
|
|
157
|
+
start = origin + self._segment_start_index * FRAME_S
|
|
158
|
+
self._on_segment(
|
|
159
|
+
Segment(
|
|
160
|
+
source=self.source,
|
|
161
|
+
pcm=b"".join(frames),
|
|
162
|
+
start=start,
|
|
163
|
+
end=start + len(frames) * FRAME_S,
|
|
164
|
+
)
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
def close(self) -> None:
|
|
168
|
+
"""Emit whatever is still buffered - call when the stream ends."""
|
|
169
|
+
if self._triggered and self._buffer:
|
|
170
|
+
self._flush(trailing_silence=0)
|