sttop 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sttop/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ """sttop - live speech-to-text monitor for the terminal."""
2
+
3
+ __version__ = "0.1.0"
4
+
5
+ SAMPLE_RATE = 16_000
6
+ FRAME_MS = 20
7
+ FRAME_SAMPLES = SAMPLE_RATE * FRAME_MS // 1000 # 320 samples
8
+ FRAME_BYTES = FRAME_SAMPLES * 2 # int16 mono
sttop/__main__.py ADDED
@@ -0,0 +1,193 @@
1
+ """Command line entry point."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ from . import __version__
10
+ from .config import CONFIG_PATH, Config, ConfigError, write_default_config
11
+ from .stt import BACKENDS
12
+
13
+
14
+ def build_parser() -> argparse.ArgumentParser:
15
+ parser = argparse.ArgumentParser(
16
+ prog="sttop",
17
+ description="Live speech-to-text monitor. Taps mic + system audio, "
18
+ "transcribes and labels speakers in real time, writes Markdown.",
19
+ )
20
+ parser.add_argument("--version", action="version", version=f"sttop {__version__}")
21
+ parser.add_argument("-c", "--config", type=Path, help=f"default: {CONFIG_PATH}")
22
+
23
+ # `--config` reads as a global option, so accept it on either side of the
24
+ # subcommand. SUPPRESS is what makes that work: without it the subcommand's
25
+ # own default would overwrite a value already given before the subcommand.
26
+ common = argparse.ArgumentParser(add_help=False)
27
+ common.add_argument(
28
+ "-c", "--config", type=Path, default=argparse.SUPPRESS, help=argparse.SUPPRESS
29
+ )
30
+
31
+ sub = parser.add_subparsers(dest="command")
32
+
33
+ def command(name: str, summary: str) -> argparse.ArgumentParser:
34
+ return sub.add_parser(name, help=summary, parents=[common])
35
+
36
+ record = command("record", "start a session (default)")
37
+ record.add_argument("-t", "--title", help="session title, used in the filename")
38
+ record.add_argument("--mic", help="mic source name or substring")
39
+ record.add_argument("--system", help="system/monitor source name or substring")
40
+ record.add_argument("-m", "--model", help="override the backend's default model")
41
+ record.add_argument("--backend", choices=list(BACKENDS))
42
+ record.add_argument(
43
+ "--language", help="force a language, e.g. es (default: autodetect)"
44
+ )
45
+ record.add_argument("--no-diarize", action="store_true", help="skip speaker id")
46
+ record.add_argument("--save-wav", action="store_true", help="keep the raw audio")
47
+
48
+ devices_cmd = command("devices", "list audio sources")
49
+ devices_cmd.add_argument(
50
+ "--test", action="store_true", help="record 1s from each and report levels"
51
+ )
52
+
53
+ command("sessions", "list recorded sessions")
54
+ command("theme", "show the detected terminal colour scheme")
55
+ command("config", "write a default config file")
56
+
57
+ return parser
58
+
59
+
60
+ #: Global options that may appear before the subcommand, and whether the option
61
+ #: swallows the token after it.
62
+ _GLOBAL_OPTIONS = {"-c": True, "--config": True, "-h": False, "--help": False,
63
+ "--version": False}
64
+
65
+
66
+ def with_default_command(argv: list[str], commands: set[str]) -> list[str]:
67
+ """Insert `record` when no subcommand was given.
68
+
69
+ `sttop -t standup` means `sttop record -t standup`. Rather than parse twice
70
+ and hope the first attempt fails cleanly - it does not, since `-t` is a
71
+ record option and argparse rejects it outright - the command is filled in
72
+ before parsing, so there is only ever one well-formed parse.
73
+ """
74
+ index = 0
75
+ while index < len(argv):
76
+ token = argv[index]
77
+ if token in commands:
78
+ return argv
79
+ option, joined, _ = token.partition("=")
80
+ if option not in _GLOBAL_OPTIONS:
81
+ break # a record option, or a positional - record starts here
82
+ # Skip the global option, plus its value when given as a separate
83
+ # token (`-c path`) rather than joined on (`--config=path`).
84
+ index += 2 if _GLOBAL_OPTIONS[option] and not joined else 1
85
+ return [*argv[:index], "record", *argv[index:]]
86
+
87
+
88
+ def cmd_devices(config: Config, args) -> int:
89
+ from .audio import capture, devices
90
+
91
+ try:
92
+ sources = devices.list_sources()
93
+ mic = devices.resolve(config.audio.mic_source, monitor=False)
94
+ system = devices.resolve(config.audio.system_source, monitor=True)
95
+ except devices.AudioError as exc:
96
+ print(f"error: {exc}", file=sys.stderr)
97
+ return 1
98
+
99
+ print(f"mic -> {mic}")
100
+ print(f"system -> {system}\n")
101
+ for source in sources:
102
+ role = "mic" if source.name == mic else "sys" if source.name == system else " "
103
+ kind = "monitor" if source.is_monitor else "input "
104
+ line = f"{role} {kind} {source.state:<10} {source.name}"
105
+ if args.test:
106
+ try:
107
+ peak = capture.check_source(source.name, seconds=1.0)
108
+ line += f" peak {peak:.3f}" + ("" if peak > 0.001 else " (silent)")
109
+ except Exception as exc:
110
+ line += f" [failed: {str(exc).splitlines()[0][:40]}]"
111
+ print(line)
112
+ return 0
113
+
114
+
115
+ def cmd_theme(config: Config) -> int:
116
+ from .terminal import detect_theme, theme_sources
117
+
118
+ for name, verdict in theme_sources():
119
+ print(f"{name:<12}{verdict or '<no answer>'}")
120
+ print(f"configured {config.ui.theme}")
121
+ print(f"\nusing {detect_theme(config.ui.theme)}")
122
+ return 0
123
+
124
+
125
+ def cmd_sessions(config: Config) -> int:
126
+ from .journal import list_sessions
127
+
128
+ directory = Path(config.sessions_dir)
129
+ paths = list_sessions(directory)
130
+ if not paths:
131
+ print(f"no sessions yet in {directory}")
132
+ return 0
133
+ for path in paths:
134
+ size = path.stat().st_size
135
+ print(f"{path.name:<52} {size / 1024:6.1f} KiB")
136
+ print(f"\n{len(paths)} session(s) in {directory}")
137
+ return 0
138
+
139
+
140
+ def cmd_record(config: Config, args) -> int:
141
+ from .tui import SttopApp
142
+
143
+ if args.mic:
144
+ config.audio.mic_source = args.mic
145
+ if args.system:
146
+ config.audio.system_source = args.system
147
+ if args.model:
148
+ config.stt.model = args.model
149
+ if args.backend:
150
+ config.stt.backend = args.backend
151
+ if args.language:
152
+ config.stt.language = args.language
153
+ if args.no_diarize:
154
+ config.diarize.enabled = False
155
+ if args.save_wav:
156
+ config.audio.save_wav = True
157
+
158
+ path = SttopApp(config, args.title).run()
159
+ if path:
160
+ print(f"transcript: {path}")
161
+ return 0
162
+
163
+
164
+ COMMANDS = {"record", "devices", "sessions", "theme", "config"}
165
+
166
+
167
+ def main(argv: list[str] | None = None) -> int:
168
+ argv = sys.argv[1:] if argv is None else argv
169
+ parser = build_parser()
170
+ args = parser.parse_args(with_default_command(argv, COMMANDS))
171
+
172
+ if args.command == "config": # writing a config must not require a valid one
173
+ path = write_default_config(args.config)
174
+ print(f"wrote {path}")
175
+ return 0
176
+
177
+ try:
178
+ config = Config.load(args.config)
179
+ except ConfigError as exc:
180
+ print(f"error: bad config: {exc}", file=sys.stderr)
181
+ return 1
182
+
183
+ if args.command == "devices":
184
+ return cmd_devices(config, args)
185
+ if args.command == "sessions":
186
+ return cmd_sessions(config)
187
+ if args.command == "theme":
188
+ return cmd_theme(config)
189
+ return cmd_record(config, args)
190
+
191
+
192
+ if __name__ == "__main__":
193
+ sys.exit(main())
@@ -0,0 +1 @@
1
+ """Audio capture and voice-activity segmentation."""
sttop/audio/capture.py ADDED
@@ -0,0 +1,228 @@
1
+ """Capture a PulseAudio/PipeWire source as 16 kHz mono PCM via ffmpeg.
2
+
3
+ ffmpeg is used instead of a PortAudio binding because monitor sources (the
4
+ "what you hear" side of the capture) are exposed cleanly by the pulse backend,
5
+ whereas PortAudio device indices for monitors are inconsistent under PipeWire.
6
+
7
+ Reading is an asyncio task, not a thread: it is blocking I/O on a pipe, which
8
+ is exactly what an event loop handles well, and it makes cancellation and
9
+ shutdown structured rather than hand-rolled from events and joins.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import asyncio
15
+ import contextlib
16
+ import shutil
17
+ import subprocess
18
+ import wave
19
+ from collections import deque
20
+ from collections.abc import Callable
21
+ from pathlib import Path
22
+
23
+ import numpy as np
24
+
25
+ from .. import FRAME_BYTES, SAMPLE_RATE
26
+ from .devices import AudioError
27
+
28
+ FrameCallback = Callable[[bytes], None]
29
+ ErrorCallback = Callable[[str], None]
30
+
31
+ #: Keep only the tail of ffmpeg's stderr - all we ever report is the last line,
32
+ #: and an unbounded buffer would grow for the whole session.
33
+ _STDERR_TAIL_LINES = 5
34
+
35
+ #: `level` is a meter reading, not a measurement: speech sits well below full
36
+ #: scale, so it is gained up to fill the bar, and smoothed asymmetrically -
37
+ #: rising fast, falling slow - so brief peaks stay visible for a frame or two.
38
+ _GAIN = 4.0
39
+ _ATTACK = 0.5
40
+ _RELEASE = 0.15
41
+
42
+
43
+ def _ffmpeg_command(pulse_source: str, *extra: str) -> list[str]:
44
+ """The one true ffmpeg invocation: one pulse source in, 16 kHz mono s16le
45
+ out on stdout. `extra` is inserted between the input and output options."""
46
+ return [
47
+ "ffmpeg",
48
+ "-hide_banner",
49
+ "-loglevel", "error",
50
+ "-nostdin",
51
+ "-f", "pulse",
52
+ "-i", pulse_source,
53
+ *extra,
54
+ "-ac", "1",
55
+ "-ar", str(SAMPLE_RATE),
56
+ "-f", "s16le",
57
+ "-",
58
+ ]
59
+
60
+
61
+ class SourceCapture:
62
+ """Streams one audio source, handing fixed-size PCM frames to a callback.
63
+
64
+ `on_frame` runs on the event loop, so it must stay cheap - VAD only. Any
65
+ real work belongs behind the queue the segmenter feeds.
66
+ """
67
+
68
+ def __init__(
69
+ self,
70
+ label: str,
71
+ pulse_source: str,
72
+ on_frame: FrameCallback,
73
+ on_error: ErrorCallback | None = None,
74
+ wav_path: Path | None = None,
75
+ ) -> None:
76
+ self.label = label
77
+ self.pulse_source = pulse_source
78
+ self._on_frame = on_frame
79
+ self._on_error = on_error
80
+ self._wav_path = wav_path
81
+
82
+ self._proc: asyncio.subprocess.Process | None = None
83
+ self._task: asyncio.Task | None = None
84
+ self._stderr_task: asyncio.Task | None = None
85
+ self._stderr_tail: deque[str] = deque(maxlen=_STDERR_TAIL_LINES)
86
+ self._wav: wave.Wave_write | None = None
87
+ self._stopping = False
88
+
89
+ #: Smoothed 0..1 loudness, for the level meters in the TUI.
90
+ self.level: float = 0.0
91
+ self.frames_seen: int = 0
92
+
93
+ async def start(self) -> None:
94
+ if not shutil.which("ffmpeg"):
95
+ raise AudioError("ffmpeg not found on PATH")
96
+
97
+ self._proc = await asyncio.create_subprocess_exec(
98
+ *_ffmpeg_command(self.pulse_source),
99
+ stdout=asyncio.subprocess.PIPE,
100
+ stderr=asyncio.subprocess.PIPE,
101
+ )
102
+ try:
103
+ self._open_wav()
104
+ except Exception:
105
+ # Half-started is not a state a caller can clean up: whoever gets
106
+ # the exception has no capture object to stop.
107
+ await self.stop()
108
+ raise
109
+
110
+ self._task = asyncio.create_task(self._pump(), name=f"capture-{self.label}")
111
+ self._stderr_task = asyncio.create_task(
112
+ self._drain_stderr(), name=f"capture-{self.label}-stderr"
113
+ )
114
+
115
+ def _open_wav(self) -> None:
116
+ if self._wav_path is None:
117
+ return
118
+ self._wav_path.parent.mkdir(parents=True, exist_ok=True)
119
+ self._wav = wave.open(str(self._wav_path), "wb")
120
+ self._wav.setnchannels(1)
121
+ self._wav.setsampwidth(2)
122
+ self._wav.setframerate(SAMPLE_RATE)
123
+
124
+ async def _pump(self) -> None:
125
+ assert self._proc is not None and self._proc.stdout is not None
126
+ try:
127
+ while True:
128
+ frame = await self._proc.stdout.readexactly(FRAME_BYTES)
129
+ self.frames_seen += 1
130
+ self._update_level(frame)
131
+ if self._wav is not None:
132
+ self._wav.writeframes(frame)
133
+ self._on_frame(frame)
134
+ except asyncio.CancelledError:
135
+ raise
136
+ except asyncio.IncompleteReadError:
137
+ # The pipe closed: either we are stopping, or the source went away.
138
+ if not self._stopping:
139
+ self._report_failure()
140
+ except Exception as exc:
141
+ # Anything else kills this task, and a dead pump is silent: frames
142
+ # stop arriving and the meter freezes at its last value, which
143
+ # reads as a working capture. Say so instead.
144
+ self._report_failure(f"{type(exc).__name__}: {exc}")
145
+
146
+ async def _drain_stderr(self) -> None:
147
+ """Read stderr continuously. Not only for the message - an unread pipe
148
+ fills at 64 KiB and would block ffmpeg, stalling capture entirely."""
149
+ pipe = self._proc.stderr if self._proc is not None else None
150
+ if pipe is None:
151
+ return
152
+ async for raw in pipe:
153
+ line = raw.decode(errors="replace").strip()
154
+ if line:
155
+ self._stderr_tail.append(line)
156
+
157
+ def _update_level(self, frame: bytes) -> None:
158
+ samples = np.frombuffer(frame, dtype=np.int16).astype(np.float32) / 32768.0
159
+ rms = float(np.sqrt(np.mean(samples * samples)))
160
+ weight = _ATTACK if rms > self.level else _RELEASE
161
+ self.level = (1 - weight) * self.level + weight * min(rms * _GAIN, 1.0)
162
+
163
+ def _report_failure(self, detail: str | None = None) -> None:
164
+ # A source that has stopped producing must not keep a live-looking
165
+ # meter, or the UI contradicts the warning printed right below it.
166
+ self.level = 0.0
167
+ if self._on_error is None:
168
+ return
169
+ if detail is None:
170
+ code = self._proc.returncode if self._proc is not None else None
171
+ detail = (
172
+ self._stderr_tail[-1] if self._stderr_tail else f"ffmpeg exited {code}"
173
+ )
174
+ self._on_error(f"[{self.label}] {detail}")
175
+
176
+ async def stop(self) -> None:
177
+ """Terminate ffmpeg, await the reader tasks, close the WAV. Idempotent.
178
+
179
+ Every step runs even if an earlier one failed: teardown that gives up
180
+ halfway leaves an ffmpeg running and a WAV missing its last seconds,
181
+ which is worse than whatever raised.
182
+ """
183
+ self._stopping = True
184
+
185
+ if self._proc is not None and self._proc.returncode is None:
186
+ with contextlib.suppress(ProcessLookupError):
187
+ self._proc.terminate()
188
+
189
+ for attr in ("_task", "_stderr_task"):
190
+ task = getattr(self, attr)
191
+ if task is not None:
192
+ task.cancel()
193
+ # The task may already have died of something other than the
194
+ # cancellation; it was reported when it happened.
195
+ with contextlib.suppress(asyncio.CancelledError, Exception):
196
+ await task
197
+ setattr(self, attr, None)
198
+
199
+ if self._proc is not None:
200
+ with contextlib.suppress(TimeoutError):
201
+ await asyncio.wait_for(self._proc.wait(), timeout=3)
202
+ if self._proc.returncode is None:
203
+ self._proc.kill()
204
+ await self._proc.wait()
205
+ self._proc = None
206
+
207
+ if self._wav is not None:
208
+ with contextlib.suppress(Exception):
209
+ self._wav.close()
210
+ self._wav = None
211
+ self.level = 0.0
212
+
213
+
214
+ def check_source(pulse_source: str, seconds: float = 1.0) -> float:
215
+ """Record briefly and report peak level - used by `sttop devices --test`."""
216
+ proc = subprocess.run(
217
+ _ffmpeg_command(pulse_source, "-t", str(seconds)),
218
+ capture_output=True,
219
+ timeout=seconds + 10,
220
+ )
221
+ if proc.returncode != 0:
222
+ raise AudioError(
223
+ proc.stderr.decode(errors="replace").strip() or "capture failed"
224
+ )
225
+ if not proc.stdout:
226
+ return 0.0
227
+ samples = np.frombuffer(proc.stdout, dtype=np.int16).astype(np.float32) / 32768.0
228
+ return float(np.max(np.abs(samples))) if samples.size else 0.0
sttop/audio/devices.py ADDED
@@ -0,0 +1,90 @@
1
+ """PulseAudio/PipeWire source discovery via pactl."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import shutil
6
+ import subprocess
7
+ from dataclasses import dataclass
8
+
9
+
10
+ class AudioError(RuntimeError):
11
+ pass
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class Source:
16
+ index: int
17
+ name: str
18
+ driver: str
19
+ spec: str
20
+ state: str
21
+
22
+ @property
23
+ def is_monitor(self) -> bool:
24
+ return self.name.endswith(".monitor")
25
+
26
+
27
+ def _pactl(*args: str) -> str:
28
+ if not shutil.which("pactl"):
29
+ raise AudioError("pactl not found - sttop needs PipeWire or PulseAudio")
30
+ try:
31
+ out = subprocess.run(
32
+ ["pactl", *args], capture_output=True, text=True, timeout=5, check=True
33
+ )
34
+ except subprocess.CalledProcessError as exc:
35
+ raise AudioError(f"pactl {' '.join(args)} failed: {exc.stderr.strip()}") from exc
36
+ except subprocess.TimeoutExpired as exc:
37
+ raise AudioError("pactl timed out - is the audio server running?") from exc
38
+ return out.stdout.strip()
39
+
40
+
41
+ def list_sources() -> list[Source]:
42
+ sources = []
43
+ for line in _pactl("list", "short", "sources").splitlines():
44
+ parts = line.split("\t")
45
+ if len(parts) < 5 or not parts[0].isdigit():
46
+ continue # a header or some other row shape we do not understand
47
+ index, name, driver, spec, state = parts[:5]
48
+ sources.append(Source(int(index), name, driver, spec, state))
49
+ return sources
50
+
51
+
52
+ def default_sink_monitor() -> str:
53
+ """The monitor source carrying whatever is currently playing on this machine."""
54
+ sink = _pactl("get-default-sink")
55
+ if not sink or sink == "@DEFAULT_SINK@":
56
+ raise AudioError("no default sink - cannot capture system audio")
57
+ return f"{sink}.monitor"
58
+
59
+
60
+ def default_source() -> str:
61
+ """The default recording source, i.e. the active microphone."""
62
+ source = _pactl("get-default-source")
63
+ if not source or source == "@DEFAULT_SOURCE@":
64
+ raise AudioError("no default source - cannot capture the microphone")
65
+ return source
66
+
67
+
68
+ def resolve(requested: str | None, *, monitor: bool) -> str:
69
+ """Resolve a configured source name, falling back to the sensible default.
70
+
71
+ A requested name may be a full source name or a unique substring of one,
72
+ so you can write `system_source = "hdmi"` instead of the full alsa id.
73
+
74
+ Blank means "no preference", the same as unset - otherwise the empty string
75
+ goes on to match every source and resolution fails as ambiguous, which is a
76
+ baffling way to punish someone for writing `mic_source = ""`.
77
+ """
78
+ if requested is None or not requested.strip():
79
+ return default_sink_monitor() if monitor else default_source()
80
+
81
+ names = [s.name for s in list_sources()]
82
+ if requested in names:
83
+ return requested
84
+
85
+ matches = [n for n in names if requested in n]
86
+ if len(matches) == 1:
87
+ return matches[0]
88
+ if not matches:
89
+ raise AudioError(f"no audio source matching {requested!r}")
90
+ raise AudioError(f"{requested!r} is ambiguous, matches: {', '.join(matches)}")
@@ -0,0 +1,170 @@
1
+ """Turn a continuous PCM frame stream into utterance-sized speech segments."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from collections import deque
7
+ from collections.abc import Callable
8
+ from dataclasses import dataclass
9
+
10
+ import webrtcvad
11
+
12
+ from .. import FRAME_MS, SAMPLE_RATE
13
+ from ..config import VadConfig
14
+
15
+ FRAME_S = FRAME_MS / 1000.0
16
+
17
+ #: Window the trigger decision is measured over, and the fraction of it that
18
+ #: must be speech before a segment opens. Measured against the *full* window,
19
+ #: not the frames seen so far, so a single stray voiced frame at stream start
20
+ #: cannot trigger. Kept separate from `pad_ms`: how much audio to keep from
21
+ #: before speech onset is a question about clipped syllables, while this is a
22
+ #: question about sensitivity, and tuning one must not silently move the other.
23
+ TRIGGER_WINDOW_MS = 300
24
+ TRIGGER_RATIO = 0.6
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class Segment:
29
+ """A contiguous run of speech from one source."""
30
+
31
+ source: str
32
+ pcm: bytes
33
+ start: float # seconds since session start
34
+ end: float
35
+
36
+ @property
37
+ def duration(self) -> float:
38
+ return self.end - self.start
39
+
40
+
41
+ class Segmenter:
42
+ """Voice-activity state machine over 20 ms frames.
43
+
44
+ Collects frames while speech is present, and emits a Segment once the
45
+ speaker has been quiet for `silence_ms` (or the segment hits its ceiling).
46
+ A rolling pre-roll buffer is prepended so the first syllable isn't clipped.
47
+ """
48
+
49
+ def __init__(
50
+ self,
51
+ source: str,
52
+ config: VadConfig,
53
+ on_segment: Callable[[Segment], None],
54
+ clock: Callable[[], float] = time.monotonic,
55
+ ) -> None:
56
+ self.source = source
57
+ self._on_segment = on_segment
58
+ self._clock = clock
59
+
60
+ self._vad = webrtcvad.Vad(config.aggressiveness)
61
+ self._pad_frames = max(1, config.pad_ms // FRAME_MS)
62
+ self._silence_frames = max(1, config.silence_ms // FRAME_MS)
63
+ self._max_frames = max(1, int(config.max_segment_s / FRAME_S))
64
+ self._min_frames = max(1, config.min_segment_ms // FRAME_MS)
65
+
66
+ self._trigger_frames = max(1, TRIGGER_WINDOW_MS // FRAME_MS)
67
+ self._preroll: deque[bytes] = deque(maxlen=self._pad_frames)
68
+ self._voiced: deque[bool] = deque(maxlen=self._trigger_frames)
69
+ self._buffer: list[bytes] = []
70
+ self._triggered = False
71
+ self._silence_run = 0
72
+ self._segment_start_index = 0
73
+ self._voiced_count = 0 # speech frames in the open segment
74
+
75
+ self._index = 0 # frames consumed since the stream opened
76
+ self._origin: float | None = None # session time of frame 0
77
+ self._paused = False
78
+
79
+ @property
80
+ def paused(self) -> bool:
81
+ return self._paused
82
+
83
+ @paused.setter
84
+ def paused(self, value: bool) -> None:
85
+ """Pausing closes any open utterance, so resuming never splices audio
86
+ from either side of the gap into a single segment."""
87
+ if value and not self._paused:
88
+ self.close()
89
+ self._paused = value
90
+
91
+ def feed(self, frame: bytes) -> None:
92
+ if self._origin is None:
93
+ self._origin = self._clock()
94
+ index = self._index
95
+ self._index += 1
96
+
97
+ if self._paused:
98
+ return
99
+
100
+ speech = self._vad.is_speech(frame, SAMPLE_RATE)
101
+
102
+ if not self._triggered:
103
+ self._preroll.append(frame)
104
+ self._voiced.append(speech)
105
+ # Enough of the recent window is speech - open a segment.
106
+ if sum(self._voiced) >= TRIGGER_RATIO * self._trigger_frames:
107
+ self._triggered = True
108
+ self._segment_start_index = index - len(self._preroll) + 1
109
+ self._buffer = list(self._preroll)
110
+ self._voiced_count = sum(self._voiced)
111
+ self._silence_run = 0
112
+ self._preroll.clear()
113
+ self._voiced.clear()
114
+ return
115
+
116
+ self._buffer.append(frame)
117
+ self._voiced_count += speech
118
+ self._silence_run = 0 if speech else self._silence_run + 1
119
+
120
+ if self._silence_run >= self._silence_frames:
121
+ self._flush(trailing_silence=self._silence_run)
122
+ elif len(self._buffer) >= self._max_frames:
123
+ # Carry the silence run across the split: the speaker may already
124
+ # be most of the way to a pause, and restarting the count would
125
+ # hold the next segment open for a further full silence_ms.
126
+ carried = self._silence_run
127
+ self._flush(trailing_silence=0)
128
+ # Long monologue: stay open so the next chunk continues immediately.
129
+ self._triggered = True
130
+ self._segment_start_index = index + 1
131
+ self._buffer = []
132
+ self._voiced_count = 0
133
+ self._silence_run = carried
134
+
135
+ def _flush(self, trailing_silence: int) -> None:
136
+ frames = self._buffer
137
+ # Drop the silence we used as an end-of-utterance signal, keeping one
138
+ # frame so the last word has a little air after it.
139
+ if trailing_silence:
140
+ keep = max(0, len(frames) - trailing_silence + 1)
141
+ frames = frames[:keep]
142
+
143
+ voiced_count = self._voiced_count
144
+ self._triggered = False
145
+ self._buffer = []
146
+ self._voiced_count = 0
147
+ self._silence_run = 0
148
+ self._preroll.clear()
149
+ self._voiced.clear()
150
+
151
+ # Measure against actual speech, not padded length - otherwise the
152
+ # pre-roll alone can push a blip over the minimum.
153
+ if voiced_count < self._min_frames:
154
+ return
155
+
156
+ origin = self._origin or 0.0
157
+ start = origin + self._segment_start_index * FRAME_S
158
+ self._on_segment(
159
+ Segment(
160
+ source=self.source,
161
+ pcm=b"".join(frames),
162
+ start=start,
163
+ end=start + len(frames) * FRAME_S,
164
+ )
165
+ )
166
+
167
+ def close(self) -> None:
168
+ """Emit whatever is still buffered - call when the stream ends."""
169
+ if self._triggered and self._buffer:
170
+ self._flush(trailing_silence=0)