beseda 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
beseda/__init__.py ADDED
File without changes
beseda/__main__.py ADDED
@@ -0,0 +1,514 @@
1
+ """Voice conversations with your coding agent: say the wake word, talk, listen to the answer."""
2
+
3
+ import argparse
4
+ import contextlib
5
+ import io
6
+ import logging
7
+ import math
8
+ import os
9
+ import queue
10
+ import subprocess
11
+ import sys
12
+ import termios
13
+ import threading
14
+ import time
15
+ import tomllib
16
+ import tty
17
+ from datetime import datetime
18
+ from pathlib import Path
19
+
20
+ from rich.console import Console, Group
21
+ from rich.live import Live
22
+ from rich.markup import escape
23
+ from rich.text import Text
24
+ from RealtimeSTT import AudioToTextRecorder
25
+ from RealtimeTTS import TextToAudioStream
26
+
27
+ from beseda.brains import BRAINS, Brain
28
+ from beseda.language import Language, available
29
+ from beseda.language import load as load_language
30
+ from beseda.recording import DialogRecorder
31
+ from beseda.stt import clean, recorder_options
32
+ from beseda.tts import TTS_ENGINES, create_engine, speakable
33
+ from beseda.wake import is_hold, is_stop, strip_wake, wake_pattern
34
+
35
+ log = logging.getLogger("beseda")
36
+
37
+ LOG_DIR = Path.home() / ".beseda" / "logs"
38
+ RECORDINGS_DIR = Path.home() / "Downloads"
39
+ CONFIG_PATH = Path.home() / ".beseda" / "config.toml"
40
+ SESSION_START_SOUND = "/System/Library/Sounds/Tink.aiff"
41
+ SESSION_END_SOUND = "/System/Library/Sounds/Bottle.aiff"
42
+
43
+
44
+ def home_relative(path: Path | str) -> str:
45
+ """~/.beseda/logs/... instead of /Users/<name>/...: the terminal often ends up in screenshots."""
46
+ return str(path).replace(str(Path.home()), "~", 1)
47
+
48
+
49
+ def setup_logging(debug: bool, keep_days: float) -> Path:
50
+ """One file per session; nothing goes to the terminal so the UI stays intact.
51
+ Logs hold everything said in the dialog, so sessions older than `keep_days` are deleted."""
52
+ LOG_DIR.mkdir(parents=True, exist_ok=True)
53
+ cutoff = time.time() - keep_days * 86400
54
+ for old in LOG_DIR.glob("beseda-*.log"):
55
+ if old.stat().st_mtime < cutoff:
56
+ old.unlink()
57
+ path = LOG_DIR / f"beseda-{datetime.now():%Y%m%d-%H%M%S}.log"
58
+ handler = logging.FileHandler(path, encoding="utf-8")
59
+ handler.setLevel(logging.DEBUG if debug else logging.INFO)
60
+ handler.setFormatter(
61
+ logging.Formatter("%(asctime)s.%(msecs)03d %(levelname)-7s [%(threadName)s] %(name)s: %(message)s", "%H:%M:%S")
62
+ )
63
+ logging.getLogger().addHandler(handler)
64
+ # Native libraries (whisper.cpp and its VAD, in RealtimeSTT's worker processes) print to file descriptor 2
65
+ # and would scramble the terminal UI. Point fd 2, inherited by those processes, at the log; Python's own
66
+ # error output keeps going to the terminal.
67
+ sys.stderr = os.fdopen(os.dup(2), "w", buffering=1)
68
+ os.dup2(handler.stream.fileno(), 2)
69
+ logging.getLogger().setLevel(logging.WARNING)
70
+ log.setLevel(logging.DEBUG if debug else logging.INFO)
71
+ threading.excepthook = lambda a: log.critical(
72
+ "uncaught exception in thread %s", a.thread.name, exc_info=(a.exc_type, a.exc_value, a.exc_traceback)
73
+ )
74
+ return path
75
+
76
+
77
+ class Turn:
78
+ """Timeline of one exchange, measured from the moment the user starts speaking."""
79
+
80
+ count = 0
81
+
82
+ def __init__(self):
83
+ Turn.count += 1
84
+ self.id = Turn.count
85
+ self.t0 = time.monotonic()
86
+ self.marks: dict[str, float] = {}
87
+ self.tools = 0
88
+ self.mark("speech_start")
89
+
90
+ def mark(self, name: str, detail: str = "") -> None:
91
+ elapsed = time.monotonic() - self.t0
92
+ first = name not in self.marks
93
+ self.marks.setdefault(name, elapsed)
94
+ if first or detail:
95
+ log.info("turn=%d +%.2fs %s %s", self.id, elapsed, name, detail)
96
+
97
+ def span(self, start: str, end: str) -> str:
98
+ if start in self.marks and end in self.marks:
99
+ return f"{self.marks[end] - self.marks[start]:.2f}s"
100
+ return "-"
101
+
102
+ def summary(self) -> str:
103
+ return (
104
+ f"stt={self.span('speech_end', 'transcribed')} "
105
+ f"llm_first_token={self.span('prompt_sent', 'first_text')} "
106
+ f"tts_first_audio={self.span('first_text', 'audio_start')} "
107
+ f"voice_to_voice={self.span('speech_end', 'audio_start')} "
108
+ f"tools={self.tools} total={self.span('speech_start', 'done')}"
109
+ )
110
+
111
+
112
+ class App:
113
+ def __init__(self, args: argparse.Namespace, language: Language, log_path: Path):
114
+ self.lang = language
115
+ self.console = Console()
116
+ self.status = "idle"
117
+ self.detail = ""
118
+ self.partial = ""
119
+ self.live = Live(self._render(), console=self.console, auto_refresh=False)
120
+ self.active = threading.Event()
121
+ self.interrupted = threading.Event()
122
+ self.stopping = threading.Event()
123
+ self.in_listen = False
124
+ self.turn: Turn | None = None
125
+ self.spoken_chars = 0
126
+ self.wake = wake_pattern(args.wake_word, language.wake_endings) if args.wake_word else None
127
+ self.wake_word = args.wake_word
128
+ self.follow_up = args.follow_up
129
+ self.hold = args.hold
130
+ # None: standby, waiting for the wake word; inf: answering; otherwise the follow-up deadline.
131
+ self.session_until: float | None = None
132
+ self.speech_started_at = 0.0
133
+ # Guards status and the session deadline: the session timer must not close a session whose next
134
+ # phrase has just started.
135
+ self.lock = threading.RLock()
136
+
137
+ log.info("start args=%s", vars(args))
138
+ with self.console.status(language.text("loading")):
139
+ started = time.monotonic()
140
+ self.brain: Brain = BRAINS[args.brain](args.model, language.voice_prompt)
141
+ engine = create_engine(args.tts, language, args.voice)
142
+ self.dialog = DialogRecorder(Path(args.record), engine.rate) if args.record else None
143
+ self.tts = TextToAudioStream(
144
+ engine,
145
+ language=language.code,
146
+ on_audio_stream_start=self._on_audio_start,
147
+ on_audio_stream_stop=self._on_audio_stop,
148
+ )
149
+ self.recorder = AudioToTextRecorder(
150
+ **recorder_options(args.whisper, args.vocabulary, language),
151
+ spinner=False,
152
+ level=logging.ERROR,
153
+ no_log_file=True,
154
+ post_speech_silence_duration=0.7,
155
+ on_recording_start=self._on_speech_start,
156
+ on_recording_stop=self._on_speech_end,
157
+ on_recorded_chunk=self.dialog.on_mic if self.dialog else None,
158
+ )
159
+ self.recorder.set_microphone(False)
160
+ log.info("components ready in %.1fs", time.monotonic() - started)
161
+ self.console.print(f"[dim]{escape(language.text('log', path=home_relative(log_path)))}[/dim]")
162
+ if self.wake:
163
+ self.console.print(language.text("intro", wake=self.wake_word.capitalize(), follow_up=f"{self.follow_up:g}"))
164
+ self.active.set() # wake word mode listens from the start, like a smart speaker
165
+
166
+ # --- UI -----------------------------------------------------------------
167
+
168
+ def _render(self) -> Group:
169
+ status = Text(self.lang.text(f"status.{self.status}"), style="bold")
170
+ if self.detail:
171
+ status.append(f" {self.detail}", style="dim")
172
+ parts = [Text(self.partial[-400:], style="cyan")] if self.partial else []
173
+ return Group(*parts, status, Text(self.lang.text("hint"), style="dim"))
174
+
175
+ def refresh(self) -> None:
176
+ self.live.update(self._render(), refresh=True)
177
+
178
+ def set_status(self, status: str, detail: str = "") -> None:
179
+ with self.lock:
180
+ if status != self.status:
181
+ log.debug("status %s -> %s %s", self.status, status, detail)
182
+ self.status, self.detail = status, detail
183
+ self.refresh()
184
+
185
+ def show(self, markup: str) -> None:
186
+ self.live.console.print(markup)
187
+
188
+ def show_user(self, text: str, note: str = "") -> None:
189
+ suffix = f" [dim]{escape(note)}[/dim]" if note else ""
190
+ self.show(f"[bold green]{escape(self.lang.text('you'))}[/] {escape(text)}{suffix}")
191
+
192
+ def duration(self, seconds: float) -> str:
193
+ return self.lang.text("minutes", n=seconds / 60) if seconds >= 60 else self.lang.text("seconds", n=seconds)
194
+
195
+ # --- recorder / tts callbacks -------------------------------------------
196
+
197
+ def _on_speech_start(self) -> None:
198
+ with self.lock:
199
+ self.speech_started_at = time.monotonic()
200
+ self.set_status("hearing")
201
+ self.turn = Turn()
202
+ if self.dialog:
203
+ self.dialog.on_speech_start()
204
+
205
+ def _on_speech_end(self) -> None:
206
+ if self.turn:
207
+ self.turn.mark("speech_end")
208
+ self.set_status("transcribing")
209
+
210
+ def _on_audio_start(self) -> None:
211
+ if self.turn:
212
+ self.turn.mark("audio_start")
213
+ self.set_status("speaking")
214
+
215
+ def _on_audio_stop(self) -> None:
216
+ if self.turn:
217
+ self.turn.mark("audio_end")
218
+
219
+ def _before_sentence(self, sentence: str) -> None:
220
+ self.spoken_chars += len(sentence)
221
+ if self.turn:
222
+ self.turn.mark("sentence_synth_start", f"chars={len(sentence)} {sentence[:60]!r}")
223
+
224
+ def _after_sentence(self, sentence: str) -> None:
225
+ if self.turn:
226
+ self.turn.mark("sentence_synth_end", f"chars={len(sentence)}")
227
+
228
+ # --- conversation -------------------------------------------------------
229
+
230
+ def conversation(self) -> None:
231
+ while not self.stopping.is_set():
232
+ self.active.wait()
233
+ try:
234
+ self.listen_and_reply()
235
+ except Exception as error:
236
+ log.exception("turn failed")
237
+ self.show(f"[red]{escape(self.lang.text('error_see_log', error=error))}[/]")
238
+
239
+ def idle_status(self) -> str:
240
+ return "standby" if self.wake and self.session_until is None else "listening"
241
+
242
+ def listen_and_reply(self) -> None:
243
+ self.set_status(self.idle_status())
244
+ self.recorder.clear_audio_queue()
245
+ self.recorder.set_microphone(True)
246
+ log.info("listening session=%s", self.session_until is not None)
247
+ self.in_listen = True
248
+ try:
249
+ text = clean(self.recorder.text())
250
+ finally:
251
+ self.in_listen = False
252
+ self.recorder.set_microphone(False) # the assistant must not hear itself
253
+ if self.stopping.is_set():
254
+ return
255
+ if not (self.active.is_set() and text):
256
+ log.info("nothing to answer: active=%s text=%r", self.active.is_set(), text)
257
+ self.turn = None
258
+ return
259
+ self.turn = self.turn or Turn()
260
+ self.turn.mark("transcribed", f"{text!r}")
261
+
262
+ if self.wake:
263
+ # Speech that began after the follow-up window closed needs the wake word again.
264
+ if self.session_until is not None and self.speech_started_at > self.session_until:
265
+ self.end_session("timeout")
266
+ addressed = strip_wake(text, self.wake)
267
+ if self.session_until is None:
268
+ if addressed is None:
269
+ log.info("ignored (no wake word): %r", text)
270
+ self.show(f"[dim]{escape(self.lang.text('ignored', text=text[:80]))}[/dim]")
271
+ self.turn = None
272
+ return
273
+ self.start_session()
274
+ if addressed is not None:
275
+ text = addressed
276
+ if not text: # just the wake word: wait for the request itself
277
+ self.session_until = time.monotonic() + self.follow_up
278
+ self.turn = None
279
+ return
280
+ if is_stop(text, self.lang):
281
+ self.show_user(text)
282
+ self.end_session("stop")
283
+ self.turn = None
284
+ return
285
+ if is_hold(text, self.lang):
286
+ log.info("hold for %gs: %r", self.hold, text)
287
+ self.show_user(text, self.lang.text("waiting", duration=self.duration(self.hold)))
288
+ self.session_until = time.monotonic() + self.hold
289
+ self.turn = None
290
+ return
291
+ self.session_until = math.inf
292
+
293
+ self.reply(text)
294
+ if self.session_until is not None:
295
+ self.session_until = time.monotonic() + self.follow_up
296
+
297
+ def start_session(self) -> None:
298
+ log.info("session start")
299
+ self.session_until = math.inf
300
+ subprocess.run(["afplay", SESSION_START_SOUND])
301
+
302
+ def end_session(self, reason: str) -> None:
303
+ with self.lock:
304
+ if self.session_until is None:
305
+ return
306
+ log.info("session end reason=%s", reason)
307
+ self.session_until = None
308
+ if self.status == "listening":
309
+ self.set_status("standby")
310
+ self.show(f"[dim]{escape(self.lang.text('session_end', reason=self.lang.text(f'end_reason.{reason}')))}[/dim]")
311
+ subprocess.Popen(["afplay", SESSION_END_SOUND])
312
+
313
+ def session_timer(self) -> None:
314
+ """Closes the follow-up window and shows the countdown while waiting for the next phrase."""
315
+ while True:
316
+ time.sleep(0.25)
317
+ with self.lock: # speech can't start between the check and the action
318
+ until = self.session_until
319
+ if until is None or math.isinf(until) or self.status != "listening":
320
+ continue
321
+ remaining = until - time.monotonic()
322
+ if remaining <= 0:
323
+ self.end_session("timeout")
324
+ else:
325
+ seconds = math.ceil(remaining)
326
+ left = f"{seconds // 60}:{seconds % 60:02d}" if seconds > 60 else self.lang.text("seconds", n=seconds)
327
+ self.set_status("listening", self.lang.text("remaining", time=left))
328
+
329
+ def reply(self, text: str) -> None:
330
+ turn = self.turn or Turn()
331
+ self.turn = turn
332
+ self.show_user(text)
333
+ self.interrupted.clear()
334
+ self.spoken_chars = 0
335
+ self.set_status("thinking")
336
+
337
+ to_speak: queue.Queue[str | None] = queue.Queue()
338
+ self.tts.feed(speakable(iter(to_speak.get, None)))
339
+ player = threading.Thread(
340
+ target=self.tts.play,
341
+ name="tts-play",
342
+ kwargs={
343
+ "language": self.lang.code,
344
+ "minimum_sentence_length": 15,
345
+ "before_sentence_synthesized": self._before_sentence,
346
+ "on_sentence_synthesized": self._after_sentence,
347
+ "on_audio_chunk": self.dialog.on_voice if self.dialog else None,
348
+ },
349
+ )
350
+ player.start()
351
+
352
+ # Drain the brain fully even after an interrupt so its next turn starts clean.
353
+ answer = ""
354
+ turn.mark("prompt_sent")
355
+ try:
356
+ for kind, value in self.brain.ask(text):
357
+ turn.mark("brain_first_event")
358
+ if self.interrupted.is_set():
359
+ continue
360
+ if kind == "text":
361
+ turn.mark("first_text")
362
+ answer += value
363
+ self.partial = answer
364
+ to_speak.put(value)
365
+ if self.status != "speaking":
366
+ self.set_status("thinking")
367
+ else:
368
+ self.refresh()
369
+ elif kind == "tool":
370
+ turn.tools += 1
371
+ turn.mark("tool", value)
372
+ self.show(f"[yellow]🔧 {escape(value)}[/]")
373
+ self.set_status("tool", value[:60])
374
+ elif kind == "error":
375
+ turn.mark("error", value)
376
+ self.show(f"[red]{escape(self.lang.text('error', error=value))}[/]")
377
+ except BaseException:
378
+ self.tts.stop() # don't finish a half-spoken answer after the brain failed
379
+ raise
380
+ finally:
381
+ # The player blocks on this queue: without the end marker it would hold the TTS forever.
382
+ to_speak.put(None)
383
+ player.join()
384
+ self.partial = ""
385
+ self.turn = None
386
+ turn.mark("brain_done", f"chars={len(answer)}")
387
+ turn.mark("done")
388
+
389
+ interrupted = self.interrupted.is_set()
390
+ if answer.strip():
391
+ suffix = f" [dim]{escape(self.lang.text('interrupted'))}[/]" if interrupted else ""
392
+ self.show(f"[bold cyan]{escape(self.lang.text('assistant'))}[/] {escape(answer.strip())}{suffix}")
393
+ if not interrupted and "audio_start" not in turn.marks:
394
+ log.error(
395
+ "turn=%d answer has %d chars, %d sent to TTS, but no audio was played",
396
+ turn.id, len(answer), self.spoken_chars,
397
+ )
398
+ self.show(f"[red]{escape(self.lang.text('not_spoken'))}[/]")
399
+ log.info("turn=%d summary %s interrupted=%s", turn.id, turn.summary(), interrupted)
400
+ self.show(f"[dim]⏱ {turn.summary()}[/dim]")
401
+
402
+ def interrupt(self) -> None:
403
+ log.info("interrupt requested status=%s", self.status)
404
+ self.interrupted.set()
405
+ self.brain.abort()
406
+ self.tts.stop()
407
+
408
+ def toggle(self) -> None:
409
+ if self.active.is_set():
410
+ log.info("conversation off")
411
+ self.active.clear()
412
+ self.interrupt()
413
+ self.end_session("pause")
414
+ self.recorder.set_microphone(False)
415
+ if self.in_listen:
416
+ self.recorder.abort() # unblock recorder.text(); blocks if called outside of it
417
+ self.set_status("idle")
418
+ else:
419
+ log.info("conversation on")
420
+ self.active.set()
421
+
422
+ # --- main loop ----------------------------------------------------------
423
+
424
+ def run(self) -> None:
425
+ fd = sys.stdin.fileno()
426
+ saved = termios.tcgetattr(fd)
427
+ tty.setcbreak(fd) # single keypresses without breaking Rich's line output
428
+ threading.Thread(target=self.conversation, name="conversation", daemon=True).start()
429
+ if self.wake:
430
+ threading.Thread(target=self.session_timer, name="session-timer", daemon=True).start()
431
+ try:
432
+ with self.live:
433
+ while (key := os.read(fd, 1)) not in (b"q", b"Q"):
434
+ if key == b" ":
435
+ self.toggle()
436
+ elif key == b"\x1b":
437
+ self.interrupt()
438
+ finally:
439
+ log.info("shutdown")
440
+ self.stopping.set()
441
+ termios.tcsetattr(fd, termios.TCSADRAIN, saved)
442
+ self.brain.close()
443
+ self.tts.stop()
444
+ # RealtimeSTT stops its mic reader process only while the mic flag is on; otherwise the orphaned
445
+ # reader keeps the microphone (and the macOS mic indicator) busy after exit.
446
+ self.recorder.set_microphone(True)
447
+ with contextlib.redirect_stdout(io.StringIO()): # RealtimeSTT prints its shutdown into the UI
448
+ self.recorder.shutdown()
449
+ if self.dialog:
450
+ self.dialog.save()
451
+ if self.dialog.path.exists():
452
+ self.console.print(escape(self.lang.text("recording_saved", path=home_relative(self.dialog.path))))
453
+
454
+
455
+ def read_config(parser: argparse.ArgumentParser) -> dict:
456
+ """config.toml values become defaults; set_defaults skips argparse's own checks, so they are repeated here."""
457
+ config = {key.replace("-", "_"): value for key, value in tomllib.loads(CONFIG_PATH.read_text()).items()}
458
+ actions = {action.dest: action for action in parser._actions}
459
+ for key, value in config.items():
460
+ action = actions.get(key)
461
+ name = key.replace("_", "-")
462
+ if action is None or key == "help":
463
+ parser.error(f"{CONFIG_PATH}: unknown key {name!r}")
464
+ if isinstance(action, argparse._StoreTrueAction):
465
+ expected = isinstance(value, bool)
466
+ elif action.type is float:
467
+ expected = isinstance(value, (int, float)) and not isinstance(value, bool)
468
+ elif action.nargs == "*":
469
+ expected = isinstance(value, list) and all(isinstance(item, str) for item in value)
470
+ elif key == "record":
471
+ expected = isinstance(value, (bool, str))
472
+ else:
473
+ expected = isinstance(value, str)
474
+ if not expected:
475
+ parser.error(f"{CONFIG_PATH}: wrong type for {name!r}: {value!r}")
476
+ if action.choices and value not in action.choices:
477
+ parser.error(f"{CONFIG_PATH}: {name} = {value!r}, expected one of {', '.join(map(str, action.choices))}")
478
+ return config
479
+
480
+
481
+ def main() -> None:
482
+ parser = argparse.ArgumentParser(prog="beseda", description=__doc__, epilog=f"Defaults for any option: {CONFIG_PATH}")
483
+ parser.add_argument("--language", choices=available(), default="ru", help="language pack: what you speak and hear")
484
+ parser.add_argument("--brain", choices=BRAINS, default="pi")
485
+ parser.add_argument("--model", help="model for the brain (default: DeepSeek V4 Flash)")
486
+ parser.add_argument("--whisper", default="small", help="Whisper model: small (fast) or turbo (more accurate, ~2 s per phrase)")
487
+ parser.add_argument("--vocabulary", nargs="*", default=[], metavar="TERM", help="terms Whisper should spell exactly like this")
488
+ parser.add_argument("--tts", choices=TTS_ENGINES, default="silero", help="speech synthesis engine")
489
+ parser.add_argument("--voice", help="TTS voice (default: the first one in the language pack)")
490
+ parser.add_argument("--wake-word", help='wake word (default: from the language pack); "" answers everything')
491
+ parser.add_argument("--follow-up", type=float, default=8, help="seconds after an answer to keep talking without the wake word")
492
+ parser.add_argument("--hold", type=float, default=120, help="seconds a hold phrase («подожди», “hold on”) extends the wait")
493
+ parser.add_argument(
494
+ "--record",
495
+ nargs="?",
496
+ const=True,
497
+ metavar="PATH",
498
+ help="record the whole dialog to a WAV with the real pauses (default: ~/Downloads/)",
499
+ )
500
+ parser.add_argument("--log-days", type=float, default=14, help="days to keep logs (they hold the full dialog text)")
501
+ parser.add_argument("--debug", action="store_true", help="verbose log: every pi event and status change")
502
+ if CONFIG_PATH.exists():
503
+ parser.set_defaults(**read_config(parser))
504
+ args = parser.parse_args()
505
+ language = load_language(args.language)
506
+ if args.wake_word is None:
507
+ args.wake_word = language.wake_word
508
+ if args.record is True:
509
+ args.record = str(RECORDINGS_DIR / f"beseda-{datetime.now():%Y%m%d-%H%M%S}.wav")
510
+ App(args, language, setup_logging(args.debug, args.log_days)).run()
511
+
512
+
513
+ if __name__ == "__main__":
514
+ main()