beseda 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- beseda/__init__.py +0 -0
- beseda/__main__.py +514 -0
- beseda/brains.py +199 -0
- beseda/language.py +74 -0
- beseda/languages/en.toml +61 -0
- beseda/languages/ru.toml +64 -0
- beseda/models.py +33 -0
- beseda/recording.py +98 -0
- beseda/samples.py +58 -0
- beseda/stt.py +64 -0
- beseda/tts.py +160 -0
- beseda/wake.py +30 -0
- beseda-0.1.0.dist-info/METADATA +213 -0
- beseda-0.1.0.dist-info/RECORD +17 -0
- beseda-0.1.0.dist-info/WHEEL +4 -0
- beseda-0.1.0.dist-info/entry_points.txt +4 -0
- beseda-0.1.0.dist-info/licenses/LICENSE +21 -0
beseda/__init__.py
ADDED
|
File without changes
|
beseda/__main__.py
ADDED
|
@@ -0,0 +1,514 @@
|
|
|
1
|
+
"""Voice conversations with your coding agent: say the wake word, talk, listen to the answer."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import contextlib
|
|
5
|
+
import io
|
|
6
|
+
import logging
|
|
7
|
+
import math
|
|
8
|
+
import os
|
|
9
|
+
import queue
|
|
10
|
+
import subprocess
|
|
11
|
+
import sys
|
|
12
|
+
import termios
|
|
13
|
+
import threading
|
|
14
|
+
import time
|
|
15
|
+
import tomllib
|
|
16
|
+
import tty
|
|
17
|
+
from datetime import datetime
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
from rich.console import Console, Group
|
|
21
|
+
from rich.live import Live
|
|
22
|
+
from rich.markup import escape
|
|
23
|
+
from rich.text import Text
|
|
24
|
+
from RealtimeSTT import AudioToTextRecorder
|
|
25
|
+
from RealtimeTTS import TextToAudioStream
|
|
26
|
+
|
|
27
|
+
from beseda.brains import BRAINS, Brain
|
|
28
|
+
from beseda.language import Language, available
|
|
29
|
+
from beseda.language import load as load_language
|
|
30
|
+
from beseda.recording import DialogRecorder
|
|
31
|
+
from beseda.stt import clean, recorder_options
|
|
32
|
+
from beseda.tts import TTS_ENGINES, create_engine, speakable
|
|
33
|
+
from beseda.wake import is_hold, is_stop, strip_wake, wake_pattern
|
|
34
|
+
|
|
35
|
+
log = logging.getLogger("beseda")
|
|
36
|
+
|
|
37
|
+
LOG_DIR = Path.home() / ".beseda" / "logs"
|
|
38
|
+
RECORDINGS_DIR = Path.home() / "Downloads"
|
|
39
|
+
CONFIG_PATH = Path.home() / ".beseda" / "config.toml"
|
|
40
|
+
SESSION_START_SOUND = "/System/Library/Sounds/Tink.aiff"
|
|
41
|
+
SESSION_END_SOUND = "/System/Library/Sounds/Bottle.aiff"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def home_relative(path: Path | str) -> str:
|
|
45
|
+
"""~/.beseda/logs/... instead of /Users/<name>/...: the terminal often ends up in screenshots."""
|
|
46
|
+
return str(path).replace(str(Path.home()), "~", 1)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def setup_logging(debug: bool, keep_days: float) -> Path:
|
|
50
|
+
"""One file per session; nothing goes to the terminal so the UI stays intact.
|
|
51
|
+
Logs hold everything said in the dialog, so sessions older than `keep_days` are deleted."""
|
|
52
|
+
LOG_DIR.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
cutoff = time.time() - keep_days * 86400
|
|
54
|
+
for old in LOG_DIR.glob("beseda-*.log"):
|
|
55
|
+
if old.stat().st_mtime < cutoff:
|
|
56
|
+
old.unlink()
|
|
57
|
+
path = LOG_DIR / f"beseda-{datetime.now():%Y%m%d-%H%M%S}.log"
|
|
58
|
+
handler = logging.FileHandler(path, encoding="utf-8")
|
|
59
|
+
handler.setLevel(logging.DEBUG if debug else logging.INFO)
|
|
60
|
+
handler.setFormatter(
|
|
61
|
+
logging.Formatter("%(asctime)s.%(msecs)03d %(levelname)-7s [%(threadName)s] %(name)s: %(message)s", "%H:%M:%S")
|
|
62
|
+
)
|
|
63
|
+
logging.getLogger().addHandler(handler)
|
|
64
|
+
# Native libraries (whisper.cpp and its VAD, in RealtimeSTT's worker processes) print to file descriptor 2
|
|
65
|
+
# and would scramble the terminal UI. Point fd 2, inherited by those processes, at the log; Python's own
|
|
66
|
+
# error output keeps going to the terminal.
|
|
67
|
+
sys.stderr = os.fdopen(os.dup(2), "w", buffering=1)
|
|
68
|
+
os.dup2(handler.stream.fileno(), 2)
|
|
69
|
+
logging.getLogger().setLevel(logging.WARNING)
|
|
70
|
+
log.setLevel(logging.DEBUG if debug else logging.INFO)
|
|
71
|
+
threading.excepthook = lambda a: log.critical(
|
|
72
|
+
"uncaught exception in thread %s", a.thread.name, exc_info=(a.exc_type, a.exc_value, a.exc_traceback)
|
|
73
|
+
)
|
|
74
|
+
return path
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class Turn:
|
|
78
|
+
"""Timeline of one exchange, measured from the moment the user starts speaking."""
|
|
79
|
+
|
|
80
|
+
count = 0
|
|
81
|
+
|
|
82
|
+
def __init__(self):
|
|
83
|
+
Turn.count += 1
|
|
84
|
+
self.id = Turn.count
|
|
85
|
+
self.t0 = time.monotonic()
|
|
86
|
+
self.marks: dict[str, float] = {}
|
|
87
|
+
self.tools = 0
|
|
88
|
+
self.mark("speech_start")
|
|
89
|
+
|
|
90
|
+
def mark(self, name: str, detail: str = "") -> None:
|
|
91
|
+
elapsed = time.monotonic() - self.t0
|
|
92
|
+
first = name not in self.marks
|
|
93
|
+
self.marks.setdefault(name, elapsed)
|
|
94
|
+
if first or detail:
|
|
95
|
+
log.info("turn=%d +%.2fs %s %s", self.id, elapsed, name, detail)
|
|
96
|
+
|
|
97
|
+
def span(self, start: str, end: str) -> str:
|
|
98
|
+
if start in self.marks and end in self.marks:
|
|
99
|
+
return f"{self.marks[end] - self.marks[start]:.2f}s"
|
|
100
|
+
return "-"
|
|
101
|
+
|
|
102
|
+
def summary(self) -> str:
|
|
103
|
+
return (
|
|
104
|
+
f"stt={self.span('speech_end', 'transcribed')} "
|
|
105
|
+
f"llm_first_token={self.span('prompt_sent', 'first_text')} "
|
|
106
|
+
f"tts_first_audio={self.span('first_text', 'audio_start')} "
|
|
107
|
+
f"voice_to_voice={self.span('speech_end', 'audio_start')} "
|
|
108
|
+
f"tools={self.tools} total={self.span('speech_start', 'done')}"
|
|
109
|
+
)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
class App:
|
|
113
|
+
def __init__(self, args: argparse.Namespace, language: Language, log_path: Path):
|
|
114
|
+
self.lang = language
|
|
115
|
+
self.console = Console()
|
|
116
|
+
self.status = "idle"
|
|
117
|
+
self.detail = ""
|
|
118
|
+
self.partial = ""
|
|
119
|
+
self.live = Live(self._render(), console=self.console, auto_refresh=False)
|
|
120
|
+
self.active = threading.Event()
|
|
121
|
+
self.interrupted = threading.Event()
|
|
122
|
+
self.stopping = threading.Event()
|
|
123
|
+
self.in_listen = False
|
|
124
|
+
self.turn: Turn | None = None
|
|
125
|
+
self.spoken_chars = 0
|
|
126
|
+
self.wake = wake_pattern(args.wake_word, language.wake_endings) if args.wake_word else None
|
|
127
|
+
self.wake_word = args.wake_word
|
|
128
|
+
self.follow_up = args.follow_up
|
|
129
|
+
self.hold = args.hold
|
|
130
|
+
# None: standby, waiting for the wake word; inf: answering; otherwise the follow-up deadline.
|
|
131
|
+
self.session_until: float | None = None
|
|
132
|
+
self.speech_started_at = 0.0
|
|
133
|
+
# Guards status and the session deadline: the session timer must not close a session whose next
|
|
134
|
+
# phrase has just started.
|
|
135
|
+
self.lock = threading.RLock()
|
|
136
|
+
|
|
137
|
+
log.info("start args=%s", vars(args))
|
|
138
|
+
with self.console.status(language.text("loading")):
|
|
139
|
+
started = time.monotonic()
|
|
140
|
+
self.brain: Brain = BRAINS[args.brain](args.model, language.voice_prompt)
|
|
141
|
+
engine = create_engine(args.tts, language, args.voice)
|
|
142
|
+
self.dialog = DialogRecorder(Path(args.record), engine.rate) if args.record else None
|
|
143
|
+
self.tts = TextToAudioStream(
|
|
144
|
+
engine,
|
|
145
|
+
language=language.code,
|
|
146
|
+
on_audio_stream_start=self._on_audio_start,
|
|
147
|
+
on_audio_stream_stop=self._on_audio_stop,
|
|
148
|
+
)
|
|
149
|
+
self.recorder = AudioToTextRecorder(
|
|
150
|
+
**recorder_options(args.whisper, args.vocabulary, language),
|
|
151
|
+
spinner=False,
|
|
152
|
+
level=logging.ERROR,
|
|
153
|
+
no_log_file=True,
|
|
154
|
+
post_speech_silence_duration=0.7,
|
|
155
|
+
on_recording_start=self._on_speech_start,
|
|
156
|
+
on_recording_stop=self._on_speech_end,
|
|
157
|
+
on_recorded_chunk=self.dialog.on_mic if self.dialog else None,
|
|
158
|
+
)
|
|
159
|
+
self.recorder.set_microphone(False)
|
|
160
|
+
log.info("components ready in %.1fs", time.monotonic() - started)
|
|
161
|
+
self.console.print(f"[dim]{escape(language.text('log', path=home_relative(log_path)))}[/dim]")
|
|
162
|
+
if self.wake:
|
|
163
|
+
self.console.print(language.text("intro", wake=self.wake_word.capitalize(), follow_up=f"{self.follow_up:g}"))
|
|
164
|
+
self.active.set() # wake word mode listens from the start, like a smart speaker
|
|
165
|
+
|
|
166
|
+
# --- UI -----------------------------------------------------------------
|
|
167
|
+
|
|
168
|
+
def _render(self) -> Group:
|
|
169
|
+
status = Text(self.lang.text(f"status.{self.status}"), style="bold")
|
|
170
|
+
if self.detail:
|
|
171
|
+
status.append(f" {self.detail}", style="dim")
|
|
172
|
+
parts = [Text(self.partial[-400:], style="cyan")] if self.partial else []
|
|
173
|
+
return Group(*parts, status, Text(self.lang.text("hint"), style="dim"))
|
|
174
|
+
|
|
175
|
+
def refresh(self) -> None:
|
|
176
|
+
self.live.update(self._render(), refresh=True)
|
|
177
|
+
|
|
178
|
+
def set_status(self, status: str, detail: str = "") -> None:
|
|
179
|
+
with self.lock:
|
|
180
|
+
if status != self.status:
|
|
181
|
+
log.debug("status %s -> %s %s", self.status, status, detail)
|
|
182
|
+
self.status, self.detail = status, detail
|
|
183
|
+
self.refresh()
|
|
184
|
+
|
|
185
|
+
def show(self, markup: str) -> None:
|
|
186
|
+
self.live.console.print(markup)
|
|
187
|
+
|
|
188
|
+
def show_user(self, text: str, note: str = "") -> None:
|
|
189
|
+
suffix = f" [dim]{escape(note)}[/dim]" if note else ""
|
|
190
|
+
self.show(f"[bold green]{escape(self.lang.text('you'))}[/] {escape(text)}{suffix}")
|
|
191
|
+
|
|
192
|
+
def duration(self, seconds: float) -> str:
|
|
193
|
+
return self.lang.text("minutes", n=seconds / 60) if seconds >= 60 else self.lang.text("seconds", n=seconds)
|
|
194
|
+
|
|
195
|
+
# --- recorder / tts callbacks -------------------------------------------
|
|
196
|
+
|
|
197
|
+
def _on_speech_start(self) -> None:
|
|
198
|
+
with self.lock:
|
|
199
|
+
self.speech_started_at = time.monotonic()
|
|
200
|
+
self.set_status("hearing")
|
|
201
|
+
self.turn = Turn()
|
|
202
|
+
if self.dialog:
|
|
203
|
+
self.dialog.on_speech_start()
|
|
204
|
+
|
|
205
|
+
def _on_speech_end(self) -> None:
|
|
206
|
+
if self.turn:
|
|
207
|
+
self.turn.mark("speech_end")
|
|
208
|
+
self.set_status("transcribing")
|
|
209
|
+
|
|
210
|
+
def _on_audio_start(self) -> None:
|
|
211
|
+
if self.turn:
|
|
212
|
+
self.turn.mark("audio_start")
|
|
213
|
+
self.set_status("speaking")
|
|
214
|
+
|
|
215
|
+
def _on_audio_stop(self) -> None:
|
|
216
|
+
if self.turn:
|
|
217
|
+
self.turn.mark("audio_end")
|
|
218
|
+
|
|
219
|
+
def _before_sentence(self, sentence: str) -> None:
|
|
220
|
+
self.spoken_chars += len(sentence)
|
|
221
|
+
if self.turn:
|
|
222
|
+
self.turn.mark("sentence_synth_start", f"chars={len(sentence)} {sentence[:60]!r}")
|
|
223
|
+
|
|
224
|
+
def _after_sentence(self, sentence: str) -> None:
|
|
225
|
+
if self.turn:
|
|
226
|
+
self.turn.mark("sentence_synth_end", f"chars={len(sentence)}")
|
|
227
|
+
|
|
228
|
+
# --- conversation -------------------------------------------------------
|
|
229
|
+
|
|
230
|
+
def conversation(self) -> None:
|
|
231
|
+
while not self.stopping.is_set():
|
|
232
|
+
self.active.wait()
|
|
233
|
+
try:
|
|
234
|
+
self.listen_and_reply()
|
|
235
|
+
except Exception as error:
|
|
236
|
+
log.exception("turn failed")
|
|
237
|
+
self.show(f"[red]{escape(self.lang.text('error_see_log', error=error))}[/]")
|
|
238
|
+
|
|
239
|
+
def idle_status(self) -> str:
|
|
240
|
+
return "standby" if self.wake and self.session_until is None else "listening"
|
|
241
|
+
|
|
242
|
+
def listen_and_reply(self) -> None:
|
|
243
|
+
self.set_status(self.idle_status())
|
|
244
|
+
self.recorder.clear_audio_queue()
|
|
245
|
+
self.recorder.set_microphone(True)
|
|
246
|
+
log.info("listening session=%s", self.session_until is not None)
|
|
247
|
+
self.in_listen = True
|
|
248
|
+
try:
|
|
249
|
+
text = clean(self.recorder.text())
|
|
250
|
+
finally:
|
|
251
|
+
self.in_listen = False
|
|
252
|
+
self.recorder.set_microphone(False) # the assistant must not hear itself
|
|
253
|
+
if self.stopping.is_set():
|
|
254
|
+
return
|
|
255
|
+
if not (self.active.is_set() and text):
|
|
256
|
+
log.info("nothing to answer: active=%s text=%r", self.active.is_set(), text)
|
|
257
|
+
self.turn = None
|
|
258
|
+
return
|
|
259
|
+
self.turn = self.turn or Turn()
|
|
260
|
+
self.turn.mark("transcribed", f"{text!r}")
|
|
261
|
+
|
|
262
|
+
if self.wake:
|
|
263
|
+
# Speech that began after the follow-up window closed needs the wake word again.
|
|
264
|
+
if self.session_until is not None and self.speech_started_at > self.session_until:
|
|
265
|
+
self.end_session("timeout")
|
|
266
|
+
addressed = strip_wake(text, self.wake)
|
|
267
|
+
if self.session_until is None:
|
|
268
|
+
if addressed is None:
|
|
269
|
+
log.info("ignored (no wake word): %r", text)
|
|
270
|
+
self.show(f"[dim]{escape(self.lang.text('ignored', text=text[:80]))}[/dim]")
|
|
271
|
+
self.turn = None
|
|
272
|
+
return
|
|
273
|
+
self.start_session()
|
|
274
|
+
if addressed is not None:
|
|
275
|
+
text = addressed
|
|
276
|
+
if not text: # just the wake word: wait for the request itself
|
|
277
|
+
self.session_until = time.monotonic() + self.follow_up
|
|
278
|
+
self.turn = None
|
|
279
|
+
return
|
|
280
|
+
if is_stop(text, self.lang):
|
|
281
|
+
self.show_user(text)
|
|
282
|
+
self.end_session("stop")
|
|
283
|
+
self.turn = None
|
|
284
|
+
return
|
|
285
|
+
if is_hold(text, self.lang):
|
|
286
|
+
log.info("hold for %gs: %r", self.hold, text)
|
|
287
|
+
self.show_user(text, self.lang.text("waiting", duration=self.duration(self.hold)))
|
|
288
|
+
self.session_until = time.monotonic() + self.hold
|
|
289
|
+
self.turn = None
|
|
290
|
+
return
|
|
291
|
+
self.session_until = math.inf
|
|
292
|
+
|
|
293
|
+
self.reply(text)
|
|
294
|
+
if self.session_until is not None:
|
|
295
|
+
self.session_until = time.monotonic() + self.follow_up
|
|
296
|
+
|
|
297
|
+
def start_session(self) -> None:
|
|
298
|
+
log.info("session start")
|
|
299
|
+
self.session_until = math.inf
|
|
300
|
+
subprocess.run(["afplay", SESSION_START_SOUND])
|
|
301
|
+
|
|
302
|
+
def end_session(self, reason: str) -> None:
|
|
303
|
+
with self.lock:
|
|
304
|
+
if self.session_until is None:
|
|
305
|
+
return
|
|
306
|
+
log.info("session end reason=%s", reason)
|
|
307
|
+
self.session_until = None
|
|
308
|
+
if self.status == "listening":
|
|
309
|
+
self.set_status("standby")
|
|
310
|
+
self.show(f"[dim]{escape(self.lang.text('session_end', reason=self.lang.text(f'end_reason.{reason}')))}[/dim]")
|
|
311
|
+
subprocess.Popen(["afplay", SESSION_END_SOUND])
|
|
312
|
+
|
|
313
|
+
def session_timer(self) -> None:
|
|
314
|
+
"""Closes the follow-up window and shows the countdown while waiting for the next phrase."""
|
|
315
|
+
while True:
|
|
316
|
+
time.sleep(0.25)
|
|
317
|
+
with self.lock: # speech can't start between the check and the action
|
|
318
|
+
until = self.session_until
|
|
319
|
+
if until is None or math.isinf(until) or self.status != "listening":
|
|
320
|
+
continue
|
|
321
|
+
remaining = until - time.monotonic()
|
|
322
|
+
if remaining <= 0:
|
|
323
|
+
self.end_session("timeout")
|
|
324
|
+
else:
|
|
325
|
+
seconds = math.ceil(remaining)
|
|
326
|
+
left = f"{seconds // 60}:{seconds % 60:02d}" if seconds > 60 else self.lang.text("seconds", n=seconds)
|
|
327
|
+
self.set_status("listening", self.lang.text("remaining", time=left))
|
|
328
|
+
|
|
329
|
+
def reply(self, text: str) -> None:
|
|
330
|
+
turn = self.turn or Turn()
|
|
331
|
+
self.turn = turn
|
|
332
|
+
self.show_user(text)
|
|
333
|
+
self.interrupted.clear()
|
|
334
|
+
self.spoken_chars = 0
|
|
335
|
+
self.set_status("thinking")
|
|
336
|
+
|
|
337
|
+
to_speak: queue.Queue[str | None] = queue.Queue()
|
|
338
|
+
self.tts.feed(speakable(iter(to_speak.get, None)))
|
|
339
|
+
player = threading.Thread(
|
|
340
|
+
target=self.tts.play,
|
|
341
|
+
name="tts-play",
|
|
342
|
+
kwargs={
|
|
343
|
+
"language": self.lang.code,
|
|
344
|
+
"minimum_sentence_length": 15,
|
|
345
|
+
"before_sentence_synthesized": self._before_sentence,
|
|
346
|
+
"on_sentence_synthesized": self._after_sentence,
|
|
347
|
+
"on_audio_chunk": self.dialog.on_voice if self.dialog else None,
|
|
348
|
+
},
|
|
349
|
+
)
|
|
350
|
+
player.start()
|
|
351
|
+
|
|
352
|
+
# Drain the brain fully even after an interrupt so its next turn starts clean.
|
|
353
|
+
answer = ""
|
|
354
|
+
turn.mark("prompt_sent")
|
|
355
|
+
try:
|
|
356
|
+
for kind, value in self.brain.ask(text):
|
|
357
|
+
turn.mark("brain_first_event")
|
|
358
|
+
if self.interrupted.is_set():
|
|
359
|
+
continue
|
|
360
|
+
if kind == "text":
|
|
361
|
+
turn.mark("first_text")
|
|
362
|
+
answer += value
|
|
363
|
+
self.partial = answer
|
|
364
|
+
to_speak.put(value)
|
|
365
|
+
if self.status != "speaking":
|
|
366
|
+
self.set_status("thinking")
|
|
367
|
+
else:
|
|
368
|
+
self.refresh()
|
|
369
|
+
elif kind == "tool":
|
|
370
|
+
turn.tools += 1
|
|
371
|
+
turn.mark("tool", value)
|
|
372
|
+
self.show(f"[yellow]🔧 {escape(value)}[/]")
|
|
373
|
+
self.set_status("tool", value[:60])
|
|
374
|
+
elif kind == "error":
|
|
375
|
+
turn.mark("error", value)
|
|
376
|
+
self.show(f"[red]{escape(self.lang.text('error', error=value))}[/]")
|
|
377
|
+
except BaseException:
|
|
378
|
+
self.tts.stop() # don't finish a half-spoken answer after the brain failed
|
|
379
|
+
raise
|
|
380
|
+
finally:
|
|
381
|
+
# The player blocks on this queue: without the end marker it would hold the TTS forever.
|
|
382
|
+
to_speak.put(None)
|
|
383
|
+
player.join()
|
|
384
|
+
self.partial = ""
|
|
385
|
+
self.turn = None
|
|
386
|
+
turn.mark("brain_done", f"chars={len(answer)}")
|
|
387
|
+
turn.mark("done")
|
|
388
|
+
|
|
389
|
+
interrupted = self.interrupted.is_set()
|
|
390
|
+
if answer.strip():
|
|
391
|
+
suffix = f" [dim]{escape(self.lang.text('interrupted'))}[/]" if interrupted else ""
|
|
392
|
+
self.show(f"[bold cyan]{escape(self.lang.text('assistant'))}[/] {escape(answer.strip())}{suffix}")
|
|
393
|
+
if not interrupted and "audio_start" not in turn.marks:
|
|
394
|
+
log.error(
|
|
395
|
+
"turn=%d answer has %d chars, %d sent to TTS, but no audio was played",
|
|
396
|
+
turn.id, len(answer), self.spoken_chars,
|
|
397
|
+
)
|
|
398
|
+
self.show(f"[red]{escape(self.lang.text('not_spoken'))}[/]")
|
|
399
|
+
log.info("turn=%d summary %s interrupted=%s", turn.id, turn.summary(), interrupted)
|
|
400
|
+
self.show(f"[dim]⏱ {turn.summary()}[/dim]")
|
|
401
|
+
|
|
402
|
+
def interrupt(self) -> None:
|
|
403
|
+
log.info("interrupt requested status=%s", self.status)
|
|
404
|
+
self.interrupted.set()
|
|
405
|
+
self.brain.abort()
|
|
406
|
+
self.tts.stop()
|
|
407
|
+
|
|
408
|
+
def toggle(self) -> None:
|
|
409
|
+
if self.active.is_set():
|
|
410
|
+
log.info("conversation off")
|
|
411
|
+
self.active.clear()
|
|
412
|
+
self.interrupt()
|
|
413
|
+
self.end_session("pause")
|
|
414
|
+
self.recorder.set_microphone(False)
|
|
415
|
+
if self.in_listen:
|
|
416
|
+
self.recorder.abort() # unblock recorder.text(); blocks if called outside of it
|
|
417
|
+
self.set_status("idle")
|
|
418
|
+
else:
|
|
419
|
+
log.info("conversation on")
|
|
420
|
+
self.active.set()
|
|
421
|
+
|
|
422
|
+
# --- main loop ----------------------------------------------------------
|
|
423
|
+
|
|
424
|
+
def run(self) -> None:
|
|
425
|
+
fd = sys.stdin.fileno()
|
|
426
|
+
saved = termios.tcgetattr(fd)
|
|
427
|
+
tty.setcbreak(fd) # single keypresses without breaking Rich's line output
|
|
428
|
+
threading.Thread(target=self.conversation, name="conversation", daemon=True).start()
|
|
429
|
+
if self.wake:
|
|
430
|
+
threading.Thread(target=self.session_timer, name="session-timer", daemon=True).start()
|
|
431
|
+
try:
|
|
432
|
+
with self.live:
|
|
433
|
+
while (key := os.read(fd, 1)) not in (b"q", b"Q"):
|
|
434
|
+
if key == b" ":
|
|
435
|
+
self.toggle()
|
|
436
|
+
elif key == b"\x1b":
|
|
437
|
+
self.interrupt()
|
|
438
|
+
finally:
|
|
439
|
+
log.info("shutdown")
|
|
440
|
+
self.stopping.set()
|
|
441
|
+
termios.tcsetattr(fd, termios.TCSADRAIN, saved)
|
|
442
|
+
self.brain.close()
|
|
443
|
+
self.tts.stop()
|
|
444
|
+
# RealtimeSTT stops its mic reader process only while the mic flag is on; otherwise the orphaned
|
|
445
|
+
# reader keeps the microphone (and the macOS mic indicator) busy after exit.
|
|
446
|
+
self.recorder.set_microphone(True)
|
|
447
|
+
with contextlib.redirect_stdout(io.StringIO()): # RealtimeSTT prints its shutdown into the UI
|
|
448
|
+
self.recorder.shutdown()
|
|
449
|
+
if self.dialog:
|
|
450
|
+
self.dialog.save()
|
|
451
|
+
if self.dialog.path.exists():
|
|
452
|
+
self.console.print(escape(self.lang.text("recording_saved", path=home_relative(self.dialog.path))))
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def read_config(parser: argparse.ArgumentParser) -> dict:
|
|
456
|
+
"""config.toml values become defaults; set_defaults skips argparse's own checks, so they are repeated here."""
|
|
457
|
+
config = {key.replace("-", "_"): value for key, value in tomllib.loads(CONFIG_PATH.read_text()).items()}
|
|
458
|
+
actions = {action.dest: action for action in parser._actions}
|
|
459
|
+
for key, value in config.items():
|
|
460
|
+
action = actions.get(key)
|
|
461
|
+
name = key.replace("_", "-")
|
|
462
|
+
if action is None or key == "help":
|
|
463
|
+
parser.error(f"{CONFIG_PATH}: unknown key {name!r}")
|
|
464
|
+
if isinstance(action, argparse._StoreTrueAction):
|
|
465
|
+
expected = isinstance(value, bool)
|
|
466
|
+
elif action.type is float:
|
|
467
|
+
expected = isinstance(value, (int, float)) and not isinstance(value, bool)
|
|
468
|
+
elif action.nargs == "*":
|
|
469
|
+
expected = isinstance(value, list) and all(isinstance(item, str) for item in value)
|
|
470
|
+
elif key == "record":
|
|
471
|
+
expected = isinstance(value, (bool, str))
|
|
472
|
+
else:
|
|
473
|
+
expected = isinstance(value, str)
|
|
474
|
+
if not expected:
|
|
475
|
+
parser.error(f"{CONFIG_PATH}: wrong type for {name!r}: {value!r}")
|
|
476
|
+
if action.choices and value not in action.choices:
|
|
477
|
+
parser.error(f"{CONFIG_PATH}: {name} = {value!r}, expected one of {', '.join(map(str, action.choices))}")
|
|
478
|
+
return config
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def main() -> None:
|
|
482
|
+
parser = argparse.ArgumentParser(prog="beseda", description=__doc__, epilog=f"Defaults for any option: {CONFIG_PATH}")
|
|
483
|
+
parser.add_argument("--language", choices=available(), default="ru", help="language pack: what you speak and hear")
|
|
484
|
+
parser.add_argument("--brain", choices=BRAINS, default="pi")
|
|
485
|
+
parser.add_argument("--model", help="model for the brain (default: DeepSeek V4 Flash)")
|
|
486
|
+
parser.add_argument("--whisper", default="small", help="Whisper model: small (fast) or turbo (more accurate, ~2 s per phrase)")
|
|
487
|
+
parser.add_argument("--vocabulary", nargs="*", default=[], metavar="TERM", help="terms Whisper should spell exactly like this")
|
|
488
|
+
parser.add_argument("--tts", choices=TTS_ENGINES, default="silero", help="speech synthesis engine")
|
|
489
|
+
parser.add_argument("--voice", help="TTS voice (default: the first one in the language pack)")
|
|
490
|
+
parser.add_argument("--wake-word", help='wake word (default: from the language pack); "" answers everything')
|
|
491
|
+
parser.add_argument("--follow-up", type=float, default=8, help="seconds after an answer to keep talking without the wake word")
|
|
492
|
+
parser.add_argument("--hold", type=float, default=120, help="seconds a hold phrase («подожди», “hold on”) extends the wait")
|
|
493
|
+
parser.add_argument(
|
|
494
|
+
"--record",
|
|
495
|
+
nargs="?",
|
|
496
|
+
const=True,
|
|
497
|
+
metavar="PATH",
|
|
498
|
+
help="record the whole dialog to a WAV with the real pauses (default: ~/Downloads/)",
|
|
499
|
+
)
|
|
500
|
+
parser.add_argument("--log-days", type=float, default=14, help="days to keep logs (they hold the full dialog text)")
|
|
501
|
+
parser.add_argument("--debug", action="store_true", help="verbose log: every pi event and status change")
|
|
502
|
+
if CONFIG_PATH.exists():
|
|
503
|
+
parser.set_defaults(**read_config(parser))
|
|
504
|
+
args = parser.parse_args()
|
|
505
|
+
language = load_language(args.language)
|
|
506
|
+
if args.wake_word is None:
|
|
507
|
+
args.wake_word = language.wake_word
|
|
508
|
+
if args.record is True:
|
|
509
|
+
args.record = str(RECORDINGS_DIR / f"beseda-{datetime.now():%Y%m%d-%H%M%S}.wav")
|
|
510
|
+
App(args, language, setup_logging(args.debug, args.log_days)).run()
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
if __name__ == "__main__":
|
|
514
|
+
main()
|