@bojackduy/opencode-voice 0.8.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +95 -21
- package/lib/chatterbox-server.js +428 -0
- package/lib/conversation.js +141 -59
- package/lib/tts.js +260 -42
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -243,6 +243,65 @@ English reply no longer flips the whole thing to the Vietnamese voice, and
|
|
|
243
243
|
vice versa. Toneless Vietnamese (no diacritics) still reads as English -
|
|
244
244
|
that is not distinguishable from English by this heuristic.
|
|
245
245
|
|
|
246
|
+
### Chatterbox engine (optional)
|
|
247
|
+
|
|
248
|
+
Piper is the default and needs no setup beyond the section above. For more
|
|
249
|
+
natural-sounding English, opt into [Chatterbox](https://github.com/resemble-ai/chatterbox)
|
|
250
|
+
(Resemble AI, MIT) as the synthesis engine. It runs as a managed sidecar
|
|
251
|
+
(`vendor/chatterbox_server.py`, stdlib HTTP only): the model loads once, the
|
|
252
|
+
plugin probes readiness, and the sidecar is killed on dispose. Playback still
|
|
253
|
+
goes through the existing `play` path, so `/tts-stop` and cancel behave the
|
|
254
|
+
same on both engines.
|
|
255
|
+
|
|
256
|
+
Setup (isolated venv, never touches your system python):
|
|
257
|
+
|
|
258
|
+
```bash
|
|
259
|
+
python3 -m venv ~/.local/share/opencode-voice/chatterbox-venv
|
|
260
|
+
~/.local/share/opencode-voice/chatterbox-venv/bin/pip install chatterbox-tts
|
|
261
|
+
# chatterbox's `perth` dependency needs pkg_resources (removed in setuptools ≥ 81):
|
|
262
|
+
~/.local/share/opencode-voice/chatterbox-venv/bin/pip install 'setuptools<81'
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
First run downloads ~7GB of weights from HuggingFace, then the sidecar takes
|
|
266
|
+
~20s to load on Apple Silicon (MPS). No venv, no weights, no wavs are ever
|
|
267
|
+
committed.
|
|
268
|
+
|
|
269
|
+
Options in `tui.json` (all under the plugin entry, next to `endpoint`):
|
|
270
|
+
|
|
271
|
+
```json
|
|
272
|
+
{
|
|
273
|
+
"ttsEngine": "chatterbox",
|
|
274
|
+
"ttsChatterboxVariant": "multilingual",
|
|
275
|
+
"ttsChatterboxVoiceRef": "/absolute/path/to/voice-5-20s.wav"
|
|
276
|
+
}
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
- `ttsEngine`: `"piper"` (default) | `"chatterbox"`. Unknown values warn and
|
|
280
|
+
use Piper.
|
|
281
|
+
- `ttsChatterboxVariant`: `"multilingual"` (default) | `"turbo"` | `"nano"`.
|
|
282
|
+
- `ttsChatterboxVoiceRef`: optional absolute path to a 5-20s reference wav
|
|
283
|
+
for zero-shot voice cloning. Unset means the model default voice; a missing
|
|
284
|
+
file warns and proceeds voiceless.
|
|
285
|
+
- `ttsChatterboxPython`: optional python binary for the sidecar. Defaults to
|
|
286
|
+
`python3` on `PATH` - set it to the venv python above.
|
|
287
|
+
|
|
288
|
+
Per-utterance Piper fallback: whenever Chatterbox cannot speak an utterance,
|
|
289
|
+
that utterance goes to Piper with a warn log (one toast per outage, not per
|
|
290
|
+
utterance; the next utterance retries Chatterbox). Cases: sidecar not
|
|
291
|
+
installed/crashed, ~15s synthesis bound exceeded, and language routing below.
|
|
292
|
+
|
|
293
|
+
Honest limits, measured on Apple Silicon (MPS) with Multilingual V3:
|
|
294
|
+
|
|
295
|
+
- Chatterbox does **not** speak Vietnamese. Its multilingual model covers 23
|
|
296
|
+
languages (`ar da de el en es fi fr he hi it ja ko ms nl no pl pt ru sv sw
|
|
297
|
+
th tr zh`) - `vi` is not one, so **every** Vietnamese utterance falls back
|
|
298
|
+
to Piper regardless of variant. Nano/Turbo are English-only by design.
|
|
299
|
+
- Latency for "Deploying now.": Piper 0.77s wall for 0.80s of audio (RTF
|
|
300
|
+
~1.0); Chatterbox ~16s wall for ~1s of audio on first synthesis after load
|
|
301
|
+
(RTF ~15, MPS warmup included), ~5-8s wall (RTF ~1.5-3) once warm. It is a
|
|
302
|
+
quality upgrade, not a speed one.
|
|
303
|
+
- Outputs carry Chatterbox's inaudible PerTh watermark.
|
|
304
|
+
|
|
246
305
|
### LLM endpoint
|
|
247
306
|
|
|
248
307
|
An OpenAI-compatible LLM endpoint is required for text normalization. For
|
|
@@ -419,36 +478,47 @@ then `s`.
|
|
|
419
478
|
|
|
420
479
|
### Voice conversation
|
|
421
480
|
|
|
422
|
-
| Command | Keybind | Description
|
|
423
|
-
| -------------------------- | ---------- |
|
|
424
|
-
| `/voice-conversation` | `leader+v` | Toggle
|
|
425
|
-
| `/voice-conversation-stop` | | Exit voice conversation mode
|
|
481
|
+
| Command | Keybind | Description |
|
|
482
|
+
| -------------------------- | ---------- | ------------------------------------------- |
|
|
483
|
+
| `/voice-conversation` | `leader+v` | Toggle push-to-talk voice conversation mode |
|
|
484
|
+
| `/voice-conversation-stop` | | Exit voice conversation mode |
|
|
426
485
|
|
|
427
|
-
One key drives the whole loop
|
|
486
|
+
One key drives the whole loop. It always means the same thing - "I want the
|
|
487
|
+
floor" - and what it does follows the toast on screen:
|
|
428
488
|
|
|
429
489
|
```
|
|
430
|
-
record -> transcribe -> normalize -> submit ->
|
|
490
|
+
press -> record -> press -> transcribe -> normalize -> submit -> speak -> press ...
|
|
431
491
|
```
|
|
432
492
|
|
|
433
|
-
| Toast shows
|
|
434
|
-
|
|
|
435
|
-
|
|
|
436
|
-
|
|
|
437
|
-
|
|
|
438
|
-
|
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
493
|
+
| Toast shows | Pressing the key does |
|
|
494
|
+
| ------------------------ | ----------------------------------------- |
|
|
495
|
+
| Press `leader+v` to talk | Open the mic (this is the resting state) |
|
|
496
|
+
| ● Recording | Finish the turn and submit |
|
|
497
|
+
| Working... / Speaking | Barge in: cut audio, drop reply, open mic |
|
|
498
|
+
| Answer on screen | Same barge-in, while the agent resumes |
|
|
499
|
+
| Transcribing... | Report busy - too short to interrupt |
|
|
500
|
+
|
|
501
|
+
The loop never reopens the mic by itself: after a reply is spoken the mode rests
|
|
502
|
+
paused, so it cannot record the room while nobody is talking. Barge-in drops
|
|
503
|
+
the pending reply on purpose - pressing means talking now, not hearing the
|
|
504
|
+
rest.
|
|
505
|
+
|
|
506
|
+
Answering a permission or a question is the one thing the key cannot do,
|
|
507
|
+
because that answer is given on screen. The mode announces the gate, keeps the
|
|
508
|
+
mic shut, then speaks the agent's answer once the agent resumes.
|
|
509
|
+
|
|
510
|
+
Exit paths: a stop phrase (`stop`, `stop stop`, `dừng lại đi`, ...),
|
|
511
|
+
`/voice-conversation-stop`, or the global `/voice-cancel` (`leader+.`). Only a
|
|
512
|
+
bare stop phrase ends the mode - full sentences mentioning stop submit
|
|
513
|
+
normally. Empty or failed turns pause instead of re-recording, so the key never
|
|
514
|
+
surprises. `/tts-stop` pauses a speaking reply and rests paused. While the mode
|
|
515
|
+
is on, the plain `/stt-record` keys act as the conversation key and auto TTS
|
|
516
|
+
stays silent (the loop speaks the reply itself).
|
|
447
517
|
|
|
448
518
|
While waiting, assistant text is spoken sentence by sentence as it streams in
|
|
449
519
|
(local cleanup, no LLM), so the answer starts before the agent turn finishes.
|
|
450
520
|
Code-like sentences are skipped; replies with nothing streamable fall back to
|
|
451
|
-
the full narrated speak.
|
|
521
|
+
the full narrated speak.
|
|
452
522
|
|
|
453
523
|
- `ttsNormalizeMode` _(optional)_ - `"llm"` (default, polished narration) or
|
|
454
524
|
`"local"` (instant deterministic cleanup, no network). Streaming speech
|
|
@@ -667,6 +737,10 @@ live-mic session was measured.
|
|
|
667
737
|
code-heavy responses, or briefly notify for confirmations
|
|
668
738
|
3. Piper synthesizes speech locally, piped through sox for playback
|
|
669
739
|
|
|
740
|
+
The LLM is only a polish layer: when the configured model is unavailable or out
|
|
741
|
+
of quota, TTS speaks the local cleanup automatically instead of going silent.
|
|
742
|
+
Set `ttsNormalizeMode: "local"` to skip narration entirely.
|
|
743
|
+
|
|
670
744
|
### Auto TTS
|
|
671
745
|
|
|
672
746
|
When enabled (`/tts-mode`), the plugin automatically speaks:
|
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
// Managed Chatterbox TTS sidecar (see vendor/chatterbox_server.py).
|
|
2
|
+
//
|
|
3
|
+
// Why a sidecar instead of a one-shot CLI: Chatterbox loads a large torch
|
|
4
|
+
// model. Loading it per utterance would never keep up with a spoken
|
|
5
|
+
// conversation, so - exactly like lib/whisper-server.js for whisper.cpp - one
|
|
6
|
+
// child process loads the model once and serves synthesis over HTTP while we
|
|
7
|
+
// own its lifetime.
|
|
8
|
+
//
|
|
9
|
+
// Follows the whisper-server conventions deliberately:
|
|
10
|
+
// - Port is probed BEFORE spawning; if anything already answers we never claim
|
|
11
|
+
// it and never kill the foreign process. Default-port callers auto-advance
|
|
12
|
+
// upward (bounded) so a second TUI lands on its own port with its own model
|
|
13
|
+
// load instead of being stuck on Piper forever.
|
|
14
|
+
// - Readiness is GET /health: 200 = ready, 503 = still loading or failed. The
|
|
15
|
+
// sidecar answers 503 with the real reason ("chatterbox-tts not installed"),
|
|
16
|
+
// so we can fail fast with an actionable message instead of waiting out the
|
|
17
|
+
// whole timeout on an install that will never succeed.
|
|
18
|
+
// - stop() only ever signals the handle this client spawned.
|
|
19
|
+
// - Every failure is a typed {code, message}, never a silent return of nothing:
|
|
20
|
+
// the caller turns it into a per-utterance Piper fallback.
|
|
21
|
+
//
|
|
22
|
+
// This client is an OPTIMIZATION. Piper is the default engine and stays the
|
|
23
|
+
// zero-setup fallback, so every failure path here ends in Piper, never silence.
|
|
24
|
+
|
|
25
|
+
import fs from "node:fs";
|
|
26
|
+
import os from "node:os";
|
|
27
|
+
import { fileURLToPath } from "node:url";
|
|
28
|
+
import { spawn } from "node:child_process";
|
|
29
|
+
|
|
30
|
+
const DEFAULT_HOST = "127.0.0.1";
|
|
31
|
+
const DEFAULT_PORT = 8120;
|
|
32
|
+
const DEFAULT_PORT_SCAN_MAX = 4;
|
|
33
|
+
// Model load on first run also downloads weights from HuggingFace, so the
|
|
34
|
+
// readiness budget is generous. It is a bound, not a promise: a failed load
|
|
35
|
+
// reports immediately via the 503 body instead of waiting this out.
|
|
36
|
+
const READY_TIMEOUT_MS = 300000;
|
|
37
|
+
const READY_POLL_MS = 500;
|
|
38
|
+
// One utterance must never stall speech indefinitely.
|
|
39
|
+
const SYNTHESIZE_TIMEOUT_MS = 15000;
|
|
40
|
+
|
|
41
|
+
const SIDECAR_PATH = fileURLToPath(new URL("../vendor/chatterbox_server.py", import.meta.url));
|
|
42
|
+
|
|
43
|
+
export const CHATTERBOX_VARIANTS = ["multilingual", "turbo", "nano"];
|
|
44
|
+
export const DEFAULT_CHATTERBOX_VARIANT = "multilingual";
|
|
45
|
+
|
|
46
|
+
// Per-variant language support, read off the real package
|
|
47
|
+
// (chatterbox-tts 0.1.7) rather than guessed.
|
|
48
|
+
//
|
|
49
|
+
// MEASURED, and worth stating plainly: the multilingual model speaks 23
|
|
50
|
+
// languages and Vietnamese is NOT one of them. Its generate() raises
|
|
51
|
+
// ValueError on language_id "vi". Nano/Turbo are English-only by design. So
|
|
52
|
+
// every variant routes a vi utterance to Piper, and Chatterbox covers English.
|
|
53
|
+
// Do not "fix" this by adding "vi" here - it would fail at synthesis time.
|
|
54
|
+
const CHATTERBOX_LANGUAGES = {
|
|
55
|
+
multilingual: new Set([
|
|
56
|
+
"ar",
|
|
57
|
+
"da",
|
|
58
|
+
"de",
|
|
59
|
+
"el",
|
|
60
|
+
"en",
|
|
61
|
+
"es",
|
|
62
|
+
"fi",
|
|
63
|
+
"fr",
|
|
64
|
+
"he",
|
|
65
|
+
"hi",
|
|
66
|
+
"it",
|
|
67
|
+
"ja",
|
|
68
|
+
"ko",
|
|
69
|
+
"ms",
|
|
70
|
+
"nl",
|
|
71
|
+
"no",
|
|
72
|
+
"pl",
|
|
73
|
+
"pt",
|
|
74
|
+
"ru",
|
|
75
|
+
"sv",
|
|
76
|
+
"sw",
|
|
77
|
+
"th",
|
|
78
|
+
"tr",
|
|
79
|
+
"zh",
|
|
80
|
+
]),
|
|
81
|
+
turbo: new Set(["en"]),
|
|
82
|
+
nano: new Set(["en"]),
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
export function isChatterboxVariant(value) {
|
|
86
|
+
return CHATTERBOX_VARIANTS.includes(value);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Can this variant speak this language? Unknown variants answer false, which
|
|
90
|
+
// routes to Piper - the safe default, since Piper always has an answer.
|
|
91
|
+
export function chatterboxSupportsLang(variant, lang) {
|
|
92
|
+
const set = CHATTERBOX_LANGUAGES[variant];
|
|
93
|
+
return !!set && set.has(lang);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// The one routing decision, pure and exported so it can be tested without a
|
|
97
|
+
// sidecar, a model, or an audio device. Returns the engine to use plus the
|
|
98
|
+
// reason, which the caller logs (and which keeps per-utterance skips quiet in
|
|
99
|
+
// the UI - no toast spam for a language Piper can already read).
|
|
100
|
+
export function resolveTtsEngine(opts, lang, warn) {
|
|
101
|
+
const configured = opts?.ttsEngine;
|
|
102
|
+
if (!configured || configured === "piper") return { engine: "piper", reason: "default" };
|
|
103
|
+
if (configured !== "chatterbox") {
|
|
104
|
+
warn?.(`Unknown ttsEngine "${configured}" - expected "piper" or "chatterbox"; using Piper`);
|
|
105
|
+
return { engine: "piper", reason: "unknown-engine" };
|
|
106
|
+
}
|
|
107
|
+
const requested = opts?.ttsChatterboxVariant;
|
|
108
|
+
if (requested && !isChatterboxVariant(requested)) {
|
|
109
|
+
warn?.(
|
|
110
|
+
`Unknown ttsChatterboxVariant "${requested}" - expected ${CHATTERBOX_VARIANTS.join(" | ")}; using "${DEFAULT_CHATTERBOX_VARIANT}"`,
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
const variant = isChatterboxVariant(requested) ? requested : DEFAULT_CHATTERBOX_VARIANT;
|
|
114
|
+
if (chatterboxSupportsLang(variant, lang)) return { engine: "chatterbox", variant, reason: "ok" };
|
|
115
|
+
warn?.(`Chatterbox variant "${variant}" cannot speak "${lang}" - using Piper for this utterance`);
|
|
116
|
+
return { engine: "piper", reason: "unsupported-language", variant };
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
export function buildChatterboxArgs({ scriptPath, host, port, variant }) {
|
|
120
|
+
return [
|
|
121
|
+
scriptPath || SIDECAR_PATH,
|
|
122
|
+
"--host",
|
|
123
|
+
host || DEFAULT_HOST,
|
|
124
|
+
"--port",
|
|
125
|
+
String(port || DEFAULT_PORT),
|
|
126
|
+
"--variant",
|
|
127
|
+
variant || DEFAULT_CHATTERBOX_VARIANT,
|
|
128
|
+
];
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
export { DEFAULT_HOST, DEFAULT_PORT, DEFAULT_PORT_SCAN_MAX, SIDECAR_PATH };
|
|
132
|
+
|
|
133
|
+
export function createChatterboxClient(options = {}) {
|
|
134
|
+
const {
|
|
135
|
+
host = DEFAULT_HOST,
|
|
136
|
+
python,
|
|
137
|
+
variant = DEFAULT_CHATTERBOX_VARIANT,
|
|
138
|
+
voiceRef,
|
|
139
|
+
exaggeration,
|
|
140
|
+
cfgWeight,
|
|
141
|
+
logger,
|
|
142
|
+
readyTimeoutMs = READY_TIMEOUT_MS,
|
|
143
|
+
readyPollMs = READY_POLL_MS,
|
|
144
|
+
synthesizeTimeoutMs = SYNTHESIZE_TIMEOUT_MS,
|
|
145
|
+
portScanMax = DEFAULT_PORT_SCAN_MAX,
|
|
146
|
+
deps = {},
|
|
147
|
+
} = options;
|
|
148
|
+
const scriptPath = options.scriptPath || SIDECAR_PATH;
|
|
149
|
+
const spawnFn = deps.spawn ?? spawn;
|
|
150
|
+
const fetchFn = deps.fetch ?? fetch;
|
|
151
|
+
|
|
152
|
+
let proc = null;
|
|
153
|
+
let ready = false;
|
|
154
|
+
let ownerPid = null;
|
|
155
|
+
// Where the port scan starts. The plugin never sets it (so production always
|
|
156
|
+
// begins at 8120), but tests bind an ephemeral port and need to reach it.
|
|
157
|
+
const basePort = options.port ?? DEFAULT_PORT;
|
|
158
|
+
let boundPort = basePort;
|
|
159
|
+
let lastError = null;
|
|
160
|
+
let device = "unknown";
|
|
161
|
+
// A missing reference wav must not kill the whole engine - the sidecar still
|
|
162
|
+
// speaks with its default voice, which is better than falling back to Piper.
|
|
163
|
+
let refPath = voiceRef ? String(voiceRef).replace(/^~(?=\/|$)/, os.homedir()) : "";
|
|
164
|
+
if (refPath && !fs.existsSync(refPath)) {
|
|
165
|
+
logger?.log(
|
|
166
|
+
"VOICE",
|
|
167
|
+
`Chatterbox voice reference not found, using default voice: ${refPath}`,
|
|
168
|
+
"warn",
|
|
169
|
+
);
|
|
170
|
+
refPath = "";
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
const baseUrlFor = (p) => `http://${host}:${p}`;
|
|
174
|
+
|
|
175
|
+
async function portResponds(p) {
|
|
176
|
+
const base = baseUrlFor(p);
|
|
177
|
+
for (const url of [`${base}/health`, base]) {
|
|
178
|
+
try {
|
|
179
|
+
const resp = await fetchFn(url, { signal: AbortSignal.timeout(1500) });
|
|
180
|
+
if (resp) return true;
|
|
181
|
+
} catch {
|
|
182
|
+
// Connection refused / timeout: try the next URL.
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return false;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// One probe. "ready" | "loading" | {failed: message} | "down". The failed
|
|
189
|
+
// shape carries the sidecar's own reason so start() can surface it verbatim.
|
|
190
|
+
async function probeOnce() {
|
|
191
|
+
try {
|
|
192
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/health`, {
|
|
193
|
+
signal: AbortSignal.timeout(2000),
|
|
194
|
+
});
|
|
195
|
+
if (!resp) return "down";
|
|
196
|
+
if (resp.status === 503) {
|
|
197
|
+
let body = {};
|
|
198
|
+
try {
|
|
199
|
+
body = await resp.json();
|
|
200
|
+
} catch {
|
|
201
|
+
// Non-JSON 503: treat as still loading rather than guessing.
|
|
202
|
+
}
|
|
203
|
+
if (body?.phase === "failed") return { failed: body?.error || "model load failed" };
|
|
204
|
+
return "loading";
|
|
205
|
+
}
|
|
206
|
+
if (resp?.ok) {
|
|
207
|
+
try {
|
|
208
|
+
const body = await resp.json();
|
|
209
|
+
if (body?.device) device = body.device;
|
|
210
|
+
} catch {
|
|
211
|
+
// 200 with an unparseable body still means the port answered /health.
|
|
212
|
+
}
|
|
213
|
+
return "ready";
|
|
214
|
+
}
|
|
215
|
+
return "down";
|
|
216
|
+
} catch {
|
|
217
|
+
return "down";
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
async function pollReady(deadline) {
|
|
222
|
+
while (Date.now() < deadline) {
|
|
223
|
+
if (!proc) return { failed: "chatterbox sidecar exited before becoming ready" };
|
|
224
|
+
const state = await probeOnce();
|
|
225
|
+
if (state === "ready") return state;
|
|
226
|
+
if (state?.failed) return state;
|
|
227
|
+
await new Promise((r) => setTimeout(r, readyPollMs));
|
|
228
|
+
}
|
|
229
|
+
return "timeout";
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// Attempt one port: probe, spawn, wait for readiness. stop() inside only ever
|
|
233
|
+
// signals the handle spawned here.
|
|
234
|
+
async function startOnPort(port) {
|
|
235
|
+
boundPort = port;
|
|
236
|
+
if (await portResponds(port)) {
|
|
237
|
+
return {
|
|
238
|
+
ok: false,
|
|
239
|
+
code: "PORT_IN_USE",
|
|
240
|
+
advancable: true,
|
|
241
|
+
message: `Port ${port} already answers - not starting a second chatterbox sidecar`,
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
const args = buildChatterboxArgs({ scriptPath, host, port, variant });
|
|
245
|
+
const bin = python || "python3";
|
|
246
|
+
try {
|
|
247
|
+
proc = spawnFn(bin, args, { stdio: ["ignore", "ignore", "pipe"] });
|
|
248
|
+
} catch (err) {
|
|
249
|
+
proc = null;
|
|
250
|
+
return {
|
|
251
|
+
ok: false,
|
|
252
|
+
code: "SPAWN_FAILED",
|
|
253
|
+
advancable: false,
|
|
254
|
+
message: `Failed to spawn ${bin}: ${err.message}`,
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
let stderr = "";
|
|
259
|
+
const owned = proc;
|
|
260
|
+
proc.stderr?.on("data", (chunk) => {
|
|
261
|
+
stderr += chunk.toString();
|
|
262
|
+
});
|
|
263
|
+
proc.on("exit", (code) => {
|
|
264
|
+
// A late event from a previous child must not wipe the current handle.
|
|
265
|
+
if (proc !== owned) return;
|
|
266
|
+
// Record why, so a sidecar that dies mid-session explains the Piper
|
|
267
|
+
// fallback instead of reporting a bare "not ready". A deliberate stop()
|
|
268
|
+
// already nulled proc, so it never reaches here.
|
|
269
|
+
lastError = { code: "EXITED", message: `chatterbox sidecar exited (code=${code})` };
|
|
270
|
+
proc = null;
|
|
271
|
+
ready = false;
|
|
272
|
+
ownerPid = null;
|
|
273
|
+
});
|
|
274
|
+
proc.on("error", (err) => {
|
|
275
|
+
if (proc !== owned) return;
|
|
276
|
+
logger?.log("VOICE", `chatterbox sidecar process error: ${err.message}`, "warn");
|
|
277
|
+
proc = null;
|
|
278
|
+
ready = false;
|
|
279
|
+
ownerPid = null;
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
const outcome = await pollReady(Date.now() + readyTimeoutMs);
|
|
283
|
+
if (outcome !== "ready") {
|
|
284
|
+
const tail = stderr.trim().slice(-300);
|
|
285
|
+
const message =
|
|
286
|
+
outcome === "timeout"
|
|
287
|
+
? `chatterbox sidecar did not become ready in time (${readyTimeoutMs}ms) stderr=${tail}`
|
|
288
|
+
: `chatterbox sidecar unavailable: ${outcome.failed ?? "unknown"}`;
|
|
289
|
+
const code = outcome === "timeout" ? "START_TIMEOUT" : "MODEL_LOAD_FAILED";
|
|
290
|
+
stop();
|
|
291
|
+
return { ok: false, code, advancable: false, message };
|
|
292
|
+
}
|
|
293
|
+
ownerPid = proc?.pid ?? null;
|
|
294
|
+
return { ok: true };
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
async function start() {
|
|
298
|
+
if (proc && ready) return true; // reuse the owned, already-ready sidecar
|
|
299
|
+
const scanMax = Math.max(1, Math.floor(portScanMax) || 1);
|
|
300
|
+
for (let i = 0; i < scanMax; i += 1) {
|
|
301
|
+
const port = basePort + i;
|
|
302
|
+
const r = await startOnPort(port);
|
|
303
|
+
if (r.ok) {
|
|
304
|
+
ready = true;
|
|
305
|
+
lastError = null;
|
|
306
|
+
logger?.log(
|
|
307
|
+
"VOICE",
|
|
308
|
+
`chatterbox sidecar ready at ${baseUrlFor(boundPort)} variant=${variant} device=${device} pid=${ownerPid}`,
|
|
309
|
+
"debug",
|
|
310
|
+
);
|
|
311
|
+
return true;
|
|
312
|
+
}
|
|
313
|
+
if (!r.advancable) {
|
|
314
|
+
lastError = { code: r.code, message: r.message };
|
|
315
|
+
logger?.log("VOICE", r.message, "warn");
|
|
316
|
+
return false;
|
|
317
|
+
}
|
|
318
|
+
logger?.log("VOICE", r.message, "warn");
|
|
319
|
+
}
|
|
320
|
+
lastError = {
|
|
321
|
+
code: "PORT_RANGE_EXHAUSTED",
|
|
322
|
+
message: `No free port in ${basePort}..${basePort + scanMax - 1} - all answer, using Piper`,
|
|
323
|
+
};
|
|
324
|
+
logger?.log("VOICE", lastError.message, "warn");
|
|
325
|
+
return false;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
// One-shot health read, for callers that want to report the engine's state
|
|
329
|
+
// (and for tests). Returns {ok, variant, device, phase, error?}.
|
|
330
|
+
async function health() {
|
|
331
|
+
try {
|
|
332
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/health`, {
|
|
333
|
+
signal: AbortSignal.timeout(2000),
|
|
334
|
+
});
|
|
335
|
+
const body = await resp.json();
|
|
336
|
+
return body;
|
|
337
|
+
} catch (err) {
|
|
338
|
+
return { ok: false, error: err.message };
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
// Synthesize one utterance. Returns {ok:true, wav:Buffer} or
|
|
343
|
+
// {ok:false, code, message} - never throws, so the caller always has a Piper
|
|
344
|
+
// fallback available. options.signal lets a cancel abort the in-flight request
|
|
345
|
+
// (the sidecar finishes its own generation and discards the result; the
|
|
346
|
+
// caller's generation guard is what actually prevents stale playback).
|
|
347
|
+
async function synthesize(text, languageId, opts = {}) {
|
|
348
|
+
if (!ready) {
|
|
349
|
+
return {
|
|
350
|
+
ok: false,
|
|
351
|
+
code: lastError?.code || "NOT_READY",
|
|
352
|
+
message: lastError?.message || "chatterbox sidecar not ready",
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
const body = {
|
|
356
|
+
text,
|
|
357
|
+
language_id: languageId || undefined,
|
|
358
|
+
exaggeration: opts.exaggeration ?? exaggeration ?? undefined,
|
|
359
|
+
cfg_weight: opts.cfgWeight ?? cfgWeight ?? undefined,
|
|
360
|
+
voice_ref: refPath || undefined,
|
|
361
|
+
};
|
|
362
|
+
try {
|
|
363
|
+
// One utterance must never stall speech indefinitely: bound the request,
|
|
364
|
+
// and let an outer cancel abort it too. On timeout we fall back to
|
|
365
|
+
// Piper for THIS utterance only - the sidecar stays up.
|
|
366
|
+
const timeout = AbortSignal.timeout(synthesizeTimeoutMs);
|
|
367
|
+
const signal = opts.signal ? AbortSignal.any([timeout, opts.signal]) : timeout;
|
|
368
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/speak`, {
|
|
369
|
+
method: "POST",
|
|
370
|
+
headers: { "Content-Type": "application/json" },
|
|
371
|
+
body: JSON.stringify(body),
|
|
372
|
+
signal,
|
|
373
|
+
});
|
|
374
|
+
if (!resp?.ok) {
|
|
375
|
+
let message = `chatterbox sidecar responded ${resp?.status}`;
|
|
376
|
+
try {
|
|
377
|
+
const data = await resp.json();
|
|
378
|
+
if (data?.error) message = data.error;
|
|
379
|
+
} catch {
|
|
380
|
+
// Keep the status-based message.
|
|
381
|
+
}
|
|
382
|
+
return { ok: false, code: "BAD_STATUS", message };
|
|
383
|
+
}
|
|
384
|
+
const audio = Buffer.from(await resp.arrayBuffer());
|
|
385
|
+
if (audio.length === 0) {
|
|
386
|
+
return { ok: false, code: "EMPTY_AUDIO", message: "chatterbox returned no audio" };
|
|
387
|
+
}
|
|
388
|
+
return { ok: true, wav: audio };
|
|
389
|
+
} catch (err) {
|
|
390
|
+
const timedOut = err?.name === "TimeoutError" || err?.name === "AbortError";
|
|
391
|
+
return {
|
|
392
|
+
ok: false,
|
|
393
|
+
code: timedOut ? "SYNTH_TIMEOUT" : "REQUEST_FAILED",
|
|
394
|
+
message: `chatterbox synthesis failed: ${err.message}`,
|
|
395
|
+
};
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
// Stop ONLY the process this client spawned (by handle). A foreign process
|
|
400
|
+
// on the same port is never signaled.
|
|
401
|
+
function stop() {
|
|
402
|
+
ready = false;
|
|
403
|
+
ownerPid = null;
|
|
404
|
+
if (proc) {
|
|
405
|
+
const owned = proc;
|
|
406
|
+
proc = null;
|
|
407
|
+
try {
|
|
408
|
+
owned.kill("SIGTERM");
|
|
409
|
+
} catch {
|
|
410
|
+
// Already gone.
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
function isRunning() {
|
|
416
|
+
return ready && proc !== null;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
function getLastError() {
|
|
420
|
+
return lastError;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
function getPort() {
|
|
424
|
+
return boundPort;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
return { start, stop, health, synthesize, isRunning, getLastError, getPort };
|
|
428
|
+
}
|
package/lib/conversation.js
CHANGED
|
@@ -1,18 +1,27 @@
|
|
|
1
|
-
// Voice conversation mode:
|
|
1
|
+
// Voice conversation mode: push-to-talk turns with barge-in.
|
|
2
2
|
//
|
|
3
|
-
// One key drives the whole loop
|
|
3
|
+
// One key drives the whole loop and its meaning is always the same: the user
|
|
4
|
+
// wants the floor. What the key actually does depends on the state shown in
|
|
4
5
|
// the toast:
|
|
5
6
|
//
|
|
6
|
-
//
|
|
7
|
+
// press -> record -> press -> transcribe -> normalize -> submit
|
|
8
|
+
// -> wait reply -> speak -> press -> record ...
|
|
7
9
|
//
|
|
10
|
+
// - paused + key: open the mic (the resting state is the user's turn)
|
|
8
11
|
// - recording + key: finish the turn and submit
|
|
9
|
-
// - speaking + key:
|
|
10
|
-
//
|
|
11
|
-
// -
|
|
12
|
+
// - speaking/waiting + key: barge in - cut the audio, drop the pending reply,
|
|
13
|
+
// open the mic (one press, never a stop-then-record dance)
|
|
14
|
+
// - processing + key: transcription is too short to interrupt, report busy
|
|
12
15
|
//
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
+
// The loop never reopens the mic by itself: after a reply is spoken the mode
|
|
17
|
+
// rests paused, so it can never record the room while nobody is talking. That
|
|
18
|
+
// auto-record is also why a question/permission gate used to eat the answer -
|
|
19
|
+
// the mic opened, the user's muttering was submitted, and the real reply was
|
|
20
|
+
// already torn down. Gates now announce and wait instead.
|
|
21
|
+
//
|
|
22
|
+
// Exit paths: a stop phrase ("stop", "dừng lại", ...), /voice-conversation-stop,
|
|
23
|
+
// or the global /voice-cancel. Empty or failed turns pause instead of
|
|
24
|
+
// auto-recording, so the key never surprises.
|
|
16
25
|
//
|
|
17
26
|
// Latency: while waiting, assistant text deltas are spoken sentence by
|
|
18
27
|
// sentence (local cleanup, no LLM), so the answer starts before the turn
|
|
@@ -151,7 +160,34 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
151
160
|
phase = "idle";
|
|
152
161
|
stop("busy");
|
|
153
162
|
toast("STT busy, conversation off", "warning");
|
|
163
|
+
return;
|
|
154
164
|
}
|
|
165
|
+
toast(`Listening - press ${keyLabel()} to send`);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// The resting state: mic closed, one press to talk. Entered by start() and
|
|
169
|
+
// after a reply is spoken, never by the loop auto-recording.
|
|
170
|
+
|
|
171
|
+
function enterPaused() {
|
|
172
|
+
phase = "paused";
|
|
173
|
+
toast(`Press ${keyLabel()} to talk`);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// One press takes the floor: cut whatever audio is queued or playing, drop
|
|
177
|
+
// the pending reply, and open the mic. The mode stays on - a press always
|
|
178
|
+
// means "I want to talk now", so it must never exit silently.
|
|
179
|
+
|
|
180
|
+
function bargeIn() {
|
|
181
|
+
logger?.log("VOICE", "Conversation barge-in", "debug");
|
|
182
|
+
discardStreamAudio();
|
|
183
|
+
clearProcessingToast();
|
|
184
|
+
// Settle the pending reply wait as "stopped" so the abandoned answer is
|
|
185
|
+
// never flushed into the mic we are about to open.
|
|
186
|
+
if (waitCancel) {
|
|
187
|
+
waitCancel("stopped");
|
|
188
|
+
waitCancel = null;
|
|
189
|
+
}
|
|
190
|
+
beginTurn();
|
|
155
191
|
}
|
|
156
192
|
|
|
157
193
|
function endReplyStream() {
|
|
@@ -221,7 +257,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
221
257
|
const { sentences, rest } = splitSpokenSentences(streamBuffer);
|
|
222
258
|
const ready = final && rest.trim() ? [...sentences, rest] : sentences;
|
|
223
259
|
if (ready.length > 0 && streamSpokenChars === 0) {
|
|
224
|
-
showProcessingToast(
|
|
260
|
+
showProcessingToast(`Speaking - press ${keyLabel()} to interrupt`);
|
|
225
261
|
}
|
|
226
262
|
for (const sentence of ready) {
|
|
227
263
|
if (!isSpeakableSentence(sentence)) continue;
|
|
@@ -234,9 +270,11 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
234
270
|
}
|
|
235
271
|
|
|
236
272
|
// Wait for the submitted turn to finish: idle, a permission/question gate,
|
|
237
|
-
// or timeout. Resolves early when stop() cancels the wait.
|
|
273
|
+
// or timeout. Resolves early when stop() cancels the wait. gates:false drops
|
|
274
|
+
// the permission/question subscriptions, so a second wait survives the gate
|
|
275
|
+
// the user is answering on screen and only ends on idle/timeout/stop.
|
|
238
276
|
|
|
239
|
-
function waitForReply(sessionID) {
|
|
277
|
+
function waitForReply(sessionID, { gates = true } = {}) {
|
|
240
278
|
let settled = false;
|
|
241
279
|
let unsubs = [];
|
|
242
280
|
let timer = null;
|
|
@@ -276,12 +314,16 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
276
314
|
finish("idle");
|
|
277
315
|
}
|
|
278
316
|
}),
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
317
|
+
...(gates
|
|
318
|
+
? [
|
|
319
|
+
api.event.on("permission.asked", (event) => {
|
|
320
|
+
if (forSession(event.properties)) finish("permission");
|
|
321
|
+
}),
|
|
322
|
+
api.event.on("question.asked", (event) => {
|
|
323
|
+
if (forSession(event.properties)) finish("question");
|
|
324
|
+
}),
|
|
325
|
+
]
|
|
326
|
+
: []),
|
|
285
327
|
];
|
|
286
328
|
timer = setTimeout(() => finish("timeout"), replyTimeoutMs);
|
|
287
329
|
});
|
|
@@ -293,6 +335,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
293
335
|
if (!active || phase !== "recording") return;
|
|
294
336
|
const myGen = generation;
|
|
295
337
|
phase = "processing";
|
|
338
|
+
showProcessingToast("Transcribing...");
|
|
296
339
|
|
|
297
340
|
const res = await stt.transcribeTurn();
|
|
298
341
|
if (!active || myGen !== generation) return;
|
|
@@ -304,12 +347,12 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
304
347
|
stop("error");
|
|
305
348
|
return;
|
|
306
349
|
}
|
|
307
|
-
pauseForRetry(
|
|
350
|
+
pauseForRetry(`Paused - press ${keyLabel()} to retry`);
|
|
308
351
|
return;
|
|
309
352
|
}
|
|
310
353
|
if (!res.text) {
|
|
311
354
|
consecErrors = 0;
|
|
312
|
-
pauseForRetry(
|
|
355
|
+
pauseForRetry(`No speech heard - press ${keyLabel()} to try again`);
|
|
313
356
|
return;
|
|
314
357
|
}
|
|
315
358
|
consecErrors = 0;
|
|
@@ -331,7 +374,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
331
374
|
phase = "waiting";
|
|
332
375
|
const sessionID = currentSessionID();
|
|
333
376
|
const turnStart = Date.now();
|
|
334
|
-
showProcessingToast(
|
|
377
|
+
showProcessingToast(`Working - press ${keyLabel()} to talk`);
|
|
335
378
|
startReplyStream(sessionID, turnStart);
|
|
336
379
|
const tWait = Date.now();
|
|
337
380
|
const outcome = await waitForReply(sessionID);
|
|
@@ -351,46 +394,78 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
351
394
|
return;
|
|
352
395
|
}
|
|
353
396
|
if (outcome === "permission" || outcome === "question") {
|
|
354
|
-
|
|
355
|
-
phase = "speaking";
|
|
356
|
-
showProcessingToast("Speaking...");
|
|
357
|
-
await tts.speak(
|
|
358
|
-
outcome === "permission"
|
|
359
|
-
? "Permission requested. Please check your screen."
|
|
360
|
-
: "A question needs your answer. Please check your screen.",
|
|
361
|
-
);
|
|
362
|
-
clearProcessingToast();
|
|
363
|
-
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
364
|
-
beginTurn();
|
|
397
|
+
await answerGateThenSpeak(myGen, sessionID, turnStart, outcome);
|
|
365
398
|
return;
|
|
366
399
|
}
|
|
367
400
|
if (outcome === "stopped") return;
|
|
368
401
|
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
402
|
+
await speakReply(myGen);
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// A permission/question gate is answered ON SCREEN, so the mic must stay
|
|
406
|
+
// shut. Announce the gate, then re-arm the reply stream and wait for the
|
|
407
|
+
// agent to resume so its post-answer reply is spoken aloud - before this,
|
|
408
|
+
// endReplyStream() had already run and the answer was dropped in silence.
|
|
372
409
|
|
|
410
|
+
async function answerGateThenSpeak(myGen, sessionID, turnStart, outcome) {
|
|
411
|
+
discardStreamAudio();
|
|
412
|
+
phase = "speaking";
|
|
413
|
+
showProcessingToast("Speaking...");
|
|
414
|
+
await tts.speak(
|
|
415
|
+
outcome === "permission"
|
|
416
|
+
? "Permission requested. Please check your screen."
|
|
417
|
+
: "A question needs your answer. Please check your screen.",
|
|
418
|
+
);
|
|
419
|
+
clearProcessingToast();
|
|
420
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
421
|
+
|
|
422
|
+
// Still "waiting": the user answers on screen, and a press here is an
|
|
423
|
+
// ordinary barge-in (drop the pending reply, open the mic).
|
|
424
|
+
phase = "waiting";
|
|
425
|
+
showProcessingToast(`Answer on screen - press ${keyLabel()} to talk`);
|
|
426
|
+
startReplyStream(sessionID, turnStart);
|
|
427
|
+
const resumed = await waitForReply(sessionID, { gates: false });
|
|
428
|
+
if (!active || myGen !== generation) return;
|
|
429
|
+
if (resumed === "stopped") return;
|
|
430
|
+
endReplyStream();
|
|
431
|
+
if (resumed === "timeout") {
|
|
432
|
+
discardStreamAudio();
|
|
433
|
+
stop("timeout");
|
|
434
|
+
toast("No reply in time, conversation off", "warning");
|
|
435
|
+
return;
|
|
436
|
+
}
|
|
437
|
+
await speakReply(myGen);
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
// Speak whatever this turn produced, then rest paused with the mic OFF.
|
|
441
|
+
// Streamed sentences are already queued, so only the tail needs flushing;
|
|
442
|
+
// the full LLM-narrated speak is the fallback when nothing streamable
|
|
443
|
+
// arrived. Returns early on any generation/phase change (barge-in, stop).
|
|
444
|
+
|
|
445
|
+
async function speakReply(myGen) {
|
|
446
|
+
phase = "speaking";
|
|
447
|
+
showProcessingToast(`Speaking - press ${keyLabel()} to interrupt`);
|
|
373
448
|
if (streamSpokenChars > 0) {
|
|
374
|
-
phase = "speaking";
|
|
375
449
|
flushStream(true);
|
|
376
450
|
await speakTail;
|
|
377
451
|
clearProcessingToast();
|
|
378
452
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
379
453
|
await delay(restartDelayMs);
|
|
380
454
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
381
|
-
|
|
455
|
+
enterPaused();
|
|
382
456
|
return;
|
|
383
457
|
}
|
|
384
|
-
|
|
385
|
-
phase = "speaking";
|
|
386
458
|
const spoken = await tts.speakAssistantTurn();
|
|
387
459
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
460
|
+
// Drop the sticky speaking status before the paused toast, or it keeps
|
|
461
|
+
// re-announcing "interrupt" once the mic is closed again.
|
|
462
|
+
clearProcessingToast();
|
|
388
463
|
if (!spoken.spoken) {
|
|
389
|
-
toast("No reply to speak
|
|
464
|
+
toast("No reply to speak", "warning");
|
|
390
465
|
}
|
|
391
466
|
await delay(restartDelayMs);
|
|
392
467
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
393
|
-
|
|
468
|
+
enterPaused();
|
|
394
469
|
}
|
|
395
470
|
|
|
396
471
|
// Pause instead of auto-recording so an empty or failed turn never feels
|
|
@@ -402,14 +477,15 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
402
477
|
if (message) toast(message, "warning");
|
|
403
478
|
}
|
|
404
479
|
|
|
405
|
-
// Stop speech and hold the mode. The next keypress records
|
|
480
|
+
// Stop speech and hold the mode, mic still closed. The next keypress records
|
|
481
|
+
// again. Only the TTS stop key lands here; the conversation key barges in.
|
|
406
482
|
|
|
407
483
|
function pauseSpeech() {
|
|
408
484
|
if (!active || phase !== "speaking") return;
|
|
409
485
|
logger?.log("VOICE", "Conversation speech paused", "debug");
|
|
410
486
|
tts.stop();
|
|
411
487
|
phase = "paused";
|
|
412
|
-
toast(
|
|
488
|
+
toast(`Speech paused - press ${keyLabel()} to talk`);
|
|
413
489
|
}
|
|
414
490
|
|
|
415
491
|
// Called by the TTS stop command while the mode is on. Returns true when the
|
|
@@ -431,7 +507,10 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
431
507
|
return false;
|
|
432
508
|
}
|
|
433
509
|
|
|
434
|
-
|
|
510
|
+
// One press, one meaning: "I want the floor". The source key (record,
|
|
511
|
+
// submit, toggle) no longer changes the outcome, and a press never exits.
|
|
512
|
+
|
|
513
|
+
function onKey() {
|
|
435
514
|
if (!active) {
|
|
436
515
|
start();
|
|
437
516
|
return;
|
|
@@ -440,21 +519,16 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
440
519
|
finishTurnFlow();
|
|
441
520
|
return;
|
|
442
521
|
}
|
|
443
|
-
if (phase === "speaking") {
|
|
444
|
-
|
|
522
|
+
if (phase === "speaking" || phase === "waiting") {
|
|
523
|
+
bargeIn();
|
|
445
524
|
return;
|
|
446
525
|
}
|
|
447
526
|
if (phase === "paused") {
|
|
448
527
|
beginTurn();
|
|
449
528
|
return;
|
|
450
529
|
}
|
|
451
|
-
// processing |
|
|
452
|
-
|
|
453
|
-
stop("cancelled");
|
|
454
|
-
toast("Conversation off");
|
|
455
|
-
} else {
|
|
456
|
-
toast("Conversation busy, please wait...", "warning");
|
|
457
|
-
}
|
|
530
|
+
// processing | idle: the transcription in flight is too short to interrupt.
|
|
531
|
+
toast("Conversation busy, please wait...", "warning");
|
|
458
532
|
}
|
|
459
533
|
|
|
460
534
|
function start() {
|
|
@@ -471,12 +545,12 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
471
545
|
generation += 1;
|
|
472
546
|
turn = 0;
|
|
473
547
|
consecErrors = 0;
|
|
474
|
-
phase = "
|
|
548
|
+
phase = "paused";
|
|
475
549
|
tts.setConversationActive(true);
|
|
476
|
-
|
|
550
|
+
// Recording toasts must name the key the user actually has bound.
|
|
551
|
+
stt.setStopHint(kb("voice.conversation"));
|
|
477
552
|
logger?.log("VOICE", "Conversation started", "debug");
|
|
478
|
-
toast(
|
|
479
|
-
beginTurn();
|
|
553
|
+
toast(`Conversation on - press ${keyLabel()} to talk`);
|
|
480
554
|
}
|
|
481
555
|
|
|
482
556
|
api.lifecycle?.onDispose?.(() => stop("dispose"));
|
|
@@ -495,16 +569,24 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
495
569
|
return v;
|
|
496
570
|
}
|
|
497
571
|
|
|
572
|
+
// Every toast names the key that acts, resolved from the user's keybinds
|
|
573
|
+
// instead of the default literal - a rebound key must never be announced as
|
|
574
|
+
// something the user does not have.
|
|
575
|
+
function keyLabel() {
|
|
576
|
+
return kb("voice.conversation") || "the key";
|
|
577
|
+
}
|
|
578
|
+
|
|
498
579
|
const commands = [
|
|
499
580
|
{
|
|
500
581
|
title: "Voice: conversation mode",
|
|
501
582
|
value: "voice.conversation",
|
|
502
583
|
category: "opencode-voice",
|
|
503
|
-
description:
|
|
584
|
+
description:
|
|
585
|
+
"Toggle voice conversation (push-to-talk: press to record, press again to send, press while it speaks to interrupt)",
|
|
504
586
|
...(kb("voice.conversation") ? { keybind: kb("voice.conversation") } : {}),
|
|
505
587
|
slash: { name: "voice-conversation" },
|
|
506
588
|
onSelect() {
|
|
507
|
-
onKey(
|
|
589
|
+
onKey();
|
|
508
590
|
},
|
|
509
591
|
},
|
|
510
592
|
{
|
package/lib/tts.js
CHANGED
|
@@ -6,6 +6,7 @@ import os from "node:os";
|
|
|
6
6
|
import { spawn } from "node:child_process";
|
|
7
7
|
import { getSessionTitle } from "./session.js";
|
|
8
8
|
import { clearProcessingToast, showProcessingToast, updateProcessingToast } from "./stt.js";
|
|
9
|
+
import { createChatterboxClient, resolveTtsEngine } from "./chatterbox-server.js";
|
|
9
10
|
|
|
10
11
|
const VOICES_DIR = path.join(os.homedir(), ".local", "share", "piper-voices");
|
|
11
12
|
|
|
@@ -66,6 +67,44 @@ export function localSpeechCleanup(text) {
|
|
|
66
67
|
return out.replace(/\s+/g, " ").trim();
|
|
67
68
|
}
|
|
68
69
|
|
|
70
|
+
// ---- Speech must never depend on the LLM ----
|
|
71
|
+
// The narrator is a cosmetic polish layer on top of local Piper, which is
|
|
72
|
+
// always available. So when the narrator fails for any reason (quota
|
|
73
|
+
// exhausted, bad auth, unreachable endpoint, missing model, non-2xx) we speak
|
|
74
|
+
// the deterministic local cleanup instead of going silent - this is the one
|
|
75
|
+
// place the decision lives, so every caller benefits.
|
|
76
|
+
//
|
|
77
|
+
// Pure so it is testable without a TUI, a model, or piper.
|
|
78
|
+
|
|
79
|
+
export function resolveSpeechText(rawText, llmResult) {
|
|
80
|
+
if (llmResult?.text) return { text: llmResult.text };
|
|
81
|
+
return { text: localSpeechCleanup(rawText), error: llmResult?.error, fellBack: true };
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// The llm-client retries with backoff (normalizeRetries in llm-client.js), so a
|
|
85
|
+
// model that is merely out of quota stalls for many seconds before admitting
|
|
86
|
+
// failure - waiting that long for a polish layer is worse than unpolished
|
|
87
|
+
// speech, so bound the wait. The timer is unref'd (same trick as
|
|
88
|
+
// wallTimeout in streaming-stt.js) and cleared on the happy path, so tests and
|
|
89
|
+
// short-lived processes never wait out the full bound.
|
|
90
|
+
const NORMALIZE_FALLBACK_TIMEOUT_MS = 12000;
|
|
91
|
+
|
|
92
|
+
async function normalizeWithinBound(promise) {
|
|
93
|
+
let timer = null;
|
|
94
|
+
const bound = new Promise((resolve) => {
|
|
95
|
+
timer = setTimeout(() => {
|
|
96
|
+
timer = null;
|
|
97
|
+
resolve({ text: null, error: "LLM normalization timed out" });
|
|
98
|
+
}, NORMALIZE_FALLBACK_TIMEOUT_MS);
|
|
99
|
+
timer.unref?.();
|
|
100
|
+
});
|
|
101
|
+
try {
|
|
102
|
+
return await Promise.race([promise, bound]);
|
|
103
|
+
} finally {
|
|
104
|
+
if (timer) clearTimeout(timer);
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
69
108
|
// Split streamed text into complete spoken sentences. Returns what is ready
|
|
70
109
|
// plus the trailing incomplete fragment to keep buffering. Sentences that
|
|
71
110
|
// look like code dumps (overlong, backticks) are skipped by the caller.
|
|
@@ -300,14 +339,48 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
300
339
|
}
|
|
301
340
|
}
|
|
302
341
|
|
|
342
|
+
// Spawn the sox `play` process every engine speaks through.
|
|
343
|
+
//
|
|
344
|
+
// Piper streams raw PCM, so the sample format must be declared up front.
|
|
345
|
+
// Chatterbox returns a complete WAV, which sox detects from the RIFF header -
|
|
346
|
+
// so it passes no formatArgs and we reuse this exact process rather than
|
|
347
|
+
// building a second audio pipeline.
|
|
348
|
+
function startPlayProcess(formatArgs, onDone) {
|
|
349
|
+
let playStderr = "";
|
|
350
|
+
playProc = spawn("play", [...formatArgs, "-q", "-"], {
|
|
351
|
+
stdio: ["pipe", "ignore", "pipe"],
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
playProc.stderr.on("data", (chunk) => {
|
|
355
|
+
playStderr += chunk.toString();
|
|
356
|
+
});
|
|
357
|
+
|
|
358
|
+
playProc.on("close", (code) => {
|
|
359
|
+
if (code !== 0 && code !== null) {
|
|
360
|
+
logger?.log?.("TTS", `play exited code=${code} stderr=${playStderr.trim()}`, "error");
|
|
361
|
+
} else {
|
|
362
|
+
logger?.log?.("TTS", "playback finished", "debug");
|
|
363
|
+
}
|
|
364
|
+
piperProc = null;
|
|
365
|
+
playProc = null;
|
|
366
|
+
onDone();
|
|
367
|
+
});
|
|
368
|
+
|
|
369
|
+
playProc.on("error", (err) => {
|
|
370
|
+
logger?.log?.("TTS", `play error: ${err.message}`, "error");
|
|
371
|
+
killProcs();
|
|
372
|
+
onDone();
|
|
373
|
+
});
|
|
374
|
+
|
|
375
|
+
return playProc;
|
|
376
|
+
}
|
|
377
|
+
|
|
303
378
|
// Spawn piper -> play and wire them. Resolves onDone when playback ends.
|
|
304
379
|
|
|
305
380
|
function spawnPlayback(voiceModel, onDone) {
|
|
306
381
|
let piperStderr = "";
|
|
307
|
-
let playStderr = "";
|
|
308
382
|
const rate = getVoiceRate(voiceModel);
|
|
309
|
-
|
|
310
|
-
"play",
|
|
383
|
+
startPlayProcess(
|
|
311
384
|
[
|
|
312
385
|
"-t",
|
|
313
386
|
"raw",
|
|
@@ -319,10 +392,8 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
319
392
|
String(PIPER_BITS),
|
|
320
393
|
"-c",
|
|
321
394
|
String(PIPER_CHANNELS),
|
|
322
|
-
"-q",
|
|
323
|
-
"-",
|
|
324
395
|
],
|
|
325
|
-
|
|
396
|
+
onDone,
|
|
326
397
|
);
|
|
327
398
|
|
|
328
399
|
piperProc = spawn("piper", ["-m", voiceModel, "--output_raw"], {
|
|
@@ -332,9 +403,6 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
332
403
|
piperProc.stderr.on("data", (chunk) => {
|
|
333
404
|
piperStderr += chunk.toString();
|
|
334
405
|
});
|
|
335
|
-
playProc.stderr.on("data", (chunk) => {
|
|
336
|
-
playStderr += chunk.toString();
|
|
337
|
-
});
|
|
338
406
|
|
|
339
407
|
piperProc.stdout.on("data", (chunk) => {
|
|
340
408
|
if (playProc?.stdin && !playProc.stdin.destroyed) {
|
|
@@ -351,27 +419,20 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
351
419
|
}
|
|
352
420
|
});
|
|
353
421
|
|
|
354
|
-
playProc.on("close", (code) => {
|
|
355
|
-
if (code !== 0 && code !== null) {
|
|
356
|
-
logger?.log?.("TTS", `play exited code=${code} stderr=${playStderr.trim()}`, "error");
|
|
357
|
-
} else {
|
|
358
|
-
logger?.log?.("TTS", "playback finished", "debug");
|
|
359
|
-
}
|
|
360
|
-
piperProc = null;
|
|
361
|
-
playProc = null;
|
|
362
|
-
onDone();
|
|
363
|
-
});
|
|
364
|
-
|
|
365
422
|
piperProc.on("error", (err) => {
|
|
366
423
|
logger?.log?.("TTS", `piper error: ${err.message}`, "error");
|
|
367
424
|
killProcs();
|
|
368
425
|
onDone();
|
|
369
426
|
});
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
// Chatterbox playback: the sidecar already produced whole WAV bytes, so this
|
|
430
|
+
// is just the same play process fed the buffer and closed.
|
|
431
|
+
function spawnWavPlayback(wavBuffer, onDone) {
|
|
432
|
+
const play = startPlayProcess([], onDone);
|
|
433
|
+
if (play?.stdin && !play.stdin.destroyed) {
|
|
434
|
+
play.stdin.end(wavBuffer);
|
|
435
|
+
}
|
|
375
436
|
}
|
|
376
437
|
|
|
377
438
|
function resolveVoiceOrWarn(line) {
|
|
@@ -389,6 +450,82 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
389
450
|
return voiceModel;
|
|
390
451
|
}
|
|
391
452
|
|
|
453
|
+
// ---- Optional Chatterbox engine (lib/chatterbox-server.js) ----
|
|
454
|
+
// Piper stays the default and stays the fallback: this is an opt-in quality
|
|
455
|
+
// upgrade, never a new dependency. Piper has a Vietnamese voice and is
|
|
456
|
+
// already installed; Chatterbox does not speak Vietnamese at all (its
|
|
457
|
+
// multilingual model covers 23 languages, vi is not one), so a vi utterance
|
|
458
|
+
// is routed to Piper regardless of variant - see resolveTtsEngine.
|
|
459
|
+
const createChatterbox = deps.createChatterboxClient ?? createChatterboxClient;
|
|
460
|
+
let chatterboxClient = null;
|
|
461
|
+
// Shared in-flight start so two utterances never load the model twice.
|
|
462
|
+
let chatterboxStarting = null;
|
|
463
|
+
// A synthesis that is waiting on the sidecar is still "speaking" as far as
|
|
464
|
+
// stop/isSpeaking are concerned, so a cancel lands exactly as it would on
|
|
465
|
+
// Piper's stdout stream.
|
|
466
|
+
let synthPending = false;
|
|
467
|
+
// One toast per availability state change, not one per utterance.
|
|
468
|
+
let chatterboxToastShown = false;
|
|
469
|
+
|
|
470
|
+
function chatterboxWarn(message) {
|
|
471
|
+
logger?.log?.("TTS", message, "warn");
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
function noteChatterboxFailure(message) {
|
|
475
|
+
logger?.log?.("TTS", message, "warn");
|
|
476
|
+
if (chatterboxToastShown) return;
|
|
477
|
+
chatterboxToastShown = true;
|
|
478
|
+
toast(`${message} - using Piper`);
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
// Recovering (or simply working) re-arms the toast, so a later outage is
|
|
482
|
+
// reported once again instead of staying silent for the rest of the session.
|
|
483
|
+
function noteChatterboxOk() {
|
|
484
|
+
chatterboxToastShown = false;
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
function getChatterboxClient() {
|
|
488
|
+
if (!chatterboxClient) {
|
|
489
|
+
chatterboxClient = createChatterbox({
|
|
490
|
+
variant: opts?.ttsChatterboxVariant,
|
|
491
|
+
python: opts?.ttsChatterboxPython,
|
|
492
|
+
voiceRef: opts?.ttsChatterboxVoiceRef,
|
|
493
|
+
logger,
|
|
494
|
+
});
|
|
495
|
+
}
|
|
496
|
+
return chatterboxClient;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
// Start the sidecar lazily, on the first utterance that actually needs it.
|
|
500
|
+
// Never at plugin load: the first run downloads model weights, so eagerly
|
|
501
|
+
// loading would mean every opencode session pays for a TTS nobody asked for.
|
|
502
|
+
async function ensureChatterboxReady() {
|
|
503
|
+
if (chatterboxClient?.isRunning()) return true;
|
|
504
|
+
if (!chatterboxStarting) {
|
|
505
|
+
chatterboxStarting = getChatterboxClient()
|
|
506
|
+
.start()
|
|
507
|
+
.then((ok) => {
|
|
508
|
+
if (ok) {
|
|
509
|
+
noteChatterboxOk();
|
|
510
|
+
return true;
|
|
511
|
+
}
|
|
512
|
+
const err = chatterboxClient.getLastError?.();
|
|
513
|
+
noteChatterboxFailure(
|
|
514
|
+
`Chatterbox unavailable (${err?.code || "START_FAILED"}: ${err?.message || "unknown"})`,
|
|
515
|
+
);
|
|
516
|
+
return false;
|
|
517
|
+
})
|
|
518
|
+
.catch((err) => {
|
|
519
|
+
noteChatterboxFailure(`Chatterbox unavailable (${err.message})`);
|
|
520
|
+
return false;
|
|
521
|
+
})
|
|
522
|
+
.finally(() => {
|
|
523
|
+
chatterboxStarting = null;
|
|
524
|
+
});
|
|
525
|
+
}
|
|
526
|
+
return chatterboxStarting;
|
|
527
|
+
}
|
|
528
|
+
|
|
392
529
|
function speak(text) {
|
|
393
530
|
const myGen = speechGen;
|
|
394
531
|
if (!text) return Promise.resolve();
|
|
@@ -397,14 +534,23 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
397
534
|
|
|
398
535
|
killProcs();
|
|
399
536
|
|
|
537
|
+
const lang = detectLang(line);
|
|
538
|
+
const route = resolveTtsEngine(opts, lang, chatterboxWarn);
|
|
539
|
+
if (route.engine === "chatterbox") {
|
|
540
|
+
return speakViaChatterbox(line, lang, route.variant, myGen);
|
|
541
|
+
}
|
|
400
542
|
// One voice for the whole call (see detectLang) - no mid-utterance
|
|
401
543
|
// switching.
|
|
544
|
+
return speakViaPiper(line, myGen);
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
// Piper synthesis path, shared by the default route and by the Chatterbox
|
|
548
|
+
// fallback so both play identically.
|
|
549
|
+
function speakViaPiper(line, myGen = speechGen) {
|
|
402
550
|
const voiceModel = resolveVoiceOrWarn(line);
|
|
403
551
|
if (!voiceModel) return Promise.resolve();
|
|
404
|
-
// A cancel (stopSpeech) during normalization invalidates this speech.
|
|
405
552
|
if (myGen !== speechGen) return Promise.resolve();
|
|
406
553
|
logger?.log?.("TTS", `Speak requested chars=${line.length} voice=${voiceModel}`, "debug");
|
|
407
|
-
|
|
408
554
|
return new Promise((resolve) => {
|
|
409
555
|
spawnPlayback(voiceModel, resolve);
|
|
410
556
|
if (piperProc?.stdin && !piperProc.stdin.destroyed) {
|
|
@@ -414,6 +560,49 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
414
560
|
});
|
|
415
561
|
}
|
|
416
562
|
|
|
563
|
+
async function speakViaChatterbox(line, lang, variant, myGen) {
|
|
564
|
+
// Claim "speaking" for the WHOLE attempt, including the sidecar startup
|
|
565
|
+
// wait: set before the first await, so /tts-stop and the global cancel act
|
|
566
|
+
// on a Chatterbox utterance the instant it is requested.
|
|
567
|
+
synthPending = true;
|
|
568
|
+
try {
|
|
569
|
+
const ready = await ensureChatterboxReady();
|
|
570
|
+
// A cancel that landed while the sidecar was loading must not speak.
|
|
571
|
+
if (myGen !== speechGen) return;
|
|
572
|
+
if (!ready) return speakViaPiper(line, myGen);
|
|
573
|
+
|
|
574
|
+
let result;
|
|
575
|
+
try {
|
|
576
|
+
result = await getChatterboxClient().synthesize(line, lang);
|
|
577
|
+
} catch (err) {
|
|
578
|
+
// The client already returns typed errors; this only guards a thrown
|
|
579
|
+
// client bug so speech still falls back instead of rejecting.
|
|
580
|
+
result = { ok: false, code: "REQUEST_FAILED", message: err.message };
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
// Same generation guard as Piper: a stop during synthesis must prevent
|
|
584
|
+
// the audio from ever reaching sox.
|
|
585
|
+
if (myGen !== speechGen) return;
|
|
586
|
+
if (!result?.ok) {
|
|
587
|
+
noteChatterboxFailure(`Chatterbox synthesis failed (${result?.code}: ${result?.message})`);
|
|
588
|
+
return speakViaPiper(line, myGen);
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
noteChatterboxOk();
|
|
592
|
+
logger?.log?.(
|
|
593
|
+
"TTS",
|
|
594
|
+
`Speak via Chatterbox chars=${line.length} lang=${lang} variant=${variant} wavBytes=${result.wav.length}`,
|
|
595
|
+
"debug",
|
|
596
|
+
);
|
|
597
|
+
return new Promise((resolve) => {
|
|
598
|
+
spawnWavPlayback(result.wav, resolve);
|
|
599
|
+
});
|
|
600
|
+
} finally {
|
|
601
|
+
// Playback, if it started, is tracked by playProc from here on.
|
|
602
|
+
synthPending = false;
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
417
606
|
// ---- Session-prefixed announcements ----
|
|
418
607
|
|
|
419
608
|
async function speakWithSessionPrefix(sessionID, message, suffix) {
|
|
@@ -429,7 +618,10 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
429
618
|
// Always invalidate: a pending normalize handoff must not speak after
|
|
430
619
|
// the cancel even when nothing is currently playing.
|
|
431
620
|
speechGen += 1;
|
|
432
|
-
|
|
621
|
+
// A waiting Chatterbox synthesis counts as playing, so /tts-stop and the
|
|
622
|
+
// global cancel kill it exactly like Piper playback.
|
|
623
|
+
const wasPlaying = piperProc !== null || playProc !== null || synthPending;
|
|
624
|
+
synthPending = false;
|
|
433
625
|
killProcs();
|
|
434
626
|
return wasPlaying;
|
|
435
627
|
}
|
|
@@ -473,14 +665,11 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
473
665
|
clearProcessingToast();
|
|
474
666
|
return;
|
|
475
667
|
}
|
|
476
|
-
if (!llmResult
|
|
668
|
+
if (!announceNormalize("Auto", llmResult)) {
|
|
477
669
|
clearProcessingToast();
|
|
478
|
-
logger?.log?.("TTS", `Auto normalization failed: ${llmResult.error}`, "warn");
|
|
479
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
480
670
|
return;
|
|
481
671
|
}
|
|
482
672
|
|
|
483
|
-
logger?.log?.("TTS", `Auto normalization succeeded chars=${llmResult.text.length}`, "debug");
|
|
484
673
|
clearProcessingToast();
|
|
485
674
|
await speakWithSessionPrefix(sessionID, llmResult.text, "Ready for your input.");
|
|
486
675
|
});
|
|
@@ -519,14 +708,11 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
519
708
|
clearProcessingToast();
|
|
520
709
|
return;
|
|
521
710
|
}
|
|
522
|
-
if (!llmResult
|
|
711
|
+
if (!announceNormalize("Manual", llmResult)) {
|
|
523
712
|
clearProcessingToast();
|
|
524
|
-
logger?.log?.("TTS", `Manual normalization failed: ${llmResult.error}`, "warn");
|
|
525
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
526
713
|
return;
|
|
527
714
|
}
|
|
528
715
|
|
|
529
|
-
logger?.log?.("TTS", `Manual normalization succeeded chars=${llmResult.text.length}`, "debug");
|
|
530
716
|
updateProcessingToast("Speaking...");
|
|
531
717
|
await speak(llmResult.text);
|
|
532
718
|
clearProcessingToast();
|
|
@@ -541,7 +727,31 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
541
727
|
|
|
542
728
|
async function normalizeOrLocal(text, systemPrompt, maxTokens) {
|
|
543
729
|
if (ttsLocal) return { text: localSpeechCleanup(text) };
|
|
544
|
-
|
|
730
|
+
const llmResult = await normalizeWithinBound(normalizeForSpeech(text, systemPrompt, maxTokens));
|
|
731
|
+
return resolveSpeechText(text, llmResult);
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
// Shared tail for all three normalize call sites: one informational toast
|
|
735
|
+
// when we are speaking the local fallback, and - when the cleanup left
|
|
736
|
+
// nothing speakable (a pure code dump) - a quiet debug log, because nothing
|
|
737
|
+
// failed there, there was just nothing to read. Returns whether to speak.
|
|
738
|
+
|
|
739
|
+
function announceNormalize(label, result) {
|
|
740
|
+
if (!result.text) {
|
|
741
|
+
logger?.log?.("TTS", `${label}: nothing speakable (${result.error})`, "debug");
|
|
742
|
+
return false;
|
|
743
|
+
}
|
|
744
|
+
if (result.fellBack) {
|
|
745
|
+
logger?.log?.(
|
|
746
|
+
"TTS",
|
|
747
|
+
`${label} narration unavailable (${result.error}); speaking local cleanup`,
|
|
748
|
+
"warn",
|
|
749
|
+
);
|
|
750
|
+
toast("Model unavailable - speaking text as-is");
|
|
751
|
+
return true;
|
|
752
|
+
}
|
|
753
|
+
logger?.log?.("TTS", `${label} normalization succeeded chars=${result.text.length}`, "debug");
|
|
754
|
+
return true;
|
|
545
755
|
}
|
|
546
756
|
|
|
547
757
|
async function speakAssistantTurn() {
|
|
@@ -564,11 +774,9 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
564
774
|
clearProcessingToast();
|
|
565
775
|
return { spoken: false };
|
|
566
776
|
}
|
|
567
|
-
if (!llmResult
|
|
777
|
+
if (!announceNormalize("Conversation", llmResult)) {
|
|
568
778
|
clearProcessingToast();
|
|
569
|
-
|
|
570
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
571
|
-
return { spoken: false, error: llmResult.error };
|
|
779
|
+
return { spoken: false };
|
|
572
780
|
}
|
|
573
781
|
|
|
574
782
|
updateProcessingToast("Speaking...");
|
|
@@ -601,7 +809,9 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
601
809
|
speak: (text) => speak(text),
|
|
602
810
|
speakAssistantTurn,
|
|
603
811
|
stop: () => stopSpeech(),
|
|
604
|
-
|
|
812
|
+
// A Chatterbox utterance is "speaking" from the moment it is handed to the
|
|
813
|
+
// sidecar, not only once audio reaches sox.
|
|
814
|
+
isSpeaking: () => piperProc !== null || playProc !== null || synthPending,
|
|
605
815
|
setConversationActive: (v) => {
|
|
606
816
|
conversationActive = !!v;
|
|
607
817
|
},
|
|
@@ -610,6 +820,14 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
610
820
|
},
|
|
611
821
|
};
|
|
612
822
|
|
|
823
|
+
// The Chatterbox sidecar holds a multi-hundred-MB torch model; leaving it
|
|
824
|
+
// running past the session would strand that memory. Same lifecycle hook the
|
|
825
|
+
// STT and live-notes servers use.
|
|
826
|
+
api.lifecycle?.onDispose?.(() => {
|
|
827
|
+
chatterboxClient?.stop();
|
|
828
|
+
chatterboxClient = null;
|
|
829
|
+
});
|
|
830
|
+
|
|
613
831
|
// ---- Commands ----
|
|
614
832
|
|
|
615
833
|
const commands = [
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bojackduy/opencode-voice",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "Speech-to-text and text-to-speech for OpenCode. Record voice prompts with whisper-cpp, hear responses via Piper TTS, with LLM normalization through any OpenAI-compatible endpoint.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"opencode",
|