@bojackduy/opencode-voice 0.9.0 → 0.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -0
- package/lib/chatterbox-server.js +428 -0
- package/lib/tts.js +193 -29
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -243,6 +243,65 @@ English reply no longer flips the whole thing to the Vietnamese voice, and
|
|
|
243
243
|
vice versa. Toneless Vietnamese (no diacritics) still reads as English -
|
|
244
244
|
that is not distinguishable from English by this heuristic.
|
|
245
245
|
|
|
246
|
+
### Chatterbox engine (optional)
|
|
247
|
+
|
|
248
|
+
Piper is the default and needs no setup beyond the section above. For more
|
|
249
|
+
natural-sounding English, opt into [Chatterbox](https://github.com/resemble-ai/chatterbox)
|
|
250
|
+
(Resemble AI, MIT) as the synthesis engine. It runs as a managed sidecar
|
|
251
|
+
(`vendor/chatterbox_server.py`, stdlib HTTP only): the model loads once, the
|
|
252
|
+
plugin probes readiness, and the sidecar is killed on dispose. Playback still
|
|
253
|
+
goes through the existing `play` path, so `/tts-stop` and cancel behave the
|
|
254
|
+
same on both engines.
|
|
255
|
+
|
|
256
|
+
Setup (isolated venv, never touches your system python):
|
|
257
|
+
|
|
258
|
+
```bash
|
|
259
|
+
python3 -m venv ~/.local/share/opencode-voice/chatterbox-venv
|
|
260
|
+
~/.local/share/opencode-voice/chatterbox-venv/bin/pip install chatterbox-tts
|
|
261
|
+
# chatterbox's `perth` dependency needs pkg_resources (removed in setuptools ≥ 81):
|
|
262
|
+
~/.local/share/opencode-voice/chatterbox-venv/bin/pip install 'setuptools<81'
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
First run downloads ~7GB of weights from HuggingFace, then the sidecar takes
|
|
266
|
+
~20s to load on Apple Silicon (MPS). No venv, no weights, no wavs are ever
|
|
267
|
+
committed.
|
|
268
|
+
|
|
269
|
+
Options in `tui.json` (all under the plugin entry, next to `endpoint`):
|
|
270
|
+
|
|
271
|
+
```json
|
|
272
|
+
{
|
|
273
|
+
"ttsEngine": "chatterbox",
|
|
274
|
+
"ttsChatterboxVariant": "multilingual",
|
|
275
|
+
"ttsChatterboxVoiceRef": "/absolute/path/to/voice-5-20s.wav"
|
|
276
|
+
}
|
|
277
|
+
```
|
|
278
|
+
|
|
279
|
+
- `ttsEngine`: `"piper"` (default) | `"chatterbox"`. Unknown values warn and
|
|
280
|
+
use Piper.
|
|
281
|
+
- `ttsChatterboxVariant`: `"multilingual"` (default) | `"turbo"` | `"nano"`.
|
|
282
|
+
- `ttsChatterboxVoiceRef`: optional absolute path to a 5-20s reference wav
|
|
283
|
+
for zero-shot voice cloning. Unset means the model default voice; a missing
|
|
284
|
+
file warns and proceeds voiceless.
|
|
285
|
+
- `ttsChatterboxPython`: optional python binary for the sidecar. Defaults to
|
|
286
|
+
`python3` on `PATH` - set it to the venv python above.
|
|
287
|
+
|
|
288
|
+
Per-utterance Piper fallback: whenever Chatterbox cannot speak an utterance,
|
|
289
|
+
that utterance goes to Piper with a warn log (one toast per outage, not per
|
|
290
|
+
utterance; the next utterance retries Chatterbox). Cases: sidecar not
|
|
291
|
+
installed/crashed, ~15s synthesis bound exceeded, and language routing below.
|
|
292
|
+
|
|
293
|
+
Honest limits, measured on Apple Silicon (MPS) with Multilingual V3:
|
|
294
|
+
|
|
295
|
+
- Chatterbox does **not** speak Vietnamese. Its multilingual model covers 23
|
|
296
|
+
languages (`ar da de el en es fi fr he hi it ja ko ms nl no pl pt ru sv sw
|
|
297
|
+
th tr zh`) - `vi` is not one, so **every** Vietnamese utterance falls back
|
|
298
|
+
to Piper regardless of variant. Nano/Turbo are English-only by design.
|
|
299
|
+
- Latency for "Deploying now.": Piper 0.77s wall for 0.80s of audio (RTF
|
|
300
|
+
~1.0); Chatterbox ~16s wall for ~1s of audio on first synthesis after load
|
|
301
|
+
(RTF ~15, MPS warmup included), ~5-8s wall (RTF ~1.5-3) once warm. It is a
|
|
302
|
+
quality upgrade, not a speed one.
|
|
303
|
+
- Outputs carry Chatterbox's inaudible PerTh watermark.
|
|
304
|
+
|
|
246
305
|
### LLM endpoint
|
|
247
306
|
|
|
248
307
|
An OpenAI-compatible LLM endpoint is required for text normalization. For
|
|
@@ -0,0 +1,428 @@
|
|
|
1
|
+
// Managed Chatterbox TTS sidecar (see vendor/chatterbox_server.py).
|
|
2
|
+
//
|
|
3
|
+
// Why a sidecar instead of a one-shot CLI: Chatterbox loads a large torch
|
|
4
|
+
// model. Loading it per utterance would never keep up with a spoken
|
|
5
|
+
// conversation, so - exactly like lib/whisper-server.js for whisper.cpp - one
|
|
6
|
+
// child process loads the model once and serves synthesis over HTTP while we
|
|
7
|
+
// own its lifetime.
|
|
8
|
+
//
|
|
9
|
+
// Follows the whisper-server conventions deliberately:
|
|
10
|
+
// - Port is probed BEFORE spawning; if anything already answers we never claim
|
|
11
|
+
// it and never kill the foreign process. Default-port callers auto-advance
|
|
12
|
+
// upward (bounded) so a second TUI lands on its own port with its own model
|
|
13
|
+
// load instead of being stuck on Piper forever.
|
|
14
|
+
// - Readiness is GET /health: 200 = ready, 503 = still loading or failed. The
|
|
15
|
+
// sidecar answers 503 with the real reason ("chatterbox-tts not installed"),
|
|
16
|
+
// so we can fail fast with an actionable message instead of waiting out the
|
|
17
|
+
// whole timeout on an install that will never succeed.
|
|
18
|
+
// - stop() only ever signals the handle this client spawned.
|
|
19
|
+
// - Every failure is a typed {code, message}, never a silent return of nothing:
|
|
20
|
+
// the caller turns it into a per-utterance Piper fallback.
|
|
21
|
+
//
|
|
22
|
+
// This client is an OPTIMIZATION. Piper is the default engine and stays the
|
|
23
|
+
// zero-setup fallback, so every failure path here ends in Piper, never silence.
|
|
24
|
+
|
|
25
|
+
import fs from "node:fs";
|
|
26
|
+
import os from "node:os";
|
|
27
|
+
import { fileURLToPath } from "node:url";
|
|
28
|
+
import { spawn } from "node:child_process";
|
|
29
|
+
|
|
30
|
+
const DEFAULT_HOST = "127.0.0.1";
|
|
31
|
+
const DEFAULT_PORT = 8120;
|
|
32
|
+
const DEFAULT_PORT_SCAN_MAX = 4;
|
|
33
|
+
// Model load on first run also downloads weights from HuggingFace, so the
|
|
34
|
+
// readiness budget is generous. It is a bound, not a promise: a failed load
|
|
35
|
+
// reports immediately via the 503 body instead of waiting this out.
|
|
36
|
+
const READY_TIMEOUT_MS = 300000;
|
|
37
|
+
const READY_POLL_MS = 500;
|
|
38
|
+
// One utterance must never stall speech indefinitely.
|
|
39
|
+
const SYNTHESIZE_TIMEOUT_MS = 15000;
|
|
40
|
+
|
|
41
|
+
const SIDECAR_PATH = fileURLToPath(new URL("../vendor/chatterbox_server.py", import.meta.url));
|
|
42
|
+
|
|
43
|
+
export const CHATTERBOX_VARIANTS = ["multilingual", "turbo", "nano"];
|
|
44
|
+
export const DEFAULT_CHATTERBOX_VARIANT = "multilingual";
|
|
45
|
+
|
|
46
|
+
// Per-variant language support, read off the real package
|
|
47
|
+
// (chatterbox-tts 0.1.7) rather than guessed.
|
|
48
|
+
//
|
|
49
|
+
// MEASURED, and worth stating plainly: the multilingual model speaks 23
|
|
50
|
+
// languages and Vietnamese is NOT one of them. Its generate() raises
|
|
51
|
+
// ValueError on language_id "vi". Nano/Turbo are English-only by design. So
|
|
52
|
+
// every variant routes a vi utterance to Piper, and Chatterbox covers English.
|
|
53
|
+
// Do not "fix" this by adding "vi" here - it would fail at synthesis time.
|
|
54
|
+
const CHATTERBOX_LANGUAGES = {
|
|
55
|
+
multilingual: new Set([
|
|
56
|
+
"ar",
|
|
57
|
+
"da",
|
|
58
|
+
"de",
|
|
59
|
+
"el",
|
|
60
|
+
"en",
|
|
61
|
+
"es",
|
|
62
|
+
"fi",
|
|
63
|
+
"fr",
|
|
64
|
+
"he",
|
|
65
|
+
"hi",
|
|
66
|
+
"it",
|
|
67
|
+
"ja",
|
|
68
|
+
"ko",
|
|
69
|
+
"ms",
|
|
70
|
+
"nl",
|
|
71
|
+
"no",
|
|
72
|
+
"pl",
|
|
73
|
+
"pt",
|
|
74
|
+
"ru",
|
|
75
|
+
"sv",
|
|
76
|
+
"sw",
|
|
77
|
+
"th",
|
|
78
|
+
"tr",
|
|
79
|
+
"zh",
|
|
80
|
+
]),
|
|
81
|
+
turbo: new Set(["en"]),
|
|
82
|
+
nano: new Set(["en"]),
|
|
83
|
+
};
|
|
84
|
+
|
|
85
|
+
export function isChatterboxVariant(value) {
|
|
86
|
+
return CHATTERBOX_VARIANTS.includes(value);
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
// Can this variant speak this language? Unknown variants answer false, which
|
|
90
|
+
// routes to Piper - the safe default, since Piper always has an answer.
|
|
91
|
+
export function chatterboxSupportsLang(variant, lang) {
|
|
92
|
+
const set = CHATTERBOX_LANGUAGES[variant];
|
|
93
|
+
return !!set && set.has(lang);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
// The one routing decision, pure and exported so it can be tested without a
|
|
97
|
+
// sidecar, a model, or an audio device. Returns the engine to use plus the
|
|
98
|
+
// reason, which the caller logs (and which keeps per-utterance skips quiet in
|
|
99
|
+
// the UI - no toast spam for a language Piper can already read).
|
|
100
|
+
export function resolveTtsEngine(opts, lang, warn) {
|
|
101
|
+
const configured = opts?.ttsEngine;
|
|
102
|
+
if (!configured || configured === "piper") return { engine: "piper", reason: "default" };
|
|
103
|
+
if (configured !== "chatterbox") {
|
|
104
|
+
warn?.(`Unknown ttsEngine "${configured}" - expected "piper" or "chatterbox"; using Piper`);
|
|
105
|
+
return { engine: "piper", reason: "unknown-engine" };
|
|
106
|
+
}
|
|
107
|
+
const requested = opts?.ttsChatterboxVariant;
|
|
108
|
+
if (requested && !isChatterboxVariant(requested)) {
|
|
109
|
+
warn?.(
|
|
110
|
+
`Unknown ttsChatterboxVariant "${requested}" - expected ${CHATTERBOX_VARIANTS.join(" | ")}; using "${DEFAULT_CHATTERBOX_VARIANT}"`,
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
const variant = isChatterboxVariant(requested) ? requested : DEFAULT_CHATTERBOX_VARIANT;
|
|
114
|
+
if (chatterboxSupportsLang(variant, lang)) return { engine: "chatterbox", variant, reason: "ok" };
|
|
115
|
+
warn?.(`Chatterbox variant "${variant}" cannot speak "${lang}" - using Piper for this utterance`);
|
|
116
|
+
return { engine: "piper", reason: "unsupported-language", variant };
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
export function buildChatterboxArgs({ scriptPath, host, port, variant }) {
|
|
120
|
+
return [
|
|
121
|
+
scriptPath || SIDECAR_PATH,
|
|
122
|
+
"--host",
|
|
123
|
+
host || DEFAULT_HOST,
|
|
124
|
+
"--port",
|
|
125
|
+
String(port || DEFAULT_PORT),
|
|
126
|
+
"--variant",
|
|
127
|
+
variant || DEFAULT_CHATTERBOX_VARIANT,
|
|
128
|
+
];
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
export { DEFAULT_HOST, DEFAULT_PORT, DEFAULT_PORT_SCAN_MAX, SIDECAR_PATH };
|
|
132
|
+
|
|
133
|
+
export function createChatterboxClient(options = {}) {
|
|
134
|
+
const {
|
|
135
|
+
host = DEFAULT_HOST,
|
|
136
|
+
python,
|
|
137
|
+
variant = DEFAULT_CHATTERBOX_VARIANT,
|
|
138
|
+
voiceRef,
|
|
139
|
+
exaggeration,
|
|
140
|
+
cfgWeight,
|
|
141
|
+
logger,
|
|
142
|
+
readyTimeoutMs = READY_TIMEOUT_MS,
|
|
143
|
+
readyPollMs = READY_POLL_MS,
|
|
144
|
+
synthesizeTimeoutMs = SYNTHESIZE_TIMEOUT_MS,
|
|
145
|
+
portScanMax = DEFAULT_PORT_SCAN_MAX,
|
|
146
|
+
deps = {},
|
|
147
|
+
} = options;
|
|
148
|
+
const scriptPath = options.scriptPath || SIDECAR_PATH;
|
|
149
|
+
const spawnFn = deps.spawn ?? spawn;
|
|
150
|
+
const fetchFn = deps.fetch ?? fetch;
|
|
151
|
+
|
|
152
|
+
let proc = null;
|
|
153
|
+
let ready = false;
|
|
154
|
+
let ownerPid = null;
|
|
155
|
+
// Where the port scan starts. The plugin never sets it (so production always
|
|
156
|
+
// begins at 8120), but tests bind an ephemeral port and need to reach it.
|
|
157
|
+
const basePort = options.port ?? DEFAULT_PORT;
|
|
158
|
+
let boundPort = basePort;
|
|
159
|
+
let lastError = null;
|
|
160
|
+
let device = "unknown";
|
|
161
|
+
// A missing reference wav must not kill the whole engine - the sidecar still
|
|
162
|
+
// speaks with its default voice, which is better than falling back to Piper.
|
|
163
|
+
let refPath = voiceRef ? String(voiceRef).replace(/^~(?=\/|$)/, os.homedir()) : "";
|
|
164
|
+
if (refPath && !fs.existsSync(refPath)) {
|
|
165
|
+
logger?.log(
|
|
166
|
+
"VOICE",
|
|
167
|
+
`Chatterbox voice reference not found, using default voice: ${refPath}`,
|
|
168
|
+
"warn",
|
|
169
|
+
);
|
|
170
|
+
refPath = "";
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
const baseUrlFor = (p) => `http://${host}:${p}`;
|
|
174
|
+
|
|
175
|
+
async function portResponds(p) {
|
|
176
|
+
const base = baseUrlFor(p);
|
|
177
|
+
for (const url of [`${base}/health`, base]) {
|
|
178
|
+
try {
|
|
179
|
+
const resp = await fetchFn(url, { signal: AbortSignal.timeout(1500) });
|
|
180
|
+
if (resp) return true;
|
|
181
|
+
} catch {
|
|
182
|
+
// Connection refused / timeout: try the next URL.
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
return false;
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
// One probe. "ready" | "loading" | {failed: message} | "down". The failed
|
|
189
|
+
// shape carries the sidecar's own reason so start() can surface it verbatim.
|
|
190
|
+
async function probeOnce() {
|
|
191
|
+
try {
|
|
192
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/health`, {
|
|
193
|
+
signal: AbortSignal.timeout(2000),
|
|
194
|
+
});
|
|
195
|
+
if (!resp) return "down";
|
|
196
|
+
if (resp.status === 503) {
|
|
197
|
+
let body = {};
|
|
198
|
+
try {
|
|
199
|
+
body = await resp.json();
|
|
200
|
+
} catch {
|
|
201
|
+
// Non-JSON 503: treat as still loading rather than guessing.
|
|
202
|
+
}
|
|
203
|
+
if (body?.phase === "failed") return { failed: body?.error || "model load failed" };
|
|
204
|
+
return "loading";
|
|
205
|
+
}
|
|
206
|
+
if (resp?.ok) {
|
|
207
|
+
try {
|
|
208
|
+
const body = await resp.json();
|
|
209
|
+
if (body?.device) device = body.device;
|
|
210
|
+
} catch {
|
|
211
|
+
// 200 with an unparseable body still means the port answered /health.
|
|
212
|
+
}
|
|
213
|
+
return "ready";
|
|
214
|
+
}
|
|
215
|
+
return "down";
|
|
216
|
+
} catch {
|
|
217
|
+
return "down";
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
async function pollReady(deadline) {
|
|
222
|
+
while (Date.now() < deadline) {
|
|
223
|
+
if (!proc) return { failed: "chatterbox sidecar exited before becoming ready" };
|
|
224
|
+
const state = await probeOnce();
|
|
225
|
+
if (state === "ready") return state;
|
|
226
|
+
if (state?.failed) return state;
|
|
227
|
+
await new Promise((r) => setTimeout(r, readyPollMs));
|
|
228
|
+
}
|
|
229
|
+
return "timeout";
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// Attempt one port: probe, spawn, wait for readiness. stop() inside only ever
|
|
233
|
+
// signals the handle spawned here.
|
|
234
|
+
async function startOnPort(port) {
|
|
235
|
+
boundPort = port;
|
|
236
|
+
if (await portResponds(port)) {
|
|
237
|
+
return {
|
|
238
|
+
ok: false,
|
|
239
|
+
code: "PORT_IN_USE",
|
|
240
|
+
advancable: true,
|
|
241
|
+
message: `Port ${port} already answers - not starting a second chatterbox sidecar`,
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
const args = buildChatterboxArgs({ scriptPath, host, port, variant });
|
|
245
|
+
const bin = python || "python3";
|
|
246
|
+
try {
|
|
247
|
+
proc = spawnFn(bin, args, { stdio: ["ignore", "ignore", "pipe"] });
|
|
248
|
+
} catch (err) {
|
|
249
|
+
proc = null;
|
|
250
|
+
return {
|
|
251
|
+
ok: false,
|
|
252
|
+
code: "SPAWN_FAILED",
|
|
253
|
+
advancable: false,
|
|
254
|
+
message: `Failed to spawn ${bin}: ${err.message}`,
|
|
255
|
+
};
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
let stderr = "";
|
|
259
|
+
const owned = proc;
|
|
260
|
+
proc.stderr?.on("data", (chunk) => {
|
|
261
|
+
stderr += chunk.toString();
|
|
262
|
+
});
|
|
263
|
+
proc.on("exit", (code) => {
|
|
264
|
+
// A late event from a previous child must not wipe the current handle.
|
|
265
|
+
if (proc !== owned) return;
|
|
266
|
+
// Record why, so a sidecar that dies mid-session explains the Piper
|
|
267
|
+
// fallback instead of reporting a bare "not ready". A deliberate stop()
|
|
268
|
+
// already nulled proc, so it never reaches here.
|
|
269
|
+
lastError = { code: "EXITED", message: `chatterbox sidecar exited (code=${code})` };
|
|
270
|
+
proc = null;
|
|
271
|
+
ready = false;
|
|
272
|
+
ownerPid = null;
|
|
273
|
+
});
|
|
274
|
+
proc.on("error", (err) => {
|
|
275
|
+
if (proc !== owned) return;
|
|
276
|
+
logger?.log("VOICE", `chatterbox sidecar process error: ${err.message}`, "warn");
|
|
277
|
+
proc = null;
|
|
278
|
+
ready = false;
|
|
279
|
+
ownerPid = null;
|
|
280
|
+
});
|
|
281
|
+
|
|
282
|
+
const outcome = await pollReady(Date.now() + readyTimeoutMs);
|
|
283
|
+
if (outcome !== "ready") {
|
|
284
|
+
const tail = stderr.trim().slice(-300);
|
|
285
|
+
const message =
|
|
286
|
+
outcome === "timeout"
|
|
287
|
+
? `chatterbox sidecar did not become ready in time (${readyTimeoutMs}ms) stderr=${tail}`
|
|
288
|
+
: `chatterbox sidecar unavailable: ${outcome.failed ?? "unknown"}`;
|
|
289
|
+
const code = outcome === "timeout" ? "START_TIMEOUT" : "MODEL_LOAD_FAILED";
|
|
290
|
+
stop();
|
|
291
|
+
return { ok: false, code, advancable: false, message };
|
|
292
|
+
}
|
|
293
|
+
ownerPid = proc?.pid ?? null;
|
|
294
|
+
return { ok: true };
|
|
295
|
+
}
|
|
296
|
+
|
|
297
|
+
async function start() {
|
|
298
|
+
if (proc && ready) return true; // reuse the owned, already-ready sidecar
|
|
299
|
+
const scanMax = Math.max(1, Math.floor(portScanMax) || 1);
|
|
300
|
+
for (let i = 0; i < scanMax; i += 1) {
|
|
301
|
+
const port = basePort + i;
|
|
302
|
+
const r = await startOnPort(port);
|
|
303
|
+
if (r.ok) {
|
|
304
|
+
ready = true;
|
|
305
|
+
lastError = null;
|
|
306
|
+
logger?.log(
|
|
307
|
+
"VOICE",
|
|
308
|
+
`chatterbox sidecar ready at ${baseUrlFor(boundPort)} variant=${variant} device=${device} pid=${ownerPid}`,
|
|
309
|
+
"debug",
|
|
310
|
+
);
|
|
311
|
+
return true;
|
|
312
|
+
}
|
|
313
|
+
if (!r.advancable) {
|
|
314
|
+
lastError = { code: r.code, message: r.message };
|
|
315
|
+
logger?.log("VOICE", r.message, "warn");
|
|
316
|
+
return false;
|
|
317
|
+
}
|
|
318
|
+
logger?.log("VOICE", r.message, "warn");
|
|
319
|
+
}
|
|
320
|
+
lastError = {
|
|
321
|
+
code: "PORT_RANGE_EXHAUSTED",
|
|
322
|
+
message: `No free port in ${basePort}..${basePort + scanMax - 1} - all answer, using Piper`,
|
|
323
|
+
};
|
|
324
|
+
logger?.log("VOICE", lastError.message, "warn");
|
|
325
|
+
return false;
|
|
326
|
+
}
|
|
327
|
+
|
|
328
|
+
// One-shot health read, for callers that want to report the engine's state
|
|
329
|
+
// (and for tests). Returns {ok, variant, device, phase, error?}.
|
|
330
|
+
async function health() {
|
|
331
|
+
try {
|
|
332
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/health`, {
|
|
333
|
+
signal: AbortSignal.timeout(2000),
|
|
334
|
+
});
|
|
335
|
+
const body = await resp.json();
|
|
336
|
+
return body;
|
|
337
|
+
} catch (err) {
|
|
338
|
+
return { ok: false, error: err.message };
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
|
|
342
|
+
// Synthesize one utterance. Returns {ok:true, wav:Buffer} or
|
|
343
|
+
// {ok:false, code, message} - never throws, so the caller always has a Piper
|
|
344
|
+
// fallback available. options.signal lets a cancel abort the in-flight request
|
|
345
|
+
// (the sidecar finishes its own generation and discards the result; the
|
|
346
|
+
// caller's generation guard is what actually prevents stale playback).
|
|
347
|
+
async function synthesize(text, languageId, opts = {}) {
|
|
348
|
+
if (!ready) {
|
|
349
|
+
return {
|
|
350
|
+
ok: false,
|
|
351
|
+
code: lastError?.code || "NOT_READY",
|
|
352
|
+
message: lastError?.message || "chatterbox sidecar not ready",
|
|
353
|
+
};
|
|
354
|
+
}
|
|
355
|
+
const body = {
|
|
356
|
+
text,
|
|
357
|
+
language_id: languageId || undefined,
|
|
358
|
+
exaggeration: opts.exaggeration ?? exaggeration ?? undefined,
|
|
359
|
+
cfg_weight: opts.cfgWeight ?? cfgWeight ?? undefined,
|
|
360
|
+
voice_ref: refPath || undefined,
|
|
361
|
+
};
|
|
362
|
+
try {
|
|
363
|
+
// One utterance must never stall speech indefinitely: bound the request,
|
|
364
|
+
// and let an outer cancel abort it too. On timeout we fall back to
|
|
365
|
+
// Piper for THIS utterance only - the sidecar stays up.
|
|
366
|
+
const timeout = AbortSignal.timeout(synthesizeTimeoutMs);
|
|
367
|
+
const signal = opts.signal ? AbortSignal.any([timeout, opts.signal]) : timeout;
|
|
368
|
+
const resp = await fetchFn(`${baseUrlFor(boundPort)}/speak`, {
|
|
369
|
+
method: "POST",
|
|
370
|
+
headers: { "Content-Type": "application/json" },
|
|
371
|
+
body: JSON.stringify(body),
|
|
372
|
+
signal,
|
|
373
|
+
});
|
|
374
|
+
if (!resp?.ok) {
|
|
375
|
+
let message = `chatterbox sidecar responded ${resp?.status}`;
|
|
376
|
+
try {
|
|
377
|
+
const data = await resp.json();
|
|
378
|
+
if (data?.error) message = data.error;
|
|
379
|
+
} catch {
|
|
380
|
+
// Keep the status-based message.
|
|
381
|
+
}
|
|
382
|
+
return { ok: false, code: "BAD_STATUS", message };
|
|
383
|
+
}
|
|
384
|
+
const audio = Buffer.from(await resp.arrayBuffer());
|
|
385
|
+
if (audio.length === 0) {
|
|
386
|
+
return { ok: false, code: "EMPTY_AUDIO", message: "chatterbox returned no audio" };
|
|
387
|
+
}
|
|
388
|
+
return { ok: true, wav: audio };
|
|
389
|
+
} catch (err) {
|
|
390
|
+
const timedOut = err?.name === "TimeoutError" || err?.name === "AbortError";
|
|
391
|
+
return {
|
|
392
|
+
ok: false,
|
|
393
|
+
code: timedOut ? "SYNTH_TIMEOUT" : "REQUEST_FAILED",
|
|
394
|
+
message: `chatterbox synthesis failed: ${err.message}`,
|
|
395
|
+
};
|
|
396
|
+
}
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
// Stop ONLY the process this client spawned (by handle). A foreign process
|
|
400
|
+
// on the same port is never signaled.
|
|
401
|
+
function stop() {
|
|
402
|
+
ready = false;
|
|
403
|
+
ownerPid = null;
|
|
404
|
+
if (proc) {
|
|
405
|
+
const owned = proc;
|
|
406
|
+
proc = null;
|
|
407
|
+
try {
|
|
408
|
+
owned.kill("SIGTERM");
|
|
409
|
+
} catch {
|
|
410
|
+
// Already gone.
|
|
411
|
+
}
|
|
412
|
+
}
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
function isRunning() {
|
|
416
|
+
return ready && proc !== null;
|
|
417
|
+
}
|
|
418
|
+
|
|
419
|
+
function getLastError() {
|
|
420
|
+
return lastError;
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
function getPort() {
|
|
424
|
+
return boundPort;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
return { start, stop, health, synthesize, isRunning, getLastError, getPort };
|
|
428
|
+
}
|
package/lib/tts.js
CHANGED
|
@@ -6,6 +6,7 @@ import os from "node:os";
|
|
|
6
6
|
import { spawn } from "node:child_process";
|
|
7
7
|
import { getSessionTitle } from "./session.js";
|
|
8
8
|
import { clearProcessingToast, showProcessingToast, updateProcessingToast } from "./stt.js";
|
|
9
|
+
import { createChatterboxClient, resolveTtsEngine } from "./chatterbox-server.js";
|
|
9
10
|
|
|
10
11
|
const VOICES_DIR = path.join(os.homedir(), ".local", "share", "piper-voices");
|
|
11
12
|
|
|
@@ -338,14 +339,48 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
338
339
|
}
|
|
339
340
|
}
|
|
340
341
|
|
|
342
|
+
// Spawn the sox `play` process every engine speaks through.
|
|
343
|
+
//
|
|
344
|
+
// Piper streams raw PCM, so the sample format must be declared up front.
|
|
345
|
+
// Chatterbox returns a complete WAV, which sox detects from the RIFF header -
|
|
346
|
+
// so it passes no formatArgs and we reuse this exact process rather than
|
|
347
|
+
// building a second audio pipeline.
|
|
348
|
+
function startPlayProcess(formatArgs, onDone) {
|
|
349
|
+
let playStderr = "";
|
|
350
|
+
playProc = spawn("play", [...formatArgs, "-q", "-"], {
|
|
351
|
+
stdio: ["pipe", "ignore", "pipe"],
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
playProc.stderr.on("data", (chunk) => {
|
|
355
|
+
playStderr += chunk.toString();
|
|
356
|
+
});
|
|
357
|
+
|
|
358
|
+
playProc.on("close", (code) => {
|
|
359
|
+
if (code !== 0 && code !== null) {
|
|
360
|
+
logger?.log?.("TTS", `play exited code=${code} stderr=${playStderr.trim()}`, "error");
|
|
361
|
+
} else {
|
|
362
|
+
logger?.log?.("TTS", "playback finished", "debug");
|
|
363
|
+
}
|
|
364
|
+
piperProc = null;
|
|
365
|
+
playProc = null;
|
|
366
|
+
onDone();
|
|
367
|
+
});
|
|
368
|
+
|
|
369
|
+
playProc.on("error", (err) => {
|
|
370
|
+
logger?.log?.("TTS", `play error: ${err.message}`, "error");
|
|
371
|
+
killProcs();
|
|
372
|
+
onDone();
|
|
373
|
+
});
|
|
374
|
+
|
|
375
|
+
return playProc;
|
|
376
|
+
}
|
|
377
|
+
|
|
341
378
|
// Spawn piper -> play and wire them. Resolves onDone when playback ends.
|
|
342
379
|
|
|
343
380
|
function spawnPlayback(voiceModel, onDone) {
|
|
344
381
|
let piperStderr = "";
|
|
345
|
-
let playStderr = "";
|
|
346
382
|
const rate = getVoiceRate(voiceModel);
|
|
347
|
-
|
|
348
|
-
"play",
|
|
383
|
+
startPlayProcess(
|
|
349
384
|
[
|
|
350
385
|
"-t",
|
|
351
386
|
"raw",
|
|
@@ -357,10 +392,8 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
357
392
|
String(PIPER_BITS),
|
|
358
393
|
"-c",
|
|
359
394
|
String(PIPER_CHANNELS),
|
|
360
|
-
"-q",
|
|
361
|
-
"-",
|
|
362
395
|
],
|
|
363
|
-
|
|
396
|
+
onDone,
|
|
364
397
|
);
|
|
365
398
|
|
|
366
399
|
piperProc = spawn("piper", ["-m", voiceModel, "--output_raw"], {
|
|
@@ -370,9 +403,6 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
370
403
|
piperProc.stderr.on("data", (chunk) => {
|
|
371
404
|
piperStderr += chunk.toString();
|
|
372
405
|
});
|
|
373
|
-
playProc.stderr.on("data", (chunk) => {
|
|
374
|
-
playStderr += chunk.toString();
|
|
375
|
-
});
|
|
376
406
|
|
|
377
407
|
piperProc.stdout.on("data", (chunk) => {
|
|
378
408
|
if (playProc?.stdin && !playProc.stdin.destroyed) {
|
|
@@ -389,27 +419,20 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
389
419
|
}
|
|
390
420
|
});
|
|
391
421
|
|
|
392
|
-
playProc.on("close", (code) => {
|
|
393
|
-
if (code !== 0 && code !== null) {
|
|
394
|
-
logger?.log?.("TTS", `play exited code=${code} stderr=${playStderr.trim()}`, "error");
|
|
395
|
-
} else {
|
|
396
|
-
logger?.log?.("TTS", "playback finished", "debug");
|
|
397
|
-
}
|
|
398
|
-
piperProc = null;
|
|
399
|
-
playProc = null;
|
|
400
|
-
onDone();
|
|
401
|
-
});
|
|
402
|
-
|
|
403
422
|
piperProc.on("error", (err) => {
|
|
404
423
|
logger?.log?.("TTS", `piper error: ${err.message}`, "error");
|
|
405
424
|
killProcs();
|
|
406
425
|
onDone();
|
|
407
426
|
});
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
427
|
+
}
|
|
428
|
+
|
|
429
|
+
// Chatterbox playback: the sidecar already produced whole WAV bytes, so this
|
|
430
|
+
// is just the same play process fed the buffer and closed.
|
|
431
|
+
function spawnWavPlayback(wavBuffer, onDone) {
|
|
432
|
+
const play = startPlayProcess([], onDone);
|
|
433
|
+
if (play?.stdin && !play.stdin.destroyed) {
|
|
434
|
+
play.stdin.end(wavBuffer);
|
|
435
|
+
}
|
|
413
436
|
}
|
|
414
437
|
|
|
415
438
|
function resolveVoiceOrWarn(line) {
|
|
@@ -427,6 +450,82 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
427
450
|
return voiceModel;
|
|
428
451
|
}
|
|
429
452
|
|
|
453
|
+
// ---- Optional Chatterbox engine (lib/chatterbox-server.js) ----
|
|
454
|
+
// Piper stays the default and stays the fallback: this is an opt-in quality
|
|
455
|
+
// upgrade, never a new dependency. Piper has a Vietnamese voice and is
|
|
456
|
+
// already installed; Chatterbox does not speak Vietnamese at all (its
|
|
457
|
+
// multilingual model covers 23 languages, vi is not one), so a vi utterance
|
|
458
|
+
// is routed to Piper regardless of variant - see resolveTtsEngine.
|
|
459
|
+
const createChatterbox = deps.createChatterboxClient ?? createChatterboxClient;
|
|
460
|
+
let chatterboxClient = null;
|
|
461
|
+
// Shared in-flight start so two utterances never load the model twice.
|
|
462
|
+
let chatterboxStarting = null;
|
|
463
|
+
// A synthesis that is waiting on the sidecar is still "speaking" as far as
|
|
464
|
+
// stop/isSpeaking are concerned, so a cancel lands exactly as it would on
|
|
465
|
+
// Piper's stdout stream.
|
|
466
|
+
let synthPending = false;
|
|
467
|
+
// One toast per availability state change, not one per utterance.
|
|
468
|
+
let chatterboxToastShown = false;
|
|
469
|
+
|
|
470
|
+
function chatterboxWarn(message) {
|
|
471
|
+
logger?.log?.("TTS", message, "warn");
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
function noteChatterboxFailure(message) {
|
|
475
|
+
logger?.log?.("TTS", message, "warn");
|
|
476
|
+
if (chatterboxToastShown) return;
|
|
477
|
+
chatterboxToastShown = true;
|
|
478
|
+
toast(`${message} - using Piper`);
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
// Recovering (or simply working) re-arms the toast, so a later outage is
|
|
482
|
+
// reported once again instead of staying silent for the rest of the session.
|
|
483
|
+
function noteChatterboxOk() {
|
|
484
|
+
chatterboxToastShown = false;
|
|
485
|
+
}
|
|
486
|
+
|
|
487
|
+
function getChatterboxClient() {
|
|
488
|
+
if (!chatterboxClient) {
|
|
489
|
+
chatterboxClient = createChatterbox({
|
|
490
|
+
variant: opts?.ttsChatterboxVariant,
|
|
491
|
+
python: opts?.ttsChatterboxPython,
|
|
492
|
+
voiceRef: opts?.ttsChatterboxVoiceRef,
|
|
493
|
+
logger,
|
|
494
|
+
});
|
|
495
|
+
}
|
|
496
|
+
return chatterboxClient;
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
// Start the sidecar lazily, on the first utterance that actually needs it.
|
|
500
|
+
// Never at plugin load: the first run downloads model weights, so eagerly
|
|
501
|
+
// loading would mean every opencode session pays for a TTS nobody asked for.
|
|
502
|
+
async function ensureChatterboxReady() {
|
|
503
|
+
if (chatterboxClient?.isRunning()) return true;
|
|
504
|
+
if (!chatterboxStarting) {
|
|
505
|
+
chatterboxStarting = getChatterboxClient()
|
|
506
|
+
.start()
|
|
507
|
+
.then((ok) => {
|
|
508
|
+
if (ok) {
|
|
509
|
+
noteChatterboxOk();
|
|
510
|
+
return true;
|
|
511
|
+
}
|
|
512
|
+
const err = chatterboxClient.getLastError?.();
|
|
513
|
+
noteChatterboxFailure(
|
|
514
|
+
`Chatterbox unavailable (${err?.code || "START_FAILED"}: ${err?.message || "unknown"})`,
|
|
515
|
+
);
|
|
516
|
+
return false;
|
|
517
|
+
})
|
|
518
|
+
.catch((err) => {
|
|
519
|
+
noteChatterboxFailure(`Chatterbox unavailable (${err.message})`);
|
|
520
|
+
return false;
|
|
521
|
+
})
|
|
522
|
+
.finally(() => {
|
|
523
|
+
chatterboxStarting = null;
|
|
524
|
+
});
|
|
525
|
+
}
|
|
526
|
+
return chatterboxStarting;
|
|
527
|
+
}
|
|
528
|
+
|
|
430
529
|
function speak(text) {
|
|
431
530
|
const myGen = speechGen;
|
|
432
531
|
if (!text) return Promise.resolve();
|
|
@@ -435,14 +534,23 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
435
534
|
|
|
436
535
|
killProcs();
|
|
437
536
|
|
|
537
|
+
const lang = detectLang(line);
|
|
538
|
+
const route = resolveTtsEngine(opts, lang, chatterboxWarn);
|
|
539
|
+
if (route.engine === "chatterbox") {
|
|
540
|
+
return speakViaChatterbox(line, lang, route.variant, myGen);
|
|
541
|
+
}
|
|
438
542
|
// One voice for the whole call (see detectLang) - no mid-utterance
|
|
439
543
|
// switching.
|
|
544
|
+
return speakViaPiper(line, myGen);
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
// Piper synthesis path, shared by the default route and by the Chatterbox
|
|
548
|
+
// fallback so both play identically.
|
|
549
|
+
function speakViaPiper(line, myGen = speechGen) {
|
|
440
550
|
const voiceModel = resolveVoiceOrWarn(line);
|
|
441
551
|
if (!voiceModel) return Promise.resolve();
|
|
442
|
-
// A cancel (stopSpeech) during normalization invalidates this speech.
|
|
443
552
|
if (myGen !== speechGen) return Promise.resolve();
|
|
444
553
|
logger?.log?.("TTS", `Speak requested chars=${line.length} voice=${voiceModel}`, "debug");
|
|
445
|
-
|
|
446
554
|
return new Promise((resolve) => {
|
|
447
555
|
spawnPlayback(voiceModel, resolve);
|
|
448
556
|
if (piperProc?.stdin && !piperProc.stdin.destroyed) {
|
|
@@ -452,6 +560,49 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
452
560
|
});
|
|
453
561
|
}
|
|
454
562
|
|
|
563
|
+
async function speakViaChatterbox(line, lang, variant, myGen) {
|
|
564
|
+
// Claim "speaking" for the WHOLE attempt, including the sidecar startup
|
|
565
|
+
// wait: set before the first await, so /tts-stop and the global cancel act
|
|
566
|
+
// on a Chatterbox utterance the instant it is requested.
|
|
567
|
+
synthPending = true;
|
|
568
|
+
try {
|
|
569
|
+
const ready = await ensureChatterboxReady();
|
|
570
|
+
// A cancel that landed while the sidecar was loading must not speak.
|
|
571
|
+
if (myGen !== speechGen) return;
|
|
572
|
+
if (!ready) return speakViaPiper(line, myGen);
|
|
573
|
+
|
|
574
|
+
let result;
|
|
575
|
+
try {
|
|
576
|
+
result = await getChatterboxClient().synthesize(line, lang);
|
|
577
|
+
} catch (err) {
|
|
578
|
+
// The client already returns typed errors; this only guards a thrown
|
|
579
|
+
// client bug so speech still falls back instead of rejecting.
|
|
580
|
+
result = { ok: false, code: "REQUEST_FAILED", message: err.message };
|
|
581
|
+
}
|
|
582
|
+
|
|
583
|
+
// Same generation guard as Piper: a stop during synthesis must prevent
|
|
584
|
+
// the audio from ever reaching sox.
|
|
585
|
+
if (myGen !== speechGen) return;
|
|
586
|
+
if (!result?.ok) {
|
|
587
|
+
noteChatterboxFailure(`Chatterbox synthesis failed (${result?.code}: ${result?.message})`);
|
|
588
|
+
return speakViaPiper(line, myGen);
|
|
589
|
+
}
|
|
590
|
+
|
|
591
|
+
noteChatterboxOk();
|
|
592
|
+
logger?.log?.(
|
|
593
|
+
"TTS",
|
|
594
|
+
`Speak via Chatterbox chars=${line.length} lang=${lang} variant=${variant} wavBytes=${result.wav.length}`,
|
|
595
|
+
"debug",
|
|
596
|
+
);
|
|
597
|
+
return new Promise((resolve) => {
|
|
598
|
+
spawnWavPlayback(result.wav, resolve);
|
|
599
|
+
});
|
|
600
|
+
} finally {
|
|
601
|
+
// Playback, if it started, is tracked by playProc from here on.
|
|
602
|
+
synthPending = false;
|
|
603
|
+
}
|
|
604
|
+
}
|
|
605
|
+
|
|
455
606
|
// ---- Session-prefixed announcements ----
|
|
456
607
|
|
|
457
608
|
async function speakWithSessionPrefix(sessionID, message, suffix) {
|
|
@@ -467,7 +618,10 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
467
618
|
// Always invalidate: a pending normalize handoff must not speak after
|
|
468
619
|
// the cancel even when nothing is currently playing.
|
|
469
620
|
speechGen += 1;
|
|
470
|
-
|
|
621
|
+
// A waiting Chatterbox synthesis counts as playing, so /tts-stop and the
|
|
622
|
+
// global cancel kill it exactly like Piper playback.
|
|
623
|
+
const wasPlaying = piperProc !== null || playProc !== null || synthPending;
|
|
624
|
+
synthPending = false;
|
|
471
625
|
killProcs();
|
|
472
626
|
return wasPlaying;
|
|
473
627
|
}
|
|
@@ -655,7 +809,9 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
655
809
|
speak: (text) => speak(text),
|
|
656
810
|
speakAssistantTurn,
|
|
657
811
|
stop: () => stopSpeech(),
|
|
658
|
-
|
|
812
|
+
// A Chatterbox utterance is "speaking" from the moment it is handed to the
|
|
813
|
+
// sidecar, not only once audio reaches sox.
|
|
814
|
+
isSpeaking: () => piperProc !== null || playProc !== null || synthPending,
|
|
659
815
|
setConversationActive: (v) => {
|
|
660
816
|
conversationActive = !!v;
|
|
661
817
|
},
|
|
@@ -664,6 +820,14 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
664
820
|
},
|
|
665
821
|
};
|
|
666
822
|
|
|
823
|
+
// The Chatterbox sidecar holds a multi-hundred-MB torch model; leaving it
|
|
824
|
+
// running past the session would strand that memory. Same lifecycle hook the
|
|
825
|
+
// STT and live-notes servers use.
|
|
826
|
+
api.lifecycle?.onDispose?.(() => {
|
|
827
|
+
chatterboxClient?.stop();
|
|
828
|
+
chatterboxClient = null;
|
|
829
|
+
});
|
|
830
|
+
|
|
667
831
|
// ---- Commands ----
|
|
668
832
|
|
|
669
833
|
const commands = [
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bojackduy/opencode-voice",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "Speech-to-text and text-to-speech for OpenCode. Record voice prompts with whisper-cpp, hear responses via Piper TTS, with LLM normalization through any OpenAI-compatible endpoint.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"opencode",
|