@bojackduy/opencode-voice 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/tts.js ADDED
@@ -0,0 +1,671 @@
1
+ // Text-to-speech: LLM normalization, Piper synthesis, sox playback.
2
+
3
+ import fs from "node:fs";
4
+ import path from "node:path";
5
+ import os from "node:os";
6
+ import { spawn } from "node:child_process";
7
+ import { getSessionTitle } from "./session.js";
8
+ import { clearProcessingToast, showProcessingToast, updateProcessingToast } from "./stt.js";
9
+
10
+ const VOICES_DIR = path.join(os.homedir(), ".local", "share", "piper-voices");
11
+
12
+ const TTS_VOICES = {
13
+ ryan: { label: "Ryan (high)", file: "en_US-ryan-high.onnx" },
14
+ bryce: { label: "Bryce (medium)", file: "en_US-bryce-medium.onnx" },
15
+ vi: { label: "Vais (vi_VN medium)", file: "vi_VN-vais1000-medium.onnx" },
16
+ vi_vivo: { label: "Vivos (vi_VN x_low)", file: "vi_VN-vivos-x_low.onnx" },
17
+ };
18
+ const DEFAULT_TTS_VOICE = "ryan";
19
+ // Vietnamese diacritics — used for auto vi/en detection
20
+ const VI_REGEX = /[àáảãạăắằẳẵặâấầẩẫậèéẻẽẹêếềểễệìíỉĩịòóỏõọôốồổỗộơớờởỡợùúủũụưứừửữựỳýỷỹỵđĐ]/;
21
+
22
+ const PIPER_RATE = 22050;
23
+ const PIPER_BITS = 16;
24
+ const PIPER_CHANNELS = 1;
25
+
26
+ // ---- Local (no-LLM) speech cleanup ----
27
+ // Deterministic markdown -> spoken-text transform. Zero network latency, so it
28
+ // backs streaming speech and ttsNormalizeMode "local". Less polished than the
29
+ // LLM narrator (no summarization), but instant.
30
+
31
+ function splitIdentifier(token) {
32
+ return token
33
+ .replace(/([a-z0-9])([A-Z])/g, "$1 $2")
34
+ .replace(/([A-Z]+)([A-Z][a-z])/g, "$1 $2")
35
+ .replace(/[_-]+/g, " ");
36
+ }
37
+
38
+ export function localSpeechCleanup(text) {
39
+ if (!text) return "";
40
+ let out = text;
41
+ out = out.replace(/```[\s\S]*?```/g, " code snippet ");
42
+ out = out.replace(/`([^`]+)`/g, (_, code) => ` ${splitIdentifier(code)} `);
43
+ out = out.replace(/!\[([^\]]*)\]\([^)]+\)/g, "$1");
44
+ out = out.replace(/\[([^\]]+)\]\(([^)]+)\)/g, "$1");
45
+ out = out.replace(/<[^>]+>/g, " ");
46
+ out = out.replace(/^#{1,6}\s+/gm, "");
47
+ out = out.replace(/^[>\s]*[-*+]\s+/gm, "");
48
+ out = out.replace(/(\*\*|__)(.*?)\1/g, "$2");
49
+ out = out.replace(/(^|\W)\*([^*\n]+)\*(?=\W|$)/g, "$1$2");
50
+ out = out
51
+ .split(/(\s+)/)
52
+ .map((tok) => {
53
+ if (/^\s+$/.test(tok) || !tok) return tok;
54
+ if (/^https?:\/\//i.test(tok)) {
55
+ const domain = tok.replace(/^https?:\/\/([^/]+).*$/i, "$1");
56
+ return domain.replace(/\./g, " dot ");
57
+ }
58
+ const base = tok.includes("/") ? tok.slice(tok.lastIndexOf("/") + 1) : tok;
59
+ if (/^\w[\w.-]*\.[a-z0-9]{1,4}$/i.test(base)) {
60
+ const dot = base.lastIndexOf(".");
61
+ return `${splitIdentifier(base.slice(0, dot))} dot ${base.slice(dot + 1)}`;
62
+ }
63
+ return splitIdentifier(tok);
64
+ })
65
+ .join("");
66
+ return out.replace(/\s+/g, " ").trim();
67
+ }
68
+
69
+ // Split streamed text into complete spoken sentences. Returns what is ready
70
+ // plus the trailing incomplete fragment to keep buffering. Sentences that
71
+ // look like code dumps (overlong, backticks) are skipped by the caller.
72
+
73
+ export function splitSpokenSentences(text) {
74
+ const flat = (text || "").replace(/\s+/g, " ");
75
+ const parts = flat.split(/(?<=[.!?…])\s+/);
76
+ if (parts.length === 0) return { sentences: [], rest: "" };
77
+ const lastComplete = /[.!?…]\s*$/.test(flat);
78
+ if (lastComplete) return { sentences: parts.filter(Boolean), rest: "" };
79
+ const rest = parts.pop() || "";
80
+ return { sentences: parts.filter(Boolean), rest };
81
+ }
82
+
83
+ export function isSpeakableSentence(sentence) {
84
+ const s = (sentence || "").trim();
85
+ if (!s || s.length > 400 || s.includes("`")) return false;
86
+ return true;
87
+ }
88
+
89
+ // Decide ONE language for the whole call (utterance or, during streaming,
90
+ // one completed sentence) and speak all of it with that single voice - no
91
+ // mid-utterance voice switching, since splicing separately synthesized clips
92
+ // sounds jarring (clicks/gaps between voices).
93
+ //
94
+ // Root-cause fix: the old rule was "any Vietnamese diacritic anywhere in the
95
+ // text -> vi voice". A single stray diacritic word in an English-dominant
96
+ // reply (e.g. a name) flipped the ENTIRE text to the Vietnamese voice, which
97
+ // then mispronounced every English word - and conversely, if the diacritics
98
+ // got normalized away, a Vietnamese reply flipped to the English voice. That
99
+ // is the "speaks Vietnamese with English (voice) and English with Vietnamese
100
+ // (voice)" bug. Fix: majority vote over words - only route to "vi" when a
101
+ // meaningful share of words actually carry Vietnamese diacritics, so one
102
+ // stray word can no longer flip the whole utterance.
103
+
104
+ const VI_LANG_WORD_THRESHOLD = 0.15;
105
+
106
+ export function detectLang(text) {
107
+ if (!text) return "en";
108
+ const words = text.split(/\s+/).filter(Boolean);
109
+ if (words.length === 0) return "en";
110
+ const viWords = words.filter((w) => VI_REGEX.test(w)).length;
111
+ return viWords / words.length >= VI_LANG_WORD_THRESHOLD ? "vi" : "en";
112
+ }
113
+
114
+ // ---- Conversation-mode hooks (wired by index.js) ----
115
+ // While voice conversation owns speaking, the TTS stop key pauses the loop
116
+ // instead of just silencing (which would auto-record again).
117
+
118
+ let conversationTtsHooks = { isActive: null, onStopKey: null };
119
+
120
+ export function __setConversationTtsHooks(hooks) {
121
+ conversationTtsHooks = { ...conversationTtsHooks, ...hooks };
122
+ }
123
+
124
+ function conversationTtsActive() {
125
+ try {
126
+ return conversationTtsHooks.isActive?.() === true;
127
+ } catch {
128
+ return false;
129
+ }
130
+ }
131
+
132
+ // ---- System prompts ----
133
+
134
+ const SYSTEM_AUTO = `You are a text-to-speech narrator for a coding assistant CLI. Your job is to convert the assistant's markdown output into natural spoken text that is useful and pleasant to listen to.
135
+
136
+ You have three modes depending on the content complexity:
137
+
138
+ 1. NARRATE - For simple explanations, short answers, and conversational responses. Convert to natural spoken text, normalizing code references for speech.
139
+ - camelCase/PascalCase identifiers: split into words (parseConfig -> "parse config")
140
+ - File paths: use just the filename (src/utils/helpers.ts -> "helpers dot ts")
141
+ - Short code snippets in backticks: read them naturally
142
+ - Keep the narrative flow intact
143
+
144
+ 2. SUMMARIZE - For responses with significant code blocks, multiple file changes, or complex technical details. Provide a brief spoken summary of what was done and tell the user to check the screen.
145
+ - Mention what was changed and why
146
+ - Do not try to describe code blocks verbatim
147
+ - End with something like "check the details on your screen" or "take a look at the output for the specifics"
148
+
149
+ 3. NOTIFY - For very short confirmations, status updates, or acknowledgments. Keep it to one brief sentence.
150
+
151
+ Choose the appropriate mode based on the content. Most responses with code blocks should use SUMMARIZE mode. Simple Q&A or short explanations use NARRATE. Build results, "done", confirmations use NOTIFY.
152
+
153
+ Output ONLY the spoken text. Nothing else. No mode labels. No commentary.`;
154
+
155
+ const SYSTEM_MANUAL = `You are a text-to-speech reader for a coding assistant. The user has explicitly requested this text be read aloud. Read the prose content faithfully and in detail.
156
+
157
+ Rules:
158
+ - Read all prose text naturally and completely
159
+ - Code identifiers: split camelCase/PascalCase/snake_case into words (parseConfig -> "parse config", my_variable -> "my variable")
160
+ - File paths: read just the filename with extension (src/utils/helpers.ts -> "helpers dot ts")
161
+ - Line references: keep as is ("line 42")
162
+ - URLs: say "a link" or just the domain name
163
+ - Code blocks: skip entirely, just say "code block" or "code snippet"
164
+ - Error codes: expand naturally (ECONNREFUSED -> "connection refused")
165
+ - Shell commands: read them naturally (npm test -> "npm test")
166
+ - List items: read each item
167
+ - Remove markdown formatting but preserve all the informational content
168
+ - Do NOT summarize. Do NOT say "check the screen". Read everything that is prose.
169
+ - Output ONLY the spoken text`;
170
+
171
+ // ---- Session helpers ----
172
+
173
+ async function getTurnAssistantText(client, api) {
174
+ const route = api.route.current;
175
+ if (route.name !== "session") return null;
176
+
177
+ const sessionID = route.params.sessionID;
178
+ const stateMessages = api.state.session.messages(sessionID);
179
+ if (!stateMessages || stateMessages.length === 0) return null;
180
+
181
+ const assistantIDs = [];
182
+ for (let i = stateMessages.length - 1; i >= 0; i--) {
183
+ if (stateMessages[i].role === "user") break;
184
+ if (stateMessages[i].role === "assistant") {
185
+ assistantIDs.unshift(stateMessages[i].id);
186
+ }
187
+ }
188
+ if (assistantIDs.length === 0) return null;
189
+
190
+ const allText = [];
191
+ for (const msgID of assistantIDs) {
192
+ try {
193
+ const fullMsg = await client.session
194
+ .message({ sessionID, messageID: msgID }, { throwOnError: true })
195
+ .then((r) => r.data);
196
+
197
+ const textParts = (fullMsg?.parts || []).filter((p) => p.type === "text");
198
+ const text = textParts
199
+ .map((p) => p.text || "")
200
+ .join("\n\n")
201
+ .trim();
202
+ if (text) allText.push(text);
203
+ } catch {
204
+ // Skip messages that fail to fetch
205
+ }
206
+ }
207
+
208
+ if (allText.length === 0) return null;
209
+
210
+ return {
211
+ lastMessageID: assistantIDs[assistantIDs.length - 1],
212
+ text: allText.join("\n\n"),
213
+ };
214
+ }
215
+
216
+ // ---- Public API for TUI plugin ----
217
+
218
+ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {}) {
219
+ if (deps.isConversationActive || deps.onConversationStopKey) {
220
+ __setConversationTtsHooks({
221
+ isActive: deps.isConversationActive,
222
+ onStopKey: deps.onConversationStopKey,
223
+ });
224
+ }
225
+ const client = api.client;
226
+ const systemAuto = prompts?.ttsAuto || SYSTEM_AUTO;
227
+ const systemManual = prompts?.ttsManual || SYSTEM_MANUAL;
228
+
229
+ function toast(message, variant = "info") {
230
+ api.ui.toast({ message, variant, duration: 3000 });
231
+ }
232
+
233
+ function getVoiceModel(text) {
234
+ const lang = detectLang(text);
235
+ if (lang === "vi") {
236
+ // prefer explicit vi voice from kv, else default vi
237
+ const viKey = kv.get("tts.voice.vi", "vi");
238
+ const viEntry = TTS_VOICES[viKey] || TTS_VOICES.vi;
239
+ const viPath = path.join(VOICES_DIR, viEntry.file);
240
+ if (fs.existsSync(viPath)) return viPath;
241
+ // fallback: any vi file present
242
+ for (const k of ["vi", "vi_vivo"]) {
243
+ const p = path.join(VOICES_DIR, TTS_VOICES[k].file);
244
+ if (fs.existsSync(p)) return p;
245
+ }
246
+ }
247
+ const voice = kv.get("tts.voice", DEFAULT_TTS_VOICE);
248
+ // if voice is a vi key but text is en, still respect en
249
+ const entry = TTS_VOICES[voice];
250
+ // if current kv is vi but text is en, use en default
251
+ if (entry && entry.file.includes("vi_VN") && lang === "en") {
252
+ return path.join(VOICES_DIR, TTS_VOICES[DEFAULT_TTS_VOICE].file);
253
+ }
254
+ return path.join(VOICES_DIR, (entry || TTS_VOICES[DEFAULT_TTS_VOICE]).file);
255
+ }
256
+
257
+ function piperOnPath() {
258
+ const pathDirs = (process.env.PATH || "").split(path.delimiter).filter(Boolean);
259
+ return pathDirs.some((dir) => fs.existsSync(path.join(dir, "piper")));
260
+ }
261
+
262
+ async function normalizeForSpeech(text, systemPrompt, maxTokens = 4096) {
263
+ logger?.log?.("TTS", `Normalizing speech chars=${text.length}`, "debug");
264
+ return complete({
265
+ system: systemPrompt,
266
+ prompt: `Convert for text-to-speech:\n\n${text}`,
267
+ config: { maxTokens },
268
+ });
269
+ }
270
+
271
+ function getVoiceRate(voicePath) {
272
+ try {
273
+ const j = JSON.parse(fs.readFileSync(`${voicePath}.json`, "utf-8"));
274
+ return j?.audio?.sample_rate || PIPER_RATE;
275
+ } catch {
276
+ return PIPER_RATE;
277
+ }
278
+ }
279
+
280
+ // ---- Audio pipeline ----
281
+
282
+ let piperProc = null;
283
+ let playProc = null;
284
+
285
+ function killProcs() {
286
+ if (piperProc) {
287
+ try {
288
+ piperProc.kill("SIGKILL");
289
+ } catch {}
290
+ piperProc = null;
291
+ }
292
+ if (playProc) {
293
+ try {
294
+ playProc.kill("SIGKILL");
295
+ } catch {}
296
+ playProc = null;
297
+ }
298
+ }
299
+
300
+ // Spawn piper -> play and wire them. Resolves onDone when playback ends.
301
+
302
+ function spawnPlayback(voiceModel, onDone) {
303
+ let piperStderr = "";
304
+ let playStderr = "";
305
+ const rate = getVoiceRate(voiceModel);
306
+ playProc = spawn(
307
+ "play",
308
+ [
309
+ "-t",
310
+ "raw",
311
+ "-r",
312
+ String(rate),
313
+ "-e",
314
+ "signed",
315
+ "-b",
316
+ String(PIPER_BITS),
317
+ "-c",
318
+ String(PIPER_CHANNELS),
319
+ "-q",
320
+ "-",
321
+ ],
322
+ { stdio: ["pipe", "ignore", "pipe"] },
323
+ );
324
+
325
+ piperProc = spawn("piper", ["-m", voiceModel, "--output_raw"], {
326
+ stdio: ["pipe", "pipe", "pipe"],
327
+ });
328
+
329
+ piperProc.stderr.on("data", (chunk) => {
330
+ piperStderr += chunk.toString();
331
+ });
332
+ playProc.stderr.on("data", (chunk) => {
333
+ playStderr += chunk.toString();
334
+ });
335
+
336
+ piperProc.stdout.on("data", (chunk) => {
337
+ if (playProc?.stdin && !playProc.stdin.destroyed) {
338
+ playProc.stdin.write(chunk);
339
+ }
340
+ });
341
+
342
+ piperProc.on("close", (code) => {
343
+ if (code !== 0 && code !== null) {
344
+ logger?.log?.("TTS", `piper exited code=${code} stderr=${piperStderr.trim()}`, "error");
345
+ }
346
+ if (playProc?.stdin && !playProc.stdin.destroyed) {
347
+ playProc.stdin.end();
348
+ }
349
+ });
350
+
351
+ playProc.on("close", (code) => {
352
+ if (code !== 0 && code !== null) {
353
+ logger?.log?.("TTS", `play exited code=${code} stderr=${playStderr.trim()}`, "error");
354
+ } else {
355
+ logger?.log?.("TTS", "playback finished", "debug");
356
+ }
357
+ piperProc = null;
358
+ playProc = null;
359
+ onDone();
360
+ });
361
+
362
+ piperProc.on("error", (err) => {
363
+ logger?.log?.("TTS", `piper error: ${err.message}`, "error");
364
+ killProcs();
365
+ onDone();
366
+ });
367
+ playProc.on("error", (err) => {
368
+ logger?.log?.("TTS", `play error: ${err.message}`, "error");
369
+ killProcs();
370
+ onDone();
371
+ });
372
+ }
373
+
374
+ function resolveVoiceOrWarn(line) {
375
+ const voiceModel = getVoiceModel(line);
376
+ if (!piperOnPath()) {
377
+ logger?.log?.("TTS", `Piper binary not found on PATH`, "warn");
378
+ toast(`Piper binary not found on PATH`, "warning");
379
+ return null;
380
+ }
381
+ if (!fs.existsSync(voiceModel)) {
382
+ logger?.log?.("TTS", `Voice model not found: ${voiceModel}`, "warn");
383
+ toast(`Voice model not found: ${voiceModel}`, "warning");
384
+ return null;
385
+ }
386
+ return voiceModel;
387
+ }
388
+
389
+ function speak(text) {
390
+ if (!text) return Promise.resolve();
391
+ const line = text.replace(/\n/g, " ").trim();
392
+ if (!line) return Promise.resolve();
393
+
394
+ killProcs();
395
+
396
+ // One voice for the whole call (see detectLang) - no mid-utterance
397
+ // switching.
398
+ const voiceModel = resolveVoiceOrWarn(line);
399
+ if (!voiceModel) return Promise.resolve();
400
+ logger?.log?.("TTS", `Speak requested chars=${line.length} voice=${voiceModel}`, "debug");
401
+
402
+ return new Promise((resolve) => {
403
+ spawnPlayback(voiceModel, resolve);
404
+ if (piperProc?.stdin && !piperProc.stdin.destroyed) {
405
+ piperProc.stdin.write(line + "\n");
406
+ piperProc.stdin.end();
407
+ }
408
+ });
409
+ }
410
+
411
+ // ---- Session-prefixed announcements ----
412
+
413
+ async function speakWithSessionPrefix(sessionID, message, suffix) {
414
+ const sessionTitle = await getSessionTitle(client, sessionID);
415
+ const parts = [];
416
+ if (sessionTitle) parts.push(`Session: ${sessionTitle}.`);
417
+ parts.push(message);
418
+ if (suffix) parts.push(suffix);
419
+ await speak(parts.join(" "));
420
+ }
421
+
422
+ function stopSpeech() {
423
+ const wasPlaying = piperProc !== null || playProc !== null;
424
+ killProcs();
425
+ return wasPlaying;
426
+ }
427
+
428
+ // ---- Auto mode ----
429
+
430
+ let lastSpokenMessageID = null;
431
+ let wasBusy = false;
432
+ // Set by the voice-conversation loop while it owns speaking, and by live
433
+ // notes while it owns the mic. Auto TTS stays out of the way (and consumes
434
+ // the busy flag) in both cases so replies are not spoken twice, and so a
435
+ // live-notes recording session is never interrupted by an unrelated auto
436
+ // TTS reply.
437
+ let conversationActive = false;
438
+ let liveNotesActive = false;
439
+
440
+ api.event.on("session.status", (event) => {
441
+ if (event.properties?.status?.type === "busy") wasBusy = true;
442
+ });
443
+
444
+ api.event.on("session.idle", async (event) => {
445
+ if (conversationActive || liveNotesActive) {
446
+ wasBusy = false;
447
+ return;
448
+ }
449
+ if (kv.get("tts.mode", "off") !== "on") return;
450
+ if (!wasBusy) return;
451
+ wasBusy = false;
452
+
453
+ const sessionID = event.properties?.sessionID;
454
+ const result = await getTurnAssistantText(client, api);
455
+ if (!result || !result.text) return;
456
+
457
+ if (result.lastMessageID === lastSpokenMessageID) return;
458
+ lastSpokenMessageID = result.lastMessageID;
459
+
460
+ showProcessingToast("Normalizing response...");
461
+ const llmResult = await normalizeOrLocal(result.text, systemAuto, 4096);
462
+ if (!llmResult.text) {
463
+ clearProcessingToast();
464
+ logger?.log?.("TTS", `Auto normalization failed: ${llmResult.error}`, "warn");
465
+ toast(`TTS normalization failed: ${llmResult.error}`, "warning");
466
+ return;
467
+ }
468
+
469
+ logger?.log?.("TTS", `Auto normalization succeeded chars=${llmResult.text.length}`, "debug");
470
+ clearProcessingToast();
471
+ await speakWithSessionPrefix(sessionID, llmResult.text, "Ready for your input.");
472
+ });
473
+
474
+ api.event.on("permission.asked", async (event) => {
475
+ if (conversationActive || liveNotesActive) return;
476
+ if (kv.get("tts.mode", "off") !== "on") return;
477
+ await speakWithSessionPrefix(
478
+ event.properties?.sessionID,
479
+ "Permission requested. Please check your screen.",
480
+ );
481
+ });
482
+
483
+ api.event.on("question.asked", async (event) => {
484
+ if (conversationActive || liveNotesActive) return;
485
+ if (kv.get("tts.mode", "off") !== "on") return;
486
+ await speakWithSessionPrefix(
487
+ event.properties?.sessionID,
488
+ "A question needs your answer. Please check your screen.",
489
+ );
490
+ });
491
+
492
+ // ---- Manual mode ----
493
+
494
+ async function speakLastResponse() {
495
+ const result = await getTurnAssistantText(client, api);
496
+ if (!result || !result.text) {
497
+ toast("No assistant response to speak", "warning");
498
+ return;
499
+ }
500
+
501
+ showProcessingToast("Normalizing response...");
502
+ const llmResult = await normalizeOrLocal(result.text, systemManual, 4096);
503
+ if (!llmResult.text) {
504
+ clearProcessingToast();
505
+ logger?.log?.("TTS", `Manual normalization failed: ${llmResult.error}`, "warn");
506
+ toast(`TTS normalization failed: ${llmResult.error}`, "warning");
507
+ return;
508
+ }
509
+
510
+ logger?.log?.("TTS", `Manual normalization succeeded chars=${llmResult.text.length}`, "debug");
511
+ updateProcessingToast("Speaking...");
512
+ await speak(llmResult.text);
513
+ clearProcessingToast();
514
+ }
515
+
516
+ // Speak the current assistant turn for the voice-conversation loop. Uses the
517
+ // auto (narrate/summarize) prompt and no session prefix - the loop already
518
+ // announces its own state via toasts.
519
+
520
+ // ttsNormalizeMode "local" skips the LLM entirely (instant, less polished).
521
+ const ttsLocal = opts?.ttsNormalizeMode === "local";
522
+
523
+ async function normalizeOrLocal(text, systemPrompt, maxTokens) {
524
+ if (ttsLocal) return { text: localSpeechCleanup(text) };
525
+ return normalizeForSpeech(text, systemPrompt, maxTokens);
526
+ }
527
+
528
+ async function speakAssistantTurn() {
529
+ const tFetch = Date.now();
530
+ const result = await getTurnAssistantText(client, api);
531
+ if (!result || !result.text) {
532
+ logger?.log?.("TTS", "Conversation: no assistant text to speak", "warn");
533
+ return { spoken: false };
534
+ }
535
+ const fetchMs = Date.now() - tFetch;
536
+
537
+ showProcessingToast("Normalizing response...");
538
+ const tNormalize = Date.now();
539
+ // Auto replies are narrated/summarized, so a tight cap is safe here (the
540
+ // manual read-aloud keeps the full 4096).
541
+ const llmResult = await normalizeOrLocal(result.text, systemAuto, 2048);
542
+ const normalizeMs = Date.now() - tNormalize;
543
+ if (!llmResult.text) {
544
+ clearProcessingToast();
545
+ logger?.log?.("TTS", `Conversation normalization failed: ${llmResult.error}`, "warn");
546
+ toast(`TTS normalization failed: ${llmResult.error}`, "warning");
547
+ return { spoken: false, error: llmResult.error };
548
+ }
549
+
550
+ updateProcessingToast("Speaking...");
551
+ const tSpeak = Date.now();
552
+ await speak(llmResult.text);
553
+ const speakMs = Date.now() - tSpeak;
554
+ clearProcessingToast();
555
+ logger?.log?.(
556
+ "TTS",
557
+ `Conversation timings fetchMs=${fetchMs} normalizeMs=${normalizeMs} speakMs=${speakMs} replyChars=${result.text.length}`,
558
+ "debug",
559
+ );
560
+ return { spoken: true };
561
+ }
562
+
563
+ const DEFAULT_KEYBINDS = {
564
+ "tts.speak-last": "<leader>]",
565
+ "tts.stop": "<leader>;",
566
+ };
567
+ function kb(value) {
568
+ const kb = opts?.keybinds;
569
+ if (!kb || typeof kb !== "object" || Array.isArray(kb)) return DEFAULT_KEYBINDS[value];
570
+ if (!Object.prototype.hasOwnProperty.call(kb, value)) return DEFAULT_KEYBINDS[value];
571
+ const v = kb[value];
572
+ if (!v || v === "none") return undefined;
573
+ return v;
574
+ }
575
+
576
+ const controller = {
577
+ speak: (text) => speak(text),
578
+ speakAssistantTurn,
579
+ stop: () => stopSpeech(),
580
+ isSpeaking: () => piperProc !== null || playProc !== null,
581
+ setConversationActive: (v) => {
582
+ conversationActive = !!v;
583
+ },
584
+ setLiveNotesActive: (v) => {
585
+ liveNotesActive = !!v;
586
+ },
587
+ };
588
+
589
+ // ---- Commands ----
590
+
591
+ const commands = [
592
+ {
593
+ title: "TTS: speak last response",
594
+ value: "tts.speak-last",
595
+ category: "opencode-voice",
596
+ description: "Read the last assistant response aloud (detailed)",
597
+ ...(kb("tts.speak-last") ? { keybind: kb("tts.speak-last") } : {}),
598
+ slash: { name: "tts-speak" },
599
+ onSelect() {
600
+ speakLastResponse();
601
+ },
602
+ },
603
+ {
604
+ title: "TTS: toggle",
605
+ value: "tts.mode",
606
+ category: "opencode-voice",
607
+ description: "Toggle auto text-to-speech on/off",
608
+ ...(kb("tts.mode") ? { keybind: kb("tts.mode") } : {}),
609
+ slash: { name: "tts-mode" },
610
+ onSelect() {
611
+ const current = kv.get("tts.mode", "off");
612
+ const next = current === "on" ? "off" : "on";
613
+ kv.set("tts.mode", next);
614
+ if (next === "off") stopSpeech();
615
+ if (next === "on") {
616
+ const enVoice =
617
+ TTS_VOICES[kv.get("tts.voice", DEFAULT_TTS_VOICE)] || TTS_VOICES[DEFAULT_TTS_VOICE];
618
+ const viPath = path.join(VOICES_DIR, TTS_VOICES.vi.file);
619
+ const hasVi = fs.existsSync(viPath);
620
+ toast(
621
+ hasVi
622
+ ? `TTS on (auto: ${enVoice.label} / ${TTS_VOICES.vi.label})`
623
+ : `TTS on (${enVoice.label})`,
624
+ );
625
+ } else {
626
+ toast("TTS off");
627
+ }
628
+ },
629
+ },
630
+ {
631
+ title: "TTS: stop playback",
632
+ value: "tts.stop",
633
+ category: "opencode-voice",
634
+ description: "Stop current TTS playback",
635
+ ...(kb("tts.stop") ? { keybind: kb("tts.stop") } : {}),
636
+ slash: { name: "tts-stop" },
637
+ onSelect() {
638
+ if (conversationTtsActive() && conversationTtsHooks.onStopKey?.()) return;
639
+ if (stopSpeech()) toast("TTS stopped");
640
+ },
641
+ },
642
+ {
643
+ title: "TTS: select voice",
644
+ value: "tts.voice",
645
+ category: "opencode-voice",
646
+ description: "Choose TTS voice",
647
+ slash: { name: "tts-voice" },
648
+ onSelect() {
649
+ const current = kv.get("tts.voice", DEFAULT_TTS_VOICE);
650
+ api.ui.dialog.replace(() =>
651
+ api.ui.DialogSelect({
652
+ title: "Select voice (auto vi/en)",
653
+ current,
654
+ options: Object.entries(TTS_VOICES).map(([key, v]) => ({
655
+ title: v.label,
656
+ value: key,
657
+ onSelect() {
658
+ if (key.startsWith("vi")) kv.set("tts.voice.vi", key);
659
+ else kv.set("tts.voice", key);
660
+ toast(`Voice: ${v.label} (auto vi/en)`);
661
+ api.ui.dialog.clear();
662
+ },
663
+ })),
664
+ }),
665
+ );
666
+ },
667
+ },
668
+ ];
669
+
670
+ return { commands, controller };
671
+ }