@bojackduy/opencode-voice 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,528 @@
1
+ // Voice conversation mode: hands-free talk loop with OpenCode.
2
+ //
3
+ // One key drives the whole loop; its meaning depends on the state shown in
4
+ // the toast:
5
+ //
6
+ // record -> transcribe -> normalize -> submit -> wait reply -> speak -> record ...
7
+ //
8
+ // - recording + key: finish the turn and submit
9
+ // - speaking + key: pause speech (press again to record the next turn)
10
+ // - paused + key: record again
11
+ // - waiting/processing + key: exit the mode
12
+ //
13
+ // Saying a stop phrase ("stop", "dừng lại", ...) ends the mode without
14
+ // submitting. Empty or failed turns pause instead of auto-recording, so the
15
+ // key never surprises.
16
+ //
17
+ // Latency: while waiting, assistant text deltas are spoken sentence by
18
+ // sentence (local cleanup, no LLM), so the answer starts before the turn
19
+ // completes. Replies with nothing streamable fall back to the full
20
+ // LLM-narrated speak.
21
+
22
+ import { clearProcessingToast, showProcessingToast } from "./stt.js";
23
+ import { isSpeakableSentence, localSpeechCleanup, splitSpokenSentences } from "./tts.js";
24
+
25
+ export const DEFAULT_STOP_PHRASES = [
26
+ "stop",
27
+ "exit",
28
+ "quit",
29
+ "stop conversation",
30
+ "exit conversation",
31
+ "end conversation",
32
+ "goodbye",
33
+ "bye",
34
+ "dừng lại",
35
+ "dừng",
36
+ "kết thúc",
37
+ "kết thúc hội thoại",
38
+ "thoát",
39
+ "tạm biệt",
40
+ ];
41
+
42
+ export function normalizePhrase(text) {
43
+ return (text || "")
44
+ .trim()
45
+ .toLowerCase()
46
+ .replace(/[.!…?]+$/u, "")
47
+ .replace(/\s+/g, " ");
48
+ }
49
+
50
+ // Small vocabulary so stuttered/short stop commands ("stop stop", "please stop
51
+ // the conversation now", "dừng lại đi") still match, while real sentences
52
+ // ("stop the server", "do not stop") fall through and get submitted.
53
+
54
+ const STOP_VOCAB = new Set(
55
+ (
56
+ "stop stops stopping please now exit quit quits quitting end ends ending finish " +
57
+ "conversation voice chat goodbye bye good the this that it turn off mode " +
58
+ "dừng lại kết thúc hội thoại thoát tạm biệt đi ra ngừng ngưng thôi"
59
+ ).split(" "),
60
+ );
61
+ const MAX_STOP_WORDS = 5;
62
+
63
+ export function matchesStopPhrase(text, phrases) {
64
+ const normalized = normalizePhrase(text);
65
+ if (!normalized) return false;
66
+ if ((phrases ?? DEFAULT_STOP_PHRASES).some((p) => normalizePhrase(p) === normalized)) return true;
67
+ // Lenient vocab match only applies to the default list; a custom list takes
68
+ // full control with exact matching.
69
+ if (phrases !== undefined) return false;
70
+ const words = normalized
71
+ .replace(/[^\p{L}\p{N}\s]/gu, "")
72
+ .split(/\s+/)
73
+ .filter(Boolean);
74
+ return (
75
+ words.length > 0 && words.length <= MAX_STOP_WORDS && words.every((w) => STOP_VOCAB.has(w))
76
+ );
77
+ }
78
+
79
+ function delay(ms) {
80
+ return new Promise((resolve) => setTimeout(resolve, ms));
81
+ }
82
+
83
+ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesActive }) {
84
+ const maxTurns = Number(opts?.conversationMaxTurns) > 0 ? Number(opts.conversationMaxTurns) : 50;
85
+ const replyTimeoutMs =
86
+ Number(opts?.conversationTimeoutMs) > 0 ? Number(opts.conversationTimeoutMs) : 300000;
87
+ const restartDelayMs =
88
+ Number(opts?.conversationRestartDelayMs) >= 0 ? Number(opts.conversationRestartDelayMs) : 350;
89
+ // Keep undefined when the user supplied nothing so matchesStopPhrase
90
+ // takes the lenient default-vocabulary path ("stop stop", "dừng lại đi").
91
+ // Passing DEFAULT_STOP_PHRASES explicitly would disable it.
92
+ const stopPhrases = Array.isArray(opts?.conversationStopPhrases)
93
+ ? opts.conversationStopPhrases
94
+ : undefined;
95
+
96
+ let active = false;
97
+ let phase = "idle"; // idle | recording | processing | waiting | speaking | paused
98
+
99
+ // Reply streaming state: text deltas heard while the agent works are spoken
100
+ // sentence by sentence through a serial queue, so the answer starts before
101
+ // the turn completes and every sentence gets its own vi/en voice.
102
+
103
+ let streamBuffer = "";
104
+ let streamSpokenChars = 0;
105
+ let streamUnsubs = [];
106
+ let speakTail = Promise.resolve();
107
+ let speakGen = 0;
108
+
109
+ function queueSpeak(sentence) {
110
+ const g = speakGen;
111
+ speakTail = speakTail.then(() => {
112
+ // Speaking must also drain: finishTurnFlow flips to speaking before
113
+ // flushing the tail, so a waiting-only guard would skip queued
114
+ // sentences/final partials under real TTS latency. speakGen still
115
+ // invalidates on pause/stop.
116
+ if (g !== speakGen || !active || (phase !== "waiting" && phase !== "speaking")) return null;
117
+ return tts.speak(sentence);
118
+ });
119
+ return speakTail;
120
+ }
121
+ let turn = 0;
122
+ let consecErrors = 0;
123
+ let generation = 0;
124
+ let waitCancel = null;
125
+
126
+ function toast(message, variant = "info") {
127
+ api.ui.toast({ message, variant, duration: 3000 });
128
+ }
129
+
130
+ function currentSessionID() {
131
+ const route = api?.route?.current;
132
+ return route?.name === "session" ? route?.params?.sessionID : null;
133
+ }
134
+
135
+ function isActive() {
136
+ return active;
137
+ }
138
+
139
+ function beginTurn() {
140
+ if (!active) return;
141
+ if (turn >= maxTurns) {
142
+ stop("done");
143
+ toast(`Conversation off - reached ${maxTurns} turns`);
144
+ return;
145
+ }
146
+ if (stt.isRecording() || stt.isProcessing() || stt.isStreaming?.()) return;
147
+ turn += 1;
148
+ phase = "recording";
149
+ logger?.log("VOICE", `Conversation turn ${turn} listening`, "debug");
150
+ if (!stt.start()) {
151
+ phase = "idle";
152
+ stop("busy");
153
+ toast("STT busy, conversation off", "warning");
154
+ }
155
+ }
156
+
157
+ function endReplyStream() {
158
+ for (const unsub of streamUnsubs) {
159
+ try {
160
+ unsub();
161
+ } catch {}
162
+ }
163
+ streamUnsubs = [];
164
+ }
165
+
166
+ // Kill streamed audio without touching the loop state. Queued sentences
167
+ // are invalidated so they never resume after a stop. The caller decides
168
+ // what happens next (legacy speak, notification, or exit).
169
+
170
+ function discardStreamAudio() {
171
+ endReplyStream();
172
+ speakGen += 1;
173
+ tts.stop();
174
+ }
175
+
176
+ function stop(reason) {
177
+ if (!active) return;
178
+ active = false;
179
+ generation += 1;
180
+ phase = "idle";
181
+ if (waitCancel) {
182
+ waitCancel("stopped");
183
+ waitCancel = null;
184
+ }
185
+ endReplyStream();
186
+ speakGen += 1;
187
+ stt.setStopHint(null);
188
+ if (stt.isRecording() || stt.isProcessing()) stt.cancel();
189
+ clearProcessingToast();
190
+ tts.stop();
191
+ tts.setConversationActive(false);
192
+ logger?.log("VOICE", `Conversation stopped reason=${reason}`, "debug");
193
+ }
194
+
195
+ function startReplyStream(sessionID, turnStart) {
196
+ streamBuffer = "";
197
+ streamSpokenChars = 0;
198
+ const unsub = api.event.on("message.part.delta", (event) => {
199
+ try {
200
+ if (!active || phase !== "waiting") return;
201
+ const props = event.properties || {};
202
+ if (sessionID && props.sessionID && props.sessionID !== sessionID) return;
203
+ if (typeof props.delta !== "string" || !props.delta) return;
204
+ if (props.field && props.field !== "text") return;
205
+ const lookupID = props.sessionID || sessionID;
206
+ const messages = api?.state?.session?.messages?.(lookupID) || [];
207
+ const msg = messages.find((m) => m?.id === props.messageID);
208
+ if (!msg || msg.role !== "assistant") return;
209
+ if ((msg.time?.created || 0) < turnStart) return;
210
+ const parts = api?.state?.part?.(props.messageID) || [];
211
+ const part = parts.find((p) => p?.id === props.partID);
212
+ if (part && part.type !== "text") return;
213
+ streamBuffer += props.delta;
214
+ flushStream(false);
215
+ } catch {}
216
+ });
217
+ streamUnsubs.push(unsub);
218
+ }
219
+
220
+ function flushStream(final = false) {
221
+ const { sentences, rest } = splitSpokenSentences(streamBuffer);
222
+ const ready = final && rest.trim() ? [...sentences, rest] : sentences;
223
+ if (ready.length > 0 && streamSpokenChars === 0) {
224
+ showProcessingToast("Speaking...");
225
+ }
226
+ for (const sentence of ready) {
227
+ if (!isSpeakableSentence(sentence)) continue;
228
+ const cleaned = localSpeechCleanup(sentence);
229
+ if (!cleaned) continue;
230
+ queueSpeak(cleaned);
231
+ streamSpokenChars += cleaned.length;
232
+ }
233
+ streamBuffer = final ? "" : rest;
234
+ }
235
+
236
+ // Wait for the submitted turn to finish: idle, a permission/question gate,
237
+ // or timeout. Resolves early when stop() cancels the wait.
238
+
239
+ function waitForReply(sessionID) {
240
+ let settled = false;
241
+ let unsubs = [];
242
+ let timer = null;
243
+
244
+ let cancelFn = null;
245
+ const done = (outcome) => {
246
+ if (settled) return null;
247
+ settled = true;
248
+ if (timer) clearTimeout(timer);
249
+ for (const unsub of unsubs) {
250
+ try {
251
+ unsub();
252
+ } catch {}
253
+ }
254
+ unsubs = [];
255
+ if (waitCancel === cancelFn) waitCancel = null;
256
+ return outcome;
257
+ };
258
+
259
+ const promise = new Promise((resolve) => {
260
+ const finish = (outcome) => {
261
+ const result = done(outcome);
262
+ if (result !== null) resolve(result);
263
+ };
264
+ cancelFn = (outcome) => finish(outcome || "stopped");
265
+ waitCancel = cancelFn;
266
+
267
+ const forSession = (props) =>
268
+ !sessionID || !props?.sessionID || props.sessionID === sessionID;
269
+
270
+ unsubs = [
271
+ api.event.on("session.idle", (event) => {
272
+ if (forSession(event.properties)) finish("idle");
273
+ }),
274
+ api.event.on("session.status", (event) => {
275
+ if (event.properties?.status?.type === "idle" && forSession(event.properties)) {
276
+ finish("idle");
277
+ }
278
+ }),
279
+ api.event.on("permission.asked", (event) => {
280
+ if (forSession(event.properties)) finish("permission");
281
+ }),
282
+ api.event.on("question.asked", (event) => {
283
+ if (forSession(event.properties)) finish("question");
284
+ }),
285
+ ];
286
+ timer = setTimeout(() => finish("timeout"), replyTimeoutMs);
287
+ });
288
+
289
+ return promise;
290
+ }
291
+
292
+ async function finishTurnFlow() {
293
+ if (!active || phase !== "recording") return;
294
+ const myGen = generation;
295
+ phase = "processing";
296
+
297
+ const res = await stt.transcribeTurn();
298
+ if (!active || myGen !== generation) return;
299
+
300
+ if (res.error) {
301
+ consecErrors += 1;
302
+ logger?.log("VOICE", `Conversation turn error consec=${consecErrors}`, "warn");
303
+ if (consecErrors >= 2) {
304
+ stop("error");
305
+ return;
306
+ }
307
+ pauseForRetry("Paused - press the key to retry");
308
+ return;
309
+ }
310
+ if (!res.text) {
311
+ consecErrors = 0;
312
+ pauseForRetry("No speech heard - press the key to try again");
313
+ return;
314
+ }
315
+ consecErrors = 0;
316
+
317
+ if (matchesStopPhrase(res.text, stopPhrases)) {
318
+ stt.discard();
319
+ stop("stop phrase");
320
+ toast("Conversation off");
321
+ return;
322
+ }
323
+
324
+ const submitted = await stt.submitTurnText(res.text);
325
+ if (!active || myGen !== generation) return;
326
+ if (submitted.error) {
327
+ stop("submit failed");
328
+ return;
329
+ }
330
+
331
+ phase = "waiting";
332
+ const sessionID = currentSessionID();
333
+ const turnStart = Date.now();
334
+ showProcessingToast("Waiting for reply...");
335
+ startReplyStream(sessionID, turnStart);
336
+ const tWait = Date.now();
337
+ const outcome = await waitForReply(sessionID);
338
+ const waitMs = Date.now() - tWait;
339
+ endReplyStream();
340
+ logger?.log(
341
+ "VOICE",
342
+ `Conversation wait outcome=${outcome} waitMs=${waitMs} streamedChars=${streamSpokenChars}`,
343
+ "debug",
344
+ );
345
+ if (!active || myGen !== generation) return;
346
+
347
+ if (outcome === "timeout") {
348
+ discardStreamAudio();
349
+ stop("timeout");
350
+ toast("No reply in time, conversation off", "warning");
351
+ return;
352
+ }
353
+ if (outcome === "permission" || outcome === "question") {
354
+ discardStreamAudio();
355
+ phase = "speaking";
356
+ showProcessingToast("Speaking...");
357
+ await tts.speak(
358
+ outcome === "permission"
359
+ ? "Permission requested. Please check your screen."
360
+ : "A question needs your answer. Please check your screen.",
361
+ );
362
+ clearProcessingToast();
363
+ if (!active || myGen !== generation || phase !== "speaking") return;
364
+ beginTurn();
365
+ return;
366
+ }
367
+ if (outcome === "stopped") return;
368
+
369
+ // Streamed path: sentences already queued; flush the tail and let the
370
+ // audio finish before listening again. Fall back to the full
371
+ // LLM-narrated speak only when nothing streamable arrived.
372
+
373
+ if (streamSpokenChars > 0) {
374
+ phase = "speaking";
375
+ flushStream(true);
376
+ await speakTail;
377
+ clearProcessingToast();
378
+ if (!active || myGen !== generation || phase !== "speaking") return;
379
+ await delay(restartDelayMs);
380
+ if (!active || myGen !== generation || phase !== "speaking") return;
381
+ beginTurn();
382
+ return;
383
+ }
384
+
385
+ phase = "speaking";
386
+ const spoken = await tts.speakAssistantTurn();
387
+ if (!active || myGen !== generation || phase !== "speaking") return;
388
+ if (!spoken.spoken) {
389
+ toast("No reply to speak, listening again", "warning");
390
+ }
391
+ await delay(restartDelayMs);
392
+ if (!active || myGen !== generation || phase !== "speaking") return;
393
+ beginTurn();
394
+ }
395
+
396
+ // Pause instead of auto-recording so an empty or failed turn never feels
397
+ // like the keypress was ignored.
398
+
399
+ function pauseForRetry(message) {
400
+ if (!active) return;
401
+ phase = "paused";
402
+ if (message) toast(message, "warning");
403
+ }
404
+
405
+ // Stop speech and hold the mode. The next keypress records again.
406
+
407
+ function pauseSpeech() {
408
+ if (!active || phase !== "speaking") return;
409
+ logger?.log("VOICE", "Conversation speech paused", "debug");
410
+ tts.stop();
411
+ phase = "paused";
412
+ toast("Paused - press the key to speak");
413
+ }
414
+
415
+ // Called by the TTS stop command while the mode is on. Returns true when the
416
+ // keypress was consumed (speech paused), so TTS skips its own handling.
417
+
418
+ function onTtsStop() {
419
+ if (!active) return false;
420
+ if (phase === "speaking") {
421
+ pauseSpeech();
422
+ return true;
423
+ }
424
+ // Mid-stream the agent is still working and cannot pause, so the stop key
425
+ // exits the mode instead of leaving a half-heard reply behind.
426
+ if (phase === "waiting" && tts.isSpeaking?.()) {
427
+ stop("cancelled");
428
+ toast("Conversation off");
429
+ return true;
430
+ }
431
+ return false;
432
+ }
433
+
434
+ function onKey(source) {
435
+ if (!active) {
436
+ start();
437
+ return;
438
+ }
439
+ if (phase === "recording") {
440
+ finishTurnFlow();
441
+ return;
442
+ }
443
+ if (phase === "speaking") {
444
+ pauseSpeech();
445
+ return;
446
+ }
447
+ if (phase === "paused") {
448
+ beginTurn();
449
+ return;
450
+ }
451
+ // processing | waiting | idle
452
+ if (source === "toggle") {
453
+ stop("cancelled");
454
+ toast("Conversation off");
455
+ } else {
456
+ toast("Conversation busy, please wait...", "warning");
457
+ }
458
+ }
459
+
460
+ function start() {
461
+ if (active) return;
462
+ if (isLiveNotesActive?.()) {
463
+ toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
464
+ return;
465
+ }
466
+ if (stt.isRecording() || stt.isProcessing() || stt.isStreaming?.()) {
467
+ toast("STT busy, try again shortly", "warning");
468
+ return;
469
+ }
470
+ active = true;
471
+ generation += 1;
472
+ turn = 0;
473
+ consecErrors = 0;
474
+ phase = "idle";
475
+ tts.setConversationActive(true);
476
+ stt.setStopHint("<leader>v");
477
+ logger?.log("VOICE", "Conversation started", "debug");
478
+ toast("Conversation on - one key: send, pause speech, resume");
479
+ beginTurn();
480
+ }
481
+
482
+ api.lifecycle?.onDispose?.(() => stop("dispose"));
483
+
484
+ const DEFAULT_KEYBINDS = {
485
+ "voice.conversation": "<leader>v",
486
+ };
487
+ function kb(value) {
488
+ const overrides = opts?.keybinds;
489
+ if (!overrides || typeof overrides !== "object" || Array.isArray(overrides)) {
490
+ return DEFAULT_KEYBINDS[value];
491
+ }
492
+ if (!Object.prototype.hasOwnProperty.call(overrides, value)) return DEFAULT_KEYBINDS[value];
493
+ const v = overrides[value];
494
+ if (!v || v === "none") return undefined;
495
+ return v;
496
+ }
497
+
498
+ const commands = [
499
+ {
500
+ title: "Voice: conversation mode",
501
+ value: "voice.conversation",
502
+ category: "opencode-voice",
503
+ description: "Toggle voice conversation (one key: send, pause speech, resume, exit)",
504
+ ...(kb("voice.conversation") ? { keybind: kb("voice.conversation") } : {}),
505
+ slash: { name: "voice-conversation" },
506
+ onSelect() {
507
+ onKey("toggle");
508
+ },
509
+ },
510
+ {
511
+ title: "Voice: stop conversation",
512
+ value: "voice.conversation-stop",
513
+ category: "opencode-voice",
514
+ description: "Exit voice conversation mode",
515
+ slash: { name: "voice-conversation-stop" },
516
+ onSelect() {
517
+ if (active) {
518
+ stop("command");
519
+ toast("Conversation off");
520
+ }
521
+ },
522
+ },
523
+ ];
524
+
525
+ const controller = { isActive, onKey, onTtsStop, start, stop };
526
+
527
+ return { commands, controller };
528
+ }