@bojackduy/opencode-voice 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +729 -0
- package/index.js +124 -0
- package/lib/audio-chunker.js +231 -0
- package/lib/audio-enhance.js +172 -0
- package/lib/conversation.js +528 -0
- package/lib/live-notes.js +622 -0
- package/lib/llm-client.js +304 -0
- package/lib/logger.js +18 -0
- package/lib/notes-writer.js +224 -0
- package/lib/session.js +102 -0
- package/lib/streaming-editor.js +308 -0
- package/lib/streaming-stt.js +1322 -0
- package/lib/streaming-transcript.js +236 -0
- package/lib/stt.js +2339 -0
- package/lib/tts.js +671 -0
- package/lib/voice-model.js +122 -0
- package/lib/whisper-server.js +471 -0
- package/package.json +47 -0
|
@@ -0,0 +1,528 @@
|
|
|
1
|
+
// Voice conversation mode: hands-free talk loop with OpenCode.
|
|
2
|
+
//
|
|
3
|
+
// One key drives the whole loop; its meaning depends on the state shown in
|
|
4
|
+
// the toast:
|
|
5
|
+
//
|
|
6
|
+
// record -> transcribe -> normalize -> submit -> wait reply -> speak -> record ...
|
|
7
|
+
//
|
|
8
|
+
// - recording + key: finish the turn and submit
|
|
9
|
+
// - speaking + key: pause speech (press again to record the next turn)
|
|
10
|
+
// - paused + key: record again
|
|
11
|
+
// - waiting/processing + key: exit the mode
|
|
12
|
+
//
|
|
13
|
+
// Saying a stop phrase ("stop", "dừng lại", ...) ends the mode without
|
|
14
|
+
// submitting. Empty or failed turns pause instead of auto-recording, so the
|
|
15
|
+
// key never surprises.
|
|
16
|
+
//
|
|
17
|
+
// Latency: while waiting, assistant text deltas are spoken sentence by
|
|
18
|
+
// sentence (local cleanup, no LLM), so the answer starts before the turn
|
|
19
|
+
// completes. Replies with nothing streamable fall back to the full
|
|
20
|
+
// LLM-narrated speak.
|
|
21
|
+
|
|
22
|
+
import { clearProcessingToast, showProcessingToast } from "./stt.js";
|
|
23
|
+
import { isSpeakableSentence, localSpeechCleanup, splitSpokenSentences } from "./tts.js";
|
|
24
|
+
|
|
25
|
+
export const DEFAULT_STOP_PHRASES = [
|
|
26
|
+
"stop",
|
|
27
|
+
"exit",
|
|
28
|
+
"quit",
|
|
29
|
+
"stop conversation",
|
|
30
|
+
"exit conversation",
|
|
31
|
+
"end conversation",
|
|
32
|
+
"goodbye",
|
|
33
|
+
"bye",
|
|
34
|
+
"dừng lại",
|
|
35
|
+
"dừng",
|
|
36
|
+
"kết thúc",
|
|
37
|
+
"kết thúc hội thoại",
|
|
38
|
+
"thoát",
|
|
39
|
+
"tạm biệt",
|
|
40
|
+
];
|
|
41
|
+
|
|
42
|
+
export function normalizePhrase(text) {
|
|
43
|
+
return (text || "")
|
|
44
|
+
.trim()
|
|
45
|
+
.toLowerCase()
|
|
46
|
+
.replace(/[.!…?]+$/u, "")
|
|
47
|
+
.replace(/\s+/g, " ");
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
// Small vocabulary so stuttered/short stop commands ("stop stop", "please stop
|
|
51
|
+
// the conversation now", "dừng lại đi") still match, while real sentences
|
|
52
|
+
// ("stop the server", "do not stop") fall through and get submitted.
|
|
53
|
+
|
|
54
|
+
const STOP_VOCAB = new Set(
|
|
55
|
+
(
|
|
56
|
+
"stop stops stopping please now exit quit quits quitting end ends ending finish " +
|
|
57
|
+
"conversation voice chat goodbye bye good the this that it turn off mode " +
|
|
58
|
+
"dừng lại kết thúc hội thoại thoát tạm biệt đi ra ngừng ngưng thôi"
|
|
59
|
+
).split(" "),
|
|
60
|
+
);
|
|
61
|
+
const MAX_STOP_WORDS = 5;
|
|
62
|
+
|
|
63
|
+
export function matchesStopPhrase(text, phrases) {
|
|
64
|
+
const normalized = normalizePhrase(text);
|
|
65
|
+
if (!normalized) return false;
|
|
66
|
+
if ((phrases ?? DEFAULT_STOP_PHRASES).some((p) => normalizePhrase(p) === normalized)) return true;
|
|
67
|
+
// Lenient vocab match only applies to the default list; a custom list takes
|
|
68
|
+
// full control with exact matching.
|
|
69
|
+
if (phrases !== undefined) return false;
|
|
70
|
+
const words = normalized
|
|
71
|
+
.replace(/[^\p{L}\p{N}\s]/gu, "")
|
|
72
|
+
.split(/\s+/)
|
|
73
|
+
.filter(Boolean);
|
|
74
|
+
return (
|
|
75
|
+
words.length > 0 && words.length <= MAX_STOP_WORDS && words.every((w) => STOP_VOCAB.has(w))
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
function delay(ms) {
|
|
80
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesActive }) {
|
|
84
|
+
const maxTurns = Number(opts?.conversationMaxTurns) > 0 ? Number(opts.conversationMaxTurns) : 50;
|
|
85
|
+
const replyTimeoutMs =
|
|
86
|
+
Number(opts?.conversationTimeoutMs) > 0 ? Number(opts.conversationTimeoutMs) : 300000;
|
|
87
|
+
const restartDelayMs =
|
|
88
|
+
Number(opts?.conversationRestartDelayMs) >= 0 ? Number(opts.conversationRestartDelayMs) : 350;
|
|
89
|
+
// Keep undefined when the user supplied nothing so matchesStopPhrase
|
|
90
|
+
// takes the lenient default-vocabulary path ("stop stop", "dừng lại đi").
|
|
91
|
+
// Passing DEFAULT_STOP_PHRASES explicitly would disable it.
|
|
92
|
+
const stopPhrases = Array.isArray(opts?.conversationStopPhrases)
|
|
93
|
+
? opts.conversationStopPhrases
|
|
94
|
+
: undefined;
|
|
95
|
+
|
|
96
|
+
let active = false;
|
|
97
|
+
let phase = "idle"; // idle | recording | processing | waiting | speaking | paused
|
|
98
|
+
|
|
99
|
+
// Reply streaming state: text deltas heard while the agent works are spoken
|
|
100
|
+
// sentence by sentence through a serial queue, so the answer starts before
|
|
101
|
+
// the turn completes and every sentence gets its own vi/en voice.
|
|
102
|
+
|
|
103
|
+
let streamBuffer = "";
|
|
104
|
+
let streamSpokenChars = 0;
|
|
105
|
+
let streamUnsubs = [];
|
|
106
|
+
let speakTail = Promise.resolve();
|
|
107
|
+
let speakGen = 0;
|
|
108
|
+
|
|
109
|
+
function queueSpeak(sentence) {
|
|
110
|
+
const g = speakGen;
|
|
111
|
+
speakTail = speakTail.then(() => {
|
|
112
|
+
// Speaking must also drain: finishTurnFlow flips to speaking before
|
|
113
|
+
// flushing the tail, so a waiting-only guard would skip queued
|
|
114
|
+
// sentences/final partials under real TTS latency. speakGen still
|
|
115
|
+
// invalidates on pause/stop.
|
|
116
|
+
if (g !== speakGen || !active || (phase !== "waiting" && phase !== "speaking")) return null;
|
|
117
|
+
return tts.speak(sentence);
|
|
118
|
+
});
|
|
119
|
+
return speakTail;
|
|
120
|
+
}
|
|
121
|
+
let turn = 0;
|
|
122
|
+
let consecErrors = 0;
|
|
123
|
+
let generation = 0;
|
|
124
|
+
let waitCancel = null;
|
|
125
|
+
|
|
126
|
+
function toast(message, variant = "info") {
|
|
127
|
+
api.ui.toast({ message, variant, duration: 3000 });
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function currentSessionID() {
|
|
131
|
+
const route = api?.route?.current;
|
|
132
|
+
return route?.name === "session" ? route?.params?.sessionID : null;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function isActive() {
|
|
136
|
+
return active;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
function beginTurn() {
|
|
140
|
+
if (!active) return;
|
|
141
|
+
if (turn >= maxTurns) {
|
|
142
|
+
stop("done");
|
|
143
|
+
toast(`Conversation off - reached ${maxTurns} turns`);
|
|
144
|
+
return;
|
|
145
|
+
}
|
|
146
|
+
if (stt.isRecording() || stt.isProcessing() || stt.isStreaming?.()) return;
|
|
147
|
+
turn += 1;
|
|
148
|
+
phase = "recording";
|
|
149
|
+
logger?.log("VOICE", `Conversation turn ${turn} listening`, "debug");
|
|
150
|
+
if (!stt.start()) {
|
|
151
|
+
phase = "idle";
|
|
152
|
+
stop("busy");
|
|
153
|
+
toast("STT busy, conversation off", "warning");
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
function endReplyStream() {
|
|
158
|
+
for (const unsub of streamUnsubs) {
|
|
159
|
+
try {
|
|
160
|
+
unsub();
|
|
161
|
+
} catch {}
|
|
162
|
+
}
|
|
163
|
+
streamUnsubs = [];
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
// Kill streamed audio without touching the loop state. Queued sentences
|
|
167
|
+
// are invalidated so they never resume after a stop. The caller decides
|
|
168
|
+
// what happens next (legacy speak, notification, or exit).
|
|
169
|
+
|
|
170
|
+
function discardStreamAudio() {
|
|
171
|
+
endReplyStream();
|
|
172
|
+
speakGen += 1;
|
|
173
|
+
tts.stop();
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function stop(reason) {
|
|
177
|
+
if (!active) return;
|
|
178
|
+
active = false;
|
|
179
|
+
generation += 1;
|
|
180
|
+
phase = "idle";
|
|
181
|
+
if (waitCancel) {
|
|
182
|
+
waitCancel("stopped");
|
|
183
|
+
waitCancel = null;
|
|
184
|
+
}
|
|
185
|
+
endReplyStream();
|
|
186
|
+
speakGen += 1;
|
|
187
|
+
stt.setStopHint(null);
|
|
188
|
+
if (stt.isRecording() || stt.isProcessing()) stt.cancel();
|
|
189
|
+
clearProcessingToast();
|
|
190
|
+
tts.stop();
|
|
191
|
+
tts.setConversationActive(false);
|
|
192
|
+
logger?.log("VOICE", `Conversation stopped reason=${reason}`, "debug");
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function startReplyStream(sessionID, turnStart) {
|
|
196
|
+
streamBuffer = "";
|
|
197
|
+
streamSpokenChars = 0;
|
|
198
|
+
const unsub = api.event.on("message.part.delta", (event) => {
|
|
199
|
+
try {
|
|
200
|
+
if (!active || phase !== "waiting") return;
|
|
201
|
+
const props = event.properties || {};
|
|
202
|
+
if (sessionID && props.sessionID && props.sessionID !== sessionID) return;
|
|
203
|
+
if (typeof props.delta !== "string" || !props.delta) return;
|
|
204
|
+
if (props.field && props.field !== "text") return;
|
|
205
|
+
const lookupID = props.sessionID || sessionID;
|
|
206
|
+
const messages = api?.state?.session?.messages?.(lookupID) || [];
|
|
207
|
+
const msg = messages.find((m) => m?.id === props.messageID);
|
|
208
|
+
if (!msg || msg.role !== "assistant") return;
|
|
209
|
+
if ((msg.time?.created || 0) < turnStart) return;
|
|
210
|
+
const parts = api?.state?.part?.(props.messageID) || [];
|
|
211
|
+
const part = parts.find((p) => p?.id === props.partID);
|
|
212
|
+
if (part && part.type !== "text") return;
|
|
213
|
+
streamBuffer += props.delta;
|
|
214
|
+
flushStream(false);
|
|
215
|
+
} catch {}
|
|
216
|
+
});
|
|
217
|
+
streamUnsubs.push(unsub);
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
function flushStream(final = false) {
|
|
221
|
+
const { sentences, rest } = splitSpokenSentences(streamBuffer);
|
|
222
|
+
const ready = final && rest.trim() ? [...sentences, rest] : sentences;
|
|
223
|
+
if (ready.length > 0 && streamSpokenChars === 0) {
|
|
224
|
+
showProcessingToast("Speaking...");
|
|
225
|
+
}
|
|
226
|
+
for (const sentence of ready) {
|
|
227
|
+
if (!isSpeakableSentence(sentence)) continue;
|
|
228
|
+
const cleaned = localSpeechCleanup(sentence);
|
|
229
|
+
if (!cleaned) continue;
|
|
230
|
+
queueSpeak(cleaned);
|
|
231
|
+
streamSpokenChars += cleaned.length;
|
|
232
|
+
}
|
|
233
|
+
streamBuffer = final ? "" : rest;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// Wait for the submitted turn to finish: idle, a permission/question gate,
|
|
237
|
+
// or timeout. Resolves early when stop() cancels the wait.
|
|
238
|
+
|
|
239
|
+
function waitForReply(sessionID) {
|
|
240
|
+
let settled = false;
|
|
241
|
+
let unsubs = [];
|
|
242
|
+
let timer = null;
|
|
243
|
+
|
|
244
|
+
let cancelFn = null;
|
|
245
|
+
const done = (outcome) => {
|
|
246
|
+
if (settled) return null;
|
|
247
|
+
settled = true;
|
|
248
|
+
if (timer) clearTimeout(timer);
|
|
249
|
+
for (const unsub of unsubs) {
|
|
250
|
+
try {
|
|
251
|
+
unsub();
|
|
252
|
+
} catch {}
|
|
253
|
+
}
|
|
254
|
+
unsubs = [];
|
|
255
|
+
if (waitCancel === cancelFn) waitCancel = null;
|
|
256
|
+
return outcome;
|
|
257
|
+
};
|
|
258
|
+
|
|
259
|
+
const promise = new Promise((resolve) => {
|
|
260
|
+
const finish = (outcome) => {
|
|
261
|
+
const result = done(outcome);
|
|
262
|
+
if (result !== null) resolve(result);
|
|
263
|
+
};
|
|
264
|
+
cancelFn = (outcome) => finish(outcome || "stopped");
|
|
265
|
+
waitCancel = cancelFn;
|
|
266
|
+
|
|
267
|
+
const forSession = (props) =>
|
|
268
|
+
!sessionID || !props?.sessionID || props.sessionID === sessionID;
|
|
269
|
+
|
|
270
|
+
unsubs = [
|
|
271
|
+
api.event.on("session.idle", (event) => {
|
|
272
|
+
if (forSession(event.properties)) finish("idle");
|
|
273
|
+
}),
|
|
274
|
+
api.event.on("session.status", (event) => {
|
|
275
|
+
if (event.properties?.status?.type === "idle" && forSession(event.properties)) {
|
|
276
|
+
finish("idle");
|
|
277
|
+
}
|
|
278
|
+
}),
|
|
279
|
+
api.event.on("permission.asked", (event) => {
|
|
280
|
+
if (forSession(event.properties)) finish("permission");
|
|
281
|
+
}),
|
|
282
|
+
api.event.on("question.asked", (event) => {
|
|
283
|
+
if (forSession(event.properties)) finish("question");
|
|
284
|
+
}),
|
|
285
|
+
];
|
|
286
|
+
timer = setTimeout(() => finish("timeout"), replyTimeoutMs);
|
|
287
|
+
});
|
|
288
|
+
|
|
289
|
+
return promise;
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
async function finishTurnFlow() {
|
|
293
|
+
if (!active || phase !== "recording") return;
|
|
294
|
+
const myGen = generation;
|
|
295
|
+
phase = "processing";
|
|
296
|
+
|
|
297
|
+
const res = await stt.transcribeTurn();
|
|
298
|
+
if (!active || myGen !== generation) return;
|
|
299
|
+
|
|
300
|
+
if (res.error) {
|
|
301
|
+
consecErrors += 1;
|
|
302
|
+
logger?.log("VOICE", `Conversation turn error consec=${consecErrors}`, "warn");
|
|
303
|
+
if (consecErrors >= 2) {
|
|
304
|
+
stop("error");
|
|
305
|
+
return;
|
|
306
|
+
}
|
|
307
|
+
pauseForRetry("Paused - press the key to retry");
|
|
308
|
+
return;
|
|
309
|
+
}
|
|
310
|
+
if (!res.text) {
|
|
311
|
+
consecErrors = 0;
|
|
312
|
+
pauseForRetry("No speech heard - press the key to try again");
|
|
313
|
+
return;
|
|
314
|
+
}
|
|
315
|
+
consecErrors = 0;
|
|
316
|
+
|
|
317
|
+
if (matchesStopPhrase(res.text, stopPhrases)) {
|
|
318
|
+
stt.discard();
|
|
319
|
+
stop("stop phrase");
|
|
320
|
+
toast("Conversation off");
|
|
321
|
+
return;
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
const submitted = await stt.submitTurnText(res.text);
|
|
325
|
+
if (!active || myGen !== generation) return;
|
|
326
|
+
if (submitted.error) {
|
|
327
|
+
stop("submit failed");
|
|
328
|
+
return;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
phase = "waiting";
|
|
332
|
+
const sessionID = currentSessionID();
|
|
333
|
+
const turnStart = Date.now();
|
|
334
|
+
showProcessingToast("Waiting for reply...");
|
|
335
|
+
startReplyStream(sessionID, turnStart);
|
|
336
|
+
const tWait = Date.now();
|
|
337
|
+
const outcome = await waitForReply(sessionID);
|
|
338
|
+
const waitMs = Date.now() - tWait;
|
|
339
|
+
endReplyStream();
|
|
340
|
+
logger?.log(
|
|
341
|
+
"VOICE",
|
|
342
|
+
`Conversation wait outcome=${outcome} waitMs=${waitMs} streamedChars=${streamSpokenChars}`,
|
|
343
|
+
"debug",
|
|
344
|
+
);
|
|
345
|
+
if (!active || myGen !== generation) return;
|
|
346
|
+
|
|
347
|
+
if (outcome === "timeout") {
|
|
348
|
+
discardStreamAudio();
|
|
349
|
+
stop("timeout");
|
|
350
|
+
toast("No reply in time, conversation off", "warning");
|
|
351
|
+
return;
|
|
352
|
+
}
|
|
353
|
+
if (outcome === "permission" || outcome === "question") {
|
|
354
|
+
discardStreamAudio();
|
|
355
|
+
phase = "speaking";
|
|
356
|
+
showProcessingToast("Speaking...");
|
|
357
|
+
await tts.speak(
|
|
358
|
+
outcome === "permission"
|
|
359
|
+
? "Permission requested. Please check your screen."
|
|
360
|
+
: "A question needs your answer. Please check your screen.",
|
|
361
|
+
);
|
|
362
|
+
clearProcessingToast();
|
|
363
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
364
|
+
beginTurn();
|
|
365
|
+
return;
|
|
366
|
+
}
|
|
367
|
+
if (outcome === "stopped") return;
|
|
368
|
+
|
|
369
|
+
// Streamed path: sentences already queued; flush the tail and let the
|
|
370
|
+
// audio finish before listening again. Fall back to the full
|
|
371
|
+
// LLM-narrated speak only when nothing streamable arrived.
|
|
372
|
+
|
|
373
|
+
if (streamSpokenChars > 0) {
|
|
374
|
+
phase = "speaking";
|
|
375
|
+
flushStream(true);
|
|
376
|
+
await speakTail;
|
|
377
|
+
clearProcessingToast();
|
|
378
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
379
|
+
await delay(restartDelayMs);
|
|
380
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
381
|
+
beginTurn();
|
|
382
|
+
return;
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
phase = "speaking";
|
|
386
|
+
const spoken = await tts.speakAssistantTurn();
|
|
387
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
388
|
+
if (!spoken.spoken) {
|
|
389
|
+
toast("No reply to speak, listening again", "warning");
|
|
390
|
+
}
|
|
391
|
+
await delay(restartDelayMs);
|
|
392
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
393
|
+
beginTurn();
|
|
394
|
+
}
|
|
395
|
+
|
|
396
|
+
// Pause instead of auto-recording so an empty or failed turn never feels
|
|
397
|
+
// like the keypress was ignored.
|
|
398
|
+
|
|
399
|
+
function pauseForRetry(message) {
|
|
400
|
+
if (!active) return;
|
|
401
|
+
phase = "paused";
|
|
402
|
+
if (message) toast(message, "warning");
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// Stop speech and hold the mode. The next keypress records again.
|
|
406
|
+
|
|
407
|
+
function pauseSpeech() {
|
|
408
|
+
if (!active || phase !== "speaking") return;
|
|
409
|
+
logger?.log("VOICE", "Conversation speech paused", "debug");
|
|
410
|
+
tts.stop();
|
|
411
|
+
phase = "paused";
|
|
412
|
+
toast("Paused - press the key to speak");
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
// Called by the TTS stop command while the mode is on. Returns true when the
|
|
416
|
+
// keypress was consumed (speech paused), so TTS skips its own handling.
|
|
417
|
+
|
|
418
|
+
function onTtsStop() {
|
|
419
|
+
if (!active) return false;
|
|
420
|
+
if (phase === "speaking") {
|
|
421
|
+
pauseSpeech();
|
|
422
|
+
return true;
|
|
423
|
+
}
|
|
424
|
+
// Mid-stream the agent is still working and cannot pause, so the stop key
|
|
425
|
+
// exits the mode instead of leaving a half-heard reply behind.
|
|
426
|
+
if (phase === "waiting" && tts.isSpeaking?.()) {
|
|
427
|
+
stop("cancelled");
|
|
428
|
+
toast("Conversation off");
|
|
429
|
+
return true;
|
|
430
|
+
}
|
|
431
|
+
return false;
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
function onKey(source) {
|
|
435
|
+
if (!active) {
|
|
436
|
+
start();
|
|
437
|
+
return;
|
|
438
|
+
}
|
|
439
|
+
if (phase === "recording") {
|
|
440
|
+
finishTurnFlow();
|
|
441
|
+
return;
|
|
442
|
+
}
|
|
443
|
+
if (phase === "speaking") {
|
|
444
|
+
pauseSpeech();
|
|
445
|
+
return;
|
|
446
|
+
}
|
|
447
|
+
if (phase === "paused") {
|
|
448
|
+
beginTurn();
|
|
449
|
+
return;
|
|
450
|
+
}
|
|
451
|
+
// processing | waiting | idle
|
|
452
|
+
if (source === "toggle") {
|
|
453
|
+
stop("cancelled");
|
|
454
|
+
toast("Conversation off");
|
|
455
|
+
} else {
|
|
456
|
+
toast("Conversation busy, please wait...", "warning");
|
|
457
|
+
}
|
|
458
|
+
}
|
|
459
|
+
|
|
460
|
+
function start() {
|
|
461
|
+
if (active) return;
|
|
462
|
+
if (isLiveNotesActive?.()) {
|
|
463
|
+
toast("Live notes recording - stop it first (/voice-notes-stop)", "warning");
|
|
464
|
+
return;
|
|
465
|
+
}
|
|
466
|
+
if (stt.isRecording() || stt.isProcessing() || stt.isStreaming?.()) {
|
|
467
|
+
toast("STT busy, try again shortly", "warning");
|
|
468
|
+
return;
|
|
469
|
+
}
|
|
470
|
+
active = true;
|
|
471
|
+
generation += 1;
|
|
472
|
+
turn = 0;
|
|
473
|
+
consecErrors = 0;
|
|
474
|
+
phase = "idle";
|
|
475
|
+
tts.setConversationActive(true);
|
|
476
|
+
stt.setStopHint("<leader>v");
|
|
477
|
+
logger?.log("VOICE", "Conversation started", "debug");
|
|
478
|
+
toast("Conversation on - one key: send, pause speech, resume");
|
|
479
|
+
beginTurn();
|
|
480
|
+
}
|
|
481
|
+
|
|
482
|
+
api.lifecycle?.onDispose?.(() => stop("dispose"));
|
|
483
|
+
|
|
484
|
+
const DEFAULT_KEYBINDS = {
|
|
485
|
+
"voice.conversation": "<leader>v",
|
|
486
|
+
};
|
|
487
|
+
function kb(value) {
|
|
488
|
+
const overrides = opts?.keybinds;
|
|
489
|
+
if (!overrides || typeof overrides !== "object" || Array.isArray(overrides)) {
|
|
490
|
+
return DEFAULT_KEYBINDS[value];
|
|
491
|
+
}
|
|
492
|
+
if (!Object.prototype.hasOwnProperty.call(overrides, value)) return DEFAULT_KEYBINDS[value];
|
|
493
|
+
const v = overrides[value];
|
|
494
|
+
if (!v || v === "none") return undefined;
|
|
495
|
+
return v;
|
|
496
|
+
}
|
|
497
|
+
|
|
498
|
+
const commands = [
|
|
499
|
+
{
|
|
500
|
+
title: "Voice: conversation mode",
|
|
501
|
+
value: "voice.conversation",
|
|
502
|
+
category: "opencode-voice",
|
|
503
|
+
description: "Toggle voice conversation (one key: send, pause speech, resume, exit)",
|
|
504
|
+
...(kb("voice.conversation") ? { keybind: kb("voice.conversation") } : {}),
|
|
505
|
+
slash: { name: "voice-conversation" },
|
|
506
|
+
onSelect() {
|
|
507
|
+
onKey("toggle");
|
|
508
|
+
},
|
|
509
|
+
},
|
|
510
|
+
{
|
|
511
|
+
title: "Voice: stop conversation",
|
|
512
|
+
value: "voice.conversation-stop",
|
|
513
|
+
category: "opencode-voice",
|
|
514
|
+
description: "Exit voice conversation mode",
|
|
515
|
+
slash: { name: "voice-conversation-stop" },
|
|
516
|
+
onSelect() {
|
|
517
|
+
if (active) {
|
|
518
|
+
stop("command");
|
|
519
|
+
toast("Conversation off");
|
|
520
|
+
}
|
|
521
|
+
},
|
|
522
|
+
},
|
|
523
|
+
];
|
|
524
|
+
|
|
525
|
+
const controller = { isActive, onKey, onTtsStop, start, stop };
|
|
526
|
+
|
|
527
|
+
return { commands, controller };
|
|
528
|
+
}
|