@bojackduy/opencode-voice 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +36 -21
- package/lib/conversation.js +141 -59
- package/lib/tts.js +67 -13
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -419,36 +419,47 @@ then `s`.
|
|
|
419
419
|
|
|
420
420
|
### Voice conversation
|
|
421
421
|
|
|
422
|
-
| Command | Keybind | Description
|
|
423
|
-
| -------------------------- | ---------- |
|
|
424
|
-
| `/voice-conversation` | `leader+v` | Toggle
|
|
425
|
-
| `/voice-conversation-stop` | | Exit voice conversation mode
|
|
422
|
+
| Command | Keybind | Description |
|
|
423
|
+
| -------------------------- | ---------- | ------------------------------------------- |
|
|
424
|
+
| `/voice-conversation` | `leader+v` | Toggle push-to-talk voice conversation mode |
|
|
425
|
+
| `/voice-conversation-stop` | | Exit voice conversation mode |
|
|
426
426
|
|
|
427
|
-
One key drives the whole loop
|
|
427
|
+
One key drives the whole loop. It always means the same thing - "I want the
|
|
428
|
+
floor" - and what it does follows the toast on screen:
|
|
428
429
|
|
|
429
430
|
```
|
|
430
|
-
record -> transcribe -> normalize -> submit ->
|
|
431
|
+
press -> record -> press -> transcribe -> normalize -> submit -> speak -> press ...
|
|
431
432
|
```
|
|
432
433
|
|
|
433
|
-
| Toast shows
|
|
434
|
-
|
|
|
435
|
-
|
|
|
436
|
-
|
|
|
437
|
-
|
|
|
438
|
-
|
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
445
|
-
|
|
446
|
-
|
|
434
|
+
| Toast shows | Pressing the key does |
|
|
435
|
+
| ------------------------ | ----------------------------------------- |
|
|
436
|
+
| Press `leader+v` to talk | Open the mic (this is the resting state) |
|
|
437
|
+
| ● Recording | Finish the turn and submit |
|
|
438
|
+
| Working... / Speaking | Barge in: cut audio, drop reply, open mic |
|
|
439
|
+
| Answer on screen | Same barge-in, while the agent resumes |
|
|
440
|
+
| Transcribing... | Report busy - too short to interrupt |
|
|
441
|
+
|
|
442
|
+
The loop never reopens the mic by itself: after a reply is spoken the mode rests
|
|
443
|
+
paused, so it cannot record the room while nobody is talking. Barge-in drops
|
|
444
|
+
the pending reply on purpose - pressing means talking now, not hearing the
|
|
445
|
+
rest.
|
|
446
|
+
|
|
447
|
+
Answering a permission or a question is the one thing the key cannot do,
|
|
448
|
+
because that answer is given on screen. The mode announces the gate, keeps the
|
|
449
|
+
mic shut, then speaks the agent's answer once the agent resumes.
|
|
450
|
+
|
|
451
|
+
Exit paths: a stop phrase (`stop`, `stop stop`, `dừng lại đi`, ...),
|
|
452
|
+
`/voice-conversation-stop`, or the global `/voice-cancel` (`leader+.`). Only a
|
|
453
|
+
bare stop phrase ends the mode - full sentences mentioning stop submit
|
|
454
|
+
normally. Empty or failed turns pause instead of re-recording, so the key never
|
|
455
|
+
surprises. `/tts-stop` pauses a speaking reply and rests paused. While the mode
|
|
456
|
+
is on, the plain `/stt-record` keys act as the conversation key and auto TTS
|
|
457
|
+
stays silent (the loop speaks the reply itself).
|
|
447
458
|
|
|
448
459
|
While waiting, assistant text is spoken sentence by sentence as it streams in
|
|
449
460
|
(local cleanup, no LLM), so the answer starts before the agent turn finishes.
|
|
450
461
|
Code-like sentences are skipped; replies with nothing streamable fall back to
|
|
451
|
-
the full narrated speak.
|
|
462
|
+
the full narrated speak.
|
|
452
463
|
|
|
453
464
|
- `ttsNormalizeMode` _(optional)_ - `"llm"` (default, polished narration) or
|
|
454
465
|
`"local"` (instant deterministic cleanup, no network). Streaming speech
|
|
@@ -667,6 +678,10 @@ live-mic session was measured.
|
|
|
667
678
|
code-heavy responses, or briefly notify for confirmations
|
|
668
679
|
3. Piper synthesizes speech locally, piped through sox for playback
|
|
669
680
|
|
|
681
|
+
The LLM is only a polish layer: when the configured model is unavailable or out
|
|
682
|
+
of quota, TTS speaks the local cleanup automatically instead of going silent.
|
|
683
|
+
Set `ttsNormalizeMode: "local"` to skip narration entirely.
|
|
684
|
+
|
|
670
685
|
### Auto TTS
|
|
671
686
|
|
|
672
687
|
When enabled (`/tts-mode`), the plugin automatically speaks:
|
package/lib/conversation.js
CHANGED
|
@@ -1,18 +1,27 @@
|
|
|
1
|
-
// Voice conversation mode:
|
|
1
|
+
// Voice conversation mode: push-to-talk turns with barge-in.
|
|
2
2
|
//
|
|
3
|
-
// One key drives the whole loop
|
|
3
|
+
// One key drives the whole loop and its meaning is always the same: the user
|
|
4
|
+
// wants the floor. What the key actually does depends on the state shown in
|
|
4
5
|
// the toast:
|
|
5
6
|
//
|
|
6
|
-
//
|
|
7
|
+
// press -> record -> press -> transcribe -> normalize -> submit
|
|
8
|
+
// -> wait reply -> speak -> press -> record ...
|
|
7
9
|
//
|
|
10
|
+
// - paused + key: open the mic (the resting state is the user's turn)
|
|
8
11
|
// - recording + key: finish the turn and submit
|
|
9
|
-
// - speaking + key:
|
|
10
|
-
//
|
|
11
|
-
// -
|
|
12
|
+
// - speaking/waiting + key: barge in - cut the audio, drop the pending reply,
|
|
13
|
+
// open the mic (one press, never a stop-then-record dance)
|
|
14
|
+
// - processing + key: transcription is too short to interrupt, report busy
|
|
12
15
|
//
|
|
13
|
-
//
|
|
14
|
-
//
|
|
15
|
-
//
|
|
16
|
+
// The loop never reopens the mic by itself: after a reply is spoken the mode
|
|
17
|
+
// rests paused, so it can never record the room while nobody is talking. That
|
|
18
|
+
// auto-record is also why a question/permission gate used to eat the answer -
|
|
19
|
+
// the mic opened, the user's muttering was submitted, and the real reply was
|
|
20
|
+
// already torn down. Gates now announce and wait instead.
|
|
21
|
+
//
|
|
22
|
+
// Exit paths: a stop phrase ("stop", "dừng lại", ...), /voice-conversation-stop,
|
|
23
|
+
// or the global /voice-cancel. Empty or failed turns pause instead of
|
|
24
|
+
// auto-recording, so the key never surprises.
|
|
16
25
|
//
|
|
17
26
|
// Latency: while waiting, assistant text deltas are spoken sentence by
|
|
18
27
|
// sentence (local cleanup, no LLM), so the answer starts before the turn
|
|
@@ -151,7 +160,34 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
151
160
|
phase = "idle";
|
|
152
161
|
stop("busy");
|
|
153
162
|
toast("STT busy, conversation off", "warning");
|
|
163
|
+
return;
|
|
154
164
|
}
|
|
165
|
+
toast(`Listening - press ${keyLabel()} to send`);
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// The resting state: mic closed, one press to talk. Entered by start() and
|
|
169
|
+
// after a reply is spoken, never by the loop auto-recording.
|
|
170
|
+
|
|
171
|
+
function enterPaused() {
|
|
172
|
+
phase = "paused";
|
|
173
|
+
toast(`Press ${keyLabel()} to talk`);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
// One press takes the floor: cut whatever audio is queued or playing, drop
|
|
177
|
+
// the pending reply, and open the mic. The mode stays on - a press always
|
|
178
|
+
// means "I want to talk now", so it must never exit silently.
|
|
179
|
+
|
|
180
|
+
function bargeIn() {
|
|
181
|
+
logger?.log("VOICE", "Conversation barge-in", "debug");
|
|
182
|
+
discardStreamAudio();
|
|
183
|
+
clearProcessingToast();
|
|
184
|
+
// Settle the pending reply wait as "stopped" so the abandoned answer is
|
|
185
|
+
// never flushed into the mic we are about to open.
|
|
186
|
+
if (waitCancel) {
|
|
187
|
+
waitCancel("stopped");
|
|
188
|
+
waitCancel = null;
|
|
189
|
+
}
|
|
190
|
+
beginTurn();
|
|
155
191
|
}
|
|
156
192
|
|
|
157
193
|
function endReplyStream() {
|
|
@@ -221,7 +257,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
221
257
|
const { sentences, rest } = splitSpokenSentences(streamBuffer);
|
|
222
258
|
const ready = final && rest.trim() ? [...sentences, rest] : sentences;
|
|
223
259
|
if (ready.length > 0 && streamSpokenChars === 0) {
|
|
224
|
-
showProcessingToast(
|
|
260
|
+
showProcessingToast(`Speaking - press ${keyLabel()} to interrupt`);
|
|
225
261
|
}
|
|
226
262
|
for (const sentence of ready) {
|
|
227
263
|
if (!isSpeakableSentence(sentence)) continue;
|
|
@@ -234,9 +270,11 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
234
270
|
}
|
|
235
271
|
|
|
236
272
|
// Wait for the submitted turn to finish: idle, a permission/question gate,
|
|
237
|
-
// or timeout. Resolves early when stop() cancels the wait.
|
|
273
|
+
// or timeout. Resolves early when stop() cancels the wait. gates:false drops
|
|
274
|
+
// the permission/question subscriptions, so a second wait survives the gate
|
|
275
|
+
// the user is answering on screen and only ends on idle/timeout/stop.
|
|
238
276
|
|
|
239
|
-
function waitForReply(sessionID) {
|
|
277
|
+
function waitForReply(sessionID, { gates = true } = {}) {
|
|
240
278
|
let settled = false;
|
|
241
279
|
let unsubs = [];
|
|
242
280
|
let timer = null;
|
|
@@ -276,12 +314,16 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
276
314
|
finish("idle");
|
|
277
315
|
}
|
|
278
316
|
}),
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
|
|
317
|
+
...(gates
|
|
318
|
+
? [
|
|
319
|
+
api.event.on("permission.asked", (event) => {
|
|
320
|
+
if (forSession(event.properties)) finish("permission");
|
|
321
|
+
}),
|
|
322
|
+
api.event.on("question.asked", (event) => {
|
|
323
|
+
if (forSession(event.properties)) finish("question");
|
|
324
|
+
}),
|
|
325
|
+
]
|
|
326
|
+
: []),
|
|
285
327
|
];
|
|
286
328
|
timer = setTimeout(() => finish("timeout"), replyTimeoutMs);
|
|
287
329
|
});
|
|
@@ -293,6 +335,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
293
335
|
if (!active || phase !== "recording") return;
|
|
294
336
|
const myGen = generation;
|
|
295
337
|
phase = "processing";
|
|
338
|
+
showProcessingToast("Transcribing...");
|
|
296
339
|
|
|
297
340
|
const res = await stt.transcribeTurn();
|
|
298
341
|
if (!active || myGen !== generation) return;
|
|
@@ -304,12 +347,12 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
304
347
|
stop("error");
|
|
305
348
|
return;
|
|
306
349
|
}
|
|
307
|
-
pauseForRetry(
|
|
350
|
+
pauseForRetry(`Paused - press ${keyLabel()} to retry`);
|
|
308
351
|
return;
|
|
309
352
|
}
|
|
310
353
|
if (!res.text) {
|
|
311
354
|
consecErrors = 0;
|
|
312
|
-
pauseForRetry(
|
|
355
|
+
pauseForRetry(`No speech heard - press ${keyLabel()} to try again`);
|
|
313
356
|
return;
|
|
314
357
|
}
|
|
315
358
|
consecErrors = 0;
|
|
@@ -331,7 +374,7 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
331
374
|
phase = "waiting";
|
|
332
375
|
const sessionID = currentSessionID();
|
|
333
376
|
const turnStart = Date.now();
|
|
334
|
-
showProcessingToast(
|
|
377
|
+
showProcessingToast(`Working - press ${keyLabel()} to talk`);
|
|
335
378
|
startReplyStream(sessionID, turnStart);
|
|
336
379
|
const tWait = Date.now();
|
|
337
380
|
const outcome = await waitForReply(sessionID);
|
|
@@ -351,46 +394,78 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
351
394
|
return;
|
|
352
395
|
}
|
|
353
396
|
if (outcome === "permission" || outcome === "question") {
|
|
354
|
-
|
|
355
|
-
phase = "speaking";
|
|
356
|
-
showProcessingToast("Speaking...");
|
|
357
|
-
await tts.speak(
|
|
358
|
-
outcome === "permission"
|
|
359
|
-
? "Permission requested. Please check your screen."
|
|
360
|
-
: "A question needs your answer. Please check your screen.",
|
|
361
|
-
);
|
|
362
|
-
clearProcessingToast();
|
|
363
|
-
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
364
|
-
beginTurn();
|
|
397
|
+
await answerGateThenSpeak(myGen, sessionID, turnStart, outcome);
|
|
365
398
|
return;
|
|
366
399
|
}
|
|
367
400
|
if (outcome === "stopped") return;
|
|
368
401
|
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
402
|
+
await speakReply(myGen);
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
// A permission/question gate is answered ON SCREEN, so the mic must stay
|
|
406
|
+
// shut. Announce the gate, then re-arm the reply stream and wait for the
|
|
407
|
+
// agent to resume so its post-answer reply is spoken aloud - before this,
|
|
408
|
+
// endReplyStream() had already run and the answer was dropped in silence.
|
|
372
409
|
|
|
410
|
+
async function answerGateThenSpeak(myGen, sessionID, turnStart, outcome) {
|
|
411
|
+
discardStreamAudio();
|
|
412
|
+
phase = "speaking";
|
|
413
|
+
showProcessingToast("Speaking...");
|
|
414
|
+
await tts.speak(
|
|
415
|
+
outcome === "permission"
|
|
416
|
+
? "Permission requested. Please check your screen."
|
|
417
|
+
: "A question needs your answer. Please check your screen.",
|
|
418
|
+
);
|
|
419
|
+
clearProcessingToast();
|
|
420
|
+
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
421
|
+
|
|
422
|
+
// Still "waiting": the user answers on screen, and a press here is an
|
|
423
|
+
// ordinary barge-in (drop the pending reply, open the mic).
|
|
424
|
+
phase = "waiting";
|
|
425
|
+
showProcessingToast(`Answer on screen - press ${keyLabel()} to talk`);
|
|
426
|
+
startReplyStream(sessionID, turnStart);
|
|
427
|
+
const resumed = await waitForReply(sessionID, { gates: false });
|
|
428
|
+
if (!active || myGen !== generation) return;
|
|
429
|
+
if (resumed === "stopped") return;
|
|
430
|
+
endReplyStream();
|
|
431
|
+
if (resumed === "timeout") {
|
|
432
|
+
discardStreamAudio();
|
|
433
|
+
stop("timeout");
|
|
434
|
+
toast("No reply in time, conversation off", "warning");
|
|
435
|
+
return;
|
|
436
|
+
}
|
|
437
|
+
await speakReply(myGen);
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
// Speak whatever this turn produced, then rest paused with the mic OFF.
|
|
441
|
+
// Streamed sentences are already queued, so only the tail needs flushing;
|
|
442
|
+
// the full LLM-narrated speak is the fallback when nothing streamable
|
|
443
|
+
// arrived. Returns early on any generation/phase change (barge-in, stop).
|
|
444
|
+
|
|
445
|
+
async function speakReply(myGen) {
|
|
446
|
+
phase = "speaking";
|
|
447
|
+
showProcessingToast(`Speaking - press ${keyLabel()} to interrupt`);
|
|
373
448
|
if (streamSpokenChars > 0) {
|
|
374
|
-
phase = "speaking";
|
|
375
449
|
flushStream(true);
|
|
376
450
|
await speakTail;
|
|
377
451
|
clearProcessingToast();
|
|
378
452
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
379
453
|
await delay(restartDelayMs);
|
|
380
454
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
381
|
-
|
|
455
|
+
enterPaused();
|
|
382
456
|
return;
|
|
383
457
|
}
|
|
384
|
-
|
|
385
|
-
phase = "speaking";
|
|
386
458
|
const spoken = await tts.speakAssistantTurn();
|
|
387
459
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
460
|
+
// Drop the sticky speaking status before the paused toast, or it keeps
|
|
461
|
+
// re-announcing "interrupt" once the mic is closed again.
|
|
462
|
+
clearProcessingToast();
|
|
388
463
|
if (!spoken.spoken) {
|
|
389
|
-
toast("No reply to speak
|
|
464
|
+
toast("No reply to speak", "warning");
|
|
390
465
|
}
|
|
391
466
|
await delay(restartDelayMs);
|
|
392
467
|
if (!active || myGen !== generation || phase !== "speaking") return;
|
|
393
|
-
|
|
468
|
+
enterPaused();
|
|
394
469
|
}
|
|
395
470
|
|
|
396
471
|
// Pause instead of auto-recording so an empty or failed turn never feels
|
|
@@ -402,14 +477,15 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
402
477
|
if (message) toast(message, "warning");
|
|
403
478
|
}
|
|
404
479
|
|
|
405
|
-
// Stop speech and hold the mode. The next keypress records
|
|
480
|
+
// Stop speech and hold the mode, mic still closed. The next keypress records
|
|
481
|
+
// again. Only the TTS stop key lands here; the conversation key barges in.
|
|
406
482
|
|
|
407
483
|
function pauseSpeech() {
|
|
408
484
|
if (!active || phase !== "speaking") return;
|
|
409
485
|
logger?.log("VOICE", "Conversation speech paused", "debug");
|
|
410
486
|
tts.stop();
|
|
411
487
|
phase = "paused";
|
|
412
|
-
toast(
|
|
488
|
+
toast(`Speech paused - press ${keyLabel()} to talk`);
|
|
413
489
|
}
|
|
414
490
|
|
|
415
491
|
// Called by the TTS stop command while the mode is on. Returns true when the
|
|
@@ -431,7 +507,10 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
431
507
|
return false;
|
|
432
508
|
}
|
|
433
509
|
|
|
434
|
-
|
|
510
|
+
// One press, one meaning: "I want the floor". The source key (record,
|
|
511
|
+
// submit, toggle) no longer changes the outcome, and a press never exits.
|
|
512
|
+
|
|
513
|
+
function onKey() {
|
|
435
514
|
if (!active) {
|
|
436
515
|
start();
|
|
437
516
|
return;
|
|
@@ -440,21 +519,16 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
440
519
|
finishTurnFlow();
|
|
441
520
|
return;
|
|
442
521
|
}
|
|
443
|
-
if (phase === "speaking") {
|
|
444
|
-
|
|
522
|
+
if (phase === "speaking" || phase === "waiting") {
|
|
523
|
+
bargeIn();
|
|
445
524
|
return;
|
|
446
525
|
}
|
|
447
526
|
if (phase === "paused") {
|
|
448
527
|
beginTurn();
|
|
449
528
|
return;
|
|
450
529
|
}
|
|
451
|
-
// processing |
|
|
452
|
-
|
|
453
|
-
stop("cancelled");
|
|
454
|
-
toast("Conversation off");
|
|
455
|
-
} else {
|
|
456
|
-
toast("Conversation busy, please wait...", "warning");
|
|
457
|
-
}
|
|
530
|
+
// processing | idle: the transcription in flight is too short to interrupt.
|
|
531
|
+
toast("Conversation busy, please wait...", "warning");
|
|
458
532
|
}
|
|
459
533
|
|
|
460
534
|
function start() {
|
|
@@ -471,12 +545,12 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
471
545
|
generation += 1;
|
|
472
546
|
turn = 0;
|
|
473
547
|
consecErrors = 0;
|
|
474
|
-
phase = "
|
|
548
|
+
phase = "paused";
|
|
475
549
|
tts.setConversationActive(true);
|
|
476
|
-
|
|
550
|
+
// Recording toasts must name the key the user actually has bound.
|
|
551
|
+
stt.setStopHint(kb("voice.conversation"));
|
|
477
552
|
logger?.log("VOICE", "Conversation started", "debug");
|
|
478
|
-
toast(
|
|
479
|
-
beginTurn();
|
|
553
|
+
toast(`Conversation on - press ${keyLabel()} to talk`);
|
|
480
554
|
}
|
|
481
555
|
|
|
482
556
|
api.lifecycle?.onDispose?.(() => stop("dispose"));
|
|
@@ -495,16 +569,24 @@ export function registerConversation(api, opts, logger, { stt, tts, isLiveNotesA
|
|
|
495
569
|
return v;
|
|
496
570
|
}
|
|
497
571
|
|
|
572
|
+
// Every toast names the key that acts, resolved from the user's keybinds
|
|
573
|
+
// instead of the default literal - a rebound key must never be announced as
|
|
574
|
+
// something the user does not have.
|
|
575
|
+
function keyLabel() {
|
|
576
|
+
return kb("voice.conversation") || "the key";
|
|
577
|
+
}
|
|
578
|
+
|
|
498
579
|
const commands = [
|
|
499
580
|
{
|
|
500
581
|
title: "Voice: conversation mode",
|
|
501
582
|
value: "voice.conversation",
|
|
502
583
|
category: "opencode-voice",
|
|
503
|
-
description:
|
|
584
|
+
description:
|
|
585
|
+
"Toggle voice conversation (push-to-talk: press to record, press again to send, press while it speaks to interrupt)",
|
|
504
586
|
...(kb("voice.conversation") ? { keybind: kb("voice.conversation") } : {}),
|
|
505
587
|
slash: { name: "voice-conversation" },
|
|
506
588
|
onSelect() {
|
|
507
|
-
onKey(
|
|
589
|
+
onKey();
|
|
508
590
|
},
|
|
509
591
|
},
|
|
510
592
|
{
|
package/lib/tts.js
CHANGED
|
@@ -66,6 +66,44 @@ export function localSpeechCleanup(text) {
|
|
|
66
66
|
return out.replace(/\s+/g, " ").trim();
|
|
67
67
|
}
|
|
68
68
|
|
|
69
|
+
// ---- Speech must never depend on the LLM ----
|
|
70
|
+
// The narrator is a cosmetic polish layer on top of local Piper, which is
|
|
71
|
+
// always available. So when the narrator fails for any reason (quota
|
|
72
|
+
// exhausted, bad auth, unreachable endpoint, missing model, non-2xx) we speak
|
|
73
|
+
// the deterministic local cleanup instead of going silent - this is the one
|
|
74
|
+
// place the decision lives, so every caller benefits.
|
|
75
|
+
//
|
|
76
|
+
// Pure so it is testable without a TUI, a model, or piper.
|
|
77
|
+
|
|
78
|
+
export function resolveSpeechText(rawText, llmResult) {
|
|
79
|
+
if (llmResult?.text) return { text: llmResult.text };
|
|
80
|
+
return { text: localSpeechCleanup(rawText), error: llmResult?.error, fellBack: true };
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
// The llm-client retries with backoff (normalizeRetries in llm-client.js), so a
|
|
84
|
+
// model that is merely out of quota stalls for many seconds before admitting
|
|
85
|
+
// failure - waiting that long for a polish layer is worse than unpolished
|
|
86
|
+
// speech, so bound the wait. The timer is unref'd (same trick as
|
|
87
|
+
// wallTimeout in streaming-stt.js) and cleared on the happy path, so tests and
|
|
88
|
+
// short-lived processes never wait out the full bound.
|
|
89
|
+
const NORMALIZE_FALLBACK_TIMEOUT_MS = 12000;
|
|
90
|
+
|
|
91
|
+
async function normalizeWithinBound(promise) {
|
|
92
|
+
let timer = null;
|
|
93
|
+
const bound = new Promise((resolve) => {
|
|
94
|
+
timer = setTimeout(() => {
|
|
95
|
+
timer = null;
|
|
96
|
+
resolve({ text: null, error: "LLM normalization timed out" });
|
|
97
|
+
}, NORMALIZE_FALLBACK_TIMEOUT_MS);
|
|
98
|
+
timer.unref?.();
|
|
99
|
+
});
|
|
100
|
+
try {
|
|
101
|
+
return await Promise.race([promise, bound]);
|
|
102
|
+
} finally {
|
|
103
|
+
if (timer) clearTimeout(timer);
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
|
|
69
107
|
// Split streamed text into complete spoken sentences. Returns what is ready
|
|
70
108
|
// plus the trailing incomplete fragment to keep buffering. Sentences that
|
|
71
109
|
// look like code dumps (overlong, backticks) are skipped by the caller.
|
|
@@ -473,14 +511,11 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
473
511
|
clearProcessingToast();
|
|
474
512
|
return;
|
|
475
513
|
}
|
|
476
|
-
if (!llmResult
|
|
514
|
+
if (!announceNormalize("Auto", llmResult)) {
|
|
477
515
|
clearProcessingToast();
|
|
478
|
-
logger?.log?.("TTS", `Auto normalization failed: ${llmResult.error}`, "warn");
|
|
479
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
480
516
|
return;
|
|
481
517
|
}
|
|
482
518
|
|
|
483
|
-
logger?.log?.("TTS", `Auto normalization succeeded chars=${llmResult.text.length}`, "debug");
|
|
484
519
|
clearProcessingToast();
|
|
485
520
|
await speakWithSessionPrefix(sessionID, llmResult.text, "Ready for your input.");
|
|
486
521
|
});
|
|
@@ -519,14 +554,11 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
519
554
|
clearProcessingToast();
|
|
520
555
|
return;
|
|
521
556
|
}
|
|
522
|
-
if (!llmResult
|
|
557
|
+
if (!announceNormalize("Manual", llmResult)) {
|
|
523
558
|
clearProcessingToast();
|
|
524
|
-
logger?.log?.("TTS", `Manual normalization failed: ${llmResult.error}`, "warn");
|
|
525
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
526
559
|
return;
|
|
527
560
|
}
|
|
528
561
|
|
|
529
|
-
logger?.log?.("TTS", `Manual normalization succeeded chars=${llmResult.text.length}`, "debug");
|
|
530
562
|
updateProcessingToast("Speaking...");
|
|
531
563
|
await speak(llmResult.text);
|
|
532
564
|
clearProcessingToast();
|
|
@@ -541,7 +573,31 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
541
573
|
|
|
542
574
|
async function normalizeOrLocal(text, systemPrompt, maxTokens) {
|
|
543
575
|
if (ttsLocal) return { text: localSpeechCleanup(text) };
|
|
544
|
-
|
|
576
|
+
const llmResult = await normalizeWithinBound(normalizeForSpeech(text, systemPrompt, maxTokens));
|
|
577
|
+
return resolveSpeechText(text, llmResult);
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
// Shared tail for all three normalize call sites: one informational toast
|
|
581
|
+
// when we are speaking the local fallback, and - when the cleanup left
|
|
582
|
+
// nothing speakable (a pure code dump) - a quiet debug log, because nothing
|
|
583
|
+
// failed there, there was just nothing to read. Returns whether to speak.
|
|
584
|
+
|
|
585
|
+
function announceNormalize(label, result) {
|
|
586
|
+
if (!result.text) {
|
|
587
|
+
logger?.log?.("TTS", `${label}: nothing speakable (${result.error})`, "debug");
|
|
588
|
+
return false;
|
|
589
|
+
}
|
|
590
|
+
if (result.fellBack) {
|
|
591
|
+
logger?.log?.(
|
|
592
|
+
"TTS",
|
|
593
|
+
`${label} narration unavailable (${result.error}); speaking local cleanup`,
|
|
594
|
+
"warn",
|
|
595
|
+
);
|
|
596
|
+
toast("Model unavailable - speaking text as-is");
|
|
597
|
+
return true;
|
|
598
|
+
}
|
|
599
|
+
logger?.log?.("TTS", `${label} normalization succeeded chars=${result.text.length}`, "debug");
|
|
600
|
+
return true;
|
|
545
601
|
}
|
|
546
602
|
|
|
547
603
|
async function speakAssistantTurn() {
|
|
@@ -564,11 +620,9 @@ export function registerTTS(api, kv, complete, prompts, opts, logger, deps = {})
|
|
|
564
620
|
clearProcessingToast();
|
|
565
621
|
return { spoken: false };
|
|
566
622
|
}
|
|
567
|
-
if (!llmResult
|
|
623
|
+
if (!announceNormalize("Conversation", llmResult)) {
|
|
568
624
|
clearProcessingToast();
|
|
569
|
-
|
|
570
|
-
toast(`TTS normalization failed: ${llmResult.error}`, "warning");
|
|
571
|
-
return { spoken: false, error: llmResult.error };
|
|
625
|
+
return { spoken: false };
|
|
572
626
|
}
|
|
573
627
|
|
|
574
628
|
updateProcessingToast("Speaking...");
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bojackduy/opencode-voice",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.9.0",
|
|
4
4
|
"description": "Speech-to-text and text-to-speech for OpenCode. Record voice prompts with whisper-cpp, hear responses via Piper TTS, with LLM normalization through any OpenAI-compatible endpoint.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"opencode",
|