dsh-live-voice 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-live-voice",
3
- "version": "0.2.0",
3
+ "version": "0.2.2",
4
4
  "type": "module",
5
5
  "scripts": {
6
6
  "typecheck": "tsc --noEmit",
@@ -13,17 +13,24 @@
13
13
  "setup:hooks": "git config core.hooksPath .githooks",
14
14
  "prepack": "npm run build"
15
15
  },
16
- "description": "Local-first voice conversations for DSH, with local speech recognition and synthesis and optional external providers.",
16
+ "description": "Local-first hands-free AI voice assistant plugin for DeepSeek Harness (DSH), with speech-to-text, text-to-speech, voice commands, and local speech engines.",
17
17
  "keywords": [
18
18
  "dsh",
19
19
  "dsh-plugin",
20
20
  "deepseek-harness",
21
21
  "local-first",
22
22
  "local-voice",
23
+ "hands-free",
24
+ "hands-free-ai",
25
+ "hands-free-assistant",
26
+ "voice-control",
27
+ "voice-commands",
23
28
  "voice-assistant",
24
29
  "voice-conversation",
30
+ "conversational-ai",
25
31
  "continuous-conversation",
26
32
  "voice-dictation",
33
+ "spoken-prompts",
27
34
  "speech-to-text",
28
35
  "text-to-speech",
29
36
  "speech-recognition",
@@ -68,16 +75,16 @@
68
75
  "url": "git+https://github.com/victorwads/dsh-live-voice.git"
69
76
  },
70
77
  "devDependencies": {
71
- "@types/node": "^26.5.1",
72
- "@types/react": "^19.3.0",
73
- "@types/react-dom": "^19.3.0",
74
- "esbuild": "^0.28.2",
75
- "jsdom": "^30.0.1",
76
- "playwright-core": "^1.63.0",
77
- "prettier": "3.9.6",
78
- "react": "^19.3.0",
79
- "react-dom": "^19.3.0",
80
- "typescript": "^7.0.2"
78
+ "@types/node": "26.6.2",
79
+ "@types/react": "19.3.0",
80
+ "@types/react-dom": "19.3.0",
81
+ "esbuild": "0.28.2",
82
+ "jsdom": "30.1.0",
83
+ "playwright-core": "1.63.0",
84
+ "prettier": "3.9.8",
85
+ "react": "19.3.0",
86
+ "react-dom": "19.3.0",
87
+ "typescript": "7.0.2"
81
88
  },
82
89
  "main": "./lib/server.js",
83
90
  "exports": {
@@ -216,25 +216,26 @@ export function createComponents(React) {
216
216
  const remaining = state.autoSendAt
217
217
  ? Math.max(1, Math.ceil((state.autoSendAt - now) / 1000))
218
218
  : null;
219
- const status = state.answeringQuestion && state.recognizing
220
- ? 'Recognizing answer…'
221
- : state.answeringQuestion && state.listening
222
- ? 'Listening for your answer…'
223
- : remaining
224
- ? `Sending in ${remaining}…`
225
- : state.starting
226
- ? 'Starting microphone…'
227
- : state.paused
228
- ? 'Speech paused'
229
- : state.speaking
230
- ? 'Speaking'
231
- : state.recognizing
232
- ? 'Recognizing speech…'
233
- : state.listening
234
- ? 'Listening — waiting for speech'
235
- : state.conversation
236
- ? 'Conversation idle'
237
- : 'Voice ready';
219
+ const status =
220
+ state.answeringQuestion && state.recognizing
221
+ ? 'Recognizing answer…'
222
+ : state.answeringQuestion && state.listening
223
+ ? 'Listening for your answer…'
224
+ : remaining
225
+ ? `Sending in ${remaining}…`
226
+ : state.starting
227
+ ? 'Starting microphone…'
228
+ : state.paused
229
+ ? 'Speech paused'
230
+ : state.speaking
231
+ ? 'Speaking'
232
+ : state.recognizing
233
+ ? 'Recognizing speech…'
234
+ : state.listening
235
+ ? 'Listening — waiting for speech'
236
+ : state.conversation
237
+ ? 'Conversation idle'
238
+ : 'Voice ready';
238
239
  return h(
239
240
  'div',
240
241
  {
@@ -884,6 +885,28 @@ export function createComponents(React) {
884
885
  null,
885
886
  'Controls how long a pause must last before captured speech is sent for recognition.',
886
887
  ),
888
+ h(
889
+ 'label',
890
+ { className: 'dlv-setting-field' },
891
+ 'Maximum continuous speech (seconds)',
892
+ h('input', {
893
+ type: 'number',
894
+ min: 10,
895
+ max: 300,
896
+ step: 1,
897
+ value: settings.recognitionMaxUtteranceSeconds ?? 60,
898
+ onChange: (event) => {
899
+ const value = Number(event.target.value);
900
+ if (Number.isInteger(value) && value >= 10 && value <= 300)
901
+ invoke('updateSettings', { recognitionMaxUtteranceSeconds: value });
902
+ },
903
+ }),
904
+ h(
905
+ 'small',
906
+ null,
907
+ 'If speech never pauses, start a new transcription chunk after this duration. Default: 60 seconds.',
908
+ ),
909
+ ),
887
910
  h(
888
911
  'div',
889
912
  {
@@ -959,6 +982,22 @@ export function createComponents(React) {
959
982
  ? 'Sending or steering a new user message stops current or paused assistant speech.'
960
983
  : 'Sending another message does not stop the assistant audio you are already hearing.',
961
984
  ),
985
+ h(
986
+ 'label',
987
+ { key: 'hold-to-talk', className: 'dlv-check' },
988
+ h('input', {
989
+ type: 'checkbox',
990
+ checked: settings.holdToTalkEnabled !== false,
991
+ onChange: (event) =>
992
+ invoke('updateSettings', { holdToTalkEnabled: event.target.checked }),
993
+ }),
994
+ ' Hold Control to talk',
995
+ ),
996
+ h(
997
+ 'p',
998
+ { key: 'hold-to-talk-description', className: 'dlv-setting-description' },
999
+ 'While a composer is open, hold Control anywhere on the page to capture speech. Release it to flush queued transcription, wait the configured send delay, queue the message, and close voice capture. Press Escape while holding to cancel.',
1000
+ ),
962
1001
  field('Listening mode', 'mode', [
963
1002
  { value: 'speaker', label: 'Speakers — gated listening' },
964
1003
  { value: 'headphones', label: 'Headphones — open microphone' },
@@ -13,7 +13,12 @@ import { QwenHttpRecognitionEngine } from '../engines/recognition/qwen-http.ts';
13
13
  import { QwenHttpSpeakingEngine } from '../engines/speaking/qwen-http.ts';
14
14
  import { createComponents } from './components.ts';
15
15
  import { styles } from './styles.ts';
16
- import { assistantMessages, addressedTurn, latestUserSequence, pendingQuestionSpeech } from './chat.ts';
16
+ import {
17
+ assistantMessages,
18
+ addressedTurn,
19
+ latestUserSequence,
20
+ pendingQuestionSpeech,
21
+ } from './chat.ts';
17
22
 
18
23
  export const inject = ['slots', 'connection', 'uiConversation', 'uiSession'];
19
24
  export function apply(ctx) {
@@ -27,16 +32,27 @@ export function apply(ctx) {
27
32
  // composer should inherit the user's explicit choice to remain in voice mode.
28
33
  let voiceModeActive = false;
29
34
  const ownership = new VoiceOwnership();
35
+ const recognitionSettingKeys = new Set([
36
+ 'recognitionEngine',
37
+ 'recognitionProcessLocally',
38
+ 'recognitionAutoInstall',
39
+ 'voiceDetectionPreset',
40
+ 'recognitionMaxUtteranceSeconds',
41
+ ]);
42
+ const changesRecognition = (next) =>
43
+ Object.keys(next).some((key) => recognitionSettingKeys.has(key));
30
44
  const recognitionFor = (settings, meter) =>
31
45
  settings.recognitionEngine === 'qwen-http'
32
46
  ? new QwenHttpRecognitionEngine({
33
47
  meter,
34
48
  voiceDetectionPreset: settings.voiceDetectionPreset,
49
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
35
50
  })
36
51
  : settings.recognitionEngine === 'whisper-http'
37
52
  ? new WhisperHttpRecognitionEngine({
38
53
  meter,
39
54
  voiceDetectionPreset: settings.voiceDetectionPreset,
55
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
40
56
  })
41
57
  : new BrowserRecognitionEngine({
42
58
  processLocally: settings.recognitionProcessLocally,
@@ -116,7 +132,8 @@ export function apply(ctx) {
116
132
  if (capture.index < capture.interaction.questions.length) {
117
133
  capture.answering = false;
118
134
  const text = pendingQuestionSpeech(capture.interaction, capture.index);
119
- if (text) run(entry.controller, entry.controller.speak(text, capture.interaction.key));
135
+ if (text)
136
+ run(entry.controller, entry.controller.speak(text, capture.interaction.key));
120
137
  } else {
121
138
  entry.questionCapture = null;
122
139
  entry.controller.patch({ answeringQuestion: false });
@@ -172,9 +189,11 @@ export function apply(ctx) {
172
189
  refresh();
173
190
  for (const listener of entry.chatListeners) listener();
174
191
  });
192
+ // DSH 0.1.6 may omit this legacy store. Keep the remaining voice controls
193
+ // available while structured-question integration is unavailable.
175
194
  const pendingInteractions = ctx.uiSession.pendingInteractions;
176
195
  const refreshPendingQuestion = (baseline = false) => {
177
- if (disposed || entry.closed) return;
196
+ if (disposed || entry.closed || !pendingInteractions) return;
178
197
  const interaction = pendingInteractions.getSnapshot().get(sessionId);
179
198
  const key = interaction?.kind === 'question' ? interaction.key : null;
180
199
  if (!key || (entry.questionCapture && entry.questionCapture.interaction.key !== key)) {
@@ -194,21 +213,27 @@ export function apply(ctx) {
194
213
  run(controller, controller.speak(text, key));
195
214
  }
196
215
  };
197
- refreshPendingQuestion(true);
198
- entry.unsubscribePendingQuestion = pendingInteractions.subscribe(refreshPendingQuestion);
216
+ if (pendingInteractions) {
217
+ refreshPendingQuestion(true);
218
+ entry.unsubscribePendingQuestion = pendingInteractions.subscribe(refreshPendingQuestion);
219
+ }
199
220
  const update = controller.updateSettings.bind(controller);
200
- controller.updateSettings = (next) => {
201
- if (disposed || entry.closed) return;
221
+ entry.applySettings = (next) => {
202
222
  update(next);
203
223
  engineBrowser.lang = controller.getSnapshot().settings.lang;
204
224
  engineQwen.lang = controller.getSnapshot().settings.lang;
225
+ run(controller, controller.refreshCapabilities());
226
+ };
227
+ controller.updateSettings = (next) => {
228
+ if (disposed || entry.closed) return;
229
+ entry.applySettings(next);
230
+ const settings = controller.getSnapshot().settings;
205
231
  try {
206
- localStorage.setItem(
207
- 'dsh-live-voice.settings',
208
- JSON.stringify(controller.getSnapshot().settings),
209
- );
232
+ localStorage.setItem('dsh-live-voice.settings', JSON.stringify(settings));
210
233
  } catch {}
211
- run(controller, controller.refreshCapabilities());
234
+ for (const other of controllers.values()) {
235
+ if (other !== entry && !other.closed) other.applySettings(settings);
236
+ }
212
237
  };
213
238
  // Cancel only this entry's queued acquisition. Global cancellation here would
214
239
  // invalidate a newer session while its predecessor is being unmounted.
@@ -220,7 +245,7 @@ export function apply(ctx) {
220
245
  return original(...args);
221
246
  };
222
247
  }
223
- for (const method of ['startDictation', 'startConversation', 'speak']) {
248
+ for (const method of ['startDictation', 'startHoldToTalk', 'startConversation', 'speak']) {
224
249
  const original = controller[method].bind(controller);
225
250
  controller[method] = (...args) => {
226
251
  if (disposed || entry.closed || !entry.refs) return Promise.resolve();
@@ -343,13 +368,11 @@ export function apply(ctx) {
343
368
  let settingsRevision = 0;
344
369
  c.updateSettings = async (next) => {
345
370
  const revision = ++settingsRevision;
346
- const previousEngine = c.getSnapshot().settings.recognitionEngine;
347
371
  update(next);
348
372
  browser.lang = c.getSnapshot().settings.lang;
349
373
  qwen.lang = c.getSnapshot().settings.lang;
350
374
  const settings = c.getSnapshot().settings;
351
- if (previousEngine !== settings.recognitionEngine)
352
- c.replaceRecognition(recognitionFor(settings, c.meter));
375
+ if (changesRecognition(next)) c.replaceRecognition(recognitionFor(settings, c.meter));
353
376
  localStorage.setItem('dsh-live-voice.settings', JSON.stringify(settings));
354
377
  run(c, c.refreshCapabilities());
355
378
  const active = [...controllers.values()];
@@ -360,11 +383,9 @@ export function apply(ctx) {
360
383
  if (revision !== settingsRevision || disposed || c.disposed) return;
361
384
  for (const entry of controllers.values()) {
362
385
  if (entry.closed) continue;
363
- if (
364
- entry.controller.getSnapshot().settings.recognitionEngine !== settings.recognitionEngine
365
- )
386
+ if (changesRecognition(next))
366
387
  entry.controller.replaceRecognition(recognitionFor(settings, entry.controller.meter));
367
- entry.controller.updateSettings(settings);
388
+ entry.applySettings(settings);
368
389
  run(entry.controller, entry.controller.refreshCapabilities());
369
390
  }
370
391
  };
@@ -480,28 +501,49 @@ export function apply(ctx) {
480
501
  for (const entry of controllers.values())
481
502
  run(entry.controller, entry.controller.endConversation());
482
503
  };
483
- const onKey = (event) => {
484
- if (
485
- disposed ||
486
- event.defaultPrevented ||
487
- !event.ctrlKey ||
488
- !event.shiftKey ||
489
- event.code !== 'Space' ||
490
- event.repeat
491
- )
492
- return;
504
+ let holdToTalk = null;
505
+ const candidate = () => {
493
506
  const candidates = [...controllers.values()].filter(
494
507
  (entry) => entry.buttons > 0 && entry.composers.size > 0,
495
508
  );
496
- if (candidates.length !== 1) return;
497
- event.preventDefault();
498
- const c = candidates[0].controller;
499
- run(
500
- c,
501
- c.getSnapshot().listening || c.getSnapshot().starting
502
- ? c.stopListening()
503
- : c.startDictation(),
504
- );
509
+ return candidates.length === 1 ? candidates[0] : null;
510
+ };
511
+ const releaseHoldToTalk = () => {
512
+ const entry = holdToTalk;
513
+ holdToTalk = null;
514
+ if (!entry || entry.closed) return;
515
+ run(entry.controller, entry.controller.releaseHoldToTalk());
516
+ };
517
+ const onKeyDown = (event) => {
518
+ if (disposed || event.defaultPrevented || event.repeat) return;
519
+ if (event.key === 'Escape' && holdToTalk) {
520
+ const entry = holdToTalk;
521
+ holdToTalk = null;
522
+ event.preventDefault();
523
+ run(entry.controller, entry.controller.cancelDictation());
524
+ return;
525
+ }
526
+ if (event.ctrlKey && event.shiftKey && event.code === 'Space') {
527
+ const entry = candidate();
528
+ if (!entry) return;
529
+ event.preventDefault();
530
+ const c = entry.controller;
531
+ run(
532
+ c,
533
+ c.getSnapshot().listening || c.getSnapshot().starting
534
+ ? c.stopListening()
535
+ : c.startDictation(),
536
+ );
537
+ return;
538
+ }
539
+ if (event.key !== 'Control' || event.altKey || event.metaKey || event.shiftKey) return;
540
+ const entry = candidate();
541
+ if (!entry || !entry.controller.getSnapshot().settings.holdToTalkEnabled) return;
542
+ holdToTalk = entry;
543
+ run(entry.controller, entry.controller.startHoldToTalk());
544
+ };
545
+ const onKeyUp = (event) => {
546
+ if (event.key === 'Control') releaseHoldToTalk();
505
547
  };
506
548
  const refreshCapabilities = () => {
507
549
  for (const entry of controllers.values())
@@ -514,7 +556,9 @@ export function apply(ctx) {
514
556
  window.speechSynthesis?.addEventListener?.('voiceschanged', refreshCapabilities);
515
557
  navigator.mediaDevices?.addEventListener?.('devicechange', refreshCapabilities);
516
558
  document.addEventListener('visibilitychange', visibilityChanged);
517
- document.addEventListener('keydown', onKey);
559
+ document.addEventListener('keydown', onKeyDown);
560
+ document.addEventListener('keyup', onKeyUp);
561
+ window.addEventListener('blur', releaseHoldToTalk);
518
562
  window.addEventListener('pagehide', stop);
519
563
  return () => {
520
564
  disposed = true;
@@ -522,7 +566,9 @@ export function apply(ctx) {
522
566
  window.speechSynthesis?.removeEventListener?.('voiceschanged', refreshCapabilities);
523
567
  navigator.mediaDevices?.removeEventListener?.('devicechange', refreshCapabilities);
524
568
  document.removeEventListener('visibilitychange', visibilityChanged);
525
- document.removeEventListener('keydown', onKey);
569
+ document.removeEventListener('keydown', onKeyDown);
570
+ document.removeEventListener('keyup', onKeyUp);
571
+ window.removeEventListener('blur', releaseHoldToTalk);
526
572
  window.removeEventListener('pagehide', stop);
527
573
  for (const entry of controllers.values()) retire(entry);
528
574
  };
@@ -46,7 +46,8 @@ export class VoiceCoordinator {
46
46
  conversation: false,
47
47
  listening: false,
48
48
  recognizing: false,
49
- muted: false,
49
+ pendingTranscriptions: 0,
50
+ muted: normalizeSettings(settings).microphoneEnabled === false,
50
51
  speaking: false,
51
52
  paused: false,
52
53
  starting: false,
@@ -60,6 +61,7 @@ export class VoiceCoordinator {
60
61
  };
61
62
  this.autoSendTimer = null;
62
63
  this.autoSendDraft = null;
64
+ this.holdToTalkRelease = false;
63
65
  this.interruptionTimer = null;
64
66
  this.interruptionTranscriptConfirmed = false;
65
67
  this.interruptionPausedSpeech = false;
@@ -116,7 +118,12 @@ export class VoiceCoordinator {
116
118
  autoInstallLocalPack: settings.recognitionAutoInstall,
117
119
  voiceDetectionPreset: settings.voiceDetectionPreset,
118
120
  });
119
- this.patch({ settings, error: null });
121
+ this.patch({
122
+ settings,
123
+ muted: settings.microphoneEnabled === false,
124
+ recognizing: settings.microphoneEnabled === false ? false : this.snapshot.recognizing,
125
+ error: null,
126
+ });
120
127
  if (settings.sendingMode === 'manual') this.cancelAutoSend();
121
128
  if (Object.hasOwn(next, 'announceAssistantMessages') && !settings.announceAssistantMessages) {
122
129
  this.queue = [];
@@ -181,20 +188,42 @@ export class VoiceCoordinator {
181
188
  startDictation() {
182
189
  return this.startListening(false);
183
190
  }
191
+ startHoldToTalk() {
192
+ this.holdToTalkRelease = false;
193
+ return this.startListening(false);
194
+ }
195
+ async releaseHoldToTalk() {
196
+ if (this.disposed || (!this.snapshot.listening && !this.snapshot.starting)) return;
197
+ this.holdToTalkRelease = true;
198
+ if (this.snapshot.starting) return;
199
+ await this.recognition.finish?.();
200
+ if (this.snapshot.pendingTranscriptions === 0) this._finishHoldToTalk();
201
+ }
202
+ _finishHoldToTalk() {
203
+ if (!this.holdToTalkRelease) return;
204
+ this.holdToTalkRelease = false;
205
+ const draft = this.composer.getDraft();
206
+ if (!draft.trim()) {
207
+ void this.stopListening();
208
+ return;
209
+ }
210
+ this.scheduleAutoSend(draft, { force: true, stopAfter: true });
211
+ }
184
212
  startConversation() {
185
213
  return this.startListening(true);
186
214
  }
187
215
  muteListening() {
188
- if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
189
216
  this.cancelAutoSend();
190
217
  this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
191
218
  this.transcript.reset();
192
- this.patch({ muted: true, recognizing: false });
219
+ this.updateSettings({ microphoneEnabled: false });
220
+ if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
193
221
  }
194
222
  resumeListeningInput() {
223
+ this.transcript.reset();
224
+ this.updateSettings({ microphoneEnabled: true });
195
225
  if (this.snapshot.listening || this.snapshot.starting) {
196
- this.transcript.reset();
197
- this.patch({ muted: false, recognizing: false });
226
+ this.patch({ recognizing: false });
198
227
  return;
199
228
  }
200
229
  return this.startListening(this.snapshot.conversation);
@@ -263,6 +292,14 @@ export class VoiceCoordinator {
263
292
  onResult: (result) => {
264
293
  if (valid()) this.onResult(result);
265
294
  },
295
+ onProcessingChange: ({ pending = 0 } = {}) => {
296
+ if (!valid()) return;
297
+ const count = Number.isSafeInteger(pending) && pending >= 0 ? pending : 0;
298
+ this.patch({ pendingTranscriptions: count });
299
+ if (count > 0) this.cancelAutoSend();
300
+ else if (this.holdToTalkRelease) this._finishHoldToTalk();
301
+ else this.maybeScheduleAutoSend();
302
+ },
266
303
  onActivity: (active) => {
267
304
  if (!valid()) return;
268
305
  if (active) {
@@ -288,6 +325,7 @@ export class VoiceCoordinator {
288
325
  return;
289
326
  }
290
327
  this.patch({ listening: true, starting: false });
328
+ if (this.holdToTalkRelease) void this.releaseHoldToTalk();
291
329
  this._drain();
292
330
  } catch (error) {
293
331
  try {
@@ -372,8 +410,10 @@ export class VoiceCoordinator {
372
410
  }
373
411
  // Pending structured questions own recognition while waiting for a hands-free
374
412
  // answer. Keep their interim and final text out of the normal chat composer.
375
- if (typeof this.composer.handleQuestionResult === 'function' &&
376
- this.composer.handleQuestionResult({ final, interim })) {
413
+ if (
414
+ typeof this.composer.handleQuestionResult === 'function' &&
415
+ this.composer.handleQuestionResult({ final, interim })
416
+ ) {
377
417
  this.cancelAutoSend();
378
418
  this.transcript.reset();
379
419
  this.patch({ recognizing: !!interim });
@@ -435,7 +475,7 @@ export class VoiceCoordinator {
435
475
  }
436
476
  const next = this.transcript.update(this.composer.getDraft(), final, true);
437
477
  this.composer.setDraft(next);
438
- if (this.snapshot.settings.sendingMode !== 'manual') this.scheduleAutoSend(next);
478
+ this.maybeScheduleAutoSend(next);
439
479
  }
440
480
  if (interim) {
441
481
  this.cancelAutoSend();
@@ -447,7 +487,16 @@ export class VoiceCoordinator {
447
487
  this.patch({ recognizing: !!interim });
448
488
  this._drain();
449
489
  }
450
- scheduleAutoSend(draft) {
490
+ maybeScheduleAutoSend(draft = this.composer.getDraft()) {
491
+ if (
492
+ this.snapshot.settings.sendingMode === 'manual' ||
493
+ this.snapshot.recognizing ||
494
+ this.snapshot.pendingTranscriptions > 0
495
+ )
496
+ return;
497
+ this.scheduleAutoSend(draft);
498
+ }
499
+ scheduleAutoSend(draft, { force = false, stopAfter = false } = {}) {
451
500
  this.cancelAutoSend();
452
501
  if (!draft.trim() || typeof this.composer.submit !== 'function') return;
453
502
  const delay = this.snapshot.settings.autoSendDelaySeconds * 1000;
@@ -460,16 +509,19 @@ export class VoiceCoordinator {
460
509
  this.patch({ autoSendAt: null });
461
510
  if (
462
511
  !this.disposed &&
463
- this.snapshot.settings.sendingMode !== 'manual' &&
512
+ (force || this.snapshot.settings.sendingMode !== 'manual') &&
464
513
  expected === this.composer.getDraft()
465
514
  ) {
466
515
  try {
467
- this.composer.submit(this.snapshot.settings.sendingMode === 'steer' ? 'steer' : 'queue');
516
+ this.composer.submit(
517
+ this.snapshot.settings.sendingMode === 'steer' && !force ? 'steer' : 'queue',
518
+ );
468
519
  // Web Speech keeps a cumulative native result list for the lifetime of
469
520
  // one recognition instance. Start a fresh instance at the turn boundary
470
521
  // so the sent utterance cannot prefix the next one.
471
522
  this.transcript.reset();
472
523
  this.recognition.reset?.();
524
+ if (stopAfter) void this.stopListening();
473
525
  } catch (error) {
474
526
  this.patch({ error: message(error) });
475
527
  }
@@ -486,10 +538,11 @@ export class VoiceCoordinator {
486
538
  if (this.autoSendDraft !== null && draft !== this.autoSendDraft) this.cancelAutoSend();
487
539
  }
488
540
  stopListening() {
541
+ this.holdToTalkRelease = false;
489
542
  this.cancelAutoSend();
490
543
  ++this.epoch;
491
544
  this.inputController?.abort();
492
- this.patch({ listening: false, recognizing: false, starting: false });
545
+ this.patch({ listening: false, recognizing: false, pendingTranscriptions: 0, starting: false });
493
546
  return this._input(async () => {
494
547
  try {
495
548
  await this._releaseInput();
@@ -36,6 +36,9 @@ export const defaultSettings = Object.freeze({
36
36
  recognitionProcessLocally: true,
37
37
  recognitionAutoInstall: true,
38
38
  voiceDetectionPreset: 'natural',
39
+ recognitionMaxUtteranceSeconds: 60,
40
+ microphoneEnabled: true,
41
+ holdToTalkEnabled: true,
39
42
  announceAssistantMessages: true,
40
43
  interruptSpeechOnUserMessage: false,
41
44
  sendingMode: 'manual',
@@ -93,6 +96,20 @@ export function normalizeSettings(value) {
93
96
  voiceDetectionPreset: Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset)
94
97
  ? source.voiceDetectionPreset
95
98
  : defaultSettings.voiceDetectionPreset,
99
+ recognitionMaxUtteranceSeconds:
100
+ Number.isInteger(source.recognitionMaxUtteranceSeconds) &&
101
+ source.recognitionMaxUtteranceSeconds >= 10 &&
102
+ source.recognitionMaxUtteranceSeconds <= 300
103
+ ? source.recognitionMaxUtteranceSeconds
104
+ : defaultSettings.recognitionMaxUtteranceSeconds,
105
+ microphoneEnabled:
106
+ typeof source.microphoneEnabled === 'boolean'
107
+ ? source.microphoneEnabled
108
+ : defaultSettings.microphoneEnabled,
109
+ holdToTalkEnabled:
110
+ typeof source.holdToTalkEnabled === 'boolean'
111
+ ? source.holdToTalkEnabled
112
+ : defaultSettings.holdToTalkEnabled,
96
113
  announceAssistantMessages:
97
114
  typeof source.announceAssistantMessages === 'boolean'
98
115
  ? source.announceAssistantMessages
@@ -141,6 +141,9 @@ export class BrowserRecognitionEngine {
141
141
  timer: null,
142
142
  recognition: null,
143
143
  speaking: false,
144
+ finishing: false,
145
+ finishPromise: null,
146
+ resolveFinish: null,
144
147
  };
145
148
  this.session = session;
146
149
  const cancelled = new Promise((resolve, reject) => {
@@ -244,6 +247,12 @@ export class BrowserRecognitionEngine {
244
247
  session.recognition = null;
245
248
  this._activity(session, false);
246
249
  if (this.session !== session) return;
250
+ if (session.finishing) {
251
+ this.session = null;
252
+ session.signal?.removeEventListener('abort', session.cancel);
253
+ session.resolveFinish?.();
254
+ return;
255
+ }
247
256
  if (session.restarts >= this.maxRestarts) {
248
257
  this._fail(
249
258
  session,
@@ -273,6 +282,34 @@ export class BrowserRecognitionEngine {
273
282
  recognition.start();
274
283
  }
275
284
 
285
+ finish() {
286
+ const session = this.session;
287
+ if (!session) return Promise.resolve();
288
+ if (session.finishPromise) return session.finishPromise;
289
+ session.finishing = true;
290
+ session.finishPromise = new Promise((resolve) => {
291
+ session.resolveFinish = resolve;
292
+ });
293
+ if (session.timer !== null) {
294
+ (this.globals.clearTimeout ?? globalThis.clearTimeout)(session.timer);
295
+ session.timer = null;
296
+ this.session = null;
297
+ session.resolveFinish();
298
+ return session.finishPromise;
299
+ }
300
+ try {
301
+ session.recognition?.stop();
302
+ if (!session.recognition) {
303
+ this.session = null;
304
+ session.resolveFinish();
305
+ }
306
+ } catch {
307
+ this.session = null;
308
+ session.resolveFinish();
309
+ }
310
+ return session.finishPromise;
311
+ }
312
+
276
313
  reset() {
277
314
  const session = this.session;
278
315
  const recognition = session?.recognition;
@@ -310,5 +347,6 @@ export class BrowserRecognitionEngine {
310
347
  /* Browser already ended capture. */
311
348
  }
312
349
  this._activity(session, false);
350
+ session.resolveFinish?.();
313
351
  }
314
352
  }