dsh-live-voice 0.2.0 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/client.js CHANGED
@@ -114,6 +114,9 @@ var defaultSettings = Object.freeze({
114
114
  recognitionProcessLocally: true,
115
115
  recognitionAutoInstall: true,
116
116
  voiceDetectionPreset: "natural",
117
+ recognitionMaxUtteranceSeconds: 60,
118
+ microphoneEnabled: true,
119
+ holdToTalkEnabled: true,
117
120
  announceAssistantMessages: true,
118
121
  interruptSpeechOnUserMessage: false,
119
122
  sendingMode: "manual",
@@ -149,6 +152,9 @@ function normalizeSettings(value) {
149
152
  recognitionProcessLocally: typeof source.recognitionProcessLocally === "boolean" ? source.recognitionProcessLocally : defaultSettings.recognitionProcessLocally,
150
153
  recognitionAutoInstall: typeof source.recognitionAutoInstall === "boolean" ? source.recognitionAutoInstall : defaultSettings.recognitionAutoInstall,
151
154
  voiceDetectionPreset: Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset) ? source.voiceDetectionPreset : defaultSettings.voiceDetectionPreset,
155
+ recognitionMaxUtteranceSeconds: Number.isInteger(source.recognitionMaxUtteranceSeconds) && source.recognitionMaxUtteranceSeconds >= 10 && source.recognitionMaxUtteranceSeconds <= 300 ? source.recognitionMaxUtteranceSeconds : defaultSettings.recognitionMaxUtteranceSeconds,
156
+ microphoneEnabled: typeof source.microphoneEnabled === "boolean" ? source.microphoneEnabled : defaultSettings.microphoneEnabled,
157
+ holdToTalkEnabled: typeof source.holdToTalkEnabled === "boolean" ? source.holdToTalkEnabled : defaultSettings.holdToTalkEnabled,
152
158
  announceAssistantMessages: typeof source.announceAssistantMessages === "boolean" ? source.announceAssistantMessages : defaultSettings.announceAssistantMessages,
153
159
  interruptSpeechOnUserMessage: typeof source.interruptSpeechOnUserMessage === "boolean" ? source.interruptSpeechOnUserMessage : defaultSettings.interruptSpeechOnUserMessage,
154
160
  sendingMode: source.sendingMode === "automatic" ? "queue" : ["manual", "queue", "steer"].includes(source.sendingMode) ? source.sendingMode : defaultSettings.sendingMode,
@@ -271,7 +277,8 @@ var VoiceCoordinator = class {
271
277
  conversation: false,
272
278
  listening: false,
273
279
  recognizing: false,
274
- muted: false,
280
+ pendingTranscriptions: 0,
281
+ muted: normalizeSettings(settings).microphoneEnabled === false,
275
282
  speaking: false,
276
283
  paused: false,
277
284
  starting: false,
@@ -285,6 +292,7 @@ var VoiceCoordinator = class {
285
292
  };
286
293
  this.autoSendTimer = null;
287
294
  this.autoSendDraft = null;
295
+ this.holdToTalkRelease = false;
288
296
  this.interruptionTimer = null;
289
297
  this.interruptionTranscriptConfirmed = false;
290
298
  this.interruptionPausedSpeech = false;
@@ -338,7 +346,12 @@ var VoiceCoordinator = class {
338
346
  autoInstallLocalPack: settings.recognitionAutoInstall,
339
347
  voiceDetectionPreset: settings.voiceDetectionPreset
340
348
  });
341
- this.patch({ settings, error: null });
349
+ this.patch({
350
+ settings,
351
+ muted: settings.microphoneEnabled === false,
352
+ recognizing: settings.microphoneEnabled === false ? false : this.snapshot.recognizing,
353
+ error: null
354
+ });
342
355
  if (settings.sendingMode === "manual") this.cancelAutoSend();
343
356
  if (Object.hasOwn(next, "announceAssistantMessages") && !settings.announceAssistantMessages) {
344
357
  this.queue = [];
@@ -397,20 +410,42 @@ var VoiceCoordinator = class {
397
410
  startDictation() {
398
411
  return this.startListening(false);
399
412
  }
413
+ startHoldToTalk() {
414
+ this.holdToTalkRelease = false;
415
+ return this.startListening(false);
416
+ }
417
+ async releaseHoldToTalk() {
418
+ if (this.disposed || !this.snapshot.listening && !this.snapshot.starting) return;
419
+ this.holdToTalkRelease = true;
420
+ if (this.snapshot.starting) return;
421
+ await this.recognition.finish?.();
422
+ if (this.snapshot.pendingTranscriptions === 0) this._finishHoldToTalk();
423
+ }
424
+ _finishHoldToTalk() {
425
+ if (!this.holdToTalkRelease) return;
426
+ this.holdToTalkRelease = false;
427
+ const draft = this.composer.getDraft();
428
+ if (!draft.trim()) {
429
+ void this.stopListening();
430
+ return;
431
+ }
432
+ this.scheduleAutoSend(draft, { force: true, stopAfter: true });
433
+ }
400
434
  startConversation() {
401
435
  return this.startListening(true);
402
436
  }
403
437
  muteListening() {
404
- if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
405
438
  this.cancelAutoSend();
406
439
  this.composer.setDraft(this.transcript.update(this.composer.getDraft(), "", true));
407
440
  this.transcript.reset();
408
- this.patch({ muted: true, recognizing: false });
441
+ this.updateSettings({ microphoneEnabled: false });
442
+ if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
409
443
  }
410
444
  resumeListeningInput() {
445
+ this.transcript.reset();
446
+ this.updateSettings({ microphoneEnabled: true });
411
447
  if (this.snapshot.listening || this.snapshot.starting) {
412
- this.transcript.reset();
413
- this.patch({ muted: false, recognizing: false });
448
+ this.patch({ recognizing: false });
414
449
  return;
415
450
  }
416
451
  return this.startListening(this.snapshot.conversation);
@@ -475,6 +510,14 @@ var VoiceCoordinator = class {
475
510
  onResult: (result) => {
476
511
  if (valid()) this.onResult(result);
477
512
  },
513
+ onProcessingChange: ({ pending = 0 } = {}) => {
514
+ if (!valid()) return;
515
+ const count = Number.isSafeInteger(pending) && pending >= 0 ? pending : 0;
516
+ this.patch({ pendingTranscriptions: count });
517
+ if (count > 0) this.cancelAutoSend();
518
+ else if (this.holdToTalkRelease) this._finishHoldToTalk();
519
+ else this.maybeScheduleAutoSend();
520
+ },
478
521
  onActivity: (active) => {
479
522
  if (!valid()) return;
480
523
  if (active) {
@@ -499,6 +542,7 @@ var VoiceCoordinator = class {
499
542
  return;
500
543
  }
501
544
  this.patch({ listening: true, starting: false });
545
+ if (this.holdToTalkRelease) void this.releaseHoldToTalk();
502
546
  this._drain();
503
547
  } catch (error) {
504
548
  try {
@@ -611,7 +655,7 @@ var VoiceCoordinator = class {
611
655
  }
612
656
  const next = this.transcript.update(this.composer.getDraft(), final, true);
613
657
  this.composer.setDraft(next);
614
- if (this.snapshot.settings.sendingMode !== "manual") this.scheduleAutoSend(next);
658
+ this.maybeScheduleAutoSend(next);
615
659
  }
616
660
  if (interim) {
617
661
  this.cancelAutoSend();
@@ -622,7 +666,12 @@ var VoiceCoordinator = class {
622
666
  this.patch({ recognizing: !!interim });
623
667
  this._drain();
624
668
  }
625
- scheduleAutoSend(draft) {
669
+ maybeScheduleAutoSend(draft = this.composer.getDraft()) {
670
+ if (this.snapshot.settings.sendingMode === "manual" || this.snapshot.recognizing || this.snapshot.pendingTranscriptions > 0)
671
+ return;
672
+ this.scheduleAutoSend(draft);
673
+ }
674
+ scheduleAutoSend(draft, { force = false, stopAfter = false } = {}) {
626
675
  this.cancelAutoSend();
627
676
  if (!draft.trim() || typeof this.composer.submit !== "function") return;
628
677
  const delay = this.snapshot.settings.autoSendDelaySeconds * 1e3;
@@ -633,11 +682,14 @@ var VoiceCoordinator = class {
633
682
  const expected = this.autoSendDraft;
634
683
  this.autoSendDraft = null;
635
684
  this.patch({ autoSendAt: null });
636
- if (!this.disposed && this.snapshot.settings.sendingMode !== "manual" && expected === this.composer.getDraft()) {
685
+ if (!this.disposed && (force || this.snapshot.settings.sendingMode !== "manual") && expected === this.composer.getDraft()) {
637
686
  try {
638
- this.composer.submit(this.snapshot.settings.sendingMode === "steer" ? "steer" : "queue");
687
+ this.composer.submit(
688
+ this.snapshot.settings.sendingMode === "steer" && !force ? "steer" : "queue"
689
+ );
639
690
  this.transcript.reset();
640
691
  this.recognition.reset?.();
692
+ if (stopAfter) void this.stopListening();
641
693
  } catch (error) {
642
694
  this.patch({ error: message(error) });
643
695
  }
@@ -654,10 +706,11 @@ var VoiceCoordinator = class {
654
706
  if (this.autoSendDraft !== null && draft !== this.autoSendDraft) this.cancelAutoSend();
655
707
  }
656
708
  stopListening() {
709
+ this.holdToTalkRelease = false;
657
710
  this.cancelAutoSend();
658
711
  ++this.epoch;
659
712
  this.inputController?.abort();
660
- this.patch({ listening: false, recognizing: false, starting: false });
713
+ this.patch({ listening: false, recognizing: false, pendingTranscriptions: 0, starting: false });
661
714
  return this._input(async () => {
662
715
  try {
663
716
  await this._releaseInput();
@@ -1381,7 +1434,10 @@ var BrowserRecognitionEngine = class {
1381
1434
  restarts: 0,
1382
1435
  timer: null,
1383
1436
  recognition: null,
1384
- speaking: false
1437
+ speaking: false,
1438
+ finishing: false,
1439
+ finishPromise: null,
1440
+ resolveFinish: null
1385
1441
  };
1386
1442
  this.session = session;
1387
1443
  const cancelled3 = new Promise((resolve, reject) => {
@@ -1479,6 +1535,12 @@ var BrowserRecognitionEngine = class {
1479
1535
  session.recognition = null;
1480
1536
  this._activity(session, false);
1481
1537
  if (this.session !== session) return;
1538
+ if (session.finishing) {
1539
+ this.session = null;
1540
+ session.signal?.removeEventListener("abort", session.cancel);
1541
+ session.resolveFinish?.();
1542
+ return;
1543
+ }
1482
1544
  if (session.restarts >= this.maxRestarts) {
1483
1545
  this._fail(
1484
1546
  session,
@@ -1507,6 +1569,33 @@ var BrowserRecognitionEngine = class {
1507
1569
  };
1508
1570
  recognition.start();
1509
1571
  }
1572
+ finish() {
1573
+ const session = this.session;
1574
+ if (!session) return Promise.resolve();
1575
+ if (session.finishPromise) return session.finishPromise;
1576
+ session.finishing = true;
1577
+ session.finishPromise = new Promise((resolve) => {
1578
+ session.resolveFinish = resolve;
1579
+ });
1580
+ if (session.timer !== null) {
1581
+ (this.globals.clearTimeout ?? globalThis.clearTimeout)(session.timer);
1582
+ session.timer = null;
1583
+ this.session = null;
1584
+ session.resolveFinish();
1585
+ return session.finishPromise;
1586
+ }
1587
+ try {
1588
+ session.recognition?.stop();
1589
+ if (!session.recognition) {
1590
+ this.session = null;
1591
+ session.resolveFinish();
1592
+ }
1593
+ } catch {
1594
+ this.session = null;
1595
+ session.resolveFinish();
1596
+ }
1597
+ return session.finishPromise;
1598
+ }
1510
1599
  reset() {
1511
1600
  const session = this.session;
1512
1601
  const recognition = session?.recognition;
@@ -1541,6 +1630,7 @@ var BrowserRecognitionEngine = class {
1541
1630
  } catch {
1542
1631
  }
1543
1632
  this._activity(session, false);
1633
+ session.resolveFinish?.();
1544
1634
  }
1545
1635
  };
1546
1636
 
@@ -1577,12 +1667,18 @@ function encodeMonoPcm16Wav(samples, inputRate) {
1577
1667
  return buffer;
1578
1668
  }
1579
1669
  var WhisperHttpRecognitionEngine = class {
1580
- constructor({ globals = globalThis, meter, voiceDetectionPreset = "natural" } = {}) {
1670
+ constructor({
1671
+ globals = globalThis,
1672
+ meter,
1673
+ voiceDetectionPreset = "natural",
1674
+ maxUtteranceSeconds = 60
1675
+ } = {}) {
1581
1676
  this.g = globals;
1582
1677
  this.meter = meter;
1583
1678
  this.session = null;
1584
1679
  this.lang = "pt-BR";
1585
1680
  this.voiceDetectionPreset = voiceDetectionPreset;
1681
+ this.maxUtteranceSeconds = maxUtteranceSeconds;
1586
1682
  this.route = ROUTE;
1587
1683
  }
1588
1684
  get segmentation() {
@@ -1608,7 +1704,14 @@ var WhisperHttpRecognitionEngine = class {
1608
1704
  };
1609
1705
  }
1610
1706
  }
1611
- async start({ lang = this.lang, signal, onResult, onActivity, onError } = {}) {
1707
+ async start({
1708
+ lang = this.lang,
1709
+ signal,
1710
+ onResult,
1711
+ onActivity,
1712
+ onError,
1713
+ onProcessingChange
1714
+ } = {}) {
1612
1715
  await this.stop();
1613
1716
  if (signal?.aborted) throw abortError3();
1614
1717
  const context = this.meter?.context, source = this.meter?.source;
@@ -1628,16 +1731,24 @@ var WhisperHttpRecognitionEngine = class {
1628
1731
  samples: 0,
1629
1732
  voiced: false,
1630
1733
  silence: 0,
1631
- inflight: /* @__PURE__ */ new Set(),
1734
+ transcriptionQueue: [],
1735
+ activeRequest: null,
1736
+ draining: false,
1632
1737
  onResult,
1633
1738
  onActivity,
1634
1739
  onError,
1740
+ onProcessingChange,
1635
1741
  lang,
1636
1742
  signal
1637
1743
  };
1638
1744
  this.session = session;
1639
1745
  const valid = () => this.session === session && !signal?.aborted;
1640
- const submit = async () => {
1746
+ const notifyProcessing = () => session.onProcessingChange?.({
1747
+ queued: session.transcriptionQueue.length,
1748
+ active: !!session.activeRequest,
1749
+ pending: session.transcriptionQueue.length + (session.activeRequest ? 1 : 0)
1750
+ });
1751
+ const enqueue = () => {
1641
1752
  if (!session.voiced || session.samples < context.sampleRate * 0.25) {
1642
1753
  session.chunks = [];
1643
1754
  session.samples = 0;
@@ -1655,29 +1766,46 @@ var WhisperHttpRecognitionEngine = class {
1655
1766
  session.samples = 0;
1656
1767
  session.voiced = false;
1657
1768
  session.silence = 0;
1658
- const request = new AbortController();
1659
- session.inflight.add(request);
1769
+ session.transcriptionQueue.push(samples);
1770
+ notifyProcessing();
1771
+ void drain();
1772
+ };
1773
+ const drain = async () => {
1774
+ if (session.draining || !valid()) return;
1775
+ session.draining = true;
1660
1776
  try {
1661
- const response = await this.g.fetch(this.route + "/transcribe", {
1662
- method: "POST",
1663
- credentials: "same-origin",
1664
- headers: {
1665
- "content-type": "audio/wav",
1666
- "x-dlv-client-id": session.operation,
1667
- "x-dlv-operation-id": id(),
1668
- "x-dlv-language": lang
1669
- },
1670
- body: encodeMonoPcm16Wav(samples, context.sampleRate),
1671
- signal: request.signal
1672
- });
1673
- const json = await response.json();
1674
- if (!response.ok || !json?.ok)
1675
- throw new Error(json?.error?.message || "HTTP transcription failed.");
1676
- if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: "" });
1677
- } catch (error) {
1678
- if (error.name !== "AbortError" && valid()) onError?.(error);
1777
+ while (valid() && session.transcriptionQueue.length) {
1778
+ const samples = session.transcriptionQueue.shift();
1779
+ const request = new AbortController();
1780
+ session.activeRequest = request;
1781
+ notifyProcessing();
1782
+ try {
1783
+ const response = await this.g.fetch(this.route + "/transcribe", {
1784
+ method: "POST",
1785
+ credentials: "same-origin",
1786
+ headers: {
1787
+ "content-type": "audio/wav",
1788
+ "x-dlv-client-id": session.operation,
1789
+ "x-dlv-operation-id": id(),
1790
+ "x-dlv-language": lang
1791
+ },
1792
+ body: encodeMonoPcm16Wav(samples, context.sampleRate),
1793
+ signal: request.signal
1794
+ });
1795
+ const json = await response.json();
1796
+ if (!response.ok || !json?.ok)
1797
+ throw new Error(json?.error?.message || "HTTP transcription failed.");
1798
+ if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: "" });
1799
+ } catch (error) {
1800
+ if (error.name !== "AbortError" && valid()) onError?.(error);
1801
+ } finally {
1802
+ if (session.activeRequest === request) session.activeRequest = null;
1803
+ notifyProcessing();
1804
+ }
1805
+ }
1679
1806
  } finally {
1680
- session.inflight.delete(request);
1807
+ session.draining = false;
1808
+ notifyProcessing();
1681
1809
  }
1682
1810
  };
1683
1811
  processor.onaudioprocess = (event) => {
@@ -1693,13 +1821,24 @@ var WhisperHttpRecognitionEngine = class {
1693
1821
  }
1694
1822
  session.chunks.push(data);
1695
1823
  session.samples += data.length;
1696
- if (session.voiced && session.silence > context.sampleRate * (this.segmentation.silenceMs / 1e3) || session.samples > context.sampleRate * 20)
1697
- void submit();
1824
+ if (session.voiced && session.silence > context.sampleRate * (this.segmentation.silenceMs / 1e3) || session.samples > context.sampleRate * this.maxUtteranceSeconds)
1825
+ enqueue();
1698
1826
  };
1699
1827
  source.connect(processor);
1828
+ session.finish = enqueue;
1700
1829
  session.abort = () => this.stop();
1701
1830
  signal?.addEventListener("abort", session.abort, { once: true });
1702
1831
  }
1832
+ finish() {
1833
+ const session = this.session;
1834
+ if (!session) return;
1835
+ session.processor.onaudioprocess = null;
1836
+ try {
1837
+ this.meter?.source?.disconnect(session.processor);
1838
+ } catch {
1839
+ }
1840
+ session.finish?.();
1841
+ }
1703
1842
  async stop() {
1704
1843
  const session = this.session;
1705
1844
  if (!session) return;
@@ -1715,8 +1854,10 @@ var WhisperHttpRecognitionEngine = class {
1715
1854
  session.gain?.disconnect();
1716
1855
  } catch {
1717
1856
  }
1718
- for (const request of session.inflight) request.abort();
1719
- session.inflight.clear();
1857
+ session.transcriptionQueue = [];
1858
+ session.activeRequest?.abort();
1859
+ session.activeRequest = null;
1860
+ session.onProcessingChange?.({ queued: 0, active: false, pending: 0 });
1720
1861
  }
1721
1862
  };
1722
1863
 
@@ -2886,6 +3027,28 @@ function createComponents(React2) {
2886
3027
  null,
2887
3028
  "Controls how long a pause must last before captured speech is sent for recognition."
2888
3029
  ),
3030
+ h(
3031
+ "label",
3032
+ { className: "dlv-setting-field" },
3033
+ "Maximum continuous speech (seconds)",
3034
+ h("input", {
3035
+ type: "number",
3036
+ min: 10,
3037
+ max: 300,
3038
+ step: 1,
3039
+ value: settings.recognitionMaxUtteranceSeconds ?? 60,
3040
+ onChange: (event) => {
3041
+ const value = Number(event.target.value);
3042
+ if (Number.isInteger(value) && value >= 10 && value <= 300)
3043
+ invoke("updateSettings", { recognitionMaxUtteranceSeconds: value });
3044
+ }
3045
+ }),
3046
+ h(
3047
+ "small",
3048
+ null,
3049
+ "If speech never pauses, start a new transcription chunk after this duration. Default: 60 seconds."
3050
+ )
3051
+ ),
2889
3052
  h(
2890
3053
  "div",
2891
3054
  {
@@ -2953,6 +3116,21 @@ function createComponents(React2) {
2953
3116
  { key: "interrupt-message-description", className: "dlv-setting-description" },
2954
3117
  settings.interruptSpeechOnUserMessage ? "Sending or steering a new user message stops current or paused assistant speech." : "Sending another message does not stop the assistant audio you are already hearing."
2955
3118
  ),
3119
+ h(
3120
+ "label",
3121
+ { key: "hold-to-talk", className: "dlv-check" },
3122
+ h("input", {
3123
+ type: "checkbox",
3124
+ checked: settings.holdToTalkEnabled !== false,
3125
+ onChange: (event) => invoke("updateSettings", { holdToTalkEnabled: event.target.checked })
3126
+ }),
3127
+ " Hold Control to talk"
3128
+ ),
3129
+ h(
3130
+ "p",
3131
+ { key: "hold-to-talk-description", className: "dlv-setting-description" },
3132
+ "While a composer is open, hold Control anywhere on the page to capture speech. Release it to flush queued transcription, wait the configured send delay, queue the message, and close voice capture. Press Escape while holding to cancel."
3133
+ ),
2956
3134
  field("Listening mode", "mode", [
2957
3135
  { value: "speaker", label: "Speakers \u2014 gated listening" },
2958
3136
  { value: "headphones", label: "Headphones \u2014 open microphone" }
@@ -3112,12 +3290,22 @@ function apply(ctx) {
3112
3290
  let disposed = false;
3113
3291
  let voiceModeActive = false;
3114
3292
  const ownership = new VoiceOwnership();
3293
+ const recognitionSettingKeys = /* @__PURE__ */ new Set([
3294
+ "recognitionEngine",
3295
+ "recognitionProcessLocally",
3296
+ "recognitionAutoInstall",
3297
+ "voiceDetectionPreset",
3298
+ "recognitionMaxUtteranceSeconds"
3299
+ ]);
3300
+ const changesRecognition = (next) => Object.keys(next).some((key) => recognitionSettingKeys.has(key));
3115
3301
  const recognitionFor = (settings, meter) => settings.recognitionEngine === "qwen-http" ? new QwenHttpRecognitionEngine({
3116
3302
  meter,
3117
- voiceDetectionPreset: settings.voiceDetectionPreset
3303
+ voiceDetectionPreset: settings.voiceDetectionPreset,
3304
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds
3118
3305
  }) : settings.recognitionEngine === "whisper-http" ? new WhisperHttpRecognitionEngine({
3119
3306
  meter,
3120
- voiceDetectionPreset: settings.voiceDetectionPreset
3307
+ voiceDetectionPreset: settings.voiceDetectionPreset,
3308
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds
3121
3309
  }) : new BrowserRecognitionEngine({
3122
3310
  processLocally: settings.recognitionProcessLocally,
3123
3311
  autoInstallLocalPack: settings.recognitionAutoInstall
@@ -3194,7 +3382,8 @@ function apply(ctx) {
3194
3382
  if (capture.index < capture.interaction.questions.length) {
3195
3383
  capture.answering = false;
3196
3384
  const text = pendingQuestionSpeech(capture.interaction, capture.index);
3197
- if (text) run(entry.controller, entry.controller.speak(text, capture.interaction.key));
3385
+ if (text)
3386
+ run(entry.controller, entry.controller.speak(text, capture.interaction.key));
3198
3387
  } else {
3199
3388
  entry.questionCapture = null;
3200
3389
  entry.controller.patch({ answeringQuestion: false });
@@ -3244,7 +3433,7 @@ function apply(ctx) {
3244
3433
  });
3245
3434
  const pendingInteractions = ctx.uiSession.pendingInteractions;
3246
3435
  const refreshPendingQuestion = (baseline = false) => {
3247
- if (disposed || entry.closed) return;
3436
+ if (disposed || entry.closed || !pendingInteractions) return;
3248
3437
  const interaction = pendingInteractions.getSnapshot().get(sessionId);
3249
3438
  const key2 = interaction?.kind === "question" ? interaction.key : null;
3250
3439
  if (!key2 || entry.questionCapture && entry.questionCapture.interaction.key !== key2) {
@@ -3264,22 +3453,28 @@ function apply(ctx) {
3264
3453
  run(controller, controller.speak(text, key2));
3265
3454
  }
3266
3455
  };
3267
- refreshPendingQuestion(true);
3268
- entry.unsubscribePendingQuestion = pendingInteractions.subscribe(refreshPendingQuestion);
3456
+ if (pendingInteractions) {
3457
+ refreshPendingQuestion(true);
3458
+ entry.unsubscribePendingQuestion = pendingInteractions.subscribe(refreshPendingQuestion);
3459
+ }
3269
3460
  const update = controller.updateSettings.bind(controller);
3270
- controller.updateSettings = (next) => {
3271
- if (disposed || entry.closed) return;
3461
+ entry.applySettings = (next) => {
3272
3462
  update(next);
3273
3463
  engineBrowser.lang = controller.getSnapshot().settings.lang;
3274
3464
  engineQwen.lang = controller.getSnapshot().settings.lang;
3465
+ run(controller, controller.refreshCapabilities());
3466
+ };
3467
+ controller.updateSettings = (next) => {
3468
+ if (disposed || entry.closed) return;
3469
+ entry.applySettings(next);
3470
+ const settings2 = controller.getSnapshot().settings;
3275
3471
  try {
3276
- localStorage.setItem(
3277
- "dsh-live-voice.settings",
3278
- JSON.stringify(controller.getSnapshot().settings)
3279
- );
3472
+ localStorage.setItem("dsh-live-voice.settings", JSON.stringify(settings2));
3280
3473
  } catch {
3281
3474
  }
3282
- run(controller, controller.refreshCapabilities());
3475
+ for (const other of controllers.values()) {
3476
+ if (other !== entry && !other.closed) other.applySettings(settings2);
3477
+ }
3283
3478
  };
3284
3479
  for (const method of ["stopListening", "cancelDictation", "endConversation", "stopSpeech"]) {
3285
3480
  const original = controller[method].bind(controller);
@@ -3289,7 +3484,7 @@ function apply(ctx) {
3289
3484
  return original(...args);
3290
3485
  };
3291
3486
  }
3292
- for (const method of ["startDictation", "startConversation", "speak"]) {
3487
+ for (const method of ["startDictation", "startHoldToTalk", "startConversation", "speak"]) {
3293
3488
  const original = controller[method].bind(controller);
3294
3489
  controller[method] = (...args) => {
3295
3490
  if (disposed || entry.closed || !entry.refs) return Promise.resolve();
@@ -3400,13 +3595,11 @@ function apply(ctx) {
3400
3595
  let settingsRevision = 0;
3401
3596
  c.updateSettings = async (next) => {
3402
3597
  const revision = ++settingsRevision;
3403
- const previousEngine = c.getSnapshot().settings.recognitionEngine;
3404
3598
  update(next);
3405
3599
  browser.lang = c.getSnapshot().settings.lang;
3406
3600
  qwen.lang = c.getSnapshot().settings.lang;
3407
3601
  const settings2 = c.getSnapshot().settings;
3408
- if (previousEngine !== settings2.recognitionEngine)
3409
- c.replaceRecognition(recognitionFor(settings2, c.meter));
3602
+ if (changesRecognition(next)) c.replaceRecognition(recognitionFor(settings2, c.meter));
3410
3603
  localStorage.setItem("dsh-live-voice.settings", JSON.stringify(settings2));
3411
3604
  run(c, c.refreshCapabilities());
3412
3605
  const active = [...controllers.values()];
@@ -3417,9 +3610,9 @@ function apply(ctx) {
3417
3610
  if (revision !== settingsRevision || disposed || c.disposed) return;
3418
3611
  for (const entry of controllers.values()) {
3419
3612
  if (entry.closed) continue;
3420
- if (entry.controller.getSnapshot().settings.recognitionEngine !== settings2.recognitionEngine)
3613
+ if (changesRecognition(next))
3421
3614
  entry.controller.replaceRecognition(recognitionFor(settings2, entry.controller.meter));
3422
- entry.controller.updateSettings(settings2);
3615
+ entry.applySettings(settings2);
3423
3616
  run(entry.controller, entry.controller.refreshCapabilities());
3424
3617
  }
3425
3618
  };
@@ -3534,19 +3727,47 @@ function apply(ctx) {
3534
3727
  for (const entry of controllers.values())
3535
3728
  run(entry.controller, entry.controller.endConversation());
3536
3729
  };
3537
- const onKey = (event) => {
3538
- if (disposed || event.defaultPrevented || !event.ctrlKey || !event.shiftKey || event.code !== "Space" || event.repeat)
3539
- return;
3730
+ let holdToTalk = null;
3731
+ const candidate = () => {
3540
3732
  const candidates = [...controllers.values()].filter(
3541
3733
  (entry) => entry.buttons > 0 && entry.composers.size > 0
3542
3734
  );
3543
- if (candidates.length !== 1) return;
3544
- event.preventDefault();
3545
- const c = candidates[0].controller;
3546
- run(
3547
- c,
3548
- c.getSnapshot().listening || c.getSnapshot().starting ? c.stopListening() : c.startDictation()
3549
- );
3735
+ return candidates.length === 1 ? candidates[0] : null;
3736
+ };
3737
+ const releaseHoldToTalk = () => {
3738
+ const entry = holdToTalk;
3739
+ holdToTalk = null;
3740
+ if (!entry || entry.closed) return;
3741
+ run(entry.controller, entry.controller.releaseHoldToTalk());
3742
+ };
3743
+ const onKeyDown = (event) => {
3744
+ if (disposed || event.defaultPrevented || event.repeat) return;
3745
+ if (event.key === "Escape" && holdToTalk) {
3746
+ const entry2 = holdToTalk;
3747
+ holdToTalk = null;
3748
+ event.preventDefault();
3749
+ run(entry2.controller, entry2.controller.cancelDictation());
3750
+ return;
3751
+ }
3752
+ if (event.ctrlKey && event.shiftKey && event.code === "Space") {
3753
+ const entry2 = candidate();
3754
+ if (!entry2) return;
3755
+ event.preventDefault();
3756
+ const c = entry2.controller;
3757
+ run(
3758
+ c,
3759
+ c.getSnapshot().listening || c.getSnapshot().starting ? c.stopListening() : c.startDictation()
3760
+ );
3761
+ return;
3762
+ }
3763
+ if (event.key !== "Control" || event.altKey || event.metaKey || event.shiftKey) return;
3764
+ const entry = candidate();
3765
+ if (!entry || !entry.controller.getSnapshot().settings.holdToTalkEnabled) return;
3766
+ holdToTalk = entry;
3767
+ run(entry.controller, entry.controller.startHoldToTalk());
3768
+ };
3769
+ const onKeyUp = (event) => {
3770
+ if (event.key === "Control") releaseHoldToTalk();
3550
3771
  };
3551
3772
  const refreshCapabilities = () => {
3552
3773
  for (const entry of controllers.values())
@@ -3559,7 +3780,9 @@ function apply(ctx) {
3559
3780
  window.speechSynthesis?.addEventListener?.("voiceschanged", refreshCapabilities);
3560
3781
  navigator.mediaDevices?.addEventListener?.("devicechange", refreshCapabilities);
3561
3782
  document.addEventListener("visibilitychange", visibilityChanged);
3562
- document.addEventListener("keydown", onKey);
3783
+ document.addEventListener("keydown", onKeyDown);
3784
+ document.addEventListener("keyup", onKeyUp);
3785
+ window.addEventListener("blur", releaseHoldToTalk);
3563
3786
  window.addEventListener("pagehide", stop);
3564
3787
  return () => {
3565
3788
  disposed = true;
@@ -3567,7 +3790,9 @@ function apply(ctx) {
3567
3790
  window.speechSynthesis?.removeEventListener?.("voiceschanged", refreshCapabilities);
3568
3791
  navigator.mediaDevices?.removeEventListener?.("devicechange", refreshCapabilities);
3569
3792
  document.removeEventListener("visibilitychange", visibilityChanged);
3570
- document.removeEventListener("keydown", onKey);
3793
+ document.removeEventListener("keydown", onKeyDown);
3794
+ document.removeEventListener("keyup", onKeyUp);
3795
+ window.removeEventListener("blur", releaseHoldToTalk);
3571
3796
  window.removeEventListener("pagehide", stop);
3572
3797
  for (const entry of controllers.values()) retire(entry);
3573
3798
  };
package/lib/server.js CHANGED
@@ -460,6 +460,9 @@ var defaultSettings = Object.freeze({
460
460
  recognitionProcessLocally: true,
461
461
  recognitionAutoInstall: true,
462
462
  voiceDetectionPreset: "natural",
463
+ recognitionMaxUtteranceSeconds: 60,
464
+ microphoneEnabled: true,
465
+ holdToTalkEnabled: true,
463
466
  announceAssistantMessages: true,
464
467
  interruptSpeechOnUserMessage: false,
465
468
  sendingMode: "manual",