dsh-live-voice 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/client.js CHANGED
@@ -114,7 +114,9 @@ var defaultSettings = Object.freeze({
114
114
  recognitionProcessLocally: true,
115
115
  recognitionAutoInstall: true,
116
116
  voiceDetectionPreset: "natural",
117
+ recognitionMaxUtteranceSeconds: 60,
117
118
  microphoneEnabled: true,
119
+ holdToTalkEnabled: true,
118
120
  announceAssistantMessages: true,
119
121
  interruptSpeechOnUserMessage: false,
120
122
  sendingMode: "manual",
@@ -150,7 +152,9 @@ function normalizeSettings(value) {
150
152
  recognitionProcessLocally: typeof source.recognitionProcessLocally === "boolean" ? source.recognitionProcessLocally : defaultSettings.recognitionProcessLocally,
151
153
  recognitionAutoInstall: typeof source.recognitionAutoInstall === "boolean" ? source.recognitionAutoInstall : defaultSettings.recognitionAutoInstall,
152
154
  voiceDetectionPreset: Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset) ? source.voiceDetectionPreset : defaultSettings.voiceDetectionPreset,
155
+ recognitionMaxUtteranceSeconds: Number.isInteger(source.recognitionMaxUtteranceSeconds) && source.recognitionMaxUtteranceSeconds >= 10 && source.recognitionMaxUtteranceSeconds <= 300 ? source.recognitionMaxUtteranceSeconds : defaultSettings.recognitionMaxUtteranceSeconds,
153
156
  microphoneEnabled: typeof source.microphoneEnabled === "boolean" ? source.microphoneEnabled : defaultSettings.microphoneEnabled,
157
+ holdToTalkEnabled: typeof source.holdToTalkEnabled === "boolean" ? source.holdToTalkEnabled : defaultSettings.holdToTalkEnabled,
154
158
  announceAssistantMessages: typeof source.announceAssistantMessages === "boolean" ? source.announceAssistantMessages : defaultSettings.announceAssistantMessages,
155
159
  interruptSpeechOnUserMessage: typeof source.interruptSpeechOnUserMessage === "boolean" ? source.interruptSpeechOnUserMessage : defaultSettings.interruptSpeechOnUserMessage,
156
160
  sendingMode: source.sendingMode === "automatic" ? "queue" : ["manual", "queue", "steer"].includes(source.sendingMode) ? source.sendingMode : defaultSettings.sendingMode,
@@ -273,6 +277,7 @@ var VoiceCoordinator = class {
273
277
  conversation: false,
274
278
  listening: false,
275
279
  recognizing: false,
280
+ pendingTranscriptions: 0,
276
281
  muted: normalizeSettings(settings).microphoneEnabled === false,
277
282
  speaking: false,
278
283
  paused: false,
@@ -287,6 +292,7 @@ var VoiceCoordinator = class {
287
292
  };
288
293
  this.autoSendTimer = null;
289
294
  this.autoSendDraft = null;
295
+ this.holdToTalkRelease = false;
290
296
  this.interruptionTimer = null;
291
297
  this.interruptionTranscriptConfirmed = false;
292
298
  this.interruptionPausedSpeech = false;
@@ -404,6 +410,27 @@ var VoiceCoordinator = class {
404
410
  startDictation() {
405
411
  return this.startListening(false);
406
412
  }
413
+ startHoldToTalk() {
414
+ this.holdToTalkRelease = false;
415
+ return this.startListening(false);
416
+ }
417
+ async releaseHoldToTalk() {
418
+ if (this.disposed || !this.snapshot.listening && !this.snapshot.starting) return;
419
+ this.holdToTalkRelease = true;
420
+ if (this.snapshot.starting) return;
421
+ await this.recognition.finish?.();
422
+ if (this.snapshot.pendingTranscriptions === 0) this._finishHoldToTalk();
423
+ }
424
+ _finishHoldToTalk() {
425
+ if (!this.holdToTalkRelease) return;
426
+ this.holdToTalkRelease = false;
427
+ const draft = this.composer.getDraft();
428
+ if (!draft.trim()) {
429
+ void this.stopListening();
430
+ return;
431
+ }
432
+ this.scheduleAutoSend(draft, { force: true, stopAfter: true });
433
+ }
407
434
  startConversation() {
408
435
  return this.startListening(true);
409
436
  }
@@ -483,6 +510,14 @@ var VoiceCoordinator = class {
483
510
  onResult: (result) => {
484
511
  if (valid()) this.onResult(result);
485
512
  },
513
+ onProcessingChange: ({ pending = 0 } = {}) => {
514
+ if (!valid()) return;
515
+ const count = Number.isSafeInteger(pending) && pending >= 0 ? pending : 0;
516
+ this.patch({ pendingTranscriptions: count });
517
+ if (count > 0) this.cancelAutoSend();
518
+ else if (this.holdToTalkRelease) this._finishHoldToTalk();
519
+ else this.maybeScheduleAutoSend();
520
+ },
486
521
  onActivity: (active) => {
487
522
  if (!valid()) return;
488
523
  if (active) {
@@ -507,6 +542,7 @@ var VoiceCoordinator = class {
507
542
  return;
508
543
  }
509
544
  this.patch({ listening: true, starting: false });
545
+ if (this.holdToTalkRelease) void this.releaseHoldToTalk();
510
546
  this._drain();
511
547
  } catch (error) {
512
548
  try {
@@ -619,7 +655,7 @@ var VoiceCoordinator = class {
619
655
  }
620
656
  const next = this.transcript.update(this.composer.getDraft(), final, true);
621
657
  this.composer.setDraft(next);
622
- if (this.snapshot.settings.sendingMode !== "manual") this.scheduleAutoSend(next);
658
+ this.maybeScheduleAutoSend(next);
623
659
  }
624
660
  if (interim) {
625
661
  this.cancelAutoSend();
@@ -630,7 +666,12 @@ var VoiceCoordinator = class {
630
666
  this.patch({ recognizing: !!interim });
631
667
  this._drain();
632
668
  }
633
- scheduleAutoSend(draft) {
669
+ maybeScheduleAutoSend(draft = this.composer.getDraft()) {
670
+ if (this.snapshot.settings.sendingMode === "manual" || this.snapshot.recognizing || this.snapshot.pendingTranscriptions > 0)
671
+ return;
672
+ this.scheduleAutoSend(draft);
673
+ }
674
+ scheduleAutoSend(draft, { force = false, stopAfter = false } = {}) {
634
675
  this.cancelAutoSend();
635
676
  if (!draft.trim() || typeof this.composer.submit !== "function") return;
636
677
  const delay = this.snapshot.settings.autoSendDelaySeconds * 1e3;
@@ -641,11 +682,14 @@ var VoiceCoordinator = class {
641
682
  const expected = this.autoSendDraft;
642
683
  this.autoSendDraft = null;
643
684
  this.patch({ autoSendAt: null });
644
- if (!this.disposed && this.snapshot.settings.sendingMode !== "manual" && expected === this.composer.getDraft()) {
685
+ if (!this.disposed && (force || this.snapshot.settings.sendingMode !== "manual") && expected === this.composer.getDraft()) {
645
686
  try {
646
- this.composer.submit(this.snapshot.settings.sendingMode === "steer" ? "steer" : "queue");
687
+ this.composer.submit(
688
+ this.snapshot.settings.sendingMode === "steer" && !force ? "steer" : "queue"
689
+ );
647
690
  this.transcript.reset();
648
691
  this.recognition.reset?.();
692
+ if (stopAfter) void this.stopListening();
649
693
  } catch (error) {
650
694
  this.patch({ error: message(error) });
651
695
  }
@@ -662,10 +706,11 @@ var VoiceCoordinator = class {
662
706
  if (this.autoSendDraft !== null && draft !== this.autoSendDraft) this.cancelAutoSend();
663
707
  }
664
708
  stopListening() {
709
+ this.holdToTalkRelease = false;
665
710
  this.cancelAutoSend();
666
711
  ++this.epoch;
667
712
  this.inputController?.abort();
668
- this.patch({ listening: false, recognizing: false, starting: false });
713
+ this.patch({ listening: false, recognizing: false, pendingTranscriptions: 0, starting: false });
669
714
  return this._input(async () => {
670
715
  try {
671
716
  await this._releaseInput();
@@ -1389,7 +1434,10 @@ var BrowserRecognitionEngine = class {
1389
1434
  restarts: 0,
1390
1435
  timer: null,
1391
1436
  recognition: null,
1392
- speaking: false
1437
+ speaking: false,
1438
+ finishing: false,
1439
+ finishPromise: null,
1440
+ resolveFinish: null
1393
1441
  };
1394
1442
  this.session = session;
1395
1443
  const cancelled3 = new Promise((resolve, reject) => {
@@ -1487,6 +1535,12 @@ var BrowserRecognitionEngine = class {
1487
1535
  session.recognition = null;
1488
1536
  this._activity(session, false);
1489
1537
  if (this.session !== session) return;
1538
+ if (session.finishing) {
1539
+ this.session = null;
1540
+ session.signal?.removeEventListener("abort", session.cancel);
1541
+ session.resolveFinish?.();
1542
+ return;
1543
+ }
1490
1544
  if (session.restarts >= this.maxRestarts) {
1491
1545
  this._fail(
1492
1546
  session,
@@ -1515,6 +1569,33 @@ var BrowserRecognitionEngine = class {
1515
1569
  };
1516
1570
  recognition.start();
1517
1571
  }
1572
+ finish() {
1573
+ const session = this.session;
1574
+ if (!session) return Promise.resolve();
1575
+ if (session.finishPromise) return session.finishPromise;
1576
+ session.finishing = true;
1577
+ session.finishPromise = new Promise((resolve) => {
1578
+ session.resolveFinish = resolve;
1579
+ });
1580
+ if (session.timer !== null) {
1581
+ (this.globals.clearTimeout ?? globalThis.clearTimeout)(session.timer);
1582
+ session.timer = null;
1583
+ this.session = null;
1584
+ session.resolveFinish();
1585
+ return session.finishPromise;
1586
+ }
1587
+ try {
1588
+ session.recognition?.stop();
1589
+ if (!session.recognition) {
1590
+ this.session = null;
1591
+ session.resolveFinish();
1592
+ }
1593
+ } catch {
1594
+ this.session = null;
1595
+ session.resolveFinish();
1596
+ }
1597
+ return session.finishPromise;
1598
+ }
1518
1599
  reset() {
1519
1600
  const session = this.session;
1520
1601
  const recognition = session?.recognition;
@@ -1549,6 +1630,7 @@ var BrowserRecognitionEngine = class {
1549
1630
  } catch {
1550
1631
  }
1551
1632
  this._activity(session, false);
1633
+ session.resolveFinish?.();
1552
1634
  }
1553
1635
  };
1554
1636
 
@@ -1585,12 +1667,18 @@ function encodeMonoPcm16Wav(samples, inputRate) {
1585
1667
  return buffer;
1586
1668
  }
1587
1669
  var WhisperHttpRecognitionEngine = class {
1588
- constructor({ globals = globalThis, meter, voiceDetectionPreset = "natural" } = {}) {
1670
+ constructor({
1671
+ globals = globalThis,
1672
+ meter,
1673
+ voiceDetectionPreset = "natural",
1674
+ maxUtteranceSeconds = 60
1675
+ } = {}) {
1589
1676
  this.g = globals;
1590
1677
  this.meter = meter;
1591
1678
  this.session = null;
1592
1679
  this.lang = "pt-BR";
1593
1680
  this.voiceDetectionPreset = voiceDetectionPreset;
1681
+ this.maxUtteranceSeconds = maxUtteranceSeconds;
1594
1682
  this.route = ROUTE;
1595
1683
  }
1596
1684
  get segmentation() {
@@ -1616,7 +1704,14 @@ var WhisperHttpRecognitionEngine = class {
1616
1704
  };
1617
1705
  }
1618
1706
  }
1619
- async start({ lang = this.lang, signal, onResult, onActivity, onError } = {}) {
1707
+ async start({
1708
+ lang = this.lang,
1709
+ signal,
1710
+ onResult,
1711
+ onActivity,
1712
+ onError,
1713
+ onProcessingChange
1714
+ } = {}) {
1620
1715
  await this.stop();
1621
1716
  if (signal?.aborted) throw abortError3();
1622
1717
  const context = this.meter?.context, source = this.meter?.source;
@@ -1636,16 +1731,24 @@ var WhisperHttpRecognitionEngine = class {
1636
1731
  samples: 0,
1637
1732
  voiced: false,
1638
1733
  silence: 0,
1639
- inflight: /* @__PURE__ */ new Set(),
1734
+ transcriptionQueue: [],
1735
+ activeRequest: null,
1736
+ draining: false,
1640
1737
  onResult,
1641
1738
  onActivity,
1642
1739
  onError,
1740
+ onProcessingChange,
1643
1741
  lang,
1644
1742
  signal
1645
1743
  };
1646
1744
  this.session = session;
1647
1745
  const valid = () => this.session === session && !signal?.aborted;
1648
- const submit = async () => {
1746
+ const notifyProcessing = () => session.onProcessingChange?.({
1747
+ queued: session.transcriptionQueue.length,
1748
+ active: !!session.activeRequest,
1749
+ pending: session.transcriptionQueue.length + (session.activeRequest ? 1 : 0)
1750
+ });
1751
+ const enqueue = () => {
1649
1752
  if (!session.voiced || session.samples < context.sampleRate * 0.25) {
1650
1753
  session.chunks = [];
1651
1754
  session.samples = 0;
@@ -1663,29 +1766,46 @@ var WhisperHttpRecognitionEngine = class {
1663
1766
  session.samples = 0;
1664
1767
  session.voiced = false;
1665
1768
  session.silence = 0;
1666
- const request = new AbortController();
1667
- session.inflight.add(request);
1769
+ session.transcriptionQueue.push(samples);
1770
+ notifyProcessing();
1771
+ void drain();
1772
+ };
1773
+ const drain = async () => {
1774
+ if (session.draining || !valid()) return;
1775
+ session.draining = true;
1668
1776
  try {
1669
- const response = await this.g.fetch(this.route + "/transcribe", {
1670
- method: "POST",
1671
- credentials: "same-origin",
1672
- headers: {
1673
- "content-type": "audio/wav",
1674
- "x-dlv-client-id": session.operation,
1675
- "x-dlv-operation-id": id(),
1676
- "x-dlv-language": lang
1677
- },
1678
- body: encodeMonoPcm16Wav(samples, context.sampleRate),
1679
- signal: request.signal
1680
- });
1681
- const json = await response.json();
1682
- if (!response.ok || !json?.ok)
1683
- throw new Error(json?.error?.message || "HTTP transcription failed.");
1684
- if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: "" });
1685
- } catch (error) {
1686
- if (error.name !== "AbortError" && valid()) onError?.(error);
1777
+ while (valid() && session.transcriptionQueue.length) {
1778
+ const samples = session.transcriptionQueue.shift();
1779
+ const request = new AbortController();
1780
+ session.activeRequest = request;
1781
+ notifyProcessing();
1782
+ try {
1783
+ const response = await this.g.fetch(this.route + "/transcribe", {
1784
+ method: "POST",
1785
+ credentials: "same-origin",
1786
+ headers: {
1787
+ "content-type": "audio/wav",
1788
+ "x-dlv-client-id": session.operation,
1789
+ "x-dlv-operation-id": id(),
1790
+ "x-dlv-language": lang
1791
+ },
1792
+ body: encodeMonoPcm16Wav(samples, context.sampleRate),
1793
+ signal: request.signal
1794
+ });
1795
+ const json = await response.json();
1796
+ if (!response.ok || !json?.ok)
1797
+ throw new Error(json?.error?.message || "HTTP transcription failed.");
1798
+ if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: "" });
1799
+ } catch (error) {
1800
+ if (error.name !== "AbortError" && valid()) onError?.(error);
1801
+ } finally {
1802
+ if (session.activeRequest === request) session.activeRequest = null;
1803
+ notifyProcessing();
1804
+ }
1805
+ }
1687
1806
  } finally {
1688
- session.inflight.delete(request);
1807
+ session.draining = false;
1808
+ notifyProcessing();
1689
1809
  }
1690
1810
  };
1691
1811
  processor.onaudioprocess = (event) => {
@@ -1701,13 +1821,24 @@ var WhisperHttpRecognitionEngine = class {
1701
1821
  }
1702
1822
  session.chunks.push(data);
1703
1823
  session.samples += data.length;
1704
- if (session.voiced && session.silence > context.sampleRate * (this.segmentation.silenceMs / 1e3) || session.samples > context.sampleRate * 20)
1705
- void submit();
1824
+ if (session.voiced && session.silence > context.sampleRate * (this.segmentation.silenceMs / 1e3) || session.samples > context.sampleRate * this.maxUtteranceSeconds)
1825
+ enqueue();
1706
1826
  };
1707
1827
  source.connect(processor);
1828
+ session.finish = enqueue;
1708
1829
  session.abort = () => this.stop();
1709
1830
  signal?.addEventListener("abort", session.abort, { once: true });
1710
1831
  }
1832
+ finish() {
1833
+ const session = this.session;
1834
+ if (!session) return;
1835
+ session.processor.onaudioprocess = null;
1836
+ try {
1837
+ this.meter?.source?.disconnect(session.processor);
1838
+ } catch {
1839
+ }
1840
+ session.finish?.();
1841
+ }
1711
1842
  async stop() {
1712
1843
  const session = this.session;
1713
1844
  if (!session) return;
@@ -1723,8 +1854,10 @@ var WhisperHttpRecognitionEngine = class {
1723
1854
  session.gain?.disconnect();
1724
1855
  } catch {
1725
1856
  }
1726
- for (const request of session.inflight) request.abort();
1727
- session.inflight.clear();
1857
+ session.transcriptionQueue = [];
1858
+ session.activeRequest?.abort();
1859
+ session.activeRequest = null;
1860
+ session.onProcessingChange?.({ queued: 0, active: false, pending: 0 });
1728
1861
  }
1729
1862
  };
1730
1863
 
@@ -2894,6 +3027,28 @@ function createComponents(React2) {
2894
3027
  null,
2895
3028
  "Controls how long a pause must last before captured speech is sent for recognition."
2896
3029
  ),
3030
+ h(
3031
+ "label",
3032
+ { className: "dlv-setting-field" },
3033
+ "Maximum continuous speech (seconds)",
3034
+ h("input", {
3035
+ type: "number",
3036
+ min: 10,
3037
+ max: 300,
3038
+ step: 1,
3039
+ value: settings.recognitionMaxUtteranceSeconds ?? 60,
3040
+ onChange: (event) => {
3041
+ const value = Number(event.target.value);
3042
+ if (Number.isInteger(value) && value >= 10 && value <= 300)
3043
+ invoke("updateSettings", { recognitionMaxUtteranceSeconds: value });
3044
+ }
3045
+ }),
3046
+ h(
3047
+ "small",
3048
+ null,
3049
+ "If speech never pauses, start a new transcription chunk after this duration. Default: 60 seconds."
3050
+ )
3051
+ ),
2897
3052
  h(
2898
3053
  "div",
2899
3054
  {
@@ -2961,6 +3116,21 @@ function createComponents(React2) {
2961
3116
  { key: "interrupt-message-description", className: "dlv-setting-description" },
2962
3117
  settings.interruptSpeechOnUserMessage ? "Sending or steering a new user message stops current or paused assistant speech." : "Sending another message does not stop the assistant audio you are already hearing."
2963
3118
  ),
3119
+ h(
3120
+ "label",
3121
+ { key: "hold-to-talk", className: "dlv-check" },
3122
+ h("input", {
3123
+ type: "checkbox",
3124
+ checked: settings.holdToTalkEnabled !== false,
3125
+ onChange: (event) => invoke("updateSettings", { holdToTalkEnabled: event.target.checked })
3126
+ }),
3127
+ " Hold Control to talk"
3128
+ ),
3129
+ h(
3130
+ "p",
3131
+ { key: "hold-to-talk-description", className: "dlv-setting-description" },
3132
+ "While a composer is open, hold Control anywhere on the page to capture speech. Release it to flush queued transcription, wait the configured send delay, queue the message, and close voice capture. Press Escape while holding to cancel."
3133
+ ),
2964
3134
  field("Listening mode", "mode", [
2965
3135
  { value: "speaker", label: "Speakers \u2014 gated listening" },
2966
3136
  { value: "headphones", label: "Headphones \u2014 open microphone" }
@@ -3120,12 +3290,22 @@ function apply(ctx) {
3120
3290
  let disposed = false;
3121
3291
  let voiceModeActive = false;
3122
3292
  const ownership = new VoiceOwnership();
3293
+ const recognitionSettingKeys = /* @__PURE__ */ new Set([
3294
+ "recognitionEngine",
3295
+ "recognitionProcessLocally",
3296
+ "recognitionAutoInstall",
3297
+ "voiceDetectionPreset",
3298
+ "recognitionMaxUtteranceSeconds"
3299
+ ]);
3300
+ const changesRecognition = (next) => Object.keys(next).some((key) => recognitionSettingKeys.has(key));
3123
3301
  const recognitionFor = (settings, meter) => settings.recognitionEngine === "qwen-http" ? new QwenHttpRecognitionEngine({
3124
3302
  meter,
3125
- voiceDetectionPreset: settings.voiceDetectionPreset
3303
+ voiceDetectionPreset: settings.voiceDetectionPreset,
3304
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds
3126
3305
  }) : settings.recognitionEngine === "whisper-http" ? new WhisperHttpRecognitionEngine({
3127
3306
  meter,
3128
- voiceDetectionPreset: settings.voiceDetectionPreset
3307
+ voiceDetectionPreset: settings.voiceDetectionPreset,
3308
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds
3129
3309
  }) : new BrowserRecognitionEngine({
3130
3310
  processLocally: settings.recognitionProcessLocally,
3131
3311
  autoInstallLocalPack: settings.recognitionAutoInstall
@@ -3202,7 +3382,8 @@ function apply(ctx) {
3202
3382
  if (capture.index < capture.interaction.questions.length) {
3203
3383
  capture.answering = false;
3204
3384
  const text = pendingQuestionSpeech(capture.interaction, capture.index);
3205
- if (text) run(entry.controller, entry.controller.speak(text, capture.interaction.key));
3385
+ if (text)
3386
+ run(entry.controller, entry.controller.speak(text, capture.interaction.key));
3206
3387
  } else {
3207
3388
  entry.questionCapture = null;
3208
3389
  entry.controller.patch({ answeringQuestion: false });
@@ -3303,7 +3484,7 @@ function apply(ctx) {
3303
3484
  return original(...args);
3304
3485
  };
3305
3486
  }
3306
- for (const method of ["startDictation", "startConversation", "speak"]) {
3487
+ for (const method of ["startDictation", "startHoldToTalk", "startConversation", "speak"]) {
3307
3488
  const original = controller[method].bind(controller);
3308
3489
  controller[method] = (...args) => {
3309
3490
  if (disposed || entry.closed || !entry.refs) return Promise.resolve();
@@ -3414,13 +3595,11 @@ function apply(ctx) {
3414
3595
  let settingsRevision = 0;
3415
3596
  c.updateSettings = async (next) => {
3416
3597
  const revision = ++settingsRevision;
3417
- const previousEngine = c.getSnapshot().settings.recognitionEngine;
3418
3598
  update(next);
3419
3599
  browser.lang = c.getSnapshot().settings.lang;
3420
3600
  qwen.lang = c.getSnapshot().settings.lang;
3421
3601
  const settings2 = c.getSnapshot().settings;
3422
- if (previousEngine !== settings2.recognitionEngine)
3423
- c.replaceRecognition(recognitionFor(settings2, c.meter));
3602
+ if (changesRecognition(next)) c.replaceRecognition(recognitionFor(settings2, c.meter));
3424
3603
  localStorage.setItem("dsh-live-voice.settings", JSON.stringify(settings2));
3425
3604
  run(c, c.refreshCapabilities());
3426
3605
  const active = [...controllers.values()];
@@ -3431,7 +3610,7 @@ function apply(ctx) {
3431
3610
  if (revision !== settingsRevision || disposed || c.disposed) return;
3432
3611
  for (const entry of controllers.values()) {
3433
3612
  if (entry.closed) continue;
3434
- if (entry.controller.getSnapshot().settings.recognitionEngine !== settings2.recognitionEngine)
3613
+ if (changesRecognition(next))
3435
3614
  entry.controller.replaceRecognition(recognitionFor(settings2, entry.controller.meter));
3436
3615
  entry.applySettings(settings2);
3437
3616
  run(entry.controller, entry.controller.refreshCapabilities());
@@ -3548,19 +3727,47 @@ function apply(ctx) {
3548
3727
  for (const entry of controllers.values())
3549
3728
  run(entry.controller, entry.controller.endConversation());
3550
3729
  };
3551
- const onKey = (event) => {
3552
- if (disposed || event.defaultPrevented || !event.ctrlKey || !event.shiftKey || event.code !== "Space" || event.repeat)
3553
- return;
3730
+ let holdToTalk = null;
3731
+ const candidate = () => {
3554
3732
  const candidates = [...controllers.values()].filter(
3555
3733
  (entry) => entry.buttons > 0 && entry.composers.size > 0
3556
3734
  );
3557
- if (candidates.length !== 1) return;
3558
- event.preventDefault();
3559
- const c = candidates[0].controller;
3560
- run(
3561
- c,
3562
- c.getSnapshot().listening || c.getSnapshot().starting ? c.stopListening() : c.startDictation()
3563
- );
3735
+ return candidates.length === 1 ? candidates[0] : null;
3736
+ };
3737
+ const releaseHoldToTalk = () => {
3738
+ const entry = holdToTalk;
3739
+ holdToTalk = null;
3740
+ if (!entry || entry.closed) return;
3741
+ run(entry.controller, entry.controller.releaseHoldToTalk());
3742
+ };
3743
+ const onKeyDown = (event) => {
3744
+ if (disposed || event.defaultPrevented || event.repeat) return;
3745
+ if (event.key === "Escape" && holdToTalk) {
3746
+ const entry2 = holdToTalk;
3747
+ holdToTalk = null;
3748
+ event.preventDefault();
3749
+ run(entry2.controller, entry2.controller.cancelDictation());
3750
+ return;
3751
+ }
3752
+ if (event.ctrlKey && event.shiftKey && event.code === "Space") {
3753
+ const entry2 = candidate();
3754
+ if (!entry2) return;
3755
+ event.preventDefault();
3756
+ const c = entry2.controller;
3757
+ run(
3758
+ c,
3759
+ c.getSnapshot().listening || c.getSnapshot().starting ? c.stopListening() : c.startDictation()
3760
+ );
3761
+ return;
3762
+ }
3763
+ if (event.key !== "Control" || event.altKey || event.metaKey || event.shiftKey) return;
3764
+ const entry = candidate();
3765
+ if (!entry || !entry.controller.getSnapshot().settings.holdToTalkEnabled) return;
3766
+ holdToTalk = entry;
3767
+ run(entry.controller, entry.controller.startHoldToTalk());
3768
+ };
3769
+ const onKeyUp = (event) => {
3770
+ if (event.key === "Control") releaseHoldToTalk();
3564
3771
  };
3565
3772
  const refreshCapabilities = () => {
3566
3773
  for (const entry of controllers.values())
@@ -3573,7 +3780,9 @@ function apply(ctx) {
3573
3780
  window.speechSynthesis?.addEventListener?.("voiceschanged", refreshCapabilities);
3574
3781
  navigator.mediaDevices?.addEventListener?.("devicechange", refreshCapabilities);
3575
3782
  document.addEventListener("visibilitychange", visibilityChanged);
3576
- document.addEventListener("keydown", onKey);
3783
+ document.addEventListener("keydown", onKeyDown);
3784
+ document.addEventListener("keyup", onKeyUp);
3785
+ window.addEventListener("blur", releaseHoldToTalk);
3577
3786
  window.addEventListener("pagehide", stop);
3578
3787
  return () => {
3579
3788
  disposed = true;
@@ -3581,7 +3790,9 @@ function apply(ctx) {
3581
3790
  window.speechSynthesis?.removeEventListener?.("voiceschanged", refreshCapabilities);
3582
3791
  navigator.mediaDevices?.removeEventListener?.("devicechange", refreshCapabilities);
3583
3792
  document.removeEventListener("visibilitychange", visibilityChanged);
3584
- document.removeEventListener("keydown", onKey);
3793
+ document.removeEventListener("keydown", onKeyDown);
3794
+ document.removeEventListener("keyup", onKeyUp);
3795
+ window.removeEventListener("blur", releaseHoldToTalk);
3585
3796
  window.removeEventListener("pagehide", stop);
3586
3797
  for (const entry of controllers.values()) retire(entry);
3587
3798
  };
package/lib/server.js CHANGED
@@ -460,7 +460,9 @@ var defaultSettings = Object.freeze({
460
460
  recognitionProcessLocally: true,
461
461
  recognitionAutoInstall: true,
462
462
  voiceDetectionPreset: "natural",
463
+ recognitionMaxUtteranceSeconds: 60,
463
464
  microphoneEnabled: true,
465
+ holdToTalkEnabled: true,
464
466
  announceAssistantMessages: true,
465
467
  interruptSpeechOnUserMessage: false,
466
468
  sendingMode: "manual",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "dsh-live-voice",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "type": "module",
5
5
  "scripts": {
6
6
  "typecheck": "tsc --noEmit",
@@ -13,7 +13,7 @@
13
13
  "setup:hooks": "git config core.hooksPath .githooks",
14
14
  "prepack": "npm run build"
15
15
  },
16
- "description": "Local-first voice conversations for DSH, with local speech recognition and synthesis and optional external providers.",
16
+ "description": "Local-first hands-free AI voice assistant plugin for DeepSeek Harness (DSH), with speech-to-text, text-to-speech, voice commands, and local speech engines.",
17
17
  "keywords": [
18
18
  "dsh",
19
19
  "dsh-plugin",