dsh-live-voice 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -216,25 +216,26 @@ export function createComponents(React) {
216
216
  const remaining = state.autoSendAt
217
217
  ? Math.max(1, Math.ceil((state.autoSendAt - now) / 1000))
218
218
  : null;
219
- const status = state.answeringQuestion && state.recognizing
220
- ? 'Recognizing answer…'
221
- : state.answeringQuestion && state.listening
222
- ? 'Listening for your answer…'
223
- : remaining
224
- ? `Sending in ${remaining}…`
225
- : state.starting
226
- ? 'Starting microphone…'
227
- : state.paused
228
- ? 'Speech paused'
229
- : state.speaking
230
- ? 'Speaking'
231
- : state.recognizing
232
- ? 'Recognizing speech…'
233
- : state.listening
234
- ? 'Listening — waiting for speech'
235
- : state.conversation
236
- ? 'Conversation idle'
237
- : 'Voice ready';
219
+ const status =
220
+ state.answeringQuestion && state.recognizing
221
+ ? 'Recognizing answer…'
222
+ : state.answeringQuestion && state.listening
223
+ ? 'Listening for your answer…'
224
+ : remaining
225
+ ? `Sending in ${remaining}…`
226
+ : state.starting
227
+ ? 'Starting microphone…'
228
+ : state.paused
229
+ ? 'Speech paused'
230
+ : state.speaking
231
+ ? 'Speaking'
232
+ : state.recognizing
233
+ ? 'Recognizing speech…'
234
+ : state.listening
235
+ ? 'Listening — waiting for speech'
236
+ : state.conversation
237
+ ? 'Conversation idle'
238
+ : 'Voice ready';
238
239
  return h(
239
240
  'div',
240
241
  {
@@ -884,6 +885,28 @@ export function createComponents(React) {
884
885
  null,
885
886
  'Controls how long a pause must last before captured speech is sent for recognition.',
886
887
  ),
888
+ h(
889
+ 'label',
890
+ { className: 'dlv-setting-field' },
891
+ 'Maximum continuous speech (seconds)',
892
+ h('input', {
893
+ type: 'number',
894
+ min: 10,
895
+ max: 300,
896
+ step: 1,
897
+ value: settings.recognitionMaxUtteranceSeconds ?? 60,
898
+ onChange: (event) => {
899
+ const value = Number(event.target.value);
900
+ if (Number.isInteger(value) && value >= 10 && value <= 300)
901
+ invoke('updateSettings', { recognitionMaxUtteranceSeconds: value });
902
+ },
903
+ }),
904
+ h(
905
+ 'small',
906
+ null,
907
+ 'If speech never pauses, start a new transcription chunk after this duration. Default: 60 seconds.',
908
+ ),
909
+ ),
887
910
  h(
888
911
  'div',
889
912
  {
@@ -959,6 +982,22 @@ export function createComponents(React) {
959
982
  ? 'Sending or steering a new user message stops current or paused assistant speech.'
960
983
  : 'Sending another message does not stop the assistant audio you are already hearing.',
961
984
  ),
985
+ h(
986
+ 'label',
987
+ { key: 'hold-to-talk', className: 'dlv-check' },
988
+ h('input', {
989
+ type: 'checkbox',
990
+ checked: settings.holdToTalkEnabled !== false,
991
+ onChange: (event) =>
992
+ invoke('updateSettings', { holdToTalkEnabled: event.target.checked }),
993
+ }),
994
+ ' Hold Control to talk',
995
+ ),
996
+ h(
997
+ 'p',
998
+ { key: 'hold-to-talk-description', className: 'dlv-setting-description' },
999
+ 'While a composer is open, hold Control anywhere on the page to capture speech. Release it to flush queued transcription, wait the configured send delay, queue the message, and close voice capture. Press Escape while holding to cancel.',
1000
+ ),
962
1001
  field('Listening mode', 'mode', [
963
1002
  { value: 'speaker', label: 'Speakers — gated listening' },
964
1003
  { value: 'headphones', label: 'Headphones — open microphone' },
@@ -13,7 +13,12 @@ import { QwenHttpRecognitionEngine } from '../engines/recognition/qwen-http.ts';
13
13
  import { QwenHttpSpeakingEngine } from '../engines/speaking/qwen-http.ts';
14
14
  import { createComponents } from './components.ts';
15
15
  import { styles } from './styles.ts';
16
- import { assistantMessages, addressedTurn, latestUserSequence, pendingQuestionSpeech } from './chat.ts';
16
+ import {
17
+ assistantMessages,
18
+ addressedTurn,
19
+ latestUserSequence,
20
+ pendingQuestionSpeech,
21
+ } from './chat.ts';
17
22
 
18
23
  export const inject = ['slots', 'connection', 'uiConversation', 'uiSession'];
19
24
  export function apply(ctx) {
@@ -27,16 +32,27 @@ export function apply(ctx) {
27
32
  // composer should inherit the user's explicit choice to remain in voice mode.
28
33
  let voiceModeActive = false;
29
34
  const ownership = new VoiceOwnership();
35
+ const recognitionSettingKeys = new Set([
36
+ 'recognitionEngine',
37
+ 'recognitionProcessLocally',
38
+ 'recognitionAutoInstall',
39
+ 'voiceDetectionPreset',
40
+ 'recognitionMaxUtteranceSeconds',
41
+ ]);
42
+ const changesRecognition = (next) =>
43
+ Object.keys(next).some((key) => recognitionSettingKeys.has(key));
30
44
  const recognitionFor = (settings, meter) =>
31
45
  settings.recognitionEngine === 'qwen-http'
32
46
  ? new QwenHttpRecognitionEngine({
33
47
  meter,
34
48
  voiceDetectionPreset: settings.voiceDetectionPreset,
49
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
35
50
  })
36
51
  : settings.recognitionEngine === 'whisper-http'
37
52
  ? new WhisperHttpRecognitionEngine({
38
53
  meter,
39
54
  voiceDetectionPreset: settings.voiceDetectionPreset,
55
+ maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
40
56
  })
41
57
  : new BrowserRecognitionEngine({
42
58
  processLocally: settings.recognitionProcessLocally,
@@ -116,7 +132,8 @@ export function apply(ctx) {
116
132
  if (capture.index < capture.interaction.questions.length) {
117
133
  capture.answering = false;
118
134
  const text = pendingQuestionSpeech(capture.interaction, capture.index);
119
- if (text) run(entry.controller, entry.controller.speak(text, capture.interaction.key));
135
+ if (text)
136
+ run(entry.controller, entry.controller.speak(text, capture.interaction.key));
120
137
  } else {
121
138
  entry.questionCapture = null;
122
139
  entry.controller.patch({ answeringQuestion: false });
@@ -228,7 +245,7 @@ export function apply(ctx) {
228
245
  return original(...args);
229
246
  };
230
247
  }
231
- for (const method of ['startDictation', 'startConversation', 'speak']) {
248
+ for (const method of ['startDictation', 'startHoldToTalk', 'startConversation', 'speak']) {
232
249
  const original = controller[method].bind(controller);
233
250
  controller[method] = (...args) => {
234
251
  if (disposed || entry.closed || !entry.refs) return Promise.resolve();
@@ -351,13 +368,11 @@ export function apply(ctx) {
351
368
  let settingsRevision = 0;
352
369
  c.updateSettings = async (next) => {
353
370
  const revision = ++settingsRevision;
354
- const previousEngine = c.getSnapshot().settings.recognitionEngine;
355
371
  update(next);
356
372
  browser.lang = c.getSnapshot().settings.lang;
357
373
  qwen.lang = c.getSnapshot().settings.lang;
358
374
  const settings = c.getSnapshot().settings;
359
- if (previousEngine !== settings.recognitionEngine)
360
- c.replaceRecognition(recognitionFor(settings, c.meter));
375
+ if (changesRecognition(next)) c.replaceRecognition(recognitionFor(settings, c.meter));
361
376
  localStorage.setItem('dsh-live-voice.settings', JSON.stringify(settings));
362
377
  run(c, c.refreshCapabilities());
363
378
  const active = [...controllers.values()];
@@ -368,9 +383,7 @@ export function apply(ctx) {
368
383
  if (revision !== settingsRevision || disposed || c.disposed) return;
369
384
  for (const entry of controllers.values()) {
370
385
  if (entry.closed) continue;
371
- if (
372
- entry.controller.getSnapshot().settings.recognitionEngine !== settings.recognitionEngine
373
- )
386
+ if (changesRecognition(next))
374
387
  entry.controller.replaceRecognition(recognitionFor(settings, entry.controller.meter));
375
388
  entry.applySettings(settings);
376
389
  run(entry.controller, entry.controller.refreshCapabilities());
@@ -488,28 +501,49 @@ export function apply(ctx) {
488
501
  for (const entry of controllers.values())
489
502
  run(entry.controller, entry.controller.endConversation());
490
503
  };
491
- const onKey = (event) => {
492
- if (
493
- disposed ||
494
- event.defaultPrevented ||
495
- !event.ctrlKey ||
496
- !event.shiftKey ||
497
- event.code !== 'Space' ||
498
- event.repeat
499
- )
500
- return;
504
+ let holdToTalk = null;
505
+ const candidate = () => {
501
506
  const candidates = [...controllers.values()].filter(
502
507
  (entry) => entry.buttons > 0 && entry.composers.size > 0,
503
508
  );
504
- if (candidates.length !== 1) return;
505
- event.preventDefault();
506
- const c = candidates[0].controller;
507
- run(
508
- c,
509
- c.getSnapshot().listening || c.getSnapshot().starting
510
- ? c.stopListening()
511
- : c.startDictation(),
512
- );
509
+ return candidates.length === 1 ? candidates[0] : null;
510
+ };
511
+ const releaseHoldToTalk = () => {
512
+ const entry = holdToTalk;
513
+ holdToTalk = null;
514
+ if (!entry || entry.closed) return;
515
+ run(entry.controller, entry.controller.releaseHoldToTalk());
516
+ };
517
+ const onKeyDown = (event) => {
518
+ if (disposed || event.defaultPrevented || event.repeat) return;
519
+ if (event.key === 'Escape' && holdToTalk) {
520
+ const entry = holdToTalk;
521
+ holdToTalk = null;
522
+ event.preventDefault();
523
+ run(entry.controller, entry.controller.cancelDictation());
524
+ return;
525
+ }
526
+ if (event.ctrlKey && event.shiftKey && event.code === 'Space') {
527
+ const entry = candidate();
528
+ if (!entry) return;
529
+ event.preventDefault();
530
+ const c = entry.controller;
531
+ run(
532
+ c,
533
+ c.getSnapshot().listening || c.getSnapshot().starting
534
+ ? c.stopListening()
535
+ : c.startDictation(),
536
+ );
537
+ return;
538
+ }
539
+ if (event.key !== 'Control' || event.altKey || event.metaKey || event.shiftKey) return;
540
+ const entry = candidate();
541
+ if (!entry || !entry.controller.getSnapshot().settings.holdToTalkEnabled) return;
542
+ holdToTalk = entry;
543
+ run(entry.controller, entry.controller.startHoldToTalk());
544
+ };
545
+ const onKeyUp = (event) => {
546
+ if (event.key === 'Control') releaseHoldToTalk();
513
547
  };
514
548
  const refreshCapabilities = () => {
515
549
  for (const entry of controllers.values())
@@ -522,7 +556,9 @@ export function apply(ctx) {
522
556
  window.speechSynthesis?.addEventListener?.('voiceschanged', refreshCapabilities);
523
557
  navigator.mediaDevices?.addEventListener?.('devicechange', refreshCapabilities);
524
558
  document.addEventListener('visibilitychange', visibilityChanged);
525
- document.addEventListener('keydown', onKey);
559
+ document.addEventListener('keydown', onKeyDown);
560
+ document.addEventListener('keyup', onKeyUp);
561
+ window.addEventListener('blur', releaseHoldToTalk);
526
562
  window.addEventListener('pagehide', stop);
527
563
  return () => {
528
564
  disposed = true;
@@ -530,7 +566,9 @@ export function apply(ctx) {
530
566
  window.speechSynthesis?.removeEventListener?.('voiceschanged', refreshCapabilities);
531
567
  navigator.mediaDevices?.removeEventListener?.('devicechange', refreshCapabilities);
532
568
  document.removeEventListener('visibilitychange', visibilityChanged);
533
- document.removeEventListener('keydown', onKey);
569
+ document.removeEventListener('keydown', onKeyDown);
570
+ document.removeEventListener('keyup', onKeyUp);
571
+ window.removeEventListener('blur', releaseHoldToTalk);
534
572
  window.removeEventListener('pagehide', stop);
535
573
  for (const entry of controllers.values()) retire(entry);
536
574
  };
@@ -46,6 +46,7 @@ export class VoiceCoordinator {
46
46
  conversation: false,
47
47
  listening: false,
48
48
  recognizing: false,
49
+ pendingTranscriptions: 0,
49
50
  muted: normalizeSettings(settings).microphoneEnabled === false,
50
51
  speaking: false,
51
52
  paused: false,
@@ -60,6 +61,7 @@ export class VoiceCoordinator {
60
61
  };
61
62
  this.autoSendTimer = null;
62
63
  this.autoSendDraft = null;
64
+ this.holdToTalkRelease = false;
63
65
  this.interruptionTimer = null;
64
66
  this.interruptionTranscriptConfirmed = false;
65
67
  this.interruptionPausedSpeech = false;
@@ -186,6 +188,27 @@ export class VoiceCoordinator {
186
188
  startDictation() {
187
189
  return this.startListening(false);
188
190
  }
191
+ startHoldToTalk() {
192
+ this.holdToTalkRelease = false;
193
+ return this.startListening(false);
194
+ }
195
+ async releaseHoldToTalk() {
196
+ if (this.disposed || (!this.snapshot.listening && !this.snapshot.starting)) return;
197
+ this.holdToTalkRelease = true;
198
+ if (this.snapshot.starting) return;
199
+ await this.recognition.finish?.();
200
+ if (this.snapshot.pendingTranscriptions === 0) this._finishHoldToTalk();
201
+ }
202
+ _finishHoldToTalk() {
203
+ if (!this.holdToTalkRelease) return;
204
+ this.holdToTalkRelease = false;
205
+ const draft = this.composer.getDraft();
206
+ if (!draft.trim()) {
207
+ void this.stopListening();
208
+ return;
209
+ }
210
+ this.scheduleAutoSend(draft, { force: true, stopAfter: true });
211
+ }
189
212
  startConversation() {
190
213
  return this.startListening(true);
191
214
  }
@@ -269,6 +292,14 @@ export class VoiceCoordinator {
269
292
  onResult: (result) => {
270
293
  if (valid()) this.onResult(result);
271
294
  },
295
+ onProcessingChange: ({ pending = 0 } = {}) => {
296
+ if (!valid()) return;
297
+ const count = Number.isSafeInteger(pending) && pending >= 0 ? pending : 0;
298
+ this.patch({ pendingTranscriptions: count });
299
+ if (count > 0) this.cancelAutoSend();
300
+ else if (this.holdToTalkRelease) this._finishHoldToTalk();
301
+ else this.maybeScheduleAutoSend();
302
+ },
272
303
  onActivity: (active) => {
273
304
  if (!valid()) return;
274
305
  if (active) {
@@ -294,6 +325,7 @@ export class VoiceCoordinator {
294
325
  return;
295
326
  }
296
327
  this.patch({ listening: true, starting: false });
328
+ if (this.holdToTalkRelease) void this.releaseHoldToTalk();
297
329
  this._drain();
298
330
  } catch (error) {
299
331
  try {
@@ -378,8 +410,10 @@ export class VoiceCoordinator {
378
410
  }
379
411
  // Pending structured questions own recognition while waiting for a hands-free
380
412
  // answer. Keep their interim and final text out of the normal chat composer.
381
- if (typeof this.composer.handleQuestionResult === 'function' &&
382
- this.composer.handleQuestionResult({ final, interim })) {
413
+ if (
414
+ typeof this.composer.handleQuestionResult === 'function' &&
415
+ this.composer.handleQuestionResult({ final, interim })
416
+ ) {
383
417
  this.cancelAutoSend();
384
418
  this.transcript.reset();
385
419
  this.patch({ recognizing: !!interim });
@@ -441,7 +475,7 @@ export class VoiceCoordinator {
441
475
  }
442
476
  const next = this.transcript.update(this.composer.getDraft(), final, true);
443
477
  this.composer.setDraft(next);
444
- if (this.snapshot.settings.sendingMode !== 'manual') this.scheduleAutoSend(next);
478
+ this.maybeScheduleAutoSend(next);
445
479
  }
446
480
  if (interim) {
447
481
  this.cancelAutoSend();
@@ -453,7 +487,16 @@ export class VoiceCoordinator {
453
487
  this.patch({ recognizing: !!interim });
454
488
  this._drain();
455
489
  }
456
- scheduleAutoSend(draft) {
490
+ maybeScheduleAutoSend(draft = this.composer.getDraft()) {
491
+ if (
492
+ this.snapshot.settings.sendingMode === 'manual' ||
493
+ this.snapshot.recognizing ||
494
+ this.snapshot.pendingTranscriptions > 0
495
+ )
496
+ return;
497
+ this.scheduleAutoSend(draft);
498
+ }
499
+ scheduleAutoSend(draft, { force = false, stopAfter = false } = {}) {
457
500
  this.cancelAutoSend();
458
501
  if (!draft.trim() || typeof this.composer.submit !== 'function') return;
459
502
  const delay = this.snapshot.settings.autoSendDelaySeconds * 1000;
@@ -466,16 +509,19 @@ export class VoiceCoordinator {
466
509
  this.patch({ autoSendAt: null });
467
510
  if (
468
511
  !this.disposed &&
469
- this.snapshot.settings.sendingMode !== 'manual' &&
512
+ (force || this.snapshot.settings.sendingMode !== 'manual') &&
470
513
  expected === this.composer.getDraft()
471
514
  ) {
472
515
  try {
473
- this.composer.submit(this.snapshot.settings.sendingMode === 'steer' ? 'steer' : 'queue');
516
+ this.composer.submit(
517
+ this.snapshot.settings.sendingMode === 'steer' && !force ? 'steer' : 'queue',
518
+ );
474
519
  // Web Speech keeps a cumulative native result list for the lifetime of
475
520
  // one recognition instance. Start a fresh instance at the turn boundary
476
521
  // so the sent utterance cannot prefix the next one.
477
522
  this.transcript.reset();
478
523
  this.recognition.reset?.();
524
+ if (stopAfter) void this.stopListening();
479
525
  } catch (error) {
480
526
  this.patch({ error: message(error) });
481
527
  }
@@ -492,10 +538,11 @@ export class VoiceCoordinator {
492
538
  if (this.autoSendDraft !== null && draft !== this.autoSendDraft) this.cancelAutoSend();
493
539
  }
494
540
  stopListening() {
541
+ this.holdToTalkRelease = false;
495
542
  this.cancelAutoSend();
496
543
  ++this.epoch;
497
544
  this.inputController?.abort();
498
- this.patch({ listening: false, recognizing: false, starting: false });
545
+ this.patch({ listening: false, recognizing: false, pendingTranscriptions: 0, starting: false });
499
546
  return this._input(async () => {
500
547
  try {
501
548
  await this._releaseInput();
@@ -36,7 +36,9 @@ export const defaultSettings = Object.freeze({
36
36
  recognitionProcessLocally: true,
37
37
  recognitionAutoInstall: true,
38
38
  voiceDetectionPreset: 'natural',
39
+ recognitionMaxUtteranceSeconds: 60,
39
40
  microphoneEnabled: true,
41
+ holdToTalkEnabled: true,
40
42
  announceAssistantMessages: true,
41
43
  interruptSpeechOnUserMessage: false,
42
44
  sendingMode: 'manual',
@@ -94,10 +96,20 @@ export function normalizeSettings(value) {
94
96
  voiceDetectionPreset: Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset)
95
97
  ? source.voiceDetectionPreset
96
98
  : defaultSettings.voiceDetectionPreset,
99
+ recognitionMaxUtteranceSeconds:
100
+ Number.isInteger(source.recognitionMaxUtteranceSeconds) &&
101
+ source.recognitionMaxUtteranceSeconds >= 10 &&
102
+ source.recognitionMaxUtteranceSeconds <= 300
103
+ ? source.recognitionMaxUtteranceSeconds
104
+ : defaultSettings.recognitionMaxUtteranceSeconds,
97
105
  microphoneEnabled:
98
106
  typeof source.microphoneEnabled === 'boolean'
99
107
  ? source.microphoneEnabled
100
108
  : defaultSettings.microphoneEnabled,
109
+ holdToTalkEnabled:
110
+ typeof source.holdToTalkEnabled === 'boolean'
111
+ ? source.holdToTalkEnabled
112
+ : defaultSettings.holdToTalkEnabled,
101
113
  announceAssistantMessages:
102
114
  typeof source.announceAssistantMessages === 'boolean'
103
115
  ? source.announceAssistantMessages
@@ -141,6 +141,9 @@ export class BrowserRecognitionEngine {
141
141
  timer: null,
142
142
  recognition: null,
143
143
  speaking: false,
144
+ finishing: false,
145
+ finishPromise: null,
146
+ resolveFinish: null,
144
147
  };
145
148
  this.session = session;
146
149
  const cancelled = new Promise((resolve, reject) => {
@@ -244,6 +247,12 @@ export class BrowserRecognitionEngine {
244
247
  session.recognition = null;
245
248
  this._activity(session, false);
246
249
  if (this.session !== session) return;
250
+ if (session.finishing) {
251
+ this.session = null;
252
+ session.signal?.removeEventListener('abort', session.cancel);
253
+ session.resolveFinish?.();
254
+ return;
255
+ }
247
256
  if (session.restarts >= this.maxRestarts) {
248
257
  this._fail(
249
258
  session,
@@ -273,6 +282,34 @@ export class BrowserRecognitionEngine {
273
282
  recognition.start();
274
283
  }
275
284
 
285
+ finish() {
286
+ const session = this.session;
287
+ if (!session) return Promise.resolve();
288
+ if (session.finishPromise) return session.finishPromise;
289
+ session.finishing = true;
290
+ session.finishPromise = new Promise((resolve) => {
291
+ session.resolveFinish = resolve;
292
+ });
293
+ if (session.timer !== null) {
294
+ (this.globals.clearTimeout ?? globalThis.clearTimeout)(session.timer);
295
+ session.timer = null;
296
+ this.session = null;
297
+ session.resolveFinish();
298
+ return session.finishPromise;
299
+ }
300
+ try {
301
+ session.recognition?.stop();
302
+ if (!session.recognition) {
303
+ this.session = null;
304
+ session.resolveFinish();
305
+ }
306
+ } catch {
307
+ this.session = null;
308
+ session.resolveFinish();
309
+ }
310
+ return session.finishPromise;
311
+ }
312
+
276
313
  reset() {
277
314
  const session = this.session;
278
315
  const recognition = session?.recognition;
@@ -310,5 +347,6 @@ export class BrowserRecognitionEngine {
310
347
  /* Browser already ended capture. */
311
348
  }
312
349
  this._activity(session, false);
350
+ session.resolveFinish?.();
313
351
  }
314
352
  }
@@ -39,12 +39,18 @@ export function encodeMonoPcm16Wav(samples, inputRate) {
39
39
  return buffer;
40
40
  }
41
41
  export class WhisperHttpRecognitionEngine {
42
- constructor({ globals = globalThis, meter, voiceDetectionPreset = 'natural' } = {}) {
42
+ constructor({
43
+ globals = globalThis,
44
+ meter,
45
+ voiceDetectionPreset = 'natural',
46
+ maxUtteranceSeconds = 60,
47
+ } = {}) {
43
48
  this.g = globals;
44
49
  this.meter = meter;
45
50
  this.session = null;
46
51
  this.lang = 'pt-BR';
47
52
  this.voiceDetectionPreset = voiceDetectionPreset;
53
+ this.maxUtteranceSeconds = maxUtteranceSeconds;
48
54
  this.route = ROUTE;
49
55
  }
50
56
  get segmentation() {
@@ -73,7 +79,14 @@ export class WhisperHttpRecognitionEngine {
73
79
  };
74
80
  }
75
81
  }
76
- async start({ lang = this.lang, signal, onResult, onActivity, onError } = {}) {
82
+ async start({
83
+ lang = this.lang,
84
+ signal,
85
+ onResult,
86
+ onActivity,
87
+ onError,
88
+ onProcessingChange,
89
+ } = {}) {
77
90
  await this.stop();
78
91
  if (signal?.aborted) throw abortError();
79
92
  const context = this.meter?.context,
@@ -95,16 +108,25 @@ export class WhisperHttpRecognitionEngine {
95
108
  samples: 0,
96
109
  voiced: false,
97
110
  silence: 0,
98
- inflight: new Set(),
111
+ transcriptionQueue: [],
112
+ activeRequest: null,
113
+ draining: false,
99
114
  onResult,
100
115
  onActivity,
101
116
  onError,
117
+ onProcessingChange,
102
118
  lang,
103
119
  signal,
104
120
  };
105
121
  this.session = session;
106
122
  const valid = () => this.session === session && !signal?.aborted;
107
- const submit = async () => {
123
+ const notifyProcessing = () =>
124
+ session.onProcessingChange?.({
125
+ queued: session.transcriptionQueue.length,
126
+ active: !!session.activeRequest,
127
+ pending: session.transcriptionQueue.length + (session.activeRequest ? 1 : 0),
128
+ });
129
+ const enqueue = () => {
108
130
  if (!session.voiced || session.samples < context.sampleRate * 0.25) {
109
131
  session.chunks = [];
110
132
  session.samples = 0;
@@ -122,29 +144,46 @@ export class WhisperHttpRecognitionEngine {
122
144
  session.samples = 0;
123
145
  session.voiced = false;
124
146
  session.silence = 0;
125
- const request = new AbortController();
126
- session.inflight.add(request);
147
+ session.transcriptionQueue.push(samples);
148
+ notifyProcessing();
149
+ void drain();
150
+ };
151
+ const drain = async () => {
152
+ if (session.draining || !valid()) return;
153
+ session.draining = true;
127
154
  try {
128
- const response = await this.g.fetch(this.route + '/transcribe', {
129
- method: 'POST',
130
- credentials: 'same-origin',
131
- headers: {
132
- 'content-type': 'audio/wav',
133
- 'x-dlv-client-id': session.operation,
134
- 'x-dlv-operation-id': id(),
135
- 'x-dlv-language': lang,
136
- },
137
- body: encodeMonoPcm16Wav(samples, context.sampleRate),
138
- signal: request.signal,
139
- });
140
- const json = await response.json();
141
- if (!response.ok || !json?.ok)
142
- throw new Error(json?.error?.message || 'HTTP transcription failed.');
143
- if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: '' });
144
- } catch (error) {
145
- if (error.name !== 'AbortError' && valid()) onError?.(error);
155
+ while (valid() && session.transcriptionQueue.length) {
156
+ const samples = session.transcriptionQueue.shift();
157
+ const request = new AbortController();
158
+ session.activeRequest = request;
159
+ notifyProcessing();
160
+ try {
161
+ const response = await this.g.fetch(this.route + '/transcribe', {
162
+ method: 'POST',
163
+ credentials: 'same-origin',
164
+ headers: {
165
+ 'content-type': 'audio/wav',
166
+ 'x-dlv-client-id': session.operation,
167
+ 'x-dlv-operation-id': id(),
168
+ 'x-dlv-language': lang,
169
+ },
170
+ body: encodeMonoPcm16Wav(samples, context.sampleRate),
171
+ signal: request.signal,
172
+ });
173
+ const json = await response.json();
174
+ if (!response.ok || !json?.ok)
175
+ throw new Error(json?.error?.message || 'HTTP transcription failed.');
176
+ if (valid() && json.value.text) onResult?.({ final: json.value.text, interim: '' });
177
+ } catch (error) {
178
+ if (error.name !== 'AbortError' && valid()) onError?.(error);
179
+ } finally {
180
+ if (session.activeRequest === request) session.activeRequest = null;
181
+ notifyProcessing();
182
+ }
183
+ }
146
184
  } finally {
147
- session.inflight.delete(request);
185
+ session.draining = false;
186
+ notifyProcessing();
148
187
  }
149
188
  };
150
189
  processor.onaudioprocess = (event) => {
@@ -164,14 +203,24 @@ export class WhisperHttpRecognitionEngine {
164
203
  if (
165
204
  (session.voiced &&
166
205
  session.silence > context.sampleRate * (this.segmentation.silenceMs / 1000)) ||
167
- session.samples > context.sampleRate * 20
206
+ session.samples > context.sampleRate * this.maxUtteranceSeconds
168
207
  )
169
- void submit();
208
+ enqueue();
170
209
  };
171
210
  source.connect(processor);
211
+ session.finish = enqueue;
172
212
  session.abort = () => this.stop();
173
213
  signal?.addEventListener('abort', session.abort, { once: true });
174
214
  }
215
+ finish() {
216
+ const session = this.session;
217
+ if (!session) return;
218
+ session.processor.onaudioprocess = null;
219
+ try {
220
+ this.meter?.source?.disconnect(session.processor);
221
+ } catch {}
222
+ session.finish?.();
223
+ }
175
224
  async stop() {
176
225
  const session = this.session;
177
226
  if (!session) return;
@@ -185,7 +234,9 @@ export class WhisperHttpRecognitionEngine {
185
234
  session.processor.disconnect();
186
235
  session.gain?.disconnect();
187
236
  } catch {}
188
- for (const request of session.inflight) request.abort();
189
- session.inflight.clear();
237
+ session.transcriptionQueue = [];
238
+ session.activeRequest?.abort();
239
+ session.activeRequest = null;
240
+ session.onProcessingChange?.({ queued: 0, active: false, pending: 0 });
190
241
  }
191
242
  }