dsh-live-voice 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +89 -1
  2. package/PLAN.md +25 -1
  3. package/README.md +4 -2
  4. package/docs/ARCHITECTURE.md +9 -16
  5. package/docs/CONFIGURATION.md +17 -3
  6. package/docs/VOICE-LIFECYCLE.md +5 -1
  7. package/lib/client.js +543 -326
  8. package/lib/server.js +158 -5
  9. package/package.json +2 -2
  10. package/src/app/AGENTS.md +7 -0
  11. package/src/app/ARCHITECTURE.md +10 -0
  12. package/src/app/client/apply.tsx +96 -40
  13. package/src/app/client/i18n/catalogs/base.ts +7 -0
  14. package/src/app/client/i18n/catalogs/en.ts +10 -0
  15. package/src/app/client/i18n/catalogs/es.ts +10 -0
  16. package/src/app/client/i18n/catalogs/fr.ts +10 -0
  17. package/src/app/client/i18n/catalogs/hi.ts +10 -0
  18. package/src/app/client/i18n/catalogs/pt-BR.ts +10 -0
  19. package/src/app/client/i18n/catalogs/zh.ts +9 -0
  20. package/src/app/server/apply.ts +4 -0
  21. package/src/app/server/registerRoutes.ts +40 -0
  22. package/src/modules/AGENTS.md +5 -0
  23. package/src/modules/conversation/ARCHITECTURE.md +7 -0
  24. package/src/modules/core/ARCHITECTURE.md +5 -0
  25. package/src/modules/core/coordinator.ts +8 -6
  26. package/src/modules/core/qwen/QwenHttpHost.ts +1 -1
  27. package/src/modules/core/qwen/QwenSettings.tsx +15 -14
  28. package/src/modules/core/settings.ts +26 -8
  29. package/src/modules/core/transcript.ts +5 -43
  30. package/src/modules/recognition/ARCHITECTURE.md +5 -0
  31. package/src/modules/recognition/engines/whisper/WhisperRecognitionEngine.ts +5 -3
  32. package/src/modules/recognition/engines/whisper/WhisperSettings.tsx +19 -20
  33. package/src/modules/settings/ARCHITECTURE.md +5 -0
  34. package/src/modules/settings/components/LiveVoiceSettings.tsx +11 -3
  35. package/src/modules/settings/models/settingsHost.ts +45 -0
  36. package/src/modules/settings/models/settingsStorage.ts +45 -10
  37. package/src/modules/settings/sections/conversation/ConversationSettingsSection.tsx +12 -6
  38. package/src/modules/settings/sections/recognition/RecognitionEngineSettings.tsx +2 -2
  39. package/src/modules/settings/sections/recognition/RecognitionFilterSettings.tsx +3 -3
  40. package/src/modules/settings/sections/recognition/RecognitionSettingsSection.tsx +7 -3
  41. package/src/modules/settings/sections/recognition/SilenceDetectionSettings.tsx +38 -8
  42. package/src/modules/settings/sections/recognition/VoiceCommandSettings.tsx +3 -3
  43. package/src/modules/settings/sections/speak/OutputFilterSettings.tsx +5 -5
  44. package/src/modules/settings/sections/speak/SpeakSettingsSection.tsx +7 -3
  45. package/src/modules/settings/sections/speak/SpeechAdvancedSettings.tsx +4 -4
  46. package/src/modules/settings/sections/speak/SpeechEngineSettings.tsx +2 -2
  47. package/src/modules/settings/services/releases.ts +10 -2
  48. package/src/modules/speak/ARCHITECTURE.md +5 -0
  49. package/src/shared/ARCHITECTURE.md +5 -0
  50. package/src/shared/design-system/forms/DraftField.tsx +79 -0
  51. package/src/shared/design-system/forms/NumberField.tsx +4 -10
  52. package/src/shared/design-system/forms/TextAreaField.tsx +4 -10
  53. package/src/shared/design-system/forms/TextField.tsx +4 -10
  54. package/src/shared/design-system/icons/icons.ts +4 -0
  55. package/src/shared/design-system/layout/SettingsSubcard.tsx +15 -3
  56. package/src/shared/design-system/layout/SettingsTabs.tsx +4 -2
  57. package/src/styles/index.ts +3 -3
@@ -100,6 +100,8 @@ const hi: LiveVoiceTranslation = {
100
100
  'dsh-live-voice.recognition.planned.vote': 'जल्द आ रहा है — रिपॉज़िटरी के इश्यू में वोट दें',
101
101
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — जल्द आ रहा है',
102
102
  'dsh-live-voice.recognition.planned.webGpu': 'ब्राउज़र में WebGPU इन्फ़रेंस — जल्द आ रहा है',
103
+ 'dsh-live-voice.recognition.presets.custom.description': 'मौन की अवधि स्वयं चुनें।',
104
+ 'dsh-live-voice.recognition.presets.custom.label': 'कस्टम',
103
105
  'dsh-live-voice.recognition.presets.long.description':
104
106
  'लंबे सोच-विचार वाले विरामों तक प्रतीक्षा करें।',
105
107
  'dsh-live-voice.recognition.presets.long.label': 'लंबा',
@@ -119,6 +121,9 @@ const hi: LiveVoiceTranslation = {
119
121
  'dsh-live-voice.recognition.qwen.hostHelp':
120
122
  'Qwen3 ASR + TTS सर्वर की पूरे होस्ट पर लागू होने वाली सेटिंग्स। कोई भी HTTP या HTTPS बेस URL दर्ज करें जिस तक DSH होस्ट पहुँच सके। ब्राउज़र प्रमाणीकरण वाले DSH रूटों के ज़रिए इसे एक्सेस करता है।',
121
123
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — HTTP API',
124
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
125
+ '100 से 10,000 ms तक पूर्णांक दर्ज करें। फ़ील्ड छोड़ने पर सहेजा जाता है। छोटे विराम बोलने को खंडों में बाँट सकते हैं; पहचान में अतिरिक्त समय लगता है।',
126
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': 'कस्टम विराम (मिलीसेकंड)',
122
127
  'dsh-live-voice.recognition.silenceDetection.duration':
123
128
  'भेजने से पहले का विराम: {milliseconds} मिलीसेकंड',
124
129
  'dsh-live-voice.recognition.silenceDetection.help':
@@ -165,6 +170,11 @@ const hi: LiveVoiceTranslation = {
165
170
  'dsh-live-voice.settings.delivery.toggle': 'स्वचालित भेजने का मोड',
166
171
  'dsh-live-voice.settings.engine.refresh': 'उपलब्ध इंजन रीफ़्रेश करें',
167
172
  'dsh-live-voice.settings.filters.title': 'फ़िल्टरिंग',
173
+ 'dsh-live-voice.settings.general.title': 'सामान्य',
174
+ 'dsh-live-voice.settings.persistence.loadError':
175
+ 'सर्वर से Live Voice सेटिंग लोड नहीं हो सकीं। फिर प्रयास करने के लिए पेज रीलोड करें।',
176
+ 'dsh-live-voice.settings.persistence.saveError':
177
+ 'सर्वर पर Live Voice सेटिंग सेव नहीं हो सकीं। फिर प्रयास करें।',
168
178
  'dsh-live-voice.settings.tabs.conversation': 'बातचीत',
169
179
  'dsh-live-voice.settings.tabs.recognition': 'वाक् पहचान',
170
180
  'dsh-live-voice.settings.tabs.speak': 'वाचन',
@@ -102,6 +102,8 @@ const ptBR: LiveVoiceTranslation = {
102
102
  'dsh-live-voice.recognition.planned.vote': 'Em breve — vote nas issues do repositório',
103
103
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — Em breve',
104
104
  'dsh-live-voice.recognition.planned.webGpu': 'Inferência WebGPU no navegador — Em breve',
105
+ 'dsh-live-voice.recognition.presets.custom.description': 'Escolha a duração do silêncio.',
106
+ 'dsh-live-voice.recognition.presets.custom.label': 'Personalizada',
105
107
  'dsh-live-voice.recognition.presets.long.description':
106
108
  'Aguarda durante pausas de pensamento mais longas.',
107
109
  'dsh-live-voice.recognition.presets.long.label': 'Longa',
@@ -120,6 +122,9 @@ const ptBR: LiveVoiceTranslation = {
120
122
  'dsh-live-voice.recognition.qwen.hostHelp':
121
123
  'Configurações de todo o host para o servidor Qwen3 ASR + TTS. Insira qualquer URL base HTTP ou HTTPS acessível a partir do host DSH. O navegador acessa através de rotas autenticadas do DSH.',
122
124
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — API HTTP',
125
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
126
+ 'Insira um número inteiro de 100 a 10.000 ms. Salvo ao sair do campo. Pausas curtas podem dividir a fala; o reconhecimento acrescenta sua própria latência.',
127
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': 'Pausa personalizada (milissegundos)',
123
128
  'dsh-live-voice.recognition.silenceDetection.duration':
124
129
  'Pausa antes de enviar: {milliseconds} ms',
125
130
  'dsh-live-voice.recognition.silenceDetection.help':
@@ -169,6 +174,11 @@ const ptBR: LiveVoiceTranslation = {
169
174
  'dsh-live-voice.settings.delivery.toggle': 'Modo de entrega automática',
170
175
  'dsh-live-voice.settings.engine.refresh': 'Atualizar mecanismos disponíveis',
171
176
  'dsh-live-voice.settings.filters.title': 'Filtragem',
177
+ 'dsh-live-voice.settings.general.title': 'Gerais',
178
+ 'dsh-live-voice.settings.persistence.loadError':
179
+ 'Não foi possível carregar as configurações do Live Voice do servidor. Recarregue para tentar novamente.',
180
+ 'dsh-live-voice.settings.persistence.saveError':
181
+ 'Não foi possível salvar as configurações do Live Voice no servidor. Tente novamente.',
172
182
  'dsh-live-voice.settings.tabs.conversation': 'Conversa',
173
183
  'dsh-live-voice.settings.tabs.recognition': 'Reconhecimento de voz',
174
184
  'dsh-live-voice.settings.tabs.speak': 'Fala',
@@ -97,6 +97,8 @@ const zh: LiveVoiceTranslation = {
97
97
  'dsh-live-voice.recognition.planned.vote': '即将推出 — 请在仓库议题中投票',
98
98
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — 即将推出',
99
99
  'dsh-live-voice.recognition.planned.webGpu': '浏览器 WebGPU 推理 — 即将推出',
100
+ 'dsh-live-voice.recognition.presets.custom.description': '自行选择静音时长。',
101
+ 'dsh-live-voice.recognition.presets.custom.label': '自定义',
100
102
  'dsh-live-voice.recognition.presets.long.description': '等待较长的思考停顿。',
101
103
  'dsh-live-voice.recognition.presets.long.label': '长',
102
104
  'dsh-live-voice.recognition.presets.natural.description': '允许短语之间出现自然停顿。',
@@ -113,6 +115,9 @@ const zh: LiveVoiceTranslation = {
113
115
  'dsh-live-voice.recognition.qwen.hostHelp':
114
116
  'Qwen3 ASR + TTS 服务器的主机级设置。请输入 DSH 主机可访问的任意 HTTP 或 HTTPS 基础 URL。浏览器通过需要身份验证的 DSH 路由访问该服务器。',
115
117
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — HTTP API',
118
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
119
+ '请输入 100 至 10,000 ms 的整数。离开输入框时保存。短停顿可能切分语音;识别还需要额外时间。',
120
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': '自定义停顿(毫秒)',
116
121
  'dsh-live-voice.recognition.silenceDetection.duration': '发送前停顿:{milliseconds} 毫秒',
117
122
  'dsh-live-voice.recognition.silenceDetection.help':
118
123
  '设置停顿持续多久后,将采集的语音发送进行识别。',
@@ -156,6 +161,10 @@ const zh: LiveVoiceTranslation = {
156
161
  'dsh-live-voice.settings.delivery.toggle': '自动发送模式',
157
162
  'dsh-live-voice.settings.engine.refresh': '刷新可用引擎',
158
163
  'dsh-live-voice.settings.filters.title': '过滤',
164
+ 'dsh-live-voice.settings.general.title': '常规',
165
+ 'dsh-live-voice.settings.persistence.loadError':
166
+ '无法从服务器加载 Live Voice 设置。请刷新后重试。',
167
+ 'dsh-live-voice.settings.persistence.saveError': '无法在服务器上保存 Live Voice 设置。请重试。',
159
168
  'dsh-live-voice.settings.tabs.conversation': '对话',
160
169
  'dsh-live-voice.settings.tabs.recognition': '语音识别',
161
170
  'dsh-live-voice.settings.tabs.speak': '语音合成',
@@ -11,6 +11,8 @@ import {
11
11
  } from '../../modules/recognition/engines/whisper/whisperRecognitionHost.js';
12
12
  import { wavToM4aAac } from '../../modules/speak/engines/audio/M4aAacTranscoder.js';
13
13
  import { SayEngine } from '../../modules/speak/engines/say/SaySpeakingEngine.js';
14
+ import { createSettingsStore } from '../../modules/settings/models/settingsHost.js';
15
+ import { registerSettingsRoute } from './registerRoutes.js';
14
16
 
15
17
  export const name = 'dsh-live-voice';
16
18
  export const inject = ['connection', 'systemPrompt'];
@@ -136,6 +138,7 @@ export function createSayHost({ engine = new SayEngine() } = {}) {
136
138
  export function apply(
137
139
  ctx,
138
140
  {
141
+ settingsStore = createSettingsStore(),
139
142
  whisperStore = createWhisperConfigStore(),
140
143
  whisperFetch = globalThis.fetch,
141
144
  qwenStore = createQwenConfigStore(),
@@ -145,6 +148,7 @@ export function apply(
145
148
  encodeHostSpeech = wavToM4aAac,
146
149
  } = {},
147
150
  ) {
151
+ registerSettingsRoute(ctx, settingsStore);
148
152
  ctx.systemPrompt.variable('live_voice_context', (assemblyContext) =>
149
153
  voiceContextStore.get(String(assemblyContext.agent?.sessionId || '')),
150
154
  );
@@ -1,4 +1,44 @@
1
1
  export const API_ROOT = '/api/dsh-live-voice';
2
+
3
+ export function registerSettingsRoute(
4
+ ctx: any,
5
+ store: { load(): Promise<unknown>; save(patch: unknown): Promise<unknown> },
6
+ ) {
7
+ const dispose = ctx.connection.fetch.register({
8
+ path: API_ROOT + '/settings',
9
+ methods: ['GET', 'PUT'],
10
+ requestBody: 'buffered',
11
+ fetch: async (request: Request) => {
12
+ const headers = { 'cache-control': 'no-store' };
13
+ let patch: unknown;
14
+ if (request.method === 'PUT') {
15
+ try {
16
+ const text = await request.text();
17
+ if (text.length > 64000) throw new Error('too-large');
18
+ patch = JSON.parse(text);
19
+ if (!patch || typeof patch !== 'object' || Array.isArray(patch))
20
+ throw new Error('invalid');
21
+ } catch {
22
+ return Response.json(
23
+ { ok: false, error: { code: 'invalid-settings' } },
24
+ { status: 400, headers },
25
+ );
26
+ }
27
+ }
28
+ try {
29
+ const value = request.method === 'GET' ? await store.load() : await store.save(patch);
30
+ return Response.json({ ok: true, value }, { headers });
31
+ } catch {
32
+ return Response.json(
33
+ { ok: false, error: { code: 'settings-unavailable' } },
34
+ { status: 500, headers },
35
+ );
36
+ }
37
+ },
38
+ });
39
+ ctx.effect(() => () => dispose(), 'dsh-live-voice: remove settings route');
40
+ }
41
+
2
42
  export const APP_VOICE_CONTEXT_PATH = API_ROOT + '/voice-context';
3
43
  export type RouteHost = {
4
44
  handle(endpoint: string, payload: unknown, signal?: AbortSignal): unknown;
@@ -0,0 +1,5 @@
1
+ # Module guidance
2
+
3
+ Read the nearest module's `ARCHITECTURE.md` before changing its behavior. `src/modules` owns product policy, feature UI, models, providers, and services. Modules may use `src/shared` and the centralized `src/app/client/i18n` API, but must not register DSH slots or host routes; composition belongs to `src/app`.
4
+
5
+ Preserve local-first STT and TTS options without claiming DSH itself is offline. Track capture/recognition, synthesis/playback, and agent generation separately. Preserve pause, resume, cancel, ownership, interruption, and stale asynchronous result semantics. Avoid logging raw audio or transcripts by default. Consult [VOICE-LIFECYCLE.md](../../docs/VOICE-LIFECYCLE.md) for the behavioral contract and [root guidance](../../AGENTS.md) for testing and authorization.
@@ -0,0 +1,7 @@
1
+ # Conversation module
2
+
3
+ Dictation is append-only at the current composer end. Final chunks preserve existing and manually edited text; provisional hypotheses never rewrite the editor. Client synchronization must accept a new manual publication even when the previous voice echo was skipped, while ignoring unchanged stale slot snapshots. Explicit Clear and normal submit actions are separate from dictation.
4
+
5
+ `models/chat.ts` interprets DSH chat turns and pending questions. `hooks/useConversationController.ts` subscribes to session state; `hooks/useConversationActions.ts` maps user actions onto the controller. `components/` renders composer controls, microphone state, playback controls, status, and waveform; `createConversationComponents.tsx` composes them for the application slot boundary.
6
+
7
+ Keep UI components session-facing rather than importing host engines or registering slots. Preserve the DSH composer contract: `useInput` provides subscribed input and `inputActions.setDraft()` updates the composer; a recognition result must not be lost to stale interim updates. History pagination must not count as a new user turn. Do not treat sending, recognition, playback, or agent generation as one state. See [voice lifecycle](../../../docs/VOICE-LIFECYCLE.md) before changing interruption or turn-taking.
@@ -0,0 +1,5 @@
1
+ # Core voice domain
2
+
3
+ `core` owns policy and reusable voice-domain state, not DSH slot/route registration or provider-specific presentation. `coordinator.ts` serializes input transitions, owns conversation state and speech queues, and invalidates stale async work with generations. `ownership.ts` arbitrates microphone and speaker use between sessions and Settings self-tests. `transcript.ts` enforces append-only final recognition text (interim hypotheses never own or rewrite composer content), `microphone.ts` manages capture and activity, `filters.ts` handles voice commands and speech text filtering, and `settings.ts` normalizes persisted values and defaults.
4
+
5
+ The `qwen/` subtree contains shared Qwen host configuration and its settings UI; recognition and speaking adapters live in their respective modules. Keep the client/host dependency graphs separate. Preserve the persisted `dsh-live-voice.settings` key and normalization compatibility. Distinguish capture, recognition, playback, and generation; do not conflate pausing with cancellation or resume obsolete speech after a new turn. Consult [voice lifecycle](../../../docs/VOICE-LIFECYCLE.md) for expected transitions.
@@ -100,6 +100,7 @@ export class VoiceCoordinator {
100
100
  processLocally: this.snapshot.settings.recognitionProcessLocally,
101
101
  autoInstallLocalPack: this.snapshot.settings.recognitionAutoInstall,
102
102
  voiceDetectionPreset: this.snapshot.settings.voiceDetectionPreset,
103
+ voiceDetectionCustomSilenceMs: this.snapshot.settings.voiceDetectionCustomSilenceMs,
103
104
  });
104
105
  this.patch({ capabilities: { ...this.snapshot.capabilities, recognition: undefined } });
105
106
  }
@@ -125,6 +126,7 @@ export class VoiceCoordinator {
125
126
  processLocally: settings.recognitionProcessLocally,
126
127
  autoInstallLocalPack: settings.recognitionAutoInstall,
127
128
  voiceDetectionPreset: settings.voiceDetectionPreset,
129
+ voiceDetectionCustomSilenceMs: settings.voiceDetectionCustomSilenceMs,
128
130
  });
129
131
  this.patch({
130
132
  settings,
@@ -223,7 +225,7 @@ export class VoiceCoordinator {
223
225
  }
224
226
  muteListening() {
225
227
  this.cancelAutoSend();
226
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
228
+ // Discard pending recognition without rewriting the composer.
227
229
  this.transcript.reset();
228
230
  this.updateSettings({ microphoneEnabled: false });
229
231
  if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
@@ -444,7 +446,7 @@ export class VoiceCoordinator {
444
446
  clear: this.snapshot.settings.voiceCommandClear,
445
447
  });
446
448
  if (command) {
447
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
449
+ // Discard pending recognition without rewriting the composer.
448
450
  this.transcript.reset();
449
451
  this.cancelAutoSend();
450
452
  this.patch({ recognizing: false });
@@ -468,7 +470,7 @@ export class VoiceCoordinator {
468
470
  return;
469
471
  }
470
472
  if (this.snapshot.muted) {
471
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
473
+ // Discard pending recognition without rewriting the composer.
472
474
  this.transcript.reset();
473
475
  this.patch({ recognizing: false });
474
476
  return;
@@ -477,7 +479,7 @@ export class VoiceCoordinator {
477
479
  this.snapshot.settings.recognitionFilterEnabled &&
478
480
  !hasMinimumWords(final, this.snapshot.settings.recognitionMinimumWords)
479
481
  ) {
480
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
482
+ // Discard pending recognition without rewriting the composer.
481
483
  this.transcript.reset();
482
484
  this.patch({ recognizing: false });
483
485
  return;
@@ -488,7 +490,7 @@ export class VoiceCoordinator {
488
490
  }
489
491
  if (interim) {
490
492
  this.cancelAutoSend();
491
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), interim));
493
+ // Interim hypotheses affect recognition status only, never composer content.
492
494
  }
493
495
  if (!interim && this.snapshot.recognizing)
494
496
  this.assistantSpeechNotBefore =
@@ -571,7 +573,7 @@ export class VoiceCoordinator {
571
573
  });
572
574
  }
573
575
  async cancelDictation() {
574
- this.composer.setDraft(this.transcript.update(this.composer.getDraft(), '', true));
576
+ // Discard pending recognition without rewriting the composer.
575
577
  await this.stopListening();
576
578
  }
577
579
  async endConversation() {
@@ -64,7 +64,7 @@ export class QwenHttpHost {
64
64
  baseUrl,
65
65
  timeoutMs = 300000,
66
66
  fetchImpl = globalThis.fetch,
67
- maxBytes = 2_000_000,
67
+ maxBytes = 50_000_000,
68
68
  maxSpeechBytes = 50_000_000,
69
69
  store,
70
70
  } = {}) {
@@ -1,5 +1,6 @@
1
1
  // @ts-nocheck
2
2
  import React from 'react';
3
+ import { TextField, NumberField } from '../../../shared/design-system/index.js';
3
4
  import { useLanguage } from '../../../app/client/i18n/index.js';
4
5
 
5
6
  const BASE = '/api/dsh-live-voice/qwen';
@@ -56,7 +57,7 @@ export function QwenSettings({ controller }) {
56
57
  'dsh-live-voice.speak.qwen.healthFailed': speak.qwen.healthFailed(),
57
58
  })[value] ?? value;
58
59
 
59
- async function run(action) {
60
+ async function run(action, config = draft) {
60
61
  active.current?.abort();
61
62
  const abort = new AbortController();
62
63
  active.current = abort;
@@ -74,7 +75,7 @@ export function QwenSettings({ controller }) {
74
75
  await controller.endConversation?.();
75
76
  const value = await qwenSettingsRequest('/config', {
76
77
  method: 'PUT',
77
- config: { ...draft, timeoutMs: Number(draft.timeoutMs) },
78
+ config: { ...config, timeoutMs: Number(config.timeoutMs) },
78
79
  signal: abort.signal,
79
80
  });
80
81
  if (!abort.signal.aborted) {
@@ -85,7 +86,7 @@ export function QwenSettings({ controller }) {
85
86
  } else {
86
87
  const value = await qwenSettingsRequest('/test', {
87
88
  method: 'POST',
88
- config: { ...draft, timeoutMs: Number(draft.timeoutMs) },
89
+ config: { ...config, timeoutMs: Number(config.timeoutMs) },
89
90
  signal: abort.signal,
90
91
  });
91
92
  if (!abort.signal.aborted) {
@@ -104,23 +105,23 @@ export function QwenSettings({ controller }) {
104
105
  void run('load');
105
106
  return () => active.current?.abort();
106
107
  }, []);
107
- const field = (label, key, type = 'text') => (
108
- <label>
109
- {label}
110
- <input
111
- type={type}
108
+ const field = (label, key, type = 'text') => {
109
+ const Field = type === 'number' ? NumberField : TextField;
110
+ return (
111
+ <Field
112
+ label={label}
112
113
  value={draft[key]}
113
114
  disabled={busy || !loaded}
114
115
  autoComplete="off"
115
116
  {...(type === 'number' ? { min: 1000, max: 600000, step: 1 } : {})}
116
- onChange={(event) => {
117
- setDraft({ ...draft, [key]: event.target.value });
118
- setMessage('dsh-live-voice.commons.connection.unsaved');
119
- setError('');
117
+ onCommit={(value) => {
118
+ const config = { ...draft, [key]: value };
119
+ setDraft(config);
120
+ void run('save', config);
120
121
  }}
121
122
  />
122
- </label>
123
- );
123
+ );
124
+ };
124
125
  return (
125
126
  <>
126
127
  <p>{recognition.qwen.hostHelp()}</p>
@@ -1,17 +1,17 @@
1
1
  // @ts-nocheck
2
2
  export const voiceDetectionPresets = Object.freeze({
3
3
  short: Object.freeze({
4
- silenceMs: 900,
4
+ silenceMs: 500,
5
5
  label: 'Short',
6
6
  description: 'Send quickly after a short pause.',
7
7
  }),
8
8
  natural: Object.freeze({
9
- silenceMs: 1500,
9
+ silenceMs: 1000,
10
10
  label: 'Natural',
11
11
  description: 'Allow normal pauses between phrases.',
12
12
  }),
13
13
  long: Object.freeze({
14
- silenceMs: 2200,
14
+ silenceMs: 2000,
15
15
  label: 'Long',
16
16
  description: 'Wait through longer thinking pauses.',
17
17
  }),
@@ -39,7 +39,8 @@ export const defaultSettings = Object.freeze({
39
39
  recognitionEngine: 'browser',
40
40
  recognitionProcessLocally: true,
41
41
  recognitionAutoInstall: true,
42
- voiceDetectionPreset: 'natural',
42
+ voiceDetectionPreset: 'short',
43
+ voiceDetectionCustomSilenceMs: 1000,
43
44
  recognitionMaxUtteranceSeconds: 60,
44
45
  microphoneEnabled: true,
45
46
  holdToTalkEnabled: true,
@@ -82,7 +83,21 @@ const normalizeCommandPhrases = (value, fallback) =>
82
83
  .join(', ')
83
84
  : fallback;
84
85
 
85
- /** Persisted browser preferences are untrusted and may belong to an older version. */
86
+ export const customSilenceMinMs = 100;
87
+ export const customSilenceMaxMs = 10000;
88
+ export const normalizeCustomSilenceMs = (value) =>
89
+ Number.isInteger(value) && value >= customSilenceMinMs && value <= customSilenceMaxMs
90
+ ? value
91
+ : defaultSettings.voiceDetectionCustomSilenceMs;
92
+ export const voiceDetectionSilenceMs = (settings) =>
93
+ settings.voiceDetectionPreset === 'custom'
94
+ ? normalizeCustomSilenceMs(settings.voiceDetectionCustomSilenceMs)
95
+ : (
96
+ voiceDetectionPresets[settings.voiceDetectionPreset] ||
97
+ voiceDetectionPresets[defaultSettings.voiceDetectionPreset]
98
+ ).silenceMs;
99
+
100
+ /** Persisted preferences are untrusted and may belong to an older version. */
86
101
  export function normalizeSettings(value) {
87
102
  const source = value && typeof value === 'object' && !Array.isArray(value) ? value : {};
88
103
  return {
@@ -100,9 +115,12 @@ export function normalizeSettings(value) {
100
115
  typeof source.recognitionAutoInstall === 'boolean'
101
116
  ? source.recognitionAutoInstall
102
117
  : defaultSettings.recognitionAutoInstall,
103
- voiceDetectionPreset: Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset)
104
- ? source.voiceDetectionPreset
105
- : defaultSettings.voiceDetectionPreset,
118
+ voiceDetectionPreset:
119
+ source.voiceDetectionPreset === 'custom' ||
120
+ Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset)
121
+ ? source.voiceDetectionPreset
122
+ : defaultSettings.voiceDetectionPreset,
123
+ voiceDetectionCustomSilenceMs: normalizeCustomSilenceMs(source.voiceDetectionCustomSilenceMs),
106
124
  recognitionMaxUtteranceSeconds:
107
125
  Number.isInteger(source.recognitionMaxUtteranceSeconds) &&
108
126
  source.recognitionMaxUtteranceSeconds >= 10 &&
@@ -1,48 +1,10 @@
1
1
  // @ts-nocheck
2
- /** Own only the current hypothesis; never restore an old snapshot over user edits. */
2
+ /** Dictation is append-only: provisional hypotheses never own composer text. */
3
3
  export class TranscriptDraft {
4
- constructor() {
5
- this.reset();
6
- }
7
- reset() {
8
- this.owned = null;
9
- this.edited = false;
10
- }
4
+ reset() {}
11
5
  update(current, hypothesis, final = false) {
12
- if (this.edited) {
13
- if (final) this.edited = false;
14
- return current;
15
- }
16
- let base = current;
17
- let at = current.length;
18
- if (this.owned) {
19
- const { start, text, before, after } = this.owned;
20
- // Only replace a hypothesis whose surrounding text has not changed.
21
- if (current === before + text + after) {
22
- base = before + after;
23
- at = start;
24
- } else if (
25
- text &&
26
- current.indexOf(text) >= 0 &&
27
- current.indexOf(text) === current.lastIndexOf(text)
28
- ) {
29
- // Edits outside our unchanged, uniquely identifiable hypothesis are safe.
30
- at = current.indexOf(text);
31
- base = current.slice(0, at) + current.slice(at + text.length);
32
- } else {
33
- // The user edited the draft: relinquish it. Do not duplicate a final
34
- // hypothesis after an edit; the edited text belongs to the user now.
35
- this.owned = null;
36
- this.edited = !final;
37
- return current;
38
- }
39
- }
40
- const before = base.slice(0, at);
41
- const after = base.slice(at);
42
- const separator = hypothesis && before && !/\s$/.test(before) ? ' ' : '';
43
- const text = separator + hypothesis;
44
- const result = before + text + after;
45
- this.owned = final ? null : { start: at, text, before, after };
46
- return result;
6
+ if (!final || !hypothesis) return current;
7
+ const separator = current && !/\s$/.test(current) ? ' ' : '';
8
+ return current + separator + hypothesis;
47
9
  }
48
10
  }
@@ -0,0 +1,5 @@
1
+ # Recognition module
2
+
3
+ `engines/browser/BrowserRecognitionEngine.ts` adapts browser speech recognition and reports local-pack/browser capabilities. `engines/qwen/QwenRecognitionEngine.ts` and `engines/whisper/WhisperRecognitionEngine.ts` adapt local-first HTTP recognition; host bridges live beside the relevant adapters (Whisper here, shared Qwen configuration in `../core/qwen`). `components/RecognitionCapabilityStatus.tsx` and engine settings UI present capability and configuration states.
4
+
5
+ Distinguish microphone capture and permission from recognition engine availability. Do not infer feature support merely from operating system. Bound audio and host requests, keep host-only code out of the browser bundle, and abort/ignore stale results after cancellation. Recognition language and voice-command phrases are independent of the interface locale. Browser simulations are not a physical microphone or model acceptance test.
@@ -1,5 +1,5 @@
1
1
  // @ts-nocheck
2
- import { voiceDetectionPresets } from '../../../core/settings.js';
2
+ import { defaultSettings, voiceDetectionSilenceMs } from '../../../core/settings.js';
3
3
 
4
4
  const ROUTE = '/api/dsh-live-voice/whisper';
5
5
  const id = () => globalThis.crypto.randomUUID();
@@ -42,7 +42,8 @@ export class WhisperHttpRecognitionEngine {
42
42
  constructor({
43
43
  globals = globalThis,
44
44
  meter,
45
- voiceDetectionPreset = 'natural',
45
+ voiceDetectionPreset = defaultSettings.voiceDetectionPreset,
46
+ voiceDetectionCustomSilenceMs = defaultSettings.voiceDetectionCustomSilenceMs,
46
47
  maxUtteranceSeconds = 60,
47
48
  } = {}) {
48
49
  this.g = globals;
@@ -50,11 +51,12 @@ export class WhisperHttpRecognitionEngine {
50
51
  this.session = null;
51
52
  this.lang = 'pt-BR';
52
53
  this.voiceDetectionPreset = voiceDetectionPreset;
54
+ this.voiceDetectionCustomSilenceMs = voiceDetectionCustomSilenceMs;
53
55
  this.maxUtteranceSeconds = maxUtteranceSeconds;
54
56
  this.route = ROUTE;
55
57
  }
56
58
  get segmentation() {
57
- return voiceDetectionPresets[this.voiceDetectionPreset] || voiceDetectionPresets.natural;
59
+ return { silenceMs: voiceDetectionSilenceMs(this) };
58
60
  }
59
61
  async capability() {
60
62
  try {
@@ -1,5 +1,6 @@
1
1
  // @ts-nocheck
2
2
  import React from 'react';
3
+ import { TextField, NumberField } from '../../../../shared/design-system/index.js';
3
4
  import { useLanguage } from '../../../../app/client/i18n/index.js';
4
5
 
5
6
  const BASE = '/api/dsh-live-voice/whisper';
@@ -55,7 +56,7 @@ export function WhisperSettings({ controller }) {
55
56
  'dsh-live-voice.recognition.whisper.requestFailed': recognition.whisper.requestFailed(),
56
57
  'dsh-live-voice.recognition.whisper.healthFailed': recognition.whisper.healthFailed(),
57
58
  })[value] ?? value;
58
- async function run(action) {
59
+ async function run(action, config = draft) {
59
60
  active.current?.abort();
60
61
  const abort = new AbortController();
61
62
  active.current = abort;
@@ -73,7 +74,7 @@ export function WhisperSettings({ controller }) {
73
74
  await controller.endConversation?.();
74
75
  const value = await whisperSettingsRequest('/config', {
75
76
  method: 'PUT',
76
- config: { ...draft, timeoutMs: Number(draft.timeoutMs) },
77
+ config: { ...config, timeoutMs: Number(config.timeoutMs) },
77
78
  signal: abort.signal,
78
79
  });
79
80
  if (!abort.signal.aborted) {
@@ -84,7 +85,7 @@ export function WhisperSettings({ controller }) {
84
85
  } else {
85
86
  const value = await whisperSettingsRequest('/test', {
86
87
  method: 'POST',
87
- config: { ...draft, timeoutMs: Number(draft.timeoutMs) },
88
+ config: { ...config, timeoutMs: Number(config.timeoutMs) },
88
89
  signal: abort.signal,
89
90
  });
90
91
  if (!abort.signal.aborted) {
@@ -103,25 +104,23 @@ export function WhisperSettings({ controller }) {
103
104
  void run('load');
104
105
  return () => active.current?.abort();
105
106
  }, []);
106
- function field(label, key, type = 'text') {
107
+ const field = (label, key, type = 'text') => {
108
+ const Field = type === 'number' ? NumberField : TextField;
107
109
  return (
108
- <label>
109
- {label}
110
- <input
111
- type={type}
112
- value={draft[key]}
113
- disabled={busy || !loaded}
114
- autoComplete="off"
115
- {...(type === 'number' ? { min: 100, max: 300000, step: 1 } : {})}
116
- onChange={(event) => {
117
- setDraft({ ...draft, [key]: event.target.value });
118
- setMessage('dsh-live-voice.commons.connection.unsaved');
119
- setError('');
120
- }}
121
- />
122
- </label>
110
+ <Field
111
+ label={label}
112
+ value={draft[key]}
113
+ disabled={busy || !loaded}
114
+ autoComplete="off"
115
+ {...(type === 'number' ? { min: 100, max: 300000, step: 1 } : {})}
116
+ onCommit={(value) => {
117
+ const config = { ...draft, [key]: value };
118
+ setDraft(config);
119
+ void run('save', config);
120
+ }}
121
+ />
123
122
  );
124
- }
123
+ };
125
124
  return (
126
125
  <>
127
126
  <p>{settings.whisper.hostHelp()}</p>
@@ -0,0 +1,5 @@
1
+ # Settings module
2
+
3
+ `components/LiveVoiceSettings.tsx` owns the tabbed Settings shell; `sections/{conversation,recognition,speak}` own feature controls. `hooks/useLiveVoiceSettings.ts` subscribes to controller state and dispatches updates, while `hooks/useAudioDevices.ts` and `hooks/useReleaseStatus.ts` handle capability-adjacent data. `models/settingsStorage.ts` owns the authenticated client transport and in-memory preferences; `models/settingsHost.ts` owns serialized, atomic host-side persistence; `services/releases.ts` supports version metadata. Shared fields, layout, icons, and feedback live in `src/shared/design-system`.
4
+
5
+ Preserve existing setting names, default values, handlers, and normalization in `../core/settings.ts`; reorganizing presentation must not silently change policy. Keep transient capability/permission messages visible even when an options section is collapsed. User-facing copy and accessible names belong in all typed catalogs under `src/app/client/i18n`; UI locale does not select recognition or speech language. The preview in `__previewjs__/LiveVoicePreviews.tsx` exercises presentation with a fake controller, not a physical microphone or authenticated DSH runtime. Run `npm test` after changes and check the preview for layout and interaction.
@@ -17,9 +17,17 @@ export function LiveVoiceSettings({ controller, onClose }: { controller: any; on
17
17
  const [activeTab, setActiveTab] = React.useState('conversation');
18
18
  const tabsId = React.useId();
19
19
  const tabs = [
20
- { id: 'speech', label: (settingsLanguage as any).tabs.speak() },
21
- { id: 'recognition', label: (settingsLanguage as any).tabs.recognition() },
22
- { id: 'conversation', label: (settingsLanguage as any).tabs.conversation() },
20
+ { id: 'speech', label: (settingsLanguage as any).tabs.speak(), icon: 'speaker' as const },
21
+ {
22
+ id: 'recognition',
23
+ label: (settingsLanguage as any).tabs.recognition(),
24
+ icon: 'mic' as const,
25
+ },
26
+ {
27
+ id: 'conversation',
28
+ label: (settingsLanguage as any).tabs.conversation(),
29
+ icon: 'send' as const,
30
+ },
23
31
  ];
24
32
  const sectionProps = {
25
33
  controller,