dsh-live-voice 0.3.0 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +89 -1
- package/PLAN.md +25 -1
- package/README.md +4 -2
- package/docs/ARCHITECTURE.md +9 -16
- package/docs/CONFIGURATION.md +17 -3
- package/docs/VOICE-LIFECYCLE.md +5 -1
- package/lib/client.js +543 -326
- package/lib/server.js +158 -5
- package/package.json +2 -2
- package/src/app/AGENTS.md +7 -0
- package/src/app/ARCHITECTURE.md +10 -0
- package/src/app/client/apply.tsx +96 -40
- package/src/app/client/i18n/catalogs/base.ts +7 -0
- package/src/app/client/i18n/catalogs/en.ts +10 -0
- package/src/app/client/i18n/catalogs/es.ts +10 -0
- package/src/app/client/i18n/catalogs/fr.ts +10 -0
- package/src/app/client/i18n/catalogs/hi.ts +10 -0
- package/src/app/client/i18n/catalogs/pt-BR.ts +10 -0
- package/src/app/client/i18n/catalogs/zh.ts +9 -0
- package/src/app/server/apply.ts +4 -0
- package/src/app/server/registerRoutes.ts +40 -0
- package/src/modules/AGENTS.md +5 -0
- package/src/modules/conversation/ARCHITECTURE.md +7 -0
- package/src/modules/core/ARCHITECTURE.md +5 -0
- package/src/modules/core/coordinator.ts +8 -6
- package/src/modules/core/qwen/QwenHttpHost.ts +1 -1
- package/src/modules/core/qwen/QwenSettings.tsx +15 -14
- package/src/modules/core/settings.ts +26 -8
- package/src/modules/core/transcript.ts +5 -43
- package/src/modules/recognition/ARCHITECTURE.md +5 -0
- package/src/modules/recognition/engines/whisper/WhisperRecognitionEngine.ts +5 -3
- package/src/modules/recognition/engines/whisper/WhisperSettings.tsx +19 -20
- package/src/modules/settings/ARCHITECTURE.md +5 -0
- package/src/modules/settings/components/LiveVoiceSettings.tsx +11 -3
- package/src/modules/settings/models/settingsHost.ts +45 -0
- package/src/modules/settings/models/settingsStorage.ts +45 -10
- package/src/modules/settings/sections/conversation/ConversationSettingsSection.tsx +12 -6
- package/src/modules/settings/sections/recognition/RecognitionEngineSettings.tsx +2 -2
- package/src/modules/settings/sections/recognition/RecognitionFilterSettings.tsx +3 -3
- package/src/modules/settings/sections/recognition/RecognitionSettingsSection.tsx +7 -3
- package/src/modules/settings/sections/recognition/SilenceDetectionSettings.tsx +38 -8
- package/src/modules/settings/sections/recognition/VoiceCommandSettings.tsx +3 -3
- package/src/modules/settings/sections/speak/OutputFilterSettings.tsx +5 -5
- package/src/modules/settings/sections/speak/SpeakSettingsSection.tsx +7 -3
- package/src/modules/settings/sections/speak/SpeechAdvancedSettings.tsx +4 -4
- package/src/modules/settings/sections/speak/SpeechEngineSettings.tsx +2 -2
- package/src/modules/settings/services/releases.ts +10 -2
- package/src/modules/speak/ARCHITECTURE.md +5 -0
- package/src/shared/ARCHITECTURE.md +5 -0
- package/src/shared/design-system/forms/DraftField.tsx +79 -0
- package/src/shared/design-system/forms/NumberField.tsx +4 -10
- package/src/shared/design-system/forms/TextAreaField.tsx +4 -10
- package/src/shared/design-system/forms/TextField.tsx +4 -10
- package/src/shared/design-system/icons/icons.ts +4 -0
- package/src/shared/design-system/layout/SettingsSubcard.tsx +15 -3
- package/src/shared/design-system/layout/SettingsTabs.tsx +4 -2
- package/src/styles/index.ts +3 -3
|
@@ -100,6 +100,8 @@ const hi: LiveVoiceTranslation = {
|
|
|
100
100
|
'dsh-live-voice.recognition.planned.vote': 'जल्द आ रहा है — रिपॉज़िटरी के इश्यू में वोट दें',
|
|
101
101
|
'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — जल्द आ रहा है',
|
|
102
102
|
'dsh-live-voice.recognition.planned.webGpu': 'ब्राउज़र में WebGPU इन्फ़रेंस — जल्द आ रहा है',
|
|
103
|
+
'dsh-live-voice.recognition.presets.custom.description': 'मौन की अवधि स्वयं चुनें।',
|
|
104
|
+
'dsh-live-voice.recognition.presets.custom.label': 'कस्टम',
|
|
103
105
|
'dsh-live-voice.recognition.presets.long.description':
|
|
104
106
|
'लंबे सोच-विचार वाले विरामों तक प्रतीक्षा करें।',
|
|
105
107
|
'dsh-live-voice.recognition.presets.long.label': 'लंबा',
|
|
@@ -119,6 +121,9 @@ const hi: LiveVoiceTranslation = {
|
|
|
119
121
|
'dsh-live-voice.recognition.qwen.hostHelp':
|
|
120
122
|
'Qwen3 ASR + TTS सर्वर की पूरे होस्ट पर लागू होने वाली सेटिंग्स। कोई भी HTTP या HTTPS बेस URL दर्ज करें जिस तक DSH होस्ट पहुँच सके। ब्राउज़र प्रमाणीकरण वाले DSH रूटों के ज़रिए इसे एक्सेस करता है।',
|
|
121
123
|
'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — HTTP API',
|
|
124
|
+
'dsh-live-voice.recognition.silenceDetection.customHelp':
|
|
125
|
+
'100 से 10,000 ms तक पूर्णांक दर्ज करें। फ़ील्ड छोड़ने पर सहेजा जाता है। छोटे विराम बोलने को खंडों में बाँट सकते हैं; पहचान में अतिरिक्त समय लगता है।',
|
|
126
|
+
'dsh-live-voice.recognition.silenceDetection.customLabel': 'कस्टम विराम (मिलीसेकंड)',
|
|
122
127
|
'dsh-live-voice.recognition.silenceDetection.duration':
|
|
123
128
|
'भेजने से पहले का विराम: {milliseconds} मिलीसेकंड',
|
|
124
129
|
'dsh-live-voice.recognition.silenceDetection.help':
|
|
@@ -165,6 +170,11 @@ const hi: LiveVoiceTranslation = {
|
|
|
165
170
|
'dsh-live-voice.settings.delivery.toggle': 'स्वचालित भेजने का मोड',
|
|
166
171
|
'dsh-live-voice.settings.engine.refresh': 'उपलब्ध इंजन रीफ़्रेश करें',
|
|
167
172
|
'dsh-live-voice.settings.filters.title': 'फ़िल्टरिंग',
|
|
173
|
+
'dsh-live-voice.settings.general.title': 'सामान्य',
|
|
174
|
+
'dsh-live-voice.settings.persistence.loadError':
|
|
175
|
+
'सर्वर से Live Voice सेटिंग लोड नहीं हो सकीं। फिर प्रयास करने के लिए पेज रीलोड करें।',
|
|
176
|
+
'dsh-live-voice.settings.persistence.saveError':
|
|
177
|
+
'सर्वर पर Live Voice सेटिंग सेव नहीं हो सकीं। फिर प्रयास करें।',
|
|
168
178
|
'dsh-live-voice.settings.tabs.conversation': 'बातचीत',
|
|
169
179
|
'dsh-live-voice.settings.tabs.recognition': 'वाक् पहचान',
|
|
170
180
|
'dsh-live-voice.settings.tabs.speak': 'वाचन',
|
|
@@ -102,6 +102,8 @@ const ptBR: LiveVoiceTranslation = {
|
|
|
102
102
|
'dsh-live-voice.recognition.planned.vote': 'Em breve — vote nas issues do repositório',
|
|
103
103
|
'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — Em breve',
|
|
104
104
|
'dsh-live-voice.recognition.planned.webGpu': 'Inferência WebGPU no navegador — Em breve',
|
|
105
|
+
'dsh-live-voice.recognition.presets.custom.description': 'Escolha a duração do silêncio.',
|
|
106
|
+
'dsh-live-voice.recognition.presets.custom.label': 'Personalizada',
|
|
105
107
|
'dsh-live-voice.recognition.presets.long.description':
|
|
106
108
|
'Aguarda durante pausas de pensamento mais longas.',
|
|
107
109
|
'dsh-live-voice.recognition.presets.long.label': 'Longa',
|
|
@@ -120,6 +122,9 @@ const ptBR: LiveVoiceTranslation = {
|
|
|
120
122
|
'dsh-live-voice.recognition.qwen.hostHelp':
|
|
121
123
|
'Configurações de todo o host para o servidor Qwen3 ASR + TTS. Insira qualquer URL base HTTP ou HTTPS acessível a partir do host DSH. O navegador acessa através de rotas autenticadas do DSH.',
|
|
122
124
|
'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — API HTTP',
|
|
125
|
+
'dsh-live-voice.recognition.silenceDetection.customHelp':
|
|
126
|
+
'Insira um número inteiro de 100 a 10.000 ms. Salvo ao sair do campo. Pausas curtas podem dividir a fala; o reconhecimento acrescenta sua própria latência.',
|
|
127
|
+
'dsh-live-voice.recognition.silenceDetection.customLabel': 'Pausa personalizada (milissegundos)',
|
|
123
128
|
'dsh-live-voice.recognition.silenceDetection.duration':
|
|
124
129
|
'Pausa antes de enviar: {milliseconds} ms',
|
|
125
130
|
'dsh-live-voice.recognition.silenceDetection.help':
|
|
@@ -169,6 +174,11 @@ const ptBR: LiveVoiceTranslation = {
|
|
|
169
174
|
'dsh-live-voice.settings.delivery.toggle': 'Modo de entrega automática',
|
|
170
175
|
'dsh-live-voice.settings.engine.refresh': 'Atualizar mecanismos disponíveis',
|
|
171
176
|
'dsh-live-voice.settings.filters.title': 'Filtragem',
|
|
177
|
+
'dsh-live-voice.settings.general.title': 'Gerais',
|
|
178
|
+
'dsh-live-voice.settings.persistence.loadError':
|
|
179
|
+
'Não foi possível carregar as configurações do Live Voice do servidor. Recarregue para tentar novamente.',
|
|
180
|
+
'dsh-live-voice.settings.persistence.saveError':
|
|
181
|
+
'Não foi possível salvar as configurações do Live Voice no servidor. Tente novamente.',
|
|
172
182
|
'dsh-live-voice.settings.tabs.conversation': 'Conversa',
|
|
173
183
|
'dsh-live-voice.settings.tabs.recognition': 'Reconhecimento de voz',
|
|
174
184
|
'dsh-live-voice.settings.tabs.speak': 'Fala',
|
|
@@ -97,6 +97,8 @@ const zh: LiveVoiceTranslation = {
|
|
|
97
97
|
'dsh-live-voice.recognition.planned.vote': '即将推出 — 请在仓库议题中投票',
|
|
98
98
|
'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — 即将推出',
|
|
99
99
|
'dsh-live-voice.recognition.planned.webGpu': '浏览器 WebGPU 推理 — 即将推出',
|
|
100
|
+
'dsh-live-voice.recognition.presets.custom.description': '自行选择静音时长。',
|
|
101
|
+
'dsh-live-voice.recognition.presets.custom.label': '自定义',
|
|
100
102
|
'dsh-live-voice.recognition.presets.long.description': '等待较长的思考停顿。',
|
|
101
103
|
'dsh-live-voice.recognition.presets.long.label': '长',
|
|
102
104
|
'dsh-live-voice.recognition.presets.natural.description': '允许短语之间出现自然停顿。',
|
|
@@ -113,6 +115,9 @@ const zh: LiveVoiceTranslation = {
|
|
|
113
115
|
'dsh-live-voice.recognition.qwen.hostHelp':
|
|
114
116
|
'Qwen3 ASR + TTS 服务器的主机级设置。请输入 DSH 主机可访问的任意 HTTP 或 HTTPS 基础 URL。浏览器通过需要身份验证的 DSH 路由访问该服务器。',
|
|
115
117
|
'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — HTTP API',
|
|
118
|
+
'dsh-live-voice.recognition.silenceDetection.customHelp':
|
|
119
|
+
'请输入 100 至 10,000 ms 的整数。离开输入框时保存。短停顿可能切分语音;识别还需要额外时间。',
|
|
120
|
+
'dsh-live-voice.recognition.silenceDetection.customLabel': '自定义停顿(毫秒)',
|
|
116
121
|
'dsh-live-voice.recognition.silenceDetection.duration': '发送前停顿:{milliseconds} 毫秒',
|
|
117
122
|
'dsh-live-voice.recognition.silenceDetection.help':
|
|
118
123
|
'设置停顿持续多久后,将采集的语音发送进行识别。',
|
|
@@ -156,6 +161,10 @@ const zh: LiveVoiceTranslation = {
|
|
|
156
161
|
'dsh-live-voice.settings.delivery.toggle': '自动发送模式',
|
|
157
162
|
'dsh-live-voice.settings.engine.refresh': '刷新可用引擎',
|
|
158
163
|
'dsh-live-voice.settings.filters.title': '过滤',
|
|
164
|
+
'dsh-live-voice.settings.general.title': '常规',
|
|
165
|
+
'dsh-live-voice.settings.persistence.loadError':
|
|
166
|
+
'无法从服务器加载 Live Voice 设置。请刷新后重试。',
|
|
167
|
+
'dsh-live-voice.settings.persistence.saveError': '无法在服务器上保存 Live Voice 设置。请重试。',
|
|
159
168
|
'dsh-live-voice.settings.tabs.conversation': '对话',
|
|
160
169
|
'dsh-live-voice.settings.tabs.recognition': '语音识别',
|
|
161
170
|
'dsh-live-voice.settings.tabs.speak': '语音合成',
|
package/src/app/server/apply.ts
CHANGED
|
@@ -11,6 +11,8 @@ import {
|
|
|
11
11
|
} from '../../modules/recognition/engines/whisper/whisperRecognitionHost.js';
|
|
12
12
|
import { wavToM4aAac } from '../../modules/speak/engines/audio/M4aAacTranscoder.js';
|
|
13
13
|
import { SayEngine } from '../../modules/speak/engines/say/SaySpeakingEngine.js';
|
|
14
|
+
import { createSettingsStore } from '../../modules/settings/models/settingsHost.js';
|
|
15
|
+
import { registerSettingsRoute } from './registerRoutes.js';
|
|
14
16
|
|
|
15
17
|
export const name = 'dsh-live-voice';
|
|
16
18
|
export const inject = ['connection', 'systemPrompt'];
|
|
@@ -136,6 +138,7 @@ export function createSayHost({ engine = new SayEngine() } = {}) {
|
|
|
136
138
|
export function apply(
|
|
137
139
|
ctx,
|
|
138
140
|
{
|
|
141
|
+
settingsStore = createSettingsStore(),
|
|
139
142
|
whisperStore = createWhisperConfigStore(),
|
|
140
143
|
whisperFetch = globalThis.fetch,
|
|
141
144
|
qwenStore = createQwenConfigStore(),
|
|
@@ -145,6 +148,7 @@ export function apply(
|
|
|
145
148
|
encodeHostSpeech = wavToM4aAac,
|
|
146
149
|
} = {},
|
|
147
150
|
) {
|
|
151
|
+
registerSettingsRoute(ctx, settingsStore);
|
|
148
152
|
ctx.systemPrompt.variable('live_voice_context', (assemblyContext) =>
|
|
149
153
|
voiceContextStore.get(String(assemblyContext.agent?.sessionId || '')),
|
|
150
154
|
);
|
|
@@ -1,4 +1,44 @@
|
|
|
1
1
|
export const API_ROOT = '/api/dsh-live-voice';
|
|
2
|
+
|
|
3
|
+
export function registerSettingsRoute(
|
|
4
|
+
ctx: any,
|
|
5
|
+
store: { load(): Promise<unknown>; save(patch: unknown): Promise<unknown> },
|
|
6
|
+
) {
|
|
7
|
+
const dispose = ctx.connection.fetch.register({
|
|
8
|
+
path: API_ROOT + '/settings',
|
|
9
|
+
methods: ['GET', 'PUT'],
|
|
10
|
+
requestBody: 'buffered',
|
|
11
|
+
fetch: async (request: Request) => {
|
|
12
|
+
const headers = { 'cache-control': 'no-store' };
|
|
13
|
+
let patch: unknown;
|
|
14
|
+
if (request.method === 'PUT') {
|
|
15
|
+
try {
|
|
16
|
+
const text = await request.text();
|
|
17
|
+
if (text.length > 64000) throw new Error('too-large');
|
|
18
|
+
patch = JSON.parse(text);
|
|
19
|
+
if (!patch || typeof patch !== 'object' || Array.isArray(patch))
|
|
20
|
+
throw new Error('invalid');
|
|
21
|
+
} catch {
|
|
22
|
+
return Response.json(
|
|
23
|
+
{ ok: false, error: { code: 'invalid-settings' } },
|
|
24
|
+
{ status: 400, headers },
|
|
25
|
+
);
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
try {
|
|
29
|
+
const value = request.method === 'GET' ? await store.load() : await store.save(patch);
|
|
30
|
+
return Response.json({ ok: true, value }, { headers });
|
|
31
|
+
} catch {
|
|
32
|
+
return Response.json(
|
|
33
|
+
{ ok: false, error: { code: 'settings-unavailable' } },
|
|
34
|
+
{ status: 500, headers },
|
|
35
|
+
);
|
|
36
|
+
}
|
|
37
|
+
},
|
|
38
|
+
});
|
|
39
|
+
ctx.effect(() => () => dispose(), 'dsh-live-voice: remove settings route');
|
|
40
|
+
}
|
|
41
|
+
|
|
2
42
|
export const APP_VOICE_CONTEXT_PATH = API_ROOT + '/voice-context';
|
|
3
43
|
export type RouteHost = {
|
|
4
44
|
handle(endpoint: string, payload: unknown, signal?: AbortSignal): unknown;
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Module guidance
|
|
2
|
+
|
|
3
|
+
Read the nearest module's `ARCHITECTURE.md` before changing its behavior. `src/modules` owns product policy, feature UI, models, providers, and services. Modules may use `src/shared` and the centralized `src/app/client/i18n` API, but must not register DSH slots or host routes; composition belongs to `src/app`.
|
|
4
|
+
|
|
5
|
+
Preserve local-first STT and TTS options without claiming DSH itself is offline. Track capture/recognition, synthesis/playback, and agent generation separately. Preserve pause, resume, cancel, ownership, interruption, and stale asynchronous result semantics. Avoid logging raw audio or transcripts by default. Consult [VOICE-LIFECYCLE.md](../../docs/VOICE-LIFECYCLE.md) for the behavioral contract and [root guidance](../../AGENTS.md) for testing and authorization.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
# Conversation module
|
|
2
|
+
|
|
3
|
+
Dictation is append-only at the current composer end. Final chunks preserve existing and manually edited text; provisional hypotheses never rewrite the editor. Client synchronization must accept a new manual publication even when the previous voice echo was skipped, while ignoring unchanged stale slot snapshots. Explicit Clear and normal submit actions are separate from dictation.
|
|
4
|
+
|
|
5
|
+
`models/chat.ts` interprets DSH chat turns and pending questions. `hooks/useConversationController.ts` subscribes to session state; `hooks/useConversationActions.ts` maps user actions onto the controller. `components/` renders composer controls, microphone state, playback controls, status, and waveform; `createConversationComponents.tsx` composes them for the application slot boundary.
|
|
6
|
+
|
|
7
|
+
Keep UI components session-facing rather than importing host engines or registering slots. Preserve the DSH composer contract: `useInput` provides subscribed input and `inputActions.setDraft()` updates the composer; a recognition result must not be lost to stale interim updates. History pagination must not count as a new user turn. Do not treat sending, recognition, playback, or agent generation as one state. See [voice lifecycle](../../../docs/VOICE-LIFECYCLE.md) before changing interruption or turn-taking.
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Core voice domain
|
|
2
|
+
|
|
3
|
+
`core` owns policy and reusable voice-domain state, not DSH slot/route registration or provider-specific presentation. `coordinator.ts` serializes input transitions, owns conversation state and speech queues, and invalidates stale async work with generations. `ownership.ts` arbitrates microphone and speaker use between sessions and Settings self-tests. `transcript.ts` enforces append-only final recognition text (interim hypotheses never own or rewrite composer content), `microphone.ts` manages capture and activity, `filters.ts` handles voice commands and speech text filtering, and `settings.ts` normalizes persisted values and defaults.
|
|
4
|
+
|
|
5
|
+
The `qwen/` subtree contains shared Qwen host configuration and its settings UI; recognition and speaking adapters live in their respective modules. Keep the client/host dependency graphs separate. Preserve the persisted `dsh-live-voice.settings` key and normalization compatibility. Distinguish capture, recognition, playback, and generation; do not conflate pausing with cancellation or resume obsolete speech after a new turn. Consult [voice lifecycle](../../../docs/VOICE-LIFECYCLE.md) for expected transitions.
|
|
@@ -100,6 +100,7 @@ export class VoiceCoordinator {
|
|
|
100
100
|
processLocally: this.snapshot.settings.recognitionProcessLocally,
|
|
101
101
|
autoInstallLocalPack: this.snapshot.settings.recognitionAutoInstall,
|
|
102
102
|
voiceDetectionPreset: this.snapshot.settings.voiceDetectionPreset,
|
|
103
|
+
voiceDetectionCustomSilenceMs: this.snapshot.settings.voiceDetectionCustomSilenceMs,
|
|
103
104
|
});
|
|
104
105
|
this.patch({ capabilities: { ...this.snapshot.capabilities, recognition: undefined } });
|
|
105
106
|
}
|
|
@@ -125,6 +126,7 @@ export class VoiceCoordinator {
|
|
|
125
126
|
processLocally: settings.recognitionProcessLocally,
|
|
126
127
|
autoInstallLocalPack: settings.recognitionAutoInstall,
|
|
127
128
|
voiceDetectionPreset: settings.voiceDetectionPreset,
|
|
129
|
+
voiceDetectionCustomSilenceMs: settings.voiceDetectionCustomSilenceMs,
|
|
128
130
|
});
|
|
129
131
|
this.patch({
|
|
130
132
|
settings,
|
|
@@ -223,7 +225,7 @@ export class VoiceCoordinator {
|
|
|
223
225
|
}
|
|
224
226
|
muteListening() {
|
|
225
227
|
this.cancelAutoSend();
|
|
226
|
-
|
|
228
|
+
// Discard pending recognition without rewriting the composer.
|
|
227
229
|
this.transcript.reset();
|
|
228
230
|
this.updateSettings({ microphoneEnabled: false });
|
|
229
231
|
if (!this.snapshot.settings.voiceCommandsEnabled) return this.stopListening();
|
|
@@ -444,7 +446,7 @@ export class VoiceCoordinator {
|
|
|
444
446
|
clear: this.snapshot.settings.voiceCommandClear,
|
|
445
447
|
});
|
|
446
448
|
if (command) {
|
|
447
|
-
|
|
449
|
+
// Discard pending recognition without rewriting the composer.
|
|
448
450
|
this.transcript.reset();
|
|
449
451
|
this.cancelAutoSend();
|
|
450
452
|
this.patch({ recognizing: false });
|
|
@@ -468,7 +470,7 @@ export class VoiceCoordinator {
|
|
|
468
470
|
return;
|
|
469
471
|
}
|
|
470
472
|
if (this.snapshot.muted) {
|
|
471
|
-
|
|
473
|
+
// Discard pending recognition without rewriting the composer.
|
|
472
474
|
this.transcript.reset();
|
|
473
475
|
this.patch({ recognizing: false });
|
|
474
476
|
return;
|
|
@@ -477,7 +479,7 @@ export class VoiceCoordinator {
|
|
|
477
479
|
this.snapshot.settings.recognitionFilterEnabled &&
|
|
478
480
|
!hasMinimumWords(final, this.snapshot.settings.recognitionMinimumWords)
|
|
479
481
|
) {
|
|
480
|
-
|
|
482
|
+
// Discard pending recognition without rewriting the composer.
|
|
481
483
|
this.transcript.reset();
|
|
482
484
|
this.patch({ recognizing: false });
|
|
483
485
|
return;
|
|
@@ -488,7 +490,7 @@ export class VoiceCoordinator {
|
|
|
488
490
|
}
|
|
489
491
|
if (interim) {
|
|
490
492
|
this.cancelAutoSend();
|
|
491
|
-
|
|
493
|
+
// Interim hypotheses affect recognition status only, never composer content.
|
|
492
494
|
}
|
|
493
495
|
if (!interim && this.snapshot.recognizing)
|
|
494
496
|
this.assistantSpeechNotBefore =
|
|
@@ -571,7 +573,7 @@ export class VoiceCoordinator {
|
|
|
571
573
|
});
|
|
572
574
|
}
|
|
573
575
|
async cancelDictation() {
|
|
574
|
-
|
|
576
|
+
// Discard pending recognition without rewriting the composer.
|
|
575
577
|
await this.stopListening();
|
|
576
578
|
}
|
|
577
579
|
async endConversation() {
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
2
|
import React from 'react';
|
|
3
|
+
import { TextField, NumberField } from '../../../shared/design-system/index.js';
|
|
3
4
|
import { useLanguage } from '../../../app/client/i18n/index.js';
|
|
4
5
|
|
|
5
6
|
const BASE = '/api/dsh-live-voice/qwen';
|
|
@@ -56,7 +57,7 @@ export function QwenSettings({ controller }) {
|
|
|
56
57
|
'dsh-live-voice.speak.qwen.healthFailed': speak.qwen.healthFailed(),
|
|
57
58
|
})[value] ?? value;
|
|
58
59
|
|
|
59
|
-
async function run(action) {
|
|
60
|
+
async function run(action, config = draft) {
|
|
60
61
|
active.current?.abort();
|
|
61
62
|
const abort = new AbortController();
|
|
62
63
|
active.current = abort;
|
|
@@ -74,7 +75,7 @@ export function QwenSettings({ controller }) {
|
|
|
74
75
|
await controller.endConversation?.();
|
|
75
76
|
const value = await qwenSettingsRequest('/config', {
|
|
76
77
|
method: 'PUT',
|
|
77
|
-
config: { ...
|
|
78
|
+
config: { ...config, timeoutMs: Number(config.timeoutMs) },
|
|
78
79
|
signal: abort.signal,
|
|
79
80
|
});
|
|
80
81
|
if (!abort.signal.aborted) {
|
|
@@ -85,7 +86,7 @@ export function QwenSettings({ controller }) {
|
|
|
85
86
|
} else {
|
|
86
87
|
const value = await qwenSettingsRequest('/test', {
|
|
87
88
|
method: 'POST',
|
|
88
|
-
config: { ...
|
|
89
|
+
config: { ...config, timeoutMs: Number(config.timeoutMs) },
|
|
89
90
|
signal: abort.signal,
|
|
90
91
|
});
|
|
91
92
|
if (!abort.signal.aborted) {
|
|
@@ -104,23 +105,23 @@ export function QwenSettings({ controller }) {
|
|
|
104
105
|
void run('load');
|
|
105
106
|
return () => active.current?.abort();
|
|
106
107
|
}, []);
|
|
107
|
-
const field = (label, key, type = 'text') =>
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
<
|
|
111
|
-
|
|
108
|
+
const field = (label, key, type = 'text') => {
|
|
109
|
+
const Field = type === 'number' ? NumberField : TextField;
|
|
110
|
+
return (
|
|
111
|
+
<Field
|
|
112
|
+
label={label}
|
|
112
113
|
value={draft[key]}
|
|
113
114
|
disabled={busy || !loaded}
|
|
114
115
|
autoComplete="off"
|
|
115
116
|
{...(type === 'number' ? { min: 1000, max: 600000, step: 1 } : {})}
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
117
|
+
onCommit={(value) => {
|
|
118
|
+
const config = { ...draft, [key]: value };
|
|
119
|
+
setDraft(config);
|
|
120
|
+
void run('save', config);
|
|
120
121
|
}}
|
|
121
122
|
/>
|
|
122
|
-
|
|
123
|
-
|
|
123
|
+
);
|
|
124
|
+
};
|
|
124
125
|
return (
|
|
125
126
|
<>
|
|
126
127
|
<p>{recognition.qwen.hostHelp()}</p>
|
|
@@ -1,17 +1,17 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
2
|
export const voiceDetectionPresets = Object.freeze({
|
|
3
3
|
short: Object.freeze({
|
|
4
|
-
silenceMs:
|
|
4
|
+
silenceMs: 500,
|
|
5
5
|
label: 'Short',
|
|
6
6
|
description: 'Send quickly after a short pause.',
|
|
7
7
|
}),
|
|
8
8
|
natural: Object.freeze({
|
|
9
|
-
silenceMs:
|
|
9
|
+
silenceMs: 1000,
|
|
10
10
|
label: 'Natural',
|
|
11
11
|
description: 'Allow normal pauses between phrases.',
|
|
12
12
|
}),
|
|
13
13
|
long: Object.freeze({
|
|
14
|
-
silenceMs:
|
|
14
|
+
silenceMs: 2000,
|
|
15
15
|
label: 'Long',
|
|
16
16
|
description: 'Wait through longer thinking pauses.',
|
|
17
17
|
}),
|
|
@@ -39,7 +39,8 @@ export const defaultSettings = Object.freeze({
|
|
|
39
39
|
recognitionEngine: 'browser',
|
|
40
40
|
recognitionProcessLocally: true,
|
|
41
41
|
recognitionAutoInstall: true,
|
|
42
|
-
voiceDetectionPreset: '
|
|
42
|
+
voiceDetectionPreset: 'short',
|
|
43
|
+
voiceDetectionCustomSilenceMs: 1000,
|
|
43
44
|
recognitionMaxUtteranceSeconds: 60,
|
|
44
45
|
microphoneEnabled: true,
|
|
45
46
|
holdToTalkEnabled: true,
|
|
@@ -82,7 +83,21 @@ const normalizeCommandPhrases = (value, fallback) =>
|
|
|
82
83
|
.join(', ')
|
|
83
84
|
: fallback;
|
|
84
85
|
|
|
85
|
-
|
|
86
|
+
export const customSilenceMinMs = 100;
|
|
87
|
+
export const customSilenceMaxMs = 10000;
|
|
88
|
+
export const normalizeCustomSilenceMs = (value) =>
|
|
89
|
+
Number.isInteger(value) && value >= customSilenceMinMs && value <= customSilenceMaxMs
|
|
90
|
+
? value
|
|
91
|
+
: defaultSettings.voiceDetectionCustomSilenceMs;
|
|
92
|
+
export const voiceDetectionSilenceMs = (settings) =>
|
|
93
|
+
settings.voiceDetectionPreset === 'custom'
|
|
94
|
+
? normalizeCustomSilenceMs(settings.voiceDetectionCustomSilenceMs)
|
|
95
|
+
: (
|
|
96
|
+
voiceDetectionPresets[settings.voiceDetectionPreset] ||
|
|
97
|
+
voiceDetectionPresets[defaultSettings.voiceDetectionPreset]
|
|
98
|
+
).silenceMs;
|
|
99
|
+
|
|
100
|
+
/** Persisted preferences are untrusted and may belong to an older version. */
|
|
86
101
|
export function normalizeSettings(value) {
|
|
87
102
|
const source = value && typeof value === 'object' && !Array.isArray(value) ? value : {};
|
|
88
103
|
return {
|
|
@@ -100,9 +115,12 @@ export function normalizeSettings(value) {
|
|
|
100
115
|
typeof source.recognitionAutoInstall === 'boolean'
|
|
101
116
|
? source.recognitionAutoInstall
|
|
102
117
|
: defaultSettings.recognitionAutoInstall,
|
|
103
|
-
voiceDetectionPreset:
|
|
104
|
-
|
|
105
|
-
|
|
118
|
+
voiceDetectionPreset:
|
|
119
|
+
source.voiceDetectionPreset === 'custom' ||
|
|
120
|
+
Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset)
|
|
121
|
+
? source.voiceDetectionPreset
|
|
122
|
+
: defaultSettings.voiceDetectionPreset,
|
|
123
|
+
voiceDetectionCustomSilenceMs: normalizeCustomSilenceMs(source.voiceDetectionCustomSilenceMs),
|
|
106
124
|
recognitionMaxUtteranceSeconds:
|
|
107
125
|
Number.isInteger(source.recognitionMaxUtteranceSeconds) &&
|
|
108
126
|
source.recognitionMaxUtteranceSeconds >= 10 &&
|
|
@@ -1,48 +1,10 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
|
-
/**
|
|
2
|
+
/** Dictation is append-only: provisional hypotheses never own composer text. */
|
|
3
3
|
export class TranscriptDraft {
|
|
4
|
-
|
|
5
|
-
this.reset();
|
|
6
|
-
}
|
|
7
|
-
reset() {
|
|
8
|
-
this.owned = null;
|
|
9
|
-
this.edited = false;
|
|
10
|
-
}
|
|
4
|
+
reset() {}
|
|
11
5
|
update(current, hypothesis, final = false) {
|
|
12
|
-
if (
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
}
|
|
16
|
-
let base = current;
|
|
17
|
-
let at = current.length;
|
|
18
|
-
if (this.owned) {
|
|
19
|
-
const { start, text, before, after } = this.owned;
|
|
20
|
-
// Only replace a hypothesis whose surrounding text has not changed.
|
|
21
|
-
if (current === before + text + after) {
|
|
22
|
-
base = before + after;
|
|
23
|
-
at = start;
|
|
24
|
-
} else if (
|
|
25
|
-
text &&
|
|
26
|
-
current.indexOf(text) >= 0 &&
|
|
27
|
-
current.indexOf(text) === current.lastIndexOf(text)
|
|
28
|
-
) {
|
|
29
|
-
// Edits outside our unchanged, uniquely identifiable hypothesis are safe.
|
|
30
|
-
at = current.indexOf(text);
|
|
31
|
-
base = current.slice(0, at) + current.slice(at + text.length);
|
|
32
|
-
} else {
|
|
33
|
-
// The user edited the draft: relinquish it. Do not duplicate a final
|
|
34
|
-
// hypothesis after an edit; the edited text belongs to the user now.
|
|
35
|
-
this.owned = null;
|
|
36
|
-
this.edited = !final;
|
|
37
|
-
return current;
|
|
38
|
-
}
|
|
39
|
-
}
|
|
40
|
-
const before = base.slice(0, at);
|
|
41
|
-
const after = base.slice(at);
|
|
42
|
-
const separator = hypothesis && before && !/\s$/.test(before) ? ' ' : '';
|
|
43
|
-
const text = separator + hypothesis;
|
|
44
|
-
const result = before + text + after;
|
|
45
|
-
this.owned = final ? null : { start: at, text, before, after };
|
|
46
|
-
return result;
|
|
6
|
+
if (!final || !hypothesis) return current;
|
|
7
|
+
const separator = current && !/\s$/.test(current) ? ' ' : '';
|
|
8
|
+
return current + separator + hypothesis;
|
|
47
9
|
}
|
|
48
10
|
}
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Recognition module
|
|
2
|
+
|
|
3
|
+
`engines/browser/BrowserRecognitionEngine.ts` adapts browser speech recognition and reports local-pack/browser capabilities. `engines/qwen/QwenRecognitionEngine.ts` and `engines/whisper/WhisperRecognitionEngine.ts` adapt local-first HTTP recognition; host bridges live beside the relevant adapters (Whisper here, shared Qwen configuration in `../core/qwen`). `components/RecognitionCapabilityStatus.tsx` and engine settings UI present capability and configuration states.
|
|
4
|
+
|
|
5
|
+
Distinguish microphone capture and permission from recognition engine availability. Do not infer feature support merely from operating system. Bound audio and host requests, keep host-only code out of the browser bundle, and abort/ignore stale results after cancellation. Recognition language and voice-command phrases are independent of the interface locale. Browser simulations are not a physical microphone or model acceptance test.
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
|
-
import {
|
|
2
|
+
import { defaultSettings, voiceDetectionSilenceMs } from '../../../core/settings.js';
|
|
3
3
|
|
|
4
4
|
const ROUTE = '/api/dsh-live-voice/whisper';
|
|
5
5
|
const id = () => globalThis.crypto.randomUUID();
|
|
@@ -42,7 +42,8 @@ export class WhisperHttpRecognitionEngine {
|
|
|
42
42
|
constructor({
|
|
43
43
|
globals = globalThis,
|
|
44
44
|
meter,
|
|
45
|
-
voiceDetectionPreset =
|
|
45
|
+
voiceDetectionPreset = defaultSettings.voiceDetectionPreset,
|
|
46
|
+
voiceDetectionCustomSilenceMs = defaultSettings.voiceDetectionCustomSilenceMs,
|
|
46
47
|
maxUtteranceSeconds = 60,
|
|
47
48
|
} = {}) {
|
|
48
49
|
this.g = globals;
|
|
@@ -50,11 +51,12 @@ export class WhisperHttpRecognitionEngine {
|
|
|
50
51
|
this.session = null;
|
|
51
52
|
this.lang = 'pt-BR';
|
|
52
53
|
this.voiceDetectionPreset = voiceDetectionPreset;
|
|
54
|
+
this.voiceDetectionCustomSilenceMs = voiceDetectionCustomSilenceMs;
|
|
53
55
|
this.maxUtteranceSeconds = maxUtteranceSeconds;
|
|
54
56
|
this.route = ROUTE;
|
|
55
57
|
}
|
|
56
58
|
get segmentation() {
|
|
57
|
-
return
|
|
59
|
+
return { silenceMs: voiceDetectionSilenceMs(this) };
|
|
58
60
|
}
|
|
59
61
|
async capability() {
|
|
60
62
|
try {
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
// @ts-nocheck
|
|
2
2
|
import React from 'react';
|
|
3
|
+
import { TextField, NumberField } from '../../../../shared/design-system/index.js';
|
|
3
4
|
import { useLanguage } from '../../../../app/client/i18n/index.js';
|
|
4
5
|
|
|
5
6
|
const BASE = '/api/dsh-live-voice/whisper';
|
|
@@ -55,7 +56,7 @@ export function WhisperSettings({ controller }) {
|
|
|
55
56
|
'dsh-live-voice.recognition.whisper.requestFailed': recognition.whisper.requestFailed(),
|
|
56
57
|
'dsh-live-voice.recognition.whisper.healthFailed': recognition.whisper.healthFailed(),
|
|
57
58
|
})[value] ?? value;
|
|
58
|
-
async function run(action) {
|
|
59
|
+
async function run(action, config = draft) {
|
|
59
60
|
active.current?.abort();
|
|
60
61
|
const abort = new AbortController();
|
|
61
62
|
active.current = abort;
|
|
@@ -73,7 +74,7 @@ export function WhisperSettings({ controller }) {
|
|
|
73
74
|
await controller.endConversation?.();
|
|
74
75
|
const value = await whisperSettingsRequest('/config', {
|
|
75
76
|
method: 'PUT',
|
|
76
|
-
config: { ...
|
|
77
|
+
config: { ...config, timeoutMs: Number(config.timeoutMs) },
|
|
77
78
|
signal: abort.signal,
|
|
78
79
|
});
|
|
79
80
|
if (!abort.signal.aborted) {
|
|
@@ -84,7 +85,7 @@ export function WhisperSettings({ controller }) {
|
|
|
84
85
|
} else {
|
|
85
86
|
const value = await whisperSettingsRequest('/test', {
|
|
86
87
|
method: 'POST',
|
|
87
|
-
config: { ...
|
|
88
|
+
config: { ...config, timeoutMs: Number(config.timeoutMs) },
|
|
88
89
|
signal: abort.signal,
|
|
89
90
|
});
|
|
90
91
|
if (!abort.signal.aborted) {
|
|
@@ -103,25 +104,23 @@ export function WhisperSettings({ controller }) {
|
|
|
103
104
|
void run('load');
|
|
104
105
|
return () => active.current?.abort();
|
|
105
106
|
}, []);
|
|
106
|
-
|
|
107
|
+
const field = (label, key, type = 'text') => {
|
|
108
|
+
const Field = type === 'number' ? NumberField : TextField;
|
|
107
109
|
return (
|
|
108
|
-
<
|
|
109
|
-
{label}
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
}}
|
|
121
|
-
/>
|
|
122
|
-
</label>
|
|
110
|
+
<Field
|
|
111
|
+
label={label}
|
|
112
|
+
value={draft[key]}
|
|
113
|
+
disabled={busy || !loaded}
|
|
114
|
+
autoComplete="off"
|
|
115
|
+
{...(type === 'number' ? { min: 100, max: 300000, step: 1 } : {})}
|
|
116
|
+
onCommit={(value) => {
|
|
117
|
+
const config = { ...draft, [key]: value };
|
|
118
|
+
setDraft(config);
|
|
119
|
+
void run('save', config);
|
|
120
|
+
}}
|
|
121
|
+
/>
|
|
123
122
|
);
|
|
124
|
-
}
|
|
123
|
+
};
|
|
125
124
|
return (
|
|
126
125
|
<>
|
|
127
126
|
<p>{settings.whisper.hostHelp()}</p>
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Settings module
|
|
2
|
+
|
|
3
|
+
`components/LiveVoiceSettings.tsx` owns the tabbed Settings shell; `sections/{conversation,recognition,speak}` own feature controls. `hooks/useLiveVoiceSettings.ts` subscribes to controller state and dispatches updates, while `hooks/useAudioDevices.ts` and `hooks/useReleaseStatus.ts` handle capability-adjacent data. `models/settingsStorage.ts` owns the authenticated client transport and in-memory preferences; `models/settingsHost.ts` owns serialized, atomic host-side persistence; `services/releases.ts` supports version metadata. Shared fields, layout, icons, and feedback live in `src/shared/design-system`.
|
|
4
|
+
|
|
5
|
+
Preserve existing setting names, default values, handlers, and normalization in `../core/settings.ts`; reorganizing presentation must not silently change policy. Keep transient capability/permission messages visible even when an options section is collapsed. User-facing copy and accessible names belong in all typed catalogs under `src/app/client/i18n`; UI locale does not select recognition or speech language. The preview in `__previewjs__/LiveVoicePreviews.tsx` exercises presentation with a fake controller, not a physical microphone or authenticated DSH runtime. Run `npm test` after changes and check the preview for layout and interaction.
|
|
@@ -17,9 +17,17 @@ export function LiveVoiceSettings({ controller, onClose }: { controller: any; on
|
|
|
17
17
|
const [activeTab, setActiveTab] = React.useState('conversation');
|
|
18
18
|
const tabsId = React.useId();
|
|
19
19
|
const tabs = [
|
|
20
|
-
{ id: 'speech', label: (settingsLanguage as any).tabs.speak() },
|
|
21
|
-
{
|
|
22
|
-
|
|
20
|
+
{ id: 'speech', label: (settingsLanguage as any).tabs.speak(), icon: 'speaker' as const },
|
|
21
|
+
{
|
|
22
|
+
id: 'recognition',
|
|
23
|
+
label: (settingsLanguage as any).tabs.recognition(),
|
|
24
|
+
icon: 'mic' as const,
|
|
25
|
+
},
|
|
26
|
+
{
|
|
27
|
+
id: 'conversation',
|
|
28
|
+
label: (settingsLanguage as any).tabs.conversation(),
|
|
29
|
+
icon: 'send' as const,
|
|
30
|
+
},
|
|
23
31
|
];
|
|
24
32
|
const sectionProps = {
|
|
25
33
|
controller,
|