dsh-live-voice 0.3.0 → 0.3.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +89 -1
  2. package/PLAN.md +25 -1
  3. package/README.md +4 -2
  4. package/docs/ARCHITECTURE.md +9 -16
  5. package/docs/CONFIGURATION.md +17 -3
  6. package/docs/VOICE-LIFECYCLE.md +5 -1
  7. package/lib/client.js +543 -326
  8. package/lib/server.js +158 -5
  9. package/package.json +2 -2
  10. package/src/app/AGENTS.md +7 -0
  11. package/src/app/ARCHITECTURE.md +10 -0
  12. package/src/app/client/apply.tsx +96 -40
  13. package/src/app/client/i18n/catalogs/base.ts +7 -0
  14. package/src/app/client/i18n/catalogs/en.ts +10 -0
  15. package/src/app/client/i18n/catalogs/es.ts +10 -0
  16. package/src/app/client/i18n/catalogs/fr.ts +10 -0
  17. package/src/app/client/i18n/catalogs/hi.ts +10 -0
  18. package/src/app/client/i18n/catalogs/pt-BR.ts +10 -0
  19. package/src/app/client/i18n/catalogs/zh.ts +9 -0
  20. package/src/app/server/apply.ts +4 -0
  21. package/src/app/server/registerRoutes.ts +40 -0
  22. package/src/modules/AGENTS.md +5 -0
  23. package/src/modules/conversation/ARCHITECTURE.md +7 -0
  24. package/src/modules/core/ARCHITECTURE.md +5 -0
  25. package/src/modules/core/coordinator.ts +8 -6
  26. package/src/modules/core/qwen/QwenHttpHost.ts +1 -1
  27. package/src/modules/core/qwen/QwenSettings.tsx +15 -14
  28. package/src/modules/core/settings.ts +26 -8
  29. package/src/modules/core/transcript.ts +5 -43
  30. package/src/modules/recognition/ARCHITECTURE.md +5 -0
  31. package/src/modules/recognition/engines/whisper/WhisperRecognitionEngine.ts +5 -3
  32. package/src/modules/recognition/engines/whisper/WhisperSettings.tsx +19 -20
  33. package/src/modules/settings/ARCHITECTURE.md +5 -0
  34. package/src/modules/settings/components/LiveVoiceSettings.tsx +11 -3
  35. package/src/modules/settings/models/settingsHost.ts +45 -0
  36. package/src/modules/settings/models/settingsStorage.ts +45 -10
  37. package/src/modules/settings/sections/conversation/ConversationSettingsSection.tsx +12 -6
  38. package/src/modules/settings/sections/recognition/RecognitionEngineSettings.tsx +2 -2
  39. package/src/modules/settings/sections/recognition/RecognitionFilterSettings.tsx +3 -3
  40. package/src/modules/settings/sections/recognition/RecognitionSettingsSection.tsx +7 -3
  41. package/src/modules/settings/sections/recognition/SilenceDetectionSettings.tsx +38 -8
  42. package/src/modules/settings/sections/recognition/VoiceCommandSettings.tsx +3 -3
  43. package/src/modules/settings/sections/speak/OutputFilterSettings.tsx +5 -5
  44. package/src/modules/settings/sections/speak/SpeakSettingsSection.tsx +7 -3
  45. package/src/modules/settings/sections/speak/SpeechAdvancedSettings.tsx +4 -4
  46. package/src/modules/settings/sections/speak/SpeechEngineSettings.tsx +2 -2
  47. package/src/modules/settings/services/releases.ts +10 -2
  48. package/src/modules/speak/ARCHITECTURE.md +5 -0
  49. package/src/shared/ARCHITECTURE.md +5 -0
  50. package/src/shared/design-system/forms/DraftField.tsx +79 -0
  51. package/src/shared/design-system/forms/NumberField.tsx +4 -10
  52. package/src/shared/design-system/forms/TextAreaField.tsx +4 -10
  53. package/src/shared/design-system/forms/TextField.tsx +4 -10
  54. package/src/shared/design-system/icons/icons.ts +4 -0
  55. package/src/shared/design-system/layout/SettingsSubcard.tsx +15 -3
  56. package/src/shared/design-system/layout/SettingsTabs.tsx +4 -2
  57. package/src/styles/index.ts +3 -3
package/lib/server.js CHANGED
@@ -7,17 +7,17 @@ import { randomUUID as randomUUID2 } from "node:crypto";
7
7
  // src/modules/core/settings.ts
8
8
  var voiceDetectionPresets = Object.freeze({
9
9
  short: Object.freeze({
10
- silenceMs: 900,
10
+ silenceMs: 500,
11
11
  label: "Short",
12
12
  description: "Send quickly after a short pause."
13
13
  }),
14
14
  natural: Object.freeze({
15
- silenceMs: 1500,
15
+ silenceMs: 1e3,
16
16
  label: "Natural",
17
17
  description: "Allow normal pauses between phrases."
18
18
  }),
19
19
  long: Object.freeze({
20
- silenceMs: 2200,
20
+ silenceMs: 2e3,
21
21
  label: "Long",
22
22
  description: "Wait through longer thinking pauses."
23
23
  })
@@ -43,7 +43,8 @@ var defaultSettings = Object.freeze({
43
43
  recognitionEngine: "browser",
44
44
  recognitionProcessLocally: true,
45
45
  recognitionAutoInstall: true,
46
- voiceDetectionPreset: "natural",
46
+ voiceDetectionPreset: "short",
47
+ voiceDetectionCustomSilenceMs: 1e3,
47
48
  recognitionMaxUtteranceSeconds: 60,
48
49
  microphoneEnabled: true,
49
50
  holdToTalkEnabled: true,
@@ -76,6 +77,73 @@ var defaultSettings = Object.freeze({
76
77
  rate: 1,
77
78
  segmentGapMs: 400
78
79
  });
80
+ var normalizeCommandPhrases = (value, fallback) => typeof value === "string" && value.length <= 1e3 && !value.includes("\0") ? value.split(/[,\r\n]+/).map((phrase) => phrase.trim()).filter(Boolean).slice(0, 20).join(", ") : fallback;
81
+ var customSilenceMinMs = 100;
82
+ var customSilenceMaxMs = 1e4;
83
+ var normalizeCustomSilenceMs = (value) => Number.isInteger(value) && value >= customSilenceMinMs && value <= customSilenceMaxMs ? value : defaultSettings.voiceDetectionCustomSilenceMs;
84
+ function normalizeSettings(value) {
85
+ const source = value && typeof value === "object" && !Array.isArray(value) ? value : {};
86
+ return {
87
+ engine: ["browser", "say", "qwen-http"].includes(source.engine) ? source.engine : defaultSettings.engine,
88
+ recognitionEngine: ["browser", "whisper-http", "qwen-http"].includes(source.recognitionEngine) ? source.recognitionEngine : defaultSettings.recognitionEngine,
89
+ recognitionProcessLocally: typeof source.recognitionProcessLocally === "boolean" ? source.recognitionProcessLocally : defaultSettings.recognitionProcessLocally,
90
+ recognitionAutoInstall: typeof source.recognitionAutoInstall === "boolean" ? source.recognitionAutoInstall : defaultSettings.recognitionAutoInstall,
91
+ voiceDetectionPreset: source.voiceDetectionPreset === "custom" || Object.hasOwn(voiceDetectionPresets, source.voiceDetectionPreset) ? source.voiceDetectionPreset : defaultSettings.voiceDetectionPreset,
92
+ voiceDetectionCustomSilenceMs: normalizeCustomSilenceMs(source.voiceDetectionCustomSilenceMs),
93
+ recognitionMaxUtteranceSeconds: Number.isInteger(source.recognitionMaxUtteranceSeconds) && source.recognitionMaxUtteranceSeconds >= 10 && source.recognitionMaxUtteranceSeconds <= 300 ? source.recognitionMaxUtteranceSeconds : defaultSettings.recognitionMaxUtteranceSeconds,
94
+ microphoneEnabled: typeof source.microphoneEnabled === "boolean" ? source.microphoneEnabled : defaultSettings.microphoneEnabled,
95
+ holdToTalkEnabled: typeof source.holdToTalkEnabled === "boolean" ? source.holdToTalkEnabled : defaultSettings.holdToTalkEnabled,
96
+ announceAssistantMessages: typeof source.announceAssistantMessages === "boolean" ? source.announceAssistantMessages : defaultSettings.announceAssistantMessages,
97
+ agentVoiceContextEnabled: typeof source.agentVoiceContextEnabled === "boolean" ? source.agentVoiceContextEnabled : defaultSettings.agentVoiceContextEnabled,
98
+ agentVoiceContext: typeof source.agentVoiceContext === "string" && source.agentVoiceContext.length <= 4e3 && !source.agentVoiceContext.includes("\0") ? source.agentVoiceContext.trim() : defaultSettings.agentVoiceContext,
99
+ interruptSpeechOnUserMessage: typeof source.interruptSpeechOnUserMessage === "boolean" ? source.interruptSpeechOnUserMessage : defaultSettings.interruptSpeechOnUserMessage,
100
+ sendingMode: source.sendingMode === "automatic" ? "queue" : ["manual", "queue", "steer"].includes(source.sendingMode) ? source.sendingMode : defaultSettings.sendingMode,
101
+ autoSendDelaySeconds: Number.isInteger(source.autoSendDelaySeconds) && source.autoSendDelaySeconds >= 2 && source.autoSendDelaySeconds <= 10 ? source.autoSendDelaySeconds : defaultSettings.autoSendDelaySeconds,
102
+ assistantSpeechDelaySeconds: Number.isInteger(source.assistantSpeechDelaySeconds) && source.assistantSpeechDelaySeconds >= 1 && source.assistantSpeechDelaySeconds <= 10 ? source.assistantSpeechDelaySeconds : defaultSettings.assistantSpeechDelaySeconds,
103
+ mode: ["speaker", "headphones"].includes(source.mode) ? source.mode : defaultSettings.mode,
104
+ lang: typeof source.lang === "string" && /^[a-z]{2,3}(?:-[A-Za-z0-9]{2,8})*$/.test(source.lang) ? source.lang : defaultSettings.lang,
105
+ recognitionLang: source.recognitionLang === "auto" || typeof source.recognitionLang === "string" && /^[a-z]{2,3}(?:-[A-Za-z0-9]{2,8})*$/.test(source.recognitionLang) ? source.recognitionLang : typeof source.lang === "string" && /^[a-z]{2,3}(?:-[A-Za-z0-9]{2,8})*$/.test(source.lang) ? source.lang : defaultSettings.recognitionLang,
106
+ voice: typeof source.voice === "string" && source.voice.length <= 200 && !source.voice.includes("\0") ? source.voice : "",
107
+ inputDeviceId: typeof source.inputDeviceId === "string" && source.inputDeviceId.length <= 500 && !source.inputDeviceId.includes("\0") ? source.inputDeviceId : "",
108
+ outputDeviceId: typeof source.outputDeviceId === "string" && source.outputDeviceId.length <= 500 && !source.outputDeviceId.includes("\0") ? source.outputDeviceId : "",
109
+ recognitionFilterEnabled: typeof source.recognitionFilterEnabled === "boolean" ? source.recognitionFilterEnabled : defaultSettings.recognitionFilterEnabled,
110
+ recognitionMinimumWords: Number.isInteger(source.recognitionMinimumWords) && source.recognitionMinimumWords >= 1 && source.recognitionMinimumWords <= 20 ? source.recognitionMinimumWords : defaultSettings.recognitionMinimumWords,
111
+ voiceCommandsEnabled: typeof source.voiceCommandsEnabled === "boolean" ? source.voiceCommandsEnabled : defaultSettings.voiceCommandsEnabled,
112
+ voiceCommandSend: normalizeCommandPhrases(
113
+ source.voiceCommandSend,
114
+ defaultSettings.voiceCommandSend
115
+ ),
116
+ voiceCommandQueue: normalizeCommandPhrases(
117
+ source.voiceCommandQueue,
118
+ defaultSettings.voiceCommandQueue
119
+ ),
120
+ voiceCommandEnd: normalizeCommandPhrases(
121
+ source.voiceCommandEnd,
122
+ defaultSettings.voiceCommandEnd
123
+ ),
124
+ voiceCommandMute: normalizeCommandPhrases(
125
+ source.voiceCommandMute,
126
+ defaultSettings.voiceCommandMute
127
+ ),
128
+ voiceCommandResume: normalizeCommandPhrases(
129
+ source.voiceCommandResume,
130
+ defaultSettings.voiceCommandResume
131
+ ),
132
+ voiceCommandStopSpeaking: normalizeCommandPhrases(
133
+ source.voiceCommandStopSpeaking,
134
+ defaultSettings.voiceCommandStopSpeaking
135
+ ),
136
+ voiceCommandClear: normalizeCommandPhrases(
137
+ source.voiceCommandClear,
138
+ defaultSettings.voiceCommandClear
139
+ ),
140
+ outputCodeFilterEnabled: typeof source.outputCodeFilterEnabled === "boolean" ? source.outputCodeFilterEnabled : defaultSettings.outputCodeFilterEnabled,
141
+ outputCodeMaxLines: Number.isInteger(source.outputCodeMaxLines) && source.outputCodeMaxLines >= 0 && source.outputCodeMaxLines <= 100 ? source.outputCodeMaxLines : defaultSettings.outputCodeMaxLines,
142
+ outputCodeNotice: typeof source.outputCodeNotice === "string" && source.outputCodeNotice.trim() && source.outputCodeNotice.length <= 300 && !source.outputCodeNotice.includes("\0") ? source.outputCodeNotice.trim() : defaultSettings.outputCodeNotice,
143
+ rate: Number.isFinite(source.rate) && source.rate >= 0.1 && source.rate <= 3 ? source.rate : 1,
144
+ segmentGapMs: Number.isFinite(source.segmentGapMs) && source.segmentGapMs >= 0 && source.segmentGapMs <= 2e3 ? Math.round(source.segmentGapMs) : 400
145
+ };
146
+ }
79
147
 
80
148
  // src/modules/recognition/engines/whisper/whisperRecognitionHost.ts
81
149
  import { readFile, mkdir, writeFile, rename, rm } from "node:fs/promises";
@@ -304,7 +372,7 @@ var QwenHttpHost = class {
304
372
  baseUrl,
305
373
  timeoutMs = 3e5,
306
374
  fetchImpl = globalThis.fetch,
307
- maxBytes = 2e6,
375
+ maxBytes = 5e7,
308
376
  maxSpeechBytes = 5e7,
309
377
  store
310
378
  } = {}) {
@@ -784,6 +852,89 @@ var SayEngine = class {
784
852
  }
785
853
  };
786
854
 
855
+ // src/modules/settings/models/settingsHost.ts
856
+ import { readFile as readFile3, mkdir as mkdir3, writeFile as writeFile3, rename as rename3, rm as rm3 } from "node:fs/promises";
857
+ import { homedir as homedir3 } from "node:os";
858
+ import { dirname as dirname3, join as join5 } from "node:path";
859
+ import { randomUUID as randomUUID3 } from "node:crypto";
860
+ function createSettingsStore(path = join5(homedir3(), ".dsh", "dsh-live-voice.settings.json")) {
861
+ let queue = Promise.resolve();
862
+ async function load() {
863
+ try {
864
+ return normalizeSettings(JSON.parse(await readFile3(path, "utf8")));
865
+ } catch (error) {
866
+ if (error.code === "ENOENT") return normalizeSettings({});
867
+ throw error;
868
+ }
869
+ }
870
+ return {
871
+ load: () => queue.then(load),
872
+ save(patch) {
873
+ const operation = queue.then(async () => {
874
+ const settings = normalizeSettings({ ...await load(), ...patch });
875
+ await mkdir3(dirname3(path), { recursive: true, mode: 448 });
876
+ const temporary = path + "." + randomUUID3() + ".tmp";
877
+ try {
878
+ await writeFile3(temporary, JSON.stringify(settings, null, 2) + "\n", {
879
+ mode: 384,
880
+ flag: "wx"
881
+ });
882
+ await rename3(temporary, path);
883
+ } finally {
884
+ await rm3(temporary, { force: true });
885
+ }
886
+ return settings;
887
+ });
888
+ queue = operation.then(
889
+ () => {
890
+ },
891
+ () => {
892
+ }
893
+ );
894
+ return operation;
895
+ }
896
+ };
897
+ }
898
+
899
+ // src/app/server/registerRoutes.ts
900
+ var API_ROOT = "/api/dsh-live-voice";
901
+ function registerSettingsRoute(ctx, store) {
902
+ const dispose = ctx.connection.fetch.register({
903
+ path: API_ROOT + "/settings",
904
+ methods: ["GET", "PUT"],
905
+ requestBody: "buffered",
906
+ fetch: async (request) => {
907
+ const headers = { "cache-control": "no-store" };
908
+ let patch;
909
+ if (request.method === "PUT") {
910
+ try {
911
+ const text = await request.text();
912
+ if (text.length > 64e3) throw new Error("too-large");
913
+ patch = JSON.parse(text);
914
+ if (!patch || typeof patch !== "object" || Array.isArray(patch))
915
+ throw new Error("invalid");
916
+ } catch {
917
+ return Response.json(
918
+ { ok: false, error: { code: "invalid-settings" } },
919
+ { status: 400, headers }
920
+ );
921
+ }
922
+ }
923
+ try {
924
+ const value = request.method === "GET" ? await store.load() : await store.save(patch);
925
+ return Response.json({ ok: true, value }, { headers });
926
+ } catch {
927
+ return Response.json(
928
+ { ok: false, error: { code: "settings-unavailable" } },
929
+ { status: 500, headers }
930
+ );
931
+ }
932
+ }
933
+ });
934
+ ctx.effect(() => () => dispose(), "dsh-live-voice: remove settings route");
935
+ }
936
+ var APP_VOICE_CONTEXT_PATH = API_ROOT + "/voice-context";
937
+
787
938
  // src/app/server/apply.ts
788
939
  var name = "dsh-live-voice";
789
940
  var inject = ["connection", "systemPrompt"];
@@ -880,6 +1031,7 @@ function createSayHost({ engine = new SayEngine() } = {}) {
880
1031
  return { handle, dispose, engine };
881
1032
  }
882
1033
  function apply(ctx, {
1034
+ settingsStore = createSettingsStore(),
883
1035
  whisperStore = createWhisperConfigStore(),
884
1036
  whisperFetch = globalThis.fetch,
885
1037
  qwenStore = createQwenConfigStore(),
@@ -888,6 +1040,7 @@ function apply(ctx, {
888
1040
  createSayEngine = () => new SayEngine(),
889
1041
  encodeHostSpeech = wavToM4aAac
890
1042
  } = {}) {
1043
+ registerSettingsRoute(ctx, settingsStore);
891
1044
  ctx.systemPrompt.variable(
892
1045
  "live_voice_context",
893
1046
  (assemblyContext) => voiceContextStore.get(String(assemblyContext.agent?.sessionId || ""))
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "name": "dsh-live-voice",
3
- "version": "0.3.0",
4
- "dshTestedVersion": "0.1.6-alpha.2",
3
+ "version": "0.3.2",
4
+ "dshTestedVersion": "0.2.0-rc.2",
5
5
  "type": "module",
6
6
  "scripts": {
7
7
  "typecheck": "tsc --noEmit",
@@ -0,0 +1,7 @@
1
+ # Application boundary
2
+
3
+ Read [ARCHITECTURE.md](ARCHITECTURE.md) before changing composition roots, routes, slots, or language registration. Read the repository [AGENTS.md](../../AGENTS.md) for product, validation, and authorization rules.
4
+
5
+ - Inspect the actual DSH extension API before changing an integration point. Preserve established slot names, IDs, ordering, injected services, authenticated same-origin routes, and browser/host separation.
6
+ - Keep UI copy in the typed catalogs under `client/i18n/`; update every supported language in alphabetical key order. Use DSH `ctx.locale`, not an independent preference. UI locale is independent of recognition/synthesis languages and configured voice-command phrases.
7
+ - Treat `package.json` as the source of truth for plugin and tested-DSH versions. Never hard-code these in the client.
@@ -0,0 +1,10 @@
1
+ # Application architecture
2
+
3
+ `src/app` is the composition boundary between DSH and the voice-domain modules. Client and server entry points remain separate; host-only Node dependencies must never enter the browser bundle.
4
+
5
+ - `client/apply.tsx` injects DSH client services, creates session controllers and voice ownership, wires engines and settings, and disposes resources. `client/slotDefinitions.ts` and `client/registerSlots.tsx` own the established slot IDs, order, and registration. Preserve those contracts when changing placement.
6
+ - `server/apply.ts` composes host engines and providers. `server/registerRoutes.ts` registers authenticated same-origin routes; do not bypass their validation or expose host access directly to browser code.
7
+ - `client/i18n` owns the only UI translation tree: typed, alphabetized catalogs for every supported locale, runtime selection synchronized with `ctx.locale`, DSH registration, and the boundary wrapping slot components. UI language must not change STT/TTS language or user-defined command phrases.
8
+ - Styles are assembled from `src/styles/index.ts`; design-system primitives live under `src/shared/design-system` rather than here.
9
+
10
+ Read [the repository overview](../../docs/ARCHITECTURE.md) for dependency direction and stable contracts, and the relevant module architecture before editing feature behavior. Read [application agent guidance](AGENTS.md) for operational constraints.
@@ -1,12 +1,7 @@
1
1
  // @ts-nocheck
2
2
  import React from 'react';
3
3
  import { createPortal } from 'react-dom';
4
- import {
5
- MicrophoneMeter,
6
- VoiceCoordinator,
7
- VoiceOwnership,
8
- normalizeSettings,
9
- } from '../../modules/core/index.js';
4
+ import { MicrophoneMeter, VoiceCoordinator, VoiceOwnership } from '../../modules/core/index.js';
10
5
  import {
11
6
  BrowserRecognitionEngine,
12
7
  QwenHttpRecognitionEngine,
@@ -26,6 +21,7 @@ import {
26
21
  import { registerLiveVoiceLocales } from './i18n/index.js';
27
22
  import { registerConversationSlots, registerSettingsSlot } from './registerSlots.js';
28
23
  import { styles } from '../../styles/index.js';
24
+ import { createSettingsClient } from '../../modules/settings/models/settingsStorage.js';
29
25
  import {
30
26
  assistantMessages,
31
27
  addressedTurn,
@@ -39,6 +35,8 @@ export function apply(ctx) {
39
35
  const SettingsPanel = createLiveVoiceSettings(t, ctx.locale);
40
36
  const controllers = new Map();
41
37
  const retiring = new Set();
38
+ const preferences = createSettingsClient();
39
+ const settingsControllers = new Set();
42
40
  let disposed = false;
43
41
  // Voice conversation mode belongs to the plugin, not to a mounted composer.
44
42
  // A route change may replace every conversation slot, but the next committed
@@ -73,6 +71,7 @@ export function apply(ctx) {
73
71
  'recognitionProcessLocally',
74
72
  'recognitionAutoInstall',
75
73
  'voiceDetectionPreset',
74
+ 'voiceDetectionCustomSilenceMs',
76
75
  'recognitionMaxUtteranceSeconds',
77
76
  ]);
78
77
  const changesRecognition = (next) =>
@@ -82,12 +81,14 @@ export function apply(ctx) {
82
81
  ? new QwenHttpRecognitionEngine({
83
82
  meter,
84
83
  voiceDetectionPreset: settings.voiceDetectionPreset,
84
+ voiceDetectionCustomSilenceMs: settings.voiceDetectionCustomSilenceMs,
85
85
  maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
86
86
  })
87
87
  : settings.recognitionEngine === 'whisper-http'
88
88
  ? new WhisperHttpRecognitionEngine({
89
89
  meter,
90
90
  voiceDetectionPreset: settings.voiceDetectionPreset,
91
+ voiceDetectionCustomSilenceMs: settings.voiceDetectionCustomSilenceMs,
91
92
  maxUtteranceSeconds: settings.recognitionMaxUtteranceSeconds,
92
93
  })
93
94
  : new BrowserRecognitionEngine({
@@ -99,6 +100,38 @@ export function apply(ctx) {
99
100
  if (!disposed && !controller.disposed)
100
101
  controller.patch({ error: error?.message || String(error) });
101
102
  });
103
+ async function savePreferences(controller, next) {
104
+ try {
105
+ return await preferences.save(next);
106
+ } catch {
107
+ const error = new Error(t('dsh-live-voice.settings.persistence.saveError'));
108
+ if (!disposed && !controller.disposed) controller.patch({ error: error.message });
109
+ throw error;
110
+ }
111
+ }
112
+ const unsubscribePreferences = preferences.subscribe((settings) => {
113
+ if (disposed) return;
114
+ for (const entry of controllers.values()) {
115
+ if (entry.closed) continue;
116
+ const previous = entry.controller.getSnapshot().settings;
117
+ if ([...recognitionSettingKeys].some((key) => previous[key] !== settings[key]))
118
+ entry.controller.replaceRecognition(recognitionFor(settings, entry.controller.meter));
119
+ entry.applySettings(settings);
120
+ }
121
+ for (const c of settingsControllers) {
122
+ if (c.disposed) continue;
123
+ const previous = c.getSnapshot().settings;
124
+ if ([...recognitionSettingKeys].some((key) => previous[key] !== settings[key]))
125
+ c.replaceRecognition(recognitionFor(settings, c.meter));
126
+ c.applySavedSettings(settings);
127
+ for (const engine of Object.values(c.engines)) engine.lang = settings.lang;
128
+ run(c, c.refreshCapabilities());
129
+ }
130
+ });
131
+ ctx.effect(
132
+ () => () => unsubscribePreferences(),
133
+ 'dsh-live-voice: remove preferences subscription',
134
+ );
102
135
  function retire(entry) {
103
136
  if (entry.closed) return;
104
137
  entry.closed = true;
@@ -128,6 +161,10 @@ export function apply(ctx) {
128
161
  key,
129
162
  draft: '',
130
163
  pendingDraft: undefined,
164
+ publishedDraft: undefined,
165
+ publishedRevision: undefined,
166
+ pendingRevision: undefined,
167
+ staleDrafts: new Set(),
131
168
  composers: new Map(),
132
169
  refs: 0,
133
170
  buttons: 0,
@@ -135,14 +172,7 @@ export function apply(ctx) {
135
172
  closed: false,
136
173
  lastVoiceContextBody: null,
137
174
  };
138
- let settings = {};
139
- try {
140
- settings = normalizeSettings(
141
- JSON.parse(localStorage.getItem('dsh-live-voice.settings') || '{}'),
142
- );
143
- } catch {
144
- settings = normalizeSettings(null);
145
- }
175
+ const settings = preferences.getSnapshot();
146
176
  const engineBrowser = new BrowserSpeakingEngine({ lang: settings.lang || 'pt-BR' });
147
177
  const engineSay = new HostAudioSpeakingEngine({
148
178
  endpoint: '/api/dsh-live-voice/say/speech',
@@ -202,12 +232,14 @@ export function apply(ctx) {
202
232
  if (disposed || entry.closed) return;
203
233
  const owner = [...entry.composers.values()].at(-1);
204
234
  if (!owner) return;
205
- // `setDraft()` updates Lexical synchronously but the subscribed InputState
206
- // can publish on a subsequent React commit. Preserve the optimistic value
207
- // until that exact publication arrives; otherwise another slot render can
208
- // make a later recognition result start from stale text.
235
+ // Preserve unacknowledged voice writes across unchanged slot renders,
236
+ // but let new editor publications (including manual edits) supersede them.
237
+ if (entry.pendingDraft === undefined) entry.staleDrafts.clear();
238
+ entry.staleDrafts.add(entry.draft);
239
+ if (entry.publishedDraft !== undefined) entry.staleDrafts.add(entry.publishedDraft);
209
240
  entry.draft = text;
210
241
  entry.pendingDraft = text;
242
+ entry.pendingRevision = entry.publishedRevision;
211
243
  owner.actions.setDraft(text);
212
244
  },
213
245
  },
@@ -286,13 +318,7 @@ export function apply(ctx) {
286
318
  controller.updateSettings = (next) => {
287
319
  if (disposed || entry.closed) return;
288
320
  entry.applySettings(next);
289
- const settings = controller.getSnapshot().settings;
290
- try {
291
- localStorage.setItem('dsh-live-voice.settings', JSON.stringify(settings));
292
- } catch {}
293
- for (const other of controllers.values()) {
294
- if (other !== entry && !other.closed) other.applySettings(settings);
295
- }
321
+ return savePreferences(controller, next);
296
322
  };
297
323
  // Cancel only this entry's queued acquisition. Global cancellation here would
298
324
  // invalidate a newer session while its predecessor is being unmounted.
@@ -322,7 +348,12 @@ export function apply(ctx) {
322
348
  return ownership.run(
323
349
  controller,
324
350
  [...controllers.values()].map((other) => other.controller).concat([...retiring]),
325
- () => {
351
+ async () => {
352
+ try {
353
+ await preferences.ready;
354
+ } catch {
355
+ throw new Error(t('dsh-live-voice.settings.persistence.loadError'));
356
+ }
326
357
  if (disposed || entry.closed || !entry.refs || request !== entry.request) return;
327
358
  if (method !== 'speak' && !entry.composers.size) return;
328
359
  const result = original(...args);
@@ -333,6 +364,10 @@ export function apply(ctx) {
333
364
  };
334
365
  }
335
366
  controllers.set(key, entry);
367
+ void preferences.ready.catch(() => {
368
+ if (!entry.closed)
369
+ controller.patch({ error: t('dsh-live-voice.settings.persistence.loadError') });
370
+ });
336
371
  run(controller, controller.refreshCapabilities());
337
372
  return entry;
338
373
  }
@@ -409,10 +444,22 @@ export function apply(ctx) {
409
444
  const published = typeof input.draft === 'string' ? input.draft : '';
410
445
  // A voice write is optimistic until React commits the editor publication.
411
446
  // Unrelated slot renders must not restore the previous draft in between.
447
+ entry.publishedDraft = published;
448
+ entry.publishedRevision = input.draftRev;
449
+ const newerRevision =
450
+ Number.isInteger(input.draftRev) &&
451
+ Number.isInteger(entry.pendingRevision) &&
452
+ input.draftRev > entry.pendingRevision;
412
453
  if (entry.pendingDraft === undefined) entry.draft = published;
413
- else if (published === entry.pendingDraft) {
454
+ else if (
455
+ published === entry.pendingDraft ||
456
+ newerRevision ||
457
+ !entry.staleDrafts.has(published)
458
+ ) {
459
+ // A new manual edit is authoritative even if the exact voice echo was skipped.
414
460
  entry.draft = published;
415
461
  entry.pendingDraft = undefined;
462
+ entry.staleDrafts.clear();
416
463
  }
417
464
  entry.controller.composerChanged(entry.draft);
418
465
  });
@@ -425,14 +472,7 @@ export function apply(ctx) {
425
472
  function Settings() {
426
473
  const [controller, setController] = React.useState(null);
427
474
  React.useEffect(() => {
428
- let settings;
429
- try {
430
- settings = normalizeSettings(
431
- JSON.parse(localStorage.getItem('dsh-live-voice.settings') || '{}'),
432
- );
433
- } catch {
434
- settings = normalizeSettings(null);
435
- }
475
+ const settings = preferences.getSnapshot();
436
476
  const browser = new BrowserSpeakingEngine({ lang: settings.lang });
437
477
  const qwen = new QwenHttpSpeakingEngine({ lang: settings.lang });
438
478
  const meter = new MicrophoneMeter();
@@ -458,9 +498,21 @@ export function apply(ctx) {
458
498
  ownership.run(
459
499
  c,
460
500
  [...controllers.values()].map((entry) => entry.controller).concat([...retiring]),
461
- () => speak(...args),
501
+ async () => {
502
+ try {
503
+ await preferences.ready;
504
+ } catch {
505
+ throw new Error(t('dsh-live-voice.settings.persistence.loadError'));
506
+ }
507
+ if (!disposed && !c.disposed) return speak(...args);
508
+ },
462
509
  );
463
510
  const update = c.updateSettings.bind(c);
511
+ c.applySavedSettings = update;
512
+ settingsControllers.add(c);
513
+ void preferences.ready.catch(() =>
514
+ c.patch({ error: t('dsh-live-voice.settings.persistence.loadError') }),
515
+ );
464
516
  let settingsRevision = 0;
465
517
  c.updateSettings = async (next) => {
466
518
  const revision = ++settingsRevision;
@@ -469,10 +521,11 @@ export function apply(ctx) {
469
521
  qwen.lang = c.getSnapshot().settings.lang;
470
522
  const settings = c.getSnapshot().settings;
471
523
  if (changesRecognition(next)) c.replaceRecognition(recognitionFor(settings, c.meter));
472
- localStorage.setItem('dsh-live-voice.settings', JSON.stringify(settings));
524
+ const saved = savePreferences(c, next);
473
525
  run(c, c.refreshCapabilities());
474
526
  const active = [...controllers.values()];
475
- await Promise.allSettled([
527
+ const [savedSettings] = await Promise.all([
528
+ saved,
476
529
  c.endConversation(),
477
530
  ...active.map((entry) => entry.controller.endConversation()),
478
531
  ]);
@@ -480,8 +533,10 @@ export function apply(ctx) {
480
533
  for (const entry of controllers.values()) {
481
534
  if (entry.closed) continue;
482
535
  if (changesRecognition(next))
483
- entry.controller.replaceRecognition(recognitionFor(settings, entry.controller.meter));
484
- entry.applySettings(settings);
536
+ entry.controller.replaceRecognition(
537
+ recognitionFor(savedSettings, entry.controller.meter),
538
+ );
539
+ entry.applySettings(savedSettings);
485
540
  run(entry.controller, entry.controller.refreshCapabilities());
486
541
  }
487
542
  };
@@ -491,6 +546,7 @@ export function apply(ctx) {
491
546
  refresh();
492
547
  return () => {
493
548
  document.removeEventListener('dsh-live-voice:capabilitieschanged', refresh);
549
+ settingsControllers.delete(c);
494
550
  void c.dispose();
495
551
  };
496
552
  }, []);
@@ -88,6 +88,8 @@ export interface LiveVoiceTranslation extends Translation {
88
88
  'dsh-live-voice.recognition.planned.vote': string;
89
89
  'dsh-live-voice.recognition.planned.voxtral': string;
90
90
  'dsh-live-voice.recognition.planned.webGpu': string;
91
+ 'dsh-live-voice.recognition.presets.custom.description': string;
92
+ 'dsh-live-voice.recognition.presets.custom.label': string;
91
93
  'dsh-live-voice.recognition.presets.long.description': string;
92
94
  'dsh-live-voice.recognition.presets.long.label': string;
93
95
  'dsh-live-voice.recognition.presets.natural.description': string;
@@ -100,6 +102,8 @@ export interface LiveVoiceTranslation extends Translation {
100
102
  'dsh-live-voice.recognition.qwen.endpointHelp': string;
101
103
  'dsh-live-voice.recognition.qwen.hostHelp': string;
102
104
  'dsh-live-voice.recognition.qwen.label': string;
105
+ 'dsh-live-voice.recognition.silenceDetection.customHelp': string;
106
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': string;
103
107
  'dsh-live-voice.recognition.silenceDetection.duration': string;
104
108
  'dsh-live-voice.recognition.silenceDetection.help': string;
105
109
  'dsh-live-voice.recognition.silenceDetection.label': string;
@@ -136,6 +140,9 @@ export interface LiveVoiceTranslation extends Translation {
136
140
  'dsh-live-voice.settings.delivery.toggle': string;
137
141
  'dsh-live-voice.settings.engine.refresh': string;
138
142
  'dsh-live-voice.settings.filters.title': string;
143
+ 'dsh-live-voice.settings.general.title': string;
144
+ 'dsh-live-voice.settings.persistence.loadError': string;
145
+ 'dsh-live-voice.settings.persistence.saveError': string;
139
146
  'dsh-live-voice.settings.tabs.conversation': string;
140
147
  'dsh-live-voice.settings.tabs.recognition': string;
141
148
  'dsh-live-voice.settings.tabs.speak': string;
@@ -101,6 +101,8 @@ const en: LiveVoiceTranslation = {
101
101
  'dsh-live-voice.recognition.planned.vote': 'Coming Soon — vote on repo issues',
102
102
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — Soon',
103
103
  'dsh-live-voice.recognition.planned.webGpu': 'Browser WebGPU Inference — Soon',
104
+ 'dsh-live-voice.recognition.presets.custom.description': 'Choose your own silence duration.',
105
+ 'dsh-live-voice.recognition.presets.custom.label': 'Custom',
104
106
  'dsh-live-voice.recognition.presets.long.description': 'Wait through longer thinking pauses.',
105
107
  'dsh-live-voice.recognition.presets.long.label': 'Long',
106
108
  'dsh-live-voice.recognition.presets.natural.description': 'Allow normal pauses between phrases.',
@@ -118,6 +120,9 @@ const en: LiveVoiceTranslation = {
118
120
  'dsh-live-voice.recognition.qwen.hostHelp':
119
121
  'Host-wide settings for the Qwen3 ASR + TTS server. Enter any HTTP or HTTPS base URL reachable from the DSH host. The browser accesses it through authenticated DSH routes.',
120
122
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — HTTP API',
123
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
124
+ 'Enter a whole number from 100 to 10,000 ms. Saved when you leave the field. Short pauses may split speech; recognition adds its own latency.',
125
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': 'Custom pause (milliseconds)',
121
126
  'dsh-live-voice.recognition.silenceDetection.duration': 'Pause before sending: {milliseconds} ms',
122
127
  'dsh-live-voice.recognition.silenceDetection.help':
123
128
  'Controls how long a pause must last before captured speech is sent for recognition.',
@@ -162,6 +167,11 @@ const en: LiveVoiceTranslation = {
162
167
  'dsh-live-voice.settings.delivery.toggle': 'Automatic delivery mode',
163
168
  'dsh-live-voice.settings.engine.refresh': 'Refresh available engines',
164
169
  'dsh-live-voice.settings.filters.title': 'Filtering',
170
+ 'dsh-live-voice.settings.general.title': 'General',
171
+ 'dsh-live-voice.settings.persistence.loadError':
172
+ 'Could not load Live Voice settings from the server. Reload to try again.',
173
+ 'dsh-live-voice.settings.persistence.saveError':
174
+ 'Could not save Live Voice settings on the server. Please try again.',
165
175
  'dsh-live-voice.settings.tabs.conversation': 'Conversation',
166
176
  'dsh-live-voice.settings.tabs.recognition': 'Speech recognition',
167
177
  'dsh-live-voice.settings.tabs.speak': 'Speech',
@@ -107,6 +107,8 @@ const es: LiveVoiceTranslation = {
107
107
  'Próximamente — vota en las incidencias del repositorio',
108
108
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — Próximamente',
109
109
  'dsh-live-voice.recognition.planned.webGpu': 'Inferencia WebGPU en el navegador — Próximamente',
110
+ 'dsh-live-voice.recognition.presets.custom.description': 'Elige la duración del silencio.',
111
+ 'dsh-live-voice.recognition.presets.custom.label': 'Personalizada',
110
112
  'dsh-live-voice.recognition.presets.long.description':
111
113
  'Espera durante pausas de reflexión más largas.',
112
114
  'dsh-live-voice.recognition.presets.long.label': 'Larga',
@@ -126,6 +128,9 @@ const es: LiveVoiceTranslation = {
126
128
  'dsh-live-voice.recognition.qwen.hostHelp':
127
129
  'Configuración para todo el host del servidor Qwen3 ASR + TTS. Introduce una URL base HTTP o HTTPS accesible desde el host de DSH. El navegador accede a ella mediante las rutas autenticadas de DSH.',
128
130
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — API HTTP',
131
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
132
+ 'Introduce un número entero de 100 a 10.000 ms. Se guarda al salir del campo. Las pausas cortas pueden dividir el habla; el reconocimiento añade su propia latencia.',
133
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': 'Pausa personalizada (milisegundos)',
129
134
  'dsh-live-voice.recognition.silenceDetection.duration':
130
135
  'Pausa antes de enviar: {milliseconds} ms',
131
136
  'dsh-live-voice.recognition.silenceDetection.help':
@@ -175,6 +180,11 @@ const es: LiveVoiceTranslation = {
175
180
  'dsh-live-voice.settings.delivery.toggle': 'Modo de envío automático',
176
181
  'dsh-live-voice.settings.engine.refresh': 'Actualizar los motores disponibles',
177
182
  'dsh-live-voice.settings.filters.title': 'Filtrado',
183
+ 'dsh-live-voice.settings.general.title': 'General',
184
+ 'dsh-live-voice.settings.persistence.loadError':
185
+ 'No se pudieron cargar los ajustes de Live Voice del servidor. Recarga para volver a intentarlo.',
186
+ 'dsh-live-voice.settings.persistence.saveError':
187
+ 'No se pudieron guardar los ajustes de Live Voice en el servidor. Inténtalo de nuevo.',
178
188
  'dsh-live-voice.settings.tabs.conversation': 'Conversación',
179
189
  'dsh-live-voice.settings.tabs.recognition': 'Reconocimiento de voz',
180
190
  'dsh-live-voice.settings.tabs.speak': 'Síntesis de voz',
@@ -109,6 +109,8 @@ const fr: LiveVoiceTranslation = {
109
109
  'dsh-live-voice.recognition.planned.voxtral': 'Voxtral Realtime — Bientôt disponible',
110
110
  'dsh-live-voice.recognition.planned.webGpu':
111
111
  'Inférence WebGPU dans le navigateur — Bientôt disponible',
112
+ 'dsh-live-voice.recognition.presets.custom.description': 'Choisissez la durée du silence.',
113
+ 'dsh-live-voice.recognition.presets.custom.label': 'Personnalisée',
112
114
  'dsh-live-voice.recognition.presets.long.description':
113
115
  'Attend pendant les pauses de réflexion plus longues.',
114
116
  'dsh-live-voice.recognition.presets.long.label': 'Longue',
@@ -129,6 +131,9 @@ const fr: LiveVoiceTranslation = {
129
131
  'dsh-live-voice.recognition.qwen.hostHelp':
130
132
  'Paramètres communs à tout l’hôte pour le serveur Qwen3 ASR + TTS. Saisissez une URL de base HTTP ou HTTPS accessible depuis l’hôte DSH. Le navigateur y accède via les routes authentifiées de DSH.',
131
133
  'dsh-live-voice.recognition.qwen.label': 'Qwen3 ASR — API HTTP',
134
+ 'dsh-live-voice.recognition.silenceDetection.customHelp':
135
+ 'Saisissez un entier de 100 à 10 000 ms. Enregistré en quittant le champ. Les pauses courtes peuvent couper la parole ; la reconnaissance ajoute sa propre latence.',
136
+ 'dsh-live-voice.recognition.silenceDetection.customLabel': 'Pause personnalisée (millisecondes)',
132
137
  'dsh-live-voice.recognition.silenceDetection.duration': 'Pause avant l’envoi : {milliseconds} ms',
133
138
  'dsh-live-voice.recognition.silenceDetection.help':
134
139
  'Détermine la durée de pause nécessaire avant que la parole capturée soit envoyée pour reconnaissance.',
@@ -178,6 +183,11 @@ const fr: LiveVoiceTranslation = {
178
183
  'dsh-live-voice.settings.delivery.toggle': 'Mode d’envoi automatique',
179
184
  'dsh-live-voice.settings.engine.refresh': 'Actualiser les moteurs disponibles',
180
185
  'dsh-live-voice.settings.filters.title': 'Filtrage',
186
+ 'dsh-live-voice.settings.general.title': 'Général',
187
+ 'dsh-live-voice.settings.persistence.loadError':
188
+ 'Impossible de charger les paramètres Live Voice depuis le serveur. Rechargez pour réessayer.',
189
+ 'dsh-live-voice.settings.persistence.saveError':
190
+ 'Impossible de sauvegarder les paramètres Live Voice sur le serveur. Réessayez.',
181
191
  'dsh-live-voice.settings.tabs.conversation': 'Conversation',
182
192
  'dsh-live-voice.settings.tabs.recognition': 'Reconnaissance vocale',
183
193
  'dsh-live-voice.settings.tabs.speak': 'Synthèse vocale',