dsh-live-voice 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (138) hide show
  1. package/CHANGELOG.md +39 -1
  2. package/DEVELOPMENT.md +13 -0
  3. package/LICENSE +201 -674
  4. package/PLAN.md +28 -0
  5. package/README.md +65 -110
  6. package/docs/ARCHITECTURE.md +40 -0
  7. package/docs/CHOOSING-AN-ENGINE.md +120 -0
  8. package/docs/CONFIGURATION.md +149 -0
  9. package/docs/REVIEW.md +35 -0
  10. package/docs/VOICE-LIFECYCLE.md +59 -0
  11. package/lib/client.js +3652 -1576
  12. package/lib/server.js +641 -453
  13. package/package.json +17 -6
  14. package/scripts/build.ts +16 -4
  15. package/scripts/preview-ui.ts +1 -1
  16. package/src/{client/index.ts → app/client/apply.tsx} +152 -59
  17. package/src/app/client/i18n/DshLanguageBoundary.tsx +42 -0
  18. package/src/app/client/i18n/catalogs/base.ts +231 -0
  19. package/src/app/client/i18n/catalogs/en.ts +258 -0
  20. package/src/app/client/i18n/catalogs/es.ts +279 -0
  21. package/src/app/client/i18n/catalogs/fr.ts +281 -0
  22. package/src/app/client/i18n/catalogs/hi.ts +263 -0
  23. package/src/app/client/i18n/catalogs/index.ts +18 -0
  24. package/src/app/client/i18n/catalogs/pt-BR.ts +270 -0
  25. package/src/app/client/i18n/catalogs/zh.ts +248 -0
  26. package/src/app/client/i18n/index.ts +4 -0
  27. package/src/app/client/i18n/registerDshLocales.ts +43 -0
  28. package/src/app/client/i18n/runtime.tsx +58 -0
  29. package/src/app/client/index.ts +3 -0
  30. package/src/app/client/registerSlots.tsx +30 -0
  31. package/src/app/client/slotDefinitions.ts +32 -0
  32. package/src/app/server/apply.ts +479 -0
  33. package/src/app/server/index.ts +2 -0
  34. package/src/app/server/registerRoutes.ts +11 -0
  35. package/src/modules/conversation/components/ConversationControls.tsx +40 -0
  36. package/src/modules/conversation/components/ConversationStatusBar.tsx +155 -0
  37. package/src/modules/conversation/components/DeliveryModeButton.tsx +43 -0
  38. package/src/modules/conversation/components/MicrophoneButton.tsx +6 -0
  39. package/src/modules/conversation/components/PlaybackControls.tsx +41 -0
  40. package/src/modules/conversation/components/SpeakButton.tsx +30 -0
  41. package/src/modules/conversation/components/Waveform.tsx +68 -0
  42. package/src/modules/conversation/components/conversationStatus.ts +17 -0
  43. package/src/modules/conversation/components/createConversationComponents.tsx +505 -0
  44. package/src/modules/conversation/components/index.ts +8 -0
  45. package/src/modules/conversation/hooks/index.ts +2 -0
  46. package/src/modules/conversation/hooks/useConversationActions.ts +27 -0
  47. package/src/modules/conversation/hooks/useConversationController.ts +15 -0
  48. package/src/modules/conversation/index.ts +3 -0
  49. package/src/{client → modules/conversation/models}/chat.ts +2 -1
  50. package/src/modules/conversation/models/index.ts +1 -0
  51. package/src/{core → modules/core}/coordinator.ts +133 -7
  52. package/src/modules/core/index.ts +6 -0
  53. package/src/{engines/qwen-http-host.ts → modules/core/qwen/QwenHttpHost.ts} +2 -2
  54. package/src/modules/core/qwen/QwenSettings.tsx +148 -0
  55. package/src/{core → modules/core}/settings.ts +23 -0
  56. package/src/modules/recognition/components/RecognitionCapabilityStatus.tsx +19 -0
  57. package/src/{engines/recognition/qwen-http.ts → modules/recognition/engines/qwen/QwenRecognitionEngine.ts} +1 -1
  58. package/src/{engines/recognition/whisper-http.ts → modules/recognition/engines/whisper/WhisperRecognitionEngine.ts} +1 -1
  59. package/src/modules/recognition/engines/whisper/WhisperSettings.tsx +150 -0
  60. package/src/modules/recognition/index.ts +4 -0
  61. package/src/modules/recognition/qwen/QwenRecognitionSettings.tsx +4 -0
  62. package/src/modules/settings/components/LiveVoiceSettings.tsx +72 -0
  63. package/src/modules/settings/components/SettingsHeader.tsx +49 -0
  64. package/src/modules/settings/components/VersionBadges.tsx +39 -0
  65. package/src/modules/settings/components/createLiveVoiceSettings.tsx +8 -0
  66. package/src/modules/settings/components/index.ts +4 -0
  67. package/src/modules/settings/hooks/index.ts +3 -0
  68. package/src/modules/settings/hooks/useAudioDevices.ts +38 -0
  69. package/src/modules/settings/hooks/useLiveVoiceSettings.ts +4 -0
  70. package/src/modules/settings/hooks/useReleaseStatus.ts +17 -0
  71. package/src/modules/settings/index.ts +4 -0
  72. package/src/modules/settings/models/index.ts +1 -0
  73. package/src/modules/settings/models/settingsStorage.ts +14 -0
  74. package/src/modules/settings/sections/conversation/ConversationDelaySettings.tsx +24 -0
  75. package/src/modules/settings/sections/conversation/ConversationSettingsSection.tsx +23 -0
  76. package/src/modules/settings/sections/conversation/DeliverySettings.tsx +38 -0
  77. package/src/modules/settings/sections/conversation/HoldToTalkSettings.tsx +20 -0
  78. package/src/modules/settings/sections/conversation/VoiceModeSettings.tsx +31 -0
  79. package/src/modules/settings/sections/conversation/index.ts +5 -0
  80. package/src/modules/settings/sections/index.ts +3 -0
  81. package/src/modules/settings/sections/recognition/RecognitionEngineSettings.tsx +109 -0
  82. package/src/modules/settings/sections/recognition/RecognitionFilterSettings.tsx +33 -0
  83. package/src/modules/settings/sections/recognition/RecognitionSettingsSection.tsx +20 -0
  84. package/src/modules/settings/sections/recognition/RecognitionStatus.tsx +19 -0
  85. package/src/modules/settings/sections/recognition/SilenceDetectionSettings.tsx +58 -0
  86. package/src/modules/settings/sections/recognition/VoiceCommandSettings.tsx +41 -0
  87. package/src/modules/settings/sections/recognition/index.ts +5 -0
  88. package/src/modules/settings/sections/speak/OutputFilterSettings.tsx +39 -0
  89. package/src/modules/settings/sections/speak/PlaybackPolicySettings.tsx +26 -0
  90. package/src/modules/settings/sections/speak/SpeakSettingsSection.tsx +16 -0
  91. package/src/modules/settings/sections/speak/SpeechAdvancedSettings.tsx +81 -0
  92. package/src/modules/settings/sections/speak/SpeechEngineSettings.tsx +80 -0
  93. package/src/modules/settings/sections/speak/index.ts +4 -0
  94. package/src/modules/settings/services/releases.ts +117 -0
  95. package/src/modules/speak/engines/audio/HostAudioEngine.ts +167 -0
  96. package/src/modules/speak/engines/audio/M4aAacTranscoder.ts +58 -0
  97. package/src/modules/speak/engines/qwen/QwenSpeakingEngine.ts +14 -0
  98. package/src/{engines/speaking/say.ts → modules/speak/engines/say/SaySpeakingEngine.ts} +26 -4
  99. package/src/modules/speak/index.ts +5 -0
  100. package/src/modules/speak/qwen/QwenSpeakingSettings.tsx +4 -0
  101. package/src/modules/speak/services/speechQueue.ts +2 -0
  102. package/src/server.ts +9 -365
  103. package/src/shared/design-system/buttons/IconButton.tsx +33 -0
  104. package/src/shared/design-system/buttons/PillButton.tsx +11 -0
  105. package/src/shared/design-system/buttons/ToggleButton.tsx +6 -0
  106. package/src/shared/design-system/buttons/index.ts +3 -0
  107. package/src/shared/design-system/feedback/ErrorMessage.tsx +27 -0
  108. package/src/shared/design-system/feedback/StatusBadge.tsx +9 -0
  109. package/src/shared/design-system/feedback/StatusMessage.tsx +4 -0
  110. package/src/shared/design-system/feedback/index.ts +3 -0
  111. package/src/shared/design-system/forms/CheckboxField.tsx +20 -0
  112. package/src/shared/design-system/forms/NumberField.tsx +12 -0
  113. package/src/shared/design-system/forms/SelectField.tsx +20 -0
  114. package/src/shared/design-system/forms/TextAreaField.tsx +12 -0
  115. package/src/shared/design-system/forms/TextField.tsx +12 -0
  116. package/src/shared/design-system/forms/index.ts +5 -0
  117. package/src/shared/design-system/icons/Icon.tsx +20 -0
  118. package/src/shared/design-system/icons/icons.ts +16 -0
  119. package/src/shared/design-system/icons/index.ts +2 -0
  120. package/src/shared/design-system/index.ts +5 -0
  121. package/src/shared/design-system/layout/SettingsCard.tsx +9 -0
  122. package/src/shared/design-system/layout/SettingsSection.tsx +9 -0
  123. package/src/shared/design-system/layout/SettingsSubcard.tsx +14 -0
  124. package/src/shared/design-system/layout/SettingsTabs.tsx +54 -0
  125. package/src/shared/design-system/layout/index.ts +4 -0
  126. package/src/{client/styles.ts → styles/index.ts} +3 -1
  127. package/src/client/components.ts +0 -1080
  128. package/src/client/qwen-settings.ts +0 -144
  129. package/src/client/whisper-settings.ts +0 -147
  130. package/src/engines/speaking/qwen-http.ts +0 -124
  131. /package/src/{core → modules/core}/filters.ts +0 -0
  132. /package/src/{core → modules/core}/microphone.ts +0 -0
  133. /package/src/{core → modules/core}/ownership.ts +0 -0
  134. /package/src/{core → modules/core}/transcript.ts +0 -0
  135. /package/src/{engines/recognition/browser.ts → modules/recognition/engines/browser/BrowserRecognitionEngine.ts} +0 -0
  136. /package/src/{engines/recognition/whisper-http-host.ts → modules/recognition/engines/whisper/whisperRecognitionHost.ts} +0 -0
  137. /package/src/{engines/speaking/browser.ts → modules/speak/engines/browser/BrowserSpeakingEngine.ts} +0 -0
  138. /package/src/{engines/speaking/say-client.ts → modules/speak/engines/say/sayClient.ts} +0 -0
package/lib/server.js CHANGED
@@ -1,249 +1,86 @@
1
- // src/engines/speaking/say.ts
2
- import { spawn as nodeSpawn } from "node:child_process";
3
- import * as nodeFs from "node:fs/promises";
4
- import { constants } from "node:fs";
5
- import { tmpdir } from "node:os";
6
- import { join } from "node:path";
7
- function failure(message, code, cause) {
8
- return Object.assign(new Error(message, cause === void 0 ? void 0 : { cause }), { code });
9
- }
10
- function aborted(reason) {
11
- return Object.assign(
12
- new Error("Speech cancelled", reason === void 0 ? void 0 : { cause: reason }),
13
- {
14
- name: "AbortError",
15
- code: "ABORT_ERR"
16
- }
17
- );
18
- }
19
- var SayEngine = class {
20
- #spawn;
21
- #fs;
22
- #platform;
23
- #tempRoot;
24
- #killAfterMs;
25
- #closeAfterMs;
26
- #tail = Promise.resolve();
27
- #requests = /* @__PURE__ */ new Set();
28
- #active = null;
29
- #unclosed = null;
30
- #state = "idle";
31
- #lastError = null;
32
- constructor({
33
- spawn = nodeSpawn,
34
- fs = nodeFs,
35
- platform = process.platform,
36
- tempRoot = tmpdir(),
37
- killAfterMs = 250,
38
- closeAfterMs = 1e3
39
- } = {}) {
40
- for (const value of [killAfterMs, closeAfterMs]) {
41
- if (!Number.isFinite(value) || value < 0 || value > 2147483647) {
42
- throw new TypeError("Cancellation deadlines must be finite nonnegative milliseconds");
43
- }
44
- }
45
- this.#spawn = spawn;
46
- this.#fs = fs;
47
- this.#platform = platform;
48
- this.#tempRoot = tempRoot;
49
- this.#killAfterMs = killAfterMs;
50
- this.#closeAfterMs = closeAfterMs;
51
- }
52
- get state() {
53
- return this.#state;
54
- }
55
- get lastError() {
56
- return this.#lastError;
57
- }
58
- /** Checks the host executable, not browser support or installed voices. */
59
- async getCapabilities() {
60
- if (this.#platform !== "darwin") {
61
- return { supported: false, pause: false, resume: false, reason: "unsupported-platform" };
62
- }
63
- try {
64
- await this.#fs.access("/usr/bin/say", constants.X_OK);
65
- return { supported: true, pause: true, resume: true, reason: null };
66
- } catch {
67
- return { supported: false, pause: false, resume: false, reason: "executable-unavailable" };
68
- }
69
- }
70
- speak(text, { voice, rate, signal } = {}) {
71
- if (typeof text !== "string") return Promise.reject(new TypeError("text must be a string"));
72
- if (voice !== void 0 && (typeof voice !== "string" || !voice.trim() || voice.includes("\0"))) {
73
- return Promise.reject(new TypeError("voice must be a nonempty string without NUL"));
74
- }
75
- if (rate !== void 0 && (!Number.isFinite(rate) || rate <= 0)) {
76
- return Promise.reject(new TypeError("rate must be a positive finite number"));
77
- }
78
- if (signal !== void 0 && (signal === null || typeof signal.addEventListener !== "function" || typeof signal.removeEventListener !== "function" || typeof signal.aborted !== "boolean")) {
79
- return Promise.reject(new TypeError("signal must be an AbortSignal"));
80
- }
81
- for (const request2 of this.#requests) this.#cancel(request2);
82
- const request = { cancelled: false, reason: void 0, cancelChild: null };
83
- const onAbort = () => this.#cancel(request, signal.reason);
84
- this.#requests.add(request);
85
- signal?.addEventListener("abort", onAbort, { once: true });
86
- if (signal?.aborted) onAbort();
87
- const result = this.#tail.then(() => this.#run(request, text, voice, rate));
88
- const settled = result.finally(() => {
89
- signal?.removeEventListener("abort", onAbort);
90
- this.#requests.delete(request);
91
- });
92
- this.#tail = settled.catch(() => {
93
- });
94
- return settled;
95
- }
96
- /** Resolves after queued requests and cleanup; rejects teardown failures. */
97
- async stop() {
98
- for (const request of this.#requests) this.#cancel(request);
99
- const pending = this.#tail;
100
- await pending;
101
- if (this.#unclosed || this.#state === "error") throw this.#lastError;
102
- }
103
- pause() {
104
- return this.#control("speaking", "paused", "SIGSTOP");
105
- }
106
- resume() {
107
- return this.#control("paused", "speaking", "SIGCONT");
108
- }
109
- #control(from, to, signal) {
110
- if (this.#platform !== "darwin" || this.#state !== from || !this.#active?.child || this.#active.cancelled)
111
- return false;
112
- try {
113
- if (!this.#active.child.kill(signal))
114
- throw failure("Unable to signal speech process", "SAY_SIGNAL_FAILED");
115
- this.#state = to;
116
- return true;
117
- } catch (error) {
118
- this.#lastError = error;
119
- return false;
120
- }
121
- }
122
- #cancel(request, reason) {
123
- if (request.cancelled) return;
124
- request.cancelled = true;
125
- request.reason = reason;
126
- request.cancelChild?.();
127
- }
128
- async #run(request, text, voice, rate) {
129
- let directory;
130
- let error;
131
- const check = () => {
132
- if (request.cancelled) throw aborted(request.reason);
133
- };
134
- try {
135
- check();
136
- if (this.#unclosed)
137
- throw failure("Previous speech process has not closed", "SAY_PROCESS_UNCLOSED");
138
- this.#active = request;
139
- this.#state = "preparing";
140
- this.#lastError = null;
141
- const capability = await this.getCapabilities();
142
- check();
143
- if (!capability.supported) throw failure("macOS say is unavailable", "SAY_UNAVAILABLE");
144
- directory = await this.#fs.mkdtemp(join(this.#tempRoot, "dsh-live-voice-say-"));
145
- check();
146
- await this.#fs.chmod(directory, 448);
147
- const file = join(directory, "speech.txt");
148
- await this.#fs.writeFile(file, text, { encoding: "utf8", mode: 384, flag: "wx" });
149
- await this.#fs.chmod(file, 384);
150
- check();
151
- const args = ["-f", file];
152
- if (voice !== void 0) args.push("-v", voice);
153
- if (rate !== void 0) args.push("-r", String(rate));
154
- const child = this.#spawn("/usr/bin/say", args, { shell: false, stdio: "ignore" });
155
- request.child = child;
156
- this.#state = "speaking";
157
- await this.#waitForClose(request, child);
158
- check();
159
- } catch (caught) {
160
- error = caught;
161
- } finally {
162
- request.cancelChild = null;
163
- if (directory) {
164
- try {
165
- await this.#fs.rm(directory, { recursive: true, force: true });
166
- } catch (cleanupError) {
167
- error = failure(
168
- "Speech temporary-file cleanup failed",
169
- "SAY_CLEANUP_FAILED",
170
- error ? new AggregateError([error, cleanupError]) : cleanupError
171
- );
172
- }
173
- }
174
- if (!error && request.cancelled) error = aborted(request.reason);
175
- if (this.#active === request) this.#active = null;
176
- if (error && error.name !== "AbortError") {
177
- this.#state = "error";
178
- this.#lastError = error;
179
- } else if (!this.#unclosed && this.#state !== "error") this.#state = "idle";
180
- }
181
- if (error) throw error;
182
- }
183
- #waitForClose(request, child) {
184
- return new Promise((resolve, reject) => {
185
- let closed = false;
186
- let processError;
187
- let killTimer;
188
- let closeTimer;
189
- const send = (signal) => {
190
- try {
191
- if (!child.kill(signal))
192
- processError ??= failure("Unable to signal speech process", "SAY_SIGNAL_FAILED");
193
- } catch (error) {
194
- processError ??= error;
195
- }
196
- };
197
- const onError = (error) => {
198
- processError ??= error;
199
- };
200
- child.on("error", onError);
201
- child.once("close", (code, signal) => {
202
- closed = true;
203
- clearTimeout(killTimer);
204
- clearTimeout(closeTimer);
205
- child.removeListener("error", onError);
206
- if (this.#unclosed === child) this.#unclosed = null;
207
- if (request.cancelled) reject(aborted(request.reason));
208
- else if (processError) reject(processError);
209
- else if (code !== 0)
210
- reject(
211
- failure("Speech process exited unsuccessfully", "SAY_EXIT_FAILED", { code, signal })
212
- );
213
- else resolve();
214
- });
215
- request.cancelChild = () => {
216
- if (closed) return;
217
- const paused = this.#state === "paused";
218
- this.#state = "stopping";
219
- if (paused) send("SIGCONT");
220
- send("SIGTERM");
221
- if (closed) return;
222
- killTimer = setTimeout(() => {
223
- send("SIGKILL");
224
- if (closed) return;
225
- closeTimer = setTimeout(() => {
226
- if (closed) return;
227
- this.#unclosed = child;
228
- reject(
229
- failure(
230
- "Speech process did not close after cancellation",
231
- "SAY_STOP_TIMEOUT",
232
- processError
233
- )
234
- );
235
- }, this.#closeAfterMs);
236
- }, this.#killAfterMs);
237
- };
238
- if (request.cancelled) request.cancelChild();
239
- });
240
- }
241
- };
1
+ // src/modules/core/qwen/QwenHttpHost.ts
2
+ import { readFile as readFile2, mkdir as mkdir2, writeFile as writeFile2, rename as rename2, rm as rm2 } from "node:fs/promises";
3
+ import { homedir as homedir2 } from "node:os";
4
+ import { dirname as dirname2, join as join2 } from "node:path";
5
+ import { randomUUID as randomUUID2 } from "node:crypto";
242
6
 
243
- // src/engines/recognition/whisper-http-host.ts
7
+ // src/modules/core/settings.ts
8
+ var voiceDetectionPresets = Object.freeze({
9
+ short: Object.freeze({
10
+ silenceMs: 900,
11
+ label: "Short",
12
+ description: "Send quickly after a short pause."
13
+ }),
14
+ natural: Object.freeze({
15
+ silenceMs: 1500,
16
+ label: "Natural",
17
+ description: "Allow normal pauses between phrases."
18
+ }),
19
+ long: Object.freeze({
20
+ silenceMs: 2200,
21
+ label: "Long",
22
+ description: "Wait through longer thinking pauses."
23
+ })
24
+ });
25
+ var qwenVoices = Object.freeze([
26
+ Object.freeze({ value: "aiden", label: "Aiden \u2014 male, American English" }),
27
+ Object.freeze({ value: "ryan", label: "Ryan \u2014 male, English" }),
28
+ Object.freeze({ value: "uncle_fu", label: "Uncle Fu \u2014 male, Chinese" }),
29
+ Object.freeze({ value: "dylan", label: "Dylan \u2014 male, Beijing Chinese" }),
30
+ Object.freeze({ value: "eric", label: "Eric \u2014 male, Sichuan Chinese" }),
31
+ Object.freeze({ value: "vivian", label: "Vivian \u2014 female, Chinese" }),
32
+ Object.freeze({ value: "serena", label: "Serena \u2014 female, Chinese" }),
33
+ Object.freeze({ value: "ono_anna", label: "Ono Anna \u2014 female, Japanese" }),
34
+ Object.freeze({ value: "sohee", label: "Sohee \u2014 female, Korean" })
35
+ ]);
36
+ var defaultQwenVoice = qwenVoices[0].value;
37
+ var isQwenVoice = (value) => qwenVoices.some((voice) => voice.value === value);
38
+ var defaultAgentVoiceContext = `Live Voice output is active. Your entire user-facing response will be spoken aloud.
39
+ Be concise and conversational. Lead with the answer or next action. Avoid unnecessary repetition, long preambles, dense lists, raw code, paths, and verbose status narration.
40
+ Do not narrate routine tool activity by default. If the user explicitly asks you to keep them informed while working, provide brief spoken progress updates only at meaningful milestones.`;
41
+ var defaultSettings = Object.freeze({
42
+ engine: "browser",
43
+ recognitionEngine: "browser",
44
+ recognitionProcessLocally: true,
45
+ recognitionAutoInstall: true,
46
+ voiceDetectionPreset: "natural",
47
+ recognitionMaxUtteranceSeconds: 60,
48
+ microphoneEnabled: true,
49
+ holdToTalkEnabled: true,
50
+ announceAssistantMessages: true,
51
+ agentVoiceContextEnabled: true,
52
+ agentVoiceContext: defaultAgentVoiceContext,
53
+ interruptSpeechOnUserMessage: false,
54
+ sendingMode: "manual",
55
+ autoSendDelaySeconds: 4,
56
+ assistantSpeechDelaySeconds: 3,
57
+ mode: "speaker",
58
+ lang: "pt-BR",
59
+ recognitionLang: "pt-BR",
60
+ voice: "",
61
+ inputDeviceId: "",
62
+ outputDeviceId: "",
63
+ recognitionFilterEnabled: true,
64
+ recognitionMinimumWords: 2,
65
+ voiceCommandsEnabled: true,
66
+ voiceCommandSend: "send, send message",
67
+ voiceCommandQueue: "queue, queue message",
68
+ voiceCommandEnd: "end, end conversation",
69
+ voiceCommandMute: "mute, stop listening",
70
+ voiceCommandResume: "resume, start listening",
71
+ voiceCommandStopSpeaking: "stop talking, stop speaking, shut up",
72
+ voiceCommandClear: "clear all, clear message",
73
+ outputCodeFilterEnabled: true,
74
+ outputCodeMaxLines: 5,
75
+ outputCodeNotice: "Look the code on out conversation",
76
+ rate: 1,
77
+ segmentGapMs: 400
78
+ });
79
+
80
+ // src/modules/recognition/engines/whisper/whisperRecognitionHost.ts
244
81
  import { readFile, mkdir, writeFile, rename, rm } from "node:fs/promises";
245
82
  import { homedir } from "node:os";
246
- import { dirname, join as join2 } from "node:path";
83
+ import { dirname, join } from "node:path";
247
84
  import { randomUUID } from "node:crypto";
248
85
  var LOOPBACK = /* @__PURE__ */ new Set(["localhost", "127.0.0.1", "::1", "[::1]"]);
249
86
  function validateWhisperConfig(value) {
@@ -257,7 +94,7 @@ function validateWhisperConfig(value) {
257
94
  throw new Error("Request timeout must be an integer between 100 and 300000 ms.");
258
95
  return { url: url.href, healthUrl: health, timeoutMs: value.timeoutMs };
259
96
  }
260
- function createWhisperConfigStore(path = join2(homedir(), ".dsh", "dsh-live-voice-whisper.json")) {
97
+ function createWhisperConfigStore(path = join(homedir(), ".dsh", "dsh-live-voice-whisper.json")) {
261
98
  return {
262
99
  async load() {
263
100
  try {
@@ -378,119 +215,46 @@ var WhisperHttpHost = class {
378
215
  try {
379
216
  const config = configuration ? validateWhisperConfig(configuration) : await this.getConfig();
380
217
  const response = await this.request(
381
- new URL(config.healthUrl, config.url),
382
- { signal },
383
- config
384
- );
385
- return response.ok ? { supported: true, local: true, streaming: false, maxBytes: this.maxBytes } : {
386
- supported: false,
387
- local: true,
388
- streaming: false,
389
- reason: "Whisper HTTP health check failed (" + response.status + ")."
390
- };
391
- } catch (error) {
392
- return {
393
- supported: false,
394
- local: true,
395
- streaming: false,
396
- reason: "Whisper HTTP server is unreachable: " + (error?.message || error)
397
- };
398
- }
399
- }
400
- async transcribe(input, { lang = "auto", signal } = {}) {
401
- const bytes = validateMonoPcm16Wav(input, { maxBytes: this.maxBytes }), form = new FormData();
402
- form.append("file", new Blob([bytes], { type: "audio/wav" }), "utterance.wav");
403
- form.append("response_format", "json");
404
- form.append("language", lang.startsWith("pt") ? "pt" : lang.startsWith("en") ? "en" : "auto");
405
- const config = await this.getConfig();
406
- return this.request(
407
- new URL(config.url),
408
- { method: "POST", body: form, signal },
409
- config,
410
- async (response) => {
411
- if (!response.ok)
412
- throw new Error("Whisper HTTP transcription failed (" + response.status + ").");
413
- const json = await response.json();
414
- return { text: clean(json?.text) };
415
- }
416
- );
417
- }
418
- };
419
-
420
- // src/engines/qwen-http-host.ts
421
- import { readFile as readFile2, mkdir as mkdir2, writeFile as writeFile2, rename as rename2, rm as rm2 } from "node:fs/promises";
422
- import { homedir as homedir2 } from "node:os";
423
- import { dirname as dirname2, join as join3 } from "node:path";
424
- import { randomUUID as randomUUID2 } from "node:crypto";
425
-
426
- // src/core/settings.ts
427
- var voiceDetectionPresets = Object.freeze({
428
- short: Object.freeze({
429
- silenceMs: 900,
430
- label: "Short",
431
- description: "Send quickly after a short pause."
432
- }),
433
- natural: Object.freeze({
434
- silenceMs: 1500,
435
- label: "Natural",
436
- description: "Allow normal pauses between phrases."
437
- }),
438
- long: Object.freeze({
439
- silenceMs: 2200,
440
- label: "Long",
441
- description: "Wait through longer thinking pauses."
442
- })
443
- });
444
- var qwenVoices = Object.freeze([
445
- Object.freeze({ value: "aiden", label: "Aiden \u2014 male, American English" }),
446
- Object.freeze({ value: "ryan", label: "Ryan \u2014 male, English" }),
447
- Object.freeze({ value: "uncle_fu", label: "Uncle Fu \u2014 male, Chinese" }),
448
- Object.freeze({ value: "dylan", label: "Dylan \u2014 male, Beijing Chinese" }),
449
- Object.freeze({ value: "eric", label: "Eric \u2014 male, Sichuan Chinese" }),
450
- Object.freeze({ value: "vivian", label: "Vivian \u2014 female, Chinese" }),
451
- Object.freeze({ value: "serena", label: "Serena \u2014 female, Chinese" }),
452
- Object.freeze({ value: "ono_anna", label: "Ono Anna \u2014 female, Japanese" }),
453
- Object.freeze({ value: "sohee", label: "Sohee \u2014 female, Korean" })
454
- ]);
455
- var defaultQwenVoice = qwenVoices[0].value;
456
- var isQwenVoice = (value) => qwenVoices.some((voice) => voice.value === value);
457
- var defaultSettings = Object.freeze({
458
- engine: "browser",
459
- recognitionEngine: "browser",
460
- recognitionProcessLocally: true,
461
- recognitionAutoInstall: true,
462
- voiceDetectionPreset: "natural",
463
- recognitionMaxUtteranceSeconds: 60,
464
- microphoneEnabled: true,
465
- holdToTalkEnabled: true,
466
- announceAssistantMessages: true,
467
- interruptSpeechOnUserMessage: false,
468
- sendingMode: "manual",
469
- autoSendDelaySeconds: 4,
470
- assistantSpeechDelaySeconds: 3,
471
- mode: "speaker",
472
- lang: "pt-BR",
473
- recognitionLang: "pt-BR",
474
- voice: "",
475
- inputDeviceId: "",
476
- outputDeviceId: "",
477
- recognitionFilterEnabled: true,
478
- recognitionMinimumWords: 2,
479
- voiceCommandsEnabled: true,
480
- voiceCommandSend: "send, send message",
481
- voiceCommandQueue: "queue, queue message",
482
- voiceCommandEnd: "end, end conversation",
483
- voiceCommandMute: "mute, stop listening",
484
- voiceCommandResume: "resume, start listening",
485
- voiceCommandStopSpeaking: "stop talking, stop speaking, shut up",
486
- voiceCommandClear: "clear all, clear message",
487
- outputCodeFilterEnabled: true,
488
- outputCodeMaxLines: 5,
489
- outputCodeNotice: "Look the code on out conversation",
490
- rate: 1
491
- });
218
+ new URL(config.healthUrl, config.url),
219
+ { signal },
220
+ config
221
+ );
222
+ return response.ok ? { supported: true, local: true, streaming: false, maxBytes: this.maxBytes } : {
223
+ supported: false,
224
+ local: true,
225
+ streaming: false,
226
+ reason: "Whisper HTTP health check failed (" + response.status + ")."
227
+ };
228
+ } catch (error) {
229
+ return {
230
+ supported: false,
231
+ local: true,
232
+ streaming: false,
233
+ reason: "Whisper HTTP server is unreachable: " + (error?.message || error)
234
+ };
235
+ }
236
+ }
237
+ async transcribe(input, { lang = "auto", signal } = {}) {
238
+ const bytes = validateMonoPcm16Wav(input, { maxBytes: this.maxBytes }), form = new FormData();
239
+ form.append("file", new Blob([bytes], { type: "audio/wav" }), "utterance.wav");
240
+ form.append("response_format", "json");
241
+ form.append("language", lang.startsWith("pt") ? "pt" : lang.startsWith("en") ? "en" : "auto");
242
+ const config = await this.getConfig();
243
+ return this.request(
244
+ new URL(config.url),
245
+ { method: "POST", body: form, signal },
246
+ config,
247
+ async (response) => {
248
+ if (!response.ok)
249
+ throw new Error("Whisper HTTP transcription failed (" + response.status + ").");
250
+ const json = await response.json();
251
+ return { text: clean(json?.text) };
252
+ }
253
+ );
254
+ }
255
+ };
492
256
 
493
- // src/engines/qwen-http-host.ts
257
+ // src/modules/core/qwen/QwenHttpHost.ts
494
258
  var clean2 = (value) => String(value ?? "").trim();
495
259
  function resolveQwenBaseUrl(value = process.env.DSH_LIVE_VOICE_QWEN_URL || "http://127.0.0.1:8080/") {
496
260
  const url = new URL(value);
@@ -509,7 +273,7 @@ function validateQwenConfig(value) {
509
273
  throw new Error("Request timeout must be an integer between 1000 and 600000 ms.");
510
274
  return { baseUrl: baseUrl.href, timeoutMs: value.timeoutMs };
511
275
  }
512
- function createQwenConfigStore(path = join3(homedir2(), ".dsh", "dsh-live-voice-qwen.json")) {
276
+ function createQwenConfigStore(path = join2(homedir2(), ".dsh", "dsh-live-voice-qwen.json")) {
513
277
  return {
514
278
  async load() {
515
279
  try {
@@ -606,109 +370,440 @@ var QwenHttpHost = class {
606
370
  }));
607
371
  return { ...response, ominix: response.body?.service === "ominix-api" };
608
372
  }
609
- async capability(signal, configuration, kind = "both") {
373
+ async capability(signal, configuration, kind = "both") {
374
+ try {
375
+ const config = configuration ? validateQwenConfig(configuration) : await this.getConfig();
376
+ const health = await this.serverInfo(signal, config);
377
+ let models = health.body?.models;
378
+ if (health.ominix && health.ok) {
379
+ const status = await this.request(
380
+ "v1/models/status",
381
+ { signal },
382
+ config,
383
+ async (response) => ({
384
+ ok: response.ok,
385
+ status: response.status,
386
+ body: response.ok ? await response.json() : null
387
+ })
388
+ );
389
+ models = {
390
+ asr: status.body?.models?.asr === "qwen3-asr",
391
+ tts: status.body?.models?.qwen3_tts === "customvoice"
392
+ };
393
+ }
394
+ const ready = kind === "asr" ? models?.asr === true : kind === "tts" ? models?.tts === true : models?.asr === true && models?.tts === true;
395
+ return health.ok && ready ? { supported: true, local: true, location: "host", streaming: false, models } : {
396
+ supported: false,
397
+ local: true,
398
+ location: "host",
399
+ streaming: false,
400
+ reason: `Qwen health check did not report ${kind === "both" ? "both ASR and TTS" : kind.toUpperCase()} ready (${health.status}).`
401
+ };
402
+ } catch (error) {
403
+ return {
404
+ supported: false,
405
+ local: true,
406
+ location: "host",
407
+ streaming: false,
408
+ reason: "Qwen speech server is unreachable: " + (error?.message || error)
409
+ };
410
+ }
411
+ }
412
+ async transcribe(input, { lang = "pt-BR", signal } = {}) {
413
+ const bytes = validateMonoPcm16Wav(input, { maxBytes: this.maxBytes }), resolved = language(lang), health = await this.serverInfo(signal);
414
+ let body, headers;
415
+ if (health.ominix) {
416
+ headers = { "content-type": "application/json" };
417
+ body = JSON.stringify({
418
+ file: Buffer.from(bytes).toString("base64"),
419
+ language: resolved,
420
+ response_format: "json"
421
+ });
422
+ } else {
423
+ const form = new FormData();
424
+ form.append("file", new Blob([bytes], { type: "audio/wav" }), "utterance.wav");
425
+ form.append("response_format", "json");
426
+ if (resolved) form.append("language", resolved);
427
+ body = form;
428
+ }
429
+ return this.request(
430
+ "v1/audio/transcriptions",
431
+ { method: "POST", headers, body, signal },
432
+ void 0,
433
+ async (response) => {
434
+ if (!response.ok) throw new Error("Qwen transcription failed (" + response.status + ").");
435
+ const json = await response.json();
436
+ return { text: clean2(json?.text) };
437
+ }
438
+ );
439
+ }
440
+ async synthesize(text, { lang = "pt-BR", signal, voice = defaultQwenVoice } = {}) {
441
+ if (typeof text !== "string" || !text.trim() || text.length > 1e5 || text.includes("\0"))
442
+ throw new Error("Speech text must contain 1\u2013100000 characters without NUL.");
443
+ if (!isQwenVoice(voice)) throw new Error("Unsupported Qwen voice.");
444
+ return this.request(
445
+ "v1/audio/speech",
446
+ {
447
+ method: "POST",
448
+ headers: { "content-type": "application/json" },
449
+ body: JSON.stringify({
450
+ model: "qwen3-tts",
451
+ input: text,
452
+ voice,
453
+ language: language(lang) || "portuguese",
454
+ response_format: "wav"
455
+ }),
456
+ signal
457
+ },
458
+ void 0,
459
+ async (response) => {
460
+ if (!response.ok) throw new Error("Qwen synthesis failed (" + response.status + ").");
461
+ const length = Number(response.headers.get("content-length") || 0);
462
+ if (length > this.maxSpeechBytes) throw new Error("Qwen speech response is too large.");
463
+ const bytes = await response.arrayBuffer();
464
+ if (bytes.byteLength < 44 || bytes.byteLength > this.maxSpeechBytes)
465
+ throw new Error("Qwen returned invalid or oversized speech audio.");
466
+ return bytes;
467
+ }
468
+ );
469
+ }
470
+ };
471
+
472
+ // src/modules/speak/engines/audio/M4aAacTranscoder.ts
473
+ import { execFile as nodeExecFile } from "node:child_process";
474
+ import * as nodeFs from "node:fs/promises";
475
+ import { tmpdir } from "node:os";
476
+ import { join as join3 } from "node:path";
477
+ import { promisify } from "node:util";
478
+ var execFile = promisify(nodeExecFile);
479
+ var aborted = (reason) => Object.assign(new Error("Speech audio conversion was cancelled."), {
480
+ name: "AbortError",
481
+ cause: reason
482
+ });
483
+ async function wavToM4aAac(input, { fs = nodeFs, run = execFile, tempRoot = tmpdir(), signal, bitrate = 64e3 } = {}) {
484
+ if (!input || !Number.isSafeInteger(input.byteLength) || input.byteLength < 44)
485
+ throw new TypeError("A valid WAV audio buffer is required.");
486
+ if (!Number.isFinite(bitrate) || bitrate < 16e3 || bitrate > 256e3)
487
+ throw new TypeError("AAC bitrate must be between 16000 and 256000 bps.");
488
+ if (signal?.aborted) throw aborted(signal.reason);
489
+ let directory, result, error;
490
+ try {
491
+ directory = await fs.mkdtemp(join3(tempRoot, "dsh-live-voice-aac-"));
492
+ await fs.chmod(directory, 448);
493
+ const source = join3(directory, "speech.wav"), target = join3(directory, "speech.m4a");
494
+ await fs.writeFile(source, input, { mode: 384, flag: "wx" });
495
+ if (signal?.aborted) throw aborted(signal.reason);
496
+ await run(
497
+ "/usr/bin/afconvert",
498
+ ["-f", "m4af", "-d", "aac", "-b", String(bitrate), source, target],
499
+ {
500
+ shell: false,
501
+ signal
502
+ }
503
+ );
504
+ if (signal?.aborted) throw aborted(signal.reason);
505
+ result = await fs.readFile(target);
506
+ if (result.byteLength < 16 || Buffer.from(result).subarray(4, 8).toString("ascii") !== "ftyp")
507
+ throw new Error("Audio conversion produced invalid M4A output.");
508
+ } catch (caught) {
509
+ error = signal?.aborted && caught?.name !== "AbortError" ? aborted(signal.reason) : caught;
510
+ } finally {
511
+ if (directory) {
512
+ try {
513
+ await fs.rm(directory, { recursive: true, force: true });
514
+ } catch (cleanupError) {
515
+ error ||= cleanupError;
516
+ }
517
+ }
518
+ }
519
+ if (error) throw error;
520
+ return result;
521
+ }
522
+
523
+ // src/modules/speak/engines/say/SaySpeakingEngine.ts
524
+ import { spawn as nodeSpawn } from "node:child_process";
525
+ import * as nodeFs2 from "node:fs/promises";
526
+ import { constants } from "node:fs";
527
+ import { tmpdir as tmpdir2 } from "node:os";
528
+ import { join as join4 } from "node:path";
529
+ function failure(message, code, cause) {
530
+ return Object.assign(new Error(message, cause === void 0 ? void 0 : { cause }), { code });
531
+ }
532
+ function aborted2(reason) {
533
+ return Object.assign(
534
+ new Error("Speech cancelled", reason === void 0 ? void 0 : { cause: reason }),
535
+ {
536
+ name: "AbortError",
537
+ code: "ABORT_ERR"
538
+ }
539
+ );
540
+ }
541
+ var SayEngine = class {
542
+ #spawn;
543
+ #fs;
544
+ #platform;
545
+ #tempRoot;
546
+ #killAfterMs;
547
+ #closeAfterMs;
548
+ #tail = Promise.resolve();
549
+ #requests = /* @__PURE__ */ new Set();
550
+ #active = null;
551
+ #unclosed = null;
552
+ #state = "idle";
553
+ #lastError = null;
554
+ constructor({
555
+ spawn = nodeSpawn,
556
+ fs = nodeFs2,
557
+ platform = process.platform,
558
+ tempRoot = tmpdir2(),
559
+ killAfterMs = 250,
560
+ closeAfterMs = 1e3
561
+ } = {}) {
562
+ for (const value of [killAfterMs, closeAfterMs]) {
563
+ if (!Number.isFinite(value) || value < 0 || value > 2147483647) {
564
+ throw new TypeError("Cancellation deadlines must be finite nonnegative milliseconds");
565
+ }
566
+ }
567
+ this.#spawn = spawn;
568
+ this.#fs = fs;
569
+ this.#platform = platform;
570
+ this.#tempRoot = tempRoot;
571
+ this.#killAfterMs = killAfterMs;
572
+ this.#closeAfterMs = closeAfterMs;
573
+ }
574
+ get state() {
575
+ return this.#state;
576
+ }
577
+ get lastError() {
578
+ return this.#lastError;
579
+ }
580
+ /** Checks the host executable, not browser support or installed voices. */
581
+ async getCapabilities() {
582
+ if (this.#platform !== "darwin") {
583
+ return {
584
+ supported: false,
585
+ pause: false,
586
+ resume: false,
587
+ audioFormat: "audio/wav",
588
+ reason: "unsupported-platform"
589
+ };
590
+ }
591
+ try {
592
+ await this.#fs.access("/usr/bin/say", constants.X_OK);
593
+ return {
594
+ supported: true,
595
+ pause: false,
596
+ resume: false,
597
+ audioFormat: "audio/wav",
598
+ reason: null
599
+ };
600
+ } catch {
601
+ return {
602
+ supported: false,
603
+ pause: false,
604
+ resume: false,
605
+ audioFormat: "audio/wav",
606
+ reason: "executable-unavailable"
607
+ };
608
+ }
609
+ }
610
+ speak(text, { voice, rate, signal } = {}) {
611
+ if (typeof text !== "string") return Promise.reject(new TypeError("text must be a string"));
612
+ if (voice !== void 0 && (typeof voice !== "string" || !voice.trim() || voice.includes("\0"))) {
613
+ return Promise.reject(new TypeError("voice must be a nonempty string without NUL"));
614
+ }
615
+ if (rate !== void 0 && (!Number.isFinite(rate) || rate <= 0)) {
616
+ return Promise.reject(new TypeError("rate must be a positive finite number"));
617
+ }
618
+ if (signal !== void 0 && (signal === null || typeof signal.addEventListener !== "function" || typeof signal.removeEventListener !== "function" || typeof signal.aborted !== "boolean")) {
619
+ return Promise.reject(new TypeError("signal must be an AbortSignal"));
620
+ }
621
+ for (const request2 of this.#requests) this.#cancel(request2);
622
+ const request = { cancelled: false, reason: void 0, cancelChild: null };
623
+ const onAbort = () => this.#cancel(request, signal.reason);
624
+ this.#requests.add(request);
625
+ signal?.addEventListener("abort", onAbort, { once: true });
626
+ if (signal?.aborted) onAbort();
627
+ const result = this.#tail.then(() => this.#run(request, text, voice, rate));
628
+ const settled = result.finally(() => {
629
+ signal?.removeEventListener("abort", onAbort);
630
+ this.#requests.delete(request);
631
+ });
632
+ this.#tail = settled.catch(() => {
633
+ });
634
+ return settled;
635
+ }
636
+ /** Resolves after queued requests and cleanup; rejects teardown failures. */
637
+ async stop() {
638
+ for (const request of this.#requests) this.#cancel(request);
639
+ const pending = this.#tail;
640
+ await pending;
641
+ if (this.#unclosed || this.#state === "error") throw this.#lastError;
642
+ }
643
+ pause() {
644
+ return this.#control("speaking", "paused", "SIGSTOP");
645
+ }
646
+ resume() {
647
+ return this.#control("paused", "speaking", "SIGCONT");
648
+ }
649
+ #control(from, to, signal) {
650
+ if (this.#platform !== "darwin" || this.#state !== from || !this.#active?.child || this.#active.cancelled)
651
+ return false;
610
652
  try {
611
- const config = configuration ? validateQwenConfig(configuration) : await this.getConfig();
612
- const health = await this.serverInfo(signal, config);
613
- let models = health.body?.models;
614
- if (health.ominix && health.ok) {
615
- const status = await this.request(
616
- "v1/models/status",
617
- { signal },
618
- config,
619
- async (response) => ({
620
- ok: response.ok,
621
- status: response.status,
622
- body: response.ok ? await response.json() : null
623
- })
624
- );
625
- models = {
626
- asr: status.body?.models?.asr === "qwen3-asr",
627
- tts: status.body?.models?.qwen3_tts === "customvoice"
628
- };
629
- }
630
- const ready = kind === "asr" ? models?.asr === true : kind === "tts" ? models?.tts === true : models?.asr === true && models?.tts === true;
631
- return health.ok && ready ? { supported: true, local: true, location: "host", streaming: false, models } : {
632
- supported: false,
633
- local: true,
634
- location: "host",
635
- streaming: false,
636
- reason: `Qwen health check did not report ${kind === "both" ? "both ASR and TTS" : kind.toUpperCase()} ready (${health.status}).`
637
- };
653
+ if (!this.#active.child.kill(signal))
654
+ throw failure("Unable to signal speech process", "SAY_SIGNAL_FAILED");
655
+ this.#state = to;
656
+ return true;
638
657
  } catch (error) {
639
- return {
640
- supported: false,
641
- local: true,
642
- location: "host",
643
- streaming: false,
644
- reason: "Qwen speech server is unreachable: " + (error?.message || error)
645
- };
658
+ this.#lastError = error;
659
+ return false;
646
660
  }
647
661
  }
648
- async transcribe(input, { lang = "pt-BR", signal } = {}) {
649
- const bytes = validateMonoPcm16Wav(input, { maxBytes: this.maxBytes }), resolved = language(lang), health = await this.serverInfo(signal);
650
- let body, headers;
651
- if (health.ominix) {
652
- headers = { "content-type": "application/json" };
653
- body = JSON.stringify({
654
- file: Buffer.from(bytes).toString("base64"),
655
- language: resolved,
656
- response_format: "json"
657
- });
658
- } else {
659
- const form = new FormData();
660
- form.append("file", new Blob([bytes], { type: "audio/wav" }), "utterance.wav");
661
- form.append("response_format", "json");
662
- if (resolved) form.append("language", resolved);
663
- body = form;
664
- }
665
- return this.request(
666
- "v1/audio/transcriptions",
667
- { method: "POST", headers, body, signal },
668
- void 0,
669
- async (response) => {
670
- if (!response.ok) throw new Error("Qwen transcription failed (" + response.status + ").");
671
- const json = await response.json();
672
- return { text: clean2(json?.text) };
673
- }
674
- );
662
+ #cancel(request, reason) {
663
+ if (request.cancelled) return;
664
+ request.cancelled = true;
665
+ request.reason = reason;
666
+ request.cancelChild?.();
675
667
  }
676
- async synthesize(text, { lang = "pt-BR", signal, voice = defaultQwenVoice } = {}) {
677
- if (typeof text !== "string" || !text.trim() || text.length > 1e5 || text.includes("\0"))
678
- throw new Error("Speech text must contain 1\u2013100000 characters without NUL.");
679
- if (!isQwenVoice(voice)) throw new Error("Unsupported Qwen voice.");
680
- return this.request(
681
- "v1/audio/speech",
682
- {
683
- method: "POST",
684
- headers: { "content-type": "application/json" },
685
- body: JSON.stringify({
686
- model: "qwen3-tts",
687
- input: text,
688
- voice,
689
- language: language(lang) || "portuguese",
690
- response_format: "wav"
691
- }),
692
- signal
693
- },
694
- void 0,
695
- async (response) => {
696
- if (!response.ok) throw new Error("Qwen synthesis failed (" + response.status + ").");
697
- const length = Number(response.headers.get("content-length") || 0);
698
- if (length > this.maxSpeechBytes) throw new Error("Qwen speech response is too large.");
699
- const bytes = await response.arrayBuffer();
700
- if (bytes.byteLength < 44 || bytes.byteLength > this.maxSpeechBytes)
701
- throw new Error("Qwen returned invalid or oversized speech audio.");
702
- return bytes;
668
+ async #run(request, text, voice, rate) {
669
+ let directory;
670
+ let error;
671
+ let result;
672
+ const check = () => {
673
+ if (request.cancelled) throw aborted2(request.reason);
674
+ };
675
+ try {
676
+ check();
677
+ if (this.#unclosed)
678
+ throw failure("Previous speech process has not closed", "SAY_PROCESS_UNCLOSED");
679
+ this.#active = request;
680
+ this.#state = "preparing";
681
+ this.#lastError = null;
682
+ const capability = await this.getCapabilities();
683
+ check();
684
+ if (!capability.supported) throw failure("macOS say is unavailable", "SAY_UNAVAILABLE");
685
+ directory = await this.#fs.mkdtemp(join4(this.#tempRoot, "dsh-live-voice-say-"));
686
+ check();
687
+ await this.#fs.chmod(directory, 448);
688
+ const file = join4(directory, "speech.txt");
689
+ const output = join4(directory, "speech.wav");
690
+ await this.#fs.writeFile(file, text, { encoding: "utf8", mode: 384, flag: "wx" });
691
+ await this.#fs.chmod(file, 384);
692
+ check();
693
+ const args = ["-f", file, "-o", output, "--file-format=WAVE", "--data-format=LEI16@22050"];
694
+ if (voice !== void 0) args.push("-v", voice);
695
+ if (rate !== void 0) args.push("-r", String(rate));
696
+ const child = this.#spawn("/usr/bin/say", args, { shell: false, stdio: "ignore" });
697
+ request.child = child;
698
+ this.#state = "speaking";
699
+ await this.#waitForClose(request, child);
700
+ check();
701
+ result = await this.#fs.readFile(output);
702
+ } catch (caught) {
703
+ error = caught;
704
+ } finally {
705
+ request.cancelChild = null;
706
+ if (directory) {
707
+ try {
708
+ await this.#fs.rm(directory, { recursive: true, force: true });
709
+ } catch (cleanupError) {
710
+ error = failure(
711
+ "Speech temporary-file cleanup failed",
712
+ "SAY_CLEANUP_FAILED",
713
+ error ? new AggregateError([error, cleanupError]) : cleanupError
714
+ );
715
+ }
703
716
  }
704
- );
717
+ if (!error && request.cancelled) error = aborted2(request.reason);
718
+ if (this.#active === request) this.#active = null;
719
+ if (error && error.name !== "AbortError") {
720
+ this.#state = "error";
721
+ this.#lastError = error;
722
+ } else if (!this.#unclosed && this.#state !== "error") this.#state = "idle";
723
+ }
724
+ if (error) throw error;
725
+ return result;
726
+ }
727
+ #waitForClose(request, child) {
728
+ return new Promise((resolve, reject) => {
729
+ let closed = false;
730
+ let processError;
731
+ let killTimer;
732
+ let closeTimer;
733
+ const send = (signal) => {
734
+ try {
735
+ if (!child.kill(signal))
736
+ processError ??= failure("Unable to signal speech process", "SAY_SIGNAL_FAILED");
737
+ } catch (error) {
738
+ processError ??= error;
739
+ }
740
+ };
741
+ const onError = (error) => {
742
+ processError ??= error;
743
+ };
744
+ child.on("error", onError);
745
+ child.once("close", (code, signal) => {
746
+ closed = true;
747
+ clearTimeout(killTimer);
748
+ clearTimeout(closeTimer);
749
+ child.removeListener("error", onError);
750
+ if (this.#unclosed === child) this.#unclosed = null;
751
+ if (request.cancelled) reject(aborted2(request.reason));
752
+ else if (processError) reject(processError);
753
+ else if (code !== 0)
754
+ reject(
755
+ failure("Speech process exited unsuccessfully", "SAY_EXIT_FAILED", { code, signal })
756
+ );
757
+ else resolve();
758
+ });
759
+ request.cancelChild = () => {
760
+ if (closed) return;
761
+ const paused = this.#state === "paused";
762
+ this.#state = "stopping";
763
+ if (paused) send("SIGCONT");
764
+ send("SIGTERM");
765
+ if (closed) return;
766
+ killTimer = setTimeout(() => {
767
+ send("SIGKILL");
768
+ if (closed) return;
769
+ closeTimer = setTimeout(() => {
770
+ if (closed) return;
771
+ this.#unclosed = child;
772
+ reject(
773
+ failure(
774
+ "Speech process did not close after cancellation",
775
+ "SAY_STOP_TIMEOUT",
776
+ processError
777
+ )
778
+ );
779
+ }, this.#closeAfterMs);
780
+ }, this.#killAfterMs);
781
+ };
782
+ if (request.cancelled) request.cancelChild();
783
+ });
705
784
  }
706
785
  };
707
786
 
708
- // src/server.ts
787
+ // src/app/server/apply.ts
709
788
  var name = "dsh-live-voice";
710
- var inject = ["connection"];
789
+ var inject = ["connection", "systemPrompt"];
711
790
  var SAY_CHANNEL = "/api/dsh-live-voice";
791
+ var VOICE_CONTEXT_PATH = SAY_CHANNEL + "/voice-context";
792
+ function createVoiceContextStore() {
793
+ const entries = /* @__PURE__ */ new Map();
794
+ return {
795
+ get(sessionId) {
796
+ return entries.get(sessionId) || "";
797
+ },
798
+ set(sessionId, text) {
799
+ if (text) entries.set(sessionId, text);
800
+ else entries.delete(sessionId);
801
+ },
802
+ clear() {
803
+ entries.clear();
804
+ }
805
+ };
806
+ }
712
807
  var ok = (value) => ({ ok: true, value });
713
808
  var fail = (code, message) => ({ ok: false, error: { code, message, details: {} } });
714
809
  var identity = (value) => typeof value === "string" && /^[a-zA-Z0-9_-]{1,128}$/.test(value);
@@ -782,15 +877,105 @@ function createSayHost({ engine = new SayEngine() } = {}) {
782
877
  await engine.stop();
783
878
  active = null;
784
879
  }
785
- return { handle, dispose };
880
+ return { handle, dispose, engine };
786
881
  }
787
882
  function apply(ctx, {
788
883
  whisperStore = createWhisperConfigStore(),
789
884
  whisperFetch = globalThis.fetch,
790
885
  qwenStore = createQwenConfigStore(),
791
- qwenFetch = globalThis.fetch
886
+ qwenFetch = globalThis.fetch,
887
+ voiceContextStore = createVoiceContextStore(),
888
+ createSayEngine = () => new SayEngine(),
889
+ encodeHostSpeech = wavToM4aAac
792
890
  } = {}) {
891
+ ctx.systemPrompt.variable(
892
+ "live_voice_context",
893
+ (assemblyContext) => voiceContextStore.get(String(assemblyContext.agent?.sessionId || ""))
894
+ );
895
+ ctx.systemPrompt.context({
896
+ name: "dsh-live-voice:spoken-output",
897
+ order: 700,
898
+ text: "{{live_voice_context}}"
899
+ });
900
+ const disposeVoiceContext = ctx.connection.fetch.register({
901
+ path: VOICE_CONTEXT_PATH,
902
+ methods: ["PUT"],
903
+ requestBody: "buffered",
904
+ fetch: async (request) => {
905
+ try {
906
+ const body = await request.json();
907
+ if (!identity(body?.sessionId) || typeof body?.context !== "string" || body.context.length > 4e3 || body.context.includes("\0"))
908
+ return Response.json(
909
+ fail("invalid-request", "A valid session ID and voice context are required."),
910
+ { status: 400 }
911
+ );
912
+ voiceContextStore.set(body.sessionId, body.active === true ? body.context.trim() : "");
913
+ return Response.json(ok({ active: body.active === true && Boolean(body.context.trim()) }));
914
+ } catch {
915
+ return Response.json(fail("invalid-request", "Invalid voice context request."), {
916
+ status: 400
917
+ });
918
+ }
919
+ }
920
+ });
921
+ ctx.effect(
922
+ () => () => {
923
+ disposeVoiceContext();
924
+ voiceContextStore.clear();
925
+ },
926
+ "dsh-live-voice: remove voice context"
927
+ );
793
928
  const host = createSayHost();
929
+ const saySyntheses = /* @__PURE__ */ new Set();
930
+ const sayCapability = ctx.connection.fetch.register({
931
+ path: SAY_CHANNEL + "/say/capabilities",
932
+ methods: ["GET"],
933
+ requestBody: "buffered",
934
+ fetch: async () => Response.json(ok({ ...await host.engine.getCapabilities(), audioFormat: "audio/mp4" }))
935
+ });
936
+ const saySpeech = ctx.connection.fetch.register({
937
+ path: SAY_CHANNEL + "/say/speech",
938
+ methods: ["POST"],
939
+ requestBody: "buffered",
940
+ fetch: async (request) => {
941
+ try {
942
+ const body = await request.json();
943
+ const engine = createSayEngine();
944
+ saySyntheses.add(engine);
945
+ try {
946
+ const wav = await engine.speak(body?.text, {
947
+ voice: body?.voice,
948
+ rate: body?.rate,
949
+ signal: request.signal
950
+ });
951
+ const bytes = await encodeHostSpeech(wav, { signal: request.signal });
952
+ return new Response(bytes, {
953
+ status: 200,
954
+ headers: { "content-type": "audio/mp4", "cache-control": "no-store" }
955
+ });
956
+ } finally {
957
+ saySyntheses.delete(engine);
958
+ }
959
+ } catch (error) {
960
+ if (error?.name === "AbortError")
961
+ return Response.json(fail("cancelled", "Speech synthesis was cancelled."), {
962
+ status: 499
963
+ });
964
+ return Response.json(fail("synthesis-failed", "Local speech synthesis failed."), {
965
+ status: 502
966
+ });
967
+ }
968
+ }
969
+ });
970
+ ctx.effect(
971
+ () => async () => {
972
+ sayCapability();
973
+ saySpeech();
974
+ await Promise.allSettled([...saySyntheses].map((engine) => engine.stop()));
975
+ saySyntheses.clear();
976
+ },
977
+ "dsh-live-voice: remove say audio routes"
978
+ );
794
979
  const whisper = new WhisperHttpHost({ store: whisperStore, fetchImpl: whisperFetch });
795
980
  const qwen = new QwenHttpHost({ store: qwenStore, fetchImpl: qwenFetch });
796
981
  for (const endpoint of ["config", "test"]) {
@@ -967,14 +1152,15 @@ function apply(ctx, {
967
1152
  fetch: async (request) => {
968
1153
  try {
969
1154
  const body = await request.json();
970
- const bytes = await qwen.synthesize(body?.text, {
1155
+ const wav = await qwen.synthesize(body?.text, {
971
1156
  lang: body?.lang || "pt-BR",
972
1157
  voice: body?.voice,
973
1158
  signal: request.signal
974
1159
  });
1160
+ const bytes = await encodeHostSpeech(wav, { signal: request.signal });
975
1161
  return new Response(bytes, {
976
1162
  status: 200,
977
- headers: { "content-type": "audio/wav", "cache-control": "no-store" }
1163
+ headers: { "content-type": "audio/mp4", "cache-control": "no-store" }
978
1164
  });
979
1165
  } catch (error) {
980
1166
  if (error?.name === "AbortError")
@@ -1017,8 +1203,10 @@ function apply(ctx, {
1017
1203
  }
1018
1204
  export {
1019
1205
  SAY_CHANNEL,
1206
+ VOICE_CONTEXT_PATH,
1020
1207
  apply,
1021
1208
  createSayHost,
1209
+ createVoiceContextStore,
1022
1210
  inject,
1023
1211
  name
1024
1212
  };