dsh-speak 1.8.0 → 1.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,614 +1,702 @@
1
- // speech-hook.js — DSH web adapter: voice-announce assistant activity
2
- // ==============================================================================
3
- // Listens to the session event stream (session/event), extracts the final reply
4
- // text, and hands it to the speech engine (engine/speak.ps1 on Windows,
5
- // engine/speak.sh on macOS) through a hidden, non-blocking child process.
6
- //
7
- // Since 1.7.0 this is the merged host of the original dsh-speak behavior and
8
- // victorwads' PR #2 (turn-level replay + host FIFO speech queue + WebSocket
9
- // state sync + native Speak settings page):
10
- //
11
- // * a host-owned FIFO speech queue: only one native speech process runs at a
12
- // time; queued items continue automatically when the current one finishes
13
- // * every eligible item (final reply, approvals, questions, optional events,
14
- // manual replay) is enqueued, so the WebSocket state (which message is
15
- // speaking, queue length) is always truthful — even for automatic replies
16
- // * `queueAllMessages` (default off) switches between two automatic modes:
17
- // - off (default): final replies are throttled/merged as before, plus the
18
- // optional event announcements; tool calls cancel pending narration
19
- // - on: every assistant/message is enqueued immediately as it arrives
20
- // * a `/dsh-speak/control` POST route (play/stop/status) and a
21
- // `/dsh-speak/ws` WebSocket publish the authoritative speech state
22
- // * a `dsh-speak` settings namespace registered through the settings SERVICE
23
- // (`ctx.inject(['settings'])` → `settings.register`); schema defaults →
24
- // patch config → UI user layer
25
- // * `enabled` master switch: when off, nothing is ever enqueued (no sound)
26
- //
27
- // Trigger semantics:
28
- // * assistant/message with a `text` block is announced (reasoning / tool_use
29
- // blocks are skipped)
30
- // * a tool/call to `ask_user_question` announces the parsed question; other
31
- // tool calls cancel the pending throttled announcement (default mode)
32
- // * `approval/asked` is announced immediately (reason, else a fixed prompt)
33
- // * optional events (turn/end, command/done, goal/change, tool/result errors,
34
- // todo/write) are announced when their toggle is on (default off)
35
- //
36
- // Configuration — prefer the Web UI (Settings → dsh-speak settings) or the
37
- // profile patch `config` block (see README.md). All keys resolve as
38
- // schema default → patch config → UI user layer.
39
- 'use strict'
40
-
41
- const { spawn } = require('child_process')
42
- const { createRequire } = require('module')
43
- const { WebSocketServer, WebSocket } = require('ws')
44
- const fs = require('fs')
45
- const os = require('os')
46
- const path = require('path')
47
-
48
- const LOG = path.join(os.tmpdir(), 'dsh-speech-hook.log')
49
- function log(...args) {
50
- try { fs.appendFileSync(LOG, `[${new Date().toISOString()}] ${args.join(' ')}\n`) } catch (e) { /* ignore */ }
51
- }
52
-
53
- const ENGINE_NAME = process.platform === 'darwin' ? 'speak.sh' : 'speak.ps1'
54
- // Settings namespace of this plugin (lowercase kebab-case; must match the
55
- // browser card's namespace in client/client.js).
56
- const SETTINGS_NS = 'dsh-speak'
57
-
58
- /**
59
- * Locate the engine script:
60
- * 1. explicit override (config `engine`)
61
- * 2. <this package>/engine/<speak.ps1|speak.sh> — repo checkout or profile install
62
- * 3. legacy file-copy location (~/.dsh/hooks/<speak.ps1|speak.sh>)
63
- */
64
- function resolveEngine(override) {
65
- if (override) return override
66
- const bundled = path.join(__dirname, '..', '..', 'engine', ENGINE_NAME)
67
- if (fs.existsSync(bundled)) return bundled
68
- return path.join(os.homedir(), '.dsh', 'hooks', ENGINE_NAME)
69
- }
70
-
71
- // Platform-aware defaults: macOS `say` has no per-utterance ceiling, so
72
- // `maxChars` defaults to 0 (unlimited) there; Windows keeps the safe 300.
73
- const DEFAULT_MAX_CHARS = process.platform === 'darwin' ? 0 : 300
74
-
75
- // ---------------------------------------------------------------------------
76
- // Settings namespace (best-effort; registered through the `settings` service)
77
- // ---------------------------------------------------------------------------
78
- // The schema mirrors every config key. Values resolve as:
79
- // schema default → patch `config` (base) → user settings layer (the UI).
80
- const SCHEMA_DEFAULTS = {
81
- enabled: true,
82
- automaticSpeech: true,
83
- cleanMarkdownFormatting: true,
84
- readInlineCode: true,
85
- codeBlocks: 'smart',
86
- codeBlockMaxChars: 300,
87
- codeBlockReplacementText: 'You can see the code in our history.',
88
- queueAllMessages: false,
89
- throttleMs: 1500,
90
- replayFullRead: false,
91
- engine: '',
92
- announceApprovals: true,
93
- announceQuestions: true,
94
- stripApprovalPrefix: true,
95
- questionGapMs: 2000,
96
- longTextMode: 'message',
97
- longTextMessage: '本次播报内容较长,请自行阅读。',
98
- maxChars: DEFAULT_MAX_CHARS,
99
- volume: 50,
100
- rate: 0,
101
- announceTurnEnd: false,
102
- announceCommandDone: false,
103
- announceGoalChange: false,
104
- announceToolErrors: false,
105
- announceTodoWrite: false,
106
- }
107
-
108
- /**
109
- * Resolve the raw settings value into the mutable `cfg` the queue reads.
110
- * Kept as a pure function so both the initial apply and settings onChange use
111
- * the same normalization (engine re-resolution, platform maxChars default).
112
- */
113
- function resolveConfig(value) {
114
- value = value || {}
115
- return {
116
- enabled: value.enabled !== false,
117
- automaticSpeech: value.automaticSpeech !== false,
118
- cleanMarkdownFormatting: value.cleanMarkdownFormatting !== false,
119
- readInlineCode: value.readInlineCode !== false,
120
- codeBlocks: ['all', 'smart', 'replace'].includes(value.codeBlocks) ? value.codeBlocks : 'smart',
121
- codeBlockMaxChars: Number(value.codeBlockMaxChars != null ? value.codeBlockMaxChars : 300),
122
- codeBlockReplacementText: String(value.codeBlockReplacementText || 'You can see the code in our history.'),
123
- queueAllMessages: value.queueAllMessages === true,
124
- throttleMs: Number(value.throttleMs != null ? value.throttleMs : 1500) || 1500,
125
- replayFullRead: value.replayFullRead === true,
126
- engine: resolveEngine(value.engine || ''),
127
- announceApprovals: value.announceApprovals !== false,
128
- announceQuestions: value.announceQuestions !== false,
129
- stripApprovalPrefix: value.stripApprovalPrefix !== false,
130
- questionGapMs: Math.max(0, Number(value.questionGapMs != null ? value.questionGapMs : 2000)) || 0,
131
- longTextMode: value.longTextMode === 'heading' ? 'heading' : 'message',
132
- longTextMessage: String(value.longTextMessage || SCHEMA_DEFAULTS.longTextMessage),
133
- maxChars: Number(value.maxChars != null ? value.maxChars : DEFAULT_MAX_CHARS) || 0,
134
- volume: Number(value.volume != null ? value.volume : 50) || 50,
135
- rate: Number(value.rate != null ? value.rate : 0) || 0,
136
- announceTurnEnd: value.announceTurnEnd === true,
137
- announceCommandDone: value.announceCommandDone === true,
138
- announceGoalChange: value.announceGoalChange === true,
139
- announceToolErrors: value.announceToolErrors === true,
140
- announceTodoWrite: value.announceTodoWrite === true,
141
- }
142
- }
143
-
144
- /**
145
- * Resolve a module specifier from the plugin's own location first, then from
146
- * the booted profile tree. `@deepseek-ai/schemastery` is a peer of this package
147
- * and lives beside it after an npm/pnpm install; the profile-tree fallback
148
- * covers the file:// install used by install.ps1 and repo checkouts.
149
- * @returns the module, or null when neither base resolves it.
150
- */
151
- function requirePeer(ctx, spec) {
152
- const bases = [__filename, ctx.baseUrl].filter(Boolean)
153
- for (const base of bases) {
154
- try { return createRequire(base)(spec) } catch (e) { /* try the next base */ }
155
- }
156
- return null
157
- }
158
-
159
- /**
160
- * Build the settings schema + entry for the settings namespace. Best-effort:
161
- * any failure (missing peer packages) returns null and the plugin keeps the
162
- * patch config.
163
- */
164
- function buildSettingsNamespace(ctx, patch) {
165
- try {
166
- const z = requirePeer(ctx, '@deepseek-ai/schemastery')
167
- if (!z) throw new Error('@deepseek-ai/schemastery 不可解析')
168
- const schema = z.object({
169
- enabled: z.boolean().default(true),
170
- automaticSpeech: z.boolean().default(true),
171
- cleanMarkdownFormatting: z.boolean().default(true),
172
- readInlineCode: z.boolean().default(true),
173
- codeBlocks: z.union(['all', 'smart', 'replace']).default('smart'),
174
- codeBlockMaxChars: z.natural().default(300),
175
- codeBlockReplacementText: z.string().default('You can see the code in our history.'),
176
- queueAllMessages: z.boolean().default(false),
177
- throttleMs: z.natural().default(1500),
178
- replayFullRead: z.boolean().default(false),
179
- engine: z.string().default(''),
180
- announceApprovals: z.boolean().default(true),
181
- announceQuestions: z.boolean().default(true),
182
- stripApprovalPrefix: z.boolean().default(true),
183
- questionGapMs: z.natural().default(2000),
184
- longTextMode: z.union(['message', 'heading']).default('message'),
185
- longTextMessage: z.string().default('本次播报内容较长,请自行阅读。'),
186
- maxChars: z.natural().default(DEFAULT_MAX_CHARS),
187
- volume: z.natural().default(50),
188
- rate: z.number().default(0),
189
- announceTurnEnd: z.boolean().default(false),
190
- announceCommandDone: z.boolean().default(false),
191
- announceGoalChange: z.boolean().default(false),
192
- announceToolErrors: z.boolean().default(false),
193
- announceTodoWrite: z.boolean().default(false),
194
- })
195
- return { schema, entry: { ...SCHEMA_DEFAULTS, ...(patch || {}) } }
196
- } catch (e) {
197
- log('settings 依赖不可用,跳过 settings namespace 注册:', e && e.message)
198
- return null
199
- }
200
- }
201
-
202
- module.exports = {
203
- apply(ctx, config) {
204
- config = config || {}
205
- let cfg = resolveConfig(config)
206
-
207
- // ---- settings namespace -------------------------------------------------
208
- // Wire the namespace through the settings SERVICE.
209
- //
210
- // dsh 0.1.2-alpha.1 deleted the `installSettingsSection` / `settingsNamespace`
211
- // convenience exports from `@deepseek-ai/dsh-settings`; what remains — and
212
- // has not changed since 0.1.0-rc.7 — is the `settings` service itself
213
- // (`ctx.settings.register(ns, schema, { base })` → `{ get, watch, update,
214
- // replace }`). Referencing the removed names is fatal: an ESM named import
215
- // of a deleted export is a module-evaluation SyntaxError that kills the host
216
- // boot, and a lazy `settingsModule.installSettingsSection(...)` call — what
217
- // this plugin used to do inside a timer callback — throws
218
- // `settingsNamespace is not a function` and crashed dsh before it served.
219
- //
220
- // `ctx.inject(['settings'])` is the graceful-degradation boundary: on a host
221
- // with no settings provider the callback never runs and the composed patch
222
- // config stands as-is.
223
- const prepared = buildSettingsNamespace(ctx, config)
224
- if (prepared) {
225
- ctx.inject(['settings'], scopedCtx => {
226
- // `scope.get()` is the live resolved value (schema default → patch
227
- // config → UI user layer), so re-deriving cfg from it on every change
228
- // is what makes a settings edit take effect without a restart.
229
- let settingsSource = () => prepared.entry
230
- const applySettings = () => {
231
- try { cfg = resolveConfig(settingsSource()) } catch (e) { log('settings 变更应用失败:', e && e.message) }
232
- }
233
- try {
234
- const scope = scopedCtx.settings.register(SETTINGS_NS, prepared.schema, { base: prepared.entry })
235
- settingsSource = () => scope.get()
236
- // Unload restores the composed entry, so a disabled plugin cannot
237
- // leave the queue reading a value nobody can see or change any more.
238
- scopedCtx.effect(() => () => {
239
- settingsSource = () => prepared.entry
240
- applySettings()
241
- })
242
- scope.watch(applySettings)
243
- applySettings()
244
- log('settings namespace 已注册:', SETTINGS_NS)
245
- } catch (e) {
246
- log('settings namespace 注册失败,继续使用 patch config:', e && e.message)
247
- }
248
- })
249
- }
250
-
251
- // ---- host-owned FIFO speech queue + WebSocket state sync (PR #2) ----
252
- let activeSpeech = null
253
- let speechToken = 0
254
- let replacement = null
255
- /** 队列项播完后的停顿定时器(多问题提问之间的间隔) */
256
- let gapTimer = null
257
- const speechQueue = []
258
- const speechSockets = new Set()
259
- const speechWss = new WebSocketServer({ noServer: true })
260
-
261
- function state() {
262
- const item = activeSpeech && activeSpeech.item
263
- return {
264
- type: 'speech-state',
265
- speaking: item !== undefined && item !== null,
266
- sessionId: item ? item.sessionId : null,
267
- turn: item ? item.turn : null,
268
- messageId: item ? item.messageId : null,
269
- source: item ? item.source : null,
270
- queueLength: speechQueue.length,
271
- }
272
- }
273
- function publishState() {
274
- const payload = JSON.stringify(state())
275
- for (const socket of speechSockets) {
276
- if (socket.readyState === WebSocket.OPEN) {
277
- try { socket.send(payload) } catch (e) { speechSockets.delete(socket) }
278
- }
279
- }
280
- }
281
- function removeTemp(tmp) { try { fs.unlinkSync(tmp) } catch (e) { /* already removed */ } }
282
-
283
- function startOne(item) {
284
- // master switch: nothing is ever spoken while disabled
285
- if (!cfg.enabled) { log('总开关关闭,跳过播报(文本长度:', item.text.length, ')'); return false }
286
- if (!item || !item.text.trim() || activeSpeech) return false
287
- const tmp = path.join(os.tmpdir(), `dsh-speech-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.txt`)
288
- try { fs.writeFileSync(tmp, item.text, 'utf8') } catch (e) { log('write temp failed:', e.message); return false }
289
- log('speech start', item.source, item.sessionId || '-', item.turn == null ? '-' : item.turn, item.messageId || '-', item.text.slice(0, 80))
290
- let child
291
- // 手动重播完整朗读:replayFullRead 打开时,重播跳过 heading 截断完整朗读
292
- const fullRead = item.manual === true && cfg.replayFullRead === true
293
- if (process.platform === 'darwin') {
294
- const args = ['-f', tmp, '-m', String(cfg.maxChars), '-M', cfg.longTextMode, '-l', cfg.longTextMessage, '-C', cfg.cleanMarkdownFormatting ? '1' : '0', '-I', cfg.readInlineCode ? '1' : '0', '-B', cfg.codeBlocks, '-K', String(cfg.codeBlockMaxChars), '-R', cfg.codeBlockReplacementText]
295
- if (cfg.rate > 0) args.push('-r', String(cfg.rate))
296
- if (fullRead) args.push('-F')
297
- child = spawn('/bin/bash', [cfg.engine].concat(args), { detached: true, stdio: 'ignore' })
298
- } else {
299
- // Windows 语速 = SAPI 刻度(-10 到 10,0 = 正常;负值变慢),直接透传,
300
- // 不要把 0/负值替换成 1(否则无法回到正常语速/减慢)
301
- child = spawn('powershell.exe', ['-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', cfg.engine, '-File', tmp, '-Volume', String(cfg.volume), '-Rate', String(cfg.rate), '-MaxChars', String(cfg.maxChars), '-LongTextMode', cfg.longTextMode, '-LongTextMessage', cfg.longTextMessage, '-CleanMarkdownFormatting', cfg.cleanMarkdownFormatting ? '1' : '0', '-ReadInlineCode', cfg.readInlineCode ? '1' : '0', '-CodeBlocks', cfg.codeBlocks, '-CodeBlockMaxChars', String(cfg.codeBlockMaxChars), '-CodeBlockReplacementText', cfg.codeBlockReplacementText, '-FullRead', fullRead ? '1' : '0'], { windowsHide: true, stdio: 'ignore' })
302
- }
303
- const token = ++speechToken
304
- activeSpeech = { process: child, tmp, token, item, gapMs: item.gapMs || 0 }
305
- publishState()
306
- const settle = () => {
307
- removeTemp(tmp)
308
- if (!activeSpeech || activeSpeech.token !== token) return
309
- const gap = activeSpeech.gapMs || 0
310
- activeSpeech = null
311
- publishState()
312
- const proceed = () => {
313
- gapTimer = null
314
- if (replacement) {
315
- const next = replacement
316
- replacement = null
317
- startOne(next)
318
- } else {
319
- startNext()
320
- }
321
- }
322
- // 队列项之间可配置停顿(如多个提问之间留 2 秒)
323
- if (gap > 0) {
324
- gapTimer = setTimeout(proceed, gap)
325
- } else {
326
- proceed()
327
- }
328
- }
329
- child.once('exit', settle)
330
- child.once('error', settle)
331
- return true
332
- }
333
- function startNext() {
334
- if (activeSpeech || replacement) return
335
- const item = speechQueue.shift()
336
- if (!item) { publishState(); return }
337
- publishState()
338
- startOne(item)
339
- }
340
- function enqueue(item) {
341
- if (!item || !item.text || !item.text.trim()) return
342
- speechQueue.push(item)
343
- publishState()
344
- startNext()
345
- }
346
- function stopActive() {
347
- const active = activeSpeech
348
- if (!active) return false
349
- try {
350
- if (process.platform === 'darwin' && active.process.pid) process.kill(-active.process.pid, 'SIGTERM')
351
- else active.process.kill()
352
- } catch (e) { log('stop speech failed:', e.message) }
353
- return true
354
- }
355
- function clearAndStop() {
356
- if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
357
- speechQueue.length = 0
358
- publishState()
359
- return stopActive()
360
- }
361
- function replaceWith(item) {
362
- if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
363
- speechQueue.length = 0
364
- replacement = item
365
- publishState()
366
- if (!activeSpeech) {
367
- const next = replacement
368
- replacement = null
369
- startOne(next)
370
- return
371
- }
372
- stopActive()
373
- }
374
- function visibleText(message) {
375
- if (!message) return ''
376
- if (typeof message.content === 'string') return message.content
377
- if (!Array.isArray(message.content)) return ''
378
- return message.content.filter(block => block && block.type === 'text' && typeof block.text === 'string').map(block => block.text).join('')
379
- }
380
- function hostItem(source, session, event, text, messageId) {
381
- const sessionValue = session && (session.id != null ? session.id : session.sessionId)
382
- return {
383
- source,
384
- sessionId: sessionValue != null ? String(sessionValue) : null,
385
- turn: event && event.data && Number.isFinite(event.data.turn) ? event.data.turn : null,
386
- messageId: messageId == null ? null : String(messageId),
387
- text,
388
- }
389
- }
390
-
391
- // ---- WebSocket + control route (PR #2) ----
392
- ctx.inject(['webServer'], webCtx => {
393
- webCtx.effect(() => webCtx.webServer.registerUpgrade({
394
- path: '/dsh-speak/ws',
395
- handler: (req, socket, head) => speechWss.handleUpgrade(req, socket, head, client => {
396
- speechSockets.add(client)
397
- client.once('close', () => speechSockets.delete(client))
398
- client.once('error', () => speechSockets.delete(client))
399
- try { client.send(JSON.stringify(state())) } catch (e) { speechSockets.delete(client) }
400
- }),
401
- }), 'dsh-speak speech-state websocket')
402
- webCtx.effect(() => webCtx.webServer.register({
403
- kind: 'exact', path: '/dsh-speak/control', handler: async (req, res) => {
404
- const reply = (status, value) => { res.writeHead(status, { 'Content-Type': 'application/json; charset=utf-8' }); res.end(JSON.stringify(value)) }
405
- if (req.method !== 'POST' || !String(req.headers['content-type'] || '').startsWith('application/json')) { reply(405, { error: 'POST application/json required' }); return }
406
- try {
407
- const chunks = []; let size = 0
408
- for await (const chunk of req) { size += chunk.length; if (size > 1024 * 1024) throw new Error('request too large'); chunks.push(chunk) }
409
- const body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}')
410
- if (body.action === 'status') { reply(200, state()); return }
411
- if (body.action === 'stop') {
412
- replacement = null
413
- clearAndStop()
414
- reply(200, state())
415
- return
416
- }
417
- if (body.action !== 'play' || typeof body.text !== 'string' || !body.text.trim()) { reply(400, { error: 'invalid control request' }); return }
418
- replaceWith({ source: 'manual', manual: true, sessionId: body.sessionId == null ? null : String(body.sessionId), turn: Number.isFinite(body.turn) ? body.turn : null, messageId: body.messageId == null ? null : String(body.messageId), text: body.text })
419
- reply(200, state())
420
- } catch (e) { reply(e.message === 'request too large' ? 413 : 400, { error: e.message }) }
421
- },
422
- }), 'dsh-speak replay control route')
423
- })
424
-
425
- ctx.effect(() => () => {
426
- replacement = null
427
- clearAndStop()
428
- for (const socket of speechSockets) { try { socket.close() } catch (e) { /* closed */ } }
429
- speechSockets.clear()
430
- try { speechWss.close() } catch (e) { /* closed */ }
431
- }, 'dsh-speak speech cleanup')
432
-
433
- // ---- session event handling ----
434
- let timer = null
435
- let pendingText = ''
436
- /** 当前回合内最后一条助手消息文本(turn/end 兜底播报用) */
437
- let lastText = ''
438
- /** 已通过节流播报过的文本(防止 turn/end 兜底重复播报) */
439
- let lastSpokenText = ''
440
- /** 最后一条助手消息 id(兜底播报时带上) */
441
- let lastMessageId = null
442
- /** cancel a pending throttled announcement (default mode, tool-call round) */
443
- function cancelPending() {
444
- if (timer) { clearTimeout(timer); timer = null }
445
- pendingText = ''
446
- }
447
-
448
- ctx.on('session/event', (session, event) => {
449
- try {
450
- const type = event && event.type
451
- if (type !== 'assistant/chunk') {
452
- log('事件 type=', type, 'surfaceOp=', event && event.surfaceOp, 'seq=', event && event.seq)
453
- }
454
- // 新回合开始:清空上一回合的兜底状态,避免跨回合残留
455
- if (type === 'turn/start') {
456
- cancelPending()
457
- lastText = ''
458
- lastSpokenText = ''
459
- lastMessageId = null
460
- return
461
- }
462
- // tool-call round: ask_user_question announces the parsed question; any
463
- // other tool call cancels the pending throttled narration
464
- if (type === 'tool/call') {
465
- if (event.data && event.data.name === 'ask_user_question' && cfg.announceQuestions) {
466
- let items = []
467
- try {
468
- const args = JSON.parse(event.data.arguments || '{}')
469
- const questions = Array.isArray(args.questions) ? args.questions : []
470
- // 每个问题单独入队播报:带"问题N"序号(多问题时)与"选项N"序号
471
- // (序号用数字,与 UI 的自动编号一致;中文 TTS 自然读成"一/二/三")
472
- items = questions.map((question, qi) => {
473
- const mode = question.multi_select ? '多选' : '单选'
474
- const opts = Array.isArray(question.options) ? question.options : []
475
- const optText = opts.map((option, oi) => {
476
- const label = option && option.label ? String(option.label) : ''
477
- return label ? `选项${oi + 1},${label}` : ''
478
- }).filter(Boolean).join(';')
479
- const head = questions.length > 1 ? `问题${qi + 1},` : ''
480
- // question 文案已含"单选/多选"字样时不再追加模式后缀,避免重复
481
- const modeSuffix = /单选|多选/.test(question.question || '') ? '' : `(${mode})`
482
- const body = [question.question || '', modeSuffix, optText ? ',' + optText : ''].join('')
483
- return (head + body).trim()
484
- }).filter(Boolean)
485
- } catch (e) { /* ignore malformed arguments */ }
486
- if (items.length > 0) {
487
- cancelPending()
488
- // 问题已单独播报,标记当前最后文本为已播,避免 turn/end 兜底重复
489
- lastSpokenText = lastText
490
- // 多条问题按 FIFO 串行播报,之间停顿 cfg.questionGapMs(默认 2 秒)
491
- const gap = cfg.questionGapMs > 0 && items.length > 1 ? cfg.questionGapMs : 0
492
- items.forEach((itemText, i) => {
493
- const item = hostItem('question', session, event, itemText, null)
494
- item.gapMs = i < items.length - 1 ? gap : 0
495
- enqueue(item)
496
- })
497
- }
498
- return
499
- }
500
- cancelPending()
501
- return
502
- }
503
- // approval requested: announce right away
504
- if (type === 'approval/asked' && cfg.announceApprovals) {
505
- cancelPending()
506
- let reason = String((event.data && event.data.reason) || '')
507
- if (cfg.stripApprovalPrefix) {
508
- // 通用剥离行首"动作标签: "前缀(英文动作短语 + 冒号,如
509
- // "Store decision fact in workspace memory (dsh-speak): <内容>"、
510
- // "escalate sandbox to danger-full-access: <原因>"),只念冒号后的
511
- // 具体内容;中文开头或无冒号的 reason 原样保留(如"删除 xxx")。
512
- reason = reason.replace(/^[A-Za-z][^::\n]*?[::]\s*/, '').trim()
513
- }
514
- enqueue(hostItem('approval', session, event, reason || '需要你的审批,请查看界面。', null))
515
- return
516
- }
517
- // 回合结束:兜底播报最终回复(被工具调用取消的节流文本在此补播,
518
- // 已播过的不重复),随后按需播报"第 N 轮对话完成"可选事件
519
- if (type === 'turn/end') {
520
- if (cfg.automaticSpeech && !cfg.queueAllMessages && lastText && lastText !== lastSpokenText) {
521
- const itemText = lastText
522
- const itemMessageId = lastMessageId
523
- cancelPending()
524
- lastText = ''
525
- lastSpokenText = itemText
526
- enqueue(hostItem('automatic', session, event, itemText, itemMessageId))
527
- }
528
- if (!cfg.announceTurnEnd) return
529
- const data = event.data
530
- const prefix = data && data.turn != null ? `第 ${data.turn} 轮对话` : '本轮对话'
531
- const kind = data && data.reason && data.reason.kind
532
- const text = ({ completed: prefix + '完成', aborted: prefix + '中断', interrupted: prefix + '中断', blocked: prefix + '被阻塞', error: prefix + '异常结束', 'max-tokens': prefix + '异常结束' })[kind] || prefix + '结束'
533
- enqueue(hostItem('turn/end', session, event, text, null))
534
- return
535
- }
536
- if (type === 'command/done' && cfg.announceCommandDone) {
537
- enqueue(hostItem('command/done', session, event, (event.data && event.data.kind) === 'error' ? '命令执行失败' : '命令执行完成', null))
538
- return
539
- }
540
- if (type === 'goal/change' && cfg.announceGoalChange) {
541
- const data = event.data
542
- const objective = data && data.goal && data.goal.objective
543
- const label = ({ create: '已创建目标', edit: '目标已更新', complete: '目标已完成', pause: '目标已暂停', resume: '目标已恢复', block: '目标已阻塞', clear: '目标已清除' })[data && data.operation] || '目标状态变化'
544
- const text = objective && ['create', 'edit', 'complete'].includes(data.operation) ? `${label}:${objective.replace(/\s+/g, ' ').trim().slice(0, 40)}` : label
545
- enqueue(hostItem('goal/change', session, event, text, null))
546
- return
547
- }
548
- if (type === 'tool/result' && cfg.announceToolErrors) {
549
- const data = event.data
550
- const err = data && data.error
551
- // 真实错误标记有两处:结构化失败身份 data.error(name/code),以及结果块上的
552
- // isError。0.1.2 起 createToolResultMessage 把结果块包进一个 ToolResultBlock
553
- // ({ type:'tool-result', toolCallId, content:[…], isError }),文字在它嵌套的
554
- // content 里;更早的版本把 isError 直接放在 text 块上。两种形状都读。
555
- //
556
- // 注意:pwsh / bash 把「命令非零退出」当作结果数据上报(`exit code: N`),
557
- // 不置 isError —— 只有基础设施失败(spawn 错误、abort)才是 isError 结果,
558
- // 所以失败的命令本身不会播报工具出错。
559
- const errText = (Array.isArray(data && data.message && data.message.content) ? data.message.content : [])
560
- .filter(block => block && block.isError === true)
561
- .map(block => {
562
- const parts = Array.isArray(block.content) ? block.content : [block]
563
- return parts.map(part => (part && (part.text || part.code)) || '').filter(Boolean).join(' ')
564
- })
565
- .filter(Boolean).join(' ')
566
- if (err || errText) {
567
- const detail = (errText || (err && err.code) || (err && err.name) || '').replace(/\s+/g, ' ').trim().slice(0, 60)
568
- // 详情只在"确实是一句中文描述"时才念:英文模板(Error: / ENOENT / 技术
569
- // code)对中文用户可读性差,应当截掉。判据是**汉字数量多于拉丁字母数量**,
570
- // 而不是"含有汉字"——后者会被路径里的中文目录名骗过:
571
- // `Error: cannot read "D:\...\第二轮测试用的不存在文件.txt"` 含 12 个汉字,
572
- // 却是纯英文报错(1.8.0 修正)。
573
- const cjkCount = (detail.match(/[\u4e00-\u9fff]/g) || []).length
574
- const latinCount = (detail.match(/[A-Za-z]/g) || []).length
575
- const readable = cjkCount > latinCount ? `:${detail}` : ''
576
- enqueue(hostItem('tool/result', session, event, `工具调用出错${readable}`, null))
577
- }
578
- return
579
- }
580
- if (type === 'todo/write' && cfg.announceTodoWrite) {
581
- const todos = Array.isArray(event.data && event.data.todos) ? event.data.todos : []
582
- const done = todos.filter(t => t && t.status === 'completed').length
583
- enqueue(hostItem('todo/write', session, event, `待办已更新:${done}/${todos.length} 完成`, null))
584
- return
585
- }
586
- if (!event || type !== 'assistant/message') return
587
- if (event.surfaceOp && event.surfaceOp !== 'append') return
588
- const message = event.data && (event.data.message || event.data)
589
- const text = visibleText(message)
590
- if (!text.trim()) return
591
-
592
- // queueAllMessages mode (PR #2): enqueue every assistant message now
593
- if (cfg.queueAllMessages && cfg.automaticSpeech) {
594
- enqueue(hostItem('automatic', session, event, text, message && message.id))
595
- return
596
- }
597
- // default mode: throttle/merge the final reply; a tool/call cancels it,
598
- // and turn/end 兜底补播 lastText(见上方 turn/end 分支)
599
- cancelPending()
600
- pendingText = text
601
- lastText = text
602
- lastMessageId = message && message.id ? String(message.id) : null
603
- timer = setTimeout(() => {
604
- if (!pendingText) return
605
- const itemText = pendingText
606
- pendingText = ''
607
- timer = null
608
- lastSpokenText = itemText
609
- enqueue(hostItem('automatic', session, event, itemText, lastMessageId))
610
- }, cfg.throttleMs)
611
- } catch (e) { log('session event speech error:', e.message) }
612
- })
613
- },
614
- }
1
+ // speech-hook.js — DSH web adapter: voice-announce assistant activity
2
+ // ==============================================================================
3
+ // Listens to the session event stream (session/event), extracts the final reply
4
+ // text, and hands it to the speech engine (engine/speak.ps1 on Windows,
5
+ // engine/speak.sh on macOS) through a hidden, non-blocking child process.
6
+ //
7
+ // Since 1.7.0 this is the merged host of the original dsh-speak behavior and
8
+ // victorwads' PR #2 (turn-level replay + host FIFO speech queue + WebSocket
9
+ // state sync + native Speak settings page):
10
+ //
11
+ // * a host-owned FIFO speech queue: only one native speech process runs at a
12
+ // time; queued items continue automatically when the current one finishes
13
+ // * every eligible item (final reply, approvals, questions, optional events,
14
+ // manual replay) is enqueued, so the WebSocket state (which message is
15
+ // speaking, queue length) is always truthful — even for automatic replies
16
+ // * `queueAllMessages` (default off) switches between two automatic modes:
17
+ // - off (default): final replies are throttled/merged as before, plus the
18
+ // optional event announcements; tool calls cancel pending narration
19
+ // - on: every assistant/message is enqueued immediately as it arrives
20
+ // * a `/dsh-speak/control` POST route (play/stop/status) and a
21
+ // `/dsh-speak/ws` WebSocket publish the authoritative speech state
22
+ // * a settings form projected from THIS module's exported `Config` (DSH >=
23
+ // 0.1.7 reads each Loader entry's own Config); schema defaults → patch
24
+ // config → UI user layer. The settings namespace is the entry id
25
+ // (`dsh-speak`), and a write commits into the live config references
26
+ // * `enabled` master switch: when off, nothing is ever enqueued (no sound)
27
+ //
28
+ // Trigger semantics:
29
+ // * assistant/message with a `text` block is announced (reasoning / tool_use
30
+ // blocks are skipped)
31
+ // * a tool/call to `ask_user_question` announces the parsed question; other
32
+ // tool calls cancel the pending throttled announcement (default mode)
33
+ // * `approval/asked` is announced immediately (reason, else a fixed prompt)
34
+ // * optional events (turn/end, command/done, goal/change, tool/result errors,
35
+ // todo/write) are announced when their toggle is on (default off)
36
+ //
37
+ // Configuration — prefer the Web UI (Settings → dsh-speak settings) or the
38
+ // profile patch `config` block (see README.md). All keys resolve as
39
+ // schema default → patch config → UI user layer.
40
+ 'use strict'
41
+
42
+ const { spawn } = require('child_process')
43
+ const { WebSocketServer, WebSocket } = require('ws')
44
+ const fs = require('fs')
45
+ const os = require('os')
46
+ const path = require('path')
47
+
48
+ const LOG = path.join(os.tmpdir(), 'dsh-speech-hook.log')
49
+ function log(...args) {
50
+ try { fs.appendFileSync(LOG, `[${new Date().toISOString()}] ${args.join(' ')}\n`) } catch (e) { /* ignore */ }
51
+ }
52
+
53
+ const ENGINE_NAME = process.platform === 'darwin' ? 'speak.sh' : 'speak.ps1'
54
+
55
+ /**
56
+ * Locate the engine script:
57
+ * 1. explicit override (config `engine`)
58
+ * 2. <this package>/engine/<speak.ps1|speak.sh> — repo checkout or profile install
59
+ * 3. legacy file-copy location (~/.dsh/hooks/<speak.ps1|speak.sh>)
60
+ */
61
+ function resolveEngine(override) {
62
+ if (override) return override
63
+ const bundled = path.join(__dirname, '..', '..', 'engine', ENGINE_NAME)
64
+ if (fs.existsSync(bundled)) return bundled
65
+ return path.join(os.homedir(), '.dsh', 'hooks', ENGINE_NAME)
66
+ }
67
+
68
+ // Platform-aware defaults: macOS `say` has no per-utterance ceiling, so
69
+ // `maxChars` defaults to 0 (unlimited) there; Windows keeps the safe 300.
70
+ const DEFAULT_MAX_CHARS = process.platform === 'darwin' ? 0 : 300
71
+
72
+ // ---------------------------------------------------------------------------
73
+ // Settings (DSH >= 0.1.7: this module's own exported `Config` is the form)
74
+ // ---------------------------------------------------------------------------
75
+ // 0.1.7 replaced the imperative `settings.register(namespace, schema, { base })`
76
+ // provider API with Config projection: the settings service reads every ACTIVE
77
+ // Loader entry's own exported `Config` schema and projects its `.volatile()`
78
+ // fields into the settings UI (`ctx.settings.describe()` on the host,
79
+ // `ctx.configForms` in the browser). There is no namespace to register any
80
+ // more — the namespace IS the Loader entry id (`dsh-speak`; see
81
+ // cordis.patch.yml), which is what client/client.js binds.
82
+ //
83
+ // What that changes here:
84
+ // * the schema must exist as a static export, built at module load (the
85
+ // Loader reads `module.exports.Config` before any context exists),
86
+ // * volatile fields reach apply() as stable references (`config.enabled
87
+ // .get()`), not plain values,
88
+ // * a settings write commits into those references in place and emits
89
+ // `loader/volatile-update` — no restart, so cfg is re-derived there.
90
+ //
91
+ // Nothing is registered from this half any more: 0.1.2-alpha.1 had deleted
92
+ // `installSettingsSection` / `settingsNamespace`, and 0.1.7 deleted the
93
+ // `settings.register` service API that had replaced them in 1.6.0. The only
94
+ // thing left to do is opt out of the shell's auto-generated page (it would
95
+ // duplicate the hand-written one in client/client.js).
96
+ //
97
+ // Values still resolve as: schema default → patch `config` → UI user layer.
98
+ // `SCHEMA_DEFAULTS` is the normalization fallback used when the schemastery
99
+ // peer is unavailable (no Config → no settings page, patch config only);
100
+ // test-settings-integration.js asserts it stays in sync with the real schema.
101
+ const SCHEMA_DEFAULTS = {
102
+ enabled: true,
103
+ automaticSpeech: true,
104
+ cleanMarkdownFormatting: true,
105
+ readInlineCode: true,
106
+ codeBlocks: 'smart',
107
+ codeBlockMaxChars: 300,
108
+ codeBlockReplacementText: 'You can see the code in our history.',
109
+ queueAllMessages: false,
110
+ throttleMs: 1500,
111
+ replayFullRead: false,
112
+ engine: '',
113
+ announceApprovals: true,
114
+ announceQuestions: true,
115
+ stripApprovalPrefix: true,
116
+ questionGapMs: 2000,
117
+ longTextMode: 'message',
118
+ longTextMessage: '本次播报内容较长,请自行阅读。',
119
+ maxChars: DEFAULT_MAX_CHARS,
120
+ volume: 50,
121
+ rate: 0,
122
+ announceTurnEnd: false,
123
+ announceCommandDone: false,
124
+ announceGoalChange: false,
125
+ announceToolErrors: false,
126
+ announceTodoWrite: false,
127
+ }
128
+
129
+ /**
130
+ * Coerce one numeric config field.
131
+ *
132
+ * An explicit value wins — including 0, which is meaningful for `throttleMs` (no
133
+ * merging), `maxChars` (macOS: unlimited) and `volume` (silence). The fallback
134
+ * applies only when the field is absent or unparseable, and the result is clamped
135
+ * to what the engine accepts, because an out-of-range SAPI `Rate` (-10..10) or
136
+ * `Volume` (0..100) makes `speak.ps1` throw — i.e. silence with no explanation.
137
+ * `||` is deliberately NOT used here: it silently turned a user's 0 into the
138
+ * fallback and let negatives through.
139
+ * @param field - config key, for the clamp diagnostic.
140
+ * @returns the usable number.
141
+ */
142
+ function configNumber(field, value, fallback, min, max) {
143
+ if (value === undefined || value === null || value === '') return fallback
144
+ const parsed = Number(value)
145
+ if (!Number.isFinite(parsed)) return fallback
146
+ const clamped = Math.min(max, Math.max(min, Math.round(parsed)))
147
+ if (clamped !== parsed) log('settings 值超出范围,已钳制:', field, parsed, '->', clamped)
148
+ return clamped
149
+ }
150
+
151
+ /**
152
+ * Resolve the raw settings value into the mutable `cfg` the queue reads.
153
+ * Kept a pure mapping (only a diagnostic line when a value had to be clamped) so
154
+ * the initial apply and every `loader/volatile-update` normalize identically
155
+ * (engine re-resolution, platform defaults, SAPI-safe ranges).
156
+ */
157
+ function resolveConfig(value) {
158
+ value = value || {}
159
+ return {
160
+ enabled: value.enabled !== false,
161
+ automaticSpeech: value.automaticSpeech !== false,
162
+ cleanMarkdownFormatting: value.cleanMarkdownFormatting !== false,
163
+ readInlineCode: value.readInlineCode !== false,
164
+ codeBlocks: ['all', 'smart', 'replace'].includes(value.codeBlocks) ? value.codeBlocks : 'smart',
165
+ codeBlockMaxChars: configNumber('codeBlockMaxChars', value.codeBlockMaxChars, 300, 0, Number.MAX_SAFE_INTEGER),
166
+ codeBlockReplacementText: String(value.codeBlockReplacementText || 'You can see the code in our history.'),
167
+ queueAllMessages: value.queueAllMessages === true,
168
+ throttleMs: configNumber('throttleMs', value.throttleMs, 1500, 0, Number.MAX_SAFE_INTEGER),
169
+ replayFullRead: value.replayFullRead === true,
170
+ engine: resolveEngine(value.engine || ''),
171
+ announceApprovals: value.announceApprovals !== false,
172
+ announceQuestions: value.announceQuestions !== false,
173
+ stripApprovalPrefix: value.stripApprovalPrefix !== false,
174
+ questionGapMs: configNumber('questionGapMs', value.questionGapMs, 2000, 0, Number.MAX_SAFE_INTEGER),
175
+ longTextMode: value.longTextMode === 'heading' ? 'heading' : 'message',
176
+ longTextMessage: String(value.longTextMessage || SCHEMA_DEFAULTS.longTextMessage),
177
+ maxChars: configNumber('maxChars', value.maxChars, DEFAULT_MAX_CHARS, 0, Number.MAX_SAFE_INTEGER),
178
+ volume: configNumber('volume', value.volume, 50, 0, 100),
179
+ // Windows: SAPI scale -10..10. macOS: words per minute (0 = engine default),
180
+ // where the engine passes -r only for a positive value anyway.
181
+ rate: process.platform === 'darwin'
182
+ ? configNumber('rate', value.rate, 0, 0, Number.MAX_SAFE_INTEGER)
183
+ : configNumber('rate', value.rate, 0, -10, 10),
184
+ announceTurnEnd: value.announceTurnEnd === true,
185
+ announceCommandDone: value.announceCommandDone === true,
186
+ announceGoalChange: value.announceGoalChange === true,
187
+ announceToolErrors: value.announceToolErrors === true,
188
+ announceTodoWrite: value.announceTodoWrite === true,
189
+ }
190
+ }
191
+
192
+ /**
193
+ * Read one config field from whatever shape the Loader handed us: a volatile
194
+ * field arrives as a reference (`{ get() }`, cosmokit `createVolatile`), a
195
+ * plugin mounted without a Config schema receives plain values, and an omitted
196
+ * field is simply `undefined`.
197
+ * @returns the current plain value, or undefined.
198
+ */
199
+ function configValue(config, key) {
200
+ if (!config) return undefined
201
+ const value = config[key]
202
+ if (value === undefined || value === null) return undefined
203
+ if (typeof value === 'object' && typeof value.get === 'function') return value.get()
204
+ return value
205
+ }
206
+
207
+ /**
208
+ * Detach every known field into the plain object `resolveConfig` normalizes.
209
+ * Read fresh on each call: the references are updated in place by a settings
210
+ * write, so the same `config` object always yields the current values.
211
+ */
212
+ function rawConfig(config) {
213
+ const raw = {}
214
+ for (const key of Object.keys(SCHEMA_DEFAULTS)) raw[key] = configValue(config, key)
215
+ return raw
216
+ }
217
+
218
+ /**
219
+ * The settings namespace this plugin owns: its Loader entry id. That id is what
220
+ * the settings service projects the form under (`settings.describe()` →
221
+ * `ctx.configForms.get(id)` in the browser) and what client/client.js binds.
222
+ * `settingsNamespace` is also the name the plugin used for its settings before
223
+ * 0.1.7, so the browser half accepts either spelling.
224
+ * @returns the entry id, or undefined when mounted without a Loader.
225
+ */
226
+ function settingsNamespace(ctx) {
227
+ try { return ctx.fiber && ctx.fiber.entry ? ctx.fiber.entry.options.id : undefined } catch (e) { return undefined }
228
+ }
229
+
230
+ /**
231
+ * Build the settings form schema — the module's static `Config` export.
232
+ *
233
+ * Resolved at load time because the Loader reads `module.exports.Config`
234
+ * immediately after importing the plugin, before any context exists (the old
235
+ * `ctx.baseUrl` fallback is therefore unavailable here, and `__filename` is
236
+ * enough: it walks the profile's own `node_modules` chain either way).
237
+ * Best-effort: an installation that cannot resolve the schemastery peer gets no
238
+ * settings page (Config stays undefined) and keeps running on the composed
239
+ * patch `config`.
240
+ * @returns the schema, or undefined when the peer is unavailable.
241
+ */
242
+ function buildConfig() {
243
+ try {
244
+ const z = require('module').createRequire(__filename)('@deepseek-ai/schemastery')
245
+ return z.object({
246
+ enabled: z.boolean().default(true).volatile(),
247
+ automaticSpeech: z.boolean().default(true).volatile(),
248
+ cleanMarkdownFormatting: z.boolean().default(true).volatile(),
249
+ readInlineCode: z.boolean().default(true).volatile(),
250
+ codeBlocks: z.union(['all', 'smart', 'replace']).default('smart').volatile(),
251
+ codeBlockMaxChars: z.natural().default(300).volatile(),
252
+ codeBlockReplacementText: z.string().default('You can see the code in our history.').volatile(),
253
+ queueAllMessages: z.boolean().default(false).volatile(),
254
+ throttleMs: z.natural().default(1500).volatile(),
255
+ replayFullRead: z.boolean().default(false).volatile(),
256
+ engine: z.string().default('').volatile(),
257
+ announceApprovals: z.boolean().default(true).volatile(),
258
+ announceQuestions: z.boolean().default(true).volatile(),
259
+ stripApprovalPrefix: z.boolean().default(true).volatile(),
260
+ questionGapMs: z.natural().default(2000).volatile(),
261
+ longTextMode: z.union(['message', 'heading']).default('message').volatile(),
262
+ longTextMessage: z.string().default('本次播报内容较长,请自行阅读。').volatile(),
263
+ maxChars: z.natural().default(DEFAULT_MAX_CHARS).volatile(),
264
+ volume: z.natural().default(50).volatile(),
265
+ rate: z.number().default(0).volatile(),
266
+ announceTurnEnd: z.boolean().default(false).volatile(),
267
+ announceCommandDone: z.boolean().default(false).volatile(),
268
+ announceGoalChange: z.boolean().default(false).volatile(),
269
+ announceToolErrors: z.boolean().default(false).volatile(),
270
+ announceTodoWrite: z.boolean().default(false).volatile(),
271
+ })
272
+ } catch (e) {
273
+ log('settings 依赖不可用,跳过 settings 表单(继续用 patch config):', e && e.message)
274
+ return undefined
275
+ }
276
+ }
277
+
278
+ module.exports = {
279
+ // Static settings form schema projected by the settings service (DSH >=
280
+ // 0.1.7). Undefined when the schemastery peer cannot be resolved.
281
+ Config: buildConfig(),
282
+
283
+ apply(ctx, config) {
284
+ // ---- duplicate-entry guard ----------------------------------------------
285
+ // One profile must mount this plugin exactly ONCE. A second row with the same
286
+ // id (or the legacy `speech-hook` id) is easy to create by accident — adding
287
+ // `dsh-speak` to `dsh.profile.bundles` while the hand-written insert row is
288
+ // still there is the usual way. Two live instances would mean two FIFO queues
289
+ // announcing everything twice and two claims on the `/dsh-speak/ws` route.
290
+ //
291
+ // The claim is keyed on `globalThis`, not on a module variable: two rows may
292
+ // name this file differently (a `file:///…` URL beside the bare package name)
293
+ // and Node would then evaluate the module twice, each copy seeing its own
294
+ // module-scope flag. The extra instance stays inert and says so in the log —
295
+ // remove the duplicate row and reload to hand ownership over.
296
+ //
297
+ // This guard protects SPEECH only. A duplicated entry id also breaks the
298
+ // settings page in a way this half cannot fix: DSH's config editor keeps only
299
+ // uniquely-ided entries, so every write is refused with
300
+ // `settings/rejected: Configuration for "dsh-speak" is overridden by a home
301
+ // patch or command-line overlay` while speech keeps working. Hence the hint in
302
+ // the log line below.
303
+ const claim = Symbol.for('dsh-speak.active')
304
+ if (globalThis[claim] !== undefined) {
305
+ log('已有实例在运行,本行不再挂载(条目 id =', settingsNamespace(ctx),
306
+ '):请删掉重复的 dsh-speak 行后重载 —— 重复的 id 还会让设置页的写入被拒(overridden by a home patch)')
307
+ return
308
+ }
309
+ globalThis[claim] = true
310
+ ctx.effect(() => () => { if (globalThis[claim] === true) delete globalThis[claim] }, 'dsh-speak: single-instance claim')
311
+
312
+ // Live configuration: the references inside `config` are updated in place
313
+ // by a settings write, so cfg is re-derived from them (see below).
314
+ const readConfig = () => resolveConfig(rawConfig(config))
315
+ let cfg = readConfig()
316
+
317
+ // ---- settings presentation ----------------------------------------------
318
+ // This plugin ships a hand-written settings page (client/client.js), so the
319
+ // entry opts out of the shell's automatically generated form — otherwise the
320
+ // same fields would appear twice. Presentation-only and non-fatal: a host
321
+ // without the settings service simply has no pages at all.
322
+ ctx.inject(['settings'], scopedCtx => {
323
+ if (!ctx.fiber) return
324
+ try {
325
+ scopedCtx.effect(() => scopedCtx.settings.configure({ auto: false }, ctx.fiber))
326
+ log('settings 表单由 settings 服务投影(entry =', settingsNamespace(ctx), ',自动页面已关闭)')
327
+ } catch (e) {
328
+ log('settings.configure 失败,保留默认页面策略:', e && e.message)
329
+ }
330
+ })
331
+
332
+ // A settings write commits into the running fiber's config references and
333
+ // emits this event; re-deriving cfg is what makes the edit audible without
334
+ // restarting dsh.
335
+ ctx.on('loader/volatile-update', () => {
336
+ try { cfg = readConfig() } catch (e) { log('settings 变更应用失败:', e && e.message) }
337
+ })
338
+
339
+ // ---- host-owned FIFO speech queue + WebSocket state sync (PR #2) ----
340
+ let activeSpeech = null
341
+ let speechToken = 0
342
+ let replacement = null
343
+ /** 队列项播完后的停顿定时器(多问题提问之间的间隔) */
344
+ let gapTimer = null
345
+ const speechQueue = []
346
+ const speechSockets = new Set()
347
+ const speechWss = new WebSocketServer({ noServer: true })
348
+
349
+ function state() {
350
+ const item = activeSpeech && activeSpeech.item
351
+ return {
352
+ type: 'speech-state',
353
+ speaking: item !== undefined && item !== null,
354
+ sessionId: item ? item.sessionId : null,
355
+ turn: item ? item.turn : null,
356
+ messageId: item ? item.messageId : null,
357
+ source: item ? item.source : null,
358
+ queueLength: speechQueue.length,
359
+ }
360
+ }
361
+ function publishState() {
362
+ const payload = JSON.stringify(state())
363
+ for (const socket of speechSockets) {
364
+ if (socket.readyState === WebSocket.OPEN) {
365
+ try { socket.send(payload) } catch (e) { speechSockets.delete(socket) }
366
+ }
367
+ }
368
+ }
369
+ function removeTemp(tmp) { try { fs.unlinkSync(tmp) } catch (e) { /* already removed */ } }
370
+
371
+ function startOne(item) {
372
+ // master switch: nothing is ever spoken while disabled
373
+ if (!cfg.enabled) { log('总开关关闭,跳过播报(文本长度:', item.text.length, ')'); return false }
374
+ if (!item || !item.text.trim() || activeSpeech) return false
375
+ const tmp = path.join(os.tmpdir(), `dsh-speech-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.txt`)
376
+ try { fs.writeFileSync(tmp, item.text, 'utf8') } catch (e) { log('write temp failed:', e.message); return false }
377
+ log('speech start', item.source, item.sessionId || '-', item.turn == null ? '-' : item.turn, item.messageId || '-', item.text.slice(0, 80))
378
+ let child
379
+ // 手动重播完整朗读:replayFullRead 打开时,重播跳过 heading 截断完整朗读
380
+ const fullRead = item.manual === true && cfg.replayFullRead === true
381
+ if (process.platform === 'darwin') {
382
+ const args = ['-f', tmp, '-m', String(cfg.maxChars), '-M', cfg.longTextMode, '-l', cfg.longTextMessage, '-C', cfg.cleanMarkdownFormatting ? '1' : '0', '-I', cfg.readInlineCode ? '1' : '0', '-B', cfg.codeBlocks, '-K', String(cfg.codeBlockMaxChars), '-R', cfg.codeBlockReplacementText]
383
+ if (cfg.rate > 0) args.push('-r', String(cfg.rate))
384
+ if (fullRead) args.push('-F')
385
+ child = spawn('/bin/bash', [cfg.engine].concat(args), { detached: true, stdio: 'ignore' })
386
+ } else {
387
+ // Windows 语速 = SAPI 刻度(-10 到 10,0 = 正常;负值变慢),直接透传,
388
+ // 不要把 0/负值替换成 1(否则无法回到正常语速/减慢)
389
+ child = spawn('powershell.exe', ['-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', cfg.engine, '-File', tmp, '-Volume', String(cfg.volume), '-Rate', String(cfg.rate), '-MaxChars', String(cfg.maxChars), '-LongTextMode', cfg.longTextMode, '-LongTextMessage', cfg.longTextMessage, '-CleanMarkdownFormatting', cfg.cleanMarkdownFormatting ? '1' : '0', '-ReadInlineCode', cfg.readInlineCode ? '1' : '0', '-CodeBlocks', cfg.codeBlocks, '-CodeBlockMaxChars', String(cfg.codeBlockMaxChars), '-CodeBlockReplacementText', cfg.codeBlockReplacementText, '-FullRead', fullRead ? '1' : '0'], { windowsHide: true, stdio: 'ignore' })
390
+ }
391
+ const token = ++speechToken
392
+ activeSpeech = { process: child, tmp, token, item, gapMs: item.gapMs || 0 }
393
+ publishState()
394
+ const settle = () => {
395
+ removeTemp(tmp)
396
+ if (!activeSpeech || activeSpeech.token !== token) return
397
+ const gap = activeSpeech.gapMs || 0
398
+ activeSpeech = null
399
+ publishState()
400
+ const proceed = () => {
401
+ gapTimer = null
402
+ if (replacement) {
403
+ const next = replacement
404
+ replacement = null
405
+ startOne(next)
406
+ } else {
407
+ startNext()
408
+ }
409
+ }
410
+ // 队列项之间可配置停顿(如多个提问之间留 2 秒)
411
+ if (gap > 0) {
412
+ gapTimer = setTimeout(proceed, gap)
413
+ } else {
414
+ proceed()
415
+ }
416
+ }
417
+ child.once('exit', settle)
418
+ child.once('error', settle)
419
+ return true
420
+ }
421
+ function startNext() {
422
+ if (activeSpeech || replacement) return
423
+ const item = speechQueue.shift()
424
+ if (!item) { publishState(); return }
425
+ publishState()
426
+ startOne(item)
427
+ }
428
+ function enqueue(item) {
429
+ if (!item || !item.text || !item.text.trim()) return
430
+ speechQueue.push(item)
431
+ publishState()
432
+ startNext()
433
+ }
434
+ function stopActive() {
435
+ const active = activeSpeech
436
+ if (!active) return false
437
+ try {
438
+ if (process.platform === 'darwin' && active.process.pid) process.kill(-active.process.pid, 'SIGTERM')
439
+ else active.process.kill()
440
+ } catch (e) { log('stop speech failed:', e.message) }
441
+ return true
442
+ }
443
+ function clearAndStop() {
444
+ if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
445
+ speechQueue.length = 0
446
+ publishState()
447
+ return stopActive()
448
+ }
449
+ function replaceWith(item) {
450
+ if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
451
+ speechQueue.length = 0
452
+ replacement = item
453
+ publishState()
454
+ if (!activeSpeech) {
455
+ const next = replacement
456
+ replacement = null
457
+ startOne(next)
458
+ return
459
+ }
460
+ stopActive()
461
+ }
462
+ function visibleText(message) {
463
+ if (!message) return ''
464
+ if (typeof message.content === 'string') return message.content
465
+ if (!Array.isArray(message.content)) return ''
466
+ return message.content.filter(block => block && block.type === 'text' && typeof block.text === 'string').map(block => block.text).join('')
467
+ }
468
+ function hostItem(source, session, event, text, messageId) {
469
+ const sessionValue = session && (session.id != null ? session.id : session.sessionId)
470
+ return {
471
+ source,
472
+ sessionId: sessionValue != null ? String(sessionValue) : null,
473
+ turn: event && event.data && Number.isFinite(event.data.turn) ? event.data.turn : null,
474
+ messageId: messageId == null ? null : String(messageId),
475
+ text,
476
+ }
477
+ }
478
+
479
+ // ---- WebSocket + control route (PR #2) ----
480
+ ctx.inject(['webServer'], webCtx => {
481
+ webCtx.effect(() => webCtx.webServer.registerUpgrade({
482
+ path: '/dsh-speak/ws',
483
+ handler: (req, socket, head) => speechWss.handleUpgrade(req, socket, head, client => {
484
+ speechSockets.add(client)
485
+ client.once('close', () => speechSockets.delete(client))
486
+ client.once('error', () => speechSockets.delete(client))
487
+ try { client.send(JSON.stringify(state())) } catch (e) { speechSockets.delete(client) }
488
+ }),
489
+ }), 'dsh-speak speech-state websocket')
490
+ webCtx.effect(() => webCtx.webServer.register({
491
+ kind: 'exact', path: '/dsh-speak/control', handler: async (req, res) => {
492
+ const reply = (status, value) => { res.writeHead(status, { 'Content-Type': 'application/json; charset=utf-8' }); res.end(JSON.stringify(value)) }
493
+ if (req.method !== 'POST' || !String(req.headers['content-type'] || '').startsWith('application/json')) { reply(405, { error: 'POST application/json required' }); return }
494
+ try {
495
+ const chunks = []; let size = 0
496
+ for await (const chunk of req) { size += chunk.length; if (size > 1024 * 1024) throw new Error('request too large'); chunks.push(chunk) }
497
+ const body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}')
498
+ if (body.action === 'status') { reply(200, state()); return }
499
+ if (body.action === 'stop') {
500
+ replacement = null
501
+ clearAndStop()
502
+ reply(200, state())
503
+ return
504
+ }
505
+ if (body.action !== 'play' || typeof body.text !== 'string' || !body.text.trim()) { reply(400, { error: 'invalid control request' }); return }
506
+ replaceWith({ source: 'manual', manual: true, sessionId: body.sessionId == null ? null : String(body.sessionId), turn: Number.isFinite(body.turn) ? body.turn : null, messageId: body.messageId == null ? null : String(body.messageId), text: body.text })
507
+ reply(200, state())
508
+ } catch (e) { reply(e.message === 'request too large' ? 413 : 400, { error: e.message }) }
509
+ },
510
+ }), 'dsh-speak replay control route')
511
+ })
512
+
513
+ ctx.effect(() => () => {
514
+ replacement = null
515
+ clearAndStop()
516
+ for (const socket of speechSockets) { try { socket.close() } catch (e) { /* closed */ } }
517
+ speechSockets.clear()
518
+ try { speechWss.close() } catch (e) { /* closed */ }
519
+ }, 'dsh-speak speech cleanup')
520
+
521
+ // ---- session event handling ----
522
+ let timer = null
523
+ let pendingText = ''
524
+ /** 当前回合内最后一条助手消息文本(turn/end 兜底播报用) */
525
+ let lastText = ''
526
+ /** 已通过节流播报过的文本(防止 turn/end 兜底重复播报) */
527
+ let lastSpokenText = ''
528
+ /** 最后一条助手消息 id(兜底播报时带上) */
529
+ let lastMessageId = null
530
+ /** cancel a pending throttled announcement (default mode, tool-call round) */
531
+ function cancelPending() {
532
+ if (timer) { clearTimeout(timer); timer = null }
533
+ pendingText = ''
534
+ }
535
+
536
+ ctx.on('session/event', (session, event) => {
537
+ try {
538
+ const type = event && event.type
539
+ if (type !== 'assistant/chunk') {
540
+ log('事件 type=', type, 'surfaceOp=', event && event.surfaceOp, 'seq=', event && event.seq)
541
+ }
542
+ // 新回合开始:清空上一回合的兜底状态,避免跨回合残留
543
+ if (type === 'turn/start') {
544
+ cancelPending()
545
+ lastText = ''
546
+ lastSpokenText = ''
547
+ lastMessageId = null
548
+ return
549
+ }
550
+ // tool-call round: ask_user_question announces the parsed question; any
551
+ // other tool call cancels the pending throttled narration
552
+ if (type === 'tool/call') {
553
+ if (event.data && event.data.name === 'ask_user_question' && cfg.announceQuestions) {
554
+ let items = []
555
+ try {
556
+ const args = JSON.parse(event.data.arguments || '{}')
557
+ const questions = Array.isArray(args.questions) ? args.questions : []
558
+ // 每个问题单独入队播报:带"问题N"序号(多问题时)与"选项N"序号
559
+ // (序号用数字,与 UI 的自动编号一致;中文 TTS 自然读成"一/二/三")
560
+ items = questions.map((question, qi) => {
561
+ const mode = question.multi_select ? '多选' : '单选'
562
+ const opts = Array.isArray(question.options) ? question.options : []
563
+ const optText = opts.map((option, oi) => {
564
+ const label = option && option.label ? String(option.label) : ''
565
+ return label ? `选项${oi + 1},${label}` : ''
566
+ }).filter(Boolean).join(';')
567
+ const head = questions.length > 1 ? `问题${qi + 1},` : ''
568
+ // question 文案已含"单选/多选"字样时不再追加模式后缀,避免重复
569
+ const modeSuffix = /单选|多选/.test(question.question || '') ? '' : `(${mode})`
570
+ const body = [question.question || '', modeSuffix, optText ? ',' + optText : ''].join('')
571
+ return (head + body).trim()
572
+ }).filter(Boolean)
573
+ } catch (e) { /* ignore malformed arguments */ }
574
+ if (items.length > 0) {
575
+ cancelPending()
576
+ // 问题已单独播报,标记当前最后文本为已播,避免 turn/end 兜底重复
577
+ lastSpokenText = lastText
578
+ // 多条问题按 FIFO 串行播报,之间停顿 cfg.questionGapMs(默认 2 秒)
579
+ const gap = cfg.questionGapMs > 0 && items.length > 1 ? cfg.questionGapMs : 0
580
+ items.forEach((itemText, i) => {
581
+ const item = hostItem('question', session, event, itemText, null)
582
+ item.gapMs = i < items.length - 1 ? gap : 0
583
+ enqueue(item)
584
+ })
585
+ }
586
+ return
587
+ }
588
+ cancelPending()
589
+ return
590
+ }
591
+ // approval requested: announce right away
592
+ if (type === 'approval/asked' && cfg.announceApprovals) {
593
+ cancelPending()
594
+ let reason = String((event.data && event.data.reason) || '')
595
+ if (cfg.stripApprovalPrefix) {
596
+ // 通用剥离行首"动作标签: "前缀(英文动作短语 + 冒号,如
597
+ // "Store decision fact in workspace memory (dsh-speak): <内容>"、
598
+ // "escalate sandbox to danger-full-access: <原因>"),只念冒号后的
599
+ // 具体内容;中文开头或无冒号的 reason 原样保留(如"删除 xxx")。
600
+ reason = reason.replace(/^[A-Za-z][^::\n]*?[::]\s*/, '').trim()
601
+ }
602
+ enqueue(hostItem('approval', session, event, reason || '需要你的审批,请查看界面。', null))
603
+ return
604
+ }
605
+ // 回合结束:兜底播报最终回复(被工具调用取消的节流文本在此补播,
606
+ // 已播过的不重复),随后按需播报"第 N 轮对话完成"可选事件
607
+ if (type === 'turn/end') {
608
+ if (cfg.automaticSpeech && !cfg.queueAllMessages && lastText && lastText !== lastSpokenText) {
609
+ const itemText = lastText
610
+ const itemMessageId = lastMessageId
611
+ cancelPending()
612
+ lastText = ''
613
+ lastSpokenText = itemText
614
+ enqueue(hostItem('automatic', session, event, itemText, itemMessageId))
615
+ }
616
+ if (!cfg.announceTurnEnd) return
617
+ const data = event.data
618
+ const prefix = data && data.turn != null ? `第 ${data.turn} 轮对话` : '本轮对话'
619
+ const kind = data && data.reason && data.reason.kind
620
+ const text = ({ completed: prefix + '完成', aborted: prefix + '中断', interrupted: prefix + '中断', blocked: prefix + '被阻塞', error: prefix + '异常结束', 'max-tokens': prefix + '异常结束' })[kind] || prefix + '结束'
621
+ enqueue(hostItem('turn/end', session, event, text, null))
622
+ return
623
+ }
624
+ if (type === 'command/done' && cfg.announceCommandDone) {
625
+ enqueue(hostItem('command/done', session, event, (event.data && event.data.kind) === 'error' ? '命令执行失败' : '命令执行完成', null))
626
+ return
627
+ }
628
+ if (type === 'goal/change' && cfg.announceGoalChange) {
629
+ const data = event.data
630
+ const objective = data && data.goal && data.goal.objective
631
+ const label = ({ create: '已创建目标', edit: '目标已更新', complete: '目标已完成', pause: '目标已暂停', resume: '目标已恢复', block: '目标已阻塞', clear: '目标已清除' })[data && data.operation] || '目标状态变化'
632
+ const text = objective && ['create', 'edit', 'complete'].includes(data.operation) ? `${label}:${objective.replace(/\s+/g, ' ').trim().slice(0, 40)}` : label
633
+ enqueue(hostItem('goal/change', session, event, text, null))
634
+ return
635
+ }
636
+ if (type === 'tool/result' && cfg.announceToolErrors) {
637
+ const data = event.data
638
+ const err = data && data.error
639
+ // 真实错误标记有两处:结构化失败身份 data.error(name/code),以及结果块上的
640
+ // isError。0.1.2 起 createToolResultMessage 把结果块包进一个 ToolResultBlock
641
+ // ({ type:'tool-result', toolCallId, content:[…], isError }),文字在它嵌套的
642
+ // content 里;更早的版本把 isError 直接放在 text 块上。两种形状都读。
643
+ //
644
+ // 注意:pwsh / bash 把「命令非零退出」当作结果数据上报(`exit code: N`),
645
+ // 不置 isError —— 只有基础设施失败(spawn 错误、abort)才是 isError 结果,
646
+ // 所以失败的命令本身不会播报工具出错。
647
+ const errText = (Array.isArray(data && data.message && data.message.content) ? data.message.content : [])
648
+ .filter(block => block && block.isError === true)
649
+ .map(block => {
650
+ const parts = Array.isArray(block.content) ? block.content : [block]
651
+ return parts.map(part => (part && (part.text || part.code)) || '').filter(Boolean).join(' ')
652
+ })
653
+ .filter(Boolean).join(' ')
654
+ if (err || errText) {
655
+ const detail = (errText || (err && err.code) || (err && err.name) || '').replace(/\s+/g, ' ').trim().slice(0, 60)
656
+ // 详情只在"确实是一句中文描述"时才念:英文模板(Error: / ENOENT / 技术
657
+ // code)对中文用户可读性差,应当截掉。判据是**汉字数量多于拉丁字母数量**,
658
+ // 而不是"含有汉字"——后者会被路径里的中文目录名骗过:
659
+ // `Error: cannot read "D:\...\第二轮测试用的不存在文件.txt"` 含 12 个汉字,
660
+ // 却是纯英文报错(1.8.0 修正)。
661
+ const cjkCount = (detail.match(/[\u4e00-\u9fff]/g) || []).length
662
+ const latinCount = (detail.match(/[A-Za-z]/g) || []).length
663
+ const readable = cjkCount > latinCount ? `:${detail}` : ''
664
+ enqueue(hostItem('tool/result', session, event, `工具调用出错${readable}`, null))
665
+ }
666
+ return
667
+ }
668
+ if (type === 'todo/write' && cfg.announceTodoWrite) {
669
+ const todos = Array.isArray(event.data && event.data.todos) ? event.data.todos : []
670
+ const done = todos.filter(t => t && t.status === 'completed').length
671
+ enqueue(hostItem('todo/write', session, event, `待办已更新:${done}/${todos.length} 完成`, null))
672
+ return
673
+ }
674
+ if (!event || type !== 'assistant/message') return
675
+ if (event.surfaceOp && event.surfaceOp !== 'append') return
676
+ const message = event.data && (event.data.message || event.data)
677
+ const text = visibleText(message)
678
+ if (!text.trim()) return
679
+
680
+ // queueAllMessages mode (PR #2): enqueue every assistant message now
681
+ if (cfg.queueAllMessages && cfg.automaticSpeech) {
682
+ enqueue(hostItem('automatic', session, event, text, message && message.id))
683
+ return
684
+ }
685
+ // default mode: throttle/merge the final reply; a tool/call cancels it,
686
+ // and turn/end 兜底补播 lastText(见上方 turn/end 分支)
687
+ cancelPending()
688
+ pendingText = text
689
+ lastText = text
690
+ lastMessageId = message && message.id ? String(message.id) : null
691
+ timer = setTimeout(() => {
692
+ if (!pendingText) return
693
+ const itemText = pendingText
694
+ pendingText = ''
695
+ timer = null
696
+ lastSpokenText = itemText
697
+ enqueue(hostItem('automatic', session, event, itemText, lastMessageId))
698
+ }, cfg.throttleMs)
699
+ } catch (e) { log('session event speech error:', e.message) }
700
+ })
701
+ },
702
+ }