dsh-speak 1.7.4 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,561 +1,614 @@
1
- // speech-hook.js — DSH web adapter: voice-announce assistant activity
2
- // ==============================================================================
3
- // Listens to the session event stream (session/event), extracts the final reply
4
- // text, and hands it to the speech engine (engine/speak.ps1 on Windows,
5
- // engine/speak.sh on macOS) through a hidden, non-blocking child process.
6
- //
7
- // Since 1.7.0 this is the merged host of the original dsh-speak behavior and
8
- // victorwads' PR #2 (turn-level replay + host FIFO speech queue + WebSocket
9
- // state sync + native Speak settings page):
10
- //
11
- // * a host-owned FIFO speech queue: only one native speech process runs at a
12
- // time; queued items continue automatically when the current one finishes
13
- // * every eligible item (final reply, approvals, questions, optional events,
14
- // manual replay) is enqueued, so the WebSocket state (which message is
15
- // speaking, queue length) is always truthful — even for automatic replies
16
- // * `queueAllMessages` (default off) switches between two automatic modes:
17
- // - off (default): final replies are throttled/merged as before, plus the
18
- // optional event announcements; tool calls cancel pending narration
19
- // - on: every assistant/message is enqueued immediately as it arrives
20
- // * a `/dsh-speak/control` POST route (play/stop/status) and a
21
- // `/dsh-speak/ws` WebSocket publish the authoritative speech state
22
- // * a `dsh-speak` settings namespace via installSettingsSection; schema
23
- // defaults → patch config → UI user layer
24
- // * `enabled` master switch: when off, nothing is ever enqueued (no sound)
25
- //
26
- // Trigger semantics:
27
- // * assistant/message with a `text` block is announced (reasoning / tool_use
28
- // blocks are skipped)
29
- // * a tool/call to `ask_user_question` announces the parsed question; other
30
- // tool calls cancel the pending throttled announcement (default mode)
31
- // * `approval/asked` is announced immediately (reason, else a fixed prompt)
32
- // * optional events (turn/end, command/done, goal/change, tool/result errors,
33
- // todo/write) are announced when their toggle is on (default off)
34
- //
35
- // Configuration — prefer the Web UI (Settings → dsh-speak settings) or the
36
- // profile patch `config` block (see README.md). All keys resolve as
37
- // schema default → patch config → UI user layer.
38
- 'use strict'
39
-
40
- const { spawn } = require('child_process')
41
- const { createRequire } = require('module')
42
- const { WebSocketServer, WebSocket } = require('ws')
43
- const fs = require('fs')
44
- const os = require('os')
45
- const path = require('path')
46
-
47
- const LOG = path.join(os.tmpdir(), 'dsh-speech-hook.log')
48
- function log(...args) {
49
- try { fs.appendFileSync(LOG, `[${new Date().toISOString()}] ${args.join(' ')}\n`) } catch (e) { /* ignore */ }
50
- }
51
-
52
- const ENGINE_NAME = process.platform === 'darwin' ? 'speak.sh' : 'speak.ps1'
53
- // Settings namespace of this plugin (lowercase kebab-case; must match the
54
- // browser card's namespace in client/client.js).
55
- const SETTINGS_NS = 'dsh-speak'
56
-
57
- /**
58
- * Locate the engine script:
59
- * 1. explicit override (config `engine`)
60
- * 2. <this package>/engine/<speak.ps1|speak.sh> — repo checkout or profile install
61
- * 3. legacy file-copy location (~/.dsh/hooks/<speak.ps1|speak.sh>)
62
- */
63
- function resolveEngine(override) {
64
- if (override) return override
65
- const bundled = path.join(__dirname, '..', '..', 'engine', ENGINE_NAME)
66
- if (fs.existsSync(bundled)) return bundled
67
- return path.join(os.homedir(), '.dsh', 'hooks', ENGINE_NAME)
68
- }
69
-
70
- // Platform-aware defaults: macOS `say` has no per-utterance ceiling, so
71
- // `maxChars` defaults to 0 (unlimited) there; Windows keeps the safe 300.
72
- const DEFAULT_MAX_CHARS = process.platform === 'darwin' ? 0 : 300
73
-
74
- // ---------------------------------------------------------------------------
75
- // Settings namespace (best-effort; see installSettingsSection in dsh-settings)
76
- // ---------------------------------------------------------------------------
77
- // The schema mirrors every config key. Values resolve as:
78
- // schema default → patch `config` (base) → user settings layer (the UI).
79
- const SCHEMA_DEFAULTS = {
80
- enabled: true,
81
- automaticSpeech: true,
82
- cleanMarkdownFormatting: true,
83
- readInlineCode: true,
84
- codeBlocks: 'smart',
85
- codeBlockMaxChars: 300,
86
- codeBlockReplacementText: 'You can see the code in our history.',
87
- queueAllMessages: false,
88
- throttleMs: 1500,
89
- replayFullRead: false,
90
- engine: '',
91
- announceApprovals: true,
92
- announceQuestions: true,
93
- stripApprovalPrefix: true,
94
- questionGapMs: 2000,
95
- longTextMode: 'message',
96
- longTextMessage: '本次播报内容较长,请自行阅读。',
97
- maxChars: DEFAULT_MAX_CHARS,
98
- volume: 50,
99
- rate: 0,
100
- announceTurnEnd: false,
101
- announceCommandDone: false,
102
- announceGoalChange: false,
103
- announceToolErrors: false,
104
- announceTodoWrite: false,
105
- }
106
-
107
- /**
108
- * Resolve the raw settings value into the mutable `cfg` the queue reads.
109
- * Kept as a pure function so both the initial apply and settings onChange use
110
- * the same normalization (engine re-resolution, platform maxChars default).
111
- */
112
- function resolveConfig(value) {
113
- value = value || {}
114
- return {
115
- enabled: value.enabled !== false,
116
- automaticSpeech: value.automaticSpeech !== false,
117
- cleanMarkdownFormatting: value.cleanMarkdownFormatting !== false,
118
- readInlineCode: value.readInlineCode !== false,
119
- codeBlocks: ['all', 'smart', 'replace'].includes(value.codeBlocks) ? value.codeBlocks : 'smart',
120
- codeBlockMaxChars: Number(value.codeBlockMaxChars != null ? value.codeBlockMaxChars : 300),
121
- codeBlockReplacementText: String(value.codeBlockReplacementText || 'You can see the code in our history.'),
122
- queueAllMessages: value.queueAllMessages === true,
123
- throttleMs: Number(value.throttleMs != null ? value.throttleMs : 1500) || 1500,
124
- replayFullRead: value.replayFullRead === true,
125
- engine: resolveEngine(value.engine || ''),
126
- announceApprovals: value.announceApprovals !== false,
127
- announceQuestions: value.announceQuestions !== false,
128
- stripApprovalPrefix: value.stripApprovalPrefix !== false,
129
- questionGapMs: Math.max(0, Number(value.questionGapMs != null ? value.questionGapMs : 2000)) || 0,
130
- longTextMode: value.longTextMode === 'heading' ? 'heading' : 'message',
131
- longTextMessage: String(value.longTextMessage || SCHEMA_DEFAULTS.longTextMessage),
132
- maxChars: Number(value.maxChars != null ? value.maxChars : DEFAULT_MAX_CHARS) || 0,
133
- volume: Number(value.volume != null ? value.volume : 50) || 50,
134
- rate: Number(value.rate != null ? value.rate : 0) || 0,
135
- announceTurnEnd: value.announceTurnEnd === true,
136
- announceCommandDone: value.announceCommandDone === true,
137
- announceGoalChange: value.announceGoalChange === true,
138
- announceToolErrors: value.announceToolErrors === true,
139
- announceTodoWrite: value.announceTodoWrite === true,
140
- }
141
- }
142
-
143
- /**
144
- * Build the settings schema + entry for installSettingsSection. Best-effort:
145
- * any failure (missing peer packages) returns null and the plugin keeps the
146
- * patch config. The registration itself happens on a timer tick in apply.
147
- */
148
- function buildSettingsNamespace(ctx, patch) {
149
- try {
150
- const profileRequire = createRequire(ctx.baseUrl || __filename)
151
- const z = profileRequire('@deepseek-ai/schemastery')
152
- const schema = z.object({
153
- enabled: z.boolean().default(true),
154
- automaticSpeech: z.boolean().default(true),
155
- cleanMarkdownFormatting: z.boolean().default(true),
156
- readInlineCode: z.boolean().default(true),
157
- codeBlocks: z.union(['all', 'smart', 'replace']).default('smart'),
158
- codeBlockMaxChars: z.natural().default(300),
159
- codeBlockReplacementText: z.string().default('You can see the code in our history.'),
160
- queueAllMessages: z.boolean().default(false),
161
- throttleMs: z.natural().default(1500),
162
- replayFullRead: z.boolean().default(false),
163
- engine: z.string().default(''),
164
- announceApprovals: z.boolean().default(true),
165
- announceQuestions: z.boolean().default(true),
166
- stripApprovalPrefix: z.boolean().default(true),
167
- questionGapMs: z.natural().default(2000),
168
- longTextMode: z.union(['message', 'heading']).default('message'),
169
- longTextMessage: z.string().default('本次播报内容较长,请自行阅读。'),
170
- maxChars: z.natural().default(DEFAULT_MAX_CHARS),
171
- volume: z.natural().default(50),
172
- rate: z.number().default(0),
173
- announceTurnEnd: z.boolean().default(false),
174
- announceCommandDone: z.boolean().default(false),
175
- announceGoalChange: z.boolean().default(false),
176
- announceToolErrors: z.boolean().default(false),
177
- announceTodoWrite: z.boolean().default(false),
178
- })
179
- return { schema, entry: { ...SCHEMA_DEFAULTS, ...(patch || {}) } }
180
- } catch (e) {
181
- log('settings 依赖不可用,跳过 settings namespace 注册:', e && e.message)
182
- return null
183
- }
184
- }
185
-
186
- module.exports = {
187
- apply(ctx, config) {
188
- config = config || {}
189
- let cfg = resolveConfig(config)
190
-
191
- // Register the settings namespace on a timer tick so apply never blocks;
192
- // cfg is replaced wholesale on settings changes.
193
- ctx.inject(['timer'], timerCtx => {
194
- timerCtx.timer.timeout(() => {
195
- const prepared = buildSettingsNamespace(ctx, config)
196
- if (!prepared) return
197
- let settingsModule
198
- try {
199
- const profileRequire = createRequire(ctx.baseUrl || __filename)
200
- settingsModule = profileRequire('@deepseek-ai/dsh-settings')
201
- } catch (e) {
202
- log('dsh-settings 不可用,跳过 settings namespace 注册:', e && e.message)
203
- return
204
- }
205
- // Keep a live source getter: installSettingsSection passes the resolved
206
- // scope thunk to setSource, and onChange must re-derive cfg from it
207
- // (installSettingsSection only calls setSource on attach/detach).
208
- let settingsSource = () => prepared.entry
209
- settingsModule.installSettingsSection(ctx, settingsModule.settingsNamespace(SETTINGS_NS), prepared.schema, prepared.entry, {
210
- setSource: source => { settingsSource = source; cfg = resolveConfig(source()) },
211
- onChange: () => {
212
- try { cfg = resolveConfig(settingsSource()) } catch (e) { log('settings 变更应用失败:', e && e.message) }
213
- log('settings 变更已应用; cfg=', JSON.stringify(cfg))
214
- },
215
- })
216
- }, 0)
217
- })
218
-
219
- // ---- host-owned FIFO speech queue + WebSocket state sync (PR #2) ----
220
- let activeSpeech = null
221
- let speechToken = 0
222
- let replacement = null
223
- /** 队列项播完后的停顿定时器(多问题提问之间的间隔) */
224
- let gapTimer = null
225
- const speechQueue = []
226
- const speechSockets = new Set()
227
- const speechWss = new WebSocketServer({ noServer: true })
228
-
229
- function state() {
230
- const item = activeSpeech && activeSpeech.item
231
- return {
232
- type: 'speech-state',
233
- speaking: item !== undefined && item !== null,
234
- sessionId: item ? item.sessionId : null,
235
- turn: item ? item.turn : null,
236
- messageId: item ? item.messageId : null,
237
- source: item ? item.source : null,
238
- queueLength: speechQueue.length,
239
- }
240
- }
241
- function publishState() {
242
- const payload = JSON.stringify(state())
243
- for (const socket of speechSockets) {
244
- if (socket.readyState === WebSocket.OPEN) {
245
- try { socket.send(payload) } catch (e) { speechSockets.delete(socket) }
246
- }
247
- }
248
- }
249
- function removeTemp(tmp) { try { fs.unlinkSync(tmp) } catch (e) { /* already removed */ } }
250
-
251
- function startOne(item) {
252
- // master switch: nothing is ever spoken while disabled
253
- if (!cfg.enabled) { log('总开关关闭,跳过播报(文本长度:', item.text.length, ')'); return false }
254
- if (!item || !item.text.trim() || activeSpeech) return false
255
- const tmp = path.join(os.tmpdir(), `dsh-speech-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.txt`)
256
- try { fs.writeFileSync(tmp, item.text, 'utf8') } catch (e) { log('write temp failed:', e.message); return false }
257
- log('speech start', item.source, item.sessionId || '-', item.turn == null ? '-' : item.turn, item.messageId || '-', item.text.slice(0, 80))
258
- let child
259
- // 手动重播完整朗读:replayFullRead 打开时,重播跳过 heading 截断完整朗读
260
- const fullRead = item.manual === true && cfg.replayFullRead === true
261
- if (process.platform === 'darwin') {
262
- const args = ['-f', tmp, '-m', String(cfg.maxChars), '-M', cfg.longTextMode, '-l', cfg.longTextMessage, '-C', cfg.cleanMarkdownFormatting ? '1' : '0', '-I', cfg.readInlineCode ? '1' : '0', '-B', cfg.codeBlocks, '-K', String(cfg.codeBlockMaxChars), '-R', cfg.codeBlockReplacementText]
263
- if (cfg.rate > 0) args.push('-r', String(cfg.rate))
264
- if (fullRead) args.push('-F')
265
- child = spawn('/bin/bash', [cfg.engine].concat(args), { detached: true, stdio: 'ignore' })
266
- } else {
267
- // Windows 语速 = SAPI 刻度(-10 到 10,0 = 正常;负值变慢),直接透传,
268
- // 不要把 0/负值替换成 1(否则无法回到正常语速/减慢)
269
- child = spawn('powershell.exe', ['-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', cfg.engine, '-File', tmp, '-Volume', String(cfg.volume), '-Rate', String(cfg.rate), '-MaxChars', String(cfg.maxChars), '-LongTextMode', cfg.longTextMode, '-LongTextMessage', cfg.longTextMessage, '-CleanMarkdownFormatting', cfg.cleanMarkdownFormatting ? '1' : '0', '-ReadInlineCode', cfg.readInlineCode ? '1' : '0', '-CodeBlocks', cfg.codeBlocks, '-CodeBlockMaxChars', String(cfg.codeBlockMaxChars), '-CodeBlockReplacementText', cfg.codeBlockReplacementText, '-FullRead', fullRead ? '1' : '0'], { windowsHide: true, stdio: 'ignore' })
270
- }
271
- const token = ++speechToken
272
- activeSpeech = { process: child, tmp, token, item, gapMs: item.gapMs || 0 }
273
- publishState()
274
- const settle = () => {
275
- removeTemp(tmp)
276
- if (!activeSpeech || activeSpeech.token !== token) return
277
- const gap = activeSpeech.gapMs || 0
278
- activeSpeech = null
279
- publishState()
280
- const proceed = () => {
281
- gapTimer = null
282
- if (replacement) {
283
- const next = replacement
284
- replacement = null
285
- startOne(next)
286
- } else {
287
- startNext()
288
- }
289
- }
290
- // 队列项之间可配置停顿(如多个提问之间留 2 秒)
291
- if (gap > 0) {
292
- gapTimer = setTimeout(proceed, gap)
293
- } else {
294
- proceed()
295
- }
296
- }
297
- child.once('exit', settle)
298
- child.once('error', settle)
299
- return true
300
- }
301
- function startNext() {
302
- if (activeSpeech || replacement) return
303
- const item = speechQueue.shift()
304
- if (!item) { publishState(); return }
305
- publishState()
306
- startOne(item)
307
- }
308
- function enqueue(item) {
309
- if (!item || !item.text || !item.text.trim()) return
310
- speechQueue.push(item)
311
- publishState()
312
- startNext()
313
- }
314
- function stopActive() {
315
- const active = activeSpeech
316
- if (!active) return false
317
- try {
318
- if (process.platform === 'darwin' && active.process.pid) process.kill(-active.process.pid, 'SIGTERM')
319
- else active.process.kill()
320
- } catch (e) { log('stop speech failed:', e.message) }
321
- return true
322
- }
323
- function clearAndStop() {
324
- if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
325
- speechQueue.length = 0
326
- publishState()
327
- return stopActive()
328
- }
329
- function replaceWith(item) {
330
- if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
331
- speechQueue.length = 0
332
- replacement = item
333
- publishState()
334
- if (!activeSpeech) {
335
- const next = replacement
336
- replacement = null
337
- startOne(next)
338
- return
339
- }
340
- stopActive()
341
- }
342
- function visibleText(message) {
343
- if (!message) return ''
344
- if (typeof message.content === 'string') return message.content
345
- if (!Array.isArray(message.content)) return ''
346
- return message.content.filter(block => block && block.type === 'text' && typeof block.text === 'string').map(block => block.text).join('')
347
- }
348
- function hostItem(source, session, event, text, messageId) {
349
- const sessionValue = session && (session.id != null ? session.id : session.sessionId)
350
- return {
351
- source,
352
- sessionId: sessionValue != null ? String(sessionValue) : null,
353
- turn: event && event.data && Number.isFinite(event.data.turn) ? event.data.turn : null,
354
- messageId: messageId == null ? null : String(messageId),
355
- text,
356
- }
357
- }
358
-
359
- // ---- WebSocket + control route (PR #2) ----
360
- ctx.inject(['webServer'], webCtx => {
361
- webCtx.effect(() => webCtx.webServer.registerUpgrade({
362
- path: '/dsh-speak/ws',
363
- handler: (req, socket, head) => speechWss.handleUpgrade(req, socket, head, client => {
364
- speechSockets.add(client)
365
- client.once('close', () => speechSockets.delete(client))
366
- client.once('error', () => speechSockets.delete(client))
367
- try { client.send(JSON.stringify(state())) } catch (e) { speechSockets.delete(client) }
368
- }),
369
- }), 'dsh-speak speech-state websocket')
370
- webCtx.effect(() => webCtx.webServer.register({
371
- kind: 'exact', path: '/dsh-speak/control', handler: async (req, res) => {
372
- const reply = (status, value) => { res.writeHead(status, { 'Content-Type': 'application/json; charset=utf-8' }); res.end(JSON.stringify(value)) }
373
- if (req.method !== 'POST' || !String(req.headers['content-type'] || '').startsWith('application/json')) { reply(405, { error: 'POST application/json required' }); return }
374
- try {
375
- const chunks = []; let size = 0
376
- for await (const chunk of req) { size += chunk.length; if (size > 1024 * 1024) throw new Error('request too large'); chunks.push(chunk) }
377
- const body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}')
378
- if (body.action === 'status') { reply(200, state()); return }
379
- if (body.action === 'stop') {
380
- replacement = null
381
- clearAndStop()
382
- reply(200, state())
383
- return
384
- }
385
- if (body.action !== 'play' || typeof body.text !== 'string' || !body.text.trim()) { reply(400, { error: 'invalid control request' }); return }
386
- replaceWith({ source: 'manual', manual: true, sessionId: body.sessionId == null ? null : String(body.sessionId), turn: Number.isFinite(body.turn) ? body.turn : null, messageId: body.messageId == null ? null : String(body.messageId), text: body.text })
387
- reply(200, state())
388
- } catch (e) { reply(e.message === 'request too large' ? 413 : 400, { error: e.message }) }
389
- },
390
- }), 'dsh-speak replay control route')
391
- })
392
-
393
- ctx.effect(() => () => {
394
- replacement = null
395
- clearAndStop()
396
- for (const socket of speechSockets) { try { socket.close() } catch (e) { /* closed */ } }
397
- speechSockets.clear()
398
- try { speechWss.close() } catch (e) { /* closed */ }
399
- }, 'dsh-speak speech cleanup')
400
-
401
- // ---- session event handling ----
402
- let timer = null
403
- let pendingText = ''
404
- /** 当前回合内最后一条助手消息文本(turn/end 兜底播报用) */
405
- let lastText = ''
406
- /** 已通过节流播报过的文本(防止 turn/end 兜底重复播报) */
407
- let lastSpokenText = ''
408
- /** 最后一条助手消息 id(兜底播报时带上) */
409
- let lastMessageId = null
410
- /** cancel a pending throttled announcement (default mode, tool-call round) */
411
- function cancelPending() {
412
- if (timer) { clearTimeout(timer); timer = null }
413
- pendingText = ''
414
- }
415
-
416
- ctx.on('session/event', (session, event) => {
417
- try {
418
- const type = event && event.type
419
- if (type !== 'assistant/chunk') {
420
- log('事件 type=', type, 'surfaceOp=', event && event.surfaceOp, 'seq=', event && event.seq)
421
- }
422
- // 新回合开始:清空上一回合的兜底状态,避免跨回合残留
423
- if (type === 'turn/start') {
424
- cancelPending()
425
- lastText = ''
426
- lastSpokenText = ''
427
- lastMessageId = null
428
- return
429
- }
430
- // tool-call round: ask_user_question announces the parsed question; any
431
- // other tool call cancels the pending throttled narration
432
- if (type === 'tool/call') {
433
- if (event.data && event.data.name === 'ask_user_question' && cfg.announceQuestions) {
434
- let items = []
435
- try {
436
- const args = JSON.parse(event.data.arguments || '{}')
437
- const questions = Array.isArray(args.questions) ? args.questions : []
438
- // 每个问题单独入队播报:带"问题N"序号(多问题时)与"选项N"序号
439
- // (序号用数字,与 UI 的自动编号一致;中文 TTS 自然读成"一/二/三")
440
- items = questions.map((question, qi) => {
441
- const mode = question.multi_select ? '多选' : '单选'
442
- const opts = Array.isArray(question.options) ? question.options : []
443
- const optText = opts.map((option, oi) => {
444
- const label = option && option.label ? String(option.label) : ''
445
- return label ? `选项${oi + 1},${label}` : ''
446
- }).filter(Boolean).join(';')
447
- const head = questions.length > 1 ? `问题${qi + 1},` : ''
448
- // question 文案已含"单选/多选"字样时不再追加模式后缀,避免重复
449
- const modeSuffix = /单选|多选/.test(question.question || '') ? '' : `(${mode})`
450
- const body = [question.question || '', modeSuffix, optText ? ',' + optText : ''].join('')
451
- return (head + body).trim()
452
- }).filter(Boolean)
453
- } catch (e) { /* ignore malformed arguments */ }
454
- if (items.length > 0) {
455
- cancelPending()
456
- // 问题已单独播报,标记当前最后文本为已播,避免 turn/end 兜底重复
457
- lastSpokenText = lastText
458
- // 多条问题按 FIFO 串行播报,之间停顿 cfg.questionGapMs(默认 2 秒)
459
- const gap = cfg.questionGapMs > 0 && items.length > 1 ? cfg.questionGapMs : 0
460
- items.forEach((itemText, i) => {
461
- const item = hostItem('question', session, event, itemText, null)
462
- item.gapMs = i < items.length - 1 ? gap : 0
463
- enqueue(item)
464
- })
465
- }
466
- return
467
- }
468
- cancelPending()
469
- return
470
- }
471
- // approval requested: announce right away
472
- if (type === 'approval/asked' && cfg.announceApprovals) {
473
- cancelPending()
474
- let reason = String((event.data && event.data.reason) || '')
475
- if (cfg.stripApprovalPrefix) reason = reason.replace(/^escalate sandbox to danger-full-access\s*:\s*/i, '').trim()
476
- enqueue(hostItem('approval', session, event, reason || '需要你的审批,请查看界面。', null))
477
- return
478
- }
479
- // 回合结束:兜底播报最终回复(被工具调用取消的节流文本在此补播,
480
- // 已播过的不重复),随后按需播报"第 N 轮对话完成"可选事件
481
- if (type === 'turn/end') {
482
- if (cfg.automaticSpeech && !cfg.queueAllMessages && lastText && lastText !== lastSpokenText) {
483
- const itemText = lastText
484
- const itemMessageId = lastMessageId
485
- cancelPending()
486
- lastText = ''
487
- lastSpokenText = itemText
488
- enqueue(hostItem('automatic', session, event, itemText, itemMessageId))
489
- }
490
- if (!cfg.announceTurnEnd) return
491
- const data = event.data
492
- const prefix = data && data.turn != null ? `第 ${data.turn} 轮对话` : '本轮对话'
493
- const kind = data && data.reason && data.reason.kind
494
- const text = ({ completed: prefix + '完成', aborted: prefix + '中断', interrupted: prefix + '中断', blocked: prefix + '被阻塞', error: prefix + '异常结束', 'max-tokens': prefix + '异常结束' })[kind] || prefix + '结束'
495
- enqueue(hostItem('turn/end', session, event, text, null))
496
- return
497
- }
498
- if (type === 'command/done' && cfg.announceCommandDone) {
499
- enqueue(hostItem('command/done', session, event, (event.data && event.data.kind) === 'error' ? '命令执行失败' : '命令执行完成', null))
500
- return
501
- }
502
- if (type === 'goal/change' && cfg.announceGoalChange) {
503
- const data = event.data
504
- const objective = data && data.goal && data.goal.objective
505
- const label = ({ create: '已创建目标', edit: '目标已更新', complete: '目标已完成', pause: '目标已暂停', resume: '目标已恢复', block: '目标已阻塞', clear: '目标已清除' })[data && data.operation] || '目标状态变化'
506
- const text = objective && ['create', 'edit', 'complete'].includes(data.operation) ? `${label}:${objective.replace(/\s+/g, ' ').trim().slice(0, 40)}` : label
507
- enqueue(hostItem('goal/change', session, event, text, null))
508
- return
509
- }
510
- if (type === 'tool/result' && cfg.announceToolErrors) {
511
- const data = event.data
512
- const err = data && data.error
513
- // 真实错误标记:error 字段(name/code)或 message 内容块 isError === true
514
- // (pwsh 等工具失败时没有 error 字段,错误文本在 isError 内容块里)
515
- const errText = (Array.isArray(data && data.message && data.message.content) ? data.message.content : [])
516
- .filter(block => block && block.isError === true)
517
- .map(block => block.text || block.code || '').filter(Boolean).join(' ')
518
- if (err || errText) {
519
- const detail = (errText || (err && err.code) || (err && err.name) || '').replace(/\s+/g, ' ').trim().slice(0, 60)
520
- // 纯英文错误详情(PowerShell 固定模板 / 技术 code)对中文用户可读性差,
521
- // 播报时截掉,只保留含中文的详情(如"文件不存在")
522
- const readable = /[\u4e00-\u9fff]/.test(detail) ? `:${detail}` : ''
523
- enqueue(hostItem('tool/result', session, event, `工具调用出错${readable}`, null))
524
- }
525
- return
526
- }
527
- if (type === 'todo/write' && cfg.announceTodoWrite) {
528
- const todos = Array.isArray(event.data && event.data.todos) ? event.data.todos : []
529
- const done = todos.filter(t => t && t.status === 'completed').length
530
- enqueue(hostItem('todo/write', session, event, `待办已更新:${done}/${todos.length} 完成`, null))
531
- return
532
- }
533
- if (!event || type !== 'assistant/message') return
534
- if (event.surfaceOp && event.surfaceOp !== 'append') return
535
- const message = event.data && (event.data.message || event.data)
536
- const text = visibleText(message)
537
- if (!text.trim()) return
538
-
539
- // queueAllMessages mode (PR #2): enqueue every assistant message now
540
- if (cfg.queueAllMessages && cfg.automaticSpeech) {
541
- enqueue(hostItem('automatic', session, event, text, message && message.id))
542
- return
543
- }
544
- // default mode: throttle/merge the final reply; a tool/call cancels it,
545
- // and turn/end 兜底补播 lastText(见上方 turn/end 分支)
546
- cancelPending()
547
- pendingText = text
548
- lastText = text
549
- lastMessageId = message && message.id ? String(message.id) : null
550
- timer = setTimeout(() => {
551
- if (!pendingText) return
552
- const itemText = pendingText
553
- pendingText = ''
554
- timer = null
555
- lastSpokenText = itemText
556
- enqueue(hostItem('automatic', session, event, itemText, lastMessageId))
557
- }, cfg.throttleMs)
558
- } catch (e) { log('session event speech error:', e.message) }
559
- })
560
- },
561
- }
1
+ // speech-hook.js — DSH web adapter: voice-announce assistant activity
2
+ // ==============================================================================
3
+ // Listens to the session event stream (session/event), extracts the final reply
4
+ // text, and hands it to the speech engine (engine/speak.ps1 on Windows,
5
+ // engine/speak.sh on macOS) through a hidden, non-blocking child process.
6
+ //
7
+ // Since 1.7.0 this is the merged host of the original dsh-speak behavior and
8
+ // victorwads' PR #2 (turn-level replay + host FIFO speech queue + WebSocket
9
+ // state sync + native Speak settings page):
10
+ //
11
+ // * a host-owned FIFO speech queue: only one native speech process runs at a
12
+ // time; queued items continue automatically when the current one finishes
13
+ // * every eligible item (final reply, approvals, questions, optional events,
14
+ // manual replay) is enqueued, so the WebSocket state (which message is
15
+ // speaking, queue length) is always truthful — even for automatic replies
16
+ // * `queueAllMessages` (default off) switches between two automatic modes:
17
+ // - off (default): final replies are throttled/merged as before, plus the
18
+ // optional event announcements; tool calls cancel pending narration
19
+ // - on: every assistant/message is enqueued immediately as it arrives
20
+ // * a `/dsh-speak/control` POST route (play/stop/status) and a
21
+ // `/dsh-speak/ws` WebSocket publish the authoritative speech state
22
+ // * a `dsh-speak` settings namespace registered through the settings SERVICE
23
+ // (`ctx.inject(['settings'])` → `settings.register`); schema defaults →
24
+ // patch config → UI user layer
25
+ // * `enabled` master switch: when off, nothing is ever enqueued (no sound)
26
+ //
27
+ // Trigger semantics:
28
+ // * assistant/message with a `text` block is announced (reasoning / tool_use
29
+ // blocks are skipped)
30
+ // * a tool/call to `ask_user_question` announces the parsed question; other
31
+ // tool calls cancel the pending throttled announcement (default mode)
32
+ // * `approval/asked` is announced immediately (reason, else a fixed prompt)
33
+ // * optional events (turn/end, command/done, goal/change, tool/result errors,
34
+ // todo/write) are announced when their toggle is on (default off)
35
+ //
36
+ // Configuration — prefer the Web UI (Settings → dsh-speak settings) or the
37
+ // profile patch `config` block (see README.md). All keys resolve as
38
+ // schema default → patch config → UI user layer.
39
+ 'use strict'
40
+
41
+ const { spawn } = require('child_process')
42
+ const { createRequire } = require('module')
43
+ const { WebSocketServer, WebSocket } = require('ws')
44
+ const fs = require('fs')
45
+ const os = require('os')
46
+ const path = require('path')
47
+
48
+ const LOG = path.join(os.tmpdir(), 'dsh-speech-hook.log')
49
+ function log(...args) {
50
+ try { fs.appendFileSync(LOG, `[${new Date().toISOString()}] ${args.join(' ')}\n`) } catch (e) { /* ignore */ }
51
+ }
52
+
53
+ const ENGINE_NAME = process.platform === 'darwin' ? 'speak.sh' : 'speak.ps1'
54
+ // Settings namespace of this plugin (lowercase kebab-case; must match the
55
+ // browser card's namespace in client/client.js).
56
+ const SETTINGS_NS = 'dsh-speak'
57
+
58
+ /**
59
+ * Locate the engine script:
60
+ * 1. explicit override (config `engine`)
61
+ * 2. <this package>/engine/<speak.ps1|speak.sh> — repo checkout or profile install
62
+ * 3. legacy file-copy location (~/.dsh/hooks/<speak.ps1|speak.sh>)
63
+ */
64
+ function resolveEngine(override) {
65
+ if (override) return override
66
+ const bundled = path.join(__dirname, '..', '..', 'engine', ENGINE_NAME)
67
+ if (fs.existsSync(bundled)) return bundled
68
+ return path.join(os.homedir(), '.dsh', 'hooks', ENGINE_NAME)
69
+ }
70
+
71
+ // Platform-aware defaults: macOS `say` has no per-utterance ceiling, so
72
+ // `maxChars` defaults to 0 (unlimited) there; Windows keeps the safe 300.
73
+ const DEFAULT_MAX_CHARS = process.platform === 'darwin' ? 0 : 300
74
+
75
+ // ---------------------------------------------------------------------------
76
+ // Settings namespace (best-effort; registered through the `settings` service)
77
+ // ---------------------------------------------------------------------------
78
+ // The schema mirrors every config key. Values resolve as:
79
+ // schema default → patch `config` (base) → user settings layer (the UI).
80
+ const SCHEMA_DEFAULTS = {
81
+ enabled: true,
82
+ automaticSpeech: true,
83
+ cleanMarkdownFormatting: true,
84
+ readInlineCode: true,
85
+ codeBlocks: 'smart',
86
+ codeBlockMaxChars: 300,
87
+ codeBlockReplacementText: 'You can see the code in our history.',
88
+ queueAllMessages: false,
89
+ throttleMs: 1500,
90
+ replayFullRead: false,
91
+ engine: '',
92
+ announceApprovals: true,
93
+ announceQuestions: true,
94
+ stripApprovalPrefix: true,
95
+ questionGapMs: 2000,
96
+ longTextMode: 'message',
97
+ longTextMessage: '本次播报内容较长,请自行阅读。',
98
+ maxChars: DEFAULT_MAX_CHARS,
99
+ volume: 50,
100
+ rate: 0,
101
+ announceTurnEnd: false,
102
+ announceCommandDone: false,
103
+ announceGoalChange: false,
104
+ announceToolErrors: false,
105
+ announceTodoWrite: false,
106
+ }
107
+
108
+ /**
109
+ * Resolve the raw settings value into the mutable `cfg` the queue reads.
110
+ * Kept as a pure function so both the initial apply and settings onChange use
111
+ * the same normalization (engine re-resolution, platform maxChars default).
112
+ */
113
+ function resolveConfig(value) {
114
+ value = value || {}
115
+ return {
116
+ enabled: value.enabled !== false,
117
+ automaticSpeech: value.automaticSpeech !== false,
118
+ cleanMarkdownFormatting: value.cleanMarkdownFormatting !== false,
119
+ readInlineCode: value.readInlineCode !== false,
120
+ codeBlocks: ['all', 'smart', 'replace'].includes(value.codeBlocks) ? value.codeBlocks : 'smart',
121
+ codeBlockMaxChars: Number(value.codeBlockMaxChars != null ? value.codeBlockMaxChars : 300),
122
+ codeBlockReplacementText: String(value.codeBlockReplacementText || 'You can see the code in our history.'),
123
+ queueAllMessages: value.queueAllMessages === true,
124
+ throttleMs: Number(value.throttleMs != null ? value.throttleMs : 1500) || 1500,
125
+ replayFullRead: value.replayFullRead === true,
126
+ engine: resolveEngine(value.engine || ''),
127
+ announceApprovals: value.announceApprovals !== false,
128
+ announceQuestions: value.announceQuestions !== false,
129
+ stripApprovalPrefix: value.stripApprovalPrefix !== false,
130
+ questionGapMs: Math.max(0, Number(value.questionGapMs != null ? value.questionGapMs : 2000)) || 0,
131
+ longTextMode: value.longTextMode === 'heading' ? 'heading' : 'message',
132
+ longTextMessage: String(value.longTextMessage || SCHEMA_DEFAULTS.longTextMessage),
133
+ maxChars: Number(value.maxChars != null ? value.maxChars : DEFAULT_MAX_CHARS) || 0,
134
+ volume: Number(value.volume != null ? value.volume : 50) || 50,
135
+ rate: Number(value.rate != null ? value.rate : 0) || 0,
136
+ announceTurnEnd: value.announceTurnEnd === true,
137
+ announceCommandDone: value.announceCommandDone === true,
138
+ announceGoalChange: value.announceGoalChange === true,
139
+ announceToolErrors: value.announceToolErrors === true,
140
+ announceTodoWrite: value.announceTodoWrite === true,
141
+ }
142
+ }
143
+
144
+ /**
145
+ * Resolve a module specifier from the plugin's own location first, then from
146
+ * the booted profile tree. `@deepseek-ai/schemastery` is a peer of this package
147
+ * and lives beside it after an npm/pnpm install; the profile-tree fallback
148
+ * covers the file:// install used by install.ps1 and repo checkouts.
149
+ * @returns the module, or null when neither base resolves it.
150
+ */
151
+ function requirePeer(ctx, spec) {
152
+ const bases = [__filename, ctx.baseUrl].filter(Boolean)
153
+ for (const base of bases) {
154
+ try { return createRequire(base)(spec) } catch (e) { /* try the next base */ }
155
+ }
156
+ return null
157
+ }
158
+
159
+ /**
160
+ * Build the settings schema + entry for the settings namespace. Best-effort:
161
+ * any failure (missing peer packages) returns null and the plugin keeps the
162
+ * patch config.
163
+ */
164
+ function buildSettingsNamespace(ctx, patch) {
165
+ try {
166
+ const z = requirePeer(ctx, '@deepseek-ai/schemastery')
167
+ if (!z) throw new Error('@deepseek-ai/schemastery 不可解析')
168
+ const schema = z.object({
169
+ enabled: z.boolean().default(true),
170
+ automaticSpeech: z.boolean().default(true),
171
+ cleanMarkdownFormatting: z.boolean().default(true),
172
+ readInlineCode: z.boolean().default(true),
173
+ codeBlocks: z.union(['all', 'smart', 'replace']).default('smart'),
174
+ codeBlockMaxChars: z.natural().default(300),
175
+ codeBlockReplacementText: z.string().default('You can see the code in our history.'),
176
+ queueAllMessages: z.boolean().default(false),
177
+ throttleMs: z.natural().default(1500),
178
+ replayFullRead: z.boolean().default(false),
179
+ engine: z.string().default(''),
180
+ announceApprovals: z.boolean().default(true),
181
+ announceQuestions: z.boolean().default(true),
182
+ stripApprovalPrefix: z.boolean().default(true),
183
+ questionGapMs: z.natural().default(2000),
184
+ longTextMode: z.union(['message', 'heading']).default('message'),
185
+ longTextMessage: z.string().default('本次播报内容较长,请自行阅读。'),
186
+ maxChars: z.natural().default(DEFAULT_MAX_CHARS),
187
+ volume: z.natural().default(50),
188
+ rate: z.number().default(0),
189
+ announceTurnEnd: z.boolean().default(false),
190
+ announceCommandDone: z.boolean().default(false),
191
+ announceGoalChange: z.boolean().default(false),
192
+ announceToolErrors: z.boolean().default(false),
193
+ announceTodoWrite: z.boolean().default(false),
194
+ })
195
+ return { schema, entry: { ...SCHEMA_DEFAULTS, ...(patch || {}) } }
196
+ } catch (e) {
197
+ log('settings 依赖不可用,跳过 settings namespace 注册:', e && e.message)
198
+ return null
199
+ }
200
+ }
201
+
202
+ module.exports = {
203
+ apply(ctx, config) {
204
+ config = config || {}
205
+ let cfg = resolveConfig(config)
206
+
207
+ // ---- settings namespace -------------------------------------------------
208
+ // Wire the namespace through the settings SERVICE.
209
+ //
210
+ // dsh 0.1.2-alpha.1 deleted the `installSettingsSection` / `settingsNamespace`
211
+ // convenience exports from `@deepseek-ai/dsh-settings`; what remains — and
212
+ // has not changed since 0.1.0-rc.7 — is the `settings` service itself
213
+ // (`ctx.settings.register(ns, schema, { base })` → `{ get, watch, update,
214
+ // replace }`). Referencing the removed names is fatal: an ESM named import
215
+ // of a deleted export is a module-evaluation SyntaxError that kills the host
216
+ // boot, and a lazy `settingsModule.installSettingsSection(...)` call — what
217
+ // this plugin used to do inside a timer callback — throws
218
+ // `settingsNamespace is not a function` and crashed dsh before it served.
219
+ //
220
+ // `ctx.inject(['settings'])` is the graceful-degradation boundary: on a host
221
+ // with no settings provider the callback never runs and the composed patch
222
+ // config stands as-is.
223
+ const prepared = buildSettingsNamespace(ctx, config)
224
+ if (prepared) {
225
+ ctx.inject(['settings'], scopedCtx => {
226
+ // `scope.get()` is the live resolved value (schema default → patch
227
+ // config → UI user layer), so re-deriving cfg from it on every change
228
+ // is what makes a settings edit take effect without a restart.
229
+ let settingsSource = () => prepared.entry
230
+ const applySettings = () => {
231
+ try { cfg = resolveConfig(settingsSource()) } catch (e) { log('settings 变更应用失败:', e && e.message) }
232
+ }
233
+ try {
234
+ const scope = scopedCtx.settings.register(SETTINGS_NS, prepared.schema, { base: prepared.entry })
235
+ settingsSource = () => scope.get()
236
+ // Unload restores the composed entry, so a disabled plugin cannot
237
+ // leave the queue reading a value nobody can see or change any more.
238
+ scopedCtx.effect(() => () => {
239
+ settingsSource = () => prepared.entry
240
+ applySettings()
241
+ })
242
+ scope.watch(applySettings)
243
+ applySettings()
244
+ log('settings namespace 已注册:', SETTINGS_NS)
245
+ } catch (e) {
246
+ log('settings namespace 注册失败,继续使用 patch config:', e && e.message)
247
+ }
248
+ })
249
+ }
250
+
251
+ // ---- host-owned FIFO speech queue + WebSocket state sync (PR #2) ----
252
+ let activeSpeech = null
253
+ let speechToken = 0
254
+ let replacement = null
255
+ /** 队列项播完后的停顿定时器(多问题提问之间的间隔) */
256
+ let gapTimer = null
257
+ const speechQueue = []
258
+ const speechSockets = new Set()
259
+ const speechWss = new WebSocketServer({ noServer: true })
260
+
261
+ function state() {
262
+ const item = activeSpeech && activeSpeech.item
263
+ return {
264
+ type: 'speech-state',
265
+ speaking: item !== undefined && item !== null,
266
+ sessionId: item ? item.sessionId : null,
267
+ turn: item ? item.turn : null,
268
+ messageId: item ? item.messageId : null,
269
+ source: item ? item.source : null,
270
+ queueLength: speechQueue.length,
271
+ }
272
+ }
273
+ function publishState() {
274
+ const payload = JSON.stringify(state())
275
+ for (const socket of speechSockets) {
276
+ if (socket.readyState === WebSocket.OPEN) {
277
+ try { socket.send(payload) } catch (e) { speechSockets.delete(socket) }
278
+ }
279
+ }
280
+ }
281
+ function removeTemp(tmp) { try { fs.unlinkSync(tmp) } catch (e) { /* already removed */ } }
282
+
283
+ function startOne(item) {
284
+ // master switch: nothing is ever spoken while disabled
285
+ if (!cfg.enabled) { log('总开关关闭,跳过播报(文本长度:', item.text.length, ')'); return false }
286
+ if (!item || !item.text.trim() || activeSpeech) return false
287
+ const tmp = path.join(os.tmpdir(), `dsh-speech-${Date.now()}-${Math.random().toString(36).slice(2, 8)}.txt`)
288
+ try { fs.writeFileSync(tmp, item.text, 'utf8') } catch (e) { log('write temp failed:', e.message); return false }
289
+ log('speech start', item.source, item.sessionId || '-', item.turn == null ? '-' : item.turn, item.messageId || '-', item.text.slice(0, 80))
290
+ let child
291
+ // 手动重播完整朗读:replayFullRead 打开时,重播跳过 heading 截断完整朗读
292
+ const fullRead = item.manual === true && cfg.replayFullRead === true
293
+ if (process.platform === 'darwin') {
294
+ const args = ['-f', tmp, '-m', String(cfg.maxChars), '-M', cfg.longTextMode, '-l', cfg.longTextMessage, '-C', cfg.cleanMarkdownFormatting ? '1' : '0', '-I', cfg.readInlineCode ? '1' : '0', '-B', cfg.codeBlocks, '-K', String(cfg.codeBlockMaxChars), '-R', cfg.codeBlockReplacementText]
295
+ if (cfg.rate > 0) args.push('-r', String(cfg.rate))
296
+ if (fullRead) args.push('-F')
297
+ child = spawn('/bin/bash', [cfg.engine].concat(args), { detached: true, stdio: 'ignore' })
298
+ } else {
299
+ // Windows 语速 = SAPI 刻度(-10 到 10,0 = 正常;负值变慢),直接透传,
300
+ // 不要把 0/负值替换成 1(否则无法回到正常语速/减慢)
301
+ child = spawn('powershell.exe', ['-NoProfile', '-ExecutionPolicy', 'Bypass', '-File', cfg.engine, '-File', tmp, '-Volume', String(cfg.volume), '-Rate', String(cfg.rate), '-MaxChars', String(cfg.maxChars), '-LongTextMode', cfg.longTextMode, '-LongTextMessage', cfg.longTextMessage, '-CleanMarkdownFormatting', cfg.cleanMarkdownFormatting ? '1' : '0', '-ReadInlineCode', cfg.readInlineCode ? '1' : '0', '-CodeBlocks', cfg.codeBlocks, '-CodeBlockMaxChars', String(cfg.codeBlockMaxChars), '-CodeBlockReplacementText', cfg.codeBlockReplacementText, '-FullRead', fullRead ? '1' : '0'], { windowsHide: true, stdio: 'ignore' })
302
+ }
303
+ const token = ++speechToken
304
+ activeSpeech = { process: child, tmp, token, item, gapMs: item.gapMs || 0 }
305
+ publishState()
306
+ const settle = () => {
307
+ removeTemp(tmp)
308
+ if (!activeSpeech || activeSpeech.token !== token) return
309
+ const gap = activeSpeech.gapMs || 0
310
+ activeSpeech = null
311
+ publishState()
312
+ const proceed = () => {
313
+ gapTimer = null
314
+ if (replacement) {
315
+ const next = replacement
316
+ replacement = null
317
+ startOne(next)
318
+ } else {
319
+ startNext()
320
+ }
321
+ }
322
+ // 队列项之间可配置停顿(如多个提问之间留 2 秒)
323
+ if (gap > 0) {
324
+ gapTimer = setTimeout(proceed, gap)
325
+ } else {
326
+ proceed()
327
+ }
328
+ }
329
+ child.once('exit', settle)
330
+ child.once('error', settle)
331
+ return true
332
+ }
333
+ function startNext() {
334
+ if (activeSpeech || replacement) return
335
+ const item = speechQueue.shift()
336
+ if (!item) { publishState(); return }
337
+ publishState()
338
+ startOne(item)
339
+ }
340
+ function enqueue(item) {
341
+ if (!item || !item.text || !item.text.trim()) return
342
+ speechQueue.push(item)
343
+ publishState()
344
+ startNext()
345
+ }
346
+ function stopActive() {
347
+ const active = activeSpeech
348
+ if (!active) return false
349
+ try {
350
+ if (process.platform === 'darwin' && active.process.pid) process.kill(-active.process.pid, 'SIGTERM')
351
+ else active.process.kill()
352
+ } catch (e) { log('stop speech failed:', e.message) }
353
+ return true
354
+ }
355
+ function clearAndStop() {
356
+ if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
357
+ speechQueue.length = 0
358
+ publishState()
359
+ return stopActive()
360
+ }
361
+ function replaceWith(item) {
362
+ if (gapTimer) { clearTimeout(gapTimer); gapTimer = null }
363
+ speechQueue.length = 0
364
+ replacement = item
365
+ publishState()
366
+ if (!activeSpeech) {
367
+ const next = replacement
368
+ replacement = null
369
+ startOne(next)
370
+ return
371
+ }
372
+ stopActive()
373
+ }
374
+ function visibleText(message) {
375
+ if (!message) return ''
376
+ if (typeof message.content === 'string') return message.content
377
+ if (!Array.isArray(message.content)) return ''
378
+ return message.content.filter(block => block && block.type === 'text' && typeof block.text === 'string').map(block => block.text).join('')
379
+ }
380
+ function hostItem(source, session, event, text, messageId) {
381
+ const sessionValue = session && (session.id != null ? session.id : session.sessionId)
382
+ return {
383
+ source,
384
+ sessionId: sessionValue != null ? String(sessionValue) : null,
385
+ turn: event && event.data && Number.isFinite(event.data.turn) ? event.data.turn : null,
386
+ messageId: messageId == null ? null : String(messageId),
387
+ text,
388
+ }
389
+ }
390
+
391
+ // ---- WebSocket + control route (PR #2) ----
392
+ ctx.inject(['webServer'], webCtx => {
393
+ webCtx.effect(() => webCtx.webServer.registerUpgrade({
394
+ path: '/dsh-speak/ws',
395
+ handler: (req, socket, head) => speechWss.handleUpgrade(req, socket, head, client => {
396
+ speechSockets.add(client)
397
+ client.once('close', () => speechSockets.delete(client))
398
+ client.once('error', () => speechSockets.delete(client))
399
+ try { client.send(JSON.stringify(state())) } catch (e) { speechSockets.delete(client) }
400
+ }),
401
+ }), 'dsh-speak speech-state websocket')
402
+ webCtx.effect(() => webCtx.webServer.register({
403
+ kind: 'exact', path: '/dsh-speak/control', handler: async (req, res) => {
404
+ const reply = (status, value) => { res.writeHead(status, { 'Content-Type': 'application/json; charset=utf-8' }); res.end(JSON.stringify(value)) }
405
+ if (req.method !== 'POST' || !String(req.headers['content-type'] || '').startsWith('application/json')) { reply(405, { error: 'POST application/json required' }); return }
406
+ try {
407
+ const chunks = []; let size = 0
408
+ for await (const chunk of req) { size += chunk.length; if (size > 1024 * 1024) throw new Error('request too large'); chunks.push(chunk) }
409
+ const body = JSON.parse(Buffer.concat(chunks).toString('utf8') || '{}')
410
+ if (body.action === 'status') { reply(200, state()); return }
411
+ if (body.action === 'stop') {
412
+ replacement = null
413
+ clearAndStop()
414
+ reply(200, state())
415
+ return
416
+ }
417
+ if (body.action !== 'play' || typeof body.text !== 'string' || !body.text.trim()) { reply(400, { error: 'invalid control request' }); return }
418
+ replaceWith({ source: 'manual', manual: true, sessionId: body.sessionId == null ? null : String(body.sessionId), turn: Number.isFinite(body.turn) ? body.turn : null, messageId: body.messageId == null ? null : String(body.messageId), text: body.text })
419
+ reply(200, state())
420
+ } catch (e) { reply(e.message === 'request too large' ? 413 : 400, { error: e.message }) }
421
+ },
422
+ }), 'dsh-speak replay control route')
423
+ })
424
+
425
+ ctx.effect(() => () => {
426
+ replacement = null
427
+ clearAndStop()
428
+ for (const socket of speechSockets) { try { socket.close() } catch (e) { /* closed */ } }
429
+ speechSockets.clear()
430
+ try { speechWss.close() } catch (e) { /* closed */ }
431
+ }, 'dsh-speak speech cleanup')
432
+
433
+ // ---- session event handling ----
434
+ let timer = null
435
+ let pendingText = ''
436
+ /** 当前回合内最后一条助手消息文本(turn/end 兜底播报用) */
437
+ let lastText = ''
438
+ /** 已通过节流播报过的文本(防止 turn/end 兜底重复播报) */
439
+ let lastSpokenText = ''
440
+ /** 最后一条助手消息 id(兜底播报时带上) */
441
+ let lastMessageId = null
442
+ /** cancel a pending throttled announcement (default mode, tool-call round) */
443
+ function cancelPending() {
444
+ if (timer) { clearTimeout(timer); timer = null }
445
+ pendingText = ''
446
+ }
447
+
448
+ ctx.on('session/event', (session, event) => {
449
+ try {
450
+ const type = event && event.type
451
+ if (type !== 'assistant/chunk') {
452
+ log('事件 type=', type, 'surfaceOp=', event && event.surfaceOp, 'seq=', event && event.seq)
453
+ }
454
+ // 新回合开始:清空上一回合的兜底状态,避免跨回合残留
455
+ if (type === 'turn/start') {
456
+ cancelPending()
457
+ lastText = ''
458
+ lastSpokenText = ''
459
+ lastMessageId = null
460
+ return
461
+ }
462
+ // tool-call round: ask_user_question announces the parsed question; any
463
+ // other tool call cancels the pending throttled narration
464
+ if (type === 'tool/call') {
465
+ if (event.data && event.data.name === 'ask_user_question' && cfg.announceQuestions) {
466
+ let items = []
467
+ try {
468
+ const args = JSON.parse(event.data.arguments || '{}')
469
+ const questions = Array.isArray(args.questions) ? args.questions : []
470
+ // 每个问题单独入队播报:带"问题N"序号(多问题时)与"选项N"序号
471
+ // (序号用数字,与 UI 的自动编号一致;中文 TTS 自然读成"一/二/三")
472
+ items = questions.map((question, qi) => {
473
+ const mode = question.multi_select ? '多选' : '单选'
474
+ const opts = Array.isArray(question.options) ? question.options : []
475
+ const optText = opts.map((option, oi) => {
476
+ const label = option && option.label ? String(option.label) : ''
477
+ return label ? `选项${oi + 1},${label}` : ''
478
+ }).filter(Boolean).join(';')
479
+ const head = questions.length > 1 ? `问题${qi + 1},` : ''
480
+ // question 文案已含"单选/多选"字样时不再追加模式后缀,避免重复
481
+ const modeSuffix = /单选|多选/.test(question.question || '') ? '' : `(${mode})`
482
+ const body = [question.question || '', modeSuffix, optText ? ',' + optText : ''].join('')
483
+ return (head + body).trim()
484
+ }).filter(Boolean)
485
+ } catch (e) { /* ignore malformed arguments */ }
486
+ if (items.length > 0) {
487
+ cancelPending()
488
+ // 问题已单独播报,标记当前最后文本为已播,避免 turn/end 兜底重复
489
+ lastSpokenText = lastText
490
+ // 多条问题按 FIFO 串行播报,之间停顿 cfg.questionGapMs(默认 2 秒)
491
+ const gap = cfg.questionGapMs > 0 && items.length > 1 ? cfg.questionGapMs : 0
492
+ items.forEach((itemText, i) => {
493
+ const item = hostItem('question', session, event, itemText, null)
494
+ item.gapMs = i < items.length - 1 ? gap : 0
495
+ enqueue(item)
496
+ })
497
+ }
498
+ return
499
+ }
500
+ cancelPending()
501
+ return
502
+ }
503
+ // approval requested: announce right away
504
+ if (type === 'approval/asked' && cfg.announceApprovals) {
505
+ cancelPending()
506
+ let reason = String((event.data && event.data.reason) || '')
507
+ if (cfg.stripApprovalPrefix) {
508
+ // 通用剥离行首"动作标签: "前缀(英文动作短语 + 冒号,如
509
+ // "Store decision fact in workspace memory (dsh-speak): <内容>"、
510
+ // "escalate sandbox to danger-full-access: <原因>"),只念冒号后的
511
+ // 具体内容;中文开头或无冒号的 reason 原样保留(如"删除 xxx")。
512
+ reason = reason.replace(/^[A-Za-z][^::\n]*?[::]\s*/, '').trim()
513
+ }
514
+ enqueue(hostItem('approval', session, event, reason || '需要你的审批,请查看界面。', null))
515
+ return
516
+ }
517
+ // 回合结束:兜底播报最终回复(被工具调用取消的节流文本在此补播,
518
+ // 已播过的不重复),随后按需播报"第 N 轮对话完成"可选事件
519
+ if (type === 'turn/end') {
520
+ if (cfg.automaticSpeech && !cfg.queueAllMessages && lastText && lastText !== lastSpokenText) {
521
+ const itemText = lastText
522
+ const itemMessageId = lastMessageId
523
+ cancelPending()
524
+ lastText = ''
525
+ lastSpokenText = itemText
526
+ enqueue(hostItem('automatic', session, event, itemText, itemMessageId))
527
+ }
528
+ if (!cfg.announceTurnEnd) return
529
+ const data = event.data
530
+ const prefix = data && data.turn != null ? `第 ${data.turn} 轮对话` : '本轮对话'
531
+ const kind = data && data.reason && data.reason.kind
532
+ const text = ({ completed: prefix + '完成', aborted: prefix + '中断', interrupted: prefix + '中断', blocked: prefix + '被阻塞', error: prefix + '异常结束', 'max-tokens': prefix + '异常结束' })[kind] || prefix + '结束'
533
+ enqueue(hostItem('turn/end', session, event, text, null))
534
+ return
535
+ }
536
+ if (type === 'command/done' && cfg.announceCommandDone) {
537
+ enqueue(hostItem('command/done', session, event, (event.data && event.data.kind) === 'error' ? '命令执行失败' : '命令执行完成', null))
538
+ return
539
+ }
540
+ if (type === 'goal/change' && cfg.announceGoalChange) {
541
+ const data = event.data
542
+ const objective = data && data.goal && data.goal.objective
543
+ const label = ({ create: '已创建目标', edit: '目标已更新', complete: '目标已完成', pause: '目标已暂停', resume: '目标已恢复', block: '目标已阻塞', clear: '目标已清除' })[data && data.operation] || '目标状态变化'
544
+ const text = objective && ['create', 'edit', 'complete'].includes(data.operation) ? `${label}:${objective.replace(/\s+/g, ' ').trim().slice(0, 40)}` : label
545
+ enqueue(hostItem('goal/change', session, event, text, null))
546
+ return
547
+ }
548
+ if (type === 'tool/result' && cfg.announceToolErrors) {
549
+ const data = event.data
550
+ const err = data && data.error
551
+ // 真实错误标记有两处:结构化失败身份 data.error(name/code),以及结果块上的
552
+ // isError。0.1.2 起 createToolResultMessage 把结果块包进一个 ToolResultBlock
553
+ // ({ type:'tool-result', toolCallId, content:[…], isError }),文字在它嵌套的
554
+ // content 里;更早的版本把 isError 直接放在 text 块上。两种形状都读。
555
+ //
556
+ // 注意:pwsh / bash 把「命令非零退出」当作结果数据上报(`exit code: N`),
557
+ // 不置 isError —— 只有基础设施失败(spawn 错误、abort)才是 isError 结果,
558
+ // 所以失败的命令本身不会播报工具出错。
559
+ const errText = (Array.isArray(data && data.message && data.message.content) ? data.message.content : [])
560
+ .filter(block => block && block.isError === true)
561
+ .map(block => {
562
+ const parts = Array.isArray(block.content) ? block.content : [block]
563
+ return parts.map(part => (part && (part.text || part.code)) || '').filter(Boolean).join(' ')
564
+ })
565
+ .filter(Boolean).join(' ')
566
+ if (err || errText) {
567
+ const detail = (errText || (err && err.code) || (err && err.name) || '').replace(/\s+/g, ' ').trim().slice(0, 60)
568
+ // 详情只在"确实是一句中文描述"时才念:英文模板(Error: / ENOENT / 技术
569
+ // code)对中文用户可读性差,应当截掉。判据是**汉字数量多于拉丁字母数量**,
570
+ // 而不是"含有汉字"——后者会被路径里的中文目录名骗过:
571
+ // `Error: cannot read "D:\...\第二轮测试用的不存在文件.txt"` 含 12 个汉字,
572
+ // 却是纯英文报错(1.8.0 修正)。
573
+ const cjkCount = (detail.match(/[\u4e00-\u9fff]/g) || []).length
574
+ const latinCount = (detail.match(/[A-Za-z]/g) || []).length
575
+ const readable = cjkCount > latinCount ? `:${detail}` : ''
576
+ enqueue(hostItem('tool/result', session, event, `工具调用出错${readable}`, null))
577
+ }
578
+ return
579
+ }
580
+ if (type === 'todo/write' && cfg.announceTodoWrite) {
581
+ const todos = Array.isArray(event.data && event.data.todos) ? event.data.todos : []
582
+ const done = todos.filter(t => t && t.status === 'completed').length
583
+ enqueue(hostItem('todo/write', session, event, `待办已更新:${done}/${todos.length} 完成`, null))
584
+ return
585
+ }
586
+ if (!event || type !== 'assistant/message') return
587
+ if (event.surfaceOp && event.surfaceOp !== 'append') return
588
+ const message = event.data && (event.data.message || event.data)
589
+ const text = visibleText(message)
590
+ if (!text.trim()) return
591
+
592
+ // queueAllMessages mode (PR #2): enqueue every assistant message now
593
+ if (cfg.queueAllMessages && cfg.automaticSpeech) {
594
+ enqueue(hostItem('automatic', session, event, text, message && message.id))
595
+ return
596
+ }
597
+ // default mode: throttle/merge the final reply; a tool/call cancels it,
598
+ // and turn/end 兜底补播 lastText(见上方 turn/end 分支)
599
+ cancelPending()
600
+ pendingText = text
601
+ lastText = text
602
+ lastMessageId = message && message.id ? String(message.id) : null
603
+ timer = setTimeout(() => {
604
+ if (!pendingText) return
605
+ const itemText = pendingText
606
+ pendingText = ''
607
+ timer = null
608
+ lastSpokenText = itemText
609
+ enqueue(hostItem('automatic', session, event, itemText, lastMessageId))
610
+ }, cfg.throttleMs)
611
+ } catch (e) { log('session event speech error:', e.message) }
612
+ })
613
+ },
614
+ }