switchroom 0.19.14 → 0.19.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/cli/switchroom.js +1 -1
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +1 -1
  4. package/telegram-plugin/bridge/bridge.ts +1 -1
  5. package/telegram-plugin/dist/bridge/bridge.js +31 -2
  6. package/telegram-plugin/dist/gateway/gateway.js +1690 -932
  7. package/telegram-plugin/dist/server.js +31 -2
  8. package/telegram-plugin/gateway/background-shell-liveness.ts +65 -0
  9. package/telegram-plugin/gateway/forward-origin.ts +6 -1
  10. package/telegram-plugin/gateway/gateway.ts +10 -57
  11. package/telegram-plugin/gateway/narrative-lane.ts +11 -0
  12. package/telegram-plugin/gateway/outbound-send-path.ts +25 -23
  13. package/telegram-plugin/gateway/outbox-listen-markup.ts +67 -0
  14. package/telegram-plugin/gateway/outbox-sweep.ts +124 -20
  15. package/telegram-plugin/gateway/rich-message-handler.ts +241 -0
  16. package/telegram-plugin/gateway/silence-poke-session-event.ts +89 -0
  17. package/telegram-plugin/gateway/stream-render.ts +107 -15
  18. package/telegram-plugin/gateway/unhandled-message.ts +14 -0
  19. package/telegram-plugin/hooks/narration-classify.d.mts +23 -0
  20. package/telegram-plugin/hooks/narration-classify.mjs +210 -0
  21. package/telegram-plugin/hooks/silent-end-scan.mjs +136 -82
  22. package/telegram-plugin/narrative-flush.ts +35 -0
  23. package/telegram-plugin/outbox.ts +73 -3
  24. package/telegram-plugin/session-tail.ts +88 -1
  25. package/telegram-plugin/shown-ledger.ts +145 -0
  26. package/telegram-plugin/silence-poke.ts +118 -1
  27. package/telegram-plugin/silent-end.ts +42 -0
  28. package/telegram-plugin/tests/background-shell-liveness.test.ts +72 -0
  29. package/telegram-plugin/tests/backstop-exactly-once.test.ts +335 -0
  30. package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +14 -0
  31. package/telegram-plugin/tests/feed-survival.test.ts +7 -1
  32. package/telegram-plugin/tests/fixtures/bg-shell-liveness-3519.jsonl +3 -0
  33. package/telegram-plugin/tests/forward-origin.test.ts +20 -0
  34. package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +290 -0
  35. package/telegram-plugin/tests/forwarded-rich-message.test.ts +305 -0
  36. package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +1 -0
  37. package/telegram-plugin/tests/narration-leak-3513.test.ts +352 -0
  38. package/telegram-plugin/tests/outbox-sweep-listen-button.test.ts +253 -0
  39. package/telegram-plugin/tests/session-tail.test.ts +91 -1
  40. package/telegram-plugin/tests/silence-poke.test.ts +280 -0
  41. package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +42 -13
  42. package/telegram-plugin/tests/silent-end.test.ts +7 -1
  43. package/telegram-plugin/tests/tts-normalize.test.ts +66 -0
  44. package/telegram-plugin/tests/turn-flush-safety.test.ts +35 -3
  45. package/telegram-plugin/tests/voice-normalize-text.test.ts +82 -1
  46. package/telegram-plugin/tts-normalize.ts +12 -0
  47. package/telegram-plugin/turn-flush-safety.ts +66 -53
  48. package/telegram-plugin/voice-normalize-text.ts +100 -0
  49. package/telegram-plugin/voice-ondemand.ts +71 -0
@@ -683,13 +683,36 @@ describe('selectFlushDeliveryText — deliver the terminal answer, strip only na
683
683
  expect(out).toBe('Here are the results:')
684
684
  })
685
685
 
686
- it('decideTurnFlush delivers the narrowed answer, not the whole blob', () => {
686
+ // #3513 follow-up — decideTurnFlush now routes through the deterministic
687
+ // `selectBackstopDelivery` coalescer, NOT the wording heuristic. With NO
688
+ // structural provenance (no `capturedBlockMeta`, so no block was followed by a
689
+ // turn-continuing tool_use) EVERY block is part of the terminal run and is
690
+ // delivered joined — fail OPEN, never drop. The wording-based narration strip
691
+ // is gone on the backstop path (it was provably incomplete; #3513). When the
692
+ // structural signal IS present it, and only it, suppresses.
693
+ it('decideTurnFlush delivers the full terminal run when NO tool-follow provenance is present (fail-open)', () => {
687
694
  const decision = decideTurnFlush({
688
695
  chatId: 'chat1',
689
696
  replyCalled: false,
690
697
  capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
691
698
  })
692
699
  expect(decision.kind).toBe('flush')
700
+ if (decision.kind === 'flush') {
701
+ expect(decision.text).toBe(`Let me check.\n\nNow let me compose the reply.\n\n${answer}`)
702
+ }
703
+ })
704
+
705
+ it('decideTurnFlush suppresses tool-followed blocks UNCONDITIONALLY (structural provenance drives the coalescer)', () => {
706
+ // The first two blocks were each followed by a turn-continuing tool_use →
707
+ // intra-turn narration → suppressed with NO length/wording gate. Only the
708
+ // terminal answer (not tool-followed) survives.
709
+ const decision = decideTurnFlush({
710
+ chatId: 'chat1',
711
+ replyCalled: false,
712
+ capturedText: ['Let me check.', 'Now let me compose the reply.', answer],
713
+ capturedBlockMeta: [true, true, false],
714
+ })
715
+ expect(decision.kind).toBe('flush')
693
716
  if (decision.kind === 'flush') {
694
717
  expect(decision.text).toBe(answer)
695
718
  expect(decision.text).not.toContain('Let me check')
@@ -832,11 +855,20 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
832
855
  // trailing ellipsis/colon — so only the structural flag could strip it.
833
856
  const wrapUp = 'Saved that to memory.'
834
857
  // meta[0]=true (a memory tool_use followed the real answer), [1]=false.
858
+ // The STANDALONE `selectFlushDeliveryText` (the old #3237 substance-gated
859
+ // function, unchanged and still exported) keeps the substantial block.
835
860
  const out = selectFlushDeliveryText([realAnswer, wrapUp], [true, false])
836
861
  expect(out).toBe(`${realAnswer}\n\n${wrapUp}`)
837
862
  expect(out).toContain('The migration completed cleanly')
838
863
 
839
- // End-to-end through decideTurnFlush.
864
+ // #3513 follow-up — decideTurnFlush now routes through
865
+ // `selectBackstopDelivery`, which suppresses a tool-followed block
866
+ // UNCONDITIONALLY (no substantive-floor carve-out). This DELIBERATELY
867
+ // overturns #3237's substance-gate ON THE BACKSTOP PATH: a real answer the
868
+ // model followed with a turn-continuing tool_use is treated as intra-turn
869
+ // and excluded; only the terminal (non-tool-followed) block is delivered.
870
+ // The deterministic contract is "the terminal run wins" — the model keeps
871
+ // its answer by ending on it or calling the reply tool.
840
872
  const decision = decideTurnFlush({
841
873
  chatId: 'chat1',
842
874
  replyCalled: false,
@@ -845,7 +877,7 @@ describe('selectFlushDeliveryText — structural provenance (followedByToolUse)
845
877
  })
846
878
  expect(decision.kind).toBe('flush')
847
879
  if (decision.kind === 'flush') {
848
- expect(decision.text).toBe(`${realAnswer}\n\n${wrapUp}`)
880
+ expect(decision.text).toBe(wrapUp)
849
881
  }
850
882
  })
851
883
 
@@ -6,7 +6,7 @@
6
6
  */
7
7
 
8
8
  import { describe, it, expect } from 'bun:test'
9
- import { normalizeForSpeech } from '../voice-normalize-text.js'
9
+ import { normalizeForSpeech, decodeHtmlEntities } from '../voice-normalize-text.js'
10
10
 
11
11
  describe('normalizeForSpeech — reported markdown/symbol cases', () => {
12
12
  it('drops stray tildes (never spoken as "tilde")', () => {
@@ -254,3 +254,84 @@ describe('normalizeForSpeech — conservative, does not mangle real words', () =
254
254
  expect(normalizeForSpeech(' \n ')).toBe('')
255
255
  })
256
256
  })
257
+
258
+ describe('normalizeForSpeech — backslash escapes & HTML entities (voice trash)', () => {
259
+ it('strips a literal backslash-b (\\b) so it is never spoken as "backslash b"', () => {
260
+ // The exact operator report: the regex/escape "\b" was read as "slash b".
261
+ const out = normalizeForSpeech('the regex \\b word boundary')
262
+ expect(out).toBe('the regex b word boundary')
263
+ expect(out).not.toContain('\\')
264
+ })
265
+
266
+ it('unescapes MarkdownV2 punctuation escapes without leaving backslashes', () => {
267
+ expect(normalizeForSpeech('done\\. next\\! wait\\-')).toBe('done. next! wait-')
268
+ expect(normalizeForSpeech('a \\* b')).not.toContain('\\')
269
+ })
270
+
271
+ it('an escaped emphasis marker collapses to the inner word, not a backslash', () => {
272
+ expect(normalizeForSpeech('Use \\*literal\\* here')).toBe('Use literal here')
273
+ })
274
+
275
+ it('drops a Windows-path-style backslash run (C:\\build)', () => {
276
+ expect(normalizeForSpeech('path C:\\build\\out done')).toBe('path C:buildout done')
277
+ })
278
+
279
+ it('decodes HTML entities so & / < are not read as "amp" / "lt"', () => {
280
+ expect(normalizeForSpeech('Tom & Jerry')).toBe('Tom and Jerry')
281
+ expect(normalizeForSpeech('5 &lt; 10 &gt; 3')).toBe('5 < 10 > 3')
282
+ expect(normalizeForSpeech('it&#39;s here')).toBe("it's here")
283
+ })
284
+
285
+ it('produces clean plain speech for a rich mixed reply (end-to-end)', () => {
286
+ const reply =
287
+ '**Bold** and `code\\b` and a [label](https://example.com/x) ' +
288
+ 'with Tom &amp; Jerry and a regex \\b\\.'
289
+ const out = normalizeForSpeech(reply)
290
+ expect(out).not.toContain('\\')
291
+ expect(out).not.toContain('`')
292
+ expect(out).not.toContain('*')
293
+ expect(out).not.toContain('&amp;')
294
+ expect(out).not.toContain('example.com')
295
+ expect(out).toContain('label')
296
+ expect(out).toContain('Tom and Jerry')
297
+ })
298
+
299
+ it('is idempotent: a second pass finds no backslashes/entities to change', () => {
300
+ const reply = 'a \\b and Tom &amp; Jerry \\. end'
301
+ const once = normalizeForSpeech(reply)
302
+ expect(normalizeForSpeech(once)).toBe(once)
303
+ })
304
+ })
305
+
306
+ describe('normalizeForSpeech — review findings (fixpoint decode, metachar, nits)', () => {
307
+ it('L2 parity: a double-encoded entity decodes to a fixpoint (depth-independent)', () => {
308
+ // Single decode and double decode must land on the SAME spoken char so the
309
+ // immediate voice-out and the lazy Listen tap never diverge.
310
+ const single = normalizeForSpeech('&amp;amp;lt;')
311
+ const doubled = normalizeForSpeech(normalizeForSpeech('&amp;amp;lt;'))
312
+ expect(single).toBe('<')
313
+ expect(doubled).toBe('<')
314
+ expect(normalizeForSpeech('&amp;amp;amp;')).toBe('&')
315
+ })
316
+
317
+ it('L1: an entity that decodes to a line-leading metachar keeps a spoken form', () => {
318
+ // &#35; → '#'. Naively re-fed to the heading stripper it would vanish; the
319
+ // user escaped it on purpose, so it must survive as spoken "hash".
320
+ expect(normalizeForSpeech('&#35; Heading')).toBe('hash Heading')
321
+ expect(normalizeForSpeech('2 &#42; 3')).toBe('2 asterisk 3')
322
+ })
323
+
324
+ it('nit: a dangling trailing backslash is dropped, never spoken', () => {
325
+ expect(normalizeForSpeech('ends here\\')).toBe('ends here')
326
+ expect(normalizeForSpeech('ends here\\')).not.toContain('\\')
327
+ })
328
+
329
+ it('nit: &#92; decodes to a backslash which is then stripped (no trash)', () => {
330
+ expect(normalizeForSpeech('X&#92;Y')).toBe('XY')
331
+ expect(normalizeForSpeech('X&#92;Y')).not.toContain('\\')
332
+ })
333
+
334
+ it('fixpoint does not over-decode entity-less text (Q&A stays literal)', () => {
335
+ expect(decodeHtmlEntities('Q&A test')).toBe('Q&A test')
336
+ })
337
+ })
@@ -38,6 +38,8 @@
38
38
  * ~5 → "about 5", > blockquote markers dropped.
39
39
  */
40
40
 
41
+ import { decodeHtmlEntities, stripBackslashEscapes } from './voice-normalize-text'
42
+
41
43
  const NULL = '\x00'
42
44
  const INLINE_PH = `${NULL}TN_INLINE`
43
45
 
@@ -197,6 +199,16 @@ export function normalizeForTts(text: string): string {
197
199
 
198
200
  let s = text.replace(/\r\n?/g, '\n')
199
201
 
202
+ // -- HTML entities → char, then backslash escapes → the escaped char. The
203
+ // last line of defence at the /tts body build: the Listen lazy path and
204
+ // the pre-synth queue can synthesize from cache entries that predate the
205
+ // normalizeForSpeech coverage, so these must be stripped here too. Both
206
+ // are idempotent — if normalizeForSpeech already ran there is nothing
207
+ // left to decode/unescape. Without this the engine speaks `\b` as
208
+ // "backslash b" and `&amp;` as "amp".
209
+ s = decodeHtmlEntities(s)
210
+ s = stripBackslashEscapes(s)
211
+
200
212
  // -- Code fences → spoken placeholder (before anything can see contents).
201
213
  s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*?\n[ \t]*\2[ \t]*(?=\n|$)/g, '$1code block omitted.')
202
214
  s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*$/g, '$1code block omitted.')
@@ -19,6 +19,13 @@
19
19
  * and can be set to `0` / `false` / `off` to disable without a rebuild.
20
20
  */
21
21
 
22
+ import {
23
+ isNarrationBlock,
24
+ isStructuralNarration,
25
+ selectBackstopDelivery,
26
+ SUBSTANTIVE_MIN_CHARS,
27
+ } from './hooks/narration-classify.mjs'
28
+
22
29
  const SILENT_MARKERS = new Set(['NO_REPLY', 'HEARTBEAT_OK'])
23
30
  // Small buffer so `NO_REPLY.` with a stray period still counts as silent.
24
31
  const SILENT_MARKER_MAX_LEN = Math.max(
@@ -134,7 +141,7 @@ export function endsWithSilentMarker(text: string | undefined): boolean {
134
141
  * `hooks/silent-end-scan.mjs` — the same bar the codebase uses everywhere to
135
142
  * recognise "this text block is a real answer, not a short narration/closer".
136
143
  */
137
- export const FLUSH_SUBSTANTIVE_MIN_CHARS = 200
144
+ export const FLUSH_SUBSTANTIVE_MIN_CHARS = SUBSTANTIVE_MIN_CHARS
138
145
 
139
146
  /**
140
147
  * Pick the text a turn-flush should actually DELIVER from the captured
@@ -203,9 +210,33 @@ export function selectFlushDeliveryText(
203
210
  .map((b, i) => ({ text: b.trim(), followedByToolUse: followedByToolUse?.[i] }))
204
211
  .filter(c => c.text.length > 0)
205
212
  if (candidates.length === 0) return ''
206
- if (candidates.length === 1) return candidates[0].text
207
- const answer = candidates[candidates.length - 1].text
208
- const preceding = candidates.slice(0, -1)
213
+
214
+ // #3513 — the STRUCTURAL rule now applies to the TERMINAL block too, not only
215
+ // preceding blocks. A lone trailing block that a `tool_use` FOLLOWED (the
216
+ // model kept acting after writing it) is intra-turn narration by construction,
217
+ // never the terminal answer — so it must not be delivered as one. This is the
218
+ // primary fix for the observed leak ("Now checking the gateway logs…"),
219
+ // wording-independent where the opener regex fails. It stays substance-gated
220
+ // (correction 3): only a SHORT or narration-shaped block that a tool followed
221
+ // is suppressed — a SUBSTANTIVE, non-narration block followed by a
222
+ // non-delivering tool (react / pin / typing / edit) is NOT structural
223
+ // narration and still delivers.
224
+ //
225
+ // Only the STRUCTURAL signal (`followedByToolUse === true`) may strip the
226
+ // terminal block; the opener/trailer heuristic remains a residual tie-breaker
227
+ // for PRECEDING blocks only, so a provenance-less single block still delivers
228
+ // verbatim (#3276 finding 7 — a colon-terminated line that IS the whole answer
229
+ // is never dropped).
230
+ let cs = candidates
231
+ while (cs.length > 1 && isStructuralNarration(cs[cs.length - 1].text, cs[cs.length - 1].followedByToolUse)) {
232
+ cs = cs.slice(0, -1)
233
+ }
234
+ if (cs.length === 1) {
235
+ return isStructuralNarration(cs[0].text, cs[0].followedByToolUse) ? '' : cs[0].text
236
+ }
237
+
238
+ const answer = cs[cs.length - 1].text
239
+ const preceding = cs.slice(0, -1)
209
240
  // Deliver only the terminal answer when every earlier block is
210
241
  // intent-narration. Per block:
211
242
  // - structural flag TRUE ⇒ a tool_use followed this block, but that alone
@@ -228,50 +259,15 @@ export function selectFlushDeliveryText(
228
259
  ? false
229
260
  : isNarrationBlock(c.text),
230
261
  )
231
- return allNarration ? answer : candidates.map(c => c.text).join('\n\n')
262
+ return allNarration ? answer : cs.map(c => c.text).join('\n\n')
232
263
  }
233
264
 
234
- /**
235
- * Narration heuristic: a block that opens with a first-person "about to do X"
236
- * phrase the model emits BEFORE composing its real answer ("Let me check…",
237
- * "I'll look it up…", "Now let me…"). Deliberately NOT length-based — the
238
- * narration that shadowed the real answer in the observed bug was a LONG
239
- * (≥200-char) "Let me pull the numbers…" block, and short blocks are frequently
240
- * legitimate multi-paragraph answer content — so gating on length either drops a
241
- * short real answer or keeps a long narration. When earlier blocks don't match
242
- * this opener we keep the full joined text (never drop content we can't
243
- * confidently attribute to narration).
244
- */
245
- const NARRATION_OPENER =
246
- /^(let me\b|lemme\b|i'?ll\b|i will\b|i am going to\b|i'?m going to\b|i'?m about to\b|going to\b|first,?\s+(?:let me|i'?ll|i will)\b|now,?\s+(?:let me|i'?ll|i will)\b|next,?\s+(?:let me|i'?ll|i will)\b|let'?s\b)/i
247
-
248
- /**
249
- * #3276 guard 8 — the `NARRATION_OPENER` regex alone misses common progress
250
- * narration that opens with a gerund/present-continuous verb and trails off
251
- * into an ellipsis or colon: "Checking now…", "Pulling the numbers:",
252
- * "Looking into that…". Relying on the opener regex leaked those blocks into
253
- * the delivered answer. This recognises a NON-terminal narration line
254
- * deterministically, WITHOUT re-gating on a min-char floor (a short real
255
- * answer like "Yes, done." must still deliver): a single short line that ends
256
- * with an ellipsis or a colon is progress narration, not the terminal answer.
257
- *
258
- * Kept conservative on purpose — only a SINGLE-line block (no internal
259
- * paragraph) under the substantive floor, ending in `…` / `...` / `:`, so a
260
- * genuine multi-paragraph answer that happens to end a paragraph with a colon
261
- * is never mistaken for narration.
262
- */
263
- const NARRATION_TRAILER = /(?:\.{3}|…|:)\s*$/
264
-
265
- function isTrailingNarrationLine(block: string): boolean {
266
- const t = block.trim()
267
- if (t.length === 0 || t.length >= FLUSH_SUBSTANTIVE_MIN_CHARS) return false
268
- if (t.includes('\n')) return false
269
- return NARRATION_TRAILER.test(t)
270
- }
271
-
272
- function isNarrationBlock(block: string): boolean {
273
- return NARRATION_OPENER.test(block.trimStart()) || isTrailingNarrationLine(block)
274
- }
265
+ // The narration heuristic (`isNarrationBlock` / openers / trailer) and the
266
+ // structural rule (`isStructuralNarration`) now live in the ONE shared
267
+ // classifier `hooks/narration-classify.mjs`, imported at the top of this file
268
+ // and by the unbundled Stop hook (`hooks/silent-end-scan.mjs`) alike — so the
269
+ // TS gateway side and the `.mjs` side can never drift (switchroom#3513
270
+ // correction 2; retires the old "MUST stay in sync" hand-synced copies).
275
271
 
276
272
  export type FlushDecision =
277
273
  | { kind: 'flush'; text: string }
@@ -283,6 +279,11 @@ export type FlushSkipReason =
283
279
  | 'no-inbound-chat'
284
280
  | 'empty-text'
285
281
  | 'silent-marker'
282
+ // A prior BACKSTOP (E2 answer-ready flush, or an out-of-process E3/E4 machine
283
+ // across a crash/restart) already delivered this turn's answer — the durable
284
+ // exactly-once-among-backstops guard (#3513 follow-up, MF2). Set by the
285
+ // caller, not `decideTurnFlush`.
286
+ | 'already-delivered'
286
287
 
287
288
  export interface FlushDecisionInput {
288
289
  /** Inbound chat the turn was servicing. `null` means system-initiated /
@@ -375,14 +376,26 @@ export function decideTurnFlush(input: FlushDecisionInput): FlushDecision {
375
376
  // sentinel — treat the whole turn as intentionally silent rather than
376
377
  // flush the prose with the sentinel glued on.
377
378
  if (endsWithSilentMarker(joined)) return { kind: 'skip', reason: 'silent-marker' }
378
- // Deliver only the substantive answer block, never the whole narration+answer
379
- // blob (see `selectFlushDeliveryText`). The silent-marker / empty guards above
380
- // still run on the full `joined` string so a partly-silent turn is classified
381
- // correctly; only the DELIVERED text is narrowed to the answer.
382
- return {
383
- kind: 'flush',
384
- text: selectFlushDeliveryText(input.capturedText, input.capturedBlockMeta),
379
+ // #3513 follow-up — deterministic, model-discipline-free coalescing. The
380
+ // DELIVERED text is the TERMINAL RUN (the suffix of blocks NOT followed by a
381
+ // turn-continuing tool_use); every tool-followed block is suppressed
382
+ // UNCONDITIONALLY (no length/wording gate), replacing #3515's substance-gated
383
+ // heuristic on this backstop path. `selectBackstopDelivery` returns null when
384
+ // nothing terminal survives (a short tool-followed fragment) — treat that as
385
+ // 'empty-text' so the turn routes to the Stop-hook re-prompt ladder rather
386
+ // than delivering an interim narration line. The silent-marker / empty guards
387
+ // above still run on the full `joined` string so a partly-silent turn is
388
+ // classified correctly.
389
+ const selected = selectBackstopDelivery(
390
+ input.capturedText.map((text, i) => ({
391
+ text,
392
+ followedByToolUse: input.capturedBlockMeta?.[i],
393
+ })),
394
+ )
395
+ if (selected == null || selected.text.trim().length === 0) {
396
+ return { kind: 'skip', reason: 'empty-text' }
385
397
  }
398
+ return { kind: 'flush', text: selected.text }
386
399
  }
387
400
 
388
401
  /**
@@ -53,6 +53,89 @@
53
53
  /** Replace a fenced code block with a spoken placeholder. */
54
54
  const CODE_BLOCK_PLACEHOLDER = 'code block omitted'
55
55
 
56
+ /** Named HTML entities the reply text realistically carries. */
57
+ const HTML_ENTITIES: Record<string, string> = {
58
+ amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ',
59
+ }
60
+
61
+ /**
62
+ * Markdown metacharacters that the block/emphasis/table stripper would
63
+ * silently consume at line-start or as a pair. When such a char arrives via
64
+ * an entity escape the user meant it LITERALLY (that is the whole point of
65
+ * escaping it), so instead of emitting the raw char — which the downstream
66
+ * stripper would then eat, losing the intent — we emit a neutral spoken form
67
+ * that survives every later pass. Deterministic; the spoken form contains no
68
+ * `&`/`;` so it can never re-enter the entity decoder.
69
+ */
70
+ const METACHAR_SPOKEN: Record<string, string> = {
71
+ '#': ' hash ',
72
+ '*': ' asterisk ',
73
+ '_': ' underscore ',
74
+ '~': ' tilde ',
75
+ '`': ' backtick ',
76
+ '|': ' bar ',
77
+ }
78
+
79
+ /** One decode pass: named + numeric entities → char (or spoken metachar). */
80
+ function decodeHtmlEntitiesOnce(input: string): string {
81
+ const toChar = (cp: number, raw: string): string => {
82
+ if (!(cp > 0 && cp <= 0x10ffff)) return raw
83
+ const ch = String.fromCodePoint(cp)
84
+ return METACHAR_SPOKEN[ch] ?? ch
85
+ }
86
+ return input
87
+ .replace(/&#x([0-9a-f]+);/gi, (m, hex: string) => toChar(parseInt(hex, 16), m))
88
+ .replace(/&#(\d+);/g, (m, dec: string) => toChar(Number(dec), m))
89
+ .replace(/&([a-z][a-z0-9]*);/gi, (m, name: string) => {
90
+ const ch = HTML_ENTITIES[name.toLowerCase()]
91
+ if (ch === undefined) return m
92
+ return METACHAR_SPOKEN[ch] ?? ch
93
+ })
94
+ }
95
+
96
+ /**
97
+ * Decode HTML entities (named + numeric) to their character so a TTS engine
98
+ * never reads `&amp;` as "amp". Unknown named entities are left untouched.
99
+ * Pure + deterministic.
100
+ *
101
+ * Iterates to a FIXPOINT: a double-encoded entity (`&amp;amp;lt;`) is decoded
102
+ * repeatedly until no entity remains, so this pass is depth-idempotent —
103
+ * applying it once yields the same result as applying it twice. That keeps
104
+ * every voice callsite in lockstep: the immediate voice-out path runs
105
+ * normalizeForSpeech THEN normalizeForTts, and the value the lazy Listen tap /
106
+ * pre-synth queue reads is itself already normalizeForSpeech'd before its own
107
+ * normalizeForTts — fixpoint decoding guarantees both speak an identical
108
+ * string regardless of how deep the original encoding was. The loop strictly
109
+ * shrinks the entity count each turn (and is capped) so it always terminates.
110
+ * Text WITHOUT a trailing `;` (e.g. `Q&A`) matches nothing and is returned
111
+ * untouched.
112
+ */
113
+ export function decodeHtmlEntities(input: string): string {
114
+ let s = input
115
+ // A fully-decodable chain shrinks by at least one entity per pass; the cap
116
+ // is a belt-and-braces guard against any pathological crafted input.
117
+ for (let i = 0; i < 10; i++) {
118
+ const next = decodeHtmlEntitiesOnce(s)
119
+ if (next === s) break
120
+ s = next
121
+ }
122
+ return s
123
+ }
124
+
125
+ /**
126
+ * Remove markdown/MarkdownV2 backslash escapes so the spoken text carries no
127
+ * literal backslashes. A backslash before ANY single character is dropped,
128
+ * keeping the character (`\.` → ".", `\*` → "*", `\b` → "b"); a dangling
129
+ * trailing backslash is dropped. Pure + deterministic + idempotent (a second
130
+ * pass finds no backslashes). A real newline is preserved (only the escaping
131
+ * backslash is consumed).
132
+ */
133
+ export function stripBackslashEscapes(input: string): string {
134
+ // `\X` → `X` for any following char (including an escaped `\\`), then drop
135
+ // any lone backslash the first pass left (an escaped backslash's survivor).
136
+ return input.replace(/\\([\s\S])/g, '$1').replace(/\\/g, '')
137
+ }
138
+
56
139
  // ---------------------------------------------------------------------------
57
140
  // Number → words helpers (small, deterministic, English cardinal only).
58
141
  // Used by the numbers/units pass. Supports 0..999_999_999 which is far more
@@ -168,6 +251,23 @@ export function normalizeForSpeech(input: string): string {
168
251
  if (!input) return ''
169
252
  let s = input.replace(/\r\n?/g, '\n')
170
253
 
254
+ // 0a. HTML entities → their character. The reply text can carry entity
255
+ // escapes (`&amp;`, `&lt;`, `&#39;`) that a TTS engine would otherwise
256
+ // read as "amp" / "lt" / a digit run. Decode BEFORE markdown/symbol
257
+ // passes so the recovered char is then handled naturally (e.g. a
258
+ // decoded `&` becomes "and" in the symbols pass).
259
+ s = decodeHtmlEntities(s)
260
+
261
+ // 0b. Backslash escapes → the escaped character. Telegram MarkdownV2 and
262
+ // CommonMark escape literal punctuation with a leading backslash
263
+ // (`\.`, `\-`, `\*`), and a backslash before a non-punctuation char
264
+ // (`\b`) is a literal backslash. Left in place the engine speaks
265
+ // "backslash b" / "slash b" — exactly the operator's "trash" report.
266
+ // Unescaping here (before the emphasis pass) restores the literal text
267
+ // so genuine `*emphasis*` markers are still stripped downstream while
268
+ // an escaped `\*` collapses to nothing spoken. Runs once; idempotent.
269
+ s = stripBackslashEscapes(s)
270
+
171
271
  // 0. Emoji & pictographs → dropped entirely, then whitespace collapsed.
172
272
  // TTS reads an emoji as its long CLDR name ("grinning face"), which is
173
273
  // noise. We also drop `:shortcode:` forms so nothing is read as
@@ -311,3 +311,74 @@ export function mayInjectListenButton(
311
311
  if (rawKeyboard == null) return true
312
312
  return !rawKeyboard.some((row) => Array.isArray(row) && row.length > 0)
313
313
  }
314
+
315
+ /** The voice-out plan fields the Listen-button decision depends on. A subset of
316
+ * the gateway's full `VoiceOutPlan` so this pure helper carries no gateway
317
+ * import. */
318
+ export interface ListenButtonVoiceOutPlan {
319
+ engine: 'kokoro' | 'openai'
320
+ voice?: string
321
+ speed: number
322
+ replyMode: 'voice+text' | 'voice-only' | 'on-demand'
323
+ ttsChunks: string[]
324
+ }
325
+
326
+ /** The decision to inject a 🔊 Listen button on a message, plus the cache
327
+ * payload the caller must persist so a tap can resynthesize. */
328
+ export interface ListenButtonPlan {
329
+ /** The single-row Listen keyboard to attach as `reply_markup`. */
330
+ replyMarkup: {
331
+ inline_keyboard: Array<Array<{ text: string; callback_data: string }>>
332
+ }
333
+ /** The freshly minted token keying the keyboard callback + cache entry. */
334
+ token: string
335
+ /** The payload the caller stores under `token` in the VoiceOnDemandCache
336
+ * (and enqueues for eager pre-synth). */
337
+ payload: VoiceOnDemandPayload
338
+ }
339
+
340
+ /**
341
+ * The SINGLE source of truth for "should this delivered message carry a 🔊
342
+ * Listen button, and if so which token/keyboard/cache payload?" — shared by
343
+ * BOTH the normal reply path (`sendReply` in outbound-send-path.ts) and the
344
+ * durable-outbox safety-net delivery (`outbox-sweep.ts` wiring), so a
345
+ * net-delivered final answer is indistinguishable from a normally-delivered
346
+ * one (switchroom #3502 regression: the sweep dropped the button).
347
+ *
348
+ * Returns null — no button — when any gate fails:
349
+ * - no voice-out plan, or the plan is not kokoro on-demand (openai on-demand
350
+ * synthesizes immediately and its taps would dead-end on the local sidecar);
351
+ * - the TTS text is empty (nothing to speak — the empty-TTS guard);
352
+ * - the agent supplied its own inline keyboard (single_use collision gate —
353
+ * see mayInjectListenButton).
354
+ *
355
+ * Pure: it mints a token and builds the keyboard/payload but performs NO side
356
+ * effects (no cache write, no eager enqueue). The caller owns those so the
357
+ * token is only persisted when it decides to actually attach the button.
358
+ */
359
+ export function planListenButton(params: {
360
+ voiceOutPlan: ListenButtonVoiceOutPlan | null
361
+ rawKeyboard: unknown[][] | undefined | null
362
+ }): ListenButtonPlan | null {
363
+ const plan = params.voiceOutPlan
364
+ // on-demand Listen button is a LOCAL-engine (kokoro) feature only.
365
+ const useOnDemandButton =
366
+ plan != null && plan.replyMode === 'on-demand' && plan.engine === 'kokoro'
367
+ if (!useOnDemandButton) return null
368
+ // Empty-TTS guard: nothing to speak → no button.
369
+ if (!(plan.ttsChunks.length > 0 && plan.ttsChunks[0]!.length > 0)) return null
370
+ // Collision gate: agent supplied its own keyboard → skip.
371
+ if (!mayInjectListenButton(params.rawKeyboard)) return null
372
+
373
+ const token = mintVoiceOnDemandToken()
374
+ return {
375
+ replyMarkup: buildListenKeyboard(token),
376
+ token,
377
+ payload: {
378
+ // ttsChunks[0] is already normalizeForSpeech(reply) (kokoro path).
379
+ text: plan.ttsChunks[0]!,
380
+ ...(plan.voice != null ? { voice: plan.voice } : {}),
381
+ speed: plan.speed,
382
+ },
383
+ }
384
+ }