switchroom 0.19.15 → 0.19.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/dist/cli/switchroom.js +1 -1
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +1 -1
  4. package/telegram-plugin/dist/bridge/bridge.js +30 -1
  5. package/telegram-plugin/dist/gateway/gateway.js +693 -433
  6. package/telegram-plugin/dist/server.js +30 -1
  7. package/telegram-plugin/gateway/background-shell-liveness.ts +65 -0
  8. package/telegram-plugin/gateway/gateway.ts +7 -58
  9. package/telegram-plugin/gateway/outbound-send-path.ts +25 -23
  10. package/telegram-plugin/gateway/outbox-listen-markup.ts +67 -0
  11. package/telegram-plugin/gateway/outbox-sweep.ts +92 -18
  12. package/telegram-plugin/gateway/rich-message-handler.ts +10 -4
  13. package/telegram-plugin/gateway/silence-poke-session-event.ts +89 -0
  14. package/telegram-plugin/session-tail.ts +88 -1
  15. package/telegram-plugin/silence-poke.ts +118 -1
  16. package/telegram-plugin/tests/background-shell-liveness.test.ts +72 -0
  17. package/telegram-plugin/tests/feed-survival.test.ts +7 -1
  18. package/telegram-plugin/tests/fixtures/bg-shell-liveness-3519.jsonl +3 -0
  19. package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +290 -0
  20. package/telegram-plugin/tests/outbox-sweep-listen-button.test.ts +253 -0
  21. package/telegram-plugin/tests/session-tail.test.ts +91 -1
  22. package/telegram-plugin/tests/silence-poke.test.ts +280 -0
  23. package/telegram-plugin/tests/tts-normalize.test.ts +66 -0
  24. package/telegram-plugin/tests/voice-normalize-text.test.ts +82 -1
  25. package/telegram-plugin/tts-normalize.ts +12 -0
  26. package/telegram-plugin/voice-normalize-text.ts +100 -0
  27. package/telegram-plugin/voice-ondemand.ts +71 -0
@@ -1,4 +1,7 @@
1
1
  import { describe, it, expect, beforeEach, afterEach } from 'vitest'
2
+ import { readFileSync } from 'fs'
3
+ import { join } from 'path'
4
+ import { projectTranscriptLine } from '../session-tail.js'
2
5
  import {
3
6
  startTurn,
4
7
  noteOutbound,
@@ -7,6 +10,9 @@ import {
7
10
  noteToolStart,
8
11
  noteToolEnd,
9
12
  noteToolLabel,
13
+ noteBackgroundShellAlive,
14
+ noteBackgroundShellDead,
15
+ __bgMarkerParserConfirmedForTests,
10
16
  endTurn,
11
17
  silenceMsForKey,
12
18
  silencePokeEnabled,
@@ -30,6 +36,7 @@ interface TestFixtures {
30
36
  function setupDeps(opts?: {
31
37
  thresholds?: Partial<typeof DEFAULT_THRESHOLDS> & { fallbackHardCeiling?: number }
32
38
  deferFallbackWhileToolInFlight?: boolean
39
+ isLegitimatelyWorking?: (key: string) => boolean
33
40
  }): TestFixtures {
34
41
  const fixtures: TestFixtures = { emitted: [], fallbacks: [] }
35
42
  __setDepsForTests({
@@ -42,6 +49,9 @@ function setupDeps(opts?: {
42
49
  ...(opts?.deferFallbackWhileToolInFlight != null
43
50
  ? { deferFallbackWhileToolInFlight: opts.deferFallbackWhileToolInFlight }
44
51
  : {}),
52
+ ...(opts?.isLegitimatelyWorking != null
53
+ ? { isLegitimatelyWorking: opts.isLegitimatelyWorking }
54
+ : {}),
45
55
  })
46
56
  return fixtures
47
57
  }
@@ -601,3 +611,273 @@ describe('silence-poke — Fix A: in-flight-tool defer', () => {
601
611
  expect(f.fallbacks).toHaveLength(0)
602
612
  })
603
613
  })
614
+
615
+ // ─── #3519: CLI-side background-bash defer — the stacked-cards regression ─────
616
+ // Root cause of the operator-visible bug: a single turn ran a foreground `Bash`
617
+ // that the claude CLI auto-moved to the background at its foreground timeout.
618
+ // The tool_result returned (draining `inFlightTools`), so every existing "still
619
+ // working" signal — `isLegitimatelyWorking()`'s foreground/async-dispatch/
620
+ // ask_user checks — went false while the process kept running and the model sat
621
+ // silent waiting on it. At 300s the framework fallback fired: it nulled
622
+ // `currentTurn` and tore down the pinned progress card (the `onFrameworkFallback`
623
+ // callback → liveness-wiring.ts:427-441 `endCurrentTurnForKey`), and the next
624
+ // tool burst minted a BRAND-NEW pinned card. Each > 300s gap repeated it →
625
+ // 2..N stacked cards on ONE turn.
626
+ //
627
+ // These are OUTCOME assertions: `fixtures.fallbacks` counts every fallback fire,
628
+ // and a fire is exactly a currentTurn-null + card teardown (the trigger for a
629
+ // re-minted card). The production wiring is reproduced faithfully:
630
+ // `isLegitimatelyWorking` IS wired (the gateway always wires it) and it returns
631
+ // FALSE for the whole gap (the CLI-side background bash is invisible to it by
632
+ // construction). Reverting the `sawBashThisTurn` defer turns these RED — the
633
+ // fallback fires on every gap and multiple cards are minted.
634
+ describe('silence-poke — #3519 background-bash defer (stacked-cards regression guard)', () => {
635
+ const PROD = {
636
+ thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
637
+ isLegitimatelyWorking: () => false, // background bash is invisible
638
+ }
639
+
640
+ it('a foreground Bash moved to background does NOT tear down the card at 300s (mid-work)', () => {
641
+ const f = setupDeps(PROD)
642
+ startTurn('c:0', 0)
643
+ // Foreground bash starts, then auto-moves to background at ~120s: its
644
+ // tool_result returns so inFlightTools drains. isLegitimatelyWorking is
645
+ // false throughout — every existing signal says "idle".
646
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
647
+ noteToolEnd('c:0', 't1', 125_000)
648
+ // The silent gap the model spends waiting on the background process.
649
+ __tickForTests(300_000)
650
+ __tickForTests(306_000) // well past the 300s base threshold
651
+ // BUG behaviour: fallback fires here (currentTurn nulled, card torn down).
652
+ // FIXED: deferred — the pinned card stays live, no re-mint.
653
+ expect(f.fallbacks).toHaveLength(0)
654
+ })
655
+
656
+ it('exactly ONE card survives a whole turn with two > 300s background-bash gaps', () => {
657
+ const f = setupDeps(PROD)
658
+
659
+ // Model the gateway's pinned-card lifecycle so the surviving card is
660
+ // asserted POSITIVELY, not merely inferred from zero teardowns. In
661
+ // production a fallback fire IS a card teardown (currentTurn nulled +
662
+ // pinned card unpinned, liveness-wiring.ts:427-441), and the next tool
663
+ // burst mints a BRAND-NEW pinned card — that re-mint is exactly the
664
+ // stacking. Here: mint one card when the turn starts, and replay a
665
+ // teardown + fresh re-mint for every fallback the silence-poke tick
666
+ // records. `mintedCards.length` is then the true number of cards that
667
+ // ever existed for the turn, and `pinnedCardId` is the live one.
668
+ const mintedCards: string[] = []
669
+ let pinnedCardId: string | null = null
670
+ const mintCard = (): void => {
671
+ const id = `card-${mintedCards.length + 1}`
672
+ mintedCards.push(id)
673
+ pinnedCardId = id
674
+ }
675
+ const syncCardsAfterTick = (): void => {
676
+ // Each recorded fallback == one teardown of the live card + a fresh
677
+ // re-mint by the resuming burst (the stacked-cards mechanism). Replay
678
+ // any teardown not yet modelled.
679
+ while (mintedCards.length - 1 < f.fallbacks.length) {
680
+ pinnedCardId = null // teardown unpins the live card
681
+ mintCard() // next tool burst mints a brand-new pinned card (a stack)
682
+ }
683
+ }
684
+
685
+ startTurn('c:0', 0)
686
+ mintCard() // the turn's first pinned progress card ("card-1")
687
+ // Gap 1 — first backgrounded find.
688
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name a', 5_000)
689
+ noteToolEnd('c:0', 't1', 125_000)
690
+ __tickForTests(306_000)
691
+ syncCardsAfterTick()
692
+ expect(f.fallbacks).toHaveLength(0) // no Card B minted (zero teardowns)
693
+ // POSITIVE: the ORIGINAL pinned card is still the live one after gap 1,
694
+ // and it is the ONLY card that has ever existed.
695
+ expect(pinnedCardId).toBe('card-1')
696
+ expect(mintedCards).toHaveLength(1)
697
+ // Model resumes and updates the SAME pinned card (production resets the
698
+ // silence clock on that render), then launches a second backgrounded find.
699
+ noteProduction('c:0', 310_000)
700
+ noteToolStart('c:0', 't2', 'Bash', 'find / -name b', 315_000)
701
+ noteToolEnd('c:0', 't2', 435_000)
702
+ // Gap 2 — > 300s of silence since the last render (310_000).
703
+ __tickForTests(620_000)
704
+ syncCardsAfterTick()
705
+ expect(f.fallbacks).toHaveLength(0) // no Card C minted (zero teardowns)
706
+ // POSITIVE: still the same single original card, updated in place across
707
+ // BOTH gaps — no second/third card was ever minted (no stacking).
708
+ expect(pinnedCardId).toBe('card-1')
709
+ expect(mintedCards).toEqual(['card-1'])
710
+ })
711
+
712
+ it('still bounded: a genuinely wedged bash-turn unwedges ONCE at the hard ceiling (not 3×)', () => {
713
+ const f = setupDeps(PROD)
714
+ startTurn('c:0', 0)
715
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
716
+ noteToolEnd('c:0', 't1', 125_000)
717
+ __tickForTests(306_000)
718
+ expect(f.fallbacks).toHaveLength(0) // deferred, not fired 3×
719
+ __tickForTests(900_000) // crosses SILENCE_FALLBACK_HARD_MS
720
+ expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
721
+ })
722
+
723
+ it('the defer is Bash-specific: a non-Bash idle turn still fires at 300s (fix is scoped)', () => {
724
+ const f = setupDeps(PROD)
725
+ startTurn('c:0', 0)
726
+ noteToolStart('c:0', 't1', 'Grep', 'foo', 5_000)
727
+ noteToolEnd('c:0', 't1', 60_000) // Grep can't be a background process
728
+ __tickForTests(306_000)
729
+ expect(f.fallbacks).toHaveLength(1) // genuine silence → unwedges normally
730
+ })
731
+
732
+ it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch', () => {
733
+ const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
734
+ process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
735
+ try {
736
+ const f = setupDeps(PROD)
737
+ startTurn('c:0', 0)
738
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
739
+ noteToolEnd('c:0', 't1', 125_000)
740
+ __tickForTests(306_000)
741
+ expect(f.fallbacks).toHaveLength(1) // defer force-disabled → legacy fire
742
+ } finally {
743
+ if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
744
+ else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
745
+ }
746
+ })
747
+ })
748
+
749
+ // ─── #3519 SHARPEN: PROVEN-alive defer, driven by REAL captured markers ───────
750
+ // The coarse `sawBashThisTurn` guard above deferred EVERY bash-turn to the 900s
751
+ // ceiling. This sharpens it: defer only while a background shell is PROVEN
752
+ // alive (its structured `backgroundTaskId` launch marker seen, no completion
753
+ // yet), and recover at ~300s the moment it finishes. The alive/dead facts here
754
+ // are PARSED FROM REAL JSONL (tests/fixtures/bg-shell-liveness-3519.jsonl —
755
+ // carrie session a6d2d33a-…, claude v2.1.197; line 109 auto-background launch,
756
+ // line 175 `<task-notification>` completion — verbatim but for the operator username
757
+ // scrubbed in both encodings per repo PII policy) through the SAME projectTranscriptLine
758
+ // path the gateway uses, then fed to silence-poke exactly as the gateway feeds
759
+ // it (tool_result.backgroundTaskId → noteBackgroundShellAlive; task_notification
760
+ // → noteBackgroundShellDead). So every fixture value is traceable to a real byte.
761
+ describe('silence-poke — #3519 sharpen: proven-alive defer (real markers)', () => {
762
+ const PROD = {
763
+ thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
764
+ isLegitimatelyWorking: () => false, // background bash invisible to it
765
+ }
766
+ const FIXTURE = join(__dirname, 'fixtures', 'bg-shell-liveness-3519.jsonl')
767
+ const fxLines = readFileSync(FIXTURE, 'utf8').split('\n').filter(l => l.length > 0)
768
+ // Derive the REAL ids straight from the fixtures via the production parser.
769
+ const aliveEv = projectTranscriptLine(fxLines[0]).find(e => e.kind === 'tool_result')
770
+ const deadEv = projectTranscriptLine(fxLines[1])[0]
771
+ const LIVE_ID = aliveEv?.kind === 'tool_result' ? aliveEv.backgroundTaskId! : ''
772
+ const DEAD_ID = deadEv.kind === 'task_notification' ? deadEv.taskId : ''
773
+
774
+ it('sanity: the fixtures really carry a matching background-shell id', () => {
775
+ expect(LIVE_ID).toBe('bxa4sv3dq')
776
+ expect(DEAD_ID).toBe('bxa4sv3dq')
777
+ })
778
+
779
+ // (a) A shell PROVEN alive across a > 300s gap must NOT tear down the card —
780
+ // and exactly ONE card survives (positive assertion, not inferred).
781
+ it('(a) live shell across a > 300s gap → no teardown, exactly one card', () => {
782
+ const f = setupDeps(PROD)
783
+ const mintedCards: string[] = []
784
+ let pinnedCardId: string | null = null
785
+ const mintCard = (): void => {
786
+ mintedCards.push(`card-${mintedCards.length + 1}`)
787
+ pinnedCardId = mintedCards[mintedCards.length - 1]
788
+ }
789
+ const syncCardsAfterTick = (): void => {
790
+ while (mintedCards.length - 1 < f.fallbacks.length) { pinnedCardId = null; mintCard() }
791
+ }
792
+
793
+ startTurn('c:0', 0)
794
+ mintCard() // the turn's first pinned progress card
795
+ // Foreground Bash auto-moved to background at ~120s: its tool_result
796
+ // returns (drains inFlightTools) carrying the real backgroundTaskId.
797
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
798
+ noteToolEnd('c:0', 't1', 125_000)
799
+ noteBackgroundShellAlive('c:0', LIVE_ID) // gateway does this from the parsed event
800
+ __tickForTests(306_000)
801
+ syncCardsAfterTick()
802
+ __tickForTests(500_000) // still alive, still silent, still well past 300s
803
+ syncCardsAfterTick()
804
+ expect(f.fallbacks).toHaveLength(0) // deferred — no teardown
805
+ expect(pinnedCardId).toBe('card-1') // POSITIVE: original card still live
806
+ expect(mintedCards).toEqual(['card-1']) // and it is the ONLY card ever minted
807
+ expect(__bgMarkerParserConfirmedForTests()).toBe(true) // marker proven working
808
+ })
809
+
810
+ // (b) THE NEW capability: once the shell FINISHES (real completion marker),
811
+ // a subsequent > 300s wedge recovers at ~300s — NOT held to 900s.
812
+ it('(b) shell went DEAD then a > 300s wedge → fallback fires at ~300s (fast recovery)', () => {
813
+ const f = setupDeps(PROD)
814
+ startTurn('c:0', 0)
815
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
816
+ noteToolEnd('c:0', 't1', 125_000)
817
+ noteBackgroundShellAlive('c:0', LIVE_ID) // backgrounded at ~120s
818
+ noteBackgroundShellDead('c:0', DEAD_ID) // real <task-notification> completed at ~200s
819
+ // Model then goes silent on a genuine wedge. Because the CLI marker PARSED
820
+ // (confirmed), the coarse sawBash degradation is OFF, so the empty
821
+ // alive-set means this is real silence → recover at the 300s base window,
822
+ // not the 900s ceiling.
823
+ __tickForTests(306_000)
824
+ expect(f.fallbacks).toHaveLength(1) // FAST recovery restored (would be 0 under the old coarse guard)
825
+ })
826
+
827
+ // (c) SAFE DEGRADATION: if a future CLI renames the markers so NOTHING parses
828
+ // (backgroundTaskId never resolves → parser never confirmed), we must not
829
+ // regress to stacking — fall back to the 900s-bounded coarse guard.
830
+ it('(c) unmatched/changed CLI marker → degrades to the 900s-bounded guard', () => {
831
+ const f = setupDeps(PROD)
832
+ // Mutate the REAL alive line so neither the structured field nor the string
833
+ // matches — exactly what a marker rename looks like. Parse it as the gateway
834
+ // would; it yields NO backgroundTaskId, so noteBackgroundShellAlive is never
835
+ // called and the parser stays unconfirmed.
836
+ const mutated = JSON.parse(fxLines[0])
837
+ delete mutated.toolUseResult.backgroundTaskId
838
+ mutated.message.content[0].content = 'Task moved to background (id withheld by a newer CLI).'
839
+ const ev = projectTranscriptLine(JSON.stringify(mutated)).find(e => e.kind === 'tool_result')
840
+ expect(ev?.kind === 'tool_result' ? ev.backgroundTaskId : 'X').toBeUndefined()
841
+
842
+ startTurn('c:0', 0)
843
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000) // sawBashThisTurn armed
844
+ noteToolEnd('c:0', 't1', 125_000)
845
+ // No alive registration (marker unparseable) → parser NOT confirmed.
846
+ expect(__bgMarkerParserConfirmedForTests()).toBe(false)
847
+ __tickForTests(306_000)
848
+ expect(f.fallbacks).toHaveLength(0) // still deferred by the coarse guard (no stacking)
849
+ __tickForTests(900_000) // crosses the hard ceiling
850
+ expect(f.fallbacks).toHaveLength(1) // bounded unwedge — never hangs forever
851
+ })
852
+
853
+ // (d) Still bounded even while genuinely alive: a shell that never reports
854
+ // dead still unwedges ONCE at the 900s ceiling (not indefinitely).
855
+ it('(d) a never-completing live shell still unwedges once at the 900s ceiling', () => {
856
+ const f = setupDeps(PROD)
857
+ startTurn('c:0', 0)
858
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
859
+ noteToolEnd('c:0', 't1', 125_000)
860
+ noteBackgroundShellAlive('c:0', LIVE_ID) // alive, never dies
861
+ __tickForTests(306_000)
862
+ expect(f.fallbacks).toHaveLength(0) // deferred while alive
863
+ __tickForTests(900_000) // hard ceiling
864
+ expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
865
+ })
866
+
867
+ it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch even with a live shell', () => {
868
+ const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
869
+ process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
870
+ try {
871
+ const f = setupDeps(PROD)
872
+ startTurn('c:0', 0)
873
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
874
+ noteToolEnd('c:0', 't1', 125_000)
875
+ noteBackgroundShellAlive('c:0', LIVE_ID)
876
+ __tickForTests(306_000)
877
+ expect(f.fallbacks).toHaveLength(1) // force-disabled → no defer at all
878
+ } finally {
879
+ if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
880
+ else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
881
+ }
882
+ })
883
+ })
@@ -4,6 +4,7 @@
4
4
  */
5
5
  import { describe, test, expect, beforeEach, afterEach } from 'bun:test'
6
6
  import { normalizeForTts, ttsNormalizeEnabled } from '../tts-normalize.js'
7
+ import { normalizeForSpeech } from '../voice-normalize-text.js'
7
8
 
8
9
  const KILL = 'SWITCHROOM_DISABLE_TTS_NORMALIZE'
9
10
 
@@ -240,3 +241,68 @@ describe('conservatism', () => {
240
241
  expect(normalizeForTts(input)).toBe(normalizeForTts(input))
241
242
  })
242
243
  })
244
+
245
+ describe('normalizeForTts — backslash escapes & HTML entities (last-line defence)', () => {
246
+ test('strips a literal \\b so the engine never speaks "backslash b"', () => {
247
+ const out = normalizeForTts('the regex \\b boundary')
248
+ expect(out).toBe('the regex b boundary')
249
+ expect(out).not.toContain('\\')
250
+ })
251
+
252
+ test('unescapes MarkdownV2 punctuation escapes (\\. \\! \\-)', () => {
253
+ expect(normalizeForTts('done\\. next\\! wait\\-')).toBe('done. next! wait-')
254
+ })
255
+
256
+ test('decodes HTML entities (&amp; &lt; &#39;)', () => {
257
+ expect(normalizeForTts('Tom &amp; Jerry')).toBe('Tom and Jerry')
258
+ expect(normalizeForTts('5 &lt; 10')).toBe('5 < 10')
259
+ expect(normalizeForTts("it&#39;s here")).toBe("it's here")
260
+ })
261
+
262
+ test('rich mixed reply → clean spoken text (no backslash/backtick/entity)', () => {
263
+ const reply =
264
+ '**Bold** and `code\\b` and a [label](https://example.com/x) ' +
265
+ 'with Tom &amp; Jerry and a regex \\b\\.'
266
+ const out = normalizeForTts(reply)
267
+ expect(out).not.toContain('\\')
268
+ expect(out).not.toContain('`')
269
+ expect(out).not.toContain('&amp;')
270
+ expect(out).toContain('label')
271
+ expect(out).toContain('Tom and Jerry')
272
+ })
273
+
274
+ test('kill switch still returns byte-identical input (escapes preserved)', () => {
275
+ process.env[KILL] = '1'
276
+ expect(normalizeForTts('a \\b &amp; b')).toBe('a \\b &amp; b')
277
+ delete process.env[KILL]
278
+ })
279
+
280
+ test('idempotent after normalizeForSpeech already unescaped', () => {
281
+ const reply = 'a \\b and Tom &amp; Jerry \\. end'
282
+ const once = normalizeForTts(reply)
283
+ expect(normalizeForTts(once)).toBe(once)
284
+ })
285
+ })
286
+
287
+ describe('normalizeForTts — review findings (fixpoint decode, metachar, nits)', () => {
288
+ test('L2 parity: immediate (speech+tts) and single-tts agree on a double-encoded entity', () => {
289
+ const x = '&amp;amp;lt;'
290
+ const immediate = normalizeForTts(normalizeForSpeech(x))
291
+ const singleTts = normalizeForTts(x)
292
+ expect(immediate).toBe('<')
293
+ expect(singleTts).toBe('<')
294
+ expect(immediate).toBe(singleTts)
295
+ expect(normalizeForTts('&amp;amp;amp;')).toBe('&')
296
+ })
297
+
298
+ test('L1: entity → line-leading metachar keeps a spoken form (hash/asterisk)', () => {
299
+ expect(normalizeForTts('&#35; Heading')).toBe('hash Heading')
300
+ expect(normalizeForTts('2 &#42; 3')).toBe('2 asterisk 3')
301
+ })
302
+
303
+ test('nit: dangling trailing backslash dropped; &#92; decodes then strips', () => {
304
+ expect(normalizeForTts('ends here\\')).toBe('ends here')
305
+ expect(normalizeForTts('X&#92;Y')).toBe('XY')
306
+ expect(normalizeForTts('X&#92;Y')).not.toContain('\\')
307
+ })
308
+ })
@@ -6,7 +6,7 @@
6
6
  */
7
7
 
8
8
  import { describe, it, expect } from 'bun:test'
9
- import { normalizeForSpeech } from '../voice-normalize-text.js'
9
+ import { normalizeForSpeech, decodeHtmlEntities } from '../voice-normalize-text.js'
10
10
 
11
11
  describe('normalizeForSpeech — reported markdown/symbol cases', () => {
12
12
  it('drops stray tildes (never spoken as "tilde")', () => {
@@ -254,3 +254,84 @@ describe('normalizeForSpeech — conservative, does not mangle real words', () =
254
254
  expect(normalizeForSpeech(' \n ')).toBe('')
255
255
  })
256
256
  })
257
+
258
+ describe('normalizeForSpeech — backslash escapes & HTML entities (voice trash)', () => {
259
+ it('strips a literal backslash-b (\\b) so it is never spoken as "backslash b"', () => {
260
+ // The exact operator report: the regex/escape "\b" was read as "slash b".
261
+ const out = normalizeForSpeech('the regex \\b word boundary')
262
+ expect(out).toBe('the regex b word boundary')
263
+ expect(out).not.toContain('\\')
264
+ })
265
+
266
+ it('unescapes MarkdownV2 punctuation escapes without leaving backslashes', () => {
267
+ expect(normalizeForSpeech('done\\. next\\! wait\\-')).toBe('done. next! wait-')
268
+ expect(normalizeForSpeech('a \\* b')).not.toContain('\\')
269
+ })
270
+
271
+ it('an escaped emphasis marker collapses to the inner word, not a backslash', () => {
272
+ expect(normalizeForSpeech('Use \\*literal\\* here')).toBe('Use literal here')
273
+ })
274
+
275
+ it('drops a Windows-path-style backslash run (C:\\build)', () => {
276
+ expect(normalizeForSpeech('path C:\\build\\out done')).toBe('path C:buildout done')
277
+ })
278
+
279
+ it('decodes HTML entities so &amp; / &lt; are not read as "amp" / "lt"', () => {
280
+ expect(normalizeForSpeech('Tom &amp; Jerry')).toBe('Tom and Jerry')
281
+ expect(normalizeForSpeech('5 &lt; 10 &gt; 3')).toBe('5 < 10 > 3')
282
+ expect(normalizeForSpeech('it&#39;s here')).toBe("it's here")
283
+ })
284
+
285
+ it('produces clean plain speech for a rich mixed reply (end-to-end)', () => {
286
+ const reply =
287
+ '**Bold** and `code\\b` and a [label](https://example.com/x) ' +
288
+ 'with Tom &amp; Jerry and a regex \\b\\.'
289
+ const out = normalizeForSpeech(reply)
290
+ expect(out).not.toContain('\\')
291
+ expect(out).not.toContain('`')
292
+ expect(out).not.toContain('*')
293
+ expect(out).not.toContain('&amp;')
294
+ expect(out).not.toContain('example.com')
295
+ expect(out).toContain('label')
296
+ expect(out).toContain('Tom and Jerry')
297
+ })
298
+
299
+ it('is idempotent: a second pass finds no backslashes/entities to change', () => {
300
+ const reply = 'a \\b and Tom &amp; Jerry \\. end'
301
+ const once = normalizeForSpeech(reply)
302
+ expect(normalizeForSpeech(once)).toBe(once)
303
+ })
304
+ })
305
+
306
+ describe('normalizeForSpeech — review findings (fixpoint decode, metachar, nits)', () => {
307
+ it('L2 parity: a double-encoded entity decodes to a fixpoint (depth-independent)', () => {
308
+ // Single decode and double decode must land on the SAME spoken char so the
309
+ // immediate voice-out and the lazy Listen tap never diverge.
310
+ const single = normalizeForSpeech('&amp;amp;lt;')
311
+ const doubled = normalizeForSpeech(normalizeForSpeech('&amp;amp;lt;'))
312
+ expect(single).toBe('<')
313
+ expect(doubled).toBe('<')
314
+ expect(normalizeForSpeech('&amp;amp;amp;')).toBe('&')
315
+ })
316
+
317
+ it('L1: an entity that decodes to a line-leading metachar keeps a spoken form', () => {
318
+ // &#35; → '#'. Naively re-fed to the heading stripper it would vanish; the
319
+ // user escaped it on purpose, so it must survive as spoken "hash".
320
+ expect(normalizeForSpeech('&#35; Heading')).toBe('hash Heading')
321
+ expect(normalizeForSpeech('2 &#42; 3')).toBe('2 asterisk 3')
322
+ })
323
+
324
+ it('nit: a dangling trailing backslash is dropped, never spoken', () => {
325
+ expect(normalizeForSpeech('ends here\\')).toBe('ends here')
326
+ expect(normalizeForSpeech('ends here\\')).not.toContain('\\')
327
+ })
328
+
329
+ it('nit: &#92; decodes to a backslash which is then stripped (no trash)', () => {
330
+ expect(normalizeForSpeech('X&#92;Y')).toBe('XY')
331
+ expect(normalizeForSpeech('X&#92;Y')).not.toContain('\\')
332
+ })
333
+
334
+ it('fixpoint does not over-decode entity-less text (Q&A stays literal)', () => {
335
+ expect(decodeHtmlEntities('Q&A test')).toBe('Q&A test')
336
+ })
337
+ })
@@ -38,6 +38,8 @@
38
38
  * ~5 → "about 5", > blockquote markers dropped.
39
39
  */
40
40
 
41
+ import { decodeHtmlEntities, stripBackslashEscapes } from './voice-normalize-text'
42
+
41
43
  const NULL = '\x00'
42
44
  const INLINE_PH = `${NULL}TN_INLINE`
43
45
 
@@ -197,6 +199,16 @@ export function normalizeForTts(text: string): string {
197
199
 
198
200
  let s = text.replace(/\r\n?/g, '\n')
199
201
 
202
+ // -- HTML entities → char, then backslash escapes → the escaped char. The
203
+ // last line of defence at the /tts body build: the Listen lazy path and
204
+ // the pre-synth queue can synthesize from cache entries that predate the
205
+ // normalizeForSpeech coverage, so these must be stripped here too. Both
206
+ // are idempotent — if normalizeForSpeech already ran there is nothing
207
+ // left to decode/unescape. Without this the engine speaks `\b` as
208
+ // "backslash b" and `&amp;` as "amp".
209
+ s = decodeHtmlEntities(s)
210
+ s = stripBackslashEscapes(s)
211
+
200
212
  // -- Code fences → spoken placeholder (before anything can see contents).
201
213
  s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*?\n[ \t]*\2[ \t]*(?=\n|$)/g, '$1code block omitted.')
202
214
  s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*$/g, '$1code block omitted.')
@@ -53,6 +53,89 @@
53
53
  /** Replace a fenced code block with a spoken placeholder. */
54
54
  const CODE_BLOCK_PLACEHOLDER = 'code block omitted'
55
55
 
56
+ /** Named HTML entities the reply text realistically carries. */
57
+ const HTML_ENTITIES: Record<string, string> = {
58
+ amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ',
59
+ }
60
+
61
+ /**
62
+ * Markdown metacharacters that the block/emphasis/table stripper would
63
+ * silently consume at line-start or as a pair. When such a char arrives via
64
+ * an entity escape the user meant it LITERALLY (that is the whole point of
65
+ * escaping it), so instead of emitting the raw char — which the downstream
66
+ * stripper would then eat, losing the intent — we emit a neutral spoken form
67
+ * that survives every later pass. Deterministic; the spoken form contains no
68
+ * `&`/`;` so it can never re-enter the entity decoder.
69
+ */
70
+ const METACHAR_SPOKEN: Record<string, string> = {
71
+ '#': ' hash ',
72
+ '*': ' asterisk ',
73
+ '_': ' underscore ',
74
+ '~': ' tilde ',
75
+ '`': ' backtick ',
76
+ '|': ' bar ',
77
+ }
78
+
79
+ /** One decode pass: named + numeric entities → char (or spoken metachar). */
80
+ function decodeHtmlEntitiesOnce(input: string): string {
81
+ const toChar = (cp: number, raw: string): string => {
82
+ if (!(cp > 0 && cp <= 0x10ffff)) return raw
83
+ const ch = String.fromCodePoint(cp)
84
+ return METACHAR_SPOKEN[ch] ?? ch
85
+ }
86
+ return input
87
+ .replace(/&#x([0-9a-f]+);/gi, (m, hex: string) => toChar(parseInt(hex, 16), m))
88
+ .replace(/&#(\d+);/g, (m, dec: string) => toChar(Number(dec), m))
89
+ .replace(/&([a-z][a-z0-9]*);/gi, (m, name: string) => {
90
+ const ch = HTML_ENTITIES[name.toLowerCase()]
91
+ if (ch === undefined) return m
92
+ return METACHAR_SPOKEN[ch] ?? ch
93
+ })
94
+ }
95
+
96
+ /**
97
+ * Decode HTML entities (named + numeric) to their character so a TTS engine
98
+ * never reads `&amp;` as "amp". Unknown named entities are left untouched.
99
+ * Pure + deterministic.
100
+ *
101
+ * Iterates to a FIXPOINT: a double-encoded entity (`&amp;amp;lt;`) is decoded
102
+ * repeatedly until no entity remains, so this pass is depth-idempotent —
103
+ * applying it once yields the same result as applying it twice. That keeps
104
+ * every voice callsite in lockstep: the immediate voice-out path runs
105
+ * normalizeForSpeech THEN normalizeForTts, and the value the lazy Listen tap /
106
+ * pre-synth queue reads is itself already normalizeForSpeech'd before its own
107
+ * normalizeForTts — fixpoint decoding guarantees both speak an identical
108
+ * string regardless of how deep the original encoding was. The loop strictly
109
+ * shrinks the entity count each turn (and is capped) so it always terminates.
110
+ * Text WITHOUT a trailing `;` (e.g. `Q&A`) matches nothing and is returned
111
+ * untouched.
112
+ */
113
+ export function decodeHtmlEntities(input: string): string {
114
+ let s = input
115
+ // A fully-decodable chain shrinks by at least one entity per pass; the cap
116
+ // is a belt-and-braces guard against any pathological crafted input.
117
+ for (let i = 0; i < 10; i++) {
118
+ const next = decodeHtmlEntitiesOnce(s)
119
+ if (next === s) break
120
+ s = next
121
+ }
122
+ return s
123
+ }
124
+
125
+ /**
126
+ * Remove markdown/MarkdownV2 backslash escapes so the spoken text carries no
127
+ * literal backslashes. A backslash before ANY single character is dropped,
128
+ * keeping the character (`\.` → ".", `\*` → "*", `\b` → "b"); a dangling
129
+ * trailing backslash is dropped. Pure + deterministic + idempotent (a second
130
+ * pass finds no backslashes). A real newline is preserved (only the escaping
131
+ * backslash is consumed).
132
+ */
133
+ export function stripBackslashEscapes(input: string): string {
134
+ // `\X` → `X` for any following char (including an escaped `\\`), then drop
135
+ // any lone backslash the first pass left (an escaped backslash's survivor).
136
+ return input.replace(/\\([\s\S])/g, '$1').replace(/\\/g, '')
137
+ }
138
+
56
139
  // ---------------------------------------------------------------------------
57
140
  // Number → words helpers (small, deterministic, English cardinal only).
58
141
  // Used by the numbers/units pass. Supports 0..999_999_999 which is far more
@@ -168,6 +251,23 @@ export function normalizeForSpeech(input: string): string {
168
251
  if (!input) return ''
169
252
  let s = input.replace(/\r\n?/g, '\n')
170
253
 
254
+ // 0a. HTML entities → their character. The reply text can carry entity
255
+ // escapes (`&amp;`, `&lt;`, `&#39;`) that a TTS engine would otherwise
256
+ // read as "amp" / "lt" / a digit run. Decode BEFORE markdown/symbol
257
+ // passes so the recovered char is then handled naturally (e.g. a
258
+ // decoded `&` becomes "and" in the symbols pass).
259
+ s = decodeHtmlEntities(s)
260
+
261
+ // 0b. Backslash escapes → the escaped character. Telegram MarkdownV2 and
262
+ // CommonMark escape literal punctuation with a leading backslash
263
+ // (`\.`, `\-`, `\*`), and a backslash before a non-punctuation char
264
+ // (`\b`) is a literal backslash. Left in place the engine speaks
265
+ // "backslash b" / "slash b" — exactly the operator's "trash" report.
266
+ // Unescaping here (before the emphasis pass) restores the literal text
267
+ // so genuine `*emphasis*` markers are still stripped downstream while
268
+ // an escaped `\*` collapses to nothing spoken. Runs once; idempotent.
269
+ s = stripBackslashEscapes(s)
270
+
171
271
  // 0. Emoji & pictographs → dropped entirely, then whitespace collapsed.
172
272
  // TTS reads an emoji as its long CLDR name ("grinning face"), which is
173
273
  // noise. We also drop `:shortcode:` forms so nothing is read as