switchroom 0.19.14 → 0.19.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/dist/cli/switchroom.js +1 -1
  2. package/dist/host-control/main.js +1 -1
  3. package/package.json +1 -1
  4. package/telegram-plugin/bridge/bridge.ts +1 -1
  5. package/telegram-plugin/dist/bridge/bridge.js +31 -2
  6. package/telegram-plugin/dist/gateway/gateway.js +1690 -932
  7. package/telegram-plugin/dist/server.js +31 -2
  8. package/telegram-plugin/gateway/background-shell-liveness.ts +65 -0
  9. package/telegram-plugin/gateway/forward-origin.ts +6 -1
  10. package/telegram-plugin/gateway/gateway.ts +10 -57
  11. package/telegram-plugin/gateway/narrative-lane.ts +11 -0
  12. package/telegram-plugin/gateway/outbound-send-path.ts +25 -23
  13. package/telegram-plugin/gateway/outbox-listen-markup.ts +67 -0
  14. package/telegram-plugin/gateway/outbox-sweep.ts +124 -20
  15. package/telegram-plugin/gateway/rich-message-handler.ts +241 -0
  16. package/telegram-plugin/gateway/silence-poke-session-event.ts +89 -0
  17. package/telegram-plugin/gateway/stream-render.ts +107 -15
  18. package/telegram-plugin/gateway/unhandled-message.ts +14 -0
  19. package/telegram-plugin/hooks/narration-classify.d.mts +23 -0
  20. package/telegram-plugin/hooks/narration-classify.mjs +210 -0
  21. package/telegram-plugin/hooks/silent-end-scan.mjs +136 -82
  22. package/telegram-plugin/narrative-flush.ts +35 -0
  23. package/telegram-plugin/outbox.ts +73 -3
  24. package/telegram-plugin/session-tail.ts +88 -1
  25. package/telegram-plugin/shown-ledger.ts +145 -0
  26. package/telegram-plugin/silence-poke.ts +118 -1
  27. package/telegram-plugin/silent-end.ts +42 -0
  28. package/telegram-plugin/tests/background-shell-liveness.test.ts +72 -0
  29. package/telegram-plugin/tests/backstop-exactly-once.test.ts +335 -0
  30. package/telegram-plugin/tests/catch-all-unhandled-message.test.ts +14 -0
  31. package/telegram-plugin/tests/feed-survival.test.ts +7 -1
  32. package/telegram-plugin/tests/fixtures/bg-shell-liveness-3519.jsonl +3 -0
  33. package/telegram-plugin/tests/forward-origin.test.ts +20 -0
  34. package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +290 -0
  35. package/telegram-plugin/tests/forwarded-rich-message.test.ts +305 -0
  36. package/telegram-plugin/tests/gateway-handler-registration-wiring.test.ts +1 -0
  37. package/telegram-plugin/tests/narration-leak-3513.test.ts +352 -0
  38. package/telegram-plugin/tests/outbox-sweep-listen-button.test.ts +253 -0
  39. package/telegram-plugin/tests/session-tail.test.ts +91 -1
  40. package/telegram-plugin/tests/silence-poke.test.ts +280 -0
  41. package/telegram-plugin/tests/silent-end-interrupt-stop-scan.test.ts +42 -13
  42. package/telegram-plugin/tests/silent-end.test.ts +7 -1
  43. package/telegram-plugin/tests/tts-normalize.test.ts +66 -0
  44. package/telegram-plugin/tests/turn-flush-safety.test.ts +35 -3
  45. package/telegram-plugin/tests/voice-normalize-text.test.ts +82 -1
  46. package/telegram-plugin/tts-normalize.ts +12 -0
  47. package/telegram-plugin/turn-flush-safety.ts +66 -53
  48. package/telegram-plugin/voice-normalize-text.ts +100 -0
  49. package/telegram-plugin/voice-ondemand.ts +71 -0
@@ -1,5 +1,5 @@
1
1
  import { describe, it, expect, afterEach } from 'vitest'
2
- import { mkdtempSync, mkdirSync, writeFileSync, appendFileSync, rmSync, utimesSync } from 'fs'
2
+ import { mkdtempSync, mkdirSync, writeFileSync, appendFileSync, rmSync, utimesSync, readFileSync } from 'fs'
3
3
  import { tmpdir } from 'os'
4
4
  import { join } from 'path'
5
5
  import {
@@ -1018,3 +1018,93 @@ describe('projectAssistantTextBlocks (shared text→narrative kernel)', () => {
1018
1018
  })
1019
1019
  })
1020
1020
  })
1021
+
1022
+ // ─── #3519 sharpen: background-shell liveness markers, from REAL captured data ─
1023
+ // Fixtures are VERBATIM lines from a real agent transcript — no hand-written
1024
+ // approximations. Provenance (cited so a reviewer can independently verify;
1025
+ // this is a SURVIVING, reachable session — the earlier 1db49136 session was
1026
+ // rotated off host, so the fixture was regenerated from this one):
1027
+ // file: /host-home/.switchroom/agents/carrie/.claude/projects/
1028
+ // -home-kenthompson--switchroom-agents-carrie/
1029
+ // a6d2d33a-a8a6-40ce-81d0-cb4bd867ac89.jsonl (claude CLI v2.1.197)
1030
+ // line 109 → ALIVE: a FOREGROUND Bash (tool_use input has NO
1031
+ // run_in_background) that the CLI auto-moved to the background at its
1032
+ // ~120s foreground window (tool_result at 00:35:56Z) — the exact
1033
+ // #3519 auto-background case. Carries BOTH the structured
1034
+ // `toolUseResult.backgroundTaskId:"bxa4sv3dq"` and the launch string
1035
+ // "Command running in background with ID: bxa4sv3dq. …".
1036
+ // line 175 → DEAD: the CLI's proactive `<task-notification>` for the SAME id
1037
+ // (`<task-id>bxa4sv3dq</task-id>`, `<status>completed</status>`),
1038
+ // enqueued as a queue-operation.
1039
+ // line 180 → the mirrored `type:"attachment"` copy of the same notification.
1040
+ // The three lines are copied into tests/fixtures/bg-shell-liveness-3519.jsonl
1041
+ // byte-for-byte EXCEPT the operator username, scrubbed in BOTH encodings — the
1042
+ // slash home path (`/home/<user>` → `~`) and the dashed tmp-path form
1043
+ // (`-home-<user>-` → `-home-user-`) — per repo PII policy
1044
+ // (scripts/check-no-pii-secrets.mjs). Every marker-bearing field —
1045
+ // backgroundTaskId, the launch string + id, the <task-notification> tags — is
1046
+ // untouched.
1047
+ describe('projectTranscriptLine — #3519 background-shell liveness (real fixtures)', () => {
1048
+ const FIXTURE = join(__dirname, 'fixtures', 'bg-shell-liveness-3519.jsonl')
1049
+ const lines = readFileSync(FIXTURE, 'utf8').split('\n').filter(l => l.length > 0)
1050
+ const [aliveLine, deadEnqueueLine] = lines
1051
+
1052
+ it('ALIVE: emits backgroundTaskId from the real structured toolUseResult field', () => {
1053
+ // The whole plumbing seam end-to-end: the raw captured bytes → the parsed
1054
+ // tool_result event actually CARRIES backgroundTaskId, so the gateway can
1055
+ // register the shell alive. If this field weren't in the real event, the
1056
+ // signal would be unavailable — this proves it is.
1057
+ const events = projectTranscriptLine(aliveLine)
1058
+ const tr = events.find(e => e.kind === 'tool_result')
1059
+ expect(tr).toBeDefined()
1060
+ expect(tr).toMatchObject({
1061
+ kind: 'tool_result',
1062
+ toolUseId: 'toolu_01B7T3y1t95oHDEqYKwSmqaW',
1063
+ backgroundTaskId: 'bxa4sv3dq',
1064
+ })
1065
+ })
1066
+
1067
+ it('DEAD: projects the real <task-notification> enqueue as a task_notification (completed)', () => {
1068
+ // Proves the DEAD seam: the CLI's completion signal — which arrives as a
1069
+ // queue-operation enqueue, NOT a tool_result — is parsed to the shell id +
1070
+ // status the gateway drops from the alive-set. It must NOT be mis-read as
1071
+ // an inbound user turn (no `enqueue` event).
1072
+ const events = projectTranscriptLine(deadEnqueueLine)
1073
+ expect(events).toEqual([
1074
+ { kind: 'task_notification', taskId: 'bxa4sv3dq', status: 'completed' },
1075
+ ])
1076
+ })
1077
+
1078
+ it('the ALIVE and DEAD ids MATCH — a real launch pairs with its real completion', () => {
1079
+ const alive = projectTranscriptLine(aliveLine).find(e => e.kind === 'tool_result')
1080
+ const dead = projectTranscriptLine(deadEnqueueLine)[0]
1081
+ expect(alive?.kind === 'tool_result' ? alive.backgroundTaskId : null)
1082
+ .toBe(dead.kind === 'task_notification' ? dead.taskId : undefined)
1083
+ })
1084
+
1085
+ it('SAFE DEGRADATION: a CLI that renamed the marker yields NO backgroundTaskId', () => {
1086
+ // Simulate a future CLI that changed the launch shape: drop the structured
1087
+ // field AND mutate the launch string. The parser must return NO id (so the
1088
+ // gateway never registers a phantom-alive shell), which is what lets the
1089
+ // silence-poke layer fall back to the 900s-bounded guard. Built by mutating
1090
+ // the REAL line, so it stays traceable.
1091
+ const obj = JSON.parse(aliveLine)
1092
+ delete obj.toolUseResult.backgroundTaskId // structured field gone
1093
+ obj.message.content[0].content =
1094
+ 'Task moved to the background (id withheld by a newer CLI).' // string changed
1095
+ const events = projectTranscriptLine(JSON.stringify(obj))
1096
+ const tr = events.find(e => e.kind === 'tool_result')
1097
+ expect(tr).toBeDefined()
1098
+ expect(tr?.kind === 'tool_result' ? tr.backgroundTaskId : 'X').toBeUndefined()
1099
+ })
1100
+
1101
+ it('secondary path: the launch STRING alone still yields the id when the field is absent', () => {
1102
+ // If a CLI keeps the human string but drops the structured field, the regex
1103
+ // fallback still recovers the id — traceable, built from the real line.
1104
+ const obj = JSON.parse(aliveLine)
1105
+ delete obj.toolUseResult.backgroundTaskId
1106
+ const events = projectTranscriptLine(JSON.stringify(obj))
1107
+ const tr = events.find(e => e.kind === 'tool_result')
1108
+ expect(tr?.kind === 'tool_result' ? tr.backgroundTaskId : null).toBe('bxa4sv3dq')
1109
+ })
1110
+ })
@@ -1,4 +1,7 @@
1
1
  import { describe, it, expect, beforeEach, afterEach } from 'vitest'
2
+ import { readFileSync } from 'fs'
3
+ import { join } from 'path'
4
+ import { projectTranscriptLine } from '../session-tail.js'
2
5
  import {
3
6
  startTurn,
4
7
  noteOutbound,
@@ -7,6 +10,9 @@ import {
7
10
  noteToolStart,
8
11
  noteToolEnd,
9
12
  noteToolLabel,
13
+ noteBackgroundShellAlive,
14
+ noteBackgroundShellDead,
15
+ __bgMarkerParserConfirmedForTests,
10
16
  endTurn,
11
17
  silenceMsForKey,
12
18
  silencePokeEnabled,
@@ -30,6 +36,7 @@ interface TestFixtures {
30
36
  function setupDeps(opts?: {
31
37
  thresholds?: Partial<typeof DEFAULT_THRESHOLDS> & { fallbackHardCeiling?: number }
32
38
  deferFallbackWhileToolInFlight?: boolean
39
+ isLegitimatelyWorking?: (key: string) => boolean
33
40
  }): TestFixtures {
34
41
  const fixtures: TestFixtures = { emitted: [], fallbacks: [] }
35
42
  __setDepsForTests({
@@ -42,6 +49,9 @@ function setupDeps(opts?: {
42
49
  ...(opts?.deferFallbackWhileToolInFlight != null
43
50
  ? { deferFallbackWhileToolInFlight: opts.deferFallbackWhileToolInFlight }
44
51
  : {}),
52
+ ...(opts?.isLegitimatelyWorking != null
53
+ ? { isLegitimatelyWorking: opts.isLegitimatelyWorking }
54
+ : {}),
45
55
  })
46
56
  return fixtures
47
57
  }
@@ -601,3 +611,273 @@ describe('silence-poke — Fix A: in-flight-tool defer', () => {
601
611
  expect(f.fallbacks).toHaveLength(0)
602
612
  })
603
613
  })
614
+
615
+ // ─── #3519: CLI-side background-bash defer — the stacked-cards regression ─────
616
+ // Root cause of the operator-visible bug: a single turn ran a foreground `Bash`
617
+ // that the claude CLI auto-moved to the background at its foreground timeout.
618
+ // The tool_result returned (draining `inFlightTools`), so every existing "still
619
+ // working" signal — `isLegitimatelyWorking()`'s foreground/async-dispatch/
620
+ // ask_user checks — went false while the process kept running and the model sat
621
+ // silent waiting on it. At 300s the framework fallback fired: it nulled
622
+ // `currentTurn` and tore down the pinned progress card (the `onFrameworkFallback`
623
+ // callback → liveness-wiring.ts:427-441 `endCurrentTurnForKey`), and the next
624
+ // tool burst minted a BRAND-NEW pinned card. Each > 300s gap repeated it →
625
+ // 2..N stacked cards on ONE turn.
626
+ //
627
+ // These are OUTCOME assertions: `fixtures.fallbacks` counts every fallback fire,
628
+ // and a fire is exactly a currentTurn-null + card teardown (the trigger for a
629
+ // re-minted card). The production wiring is reproduced faithfully:
630
+ // `isLegitimatelyWorking` IS wired (the gateway always wires it) and it returns
631
+ // FALSE for the whole gap (the CLI-side background bash is invisible to it by
632
+ // construction). Reverting the `sawBashThisTurn` defer turns these RED — the
633
+ // fallback fires on every gap and multiple cards are minted.
634
+ describe('silence-poke — #3519 background-bash defer (stacked-cards regression guard)', () => {
635
+ const PROD = {
636
+ thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
637
+ isLegitimatelyWorking: () => false, // background bash is invisible
638
+ }
639
+
640
+ it('a foreground Bash moved to background does NOT tear down the card at 300s (mid-work)', () => {
641
+ const f = setupDeps(PROD)
642
+ startTurn('c:0', 0)
643
+ // Foreground bash starts, then auto-moves to background at ~120s: its
644
+ // tool_result returns so inFlightTools drains. isLegitimatelyWorking is
645
+ // false throughout — every existing signal says "idle".
646
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
647
+ noteToolEnd('c:0', 't1', 125_000)
648
+ // The silent gap the model spends waiting on the background process.
649
+ __tickForTests(300_000)
650
+ __tickForTests(306_000) // well past the 300s base threshold
651
+ // BUG behaviour: fallback fires here (currentTurn nulled, card torn down).
652
+ // FIXED: deferred — the pinned card stays live, no re-mint.
653
+ expect(f.fallbacks).toHaveLength(0)
654
+ })
655
+
656
+ it('exactly ONE card survives a whole turn with two > 300s background-bash gaps', () => {
657
+ const f = setupDeps(PROD)
658
+
659
+ // Model the gateway's pinned-card lifecycle so the surviving card is
660
+ // asserted POSITIVELY, not merely inferred from zero teardowns. In
661
+ // production a fallback fire IS a card teardown (currentTurn nulled +
662
+ // pinned card unpinned, liveness-wiring.ts:427-441), and the next tool
663
+ // burst mints a BRAND-NEW pinned card — that re-mint is exactly the
664
+ // stacking. Here: mint one card when the turn starts, and replay a
665
+ // teardown + fresh re-mint for every fallback the silence-poke tick
666
+ // records. `mintedCards.length` is then the true number of cards that
667
+ // ever existed for the turn, and `pinnedCardId` is the live one.
668
+ const mintedCards: string[] = []
669
+ let pinnedCardId: string | null = null
670
+ const mintCard = (): void => {
671
+ const id = `card-${mintedCards.length + 1}`
672
+ mintedCards.push(id)
673
+ pinnedCardId = id
674
+ }
675
+ const syncCardsAfterTick = (): void => {
676
+ // Each recorded fallback == one teardown of the live card + a fresh
677
+ // re-mint by the resuming burst (the stacked-cards mechanism). Replay
678
+ // any teardown not yet modelled.
679
+ while (mintedCards.length - 1 < f.fallbacks.length) {
680
+ pinnedCardId = null // teardown unpins the live card
681
+ mintCard() // next tool burst mints a brand-new pinned card (a stack)
682
+ }
683
+ }
684
+
685
+ startTurn('c:0', 0)
686
+ mintCard() // the turn's first pinned progress card ("card-1")
687
+ // Gap 1 — first backgrounded find.
688
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name a', 5_000)
689
+ noteToolEnd('c:0', 't1', 125_000)
690
+ __tickForTests(306_000)
691
+ syncCardsAfterTick()
692
+ expect(f.fallbacks).toHaveLength(0) // no Card B minted (zero teardowns)
693
+ // POSITIVE: the ORIGINAL pinned card is still the live one after gap 1,
694
+ // and it is the ONLY card that has ever existed.
695
+ expect(pinnedCardId).toBe('card-1')
696
+ expect(mintedCards).toHaveLength(1)
697
+ // Model resumes and updates the SAME pinned card (production resets the
698
+ // silence clock on that render), then launches a second backgrounded find.
699
+ noteProduction('c:0', 310_000)
700
+ noteToolStart('c:0', 't2', 'Bash', 'find / -name b', 315_000)
701
+ noteToolEnd('c:0', 't2', 435_000)
702
+ // Gap 2 — > 300s of silence since the last render (310_000).
703
+ __tickForTests(620_000)
704
+ syncCardsAfterTick()
705
+ expect(f.fallbacks).toHaveLength(0) // no Card C minted (zero teardowns)
706
+ // POSITIVE: still the same single original card, updated in place across
707
+ // BOTH gaps — no second/third card was ever minted (no stacking).
708
+ expect(pinnedCardId).toBe('card-1')
709
+ expect(mintedCards).toEqual(['card-1'])
710
+ })
711
+
712
+ it('still bounded: a genuinely wedged bash-turn unwedges ONCE at the hard ceiling (not 3×)', () => {
713
+ const f = setupDeps(PROD)
714
+ startTurn('c:0', 0)
715
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
716
+ noteToolEnd('c:0', 't1', 125_000)
717
+ __tickForTests(306_000)
718
+ expect(f.fallbacks).toHaveLength(0) // deferred, not fired 3×
719
+ __tickForTests(900_000) // crosses SILENCE_FALLBACK_HARD_MS
720
+ expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
721
+ })
722
+
723
+ it('the defer is Bash-specific: a non-Bash idle turn still fires at 300s (fix is scoped)', () => {
724
+ const f = setupDeps(PROD)
725
+ startTurn('c:0', 0)
726
+ noteToolStart('c:0', 't1', 'Grep', 'foo', 5_000)
727
+ noteToolEnd('c:0', 't1', 60_000) // Grep can't be a background process
728
+ __tickForTests(306_000)
729
+ expect(f.fallbacks).toHaveLength(1) // genuine silence → unwedges normally
730
+ })
731
+
732
+ it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch', () => {
733
+ const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
734
+ process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
735
+ try {
736
+ const f = setupDeps(PROD)
737
+ startTurn('c:0', 0)
738
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
739
+ noteToolEnd('c:0', 't1', 125_000)
740
+ __tickForTests(306_000)
741
+ expect(f.fallbacks).toHaveLength(1) // defer force-disabled → legacy fire
742
+ } finally {
743
+ if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
744
+ else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
745
+ }
746
+ })
747
+ })
748
+
749
+ // ─── #3519 SHARPEN: PROVEN-alive defer, driven by REAL captured markers ───────
750
+ // The coarse `sawBashThisTurn` guard above deferred EVERY bash-turn to the 900s
751
+ // ceiling. This sharpens it: defer only while a background shell is PROVEN
752
+ // alive (its structured `backgroundTaskId` launch marker seen, no completion
753
+ // yet), and recover at ~300s the moment it finishes. The alive/dead facts here
754
+ // are PARSED FROM REAL JSONL (tests/fixtures/bg-shell-liveness-3519.jsonl —
755
+ // carrie session a6d2d33a-…, claude v2.1.197; line 109 auto-background launch,
756
+ // line 175 `<task-notification>` completion — verbatim but for the operator username
757
+ // scrubbed in both encodings per repo PII policy) through the SAME projectTranscriptLine
758
+ // path the gateway uses, then fed to silence-poke exactly as the gateway feeds
759
+ // it (tool_result.backgroundTaskId → noteBackgroundShellAlive; task_notification
760
+ // → noteBackgroundShellDead). So every fixture value is traceable to a real byte.
761
+ describe('silence-poke — #3519 sharpen: proven-alive defer (real markers)', () => {
762
+ const PROD = {
763
+ thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
764
+ isLegitimatelyWorking: () => false, // background bash invisible to it
765
+ }
766
+ const FIXTURE = join(__dirname, 'fixtures', 'bg-shell-liveness-3519.jsonl')
767
+ const fxLines = readFileSync(FIXTURE, 'utf8').split('\n').filter(l => l.length > 0)
768
+ // Derive the REAL ids straight from the fixtures via the production parser.
769
+ const aliveEv = projectTranscriptLine(fxLines[0]).find(e => e.kind === 'tool_result')
770
+ const deadEv = projectTranscriptLine(fxLines[1])[0]
771
+ const LIVE_ID = aliveEv?.kind === 'tool_result' ? aliveEv.backgroundTaskId! : ''
772
+ const DEAD_ID = deadEv.kind === 'task_notification' ? deadEv.taskId : ''
773
+
774
+ it('sanity: the fixtures really carry a matching background-shell id', () => {
775
+ expect(LIVE_ID).toBe('bxa4sv3dq')
776
+ expect(DEAD_ID).toBe('bxa4sv3dq')
777
+ })
778
+
779
+ // (a) A shell PROVEN alive across a > 300s gap must NOT tear down the card —
780
+ // and exactly ONE card survives (positive assertion, not inferred).
781
+ it('(a) live shell across a > 300s gap → no teardown, exactly one card', () => {
782
+ const f = setupDeps(PROD)
783
+ const mintedCards: string[] = []
784
+ let pinnedCardId: string | null = null
785
+ const mintCard = (): void => {
786
+ mintedCards.push(`card-${mintedCards.length + 1}`)
787
+ pinnedCardId = mintedCards[mintedCards.length - 1]
788
+ }
789
+ const syncCardsAfterTick = (): void => {
790
+ while (mintedCards.length - 1 < f.fallbacks.length) { pinnedCardId = null; mintCard() }
791
+ }
792
+
793
+ startTurn('c:0', 0)
794
+ mintCard() // the turn's first pinned progress card
795
+ // Foreground Bash auto-moved to background at ~120s: its tool_result
796
+ // returns (drains inFlightTools) carrying the real backgroundTaskId.
797
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
798
+ noteToolEnd('c:0', 't1', 125_000)
799
+ noteBackgroundShellAlive('c:0', LIVE_ID) // gateway does this from the parsed event
800
+ __tickForTests(306_000)
801
+ syncCardsAfterTick()
802
+ __tickForTests(500_000) // still alive, still silent, still well past 300s
803
+ syncCardsAfterTick()
804
+ expect(f.fallbacks).toHaveLength(0) // deferred — no teardown
805
+ expect(pinnedCardId).toBe('card-1') // POSITIVE: original card still live
806
+ expect(mintedCards).toEqual(['card-1']) // and it is the ONLY card ever minted
807
+ expect(__bgMarkerParserConfirmedForTests()).toBe(true) // marker proven working
808
+ })
809
+
810
+ // (b) THE NEW capability: once the shell FINISHES (real completion marker),
811
+ // a subsequent > 300s wedge recovers at ~300s — NOT held to 900s.
812
+ it('(b) shell went DEAD then a > 300s wedge → fallback fires at ~300s (fast recovery)', () => {
813
+ const f = setupDeps(PROD)
814
+ startTurn('c:0', 0)
815
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
816
+ noteToolEnd('c:0', 't1', 125_000)
817
+ noteBackgroundShellAlive('c:0', LIVE_ID) // backgrounded at ~120s
818
+ noteBackgroundShellDead('c:0', DEAD_ID) // real <task-notification> completed at ~200s
819
+ // Model then goes silent on a genuine wedge. Because the CLI marker PARSED
820
+ // (confirmed), the coarse sawBash degradation is OFF, so the empty
821
+ // alive-set means this is real silence → recover at the 300s base window,
822
+ // not the 900s ceiling.
823
+ __tickForTests(306_000)
824
+ expect(f.fallbacks).toHaveLength(1) // FAST recovery restored (would be 0 under the old coarse guard)
825
+ })
826
+
827
+ // (c) SAFE DEGRADATION: if a future CLI renames the markers so NOTHING parses
828
+ // (backgroundTaskId never resolves → parser never confirmed), we must not
829
+ // regress to stacking — fall back to the 900s-bounded coarse guard.
830
+ it('(c) unmatched/changed CLI marker → degrades to the 900s-bounded guard', () => {
831
+ const f = setupDeps(PROD)
832
+ // Mutate the REAL alive line so neither the structured field nor the string
833
+ // matches — exactly what a marker rename looks like. Parse it as the gateway
834
+ // would; it yields NO backgroundTaskId, so noteBackgroundShellAlive is never
835
+ // called and the parser stays unconfirmed.
836
+ const mutated = JSON.parse(fxLines[0])
837
+ delete mutated.toolUseResult.backgroundTaskId
838
+ mutated.message.content[0].content = 'Task moved to background (id withheld by a newer CLI).'
839
+ const ev = projectTranscriptLine(JSON.stringify(mutated)).find(e => e.kind === 'tool_result')
840
+ expect(ev?.kind === 'tool_result' ? ev.backgroundTaskId : 'X').toBeUndefined()
841
+
842
+ startTurn('c:0', 0)
843
+ noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000) // sawBashThisTurn armed
844
+ noteToolEnd('c:0', 't1', 125_000)
845
+ // No alive registration (marker unparseable) → parser NOT confirmed.
846
+ expect(__bgMarkerParserConfirmedForTests()).toBe(false)
847
+ __tickForTests(306_000)
848
+ expect(f.fallbacks).toHaveLength(0) // still deferred by the coarse guard (no stacking)
849
+ __tickForTests(900_000) // crosses the hard ceiling
850
+ expect(f.fallbacks).toHaveLength(1) // bounded unwedge — never hangs forever
851
+ })
852
+
853
+ // (d) Still bounded even while genuinely alive: a shell that never reports
854
+ // dead still unwedges ONCE at the 900s ceiling (not indefinitely).
855
+ it('(d) a never-completing live shell still unwedges once at the 900s ceiling', () => {
856
+ const f = setupDeps(PROD)
857
+ startTurn('c:0', 0)
858
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
859
+ noteToolEnd('c:0', 't1', 125_000)
860
+ noteBackgroundShellAlive('c:0', LIVE_ID) // alive, never dies
861
+ __tickForTests(306_000)
862
+ expect(f.fallbacks).toHaveLength(0) // deferred while alive
863
+ __tickForTests(900_000) // hard ceiling
864
+ expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
865
+ })
866
+
867
+ it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch even with a live shell', () => {
868
+ const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
869
+ process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
870
+ try {
871
+ const f = setupDeps(PROD)
872
+ startTurn('c:0', 0)
873
+ noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
874
+ noteToolEnd('c:0', 't1', 125_000)
875
+ noteBackgroundShellAlive('c:0', LIVE_ID)
876
+ __tickForTests(306_000)
877
+ expect(f.fallbacks).toHaveLength(1) // force-disabled → no defer at all
878
+ } finally {
879
+ if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
880
+ else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
881
+ }
882
+ })
883
+ })
@@ -404,10 +404,13 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
404
404
  const NARRATION_B = "Now let me query the second source; it is slower than expected, hang tight…" // opener + …
405
405
  const NARRATION_C = "I'll cross-reference the last set of figures against the ledger before I summarise…" // opener + …
406
406
 
407
- it('zero-delivery turn of only NARRATION blocks → block WITHOUT pendingText (#3228 Finding 2 / flush parity)', () => {
408
- // Combined length of the three blocks is ≥ 200, so the OLD joined-floor code
409
- // would join them and set pendingText (masquerade). The NEW code mirrors the
410
- // flush's narration strip: every block is narration → nothing to deliver.
407
+ it('narration interleaved with work-tools, ending on a terminal narration line → only the terminal block, never the join (#3513 structural coalescer)', () => {
408
+ // #3513 follow-up — A and B are each followed by a turn-continuing tool_use
409
+ // (Bash / Read) → structurally intra-turn → UNCONDITIONALLY suppressed.
410
+ // C is the terminal block (no tool after it), so the deterministic coalescer
411
+ // delivers C ALONE. The #3228 Finding-2 join-masquerade (concatenating short
412
+ // narration to cross the 200-char floor) is STILL prevented: the 200-floor
413
+ // `pendingText` stays undefined, and the delivered text is never the join.
411
414
  expect((NARRATION_A + '\n\n' + NARRATION_B + '\n\n' + NARRATION_C).length)
412
415
  .toBeGreaterThanOrEqual(200)
413
416
  const text = jsonl(
@@ -421,8 +424,12 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
421
424
  const r = scanTurnForFinalReply(text)
422
425
  expect(r.decided).toBe('block')
423
426
  expect(r.reason).toBe('no-final-reply')
424
- // The masquerade must NOT happen — no answer to deliver, take the re-prompt.
425
- expect(r.pendingText).toBeUndefined()
427
+ // Never the join of the suppressed A/B blocks (masquerade guard preserved).
428
+ expect(r.pendingText).not.toContain(NARRATION_A)
429
+ expect(r.pendingText).not.toContain(NARRATION_B)
430
+ // The terminal block is the coalescer's selection (delivered via the
431
+ // zero-reply lowered floor); the 200-char masquerade path yields nothing.
432
+ expect(r.pendingText).toBe(NARRATION_C)
426
433
  })
427
434
 
428
435
  it('zero-delivery turn of MULTIPLE sub-200 REAL-CONTENT blocks → JOINED pendingText (review item 3 drop fix)', () => {
@@ -445,15 +452,34 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
445
452
  expect(r.hasTrailingProse).toBe(true)
446
453
  })
447
454
 
448
- it('leading narration + a real sub-200 answer block → only the answer is delivered (narration stripped)', () => {
455
+ it('leading narration FOLLOWED BY A TOOL + a real sub-200 answer block → only the answer is delivered (structural strip)', () => {
456
+ // #3513 follow-up — the narration is stripped by the STRUCTURAL signal (a
457
+ // work-tool followed it), not by wording. The terminal answer block (no tool
458
+ // after it) is the coalescer's selection.
449
459
  const NARR = 'Let me pull the numbers first…' // narration opener + …
450
460
  const ANSWER = 'Revenue was up 12% quarter-over-quarter, driven mostly by the new enterprise tier.' // ~85, real
451
- const text = jsonl(ENQUEUE, assistantText(NARR), assistantText(ANSWER))
461
+ const text = jsonl(
462
+ ENQUEUE,
463
+ assistantText(NARR),
464
+ assistantToolUse('Bash', { command: 'psql -c "select ..."' }),
465
+ assistantText(ANSWER),
466
+ )
452
467
  const r = scanTurnForFinalReply(text)
453
468
  expect(r.decided).toBe('block')
454
469
  expect(r.pendingText).toBe(ANSWER)
455
470
  })
456
471
 
472
+ it('leading narration with NO tool + a real answer → both delivered joined (fail-open; no wording strip on the backstop path, #3513)', () => {
473
+ // Without a structural continuation signal the coalescer delivers the full
474
+ // terminal run — the wording-based strip is gone (it was provably incomplete).
475
+ const NARR = 'Let me pull the numbers first…'
476
+ const ANSWER = 'Revenue was up 12% quarter-over-quarter, driven mostly by the new enterprise tier.'
477
+ const text = jsonl(ENQUEUE, assistantText(NARR), assistantText(ANSWER))
478
+ const r = scanTurnForFinalReply(text)
479
+ expect(r.decided).toBe('block')
480
+ expect(r.pendingText).toBe(`${NARR}\n\n${ANSWER}`)
481
+ })
482
+
457
483
  it('trailing narration after a delivered reply, all sub-floor → block only if a real ≥floor block exists; here → allow, no pendingText', () => {
458
484
  // Each trailing block is sub-floor, so `sawUndeliveredTextAfterAllow` is
459
485
  // false → allow. (The old joined-floor logic never affected the block
@@ -469,18 +495,21 @@ describe('scanTurnForFinalReply — pendingText is a single substantive block, n
469
495
  expect(r.pendingText).toBeUndefined()
470
496
  })
471
497
 
472
- it('a genuine ≥floor answer followed by a SHORT closer → pendingText is the answer, not the closer', () => {
473
- // "big answer then short closer" — the LAST substantive (≥floor) block is
474
- // the answer; the trailing short closer must not displace it.
498
+ it('a genuine ≥floor answer followed by a SHORT closer → both delivered as the terminal run (answer never dropped, #3513)', () => {
499
+ // #3513 follow-up — "big answer then short closer", NO tool after either
500
+ // block, so both are the terminal run and are delivered JOINED. The #3228
501
+ // concern (a short closer DISPLACING the answer) cannot happen: the coalescer
502
+ // never drops the answer — it delivers the whole terminal run.
475
503
  const answer = 'Here is the real answer you were waiting for: ' + 'A'.repeat(300)
504
+ const closer = 'Let me know if you need anything else.'
476
505
  const text = jsonl(
477
506
  ENQUEUE,
478
507
  assistantText(answer),
479
- assistantText('Let me know if you need anything else.'),
508
+ assistantText(closer),
480
509
  )
481
510
  const r = scanTurnForFinalReply(text)
482
511
  expect(r.decided).toBe('block')
483
- expect(r.pendingText).toBe(answer)
512
+ expect(r.pendingText).toBe(`${answer}\n\n${closer}`)
484
513
  })
485
514
 
486
515
  it('substance-floor boundary: a single trailing block at 199/200/201 chars', () => {
@@ -787,11 +787,17 @@ describe('silent-end-interrupt-stop hook — integration (#1775: transcript-scan
787
787
  expect(lowered.text).toBe(joined)
788
788
  })
789
789
 
790
- it('review item 3: NARRATION-only multi-block → still NO pendingText (no masquerade, #3228 Finding 2)', () => {
790
+ it('review item 3: intra-turn narration blocks each FOLLOWED BY A TOOL → NO pendingText (structural suppression, #3513)', () => {
791
+ // #3513 follow-up — each narration block is followed by a turn-continuing
792
+ // tool_use, so the deterministic coalescer suppresses BOTH unconditionally.
793
+ // The terminal run is empty and the last (tool-followed) block is sub-200 →
794
+ // the empty-terminal corner returns null → no pendingText, no masquerade.
791
795
  const transcript = writeTranscript([
792
796
  ENQUEUE,
793
797
  { type: 'assistant', message: { content: [{ type: 'text', text: 'Let me check the first source now, scanning the rows…' }] } },
798
+ { type: 'assistant', message: { content: [{ type: 'tool_use', id: 't1', name: 'Bash', input: { command: 'ls' } }] } },
794
799
  { type: 'assistant', message: { content: [{ type: 'text', text: "Now let me query the second source; hang tight…" }] } },
800
+ { type: 'assistant', message: { content: [{ type: 'tool_use', id: 't2', name: 'Read', input: { file_path: '/tmp/x' } }] } },
795
801
  ])
796
802
  const r = runHook({ session_id: 's', transcript_path: transcript, hook_event_name: 'Stop' })
797
803
  expect(JSON.parse(r.stdout.trim()).decision).toBe('block')
@@ -4,6 +4,7 @@
4
4
  */
5
5
  import { describe, test, expect, beforeEach, afterEach } from 'bun:test'
6
6
  import { normalizeForTts, ttsNormalizeEnabled } from '../tts-normalize.js'
7
+ import { normalizeForSpeech } from '../voice-normalize-text.js'
7
8
 
8
9
  const KILL = 'SWITCHROOM_DISABLE_TTS_NORMALIZE'
9
10
 
@@ -240,3 +241,68 @@ describe('conservatism', () => {
240
241
  expect(normalizeForTts(input)).toBe(normalizeForTts(input))
241
242
  })
242
243
  })
244
+
245
+ describe('normalizeForTts — backslash escapes & HTML entities (last-line defence)', () => {
246
+ test('strips a literal \\b so the engine never speaks "backslash b"', () => {
247
+ const out = normalizeForTts('the regex \\b boundary')
248
+ expect(out).toBe('the regex b boundary')
249
+ expect(out).not.toContain('\\')
250
+ })
251
+
252
+ test('unescapes MarkdownV2 punctuation escapes (\\. \\! \\-)', () => {
253
+ expect(normalizeForTts('done\\. next\\! wait\\-')).toBe('done. next! wait-')
254
+ })
255
+
256
+ test('decodes HTML entities (&amp; &lt; &#39;)', () => {
257
+ expect(normalizeForTts('Tom &amp; Jerry')).toBe('Tom and Jerry')
258
+ expect(normalizeForTts('5 &lt; 10')).toBe('5 < 10')
259
+ expect(normalizeForTts("it&#39;s here")).toBe("it's here")
260
+ })
261
+
262
+ test('rich mixed reply → clean spoken text (no backslash/backtick/entity)', () => {
263
+ const reply =
264
+ '**Bold** and `code\\b` and a [label](https://example.com/x) ' +
265
+ 'with Tom &amp; Jerry and a regex \\b\\.'
266
+ const out = normalizeForTts(reply)
267
+ expect(out).not.toContain('\\')
268
+ expect(out).not.toContain('`')
269
+ expect(out).not.toContain('&amp;')
270
+ expect(out).toContain('label')
271
+ expect(out).toContain('Tom and Jerry')
272
+ })
273
+
274
+ test('kill switch still returns byte-identical input (escapes preserved)', () => {
275
+ process.env[KILL] = '1'
276
+ expect(normalizeForTts('a \\b &amp; b')).toBe('a \\b &amp; b')
277
+ delete process.env[KILL]
278
+ })
279
+
280
+ test('idempotent after normalizeForSpeech already unescaped', () => {
281
+ const reply = 'a \\b and Tom &amp; Jerry \\. end'
282
+ const once = normalizeForTts(reply)
283
+ expect(normalizeForTts(once)).toBe(once)
284
+ })
285
+ })
286
+
287
+ describe('normalizeForTts — review findings (fixpoint decode, metachar, nits)', () => {
288
+ test('L2 parity: immediate (speech+tts) and single-tts agree on a double-encoded entity', () => {
289
+ const x = '&amp;amp;lt;'
290
+ const immediate = normalizeForTts(normalizeForSpeech(x))
291
+ const singleTts = normalizeForTts(x)
292
+ expect(immediate).toBe('<')
293
+ expect(singleTts).toBe('<')
294
+ expect(immediate).toBe(singleTts)
295
+ expect(normalizeForTts('&amp;amp;amp;')).toBe('&')
296
+ })
297
+
298
+ test('L1: entity → line-leading metachar keeps a spoken form (hash/asterisk)', () => {
299
+ expect(normalizeForTts('&#35; Heading')).toBe('hash Heading')
300
+ expect(normalizeForTts('2 &#42; 3')).toBe('2 asterisk 3')
301
+ })
302
+
303
+ test('nit: dangling trailing backslash dropped; &#92; decodes then strips', () => {
304
+ expect(normalizeForTts('ends here\\')).toBe('ends here')
305
+ expect(normalizeForTts('X&#92;Y')).toBe('XY')
306
+ expect(normalizeForTts('X&#92;Y')).not.toContain('\\')
307
+ })
308
+ })