switchroom 0.19.15 → 0.19.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/switchroom.js +1 -1
- package/dist/host-control/main.js +1 -1
- package/package.json +1 -1
- package/telegram-plugin/dist/bridge/bridge.js +30 -1
- package/telegram-plugin/dist/gateway/gateway.js +693 -433
- package/telegram-plugin/dist/server.js +30 -1
- package/telegram-plugin/gateway/background-shell-liveness.ts +65 -0
- package/telegram-plugin/gateway/gateway.ts +7 -58
- package/telegram-plugin/gateway/outbound-send-path.ts +25 -23
- package/telegram-plugin/gateway/outbox-listen-markup.ts +67 -0
- package/telegram-plugin/gateway/outbox-sweep.ts +92 -18
- package/telegram-plugin/gateway/rich-message-handler.ts +10 -4
- package/telegram-plugin/gateway/silence-poke-session-event.ts +89 -0
- package/telegram-plugin/session-tail.ts +88 -1
- package/telegram-plugin/silence-poke.ts +118 -1
- package/telegram-plugin/tests/background-shell-liveness.test.ts +72 -0
- package/telegram-plugin/tests/feed-survival.test.ts +7 -1
- package/telegram-plugin/tests/fixtures/bg-shell-liveness-3519.jsonl +3 -0
- package/telegram-plugin/tests/forwarded-rich-message-coalesce.test.ts +290 -0
- package/telegram-plugin/tests/outbox-sweep-listen-button.test.ts +253 -0
- package/telegram-plugin/tests/session-tail.test.ts +91 -1
- package/telegram-plugin/tests/silence-poke.test.ts +280 -0
- package/telegram-plugin/tests/tts-normalize.test.ts +66 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +82 -1
- package/telegram-plugin/tts-normalize.ts +12 -0
- package/telegram-plugin/voice-normalize-text.ts +100 -0
- package/telegram-plugin/voice-ondemand.ts +71 -0
|
@@ -1,4 +1,7 @@
|
|
|
1
1
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
2
|
+
import { readFileSync } from 'fs'
|
|
3
|
+
import { join } from 'path'
|
|
4
|
+
import { projectTranscriptLine } from '../session-tail.js'
|
|
2
5
|
import {
|
|
3
6
|
startTurn,
|
|
4
7
|
noteOutbound,
|
|
@@ -7,6 +10,9 @@ import {
|
|
|
7
10
|
noteToolStart,
|
|
8
11
|
noteToolEnd,
|
|
9
12
|
noteToolLabel,
|
|
13
|
+
noteBackgroundShellAlive,
|
|
14
|
+
noteBackgroundShellDead,
|
|
15
|
+
__bgMarkerParserConfirmedForTests,
|
|
10
16
|
endTurn,
|
|
11
17
|
silenceMsForKey,
|
|
12
18
|
silencePokeEnabled,
|
|
@@ -30,6 +36,7 @@ interface TestFixtures {
|
|
|
30
36
|
function setupDeps(opts?: {
|
|
31
37
|
thresholds?: Partial<typeof DEFAULT_THRESHOLDS> & { fallbackHardCeiling?: number }
|
|
32
38
|
deferFallbackWhileToolInFlight?: boolean
|
|
39
|
+
isLegitimatelyWorking?: (key: string) => boolean
|
|
33
40
|
}): TestFixtures {
|
|
34
41
|
const fixtures: TestFixtures = { emitted: [], fallbacks: [] }
|
|
35
42
|
__setDepsForTests({
|
|
@@ -42,6 +49,9 @@ function setupDeps(opts?: {
|
|
|
42
49
|
...(opts?.deferFallbackWhileToolInFlight != null
|
|
43
50
|
? { deferFallbackWhileToolInFlight: opts.deferFallbackWhileToolInFlight }
|
|
44
51
|
: {}),
|
|
52
|
+
...(opts?.isLegitimatelyWorking != null
|
|
53
|
+
? { isLegitimatelyWorking: opts.isLegitimatelyWorking }
|
|
54
|
+
: {}),
|
|
45
55
|
})
|
|
46
56
|
return fixtures
|
|
47
57
|
}
|
|
@@ -601,3 +611,273 @@ describe('silence-poke — Fix A: in-flight-tool defer', () => {
|
|
|
601
611
|
expect(f.fallbacks).toHaveLength(0)
|
|
602
612
|
})
|
|
603
613
|
})
|
|
614
|
+
|
|
615
|
+
// ─── #3519: CLI-side background-bash defer — the stacked-cards regression ─────
|
|
616
|
+
// Root cause of the operator-visible bug: a single turn ran a foreground `Bash`
|
|
617
|
+
// that the claude CLI auto-moved to the background at its foreground timeout.
|
|
618
|
+
// The tool_result returned (draining `inFlightTools`), so every existing "still
|
|
619
|
+
// working" signal — `isLegitimatelyWorking()`'s foreground/async-dispatch/
|
|
620
|
+
// ask_user checks — went false while the process kept running and the model sat
|
|
621
|
+
// silent waiting on it. At 300s the framework fallback fired: it nulled
|
|
622
|
+
// `currentTurn` and tore down the pinned progress card (the `onFrameworkFallback`
|
|
623
|
+
// callback → liveness-wiring.ts:427-441 `endCurrentTurnForKey`), and the next
|
|
624
|
+
// tool burst minted a BRAND-NEW pinned card. Each > 300s gap repeated it →
|
|
625
|
+
// 2..N stacked cards on ONE turn.
|
|
626
|
+
//
|
|
627
|
+
// These are OUTCOME assertions: `fixtures.fallbacks` counts every fallback fire,
|
|
628
|
+
// and a fire is exactly a currentTurn-null + card teardown (the trigger for a
|
|
629
|
+
// re-minted card). The production wiring is reproduced faithfully:
|
|
630
|
+
// `isLegitimatelyWorking` IS wired (the gateway always wires it) and it returns
|
|
631
|
+
// FALSE for the whole gap (the CLI-side background bash is invisible to it by
|
|
632
|
+
// construction). Reverting the `sawBashThisTurn` defer turns these RED — the
|
|
633
|
+
// fallback fires on every gap and multiple cards are minted.
|
|
634
|
+
describe('silence-poke — #3519 background-bash defer (stacked-cards regression guard)', () => {
|
|
635
|
+
const PROD = {
|
|
636
|
+
thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
|
|
637
|
+
isLegitimatelyWorking: () => false, // background bash is invisible
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
it('a foreground Bash moved to background does NOT tear down the card at 300s (mid-work)', () => {
|
|
641
|
+
const f = setupDeps(PROD)
|
|
642
|
+
startTurn('c:0', 0)
|
|
643
|
+
// Foreground bash starts, then auto-moves to background at ~120s: its
|
|
644
|
+
// tool_result returns so inFlightTools drains. isLegitimatelyWorking is
|
|
645
|
+
// false throughout — every existing signal says "idle".
|
|
646
|
+
noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
|
|
647
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
648
|
+
// The silent gap the model spends waiting on the background process.
|
|
649
|
+
__tickForTests(300_000)
|
|
650
|
+
__tickForTests(306_000) // well past the 300s base threshold
|
|
651
|
+
// BUG behaviour: fallback fires here (currentTurn nulled, card torn down).
|
|
652
|
+
// FIXED: deferred — the pinned card stays live, no re-mint.
|
|
653
|
+
expect(f.fallbacks).toHaveLength(0)
|
|
654
|
+
})
|
|
655
|
+
|
|
656
|
+
it('exactly ONE card survives a whole turn with two > 300s background-bash gaps', () => {
|
|
657
|
+
const f = setupDeps(PROD)
|
|
658
|
+
|
|
659
|
+
// Model the gateway's pinned-card lifecycle so the surviving card is
|
|
660
|
+
// asserted POSITIVELY, not merely inferred from zero teardowns. In
|
|
661
|
+
// production a fallback fire IS a card teardown (currentTurn nulled +
|
|
662
|
+
// pinned card unpinned, liveness-wiring.ts:427-441), and the next tool
|
|
663
|
+
// burst mints a BRAND-NEW pinned card — that re-mint is exactly the
|
|
664
|
+
// stacking. Here: mint one card when the turn starts, and replay a
|
|
665
|
+
// teardown + fresh re-mint for every fallback the silence-poke tick
|
|
666
|
+
// records. `mintedCards.length` is then the true number of cards that
|
|
667
|
+
// ever existed for the turn, and `pinnedCardId` is the live one.
|
|
668
|
+
const mintedCards: string[] = []
|
|
669
|
+
let pinnedCardId: string | null = null
|
|
670
|
+
const mintCard = (): void => {
|
|
671
|
+
const id = `card-${mintedCards.length + 1}`
|
|
672
|
+
mintedCards.push(id)
|
|
673
|
+
pinnedCardId = id
|
|
674
|
+
}
|
|
675
|
+
const syncCardsAfterTick = (): void => {
|
|
676
|
+
// Each recorded fallback == one teardown of the live card + a fresh
|
|
677
|
+
// re-mint by the resuming burst (the stacked-cards mechanism). Replay
|
|
678
|
+
// any teardown not yet modelled.
|
|
679
|
+
while (mintedCards.length - 1 < f.fallbacks.length) {
|
|
680
|
+
pinnedCardId = null // teardown unpins the live card
|
|
681
|
+
mintCard() // next tool burst mints a brand-new pinned card (a stack)
|
|
682
|
+
}
|
|
683
|
+
}
|
|
684
|
+
|
|
685
|
+
startTurn('c:0', 0)
|
|
686
|
+
mintCard() // the turn's first pinned progress card ("card-1")
|
|
687
|
+
// Gap 1 — first backgrounded find.
|
|
688
|
+
noteToolStart('c:0', 't1', 'Bash', 'find / -name a', 5_000)
|
|
689
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
690
|
+
__tickForTests(306_000)
|
|
691
|
+
syncCardsAfterTick()
|
|
692
|
+
expect(f.fallbacks).toHaveLength(0) // no Card B minted (zero teardowns)
|
|
693
|
+
// POSITIVE: the ORIGINAL pinned card is still the live one after gap 1,
|
|
694
|
+
// and it is the ONLY card that has ever existed.
|
|
695
|
+
expect(pinnedCardId).toBe('card-1')
|
|
696
|
+
expect(mintedCards).toHaveLength(1)
|
|
697
|
+
// Model resumes and updates the SAME pinned card (production resets the
|
|
698
|
+
// silence clock on that render), then launches a second backgrounded find.
|
|
699
|
+
noteProduction('c:0', 310_000)
|
|
700
|
+
noteToolStart('c:0', 't2', 'Bash', 'find / -name b', 315_000)
|
|
701
|
+
noteToolEnd('c:0', 't2', 435_000)
|
|
702
|
+
// Gap 2 — > 300s of silence since the last render (310_000).
|
|
703
|
+
__tickForTests(620_000)
|
|
704
|
+
syncCardsAfterTick()
|
|
705
|
+
expect(f.fallbacks).toHaveLength(0) // no Card C minted (zero teardowns)
|
|
706
|
+
// POSITIVE: still the same single original card, updated in place across
|
|
707
|
+
// BOTH gaps — no second/third card was ever minted (no stacking).
|
|
708
|
+
expect(pinnedCardId).toBe('card-1')
|
|
709
|
+
expect(mintedCards).toEqual(['card-1'])
|
|
710
|
+
})
|
|
711
|
+
|
|
712
|
+
it('still bounded: a genuinely wedged bash-turn unwedges ONCE at the hard ceiling (not 3×)', () => {
|
|
713
|
+
const f = setupDeps(PROD)
|
|
714
|
+
startTurn('c:0', 0)
|
|
715
|
+
noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
|
|
716
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
717
|
+
__tickForTests(306_000)
|
|
718
|
+
expect(f.fallbacks).toHaveLength(0) // deferred, not fired 3×
|
|
719
|
+
__tickForTests(900_000) // crosses SILENCE_FALLBACK_HARD_MS
|
|
720
|
+
expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
|
|
721
|
+
})
|
|
722
|
+
|
|
723
|
+
it('the defer is Bash-specific: a non-Bash idle turn still fires at 300s (fix is scoped)', () => {
|
|
724
|
+
const f = setupDeps(PROD)
|
|
725
|
+
startTurn('c:0', 0)
|
|
726
|
+
noteToolStart('c:0', 't1', 'Grep', 'foo', 5_000)
|
|
727
|
+
noteToolEnd('c:0', 't1', 60_000) // Grep can't be a background process
|
|
728
|
+
__tickForTests(306_000)
|
|
729
|
+
expect(f.fallbacks).toHaveLength(1) // genuine silence → unwedges normally
|
|
730
|
+
})
|
|
731
|
+
|
|
732
|
+
it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch', () => {
|
|
733
|
+
const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
|
|
734
|
+
process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
|
|
735
|
+
try {
|
|
736
|
+
const f = setupDeps(PROD)
|
|
737
|
+
startTurn('c:0', 0)
|
|
738
|
+
noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
|
|
739
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
740
|
+
__tickForTests(306_000)
|
|
741
|
+
expect(f.fallbacks).toHaveLength(1) // defer force-disabled → legacy fire
|
|
742
|
+
} finally {
|
|
743
|
+
if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
|
|
744
|
+
else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
|
|
745
|
+
}
|
|
746
|
+
})
|
|
747
|
+
})
|
|
748
|
+
|
|
749
|
+
// ─── #3519 SHARPEN: PROVEN-alive defer, driven by REAL captured markers ───────
|
|
750
|
+
// The coarse `sawBashThisTurn` guard above deferred EVERY bash-turn to the 900s
|
|
751
|
+
// ceiling. This sharpens it: defer only while a background shell is PROVEN
|
|
752
|
+
// alive (its structured `backgroundTaskId` launch marker seen, no completion
|
|
753
|
+
// yet), and recover at ~300s the moment it finishes. The alive/dead facts here
|
|
754
|
+
// are PARSED FROM REAL JSONL (tests/fixtures/bg-shell-liveness-3519.jsonl —
|
|
755
|
+
// carrie session a6d2d33a-…, claude v2.1.197; line 109 auto-background launch,
|
|
756
|
+
// line 175 `<task-notification>` completion — verbatim but for the operator username
|
|
757
|
+
// scrubbed in both encodings per repo PII policy) through the SAME projectTranscriptLine
|
|
758
|
+
// path the gateway uses, then fed to silence-poke exactly as the gateway feeds
|
|
759
|
+
// it (tool_result.backgroundTaskId → noteBackgroundShellAlive; task_notification
|
|
760
|
+
// → noteBackgroundShellDead). So every fixture value is traceable to a real byte.
|
|
761
|
+
describe('silence-poke — #3519 sharpen: proven-alive defer (real markers)', () => {
|
|
762
|
+
const PROD = {
|
|
763
|
+
thresholds: { fallbackHardCeiling: 900_000 }, // SILENCE_FALLBACK_HARD_MS
|
|
764
|
+
isLegitimatelyWorking: () => false, // background bash invisible to it
|
|
765
|
+
}
|
|
766
|
+
const FIXTURE = join(__dirname, 'fixtures', 'bg-shell-liveness-3519.jsonl')
|
|
767
|
+
const fxLines = readFileSync(FIXTURE, 'utf8').split('\n').filter(l => l.length > 0)
|
|
768
|
+
// Derive the REAL ids straight from the fixtures via the production parser.
|
|
769
|
+
const aliveEv = projectTranscriptLine(fxLines[0]).find(e => e.kind === 'tool_result')
|
|
770
|
+
const deadEv = projectTranscriptLine(fxLines[1])[0]
|
|
771
|
+
const LIVE_ID = aliveEv?.kind === 'tool_result' ? aliveEv.backgroundTaskId! : ''
|
|
772
|
+
const DEAD_ID = deadEv.kind === 'task_notification' ? deadEv.taskId : ''
|
|
773
|
+
|
|
774
|
+
it('sanity: the fixtures really carry a matching background-shell id', () => {
|
|
775
|
+
expect(LIVE_ID).toBe('bxa4sv3dq')
|
|
776
|
+
expect(DEAD_ID).toBe('bxa4sv3dq')
|
|
777
|
+
})
|
|
778
|
+
|
|
779
|
+
// (a) A shell PROVEN alive across a > 300s gap must NOT tear down the card —
|
|
780
|
+
// and exactly ONE card survives (positive assertion, not inferred).
|
|
781
|
+
it('(a) live shell across a > 300s gap → no teardown, exactly one card', () => {
|
|
782
|
+
const f = setupDeps(PROD)
|
|
783
|
+
const mintedCards: string[] = []
|
|
784
|
+
let pinnedCardId: string | null = null
|
|
785
|
+
const mintCard = (): void => {
|
|
786
|
+
mintedCards.push(`card-${mintedCards.length + 1}`)
|
|
787
|
+
pinnedCardId = mintedCards[mintedCards.length - 1]
|
|
788
|
+
}
|
|
789
|
+
const syncCardsAfterTick = (): void => {
|
|
790
|
+
while (mintedCards.length - 1 < f.fallbacks.length) { pinnedCardId = null; mintCard() }
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
startTurn('c:0', 0)
|
|
794
|
+
mintCard() // the turn's first pinned progress card
|
|
795
|
+
// Foreground Bash auto-moved to background at ~120s: its tool_result
|
|
796
|
+
// returns (drains inFlightTools) carrying the real backgroundTaskId.
|
|
797
|
+
noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
|
|
798
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
799
|
+
noteBackgroundShellAlive('c:0', LIVE_ID) // gateway does this from the parsed event
|
|
800
|
+
__tickForTests(306_000)
|
|
801
|
+
syncCardsAfterTick()
|
|
802
|
+
__tickForTests(500_000) // still alive, still silent, still well past 300s
|
|
803
|
+
syncCardsAfterTick()
|
|
804
|
+
expect(f.fallbacks).toHaveLength(0) // deferred — no teardown
|
|
805
|
+
expect(pinnedCardId).toBe('card-1') // POSITIVE: original card still live
|
|
806
|
+
expect(mintedCards).toEqual(['card-1']) // and it is the ONLY card ever minted
|
|
807
|
+
expect(__bgMarkerParserConfirmedForTests()).toBe(true) // marker proven working
|
|
808
|
+
})
|
|
809
|
+
|
|
810
|
+
// (b) THE NEW capability: once the shell FINISHES (real completion marker),
|
|
811
|
+
// a subsequent > 300s wedge recovers at ~300s — NOT held to 900s.
|
|
812
|
+
it('(b) shell went DEAD then a > 300s wedge → fallback fires at ~300s (fast recovery)', () => {
|
|
813
|
+
const f = setupDeps(PROD)
|
|
814
|
+
startTurn('c:0', 0)
|
|
815
|
+
noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000)
|
|
816
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
817
|
+
noteBackgroundShellAlive('c:0', LIVE_ID) // backgrounded at ~120s
|
|
818
|
+
noteBackgroundShellDead('c:0', DEAD_ID) // real <task-notification> completed at ~200s
|
|
819
|
+
// Model then goes silent on a genuine wedge. Because the CLI marker PARSED
|
|
820
|
+
// (confirmed), the coarse sawBash degradation is OFF, so the empty
|
|
821
|
+
// alive-set means this is real silence → recover at the 300s base window,
|
|
822
|
+
// not the 900s ceiling.
|
|
823
|
+
__tickForTests(306_000)
|
|
824
|
+
expect(f.fallbacks).toHaveLength(1) // FAST recovery restored (would be 0 under the old coarse guard)
|
|
825
|
+
})
|
|
826
|
+
|
|
827
|
+
// (c) SAFE DEGRADATION: if a future CLI renames the markers so NOTHING parses
|
|
828
|
+
// (backgroundTaskId never resolves → parser never confirmed), we must not
|
|
829
|
+
// regress to stacking — fall back to the 900s-bounded coarse guard.
|
|
830
|
+
it('(c) unmatched/changed CLI marker → degrades to the 900s-bounded guard', () => {
|
|
831
|
+
const f = setupDeps(PROD)
|
|
832
|
+
// Mutate the REAL alive line so neither the structured field nor the string
|
|
833
|
+
// matches — exactly what a marker rename looks like. Parse it as the gateway
|
|
834
|
+
// would; it yields NO backgroundTaskId, so noteBackgroundShellAlive is never
|
|
835
|
+
// called and the parser stays unconfirmed.
|
|
836
|
+
const mutated = JSON.parse(fxLines[0])
|
|
837
|
+
delete mutated.toolUseResult.backgroundTaskId
|
|
838
|
+
mutated.message.content[0].content = 'Task moved to background (id withheld by a newer CLI).'
|
|
839
|
+
const ev = projectTranscriptLine(JSON.stringify(mutated)).find(e => e.kind === 'tool_result')
|
|
840
|
+
expect(ev?.kind === 'tool_result' ? ev.backgroundTaskId : 'X').toBeUndefined()
|
|
841
|
+
|
|
842
|
+
startTurn('c:0', 0)
|
|
843
|
+
noteToolStart('c:0', 't1', 'Bash', 'find / -name x', 5_000) // sawBashThisTurn armed
|
|
844
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
845
|
+
// No alive registration (marker unparseable) → parser NOT confirmed.
|
|
846
|
+
expect(__bgMarkerParserConfirmedForTests()).toBe(false)
|
|
847
|
+
__tickForTests(306_000)
|
|
848
|
+
expect(f.fallbacks).toHaveLength(0) // still deferred by the coarse guard (no stacking)
|
|
849
|
+
__tickForTests(900_000) // crosses the hard ceiling
|
|
850
|
+
expect(f.fallbacks).toHaveLength(1) // bounded unwedge — never hangs forever
|
|
851
|
+
})
|
|
852
|
+
|
|
853
|
+
// (d) Still bounded even while genuinely alive: a shell that never reports
|
|
854
|
+
// dead still unwedges ONCE at the 900s ceiling (not indefinitely).
|
|
855
|
+
it('(d) a never-completing live shell still unwedges once at the 900s ceiling', () => {
|
|
856
|
+
const f = setupDeps(PROD)
|
|
857
|
+
startTurn('c:0', 0)
|
|
858
|
+
noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
|
|
859
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
860
|
+
noteBackgroundShellAlive('c:0', LIVE_ID) // alive, never dies
|
|
861
|
+
__tickForTests(306_000)
|
|
862
|
+
expect(f.fallbacks).toHaveLength(0) // deferred while alive
|
|
863
|
+
__tickForTests(900_000) // hard ceiling
|
|
864
|
+
expect(f.fallbacks).toHaveLength(1) // exactly one bounded unwedge
|
|
865
|
+
})
|
|
866
|
+
|
|
867
|
+
it('honours the SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS=0 kill switch even with a live shell', () => {
|
|
868
|
+
const prev = process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
|
|
869
|
+
process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = '0'
|
|
870
|
+
try {
|
|
871
|
+
const f = setupDeps(PROD)
|
|
872
|
+
startTurn('c:0', 0)
|
|
873
|
+
noteToolStart('c:0', 't1', 'Bash', 'find /', 5_000)
|
|
874
|
+
noteToolEnd('c:0', 't1', 125_000)
|
|
875
|
+
noteBackgroundShellAlive('c:0', LIVE_ID)
|
|
876
|
+
__tickForTests(306_000)
|
|
877
|
+
expect(f.fallbacks).toHaveLength(1) // force-disabled → no defer at all
|
|
878
|
+
} finally {
|
|
879
|
+
if (prev != null) process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS = prev
|
|
880
|
+
else delete process.env.SWITCHROOM_SILENCE_DEFER_INFLIGHT_TOOLS
|
|
881
|
+
}
|
|
882
|
+
})
|
|
883
|
+
})
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
*/
|
|
5
5
|
import { describe, test, expect, beforeEach, afterEach } from 'bun:test'
|
|
6
6
|
import { normalizeForTts, ttsNormalizeEnabled } from '../tts-normalize.js'
|
|
7
|
+
import { normalizeForSpeech } from '../voice-normalize-text.js'
|
|
7
8
|
|
|
8
9
|
const KILL = 'SWITCHROOM_DISABLE_TTS_NORMALIZE'
|
|
9
10
|
|
|
@@ -240,3 +241,68 @@ describe('conservatism', () => {
|
|
|
240
241
|
expect(normalizeForTts(input)).toBe(normalizeForTts(input))
|
|
241
242
|
})
|
|
242
243
|
})
|
|
244
|
+
|
|
245
|
+
describe('normalizeForTts — backslash escapes & HTML entities (last-line defence)', () => {
|
|
246
|
+
test('strips a literal \\b so the engine never speaks "backslash b"', () => {
|
|
247
|
+
const out = normalizeForTts('the regex \\b boundary')
|
|
248
|
+
expect(out).toBe('the regex b boundary')
|
|
249
|
+
expect(out).not.toContain('\\')
|
|
250
|
+
})
|
|
251
|
+
|
|
252
|
+
test('unescapes MarkdownV2 punctuation escapes (\\. \\! \\-)', () => {
|
|
253
|
+
expect(normalizeForTts('done\\. next\\! wait\\-')).toBe('done. next! wait-')
|
|
254
|
+
})
|
|
255
|
+
|
|
256
|
+
test('decodes HTML entities (& < ')', () => {
|
|
257
|
+
expect(normalizeForTts('Tom & Jerry')).toBe('Tom and Jerry')
|
|
258
|
+
expect(normalizeForTts('5 < 10')).toBe('5 < 10')
|
|
259
|
+
expect(normalizeForTts("it's here")).toBe("it's here")
|
|
260
|
+
})
|
|
261
|
+
|
|
262
|
+
test('rich mixed reply → clean spoken text (no backslash/backtick/entity)', () => {
|
|
263
|
+
const reply =
|
|
264
|
+
'**Bold** and `code\\b` and a [label](https://example.com/x) ' +
|
|
265
|
+
'with Tom & Jerry and a regex \\b\\.'
|
|
266
|
+
const out = normalizeForTts(reply)
|
|
267
|
+
expect(out).not.toContain('\\')
|
|
268
|
+
expect(out).not.toContain('`')
|
|
269
|
+
expect(out).not.toContain('&')
|
|
270
|
+
expect(out).toContain('label')
|
|
271
|
+
expect(out).toContain('Tom and Jerry')
|
|
272
|
+
})
|
|
273
|
+
|
|
274
|
+
test('kill switch still returns byte-identical input (escapes preserved)', () => {
|
|
275
|
+
process.env[KILL] = '1'
|
|
276
|
+
expect(normalizeForTts('a \\b & b')).toBe('a \\b & b')
|
|
277
|
+
delete process.env[KILL]
|
|
278
|
+
})
|
|
279
|
+
|
|
280
|
+
test('idempotent after normalizeForSpeech already unescaped', () => {
|
|
281
|
+
const reply = 'a \\b and Tom & Jerry \\. end'
|
|
282
|
+
const once = normalizeForTts(reply)
|
|
283
|
+
expect(normalizeForTts(once)).toBe(once)
|
|
284
|
+
})
|
|
285
|
+
})
|
|
286
|
+
|
|
287
|
+
describe('normalizeForTts — review findings (fixpoint decode, metachar, nits)', () => {
|
|
288
|
+
test('L2 parity: immediate (speech+tts) and single-tts agree on a double-encoded entity', () => {
|
|
289
|
+
const x = '&amp;lt;'
|
|
290
|
+
const immediate = normalizeForTts(normalizeForSpeech(x))
|
|
291
|
+
const singleTts = normalizeForTts(x)
|
|
292
|
+
expect(immediate).toBe('<')
|
|
293
|
+
expect(singleTts).toBe('<')
|
|
294
|
+
expect(immediate).toBe(singleTts)
|
|
295
|
+
expect(normalizeForTts('&amp;amp;')).toBe('&')
|
|
296
|
+
})
|
|
297
|
+
|
|
298
|
+
test('L1: entity → line-leading metachar keeps a spoken form (hash/asterisk)', () => {
|
|
299
|
+
expect(normalizeForTts('# Heading')).toBe('hash Heading')
|
|
300
|
+
expect(normalizeForTts('2 * 3')).toBe('2 asterisk 3')
|
|
301
|
+
})
|
|
302
|
+
|
|
303
|
+
test('nit: dangling trailing backslash dropped; \ decodes then strips', () => {
|
|
304
|
+
expect(normalizeForTts('ends here\\')).toBe('ends here')
|
|
305
|
+
expect(normalizeForTts('X\Y')).toBe('XY')
|
|
306
|
+
expect(normalizeForTts('X\Y')).not.toContain('\\')
|
|
307
|
+
})
|
|
308
|
+
})
|
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
*/
|
|
7
7
|
|
|
8
8
|
import { describe, it, expect } from 'bun:test'
|
|
9
|
-
import { normalizeForSpeech } from '../voice-normalize-text.js'
|
|
9
|
+
import { normalizeForSpeech, decodeHtmlEntities } from '../voice-normalize-text.js'
|
|
10
10
|
|
|
11
11
|
describe('normalizeForSpeech — reported markdown/symbol cases', () => {
|
|
12
12
|
it('drops stray tildes (never spoken as "tilde")', () => {
|
|
@@ -254,3 +254,84 @@ describe('normalizeForSpeech — conservative, does not mangle real words', () =
|
|
|
254
254
|
expect(normalizeForSpeech(' \n ')).toBe('')
|
|
255
255
|
})
|
|
256
256
|
})
|
|
257
|
+
|
|
258
|
+
describe('normalizeForSpeech — backslash escapes & HTML entities (voice trash)', () => {
|
|
259
|
+
it('strips a literal backslash-b (\\b) so it is never spoken as "backslash b"', () => {
|
|
260
|
+
// The exact operator report: the regex/escape "\b" was read as "slash b".
|
|
261
|
+
const out = normalizeForSpeech('the regex \\b word boundary')
|
|
262
|
+
expect(out).toBe('the regex b word boundary')
|
|
263
|
+
expect(out).not.toContain('\\')
|
|
264
|
+
})
|
|
265
|
+
|
|
266
|
+
it('unescapes MarkdownV2 punctuation escapes without leaving backslashes', () => {
|
|
267
|
+
expect(normalizeForSpeech('done\\. next\\! wait\\-')).toBe('done. next! wait-')
|
|
268
|
+
expect(normalizeForSpeech('a \\* b')).not.toContain('\\')
|
|
269
|
+
})
|
|
270
|
+
|
|
271
|
+
it('an escaped emphasis marker collapses to the inner word, not a backslash', () => {
|
|
272
|
+
expect(normalizeForSpeech('Use \\*literal\\* here')).toBe('Use literal here')
|
|
273
|
+
})
|
|
274
|
+
|
|
275
|
+
it('drops a Windows-path-style backslash run (C:\\build)', () => {
|
|
276
|
+
expect(normalizeForSpeech('path C:\\build\\out done')).toBe('path C:buildout done')
|
|
277
|
+
})
|
|
278
|
+
|
|
279
|
+
it('decodes HTML entities so & / < are not read as "amp" / "lt"', () => {
|
|
280
|
+
expect(normalizeForSpeech('Tom & Jerry')).toBe('Tom and Jerry')
|
|
281
|
+
expect(normalizeForSpeech('5 < 10 > 3')).toBe('5 < 10 > 3')
|
|
282
|
+
expect(normalizeForSpeech('it's here')).toBe("it's here")
|
|
283
|
+
})
|
|
284
|
+
|
|
285
|
+
it('produces clean plain speech for a rich mixed reply (end-to-end)', () => {
|
|
286
|
+
const reply =
|
|
287
|
+
'**Bold** and `code\\b` and a [label](https://example.com/x) ' +
|
|
288
|
+
'with Tom & Jerry and a regex \\b\\.'
|
|
289
|
+
const out = normalizeForSpeech(reply)
|
|
290
|
+
expect(out).not.toContain('\\')
|
|
291
|
+
expect(out).not.toContain('`')
|
|
292
|
+
expect(out).not.toContain('*')
|
|
293
|
+
expect(out).not.toContain('&')
|
|
294
|
+
expect(out).not.toContain('example.com')
|
|
295
|
+
expect(out).toContain('label')
|
|
296
|
+
expect(out).toContain('Tom and Jerry')
|
|
297
|
+
})
|
|
298
|
+
|
|
299
|
+
it('is idempotent: a second pass finds no backslashes/entities to change', () => {
|
|
300
|
+
const reply = 'a \\b and Tom & Jerry \\. end'
|
|
301
|
+
const once = normalizeForSpeech(reply)
|
|
302
|
+
expect(normalizeForSpeech(once)).toBe(once)
|
|
303
|
+
})
|
|
304
|
+
})
|
|
305
|
+
|
|
306
|
+
describe('normalizeForSpeech — review findings (fixpoint decode, metachar, nits)', () => {
|
|
307
|
+
it('L2 parity: a double-encoded entity decodes to a fixpoint (depth-independent)', () => {
|
|
308
|
+
// Single decode and double decode must land on the SAME spoken char so the
|
|
309
|
+
// immediate voice-out and the lazy Listen tap never diverge.
|
|
310
|
+
const single = normalizeForSpeech('&amp;lt;')
|
|
311
|
+
const doubled = normalizeForSpeech(normalizeForSpeech('&amp;lt;'))
|
|
312
|
+
expect(single).toBe('<')
|
|
313
|
+
expect(doubled).toBe('<')
|
|
314
|
+
expect(normalizeForSpeech('&amp;amp;')).toBe('&')
|
|
315
|
+
})
|
|
316
|
+
|
|
317
|
+
it('L1: an entity that decodes to a line-leading metachar keeps a spoken form', () => {
|
|
318
|
+
// # → '#'. Naively re-fed to the heading stripper it would vanish; the
|
|
319
|
+
// user escaped it on purpose, so it must survive as spoken "hash".
|
|
320
|
+
expect(normalizeForSpeech('# Heading')).toBe('hash Heading')
|
|
321
|
+
expect(normalizeForSpeech('2 * 3')).toBe('2 asterisk 3')
|
|
322
|
+
})
|
|
323
|
+
|
|
324
|
+
it('nit: a dangling trailing backslash is dropped, never spoken', () => {
|
|
325
|
+
expect(normalizeForSpeech('ends here\\')).toBe('ends here')
|
|
326
|
+
expect(normalizeForSpeech('ends here\\')).not.toContain('\\')
|
|
327
|
+
})
|
|
328
|
+
|
|
329
|
+
it('nit: \ decodes to a backslash which is then stripped (no trash)', () => {
|
|
330
|
+
expect(normalizeForSpeech('X\Y')).toBe('XY')
|
|
331
|
+
expect(normalizeForSpeech('X\Y')).not.toContain('\\')
|
|
332
|
+
})
|
|
333
|
+
|
|
334
|
+
it('fixpoint does not over-decode entity-less text (Q&A stays literal)', () => {
|
|
335
|
+
expect(decodeHtmlEntities('Q&A test')).toBe('Q&A test')
|
|
336
|
+
})
|
|
337
|
+
})
|
|
@@ -38,6 +38,8 @@
|
|
|
38
38
|
* ~5 → "about 5", > blockquote markers dropped.
|
|
39
39
|
*/
|
|
40
40
|
|
|
41
|
+
import { decodeHtmlEntities, stripBackslashEscapes } from './voice-normalize-text'
|
|
42
|
+
|
|
41
43
|
const NULL = '\x00'
|
|
42
44
|
const INLINE_PH = `${NULL}TN_INLINE`
|
|
43
45
|
|
|
@@ -197,6 +199,16 @@ export function normalizeForTts(text: string): string {
|
|
|
197
199
|
|
|
198
200
|
let s = text.replace(/\r\n?/g, '\n')
|
|
199
201
|
|
|
202
|
+
// -- HTML entities → char, then backslash escapes → the escaped char. The
|
|
203
|
+
// last line of defence at the /tts body build: the Listen lazy path and
|
|
204
|
+
// the pre-synth queue can synthesize from cache entries that predate the
|
|
205
|
+
// normalizeForSpeech coverage, so these must be stripped here too. Both
|
|
206
|
+
// are idempotent — if normalizeForSpeech already ran there is nothing
|
|
207
|
+
// left to decode/unescape. Without this the engine speaks `\b` as
|
|
208
|
+
// "backslash b" and `&` as "amp".
|
|
209
|
+
s = decodeHtmlEntities(s)
|
|
210
|
+
s = stripBackslashEscapes(s)
|
|
211
|
+
|
|
200
212
|
// -- Code fences → spoken placeholder (before anything can see contents).
|
|
201
213
|
s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*?\n[ \t]*\2[ \t]*(?=\n|$)/g, '$1code block omitted.')
|
|
202
214
|
s = s.replace(/(^|\n)[ \t]*(`{3,}|~{3,})[^\n]*\n[\s\S]*$/g, '$1code block omitted.')
|
|
@@ -53,6 +53,89 @@
|
|
|
53
53
|
/** Replace a fenced code block with a spoken placeholder. */
|
|
54
54
|
const CODE_BLOCK_PLACEHOLDER = 'code block omitted'
|
|
55
55
|
|
|
56
|
+
/** Named HTML entities the reply text realistically carries. */
|
|
57
|
+
const HTML_ENTITIES: Record<string, string> = {
|
|
58
|
+
amp: '&', lt: '<', gt: '>', quot: '"', apos: "'", nbsp: ' ',
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Markdown metacharacters that the block/emphasis/table stripper would
|
|
63
|
+
* silently consume at line-start or as a pair. When such a char arrives via
|
|
64
|
+
* an entity escape the user meant it LITERALLY (that is the whole point of
|
|
65
|
+
* escaping it), so instead of emitting the raw char — which the downstream
|
|
66
|
+
* stripper would then eat, losing the intent — we emit a neutral spoken form
|
|
67
|
+
* that survives every later pass. Deterministic; the spoken form contains no
|
|
68
|
+
* `&`/`;` so it can never re-enter the entity decoder.
|
|
69
|
+
*/
|
|
70
|
+
const METACHAR_SPOKEN: Record<string, string> = {
|
|
71
|
+
'#': ' hash ',
|
|
72
|
+
'*': ' asterisk ',
|
|
73
|
+
'_': ' underscore ',
|
|
74
|
+
'~': ' tilde ',
|
|
75
|
+
'`': ' backtick ',
|
|
76
|
+
'|': ' bar ',
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** One decode pass: named + numeric entities → char (or spoken metachar). */
|
|
80
|
+
function decodeHtmlEntitiesOnce(input: string): string {
|
|
81
|
+
const toChar = (cp: number, raw: string): string => {
|
|
82
|
+
if (!(cp > 0 && cp <= 0x10ffff)) return raw
|
|
83
|
+
const ch = String.fromCodePoint(cp)
|
|
84
|
+
return METACHAR_SPOKEN[ch] ?? ch
|
|
85
|
+
}
|
|
86
|
+
return input
|
|
87
|
+
.replace(/&#x([0-9a-f]+);/gi, (m, hex: string) => toChar(parseInt(hex, 16), m))
|
|
88
|
+
.replace(/&#(\d+);/g, (m, dec: string) => toChar(Number(dec), m))
|
|
89
|
+
.replace(/&([a-z][a-z0-9]*);/gi, (m, name: string) => {
|
|
90
|
+
const ch = HTML_ENTITIES[name.toLowerCase()]
|
|
91
|
+
if (ch === undefined) return m
|
|
92
|
+
return METACHAR_SPOKEN[ch] ?? ch
|
|
93
|
+
})
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Decode HTML entities (named + numeric) to their character so a TTS engine
|
|
98
|
+
* never reads `&` as "amp". Unknown named entities are left untouched.
|
|
99
|
+
* Pure + deterministic.
|
|
100
|
+
*
|
|
101
|
+
* Iterates to a FIXPOINT: a double-encoded entity (`&amp;lt;`) is decoded
|
|
102
|
+
* repeatedly until no entity remains, so this pass is depth-idempotent —
|
|
103
|
+
* applying it once yields the same result as applying it twice. That keeps
|
|
104
|
+
* every voice callsite in lockstep: the immediate voice-out path runs
|
|
105
|
+
* normalizeForSpeech THEN normalizeForTts, and the value the lazy Listen tap /
|
|
106
|
+
* pre-synth queue reads is itself already normalizeForSpeech'd before its own
|
|
107
|
+
* normalizeForTts — fixpoint decoding guarantees both speak an identical
|
|
108
|
+
* string regardless of how deep the original encoding was. The loop strictly
|
|
109
|
+
* shrinks the entity count each turn (and is capped) so it always terminates.
|
|
110
|
+
* Text WITHOUT a trailing `;` (e.g. `Q&A`) matches nothing and is returned
|
|
111
|
+
* untouched.
|
|
112
|
+
*/
|
|
113
|
+
export function decodeHtmlEntities(input: string): string {
|
|
114
|
+
let s = input
|
|
115
|
+
// A fully-decodable chain shrinks by at least one entity per pass; the cap
|
|
116
|
+
// is a belt-and-braces guard against any pathological crafted input.
|
|
117
|
+
for (let i = 0; i < 10; i++) {
|
|
118
|
+
const next = decodeHtmlEntitiesOnce(s)
|
|
119
|
+
if (next === s) break
|
|
120
|
+
s = next
|
|
121
|
+
}
|
|
122
|
+
return s
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Remove markdown/MarkdownV2 backslash escapes so the spoken text carries no
|
|
127
|
+
* literal backslashes. A backslash before ANY single character is dropped,
|
|
128
|
+
* keeping the character (`\.` → ".", `\*` → "*", `\b` → "b"); a dangling
|
|
129
|
+
* trailing backslash is dropped. Pure + deterministic + idempotent (a second
|
|
130
|
+
* pass finds no backslashes). A real newline is preserved (only the escaping
|
|
131
|
+
* backslash is consumed).
|
|
132
|
+
*/
|
|
133
|
+
export function stripBackslashEscapes(input: string): string {
|
|
134
|
+
// `\X` → `X` for any following char (including an escaped `\\`), then drop
|
|
135
|
+
// any lone backslash the first pass left (an escaped backslash's survivor).
|
|
136
|
+
return input.replace(/\\([\s\S])/g, '$1').replace(/\\/g, '')
|
|
137
|
+
}
|
|
138
|
+
|
|
56
139
|
// ---------------------------------------------------------------------------
|
|
57
140
|
// Number → words helpers (small, deterministic, English cardinal only).
|
|
58
141
|
// Used by the numbers/units pass. Supports 0..999_999_999 which is far more
|
|
@@ -168,6 +251,23 @@ export function normalizeForSpeech(input: string): string {
|
|
|
168
251
|
if (!input) return ''
|
|
169
252
|
let s = input.replace(/\r\n?/g, '\n')
|
|
170
253
|
|
|
254
|
+
// 0a. HTML entities → their character. The reply text can carry entity
|
|
255
|
+
// escapes (`&`, `<`, `'`) that a TTS engine would otherwise
|
|
256
|
+
// read as "amp" / "lt" / a digit run. Decode BEFORE markdown/symbol
|
|
257
|
+
// passes so the recovered char is then handled naturally (e.g. a
|
|
258
|
+
// decoded `&` becomes "and" in the symbols pass).
|
|
259
|
+
s = decodeHtmlEntities(s)
|
|
260
|
+
|
|
261
|
+
// 0b. Backslash escapes → the escaped character. Telegram MarkdownV2 and
|
|
262
|
+
// CommonMark escape literal punctuation with a leading backslash
|
|
263
|
+
// (`\.`, `\-`, `\*`), and a backslash before a non-punctuation char
|
|
264
|
+
// (`\b`) is a literal backslash. Left in place the engine speaks
|
|
265
|
+
// "backslash b" / "slash b" — exactly the operator's "trash" report.
|
|
266
|
+
// Unescaping here (before the emphasis pass) restores the literal text
|
|
267
|
+
// so genuine `*emphasis*` markers are still stripped downstream while
|
|
268
|
+
// an escaped `\*` collapses to nothing spoken. Runs once; idempotent.
|
|
269
|
+
s = stripBackslashEscapes(s)
|
|
270
|
+
|
|
171
271
|
// 0. Emoji & pictographs → dropped entirely, then whitespace collapsed.
|
|
172
272
|
// TTS reads an emoji as its long CLDR name ("grinning face"), which is
|
|
173
273
|
// noise. We also drop `:shortcode:` forms so nothing is read as
|