switchroom 0.21.7 → 0.21.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/bin/tmp-reaper.sh +234 -0
  2. package/dist/agent-scheduler/index.js +1 -1
  3. package/dist/auth-broker/index.js +2 -2
  4. package/dist/cli/notion-write-pretool.mjs +1 -1
  5. package/dist/cli/switchroom.js +3421 -2744
  6. package/dist/host-control/main.js +177 -13
  7. package/dist/vault/approvals/kernel-server.js +2 -2
  8. package/dist/vault/broker/server.js +2 -2
  9. package/package.json +5 -4
  10. package/profiles/_base/start.sh.hbs +115 -0
  11. package/profiles/_shared/local-time.md.hbs +6 -0
  12. package/profiles/default/CLAUDE.md.hbs +0 -12
  13. package/telegram-plugin/dist/gateway/gateway.js +1017 -465
  14. package/telegram-plugin/gateway/agent-process-liveness.ts +558 -0
  15. package/telegram-plugin/gateway/approval-hold.ts +32 -1
  16. package/telegram-plugin/gateway/approval-outcome-sources.ts +274 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +21 -9
  18. package/telegram-plugin/gateway/callback-query-handlers.ts +87 -15
  19. package/telegram-plugin/gateway/eval-case-proposal-inbound-builders.ts +197 -0
  20. package/telegram-plugin/gateway/gateway.ts +12 -10
  21. package/telegram-plugin/gateway/pending-inbound-buffer.ts +167 -11
  22. package/telegram-plugin/gateway/self-improve-proposal-wiring.test.ts +333 -0
  23. package/telegram-plugin/gateway/self-improve-proposal-wiring.ts +152 -3
  24. package/telegram-plugin/gateway/subagent-handback-marker.ts +19 -0
  25. package/telegram-plugin/tests/agent-process-liveness.test.ts +406 -0
  26. package/telegram-plugin/tests/approval-hold-record.test.ts +21 -8
  27. package/telegram-plugin/tests/boot-resume-gateway-only-respawn.test.ts +752 -0
  28. package/telegram-plugin/tests/boot-resume-guard-wiring.test.ts +203 -0
  29. package/telegram-plugin/tests/callback-query-handlers.test.ts +143 -1
  30. package/telegram-plugin/tests/eval-case-proposal-inbound-builders.test.ts +144 -0
  31. package/telegram-plugin/tests/hermes-messages-paging.test.ts +149 -0
  32. package/telegram-plugin/tests/hermes-session-search.test.ts +146 -0
  33. package/telegram-plugin/tests/pending-inbound-buffer.test.ts +443 -2
  34. package/telegram-plugin/tests/subagent-handback-marker.test.ts +14 -0
@@ -0,0 +1,203 @@
1
+ /**
2
+ * switchroom#4641 — wiring pin for the boot-resume generation guard.
3
+ *
4
+ * `boot-resume-gateway-only-respawn.test.ts` proves the OUTCOME (a stamped
5
+ * generation token yields zero `resume_interrupted` spool entries, an unstamped
6
+ * one still resumes) over the real registry, builders and spool, but it
7
+ * supplies the ORDERING itself: gateway.ts is a 24k-line module whose boot
8
+ * block cannot be imported without booting a gateway. This file closes that gap
9
+ * by pinning gateway.ts's own wiring, so a refactor cannot move the guard below
10
+ * the reaper, drop it, or move the token stamp back inside the block while
11
+ * the outcome suite stays green against its own copy of the sequence.
12
+ *
13
+ * Pinned:
14
+ * 1. the label exists and is on the `isGatewayMain` boot-registry block,
15
+ * 2. the guard runs and `break`s out of that label,
16
+ * 3. it precedes the orphan-turn reaper, the bridge-dead marker consumption,
17
+ * the interrupted-turn finder, the resume builders, the crash-redelivery
18
+ * capture and `writePendingTurnEnv` — i.e. `break` skips ALL of them,
19
+ * 4. the break carries a comment naming those skipped effects (the reviewer
20
+ * MEDIUM: the skip is wider than the resume path and was undocumented),
21
+ * 5. THE STAMP FOLLOWS THE DURABLE PUT. The `bootResumeInit` block only
22
+ * builds `bootResumeInbound` in MEMORY; the resume becomes crash-
23
+ * survivable ~8k lines later at `inboundSpool.put(...)`. Stamping at the
24
+ * tail of the block (the shape this PR's first revision shipped) means a
25
+ * crash in between leaves a token, no `agent-process.json`, and a
26
+ * successor that suppresses a turn already stamped `ended_via='restart'`
27
+ * — silent, permanent work loss. So: the stamp must NOT appear inside the
28
+ * block at all, must come after the put and after `markTurnResumed`, and
29
+ * must be UNCONDITIONAL (top-level, gated only on `isGatewayMain`) so a
30
+ * boot with nothing to resume still writes a token. Same rule
31
+ * `markTurnResumed` already obeys — see turns-schema.ts's "the caller
32
+ * must stamp only AFTER the resume inbound is durably spooled".
33
+ */
34
+
35
+ import { describe, it, expect } from 'vitest'
36
+ import { readFileSync } from 'node:fs'
37
+ import { fileURLToPath } from 'node:url'
38
+ import { dirname, resolve } from 'node:path'
39
+
40
+ const __dirname = dirname(fileURLToPath(import.meta.url))
41
+ const SRC = readFileSync(
42
+ resolve(__dirname, '..', 'gateway', 'gateway.ts'),
43
+ 'utf8',
44
+ )
45
+
46
+ /** First index of `needle`, asserted present. */
47
+ function idx(needle: string): number {
48
+ const i = SRC.indexOf(needle)
49
+ expect(i, `gateway.ts should contain ${JSON.stringify(needle)}`).toBeGreaterThan(-1)
50
+ return i
51
+ }
52
+
53
+ describe('#4641 boot-resume guard wiring in gateway.ts', () => {
54
+ it('labels the boot-registry block so the guard can exit it', () => {
55
+ expect(SRC).toContain('bootResumeInit: if (isGatewayMain) try {')
56
+ })
57
+
58
+ it('calls the guard and breaks out of the whole block', () => {
59
+ expect(SRC).toMatch(
60
+ /if \(shouldSkipBootResumeForGatewayOnlyRespawn\(STATE_DIR\)\) break bootResumeInit/,
61
+ )
62
+ expect(SRC).toContain(
63
+ "import { shouldSkipBootResumeForGatewayOnlyRespawn, markBootResumeComplete } from './agent-process-liveness.js'",
64
+ )
65
+ })
66
+
67
+ it('runs the guard BEFORE every side effect a gateway-only respawn must skip', () => {
68
+ const guard = idx('shouldSkipBootResumeForGatewayOnlyRespawn(STATE_DIR)')
69
+ // The orphan-turn reaper — what stamped the live turn `ended_via='restart'`.
70
+ expect(guard).toBeLessThan(idx('markOrphanedWithTimeoutClassification(turnsDb'))
71
+ // #3038 bridge-dead escalation marker consumption.
72
+ expect(guard).toBeLessThan(idx('consumeBridgeDeadEscalationMarker('))
73
+ // The interrupted-turn finder + the synthetic that lied to the session.
74
+ expect(guard).toBeLessThan(idx('findLatestTurnIfInterrupted(turnsDb)'))
75
+ expect(guard).toBeLessThan(idx('buildResumeInterruptedInbound({'))
76
+ // The sub-agent "killed by the restart" list.
77
+ expect(guard).toBeLessThan(idx('listNonTerminalSubagentsForTurn(turnsDb'))
78
+ // The crash-redelivery candidate capture.
79
+ expect(guard).toBeLessThan(idx('pendingRedelivery = { turn: pending'))
80
+ // The bridge-dead idle notice.
81
+ expect(guard).toBeLessThan(idx('buildBridgeDeadIdleNoticeInbound({'))
82
+ // The one-shot wake-audit env file.
83
+ expect(guard).toBeLessThan(idx('writePendingTurnEnv(agentDir, pending)'))
84
+ })
85
+
86
+ it('documents at the break every effect it skips beyond the resume path', () => {
87
+ // The reviewer MEDIUM: `break bootResumeInit` silently skipped the
88
+ // bridge-dead marker (whose own comment claims it is ALWAYS cleared), the
89
+ // idle notice and the redelivery capture. The break must name them, so the
90
+ // next reader of gateway.ts:1721 is not misled by that stale invariant.
91
+ const line = SRC.split('\n').find((l) => l.includes('break bootResumeInit'))
92
+ expect(line, 'gateway.ts should contain the guarded break').toBeDefined()
93
+ for (const named of [
94
+ 'reaper',
95
+ 'consumeBridgeDeadEscalationMarker',
96
+ 'idle notice',
97
+ 'pendingRedelivery',
98
+ 'writePendingTurnEnv',
99
+ ]) {
100
+ expect(line!, `the break comment must name ${named}`).toContain(named)
101
+ }
102
+ })
103
+
104
+ it('stamps the generation token AFTER the durable spool put, never inside the block', () => {
105
+ // The invariant this pins, and why it is NOT "the stamp is last in the
106
+ // block": the block builds `bootResumeInbound` in memory only. Stamping at
107
+ // its tail marks the generation done while the resume is still nowhere on
108
+ // disk, so a crash before the put leaves a token + no `agent-process.json`
109
+ // + a turn stamped `ended_via='restart'` and `resumed_at` NULL — and the
110
+ // successor returns `gateway-only-respawn-no-record` and drops it forever.
111
+ const durablePut = idx('inboundSpool.put(bootResumeInbound.agent, bootResumeInbound.msg)')
112
+ const stamps = [...SRC.matchAll(/markBootResumeComplete\(STATE_DIR\)/g)].map((m) => m.index!)
113
+ expect(stamps.length, 'exactly one token stamp site in gateway.ts').toBe(1)
114
+ const stamp = stamps[0]!
115
+
116
+ // 1. It is OUTSIDE the bootResumeInit block: the block's `} catch (err) {`
117
+ // closes long before the stamp.
118
+ const blockStart = idx('bootResumeInit: if (isGatewayMain) try {')
119
+ const blockEnd = SRC.indexOf('} catch (err) {', blockStart)
120
+ expect(blockEnd).toBeGreaterThan(blockStart)
121
+ expect(
122
+ stamp,
123
+ 'the token stamp must NOT live inside the bootResumeInit block — the block only builds the resume in memory',
124
+ ).toBeGreaterThan(blockEnd)
125
+
126
+ // 2. It follows the durable put, and the at-most-once `markTurnResumed`
127
+ // ledger that obeys the identical ordering rule.
128
+ expect(durablePut, 'the durable spool put must precede the token stamp').toBeLessThan(stamp)
129
+ expect(idx('markTurnResumed(turnsDb, resumeTurnKey)')).toBeLessThan(stamp)
130
+
131
+ // 3. It is top-level (zero indentation, so not nested inside
132
+ // `if (isGatewayMain && bootResumeInbound != null)`) and independent of
133
+ // whether there was anything to resume — a boot that found nothing must
134
+ // still stamp, or a later gateway-only respawn reaps a
135
+ // meanwhile-started live turn — but it IS gated on the block not having
136
+ // thrown. gateway.ts's block-level `catch` swallows and lets init
137
+ // continue, so an ungated stamp marks the generation done after the
138
+ // reaper has already durably ended a turn that was never spooled.
139
+ const stampLine = SRC.split('\n').find((l) => l.includes('markBootResumeComplete(STATE_DIR)'))!
140
+ expect(stampLine).toMatch(
141
+ /^if \(isGatewayMain && !bootResumeThrew\) markBootResumeComplete\(STATE_DIR\)/,
142
+ )
143
+
144
+ // 4. Nothing durable-resume-related may sneak between the put and the
145
+ // stamp except the resumed_at ledger — i.e. the stamp closes that
146
+ // sequence rather than floating somewhere later in module init.
147
+ const between = SRC.slice(durablePut + 1, stamp)
148
+ expect(between, 'no second spool put may separate the durable put from the stamp')
149
+ .not.toContain('inboundSpool.put(bootResumeInbound')
150
+ expect(
151
+ between.split('\n').length,
152
+ 'the stamp must sit immediately after the resume-commit block, not drift away from it',
153
+ ).toBeLessThan(40)
154
+ })
155
+
156
+ it('does not re-join the stamp onto writePendingTurnEnv inside the block', () => {
157
+ // The exact regression shape: `writePendingTurnEnv(agentDir, pending); markBootResumeComplete(STATE_DIR)`
158
+ // — joined onto one line to satisfy the gateway line ratchet. The ratchet
159
+ // is not a licence to stamp early.
160
+ const envLine = SRC.split('\n').find((l) => l.includes('writePendingTurnEnv(agentDir, pending)'))!
161
+ expect(
162
+ envLine,
163
+ 'the generation token must not be stamped alongside writePendingTurnEnv (still in-memory-only territory)',
164
+ ).not.toContain('markBootResumeComplete')
165
+ })
166
+
167
+ it("sets bootResumeThrew in the block's own catch, and nowhere else", () => {
168
+ // The catch is a SWALLOW: it logs, nulls turnsDb and lets module init run
169
+ // on to the stamp. By then the reaper may already have durably written
170
+ // `ended_via='restart'` with nothing spooled, so the catch must record the
171
+ // failure and the stamp must honour it. Pinned by source text because the
172
+ // block is module-init code this suite cannot import.
173
+ const blockStart = idx('bootResumeInit: if (isGatewayMain) try {')
174
+ const catchAt = SRC.indexOf('} catch (err) {', blockStart)
175
+ const stamp = idx('markBootResumeComplete(STATE_DIR)')
176
+
177
+ const sets = [...SRC.matchAll(/bootResumeThrew = true/g)].map((m) => m.index!)
178
+ expect(sets.length, 'exactly one assignment site').toBe(1)
179
+ expect(sets[0]!, "the assignment must be in the block's catch").toBeGreaterThan(catchAt)
180
+ expect(sets[0]!, 'and before the stamp reads it').toBeLessThan(stamp)
181
+
182
+ // The catch must not early-return/exit instead: init continues, which is
183
+ // precisely why the flag is needed.
184
+ const catchBody = SRC.slice(catchAt, SRC.indexOf('\n}\n', catchAt) + 2)
185
+ expect(catchBody).toContain('turnsDb = null')
186
+ expect(catchBody).not.toContain('process.exit')
187
+
188
+ const decl = SRC.indexOf('let bootResumeThrew = false')
189
+ expect(decl, 'declared before the block so the stamp can read it').toBeGreaterThan(-1)
190
+ expect(decl).toBeLessThan(blockStart)
191
+ })
192
+
193
+ it('keeps the guard AFTER the registry is opened, so turn tracking still works', () => {
194
+ // A gateway-only respawn still needs a live turnsDb for the rest of the
195
+ // gateway; the guard skips the boot-resume side effects, not the DB.
196
+ expect(idx('turnsDb = openTurnsDb(agentDir)')).toBeLessThan(
197
+ idx('shouldSkipBootResumeForGatewayOnlyRespawn(STATE_DIR)'),
198
+ )
199
+ expect(idx('applySubagentsSchema(turnsDb)')).toBeLessThan(
200
+ idx('shouldSkipBootResumeForGatewayOnlyRespawn(STATE_DIR)'),
201
+ )
202
+ })
203
+ })
@@ -25,7 +25,7 @@
25
25
  * broker call, which is exactly the behavior the extraction must not change.
26
26
  */
27
27
 
28
- import { describe, it, expect, vi, beforeEach } from 'vitest'
28
+ import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'
29
29
  import type { Context } from 'grammy'
30
30
  import {
31
31
  createCallbackQueryHandlers,
@@ -42,6 +42,10 @@ import { createSweepableCardStore } from '../gateway/approval-card-stores.js'
42
42
  import { createSweepableStore } from '../gateway/pending-state-stores.js'
43
43
  import { StagingMap } from '../secret-detect/staging.js'
44
44
  import { InlineKeyboard } from 'grammy'
45
+ import { mkdtempSync, rmSync } from 'node:fs'
46
+ import { tmpdir } from 'node:os'
47
+ import { join } from 'node:path'
48
+ import { enqueueEvalCaseProposal } from '../../src/self-improve/eval-case-proposals.js'
45
49
 
46
50
  // Mock the auth-broker client so the `auth:use:` swap path is observable
47
51
  // (spy on setActive) without a live UDS broker. Only handleAuthDashboardCallback
@@ -625,6 +629,144 @@ describe('handleSkillProposalCallback', () => {
625
629
  })
626
630
  })
627
631
 
632
+ // ── evcase:* — eval-case proposal ───────────────────────────────────────
633
+ //
634
+ // The defect these pin: this handler was a copy of the skill handler that
635
+ // dropped the `deliverResumeSyntheticOrBuffer` call, so BOTH exits — dismiss
636
+ // and approve — edited the card and returned without telling the proposing
637
+ // agent anything. The agent had been steered to end its turn and wait for a
638
+ // wake-up that no code path sent. On unmodified code every assertion below on
639
+ // `deliverResumeSyntheticOrBuffer` fails with 0 calls.
640
+ //
641
+ // `handleEvalCaseProposalCallback` reads the proposal store off disk (not from
642
+ // an injected store), so these drive a REAL tmp state dir via the real
643
+ // `enqueueEvalCaseProposal`. `TELEGRAM_STATE_DIR` is NOT one of
644
+ // `agent-state-dir-guard`'s GUARDED_STATE_DIR_VARS, so the guard neither
645
+ // redirects nor blocks it — setting it here is the whole isolation.
646
+
647
+ describe('handleEvalCaseProposalCallback', () => {
648
+ let stateDir: string
649
+ let prevStateDir: string | undefined
650
+ let prevAgent: string | undefined
651
+
652
+ beforeEach(() => {
653
+ stateDir = mkdtempSync(join(tmpdir(), 'evcase-cb-'))
654
+ prevStateDir = process.env.TELEGRAM_STATE_DIR
655
+ prevAgent = process.env.SWITCHROOM_AGENT_NAME
656
+ process.env.TELEGRAM_STATE_DIR = stateDir
657
+ process.env.SWITCHROOM_AGENT_NAME = 'test-agent'
658
+ })
659
+
660
+ afterEach(() => {
661
+ if (prevStateDir == null) delete process.env.TELEGRAM_STATE_DIR
662
+ else process.env.TELEGRAM_STATE_DIR = prevStateDir
663
+ if (prevAgent == null) delete process.env.SWITCHROOM_AGENT_NAME
664
+ else process.env.SWITCHROOM_AGENT_NAME = prevAgent
665
+ rmSync(stateDir, { recursive: true, force: true })
666
+ })
667
+
668
+ function seedProposal(over: { heldOut?: boolean } = {}) {
669
+ return enqueueEvalCaseProposal(stateDir, {
670
+ skill_slug: 'deploy-checklist',
671
+ skill_dir: join(stateDir, 'skills', 'deploy-checklist'),
672
+ case: { prompt: 'the correction, as an input' },
673
+ fingerprint: 'fp-1',
674
+ held_out: over.heldOut === true,
675
+ })
676
+ }
677
+
678
+ it('dismiss wakes the agent with an eval_case_rejected inbound', async () => {
679
+ const { deps } = makeDeps()
680
+ const h = createCallbackQueryHandlers(deps)
681
+ const p = seedProposal()
682
+ const { ctx } = makeCtx()
683
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:deny:${p.id}`)
684
+ expect(deps.deliverResumeSyntheticOrBuffer).toHaveBeenCalledWith(
685
+ 'test-agent',
686
+ expect.objectContaining({
687
+ meta: expect.objectContaining({
688
+ source: 'eval_case_rejected',
689
+ proposal_id: p.id,
690
+ skill_slug: 'deploy-checklist',
691
+ }),
692
+ }),
693
+ )
694
+ })
695
+
696
+ it('approve with a successful applier wakes the agent with eval_case_applied', async () => {
697
+ const runEvalCaseApply = vi.fn(() => ({ ok: true, out: 'added 1 case' }))
698
+ const { deps } = makeDeps({ runEvalCaseApply })
699
+ const h = createCallbackQueryHandlers(deps)
700
+ const p = seedProposal()
701
+ const { ctx } = makeCtx()
702
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:approve:${p.id}`)
703
+ expect(runEvalCaseApply).toHaveBeenCalledWith(p.id)
704
+ expect(deps.deliverResumeSyntheticOrBuffer).toHaveBeenCalledWith(
705
+ 'test-agent',
706
+ expect.objectContaining({
707
+ meta: expect.objectContaining({ source: 'eval_case_applied', proposal_id: p.id }),
708
+ }),
709
+ )
710
+ })
711
+
712
+ it('approve with a failing applier wakes the agent with eval_case_apply_failed + the output', async () => {
713
+ const runEvalCaseApply = vi.fn(() => ({ ok: false, out: 'ENOENT: skill dir missing' }))
714
+ const { deps, injected } = makeDeps({ runEvalCaseApply })
715
+ const h = createCallbackQueryHandlers(deps)
716
+ const p = seedProposal()
717
+ const { ctx } = makeCtx()
718
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:approve:${p.id}`)
719
+ expect(deps.deliverResumeSyntheticOrBuffer).toHaveBeenCalledWith(
720
+ 'test-agent',
721
+ expect.objectContaining({
722
+ meta: expect.objectContaining({ source: 'eval_case_apply_failed', proposal_id: p.id }),
723
+ }),
724
+ )
725
+ // The applier's own diagnosis reaches the agent, not just "it failed".
726
+ expect(injected[0]?.text).toContain('ENOENT: skill dir missing')
727
+ })
728
+
729
+ it('carries the card’s forum topic into the outcome inbound', async () => {
730
+ const { deps } = makeDeps()
731
+ const h = createCallbackQueryHandlers(deps)
732
+ const p = seedProposal()
733
+ const { ctx } = makeCtx({ threadId: 7 })
734
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:deny:${p.id}`)
735
+ expect(deps.deliverResumeSyntheticOrBuffer).toHaveBeenCalledWith(
736
+ 'test-agent',
737
+ expect.objectContaining({
738
+ threadId: 7,
739
+ meta: expect.objectContaining({ message_thread_id: '7' }),
740
+ }),
741
+ )
742
+ })
743
+
744
+ it('a second tap injects nothing — the non-pending guard runs before every injection', async () => {
745
+ const runEvalCaseApply = vi.fn(() => ({ ok: true, out: '' }))
746
+ const { deps } = makeDeps({ runEvalCaseApply })
747
+ const h = createCallbackQueryHandlers(deps)
748
+ const p = seedProposal()
749
+ const { ctx } = makeCtx()
750
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:approve:${p.id}`)
751
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:approve:${p.id}`)
752
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:deny:${p.id}`)
753
+ expect(deps.deliverResumeSyntheticOrBuffer).toHaveBeenCalledTimes(1)
754
+ expect(runEvalCaseApply).toHaveBeenCalledTimes(1)
755
+ })
756
+
757
+ it('a non-allowlisted tapper neither applies nor wakes the agent', async () => {
758
+ const runEvalCaseApply = vi.fn(() => ({ ok: true, out: '' }))
759
+ const { deps } = makeDeps({ runEvalCaseApply })
760
+ const h = createCallbackQueryHandlers(deps)
761
+ const p = seedProposal()
762
+ const { ctx, raw } = makeCtx({ senderId: '999' })
763
+ await h.handleEvalCaseProposalCallback(ctx, `evcase:approve:${p.id}`)
764
+ expect(raw.answerCallbackQuery).toHaveBeenCalledWith({ text: 'Not authorized.' })
765
+ expect(runEvalCaseApply).not.toHaveBeenCalled()
766
+ expect(deps.deliverResumeSyntheticOrBuffer).not.toHaveBeenCalled()
767
+ })
768
+ })
769
+
628
770
  // ── op:* — operator-event card actions ──────────────────────────────────
629
771
 
630
772
  describe('handleOperatorEventCallback', () => {
@@ -0,0 +1,144 @@
1
+ /**
2
+ * Fixture pins for the eval-case outcome inbounds.
3
+ *
4
+ * `meta.source` is what the bridge keys on to render the `<channel source=…>`
5
+ * block that wakes the proposing agent, and the `meta.*` fields are the
6
+ * forensic anchor back to the proposal + the deciding operator. A silent change
7
+ * to either would put the agent back where this fix found it — steered to end
8
+ * its turn and wait for a wake-up nothing recognises. Mirrors
9
+ * `skill-proposal-card.test.ts`'s `buildSkillProposalApplyInbound` block.
10
+ */
11
+
12
+ import { describe, it, expect } from 'vitest'
13
+ import {
14
+ buildEvalCaseAppliedInbound,
15
+ buildEvalCaseRejectedInbound,
16
+ buildEvalCaseApplyFailedInbound,
17
+ } from '../gateway/eval-case-proposal-inbound-builders.js'
18
+
19
+ const CTX = { agent: 'klanker', chat_id: '12345' }
20
+
21
+ describe('buildEvalCaseAppliedInbound', () => {
22
+ it('pins source and meta for an applied case', () => {
23
+ const inb = buildEvalCaseAppliedInbound({
24
+ ctx: CTX,
25
+ proposalId: 'p1',
26
+ skillSlug: 'deploy-checklist',
27
+ heldOut: false,
28
+ operatorId: '999',
29
+ nowMs: 1000,
30
+ })
31
+ expect(inb.meta.source).toBe('eval_case_applied')
32
+ expect(inb.meta.agent).toBe('klanker')
33
+ expect(inb.meta.proposal_id).toBe('p1')
34
+ expect(inb.meta.skill_slug).toBe('deploy-checklist')
35
+ expect(inb.meta.held_out).toBe('false')
36
+ expect(inb.meta.operator_id).toBe('999')
37
+ expect(inb.chatId).toBe('12345')
38
+ expect(inb.user).toBe('self-improve')
39
+ expect(inb.messageId).toBe(1000)
40
+ expect(inb.ts).toBe(1000)
41
+ // The agent must NOT re-write a case the gateway already applied.
42
+ expect(inb.text).toContain('do NOT write it yourself')
43
+ expect(inb.text).toContain('evals.json')
44
+ })
45
+
46
+ it('names the held-out sink instead of evals.json when held_out', () => {
47
+ const inb = buildEvalCaseAppliedInbound({
48
+ ctx: CTX,
49
+ proposalId: 'p1',
50
+ skillSlug: 'deploy-checklist',
51
+ heldOut: true,
52
+ operatorId: '999',
53
+ })
54
+ expect(inb.meta.held_out).toBe('true')
55
+ expect(inb.text).toContain('held-out sink')
56
+ expect(inb.text).not.toContain('evals.json')
57
+ })
58
+
59
+ it('carries the forum topic through to threadId and meta', () => {
60
+ const inb = buildEvalCaseAppliedInbound({
61
+ ctx: { ...CTX, threadId: 7 },
62
+ proposalId: 'p1',
63
+ skillSlug: 's',
64
+ heldOut: false,
65
+ operatorId: '999',
66
+ })
67
+ expect(inb.threadId).toBe(7)
68
+ expect(inb.meta.message_thread_id).toBe('7')
69
+ })
70
+
71
+ it('omits threadId/message_thread_id entirely for a DM proposal', () => {
72
+ const inb = buildEvalCaseAppliedInbound({
73
+ ctx: CTX,
74
+ proposalId: 'p1',
75
+ skillSlug: 's',
76
+ heldOut: false,
77
+ operatorId: '999',
78
+ })
79
+ expect('threadId' in inb).toBe(false)
80
+ expect('message_thread_id' in inb.meta).toBe(false)
81
+ })
82
+ })
83
+
84
+ describe('buildEvalCaseRejectedInbound', () => {
85
+ it('pins source and tells the agent nothing was written', () => {
86
+ const inb = buildEvalCaseRejectedInbound({
87
+ ctx: CTX,
88
+ proposalId: 'p2',
89
+ skillSlug: 'deploy-checklist',
90
+ heldOut: false,
91
+ operatorId: '999',
92
+ nowMs: 2000,
93
+ })
94
+ expect(inb.meta.source).toBe('eval_case_rejected')
95
+ expect(inb.meta.proposal_id).toBe('p2')
96
+ expect(inb.user).toBe('self-improve')
97
+ expect(inb.text).toContain('NOTHING was written')
98
+ expect(inb.text).toContain('Do NOT re-propose')
99
+ })
100
+ })
101
+
102
+ describe('buildEvalCaseApplyFailedInbound', () => {
103
+ it('pins source and refuses to let the agent assume the case landed', () => {
104
+ const inb = buildEvalCaseApplyFailedInbound({
105
+ ctx: CTX,
106
+ proposalId: 'p3',
107
+ skillSlug: 'deploy-checklist',
108
+ heldOut: false,
109
+ operatorId: '999',
110
+ applyOut: 'ENOENT: skill dir missing',
111
+ nowMs: 3000,
112
+ })
113
+ expect(inb.meta.source).toBe('eval_case_apply_failed')
114
+ expect(inb.meta.proposal_id).toBe('p3')
115
+ expect(inb.text).toContain('was NOT')
116
+ expect(inb.text).toContain('do NOT assume')
117
+ expect(inb.text).toContain('ENOENT: skill dir missing')
118
+ })
119
+
120
+ it('truncates a runaway applier dump so it cannot dominate the woken turn', () => {
121
+ const inb = buildEvalCaseApplyFailedInbound({
122
+ ctx: CTX,
123
+ proposalId: 'p3',
124
+ skillSlug: 's',
125
+ heldOut: false,
126
+ operatorId: '999',
127
+ applyOut: 'x'.repeat(5000),
128
+ })
129
+ expect(inb.text).toContain('x'.repeat(500))
130
+ expect(inb.text).not.toContain('x'.repeat(501))
131
+ })
132
+
133
+ it('says "(no output)" rather than emitting an empty line', () => {
134
+ const inb = buildEvalCaseApplyFailedInbound({
135
+ ctx: CTX,
136
+ proposalId: 'p3',
137
+ skillSlug: 's',
138
+ heldOut: false,
139
+ operatorId: '999',
140
+ applyOut: ' \n ',
141
+ })
142
+ expect(inb.text).toContain('(no output)')
143
+ })
144
+ })
@@ -0,0 +1,149 @@
1
+ /**
2
+ * `GET /api/sessions/:id/messages` — ordering and pagination, against a REAL
3
+ * turns DB.
4
+ *
5
+ * Separate from `hermes-rest-parity.test.ts` for one reason: that fixture
6
+ * points `SWITCHROOM_AGENTS_DIR` at an empty tmpdir, so every session has zero
7
+ * messages and an assertion about ORDER or about which slice came back passes
8
+ * vacuously there. Ordering and offset windowing are exactly the properties
9
+ * that need real rows, and the turns DB is `bun:sqlite` — hence a bun test
10
+ * (vitest excludes it; see vitest.config.ts). It lives under
11
+ * telegram-plugin/tests/ rather than beside the adapter because that is the only
12
+ * tree the bun CI job walks — `bun test` runs with cwd telegram-plugin/, so a
13
+ * root-level src/ bun test is executed by NEITHER runner
14
+ * (scripts/check-test-runner-coverage.mjs, and the debt list it guards).
15
+ *
16
+ * Upstream contract being asserted (`hermes_cli/web_routers/sessions.py:601-651`
17
+ * at 9da6d455c9e1f2bf74bb9f47766ee9fc52e17bfb):
18
+ * - messages come back CHRONOLOGICAL within the page;
19
+ * - `order=oldest` windows from the start, `order=latest` from the end;
20
+ * - `pagination.returned` is the real row count of that window.
21
+ *
22
+ * The bug each case guards, stated so a future edit cannot weaken it into a
23
+ * code-path check: the adapter read `listTurnsForAgent` (newest-first) and
24
+ * projected it without reversing, so the transcript arrived backwards; and it
25
+ * emitted no `pagination`, which `getAllSessionMessages` (hermes.ts:713-750)
26
+ * reads as "legacy backend, that was everything" and stops paging on.
27
+ */
28
+
29
+ import { test, expect, beforeEach, afterEach } from "bun:test";
30
+ import { mkdtempSync, rmSync, mkdirSync } from "node:fs";
31
+ import { tmpdir } from "node:os";
32
+ import { join } from "node:path";
33
+ import { openTurnsDb } from "../registry/turns-schema.js";
34
+ import { handleHermesRest, type HermesRestResult } from "../../src/web/hermes-adapter.js";
35
+ import type { SwitchroomConfig } from "../../src/config/schema.js";
36
+
37
+ const AGENT = "alpha";
38
+ let agentsDir = "";
39
+
40
+ const CONFIG = { agents: { [AGENT]: {} } } as unknown as SwitchroomConfig;
41
+
42
+ /** Seed `n` completed turns with explicit, strictly-increasing start times. */
43
+ function seedTurns(n: number) {
44
+ const dir = join(agentsDir, AGENT);
45
+ mkdirSync(dir, { recursive: true });
46
+ const db = openTurnsDb(dir);
47
+ try {
48
+ for (let i = 0; i < n; i++) {
49
+ const ts = 1_700_000_000_000 + i * 1000;
50
+ db.prepare(
51
+ `INSERT INTO turns
52
+ (turn_key, chat_id, thread_id, started_at, ended_at, ended_via,
53
+ user_prompt_preview, assistant_reply_preview, created_at, updated_at)
54
+ VALUES (?, ?, NULL, ?, ?, 'stop', ?, ?, ?, ?)`,
55
+ ).run(`t${i}`, "chat", ts, ts + 500, `u${i}`, `a${i}`, ts, ts);
56
+ }
57
+ } finally {
58
+ db.close();
59
+ }
60
+ }
61
+
62
+ async function messages(search: string): Promise<Record<string, unknown>> {
63
+ const res = (await handleHermesRest(
64
+ "GET",
65
+ `/api/sessions/${AGENT}/messages`,
66
+ CONFIG,
67
+ search,
68
+ )) as HermesRestResult;
69
+ expect(res).not.toBeNull();
70
+ expect(res.status).toBe(200);
71
+ return res.body as Record<string, unknown>;
72
+ }
73
+
74
+ function contents(body: Record<string, unknown>): string[] {
75
+ return (body.messages as { content: string }[]).map((m) => m.content);
76
+ }
77
+
78
+ beforeEach(() => {
79
+ agentsDir = mkdtempSync(join(tmpdir(), "sr-hermes-msgs-"));
80
+ process.env.SWITCHROOM_AGENTS_DIR = agentsDir;
81
+ });
82
+
83
+ afterEach(() => {
84
+ rmSync(agentsDir, { recursive: true, force: true });
85
+ });
86
+
87
+ test("messages are chronological, not the newest-first order the turns DB reads in", async () => {
88
+ seedTurns(3);
89
+ const body = await messages("?limit=500&offset=0&order=oldest");
90
+ // u0/a0 is the OLDEST turn. Before the fix this list started at u2.
91
+ expect(contents(body)).toEqual(["u0", "a0", "u1", "a1", "u2", "a2"]);
92
+ });
93
+
94
+ test("order=oldest windows from the start and reports what it returned", async () => {
95
+ seedTurns(3);
96
+ const body = await messages("?limit=2&offset=2&order=oldest");
97
+ expect(contents(body)).toEqual(["u1", "a1"]);
98
+ expect(body.pagination).toEqual({ limit: 2, offset: 2, order: "oldest", returned: 2 });
99
+ });
100
+
101
+ test("order=latest windows from the END, still chronological inside the page", async () => {
102
+ seedTurns(3);
103
+ const body = await messages("?limit=3&order=latest");
104
+ expect(contents(body)).toEqual(["a1", "u2", "a2"]);
105
+ expect(body.pagination).toEqual({ limit: 3, offset: 0, order: "latest", returned: 3 });
106
+ });
107
+
108
+ test("paging with the desktop's own loop terminates having read every message once", async () => {
109
+ seedTurns(5); // 10 messages
110
+ // getAllSessionMessages (hermes.ts:713-750): limit=500, order=oldest, advance
111
+ // offset by the page length, stop when the page is short of `limit`.
112
+ const seen: string[] = [];
113
+ let offset = 0;
114
+ for (let guard = 0; guard < 10; guard++) {
115
+ const page = await messages(`?limit=4&offset=${offset}&order=oldest`);
116
+ const pagination = page.pagination as { limit: number };
117
+ seen.push(...contents(page));
118
+ if (contents(page).length === 0 || contents(page).length < pagination.limit) break;
119
+ offset += contents(page).length;
120
+ }
121
+ expect(seen).toEqual(["u0", "a0", "u1", "a1", "u2", "a2", "u3", "a3", "u4", "a4"]);
122
+ });
123
+
124
+ test("an omitted limit defaults to the latest 500, an explicit one defaults to oldest", async () => {
125
+ seedTurns(1);
126
+ expect((await messages("")).pagination).toEqual({
127
+ limit: 500,
128
+ offset: 0,
129
+ order: "latest",
130
+ returned: 2,
131
+ });
132
+ expect((await messages("?limit=10")).pagination).toEqual({
133
+ limit: 10,
134
+ offset: 0,
135
+ order: "oldest",
136
+ returned: 2,
137
+ });
138
+ });
139
+
140
+ test("an unrecognised order is a 400, not a silently-reinterpreted page", async () => {
141
+ seedTurns(1);
142
+ const res = await handleHermesRest(
143
+ "GET",
144
+ `/api/sessions/${AGENT}/messages`,
145
+ CONFIG,
146
+ "?order=sideways",
147
+ );
148
+ expect(res?.status).toBe(400);
149
+ });