switchroom 0.21.7 → 0.21.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/bin/tmp-reaper.sh +234 -0
  2. package/dist/agent-scheduler/index.js +1 -1
  3. package/dist/auth-broker/index.js +2 -2
  4. package/dist/cli/notion-write-pretool.mjs +1 -1
  5. package/dist/cli/switchroom.js +3421 -2744
  6. package/dist/host-control/main.js +177 -13
  7. package/dist/vault/approvals/kernel-server.js +2 -2
  8. package/dist/vault/broker/server.js +2 -2
  9. package/package.json +5 -4
  10. package/profiles/_base/start.sh.hbs +115 -0
  11. package/profiles/_shared/local-time.md.hbs +6 -0
  12. package/profiles/default/CLAUDE.md.hbs +0 -12
  13. package/telegram-plugin/dist/gateway/gateway.js +1017 -465
  14. package/telegram-plugin/gateway/agent-process-liveness.ts +558 -0
  15. package/telegram-plugin/gateway/approval-hold.ts +32 -1
  16. package/telegram-plugin/gateway/approval-outcome-sources.ts +274 -0
  17. package/telegram-plugin/gateway/bridge-dead-watchdog.ts +21 -9
  18. package/telegram-plugin/gateway/callback-query-handlers.ts +87 -15
  19. package/telegram-plugin/gateway/eval-case-proposal-inbound-builders.ts +197 -0
  20. package/telegram-plugin/gateway/gateway.ts +12 -10
  21. package/telegram-plugin/gateway/pending-inbound-buffer.ts +167 -11
  22. package/telegram-plugin/gateway/self-improve-proposal-wiring.test.ts +333 -0
  23. package/telegram-plugin/gateway/self-improve-proposal-wiring.ts +152 -3
  24. package/telegram-plugin/gateway/subagent-handback-marker.ts +19 -0
  25. package/telegram-plugin/tests/agent-process-liveness.test.ts +406 -0
  26. package/telegram-plugin/tests/approval-hold-record.test.ts +21 -8
  27. package/telegram-plugin/tests/boot-resume-gateway-only-respawn.test.ts +752 -0
  28. package/telegram-plugin/tests/boot-resume-guard-wiring.test.ts +203 -0
  29. package/telegram-plugin/tests/callback-query-handlers.test.ts +143 -1
  30. package/telegram-plugin/tests/eval-case-proposal-inbound-builders.test.ts +144 -0
  31. package/telegram-plugin/tests/hermes-messages-paging.test.ts +149 -0
  32. package/telegram-plugin/tests/hermes-session-search.test.ts +146 -0
  33. package/telegram-plugin/tests/pending-inbound-buffer.test.ts +443 -2
  34. package/telegram-plugin/tests/subagent-handback-marker.test.ts +14 -0
@@ -0,0 +1,146 @@
1
+ /**
2
+ * `GET /api/sessions/search` — that it actually SEARCHES.
3
+ *
4
+ * `hermes-rest-parity.test.ts` can only assert the contract floor here
5
+ * (`Array.isArray(results)`), because it points `SWITCHROOM_AGENTS_DIR` at an
6
+ * empty tmpdir: with no turns anywhere, "found the right rows", "found nothing
7
+ * for a miss" and "hardcoded `{results: []}`" are indistinguishable. A
8
+ * hardcoded empty list would pass that floor forever, and an always-200 empty
9
+ * body is the exact bug class the desktop cannot detect —
10
+ * `isEndpointMissingError` only matches 404s.
11
+ *
12
+ * So this file seeds a real turns registry (`bun:sqlite`, hence a bun test —
13
+ * vitest excludes it; it lives under telegram-plugin/tests/ because that is the
14
+ * only tree the bun CI job walks) and asserts the discriminating outcomes:
15
+ * a query that should hit, hits; a query that should miss, misses.
16
+ */
17
+
18
+ import { test, expect, beforeEach, afterEach } from "bun:test";
19
+ import { mkdtempSync, rmSync, mkdirSync } from "node:fs";
20
+ import { tmpdir } from "node:os";
21
+ import { join } from "node:path";
22
+ import { openTurnsDb } from "../registry/turns-schema.js";
23
+ import { handleHermesRest, type HermesRestResult } from "../../src/web/hermes-adapter.js";
24
+ import type { SwitchroomConfig } from "../../src/config/schema.js";
25
+
26
+ const CONFIG = {
27
+ agents: { alpha: { model: "sonnet" }, beta: {} },
28
+ } as unknown as SwitchroomConfig;
29
+
30
+ let agentsDir = "";
31
+
32
+ function seed(agent: string, rows: { user: string; assistant: string }[]) {
33
+ const dir = join(agentsDir, agent);
34
+ mkdirSync(dir, { recursive: true });
35
+ const db = openTurnsDb(dir);
36
+ try {
37
+ rows.forEach((r, i) => {
38
+ const ts = 1_700_000_000_000 + i * 1000;
39
+ db.prepare(
40
+ `INSERT INTO turns
41
+ (turn_key, chat_id, thread_id, started_at, ended_at, ended_via,
42
+ user_prompt_preview, assistant_reply_preview, created_at, updated_at)
43
+ VALUES (?, ?, NULL, ?, ?, 'stop', ?, ?, ?, ?)`,
44
+ ).run(`${agent}-t${i}`, "chat", ts, ts + 500, r.user, r.assistant, ts, ts);
45
+ });
46
+ } finally {
47
+ db.close();
48
+ }
49
+ }
50
+
51
+ type Result = {
52
+ session_id: string;
53
+ lineage_root: string;
54
+ snippet: string;
55
+ role: string | null;
56
+ source: string | null;
57
+ model: string | null;
58
+ session_started: number | null;
59
+ };
60
+
61
+ async function search(query: string): Promise<Result[]> {
62
+ const res = (await handleHermesRest(
63
+ "GET",
64
+ "/api/sessions/search",
65
+ CONFIG,
66
+ query,
67
+ )) as HermesRestResult;
68
+ expect(res).not.toBeNull();
69
+ // The pre-fix behaviour: the /api/sessions/:id regex matched "search" as an
70
+ // id and answered 404 {error:'Unknown session'}, which threw the panel.
71
+ expect(res.status).toBe(200);
72
+ return (res.body as { results: Result[] }).results;
73
+ }
74
+
75
+ beforeEach(() => {
76
+ agentsDir = mkdtempSync(join(tmpdir(), "sr-hermes-search-"));
77
+ process.env.SWITCHROOM_AGENTS_DIR = agentsDir;
78
+ });
79
+
80
+ afterEach(() => {
81
+ rmSync(agentsDir, { recursive: true, force: true });
82
+ });
83
+
84
+ test("a session-id match comes back, with role null and the id as its own lineage root", async () => {
85
+ const results = await search("?q=alph");
86
+ expect(results.map((r) => r.session_id)).toEqual(["alpha"]);
87
+ expect(results[0].role).toBeNull();
88
+ expect(results[0].lineage_root).toBe("alpha");
89
+ });
90
+
91
+ test("a message-content match comes back tagged with the role it matched on", async () => {
92
+ seed("alpha", [{ user: "deploy the gateway", assistant: "done" }]);
93
+ seed("beta", [{ user: "unrelated", assistant: "also unrelated" }]);
94
+
95
+ const onUser = await search("?q=deploy");
96
+ expect(onUser.map((r) => r.session_id)).toEqual(["alpha"]);
97
+ expect(onUser[0].role).toBe("user");
98
+ expect(onUser[0].snippet).toBe("deploy the gateway");
99
+
100
+ const onAssistant = await search("?q=also%20unrelated");
101
+ expect(onAssistant.map((r) => r.session_id)).toEqual(["beta"]);
102
+ expect(onAssistant[0].role).toBe("assistant");
103
+ expect(onAssistant[0].snippet).toBe("also unrelated");
104
+ });
105
+
106
+ test("a query that matches nothing returns nothing — not the whole fleet", async () => {
107
+ seed("alpha", [{ user: "deploy the gateway", assistant: "done" }]);
108
+ // The discriminating case: a stub returning every session, and a stub
109
+ // returning none, are told apart only by running both this and the hit above.
110
+ expect(await search("?q=zzz-no-such-token")).toEqual([]);
111
+ });
112
+
113
+ test("an empty or whitespace query is short-circuited, not treated as match-everything", async () => {
114
+ seed("alpha", [{ user: "deploy", assistant: "done" }]);
115
+ expect(await search("?q=")).toEqual([]);
116
+ expect(await search("?q=%20%20")).toEqual([]);
117
+ expect(await search("")).toEqual([]);
118
+ });
119
+
120
+ test("one row per session even when several turns match", async () => {
121
+ seed("alpha", [
122
+ { user: "deploy one", assistant: "ok" },
123
+ { user: "deploy two", assistant: "ok" },
124
+ { user: "deploy three", assistant: "ok" },
125
+ ]);
126
+ const results = await search("?q=deploy");
127
+ expect(results.length).toBe(1);
128
+ expect(results[0].session_id).toBe("alpha");
129
+ });
130
+
131
+ test("?limit= caps the result set", async () => {
132
+ seed("alpha", [{ user: "shared token", assistant: "ok" }]);
133
+ seed("beta", [{ user: "shared token", assistant: "ok" }]);
134
+ expect((await search("?q=shared&limit=1")).length).toBe(1);
135
+ expect((await search("?q=shared")).length).toBe(2);
136
+ });
137
+
138
+ test("id matches are ordered ahead of content matches", async () => {
139
+ // beta matches by id; alpha only by content. Upstream surfaces direct
140
+ // session-id hits first (sessions.py:169-383) and the panel relies on it.
141
+ seed("alpha", [{ user: "beta is mentioned here", assistant: "ok" }]);
142
+ const results = await search("?q=beta");
143
+ expect(results.map((r) => r.session_id)).toEqual(["beta", "alpha"]);
144
+ expect(results[0].role).toBeNull();
145
+ expect(results[1].role).toBe("user");
146
+ });
@@ -7,7 +7,8 @@
7
7
  */
8
8
 
9
9
  import { describe, it, expect } from 'vitest'
10
- import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, DEFAULT_PENDING_INBOUND_CAP } from '../gateway/pending-inbound-buffer.js'
10
+ import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, selectEvictionVictim, DEFAULT_PENDING_INBOUND_CAP, APPROVAL_OUTCOME_PROTECTION_MS } from '../gateway/pending-inbound-buffer.js'
11
+ import { APPROVAL_OUTCOME_SOURCES, APPROVAL_OUTCOME_DROPPED_SOURCE, isApprovalOutcome, createApprovalOutcomeDropNotifier } from '../gateway/approval-outcome-sources.js'
11
12
  import type { InboundMessage } from '../gateway/ipc-protocol.js'
12
13
  import { ObligationLedger } from '../gateway/obligation-ledger.js'
13
14
  import { makeRepresentRedeliveryGuard } from '../gateway/represent-delivery-guard.js'
@@ -171,7 +172,11 @@ describe('pending-inbound-buffer', () => {
171
172
  const buf = createPendingInboundBuffer({ capPerAgent: 1, log: (l) => logs.push(l) })
172
173
  buf.push('a', inbound('m1', 1))
173
174
  buf.push('a', inbound('m2', 2)) // evicts m1
174
- expect(logs.some((l) => l.includes('cap=1') && l.includes('dropped oldest'))).toBe(true)
175
+ // PR D: victim selection is tiered, so the line names the index + tier
176
+ // rather than claiming "oldest" (it is the oldest of its tier).
177
+ expect(
178
+ logs.some((l) => l.includes('cap=1') && l.includes('dropped entry idx=0 reason=non-outcome')),
179
+ ).toBe(true)
175
180
  expect(logs.some((l) => l.includes('m1'))).toBe(true)
176
181
  })
177
182
 
@@ -849,3 +854,439 @@ describe('redeliverBufferedInbound — beforeRedeliver fail-open on throw', () =
849
854
  expect(buf.depth('a')).toBe(0) // nothing stranded
850
855
  })
851
856
  })
857
+
858
+ /**
859
+ * PR D — cap eviction must not pick an APPROVAL OUTCOME while anything else is
860
+ * droppable. The victim used to be an unconditional `q.shift()`, and the oldest
861
+ * entry is exactly the one most likely to be a synthetic approval outcome that
862
+ * has been waiting through an entire turn. An ordinary chat message is
863
+ * resendable; a `vault_grant_approved` is not — the operator's tap already
864
+ * happened, so "please resend" is meaningless and the agent blocks forever.
865
+ */
866
+ describe('pending-inbound-buffer — approval-outcome eviction protection (PR D)', () => {
867
+ /** A button tap: NO meta.source at all, only meta.button_callback. */
868
+ function tap(ts: number): InboundMessage {
869
+ return {
870
+ type: 'inbound',
871
+ chatId: 'c1',
872
+ messageId: ts,
873
+ user: 'alice',
874
+ userId: 42,
875
+ ts,
876
+ text: '[user tapped button: Approve]',
877
+ meta: { button_callback: 'true', button_callback_data: 'ag:ok', button_text: 'Approve' },
878
+ }
879
+ }
880
+
881
+ it('an outcome at the HEAD survives an overflow of ordinary messages', () => {
882
+ const buf = createPendingInboundBuffer({ log: () => {} }) // cap 32
883
+ buf.push('a', inbound('vault_grant_approved', 1))
884
+ for (let i = 0; i < 31; i++) buf.push('a', userMsg({ text: `m${i}`, ts: 100 + i }))
885
+ buf.push('a', userMsg({ text: 'overflow', ts: 999 }))
886
+ const drained = buf.drain('a')
887
+ expect(drained.some((m) => m.meta?.source === 'vault_grant_approved')).toBe(true)
888
+ // The oldest ORDINARY message went instead, and nothing else was lost.
889
+ expect(drained).toHaveLength(32)
890
+ expect(drained.map((m) => m.text)).not.toContain('m0')
891
+ expect(drained.map((m) => m.text)).toContain('overflow')
892
+ })
893
+
894
+ it('a sustained ordinary-message burst never evicts the buffered outcome', () => {
895
+ const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
896
+ buf.push('a', inbound('secret_provided', 1))
897
+ for (let i = 0; i < 40; i++) buf.push('a', userMsg({ text: `u${i}`, ts: 100 + i }))
898
+ const drained = buf.drain('a')
899
+ expect(drained[0]?.meta?.source).toBe('secret_provided')
900
+ expect(drained.map((m) => m.text).slice(1)).toEqual(['u37', 'u38', 'u39'])
901
+ })
902
+
903
+ it('a button tap is protected too — it carries NO meta.source', () => {
904
+ const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
905
+ buf.push('a', tap(1))
906
+ buf.push('a', userMsg({ text: 'u1', ts: 2 }))
907
+ buf.push('a', userMsg({ text: 'u2', ts: 3 }))
908
+ buf.push('a', userMsg({ text: 'u3', ts: 4 }))
909
+ const drained = buf.drain('a')
910
+ expect(drained[0]?.meta?.button_callback).toBe('true')
911
+ expect(drained.map((m) => m.text).slice(1)).toEqual(['u2', 'u3'])
912
+ })
913
+
914
+ it('survivors keep insertion order after a mid-queue eviction (FIFO contract)', () => {
915
+ const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
916
+ buf.push('a', inbound('vault_grant_approved', 1))
917
+ buf.push('a', userMsg({ text: 'u1', ts: 2 }))
918
+ buf.push('a', userMsg({ text: 'u2', ts: 3 }))
919
+ buf.push('a', userMsg({ text: 'u3', ts: 4 }))
920
+ buf.push('a', userMsg({ text: 'u4', ts: 5 })) // evicts u1 (index 1), not the head
921
+ expect(buf.drain('a').map((m) => m.meta?.source ?? m.text)).toEqual([
922
+ 'vault_grant_approved', 'u2', 'u3', 'u4',
923
+ ])
924
+ })
925
+
926
+ it('onEvict (not onEvictCritical) fires for an ordinary victim', () => {
927
+ const ordinary: InboundMessage[] = []
928
+ const critical: InboundMessage[] = []
929
+ const buf = createPendingInboundBuffer({
930
+ capPerAgent: 2,
931
+ log: () => {},
932
+ onEvict: (_a, m) => ordinary.push(m),
933
+ onEvictCritical: (_a, m) => critical.push(m),
934
+ })
935
+ buf.push('a', inbound('vault_grant_approved', 1))
936
+ buf.push('a', userMsg({ text: 'u1', ts: 2 }))
937
+ buf.push('a', userMsg({ text: 'u2', ts: 3 }))
938
+ expect(ordinary.map((m) => m.text)).toEqual(['u1'])
939
+ expect(critical).toHaveLength(0)
940
+ })
941
+
942
+ it('all-outcomes fallback: drops the oldest and fires onEvictCritical, NOT onEvict', () => {
943
+ const ordinary: InboundMessage[] = []
944
+ const critical: { agent: string; msg: InboundMessage }[] = []
945
+ const buf = createPendingInboundBuffer({
946
+ capPerAgent: 3,
947
+ log: () => {},
948
+ now: () => 1000, // all three below are well inside the protection window
949
+ onEvict: (_a, m) => ordinary.push(m),
950
+ onEvictCritical: (agent, msg) => critical.push({ agent, msg }),
951
+ })
952
+ buf.push('a', inbound('vault_grant_approved', 900))
953
+ buf.push('a', inbound('secret_provided', 950))
954
+ buf.push('a', inbound('skill_proposal_apply', 980))
955
+ expect(critical).toHaveLength(0)
956
+ buf.push('a', inbound('mental_model_proposal_applied', 999))
957
+ expect(ordinary).toHaveLength(0)
958
+ expect(critical).toHaveLength(1)
959
+ expect(critical[0]!.agent).toBe('a')
960
+ expect(critical[0]!.msg.meta?.source).toBe('vault_grant_approved')
961
+ expect(critical[0]!.msg.ts).toBe(900)
962
+ // The newest outcome is in and the other two are intact.
963
+ expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
964
+ 'secret_provided', 'skill_proposal_apply', 'mental_model_proposal_applied',
965
+ ])
966
+ })
967
+
968
+ it('a STALE outcome is evictable — protection is bounded, so it cannot pin the buffer', () => {
969
+ const critical: InboundMessage[] = []
970
+ const nowMs = 100 * 60 * 1000
971
+ const buf = createPendingInboundBuffer({
972
+ capPerAgent: 2,
973
+ log: () => {},
974
+ now: () => nowMs,
975
+ onEvictCritical: (_a, m) => critical.push(m),
976
+ })
977
+ // Stale: older than the 15-min protection window (already spool-escalated).
978
+ buf.push('a', inbound('vault_grant_timeout', nowMs - 20 * 60 * 1000))
979
+ // Fresh.
980
+ buf.push('a', inbound('secret_provided', nowMs - 1000))
981
+ buf.push('a', inbound('vault_grant_approved', nowMs))
982
+ expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_timeout'])
983
+ // The FRESH outcomes both survived — staleness, not arrival order, decided it.
984
+ expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
985
+ 'secret_provided', 'vault_grant_approved',
986
+ ])
987
+ })
988
+
989
+ it('APPROVAL_OUTCOME_PROTECTION_MS matches the spool escalation window', () => {
990
+ expect(APPROVAL_OUTCOME_PROTECTION_MS).toBe(15 * 60 * 1000)
991
+ })
992
+
993
+ it('cap of 1 with a single fresh outcome: evicts it and reports critical', () => {
994
+ const critical: InboundMessage[] = []
995
+ const buf = createPendingInboundBuffer({
996
+ capPerAgent: 1,
997
+ log: () => {},
998
+ now: () => 1000,
999
+ onEvictCritical: (_a, m) => critical.push(m),
1000
+ })
1001
+ buf.push('a', inbound('vault_grant_approved', 900))
1002
+ buf.push('a', inbound('secret_provided', 950))
1003
+ expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_approved'])
1004
+ expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
1005
+ })
1006
+
1007
+ it('a throwing onEvictCritical never breaks the push hot path', () => {
1008
+ const buf = createPendingInboundBuffer({
1009
+ capPerAgent: 1,
1010
+ log: () => {},
1011
+ onEvictCritical: () => { throw new Error('notice failed') },
1012
+ })
1013
+ buf.push('a', inbound('vault_grant_approved', 1))
1014
+ expect(() => buf.push('a', inbound('secret_provided', 2))).not.toThrow()
1015
+ expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
1016
+ })
1017
+
1018
+ describe('selectEvictionVictim tiers', () => {
1019
+ const out = (ts: number) => inbound('vault_grant_approved', ts)
1020
+ const ord = (ts: number) => userMsg({ text: `u${ts}`, ts })
1021
+
1022
+ it('empty queue is safe (index 0, no crash at the call site)', () => {
1023
+ expect(selectEvictionVictim([], 0)).toEqual({ index: 0, reason: 'all-outcomes' })
1024
+ })
1025
+
1026
+ it('picks the FIRST non-outcome, not merely any non-outcome', () => {
1027
+ expect(selectEvictionVictim([out(1), out(2), ord(3), ord(4)], 5)).toEqual({
1028
+ index: 2, reason: 'non-outcome',
1029
+ })
1030
+ })
1031
+
1032
+ it('prefers an ordinary message over a STALE outcome', () => {
1033
+ const nowMs = 60 * 60 * 1000
1034
+ expect(selectEvictionVictim([out(1), ord(nowMs)], nowMs)).toEqual({
1035
+ index: 1, reason: 'non-outcome',
1036
+ })
1037
+ })
1038
+
1039
+ // Tier 2 vs tier 3. These two tiers agree on the victim whenever the queue
1040
+ // is in `ts`-ascending order (the oldest entry is then also the stalest),
1041
+ // so an in-order fixture CANNOT tell them apart — disabling the age bound
1042
+ // entirely leaves such a test green (verified by mutation). The
1043
+ // discriminating case is a queue whose head is FRESH and whose later entry
1044
+ // is STALE.
1045
+ it('prefers a STALE outcome over a fresher one queued ahead of it', () => {
1046
+ const nowMs = 100 * 60 * 1000
1047
+ const fresh = out(nowMs)
1048
+ const stale = out(nowMs - 20 * 60 * 1000) // past the 15-min window
1049
+ expect(selectEvictionVictim([fresh, stale], nowMs)).toEqual({
1050
+ index: 1, reason: 'stale-outcome',
1051
+ })
1052
+ })
1053
+
1054
+ // The `reason` is not cosmetic: it is what the eviction log line reports,
1055
+ // and it is the only signal distinguishing "dropped an outcome the spool
1056
+ // already escalated" from "dropped a live one because there was nothing
1057
+ // else". Pin it in the ordinary in-order shape too.
1058
+ it('reports stale-outcome (not all-outcomes) when the victim is past its window', () => {
1059
+ const nowMs = 100 * 60 * 1000
1060
+ expect(
1061
+ selectEvictionVictim([out(nowMs - 20 * 60 * 1000), out(nowMs)], nowMs).reason,
1062
+ ).toBe('stale-outcome')
1063
+ })
1064
+
1065
+ // The bound must actually be a bound: an outcome one millisecond inside
1066
+ // the window is still protected, so tier 3 is what fires.
1067
+ it('an outcome just INSIDE the window is not stale — falls through to tier 3', () => {
1068
+ const nowMs = 100 * 60 * 1000
1069
+ const justInside = out(nowMs - (APPROVAL_OUTCOME_PROTECTION_MS - 1))
1070
+ expect(selectEvictionVictim([out(nowMs), justInside], nowMs)).toEqual({
1071
+ index: 0, reason: 'all-outcomes',
1072
+ })
1073
+ })
1074
+ })
1075
+ })
1076
+
1077
+ /**
1078
+ * PR D — the notifier the gateway wires to `onEvictCritical`. This drives the
1079
+ * REAL production factory (gateway.ts calls exactly this), so the re-entrancy
1080
+ * and termination properties are pinned where they live, not in a test copy.
1081
+ */
1082
+ describe('approval-outcome drop notifier (PR D)', () => {
1083
+ /** Run every queued microtask to quiescence. */
1084
+ const settle = async (): Promise<void> => {
1085
+ for (let i = 0; i < 50; i++) await Promise.resolve()
1086
+ }
1087
+
1088
+ it('the approval_outcome_dropped notice lands in the buffer and names the source', async () => {
1089
+ // Wire it exactly as gateway.ts does: the buffer's onEvictCritical calls
1090
+ // the notifier, and the notifier pushes back into the same buffer.
1091
+ const buf = createPendingInboundBuffer({
1092
+ capPerAgent: 3,
1093
+ log: () => {},
1094
+ now: () => 1000,
1095
+ onEvictCritical: (a, m) => notifier(a, m),
1096
+ })
1097
+ const notifier = createApprovalOutcomeDropNotifier({
1098
+ push: (agent, msg) => { buf.push(agent, msg) },
1099
+ log: () => {},
1100
+ })
1101
+ buf.push('a', inbound('vault_grant_approved', 900))
1102
+ buf.push('a', inbound('secret_provided', 950))
1103
+ buf.push('a', inbound('skill_proposal_apply', 960))
1104
+ buf.push('a', inbound('mental_model_proposal_applied', 970)) // critical evict
1105
+ // Nothing enqueued synchronously — the notice must NOT ride the push frame.
1106
+ expect(buf.depth('a')).toBe(3)
1107
+ await settle()
1108
+ const drained = buf.drain('a')
1109
+ const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
1110
+ expect(notice).toBeDefined()
1111
+ // The notice push itself ran against a queue still full of outcomes, so it
1112
+ // evicted one more. The surviving notice must name BOTH — an earlier notice
1113
+ // naming only the first drop is exactly what the next overflow evicts.
1114
+ expect(notice!.meta?.dropped_sources).toBe('vault_grant_approved, secret_provided')
1115
+ expect(notice!.text).toContain('vault_grant_approved')
1116
+ // Every outcome that was NOT dropped is still buffered.
1117
+ expect(drained.map((m) => m.meta?.source)).toEqual([
1118
+ 'skill_proposal_apply', 'mental_model_proposal_applied', 'approval_outcome_dropped',
1119
+ ])
1120
+ })
1121
+
1122
+ // The guarantee is "a dropped approval outcome is never silent". If it
1123
+ // depended on each construction site passing a callback it would be
1124
+ // discipline, not a mechanism — so the notifier defaults ON and this pins it.
1125
+ it('is wired BY DEFAULT — a bare buffer still enqueues the notice', async () => {
1126
+ const buf = createPendingInboundBuffer({
1127
+ capPerAgent: 3,
1128
+ log: () => {},
1129
+ now: () => 1000,
1130
+ // NO onEvictCritical passed.
1131
+ })
1132
+ buf.push('a', inbound('vault_grant_approved', 900))
1133
+ buf.push('a', inbound('secret_provided', 950))
1134
+ buf.push('a', inbound('skill_proposal_apply', 960))
1135
+ buf.push('a', inbound('vault_grant_denied', 970))
1136
+ await settle()
1137
+ const drained = buf.drain('a')
1138
+ const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
1139
+ expect(notice).toBeDefined()
1140
+ expect(notice!.meta?.dropped_sources).toContain('vault_grant_approved')
1141
+ })
1142
+
1143
+ it('an explicit no-op onEvictCritical opts out of the notice', async () => {
1144
+ const buf = createPendingInboundBuffer({
1145
+ capPerAgent: 3,
1146
+ log: () => {},
1147
+ now: () => 1000,
1148
+ onEvictCritical: () => {},
1149
+ })
1150
+ for (let i = 0; i < 6; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
1151
+ await settle()
1152
+ expect(buf.drain('a').some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(false)
1153
+ })
1154
+
1155
+ it('a burst of critical evictions terminates and does not spin the microtask queue', async () => {
1156
+ let pushes = 0
1157
+ const buf = createPendingInboundBuffer({
1158
+ capPerAgent: 3,
1159
+ log: () => {},
1160
+ now: () => 1000,
1161
+ onEvictCritical: (a, m) => notifier(a, m),
1162
+ })
1163
+ const notifier = createApprovalOutcomeDropNotifier({
1164
+ push: (agent, msg) => { pushes++; buf.push(agent, msg) },
1165
+ log: () => {},
1166
+ })
1167
+ for (let i = 0; i < 10; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
1168
+ await settle()
1169
+ // Bounded: at most one notice per burst plus the follow-up hop, never one
1170
+ // notice per evicted outcome (that is the recursion this guards).
1171
+ expect(pushes).toBeGreaterThan(0)
1172
+ expect(pushes).toBeLessThanOrEqual(3)
1173
+ // A notice is resident, and it is NOT itself an approval outcome — which is
1174
+ // what makes it the preferred victim next time and bounds the chain.
1175
+ const drained = buf.drain('a')
1176
+ expect(drained.some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(true)
1177
+ })
1178
+
1179
+ it('a burst coalesces into ONE notice naming every dropped source', async () => {
1180
+ const notices: InboundMessage[] = []
1181
+ const notifier = createApprovalOutcomeDropNotifier({
1182
+ push: (_agent, msg) => { notices.push(msg) },
1183
+ log: () => {},
1184
+ })
1185
+ notifier('a', inbound('vault_grant_approved', 1))
1186
+ notifier('a', inbound('secret_declined', 2))
1187
+ notifier('a', inbound('skill_proposal_apply', 3))
1188
+ expect(notices).toHaveLength(0) // deferred, never same-frame
1189
+ await settle()
1190
+ expect(notices).toHaveLength(1)
1191
+ expect(notices[0]!.meta?.dropped_sources).toBe(
1192
+ 'vault_grant_approved, secret_declined, skill_proposal_apply',
1193
+ )
1194
+ expect(notices[0]!.meta?.dropped_count).toBe('3')
1195
+ })
1196
+
1197
+ it('a button tap victim is labelled button_callback, not "-"', async () => {
1198
+ const notices: InboundMessage[] = []
1199
+ const notifier = createApprovalOutcomeDropNotifier({
1200
+ push: (_agent, msg) => { notices.push(msg) },
1201
+ log: () => {},
1202
+ })
1203
+ notifier('a', {
1204
+ type: 'inbound', chatId: 'c1', messageId: 1, user: 'alice', userId: 42, ts: 1,
1205
+ text: '[user tapped button: Approve]', meta: { button_callback: 'true' },
1206
+ })
1207
+ await settle()
1208
+ expect(notices[0]!.meta?.dropped_sources).toBe('button_callback')
1209
+ })
1210
+ })
1211
+
1212
+ describe('isApprovalOutcome (PR D)', () => {
1213
+ const withMeta = (meta: Record<string, string> | undefined): InboundMessage => ({
1214
+ type: 'inbound', chatId: 'c1', messageId: 1, user: 'u', userId: 1, ts: 1, text: 't',
1215
+ ...(meta != null ? { meta } : {}),
1216
+ } as InboundMessage)
1217
+
1218
+ it('true for every registered source', () => {
1219
+ for (const s of APPROVAL_OUTCOME_SOURCES) {
1220
+ expect(isApprovalOutcome(withMeta({ source: s }))).toBe(true)
1221
+ }
1222
+ })
1223
+
1224
+ it('true for a button tap with no source at all', () => {
1225
+ expect(isApprovalOutcome(withMeta({ button_callback: 'true' }))).toBe(true)
1226
+ })
1227
+
1228
+ it('false for ordinary messages and non-outcome system sources', () => {
1229
+ expect(isApprovalOutcome(withMeta({}))).toBe(false)
1230
+ expect(isApprovalOutcome(withMeta(undefined))).toBe(false)
1231
+ for (const s of [
1232
+ 'missed_approval_retry', 'obligation_represent', 'subagent_handback',
1233
+ 'subagent_progress', 'resume_interrupted', 'warmup', 'reaction', 'cron',
1234
+ 'approval_outcome_dropped',
1235
+ ]) {
1236
+ expect(isApprovalOutcome(withMeta({ source: s }))).toBe(false)
1237
+ }
1238
+ })
1239
+
1240
+ // #4664. `eval_case_suppressed` is the one protected member that is not a
1241
+ // verdict card, and it is the one with the least margin for error: no sweep
1242
+ // regenerates it (it fires exactly once, at suppression time, and posts no
1243
+ // card), so an eviction is a PERMANENT block on the agent's turn. Asserted
1244
+ // through the buffer, not just the set, so the pin fails on the behaviour
1245
+ // rather than on a membership literal.
1246
+ it('a buffered eval_case_suppressed notice survives a cap overflow', () => {
1247
+ const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
1248
+ buf.push('a', inbound('eval_case_suppressed', 1))
1249
+ buf.push('a', userMsg({ text: 'first', ts: 2 }))
1250
+ buf.push('a', userMsg({ text: 'second', ts: 3 }))
1251
+ // Overflow: the resendable user message must go, not the one-shot notice.
1252
+ buf.push('a', userMsg({ text: 'third', ts: 4 }))
1253
+ const drained = buf.drain('a')
1254
+ expect(drained.map((m) => m.meta?.source ?? m.text)).toEqual([
1255
+ 'eval_case_suppressed', 'second', 'third',
1256
+ ])
1257
+ })
1258
+
1259
+ it('the dropped-notice source is deliberately NOT protected', () => {
1260
+ expect(APPROVAL_OUTCOME_SOURCES.has(APPROVAL_OUTCOME_DROPPED_SOURCE)).toBe(false)
1261
+ })
1262
+
1263
+ // Drift pin on the registry itself. `Object.freeze` on a Set does NOT block
1264
+ // `.add` (Set data lives in internal slots), so asserting `isFrozen` would be
1265
+ // a placebo — immutability is enforced at compile time by the `ReadonlySet`
1266
+ // type. What IS worth pinning is membership: silently dropping a source here
1267
+ // makes that verdict class evictable again with no test going red.
1268
+ it('the registry contents are exactly the audited set', () => {
1269
+ expect([...APPROVAL_OUTCOME_SOURCES].sort()).toEqual([
1270
+ 'eval_case_applied',
1271
+ 'eval_case_apply_failed',
1272
+ 'eval_case_rejected',
1273
+ 'eval_case_suppressed',
1274
+ 'mental_model_proposal_applied',
1275
+ 'mental_model_proposal_denied',
1276
+ 'mental_model_proposal_failed',
1277
+ 'mental_model_propose_timeout',
1278
+ 'secret_declined',
1279
+ 'secret_provide_failed',
1280
+ 'secret_provided',
1281
+ 'secret_request_timeout',
1282
+ 'skill_proposal_apply',
1283
+ 'vault_grant_approved',
1284
+ 'vault_grant_denied',
1285
+ 'vault_grant_timeout',
1286
+ 'vault_save_completed',
1287
+ 'vault_save_discarded',
1288
+ 'vault_save_failed',
1289
+ 'vault_save_timeout',
1290
+ ])
1291
+ })
1292
+ })
@@ -96,6 +96,20 @@ describe('inbound source classification (F1 durability)', () => {
96
96
  expect(stampsHandbackMarker('reaction')).toBe(false)
97
97
  expect(stampsHandbackMarker('resume_interrupted')).toBe(false)
98
98
  expect(stampsHandbackMarker('vault_grant_approved')).toBe(false)
99
+ // #4662 — the eval-case outcome inbounds. The exhaustiveness test below pins
100
+ // only their PRESENCE in the registry; these pin the VALUE, which is the part
101
+ // with runtime consequence: classified `true` (or left to the fail-safe
102
+ // default) they would hold the content gate chat-wide for 60 s after every
103
+ // operator tap on an eval-case card.
104
+ expect(stampsHandbackMarker('eval_case_applied')).toBe(false)
105
+ expect(stampsHandbackMarker('eval_case_rejected')).toBe(false)
106
+ expect(stampsHandbackMarker('eval_case_apply_failed')).toBe(false)
107
+ // #4664 — the suppressed-proposal notice. The exhaustiveness test below pins
108
+ // only its PRESENCE; this pins the VALUE, which is the part with runtime
109
+ // consequence: classified `true` (or left to the fail-safe default) it would
110
+ // hold the content gate chat-wide for 60 s every time a duplicate eval case
111
+ // is declined.
112
+ expect(stampsHandbackMarker('eval_case_suppressed')).toBe(false)
99
113
  })
100
114
 
101
115
  it('a normal user inbound (no meta.source) never stamps', () => {