switchroom 0.21.7 → 0.21.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/tmp-reaper.sh +234 -0
- package/dist/agent-scheduler/index.js +1 -1
- package/dist/auth-broker/index.js +2 -2
- package/dist/cli/notion-write-pretool.mjs +1 -1
- package/dist/cli/switchroom.js +3421 -2744
- package/dist/host-control/main.js +177 -13
- package/dist/vault/approvals/kernel-server.js +2 -2
- package/dist/vault/broker/server.js +2 -2
- package/package.json +5 -4
- package/profiles/_base/start.sh.hbs +115 -0
- package/profiles/_shared/local-time.md.hbs +6 -0
- package/profiles/default/CLAUDE.md.hbs +0 -12
- package/telegram-plugin/dist/gateway/gateway.js +1017 -465
- package/telegram-plugin/gateway/agent-process-liveness.ts +558 -0
- package/telegram-plugin/gateway/approval-hold.ts +32 -1
- package/telegram-plugin/gateway/approval-outcome-sources.ts +274 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +21 -9
- package/telegram-plugin/gateway/callback-query-handlers.ts +87 -15
- package/telegram-plugin/gateway/eval-case-proposal-inbound-builders.ts +197 -0
- package/telegram-plugin/gateway/gateway.ts +12 -10
- package/telegram-plugin/gateway/pending-inbound-buffer.ts +167 -11
- package/telegram-plugin/gateway/self-improve-proposal-wiring.test.ts +333 -0
- package/telegram-plugin/gateway/self-improve-proposal-wiring.ts +152 -3
- package/telegram-plugin/gateway/subagent-handback-marker.ts +19 -0
- package/telegram-plugin/tests/agent-process-liveness.test.ts +406 -0
- package/telegram-plugin/tests/approval-hold-record.test.ts +21 -8
- package/telegram-plugin/tests/boot-resume-gateway-only-respawn.test.ts +752 -0
- package/telegram-plugin/tests/boot-resume-guard-wiring.test.ts +203 -0
- package/telegram-plugin/tests/callback-query-handlers.test.ts +143 -1
- package/telegram-plugin/tests/eval-case-proposal-inbound-builders.test.ts +144 -0
- package/telegram-plugin/tests/hermes-messages-paging.test.ts +149 -0
- package/telegram-plugin/tests/hermes-session-search.test.ts +146 -0
- package/telegram-plugin/tests/pending-inbound-buffer.test.ts +443 -2
- package/telegram-plugin/tests/subagent-handback-marker.test.ts +14 -0
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `GET /api/sessions/search` — that it actually SEARCHES.
|
|
3
|
+
*
|
|
4
|
+
* `hermes-rest-parity.test.ts` can only assert the contract floor here
|
|
5
|
+
* (`Array.isArray(results)`), because it points `SWITCHROOM_AGENTS_DIR` at an
|
|
6
|
+
* empty tmpdir: with no turns anywhere, "found the right rows", "found nothing
|
|
7
|
+
* for a miss" and "hardcoded `{results: []}`" are indistinguishable. A
|
|
8
|
+
* hardcoded empty list would pass that floor forever, and an always-200 empty
|
|
9
|
+
* body is the exact bug class the desktop cannot detect —
|
|
10
|
+
* `isEndpointMissingError` only matches 404s.
|
|
11
|
+
*
|
|
12
|
+
* So this file seeds a real turns registry (`bun:sqlite`, hence a bun test —
|
|
13
|
+
* vitest excludes it; it lives under telegram-plugin/tests/ because that is the
|
|
14
|
+
* only tree the bun CI job walks) and asserts the discriminating outcomes:
|
|
15
|
+
* a query that should hit, hits; a query that should miss, misses.
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { test, expect, beforeEach, afterEach } from "bun:test";
|
|
19
|
+
import { mkdtempSync, rmSync, mkdirSync } from "node:fs";
|
|
20
|
+
import { tmpdir } from "node:os";
|
|
21
|
+
import { join } from "node:path";
|
|
22
|
+
import { openTurnsDb } from "../registry/turns-schema.js";
|
|
23
|
+
import { handleHermesRest, type HermesRestResult } from "../../src/web/hermes-adapter.js";
|
|
24
|
+
import type { SwitchroomConfig } from "../../src/config/schema.js";
|
|
25
|
+
|
|
26
|
+
const CONFIG = {
|
|
27
|
+
agents: { alpha: { model: "sonnet" }, beta: {} },
|
|
28
|
+
} as unknown as SwitchroomConfig;
|
|
29
|
+
|
|
30
|
+
let agentsDir = "";
|
|
31
|
+
|
|
32
|
+
function seed(agent: string, rows: { user: string; assistant: string }[]) {
|
|
33
|
+
const dir = join(agentsDir, agent);
|
|
34
|
+
mkdirSync(dir, { recursive: true });
|
|
35
|
+
const db = openTurnsDb(dir);
|
|
36
|
+
try {
|
|
37
|
+
rows.forEach((r, i) => {
|
|
38
|
+
const ts = 1_700_000_000_000 + i * 1000;
|
|
39
|
+
db.prepare(
|
|
40
|
+
`INSERT INTO turns
|
|
41
|
+
(turn_key, chat_id, thread_id, started_at, ended_at, ended_via,
|
|
42
|
+
user_prompt_preview, assistant_reply_preview, created_at, updated_at)
|
|
43
|
+
VALUES (?, ?, NULL, ?, ?, 'stop', ?, ?, ?, ?)`,
|
|
44
|
+
).run(`${agent}-t${i}`, "chat", ts, ts + 500, r.user, r.assistant, ts, ts);
|
|
45
|
+
});
|
|
46
|
+
} finally {
|
|
47
|
+
db.close();
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
type Result = {
|
|
52
|
+
session_id: string;
|
|
53
|
+
lineage_root: string;
|
|
54
|
+
snippet: string;
|
|
55
|
+
role: string | null;
|
|
56
|
+
source: string | null;
|
|
57
|
+
model: string | null;
|
|
58
|
+
session_started: number | null;
|
|
59
|
+
};
|
|
60
|
+
|
|
61
|
+
async function search(query: string): Promise<Result[]> {
|
|
62
|
+
const res = (await handleHermesRest(
|
|
63
|
+
"GET",
|
|
64
|
+
"/api/sessions/search",
|
|
65
|
+
CONFIG,
|
|
66
|
+
query,
|
|
67
|
+
)) as HermesRestResult;
|
|
68
|
+
expect(res).not.toBeNull();
|
|
69
|
+
// The pre-fix behaviour: the /api/sessions/:id regex matched "search" as an
|
|
70
|
+
// id and answered 404 {error:'Unknown session'}, which threw the panel.
|
|
71
|
+
expect(res.status).toBe(200);
|
|
72
|
+
return (res.body as { results: Result[] }).results;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
beforeEach(() => {
|
|
76
|
+
agentsDir = mkdtempSync(join(tmpdir(), "sr-hermes-search-"));
|
|
77
|
+
process.env.SWITCHROOM_AGENTS_DIR = agentsDir;
|
|
78
|
+
});
|
|
79
|
+
|
|
80
|
+
afterEach(() => {
|
|
81
|
+
rmSync(agentsDir, { recursive: true, force: true });
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test("a session-id match comes back, with role null and the id as its own lineage root", async () => {
|
|
85
|
+
const results = await search("?q=alph");
|
|
86
|
+
expect(results.map((r) => r.session_id)).toEqual(["alpha"]);
|
|
87
|
+
expect(results[0].role).toBeNull();
|
|
88
|
+
expect(results[0].lineage_root).toBe("alpha");
|
|
89
|
+
});
|
|
90
|
+
|
|
91
|
+
test("a message-content match comes back tagged with the role it matched on", async () => {
|
|
92
|
+
seed("alpha", [{ user: "deploy the gateway", assistant: "done" }]);
|
|
93
|
+
seed("beta", [{ user: "unrelated", assistant: "also unrelated" }]);
|
|
94
|
+
|
|
95
|
+
const onUser = await search("?q=deploy");
|
|
96
|
+
expect(onUser.map((r) => r.session_id)).toEqual(["alpha"]);
|
|
97
|
+
expect(onUser[0].role).toBe("user");
|
|
98
|
+
expect(onUser[0].snippet).toBe("deploy the gateway");
|
|
99
|
+
|
|
100
|
+
const onAssistant = await search("?q=also%20unrelated");
|
|
101
|
+
expect(onAssistant.map((r) => r.session_id)).toEqual(["beta"]);
|
|
102
|
+
expect(onAssistant[0].role).toBe("assistant");
|
|
103
|
+
expect(onAssistant[0].snippet).toBe("also unrelated");
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
test("a query that matches nothing returns nothing — not the whole fleet", async () => {
|
|
107
|
+
seed("alpha", [{ user: "deploy the gateway", assistant: "done" }]);
|
|
108
|
+
// The discriminating case: a stub returning every session, and a stub
|
|
109
|
+
// returning none, are told apart only by running both this and the hit above.
|
|
110
|
+
expect(await search("?q=zzz-no-such-token")).toEqual([]);
|
|
111
|
+
});
|
|
112
|
+
|
|
113
|
+
test("an empty or whitespace query is short-circuited, not treated as match-everything", async () => {
|
|
114
|
+
seed("alpha", [{ user: "deploy", assistant: "done" }]);
|
|
115
|
+
expect(await search("?q=")).toEqual([]);
|
|
116
|
+
expect(await search("?q=%20%20")).toEqual([]);
|
|
117
|
+
expect(await search("")).toEqual([]);
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
test("one row per session even when several turns match", async () => {
|
|
121
|
+
seed("alpha", [
|
|
122
|
+
{ user: "deploy one", assistant: "ok" },
|
|
123
|
+
{ user: "deploy two", assistant: "ok" },
|
|
124
|
+
{ user: "deploy three", assistant: "ok" },
|
|
125
|
+
]);
|
|
126
|
+
const results = await search("?q=deploy");
|
|
127
|
+
expect(results.length).toBe(1);
|
|
128
|
+
expect(results[0].session_id).toBe("alpha");
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
test("?limit= caps the result set", async () => {
|
|
132
|
+
seed("alpha", [{ user: "shared token", assistant: "ok" }]);
|
|
133
|
+
seed("beta", [{ user: "shared token", assistant: "ok" }]);
|
|
134
|
+
expect((await search("?q=shared&limit=1")).length).toBe(1);
|
|
135
|
+
expect((await search("?q=shared")).length).toBe(2);
|
|
136
|
+
});
|
|
137
|
+
|
|
138
|
+
test("id matches are ordered ahead of content matches", async () => {
|
|
139
|
+
// beta matches by id; alpha only by content. Upstream surfaces direct
|
|
140
|
+
// session-id hits first (sessions.py:169-383) and the panel relies on it.
|
|
141
|
+
seed("alpha", [{ user: "beta is mentioned here", assistant: "ok" }]);
|
|
142
|
+
const results = await search("?q=beta");
|
|
143
|
+
expect(results.map((r) => r.session_id)).toEqual(["beta", "alpha"]);
|
|
144
|
+
expect(results[0].role).toBeNull();
|
|
145
|
+
expect(results[1].role).toBe("user");
|
|
146
|
+
});
|
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { describe, it, expect } from 'vitest'
|
|
10
|
-
import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, DEFAULT_PENDING_INBOUND_CAP } from '../gateway/pending-inbound-buffer.js'
|
|
10
|
+
import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, selectEvictionVictim, DEFAULT_PENDING_INBOUND_CAP, APPROVAL_OUTCOME_PROTECTION_MS } from '../gateway/pending-inbound-buffer.js'
|
|
11
|
+
import { APPROVAL_OUTCOME_SOURCES, APPROVAL_OUTCOME_DROPPED_SOURCE, isApprovalOutcome, createApprovalOutcomeDropNotifier } from '../gateway/approval-outcome-sources.js'
|
|
11
12
|
import type { InboundMessage } from '../gateway/ipc-protocol.js'
|
|
12
13
|
import { ObligationLedger } from '../gateway/obligation-ledger.js'
|
|
13
14
|
import { makeRepresentRedeliveryGuard } from '../gateway/represent-delivery-guard.js'
|
|
@@ -171,7 +172,11 @@ describe('pending-inbound-buffer', () => {
|
|
|
171
172
|
const buf = createPendingInboundBuffer({ capPerAgent: 1, log: (l) => logs.push(l) })
|
|
172
173
|
buf.push('a', inbound('m1', 1))
|
|
173
174
|
buf.push('a', inbound('m2', 2)) // evicts m1
|
|
174
|
-
|
|
175
|
+
// PR D: victim selection is tiered, so the line names the index + tier
|
|
176
|
+
// rather than claiming "oldest" (it is the oldest of its tier).
|
|
177
|
+
expect(
|
|
178
|
+
logs.some((l) => l.includes('cap=1') && l.includes('dropped entry idx=0 reason=non-outcome')),
|
|
179
|
+
).toBe(true)
|
|
175
180
|
expect(logs.some((l) => l.includes('m1'))).toBe(true)
|
|
176
181
|
})
|
|
177
182
|
|
|
@@ -849,3 +854,439 @@ describe('redeliverBufferedInbound — beforeRedeliver fail-open on throw', () =
|
|
|
849
854
|
expect(buf.depth('a')).toBe(0) // nothing stranded
|
|
850
855
|
})
|
|
851
856
|
})
|
|
857
|
+
|
|
858
|
+
/**
|
|
859
|
+
* PR D — cap eviction must not pick an APPROVAL OUTCOME while anything else is
|
|
860
|
+
* droppable. The victim used to be an unconditional `q.shift()`, and the oldest
|
|
861
|
+
* entry is exactly the one most likely to be a synthetic approval outcome that
|
|
862
|
+
* has been waiting through an entire turn. An ordinary chat message is
|
|
863
|
+
* resendable; a `vault_grant_approved` is not — the operator's tap already
|
|
864
|
+
* happened, so "please resend" is meaningless and the agent blocks forever.
|
|
865
|
+
*/
|
|
866
|
+
describe('pending-inbound-buffer — approval-outcome eviction protection (PR D)', () => {
|
|
867
|
+
/** A button tap: NO meta.source at all, only meta.button_callback. */
|
|
868
|
+
function tap(ts: number): InboundMessage {
|
|
869
|
+
return {
|
|
870
|
+
type: 'inbound',
|
|
871
|
+
chatId: 'c1',
|
|
872
|
+
messageId: ts,
|
|
873
|
+
user: 'alice',
|
|
874
|
+
userId: 42,
|
|
875
|
+
ts,
|
|
876
|
+
text: '[user tapped button: Approve]',
|
|
877
|
+
meta: { button_callback: 'true', button_callback_data: 'ag:ok', button_text: 'Approve' },
|
|
878
|
+
}
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
it('an outcome at the HEAD survives an overflow of ordinary messages', () => {
|
|
882
|
+
const buf = createPendingInboundBuffer({ log: () => {} }) // cap 32
|
|
883
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
884
|
+
for (let i = 0; i < 31; i++) buf.push('a', userMsg({ text: `m${i}`, ts: 100 + i }))
|
|
885
|
+
buf.push('a', userMsg({ text: 'overflow', ts: 999 }))
|
|
886
|
+
const drained = buf.drain('a')
|
|
887
|
+
expect(drained.some((m) => m.meta?.source === 'vault_grant_approved')).toBe(true)
|
|
888
|
+
// The oldest ORDINARY message went instead, and nothing else was lost.
|
|
889
|
+
expect(drained).toHaveLength(32)
|
|
890
|
+
expect(drained.map((m) => m.text)).not.toContain('m0')
|
|
891
|
+
expect(drained.map((m) => m.text)).toContain('overflow')
|
|
892
|
+
})
|
|
893
|
+
|
|
894
|
+
it('a sustained ordinary-message burst never evicts the buffered outcome', () => {
|
|
895
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
|
|
896
|
+
buf.push('a', inbound('secret_provided', 1))
|
|
897
|
+
for (let i = 0; i < 40; i++) buf.push('a', userMsg({ text: `u${i}`, ts: 100 + i }))
|
|
898
|
+
const drained = buf.drain('a')
|
|
899
|
+
expect(drained[0]?.meta?.source).toBe('secret_provided')
|
|
900
|
+
expect(drained.map((m) => m.text).slice(1)).toEqual(['u37', 'u38', 'u39'])
|
|
901
|
+
})
|
|
902
|
+
|
|
903
|
+
it('a button tap is protected too — it carries NO meta.source', () => {
|
|
904
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
|
|
905
|
+
buf.push('a', tap(1))
|
|
906
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
907
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
908
|
+
buf.push('a', userMsg({ text: 'u3', ts: 4 }))
|
|
909
|
+
const drained = buf.drain('a')
|
|
910
|
+
expect(drained[0]?.meta?.button_callback).toBe('true')
|
|
911
|
+
expect(drained.map((m) => m.text).slice(1)).toEqual(['u2', 'u3'])
|
|
912
|
+
})
|
|
913
|
+
|
|
914
|
+
it('survivors keep insertion order after a mid-queue eviction (FIFO contract)', () => {
|
|
915
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
|
|
916
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
917
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
918
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
919
|
+
buf.push('a', userMsg({ text: 'u3', ts: 4 }))
|
|
920
|
+
buf.push('a', userMsg({ text: 'u4', ts: 5 })) // evicts u1 (index 1), not the head
|
|
921
|
+
expect(buf.drain('a').map((m) => m.meta?.source ?? m.text)).toEqual([
|
|
922
|
+
'vault_grant_approved', 'u2', 'u3', 'u4',
|
|
923
|
+
])
|
|
924
|
+
})
|
|
925
|
+
|
|
926
|
+
it('onEvict (not onEvictCritical) fires for an ordinary victim', () => {
|
|
927
|
+
const ordinary: InboundMessage[] = []
|
|
928
|
+
const critical: InboundMessage[] = []
|
|
929
|
+
const buf = createPendingInboundBuffer({
|
|
930
|
+
capPerAgent: 2,
|
|
931
|
+
log: () => {},
|
|
932
|
+
onEvict: (_a, m) => ordinary.push(m),
|
|
933
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
934
|
+
})
|
|
935
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
936
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
937
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
938
|
+
expect(ordinary.map((m) => m.text)).toEqual(['u1'])
|
|
939
|
+
expect(critical).toHaveLength(0)
|
|
940
|
+
})
|
|
941
|
+
|
|
942
|
+
it('all-outcomes fallback: drops the oldest and fires onEvictCritical, NOT onEvict', () => {
|
|
943
|
+
const ordinary: InboundMessage[] = []
|
|
944
|
+
const critical: { agent: string; msg: InboundMessage }[] = []
|
|
945
|
+
const buf = createPendingInboundBuffer({
|
|
946
|
+
capPerAgent: 3,
|
|
947
|
+
log: () => {},
|
|
948
|
+
now: () => 1000, // all three below are well inside the protection window
|
|
949
|
+
onEvict: (_a, m) => ordinary.push(m),
|
|
950
|
+
onEvictCritical: (agent, msg) => critical.push({ agent, msg }),
|
|
951
|
+
})
|
|
952
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
953
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
954
|
+
buf.push('a', inbound('skill_proposal_apply', 980))
|
|
955
|
+
expect(critical).toHaveLength(0)
|
|
956
|
+
buf.push('a', inbound('mental_model_proposal_applied', 999))
|
|
957
|
+
expect(ordinary).toHaveLength(0)
|
|
958
|
+
expect(critical).toHaveLength(1)
|
|
959
|
+
expect(critical[0]!.agent).toBe('a')
|
|
960
|
+
expect(critical[0]!.msg.meta?.source).toBe('vault_grant_approved')
|
|
961
|
+
expect(critical[0]!.msg.ts).toBe(900)
|
|
962
|
+
// The newest outcome is in and the other two are intact.
|
|
963
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
|
|
964
|
+
'secret_provided', 'skill_proposal_apply', 'mental_model_proposal_applied',
|
|
965
|
+
])
|
|
966
|
+
})
|
|
967
|
+
|
|
968
|
+
it('a STALE outcome is evictable — protection is bounded, so it cannot pin the buffer', () => {
|
|
969
|
+
const critical: InboundMessage[] = []
|
|
970
|
+
const nowMs = 100 * 60 * 1000
|
|
971
|
+
const buf = createPendingInboundBuffer({
|
|
972
|
+
capPerAgent: 2,
|
|
973
|
+
log: () => {},
|
|
974
|
+
now: () => nowMs,
|
|
975
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
976
|
+
})
|
|
977
|
+
// Stale: older than the 15-min protection window (already spool-escalated).
|
|
978
|
+
buf.push('a', inbound('vault_grant_timeout', nowMs - 20 * 60 * 1000))
|
|
979
|
+
// Fresh.
|
|
980
|
+
buf.push('a', inbound('secret_provided', nowMs - 1000))
|
|
981
|
+
buf.push('a', inbound('vault_grant_approved', nowMs))
|
|
982
|
+
expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_timeout'])
|
|
983
|
+
// The FRESH outcomes both survived — staleness, not arrival order, decided it.
|
|
984
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
|
|
985
|
+
'secret_provided', 'vault_grant_approved',
|
|
986
|
+
])
|
|
987
|
+
})
|
|
988
|
+
|
|
989
|
+
it('APPROVAL_OUTCOME_PROTECTION_MS matches the spool escalation window', () => {
|
|
990
|
+
expect(APPROVAL_OUTCOME_PROTECTION_MS).toBe(15 * 60 * 1000)
|
|
991
|
+
})
|
|
992
|
+
|
|
993
|
+
it('cap of 1 with a single fresh outcome: evicts it and reports critical', () => {
|
|
994
|
+
const critical: InboundMessage[] = []
|
|
995
|
+
const buf = createPendingInboundBuffer({
|
|
996
|
+
capPerAgent: 1,
|
|
997
|
+
log: () => {},
|
|
998
|
+
now: () => 1000,
|
|
999
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
1000
|
+
})
|
|
1001
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1002
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1003
|
+
expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_approved'])
|
|
1004
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
|
|
1005
|
+
})
|
|
1006
|
+
|
|
1007
|
+
it('a throwing onEvictCritical never breaks the push hot path', () => {
|
|
1008
|
+
const buf = createPendingInboundBuffer({
|
|
1009
|
+
capPerAgent: 1,
|
|
1010
|
+
log: () => {},
|
|
1011
|
+
onEvictCritical: () => { throw new Error('notice failed') },
|
|
1012
|
+
})
|
|
1013
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
1014
|
+
expect(() => buf.push('a', inbound('secret_provided', 2))).not.toThrow()
|
|
1015
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
|
|
1016
|
+
})
|
|
1017
|
+
|
|
1018
|
+
describe('selectEvictionVictim tiers', () => {
|
|
1019
|
+
const out = (ts: number) => inbound('vault_grant_approved', ts)
|
|
1020
|
+
const ord = (ts: number) => userMsg({ text: `u${ts}`, ts })
|
|
1021
|
+
|
|
1022
|
+
it('empty queue is safe (index 0, no crash at the call site)', () => {
|
|
1023
|
+
expect(selectEvictionVictim([], 0)).toEqual({ index: 0, reason: 'all-outcomes' })
|
|
1024
|
+
})
|
|
1025
|
+
|
|
1026
|
+
it('picks the FIRST non-outcome, not merely any non-outcome', () => {
|
|
1027
|
+
expect(selectEvictionVictim([out(1), out(2), ord(3), ord(4)], 5)).toEqual({
|
|
1028
|
+
index: 2, reason: 'non-outcome',
|
|
1029
|
+
})
|
|
1030
|
+
})
|
|
1031
|
+
|
|
1032
|
+
it('prefers an ordinary message over a STALE outcome', () => {
|
|
1033
|
+
const nowMs = 60 * 60 * 1000
|
|
1034
|
+
expect(selectEvictionVictim([out(1), ord(nowMs)], nowMs)).toEqual({
|
|
1035
|
+
index: 1, reason: 'non-outcome',
|
|
1036
|
+
})
|
|
1037
|
+
})
|
|
1038
|
+
|
|
1039
|
+
// Tier 2 vs tier 3. These two tiers agree on the victim whenever the queue
|
|
1040
|
+
// is in `ts`-ascending order (the oldest entry is then also the stalest),
|
|
1041
|
+
// so an in-order fixture CANNOT tell them apart — disabling the age bound
|
|
1042
|
+
// entirely leaves such a test green (verified by mutation). The
|
|
1043
|
+
// discriminating case is a queue whose head is FRESH and whose later entry
|
|
1044
|
+
// is STALE.
|
|
1045
|
+
it('prefers a STALE outcome over a fresher one queued ahead of it', () => {
|
|
1046
|
+
const nowMs = 100 * 60 * 1000
|
|
1047
|
+
const fresh = out(nowMs)
|
|
1048
|
+
const stale = out(nowMs - 20 * 60 * 1000) // past the 15-min window
|
|
1049
|
+
expect(selectEvictionVictim([fresh, stale], nowMs)).toEqual({
|
|
1050
|
+
index: 1, reason: 'stale-outcome',
|
|
1051
|
+
})
|
|
1052
|
+
})
|
|
1053
|
+
|
|
1054
|
+
// The `reason` is not cosmetic: it is what the eviction log line reports,
|
|
1055
|
+
// and it is the only signal distinguishing "dropped an outcome the spool
|
|
1056
|
+
// already escalated" from "dropped a live one because there was nothing
|
|
1057
|
+
// else". Pin it in the ordinary in-order shape too.
|
|
1058
|
+
it('reports stale-outcome (not all-outcomes) when the victim is past its window', () => {
|
|
1059
|
+
const nowMs = 100 * 60 * 1000
|
|
1060
|
+
expect(
|
|
1061
|
+
selectEvictionVictim([out(nowMs - 20 * 60 * 1000), out(nowMs)], nowMs).reason,
|
|
1062
|
+
).toBe('stale-outcome')
|
|
1063
|
+
})
|
|
1064
|
+
|
|
1065
|
+
// The bound must actually be a bound: an outcome one millisecond inside
|
|
1066
|
+
// the window is still protected, so tier 3 is what fires.
|
|
1067
|
+
it('an outcome just INSIDE the window is not stale — falls through to tier 3', () => {
|
|
1068
|
+
const nowMs = 100 * 60 * 1000
|
|
1069
|
+
const justInside = out(nowMs - (APPROVAL_OUTCOME_PROTECTION_MS - 1))
|
|
1070
|
+
expect(selectEvictionVictim([out(nowMs), justInside], nowMs)).toEqual({
|
|
1071
|
+
index: 0, reason: 'all-outcomes',
|
|
1072
|
+
})
|
|
1073
|
+
})
|
|
1074
|
+
})
|
|
1075
|
+
})
|
|
1076
|
+
|
|
1077
|
+
/**
|
|
1078
|
+
* PR D — the notifier the gateway wires to `onEvictCritical`. This drives the
|
|
1079
|
+
* REAL production factory (gateway.ts calls exactly this), so the re-entrancy
|
|
1080
|
+
* and termination properties are pinned where they live, not in a test copy.
|
|
1081
|
+
*/
|
|
1082
|
+
describe('approval-outcome drop notifier (PR D)', () => {
|
|
1083
|
+
/** Run every queued microtask to quiescence. */
|
|
1084
|
+
const settle = async (): Promise<void> => {
|
|
1085
|
+
for (let i = 0; i < 50; i++) await Promise.resolve()
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1088
|
+
it('the approval_outcome_dropped notice lands in the buffer and names the source', async () => {
|
|
1089
|
+
// Wire it exactly as gateway.ts does: the buffer's onEvictCritical calls
|
|
1090
|
+
// the notifier, and the notifier pushes back into the same buffer.
|
|
1091
|
+
const buf = createPendingInboundBuffer({
|
|
1092
|
+
capPerAgent: 3,
|
|
1093
|
+
log: () => {},
|
|
1094
|
+
now: () => 1000,
|
|
1095
|
+
onEvictCritical: (a, m) => notifier(a, m),
|
|
1096
|
+
})
|
|
1097
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1098
|
+
push: (agent, msg) => { buf.push(agent, msg) },
|
|
1099
|
+
log: () => {},
|
|
1100
|
+
})
|
|
1101
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1102
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1103
|
+
buf.push('a', inbound('skill_proposal_apply', 960))
|
|
1104
|
+
buf.push('a', inbound('mental_model_proposal_applied', 970)) // critical evict
|
|
1105
|
+
// Nothing enqueued synchronously — the notice must NOT ride the push frame.
|
|
1106
|
+
expect(buf.depth('a')).toBe(3)
|
|
1107
|
+
await settle()
|
|
1108
|
+
const drained = buf.drain('a')
|
|
1109
|
+
const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
|
|
1110
|
+
expect(notice).toBeDefined()
|
|
1111
|
+
// The notice push itself ran against a queue still full of outcomes, so it
|
|
1112
|
+
// evicted one more. The surviving notice must name BOTH — an earlier notice
|
|
1113
|
+
// naming only the first drop is exactly what the next overflow evicts.
|
|
1114
|
+
expect(notice!.meta?.dropped_sources).toBe('vault_grant_approved, secret_provided')
|
|
1115
|
+
expect(notice!.text).toContain('vault_grant_approved')
|
|
1116
|
+
// Every outcome that was NOT dropped is still buffered.
|
|
1117
|
+
expect(drained.map((m) => m.meta?.source)).toEqual([
|
|
1118
|
+
'skill_proposal_apply', 'mental_model_proposal_applied', 'approval_outcome_dropped',
|
|
1119
|
+
])
|
|
1120
|
+
})
|
|
1121
|
+
|
|
1122
|
+
// The guarantee is "a dropped approval outcome is never silent". If it
|
|
1123
|
+
// depended on each construction site passing a callback it would be
|
|
1124
|
+
// discipline, not a mechanism — so the notifier defaults ON and this pins it.
|
|
1125
|
+
it('is wired BY DEFAULT — a bare buffer still enqueues the notice', async () => {
|
|
1126
|
+
const buf = createPendingInboundBuffer({
|
|
1127
|
+
capPerAgent: 3,
|
|
1128
|
+
log: () => {},
|
|
1129
|
+
now: () => 1000,
|
|
1130
|
+
// NO onEvictCritical passed.
|
|
1131
|
+
})
|
|
1132
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1133
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1134
|
+
buf.push('a', inbound('skill_proposal_apply', 960))
|
|
1135
|
+
buf.push('a', inbound('vault_grant_denied', 970))
|
|
1136
|
+
await settle()
|
|
1137
|
+
const drained = buf.drain('a')
|
|
1138
|
+
const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
|
|
1139
|
+
expect(notice).toBeDefined()
|
|
1140
|
+
expect(notice!.meta?.dropped_sources).toContain('vault_grant_approved')
|
|
1141
|
+
})
|
|
1142
|
+
|
|
1143
|
+
it('an explicit no-op onEvictCritical opts out of the notice', async () => {
|
|
1144
|
+
const buf = createPendingInboundBuffer({
|
|
1145
|
+
capPerAgent: 3,
|
|
1146
|
+
log: () => {},
|
|
1147
|
+
now: () => 1000,
|
|
1148
|
+
onEvictCritical: () => {},
|
|
1149
|
+
})
|
|
1150
|
+
for (let i = 0; i < 6; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
|
|
1151
|
+
await settle()
|
|
1152
|
+
expect(buf.drain('a').some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(false)
|
|
1153
|
+
})
|
|
1154
|
+
|
|
1155
|
+
it('a burst of critical evictions terminates and does not spin the microtask queue', async () => {
|
|
1156
|
+
let pushes = 0
|
|
1157
|
+
const buf = createPendingInboundBuffer({
|
|
1158
|
+
capPerAgent: 3,
|
|
1159
|
+
log: () => {},
|
|
1160
|
+
now: () => 1000,
|
|
1161
|
+
onEvictCritical: (a, m) => notifier(a, m),
|
|
1162
|
+
})
|
|
1163
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1164
|
+
push: (agent, msg) => { pushes++; buf.push(agent, msg) },
|
|
1165
|
+
log: () => {},
|
|
1166
|
+
})
|
|
1167
|
+
for (let i = 0; i < 10; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
|
|
1168
|
+
await settle()
|
|
1169
|
+
// Bounded: at most one notice per burst plus the follow-up hop, never one
|
|
1170
|
+
// notice per evicted outcome (that is the recursion this guards).
|
|
1171
|
+
expect(pushes).toBeGreaterThan(0)
|
|
1172
|
+
expect(pushes).toBeLessThanOrEqual(3)
|
|
1173
|
+
// A notice is resident, and it is NOT itself an approval outcome — which is
|
|
1174
|
+
// what makes it the preferred victim next time and bounds the chain.
|
|
1175
|
+
const drained = buf.drain('a')
|
|
1176
|
+
expect(drained.some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(true)
|
|
1177
|
+
})
|
|
1178
|
+
|
|
1179
|
+
it('a burst coalesces into ONE notice naming every dropped source', async () => {
|
|
1180
|
+
const notices: InboundMessage[] = []
|
|
1181
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1182
|
+
push: (_agent, msg) => { notices.push(msg) },
|
|
1183
|
+
log: () => {},
|
|
1184
|
+
})
|
|
1185
|
+
notifier('a', inbound('vault_grant_approved', 1))
|
|
1186
|
+
notifier('a', inbound('secret_declined', 2))
|
|
1187
|
+
notifier('a', inbound('skill_proposal_apply', 3))
|
|
1188
|
+
expect(notices).toHaveLength(0) // deferred, never same-frame
|
|
1189
|
+
await settle()
|
|
1190
|
+
expect(notices).toHaveLength(1)
|
|
1191
|
+
expect(notices[0]!.meta?.dropped_sources).toBe(
|
|
1192
|
+
'vault_grant_approved, secret_declined, skill_proposal_apply',
|
|
1193
|
+
)
|
|
1194
|
+
expect(notices[0]!.meta?.dropped_count).toBe('3')
|
|
1195
|
+
})
|
|
1196
|
+
|
|
1197
|
+
it('a button tap victim is labelled button_callback, not "-"', async () => {
|
|
1198
|
+
const notices: InboundMessage[] = []
|
|
1199
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1200
|
+
push: (_agent, msg) => { notices.push(msg) },
|
|
1201
|
+
log: () => {},
|
|
1202
|
+
})
|
|
1203
|
+
notifier('a', {
|
|
1204
|
+
type: 'inbound', chatId: 'c1', messageId: 1, user: 'alice', userId: 42, ts: 1,
|
|
1205
|
+
text: '[user tapped button: Approve]', meta: { button_callback: 'true' },
|
|
1206
|
+
})
|
|
1207
|
+
await settle()
|
|
1208
|
+
expect(notices[0]!.meta?.dropped_sources).toBe('button_callback')
|
|
1209
|
+
})
|
|
1210
|
+
})
|
|
1211
|
+
|
|
1212
|
+
describe('isApprovalOutcome (PR D)', () => {
|
|
1213
|
+
const withMeta = (meta: Record<string, string> | undefined): InboundMessage => ({
|
|
1214
|
+
type: 'inbound', chatId: 'c1', messageId: 1, user: 'u', userId: 1, ts: 1, text: 't',
|
|
1215
|
+
...(meta != null ? { meta } : {}),
|
|
1216
|
+
} as InboundMessage)
|
|
1217
|
+
|
|
1218
|
+
it('true for every registered source', () => {
|
|
1219
|
+
for (const s of APPROVAL_OUTCOME_SOURCES) {
|
|
1220
|
+
expect(isApprovalOutcome(withMeta({ source: s }))).toBe(true)
|
|
1221
|
+
}
|
|
1222
|
+
})
|
|
1223
|
+
|
|
1224
|
+
it('true for a button tap with no source at all', () => {
|
|
1225
|
+
expect(isApprovalOutcome(withMeta({ button_callback: 'true' }))).toBe(true)
|
|
1226
|
+
})
|
|
1227
|
+
|
|
1228
|
+
it('false for ordinary messages and non-outcome system sources', () => {
|
|
1229
|
+
expect(isApprovalOutcome(withMeta({}))).toBe(false)
|
|
1230
|
+
expect(isApprovalOutcome(withMeta(undefined))).toBe(false)
|
|
1231
|
+
for (const s of [
|
|
1232
|
+
'missed_approval_retry', 'obligation_represent', 'subagent_handback',
|
|
1233
|
+
'subagent_progress', 'resume_interrupted', 'warmup', 'reaction', 'cron',
|
|
1234
|
+
'approval_outcome_dropped',
|
|
1235
|
+
]) {
|
|
1236
|
+
expect(isApprovalOutcome(withMeta({ source: s }))).toBe(false)
|
|
1237
|
+
}
|
|
1238
|
+
})
|
|
1239
|
+
|
|
1240
|
+
// #4664. `eval_case_suppressed` is the one protected member that is not a
|
|
1241
|
+
// verdict card, and it is the one with the least margin for error: no sweep
|
|
1242
|
+
// regenerates it (it fires exactly once, at suppression time, and posts no
|
|
1243
|
+
// card), so an eviction is a PERMANENT block on the agent's turn. Asserted
|
|
1244
|
+
// through the buffer, not just the set, so the pin fails on the behaviour
|
|
1245
|
+
// rather than on a membership literal.
|
|
1246
|
+
it('a buffered eval_case_suppressed notice survives a cap overflow', () => {
|
|
1247
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
|
|
1248
|
+
buf.push('a', inbound('eval_case_suppressed', 1))
|
|
1249
|
+
buf.push('a', userMsg({ text: 'first', ts: 2 }))
|
|
1250
|
+
buf.push('a', userMsg({ text: 'second', ts: 3 }))
|
|
1251
|
+
// Overflow: the resendable user message must go, not the one-shot notice.
|
|
1252
|
+
buf.push('a', userMsg({ text: 'third', ts: 4 }))
|
|
1253
|
+
const drained = buf.drain('a')
|
|
1254
|
+
expect(drained.map((m) => m.meta?.source ?? m.text)).toEqual([
|
|
1255
|
+
'eval_case_suppressed', 'second', 'third',
|
|
1256
|
+
])
|
|
1257
|
+
})
|
|
1258
|
+
|
|
1259
|
+
it('the dropped-notice source is deliberately NOT protected', () => {
|
|
1260
|
+
expect(APPROVAL_OUTCOME_SOURCES.has(APPROVAL_OUTCOME_DROPPED_SOURCE)).toBe(false)
|
|
1261
|
+
})
|
|
1262
|
+
|
|
1263
|
+
// Drift pin on the registry itself. `Object.freeze` on a Set does NOT block
|
|
1264
|
+
// `.add` (Set data lives in internal slots), so asserting `isFrozen` would be
|
|
1265
|
+
// a placebo — immutability is enforced at compile time by the `ReadonlySet`
|
|
1266
|
+
// type. What IS worth pinning is membership: silently dropping a source here
|
|
1267
|
+
// makes that verdict class evictable again with no test going red.
|
|
1268
|
+
it('the registry contents are exactly the audited set', () => {
|
|
1269
|
+
expect([...APPROVAL_OUTCOME_SOURCES].sort()).toEqual([
|
|
1270
|
+
'eval_case_applied',
|
|
1271
|
+
'eval_case_apply_failed',
|
|
1272
|
+
'eval_case_rejected',
|
|
1273
|
+
'eval_case_suppressed',
|
|
1274
|
+
'mental_model_proposal_applied',
|
|
1275
|
+
'mental_model_proposal_denied',
|
|
1276
|
+
'mental_model_proposal_failed',
|
|
1277
|
+
'mental_model_propose_timeout',
|
|
1278
|
+
'secret_declined',
|
|
1279
|
+
'secret_provide_failed',
|
|
1280
|
+
'secret_provided',
|
|
1281
|
+
'secret_request_timeout',
|
|
1282
|
+
'skill_proposal_apply',
|
|
1283
|
+
'vault_grant_approved',
|
|
1284
|
+
'vault_grant_denied',
|
|
1285
|
+
'vault_grant_timeout',
|
|
1286
|
+
'vault_save_completed',
|
|
1287
|
+
'vault_save_discarded',
|
|
1288
|
+
'vault_save_failed',
|
|
1289
|
+
'vault_save_timeout',
|
|
1290
|
+
])
|
|
1291
|
+
})
|
|
1292
|
+
})
|
|
@@ -96,6 +96,20 @@ describe('inbound source classification (F1 durability)', () => {
|
|
|
96
96
|
expect(stampsHandbackMarker('reaction')).toBe(false)
|
|
97
97
|
expect(stampsHandbackMarker('resume_interrupted')).toBe(false)
|
|
98
98
|
expect(stampsHandbackMarker('vault_grant_approved')).toBe(false)
|
|
99
|
+
// #4662 — the eval-case outcome inbounds. The exhaustiveness test below pins
|
|
100
|
+
// only their PRESENCE in the registry; these pin the VALUE, which is the part
|
|
101
|
+
// with runtime consequence: classified `true` (or left to the fail-safe
|
|
102
|
+
// default) they would hold the content gate chat-wide for 60 s after every
|
|
103
|
+
// operator tap on an eval-case card.
|
|
104
|
+
expect(stampsHandbackMarker('eval_case_applied')).toBe(false)
|
|
105
|
+
expect(stampsHandbackMarker('eval_case_rejected')).toBe(false)
|
|
106
|
+
expect(stampsHandbackMarker('eval_case_apply_failed')).toBe(false)
|
|
107
|
+
// #4664 — the suppressed-proposal notice. The exhaustiveness test below pins
|
|
108
|
+
// only its PRESENCE; this pins the VALUE, which is the part with runtime
|
|
109
|
+
// consequence: classified `true` (or left to the fail-safe default) it would
|
|
110
|
+
// hold the content gate chat-wide for 60 s every time a duplicate eval case
|
|
111
|
+
// is declined.
|
|
112
|
+
expect(stampsHandbackMarker('eval_case_suppressed')).toBe(false)
|
|
99
113
|
})
|
|
100
114
|
|
|
101
115
|
it('a normal user inbound (no meta.source) never stamps', () => {
|