switchroom 0.21.6 → 0.21.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/autoaccept.exp +21 -1
- package/bin/tmp-reaper.sh +234 -0
- package/dist/agent-scheduler/index.js +1 -1
- package/dist/auth-broker/index.js +2 -2
- package/dist/cli/autoaccept-poll.js +95 -7
- package/dist/cli/notion-write-pretool.mjs +1 -1
- package/dist/cli/switchroom.js +3432 -2751
- package/dist/host-control/main.js +177 -13
- package/dist/vault/approvals/kernel-server.js +2 -2
- package/dist/vault/broker/server.js +2 -2
- package/package.json +6 -4
- package/profiles/_base/start.sh.hbs +115 -0
- package/profiles/_shared/local-time.md.hbs +6 -0
- package/profiles/default/CLAUDE.md.hbs +0 -12
- package/telegram-plugin/bunfig.toml +12 -4
- package/telegram-plugin/dist/gateway/gateway.js +1017 -465
- package/telegram-plugin/gateway/agent-process-liveness.ts +558 -0
- package/telegram-plugin/gateway/approval-hold.ts +32 -1
- package/telegram-plugin/gateway/approval-outcome-sources.ts +274 -0
- package/telegram-plugin/gateway/bridge-dead-watchdog.ts +21 -9
- package/telegram-plugin/gateway/callback-query-handlers.ts +87 -15
- package/telegram-plugin/gateway/eval-case-proposal-inbound-builders.ts +197 -0
- package/telegram-plugin/gateway/gateway.ts +12 -10
- package/telegram-plugin/gateway/pending-inbound-buffer.ts +167 -11
- package/telegram-plugin/gateway/self-improve-proposal-wiring.test.ts +333 -0
- package/telegram-plugin/gateway/self-improve-proposal-wiring.ts +152 -3
- package/telegram-plugin/gateway/subagent-handback-marker.ts +19 -0
- package/telegram-plugin/tests/agent-process-liveness.test.ts +406 -0
- package/telegram-plugin/tests/approval-hold-record.test.ts +21 -8
- package/telegram-plugin/tests/boot-resume-gateway-only-respawn.test.ts +752 -0
- package/telegram-plugin/tests/boot-resume-guard-wiring.test.ts +203 -0
- package/telegram-plugin/tests/callback-query-handlers.test.ts +143 -1
- package/telegram-plugin/tests/eval-case-proposal-inbound-builders.test.ts +144 -0
- package/telegram-plugin/tests/framework-fallback-drains-parked.test.ts +7 -1
- package/telegram-plugin/tests/hermes-messages-paging.test.ts +149 -0
- package/telegram-plugin/tests/hermes-session-search.test.ts +146 -0
- package/telegram-plugin/tests/parked-turn-start-preload.test.ts +65 -0
- package/telegram-plugin/tests/pending-inbound-buffer.test.ts +443 -2
- package/telegram-plugin/tests/queued-card-surface.test.ts +9 -1
- package/telegram-plugin/tests/stream-render-golden.test.ts +10 -1
- package/telegram-plugin/tests/subagent-handback-marker.test.ts +14 -0
- package/telegram-plugin/tests/turn-mint-defers-until-dequeue.test.ts +7 -1
- package/telegram-plugin/tests/turn-supersede-finalizes-prior-card.test.ts +7 -1
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { describe, it, expect } from 'vitest'
|
|
10
|
-
import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, DEFAULT_PENDING_INBOUND_CAP } from '../gateway/pending-inbound-buffer.js'
|
|
10
|
+
import { createPendingInboundBuffer, redeliverBufferedInbound, idleDrainTick, planBufferedRedelivery, selectEvictionVictim, DEFAULT_PENDING_INBOUND_CAP, APPROVAL_OUTCOME_PROTECTION_MS } from '../gateway/pending-inbound-buffer.js'
|
|
11
|
+
import { APPROVAL_OUTCOME_SOURCES, APPROVAL_OUTCOME_DROPPED_SOURCE, isApprovalOutcome, createApprovalOutcomeDropNotifier } from '../gateway/approval-outcome-sources.js'
|
|
11
12
|
import type { InboundMessage } from '../gateway/ipc-protocol.js'
|
|
12
13
|
import { ObligationLedger } from '../gateway/obligation-ledger.js'
|
|
13
14
|
import { makeRepresentRedeliveryGuard } from '../gateway/represent-delivery-guard.js'
|
|
@@ -171,7 +172,11 @@ describe('pending-inbound-buffer', () => {
|
|
|
171
172
|
const buf = createPendingInboundBuffer({ capPerAgent: 1, log: (l) => logs.push(l) })
|
|
172
173
|
buf.push('a', inbound('m1', 1))
|
|
173
174
|
buf.push('a', inbound('m2', 2)) // evicts m1
|
|
174
|
-
|
|
175
|
+
// PR D: victim selection is tiered, so the line names the index + tier
|
|
176
|
+
// rather than claiming "oldest" (it is the oldest of its tier).
|
|
177
|
+
expect(
|
|
178
|
+
logs.some((l) => l.includes('cap=1') && l.includes('dropped entry idx=0 reason=non-outcome')),
|
|
179
|
+
).toBe(true)
|
|
175
180
|
expect(logs.some((l) => l.includes('m1'))).toBe(true)
|
|
176
181
|
})
|
|
177
182
|
|
|
@@ -849,3 +854,439 @@ describe('redeliverBufferedInbound — beforeRedeliver fail-open on throw', () =
|
|
|
849
854
|
expect(buf.depth('a')).toBe(0) // nothing stranded
|
|
850
855
|
})
|
|
851
856
|
})
|
|
857
|
+
|
|
858
|
+
/**
|
|
859
|
+
* PR D — cap eviction must not pick an APPROVAL OUTCOME while anything else is
|
|
860
|
+
* droppable. The victim used to be an unconditional `q.shift()`, and the oldest
|
|
861
|
+
* entry is exactly the one most likely to be a synthetic approval outcome that
|
|
862
|
+
* has been waiting through an entire turn. An ordinary chat message is
|
|
863
|
+
* resendable; a `vault_grant_approved` is not — the operator's tap already
|
|
864
|
+
* happened, so "please resend" is meaningless and the agent blocks forever.
|
|
865
|
+
*/
|
|
866
|
+
describe('pending-inbound-buffer — approval-outcome eviction protection (PR D)', () => {
|
|
867
|
+
/** A button tap: NO meta.source at all, only meta.button_callback. */
|
|
868
|
+
function tap(ts: number): InboundMessage {
|
|
869
|
+
return {
|
|
870
|
+
type: 'inbound',
|
|
871
|
+
chatId: 'c1',
|
|
872
|
+
messageId: ts,
|
|
873
|
+
user: 'alice',
|
|
874
|
+
userId: 42,
|
|
875
|
+
ts,
|
|
876
|
+
text: '[user tapped button: Approve]',
|
|
877
|
+
meta: { button_callback: 'true', button_callback_data: 'ag:ok', button_text: 'Approve' },
|
|
878
|
+
}
|
|
879
|
+
}
|
|
880
|
+
|
|
881
|
+
it('an outcome at the HEAD survives an overflow of ordinary messages', () => {
|
|
882
|
+
const buf = createPendingInboundBuffer({ log: () => {} }) // cap 32
|
|
883
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
884
|
+
for (let i = 0; i < 31; i++) buf.push('a', userMsg({ text: `m${i}`, ts: 100 + i }))
|
|
885
|
+
buf.push('a', userMsg({ text: 'overflow', ts: 999 }))
|
|
886
|
+
const drained = buf.drain('a')
|
|
887
|
+
expect(drained.some((m) => m.meta?.source === 'vault_grant_approved')).toBe(true)
|
|
888
|
+
// The oldest ORDINARY message went instead, and nothing else was lost.
|
|
889
|
+
expect(drained).toHaveLength(32)
|
|
890
|
+
expect(drained.map((m) => m.text)).not.toContain('m0')
|
|
891
|
+
expect(drained.map((m) => m.text)).toContain('overflow')
|
|
892
|
+
})
|
|
893
|
+
|
|
894
|
+
it('a sustained ordinary-message burst never evicts the buffered outcome', () => {
|
|
895
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
|
|
896
|
+
buf.push('a', inbound('secret_provided', 1))
|
|
897
|
+
for (let i = 0; i < 40; i++) buf.push('a', userMsg({ text: `u${i}`, ts: 100 + i }))
|
|
898
|
+
const drained = buf.drain('a')
|
|
899
|
+
expect(drained[0]?.meta?.source).toBe('secret_provided')
|
|
900
|
+
expect(drained.map((m) => m.text).slice(1)).toEqual(['u37', 'u38', 'u39'])
|
|
901
|
+
})
|
|
902
|
+
|
|
903
|
+
it('a button tap is protected too — it carries NO meta.source', () => {
|
|
904
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
|
|
905
|
+
buf.push('a', tap(1))
|
|
906
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
907
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
908
|
+
buf.push('a', userMsg({ text: 'u3', ts: 4 }))
|
|
909
|
+
const drained = buf.drain('a')
|
|
910
|
+
expect(drained[0]?.meta?.button_callback).toBe('true')
|
|
911
|
+
expect(drained.map((m) => m.text).slice(1)).toEqual(['u2', 'u3'])
|
|
912
|
+
})
|
|
913
|
+
|
|
914
|
+
it('survivors keep insertion order after a mid-queue eviction (FIFO contract)', () => {
|
|
915
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 4, log: () => {} })
|
|
916
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
917
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
918
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
919
|
+
buf.push('a', userMsg({ text: 'u3', ts: 4 }))
|
|
920
|
+
buf.push('a', userMsg({ text: 'u4', ts: 5 })) // evicts u1 (index 1), not the head
|
|
921
|
+
expect(buf.drain('a').map((m) => m.meta?.source ?? m.text)).toEqual([
|
|
922
|
+
'vault_grant_approved', 'u2', 'u3', 'u4',
|
|
923
|
+
])
|
|
924
|
+
})
|
|
925
|
+
|
|
926
|
+
it('onEvict (not onEvictCritical) fires for an ordinary victim', () => {
|
|
927
|
+
const ordinary: InboundMessage[] = []
|
|
928
|
+
const critical: InboundMessage[] = []
|
|
929
|
+
const buf = createPendingInboundBuffer({
|
|
930
|
+
capPerAgent: 2,
|
|
931
|
+
log: () => {},
|
|
932
|
+
onEvict: (_a, m) => ordinary.push(m),
|
|
933
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
934
|
+
})
|
|
935
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
936
|
+
buf.push('a', userMsg({ text: 'u1', ts: 2 }))
|
|
937
|
+
buf.push('a', userMsg({ text: 'u2', ts: 3 }))
|
|
938
|
+
expect(ordinary.map((m) => m.text)).toEqual(['u1'])
|
|
939
|
+
expect(critical).toHaveLength(0)
|
|
940
|
+
})
|
|
941
|
+
|
|
942
|
+
it('all-outcomes fallback: drops the oldest and fires onEvictCritical, NOT onEvict', () => {
|
|
943
|
+
const ordinary: InboundMessage[] = []
|
|
944
|
+
const critical: { agent: string; msg: InboundMessage }[] = []
|
|
945
|
+
const buf = createPendingInboundBuffer({
|
|
946
|
+
capPerAgent: 3,
|
|
947
|
+
log: () => {},
|
|
948
|
+
now: () => 1000, // all three below are well inside the protection window
|
|
949
|
+
onEvict: (_a, m) => ordinary.push(m),
|
|
950
|
+
onEvictCritical: (agent, msg) => critical.push({ agent, msg }),
|
|
951
|
+
})
|
|
952
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
953
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
954
|
+
buf.push('a', inbound('skill_proposal_apply', 980))
|
|
955
|
+
expect(critical).toHaveLength(0)
|
|
956
|
+
buf.push('a', inbound('mental_model_proposal_applied', 999))
|
|
957
|
+
expect(ordinary).toHaveLength(0)
|
|
958
|
+
expect(critical).toHaveLength(1)
|
|
959
|
+
expect(critical[0]!.agent).toBe('a')
|
|
960
|
+
expect(critical[0]!.msg.meta?.source).toBe('vault_grant_approved')
|
|
961
|
+
expect(critical[0]!.msg.ts).toBe(900)
|
|
962
|
+
// The newest outcome is in and the other two are intact.
|
|
963
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
|
|
964
|
+
'secret_provided', 'skill_proposal_apply', 'mental_model_proposal_applied',
|
|
965
|
+
])
|
|
966
|
+
})
|
|
967
|
+
|
|
968
|
+
it('a STALE outcome is evictable — protection is bounded, so it cannot pin the buffer', () => {
|
|
969
|
+
const critical: InboundMessage[] = []
|
|
970
|
+
const nowMs = 100 * 60 * 1000
|
|
971
|
+
const buf = createPendingInboundBuffer({
|
|
972
|
+
capPerAgent: 2,
|
|
973
|
+
log: () => {},
|
|
974
|
+
now: () => nowMs,
|
|
975
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
976
|
+
})
|
|
977
|
+
// Stale: older than the 15-min protection window (already spool-escalated).
|
|
978
|
+
buf.push('a', inbound('vault_grant_timeout', nowMs - 20 * 60 * 1000))
|
|
979
|
+
// Fresh.
|
|
980
|
+
buf.push('a', inbound('secret_provided', nowMs - 1000))
|
|
981
|
+
buf.push('a', inbound('vault_grant_approved', nowMs))
|
|
982
|
+
expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_timeout'])
|
|
983
|
+
// The FRESH outcomes both survived — staleness, not arrival order, decided it.
|
|
984
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual([
|
|
985
|
+
'secret_provided', 'vault_grant_approved',
|
|
986
|
+
])
|
|
987
|
+
})
|
|
988
|
+
|
|
989
|
+
it('APPROVAL_OUTCOME_PROTECTION_MS matches the spool escalation window', () => {
|
|
990
|
+
expect(APPROVAL_OUTCOME_PROTECTION_MS).toBe(15 * 60 * 1000)
|
|
991
|
+
})
|
|
992
|
+
|
|
993
|
+
it('cap of 1 with a single fresh outcome: evicts it and reports critical', () => {
|
|
994
|
+
const critical: InboundMessage[] = []
|
|
995
|
+
const buf = createPendingInboundBuffer({
|
|
996
|
+
capPerAgent: 1,
|
|
997
|
+
log: () => {},
|
|
998
|
+
now: () => 1000,
|
|
999
|
+
onEvictCritical: (_a, m) => critical.push(m),
|
|
1000
|
+
})
|
|
1001
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1002
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1003
|
+
expect(critical.map((m) => m.meta?.source)).toEqual(['vault_grant_approved'])
|
|
1004
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
|
|
1005
|
+
})
|
|
1006
|
+
|
|
1007
|
+
it('a throwing onEvictCritical never breaks the push hot path', () => {
|
|
1008
|
+
const buf = createPendingInboundBuffer({
|
|
1009
|
+
capPerAgent: 1,
|
|
1010
|
+
log: () => {},
|
|
1011
|
+
onEvictCritical: () => { throw new Error('notice failed') },
|
|
1012
|
+
})
|
|
1013
|
+
buf.push('a', inbound('vault_grant_approved', 1))
|
|
1014
|
+
expect(() => buf.push('a', inbound('secret_provided', 2))).not.toThrow()
|
|
1015
|
+
expect(buf.drain('a').map((m) => m.meta?.source)).toEqual(['secret_provided'])
|
|
1016
|
+
})
|
|
1017
|
+
|
|
1018
|
+
describe('selectEvictionVictim tiers', () => {
|
|
1019
|
+
const out = (ts: number) => inbound('vault_grant_approved', ts)
|
|
1020
|
+
const ord = (ts: number) => userMsg({ text: `u${ts}`, ts })
|
|
1021
|
+
|
|
1022
|
+
it('empty queue is safe (index 0, no crash at the call site)', () => {
|
|
1023
|
+
expect(selectEvictionVictim([], 0)).toEqual({ index: 0, reason: 'all-outcomes' })
|
|
1024
|
+
})
|
|
1025
|
+
|
|
1026
|
+
it('picks the FIRST non-outcome, not merely any non-outcome', () => {
|
|
1027
|
+
expect(selectEvictionVictim([out(1), out(2), ord(3), ord(4)], 5)).toEqual({
|
|
1028
|
+
index: 2, reason: 'non-outcome',
|
|
1029
|
+
})
|
|
1030
|
+
})
|
|
1031
|
+
|
|
1032
|
+
it('prefers an ordinary message over a STALE outcome', () => {
|
|
1033
|
+
const nowMs = 60 * 60 * 1000
|
|
1034
|
+
expect(selectEvictionVictim([out(1), ord(nowMs)], nowMs)).toEqual({
|
|
1035
|
+
index: 1, reason: 'non-outcome',
|
|
1036
|
+
})
|
|
1037
|
+
})
|
|
1038
|
+
|
|
1039
|
+
// Tier 2 vs tier 3. These two tiers agree on the victim whenever the queue
|
|
1040
|
+
// is in `ts`-ascending order (the oldest entry is then also the stalest),
|
|
1041
|
+
// so an in-order fixture CANNOT tell them apart — disabling the age bound
|
|
1042
|
+
// entirely leaves such a test green (verified by mutation). The
|
|
1043
|
+
// discriminating case is a queue whose head is FRESH and whose later entry
|
|
1044
|
+
// is STALE.
|
|
1045
|
+
it('prefers a STALE outcome over a fresher one queued ahead of it', () => {
|
|
1046
|
+
const nowMs = 100 * 60 * 1000
|
|
1047
|
+
const fresh = out(nowMs)
|
|
1048
|
+
const stale = out(nowMs - 20 * 60 * 1000) // past the 15-min window
|
|
1049
|
+
expect(selectEvictionVictim([fresh, stale], nowMs)).toEqual({
|
|
1050
|
+
index: 1, reason: 'stale-outcome',
|
|
1051
|
+
})
|
|
1052
|
+
})
|
|
1053
|
+
|
|
1054
|
+
// The `reason` is not cosmetic: it is what the eviction log line reports,
|
|
1055
|
+
// and it is the only signal distinguishing "dropped an outcome the spool
|
|
1056
|
+
// already escalated" from "dropped a live one because there was nothing
|
|
1057
|
+
// else". Pin it in the ordinary in-order shape too.
|
|
1058
|
+
it('reports stale-outcome (not all-outcomes) when the victim is past its window', () => {
|
|
1059
|
+
const nowMs = 100 * 60 * 1000
|
|
1060
|
+
expect(
|
|
1061
|
+
selectEvictionVictim([out(nowMs - 20 * 60 * 1000), out(nowMs)], nowMs).reason,
|
|
1062
|
+
).toBe('stale-outcome')
|
|
1063
|
+
})
|
|
1064
|
+
|
|
1065
|
+
// The bound must actually be a bound: an outcome one millisecond inside
|
|
1066
|
+
// the window is still protected, so tier 3 is what fires.
|
|
1067
|
+
it('an outcome just INSIDE the window is not stale — falls through to tier 3', () => {
|
|
1068
|
+
const nowMs = 100 * 60 * 1000
|
|
1069
|
+
const justInside = out(nowMs - (APPROVAL_OUTCOME_PROTECTION_MS - 1))
|
|
1070
|
+
expect(selectEvictionVictim([out(nowMs), justInside], nowMs)).toEqual({
|
|
1071
|
+
index: 0, reason: 'all-outcomes',
|
|
1072
|
+
})
|
|
1073
|
+
})
|
|
1074
|
+
})
|
|
1075
|
+
})
|
|
1076
|
+
|
|
1077
|
+
/**
|
|
1078
|
+
* PR D — the notifier the gateway wires to `onEvictCritical`. This drives the
|
|
1079
|
+
* REAL production factory (gateway.ts calls exactly this), so the re-entrancy
|
|
1080
|
+
* and termination properties are pinned where they live, not in a test copy.
|
|
1081
|
+
*/
|
|
1082
|
+
describe('approval-outcome drop notifier (PR D)', () => {
|
|
1083
|
+
/** Run every queued microtask to quiescence. */
|
|
1084
|
+
const settle = async (): Promise<void> => {
|
|
1085
|
+
for (let i = 0; i < 50; i++) await Promise.resolve()
|
|
1086
|
+
}
|
|
1087
|
+
|
|
1088
|
+
it('the approval_outcome_dropped notice lands in the buffer and names the source', async () => {
|
|
1089
|
+
// Wire it exactly as gateway.ts does: the buffer's onEvictCritical calls
|
|
1090
|
+
// the notifier, and the notifier pushes back into the same buffer.
|
|
1091
|
+
const buf = createPendingInboundBuffer({
|
|
1092
|
+
capPerAgent: 3,
|
|
1093
|
+
log: () => {},
|
|
1094
|
+
now: () => 1000,
|
|
1095
|
+
onEvictCritical: (a, m) => notifier(a, m),
|
|
1096
|
+
})
|
|
1097
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1098
|
+
push: (agent, msg) => { buf.push(agent, msg) },
|
|
1099
|
+
log: () => {},
|
|
1100
|
+
})
|
|
1101
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1102
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1103
|
+
buf.push('a', inbound('skill_proposal_apply', 960))
|
|
1104
|
+
buf.push('a', inbound('mental_model_proposal_applied', 970)) // critical evict
|
|
1105
|
+
// Nothing enqueued synchronously — the notice must NOT ride the push frame.
|
|
1106
|
+
expect(buf.depth('a')).toBe(3)
|
|
1107
|
+
await settle()
|
|
1108
|
+
const drained = buf.drain('a')
|
|
1109
|
+
const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
|
|
1110
|
+
expect(notice).toBeDefined()
|
|
1111
|
+
// The notice push itself ran against a queue still full of outcomes, so it
|
|
1112
|
+
// evicted one more. The surviving notice must name BOTH — an earlier notice
|
|
1113
|
+
// naming only the first drop is exactly what the next overflow evicts.
|
|
1114
|
+
expect(notice!.meta?.dropped_sources).toBe('vault_grant_approved, secret_provided')
|
|
1115
|
+
expect(notice!.text).toContain('vault_grant_approved')
|
|
1116
|
+
// Every outcome that was NOT dropped is still buffered.
|
|
1117
|
+
expect(drained.map((m) => m.meta?.source)).toEqual([
|
|
1118
|
+
'skill_proposal_apply', 'mental_model_proposal_applied', 'approval_outcome_dropped',
|
|
1119
|
+
])
|
|
1120
|
+
})
|
|
1121
|
+
|
|
1122
|
+
// The guarantee is "a dropped approval outcome is never silent". If it
|
|
1123
|
+
// depended on each construction site passing a callback it would be
|
|
1124
|
+
// discipline, not a mechanism — so the notifier defaults ON and this pins it.
|
|
1125
|
+
it('is wired BY DEFAULT — a bare buffer still enqueues the notice', async () => {
|
|
1126
|
+
const buf = createPendingInboundBuffer({
|
|
1127
|
+
capPerAgent: 3,
|
|
1128
|
+
log: () => {},
|
|
1129
|
+
now: () => 1000,
|
|
1130
|
+
// NO onEvictCritical passed.
|
|
1131
|
+
})
|
|
1132
|
+
buf.push('a', inbound('vault_grant_approved', 900))
|
|
1133
|
+
buf.push('a', inbound('secret_provided', 950))
|
|
1134
|
+
buf.push('a', inbound('skill_proposal_apply', 960))
|
|
1135
|
+
buf.push('a', inbound('vault_grant_denied', 970))
|
|
1136
|
+
await settle()
|
|
1137
|
+
const drained = buf.drain('a')
|
|
1138
|
+
const notice = drained.find((m) => m.meta?.source === 'approval_outcome_dropped')
|
|
1139
|
+
expect(notice).toBeDefined()
|
|
1140
|
+
expect(notice!.meta?.dropped_sources).toContain('vault_grant_approved')
|
|
1141
|
+
})
|
|
1142
|
+
|
|
1143
|
+
it('an explicit no-op onEvictCritical opts out of the notice', async () => {
|
|
1144
|
+
const buf = createPendingInboundBuffer({
|
|
1145
|
+
capPerAgent: 3,
|
|
1146
|
+
log: () => {},
|
|
1147
|
+
now: () => 1000,
|
|
1148
|
+
onEvictCritical: () => {},
|
|
1149
|
+
})
|
|
1150
|
+
for (let i = 0; i < 6; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
|
|
1151
|
+
await settle()
|
|
1152
|
+
expect(buf.drain('a').some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(false)
|
|
1153
|
+
})
|
|
1154
|
+
|
|
1155
|
+
it('a burst of critical evictions terminates and does not spin the microtask queue', async () => {
|
|
1156
|
+
let pushes = 0
|
|
1157
|
+
const buf = createPendingInboundBuffer({
|
|
1158
|
+
capPerAgent: 3,
|
|
1159
|
+
log: () => {},
|
|
1160
|
+
now: () => 1000,
|
|
1161
|
+
onEvictCritical: (a, m) => notifier(a, m),
|
|
1162
|
+
})
|
|
1163
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1164
|
+
push: (agent, msg) => { pushes++; buf.push(agent, msg) },
|
|
1165
|
+
log: () => {},
|
|
1166
|
+
})
|
|
1167
|
+
for (let i = 0; i < 10; i++) buf.push('a', inbound('vault_grant_approved', 900 + i))
|
|
1168
|
+
await settle()
|
|
1169
|
+
// Bounded: at most one notice per burst plus the follow-up hop, never one
|
|
1170
|
+
// notice per evicted outcome (that is the recursion this guards).
|
|
1171
|
+
expect(pushes).toBeGreaterThan(0)
|
|
1172
|
+
expect(pushes).toBeLessThanOrEqual(3)
|
|
1173
|
+
// A notice is resident, and it is NOT itself an approval outcome — which is
|
|
1174
|
+
// what makes it the preferred victim next time and bounds the chain.
|
|
1175
|
+
const drained = buf.drain('a')
|
|
1176
|
+
expect(drained.some((m) => m.meta?.source === 'approval_outcome_dropped')).toBe(true)
|
|
1177
|
+
})
|
|
1178
|
+
|
|
1179
|
+
it('a burst coalesces into ONE notice naming every dropped source', async () => {
|
|
1180
|
+
const notices: InboundMessage[] = []
|
|
1181
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1182
|
+
push: (_agent, msg) => { notices.push(msg) },
|
|
1183
|
+
log: () => {},
|
|
1184
|
+
})
|
|
1185
|
+
notifier('a', inbound('vault_grant_approved', 1))
|
|
1186
|
+
notifier('a', inbound('secret_declined', 2))
|
|
1187
|
+
notifier('a', inbound('skill_proposal_apply', 3))
|
|
1188
|
+
expect(notices).toHaveLength(0) // deferred, never same-frame
|
|
1189
|
+
await settle()
|
|
1190
|
+
expect(notices).toHaveLength(1)
|
|
1191
|
+
expect(notices[0]!.meta?.dropped_sources).toBe(
|
|
1192
|
+
'vault_grant_approved, secret_declined, skill_proposal_apply',
|
|
1193
|
+
)
|
|
1194
|
+
expect(notices[0]!.meta?.dropped_count).toBe('3')
|
|
1195
|
+
})
|
|
1196
|
+
|
|
1197
|
+
it('a button tap victim is labelled button_callback, not "-"', async () => {
|
|
1198
|
+
const notices: InboundMessage[] = []
|
|
1199
|
+
const notifier = createApprovalOutcomeDropNotifier({
|
|
1200
|
+
push: (_agent, msg) => { notices.push(msg) },
|
|
1201
|
+
log: () => {},
|
|
1202
|
+
})
|
|
1203
|
+
notifier('a', {
|
|
1204
|
+
type: 'inbound', chatId: 'c1', messageId: 1, user: 'alice', userId: 42, ts: 1,
|
|
1205
|
+
text: '[user tapped button: Approve]', meta: { button_callback: 'true' },
|
|
1206
|
+
})
|
|
1207
|
+
await settle()
|
|
1208
|
+
expect(notices[0]!.meta?.dropped_sources).toBe('button_callback')
|
|
1209
|
+
})
|
|
1210
|
+
})
|
|
1211
|
+
|
|
1212
|
+
describe('isApprovalOutcome (PR D)', () => {
|
|
1213
|
+
const withMeta = (meta: Record<string, string> | undefined): InboundMessage => ({
|
|
1214
|
+
type: 'inbound', chatId: 'c1', messageId: 1, user: 'u', userId: 1, ts: 1, text: 't',
|
|
1215
|
+
...(meta != null ? { meta } : {}),
|
|
1216
|
+
} as InboundMessage)
|
|
1217
|
+
|
|
1218
|
+
it('true for every registered source', () => {
|
|
1219
|
+
for (const s of APPROVAL_OUTCOME_SOURCES) {
|
|
1220
|
+
expect(isApprovalOutcome(withMeta({ source: s }))).toBe(true)
|
|
1221
|
+
}
|
|
1222
|
+
})
|
|
1223
|
+
|
|
1224
|
+
it('true for a button tap with no source at all', () => {
|
|
1225
|
+
expect(isApprovalOutcome(withMeta({ button_callback: 'true' }))).toBe(true)
|
|
1226
|
+
})
|
|
1227
|
+
|
|
1228
|
+
it('false for ordinary messages and non-outcome system sources', () => {
|
|
1229
|
+
expect(isApprovalOutcome(withMeta({}))).toBe(false)
|
|
1230
|
+
expect(isApprovalOutcome(withMeta(undefined))).toBe(false)
|
|
1231
|
+
for (const s of [
|
|
1232
|
+
'missed_approval_retry', 'obligation_represent', 'subagent_handback',
|
|
1233
|
+
'subagent_progress', 'resume_interrupted', 'warmup', 'reaction', 'cron',
|
|
1234
|
+
'approval_outcome_dropped',
|
|
1235
|
+
]) {
|
|
1236
|
+
expect(isApprovalOutcome(withMeta({ source: s }))).toBe(false)
|
|
1237
|
+
}
|
|
1238
|
+
})
|
|
1239
|
+
|
|
1240
|
+
// #4664. `eval_case_suppressed` is the one protected member that is not a
|
|
1241
|
+
// verdict card, and it is the one with the least margin for error: no sweep
|
|
1242
|
+
// regenerates it (it fires exactly once, at suppression time, and posts no
|
|
1243
|
+
// card), so an eviction is a PERMANENT block on the agent's turn. Asserted
|
|
1244
|
+
// through the buffer, not just the set, so the pin fails on the behaviour
|
|
1245
|
+
// rather than on a membership literal.
|
|
1246
|
+
it('a buffered eval_case_suppressed notice survives a cap overflow', () => {
|
|
1247
|
+
const buf = createPendingInboundBuffer({ capPerAgent: 3, log: () => {} })
|
|
1248
|
+
buf.push('a', inbound('eval_case_suppressed', 1))
|
|
1249
|
+
buf.push('a', userMsg({ text: 'first', ts: 2 }))
|
|
1250
|
+
buf.push('a', userMsg({ text: 'second', ts: 3 }))
|
|
1251
|
+
// Overflow: the resendable user message must go, not the one-shot notice.
|
|
1252
|
+
buf.push('a', userMsg({ text: 'third', ts: 4 }))
|
|
1253
|
+
const drained = buf.drain('a')
|
|
1254
|
+
expect(drained.map((m) => m.meta?.source ?? m.text)).toEqual([
|
|
1255
|
+
'eval_case_suppressed', 'second', 'third',
|
|
1256
|
+
])
|
|
1257
|
+
})
|
|
1258
|
+
|
|
1259
|
+
it('the dropped-notice source is deliberately NOT protected', () => {
|
|
1260
|
+
expect(APPROVAL_OUTCOME_SOURCES.has(APPROVAL_OUTCOME_DROPPED_SOURCE)).toBe(false)
|
|
1261
|
+
})
|
|
1262
|
+
|
|
1263
|
+
// Drift pin on the registry itself. `Object.freeze` on a Set does NOT block
|
|
1264
|
+
// `.add` (Set data lives in internal slots), so asserting `isFrozen` would be
|
|
1265
|
+
// a placebo — immutability is enforced at compile time by the `ReadonlySet`
|
|
1266
|
+
// type. What IS worth pinning is membership: silently dropping a source here
|
|
1267
|
+
// makes that verdict class evictable again with no test going red.
|
|
1268
|
+
it('the registry contents are exactly the audited set', () => {
|
|
1269
|
+
expect([...APPROVAL_OUTCOME_SOURCES].sort()).toEqual([
|
|
1270
|
+
'eval_case_applied',
|
|
1271
|
+
'eval_case_apply_failed',
|
|
1272
|
+
'eval_case_rejected',
|
|
1273
|
+
'eval_case_suppressed',
|
|
1274
|
+
'mental_model_proposal_applied',
|
|
1275
|
+
'mental_model_proposal_denied',
|
|
1276
|
+
'mental_model_proposal_failed',
|
|
1277
|
+
'mental_model_propose_timeout',
|
|
1278
|
+
'secret_declined',
|
|
1279
|
+
'secret_provide_failed',
|
|
1280
|
+
'secret_provided',
|
|
1281
|
+
'secret_request_timeout',
|
|
1282
|
+
'skill_proposal_apply',
|
|
1283
|
+
'vault_grant_approved',
|
|
1284
|
+
'vault_grant_denied',
|
|
1285
|
+
'vault_grant_timeout',
|
|
1286
|
+
'vault_save_completed',
|
|
1287
|
+
'vault_save_discarded',
|
|
1288
|
+
'vault_save_failed',
|
|
1289
|
+
'vault_save_timeout',
|
|
1290
|
+
])
|
|
1291
|
+
})
|
|
1292
|
+
})
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
* • remove → the card is finalized as "folded into the current task".
|
|
15
15
|
* • TTL → the card is finalized as timed-out, never left frozen.
|
|
16
16
|
*/
|
|
17
|
-
import { describe, it, expect, beforeEach } from 'vitest'
|
|
17
|
+
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
18
18
|
import {
|
|
19
19
|
handleSessionEvent,
|
|
20
20
|
__resetParkedTurnStartsForTest,
|
|
@@ -58,9 +58,17 @@ function withRecordingBot(h: Harness) {
|
|
|
58
58
|
* stores the card id on the parked envelope) has run. */
|
|
59
59
|
const settle = () => new Promise((r) => setTimeout(r, 0))
|
|
60
60
|
|
|
61
|
+
// The parked store is module-scope and `bun test` runs all ~657 files in ONE
|
|
62
|
+
// process, so resetting on ENTRY alone is not enough: a case that ends mid-park
|
|
63
|
+
// leaves the entry behind for every later FILE. #4611 — this suite's last case
|
|
64
|
+
// parks msg 502 and never dequeues, and the leftover made the obligation sweep
|
|
65
|
+
// read the session as busy for the rest of the run, failing represent-guard.
|
|
61
66
|
beforeEach(() => {
|
|
62
67
|
__resetParkedTurnStartsForTest()
|
|
63
68
|
})
|
|
69
|
+
afterEach(() => {
|
|
70
|
+
__resetParkedTurnStartsForTest()
|
|
71
|
+
})
|
|
64
72
|
|
|
65
73
|
describe('Part B — queued card is posted at park and adopted on dequeue', () => {
|
|
66
74
|
it('parks with a reply-anchored "Queued" card, then EDITS that same card in place on dequeue', async () => {
|
|
@@ -19,7 +19,7 @@
|
|
|
19
19
|
* the duplicate. This is the exact duplicate-reply class the shared-singleton
|
|
20
20
|
* injection (never a re-`new`) exists to kill.
|
|
21
21
|
*/
|
|
22
|
-
import { describe, it, expect, beforeEach } from 'vitest'
|
|
22
|
+
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
23
23
|
import { readFileSync } from 'node:fs'
|
|
24
24
|
import { tmpdir } from 'node:os'
|
|
25
25
|
import {
|
|
@@ -47,6 +47,15 @@ import {
|
|
|
47
47
|
import type { CurrentTurn } from '../gateway/gateway.js'
|
|
48
48
|
import type { ReplyOwnerTier } from '../reply-owner-resolve.js'
|
|
49
49
|
|
|
50
|
+
// The parked turn-start store is module-scope and `bun test` runs every file in
|
|
51
|
+
// ONE process, so the two describe-scoped `beforeEach` resets below clean ENTRY
|
|
52
|
+
// only — a case that ends mid-park still leaks into the next FILE, where the
|
|
53
|
+
// obligation sweep reads the leftover as a busy session (#4611). File-level so
|
|
54
|
+
// it covers every case here, not just the two blocks that reset on entry.
|
|
55
|
+
afterEach(() => {
|
|
56
|
+
__resetParkedTurnStartsForTest()
|
|
57
|
+
})
|
|
58
|
+
|
|
50
59
|
/** The owner-resolution shape `resolveReplyOwnerTurn` returns, including the
|
|
51
60
|
* candidate set the content-gate bypass corroborates against. These fixtures
|
|
52
61
|
* never exercise the supersede path, so the candidates mirror the resolved turn
|
|
@@ -96,6 +96,20 @@ describe('inbound source classification (F1 durability)', () => {
|
|
|
96
96
|
expect(stampsHandbackMarker('reaction')).toBe(false)
|
|
97
97
|
expect(stampsHandbackMarker('resume_interrupted')).toBe(false)
|
|
98
98
|
expect(stampsHandbackMarker('vault_grant_approved')).toBe(false)
|
|
99
|
+
// #4662 — the eval-case outcome inbounds. The exhaustiveness test below pins
|
|
100
|
+
// only their PRESENCE in the registry; these pin the VALUE, which is the part
|
|
101
|
+
// with runtime consequence: classified `true` (or left to the fail-safe
|
|
102
|
+
// default) they would hold the content gate chat-wide for 60 s after every
|
|
103
|
+
// operator tap on an eval-case card.
|
|
104
|
+
expect(stampsHandbackMarker('eval_case_applied')).toBe(false)
|
|
105
|
+
expect(stampsHandbackMarker('eval_case_rejected')).toBe(false)
|
|
106
|
+
expect(stampsHandbackMarker('eval_case_apply_failed')).toBe(false)
|
|
107
|
+
// #4664 — the suppressed-proposal notice. The exhaustiveness test below pins
|
|
108
|
+
// only its PRESENCE; this pins the VALUE, which is the part with runtime
|
|
109
|
+
// consequence: classified `true` (or left to the fail-safe default) it would
|
|
110
|
+
// hold the content gate chat-wide for 60 s every time a duplicate eval case
|
|
111
|
+
// is declined.
|
|
112
|
+
expect(stampsHandbackMarker('eval_case_suppressed')).toBe(false)
|
|
99
113
|
})
|
|
100
114
|
|
|
101
115
|
it('a normal user inbound (no meta.source) never stamps', () => {
|
|
@@ -20,7 +20,7 @@
|
|
|
20
20
|
* These drive the REAL `handleSessionEvent` (extracted-module golden-harness
|
|
21
21
|
* oracle, same standard as `stream-render-golden.test.ts`).
|
|
22
22
|
*/
|
|
23
|
-
import { describe, it, expect, beforeEach, vi } from 'vitest'
|
|
23
|
+
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'
|
|
24
24
|
import {
|
|
25
25
|
handleSessionEvent,
|
|
26
26
|
__resetParkedTurnStartsForTest,
|
|
@@ -29,9 +29,15 @@ import {
|
|
|
29
29
|
import { projectTranscriptLine } from '../session-tail.js'
|
|
30
30
|
import { CHAT, enqueue, inbound, makeHarness } from './turn-mint-harness.js'
|
|
31
31
|
|
|
32
|
+
// Module-scope store + one-process `bun test` sweep: resetting on ENTRY alone
|
|
33
|
+
// leaves a mid-park case's entry behind for every later FILE, which makes the
|
|
34
|
+
// obligation sweep read the session as busy for the rest of the run (#4611).
|
|
32
35
|
beforeEach(() => {
|
|
33
36
|
__resetParkedTurnStartsForTest()
|
|
34
37
|
})
|
|
38
|
+
afterEach(() => {
|
|
39
|
+
__resetParkedTurnStartsForTest()
|
|
40
|
+
})
|
|
35
41
|
|
|
36
42
|
describe('#3927 FIX A — a mid-turn enqueue parks; the turn mints on dequeue', () => {
|
|
37
43
|
it('a second enqueue while turn A is live opens NO card, does not steal the slot, ' +
|
|
@@ -25,7 +25,7 @@
|
|
|
25
25
|
* mints straight on top of it. That is the exact orphan carrie hit, and it is
|
|
26
26
|
* what FIX B finalizes.
|
|
27
27
|
*/
|
|
28
|
-
import { describe, it, expect, beforeEach } from 'vitest'
|
|
28
|
+
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
29
29
|
import {
|
|
30
30
|
handleSessionEvent,
|
|
31
31
|
__resetParkedTurnStartsForTest,
|
|
@@ -45,9 +45,15 @@ function syntheticEnqueue(chatId: string, text: string) {
|
|
|
45
45
|
}
|
|
46
46
|
}
|
|
47
47
|
|
|
48
|
+
// Module-scope store + one-process `bun test` sweep: resetting on ENTRY alone
|
|
49
|
+
// leaves a mid-park case's entry behind for every later FILE, which makes the
|
|
50
|
+
// obligation sweep read the session as busy for the rest of the run (#4611).
|
|
48
51
|
beforeEach(() => {
|
|
49
52
|
__resetParkedTurnStartsForTest()
|
|
50
53
|
})
|
|
54
|
+
afterEach(() => {
|
|
55
|
+
__resetParkedTurnStartsForTest()
|
|
56
|
+
})
|
|
51
57
|
|
|
52
58
|
describe('#3927 FIX B — superseding a live turn finalizes its card', () => {
|
|
53
59
|
it('a dequeue-driven mint on top of a never-ended turn calls clearActivitySummary ' +
|