switchroom 0.19.1 → 0.19.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent-scheduler/index.js +31 -1
- package/dist/auth-broker/index.js +565 -48
- package/dist/cli/autoaccept-poll.js +31 -1
- package/dist/cli/drive-write-pretool.mjs +32 -2
- package/dist/cli/ms-365-write-pretool.mjs +32 -2
- package/dist/cli/switchroom.js +1148 -274
- package/dist/host-control/main.js +3 -3
- package/dist/vault/approvals/kernel-server.js +2 -2
- package/dist/vault/broker/server.js +2 -2
- package/package.json +3 -2
- package/profiles/_base/start.sh.hbs +1 -0
- package/profiles/default/CLAUDE.md.hbs +8 -0
- package/skills/mental-model-curator/SKILL.md +68 -2
- package/skills/switchroom-cli/SKILL.md +25 -0
- package/telegram-plugin/auth-snapshot-format.ts +143 -12
- package/telegram-plugin/dist/bridge/bridge.js +8 -2
- package/telegram-plugin/dist/gateway/gateway.js +1427 -689
- package/telegram-plugin/dist/server.js +8 -2
- package/telegram-plugin/external-spend.ts +135 -0
- package/telegram-plugin/flushed-turn-supersede.ts +117 -13
- package/telegram-plugin/gateway/auth-add-flow.ts +215 -6
- package/telegram-plugin/gateway/auth-command.ts +138 -5
- package/telegram-plugin/gateway/gateway.ts +141 -158
- package/telegram-plugin/gateway/inbound-interceptors.ts +13 -3
- package/telegram-plugin/gateway/model-command.ts +309 -1
- package/telegram-plugin/gateway/narrative-lane.ts +23 -9
- package/telegram-plugin/gateway/outbound-send-path.ts +68 -15
- package/telegram-plugin/gateway/session-model-source.ts +90 -10
- package/telegram-plugin/gateway/status-pin-store.ts +64 -4
- package/telegram-plugin/gateway/stream-render.ts +22 -5
- package/telegram-plugin/gateway/usage-mask.ts +29 -0
- package/telegram-plugin/hooks/subagent-tracker-pretool.mjs +19 -2
- package/telegram-plugin/quota-bar-format.ts +78 -12
- package/telegram-plugin/quota-check.ts +17 -2
- package/telegram-plugin/reply-owner-resolve.ts +76 -11
- package/telegram-plugin/session-tail.ts +27 -3
- package/telegram-plugin/tests/activity-card-wiring.test.ts +47 -0
- package/telegram-plugin/tests/auth-add-flow.test.ts +367 -5
- package/telegram-plugin/tests/auth-snapshot-format.test.ts +41 -0
- package/telegram-plugin/tests/external-spend.test.ts +168 -0
- package/telegram-plugin/tests/flushed-turn-supersede.test.ts +117 -0
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +219 -29
- package/telegram-plugin/tests/model-command.test.ts +220 -0
- package/telegram-plugin/tests/quota-bar-format.test.ts +43 -0
- package/telegram-plugin/tests/quota-check.test.ts +57 -0
- package/telegram-plugin/tests/reply-owner-resolve.test.ts +257 -13
- package/telegram-plugin/tests/send-reply-golden.test.ts +154 -0
- package/telegram-plugin/tests/session-model-source.test.ts +142 -0
- package/telegram-plugin/tests/session-tail-first-attach.test.ts +115 -2
- package/telegram-plugin/tests/status-pin-store.test.ts +198 -0
- package/telegram-plugin/tests/subagent-tracker-hooks.test.ts +50 -0
- package/telegram-plugin/tests/usage-footer-freshness.test.ts +141 -0
- package/telegram-plugin/tests/usage-mask.test.ts +35 -0
- package/telegram-plugin/tests/worker-feed-dispatch.test.ts +27 -0
- package/telegram-plugin/tests/worker-feed-pin-persistence.test.ts +131 -1
- package/vendor/hindsight-memory/CHANGELOG.md +102 -0
- package/vendor/hindsight-memory/README.md +2 -1
- package/vendor/hindsight-memory/hooks/hooks.json +12 -0
- package/vendor/hindsight-memory/scripts/directive_verify.py +100 -3
- package/vendor/hindsight-memory/scripts/lib/config.py +150 -1
- package/vendor/hindsight-memory/scripts/lib/content.py +55 -5
- package/vendor/hindsight-memory/scripts/lib/directives.py +152 -15
- package/vendor/hindsight-memory/scripts/lib/parallel_recall.py +142 -0
- package/vendor/hindsight-memory/scripts/lib/state.py +31 -0
- package/vendor/hindsight-memory/scripts/recall.py +789 -143
- package/vendor/hindsight-memory/scripts/reconcile_tail.py +22 -1
- package/vendor/hindsight-memory/scripts/retain.py +71 -2
- package/vendor/hindsight-memory/scripts/subagent_retain.py +501 -0
- package/vendor/hindsight-memory/scripts/tests/test_directive_verify.py +169 -0
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +177 -0
- package/vendor/hindsight-memory/scripts/tests/test_lesson_tagging.py +200 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_context_turns_default.py +200 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_envelope_strip_telemetry.py +477 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +51 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_parallel_deadline.py +409 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_tag_weights.py +96 -0
- package/vendor/hindsight-memory/scripts/tests/test_recall_transcript_fallback.py +413 -0
- package/vendor/hindsight-memory/scripts/tests/test_reconcile_durability.py +49 -0
- package/vendor/hindsight-memory/scripts/tests/test_subagent_retain.py +439 -0
- package/vendor/hindsight-memory/settings.json +3 -1
|
@@ -10,6 +10,8 @@
|
|
|
10
10
|
*/
|
|
11
11
|
import { describe, it, expect } from 'vitest'
|
|
12
12
|
import { createSessionModelSource } from '../gateway/session-model-source.js'
|
|
13
|
+
import type { SessionModelDivergence } from '../gateway/session-model-source.js'
|
|
14
|
+
import { servedModelMatchesRequested } from '../gateway/model-command.js'
|
|
13
15
|
|
|
14
16
|
describe('createSessionModelSource — freshest observation wins', () => {
|
|
15
17
|
it('returns null when neither source has reported', () => {
|
|
@@ -76,3 +78,143 @@ describe('createSessionModelSource — freshest observation wins', () => {
|
|
|
76
78
|
expect(s.getOverride()).toBe('sr-glm-5')
|
|
77
79
|
})
|
|
78
80
|
})
|
|
81
|
+
|
|
82
|
+
// ── #3427 item 4: requested-vs-served divergence tripwire ────────────────────
|
|
83
|
+
//
|
|
84
|
+
// `--fallback-model` masks an invalid requested id: the override carries the
|
|
85
|
+
// requested token while claude silently serves the fallback. The FIRST live
|
|
86
|
+
// transcript observation of the post-relaunch session is the earliest
|
|
87
|
+
// deterministic verification point — these assert the handler FIRES on a
|
|
88
|
+
// mismatch (with the right payload), fires at most once per armed override,
|
|
89
|
+
// and — the #3437 H1/H2 false-positive contract — NEVER fires from a
|
|
90
|
+
// command-time (unarmed) override set or from a first-attach replay line.
|
|
91
|
+
|
|
92
|
+
describe('createSessionModelSource — divergence tripwire (#3427)', () => {
|
|
93
|
+
function armed() {
|
|
94
|
+
const fired: SessionModelDivergence[] = []
|
|
95
|
+
const s = createSessionModelSource({ servedMatchesRequested: servedModelMatchesRequested })
|
|
96
|
+
s.setDivergenceHandler((d) => fired.push(d))
|
|
97
|
+
return { s, fired }
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
it('fires ONCE with requested+served when the first post-boot line serves a different model', () => {
|
|
101
|
+
const { s, fired } = armed()
|
|
102
|
+
s.setOverride('claude-sonnet-9', { verify: true }) // boot rehydration: invalid id, fallback will substitute
|
|
103
|
+
s.noteTranscriptModel('claude-opus-4-8') // first assistant line: the fallback
|
|
104
|
+
expect(fired).toEqual([{ requested: 'claude-sonnet-9', served: 'claude-opus-4-8' }])
|
|
105
|
+
// Subsequent lines do not re-fire (once per armed override).
|
|
106
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
107
|
+
expect(fired).toHaveLength(1)
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
it('does NOT fire when the served model satisfies the requested token (alias → full id)', () => {
|
|
111
|
+
const { s, fired } = armed()
|
|
112
|
+
s.setOverride('sonnet', { verify: true })
|
|
113
|
+
s.noteTranscriptModel('claude-sonnet-5')
|
|
114
|
+
expect(fired).toHaveLength(0)
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
it('verification is consumed by the FIRST live observation — a later different model is a normal switch, not a divergence', () => {
|
|
118
|
+
const { s, fired } = armed()
|
|
119
|
+
s.setOverride('claude-opus-4-8', { verify: true })
|
|
120
|
+
s.noteTranscriptModel('claude-opus-4-8') // verified OK
|
|
121
|
+
s.noteTranscriptModel('claude-sonnet-5') // e.g. a native in-session switch
|
|
122
|
+
expect(fired).toHaveLength(0)
|
|
123
|
+
})
|
|
124
|
+
|
|
125
|
+
it('re-arms on every verify-set', () => {
|
|
126
|
+
const { s, fired } = armed()
|
|
127
|
+
s.setOverride('claude-opus-4-8', { verify: true })
|
|
128
|
+
s.noteTranscriptModel('claude-opus-4-8') // ok
|
|
129
|
+
s.setOverride('claude-sonnet-9', { verify: true }) // next apply-boot, bogus id
|
|
130
|
+
s.noteTranscriptModel('claude-opus-4-8') // fallback again
|
|
131
|
+
expect(fired).toEqual([{ requested: 'claude-sonnet-9', served: 'claude-opus-4-8' }])
|
|
132
|
+
})
|
|
133
|
+
|
|
134
|
+
it('clearing the override disarms (nothing to verify)', () => {
|
|
135
|
+
const { s, fired } = armed()
|
|
136
|
+
s.setOverride('claude-sonnet-9', { verify: true })
|
|
137
|
+
s.setOverride(null)
|
|
138
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
139
|
+
expect(fired).toHaveLength(0)
|
|
140
|
+
})
|
|
141
|
+
|
|
142
|
+
// H1 (#3437 review blocker): the command-time scheduleModelRelaunch record —
|
|
143
|
+
// setOverride(model) with NO verify — must not arm. The prior boot may have
|
|
144
|
+
// left the handler registered and its own arm consumed; an assistant line in
|
|
145
|
+
// the pre-restart window is served by the OLD model and would otherwise
|
|
146
|
+
// false-accuse a perfectly valid NEW token.
|
|
147
|
+
it('H1: a command-time setOverride (no verify) NEVER arms — no false DIVERGENCE in the pre-restart window', () => {
|
|
148
|
+
const { s, fired } = armed()
|
|
149
|
+
// Prior apply-boot: armed, verified OK on the first line.
|
|
150
|
+
s.setOverride('claude-opus-4-8', { verify: true })
|
|
151
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
152
|
+
// Operator issues /model <valid new id>; scheduleModelRelaunch records it
|
|
153
|
+
// pre-restart (status honesty) WITHOUT verify.
|
|
154
|
+
s.setOverride('claude-sonnet-5')
|
|
155
|
+
// The still-running OLD session emits another assistant line before the
|
|
156
|
+
// restart lands — served by the OLD model. Must NOT fire.
|
|
157
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
158
|
+
expect(fired).toHaveLength(0)
|
|
159
|
+
// …and the override record itself is intact for /status honesty.
|
|
160
|
+
expect(s.getOverride()).toBe('claude-sonnet-5')
|
|
161
|
+
})
|
|
162
|
+
|
|
163
|
+
it('H1: a verify-arm is CLEARED by a later plain set (the newest write wins, unarmed)', () => {
|
|
164
|
+
const { s, fired } = armed()
|
|
165
|
+
s.setOverride('claude-sonnet-9', { verify: true }) // armed, not yet verified
|
|
166
|
+
s.setOverride('claude-haiku-4-5') // command-time re-set before any line
|
|
167
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
168
|
+
expect(fired).toHaveLength(0)
|
|
169
|
+
})
|
|
170
|
+
|
|
171
|
+
// H2 (#3437 review blocker): the session-tail's first-attach replay delivers
|
|
172
|
+
// the PRIOR session's in-flight turn AFTER boot — OLD-model lines. They must
|
|
173
|
+
// neither fire nor consume verification; the first LIVE line still verifies.
|
|
174
|
+
it('H2: replayed observations neither fire nor consume — the first LIVE line still verifies (valid switch, no card)', () => {
|
|
175
|
+
const { s, fired } = armed()
|
|
176
|
+
s.setOverride('claude-sonnet-5', { verify: true }) // apply-boot onto a VALID id
|
|
177
|
+
// Boot replay of the pre-relaunch in-flight turn (old model claude-opus-4-8).
|
|
178
|
+
s.noteTranscriptModel('claude-opus-4-8', { replayed: true })
|
|
179
|
+
s.noteTranscriptModel('claude-opus-4-8', { replayed: true })
|
|
180
|
+
expect(fired).toHaveLength(0) // the false-positive path the review flagged
|
|
181
|
+
// First LIVE line of the new session: the requested model. Verified clean.
|
|
182
|
+
s.noteTranscriptModel('claude-sonnet-5')
|
|
183
|
+
expect(fired).toHaveLength(0)
|
|
184
|
+
})
|
|
185
|
+
|
|
186
|
+
it('H2: a genuinely bogus id still fires on the first LIVE line after replay', () => {
|
|
187
|
+
const { s, fired } = armed()
|
|
188
|
+
s.setOverride('claude-sonnet-9', { verify: true })
|
|
189
|
+
s.noteTranscriptModel('claude-opus-4-8', { replayed: true }) // replayed old-turn line: ignored
|
|
190
|
+
expect(fired).toHaveLength(0)
|
|
191
|
+
s.noteTranscriptModel('claude-opus-4-8') // first LIVE line: the fallback
|
|
192
|
+
expect(fired).toEqual([{ requested: 'claude-sonnet-9', served: 'claude-opus-4-8' }])
|
|
193
|
+
})
|
|
194
|
+
|
|
195
|
+
it('replayed observations still update /status freshness exactly as before', () => {
|
|
196
|
+
const { s } = armed()
|
|
197
|
+
s.setOverride('claude-sonnet-5', { verify: true })
|
|
198
|
+
s.noteTranscriptModel('claude-opus-4-8', { replayed: true })
|
|
199
|
+
// Freshness contract unchanged: the newer observation (transcript) wins.
|
|
200
|
+
expect(s.resolve()).toEqual({ model: 'claude-opus-4-8', source: 'transcript' })
|
|
201
|
+
})
|
|
202
|
+
|
|
203
|
+
it('never fires without a comparator (default construction) — old callers unchanged', () => {
|
|
204
|
+
const fired: SessionModelDivergence[] = []
|
|
205
|
+
const s = createSessionModelSource()
|
|
206
|
+
s.setDivergenceHandler((d) => fired.push(d))
|
|
207
|
+
s.setOverride('claude-sonnet-9', { verify: true })
|
|
208
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
209
|
+
expect(fired).toHaveLength(0)
|
|
210
|
+
})
|
|
211
|
+
|
|
212
|
+
it('a handler that clears the override does not break resolution', () => {
|
|
213
|
+
const s = createSessionModelSource({ servedMatchesRequested: servedModelMatchesRequested })
|
|
214
|
+
s.setDivergenceHandler(() => s.setOverride(null))
|
|
215
|
+
s.setOverride('claude-sonnet-9', { verify: true })
|
|
216
|
+
s.noteTranscriptModel('claude-opus-4-8')
|
|
217
|
+
expect(s.getOverride()).toBeNull()
|
|
218
|
+
expect(s.resolve()).toEqual({ model: 'claude-opus-4-8', source: 'transcript' })
|
|
219
|
+
})
|
|
220
|
+
})
|
|
@@ -1,8 +1,13 @@
|
|
|
1
1
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest'
|
|
2
|
-
import { mkdtempSync, rmSync, writeFileSync, statSync } from 'node:fs'
|
|
2
|
+
import { mkdtempSync, mkdirSync, rmSync, writeFileSync, appendFileSync, statSync } from 'node:fs'
|
|
3
3
|
import { tmpdir } from 'node:os'
|
|
4
4
|
import { join } from 'node:path'
|
|
5
|
-
import {
|
|
5
|
+
import {
|
|
6
|
+
computeFirstAttachCursor,
|
|
7
|
+
getProjectsDirForCwd,
|
|
8
|
+
startSessionTail,
|
|
9
|
+
type SessionEvent,
|
|
10
|
+
} from '../session-tail.js'
|
|
6
11
|
|
|
7
12
|
/**
|
|
8
13
|
* computeFirstAttachCursor: on first attach to a transcript, seek to EOF
|
|
@@ -63,3 +68,111 @@ describe('computeFirstAttachCursor', () => {
|
|
|
63
68
|
expect(computeFirstAttachCursor(missing, 0)).toBe(0)
|
|
64
69
|
})
|
|
65
70
|
})
|
|
71
|
+
|
|
72
|
+
// ── #3427 H2: first-attach replay marks `model` events as replayed ───────────
|
|
73
|
+
//
|
|
74
|
+
// A /model relaunch during an active turn is EXACTLY the shape that triggers
|
|
75
|
+
// the in-flight-turn replay above: the new gateway attaches to the OLD
|
|
76
|
+
// session's JSONL and replays its enqueue + assistant lines — which carry the
|
|
77
|
+
// PRE-relaunch model. Those replayed `model` observations must be flagged so
|
|
78
|
+
// the divergence tripwire (session-model-source) ignores them; live lines
|
|
79
|
+
// appended after attach must NOT be flagged. Outcome-asserted against the
|
|
80
|
+
// real tailer, not the source.
|
|
81
|
+
|
|
82
|
+
describe('startSessionTail — replayed model events are flagged (#3427 H2)', () => {
|
|
83
|
+
const tempDirs: string[] = []
|
|
84
|
+
afterEach(() => {
|
|
85
|
+
for (const d of tempDirs) {
|
|
86
|
+
try { rmSync(d, { recursive: true, force: true }) } catch { /* ignore */ }
|
|
87
|
+
}
|
|
88
|
+
tempDirs.length = 0
|
|
89
|
+
})
|
|
90
|
+
|
|
91
|
+
function mkProjectsDir(): { claudeHome: string; cwd: string; projectsDir: string } {
|
|
92
|
+
const base = mkdtempSync(join(tmpdir(), 'first-attach-replayed-'))
|
|
93
|
+
tempDirs.push(base)
|
|
94
|
+
const cwd = join(base, 'agent')
|
|
95
|
+
const claudeHome = join(base, 'claude-home')
|
|
96
|
+
const projectsDir = getProjectsDirForCwd(cwd, claudeHome)
|
|
97
|
+
mkdirSync(projectsDir, { recursive: true })
|
|
98
|
+
return { claudeHome, cwd, projectsDir }
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const wait = (ms: number): Promise<void> => new Promise((r) => setTimeout(r, ms))
|
|
102
|
+
|
|
103
|
+
const assistantModelLine = (model: string): string =>
|
|
104
|
+
JSON.stringify({
|
|
105
|
+
type: 'assistant',
|
|
106
|
+
message: { model, content: [{ type: 'text', text: 'hi' }] },
|
|
107
|
+
}) + '\n'
|
|
108
|
+
|
|
109
|
+
function modelEvents(events: SessionEvent[]): Array<{ model: string; replayed?: boolean }> {
|
|
110
|
+
return events.filter((e): e is Extract<SessionEvent, { kind: 'model' }> => e.kind === 'model')
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
it('mid-turn restart: replayed OLD-model lines carry replayed:true; a live line appended after attach does not', async () => {
|
|
114
|
+
const { claudeHome, cwd, projectsDir } = mkProjectsDir()
|
|
115
|
+
const file = join(projectsDir, 'sess.jsonl')
|
|
116
|
+
// The pre-relaunch session ended mid-turn: enqueue with no turn_duration,
|
|
117
|
+
// followed by an assistant line served by the OLD model.
|
|
118
|
+
writeFileSync(
|
|
119
|
+
file,
|
|
120
|
+
ENQUEUE + '\n' + DEQUEUE + '\n' + assistantModelLine('claude-opus-4-8'),
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
const events: SessionEvent[] = []
|
|
124
|
+
const handle = startSessionTail({
|
|
125
|
+
cwd,
|
|
126
|
+
claudeHome,
|
|
127
|
+
rescanIntervalMs: 50,
|
|
128
|
+
onEvent: (ev) => { events.push(ev) },
|
|
129
|
+
})
|
|
130
|
+
try {
|
|
131
|
+
await wait(200) // attach + replay the in-flight turn
|
|
132
|
+
const replayedBatch = modelEvents(events)
|
|
133
|
+
expect(replayedBatch.length).toBeGreaterThan(0)
|
|
134
|
+
// THE H2 false-positive shape: the old model arrives AFTER boot — but
|
|
135
|
+
// flagged, so the divergence tripwire skips it.
|
|
136
|
+
expect(replayedBatch[0]).toMatchObject({ model: 'claude-opus-4-8', replayed: true })
|
|
137
|
+
|
|
138
|
+
// The live session now writes its first assistant line (new model).
|
|
139
|
+
appendFileSync(file, assistantModelLine('claude-sonnet-5'))
|
|
140
|
+
await wait(200)
|
|
141
|
+
const all = modelEvents(events)
|
|
142
|
+
const live = all[all.length - 1]
|
|
143
|
+
expect(live.model).toBe('claude-sonnet-5')
|
|
144
|
+
expect(live.replayed).not.toBe(true)
|
|
145
|
+
} finally {
|
|
146
|
+
handle.stop()
|
|
147
|
+
}
|
|
148
|
+
})
|
|
149
|
+
|
|
150
|
+
it('completed-turn attach (no replay): appended model lines are never flagged', async () => {
|
|
151
|
+
const { claudeHome, cwd, projectsDir } = mkProjectsDir()
|
|
152
|
+
const file = join(projectsDir, 'sess.jsonl')
|
|
153
|
+
writeFileSync(
|
|
154
|
+
file,
|
|
155
|
+
ENQUEUE + '\n' + assistantModelLine('claude-opus-4-8') + TURN_DURATION + '\n',
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
const events: SessionEvent[] = []
|
|
159
|
+
const handle = startSessionTail({
|
|
160
|
+
cwd,
|
|
161
|
+
claudeHome,
|
|
162
|
+
rescanIntervalMs: 50,
|
|
163
|
+
onEvent: (ev) => { events.push(ev) },
|
|
164
|
+
})
|
|
165
|
+
try {
|
|
166
|
+
await wait(200) // attach seeks to EOF — no replay
|
|
167
|
+
expect(modelEvents(events)).toHaveLength(0)
|
|
168
|
+
appendFileSync(file, assistantModelLine('claude-sonnet-5'))
|
|
169
|
+
await wait(200)
|
|
170
|
+
const all = modelEvents(events)
|
|
171
|
+
expect(all.length).toBeGreaterThan(0)
|
|
172
|
+
expect(all[all.length - 1].model).toBe('claude-sonnet-5')
|
|
173
|
+
expect(all[all.length - 1].replayed).not.toBe(true)
|
|
174
|
+
} finally {
|
|
175
|
+
handle.stop()
|
|
176
|
+
}
|
|
177
|
+
})
|
|
178
|
+
})
|
|
@@ -7,10 +7,13 @@ import {
|
|
|
7
7
|
pinnedMessageIsOurs,
|
|
8
8
|
reconcileAndPersistStatusPin,
|
|
9
9
|
runStatusPinBootCleanup,
|
|
10
|
+
withPinReconcileLock,
|
|
10
11
|
type PersistedStatusPin,
|
|
11
12
|
type StatusPinStoreFsSeam,
|
|
12
13
|
type TrackedStatusPin,
|
|
13
14
|
} from "../gateway/status-pin-store.js";
|
|
15
|
+
import { decidePinAction, type PinState, type DesiredPin } from "../status-pin.js";
|
|
16
|
+
import { reconcilePin, type PinBotApi } from "../status-pin-driver.js";
|
|
14
17
|
|
|
15
18
|
/** In-memory fs seam with an atomic rename, so the store's tmp→rename
|
|
16
19
|
* crash-safety contract is exercised without touching the real disk. */
|
|
@@ -718,4 +721,199 @@ describe("reconcileAndPersistStatusPin — persist-before-pin ordering", () => {
|
|
|
718
721
|
expect(next).toBeNull();
|
|
719
722
|
expect(loadStatusPins(PATH, fs)).toEqual([]);
|
|
720
723
|
});
|
|
724
|
+
|
|
725
|
+
// F1 (invisible-worker-cards review): a `clear` op is emitted for BOTH a real
|
|
726
|
+
// unpin AND a `noop: already pinned`. For a noop, reconcilePin issues NO
|
|
727
|
+
// Telegram call and returns the LIVE claim (non-null). The pre-fix clear
|
|
728
|
+
// branch dropped the row unconditionally, so the worker feed's per-edit
|
|
729
|
+
// syncPin (every steady-state edit maps to a noop-clear) erased the durable
|
|
730
|
+
// row for a still-live pin — a crash after that left a stuck pinned card boot
|
|
731
|
+
// cleanup could never see. The row MUST be preserved when applyPin returns a
|
|
732
|
+
// live claim.
|
|
733
|
+
it("clear op with a LIVE claim (noop-already-pinned) PRESERVES the durable row", async () => {
|
|
734
|
+
const { fs, calls } = memFs();
|
|
735
|
+
persistStatusPins(PATH, fs, [pin({ pinKey: "wk:group:g1", messageId: 715 })]);
|
|
736
|
+
const before = calls.length;
|
|
737
|
+
// applyPin returns the live claim unchanged (what reconcilePin does on noop).
|
|
738
|
+
const next = await reconcileAndPersistStatusPin({
|
|
739
|
+
path: PATH,
|
|
740
|
+
fs,
|
|
741
|
+
pinKey: "wk:group:g1",
|
|
742
|
+
chatId: "-100123",
|
|
743
|
+
op: { kind: "clear" },
|
|
744
|
+
applyPin: async () => ({ messageId: 715 }),
|
|
745
|
+
log: () => {},
|
|
746
|
+
});
|
|
747
|
+
expect(next).toEqual({ messageId: 715 });
|
|
748
|
+
// Row survives (rewritten confirmed, no `pending`).
|
|
749
|
+
expect(loadStatusPins(PATH, fs)).toEqual([
|
|
750
|
+
{ pinKey: "wk:group:g1", chatId: "-100123", messageId: 715 },
|
|
751
|
+
]);
|
|
752
|
+
// Sanity: at least one disk write happened (the confirm rewrite) — the fix
|
|
753
|
+
// does not skip persistence, it changes the CONTENT from delete to upsert.
|
|
754
|
+
expect(calls.length).toBeGreaterThan(before);
|
|
755
|
+
});
|
|
756
|
+
|
|
757
|
+
it("repeated steady-state noop-clears keep the row present across many edits", async () => {
|
|
758
|
+
const { fs } = memFs();
|
|
759
|
+
persistStatusPins(PATH, fs, [pin({ pinKey: "wk:group:g1", messageId: 715 })]);
|
|
760
|
+
for (let i = 0; i < 5; i++) {
|
|
761
|
+
await reconcileAndPersistStatusPin({
|
|
762
|
+
path: PATH,
|
|
763
|
+
fs,
|
|
764
|
+
pinKey: "wk:group:g1",
|
|
765
|
+
chatId: "-100123",
|
|
766
|
+
op: { kind: "clear" },
|
|
767
|
+
applyPin: async () => ({ messageId: 715 }),
|
|
768
|
+
log: () => {},
|
|
769
|
+
});
|
|
770
|
+
}
|
|
771
|
+
expect(loadStatusPins(PATH, fs)).toEqual([
|
|
772
|
+
{ pinKey: "wk:group:g1", chatId: "-100123", messageId: 715 },
|
|
773
|
+
]);
|
|
774
|
+
});
|
|
775
|
+
});
|
|
776
|
+
|
|
777
|
+
/**
|
|
778
|
+
* F2 (invisible-worker-cards review): the gateway's status-pin reconcile reads
|
|
779
|
+
* its `prev` claim from an in-memory map at the top of the reconcile, then
|
|
780
|
+
* awaits. Two overlapping reconciles for ONE key with OPPOSITE desires could
|
|
781
|
+
* both capture a stale `prev` → a turn-end `{pinned:false}` clears the durable
|
|
782
|
+
* row while a flood-delayed open-pin lands, leaving a stuck pin with no row.
|
|
783
|
+
*
|
|
784
|
+
* These tests exercise the REAL fix primitives — `withPinReconcileLock` +
|
|
785
|
+
* `reconcileAndPersistStatusPin` + `reconcilePin` + `decidePinAction` — composed
|
|
786
|
+
* exactly as the gateway wires them (read prev INSIDE the per-key lock), and
|
|
787
|
+
* prove the row survives. The `noLock` control models the pre-fix ordering
|
|
788
|
+
* (prev read OUTSIDE the lock) and shows it regresses — so the lock, not luck,
|
|
789
|
+
* is what fixes it.
|
|
790
|
+
*/
|
|
791
|
+
describe("status-pin-store — F2 per-key reconcile serialization", () => {
|
|
792
|
+
// A gateway-faithful reconcile: an in-memory claim map (the `prev` source read
|
|
793
|
+
// at the TOP of the reconcile), the real decide→op mapping, and
|
|
794
|
+
// reconcileAndPersistStatusPin driving reconcilePin against a call-counting
|
|
795
|
+
// pin API. An optional gate lets a test hold ONE reconcile's applyPin open to
|
|
796
|
+
// force the exact interleave. `serialize` toggles the F2 fix.
|
|
797
|
+
function makeReconciler(fs: StatusPinStoreFsSeam, opts: { serialize: boolean }) {
|
|
798
|
+
const claims = new Map<string, PinState>();
|
|
799
|
+
const pinCalls: number[] = [];
|
|
800
|
+
const unpinCalls: number[] = [];
|
|
801
|
+
const api: PinBotApi = {
|
|
802
|
+
pinChatMessage: async (_c, id) => {
|
|
803
|
+
pinCalls.push(id);
|
|
804
|
+
},
|
|
805
|
+
unpinChatMessage: async (_c, id) => {
|
|
806
|
+
unpinCalls.push(id);
|
|
807
|
+
},
|
|
808
|
+
};
|
|
809
|
+
async function core(
|
|
810
|
+
pinKey: string,
|
|
811
|
+
chatId: string,
|
|
812
|
+
desired: DesiredPin,
|
|
813
|
+
gate?: Promise<void>,
|
|
814
|
+
) {
|
|
815
|
+
const prev = claims.get(pinKey) ?? null; // read BEFORE the store op (the F2 window)
|
|
816
|
+
const action = decidePinAction(prev, desired);
|
|
817
|
+
const op =
|
|
818
|
+
action.kind === "pin"
|
|
819
|
+
? ({ kind: "pin", messageId: action.messageId } as const)
|
|
820
|
+
: ({ kind: "clear" } as const);
|
|
821
|
+
const next = await reconcileAndPersistStatusPin({
|
|
822
|
+
path: PATH,
|
|
823
|
+
fs,
|
|
824
|
+
pinKey,
|
|
825
|
+
chatId,
|
|
826
|
+
op,
|
|
827
|
+
applyPin: async () => {
|
|
828
|
+
if (gate) await gate; // hold the pin open inside the store lock
|
|
829
|
+
return reconcilePin({ api, chatId, prevState: prev, desired });
|
|
830
|
+
},
|
|
831
|
+
log: () => {},
|
|
832
|
+
});
|
|
833
|
+
if (next == null) claims.delete(pinKey);
|
|
834
|
+
else claims.set(pinKey, next);
|
|
835
|
+
}
|
|
836
|
+
function reconcile(
|
|
837
|
+
pinKey: string,
|
|
838
|
+
chatId: string,
|
|
839
|
+
desired: DesiredPin,
|
|
840
|
+
gate?: Promise<void>,
|
|
841
|
+
) {
|
|
842
|
+
return opts.serialize
|
|
843
|
+
? withPinReconcileLock(pinKey, () => core(pinKey, chatId, desired, gate))
|
|
844
|
+
: core(pinKey, chatId, desired, gate);
|
|
845
|
+
}
|
|
846
|
+
return { reconcile, claims, pinCalls, unpinCalls };
|
|
847
|
+
}
|
|
848
|
+
|
|
849
|
+
// The exact review sequence: a flood-delayed OPEN-pin (A) is in-flight inside
|
|
850
|
+
// the store lock; a turn-end UNPIN (B) starts during A's await window and
|
|
851
|
+
// reads prev=null (A has not set the claim yet). Pre-fix, B's clear then drops
|
|
852
|
+
// the disk row A wrote — stranding a live in-memory claim with NO durable row.
|
|
853
|
+
async function runStrandingSequence(serialize: boolean) {
|
|
854
|
+
const { fs } = memFs();
|
|
855
|
+
const r = makeReconciler(fs, { serialize });
|
|
856
|
+
let releaseA!: () => void;
|
|
857
|
+
const gateA = new Promise<void>((res) => {
|
|
858
|
+
releaseA = res;
|
|
859
|
+
});
|
|
860
|
+
// A: open-pin 900, held open at applyPin.
|
|
861
|
+
const aDone = r.reconcile("fg:c:3", "-100123", { pinned: true, messageId: 900 }, gateA);
|
|
862
|
+
// Let A reach its applyPin await (pending row on disk; claim NOT yet set).
|
|
863
|
+
await Promise.resolve();
|
|
864
|
+
await Promise.resolve();
|
|
865
|
+
// B: turn-end unpin, starts now — reads prev from the in-memory claim map.
|
|
866
|
+
const bDone = r.reconcile("fg:c:3", "-100123", { pinned: false });
|
|
867
|
+
await Promise.resolve();
|
|
868
|
+
releaseA();
|
|
869
|
+
await Promise.all([aDone, bDone]);
|
|
870
|
+
return { fs, claims: r.claims, unpinCalls: r.unpinCalls, pinCalls: r.pinCalls };
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
it("serialized: turn-end unpin waits for the in-flight open-pin, so the pinned message is genuinely UNPINNED (not stranded)", async () => {
|
|
874
|
+
const { fs, claims, pinCalls, unpinCalls } = await runStrandingSequence(true);
|
|
875
|
+
const rows = loadStatusPins(PATH, fs);
|
|
876
|
+
// Under serialization B runs only after A fully settles, so B reads
|
|
877
|
+
// prev={900}, issues a REAL unpin of the pinned message, and clears the row
|
|
878
|
+
// and claim together. The Telegram pin is cleaned up — nothing stuck.
|
|
879
|
+
expect(pinCalls).toEqual([900]); // A pinned it
|
|
880
|
+
expect(unpinCalls).toEqual([900]); // B unpinned it (the fix)
|
|
881
|
+
expect(rows).toEqual([]);
|
|
882
|
+
expect(claims.get("fg:c:3") ?? null).toBeNull();
|
|
883
|
+
});
|
|
884
|
+
|
|
885
|
+
it("pre-fix control (no per-key lock): the same interleave STRANDS the pinned message — pinned, row cleared, NEVER unpinned", async () => {
|
|
886
|
+
const { fs, unpinCalls, pinCalls } = await runStrandingSequence(false);
|
|
887
|
+
const rows = loadStatusPins(PATH, fs);
|
|
888
|
+
// Reproduces the F2 defect the lock removes: B read prev=null during A's
|
|
889
|
+
// await window, so its clear was a NO-OP unpin (issued no Telegram unpin)
|
|
890
|
+
// yet still dropped the durable row A wrote — while A's pin call landed. The
|
|
891
|
+
// message 900 stays pinned on Telegram with NO durable row and no unpin ever
|
|
892
|
+
// issued → stuck until a boot cleanup that can no longer see it.
|
|
893
|
+
expect(pinCalls).toEqual([900]); // message 900 WAS pinned
|
|
894
|
+
expect(unpinCalls).toEqual([]); // …but never unpinned (no-op clear on prev=null)
|
|
895
|
+
expect(rows).toEqual([]); // …and the durable row is gone → unreapable
|
|
896
|
+
});
|
|
897
|
+
|
|
898
|
+
it("serialized steady-state noop reconciles add ZERO pin/unpin API calls (finn-flood guard)", async () => {
|
|
899
|
+
const { fs } = memFs();
|
|
900
|
+
const r = makeReconciler(fs, { serialize: true });
|
|
901
|
+
// First open pins exactly once.
|
|
902
|
+
await r.reconcile("wk:group:g1", "-100123", { pinned: true, messageId: 900 });
|
|
903
|
+
expect(r.pinCalls).toEqual([900]);
|
|
904
|
+
expect(r.unpinCalls).toEqual([]);
|
|
905
|
+
|
|
906
|
+
// 20 steady-state edits — each a re-pin of the SAME id → decidePinAction
|
|
907
|
+
// noop → reconcilePin issues NO Telegram call. The serialization + F1
|
|
908
|
+
// row-preservation must not add a single pin/unpin API call.
|
|
909
|
+
for (let i = 0; i < 20; i++) {
|
|
910
|
+
await r.reconcile("wk:group:g1", "-100123", { pinned: true, messageId: 900 });
|
|
911
|
+
}
|
|
912
|
+
expect(r.pinCalls).toEqual([900]); // still exactly one pin, ever
|
|
913
|
+
expect(r.unpinCalls).toEqual([]); // never unpinned
|
|
914
|
+
// The durable row is intact throughout (F1).
|
|
915
|
+
expect(loadStatusPins(PATH, fs)).toEqual([
|
|
916
|
+
{ pinKey: "wk:group:g1", chatId: "-100123", messageId: 900 },
|
|
917
|
+
]);
|
|
918
|
+
});
|
|
721
919
|
});
|
|
@@ -147,6 +147,56 @@ describe('subagent-tracker-pretool', () => {
|
|
|
147
147
|
expect(row?.model ?? null).toBeNull()
|
|
148
148
|
})
|
|
149
149
|
|
|
150
|
+
// F3 (progress-card fork model): a fork dispatch inherits the parent's model
|
|
151
|
+
// and IGNORES tool_input.model, so seeding the row's first-paint model from
|
|
152
|
+
// that ignored override made the worker card show a WRONG model (e.g. "sonnet"
|
|
153
|
+
// while the fork runs Opus) until the transcript overwrote it. The seed must
|
|
154
|
+
// be suppressed for forks — model stays NULL and the card omits it until the
|
|
155
|
+
// watcher records the real model from the fork's own transcript.
|
|
156
|
+
it('leaves model null for a FORK dispatch even when tool_input.model is set (F3)', () => {
|
|
157
|
+
const event = {
|
|
158
|
+
session_id: 'sess-fork',
|
|
159
|
+
tool_name: 'Agent',
|
|
160
|
+
tool_use_id: 'toolu_fork001',
|
|
161
|
+
tool_input: {
|
|
162
|
+
subagent_type: 'fork',
|
|
163
|
+
description: 'Fork the session',
|
|
164
|
+
run_in_background: true,
|
|
165
|
+
model: 'sonnet', // override a fork ignores — must NOT be persisted
|
|
166
|
+
},
|
|
167
|
+
}
|
|
168
|
+
const result = runHook(PRETOOL_SCRIPT, event)
|
|
169
|
+
expect(result.status).toBe(0)
|
|
170
|
+
|
|
171
|
+
const db = openDb()
|
|
172
|
+
const row = db.prepare('SELECT model FROM subagents WHERE id = ?').get('toolu_fork001') as
|
|
173
|
+
| { model: string | null }
|
|
174
|
+
| undefined
|
|
175
|
+
expect(row?.model ?? null).toBeNull()
|
|
176
|
+
})
|
|
177
|
+
|
|
178
|
+
it('still persists tool_input.model for a NON-fork dispatch (fork suppression is scoped)', () => {
|
|
179
|
+
const event = {
|
|
180
|
+
session_id: 'sess-nonfork',
|
|
181
|
+
tool_name: 'Agent',
|
|
182
|
+
tool_use_id: 'toolu_nonfork001',
|
|
183
|
+
tool_input: {
|
|
184
|
+
subagent_type: 'researcher',
|
|
185
|
+
description: 'Research with a pinned model',
|
|
186
|
+
run_in_background: true,
|
|
187
|
+
model: 'claude-opus-4-8',
|
|
188
|
+
},
|
|
189
|
+
}
|
|
190
|
+
const result = runHook(PRETOOL_SCRIPT, event)
|
|
191
|
+
expect(result.status).toBe(0)
|
|
192
|
+
|
|
193
|
+
const db = openDb()
|
|
194
|
+
const row = db.prepare('SELECT model FROM subagents WHERE id = ?').get('toolu_nonfork001') as
|
|
195
|
+
| { model: string | null }
|
|
196
|
+
| undefined
|
|
197
|
+
expect(row?.model).toBe('claude-opus-4-8')
|
|
198
|
+
})
|
|
199
|
+
|
|
150
200
|
it('does not write a row when tool_name is not Agent', () => {
|
|
151
201
|
const event = {
|
|
152
202
|
session_id: 'sess-abc123',
|