switchroom 0.17.10 ā 0.18.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/workspace-dynamic-hook.sh +12 -13
- package/dist/agent-scheduler/index.js +27 -1
- package/dist/auth-broker/index.js +6161 -151
- package/dist/cli/notion-write-pretool.mjs +29 -2
- package/dist/cli/switchroom.js +578 -454
- package/dist/host-control/main.js +6182 -172
- package/dist/vault/approvals/kernel-server.js +5891 -164
- package/dist/vault/broker/server.js +6597 -881
- package/package.json +1 -1
- package/profiles/_base/settings.json.hbs +2 -2
- package/profiles/_base/start.sh.hbs +170 -21
- package/profiles/coding/CLAUDE.md.hbs +1 -1
- package/profiles/default/CLAUDE.md +2 -2
- package/profiles/default/CLAUDE.md.hbs +2 -2
- package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
- package/profiles/health-coach/CLAUDE.md.hbs +1 -1
- package/telegram-plugin/auth-snapshot-format.ts +22 -24
- package/telegram-plugin/context-exhaustion.ts +124 -0
- package/telegram-plugin/dist/gateway/gateway.js +24086 -8727
- package/telegram-plugin/gateway/activity-card-store.ts +76 -0
- package/telegram-plugin/gateway/gateway.ts +480 -85
- package/telegram-plugin/gateway/inbound-delivery-gate.ts +26 -0
- package/telegram-plugin/gateway/model-command.ts +70 -10
- package/telegram-plugin/package.json +6 -0
- package/telegram-plugin/quota-watch.ts +4 -6
- package/telegram-plugin/registry/turns-schema.test.ts +97 -0
- package/telegram-plugin/registry/turns-schema.ts +78 -0
- package/telegram-plugin/render/ir.ts +209 -0
- package/telegram-plugin/render/parse.ts +363 -0
- package/telegram-plugin/render/render.ts +440 -0
- package/telegram-plugin/render/rich-render.ts +72 -0
- package/telegram-plugin/stream-controller.ts +14 -3
- package/telegram-plugin/tests/activity-card-store.test.ts +94 -0
- package/telegram-plugin/tests/auth-command-format2.test.ts +1 -1
- package/telegram-plugin/tests/auth-snapshot-format.test.ts +30 -16
- package/telegram-plugin/tests/claude-code-event-contract.test.ts +48 -0
- package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +11 -0
- package/telegram-plugin/tests/feed-survival.test.ts +39 -0
- package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +81 -0
- package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +82 -0
- package/telegram-plugin/tests/liveness-tracker.test.ts +228 -0
- package/telegram-plugin/tests/model-command.test.ts +193 -16
- package/telegram-plugin/tests/narrative-render.test.ts +125 -0
- package/telegram-plugin/tests/orphaned-reply-rearm.test.ts +123 -163
- package/telegram-plugin/tests/quota-watch.test.ts +1 -4
- package/telegram-plugin/tests/rapid-fire-delivery-ordering.test.ts +149 -0
- package/telegram-plugin/tests/render/parse-torture.test.ts +136 -0
- package/telegram-plugin/tests/render/parse.test.ts +393 -0
- package/telegram-plugin/tests/render/render.test.ts +436 -0
- package/telegram-plugin/tests/render/rich-render.test.ts +85 -0
- package/telegram-plugin/tests/telegram-activity-visibility-integration.test.ts +155 -1
- package/telegram-plugin/tests/worktree-watch-cwds.test.ts +98 -3
- package/telegram-plugin/turn-liveness-floor.ts +35 -1
- package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +99 -7
- package/telegram-plugin/worktree-watch-cwds.ts +92 -17
- package/vendor/hindsight-memory/scripts/lib/client.py +11 -1
- package/vendor/hindsight-memory/scripts/lib/config.py +9 -2
- package/vendor/hindsight-memory/scripts/recall.py +64 -6
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +1 -0
- package/vendor/hindsight-memory/tests/test_client.py +43 -0
- package/vendor/hindsight-memory/tests/test_recall_precision.py +114 -0
|
@@ -262,11 +262,13 @@ describe('renderAuthSnapshotFormat2', () => {
|
|
|
262
262
|
return rows;
|
|
263
263
|
}
|
|
264
264
|
|
|
265
|
-
it('renders a GFM table with the State/Account/5h/7d/
|
|
265
|
+
it('renders a GFM table with the State/Account/5h/5h resets/7d/7d resets header', () => {
|
|
266
266
|
const out = renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' });
|
|
267
267
|
expect(out).toContain('š **Auth ā fleet status**');
|
|
268
|
-
expect(out).toContain('| State | Account | 5h | 7d |
|
|
269
|
-
expect(out).toContain('| --- | --- | --- | --- | --- |');
|
|
268
|
+
expect(out).toContain('| State | Account | 5h | 5h resets | 7d | 7d resets |');
|
|
269
|
+
expect(out).toContain('| --- | --- | --- | --- | --- | --- |');
|
|
270
|
+
// The single collapsed Status column is gone.
|
|
271
|
+
expect(out).not.toContain('| 7d | Status |');
|
|
270
272
|
// No legacy group headers / health-section titles remain.
|
|
271
273
|
expect(out).not.toContain('**BLOCKED**');
|
|
272
274
|
expect(out).not.toContain('**HEALTHY**');
|
|
@@ -307,16 +309,22 @@ describe('renderAuthSnapshotFormat2', () => {
|
|
|
307
309
|
expect(bob).toBeLessThan(alice);
|
|
308
310
|
});
|
|
309
311
|
|
|
310
|
-
it('
|
|
312
|
+
it('reset columns show a "<time> (in ...)" cell for a blocked account (7d binding window)', () => {
|
|
311
313
|
const rows = tableRows(renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' }));
|
|
312
314
|
const bob = rows.find((r) => r[1].includes('bob@example.com'))!;
|
|
313
|
-
|
|
315
|
+
// bob is 7d-maxed; its 7d reset (2026-05-17T10:00Z) is ~2 days out and still
|
|
316
|
+
// renders in its own reset cell ā no "back" prefix in the new per-window shape.
|
|
317
|
+
expect(bob[5]).toMatch(/^.* \(in .+\)$/);
|
|
318
|
+
expect(bob[5]).not.toContain('back');
|
|
314
319
|
});
|
|
315
320
|
|
|
316
|
-
it('
|
|
321
|
+
it('reset columns show a "<time> (in ...)" cell for each window of a healthy account', () => {
|
|
317
322
|
const rows = tableRows(renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' }));
|
|
318
323
|
const you = rows.find((r) => r[1].includes('you@example.com'))!;
|
|
319
|
-
|
|
324
|
+
// 5h resets cell [3] and 7d resets cell [5] both carry a relative hint.
|
|
325
|
+
expect(you[3]).toMatch(/^.* \(in .+\)$/);
|
|
326
|
+
expect(you[5]).toMatch(/^.* \(in .+\)$/);
|
|
327
|
+
expect(you[3]).not.toContain('refills');
|
|
320
328
|
});
|
|
321
329
|
|
|
322
330
|
it('NEVER displays a percentage over 100% even on an over-cap account', () => {
|
|
@@ -330,13 +338,15 @@ describe('renderAuthSnapshotFormat2', () => {
|
|
|
330
338
|
];
|
|
331
339
|
const rows = tableRows(renderAuthSnapshotFormat2(overSnaps, { now: NOW, tz: 'UTC' }));
|
|
332
340
|
expect(rows[0][2]).toBe('100%'); // 5h
|
|
333
|
-
expect(rows[0][
|
|
334
|
-
// And the blocked state still surfaces via the emoji (š“)
|
|
341
|
+
expect(rows[0][4]).toBe('100%'); // 7d
|
|
342
|
+
// And the blocked state still surfaces via the emoji (š“). No resets known
|
|
343
|
+
// (fixture has no reset timestamps) ā both reset cells degrade to "ā".
|
|
335
344
|
expect(rows[0][0]).toBe('š“');
|
|
336
|
-
expect(rows[0][
|
|
345
|
+
expect(rows[0][3]).toBe('ā'); // 5h resets
|
|
346
|
+
expect(rows[0][5]).toBe('ā'); // 7d resets
|
|
337
347
|
});
|
|
338
348
|
|
|
339
|
-
it('renders dates in the
|
|
349
|
+
it('renders dates in the reset columns only when NOT today (user tz)', () => {
|
|
340
350
|
// now = Fri 3:00 PM Melbourne. A reset later the SAME Melbourne day shows
|
|
341
351
|
// time-only; a reset on the next day shows the weekday.
|
|
342
352
|
const MEL = 'Australia/Melbourne';
|
|
@@ -349,8 +359,9 @@ describe('renderAuthSnapshotFormat2', () => {
|
|
|
349
359
|
}) }),
|
|
350
360
|
];
|
|
351
361
|
const todayRows = tableRows(renderAuthSnapshotFormat2(sameDay, { now: NOW_MEL, tz: MEL }));
|
|
352
|
-
|
|
353
|
-
expect(todayRows[0][
|
|
362
|
+
// 5h resets cell [3] ā same Melbourne day ā time only.
|
|
363
|
+
expect(todayRows[0][3]).toContain('9:00 PM');
|
|
364
|
+
expect(todayRows[0][3]).not.toMatch(/Mon|Tue|Wed|Thu|Fri|Sat|Sun/);
|
|
354
365
|
|
|
355
366
|
const nextDay = [
|
|
356
367
|
snap({ label: 'tomorrow@example.com', isActive: true, quota: quota({
|
|
@@ -360,7 +371,8 @@ describe('renderAuthSnapshotFormat2', () => {
|
|
|
360
371
|
}) }),
|
|
361
372
|
];
|
|
362
373
|
const tomorrowRows = tableRows(renderAuthSnapshotFormat2(nextDay, { now: NOW_MEL, tz: MEL }));
|
|
363
|
-
|
|
374
|
+
// 5h resets cell [3] ā next Melbourne day ā weekday prefix.
|
|
375
|
+
expect(tomorrowRows[0][3]).toContain('Sat 1:00 AM');
|
|
364
376
|
});
|
|
365
377
|
|
|
366
378
|
it('emits a recommendation footer that names a healthy alternative when active is throttling', () => {
|
|
@@ -1120,14 +1132,16 @@ describe('#2494 ā renderAuthSnapshotFormat2 row rendering (out_of_credits demo
|
|
|
1120
1132
|
expect(allText).toContain('Switch fleet ā carol@example.com');
|
|
1121
1133
|
});
|
|
1122
1134
|
|
|
1123
|
-
it('quota-exhausted row shows a š“ +
|
|
1135
|
+
it('quota-exhausted row shows a š“ + its 5h reset time in the reset cell, not "billing disabled"', () => {
|
|
1124
1136
|
const futureReset = new Date(NOW.getTime() + 45 * 60_000);
|
|
1125
1137
|
const out = renderAuthSnapshotFormat2(
|
|
1126
1138
|
[snap({ label: 'ex@x', isActive: true, quota: quota({ fiveHourUtilizationPct: 100, fiveHourResetAt: futureReset }) })],
|
|
1127
1139
|
{ now: NOW },
|
|
1128
1140
|
);
|
|
1129
1141
|
expect(out).toContain('| š“ |');
|
|
1130
|
-
|
|
1142
|
+
// The 5h reset (45m out) renders in its own reset cell ā no "back" prefix.
|
|
1143
|
+
expect(out).toMatch(/\(in 45m\)/);
|
|
1144
|
+
expect(out).not.toContain('back ');
|
|
1131
1145
|
expect(out).not.toContain('billing disabled');
|
|
1132
1146
|
expect(out).not.toContain('overage off');
|
|
1133
1147
|
});
|
|
@@ -137,6 +137,54 @@ describe('Claude Code event-stream contract (canary)', () => {
|
|
|
137
137
|
).toBe(false)
|
|
138
138
|
})
|
|
139
139
|
|
|
140
|
+
it('CANARY (Fix 3 precondition): Task tool_use still projects {id, name, input} ā the shape the foreground-sub-agent-tracking / post-answer-liveness-freeze fix keys "still dispatched" off of', () => {
|
|
141
|
+
// The Fix 3 gap-closer (turn-liveness-floor.ts `stillDispatched`) is only
|
|
142
|
+
// as good as `turn.foregroundSubAgents` staying populated for the
|
|
143
|
+
// lifetime of a real Task dispatch. That population is driven by THIS
|
|
144
|
+
// exact tool_use projection (toolName === 'Task') plus the matching
|
|
145
|
+
// tool_result closing it out ā if a future Claude Code release renames
|
|
146
|
+
// `name`/`input`/`id` or nests them differently, the fix silently goes
|
|
147
|
+
// inert (stillDispatched always false, freeze regresses) rather than
|
|
148
|
+
// failing loudly. Pin the shape here so drift fails CI instead.
|
|
149
|
+
const line = JSON.stringify({
|
|
150
|
+
type: 'assistant',
|
|
151
|
+
message: {
|
|
152
|
+
id: 'msg_task',
|
|
153
|
+
stop_reason: 'tool_use',
|
|
154
|
+
content: [
|
|
155
|
+
{
|
|
156
|
+
type: 'tool_use',
|
|
157
|
+
id: 'toolu_task_1',
|
|
158
|
+
name: 'Task',
|
|
159
|
+
input: { description: 'Investigate the freeze bug', subagent_type: 'worker' },
|
|
160
|
+
},
|
|
161
|
+
],
|
|
162
|
+
},
|
|
163
|
+
})
|
|
164
|
+
const evs = projectTranscriptLine(line)
|
|
165
|
+
expect(evs).toEqual([
|
|
166
|
+
{
|
|
167
|
+
kind: 'tool_use',
|
|
168
|
+
toolName: 'Task',
|
|
169
|
+
toolUseId: 'toolu_task_1',
|
|
170
|
+
input: { description: 'Investigate the freeze bug', subagent_type: 'worker' },
|
|
171
|
+
},
|
|
172
|
+
])
|
|
173
|
+
// The matching tool_result (Task returning) must still carry tool_use_id
|
|
174
|
+
// so the gateway can close out the foreground-sub-agent tracking entry.
|
|
175
|
+
const resultLine = JSON.stringify({
|
|
176
|
+
type: 'user',
|
|
177
|
+
message: { content: [{ type: 'tool_result', tool_use_id: 'toolu_task_1', is_error: false, content: 'done' }] },
|
|
178
|
+
})
|
|
179
|
+
const resultEvs = projectTranscriptLine(resultLine)
|
|
180
|
+
expect(resultEvs).toHaveLength(1)
|
|
181
|
+
// `is_error: false` projects to `isError: undefined` (only `true` is
|
|
182
|
+
// ever stamped ā see session-tail.ts's `c.is_error === true ? true :
|
|
183
|
+
// undefined`), so assert the falsy/absent form, not a literal `false`.
|
|
184
|
+
expect(resultEvs[0]).toMatchObject({ kind: 'tool_result', toolUseId: 'toolu_task_1' })
|
|
185
|
+
expect((resultEvs[0] as { isError?: boolean }).isError).toBeFalsy()
|
|
186
|
+
})
|
|
187
|
+
|
|
140
188
|
it('sub-agent kickoff: first user message string prompt fires sub_agent_started', () => {
|
|
141
189
|
const st = { hasEmittedStart: false }
|
|
142
190
|
const line = JSON.stringify({
|
|
@@ -153,6 +153,17 @@ describe('H-2: feedHeartbeatTick post-answer background-agent liveness (Fix 2 /
|
|
|
153
153
|
expect(postAnswerBlock).toMatch(/livenessVerdict\s*!==\s*'emit'/)
|
|
154
154
|
})
|
|
155
155
|
|
|
156
|
+
it('Fix 3: post-answer branch feeds stillDispatched from turn.foregroundSubAgents so a tracked foreground sub-agent bypasses the staleness cap', () => {
|
|
157
|
+
const body = feedHeartbeatTickSrc()
|
|
158
|
+
const afterPostAnswer = body.split('if (turn.finalAnswerDelivered)')[1] ?? ''
|
|
159
|
+
const postAnswerBlock = afterPostAnswer.split('\n }\n')[0] ?? ''
|
|
160
|
+
// A positive foreground-tracking signal must be computed and threaded
|
|
161
|
+
// into the pure decision ā this is what stops the card freezing mid-
|
|
162
|
+
// delegation while a foreground Task/Agent is still outstanding.
|
|
163
|
+
expect(postAnswerBlock).toMatch(/stillDispatched\s*=\s*turn\.foregroundSubAgents\.size\s*>\s*0/)
|
|
164
|
+
expect(postAnswerBlock).toMatch(/stillDispatched,/)
|
|
165
|
+
})
|
|
166
|
+
|
|
156
167
|
it('staleness cap (concern 3) is parsed default-ON (30s) from SWITCHROOM_POST_ANSWER_LIVENESS_STALE_MS', () => {
|
|
157
168
|
// The cap const must default to 30_000 when the env is unset (the `|| 30_000`
|
|
158
169
|
// fallback over the positive-or-0 parse) so the post-answer card stops
|
|
@@ -47,6 +47,8 @@ import { ToolFlightTracker } from '../gateway/interrupt-defer.js'
|
|
|
47
47
|
import {
|
|
48
48
|
ORPHANED_REPLY_TIMEOUT_MS,
|
|
49
49
|
ORPHANED_REPLY_MAX_REARMS,
|
|
50
|
+
ORPHANED_REPLY_STREAM_WINDOW_MS,
|
|
51
|
+
LivenessTracker,
|
|
50
52
|
} from '../context-exhaustion.js'
|
|
51
53
|
|
|
52
54
|
// āāā Helpers āāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāāā
|
|
@@ -524,3 +526,40 @@ describe('silence-poke ā hard ceiling bounds the defer', () => {
|
|
|
524
526
|
expect(__getStateForTests('chat:0')!.fallbackFired).toBe(true)
|
|
525
527
|
})
|
|
526
528
|
})
|
|
529
|
+
|
|
530
|
+
// āāā Thinking-pause liveness via the REAL LivenessTracker (orphaned-reply fix) ā
|
|
531
|
+
|
|
532
|
+
describe('silence-poke ā thinking-pause liveness via the real LivenessTracker', () => {
|
|
533
|
+
// Drives the REAL LivenessTracker as the isLegitimatelyWorking source: a
|
|
534
|
+
// genuine stream event within ORPHANED_REPLY_STREAM_WINDOW_MS keeps the feed
|
|
535
|
+
// alive (a model reasoning pause is survivable); a gap longer than the window
|
|
536
|
+
// with no work tears the feed down.
|
|
537
|
+
it('a stream event within the window keeps the feed alive (thinking-pause survivable)', () => {
|
|
538
|
+
let clockNow = 0
|
|
539
|
+
const tracker = new LivenessTracker(0)
|
|
540
|
+
const f = setupSilenceDeps({
|
|
541
|
+
thresholds: { fallback: 300_000, fallbackHardCeiling: 900_000 },
|
|
542
|
+
isLegitimatelyWorking: () => tracker.recentlyStreaming(clockNow, ORPHANED_REPLY_STREAM_WINDOW_MS),
|
|
543
|
+
})
|
|
544
|
+
startTurn('chat:0', 0)
|
|
545
|
+
// A genuine stream event lands mid-turn (e.g. the model resumes after a
|
|
546
|
+
// reasoning pause) ā only 50s before the 300s fallback tick.
|
|
547
|
+
tracker.onStreamEvent('text', undefined, 250_000)
|
|
548
|
+
clockNow = 300_000
|
|
549
|
+
__tickForTests(300_000) // 50s since last stream < 120s window ā deferred
|
|
550
|
+
expect(f.fallbacks).toHaveLength(0)
|
|
551
|
+
})
|
|
552
|
+
|
|
553
|
+
it('no stream event for longer than the window tears the feed down', () => {
|
|
554
|
+
let clockNow = 0
|
|
555
|
+
const tracker = new LivenessTracker(0) // seed at t=0, nothing after
|
|
556
|
+
const f = setupSilenceDeps({
|
|
557
|
+
thresholds: { fallback: 300_000, fallbackHardCeiling: 900_000 },
|
|
558
|
+
isLegitimatelyWorking: () => tracker.recentlyStreaming(clockNow, ORPHANED_REPLY_STREAM_WINDOW_MS),
|
|
559
|
+
})
|
|
560
|
+
startTurn('chat:0', 0)
|
|
561
|
+
clockNow = 300_000
|
|
562
|
+
__tickForTests(300_000) // 300s since last stream >> 120s window ā fires
|
|
563
|
+
expect(f.fallbacks).toHaveLength(1)
|
|
564
|
+
})
|
|
565
|
+
})
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural pins for the session-only model relaunch wiring in gateway.ts.
|
|
3
|
+
*
|
|
4
|
+
* The behaviour lives in un-exported inline closures (buildModelDeps's
|
|
5
|
+
* scheduleModelRelaunch, the model-menu callback sr-* branch, and the boot
|
|
6
|
+
* re-hydration / alert-sentinel block inside the startup IIFE), so ā mirroring
|
|
7
|
+
* the other gateway-*.test.ts source-pins ā we assert on the source structure.
|
|
8
|
+
* The end-to-end behaviour of the carrier itself is exercised in
|
|
9
|
+
* tests/scaffold.session-model-override.test.ts (rendered start.sh) and the
|
|
10
|
+
* handler contract in telegram-plugin/tests/model-command.test.ts.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
import { describe, it, expect } from 'vitest'
|
|
14
|
+
import { readFileSync } from 'node:fs'
|
|
15
|
+
import { fileURLToPath } from 'node:url'
|
|
16
|
+
import { dirname, resolve } from 'node:path'
|
|
17
|
+
|
|
18
|
+
const __dirname = dirname(fileURLToPath(import.meta.url))
|
|
19
|
+
const GATEWAY_SRC = readFileSync(resolve(__dirname, '..', 'gateway', 'gateway.ts'), 'utf8')
|
|
20
|
+
|
|
21
|
+
describe('gateway: scheduleModelRelaunch dep', () => {
|
|
22
|
+
it('writes the carrier with exact "<model>\\n" bytes', () => {
|
|
23
|
+
const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
|
|
24
|
+
expect(idx).toBeGreaterThan(0)
|
|
25
|
+
const win = GATEWAY_SRC.slice(idx, idx + 900)
|
|
26
|
+
expect(win).toMatch(/writeFileSync\(\s*join\(agentDir, '\.session-model-override'\),\s*`\$\{model\}\\n`/)
|
|
27
|
+
})
|
|
28
|
+
|
|
29
|
+
it('sets the in-memory activeSessionModelOverride before dispatching the restart', () => {
|
|
30
|
+
const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
|
|
31
|
+
const win = GATEWAY_SRC.slice(idx, idx + 900)
|
|
32
|
+
const setIdx = win.indexOf('activeSessionModelOverride = model')
|
|
33
|
+
const restartIdx = win.indexOf('deps.scheduleRestart(reason)')
|
|
34
|
+
expect(setIdx).toBeGreaterThan(0)
|
|
35
|
+
expect(restartIdx).toBeGreaterThan(setIdx)
|
|
36
|
+
})
|
|
37
|
+
|
|
38
|
+
it('reuses the same scheduleRestart dispatch (not a bespoke restart path)', () => {
|
|
39
|
+
const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
|
|
40
|
+
const win = GATEWAY_SRC.slice(idx, idx + 900)
|
|
41
|
+
expect(win).toContain('await deps.scheduleRestart(reason)')
|
|
42
|
+
})
|
|
43
|
+
})
|
|
44
|
+
|
|
45
|
+
describe('gateway: model-menu sr-* target relaunches via the carrier', () => {
|
|
46
|
+
it('the sr-* callback branch calls scheduleModelRelaunch, not inject', () => {
|
|
47
|
+
const idx = GATEWAY_SRC.indexOf('if (data.startsWith(MODEL_CALLBACK_SR))')
|
|
48
|
+
expect(idx).toBeGreaterThan(0)
|
|
49
|
+
const win = GATEWAY_SRC.slice(idx, idx + 1400)
|
|
50
|
+
expect(win).toContain('modelDeps.scheduleModelRelaunch(srName')
|
|
51
|
+
// It must return before falling through to handleModelMenuCallback (which
|
|
52
|
+
// would inject an sr-* id claude's picker rejects).
|
|
53
|
+
expect(win).toMatch(/scheduleModelRelaunch[\s\S]*?\n\s*return\n/)
|
|
54
|
+
})
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
describe('gateway boot: session-model re-hydration + LiteLLM-down alert', () => {
|
|
58
|
+
it('re-hydrates activeSessionModelOverride from .active-session-model', () => {
|
|
59
|
+
const idx = GATEWAY_SRC.indexOf("join(smAgentDir, '.active-session-model')")
|
|
60
|
+
expect(idx).toBeGreaterThan(0)
|
|
61
|
+
const win = GATEWAY_SRC.slice(idx - 200, idx + 1400)
|
|
62
|
+
// Only an override when the launched model differs from the configured one.
|
|
63
|
+
expect(win).toMatch(/launched\.length > 0 && launched !== configured \? launched : null/)
|
|
64
|
+
// The configured value must be resolved through resolveMainModel (the SAME
|
|
65
|
+
// resolver start.sh's scaffold uses) so an unset/`default` model config does
|
|
66
|
+
// not get flagged as a phantom session override on an ordinary restart.
|
|
67
|
+
expect(win).toContain('resolveMainModel(raw ?? undefined)')
|
|
68
|
+
})
|
|
69
|
+
|
|
70
|
+
it('consumes the .session-model-alert sentinel, notifies ALL operators, and deletes it', () => {
|
|
71
|
+
const idx = GATEWAY_SRC.indexOf("join(smAgentDir, '.session-model-alert')")
|
|
72
|
+
expect(idx).toBeGreaterThan(0)
|
|
73
|
+
const win = GATEWAY_SRC.slice(idx, idx + 1100)
|
|
74
|
+
expect(win).toContain('unlinkSync(alertPath)')
|
|
75
|
+
// Broadcasts to every operator in allowFrom, not just allowFrom[0].
|
|
76
|
+
expect(win).toContain('const operators = loadAccess().allowFrom')
|
|
77
|
+
expect(win).toContain('for (const operator of operators)')
|
|
78
|
+
expect(win).toContain('lockedBot.api')
|
|
79
|
+
expect(win).toContain('.sendMessage(operator')
|
|
80
|
+
})
|
|
81
|
+
})
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural pin: the delivery state machine's `inbound` event MUST be
|
|
3
|
+
* emitted from `handleInbound` only AFTER the intercept/early-return
|
|
4
|
+
* gauntlet ā never eagerly at handler entry.
|
|
5
|
+
*
|
|
6
|
+
* The wedge this guards (overlord `/usage` went dead, 2026-07-08). The
|
|
7
|
+
* inbound emit drives the now-AUTHORITATIVE turn-in-flight gate
|
|
8
|
+
* (turnInFlightForGate ā isMachineInTurn). When it fired at handler entry
|
|
9
|
+
* ā before the permission-reply / `/auth` paste-back / interrupt-empty /
|
|
10
|
+
* secret-detect-drop intercepts ā an intercepted message (e.g. an approval
|
|
11
|
+
* reply) drove the machine idleā`bridge_alive_in_turn`, then early-returned
|
|
12
|
+
* as an intercept: no delivery, no claudeBusyKeys mark, and critically no
|
|
13
|
+
* `turnEnd`. The machine held the gate closed until the 5-min TTL tick
|
|
14
|
+
* force-cleared it, buffering every inbound (including `/usage`) meanwhile ā
|
|
15
|
+
* the dangerous `machine_over_holds` divergence gate-parity-probe.ts flags.
|
|
16
|
+
*
|
|
17
|
+
* The behaviour lives inside the un-exported `handleInbound` closure, so ā
|
|
18
|
+
* mirroring the other gateway-*.test.ts source-pins ā we assert on source
|
|
19
|
+
* structure: the single inbound `shadowEmit` must sit below the intercepts
|
|
20
|
+
* and carry the real `isSteering` classification. Machine self-heal + gate
|
|
21
|
+
* accessors are behaviour-tested in inbound-delivery-cutover-gate.test.ts.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { describe, it, expect } from 'vitest'
|
|
25
|
+
import { readFileSync } from 'node:fs'
|
|
26
|
+
import { fileURLToPath } from 'node:url'
|
|
27
|
+
import { dirname, resolve } from 'node:path'
|
|
28
|
+
|
|
29
|
+
const __dirname = dirname(fileURLToPath(import.meta.url))
|
|
30
|
+
const GATEWAY_SRC = readFileSync(resolve(__dirname, '..', 'gateway', 'gateway.ts'), 'utf8')
|
|
31
|
+
|
|
32
|
+
/** Byte offset of the sole `shadowEmit({ kind: 'inbound' ... })` call. */
|
|
33
|
+
function inboundEmitOffset(): number {
|
|
34
|
+
const m = GATEWAY_SRC.match(/shadowEmit\(\{\s*\n\s*kind:\s*'inbound'/)
|
|
35
|
+
expect(m, 'expected exactly one multi-line inbound shadowEmit').not.toBeNull()
|
|
36
|
+
const idx = GATEWAY_SRC.indexOf(m![0])
|
|
37
|
+
// There must be only ONE such emit ā a second eager site would reintroduce
|
|
38
|
+
// the wedge.
|
|
39
|
+
const second = GATEWAY_SRC.indexOf(m![0], idx + 1)
|
|
40
|
+
expect(second, 'a second inbound shadowEmit would reintroduce the eager-emit wedge').toBe(-1)
|
|
41
|
+
return idx
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
describe('gateway: inbound machine-emit is deferred past the intercepts', () => {
|
|
45
|
+
it('carries the DEFERRED_INBOUND_EMIT marker so the intent is greppable', () => {
|
|
46
|
+
expect(GATEWAY_SRC).toContain('DEFERRED_INBOUND_EMIT')
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
it('emits the machine `inbound` event AFTER the permission-reply intercept', () => {
|
|
50
|
+
const emitIdx = inboundEmitOffset()
|
|
51
|
+
// The permission-reply intercept is one of the early-return paths that
|
|
52
|
+
// must NOT drive the machine into a turn.
|
|
53
|
+
const permIdx = GATEWAY_SRC.indexOf('const permMatch = PERMISSION_REPLY_RE.exec(text)')
|
|
54
|
+
expect(permIdx).toBeGreaterThan(0)
|
|
55
|
+
expect(emitIdx).toBeGreaterThan(permIdx)
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
it('emits AFTER the `/auth add` paste-back intercept', () => {
|
|
59
|
+
const emitIdx = inboundEmitOffset()
|
|
60
|
+
const authIdx = GATEWAY_SRC.indexOf('pendingAuthAddFlows.get(interceptKey)')
|
|
61
|
+
expect(authIdx).toBeGreaterThan(0)
|
|
62
|
+
expect(emitIdx).toBeGreaterThan(authIdx)
|
|
63
|
+
})
|
|
64
|
+
|
|
65
|
+
it('emits AFTER isSteering is classified, and passes the real value (not a hardcoded false)', () => {
|
|
66
|
+
const emitIdx = inboundEmitOffset()
|
|
67
|
+
const steerIdx = GATEWAY_SRC.indexOf('isSteering = priorTurnInFlight && isSteerPrefix')
|
|
68
|
+
expect(steerIdx).toBeGreaterThan(0)
|
|
69
|
+
expect(emitIdx).toBeGreaterThan(steerIdx)
|
|
70
|
+
// The emit's msg object must forward the live `isSteering` binding, so a
|
|
71
|
+
// mid-turn steer is not mis-modelled as a fresh turn.
|
|
72
|
+
const win = GATEWAY_SRC.slice(emitIdx, emitIdx + 260)
|
|
73
|
+
expect(win).toMatch(/msg:\s*\{[\s\S]*?\bisSteering,[\s\S]*?\}/)
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
it('emits BEFORE the deliver-or-buffer decision (reserveInboundDelivery)', () => {
|
|
77
|
+
const emitIdx = inboundEmitOffset()
|
|
78
|
+
const gateIdx = GATEWAY_SRC.indexOf('const deliveryGate = reserveInboundDelivery({')
|
|
79
|
+
expect(gateIdx).toBeGreaterThan(0)
|
|
80
|
+
expect(emitIdx).toBeLessThan(gateIdx)
|
|
81
|
+
})
|
|
82
|
+
})
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* liveness-tracker.test.ts ā drives the REAL LivenessTracker (the testability
|
|
3
|
+
* seam of the orphaned-reply thinking-pause fix).
|
|
4
|
+
*
|
|
5
|
+
* Root cause being fixed: the orphaned-reply fuse (ORPHANED_REPLY_TIMEOUT_MS =
|
|
6
|
+
* 30 s) was reset ONLY by `tool_label` / `text` events. During a long model
|
|
7
|
+
* reasoning pause the gateway sees NO such events (thinking carries no text),
|
|
8
|
+
* so the fuse ran down and force-ended a genuinely-live turn mid-work.
|
|
9
|
+
*
|
|
10
|
+
* The fix stamps `lastStreamEventAt` on ANY genuine stream event; if a genuine
|
|
11
|
+
* event landed within `windowMs` (default 120 s) the turn is "recently
|
|
12
|
+
* streaming" and the fuse re-arms instead of firing. A genuine multi-minute
|
|
13
|
+
* hang (no events at all) still fires.
|
|
14
|
+
*
|
|
15
|
+
* These tests use the REAL class ā no replica helpers.
|
|
16
|
+
*
|
|
17
|
+
* NOTE on onStreamEvent's signature: it is `(kind, durationMs, now)`. The
|
|
18
|
+
* tracker is time-only ā it never reads text content (the "Prompt is too long"
|
|
19
|
+
* marker matters to the gateway's context-exhaustion detection, not to the
|
|
20
|
+
* tracker). Where the fix spec wrote `onStreamEvent('text','Prompt is too
|
|
21
|
+
* long',0)` we pass `undefined` as durationMs (it is not a turn_end).
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
import { describe, it, expect } from 'vitest'
|
|
25
|
+
import {
|
|
26
|
+
LivenessTracker,
|
|
27
|
+
ORPHANED_REPLY_STREAM_WINDOW_MS,
|
|
28
|
+
ORPHANED_REPLY_TIMEOUT_MS,
|
|
29
|
+
ORPHANED_REPLY_MAX_REARMS,
|
|
30
|
+
} from '../context-exhaustion.js'
|
|
31
|
+
|
|
32
|
+
const W = ORPHANED_REPLY_STREAM_WINDOW_MS // 120_000
|
|
33
|
+
const MAX = ORPHANED_REPLY_MAX_REARMS // 20
|
|
34
|
+
const FUSE = ORPHANED_REPLY_TIMEOUT_MS // 30_000
|
|
35
|
+
|
|
36
|
+
function expiry(
|
|
37
|
+
t: LivenessTracker,
|
|
38
|
+
now: number,
|
|
39
|
+
opts?: { working?: boolean; humanWaiting?: boolean },
|
|
40
|
+
) {
|
|
41
|
+
return t.decideOnExpiry({
|
|
42
|
+
working: opts?.working ?? false,
|
|
43
|
+
humanWaiting: opts?.humanWaiting ?? false,
|
|
44
|
+
now,
|
|
45
|
+
windowMs: W,
|
|
46
|
+
maxRearms: MAX,
|
|
47
|
+
})
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
describe('LivenessTracker ā pinned constants', () => {
|
|
51
|
+
it('window is 120 s and fuse tick is 30 s', () => {
|
|
52
|
+
expect(W).toBe(120_000)
|
|
53
|
+
expect(FUSE).toBe(30_000)
|
|
54
|
+
expect(MAX).toBe(20)
|
|
55
|
+
})
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
describe('T1 ā thinking-pause survival', () => {
|
|
59
|
+
it('textātool_useātool_result then a 45s stream gap keeps re-arming (gap < window)', () => {
|
|
60
|
+
const t = new LivenessTracker(0)
|
|
61
|
+
t.onStreamEvent('text', undefined, 0)
|
|
62
|
+
t.onStreamEvent('tool_use', undefined, 1_000)
|
|
63
|
+
t.onStreamEvent('tool_result', undefined, 2_000)
|
|
64
|
+
|
|
65
|
+
// +30s past the last event (t=32s): recently streaming ā re-arm.
|
|
66
|
+
const d1 = expiry(t, 32_000)
|
|
67
|
+
expect(d1).toEqual({ rearm: true, countsAgainstCap: true })
|
|
68
|
+
expect(t.orphanedReplyRearmCount).toBe(1)
|
|
69
|
+
|
|
70
|
+
// +45s past the last event (t=47s): still within the 120s window ā re-arm.
|
|
71
|
+
const d2 = expiry(t, 47_000)
|
|
72
|
+
expect(d2).toEqual({ rearm: true, countsAgainstCap: true })
|
|
73
|
+
expect(t.orphanedReplyRearmCount).toBe(2)
|
|
74
|
+
})
|
|
75
|
+
|
|
76
|
+
it('the next genuine stream event ZEROES the rearm counter and re-stamps liveness', () => {
|
|
77
|
+
const t = new LivenessTracker(0)
|
|
78
|
+
t.onStreamEvent('tool_result', undefined, 2_000)
|
|
79
|
+
expiry(t, 32_000)
|
|
80
|
+
expiry(t, 47_000)
|
|
81
|
+
expect(t.orphanedReplyRearmCount).toBe(2)
|
|
82
|
+
|
|
83
|
+
t.onStreamEvent('tool_label', undefined, 48_000)
|
|
84
|
+
expect(t.orphanedReplyRearmCount).toBe(0)
|
|
85
|
+
expect(t.lastStreamEventAt).toBe(48_000)
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
it('a synthetic turn_end (durationMs === -1) does NOT stamp or reset (F4 boundary)', () => {
|
|
89
|
+
const t = new LivenessTracker(0)
|
|
90
|
+
t.onStreamEvent('tool_label', undefined, 48_000)
|
|
91
|
+
expiry(t, 60_000) // counter ā 1
|
|
92
|
+
expect(t.orphanedReplyRearmCount).toBe(1)
|
|
93
|
+
|
|
94
|
+
t.onStreamEvent('turn_end', -1, 70_000)
|
|
95
|
+
expect(t.lastStreamEventAt).toBe(48_000) // unchanged ā synthetic excluded
|
|
96
|
+
expect(t.orphanedReplyRearmCount).toBe(1) // not reset
|
|
97
|
+
|
|
98
|
+
// A REAL turn_end (durationMs >= 0) DOES stamp ā harmless, the turn is
|
|
99
|
+
// ending anyway (spec #3).
|
|
100
|
+
t.onStreamEvent('turn_end', 0, 71_000)
|
|
101
|
+
expect(t.lastStreamEventAt).toBe(71_000)
|
|
102
|
+
expect(t.orphanedReplyRearmCount).toBe(0)
|
|
103
|
+
})
|
|
104
|
+
})
|
|
105
|
+
|
|
106
|
+
describe('T2 ā production repro (48s tool_result, 78s expiry)', () => {
|
|
107
|
+
it('text@0, tool_use@1s, tool_result@48s ā expiry@78s re-arms (78-48=30 < 120)', () => {
|
|
108
|
+
const t = new LivenessTracker(0)
|
|
109
|
+
t.onStreamEvent('text', undefined, 0)
|
|
110
|
+
t.onStreamEvent('tool_use', undefined, 1_000)
|
|
111
|
+
t.onStreamEvent('tool_result', undefined, 48_000)
|
|
112
|
+
|
|
113
|
+
const d = expiry(t, 78_000)
|
|
114
|
+
expect(d.rearm).toBe(true)
|
|
115
|
+
expect(t.orphanedReplyRearmCount).toBe(1)
|
|
116
|
+
|
|
117
|
+
// tool_label@80s zeroes the counter (progress observed).
|
|
118
|
+
t.onStreamEvent('tool_label', undefined, 80_000)
|
|
119
|
+
expect(t.orphanedReplyRearmCount).toBe(0)
|
|
120
|
+
})
|
|
121
|
+
})
|
|
122
|
+
|
|
123
|
+
describe('T3 ā long active turn (A2: counter reset is load-bearing)', () => {
|
|
124
|
+
it('tool_label every 60s + expiry every 60s+30s NEVER fires across 15 min; counter never exceeds 1', () => {
|
|
125
|
+
const t = new LivenessTracker(0)
|
|
126
|
+
for (let k = 0; k < 15; k++) {
|
|
127
|
+
t.onStreamEvent('tool_label', undefined, k * 60_000) // progress ā reset
|
|
128
|
+
const d = expiry(t, k * 60_000 + 30_000, { working: true })
|
|
129
|
+
expect(d.rearm).toBe(true) // never fires
|
|
130
|
+
expect(t.orphanedReplyRearmCount).toBeLessThanOrEqual(1)
|
|
131
|
+
}
|
|
132
|
+
})
|
|
133
|
+
|
|
134
|
+
it('WITHOUT the progress reset the SAME cadence hits cap=20 near t=600s (proves reset load-bearing)', () => {
|
|
135
|
+
const t = new LivenessTracker(0)
|
|
136
|
+
let firedAt = -1
|
|
137
|
+
for (let k = 1; k <= 25; k++) {
|
|
138
|
+
const now = k * 30_000
|
|
139
|
+
// working:true so it would re-arm indefinitely IF the counter didn't
|
|
140
|
+
// accumulate ā but with no onStreamEvent reset it climbs to the cap.
|
|
141
|
+
const d = expiry(t, now, { working: true })
|
|
142
|
+
if (!d.rearm) {
|
|
143
|
+
firedAt = now
|
|
144
|
+
break
|
|
145
|
+
}
|
|
146
|
+
}
|
|
147
|
+
// 20th re-arm lands at t=600s; the 21st expiry (t=630s) fires at the cap.
|
|
148
|
+
expect(t.orphanedReplyRearmCount).toBe(MAX)
|
|
149
|
+
expect(firedAt).toBeGreaterThanOrEqual(600_000)
|
|
150
|
+
})
|
|
151
|
+
})
|
|
152
|
+
|
|
153
|
+
describe('T4 ā genuine hang caught (no events at all)', () => {
|
|
154
|
+
it('tool_result@0 then silence: re-arms via recentlyStreaming @30/60/90s, FIRES @120s (window lapsed)', () => {
|
|
155
|
+
const t = new LivenessTracker(0)
|
|
156
|
+
t.onStreamEvent('tool_result', undefined, 0)
|
|
157
|
+
|
|
158
|
+
for (const now of [30_000, 60_000, 90_000]) {
|
|
159
|
+
const d = expiry(t, now) // working:false ā kept alive only by recentlyStreaming
|
|
160
|
+
expect(d.rearm).toBe(true)
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// t=120s: now - lastStreamEventAt = 120_000 >= window ā not recently
|
|
164
|
+
// streaming, working false ā fail-safe FIRE.
|
|
165
|
+
const d = expiry(t, 120_000)
|
|
166
|
+
expect(d).toEqual({ rearm: false, countsAgainstCap: false })
|
|
167
|
+
})
|
|
168
|
+
})
|
|
169
|
+
|
|
170
|
+
describe('T5 ā context-exhaustion recovery COMPLETES (not suppressed forever)', () => {
|
|
171
|
+
it('a "Prompt is too long" text event stamps liveness; defensive-guard predicate is true@30s, false@>=120s', () => {
|
|
172
|
+
const t = new LivenessTracker(0)
|
|
173
|
+
// The marker is a genuine `text` event (content is not a tracker concern).
|
|
174
|
+
t.onStreamEvent('text', undefined, 0)
|
|
175
|
+
|
|
176
|
+
// Context exhaustion means no tool is in flight: working === false. The
|
|
177
|
+
// defensive-guard suppression predicate is `working || recentlyStreaming`.
|
|
178
|
+
const working = false
|
|
179
|
+
|
|
180
|
+
// @30s: recentlyStreaming true ā synthetic turn_end is suppressed. This
|
|
181
|
+
// DOCUMENTS the accepted ~30sā~120s recovery-latency growth (F3).
|
|
182
|
+
expect(working || t.recentlyStreaming(30_000, W)).toBe(true)
|
|
183
|
+
|
|
184
|
+
// @120s: window lapsed ā NOT suppressed ā teardown COMPLETES.
|
|
185
|
+
expect(working || t.recentlyStreaming(120_000, W)).toBe(false)
|
|
186
|
+
|
|
187
|
+
// @150s: still not suppressed ā recovery is delayed, never forever.
|
|
188
|
+
expect(working || t.recentlyStreaming(150_000, W)).toBe(false)
|
|
189
|
+
})
|
|
190
|
+
})
|
|
191
|
+
|
|
192
|
+
describe('T7 ā cap semantics', () => {
|
|
193
|
+
it('interleaved expiry/genuine-event: count 1 ā 0 ā 1 ā 2', () => {
|
|
194
|
+
const t = new LivenessTracker(0)
|
|
195
|
+
|
|
196
|
+
expect(expiry(t, 10_000).rearm).toBe(true)
|
|
197
|
+
expect(t.orphanedReplyRearmCount).toBe(1)
|
|
198
|
+
|
|
199
|
+
t.onStreamEvent('tool_label', undefined, 11_000) // genuine event resets
|
|
200
|
+
expect(t.orphanedReplyRearmCount).toBe(0)
|
|
201
|
+
|
|
202
|
+
expiry(t, 20_000)
|
|
203
|
+
expect(t.orphanedReplyRearmCount).toBe(1)
|
|
204
|
+
|
|
205
|
+
expiry(t, 30_000) // consecutive silence (last event 11_000, gap 19s < 120s)
|
|
206
|
+
expect(t.orphanedReplyRearmCount).toBe(2)
|
|
207
|
+
})
|
|
208
|
+
|
|
209
|
+
it('21 consecutive recently-streaming expiries (no reset): 21st fires at the cap', () => {
|
|
210
|
+
const t = new LivenessTracker(0)
|
|
211
|
+
let last: { rearm: boolean; countsAgainstCap: boolean } | undefined
|
|
212
|
+
for (let k = 1; k <= 21; k++) {
|
|
213
|
+
// now = k*1000 keeps the gap from the seed (0) under the 120s window,
|
|
214
|
+
// so recentlyStreaming stays true throughout ā the ONLY bound is the cap.
|
|
215
|
+
last = expiry(t, k * 1_000)
|
|
216
|
+
}
|
|
217
|
+
expect(last).toEqual({ rearm: false, countsAgainstCap: true })
|
|
218
|
+
expect(t.orphanedReplyRearmCount).toBe(MAX)
|
|
219
|
+
|
|
220
|
+
// humanWaiting past the cap ā uncapped re-arm.
|
|
221
|
+
const dHuman = expiry(t, 22_000, { humanWaiting: true })
|
|
222
|
+
expect(dHuman).toEqual({ rearm: true, countsAgainstCap: false })
|
|
223
|
+
|
|
224
|
+
// recentlyStreaming false + working false + humanWaiting false ā fail-safe FIRE.
|
|
225
|
+
const dFire = expiry(t, 200_000)
|
|
226
|
+
expect(dFire).toEqual({ rearm: false, countsAgainstCap: false })
|
|
227
|
+
})
|
|
228
|
+
})
|