switchroom 0.17.10 → 0.18.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/bin/workspace-dynamic-hook.sh +12 -13
  2. package/dist/agent-scheduler/index.js +27 -1
  3. package/dist/auth-broker/index.js +6161 -151
  4. package/dist/cli/notion-write-pretool.mjs +29 -2
  5. package/dist/cli/switchroom.js +578 -454
  6. package/dist/host-control/main.js +6182 -172
  7. package/dist/vault/approvals/kernel-server.js +5891 -164
  8. package/dist/vault/broker/server.js +6597 -881
  9. package/package.json +1 -1
  10. package/profiles/_base/settings.json.hbs +2 -2
  11. package/profiles/_base/start.sh.hbs +170 -21
  12. package/profiles/coding/CLAUDE.md.hbs +1 -1
  13. package/profiles/default/CLAUDE.md +2 -2
  14. package/profiles/default/CLAUDE.md.hbs +2 -2
  15. package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
  16. package/profiles/health-coach/CLAUDE.md.hbs +1 -1
  17. package/telegram-plugin/auth-snapshot-format.ts +22 -24
  18. package/telegram-plugin/context-exhaustion.ts +124 -0
  19. package/telegram-plugin/dist/gateway/gateway.js +24086 -8727
  20. package/telegram-plugin/gateway/activity-card-store.ts +76 -0
  21. package/telegram-plugin/gateway/gateway.ts +480 -85
  22. package/telegram-plugin/gateway/inbound-delivery-gate.ts +26 -0
  23. package/telegram-plugin/gateway/model-command.ts +70 -10
  24. package/telegram-plugin/package.json +6 -0
  25. package/telegram-plugin/quota-watch.ts +4 -6
  26. package/telegram-plugin/registry/turns-schema.test.ts +97 -0
  27. package/telegram-plugin/registry/turns-schema.ts +78 -0
  28. package/telegram-plugin/render/ir.ts +209 -0
  29. package/telegram-plugin/render/parse.ts +363 -0
  30. package/telegram-plugin/render/render.ts +440 -0
  31. package/telegram-plugin/render/rich-render.ts +72 -0
  32. package/telegram-plugin/stream-controller.ts +14 -3
  33. package/telegram-plugin/tests/activity-card-store.test.ts +94 -0
  34. package/telegram-plugin/tests/auth-command-format2.test.ts +1 -1
  35. package/telegram-plugin/tests/auth-snapshot-format.test.ts +30 -16
  36. package/telegram-plugin/tests/claude-code-event-contract.test.ts +48 -0
  37. package/telegram-plugin/tests/feed-heartbeat-liveness-open.test.ts +11 -0
  38. package/telegram-plugin/tests/feed-survival.test.ts +39 -0
  39. package/telegram-plugin/tests/gateway-session-model-relaunch.test.ts +81 -0
  40. package/telegram-plugin/tests/inbound-emit-after-intercepts.test.ts +82 -0
  41. package/telegram-plugin/tests/liveness-tracker.test.ts +228 -0
  42. package/telegram-plugin/tests/model-command.test.ts +193 -16
  43. package/telegram-plugin/tests/narrative-render.test.ts +125 -0
  44. package/telegram-plugin/tests/orphaned-reply-rearm.test.ts +123 -163
  45. package/telegram-plugin/tests/quota-watch.test.ts +1 -4
  46. package/telegram-plugin/tests/rapid-fire-delivery-ordering.test.ts +149 -0
  47. package/telegram-plugin/tests/render/parse-torture.test.ts +136 -0
  48. package/telegram-plugin/tests/render/parse.test.ts +393 -0
  49. package/telegram-plugin/tests/render/render.test.ts +436 -0
  50. package/telegram-plugin/tests/render/rich-render.test.ts +85 -0
  51. package/telegram-plugin/tests/telegram-activity-visibility-integration.test.ts +155 -1
  52. package/telegram-plugin/tests/worktree-watch-cwds.test.ts +98 -3
  53. package/telegram-plugin/turn-liveness-floor.ts +35 -1
  54. package/telegram-plugin/uat/scenarios/jtbd-rich-formatting-render-dm.test.ts +99 -7
  55. package/telegram-plugin/worktree-watch-cwds.ts +92 -17
  56. package/vendor/hindsight-memory/scripts/lib/client.py +11 -1
  57. package/vendor/hindsight-memory/scripts/lib/config.py +9 -2
  58. package/vendor/hindsight-memory/scripts/recall.py +64 -6
  59. package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +1 -0
  60. package/vendor/hindsight-memory/tests/test_client.py +43 -0
  61. package/vendor/hindsight-memory/tests/test_recall_precision.py +114 -0
@@ -262,11 +262,13 @@ describe('renderAuthSnapshotFormat2', () => {
262
262
  return rows;
263
263
  }
264
264
 
265
- it('renders a GFM table with the State/Account/5h/7d/Status header', () => {
265
+ it('renders a GFM table with the State/Account/5h/5h resets/7d/7d resets header', () => {
266
266
  const out = renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' });
267
267
  expect(out).toContain('šŸ”‹ **Auth — fleet status**');
268
- expect(out).toContain('| State | Account | 5h | 7d | Status |');
269
- expect(out).toContain('| --- | --- | --- | --- | --- |');
268
+ expect(out).toContain('| State | Account | 5h | 5h resets | 7d | 7d resets |');
269
+ expect(out).toContain('| --- | --- | --- | --- | --- | --- |');
270
+ // The single collapsed Status column is gone.
271
+ expect(out).not.toContain('| 7d | Status |');
270
272
  // No legacy group headers / health-section titles remain.
271
273
  expect(out).not.toContain('**BLOCKED**');
272
274
  expect(out).not.toContain('**HEALTHY**');
@@ -307,16 +309,22 @@ describe('renderAuthSnapshotFormat2', () => {
307
309
  expect(bob).toBeLessThan(alice);
308
310
  });
309
311
 
310
- it('Status column shows "back <when>" for a blocked account', () => {
312
+ it('reset columns show a "<time> (in ...)" cell for a blocked account (7d binding window)', () => {
311
313
  const rows = tableRows(renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' }));
312
314
  const bob = rows.find((r) => r[1].includes('bob@example.com'))!;
313
- expect(bob[4]).toMatch(/^back .* \(in .+\)$/);
315
+ // bob is 7d-maxed; its 7d reset (2026-05-17T10:00Z) is ~2 days out and still
316
+ // renders in its own reset cell — no "back" prefix in the new per-window shape.
317
+ expect(bob[5]).toMatch(/^.* \(in .+\)$/);
318
+ expect(bob[5]).not.toContain('back');
314
319
  });
315
320
 
316
- it('Status column shows "refills <when>" for a healthy account', () => {
321
+ it('reset columns show a "<time> (in ...)" cell for each window of a healthy account', () => {
317
322
  const rows = tableRows(renderAuthSnapshotFormat2(fixtureSnaps, { now: NOW, tz: 'UTC' }));
318
323
  const you = rows.find((r) => r[1].includes('you@example.com'))!;
319
- expect(you[4]).toMatch(/^refills .* \(in .+\)$/);
324
+ // 5h resets cell [3] and 7d resets cell [5] both carry a relative hint.
325
+ expect(you[3]).toMatch(/^.* \(in .+\)$/);
326
+ expect(you[5]).toMatch(/^.* \(in .+\)$/);
327
+ expect(you[3]).not.toContain('refills');
320
328
  });
321
329
 
322
330
  it('NEVER displays a percentage over 100% even on an over-cap account', () => {
@@ -330,13 +338,15 @@ describe('renderAuthSnapshotFormat2', () => {
330
338
  ];
331
339
  const rows = tableRows(renderAuthSnapshotFormat2(overSnaps, { now: NOW, tz: 'UTC' }));
332
340
  expect(rows[0][2]).toBe('100%'); // 5h
333
- expect(rows[0][3]).toBe('100%'); // 7d
334
- // And the blocked state still surfaces via the emoji (šŸ”“) + Status.
341
+ expect(rows[0][4]).toBe('100%'); // 7d
342
+ // And the blocked state still surfaces via the emoji (šŸ”“). No resets known
343
+ // (fixture has no reset timestamps) → both reset cells degrade to "—".
335
344
  expect(rows[0][0]).toBe('šŸ”“');
336
- expect(rows[0][4]).toMatch(/^back/);
345
+ expect(rows[0][3]).toBe('—'); // 5h resets
346
+ expect(rows[0][5]).toBe('—'); // 7d resets
337
347
  });
338
348
 
339
- it('renders dates in the Status column only when NOT today (user tz)', () => {
349
+ it('renders dates in the reset columns only when NOT today (user tz)', () => {
340
350
  // now = Fri 3:00 PM Melbourne. A reset later the SAME Melbourne day shows
341
351
  // time-only; a reset on the next day shows the weekday.
342
352
  const MEL = 'Australia/Melbourne';
@@ -349,8 +359,9 @@ describe('renderAuthSnapshotFormat2', () => {
349
359
  }) }),
350
360
  ];
351
361
  const todayRows = tableRows(renderAuthSnapshotFormat2(sameDay, { now: NOW_MEL, tz: MEL }));
352
- expect(todayRows[0][4]).toContain('9:00 PM');
353
- expect(todayRows[0][4]).not.toMatch(/Mon|Tue|Wed|Thu|Fri|Sat|Sun/);
362
+ // 5h resets cell [3] — same Melbourne day → time only.
363
+ expect(todayRows[0][3]).toContain('9:00 PM');
364
+ expect(todayRows[0][3]).not.toMatch(/Mon|Tue|Wed|Thu|Fri|Sat|Sun/);
354
365
 
355
366
  const nextDay = [
356
367
  snap({ label: 'tomorrow@example.com', isActive: true, quota: quota({
@@ -360,7 +371,8 @@ describe('renderAuthSnapshotFormat2', () => {
360
371
  }) }),
361
372
  ];
362
373
  const tomorrowRows = tableRows(renderAuthSnapshotFormat2(nextDay, { now: NOW_MEL, tz: MEL }));
363
- expect(tomorrowRows[0][4]).toContain('Sat 1:00 AM');
374
+ // 5h resets cell [3] — next Melbourne day → weekday prefix.
375
+ expect(tomorrowRows[0][3]).toContain('Sat 1:00 AM');
364
376
  });
365
377
 
366
378
  it('emits a recommendation footer that names a healthy alternative when active is throttling', () => {
@@ -1120,14 +1132,16 @@ describe('#2494 — renderAuthSnapshotFormat2 row rendering (out_of_credits demo
1120
1132
  expect(allText).toContain('Switch fleet → carol@example.com');
1121
1133
  });
1122
1134
 
1123
- it('quota-exhausted row shows a šŸ”“ + "back <when>" Status, not "billing disabled"', () => {
1135
+ it('quota-exhausted row shows a šŸ”“ + its 5h reset time in the reset cell, not "billing disabled"', () => {
1124
1136
  const futureReset = new Date(NOW.getTime() + 45 * 60_000);
1125
1137
  const out = renderAuthSnapshotFormat2(
1126
1138
  [snap({ label: 'ex@x', isActive: true, quota: quota({ fiveHourUtilizationPct: 100, fiveHourResetAt: futureReset }) })],
1127
1139
  { now: NOW },
1128
1140
  );
1129
1141
  expect(out).toContain('| šŸ”“ |');
1130
- expect(out).toMatch(/back .* \(in 45m\)/);
1142
+ // The 5h reset (45m out) renders in its own reset cell — no "back" prefix.
1143
+ expect(out).toMatch(/\(in 45m\)/);
1144
+ expect(out).not.toContain('back ');
1131
1145
  expect(out).not.toContain('billing disabled');
1132
1146
  expect(out).not.toContain('overage off');
1133
1147
  });
@@ -137,6 +137,54 @@ describe('Claude Code event-stream contract (canary)', () => {
137
137
  ).toBe(false)
138
138
  })
139
139
 
140
+ it('CANARY (Fix 3 precondition): Task tool_use still projects {id, name, input} — the shape the foreground-sub-agent-tracking / post-answer-liveness-freeze fix keys "still dispatched" off of', () => {
141
+ // The Fix 3 gap-closer (turn-liveness-floor.ts `stillDispatched`) is only
142
+ // as good as `turn.foregroundSubAgents` staying populated for the
143
+ // lifetime of a real Task dispatch. That population is driven by THIS
144
+ // exact tool_use projection (toolName === 'Task') plus the matching
145
+ // tool_result closing it out — if a future Claude Code release renames
146
+ // `name`/`input`/`id` or nests them differently, the fix silently goes
147
+ // inert (stillDispatched always false, freeze regresses) rather than
148
+ // failing loudly. Pin the shape here so drift fails CI instead.
149
+ const line = JSON.stringify({
150
+ type: 'assistant',
151
+ message: {
152
+ id: 'msg_task',
153
+ stop_reason: 'tool_use',
154
+ content: [
155
+ {
156
+ type: 'tool_use',
157
+ id: 'toolu_task_1',
158
+ name: 'Task',
159
+ input: { description: 'Investigate the freeze bug', subagent_type: 'worker' },
160
+ },
161
+ ],
162
+ },
163
+ })
164
+ const evs = projectTranscriptLine(line)
165
+ expect(evs).toEqual([
166
+ {
167
+ kind: 'tool_use',
168
+ toolName: 'Task',
169
+ toolUseId: 'toolu_task_1',
170
+ input: { description: 'Investigate the freeze bug', subagent_type: 'worker' },
171
+ },
172
+ ])
173
+ // The matching tool_result (Task returning) must still carry tool_use_id
174
+ // so the gateway can close out the foreground-sub-agent tracking entry.
175
+ const resultLine = JSON.stringify({
176
+ type: 'user',
177
+ message: { content: [{ type: 'tool_result', tool_use_id: 'toolu_task_1', is_error: false, content: 'done' }] },
178
+ })
179
+ const resultEvs = projectTranscriptLine(resultLine)
180
+ expect(resultEvs).toHaveLength(1)
181
+ // `is_error: false` projects to `isError: undefined` (only `true` is
182
+ // ever stamped — see session-tail.ts's `c.is_error === true ? true :
183
+ // undefined`), so assert the falsy/absent form, not a literal `false`.
184
+ expect(resultEvs[0]).toMatchObject({ kind: 'tool_result', toolUseId: 'toolu_task_1' })
185
+ expect((resultEvs[0] as { isError?: boolean }).isError).toBeFalsy()
186
+ })
187
+
140
188
  it('sub-agent kickoff: first user message string prompt fires sub_agent_started', () => {
141
189
  const st = { hasEmittedStart: false }
142
190
  const line = JSON.stringify({
@@ -153,6 +153,17 @@ describe('H-2: feedHeartbeatTick post-answer background-agent liveness (Fix 2 /
153
153
  expect(postAnswerBlock).toMatch(/livenessVerdict\s*!==\s*'emit'/)
154
154
  })
155
155
 
156
+ it('Fix 3: post-answer branch feeds stillDispatched from turn.foregroundSubAgents so a tracked foreground sub-agent bypasses the staleness cap', () => {
157
+ const body = feedHeartbeatTickSrc()
158
+ const afterPostAnswer = body.split('if (turn.finalAnswerDelivered)')[1] ?? ''
159
+ const postAnswerBlock = afterPostAnswer.split('\n }\n')[0] ?? ''
160
+ // A positive foreground-tracking signal must be computed and threaded
161
+ // into the pure decision — this is what stops the card freezing mid-
162
+ // delegation while a foreground Task/Agent is still outstanding.
163
+ expect(postAnswerBlock).toMatch(/stillDispatched\s*=\s*turn\.foregroundSubAgents\.size\s*>\s*0/)
164
+ expect(postAnswerBlock).toMatch(/stillDispatched,/)
165
+ })
166
+
156
167
  it('staleness cap (concern 3) is parsed default-ON (30s) from SWITCHROOM_POST_ANSWER_LIVENESS_STALE_MS', () => {
157
168
  // The cap const must default to 30_000 when the env is unset (the `|| 30_000`
158
169
  // fallback over the positive-or-0 parse) so the post-answer card stops
@@ -47,6 +47,8 @@ import { ToolFlightTracker } from '../gateway/interrupt-defer.js'
47
47
  import {
48
48
  ORPHANED_REPLY_TIMEOUT_MS,
49
49
  ORPHANED_REPLY_MAX_REARMS,
50
+ ORPHANED_REPLY_STREAM_WINDOW_MS,
51
+ LivenessTracker,
50
52
  } from '../context-exhaustion.js'
51
53
 
52
54
  // ─── Helpers ──────────────────────────────────────────────────────────────────
@@ -524,3 +526,40 @@ describe('silence-poke — hard ceiling bounds the defer', () => {
524
526
  expect(__getStateForTests('chat:0')!.fallbackFired).toBe(true)
525
527
  })
526
528
  })
529
+
530
+ // ─── Thinking-pause liveness via the REAL LivenessTracker (orphaned-reply fix) ─
531
+
532
+ describe('silence-poke — thinking-pause liveness via the real LivenessTracker', () => {
533
+ // Drives the REAL LivenessTracker as the isLegitimatelyWorking source: a
534
+ // genuine stream event within ORPHANED_REPLY_STREAM_WINDOW_MS keeps the feed
535
+ // alive (a model reasoning pause is survivable); a gap longer than the window
536
+ // with no work tears the feed down.
537
+ it('a stream event within the window keeps the feed alive (thinking-pause survivable)', () => {
538
+ let clockNow = 0
539
+ const tracker = new LivenessTracker(0)
540
+ const f = setupSilenceDeps({
541
+ thresholds: { fallback: 300_000, fallbackHardCeiling: 900_000 },
542
+ isLegitimatelyWorking: () => tracker.recentlyStreaming(clockNow, ORPHANED_REPLY_STREAM_WINDOW_MS),
543
+ })
544
+ startTurn('chat:0', 0)
545
+ // A genuine stream event lands mid-turn (e.g. the model resumes after a
546
+ // reasoning pause) — only 50s before the 300s fallback tick.
547
+ tracker.onStreamEvent('text', undefined, 250_000)
548
+ clockNow = 300_000
549
+ __tickForTests(300_000) // 50s since last stream < 120s window → deferred
550
+ expect(f.fallbacks).toHaveLength(0)
551
+ })
552
+
553
+ it('no stream event for longer than the window tears the feed down', () => {
554
+ let clockNow = 0
555
+ const tracker = new LivenessTracker(0) // seed at t=0, nothing after
556
+ const f = setupSilenceDeps({
557
+ thresholds: { fallback: 300_000, fallbackHardCeiling: 900_000 },
558
+ isLegitimatelyWorking: () => tracker.recentlyStreaming(clockNow, ORPHANED_REPLY_STREAM_WINDOW_MS),
559
+ })
560
+ startTurn('chat:0', 0)
561
+ clockNow = 300_000
562
+ __tickForTests(300_000) // 300s since last stream >> 120s window → fires
563
+ expect(f.fallbacks).toHaveLength(1)
564
+ })
565
+ })
@@ -0,0 +1,81 @@
1
+ /**
2
+ * Structural pins for the session-only model relaunch wiring in gateway.ts.
3
+ *
4
+ * The behaviour lives in un-exported inline closures (buildModelDeps's
5
+ * scheduleModelRelaunch, the model-menu callback sr-* branch, and the boot
6
+ * re-hydration / alert-sentinel block inside the startup IIFE), so — mirroring
7
+ * the other gateway-*.test.ts source-pins — we assert on the source structure.
8
+ * The end-to-end behaviour of the carrier itself is exercised in
9
+ * tests/scaffold.session-model-override.test.ts (rendered start.sh) and the
10
+ * handler contract in telegram-plugin/tests/model-command.test.ts.
11
+ */
12
+
13
+ import { describe, it, expect } from 'vitest'
14
+ import { readFileSync } from 'node:fs'
15
+ import { fileURLToPath } from 'node:url'
16
+ import { dirname, resolve } from 'node:path'
17
+
18
+ const __dirname = dirname(fileURLToPath(import.meta.url))
19
+ const GATEWAY_SRC = readFileSync(resolve(__dirname, '..', 'gateway', 'gateway.ts'), 'utf8')
20
+
21
+ describe('gateway: scheduleModelRelaunch dep', () => {
22
+ it('writes the carrier with exact "<model>\\n" bytes', () => {
23
+ const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
24
+ expect(idx).toBeGreaterThan(0)
25
+ const win = GATEWAY_SRC.slice(idx, idx + 900)
26
+ expect(win).toMatch(/writeFileSync\(\s*join\(agentDir, '\.session-model-override'\),\s*`\$\{model\}\\n`/)
27
+ })
28
+
29
+ it('sets the in-memory activeSessionModelOverride before dispatching the restart', () => {
30
+ const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
31
+ const win = GATEWAY_SRC.slice(idx, idx + 900)
32
+ const setIdx = win.indexOf('activeSessionModelOverride = model')
33
+ const restartIdx = win.indexOf('deps.scheduleRestart(reason)')
34
+ expect(setIdx).toBeGreaterThan(0)
35
+ expect(restartIdx).toBeGreaterThan(setIdx)
36
+ })
37
+
38
+ it('reuses the same scheduleRestart dispatch (not a bespoke restart path)', () => {
39
+ const idx = GATEWAY_SRC.indexOf('scheduleModelRelaunch: async')
40
+ const win = GATEWAY_SRC.slice(idx, idx + 900)
41
+ expect(win).toContain('await deps.scheduleRestart(reason)')
42
+ })
43
+ })
44
+
45
+ describe('gateway: model-menu sr-* target relaunches via the carrier', () => {
46
+ it('the sr-* callback branch calls scheduleModelRelaunch, not inject', () => {
47
+ const idx = GATEWAY_SRC.indexOf('if (data.startsWith(MODEL_CALLBACK_SR))')
48
+ expect(idx).toBeGreaterThan(0)
49
+ const win = GATEWAY_SRC.slice(idx, idx + 1400)
50
+ expect(win).toContain('modelDeps.scheduleModelRelaunch(srName')
51
+ // It must return before falling through to handleModelMenuCallback (which
52
+ // would inject an sr-* id claude's picker rejects).
53
+ expect(win).toMatch(/scheduleModelRelaunch[\s\S]*?\n\s*return\n/)
54
+ })
55
+ })
56
+
57
+ describe('gateway boot: session-model re-hydration + LiteLLM-down alert', () => {
58
+ it('re-hydrates activeSessionModelOverride from .active-session-model', () => {
59
+ const idx = GATEWAY_SRC.indexOf("join(smAgentDir, '.active-session-model')")
60
+ expect(idx).toBeGreaterThan(0)
61
+ const win = GATEWAY_SRC.slice(idx - 200, idx + 1400)
62
+ // Only an override when the launched model differs from the configured one.
63
+ expect(win).toMatch(/launched\.length > 0 && launched !== configured \? launched : null/)
64
+ // The configured value must be resolved through resolveMainModel (the SAME
65
+ // resolver start.sh's scaffold uses) so an unset/`default` model config does
66
+ // not get flagged as a phantom session override on an ordinary restart.
67
+ expect(win).toContain('resolveMainModel(raw ?? undefined)')
68
+ })
69
+
70
+ it('consumes the .session-model-alert sentinel, notifies ALL operators, and deletes it', () => {
71
+ const idx = GATEWAY_SRC.indexOf("join(smAgentDir, '.session-model-alert')")
72
+ expect(idx).toBeGreaterThan(0)
73
+ const win = GATEWAY_SRC.slice(idx, idx + 1100)
74
+ expect(win).toContain('unlinkSync(alertPath)')
75
+ // Broadcasts to every operator in allowFrom, not just allowFrom[0].
76
+ expect(win).toContain('const operators = loadAccess().allowFrom')
77
+ expect(win).toContain('for (const operator of operators)')
78
+ expect(win).toContain('lockedBot.api')
79
+ expect(win).toContain('.sendMessage(operator')
80
+ })
81
+ })
@@ -0,0 +1,82 @@
1
+ /**
2
+ * Structural pin: the delivery state machine's `inbound` event MUST be
3
+ * emitted from `handleInbound` only AFTER the intercept/early-return
4
+ * gauntlet — never eagerly at handler entry.
5
+ *
6
+ * The wedge this guards (overlord `/usage` went dead, 2026-07-08). The
7
+ * inbound emit drives the now-AUTHORITATIVE turn-in-flight gate
8
+ * (turnInFlightForGate → isMachineInTurn). When it fired at handler entry
9
+ * — before the permission-reply / `/auth` paste-back / interrupt-empty /
10
+ * secret-detect-drop intercepts — an intercepted message (e.g. an approval
11
+ * reply) drove the machine idle→`bridge_alive_in_turn`, then early-returned
12
+ * as an intercept: no delivery, no claudeBusyKeys mark, and critically no
13
+ * `turnEnd`. The machine held the gate closed until the 5-min TTL tick
14
+ * force-cleared it, buffering every inbound (including `/usage`) meanwhile —
15
+ * the dangerous `machine_over_holds` divergence gate-parity-probe.ts flags.
16
+ *
17
+ * The behaviour lives inside the un-exported `handleInbound` closure, so —
18
+ * mirroring the other gateway-*.test.ts source-pins — we assert on source
19
+ * structure: the single inbound `shadowEmit` must sit below the intercepts
20
+ * and carry the real `isSteering` classification. Machine self-heal + gate
21
+ * accessors are behaviour-tested in inbound-delivery-cutover-gate.test.ts.
22
+ */
23
+
24
+ import { describe, it, expect } from 'vitest'
25
+ import { readFileSync } from 'node:fs'
26
+ import { fileURLToPath } from 'node:url'
27
+ import { dirname, resolve } from 'node:path'
28
+
29
+ const __dirname = dirname(fileURLToPath(import.meta.url))
30
+ const GATEWAY_SRC = readFileSync(resolve(__dirname, '..', 'gateway', 'gateway.ts'), 'utf8')
31
+
32
+ /** Byte offset of the sole `shadowEmit({ kind: 'inbound' ... })` call. */
33
+ function inboundEmitOffset(): number {
34
+ const m = GATEWAY_SRC.match(/shadowEmit\(\{\s*\n\s*kind:\s*'inbound'/)
35
+ expect(m, 'expected exactly one multi-line inbound shadowEmit').not.toBeNull()
36
+ const idx = GATEWAY_SRC.indexOf(m![0])
37
+ // There must be only ONE such emit — a second eager site would reintroduce
38
+ // the wedge.
39
+ const second = GATEWAY_SRC.indexOf(m![0], idx + 1)
40
+ expect(second, 'a second inbound shadowEmit would reintroduce the eager-emit wedge').toBe(-1)
41
+ return idx
42
+ }
43
+
44
+ describe('gateway: inbound machine-emit is deferred past the intercepts', () => {
45
+ it('carries the DEFERRED_INBOUND_EMIT marker so the intent is greppable', () => {
46
+ expect(GATEWAY_SRC).toContain('DEFERRED_INBOUND_EMIT')
47
+ })
48
+
49
+ it('emits the machine `inbound` event AFTER the permission-reply intercept', () => {
50
+ const emitIdx = inboundEmitOffset()
51
+ // The permission-reply intercept is one of the early-return paths that
52
+ // must NOT drive the machine into a turn.
53
+ const permIdx = GATEWAY_SRC.indexOf('const permMatch = PERMISSION_REPLY_RE.exec(text)')
54
+ expect(permIdx).toBeGreaterThan(0)
55
+ expect(emitIdx).toBeGreaterThan(permIdx)
56
+ })
57
+
58
+ it('emits AFTER the `/auth add` paste-back intercept', () => {
59
+ const emitIdx = inboundEmitOffset()
60
+ const authIdx = GATEWAY_SRC.indexOf('pendingAuthAddFlows.get(interceptKey)')
61
+ expect(authIdx).toBeGreaterThan(0)
62
+ expect(emitIdx).toBeGreaterThan(authIdx)
63
+ })
64
+
65
+ it('emits AFTER isSteering is classified, and passes the real value (not a hardcoded false)', () => {
66
+ const emitIdx = inboundEmitOffset()
67
+ const steerIdx = GATEWAY_SRC.indexOf('isSteering = priorTurnInFlight && isSteerPrefix')
68
+ expect(steerIdx).toBeGreaterThan(0)
69
+ expect(emitIdx).toBeGreaterThan(steerIdx)
70
+ // The emit's msg object must forward the live `isSteering` binding, so a
71
+ // mid-turn steer is not mis-modelled as a fresh turn.
72
+ const win = GATEWAY_SRC.slice(emitIdx, emitIdx + 260)
73
+ expect(win).toMatch(/msg:\s*\{[\s\S]*?\bisSteering,[\s\S]*?\}/)
74
+ })
75
+
76
+ it('emits BEFORE the deliver-or-buffer decision (reserveInboundDelivery)', () => {
77
+ const emitIdx = inboundEmitOffset()
78
+ const gateIdx = GATEWAY_SRC.indexOf('const deliveryGate = reserveInboundDelivery({')
79
+ expect(gateIdx).toBeGreaterThan(0)
80
+ expect(emitIdx).toBeLessThan(gateIdx)
81
+ })
82
+ })
@@ -0,0 +1,228 @@
1
+ /**
2
+ * liveness-tracker.test.ts — drives the REAL LivenessTracker (the testability
3
+ * seam of the orphaned-reply thinking-pause fix).
4
+ *
5
+ * Root cause being fixed: the orphaned-reply fuse (ORPHANED_REPLY_TIMEOUT_MS =
6
+ * 30 s) was reset ONLY by `tool_label` / `text` events. During a long model
7
+ * reasoning pause the gateway sees NO such events (thinking carries no text),
8
+ * so the fuse ran down and force-ended a genuinely-live turn mid-work.
9
+ *
10
+ * The fix stamps `lastStreamEventAt` on ANY genuine stream event; if a genuine
11
+ * event landed within `windowMs` (default 120 s) the turn is "recently
12
+ * streaming" and the fuse re-arms instead of firing. A genuine multi-minute
13
+ * hang (no events at all) still fires.
14
+ *
15
+ * These tests use the REAL class — no replica helpers.
16
+ *
17
+ * NOTE on onStreamEvent's signature: it is `(kind, durationMs, now)`. The
18
+ * tracker is time-only — it never reads text content (the "Prompt is too long"
19
+ * marker matters to the gateway's context-exhaustion detection, not to the
20
+ * tracker). Where the fix spec wrote `onStreamEvent('text','Prompt is too
21
+ * long',0)` we pass `undefined` as durationMs (it is not a turn_end).
22
+ */
23
+
24
+ import { describe, it, expect } from 'vitest'
25
+ import {
26
+ LivenessTracker,
27
+ ORPHANED_REPLY_STREAM_WINDOW_MS,
28
+ ORPHANED_REPLY_TIMEOUT_MS,
29
+ ORPHANED_REPLY_MAX_REARMS,
30
+ } from '../context-exhaustion.js'
31
+
32
+ const W = ORPHANED_REPLY_STREAM_WINDOW_MS // 120_000
33
+ const MAX = ORPHANED_REPLY_MAX_REARMS // 20
34
+ const FUSE = ORPHANED_REPLY_TIMEOUT_MS // 30_000
35
+
36
+ function expiry(
37
+ t: LivenessTracker,
38
+ now: number,
39
+ opts?: { working?: boolean; humanWaiting?: boolean },
40
+ ) {
41
+ return t.decideOnExpiry({
42
+ working: opts?.working ?? false,
43
+ humanWaiting: opts?.humanWaiting ?? false,
44
+ now,
45
+ windowMs: W,
46
+ maxRearms: MAX,
47
+ })
48
+ }
49
+
50
+ describe('LivenessTracker — pinned constants', () => {
51
+ it('window is 120 s and fuse tick is 30 s', () => {
52
+ expect(W).toBe(120_000)
53
+ expect(FUSE).toBe(30_000)
54
+ expect(MAX).toBe(20)
55
+ })
56
+ })
57
+
58
+ describe('T1 — thinking-pause survival', () => {
59
+ it('text→tool_use→tool_result then a 45s stream gap keeps re-arming (gap < window)', () => {
60
+ const t = new LivenessTracker(0)
61
+ t.onStreamEvent('text', undefined, 0)
62
+ t.onStreamEvent('tool_use', undefined, 1_000)
63
+ t.onStreamEvent('tool_result', undefined, 2_000)
64
+
65
+ // +30s past the last event (t=32s): recently streaming → re-arm.
66
+ const d1 = expiry(t, 32_000)
67
+ expect(d1).toEqual({ rearm: true, countsAgainstCap: true })
68
+ expect(t.orphanedReplyRearmCount).toBe(1)
69
+
70
+ // +45s past the last event (t=47s): still within the 120s window → re-arm.
71
+ const d2 = expiry(t, 47_000)
72
+ expect(d2).toEqual({ rearm: true, countsAgainstCap: true })
73
+ expect(t.orphanedReplyRearmCount).toBe(2)
74
+ })
75
+
76
+ it('the next genuine stream event ZEROES the rearm counter and re-stamps liveness', () => {
77
+ const t = new LivenessTracker(0)
78
+ t.onStreamEvent('tool_result', undefined, 2_000)
79
+ expiry(t, 32_000)
80
+ expiry(t, 47_000)
81
+ expect(t.orphanedReplyRearmCount).toBe(2)
82
+
83
+ t.onStreamEvent('tool_label', undefined, 48_000)
84
+ expect(t.orphanedReplyRearmCount).toBe(0)
85
+ expect(t.lastStreamEventAt).toBe(48_000)
86
+ })
87
+
88
+ it('a synthetic turn_end (durationMs === -1) does NOT stamp or reset (F4 boundary)', () => {
89
+ const t = new LivenessTracker(0)
90
+ t.onStreamEvent('tool_label', undefined, 48_000)
91
+ expiry(t, 60_000) // counter → 1
92
+ expect(t.orphanedReplyRearmCount).toBe(1)
93
+
94
+ t.onStreamEvent('turn_end', -1, 70_000)
95
+ expect(t.lastStreamEventAt).toBe(48_000) // unchanged — synthetic excluded
96
+ expect(t.orphanedReplyRearmCount).toBe(1) // not reset
97
+
98
+ // A REAL turn_end (durationMs >= 0) DOES stamp — harmless, the turn is
99
+ // ending anyway (spec #3).
100
+ t.onStreamEvent('turn_end', 0, 71_000)
101
+ expect(t.lastStreamEventAt).toBe(71_000)
102
+ expect(t.orphanedReplyRearmCount).toBe(0)
103
+ })
104
+ })
105
+
106
+ describe('T2 — production repro (48s tool_result, 78s expiry)', () => {
107
+ it('text@0, tool_use@1s, tool_result@48s → expiry@78s re-arms (78-48=30 < 120)', () => {
108
+ const t = new LivenessTracker(0)
109
+ t.onStreamEvent('text', undefined, 0)
110
+ t.onStreamEvent('tool_use', undefined, 1_000)
111
+ t.onStreamEvent('tool_result', undefined, 48_000)
112
+
113
+ const d = expiry(t, 78_000)
114
+ expect(d.rearm).toBe(true)
115
+ expect(t.orphanedReplyRearmCount).toBe(1)
116
+
117
+ // tool_label@80s zeroes the counter (progress observed).
118
+ t.onStreamEvent('tool_label', undefined, 80_000)
119
+ expect(t.orphanedReplyRearmCount).toBe(0)
120
+ })
121
+ })
122
+
123
+ describe('T3 — long active turn (A2: counter reset is load-bearing)', () => {
124
+ it('tool_label every 60s + expiry every 60s+30s NEVER fires across 15 min; counter never exceeds 1', () => {
125
+ const t = new LivenessTracker(0)
126
+ for (let k = 0; k < 15; k++) {
127
+ t.onStreamEvent('tool_label', undefined, k * 60_000) // progress → reset
128
+ const d = expiry(t, k * 60_000 + 30_000, { working: true })
129
+ expect(d.rearm).toBe(true) // never fires
130
+ expect(t.orphanedReplyRearmCount).toBeLessThanOrEqual(1)
131
+ }
132
+ })
133
+
134
+ it('WITHOUT the progress reset the SAME cadence hits cap=20 near t=600s (proves reset load-bearing)', () => {
135
+ const t = new LivenessTracker(0)
136
+ let firedAt = -1
137
+ for (let k = 1; k <= 25; k++) {
138
+ const now = k * 30_000
139
+ // working:true so it would re-arm indefinitely IF the counter didn't
140
+ // accumulate — but with no onStreamEvent reset it climbs to the cap.
141
+ const d = expiry(t, now, { working: true })
142
+ if (!d.rearm) {
143
+ firedAt = now
144
+ break
145
+ }
146
+ }
147
+ // 20th re-arm lands at t=600s; the 21st expiry (t=630s) fires at the cap.
148
+ expect(t.orphanedReplyRearmCount).toBe(MAX)
149
+ expect(firedAt).toBeGreaterThanOrEqual(600_000)
150
+ })
151
+ })
152
+
153
+ describe('T4 — genuine hang caught (no events at all)', () => {
154
+ it('tool_result@0 then silence: re-arms via recentlyStreaming @30/60/90s, FIRES @120s (window lapsed)', () => {
155
+ const t = new LivenessTracker(0)
156
+ t.onStreamEvent('tool_result', undefined, 0)
157
+
158
+ for (const now of [30_000, 60_000, 90_000]) {
159
+ const d = expiry(t, now) // working:false — kept alive only by recentlyStreaming
160
+ expect(d.rearm).toBe(true)
161
+ }
162
+
163
+ // t=120s: now - lastStreamEventAt = 120_000 >= window → not recently
164
+ // streaming, working false → fail-safe FIRE.
165
+ const d = expiry(t, 120_000)
166
+ expect(d).toEqual({ rearm: false, countsAgainstCap: false })
167
+ })
168
+ })
169
+
170
+ describe('T5 — context-exhaustion recovery COMPLETES (not suppressed forever)', () => {
171
+ it('a "Prompt is too long" text event stamps liveness; defensive-guard predicate is true@30s, false@>=120s', () => {
172
+ const t = new LivenessTracker(0)
173
+ // The marker is a genuine `text` event (content is not a tracker concern).
174
+ t.onStreamEvent('text', undefined, 0)
175
+
176
+ // Context exhaustion means no tool is in flight: working === false. The
177
+ // defensive-guard suppression predicate is `working || recentlyStreaming`.
178
+ const working = false
179
+
180
+ // @30s: recentlyStreaming true → synthetic turn_end is suppressed. This
181
+ // DOCUMENTS the accepted ~30s→~120s recovery-latency growth (F3).
182
+ expect(working || t.recentlyStreaming(30_000, W)).toBe(true)
183
+
184
+ // @120s: window lapsed → NOT suppressed → teardown COMPLETES.
185
+ expect(working || t.recentlyStreaming(120_000, W)).toBe(false)
186
+
187
+ // @150s: still not suppressed — recovery is delayed, never forever.
188
+ expect(working || t.recentlyStreaming(150_000, W)).toBe(false)
189
+ })
190
+ })
191
+
192
+ describe('T7 — cap semantics', () => {
193
+ it('interleaved expiry/genuine-event: count 1 → 0 → 1 → 2', () => {
194
+ const t = new LivenessTracker(0)
195
+
196
+ expect(expiry(t, 10_000).rearm).toBe(true)
197
+ expect(t.orphanedReplyRearmCount).toBe(1)
198
+
199
+ t.onStreamEvent('tool_label', undefined, 11_000) // genuine event resets
200
+ expect(t.orphanedReplyRearmCount).toBe(0)
201
+
202
+ expiry(t, 20_000)
203
+ expect(t.orphanedReplyRearmCount).toBe(1)
204
+
205
+ expiry(t, 30_000) // consecutive silence (last event 11_000, gap 19s < 120s)
206
+ expect(t.orphanedReplyRearmCount).toBe(2)
207
+ })
208
+
209
+ it('21 consecutive recently-streaming expiries (no reset): 21st fires at the cap', () => {
210
+ const t = new LivenessTracker(0)
211
+ let last: { rearm: boolean; countsAgainstCap: boolean } | undefined
212
+ for (let k = 1; k <= 21; k++) {
213
+ // now = k*1000 keeps the gap from the seed (0) under the 120s window,
214
+ // so recentlyStreaming stays true throughout — the ONLY bound is the cap.
215
+ last = expiry(t, k * 1_000)
216
+ }
217
+ expect(last).toEqual({ rearm: false, countsAgainstCap: true })
218
+ expect(t.orphanedReplyRearmCount).toBe(MAX)
219
+
220
+ // humanWaiting past the cap → uncapped re-arm.
221
+ const dHuman = expiry(t, 22_000, { humanWaiting: true })
222
+ expect(dHuman).toEqual({ rearm: true, countsAgainstCap: false })
223
+
224
+ // recentlyStreaming false + working false + humanWaiting false → fail-safe FIRE.
225
+ const dFire = expiry(t, 200_000)
226
+ expect(dFire).toEqual({ rearm: false, countsAgainstCap: false })
227
+ })
228
+ })