claude-slack-channel-bots 0.8.1 → 0.8.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "claude-slack-channel-bots",
3
- "version": "0.8.1",
3
+ "version": "0.8.2",
4
4
  "description": "Multi-session Slack-to-Claude bridge — run multiple Claude Code bots across Slack channels via Socket Mode",
5
5
  "type": "module",
6
6
  "bin": {
@@ -31,6 +31,7 @@ import {
31
31
  ErrNoSessionId,
32
32
  ErrSpawnNotFound,
33
33
  ErrSpawnNotResumable,
34
+ ErrTmuxSendKeys,
34
35
  ErrTmuxSessionCreate,
35
36
  } from 'agent-director'
36
37
  import type { ListRow, SpawnParams } from 'agent-director'
@@ -180,9 +181,50 @@ export function flushSpawnFailureQueue(web: WebClient): void {
180
181
  // reconnectMcp — send `/mcp reconnect <server-name>` via library sendKeys
181
182
  // ---------------------------------------------------------------------------
182
183
 
184
+ /**
185
+ * Ensure a tmux server exists on the default socket. Injectable seam so unit
186
+ * tests can assert the b.rmy self-heal path without spawning real processes.
187
+ * Default impl runs `tmux start-server` best-effort — never throws.
188
+ *
189
+ * b.rmy: after a container restart, /tmp (and thus the tmux socket) is wiped
190
+ * and nothing auto-starts tmux at boot, so every startup reconnect fails with
191
+ * `ErrTmuxSendKeys` ("no server running"). Starting the server before a retry
192
+ * gives the reconnect a chance to succeed instead of failing terminally.
193
+ */
194
+ export type TmuxServerEnsurer = () => Promise<void>
195
+
196
+ const defaultEnsureTmuxServer: TmuxServerEnsurer = async (): Promise<void> => {
197
+ const { spawn } = await import('child_process')
198
+ await new Promise<void>((resolve) => {
199
+ try {
200
+ const child = spawn('tmux', ['start-server'], { stdio: 'ignore' })
201
+ child.on('error', () => resolve()) // tmux missing — best-effort
202
+ child.on('close', () => resolve())
203
+ } catch {
204
+ resolve()
205
+ }
206
+ })
207
+ }
208
+
209
+ let _ensureTmuxServer: TmuxServerEnsurer = defaultEnsureTmuxServer
210
+
211
+ /** Test-only seam: override the tmux-server ensurer. */
212
+ export function _setTmuxServerEnsurer(fn: TmuxServerEnsurer): void {
213
+ _ensureTmuxServer = fn
214
+ }
215
+
216
+ /** Test-only seam: restore the default tmux-server ensurer. */
217
+ export function _resetTmuxServerEnsurer(): void {
218
+ _ensureTmuxServer = defaultEnsureTmuxServer
219
+ }
220
+
183
221
  /**
184
222
  * Send `/mcp reconnect <MCP_SERVER_NAME>` to the spawn's pane. Library's
185
223
  * sendKeys appends Enter automatically per its contract.
224
+ *
225
+ * b.rmy self-heal: on `ErrTmuxSendKeys` (no tmux server / session — the
226
+ * post-reboot field failure), ensure a tmux server exists and retry the
227
+ * send-keys ONCE. The retry's outcome is the returned outcome.
186
228
  */
187
229
  export async function reconnectMcp(
188
230
  channelId: string,
@@ -191,14 +233,33 @@ export async function reconnectMcp(
191
233
  ): Promise<boolean> {
192
234
  const claude_instance_id = instanceIdFor(channelId, routingConfig?.routes[channelId]?.normalizedName)
193
235
  console.error(`[slack] reconnecting MCP server "${MCP_SERVER_NAME}": channel=${channelId}`)
194
- try {
195
- await withOutageDetection(channelId, undefined, (client) => client.sendKeys({
236
+ const sendReconnect = (): Promise<unknown> =>
237
+ withOutageDetection(channelId, undefined, (client) => client.sendKeys({
196
238
  claude_instance_id,
197
239
  text: `/mcp reconnect ${MCP_SERVER_NAME}`,
198
240
  }))
241
+ try {
242
+ await sendReconnect()
199
243
  return true
200
244
  } catch (err) {
201
245
  if (err instanceof ErrSystemInstallDisappeared || err instanceof ErrTmuxNotAvailable) return false
246
+ if (err instanceof ErrTmuxSendKeys) {
247
+ console.error(
248
+ `[slack] reconnectMcp: ErrTmuxSendKeys for channel=${channelId} — ensuring tmux server exists and retrying send-keys once`,
249
+ )
250
+ await _ensureTmuxServer()
251
+ try {
252
+ await sendReconnect()
253
+ console.error(`[slack] reconnectMcp: retry succeeded after ErrTmuxSendKeys for channel=${channelId}`)
254
+ return true
255
+ } catch (err2) {
256
+ if (err2 instanceof ErrSystemInstallDisappeared || err2 instanceof ErrTmuxNotAvailable) return false
257
+ const e2 = err2 instanceof AgentDirectorError ? err2 : new AgentDirectorError('send-keys', 'UnknownError', String(err2))
258
+ console.error(`[slack] reconnectMcp: retry after ErrTmuxSendKeys failed for channel=${channelId}: ${e2.errName}`)
259
+ postSpawnFailureToChannel(channelId, e2, web)
260
+ return false
261
+ }
262
+ }
202
263
  const e = err instanceof AgentDirectorError ? err : new AgentDirectorError('send-keys', 'UnknownError', String(err))
203
264
  console.error(`[slack] reconnectMcp: send-keys failed for channel=${channelId}: ${e.errName}`)
204
265
  postSpawnFailureToChannel(channelId, e, web)
@@ -947,12 +1008,27 @@ export async function spawnForRoute(
947
1008
  }
948
1009
 
949
1010
  if (state === 'waiting') {
950
- await reconnectMcp(channelId, web, routingConfig)
1011
+ // b.rmy: propagate the reconnect outcome — a failed reconnect must count
1012
+ // as `failed` so startupSessionManager's ok/failed totals reflect reality
1013
+ // (previously this reported 'reconnected' unconditionally, masking a
1014
+ // total post-reboot outage as "0 failed").
1015
+ if (!(await reconnectMcp(channelId, web, routingConfig))) {
1016
+ console.error(`[slack] spawnForRoute: reconnect failed for channel=${channelId}`)
1017
+ if (isStartup) recordStartupError('spawn-failed', `reconnect failed for channel=${channelId} (state=waiting)`)
1018
+ return { channelId, action: 'failed' }
1019
+ }
951
1020
  return { channelId, action: 'reconnected' }
952
1021
  }
953
1022
 
954
1023
  if (state === 'working') {
955
- await waitForWaitingAndReconnect(channelId, routingConfig, web)
1024
+ // b.rmy: same outcome propagation as the `waiting` branch. Note
1025
+ // waitForWaitingAndReconnect returns true on timeout/terminal transitions
1026
+ // by design (long turns aren't errors) — only real failures reach here.
1027
+ if (!(await waitForWaitingAndReconnect(channelId, routingConfig, web))) {
1028
+ console.error(`[slack] spawnForRoute: reconnect failed for channel=${channelId}`)
1029
+ if (isStartup) recordStartupError('spawn-failed', `reconnect failed for channel=${channelId} (state=working)`)
1030
+ return { channelId, action: 'failed' }
1031
+ }
956
1032
  return { channelId, action: 'reconnected' }
957
1033
  }
958
1034