@songsid/agend 2.1.1-beta.15 → 2.1.1-beta.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -26,7 +26,7 @@ import { processAttachments } from "./channel/attachment-handler.js";
26
26
  import { routeToolCall } from "./channel/tool-router.js";
27
27
  import { Scheduler } from "./scheduler/index.js";
28
28
  import { DEFAULT_SCHEDULER_CONFIG } from "./scheduler/index.js";
29
- import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext } from "./topic-commands.js";
29
+ import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext, forgetInstanceContext } from "./topic-commands.js";
30
30
  import { DailySummary } from "./daily-summary.js";
31
31
  import { WebhookEmitter } from "./webhook-emitter.js";
32
32
  import { TmuxControlClient } from "./tmux-control.js";
@@ -1134,8 +1134,16 @@ export class FleetManager {
1134
1134
  }
1135
1135
  }
1136
1136
  }
1137
- // Signal systemd: generals ready
1138
- sdNotify("READY=1");
1137
+ // The systemd watchdog answers exactly one question: is this process still
1138
+ // turning its event loop? Pinging from a timer proves that, and after the
1139
+ // blocking child-process calls were made async it is a meaningful signal —
1140
+ // a deadlocked or frozen fleet stops pinging and systemd restarts it.
1141
+ //
1142
+ // It deliberately does NOT gate on fleet health. "No adapter connected" or
1143
+ // "an instance crashed" must not kill the process: the fleet would be restarted
1144
+ // into the same broken state, and a user who has legitimately stopped every
1145
+ // instance would get a restart loop. Those conditions surface through /health
1146
+ // (which now returns 503) and through the General-topic notifications instead.
1139
1147
  this.watchdogTimer = setInterval(() => sdNotify("WATCHDOG=1"), 30_000);
1140
1148
  // EventLog.prune() existed but was never called, so `events` and `activity`
1141
1149
  // grew without bound for the life of the install. Prune once at startup and
@@ -1278,6 +1286,15 @@ export class FleetManager {
1278
1286
  // rest of startup finishes. Replay one coalesced reload only after all
1279
1287
  // startup-owned lifecycle work and signal handlers are in place.
1280
1288
  this.finishStartup();
1289
+ // Tell systemd we are ready only now. This used to fire right after the
1290
+ // generals started — before adapters, classic instances, topic creation and the
1291
+ // health server — so `systemctl start` returned success while the fleet was
1292
+ // still deaf: no path existed for a user message to arrive.
1293
+ sdNotify("READY=1");
1294
+ const health = this.getFleetHealth();
1295
+ if (health.status !== "ok") {
1296
+ this.logger.warn({ health }, "Fleet started with problems — see /health");
1297
+ }
1281
1298
  }
1282
1299
  /**
1283
1300
  * Delete inbox files older than retentionDays (by mtime). Cleans the shared
@@ -1394,6 +1411,66 @@ export class FleetManager {
1394
1411
  getAdapterStates() {
1395
1412
  return this.adapterState;
1396
1413
  }
1414
+ /**
1415
+ * Real, checkable fleet health for `/health` and the operator.
1416
+ *
1417
+ * `status` is:
1418
+ * - `ok` — at least one adapter connected and every configured instance
1419
+ * that should be running is running
1420
+ * - `degraded` — reachable, but something the operator should look at (an
1421
+ * adapter retrying, an instance crashed or stopped)
1422
+ * - `down` — the fleet cannot do its job: no adapter is connected, so no
1423
+ * message can arrive or be answered
1424
+ *
1425
+ * Deliberately does NOT gate the systemd watchdog — see the comment at the
1426
+ * WATCHDOG timer for why.
1427
+ */
1428
+ getFleetHealth() {
1429
+ const names = Object.keys(this.fleetConfig?.instances ?? {});
1430
+ const counts = { configured: names.length, running: 0, crashed: 0, paused: 0, stopped: 0 };
1431
+ for (const name of names) {
1432
+ const state = this.getInstanceStatus(name);
1433
+ if (state === "running")
1434
+ counts.running++;
1435
+ else if (state === "crashed")
1436
+ counts.crashed++;
1437
+ else if (state === "paused")
1438
+ counts.paused++;
1439
+ else
1440
+ counts.stopped++;
1441
+ }
1442
+ const states = {};
1443
+ let connected = 0;
1444
+ for (const [id, state] of this.adapterState) {
1445
+ states[id] = state.status;
1446
+ if (state.status === "connected")
1447
+ connected++;
1448
+ }
1449
+ const problems = [];
1450
+ if (this.adapterState.size > 0 && connected === 0)
1451
+ problems.push("no channel adapter is connected");
1452
+ if (counts.crashed > 0)
1453
+ problems.push(`${counts.crashed} instance(s) crashed`);
1454
+ for (const [id, state] of this.adapterState) {
1455
+ if (state.status !== "connected")
1456
+ problems.push(`adapter ${id} is ${state.status}`);
1457
+ }
1458
+ if (!this.startupComplete)
1459
+ problems.push("startup has not completed");
1460
+ // "down" is reserved for "cannot receive or answer a message at all". A fleet
1461
+ // with adapters configured but none connected is exactly that.
1462
+ const status = this.adapterState.size > 0 && connected === 0
1463
+ ? "down"
1464
+ : problems.length > 0 ? "degraded" : "ok";
1465
+ return {
1466
+ status,
1467
+ uptime: Math.floor((Date.now() - this.startedAt) / 1000),
1468
+ instances: counts,
1469
+ adapters: { total: this.adapterState.size, connected, states },
1470
+ startupComplete: this.startupComplete,
1471
+ problems,
1472
+ };
1473
+ }
1397
1474
  /** Start the primary adapter (backward-compatible, sets this.adapter) */
1398
1475
  async startSingleAdapter(fleet, channelConfig) {
1399
1476
  const botToken = process.env[channelConfig.bot_token_env];
@@ -2130,6 +2207,12 @@ export class FleetManager {
2130
2207
  if (this.adapterRestarting.has(id))
2131
2208
  return;
2132
2209
  this.adapterRestarting.add(id);
2210
+ // Reflect reality in adapterState throughout. This loop used to leave the state
2211
+ // untouched, so getAdapterStates() — and therefore /health and the dashboard —
2212
+ // kept reporting "connected" for an adapter that had been down for hours. An
2213
+ // adapter's true status was simply not knowable from inside the process.
2214
+ const previous = this.adapterState.get(id);
2215
+ this.adapterState.set(id, { status: "retrying", retryCount: 0, lastError: previous?.lastError });
2133
2216
  try {
2134
2217
  for (let attempt = 1;; attempt++) {
2135
2218
  if (this.ipcStoppingInstances.has("__fleet_stopping__"))
@@ -2142,9 +2225,16 @@ export class FleetManager {
2142
2225
  await adapter.stop().catch(() => { });
2143
2226
  await adapter.start();
2144
2227
  this.logger.info({ id, attempt }, "Adapter restarted successfully");
2228
+ this.adapterState.set(id, { status: "connected", retryCount: 0 });
2145
2229
  return;
2146
2230
  }
2147
- catch { /* retry */ }
2231
+ catch (err) {
2232
+ this.adapterState.set(id, {
2233
+ status: "retrying",
2234
+ retryCount: attempt,
2235
+ lastError: err?.message ?? String(err),
2236
+ });
2237
+ }
2148
2238
  if (attempt % 10 === 0) {
2149
2239
  this.logger.warn({ id, attempt }, "Adapter restart still failing");
2150
2240
  }
@@ -3351,6 +3441,9 @@ export class FleetManager {
3351
3441
  }
3352
3442
  }
3353
3443
  async removeInstance(name) {
3444
+ // Drop cached pane context — the map is keyed by instance name and nothing
3445
+ // else evicted deleted entries, so it grew for the life of the process.
3446
+ forgetInstanceContext(name);
3354
3447
  // Clean up schedules (scheduler is fleet-level, not lifecycle-level)
3355
3448
  const config = this.fleetConfig?.instances[name];
3356
3449
  if (this.scheduler && config?.topic_id) {
@@ -5609,15 +5702,13 @@ When users create specialized instances, suggest these configurations:
5609
5702
  }
5610
5703
  }
5611
5704
  if (req.method === "GET" && req.url === "/health") {
5612
- const instanceCount = this.fleetConfig?.instances
5613
- ? Object.keys(this.fleetConfig.instances).length
5614
- : 0;
5615
- res.writeHead(200);
5616
- res.end(JSON.stringify({
5617
- status: "ok",
5618
- instances: instanceCount,
5619
- uptime: Math.floor((Date.now() - this.startedAt) / 1000),
5620
- }));
5705
+ const health = this.getFleetHealth();
5706
+ // 503 when the fleet cannot do its job, so an external monitor sees it.
5707
+ // This used to always answer 200 "ok" with a count of CONFIGURED instances,
5708
+ // so every agent could be dead and every adapter down and it still looked
5709
+ // green.
5710
+ res.writeHead(health.status === "ok" ? 200 : 503);
5711
+ res.end(JSON.stringify(health));
5621
5712
  return;
5622
5713
  }
5623
5714
  if (req.method === "GET" && req.url === "/status") {