@songsid/agend 2.1.1-beta.14 → 2.1.1-beta.16
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +8 -2
- package/dist/cli.js.map +1 -1
- package/dist/fleet-manager.d.ts +43 -0
- package/dist/fleet-manager.js +163 -40
- package/dist/fleet-manager.js.map +1 -1
- package/dist/topic-commands.d.ts +15 -2
- package/dist/topic-commands.js +78 -19
- package/dist/topic-commands.js.map +1 -1
- package/dist/web-api.js +5 -2
- package/dist/web-api.js.map +1 -1
- package/package.json +1 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -26,7 +26,7 @@ import { processAttachments } from "./channel/attachment-handler.js";
|
|
|
26
26
|
import { routeToolCall } from "./channel/tool-router.js";
|
|
27
27
|
import { Scheduler } from "./scheduler/index.js";
|
|
28
28
|
import { DEFAULT_SCHEDULER_CONFIG } from "./scheduler/index.js";
|
|
29
|
-
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext } from "./topic-commands.js";
|
|
29
|
+
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext, forgetInstanceContext } from "./topic-commands.js";
|
|
30
30
|
import { DailySummary } from "./daily-summary.js";
|
|
31
31
|
import { WebhookEmitter } from "./webhook-emitter.js";
|
|
32
32
|
import { TmuxControlClient } from "./tmux-control.js";
|
|
@@ -1134,8 +1134,16 @@ export class FleetManager {
|
|
|
1134
1134
|
}
|
|
1135
1135
|
}
|
|
1136
1136
|
}
|
|
1137
|
-
//
|
|
1138
|
-
|
|
1137
|
+
// The systemd watchdog answers exactly one question: is this process still
|
|
1138
|
+
// turning its event loop? Pinging from a timer proves that, and after the
|
|
1139
|
+
// blocking child-process calls were made async it is a meaningful signal —
|
|
1140
|
+
// a deadlocked or frozen fleet stops pinging and systemd restarts it.
|
|
1141
|
+
//
|
|
1142
|
+
// It deliberately does NOT gate on fleet health. "No adapter connected" or
|
|
1143
|
+
// "an instance crashed" must not kill the process: the fleet would be restarted
|
|
1144
|
+
// into the same broken state, and a user who has legitimately stopped every
|
|
1145
|
+
// instance would get a restart loop. Those conditions surface through /health
|
|
1146
|
+
// (which now returns 503) and through the General-topic notifications instead.
|
|
1139
1147
|
this.watchdogTimer = setInterval(() => sdNotify("WATCHDOG=1"), 30_000);
|
|
1140
1148
|
// EventLog.prune() existed but was never called, so `events` and `activity`
|
|
1141
1149
|
// grew without bound for the life of the install. Prune once at startup and
|
|
@@ -1278,6 +1286,15 @@ export class FleetManager {
|
|
|
1278
1286
|
// rest of startup finishes. Replay one coalesced reload only after all
|
|
1279
1287
|
// startup-owned lifecycle work and signal handlers are in place.
|
|
1280
1288
|
this.finishStartup();
|
|
1289
|
+
// Tell systemd we are ready only now. This used to fire right after the
|
|
1290
|
+
// generals started — before adapters, classic instances, topic creation and the
|
|
1291
|
+
// health server — so `systemctl start` returned success while the fleet was
|
|
1292
|
+
// still deaf: no path existed for a user message to arrive.
|
|
1293
|
+
sdNotify("READY=1");
|
|
1294
|
+
const health = this.getFleetHealth();
|
|
1295
|
+
if (health.status !== "ok") {
|
|
1296
|
+
this.logger.warn({ health }, "Fleet started with problems — see /health");
|
|
1297
|
+
}
|
|
1281
1298
|
}
|
|
1282
1299
|
/**
|
|
1283
1300
|
* Delete inbox files older than retentionDays (by mtime). Cleans the shared
|
|
@@ -1394,6 +1411,66 @@ export class FleetManager {
|
|
|
1394
1411
|
getAdapterStates() {
|
|
1395
1412
|
return this.adapterState;
|
|
1396
1413
|
}
|
|
1414
|
+
/**
|
|
1415
|
+
* Real, checkable fleet health for `/health` and the operator.
|
|
1416
|
+
*
|
|
1417
|
+
* `status` is:
|
|
1418
|
+
* - `ok` — at least one adapter connected and every configured instance
|
|
1419
|
+
* that should be running is running
|
|
1420
|
+
* - `degraded` — reachable, but something the operator should look at (an
|
|
1421
|
+
* adapter retrying, an instance crashed or stopped)
|
|
1422
|
+
* - `down` — the fleet cannot do its job: no adapter is connected, so no
|
|
1423
|
+
* message can arrive or be answered
|
|
1424
|
+
*
|
|
1425
|
+
* Deliberately does NOT gate the systemd watchdog — see the comment at the
|
|
1426
|
+
* WATCHDOG timer for why.
|
|
1427
|
+
*/
|
|
1428
|
+
getFleetHealth() {
|
|
1429
|
+
const names = Object.keys(this.fleetConfig?.instances ?? {});
|
|
1430
|
+
const counts = { configured: names.length, running: 0, crashed: 0, paused: 0, stopped: 0 };
|
|
1431
|
+
for (const name of names) {
|
|
1432
|
+
const state = this.getInstanceStatus(name);
|
|
1433
|
+
if (state === "running")
|
|
1434
|
+
counts.running++;
|
|
1435
|
+
else if (state === "crashed")
|
|
1436
|
+
counts.crashed++;
|
|
1437
|
+
else if (state === "paused")
|
|
1438
|
+
counts.paused++;
|
|
1439
|
+
else
|
|
1440
|
+
counts.stopped++;
|
|
1441
|
+
}
|
|
1442
|
+
const states = {};
|
|
1443
|
+
let connected = 0;
|
|
1444
|
+
for (const [id, state] of this.adapterState) {
|
|
1445
|
+
states[id] = state.status;
|
|
1446
|
+
if (state.status === "connected")
|
|
1447
|
+
connected++;
|
|
1448
|
+
}
|
|
1449
|
+
const problems = [];
|
|
1450
|
+
if (this.adapterState.size > 0 && connected === 0)
|
|
1451
|
+
problems.push("no channel adapter is connected");
|
|
1452
|
+
if (counts.crashed > 0)
|
|
1453
|
+
problems.push(`${counts.crashed} instance(s) crashed`);
|
|
1454
|
+
for (const [id, state] of this.adapterState) {
|
|
1455
|
+
if (state.status !== "connected")
|
|
1456
|
+
problems.push(`adapter ${id} is ${state.status}`);
|
|
1457
|
+
}
|
|
1458
|
+
if (!this.startupComplete)
|
|
1459
|
+
problems.push("startup has not completed");
|
|
1460
|
+
// "down" is reserved for "cannot receive or answer a message at all". A fleet
|
|
1461
|
+
// with adapters configured but none connected is exactly that.
|
|
1462
|
+
const status = this.adapterState.size > 0 && connected === 0
|
|
1463
|
+
? "down"
|
|
1464
|
+
: problems.length > 0 ? "degraded" : "ok";
|
|
1465
|
+
return {
|
|
1466
|
+
status,
|
|
1467
|
+
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
1468
|
+
instances: counts,
|
|
1469
|
+
adapters: { total: this.adapterState.size, connected, states },
|
|
1470
|
+
startupComplete: this.startupComplete,
|
|
1471
|
+
problems,
|
|
1472
|
+
};
|
|
1473
|
+
}
|
|
1397
1474
|
/** Start the primary adapter (backward-compatible, sets this.adapter) */
|
|
1398
1475
|
async startSingleAdapter(fleet, channelConfig) {
|
|
1399
1476
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
@@ -1594,17 +1671,7 @@ export class FleetManager {
|
|
|
1594
1671
|
await data.respond(t("not_authorized"));
|
|
1595
1672
|
return;
|
|
1596
1673
|
}
|
|
1597
|
-
|
|
1598
|
-
const { execSync } = await import("node:child_process");
|
|
1599
|
-
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
1600
|
-
const result = execSync(`agend backend doctor ${backend}`, { timeout: 30_000, encoding: "utf-8" });
|
|
1601
|
-
const clean = result.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1602
|
-
await data.respond(clean || "No output");
|
|
1603
|
-
}
|
|
1604
|
-
catch (err) {
|
|
1605
|
-
const output = (err.stdout ?? err.message ?? "Doctor failed").replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1606
|
-
await data.respond(output);
|
|
1607
|
-
}
|
|
1674
|
+
await data.respond(await this.runBackendDoctor());
|
|
1608
1675
|
}
|
|
1609
1676
|
else if (data.command === "status") {
|
|
1610
1677
|
const text = await this.topicCommands.getStatusText();
|
|
@@ -1876,17 +1943,7 @@ export class FleetManager {
|
|
|
1876
1943
|
await data.respond(t("not_authorized"));
|
|
1877
1944
|
return;
|
|
1878
1945
|
}
|
|
1879
|
-
|
|
1880
|
-
const { execSync } = await import("node:child_process");
|
|
1881
|
-
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
1882
|
-
const result = execSync(`agend backend doctor ${backend}`, { timeout: 30_000, encoding: "utf-8" });
|
|
1883
|
-
const clean = result.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1884
|
-
await data.respond(clean || "No output");
|
|
1885
|
-
}
|
|
1886
|
-
catch (err) {
|
|
1887
|
-
const output = (err.stdout ?? err.message ?? "Doctor failed").replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1888
|
-
await data.respond(output);
|
|
1889
|
-
}
|
|
1946
|
+
await data.respond(await this.runBackendDoctor());
|
|
1890
1947
|
}
|
|
1891
1948
|
else if (data.command === "status") {
|
|
1892
1949
|
const text = await this.topicCommands.getStatusText();
|
|
@@ -2114,9 +2171,22 @@ export class FleetManager {
|
|
|
2114
2171
|
if (existsSync(windowIdPath)) {
|
|
2115
2172
|
const windowId = readFileSync(windowIdPath, "utf-8").trim();
|
|
2116
2173
|
if (windowId) {
|
|
2174
|
+
// Async with an explicit timeout: this was execSync with NO timeout at
|
|
2175
|
+
// all, so a wedged tmux server blocked the whole fleet event loop
|
|
2176
|
+
// indefinitely — while we were here to diagnose a lost connection.
|
|
2177
|
+
// A timeout is also the correct signal: an unresponsive tmux server
|
|
2178
|
+
// means we cannot verify the pane, which is treated as dead (the same
|
|
2179
|
+
// conclusion the old code reached only by throwing).
|
|
2117
2180
|
try {
|
|
2118
|
-
const {
|
|
2119
|
-
|
|
2181
|
+
const { execFile } = await import("node:child_process");
|
|
2182
|
+
const { promisify } = await import("node:util");
|
|
2183
|
+
const { getTmuxSocketName } = await import("./paths.js");
|
|
2184
|
+
// Honour socket isolation: without -L this queried the user's default
|
|
2185
|
+
// tmux server instead of the fleet's, so under a custom AGEND_HOME the
|
|
2186
|
+
// check was meaningless (it reported every pane dead).
|
|
2187
|
+
const socket = getTmuxSocketName();
|
|
2188
|
+
const args = socket ? ["-L", socket, "list-panes", "-t", windowId] : ["list-panes", "-t", windowId];
|
|
2189
|
+
await promisify(execFile)("tmux", args, { timeout: 5_000 });
|
|
2120
2190
|
}
|
|
2121
2191
|
catch {
|
|
2122
2192
|
// Pane dead — respawn
|
|
@@ -2137,6 +2207,12 @@ export class FleetManager {
|
|
|
2137
2207
|
if (this.adapterRestarting.has(id))
|
|
2138
2208
|
return;
|
|
2139
2209
|
this.adapterRestarting.add(id);
|
|
2210
|
+
// Reflect reality in adapterState throughout. This loop used to leave the state
|
|
2211
|
+
// untouched, so getAdapterStates() — and therefore /health and the dashboard —
|
|
2212
|
+
// kept reporting "connected" for an adapter that had been down for hours. An
|
|
2213
|
+
// adapter's true status was simply not knowable from inside the process.
|
|
2214
|
+
const previous = this.adapterState.get(id);
|
|
2215
|
+
this.adapterState.set(id, { status: "retrying", retryCount: 0, lastError: previous?.lastError });
|
|
2140
2216
|
try {
|
|
2141
2217
|
for (let attempt = 1;; attempt++) {
|
|
2142
2218
|
if (this.ipcStoppingInstances.has("__fleet_stopping__"))
|
|
@@ -2149,9 +2225,16 @@ export class FleetManager {
|
|
|
2149
2225
|
await adapter.stop().catch(() => { });
|
|
2150
2226
|
await adapter.start();
|
|
2151
2227
|
this.logger.info({ id, attempt }, "Adapter restarted successfully");
|
|
2228
|
+
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
2152
2229
|
return;
|
|
2153
2230
|
}
|
|
2154
|
-
catch {
|
|
2231
|
+
catch (err) {
|
|
2232
|
+
this.adapterState.set(id, {
|
|
2233
|
+
status: "retrying",
|
|
2234
|
+
retryCount: attempt,
|
|
2235
|
+
lastError: err?.message ?? String(err),
|
|
2236
|
+
});
|
|
2237
|
+
}
|
|
2155
2238
|
if (attempt % 10 === 0) {
|
|
2156
2239
|
this.logger.warn({ id, attempt }, "Adapter restart still failing");
|
|
2157
2240
|
}
|
|
@@ -3358,6 +3441,9 @@ export class FleetManager {
|
|
|
3358
3441
|
}
|
|
3359
3442
|
}
|
|
3360
3443
|
async removeInstance(name) {
|
|
3444
|
+
// Drop cached pane context — the map is keyed by instance name and nothing
|
|
3445
|
+
// else evicted deleted entries, so it grew for the life of the process.
|
|
3446
|
+
forgetInstanceContext(name);
|
|
3361
3447
|
// Clean up schedules (scheduler is fleet-level, not lifecycle-level)
|
|
3362
3448
|
const config = this.fleetConfig?.instances[name];
|
|
3363
3449
|
if (this.scheduler && config?.topic_id) {
|
|
@@ -3476,6 +3562,35 @@ export class FleetManager {
|
|
|
3476
3562
|
* So: try, move a bad file aside and retry once with a fresh one, and if even
|
|
3477
3563
|
* that fails carry on without an event log.
|
|
3478
3564
|
*/
|
|
3565
|
+
/**
|
|
3566
|
+
* Run `agend backend doctor` for the fleet's default backend and return its
|
|
3567
|
+
* cleaned output.
|
|
3568
|
+
*
|
|
3569
|
+
* Async on purpose: this was `execSync` with a 30s timeout, reachable by any
|
|
3570
|
+
* allowlisted user through `/doctor`. While it ran, the entire fleet event loop
|
|
3571
|
+
* was frozen — no IPC, no adapter, no message delivery, no health responses,
|
|
3572
|
+
* and critically no WATCHDOG ping, so a slow doctor could push past
|
|
3573
|
+
* WatchdogSec and have systemd SIGABRT the fleet.
|
|
3574
|
+
*/
|
|
3575
|
+
async runBackendDoctor() {
|
|
3576
|
+
const stripAnsi = (s) => s.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
3577
|
+
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
3578
|
+
try {
|
|
3579
|
+
const { execFile } = await import("node:child_process");
|
|
3580
|
+
const { promisify } = await import("node:util");
|
|
3581
|
+
// execFile with an argv array — no shell, so the backend name cannot be
|
|
3582
|
+
// interpreted as a command even if config is malformed.
|
|
3583
|
+
const { stdout } = await promisify(execFile)("agend", ["backend", "doctor", backend], {
|
|
3584
|
+
timeout: 30_000,
|
|
3585
|
+
encoding: "utf-8",
|
|
3586
|
+
});
|
|
3587
|
+
return stripAnsi(stdout) || "No output";
|
|
3588
|
+
}
|
|
3589
|
+
catch (err) {
|
|
3590
|
+
const e = err;
|
|
3591
|
+
return stripAnsi(e.stdout ?? e.message ?? "Doctor failed");
|
|
3592
|
+
}
|
|
3593
|
+
}
|
|
3479
3594
|
/** Drop event/activity rows older than the retention window. Best-effort. */
|
|
3480
3595
|
pruneEventLog() {
|
|
3481
3596
|
try {
|
|
@@ -5441,10 +5556,20 @@ When users create specialized instances, suggest these configurations:
|
|
|
5441
5556
|
// ── Update check ────────────────────────────────────────────────────
|
|
5442
5557
|
async checkForUpdates() {
|
|
5443
5558
|
try {
|
|
5444
|
-
|
|
5559
|
+
// Both npm lookups are async: as execSync they froze the fleet event loop for
|
|
5560
|
+
// up to 15s each, and on a beta build BOTH ran — 30s with no WATCHDOG ping,
|
|
5561
|
+
// past WatchdogSec's half-interval and enough for systemd to SIGABRT the fleet
|
|
5562
|
+
// for a background version check.
|
|
5563
|
+
const { execFile } = await import("node:child_process");
|
|
5564
|
+
const { promisify } = await import("node:util");
|
|
5565
|
+
const execFileP = promisify(execFile);
|
|
5566
|
+
const npmVersion = async (spec) => {
|
|
5567
|
+
const { stdout } = await execFileP("npm", ["view", spec, "version"], { timeout: 15_000 });
|
|
5568
|
+
return stdout.toString().trim();
|
|
5569
|
+
};
|
|
5445
5570
|
const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "..", "package.json");
|
|
5446
5571
|
const currentVersion = JSON.parse(readFileSync(pkgPath, "utf-8")).version ?? "0.0.0";
|
|
5447
|
-
const latest =
|
|
5572
|
+
const latest = await npmVersion("@songsid/agend");
|
|
5448
5573
|
let target = latest;
|
|
5449
5574
|
if (currentVersion.includes("-beta")) {
|
|
5450
5575
|
// Beta users track the @beta channel (never fall back to @latest, which is
|
|
@@ -5452,7 +5577,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
5452
5577
|
// of beta/latest is the newest.
|
|
5453
5578
|
let beta = "";
|
|
5454
5579
|
try {
|
|
5455
|
-
beta =
|
|
5580
|
+
beta = await npmVersion("@songsid/agend@beta");
|
|
5456
5581
|
}
|
|
5457
5582
|
catch { /* no beta tag */ }
|
|
5458
5583
|
target = beta || latest;
|
|
@@ -5577,15 +5702,13 @@ When users create specialized instances, suggest these configurations:
|
|
|
5577
5702
|
}
|
|
5578
5703
|
}
|
|
5579
5704
|
if (req.method === "GET" && req.url === "/health") {
|
|
5580
|
-
const
|
|
5581
|
-
|
|
5582
|
-
|
|
5583
|
-
|
|
5584
|
-
|
|
5585
|
-
|
|
5586
|
-
|
|
5587
|
-
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
5588
|
-
}));
|
|
5705
|
+
const health = this.getFleetHealth();
|
|
5706
|
+
// 503 when the fleet cannot do its job, so an external monitor sees it.
|
|
5707
|
+
// This used to always answer 200 "ok" with a count of CONFIGURED instances,
|
|
5708
|
+
// so every agent could be dead and every adapter down and it still looked
|
|
5709
|
+
// green.
|
|
5710
|
+
res.writeHead(health.status === "ok" ? 200 : 503);
|
|
5711
|
+
res.end(JSON.stringify(health));
|
|
5589
5712
|
return;
|
|
5590
5713
|
}
|
|
5591
5714
|
if (req.method === "GET" && req.url === "/status") {
|