@songsid/agend 2.1.1-beta.15 → 2.1.1-beta.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/channel/ipc-timeouts.d.ts +40 -0
- package/dist/channel/ipc-timeouts.js +58 -0
- package/dist/channel/ipc-timeouts.js.map +1 -0
- package/dist/channel/mcp-server.js +5 -4
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/cli.js +8 -2
- package/dist/cli.js.map +1 -1
- package/dist/daemon.d.ts +13 -0
- package/dist/daemon.js +32 -9
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-manager.d.ts +32 -0
- package/dist/fleet-manager.js +120 -14
- package/dist/fleet-manager.js.map +1 -1
- package/dist/instance-lifecycle.js +9 -0
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/outbound-handlers.d.ts +7 -0
- package/dist/outbound-handlers.js +62 -4
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/topic-commands.d.ts +15 -2
- package/dist/topic-commands.js +62 -12
- package/dist/topic-commands.js.map +1 -1
- package/package.json +1 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -26,7 +26,7 @@ import { processAttachments } from "./channel/attachment-handler.js";
|
|
|
26
26
|
import { routeToolCall } from "./channel/tool-router.js";
|
|
27
27
|
import { Scheduler } from "./scheduler/index.js";
|
|
28
28
|
import { DEFAULT_SCHEDULER_CONFIG } from "./scheduler/index.js";
|
|
29
|
-
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext } from "./topic-commands.js";
|
|
29
|
+
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext, forgetInstanceContext } from "./topic-commands.js";
|
|
30
30
|
import { DailySummary } from "./daily-summary.js";
|
|
31
31
|
import { WebhookEmitter } from "./webhook-emitter.js";
|
|
32
32
|
import { TmuxControlClient } from "./tmux-control.js";
|
|
@@ -218,7 +218,13 @@ export class FleetManager {
|
|
|
218
218
|
}
|
|
219
219
|
this.reloadPending = false;
|
|
220
220
|
this.reconcileInFlight = this.reconcileInstances()
|
|
221
|
-
.catch(err =>
|
|
221
|
+
.catch(err => {
|
|
222
|
+
// Almost always a YAML parse error. Log-only meant the user edited
|
|
223
|
+
// fleet.yaml, sent SIGHUP, and got no reaction and no explanation.
|
|
224
|
+
this.logger.error({ err }, "SIGHUP config reload failed");
|
|
225
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
226
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload FAILED — ${message}\nThe previous configuration is still running.`);
|
|
227
|
+
})
|
|
222
228
|
.finally(() => {
|
|
223
229
|
this.reconcileInFlight = null;
|
|
224
230
|
if (this.reloadPending && this.startupComplete) {
|
|
@@ -1134,8 +1140,16 @@ export class FleetManager {
|
|
|
1134
1140
|
}
|
|
1135
1141
|
}
|
|
1136
1142
|
}
|
|
1137
|
-
//
|
|
1138
|
-
|
|
1143
|
+
// The systemd watchdog answers exactly one question: is this process still
|
|
1144
|
+
// turning its event loop? Pinging from a timer proves that, and after the
|
|
1145
|
+
// blocking child-process calls were made async it is a meaningful signal —
|
|
1146
|
+
// a deadlocked or frozen fleet stops pinging and systemd restarts it.
|
|
1147
|
+
//
|
|
1148
|
+
// It deliberately does NOT gate on fleet health. "No adapter connected" or
|
|
1149
|
+
// "an instance crashed" must not kill the process: the fleet would be restarted
|
|
1150
|
+
// into the same broken state, and a user who has legitimately stopped every
|
|
1151
|
+
// instance would get a restart loop. Those conditions surface through /health
|
|
1152
|
+
// (which now returns 503) and through the General-topic notifications instead.
|
|
1139
1153
|
this.watchdogTimer = setInterval(() => sdNotify("WATCHDOG=1"), 30_000);
|
|
1140
1154
|
// EventLog.prune() existed but was never called, so `events` and `activity`
|
|
1141
1155
|
// grew without bound for the life of the install. Prune once at startup and
|
|
@@ -1278,6 +1292,15 @@ export class FleetManager {
|
|
|
1278
1292
|
// rest of startup finishes. Replay one coalesced reload only after all
|
|
1279
1293
|
// startup-owned lifecycle work and signal handlers are in place.
|
|
1280
1294
|
this.finishStartup();
|
|
1295
|
+
// Tell systemd we are ready only now. This used to fire right after the
|
|
1296
|
+
// generals started — before adapters, classic instances, topic creation and the
|
|
1297
|
+
// health server — so `systemctl start` returned success while the fleet was
|
|
1298
|
+
// still deaf: no path existed for a user message to arrive.
|
|
1299
|
+
sdNotify("READY=1");
|
|
1300
|
+
const health = this.getFleetHealth();
|
|
1301
|
+
if (health.status !== "ok") {
|
|
1302
|
+
this.logger.warn({ health }, "Fleet started with problems — see /health");
|
|
1303
|
+
}
|
|
1281
1304
|
}
|
|
1282
1305
|
/**
|
|
1283
1306
|
* Delete inbox files older than retentionDays (by mtime). Cleans the shared
|
|
@@ -1394,6 +1417,66 @@ export class FleetManager {
|
|
|
1394
1417
|
getAdapterStates() {
|
|
1395
1418
|
return this.adapterState;
|
|
1396
1419
|
}
|
|
1420
|
+
/**
|
|
1421
|
+
* Real, checkable fleet health for `/health` and the operator.
|
|
1422
|
+
*
|
|
1423
|
+
* `status` is:
|
|
1424
|
+
* - `ok` — at least one adapter connected and every configured instance
|
|
1425
|
+
* that should be running is running
|
|
1426
|
+
* - `degraded` — reachable, but something the operator should look at (an
|
|
1427
|
+
* adapter retrying, an instance crashed or stopped)
|
|
1428
|
+
* - `down` — the fleet cannot do its job: no adapter is connected, so no
|
|
1429
|
+
* message can arrive or be answered
|
|
1430
|
+
*
|
|
1431
|
+
* Deliberately does NOT gate the systemd watchdog — see the comment at the
|
|
1432
|
+
* WATCHDOG timer for why.
|
|
1433
|
+
*/
|
|
1434
|
+
getFleetHealth() {
|
|
1435
|
+
const names = Object.keys(this.fleetConfig?.instances ?? {});
|
|
1436
|
+
const counts = { configured: names.length, running: 0, crashed: 0, paused: 0, stopped: 0 };
|
|
1437
|
+
for (const name of names) {
|
|
1438
|
+
const state = this.getInstanceStatus(name);
|
|
1439
|
+
if (state === "running")
|
|
1440
|
+
counts.running++;
|
|
1441
|
+
else if (state === "crashed")
|
|
1442
|
+
counts.crashed++;
|
|
1443
|
+
else if (state === "paused")
|
|
1444
|
+
counts.paused++;
|
|
1445
|
+
else
|
|
1446
|
+
counts.stopped++;
|
|
1447
|
+
}
|
|
1448
|
+
const states = {};
|
|
1449
|
+
let connected = 0;
|
|
1450
|
+
for (const [id, state] of this.adapterState) {
|
|
1451
|
+
states[id] = state.status;
|
|
1452
|
+
if (state.status === "connected")
|
|
1453
|
+
connected++;
|
|
1454
|
+
}
|
|
1455
|
+
const problems = [];
|
|
1456
|
+
if (this.adapterState.size > 0 && connected === 0)
|
|
1457
|
+
problems.push("no channel adapter is connected");
|
|
1458
|
+
if (counts.crashed > 0)
|
|
1459
|
+
problems.push(`${counts.crashed} instance(s) crashed`);
|
|
1460
|
+
for (const [id, state] of this.adapterState) {
|
|
1461
|
+
if (state.status !== "connected")
|
|
1462
|
+
problems.push(`adapter ${id} is ${state.status}`);
|
|
1463
|
+
}
|
|
1464
|
+
if (!this.startupComplete)
|
|
1465
|
+
problems.push("startup has not completed");
|
|
1466
|
+
// "down" is reserved for "cannot receive or answer a message at all". A fleet
|
|
1467
|
+
// with adapters configured but none connected is exactly that.
|
|
1468
|
+
const status = this.adapterState.size > 0 && connected === 0
|
|
1469
|
+
? "down"
|
|
1470
|
+
: problems.length > 0 ? "degraded" : "ok";
|
|
1471
|
+
return {
|
|
1472
|
+
status,
|
|
1473
|
+
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
1474
|
+
instances: counts,
|
|
1475
|
+
adapters: { total: this.adapterState.size, connected, states },
|
|
1476
|
+
startupComplete: this.startupComplete,
|
|
1477
|
+
problems,
|
|
1478
|
+
};
|
|
1479
|
+
}
|
|
1397
1480
|
/** Start the primary adapter (backward-compatible, sets this.adapter) */
|
|
1398
1481
|
async startSingleAdapter(fleet, channelConfig) {
|
|
1399
1482
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
@@ -2130,6 +2213,12 @@ export class FleetManager {
|
|
|
2130
2213
|
if (this.adapterRestarting.has(id))
|
|
2131
2214
|
return;
|
|
2132
2215
|
this.adapterRestarting.add(id);
|
|
2216
|
+
// Reflect reality in adapterState throughout. This loop used to leave the state
|
|
2217
|
+
// untouched, so getAdapterStates() — and therefore /health and the dashboard —
|
|
2218
|
+
// kept reporting "connected" for an adapter that had been down for hours. An
|
|
2219
|
+
// adapter's true status was simply not knowable from inside the process.
|
|
2220
|
+
const previous = this.adapterState.get(id);
|
|
2221
|
+
this.adapterState.set(id, { status: "retrying", retryCount: 0, lastError: previous?.lastError });
|
|
2133
2222
|
try {
|
|
2134
2223
|
for (let attempt = 1;; attempt++) {
|
|
2135
2224
|
if (this.ipcStoppingInstances.has("__fleet_stopping__"))
|
|
@@ -2142,9 +2231,16 @@ export class FleetManager {
|
|
|
2142
2231
|
await adapter.stop().catch(() => { });
|
|
2143
2232
|
await adapter.start();
|
|
2144
2233
|
this.logger.info({ id, attempt }, "Adapter restarted successfully");
|
|
2234
|
+
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
2145
2235
|
return;
|
|
2146
2236
|
}
|
|
2147
|
-
catch {
|
|
2237
|
+
catch (err) {
|
|
2238
|
+
this.adapterState.set(id, {
|
|
2239
|
+
status: "retrying",
|
|
2240
|
+
retryCount: attempt,
|
|
2241
|
+
lastError: err?.message ?? String(err),
|
|
2242
|
+
});
|
|
2243
|
+
}
|
|
2148
2244
|
if (attempt % 10 === 0) {
|
|
2149
2245
|
this.logger.warn({ id, attempt }, "Adapter restart still failing");
|
|
2150
2246
|
}
|
|
@@ -3351,6 +3447,9 @@ export class FleetManager {
|
|
|
3351
3447
|
}
|
|
3352
3448
|
}
|
|
3353
3449
|
async removeInstance(name) {
|
|
3450
|
+
// Drop cached pane context — the map is keyed by instance name and nothing
|
|
3451
|
+
// else evicted deleted entries, so it grew for the life of the process.
|
|
3452
|
+
forgetInstanceContext(name);
|
|
3354
3453
|
// Clean up schedules (scheduler is fleet-level, not lifecycle-level)
|
|
3355
3454
|
const config = this.fleetConfig?.instances[name];
|
|
3356
3455
|
if (this.scheduler && config?.topic_id) {
|
|
@@ -5264,6 +5363,15 @@ When users create specialized instances, suggest these configurations:
|
|
|
5264
5363
|
removedRatio,
|
|
5265
5364
|
validationErrors: validation.errors,
|
|
5266
5365
|
}, "Refusing unsafe fleet config reload; running configuration was kept");
|
|
5366
|
+
// Tell the operator. A silently ignored config edit is the most confusing
|
|
5367
|
+
// possible outcome: they change fleet.yaml, send SIGHUP, and nothing happens
|
|
5368
|
+
// with no explanation anywhere they are looking.
|
|
5369
|
+
const why = !validation.valid
|
|
5370
|
+
? `validation failed:\n${validation.errors.map(e => `• ${e.path}: ${e.message}`).join("\n")}`
|
|
5371
|
+
: unsafeEmpty
|
|
5372
|
+
? `it removed every instance (${oldCount} → 0)`
|
|
5373
|
+
: `it removed more than half the instances (${oldCount} → ${newCount})`;
|
|
5374
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload REJECTED — ${why}\nThe previous configuration is still running.`);
|
|
5267
5375
|
return;
|
|
5268
5376
|
}
|
|
5269
5377
|
this.routing.rebuild(this.fleetConfig);
|
|
@@ -5609,15 +5717,13 @@ When users create specialized instances, suggest these configurations:
|
|
|
5609
5717
|
}
|
|
5610
5718
|
}
|
|
5611
5719
|
if (req.method === "GET" && req.url === "/health") {
|
|
5612
|
-
const
|
|
5613
|
-
|
|
5614
|
-
|
|
5615
|
-
|
|
5616
|
-
|
|
5617
|
-
|
|
5618
|
-
|
|
5619
|
-
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
5620
|
-
}));
|
|
5720
|
+
const health = this.getFleetHealth();
|
|
5721
|
+
// 503 when the fleet cannot do its job, so an external monitor sees it.
|
|
5722
|
+
// This used to always answer 200 "ok" with a count of CONFIGURED instances,
|
|
5723
|
+
// so every agent could be dead and every adapter down and it still looked
|
|
5724
|
+
// green.
|
|
5725
|
+
res.writeHead(health.status === "ok" ? 200 : 503);
|
|
5726
|
+
res.end(JSON.stringify(health));
|
|
5621
5727
|
return;
|
|
5622
5728
|
}
|
|
5623
5729
|
if (req.method === "GET" && req.url === "/status") {
|