@songsid/agend 2.1.1-beta.2 → 2.1.1-beta.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/README.zh-TW.md +1 -0
- package/dist/access-path.js +15 -6
- package/dist/access-path.js.map +1 -1
- package/dist/agent-cli-instructions.md +9 -0
- package/dist/backend/antigravity.d.ts +4 -1
- package/dist/backend/antigravity.js +24 -13
- package/dist/backend/antigravity.js.map +1 -1
- package/dist/backend/claude-code.d.ts +37 -0
- package/dist/backend/claude-code.js +47 -8
- package/dist/backend/claude-code.js.map +1 -1
- package/dist/backend/codex.d.ts +17 -0
- package/dist/backend/codex.js +279 -62
- package/dist/backend/codex.js.map +1 -1
- package/dist/backend/gemini-cli.js +4 -7
- package/dist/backend/gemini-cli.js.map +1 -1
- package/dist/backend/grok.d.ts +24 -0
- package/dist/backend/grok.js +57 -8
- package/dist/backend/grok.js.map +1 -1
- package/dist/backend/kiro.d.ts +27 -0
- package/dist/backend/kiro.js +96 -15
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/opencode.js +4 -7
- package/dist/backend/opencode.js.map +1 -1
- package/dist/backend/types.d.ts +44 -0
- package/dist/backend/types.js +3 -3
- package/dist/backend/types.js.map +1 -1
- package/dist/channel/adapters/discord.d.ts +15 -0
- package/dist/channel/adapters/discord.js +85 -6
- package/dist/channel/adapters/discord.js.map +1 -1
- package/dist/channel/adapters/telegram.d.ts +21 -0
- package/dist/channel/adapters/telegram.js +132 -15
- package/dist/channel/adapters/telegram.js.map +1 -1
- package/dist/channel/ipc-bridge.d.ts +17 -1
- package/dist/channel/ipc-bridge.js +53 -17
- package/dist/channel/ipc-bridge.js.map +1 -1
- package/dist/channel/ipc-timeouts.d.ts +40 -0
- package/dist/channel/ipc-timeouts.js +58 -0
- package/dist/channel/ipc-timeouts.js.map +1 -0
- package/dist/channel/mcp-server.js +30 -14
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/channel/mcp-tools.js +25 -0
- package/dist/channel/mcp-tools.js.map +1 -1
- package/dist/channel/message-queue.d.ts +1 -0
- package/dist/channel/message-queue.js +49 -17
- package/dist/channel/message-queue.js.map +1 -1
- package/dist/channel/reconnect-backoff.d.ts +17 -0
- package/dist/channel/reconnect-backoff.js +21 -0
- package/dist/channel/reconnect-backoff.js.map +1 -0
- package/dist/channel/types.d.ts +29 -0
- package/dist/classic-channel-manager.js +1 -5
- package/dist/classic-channel-manager.js.map +1 -1
- package/dist/cli.js +148 -60
- package/dist/cli.js.map +1 -1
- package/dist/completion.d.ts +27 -0
- package/dist/completion.js +121 -0
- package/dist/completion.js.map +1 -0
- package/dist/config-validator.js +27 -0
- package/dist/config-validator.js.map +1 -1
- package/dist/config.js +6 -0
- package/dist/config.js.map +1 -1
- package/dist/cost-guard.d.ts +3 -1
- package/dist/cost-guard.js +3 -1
- package/dist/cost-guard.js.map +1 -1
- package/dist/daemon.d.ts +230 -13
- package/dist/daemon.js +975 -335
- package/dist/daemon.js.map +1 -1
- package/dist/event-log.d.ts +31 -0
- package/dist/event-log.js +96 -0
- package/dist/event-log.js.map +1 -1
- package/dist/fleet-context.d.ts +2 -0
- package/dist/fleet-manager.d.ts +209 -4
- package/dist/fleet-manager.js +777 -92
- package/dist/fleet-manager.js.map +1 -1
- package/dist/general-knowledge/skills/cross-instance-messaging/SKILL.md +22 -0
- package/dist/general-knowledge/skills/model-discovery/SKILL.md +14 -23
- package/dist/general-knowledge/skills/session-management/SKILL.md +15 -34
- package/dist/hang-detector.d.ts +19 -13
- package/dist/hang-detector.js +19 -49
- package/dist/hang-detector.js.map +1 -1
- package/dist/instance-lifecycle.d.ts +21 -1
- package/dist/instance-lifecycle.js +141 -7
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/instructions.d.ts +5 -0
- package/dist/instructions.js +10 -5
- package/dist/instructions.js.map +1 -1
- package/dist/locale.js +2 -0
- package/dist/locale.js.map +1 -1
- package/dist/logger.js +14 -0
- package/dist/logger.js.map +1 -1
- package/dist/mcp-liveness.d.ts +21 -0
- package/dist/mcp-liveness.js +27 -0
- package/dist/mcp-liveness.js.map +1 -0
- package/dist/outbound-handlers.d.ts +16 -0
- package/dist/outbound-handlers.js +104 -26
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/outbound-schemas.d.ts +2 -0
- package/dist/outbound-schemas.js +10 -2
- package/dist/outbound-schemas.js.map +1 -1
- package/dist/pane-write-lock.d.ts +48 -0
- package/dist/pane-write-lock.js +73 -0
- package/dist/pane-write-lock.js.map +1 -0
- package/dist/process-memory.d.ts +31 -0
- package/dist/process-memory.js +79 -0
- package/dist/process-memory.js.map +1 -0
- package/dist/quickstart.js +17 -16
- package/dist/quickstart.js.map +1 -1
- package/dist/scheduler/db.js +3 -0
- package/dist/scheduler/db.js.map +1 -1
- package/dist/sd-notify.d.ts +45 -0
- package/dist/sd-notify.js +74 -3
- package/dist/sd-notify.js.map +1 -1
- package/dist/secret-file.d.ts +33 -0
- package/dist/secret-file.js +36 -0
- package/dist/secret-file.js.map +1 -0
- package/dist/setup-wizard.js +9 -7
- package/dist/setup-wizard.js.map +1 -1
- package/dist/tmux-control.d.ts +58 -5
- package/dist/tmux-control.js +107 -13
- package/dist/tmux-control.js.map +1 -1
- package/dist/tmux-manager.d.ts +47 -2
- package/dist/tmux-manager.js +103 -8
- package/dist/tmux-manager.js.map +1 -1
- package/dist/topic-commands.d.ts +35 -1
- package/dist/topic-commands.js +167 -66
- package/dist/topic-commands.js.map +1 -1
- package/dist/tui-glyphs.d.ts +25 -0
- package/dist/tui-glyphs.js +26 -0
- package/dist/tui-glyphs.js.map +1 -0
- package/dist/types.d.ts +19 -0
- package/dist/ui/view.html +291 -33
- package/dist/update-check.d.ts +22 -0
- package/dist/update-check.js +44 -0
- package/dist/update-check.js.map +1 -0
- package/dist/usage/providers.d.ts +50 -0
- package/dist/usage/providers.js +611 -0
- package/dist/usage/providers.js.map +1 -0
- package/dist/usage/usage-api.d.ts +35 -0
- package/dist/usage/usage-api.js +52 -0
- package/dist/usage/usage-api.js.map +1 -0
- package/dist/view-api.d.ts +10 -0
- package/dist/view-api.js +67 -14
- package/dist/view-api.js.map +1 -1
- package/dist/web-api.js +5 -2
- package/dist/web-api.js.map +1 -1
- package/package.json +4 -1
- package/templates/systemd.service.ejs +9 -1
- package/dist/channel/tool-tracker.d.ts +0 -13
- package/dist/channel/tool-tracker.js +0 -58
- package/dist/channel/tool-tracker.js.map +0 -1
- package/dist/daemon-entry.d.ts +0 -1
- package/dist/daemon-entry.js +0 -30
- package/dist/daemon-entry.js.map +0 -1
- package/dist/fleet-system-prompt.d.ts +0 -11
- package/dist/fleet-system-prompt.js +0 -61
- package/dist/fleet-system-prompt.js.map +0 -1
package/dist/fleet-manager.js
CHANGED
|
@@ -4,7 +4,8 @@ import { createServer } from "node:http";
|
|
|
4
4
|
import { join, dirname, basename } from "node:path";
|
|
5
5
|
import { fileURLToPath } from "node:url";
|
|
6
6
|
import { getAgendHome, ensureWorkspaceGit } from "./paths.js";
|
|
7
|
-
import { sdNotify } from "./sd-notify.js";
|
|
7
|
+
import { sdNotify, sdNotifyBlocking } from "./sd-notify.js";
|
|
8
|
+
import { readFleetMemory } from "./process-memory.js";
|
|
8
9
|
import { isScalar, parseDocument } from "yaml";
|
|
9
10
|
const __filename = fileURLToPath(import.meta.url);
|
|
10
11
|
const __dirname = dirname(__filename);
|
|
@@ -21,12 +22,12 @@ import { IpcClient } from "./channel/ipc-bridge.js";
|
|
|
21
22
|
import { createAdapter } from "./channel/factory.js";
|
|
22
23
|
import { createBackend } from "./backend/factory.js";
|
|
23
24
|
import { isModelCompatible } from "./backend/types.js";
|
|
24
|
-
import { createLogger } from "./logger.js";
|
|
25
|
+
import { createLogger, rotateLogIfNeeded } from "./logger.js";
|
|
25
26
|
import { processAttachments } from "./channel/attachment-handler.js";
|
|
26
27
|
import { routeToolCall } from "./channel/tool-router.js";
|
|
27
28
|
import { Scheduler } from "./scheduler/index.js";
|
|
28
29
|
import { DEFAULT_SCHEDULER_CONFIG } from "./scheduler/index.js";
|
|
29
|
-
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG } from "./topic-commands.js";
|
|
30
|
+
import { TopicCommands, saveCommandForBackend, parseSaveFilename, parsePauseWakeCommand, SAVE_FILENAME_RE, SAVE_UNSUPPORTED_MSG, resolveInstanceContext, forgetInstanceContext } from "./topic-commands.js";
|
|
30
31
|
import { DailySummary } from "./daily-summary.js";
|
|
31
32
|
import { WebhookEmitter } from "./webhook-emitter.js";
|
|
32
33
|
import { TmuxControlClient } from "./tmux-control.js";
|
|
@@ -38,6 +39,7 @@ import { StatuslineWatcher } from "./statusline-watcher.js";
|
|
|
38
39
|
import { outboundHandlers } from "./outbound-handlers.js";
|
|
39
40
|
import { handleWebRequest, broadcastSseEvent } from "./web-api.js";
|
|
40
41
|
import { handleViewRequest, isViewPath } from "./view-api.js";
|
|
42
|
+
import { handleUsageRequest, isUsagePath } from "./usage/usage-api.js";
|
|
41
43
|
import { handleSettingsRequest } from "./settings-api.js";
|
|
42
44
|
import { setLocale, detectLocale, t } from "./locale.js";
|
|
43
45
|
import { handleAgentRequest } from "./agent-endpoint.js";
|
|
@@ -83,6 +85,27 @@ const CANCEL_BTN_MAX_RETRIES = 3;
|
|
|
83
85
|
* buttons no clear trigger reached (e.g. a scheduled/HTTP turn that never called
|
|
84
86
|
* reply). 5min (not the old 2s idle-watch) so Thinking isn't misread as idle. */
|
|
85
87
|
const CANCEL_BTN_IDLE_CHECK_INTERVAL_MS = 5 * 60_000;
|
|
88
|
+
/**
|
|
89
|
+
* How often the cancel button's text is refreshed with elapsed working time.
|
|
90
|
+
*
|
|
91
|
+
* One edit per working instance per interval — at 60s that is trivial for both
|
|
92
|
+
* platforms' rate limits, and it reads as a live counter rather than a stale
|
|
93
|
+
* snapshot. Nothing new is posted, so the channel is never spammed: there is
|
|
94
|
+
* exactly one progress message per turn, and it is the cancel button itself.
|
|
95
|
+
*/
|
|
96
|
+
const PROGRESS_UPDATE_INTERVAL_MS = 60_000;
|
|
97
|
+
/** Elapsed time is only shown once work has clearly outlasted a quick answer. */
|
|
98
|
+
const PROGRESS_MIN_ELAPSED_MS = 2 * 60_000;
|
|
99
|
+
/** How much of a tool summary the progress line will show before eliding. */
|
|
100
|
+
const PROGRESS_ACTIVITY_MAX_CHARS = 48;
|
|
101
|
+
/**
|
|
102
|
+
* How long a delivery waits out a disconnected instance IPC before giving up.
|
|
103
|
+
*
|
|
104
|
+
* Sized for a daemon restart (socket close → respawn → CLI ready), which is the
|
|
105
|
+
* event this exists for. Past it the delivery fails loudly as it always did.
|
|
106
|
+
*/
|
|
107
|
+
const IPC_RECONNECT_GRACE_MS = 30_000;
|
|
108
|
+
const IPC_RECONNECT_POLL_MS = 250;
|
|
86
109
|
const CLASSIC_BACKEND_SELECTION_TIMEOUT_MS = 60_000;
|
|
87
110
|
const CLASSIC_BACKEND_CALLBACK_PREFIX = "classic-backend:";
|
|
88
111
|
const MODEL_SELECT_CALLBACK_PREFIX = "model-select:";
|
|
@@ -93,7 +116,7 @@ export class FleetManager {
|
|
|
93
116
|
static sighupHandlerInstalled = false;
|
|
94
117
|
children = new Map();
|
|
95
118
|
lifecycle;
|
|
96
|
-
/**
|
|
119
|
+
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
97
120
|
get daemons() { return this.lifecycle.daemons; }
|
|
98
121
|
fleetConfig = null;
|
|
99
122
|
rawFleetConfig = {};
|
|
@@ -151,6 +174,10 @@ export class FleetManager {
|
|
|
151
174
|
// reply, on cancel, or when a newer button supersedes it for the same
|
|
152
175
|
// instance. Per-button tracking means a failed delete never strands a button.
|
|
153
176
|
cancelButtons = new Map();
|
|
177
|
+
/** instanceName → what it is doing right now, when the backend can tell us. */
|
|
178
|
+
instanceActivity = new Map();
|
|
179
|
+
/** instanceName → tail of deliveries waiting for its IPC to come back. */
|
|
180
|
+
ipcWaitTails = new Map();
|
|
154
181
|
// Last user message delivered to each instance — used to react ✅ on completion.
|
|
155
182
|
lastInboundMsg = new Map();
|
|
156
183
|
topicArchiver;
|
|
@@ -163,6 +190,10 @@ export class FleetManager {
|
|
|
163
190
|
failoverActive = new Map(); // instance → current failover model
|
|
164
191
|
// IPC reconnect: tracks instances being intentionally stopped (skip reconnect)
|
|
165
192
|
ipcStoppingInstances = new Set();
|
|
193
|
+
/** Coalesce concurrent connection attempts for the same daemon socket. */
|
|
194
|
+
ipcConnectInFlight = new Map();
|
|
195
|
+
/** At most one reconnect/backoff loop may exist per instance. */
|
|
196
|
+
ipcReconnectInFlight = new Map();
|
|
166
197
|
// Adapter restart: prevents re-entrant restart attempts
|
|
167
198
|
adapterRestarting = new Set();
|
|
168
199
|
// Adapter isolation: track state per adapter for retry + visibility
|
|
@@ -172,6 +203,10 @@ export class FleetManager {
|
|
|
172
203
|
healthServer = null;
|
|
173
204
|
healthPortRetried = false;
|
|
174
205
|
updateCheckTimer = null;
|
|
206
|
+
eventLogPruneTimer = null;
|
|
207
|
+
logRotateTimer = null;
|
|
208
|
+
/** Days of event/activity history to keep. */
|
|
209
|
+
static EVENT_LOG_RETENTION_DAYS = 30;
|
|
175
210
|
watchdogTimer = null;
|
|
176
211
|
startedAt = 0;
|
|
177
212
|
// Mirror topic: buffer cross-instance messages, flush every 3s
|
|
@@ -210,7 +245,13 @@ export class FleetManager {
|
|
|
210
245
|
}
|
|
211
246
|
this.reloadPending = false;
|
|
212
247
|
this.reconcileInFlight = this.reconcileInstances()
|
|
213
|
-
.catch(err =>
|
|
248
|
+
.catch(err => {
|
|
249
|
+
// Almost always a YAML parse error. Log-only meant the user edited
|
|
250
|
+
// fleet.yaml, sent SIGHUP, and got no reaction and no explanation.
|
|
251
|
+
this.logger.error({ err }, "SIGHUP config reload failed");
|
|
252
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
253
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload FAILED — ${message}\nThe previous configuration is still running.`);
|
|
254
|
+
})
|
|
214
255
|
.finally(() => {
|
|
215
256
|
this.reconcileInFlight = null;
|
|
216
257
|
if (this.reloadPending && this.startupComplete) {
|
|
@@ -473,8 +514,12 @@ export class FleetManager {
|
|
|
473
514
|
// warm_cap: a fresh transition into idle may free this instance for eviction,
|
|
474
515
|
// or (more usefully) reveal that the fleet is now over cap. Only fire on the
|
|
475
516
|
// edge into idle, not on every idle heartbeat.
|
|
476
|
-
if (state === "idle" && previous?.state !== "idle")
|
|
517
|
+
if (state === "idle" && previous?.state !== "idle") {
|
|
477
518
|
this.enforceWarmCap();
|
|
519
|
+
// The turn is genuinely over — retire the cancel/progress button now rather
|
|
520
|
+
// than waiting for the 5-minute idle backstop to notice.
|
|
521
|
+
this.retireInstanceButtons(name);
|
|
522
|
+
}
|
|
478
523
|
}
|
|
479
524
|
cacheInstanceProcessStatus(name, status) {
|
|
480
525
|
if (status === "running") {
|
|
@@ -579,12 +624,70 @@ export class FleetManager {
|
|
|
579
624
|
if (!idle) {
|
|
580
625
|
this.logger.warn({ instanceName, timeoutMs }, "Idle gate timed out; forcing delivery");
|
|
581
626
|
}
|
|
582
|
-
|
|
583
|
-
if (!ipc?.connected)
|
|
584
|
-
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
585
|
-
ipc.send(payload);
|
|
627
|
+
await this.sendWhenConnected(instanceName, payload);
|
|
586
628
|
this.lastDeliveryAt.set(instanceName, Date.now());
|
|
587
629
|
}
|
|
630
|
+
/**
|
|
631
|
+
* Hand a payload to an instance's IPC, waiting out a *transient* disconnect.
|
|
632
|
+
*
|
|
633
|
+
* A daemon that is restarting — `/restart`, crash recovery, a model switch —
|
|
634
|
+
* drops its socket for a few seconds. Any message arriving in that window used
|
|
635
|
+
* to fail instantly: the caller logged a warning, put ❌ on the user's message,
|
|
636
|
+
* and the message was gone. The user had to notice the ❌ and retype it. That is
|
|
637
|
+
* the "instance 訊息不容易掉" goal failing on the most predictable event there is.
|
|
638
|
+
*
|
|
639
|
+
* The wait is bounded. If the instance is genuinely down, this still throws and
|
|
640
|
+
* the ❌ still appears — just for a real failure rather than a restart.
|
|
641
|
+
*
|
|
642
|
+
* Ordering is preserved by serialising behind any waiter already queued for this
|
|
643
|
+
* instance, *including* when the socket happens to be up: otherwise a message
|
|
644
|
+
* arriving after the reconnect could overtake one that has been waiting for it.
|
|
645
|
+
*/
|
|
646
|
+
async sendWhenConnected(instanceName, payload) {
|
|
647
|
+
const queued = this.ipcWaitTails.get(instanceName);
|
|
648
|
+
if (!queued) {
|
|
649
|
+
const ipc = this.instanceIpcClients.get(instanceName);
|
|
650
|
+
if (ipc?.connected && ipc.send(payload))
|
|
651
|
+
return;
|
|
652
|
+
}
|
|
653
|
+
const attempt = (queued ?? Promise.resolve())
|
|
654
|
+
.catch(() => { })
|
|
655
|
+
.then(() => this.sendAfterIpcReturns(instanceName, payload));
|
|
656
|
+
// The chain stores a settled-either-way promise so one failed delivery cannot
|
|
657
|
+
// wedge every later one, and so `queued` above is safe to await unguarded.
|
|
658
|
+
const tail = attempt.catch(() => { });
|
|
659
|
+
this.ipcWaitTails.set(instanceName, tail);
|
|
660
|
+
try {
|
|
661
|
+
await attempt;
|
|
662
|
+
}
|
|
663
|
+
finally {
|
|
664
|
+
// Only the last waiter clears the chain; while a queue is still draining the
|
|
665
|
+
// map must keep pointing at it or ordering is lost.
|
|
666
|
+
if (this.ipcWaitTails.get(instanceName) === tail) {
|
|
667
|
+
this.ipcWaitTails.delete(instanceName);
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
}
|
|
671
|
+
/** Poll for the instance's IPC to come back, then send. Throws if it does not. */
|
|
672
|
+
async sendAfterIpcReturns(instanceName, payload) {
|
|
673
|
+
const deadline = Date.now() + IPC_RECONNECT_GRACE_MS;
|
|
674
|
+
let warned = false;
|
|
675
|
+
for (;;) {
|
|
676
|
+
// Re-read every round: a reconnect replaces the IpcClient object entirely,
|
|
677
|
+
// so a cached reference would stay dead forever.
|
|
678
|
+
const ipc = this.instanceIpcClients.get(instanceName);
|
|
679
|
+
if (ipc?.connected && ipc.send(payload))
|
|
680
|
+
return;
|
|
681
|
+
if (Date.now() >= deadline) {
|
|
682
|
+
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
683
|
+
}
|
|
684
|
+
if (!warned) {
|
|
685
|
+
warned = true;
|
|
686
|
+
this.logger.info({ instanceName }, "Instance IPC is down — holding delivery until it reconnects");
|
|
687
|
+
}
|
|
688
|
+
await new Promise(resolve => setTimeout(resolve, IPC_RECONNECT_POLL_MS));
|
|
689
|
+
}
|
|
690
|
+
}
|
|
588
691
|
/** Single delivery facade: wake paused CLIs and serialize non-user work behind idle. */
|
|
589
692
|
async deliverToInstance(instanceName, payload, options = {}) {
|
|
590
693
|
const meta = payload.meta && typeof payload.meta === "object"
|
|
@@ -600,10 +703,7 @@ export class FleetManager {
|
|
|
600
703
|
await this.lifecycle.wake(instanceName, 30_000);
|
|
601
704
|
this.enforceWarmCap(instanceName); // woke one → evict a different LRU idle if over cap
|
|
602
705
|
}
|
|
603
|
-
|
|
604
|
-
if (!ipc?.connected)
|
|
605
|
-
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
606
|
-
ipc.send(payload);
|
|
706
|
+
await this.sendWhenConnected(instanceName, payload);
|
|
607
707
|
// A cross-instance item arriving before the daemon observes this turn as
|
|
608
708
|
// working must not trust the stale idle snapshot from before the send.
|
|
609
709
|
this.lastDeliveryAt.set(instanceName, Date.now());
|
|
@@ -651,7 +751,7 @@ export class FleetManager {
|
|
|
651
751
|
await new Promise(resolve => setTimeout(resolve, 250));
|
|
652
752
|
await this.startClassicInstance(instanceName, this.classicChannels.getBackendByInstance(instanceName, this.fleetConfig?.defaults?.backend), this.classicChannels.getPreTaskCommand(channel.channelId, channel.adapterId), this.classicChannels.getModel(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.model), this.classicChannels.getAutoPauseAfter(channel.channelId, channel.adapterId, this.fleetConfig?.defaults?.auto_pause_after));
|
|
653
753
|
}
|
|
654
|
-
async startInstance(name, config, topicMode) {
|
|
754
|
+
async startInstance(name, config, topicMode, kind = "fleet-topic") {
|
|
655
755
|
if (this.lifecycle.isPaused(name)) {
|
|
656
756
|
this.logger.info({ name }, "Persisted paused instance — skipping startup");
|
|
657
757
|
return;
|
|
@@ -667,7 +767,11 @@ export class FleetManager {
|
|
|
667
767
|
this.ensureGeneralInstructions(config.working_directory, config.backend);
|
|
668
768
|
}
|
|
669
769
|
this.instanceProcessStatus.delete(name);
|
|
670
|
-
await this.lifecycle.start(name, config, topicMode
|
|
770
|
+
await this.lifecycle.start(name, config, topicMode, {
|
|
771
|
+
kind,
|
|
772
|
+
backend: config.backend ?? this.fleetConfig?.defaults?.backend ?? "claude-code",
|
|
773
|
+
model: this.resolveInstanceModel(name).display,
|
|
774
|
+
});
|
|
671
775
|
// Auto-connect IPC — daemon.start() ensures socket is ready before resolving
|
|
672
776
|
await this.connectIpcToInstance(name);
|
|
673
777
|
}
|
|
@@ -845,7 +949,6 @@ export class FleetManager {
|
|
|
845
949
|
this.configPath = configPath;
|
|
846
950
|
this.loadEnvFile();
|
|
847
951
|
// Rotate fleet.log if oversized (before any logging)
|
|
848
|
-
const { rotateLogIfNeeded } = await import("./logger.js");
|
|
849
952
|
rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
|
|
850
953
|
const fleet = this.loadConfig(configPath);
|
|
851
954
|
setLocale(detectLocale(fleet)); // user-facing text language (fleet.yaml defaults.locale / timezone)
|
|
@@ -889,7 +992,7 @@ export class FleetManager {
|
|
|
889
992
|
}
|
|
890
993
|
const pidPath = join(this.dataDir, "fleet.pid");
|
|
891
994
|
writeFileSync(pidPath, String(process.pid), "utf-8");
|
|
892
|
-
this.eventLog =
|
|
995
|
+
this.eventLog = this.openEventLog();
|
|
893
996
|
// Initialize classic channel manager. The primary adapter (channels[0])
|
|
894
997
|
// migrates legacy single-bot entries and names without a suffix. Classic
|
|
895
998
|
// routing does NOT go through the routing engine (single-key, can't hold two
|
|
@@ -978,6 +1081,10 @@ export class FleetManager {
|
|
|
978
1081
|
// Rotate fleet.log daily too (besides the startup size check above), so a
|
|
979
1082
|
// long-running fleet doesn't accumulate an unbounded log.
|
|
980
1083
|
rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
|
|
1084
|
+
// Instance output.log is pipe-pane (TUI ANSI). Daemon health ticks rotate a
|
|
1085
|
+
// running instance's own log; this sweep is the safety net for every other
|
|
1086
|
+
// kind. One implementation, so the two cannot cover different sets.
|
|
1087
|
+
this.rotateAllInstanceLogs();
|
|
981
1088
|
}, () => {
|
|
982
1089
|
const instances = Object.keys(this.fleetConfig?.instances ?? {});
|
|
983
1090
|
const costMap = new Map();
|
|
@@ -1117,9 +1224,29 @@ export class FleetManager {
|
|
|
1117
1224
|
}
|
|
1118
1225
|
}
|
|
1119
1226
|
}
|
|
1120
|
-
//
|
|
1121
|
-
|
|
1227
|
+
// The systemd watchdog answers exactly one question: is this process still
|
|
1228
|
+
// turning its event loop? Pinging from a timer proves that, and after the
|
|
1229
|
+
// blocking child-process calls were made async it is a meaningful signal —
|
|
1230
|
+
// a deadlocked or frozen fleet stops pinging and systemd restarts it.
|
|
1231
|
+
//
|
|
1232
|
+
// It deliberately does NOT gate on fleet health. "No adapter connected" or
|
|
1233
|
+
// "an instance crashed" must not kill the process: the fleet would be restarted
|
|
1234
|
+
// into the same broken state, and a user who has legitimately stopped every
|
|
1235
|
+
// instance would get a restart loop. Those conditions surface through /health
|
|
1236
|
+
// (which now returns 503) and through the General-topic notifications instead.
|
|
1122
1237
|
this.watchdogTimer = setInterval(() => sdNotify("WATCHDOG=1"), 30_000);
|
|
1238
|
+
// EventLog.prune() existed but was never called, so `events` and `activity`
|
|
1239
|
+
// grew without bound for the life of the install. Prune once at startup and
|
|
1240
|
+
// daily after that; the timer is unref'd so it never holds the loop open.
|
|
1241
|
+
this.pruneEventLog();
|
|
1242
|
+
this.eventLogPruneTimer = setInterval(() => this.pruneEventLog(), 24 * 60 * 60_000);
|
|
1243
|
+
this.eventLogPruneTimer.unref?.();
|
|
1244
|
+
// Same shape for pipe-pane logs, and for the same reason: the only sweep that
|
|
1245
|
+
// covered them lived inside the daily-summary callback, so it did not run at
|
|
1246
|
+
// all when summaries were off.
|
|
1247
|
+
this.rotateAllInstanceLogs();
|
|
1248
|
+
this.logRotateTimer = setInterval(() => this.rotateAllInstanceLogs(), 24 * 60 * 60_000);
|
|
1249
|
+
this.logRotateTimer.unref?.();
|
|
1123
1250
|
// Phase 2: Start remaining instances with staggered concurrency
|
|
1124
1251
|
if (others.length > 0) {
|
|
1125
1252
|
await this.startInstancesWithConcurrency(others, topicMode);
|
|
@@ -1255,6 +1382,15 @@ export class FleetManager {
|
|
|
1255
1382
|
// rest of startup finishes. Replay one coalesced reload only after all
|
|
1256
1383
|
// startup-owned lifecycle work and signal handlers are in place.
|
|
1257
1384
|
this.finishStartup();
|
|
1385
|
+
// Tell systemd we are ready only now. This used to fire right after the
|
|
1386
|
+
// generals started — before adapters, classic instances, topic creation and the
|
|
1387
|
+
// health server — so `systemctl start` returned success while the fleet was
|
|
1388
|
+
// still deaf: no path existed for a user message to arrive.
|
|
1389
|
+
sdNotify("READY=1");
|
|
1390
|
+
const health = this.getFleetHealth();
|
|
1391
|
+
if (health.status !== "ok") {
|
|
1392
|
+
this.logger.warn({ health }, "Fleet started with problems — see /health");
|
|
1393
|
+
}
|
|
1258
1394
|
}
|
|
1259
1395
|
/**
|
|
1260
1396
|
* Delete inbox files older than retentionDays (by mtime). Cleans the shared
|
|
@@ -1371,6 +1507,67 @@ export class FleetManager {
|
|
|
1371
1507
|
getAdapterStates() {
|
|
1372
1508
|
return this.adapterState;
|
|
1373
1509
|
}
|
|
1510
|
+
/**
|
|
1511
|
+
* Real, checkable fleet health for `/health` and the operator.
|
|
1512
|
+
*
|
|
1513
|
+
* `status` is:
|
|
1514
|
+
* - `ok` — at least one adapter connected and every configured instance
|
|
1515
|
+
* that should be running is running
|
|
1516
|
+
* - `degraded` — reachable, but something the operator should look at (an
|
|
1517
|
+
* adapter retrying, an instance crashed or stopped)
|
|
1518
|
+
* - `down` — the fleet cannot do its job: no adapter is connected, so no
|
|
1519
|
+
* message can arrive or be answered
|
|
1520
|
+
*
|
|
1521
|
+
* Deliberately does NOT gate the systemd watchdog — see the comment at the
|
|
1522
|
+
* WATCHDOG timer for why.
|
|
1523
|
+
*/
|
|
1524
|
+
getFleetHealth() {
|
|
1525
|
+
const names = Object.keys(this.fleetConfig?.instances ?? {});
|
|
1526
|
+
const counts = { configured: names.length, running: 0, crashed: 0, paused: 0, stopped: 0 };
|
|
1527
|
+
for (const name of names) {
|
|
1528
|
+
const state = this.getInstanceStatus(name);
|
|
1529
|
+
if (state === "running")
|
|
1530
|
+
counts.running++;
|
|
1531
|
+
else if (state === "crashed")
|
|
1532
|
+
counts.crashed++;
|
|
1533
|
+
else if (state === "paused")
|
|
1534
|
+
counts.paused++;
|
|
1535
|
+
else
|
|
1536
|
+
counts.stopped++;
|
|
1537
|
+
}
|
|
1538
|
+
const states = {};
|
|
1539
|
+
let connected = 0;
|
|
1540
|
+
for (const [id, state] of this.adapterState) {
|
|
1541
|
+
states[id] = state.status;
|
|
1542
|
+
if (state.status === "connected")
|
|
1543
|
+
connected++;
|
|
1544
|
+
}
|
|
1545
|
+
const problems = [];
|
|
1546
|
+
if (this.adapterState.size > 0 && connected === 0)
|
|
1547
|
+
problems.push("no channel adapter is connected");
|
|
1548
|
+
if (counts.crashed > 0)
|
|
1549
|
+
problems.push(`${counts.crashed} instance(s) crashed`);
|
|
1550
|
+
for (const [id, state] of this.adapterState) {
|
|
1551
|
+
if (state.status !== "connected")
|
|
1552
|
+
problems.push(`adapter ${id} is ${state.status}`);
|
|
1553
|
+
}
|
|
1554
|
+
if (!this.startupComplete)
|
|
1555
|
+
problems.push("startup has not completed");
|
|
1556
|
+
// "down" is reserved for "cannot receive or answer a message at all". A fleet
|
|
1557
|
+
// with adapters configured but none connected is exactly that.
|
|
1558
|
+
const status = this.adapterState.size > 0 && connected === 0
|
|
1559
|
+
? "down"
|
|
1560
|
+
: problems.length > 0 ? "degraded" : "ok";
|
|
1561
|
+
return {
|
|
1562
|
+
status,
|
|
1563
|
+
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
1564
|
+
instances: counts,
|
|
1565
|
+
adapters: { total: this.adapterState.size, connected, states },
|
|
1566
|
+
startupComplete: this.startupComplete,
|
|
1567
|
+
memory: readFleetMemory(),
|
|
1568
|
+
problems,
|
|
1569
|
+
};
|
|
1570
|
+
}
|
|
1374
1571
|
/** Start the primary adapter (backward-compatible, sets this.adapter) */
|
|
1375
1572
|
async startSingleAdapter(fleet, channelConfig) {
|
|
1376
1573
|
const botToken = process.env[channelConfig.bot_token_env];
|
|
@@ -1397,6 +1594,9 @@ export class FleetManager {
|
|
|
1397
1594
|
this.adapter.on("message", safeHandler(async (msg) => {
|
|
1398
1595
|
await this.handleInboundMessage(msg);
|
|
1399
1596
|
}, this.logger, "adapter.message"));
|
|
1597
|
+
this.adapter.on("reaction", safeHandler(async (r) => {
|
|
1598
|
+
await this.handleInboundReaction(r);
|
|
1599
|
+
}, this.logger, "adapter.reaction"));
|
|
1400
1600
|
this.adapter.on("callback_query", safeHandler(async (data) => {
|
|
1401
1601
|
if (await this.handleClassicBackendSelection(data))
|
|
1402
1602
|
return;
|
|
@@ -1571,17 +1771,7 @@ export class FleetManager {
|
|
|
1571
1771
|
await data.respond(t("not_authorized"));
|
|
1572
1772
|
return;
|
|
1573
1773
|
}
|
|
1574
|
-
|
|
1575
|
-
const { execSync } = await import("node:child_process");
|
|
1576
|
-
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
1577
|
-
const result = execSync(`agend backend doctor ${backend}`, { timeout: 30_000, encoding: "utf-8" });
|
|
1578
|
-
const clean = result.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1579
|
-
await data.respond(clean || "No output");
|
|
1580
|
-
}
|
|
1581
|
-
catch (err) {
|
|
1582
|
-
const output = (err.stdout ?? err.message ?? "Doctor failed").replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1583
|
-
await data.respond(output);
|
|
1584
|
-
}
|
|
1774
|
+
await data.respond(await this.runBackendDoctor());
|
|
1585
1775
|
}
|
|
1586
1776
|
else if (data.command === "status") {
|
|
1587
1777
|
const text = await this.topicCommands.getStatusText();
|
|
@@ -1692,6 +1882,9 @@ export class FleetManager {
|
|
|
1692
1882
|
adapter.on("message", safeHandler(async (msg) => {
|
|
1693
1883
|
await this.handleInboundMessage(msg);
|
|
1694
1884
|
}, this.logger, `adapter[${adapterId}].message`));
|
|
1885
|
+
adapter.on("reaction", safeHandler(async (r) => {
|
|
1886
|
+
await this.handleInboundReaction(r);
|
|
1887
|
+
}, this.logger, `adapter[${adapterId}].reaction`));
|
|
1695
1888
|
adapter.on("callback_query", safeHandler(async (data) => {
|
|
1696
1889
|
if (await this.handleClassicBackendSelection(data))
|
|
1697
1890
|
return;
|
|
@@ -1853,17 +2046,7 @@ export class FleetManager {
|
|
|
1853
2046
|
await data.respond(t("not_authorized"));
|
|
1854
2047
|
return;
|
|
1855
2048
|
}
|
|
1856
|
-
|
|
1857
|
-
const { execSync } = await import("node:child_process");
|
|
1858
|
-
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
1859
|
-
const result = execSync(`agend backend doctor ${backend}`, { timeout: 30_000, encoding: "utf-8" });
|
|
1860
|
-
const clean = result.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1861
|
-
await data.respond(clean || "No output");
|
|
1862
|
-
}
|
|
1863
|
-
catch (err) {
|
|
1864
|
-
const output = (err.stdout ?? err.message ?? "Doctor failed").replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
1865
|
-
await data.respond(output);
|
|
1866
|
-
}
|
|
2049
|
+
await data.respond(await this.runBackendDoctor());
|
|
1867
2050
|
}
|
|
1868
2051
|
else if (data.command === "status") {
|
|
1869
2052
|
const text = await this.topicCommands.getStatusText();
|
|
@@ -1931,19 +2114,38 @@ export class FleetManager {
|
|
|
1931
2114
|
this.logger.info({ adapterId, type: channelConfig.type }, "Additional adapter started");
|
|
1932
2115
|
}
|
|
1933
2116
|
/** Connect IPC to a single instance with all handlers */
|
|
1934
|
-
|
|
2117
|
+
connectIpcToInstance(name) {
|
|
2118
|
+
const inFlight = this.ipcConnectInFlight.get(name);
|
|
2119
|
+
if (inFlight)
|
|
2120
|
+
return inFlight;
|
|
2121
|
+
const connection = this.connectIpcToInstanceInternal(name)
|
|
2122
|
+
.finally(() => {
|
|
2123
|
+
if (this.ipcConnectInFlight.get(name) === connection) {
|
|
2124
|
+
this.ipcConnectInFlight.delete(name);
|
|
2125
|
+
}
|
|
2126
|
+
});
|
|
2127
|
+
this.ipcConnectInFlight.set(name, connection);
|
|
2128
|
+
return connection;
|
|
2129
|
+
}
|
|
2130
|
+
async connectIpcToInstanceInternal(name) {
|
|
1935
2131
|
// Close existing client to prevent socket leak on reconnect
|
|
1936
2132
|
const existing = this.instanceIpcClients.get(name);
|
|
1937
2133
|
if (existing) {
|
|
1938
|
-
|
|
2134
|
+
// Remove application listeners before destroying the socket. Even if a
|
|
2135
|
+
// future regression creates two clients, the replaced one cannot keep
|
|
2136
|
+
// handling fleet_outbound messages as an orphan.
|
|
2137
|
+
existing.removeAllListeners();
|
|
1939
2138
|
try {
|
|
1940
|
-
existing.close();
|
|
2139
|
+
await existing.close();
|
|
1941
2140
|
}
|
|
1942
2141
|
catch (err) {
|
|
1943
2142
|
this.logger.debug({ err, name }, "IPC client close failed (likely already closed)");
|
|
1944
2143
|
}
|
|
1945
|
-
|
|
1946
|
-
|
|
2144
|
+
finally {
|
|
2145
|
+
if (this.instanceIpcClients.get(name) === existing) {
|
|
2146
|
+
this.instanceIpcClients.delete(name);
|
|
2147
|
+
}
|
|
2148
|
+
}
|
|
1947
2149
|
}
|
|
1948
2150
|
const sockPath = join(this.getInstanceDir(name), "channel.sock");
|
|
1949
2151
|
if (!existsSync(sockPath))
|
|
@@ -2002,6 +2204,9 @@ export class FleetManager {
|
|
|
2002
2204
|
else if (msg.type === "instance_process_state") {
|
|
2003
2205
|
this.cacheInstanceProcessStatus(name, msg.status);
|
|
2004
2206
|
}
|
|
2207
|
+
else if (msg.type === "instance_activity") {
|
|
2208
|
+
this.cacheInstanceActivity(name, msg.activity);
|
|
2209
|
+
}
|
|
2005
2210
|
else if (msg.type === "instance_state" || msg.type === "instance_state_response") {
|
|
2006
2211
|
this.cacheInstanceExecutionState(name, msg);
|
|
2007
2212
|
if (msg.type === "instance_state_response") {
|
|
@@ -2021,6 +2226,10 @@ export class FleetManager {
|
|
|
2021
2226
|
}
|
|
2022
2227
|
// Auto-reconnect on disconnect (unless intentionally stopping)
|
|
2023
2228
|
ipc.on("disconnect", () => {
|
|
2229
|
+
// A delayed event from a replaced/stale client must never delete the
|
|
2230
|
+
// current connection or start another reconnect loop.
|
|
2231
|
+
if (this.instanceIpcClients.get(name) !== ipc)
|
|
2232
|
+
return;
|
|
2024
2233
|
this.instanceIpcClients.delete(name);
|
|
2025
2234
|
if (this.ipcStoppingInstances.has(name))
|
|
2026
2235
|
return;
|
|
@@ -2032,7 +2241,20 @@ export class FleetManager {
|
|
|
2032
2241
|
}
|
|
2033
2242
|
}
|
|
2034
2243
|
/** Attempt IPC reconnection with exponential backoff */
|
|
2035
|
-
|
|
2244
|
+
ipcReconnect(name) {
|
|
2245
|
+
const inFlight = this.ipcReconnectInFlight.get(name);
|
|
2246
|
+
if (inFlight)
|
|
2247
|
+
return inFlight;
|
|
2248
|
+
const reconnect = this.runIpcReconnect(name)
|
|
2249
|
+
.finally(() => {
|
|
2250
|
+
if (this.ipcReconnectInFlight.get(name) === reconnect) {
|
|
2251
|
+
this.ipcReconnectInFlight.delete(name);
|
|
2252
|
+
}
|
|
2253
|
+
});
|
|
2254
|
+
this.ipcReconnectInFlight.set(name, reconnect);
|
|
2255
|
+
return reconnect;
|
|
2256
|
+
}
|
|
2257
|
+
async runIpcReconnect(name) {
|
|
2036
2258
|
for (let attempt = 1;; attempt++) {
|
|
2037
2259
|
if (this.ipcStoppingInstances.has(name) || !this.daemons.has(name))
|
|
2038
2260
|
return;
|
|
@@ -2055,9 +2277,22 @@ export class FleetManager {
|
|
|
2055
2277
|
if (existsSync(windowIdPath)) {
|
|
2056
2278
|
const windowId = readFileSync(windowIdPath, "utf-8").trim();
|
|
2057
2279
|
if (windowId) {
|
|
2280
|
+
// Async with an explicit timeout: this was execSync with NO timeout at
|
|
2281
|
+
// all, so a wedged tmux server blocked the whole fleet event loop
|
|
2282
|
+
// indefinitely — while we were here to diagnose a lost connection.
|
|
2283
|
+
// A timeout is also the correct signal: an unresponsive tmux server
|
|
2284
|
+
// means we cannot verify the pane, which is treated as dead (the same
|
|
2285
|
+
// conclusion the old code reached only by throwing).
|
|
2058
2286
|
try {
|
|
2059
|
-
const {
|
|
2060
|
-
|
|
2287
|
+
const { execFile } = await import("node:child_process");
|
|
2288
|
+
const { promisify } = await import("node:util");
|
|
2289
|
+
const { getTmuxSocketName } = await import("./paths.js");
|
|
2290
|
+
// Honour socket isolation: without -L this queried the user's default
|
|
2291
|
+
// tmux server instead of the fleet's, so under a custom AGEND_HOME the
|
|
2292
|
+
// check was meaningless (it reported every pane dead).
|
|
2293
|
+
const socket = getTmuxSocketName();
|
|
2294
|
+
const args = socket ? ["-L", socket, "list-panes", "-t", windowId] : ["list-panes", "-t", windowId];
|
|
2295
|
+
await promisify(execFile)("tmux", args, { timeout: 5_000 });
|
|
2061
2296
|
}
|
|
2062
2297
|
catch {
|
|
2063
2298
|
// Pane dead — respawn
|
|
@@ -2078,6 +2313,12 @@ export class FleetManager {
|
|
|
2078
2313
|
if (this.adapterRestarting.has(id))
|
|
2079
2314
|
return;
|
|
2080
2315
|
this.adapterRestarting.add(id);
|
|
2316
|
+
// Reflect reality in adapterState throughout. This loop used to leave the state
|
|
2317
|
+
// untouched, so getAdapterStates() — and therefore /health and the dashboard —
|
|
2318
|
+
// kept reporting "connected" for an adapter that had been down for hours. An
|
|
2319
|
+
// adapter's true status was simply not knowable from inside the process.
|
|
2320
|
+
const previous = this.adapterState.get(id);
|
|
2321
|
+
this.adapterState.set(id, { status: "retrying", retryCount: 0, lastError: previous?.lastError });
|
|
2081
2322
|
try {
|
|
2082
2323
|
for (let attempt = 1;; attempt++) {
|
|
2083
2324
|
if (this.ipcStoppingInstances.has("__fleet_stopping__"))
|
|
@@ -2090,9 +2331,16 @@ export class FleetManager {
|
|
|
2090
2331
|
await adapter.stop().catch(() => { });
|
|
2091
2332
|
await adapter.start();
|
|
2092
2333
|
this.logger.info({ id, attempt }, "Adapter restarted successfully");
|
|
2334
|
+
this.adapterState.set(id, { status: "connected", retryCount: 0 });
|
|
2093
2335
|
return;
|
|
2094
2336
|
}
|
|
2095
|
-
catch {
|
|
2337
|
+
catch (err) {
|
|
2338
|
+
this.adapterState.set(id, {
|
|
2339
|
+
status: "retrying",
|
|
2340
|
+
retryCount: attempt,
|
|
2341
|
+
lastError: err?.message ?? String(err),
|
|
2342
|
+
});
|
|
2343
|
+
}
|
|
2096
2344
|
if (attempt % 10 === 0) {
|
|
2097
2345
|
this.logger.warn({ id, attempt }, "Adapter restart still failing");
|
|
2098
2346
|
}
|
|
@@ -2128,6 +2376,45 @@ export class FleetManager {
|
|
|
2128
2376
|
}
|
|
2129
2377
|
return generals[0];
|
|
2130
2378
|
}
|
|
2379
|
+
/**
|
|
2380
|
+
* A user reacted to one of the bot's messages (#408).
|
|
2381
|
+
*
|
|
2382
|
+
* A reaction is context, not a message (#432, reworking #413): it never triggers
|
|
2383
|
+
* an agent turn and never wakes anything. It is queued in the event log and rides
|
|
2384
|
+
* into the instance's NEXT real message as one compact leading line —
|
|
2385
|
+
* `[Recent reactions: 👍×2 from hanhanv]` — after which it is marked consumed.
|
|
2386
|
+
* No pending reactions → no line → zero context spent, which is the common case.
|
|
2387
|
+
*/
|
|
2388
|
+
async handleInboundReaction(r) {
|
|
2389
|
+
const instanceName = this.resolveSlashTarget(r.threadId ?? r.chatId, r.adapterId);
|
|
2390
|
+
if (!instanceName) {
|
|
2391
|
+
this.logger.debug({ emoji: r.emoji, chatId: r.chatId }, "Reaction in an unrouted channel — ignoring");
|
|
2392
|
+
return;
|
|
2393
|
+
}
|
|
2394
|
+
this.eventLog?.logActivity("reaction", r.username, `${r.emoji} ${r.action}`, instanceName);
|
|
2395
|
+
if (r.action === "add") {
|
|
2396
|
+
this.eventLog?.addReaction(instanceName, r.messageId, r.username, r.emoji);
|
|
2397
|
+
}
|
|
2398
|
+
else {
|
|
2399
|
+
// Withdrawn before anyone saw it → it never happened. See removeReaction.
|
|
2400
|
+
this.eventLog?.removeReaction(instanceName, r.messageId, r.username, r.emoji);
|
|
2401
|
+
}
|
|
2402
|
+
}
|
|
2403
|
+
/**
|
|
2404
|
+
* The queued-reaction summary for an instance's next real message, or {} when
|
|
2405
|
+
* nothing is pending (the common case must add zero context). The consume
|
|
2406
|
+
* callback is separate from the fetch so reactions are only marked once the
|
|
2407
|
+
* message actually went out — a failed delivery keeps them queued.
|
|
2408
|
+
*/
|
|
2409
|
+
pendingReactionsMeta(instanceName) {
|
|
2410
|
+
const pending = this.eventLog?.pendingReactions(instanceName);
|
|
2411
|
+
if (!pending)
|
|
2412
|
+
return { meta: {}, consume: () => { } };
|
|
2413
|
+
return {
|
|
2414
|
+
meta: { pending_reactions: pending.summary },
|
|
2415
|
+
consume: () => this.eventLog?.markReactionsConsumed(instanceName, pending.maxId),
|
|
2416
|
+
};
|
|
2417
|
+
}
|
|
2131
2418
|
async handleInboundMessage(msg) {
|
|
2132
2419
|
const threadId = msg.threadId || undefined;
|
|
2133
2420
|
this.logger.debug({ source: msg.source, chatId: msg.chatId, threadId, userId: msg.userId, isBotMessage: msg.isBotMessage, textLen: (msg.text ?? "").length, text: (msg.text ?? "").slice(0, 80) }, "handleInboundMessage entry");
|
|
@@ -2465,6 +2752,7 @@ export class FleetManager {
|
|
|
2465
2752
|
}
|
|
2466
2753
|
this.warnIfRateLimited(generalInstance, msg);
|
|
2467
2754
|
const { text, extraMeta } = await processAttachments(msg, inboundAdapter, this.logger, generalInstance);
|
|
2755
|
+
const generalReactions = this.pendingReactionsMeta(generalInstance);
|
|
2468
2756
|
try {
|
|
2469
2757
|
await this.deliverToInstance(generalInstance, {
|
|
2470
2758
|
type: "fleet_inbound",
|
|
@@ -2480,9 +2768,11 @@ export class FleetManager {
|
|
|
2480
2768
|
adapter_id: msg.adapterId,
|
|
2481
2769
|
source: msg.source,
|
|
2482
2770
|
...(msg.replyToText ? { reply_to_text: msg.replyToText } : {}),
|
|
2771
|
+
...generalReactions.meta,
|
|
2483
2772
|
...extraMeta,
|
|
2484
2773
|
},
|
|
2485
2774
|
});
|
|
2775
|
+
generalReactions.consume();
|
|
2486
2776
|
this.lastInboundUser.set(generalInstance, msg.username);
|
|
2487
2777
|
this.logger.info(`${msg.username} → ${generalInstance}: ${(text ?? "").slice(0, 100)}`);
|
|
2488
2778
|
this.eventLog?.logActivity("message", msg.username, (text ?? "").slice(0, 200), generalInstance);
|
|
@@ -2558,6 +2848,7 @@ export class FleetManager {
|
|
|
2558
2848
|
this.setTopicIcon(instanceName, "blue");
|
|
2559
2849
|
this.warnIfRateLimited(instanceName, msg);
|
|
2560
2850
|
const { text, extraMeta } = await processAttachments(msg, inboundAdapter, this.logger, instanceName);
|
|
2851
|
+
const reactions = this.pendingReactionsMeta(instanceName);
|
|
2561
2852
|
try {
|
|
2562
2853
|
await this.deliverToInstance(instanceName, {
|
|
2563
2854
|
type: "fleet_inbound",
|
|
@@ -2573,9 +2864,13 @@ export class FleetManager {
|
|
|
2573
2864
|
adapter_id: msg.adapterId,
|
|
2574
2865
|
source: msg.source,
|
|
2575
2866
|
...(msg.replyToText ? { reply_to_text: msg.replyToText } : {}),
|
|
2867
|
+
...reactions.meta,
|
|
2576
2868
|
...extraMeta,
|
|
2577
2869
|
},
|
|
2578
2870
|
});
|
|
2871
|
+
// Only after the message actually went out. A failed delivery keeps the
|
|
2872
|
+
// reactions queued for the retry / the next message.
|
|
2873
|
+
reactions.consume();
|
|
2579
2874
|
}
|
|
2580
2875
|
catch (err) {
|
|
2581
2876
|
this.logger.warn({ err: err.message, instanceName }, "Wake/delivery failed");
|
|
@@ -2667,8 +2962,18 @@ export class FleetManager {
|
|
|
2667
2962
|
// Route standard channel tools (reply, react, edit_message, download_attachment)
|
|
2668
2963
|
if (routeToolCall(outAdapter, tool, args, threadId, respond)) {
|
|
2669
2964
|
if (tool === "reply") {
|
|
2670
|
-
//
|
|
2671
|
-
|
|
2965
|
+
// A reply is NOT proof the turn is over: on multi-step work an agent
|
|
2966
|
+
// replies ("starting…") and keeps going for many minutes. Retiring the
|
|
2967
|
+
// button here left the channel looking idle with no way to cancel and no
|
|
2968
|
+
// sign anything was happening (#410). Idle state owns retirement now; if the
|
|
2969
|
+
// instance is still working, move the button below the new reply so it stays
|
|
2970
|
+
// the last thing in the channel.
|
|
2971
|
+
if (this.getInstanceIdle(instanceName)) {
|
|
2972
|
+
this.clearCancelButton(instanceName);
|
|
2973
|
+
}
|
|
2974
|
+
else {
|
|
2975
|
+
void this.sendCancelButton(instanceName);
|
|
2976
|
+
}
|
|
2672
2977
|
this.reactDone(instanceName);
|
|
2673
2978
|
const replyTo = this.lastInboundUser.get(instanceName) ?? "user";
|
|
2674
2979
|
this.logger.info(`${instanceName} → ${replyTo}: ${(args.text ?? "").slice(0, 100)}`);
|
|
@@ -3299,6 +3604,9 @@ export class FleetManager {
|
|
|
3299
3604
|
}
|
|
3300
3605
|
}
|
|
3301
3606
|
async removeInstance(name) {
|
|
3607
|
+
// Drop cached pane context — the map is keyed by instance name and nothing
|
|
3608
|
+
// else evicted deleted entries, so it grew for the life of the process.
|
|
3609
|
+
forgetInstanceContext(name);
|
|
3302
3610
|
// Clean up schedules (scheduler is fleet-level, not lifecycle-level)
|
|
3303
3611
|
const config = this.fleetConfig?.instances[name];
|
|
3304
3612
|
if (this.scheduler && config?.topic_id) {
|
|
@@ -3404,6 +3712,171 @@ export class FleetManager {
|
|
|
3404
3712
|
this.collabInstances.add(instanceName);
|
|
3405
3713
|
return true;
|
|
3406
3714
|
}
|
|
3715
|
+
/**
|
|
3716
|
+
* Open the event log, tolerating a corrupt file.
|
|
3717
|
+
*
|
|
3718
|
+
* `events.db` holds history only — event rows and the activity feed. Nothing the
|
|
3719
|
+
* fleet needs to run depends on it, and every consumer already uses
|
|
3720
|
+
* `this.eventLog?.`. An unguarded `new EventLog(...)` here meant a corrupt or
|
|
3721
|
+
* unreadable history file (a truncated WAL after a hard kill, a full disk)
|
|
3722
|
+
* threw during startAll and the WHOLE FLEET FAILED TO BOOT — trading every
|
|
3723
|
+
* running agent for a file whose only job is reporting.
|
|
3724
|
+
*
|
|
3725
|
+
* So: try, move a bad file aside and retry once with a fresh one, and if even
|
|
3726
|
+
* that fails carry on without an event log.
|
|
3727
|
+
*/
|
|
3728
|
+
/**
|
|
3729
|
+
* Run `agend backend doctor` for the fleet's default backend and return its
|
|
3730
|
+
* cleaned output.
|
|
3731
|
+
*
|
|
3732
|
+
* Async on purpose: this was `execSync` with a 30s timeout, reachable by any
|
|
3733
|
+
* allowlisted user through `/doctor`. While it ran, the entire fleet event loop
|
|
3734
|
+
* was frozen — no IPC, no adapter, no message delivery, no health responses,
|
|
3735
|
+
* and critically no WATCHDOG ping, so a slow doctor could push past
|
|
3736
|
+
* WatchdogSec and have systemd SIGABRT the fleet.
|
|
3737
|
+
*/
|
|
3738
|
+
async runBackendDoctor() {
|
|
3739
|
+
const stripAnsi = (s) => s.replace(/\x1b\[[0-9;]*[a-zA-Z]/g, "");
|
|
3740
|
+
const backend = this.fleetConfig?.defaults?.backend || "claude-code";
|
|
3741
|
+
try {
|
|
3742
|
+
const { execFile } = await import("node:child_process");
|
|
3743
|
+
const { promisify } = await import("node:util");
|
|
3744
|
+
// execFile with an argv array — no shell, so the backend name cannot be
|
|
3745
|
+
// interpreted as a command even if config is malformed.
|
|
3746
|
+
const { stdout } = await promisify(execFile)("agend", ["backend", "doctor", backend], {
|
|
3747
|
+
timeout: 30_000,
|
|
3748
|
+
encoding: "utf-8",
|
|
3749
|
+
});
|
|
3750
|
+
return stripAnsi(stdout) || "No output";
|
|
3751
|
+
}
|
|
3752
|
+
catch (err) {
|
|
3753
|
+
const e = err;
|
|
3754
|
+
return stripAnsi(e.stdout ?? e.message ?? "Doctor failed");
|
|
3755
|
+
}
|
|
3756
|
+
}
|
|
3757
|
+
/** Drop event/activity rows older than the retention window. Best-effort. */
|
|
3758
|
+
/**
|
|
3759
|
+
* Cap every instance's pipe-pane log, walking the instances **directory** rather
|
|
3760
|
+
* than the config.
|
|
3761
|
+
*
|
|
3762
|
+
* A running instance rotates its own log on each health tick, so the ones that
|
|
3763
|
+
* need this are the ones nothing else looks at:
|
|
3764
|
+
*
|
|
3765
|
+
* - deleted instances, whose directory outlives the config entry. Nothing ever
|
|
3766
|
+
* touched these again. On the machine this was found on, one held 122 MB and
|
|
3767
|
+
* another 74 MB, out of 622 MB of pipe-pane logs in total.
|
|
3768
|
+
* - classic instances, which live in classicChannels, not fleetConfig.instances,
|
|
3769
|
+
* and so were never in the old config-driven loop at all.
|
|
3770
|
+
* - stopped instances, which have no health tick running.
|
|
3771
|
+
*
|
|
3772
|
+
* pipe-pane writes raw TUI output, so a wedged splash screen can emit ANSI frames
|
|
3773
|
+
* at animation rate. Unbounded growth here fills the disk, which takes the whole
|
|
3774
|
+
* fleet down rather than one instance.
|
|
3775
|
+
*/
|
|
3776
|
+
rotateAllInstanceLogs() {
|
|
3777
|
+
const root = join(this.dataDir, "instances");
|
|
3778
|
+
let entries;
|
|
3779
|
+
try {
|
|
3780
|
+
entries = readdirSync(root, { withFileTypes: true });
|
|
3781
|
+
}
|
|
3782
|
+
catch {
|
|
3783
|
+
return; // no instances directory yet
|
|
3784
|
+
}
|
|
3785
|
+
for (const entry of entries) {
|
|
3786
|
+
if (!entry.isDirectory())
|
|
3787
|
+
continue;
|
|
3788
|
+
// rotateLogIfNeeded is already best-effort and returns early on a missing
|
|
3789
|
+
// file, so a directory without a pipe-pane log costs one stat.
|
|
3790
|
+
rotateLogIfNeeded(join(root, entry.name, "output.log"));
|
|
3791
|
+
}
|
|
3792
|
+
}
|
|
3793
|
+
pruneEventLog() {
|
|
3794
|
+
try {
|
|
3795
|
+
this.eventLog?.prune(FleetManager.EVENT_LOG_RETENTION_DAYS);
|
|
3796
|
+
}
|
|
3797
|
+
catch (err) {
|
|
3798
|
+
this.logger.warn({ err }, "Event log prune failed");
|
|
3799
|
+
}
|
|
3800
|
+
}
|
|
3801
|
+
openEventLog() {
|
|
3802
|
+
const dbPath = join(this.dataDir, "events.db");
|
|
3803
|
+
try {
|
|
3804
|
+
return new EventLog(dbPath);
|
|
3805
|
+
}
|
|
3806
|
+
catch (err) {
|
|
3807
|
+
this.logger.error({ err, dbPath }, "events.db unusable — moving it aside and starting a fresh one");
|
|
3808
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, "-");
|
|
3809
|
+
for (const suffix of ["", "-wal", "-shm"]) {
|
|
3810
|
+
try {
|
|
3811
|
+
renameSync(`${dbPath}${suffix}`, `${dbPath}${suffix}.corrupt-${stamp}`);
|
|
3812
|
+
}
|
|
3813
|
+
catch { /* may not exist */ }
|
|
3814
|
+
}
|
|
3815
|
+
try {
|
|
3816
|
+
return new EventLog(dbPath);
|
|
3817
|
+
}
|
|
3818
|
+
catch (retryErr) {
|
|
3819
|
+
// History is worth losing; a fleet that won't start is not.
|
|
3820
|
+
this.logger.error({ err: retryErr, dbPath }, "Could not open a fresh events.db — continuing without event logging");
|
|
3821
|
+
return null;
|
|
3822
|
+
}
|
|
3823
|
+
}
|
|
3824
|
+
}
|
|
3825
|
+
/**
|
|
3826
|
+
* Report a fleet-level fault (not attributable to one instance) to the General
|
|
3827
|
+
* topic, so the operator learns about it without reading daemon.log.
|
|
3828
|
+
*
|
|
3829
|
+
* Throttled per distinct message: an unhandled rejection typically comes from a
|
|
3830
|
+
* loop (a poller, a repeating timer), and one channel message per occurrence
|
|
3831
|
+
* would bury the topic — which is worse than silence. First occurrence goes out
|
|
3832
|
+
* immediately, repeats are suppressed for THROTTLE_MS and then re-sent with a
|
|
3833
|
+
* count.
|
|
3834
|
+
*
|
|
3835
|
+
* The log line is written by the caller regardless: if every adapter is down,
|
|
3836
|
+
* the only notification path is the one that is broken.
|
|
3837
|
+
*/
|
|
3838
|
+
notifyFleetError(text) {
|
|
3839
|
+
const now = Date.now();
|
|
3840
|
+
const key = text.slice(0, 200);
|
|
3841
|
+
const seen = this.fleetErrorNotices.get(key);
|
|
3842
|
+
if (seen && now - seen.at < FleetManager.FLEET_ERROR_THROTTLE_MS) {
|
|
3843
|
+
seen.suppressed++;
|
|
3844
|
+
return;
|
|
3845
|
+
}
|
|
3846
|
+
const suppressed = seen?.suppressed ?? 0;
|
|
3847
|
+
this.fleetErrorNotices.set(key, { at: now, suppressed: 0 });
|
|
3848
|
+
// Bound the map: it is keyed by message text, and a message with a varying
|
|
3849
|
+
// suffix (a path, an id) would otherwise grow it without limit.
|
|
3850
|
+
if (this.fleetErrorNotices.size > 100) {
|
|
3851
|
+
const oldest = this.fleetErrorNotices.keys().next().value;
|
|
3852
|
+
if (oldest !== undefined)
|
|
3853
|
+
this.fleetErrorNotices.delete(oldest);
|
|
3854
|
+
}
|
|
3855
|
+
const body = suppressed > 0
|
|
3856
|
+
? `${text}\n(plus ${suppressed} more in the last ${Math.round(FleetManager.FLEET_ERROR_THROTTLE_MS / 60_000)}m)`
|
|
3857
|
+
: text;
|
|
3858
|
+
// Resolved from config, NOT findGeneralInstance(): that requires a live daemon,
|
|
3859
|
+
// and a fleet-level fault is exactly when the General may be down. The topic
|
|
3860
|
+
// itself still exists, and notifyInstanceTopic only needs adapter + group +
|
|
3861
|
+
// topic_id to post into it.
|
|
3862
|
+
const general = Object.entries(this.fleetConfig?.instances ?? {})
|
|
3863
|
+
.find(([, config]) => config.general_topic === true)?.[0];
|
|
3864
|
+
if (general) {
|
|
3865
|
+
this.notifyInstanceTopic(general, body);
|
|
3866
|
+
return;
|
|
3867
|
+
}
|
|
3868
|
+
// No General instance — fall back to the primary channel's group.
|
|
3869
|
+
const channelCfg = this.getChannelConfig();
|
|
3870
|
+
const groupId = channelCfg?.group_id;
|
|
3871
|
+
if (this.adapter && groupId) {
|
|
3872
|
+
this.adapter.sendText(String(groupId), body)
|
|
3873
|
+
.catch(err => this.logger.warn({ err }, "Failed to send fleet error notification"));
|
|
3874
|
+
return;
|
|
3875
|
+
}
|
|
3876
|
+
this.logger.warn({ text: body }, "Fleet error had no notification target (no General instance, no adapter)");
|
|
3877
|
+
}
|
|
3878
|
+
static FLEET_ERROR_THROTTLE_MS = 10 * 60_000;
|
|
3879
|
+
fleetErrorNotices = new Map();
|
|
3407
3880
|
notifyInstanceTopic(instanceName, text, extraOpts) {
|
|
3408
3881
|
const adapter = this.getAdapterForInstance(instanceName) ?? this.adapter;
|
|
3409
3882
|
if (!adapter)
|
|
@@ -3533,7 +4006,12 @@ export class FleetManager {
|
|
|
3533
4006
|
threadId: sent.threadId ?? threadId,
|
|
3534
4007
|
correlationId,
|
|
3535
4008
|
retryCount: 0,
|
|
4009
|
+
// Elapsed time is measured from when this button was posted — i.e. from
|
|
4010
|
+
// when the work was handed over — not from the pane's working transition,
|
|
4011
|
+
// which resets if the CLI blips idle mid-turn.
|
|
4012
|
+
startedAt: Date.now(),
|
|
3536
4013
|
};
|
|
4014
|
+
this.startProgressTicker(entry);
|
|
3537
4015
|
// Idle-check backstop: every 5min, if the instance is idle, retire the
|
|
3538
4016
|
// button. Covers turns that end without hitting a clear trigger (reply /
|
|
3539
4017
|
// cancel / correlation). Cleared in discardButton when the entry is removed.
|
|
@@ -3554,6 +4032,97 @@ export class FleetManager {
|
|
|
3554
4032
|
this.logger.warn({ err: e.message, instanceName }, "Failed to send cancel button");
|
|
3555
4033
|
}
|
|
3556
4034
|
}
|
|
4035
|
+
/**
|
|
4036
|
+
* The cancel button's text for a given elapsed time.
|
|
4037
|
+
*
|
|
4038
|
+
* Below the threshold it keeps the original wording, so a normal quick answer
|
|
4039
|
+
* looks exactly as it did before. Past it, the button doubles as the live
|
|
4040
|
+
* progress indicator (#409) — the channel showed nothing at all during long work,
|
|
4041
|
+
* and once the agent had replied once there was no sign it was still going.
|
|
4042
|
+
*/
|
|
4043
|
+
static progressText(elapsedMs, activity) {
|
|
4044
|
+
if (elapsedMs < PROGRESS_MIN_ELAPSED_MS)
|
|
4045
|
+
return "👀 處理中…";
|
|
4046
|
+
const totalSeconds = Math.floor(elapsedMs / 1000);
|
|
4047
|
+
const minutes = Math.floor(totalSeconds / 60);
|
|
4048
|
+
const seconds = totalSeconds % 60;
|
|
4049
|
+
const elapsed = minutes >= 60
|
|
4050
|
+
? `${Math.floor(minutes / 60)}h ${minutes % 60}m`
|
|
4051
|
+
: `${minutes}m ${String(seconds).padStart(2, "0")}s`;
|
|
4052
|
+
const detail = FleetManager.sanitizeActivity(activity);
|
|
4053
|
+
return detail
|
|
4054
|
+
? `⏳ 處理中… (已進行 ${elapsed} · ${detail})`
|
|
4055
|
+
: `⏳ 處理中… (已進行 ${elapsed})`;
|
|
4056
|
+
}
|
|
4057
|
+
/**
|
|
4058
|
+
* Make a tool summary safe to paste into a channel message.
|
|
4059
|
+
*
|
|
4060
|
+
* The text is agent-controlled (it is built from tool inputs — file paths,
|
|
4061
|
+
* shell commands), so it gets flattened to one line, capped, and stripped of
|
|
4062
|
+
* the two Discord mass-mention triggers. Neither channel renders it with a
|
|
4063
|
+
* parse mode, so no markup escaping is needed beyond that.
|
|
4064
|
+
*/
|
|
4065
|
+
static sanitizeActivity(activity) {
|
|
4066
|
+
if (!activity)
|
|
4067
|
+
return null;
|
|
4068
|
+
const flat = activity.replace(/\s+/g, " ").replace(/@(everyone|here)/g, "@$1").trim();
|
|
4069
|
+
if (!flat)
|
|
4070
|
+
return null;
|
|
4071
|
+
return flat.length > PROGRESS_ACTIVITY_MAX_CHARS
|
|
4072
|
+
? `${flat.slice(0, PROGRESS_ACTIVITY_MAX_CHARS - 1)}…`
|
|
4073
|
+
: flat;
|
|
4074
|
+
}
|
|
4075
|
+
/**
|
|
4076
|
+
* Remember what an instance is currently doing, for the progress line.
|
|
4077
|
+
*
|
|
4078
|
+
* Best-effort by design: only backends that expose a live activity feed report
|
|
4079
|
+
* anything, and the progress line simply omits the detail for the rest. It is
|
|
4080
|
+
* never used to decide anything — purely what the user is shown.
|
|
4081
|
+
*/
|
|
4082
|
+
cacheInstanceActivity(name, activity) {
|
|
4083
|
+
if (activity)
|
|
4084
|
+
this.instanceActivity.set(name, activity);
|
|
4085
|
+
else
|
|
4086
|
+
this.instanceActivity.delete(name);
|
|
4087
|
+
}
|
|
4088
|
+
/**
|
|
4089
|
+
* Refresh the button's text in place while the instance keeps working.
|
|
4090
|
+
*
|
|
4091
|
+
* Uses `editAlert`, NOT `editMessage`: on Telegram the latter omits reply_markup,
|
|
4092
|
+
* and the Bot API treats that as "clear the keyboard" — so editing with it would
|
|
4093
|
+
* delete the very cancel button this is trying to keep alive.
|
|
4094
|
+
*/
|
|
4095
|
+
startProgressTicker(entry) {
|
|
4096
|
+
entry.progressTimer = setInterval(() => {
|
|
4097
|
+
if (!this.cancelButtons.has(entry.messageId)) {
|
|
4098
|
+
clearInterval(entry.progressTimer);
|
|
4099
|
+
return;
|
|
4100
|
+
}
|
|
4101
|
+
// Idle means the turn ended; the idle-edge handler retires the button.
|
|
4102
|
+
if (this.getInstanceIdle(entry.instanceName))
|
|
4103
|
+
return;
|
|
4104
|
+
const text = FleetManager.progressText(Date.now() - (entry.startedAt ?? Date.now()), this.instanceActivity.get(entry.instanceName));
|
|
4105
|
+
if (text === entry.lastProgressText)
|
|
4106
|
+
return; // nothing changed — skip the API call
|
|
4107
|
+
const adapter = this.getAdapterForInstance(entry.instanceName) ?? this.adapter;
|
|
4108
|
+
if (!adapter?.editAlert)
|
|
4109
|
+
return;
|
|
4110
|
+
entry.lastProgressText = text;
|
|
4111
|
+
adapter.editAlert(entry.chatId, entry.messageId, {
|
|
4112
|
+
type: "cancel",
|
|
4113
|
+
instanceName: entry.instanceName,
|
|
4114
|
+
message: text,
|
|
4115
|
+
choices: [{ id: `cancel:${entry.instanceName}`, label: t("cancel.button") }],
|
|
4116
|
+
}, entry.threadId ? { threadId: entry.threadId } : undefined)
|
|
4117
|
+
.catch(err => {
|
|
4118
|
+
// A failed progress edit must never escalate: the button still works and
|
|
4119
|
+
// the next tick retries. Common causes are a deleted message or a
|
|
4120
|
+
// rate limit.
|
|
4121
|
+
this.logger.debug({ err, instanceName: entry.instanceName }, "Progress edit failed");
|
|
4122
|
+
});
|
|
4123
|
+
}, PROGRESS_UPDATE_INTERVAL_MS);
|
|
4124
|
+
entry.progressTimer.unref?.();
|
|
4125
|
+
}
|
|
3557
4126
|
/** Retire (delete) every cancel button belonging to an instance. */
|
|
3558
4127
|
retireInstanceButtons(instanceName) {
|
|
3559
4128
|
// Snapshot first — retireButton may delete entries from the map on success.
|
|
@@ -3585,6 +4154,8 @@ export class FleetManager {
|
|
|
3585
4154
|
clearTimeout(entry.retryTimer);
|
|
3586
4155
|
if (entry.idleCheckTimer)
|
|
3587
4156
|
clearInterval(entry.idleCheckTimer);
|
|
4157
|
+
if (entry.progressTimer)
|
|
4158
|
+
clearInterval(entry.progressTimer);
|
|
3588
4159
|
this.cancelButtons.delete(entry.messageId);
|
|
3589
4160
|
}
|
|
3590
4161
|
/** Re-attempt a failed button delete up to CANCEL_BTN_MAX_RETRIES times. */
|
|
@@ -4300,22 +4871,49 @@ When users create specialized instances, suggest these configurations:
|
|
|
4300
4871
|
catch { /* missing / stale / corrupt */ }
|
|
4301
4872
|
return null;
|
|
4302
4873
|
}
|
|
4303
|
-
/**
|
|
4304
|
-
|
|
4874
|
+
/**
|
|
4875
|
+
* Resolve the effective model for a fleet or ClassicBot instance, plus where it
|
|
4876
|
+
* came from. Single source of truth for `/model` and `/ctx` — precedence:
|
|
4877
|
+
* per-instance → fleet defaults → classic channel → CLI's own default (from the
|
|
4878
|
+
* cli-env probe cache) → unresolved.
|
|
4879
|
+
*/
|
|
4880
|
+
resolveInstanceModel(instanceName) {
|
|
4881
|
+
const done = (model, source, reason) => ({
|
|
4882
|
+
model,
|
|
4883
|
+
source,
|
|
4884
|
+
reason,
|
|
4885
|
+
// Make an inherited CLI default legible instead of the bare word "default".
|
|
4886
|
+
display: source === "cli-default" ? `${model} (default)`
|
|
4887
|
+
: source === "unresolved" ? `default (${reason ?? "unresolved"})`
|
|
4888
|
+
: model,
|
|
4889
|
+
});
|
|
4305
4890
|
const fleetInstance = this.fleetConfig?.instances[instanceName];
|
|
4306
4891
|
if (fleetInstance) {
|
|
4307
|
-
|
|
4308
|
-
|
|
4309
|
-
|
|
4892
|
+
if (fleetInstance.model?.trim())
|
|
4893
|
+
return done(fleetInstance.model.trim(), "instance");
|
|
4894
|
+
const fleetDefault = this.fleetConfig?.defaults?.model;
|
|
4895
|
+
if (fleetDefault?.trim())
|
|
4896
|
+
return done(fleetDefault.trim(), "fleet-default");
|
|
4310
4897
|
}
|
|
4311
4898
|
const classic = this.classicChannels?.getAll().find(ch => ch.instanceName === instanceName);
|
|
4312
4899
|
if (classic) {
|
|
4313
4900
|
const classicModel = this.classicChannels?.getModel(classic.channelId, classic.adapterId, this.fleetConfig?.defaults?.model);
|
|
4314
4901
|
if (classicModel?.trim())
|
|
4315
|
-
return classicModel.trim();
|
|
4316
|
-
}
|
|
4317
|
-
|
|
4318
|
-
|
|
4902
|
+
return done(classicModel.trim(), "classic");
|
|
4903
|
+
}
|
|
4904
|
+
// Nothing configured → show what the CLI itself defaults to (kiro default_model,
|
|
4905
|
+
// grok "Default model:", codex config.toml, agy settings.json), cached by the probe.
|
|
4906
|
+
const cliEnv = this.readCliEnv(this.backendNameForInstance(instanceName));
|
|
4907
|
+
const cachedModel = cliEnv?.currentModel;
|
|
4908
|
+
if (cachedModel?.trim())
|
|
4909
|
+
return done(cachedModel.trim(), "cli-default");
|
|
4910
|
+
// Say WHY it's unresolved: no fresh probe yet vs. the CLI not exposing a default
|
|
4911
|
+
// (e.g. claude-code's default is account-side, opencode's is provider-side).
|
|
4912
|
+
return done("default", "unresolved", cliEnv ? "this CLI does not report a default" : "not probed yet");
|
|
4913
|
+
}
|
|
4914
|
+
/** Human-readable effective model, e.g. `auto (default)`. Used by /ctx. */
|
|
4915
|
+
modelDisplayForInstance(instanceName) {
|
|
4916
|
+
return this.resolveInstanceModel(instanceName).display;
|
|
4319
4917
|
}
|
|
4320
4918
|
modelChoiceLabel(option, currentModel) {
|
|
4321
4919
|
const label = option.description ? `${option.label} — ${option.description}` : option.label;
|
|
@@ -4395,7 +4993,8 @@ When users create specialized instances, suggest these configurations:
|
|
|
4395
4993
|
await data.respond(`No model list available for ${name}. Type \`/model <name>\` to set one directly.`);
|
|
4396
4994
|
return;
|
|
4397
4995
|
}
|
|
4398
|
-
|
|
4996
|
+
// Raw id for ✓-matching options; display resolves an inherited CLI default.
|
|
4997
|
+
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(name);
|
|
4399
4998
|
const nonce = randomBytes(6).toString("hex");
|
|
4400
4999
|
const choices = options.slice(0, 25).map(o => ({
|
|
4401
5000
|
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
@@ -4405,7 +5004,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
4405
5004
|
timer.unref?.();
|
|
4406
5005
|
this.pendingModelSelects.set(nonce, { instanceName: name, model: "", userId: data.userId, channelId: data.channelId, timer, respond: data.respond });
|
|
4407
5006
|
try {
|
|
4408
|
-
await data.respondChoices(`Current model: **${
|
|
5007
|
+
await data.respondChoices(`Current model: **${currentDisplay}**\nSelect a new model:`, choices);
|
|
4409
5008
|
}
|
|
4410
5009
|
catch (err) {
|
|
4411
5010
|
this.pendingModelSelects.delete(nonce);
|
|
@@ -4424,7 +5023,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
4424
5023
|
if (options.length === 0) {
|
|
4425
5024
|
return `No model list available for ${instanceName}. Use \`/model <name>\` to set one directly.`;
|
|
4426
5025
|
}
|
|
4427
|
-
const currentModel = this.
|
|
5026
|
+
const { model: currentModel, display: currentDisplay } = this.resolveInstanceModel(instanceName);
|
|
4428
5027
|
const nonce = randomBytes(6).toString("hex");
|
|
4429
5028
|
const choices = options.slice(0, 25).map(o => ({
|
|
4430
5029
|
id: `${MODEL_SELECT_CALLBACK_PREFIX}${nonce}:${o.id}`,
|
|
@@ -4444,7 +5043,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
4444
5043
|
timer.unref?.();
|
|
4445
5044
|
this.pendingModelSelects.set(nonce, { instanceName, model: "", userId, channelId, timer, respond, adapter, adapterChatId: chatId, adapterThreadId: threadId });
|
|
4446
5045
|
try {
|
|
4447
|
-
const menuMessageId = await adapter.promptUser(chatId, `Current model: ${
|
|
5046
|
+
const menuMessageId = await adapter.promptUser(chatId, `Current model: ${currentDisplay}\nSelect a new model:`, choices, { threadId });
|
|
4448
5047
|
const pending = this.pendingModelSelects.get(nonce);
|
|
4449
5048
|
if (pending)
|
|
4450
5049
|
pending.menuMessageId = menuMessageId;
|
|
@@ -4503,9 +5102,19 @@ When users create specialized instances, suggest these configurations:
|
|
|
4503
5102
|
// DC path: respond immediately with progress text
|
|
4504
5103
|
await pending.respond(progressText).catch(() => { });
|
|
4505
5104
|
}
|
|
4506
|
-
// Apply model in background — don't await here (keeps callback handler fast)
|
|
5105
|
+
// Apply model in background — don't await here (keeps callback handler fast).
|
|
5106
|
+
// Guarded: applyModel() restarts the instance, and an unguarded rejection here
|
|
5107
|
+
// meant a user picking from the /model menu could take the whole fleet down.
|
|
5108
|
+
// On failure the user gets told, rather than the click silently doing nothing.
|
|
4507
5109
|
void (async () => {
|
|
4508
|
-
|
|
5110
|
+
let result;
|
|
5111
|
+
try {
|
|
5112
|
+
result = await this.applyModel(pending.instanceName, model);
|
|
5113
|
+
}
|
|
5114
|
+
catch (err) {
|
|
5115
|
+
this.logger.error({ err, instance: pending.instanceName, model }, "Model switch failed");
|
|
5116
|
+
result = `Model switch to \`${model}\` failed: ${err instanceof Error ? err.message : String(err)}`;
|
|
5117
|
+
}
|
|
4509
5118
|
if (pending.adapter && pending.adapterChatId) {
|
|
4510
5119
|
if (progressMsgId) {
|
|
4511
5120
|
pending.adapter.editMessage(pending.adapterChatId, progressMsgId, result, pending.adapterThreadId).catch(() => {
|
|
@@ -4743,7 +5352,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
4743
5352
|
...(preTaskCommand ? { pre_task_command: preTaskCommand } : {}),
|
|
4744
5353
|
};
|
|
4745
5354
|
const topicMode = this.fleetConfig?.channel?.mode === "topic";
|
|
4746
|
-
await this.startInstance(instanceName, config, topicMode);
|
|
5355
|
+
await this.startInstance(instanceName, config, topicMode, "classic");
|
|
4747
5356
|
}
|
|
4748
5357
|
/** Handle /start slash command — register classic channel */
|
|
4749
5358
|
async handleClassicStart(channelId, channelName, userId, guildId, adapterId, backend) {
|
|
@@ -4785,11 +5394,24 @@ When users create specialized instances, suggest these configurations:
|
|
|
4785
5394
|
this.logger.info({ channelId, adapterId, instanceName: ch.instanceName }, "Classic channel stopped");
|
|
4786
5395
|
return t("classic.stopped");
|
|
4787
5396
|
}
|
|
4788
|
-
|
|
5397
|
+
/**
|
|
5398
|
+
* Idempotent while in flight: SIGINT and SIGTERM share one handler and the
|
|
5399
|
+
* uncaughtException path calls this too, so overlapping runs were possible —
|
|
5400
|
+
* each snapshotting the daemon map and calling stop() on the same daemons
|
|
5401
|
+
* concurrently. Deliberately NOT `async`, so callers receive the same promise
|
|
5402
|
+
* object rather than a fresh wrapper around it. The latch clears when the run
|
|
5403
|
+
* settles, so a later genuine stop (after a restart) still does the work.
|
|
5404
|
+
*/
|
|
5405
|
+
stopAll() {
|
|
5406
|
+
this.stopAllInFlight ??= this.doStopAll().finally(() => { this.stopAllInFlight = null; });
|
|
5407
|
+
return this.stopAllInFlight;
|
|
5408
|
+
}
|
|
5409
|
+
stopAllInFlight = null;
|
|
5410
|
+
async doStopAll() {
|
|
4789
5411
|
this.startupComplete = false;
|
|
4790
5412
|
this.reloadPending = false;
|
|
4791
5413
|
this.ipcStoppingInstances.add("__fleet_stopping__");
|
|
4792
|
-
|
|
5414
|
+
sdNotifyBlocking("STOPPING=1");
|
|
4793
5415
|
if (this.watchdogTimer) {
|
|
4794
5416
|
clearInterval(this.watchdogTimer);
|
|
4795
5417
|
this.watchdogTimer = null;
|
|
@@ -4809,6 +5431,26 @@ When users create specialized instances, suggest these configurations:
|
|
|
4809
5431
|
clearInterval(this.updateCheckTimer);
|
|
4810
5432
|
this.updateCheckTimer = null;
|
|
4811
5433
|
}
|
|
5434
|
+
if (this.eventLogPruneTimer) {
|
|
5435
|
+
clearInterval(this.eventLogPruneTimer);
|
|
5436
|
+
this.eventLogPruneTimer = null;
|
|
5437
|
+
}
|
|
5438
|
+
if (this.logRotateTimer) {
|
|
5439
|
+
clearInterval(this.logRotateTimer);
|
|
5440
|
+
this.logRotateTimer = null;
|
|
5441
|
+
}
|
|
5442
|
+
// Cancel-button timers were never cleared here. The idle-check interval is not
|
|
5443
|
+
// unref'd, so it held the event loop open past shutdown and kept retrying
|
|
5444
|
+
// deletes against an adapter that was already gone.
|
|
5445
|
+
for (const entry of [...this.cancelButtons.values()]) {
|
|
5446
|
+
if (entry.retryTimer)
|
|
5447
|
+
clearTimeout(entry.retryTimer);
|
|
5448
|
+
if (entry.idleCheckTimer)
|
|
5449
|
+
clearInterval(entry.idleCheckTimer);
|
|
5450
|
+
if (entry.progressTimer)
|
|
5451
|
+
clearInterval(entry.progressTimer);
|
|
5452
|
+
}
|
|
5453
|
+
this.cancelButtons.clear();
|
|
4812
5454
|
if (this.topicCleanupTimer) {
|
|
4813
5455
|
clearInterval(this.topicCleanupTimer);
|
|
4814
5456
|
this.topicCleanupTimer = null;
|
|
@@ -5027,6 +5669,15 @@ When users create specialized instances, suggest these configurations:
|
|
|
5027
5669
|
removedRatio,
|
|
5028
5670
|
validationErrors: validation.errors,
|
|
5029
5671
|
}, "Refusing unsafe fleet config reload; running configuration was kept");
|
|
5672
|
+
// Tell the operator. A silently ignored config edit is the most confusing
|
|
5673
|
+
// possible outcome: they change fleet.yaml, send SIGHUP, and nothing happens
|
|
5674
|
+
// with no explanation anywhere they are looking.
|
|
5675
|
+
const why = !validation.valid
|
|
5676
|
+
? `validation failed:\n${validation.errors.map(e => `• ${e.path}: ${e.message}`).join("\n")}`
|
|
5677
|
+
: unsafeEmpty
|
|
5678
|
+
? `it removed every instance (${oldCount} → 0)`
|
|
5679
|
+
: `it removed more than half the instances (${oldCount} → ${newCount})`;
|
|
5680
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload REJECTED — ${why}\nThe previous configuration is still running.`);
|
|
5030
5681
|
return;
|
|
5031
5682
|
}
|
|
5032
5683
|
this.routing.rebuild(this.fleetConfig);
|
|
@@ -5226,10 +5877,20 @@ When users create specialized instances, suggest these configurations:
|
|
|
5226
5877
|
// ── Update check ────────────────────────────────────────────────────
|
|
5227
5878
|
async checkForUpdates() {
|
|
5228
5879
|
try {
|
|
5229
|
-
|
|
5880
|
+
// Both npm lookups are async: as execSync they froze the fleet event loop for
|
|
5881
|
+
// up to 15s each, and on a beta build BOTH ran — 30s with no WATCHDOG ping,
|
|
5882
|
+
// past WatchdogSec's half-interval and enough for systemd to SIGABRT the fleet
|
|
5883
|
+
// for a background version check.
|
|
5884
|
+
const { execFile } = await import("node:child_process");
|
|
5885
|
+
const { promisify } = await import("node:util");
|
|
5886
|
+
const execFileP = promisify(execFile);
|
|
5887
|
+
const npmVersion = async (spec) => {
|
|
5888
|
+
const { stdout } = await execFileP("npm", ["view", spec, "version"], { timeout: 15_000 });
|
|
5889
|
+
return stdout.toString().trim();
|
|
5890
|
+
};
|
|
5230
5891
|
const pkgPath = join(dirname(fileURLToPath(import.meta.url)), "..", "package.json");
|
|
5231
5892
|
const currentVersion = JSON.parse(readFileSync(pkgPath, "utf-8")).version ?? "0.0.0";
|
|
5232
|
-
const latest =
|
|
5893
|
+
const latest = await npmVersion("@songsid/agend");
|
|
5233
5894
|
let target = latest;
|
|
5234
5895
|
if (currentVersion.includes("-beta")) {
|
|
5235
5896
|
// Beta users track the @beta channel (never fall back to @latest, which is
|
|
@@ -5237,7 +5898,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
5237
5898
|
// of beta/latest is the newest.
|
|
5238
5899
|
let beta = "";
|
|
5239
5900
|
try {
|
|
5240
|
-
beta =
|
|
5901
|
+
beta = await npmVersion("@songsid/agend@beta");
|
|
5241
5902
|
}
|
|
5242
5903
|
catch { /* no beta tag */ }
|
|
5243
5904
|
target = beta || latest;
|
|
@@ -5344,6 +6005,10 @@ When users create specialized instances, suggest these configurations:
|
|
|
5344
6005
|
// /view routes accept the read-only view.token (or web.token) and do
|
|
5345
6006
|
// their own per-method auth in view-api.ts — skip the web-token gate.
|
|
5346
6007
|
}
|
|
6008
|
+
else if (isUsagePath(new URL(req.url ?? "/", `http://localhost:${port}`).pathname)) {
|
|
6009
|
+
// /api/ai-usage is read-only GET data for the /view Usage panel — open
|
|
6010
|
+
// like the other /view data routes (usage-api.ts rejects non-GET).
|
|
6011
|
+
}
|
|
5347
6012
|
else {
|
|
5348
6013
|
// All other endpoints require a valid token (query ?token= or X-Agend-Token header).
|
|
5349
6014
|
// /ui/* will also re-check in web-api.ts, which is harmless.
|
|
@@ -5358,34 +6023,34 @@ When users create specialized instances, suggest these configurations:
|
|
|
5358
6023
|
}
|
|
5359
6024
|
}
|
|
5360
6025
|
if (req.method === "GET" && req.url === "/health") {
|
|
5361
|
-
const
|
|
5362
|
-
|
|
5363
|
-
|
|
5364
|
-
|
|
5365
|
-
|
|
5366
|
-
|
|
5367
|
-
|
|
5368
|
-
uptime: Math.floor((Date.now() - this.startedAt) / 1000),
|
|
5369
|
-
}));
|
|
6026
|
+
const health = this.getFleetHealth();
|
|
6027
|
+
// 503 when the fleet cannot do its job, so an external monitor sees it.
|
|
6028
|
+
// This used to always answer 200 "ok" with a count of CONFIGURED instances,
|
|
6029
|
+
// so every agent could be dead and every adapter down and it still looked
|
|
6030
|
+
// green.
|
|
6031
|
+
res.writeHead(health.status === "ok" ? 200 : 503);
|
|
6032
|
+
res.end(JSON.stringify(health));
|
|
5370
6033
|
return;
|
|
5371
6034
|
}
|
|
5372
6035
|
if (req.method === "GET" && req.url === "/status") {
|
|
5373
6036
|
const instances = Object.keys(this.fleetConfig?.instances ?? {}).map(name => {
|
|
5374
6037
|
const statusFile = join(this.getInstanceDir(name), "statusline.json");
|
|
5375
|
-
let context_pct = 0;
|
|
5376
6038
|
let cost = 0;
|
|
5377
6039
|
try {
|
|
5378
6040
|
const data = JSON.parse(readFileSync(statusFile, "utf-8"));
|
|
5379
|
-
context_pct = data.context_window?.used_percentage ?? 0;
|
|
5380
6041
|
cost = data.cost?.total_cost_usd ?? 0;
|
|
5381
6042
|
}
|
|
5382
6043
|
catch (err) {
|
|
5383
6044
|
this.logger.debug({ err, name }, "statusline.json read failed (/status)");
|
|
5384
6045
|
}
|
|
6046
|
+
const backend = this.fleetConfig?.instances[name]?.backend
|
|
6047
|
+
?? this.fleetConfig?.defaults?.backend
|
|
6048
|
+
?? "claude-code";
|
|
6049
|
+
const { context } = resolveInstanceContext(this.dataDir, name, backend);
|
|
5385
6050
|
return {
|
|
5386
6051
|
name,
|
|
5387
6052
|
status: this.getInstanceStatus(name),
|
|
5388
|
-
context_pct,
|
|
6053
|
+
context_pct: context ?? 0,
|
|
5389
6054
|
cost,
|
|
5390
6055
|
};
|
|
5391
6056
|
});
|
|
@@ -5500,7 +6165,10 @@ When users create specialized instances, suggest these configurations:
|
|
|
5500
6165
|
res.writeHead(500);
|
|
5501
6166
|
res.end(JSON.stringify({ error: `Start failed: ${err.message}` }));
|
|
5502
6167
|
}
|
|
5503
|
-
|
|
6168
|
+
// The inner catch can itself throw (writeHead after a successful
|
|
6169
|
+
// writeHead is ERR_HTTP_HEADERS_SENT), and that rejection escapes the
|
|
6170
|
+
// IIFE. Same for the two handlers below.
|
|
6171
|
+
})().catch(err => this.logger.error({ err, name }, "HTTP start handler failed"));
|
|
5504
6172
|
return;
|
|
5505
6173
|
}
|
|
5506
6174
|
// Instance restart (immediate, no idle wait)
|
|
@@ -5521,7 +6189,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
5521
6189
|
res.writeHead(status);
|
|
5522
6190
|
res.end(JSON.stringify({ error: `Restart failed: ${err.message}` }));
|
|
5523
6191
|
}
|
|
5524
|
-
})();
|
|
6192
|
+
})().catch(err => this.logger.error({ err, name }, "HTTP restart handler failed"));
|
|
5525
6193
|
return;
|
|
5526
6194
|
}
|
|
5527
6195
|
if (req.method === "POST" && req.url?.startsWith("/stop/")) {
|
|
@@ -5544,7 +6212,7 @@ When users create specialized instances, suggest these configurations:
|
|
|
5544
6212
|
res.writeHead(500);
|
|
5545
6213
|
res.end(JSON.stringify({ error: `Stop failed: ${err.message}` }));
|
|
5546
6214
|
}
|
|
5547
|
-
})();
|
|
6215
|
+
})().catch(err => this.logger.error({ err, name }, "HTTP stop handler failed"));
|
|
5548
6216
|
return;
|
|
5549
6217
|
}
|
|
5550
6218
|
// ── Agent CLI endpoint ─────
|
|
@@ -5556,6 +6224,8 @@ When users create specialized instances, suggest these configurations:
|
|
|
5556
6224
|
const url = new URL(req.url ?? "/", `http://localhost:${port}`);
|
|
5557
6225
|
if (handleViewRequest(req, res, url, this))
|
|
5558
6226
|
return;
|
|
6227
|
+
if (handleUsageRequest(req, res, url, this))
|
|
6228
|
+
return;
|
|
5559
6229
|
if (handleSettingsRequest(req, res, url, this))
|
|
5560
6230
|
return;
|
|
5561
6231
|
if (handleWebRequest(req, res, url, this))
|
|
@@ -5602,20 +6272,35 @@ When users create specialized instances, suggest these configurations:
|
|
|
5602
6272
|
this.logger.info({ url: `http://localhost:${port}/view?token=${this.viewToken}` }, "Web View available");
|
|
5603
6273
|
}
|
|
5604
6274
|
getUiStatus() {
|
|
5605
|
-
const
|
|
6275
|
+
const fleetNames = Object.keys(this.fleetConfig?.instances ?? {});
|
|
6276
|
+
// Classic rooms live only in classicBot.yaml — /api/profiles merges them into
|
|
6277
|
+
// the View roster, but previously getUiStatus skipped them so context_pct was
|
|
6278
|
+
// always 0 (live map miss → l?.context_pct ?? 0).
|
|
6279
|
+
const classicOnly = (this.classicChannels?.getAll() ?? [])
|
|
6280
|
+
.map(ch => ch.instanceName)
|
|
6281
|
+
.filter(name => !fleetNames.includes(name));
|
|
6282
|
+
const names = [...fleetNames, ...classicOnly];
|
|
6283
|
+
const instances = names.map(name => {
|
|
5606
6284
|
const statusFile = join(this.getInstanceDir(name), "statusline.json");
|
|
5607
|
-
let context_pct = 0;
|
|
5608
6285
|
let cost = 0;
|
|
5609
6286
|
let model = "";
|
|
5610
6287
|
try {
|
|
5611
6288
|
const data = JSON.parse(readFileSync(statusFile, "utf-8"));
|
|
5612
|
-
context_pct = data.context_window?.used_percentage ?? 0;
|
|
5613
6289
|
cost = data.cost?.total_cost_usd ?? 0;
|
|
5614
6290
|
model = data.model?.display_name ?? "";
|
|
5615
6291
|
}
|
|
5616
6292
|
catch (err) {
|
|
5617
6293
|
this.logger.debug({ err, name }, "statusline.json read failed (getUiStatus)");
|
|
5618
6294
|
}
|
|
6295
|
+
// Align with /ctx: statusline for claude-code, pane scrape for kiro/grok/codex.
|
|
6296
|
+
const classic = classicOnly.includes(name);
|
|
6297
|
+
const backend = classic
|
|
6298
|
+
? this.classicChannels.getBackendByInstance(name, this.fleetConfig?.defaults?.backend)
|
|
6299
|
+
: (this.fleetConfig?.instances[name]?.backend
|
|
6300
|
+
?? this.fleetConfig?.defaults?.backend
|
|
6301
|
+
?? "claude-code");
|
|
6302
|
+
const { context } = resolveInstanceContext(this.dataDir, name, backend);
|
|
6303
|
+
const context_pct = context ?? 0;
|
|
5619
6304
|
return { name, status: this.getInstanceStatus(name), context_pct, cost, model };
|
|
5620
6305
|
});
|
|
5621
6306
|
return {
|