@songsid/agend 2.1.1-beta.16 → 2.1.1-beta.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/claude-code.d.ts +37 -0
- package/dist/backend/claude-code.js +40 -1
- package/dist/backend/claude-code.js.map +1 -1
- package/dist/backend/kiro.d.ts +20 -0
- package/dist/backend/kiro.js +30 -0
- package/dist/backend/kiro.js.map +1 -1
- package/dist/backend/types.d.ts +27 -0
- package/dist/backend/types.js.map +1 -1
- package/dist/channel/adapters/discord.d.ts +15 -0
- package/dist/channel/adapters/discord.js +81 -1
- package/dist/channel/adapters/discord.js.map +1 -1
- package/dist/channel/adapters/telegram.d.ts +21 -0
- package/dist/channel/adapters/telegram.js +82 -0
- package/dist/channel/adapters/telegram.js.map +1 -1
- package/dist/channel/ipc-bridge.d.ts +9 -1
- package/dist/channel/ipc-bridge.js +12 -3
- package/dist/channel/ipc-bridge.js.map +1 -1
- package/dist/channel/ipc-timeouts.d.ts +40 -0
- package/dist/channel/ipc-timeouts.js +58 -0
- package/dist/channel/ipc-timeouts.js.map +1 -0
- package/dist/channel/mcp-server.js +5 -4
- package/dist/channel/mcp-server.js.map +1 -1
- package/dist/channel/types.d.ts +29 -0
- package/dist/daemon.d.ts +114 -5
- package/dist/daemon.js +336 -59
- package/dist/daemon.js.map +1 -1
- package/dist/fleet-manager.d.ts +92 -1
- package/dist/fleet-manager.js +346 -20
- package/dist/fleet-manager.js.map +1 -1
- package/dist/hang-detector.d.ts +19 -13
- package/dist/hang-detector.js +19 -49
- package/dist/hang-detector.js.map +1 -1
- package/dist/instance-lifecycle.js +9 -0
- package/dist/instance-lifecycle.js.map +1 -1
- package/dist/outbound-handlers.d.ts +7 -0
- package/dist/outbound-handlers.js +62 -4
- package/dist/outbound-handlers.js.map +1 -1
- package/dist/pane-write-lock.d.ts +48 -0
- package/dist/pane-write-lock.js +73 -0
- package/dist/pane-write-lock.js.map +1 -0
- package/dist/tmux-control.d.ts +48 -1
- package/dist/tmux-control.js +79 -6
- package/dist/tmux-control.js.map +1 -1
- package/package.json +1 -1
- package/dist/channel/tool-tracker.d.ts +0 -13
- package/dist/channel/tool-tracker.js +0 -58
- package/dist/channel/tool-tracker.js.map +0 -1
- package/dist/daemon-entry.d.ts +0 -1
- package/dist/daemon-entry.js +0 -30
- package/dist/daemon-entry.js.map +0 -1
package/dist/fleet-manager.d.ts
CHANGED
|
@@ -48,7 +48,7 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
48
48
|
private static sighupHandlerInstalled;
|
|
49
49
|
private children;
|
|
50
50
|
readonly lifecycle: InstanceLifecycle;
|
|
51
|
-
/**
|
|
51
|
+
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
52
52
|
get daemons(): Map<string, import("./daemon.js").Daemon>;
|
|
53
53
|
fleetConfig: FleetConfig | null;
|
|
54
54
|
private rawFleetConfig;
|
|
@@ -98,6 +98,10 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
98
98
|
private instanceIdleWaiters;
|
|
99
99
|
private lastInboundUser;
|
|
100
100
|
private cancelButtons;
|
|
101
|
+
/** instanceName → what it is doing right now, when the backend can tell us. */
|
|
102
|
+
private instanceActivity;
|
|
103
|
+
/** instanceName → tail of deliveries waiting for its IPC to come back. */
|
|
104
|
+
private ipcWaitTails;
|
|
101
105
|
private lastInboundMsg;
|
|
102
106
|
private topicArchiver;
|
|
103
107
|
controlClient: TmuxControlClient | null;
|
|
@@ -118,6 +122,7 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
118
122
|
private healthPortRetried;
|
|
119
123
|
private updateCheckTimer;
|
|
120
124
|
private eventLogPruneTimer;
|
|
125
|
+
private logRotateTimer;
|
|
121
126
|
/** Days of event/activity history to keep. */
|
|
122
127
|
private static readonly EVENT_LOG_RETENTION_DAYS;
|
|
123
128
|
private watchdogTimer;
|
|
@@ -200,6 +205,25 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
200
205
|
private enforceWarmCap;
|
|
201
206
|
private waitForInstanceIdle;
|
|
202
207
|
private deliverWithIdleGate;
|
|
208
|
+
/**
|
|
209
|
+
* Hand a payload to an instance's IPC, waiting out a *transient* disconnect.
|
|
210
|
+
*
|
|
211
|
+
* A daemon that is restarting — `/restart`, crash recovery, a model switch —
|
|
212
|
+
* drops its socket for a few seconds. Any message arriving in that window used
|
|
213
|
+
* to fail instantly: the caller logged a warning, put ❌ on the user's message,
|
|
214
|
+
* and the message was gone. The user had to notice the ❌ and retype it. That is
|
|
215
|
+
* the "instance 訊息不容易掉" goal failing on the most predictable event there is.
|
|
216
|
+
*
|
|
217
|
+
* The wait is bounded. If the instance is genuinely down, this still throws and
|
|
218
|
+
* the ❌ still appears — just for a real failure rather than a restart.
|
|
219
|
+
*
|
|
220
|
+
* Ordering is preserved by serialising behind any waiter already queued for this
|
|
221
|
+
* instance, *including* when the socket happens to be up: otherwise a message
|
|
222
|
+
* arriving after the reconnect could overtake one that has been waiting for it.
|
|
223
|
+
*/
|
|
224
|
+
private sendWhenConnected;
|
|
225
|
+
/** Poll for the instance's IPC to come back, then send. Throws if it does not. */
|
|
226
|
+
private sendAfterIpcReturns;
|
|
203
227
|
/** Single delivery facade: wake paused CLIs and serialize non-user work behind idle. */
|
|
204
228
|
deliverToInstance(instanceName: string, payload: Record<string, unknown>, options?: DeliveryOptions): Promise<void>;
|
|
205
229
|
/** Fleet admin is an explicit config allowlist entry, not merely an open/paired user. */
|
|
@@ -300,6 +324,20 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
300
324
|
private restartAdapter;
|
|
301
325
|
/** Handle inbound message — transcribe voice if present, then route */
|
|
302
326
|
private findGeneralInstance;
|
|
327
|
+
/**
|
|
328
|
+
* A user reacted to one of the bot's messages (#408).
|
|
329
|
+
*
|
|
330
|
+
* Delivered as a normal inbound so it reuses routing, dedup, the idle gate and
|
|
331
|
+
* delivery confirmation — nothing new is needed on that path. `message_id`
|
|
332
|
+
* identifies the bot message that was reacted to, and since the inbound block now
|
|
333
|
+
* renders message_id, the agent can tell WHICH of its messages this refers to.
|
|
334
|
+
*
|
|
335
|
+
* Policy: only the approval emojis wake an instance. A 👍 costing a full agent
|
|
336
|
+
* turn is acceptable when it means "approved"; every other emoji is chatter and
|
|
337
|
+
* must not spend a turn, so those are delivered only to an instance that is
|
|
338
|
+
* already awake (`no_wake`) and dropped for a paused one.
|
|
339
|
+
*/
|
|
340
|
+
private handleInboundReaction;
|
|
303
341
|
private handleInboundMessage;
|
|
304
342
|
/** Handle outbound tool calls from a daemon instance */
|
|
305
343
|
/** Warn (but don't block) when rate limits are high. 30-min debounce per instance. */
|
|
@@ -373,6 +411,25 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
373
411
|
*/
|
|
374
412
|
private runBackendDoctor;
|
|
375
413
|
/** Drop event/activity rows older than the retention window. Best-effort. */
|
|
414
|
+
/**
|
|
415
|
+
* Cap every instance's pipe-pane log, walking the instances **directory** rather
|
|
416
|
+
* than the config.
|
|
417
|
+
*
|
|
418
|
+
* A running instance rotates its own log on each health tick, so the ones that
|
|
419
|
+
* need this are the ones nothing else looks at:
|
|
420
|
+
*
|
|
421
|
+
* - deleted instances, whose directory outlives the config entry. Nothing ever
|
|
422
|
+
* touched these again. On the machine this was found on, one held 122 MB and
|
|
423
|
+
* another 74 MB, out of 622 MB of pipe-pane logs in total.
|
|
424
|
+
* - classic instances, which live in classicChannels, not fleetConfig.instances,
|
|
425
|
+
* and so were never in the old config-driven loop at all.
|
|
426
|
+
* - stopped instances, which have no health tick running.
|
|
427
|
+
*
|
|
428
|
+
* pipe-pane writes raw TUI output, so a wedged splash screen can emit ANSI frames
|
|
429
|
+
* at animation rate. Unbounded growth here fills the disk, which takes the whole
|
|
430
|
+
* fleet down rather than one instance.
|
|
431
|
+
*/
|
|
432
|
+
private rotateAllInstanceLogs;
|
|
376
433
|
private pruneEventLog;
|
|
377
434
|
private openEventLog;
|
|
378
435
|
/**
|
|
@@ -402,6 +459,40 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
|
|
|
402
459
|
/** Whether the instance currently has at least one live cancel button. */
|
|
403
460
|
private hasCancelButton;
|
|
404
461
|
sendCancelButton(instanceName: string, correlationId?: string): Promise<void>;
|
|
462
|
+
/**
|
|
463
|
+
* The cancel button's text for a given elapsed time.
|
|
464
|
+
*
|
|
465
|
+
* Below the threshold it keeps the original wording, so a normal quick answer
|
|
466
|
+
* looks exactly as it did before. Past it, the button doubles as the live
|
|
467
|
+
* progress indicator (#409) — the channel showed nothing at all during long work,
|
|
468
|
+
* and once the agent had replied once there was no sign it was still going.
|
|
469
|
+
*/
|
|
470
|
+
static progressText(elapsedMs: number, activity?: string | null): string;
|
|
471
|
+
/**
|
|
472
|
+
* Make a tool summary safe to paste into a channel message.
|
|
473
|
+
*
|
|
474
|
+
* The text is agent-controlled (it is built from tool inputs — file paths,
|
|
475
|
+
* shell commands), so it gets flattened to one line, capped, and stripped of
|
|
476
|
+
* the two Discord mass-mention triggers. Neither channel renders it with a
|
|
477
|
+
* parse mode, so no markup escaping is needed beyond that.
|
|
478
|
+
*/
|
|
479
|
+
private static sanitizeActivity;
|
|
480
|
+
/**
|
|
481
|
+
* Remember what an instance is currently doing, for the progress line.
|
|
482
|
+
*
|
|
483
|
+
* Best-effort by design: only backends that expose a live activity feed report
|
|
484
|
+
* anything, and the progress line simply omits the detail for the rest. It is
|
|
485
|
+
* never used to decide anything — purely what the user is shown.
|
|
486
|
+
*/
|
|
487
|
+
private cacheInstanceActivity;
|
|
488
|
+
/**
|
|
489
|
+
* Refresh the button's text in place while the instance keeps working.
|
|
490
|
+
*
|
|
491
|
+
* Uses `editAlert`, NOT `editMessage`: on Telegram the latter omits reply_markup,
|
|
492
|
+
* and the Bot API treats that as "clear the keyboard" — so editing with it would
|
|
493
|
+
* delete the very cancel button this is trying to keep alive.
|
|
494
|
+
*/
|
|
495
|
+
private startProgressTicker;
|
|
405
496
|
/** Retire (delete) every cancel button belonging to an instance. */
|
|
406
497
|
private retireInstanceButtons;
|
|
407
498
|
/** Begin retiring one button (delete + bounded retry on failure). Idempotent:
|
package/dist/fleet-manager.js
CHANGED
|
@@ -21,7 +21,7 @@ import { IpcClient } from "./channel/ipc-bridge.js";
|
|
|
21
21
|
import { createAdapter } from "./channel/factory.js";
|
|
22
22
|
import { createBackend } from "./backend/factory.js";
|
|
23
23
|
import { isModelCompatible } from "./backend/types.js";
|
|
24
|
-
import { createLogger } from "./logger.js";
|
|
24
|
+
import { createLogger, rotateLogIfNeeded } from "./logger.js";
|
|
25
25
|
import { processAttachments } from "./channel/attachment-handler.js";
|
|
26
26
|
import { routeToolCall } from "./channel/tool-router.js";
|
|
27
27
|
import { Scheduler } from "./scheduler/index.js";
|
|
@@ -84,6 +84,33 @@ const CANCEL_BTN_MAX_RETRIES = 3;
|
|
|
84
84
|
* buttons no clear trigger reached (e.g. a scheduled/HTTP turn that never called
|
|
85
85
|
* reply). 5min (not the old 2s idle-watch) so Thinking isn't misread as idle. */
|
|
86
86
|
const CANCEL_BTN_IDLE_CHECK_INTERVAL_MS = 5 * 60_000;
|
|
87
|
+
/**
|
|
88
|
+
* How often the cancel button's text is refreshed with elapsed working time.
|
|
89
|
+
*
|
|
90
|
+
* One edit per working instance per interval — at 60s that is trivial for both
|
|
91
|
+
* platforms' rate limits, and it reads as a live counter rather than a stale
|
|
92
|
+
* snapshot. Nothing new is posted, so the channel is never spammed: there is
|
|
93
|
+
* exactly one progress message per turn, and it is the cancel button itself.
|
|
94
|
+
*/
|
|
95
|
+
const PROGRESS_UPDATE_INTERVAL_MS = 60_000;
|
|
96
|
+
/** Elapsed time is only shown once work has clearly outlasted a quick answer. */
|
|
97
|
+
const PROGRESS_MIN_ELAPSED_MS = 2 * 60_000;
|
|
98
|
+
/** How much of a tool summary the progress line will show before eliding. */
|
|
99
|
+
const PROGRESS_ACTIVITY_MAX_CHARS = 48;
|
|
100
|
+
/**
|
|
101
|
+
* How long a delivery waits out a disconnected instance IPC before giving up.
|
|
102
|
+
*
|
|
103
|
+
* Sized for a daemon restart (socket close → respawn → CLI ready), which is the
|
|
104
|
+
* event this exists for. Past it the delivery fails loudly as it always did.
|
|
105
|
+
*/
|
|
106
|
+
const IPC_RECONNECT_GRACE_MS = 30_000;
|
|
107
|
+
const IPC_RECONNECT_POLL_MS = 250;
|
|
108
|
+
/**
|
|
109
|
+
* Reactions that count as an explicit approve/reject signal, and are therefore worth
|
|
110
|
+
* waking a paused instance for. Everything else is chatter: delivered only if the
|
|
111
|
+
* instance is already awake, never worth a wake-up plus a full agent turn.
|
|
112
|
+
*/
|
|
113
|
+
const REACTION_APPROVAL_EMOJIS = new Set(["👍", "👎"]);
|
|
87
114
|
const CLASSIC_BACKEND_SELECTION_TIMEOUT_MS = 60_000;
|
|
88
115
|
const CLASSIC_BACKEND_CALLBACK_PREFIX = "classic-backend:";
|
|
89
116
|
const MODEL_SELECT_CALLBACK_PREFIX = "model-select:";
|
|
@@ -94,7 +121,7 @@ export class FleetManager {
|
|
|
94
121
|
static sighupHandlerInstalled = false;
|
|
95
122
|
children = new Map();
|
|
96
123
|
lifecycle;
|
|
97
|
-
/**
|
|
124
|
+
/** Live view of lifecycle.daemons — used throughout; not deprecated. */
|
|
98
125
|
get daemons() { return this.lifecycle.daemons; }
|
|
99
126
|
fleetConfig = null;
|
|
100
127
|
rawFleetConfig = {};
|
|
@@ -152,6 +179,10 @@ export class FleetManager {
|
|
|
152
179
|
// reply, on cancel, or when a newer button supersedes it for the same
|
|
153
180
|
// instance. Per-button tracking means a failed delete never strands a button.
|
|
154
181
|
cancelButtons = new Map();
|
|
182
|
+
/** instanceName → what it is doing right now, when the backend can tell us. */
|
|
183
|
+
instanceActivity = new Map();
|
|
184
|
+
/** instanceName → tail of deliveries waiting for its IPC to come back. */
|
|
185
|
+
ipcWaitTails = new Map();
|
|
155
186
|
// Last user message delivered to each instance — used to react ✅ on completion.
|
|
156
187
|
lastInboundMsg = new Map();
|
|
157
188
|
topicArchiver;
|
|
@@ -178,6 +209,7 @@ export class FleetManager {
|
|
|
178
209
|
healthPortRetried = false;
|
|
179
210
|
updateCheckTimer = null;
|
|
180
211
|
eventLogPruneTimer = null;
|
|
212
|
+
logRotateTimer = null;
|
|
181
213
|
/** Days of event/activity history to keep. */
|
|
182
214
|
static EVENT_LOG_RETENTION_DAYS = 30;
|
|
183
215
|
watchdogTimer = null;
|
|
@@ -218,7 +250,13 @@ export class FleetManager {
|
|
|
218
250
|
}
|
|
219
251
|
this.reloadPending = false;
|
|
220
252
|
this.reconcileInFlight = this.reconcileInstances()
|
|
221
|
-
.catch(err =>
|
|
253
|
+
.catch(err => {
|
|
254
|
+
// Almost always a YAML parse error. Log-only meant the user edited
|
|
255
|
+
// fleet.yaml, sent SIGHUP, and got no reaction and no explanation.
|
|
256
|
+
this.logger.error({ err }, "SIGHUP config reload failed");
|
|
257
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
258
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload FAILED — ${message}\nThe previous configuration is still running.`);
|
|
259
|
+
})
|
|
222
260
|
.finally(() => {
|
|
223
261
|
this.reconcileInFlight = null;
|
|
224
262
|
if (this.reloadPending && this.startupComplete) {
|
|
@@ -481,8 +519,12 @@ export class FleetManager {
|
|
|
481
519
|
// warm_cap: a fresh transition into idle may free this instance for eviction,
|
|
482
520
|
// or (more usefully) reveal that the fleet is now over cap. Only fire on the
|
|
483
521
|
// edge into idle, not on every idle heartbeat.
|
|
484
|
-
if (state === "idle" && previous?.state !== "idle")
|
|
522
|
+
if (state === "idle" && previous?.state !== "idle") {
|
|
485
523
|
this.enforceWarmCap();
|
|
524
|
+
// The turn is genuinely over — retire the cancel/progress button now rather
|
|
525
|
+
// than waiting for the 5-minute idle backstop to notice.
|
|
526
|
+
this.retireInstanceButtons(name);
|
|
527
|
+
}
|
|
486
528
|
}
|
|
487
529
|
cacheInstanceProcessStatus(name, status) {
|
|
488
530
|
if (status === "running") {
|
|
@@ -587,12 +629,70 @@ export class FleetManager {
|
|
|
587
629
|
if (!idle) {
|
|
588
630
|
this.logger.warn({ instanceName, timeoutMs }, "Idle gate timed out; forcing delivery");
|
|
589
631
|
}
|
|
590
|
-
|
|
591
|
-
if (!ipc?.connected)
|
|
592
|
-
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
593
|
-
ipc.send(payload);
|
|
632
|
+
await this.sendWhenConnected(instanceName, payload);
|
|
594
633
|
this.lastDeliveryAt.set(instanceName, Date.now());
|
|
595
634
|
}
|
|
635
|
+
/**
|
|
636
|
+
* Hand a payload to an instance's IPC, waiting out a *transient* disconnect.
|
|
637
|
+
*
|
|
638
|
+
* A daemon that is restarting — `/restart`, crash recovery, a model switch —
|
|
639
|
+
* drops its socket for a few seconds. Any message arriving in that window used
|
|
640
|
+
* to fail instantly: the caller logged a warning, put ❌ on the user's message,
|
|
641
|
+
* and the message was gone. The user had to notice the ❌ and retype it. That is
|
|
642
|
+
* the "instance 訊息不容易掉" goal failing on the most predictable event there is.
|
|
643
|
+
*
|
|
644
|
+
* The wait is bounded. If the instance is genuinely down, this still throws and
|
|
645
|
+
* the ❌ still appears — just for a real failure rather than a restart.
|
|
646
|
+
*
|
|
647
|
+
* Ordering is preserved by serialising behind any waiter already queued for this
|
|
648
|
+
* instance, *including* when the socket happens to be up: otherwise a message
|
|
649
|
+
* arriving after the reconnect could overtake one that has been waiting for it.
|
|
650
|
+
*/
|
|
651
|
+
async sendWhenConnected(instanceName, payload) {
|
|
652
|
+
const queued = this.ipcWaitTails.get(instanceName);
|
|
653
|
+
if (!queued) {
|
|
654
|
+
const ipc = this.instanceIpcClients.get(instanceName);
|
|
655
|
+
if (ipc?.connected && ipc.send(payload))
|
|
656
|
+
return;
|
|
657
|
+
}
|
|
658
|
+
const attempt = (queued ?? Promise.resolve())
|
|
659
|
+
.catch(() => { })
|
|
660
|
+
.then(() => this.sendAfterIpcReturns(instanceName, payload));
|
|
661
|
+
// The chain stores a settled-either-way promise so one failed delivery cannot
|
|
662
|
+
// wedge every later one, and so `queued` above is safe to await unguarded.
|
|
663
|
+
const tail = attempt.catch(() => { });
|
|
664
|
+
this.ipcWaitTails.set(instanceName, tail);
|
|
665
|
+
try {
|
|
666
|
+
await attempt;
|
|
667
|
+
}
|
|
668
|
+
finally {
|
|
669
|
+
// Only the last waiter clears the chain; while a queue is still draining the
|
|
670
|
+
// map must keep pointing at it or ordering is lost.
|
|
671
|
+
if (this.ipcWaitTails.get(instanceName) === tail) {
|
|
672
|
+
this.ipcWaitTails.delete(instanceName);
|
|
673
|
+
}
|
|
674
|
+
}
|
|
675
|
+
}
|
|
676
|
+
/** Poll for the instance's IPC to come back, then send. Throws if it does not. */
|
|
677
|
+
async sendAfterIpcReturns(instanceName, payload) {
|
|
678
|
+
const deadline = Date.now() + IPC_RECONNECT_GRACE_MS;
|
|
679
|
+
let warned = false;
|
|
680
|
+
for (;;) {
|
|
681
|
+
// Re-read every round: a reconnect replaces the IpcClient object entirely,
|
|
682
|
+
// so a cached reference would stay dead forever.
|
|
683
|
+
const ipc = this.instanceIpcClients.get(instanceName);
|
|
684
|
+
if (ipc?.connected && ipc.send(payload))
|
|
685
|
+
return;
|
|
686
|
+
if (Date.now() >= deadline) {
|
|
687
|
+
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
688
|
+
}
|
|
689
|
+
if (!warned) {
|
|
690
|
+
warned = true;
|
|
691
|
+
this.logger.info({ instanceName }, "Instance IPC is down — holding delivery until it reconnects");
|
|
692
|
+
}
|
|
693
|
+
await new Promise(resolve => setTimeout(resolve, IPC_RECONNECT_POLL_MS));
|
|
694
|
+
}
|
|
695
|
+
}
|
|
596
696
|
/** Single delivery facade: wake paused CLIs and serialize non-user work behind idle. */
|
|
597
697
|
async deliverToInstance(instanceName, payload, options = {}) {
|
|
598
698
|
const meta = payload.meta && typeof payload.meta === "object"
|
|
@@ -608,10 +708,7 @@ export class FleetManager {
|
|
|
608
708
|
await this.lifecycle.wake(instanceName, 30_000);
|
|
609
709
|
this.enforceWarmCap(instanceName); // woke one → evict a different LRU idle if over cap
|
|
610
710
|
}
|
|
611
|
-
|
|
612
|
-
if (!ipc?.connected)
|
|
613
|
-
throw new Error(`Instance '${instanceName}' IPC is unavailable`);
|
|
614
|
-
ipc.send(payload);
|
|
711
|
+
await this.sendWhenConnected(instanceName, payload);
|
|
615
712
|
// A cross-instance item arriving before the daemon observes this turn as
|
|
616
713
|
// working must not trust the stale idle snapshot from before the send.
|
|
617
714
|
this.lastDeliveryAt.set(instanceName, Date.now());
|
|
@@ -857,7 +954,6 @@ export class FleetManager {
|
|
|
857
954
|
this.configPath = configPath;
|
|
858
955
|
this.loadEnvFile();
|
|
859
956
|
// Rotate fleet.log if oversized (before any logging)
|
|
860
|
-
const { rotateLogIfNeeded } = await import("./logger.js");
|
|
861
957
|
rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
|
|
862
958
|
const fleet = this.loadConfig(configPath);
|
|
863
959
|
setLocale(detectLocale(fleet)); // user-facing text language (fleet.yaml defaults.locale / timezone)
|
|
@@ -990,11 +1086,10 @@ export class FleetManager {
|
|
|
990
1086
|
// Rotate fleet.log daily too (besides the startup size check above), so a
|
|
991
1087
|
// long-running fleet doesn't accumulate an unbounded log.
|
|
992
1088
|
rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
|
|
993
|
-
// Instance output.log is pipe-pane (TUI ANSI). Daemon health ticks
|
|
994
|
-
//
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
}
|
|
1089
|
+
// Instance output.log is pipe-pane (TUI ANSI). Daemon health ticks rotate a
|
|
1090
|
+
// running instance's own log; this sweep is the safety net for every other
|
|
1091
|
+
// kind. One implementation, so the two cannot cover different sets.
|
|
1092
|
+
this.rotateAllInstanceLogs();
|
|
998
1093
|
}, () => {
|
|
999
1094
|
const instances = Object.keys(this.fleetConfig?.instances ?? {});
|
|
1000
1095
|
const costMap = new Map();
|
|
@@ -1151,6 +1246,12 @@ export class FleetManager {
|
|
|
1151
1246
|
this.pruneEventLog();
|
|
1152
1247
|
this.eventLogPruneTimer = setInterval(() => this.pruneEventLog(), 24 * 60 * 60_000);
|
|
1153
1248
|
this.eventLogPruneTimer.unref?.();
|
|
1249
|
+
// Same shape for pipe-pane logs, and for the same reason: the only sweep that
|
|
1250
|
+
// covered them lived inside the daily-summary callback, so it did not run at
|
|
1251
|
+
// all when summaries were off.
|
|
1252
|
+
this.rotateAllInstanceLogs();
|
|
1253
|
+
this.logRotateTimer = setInterval(() => this.rotateAllInstanceLogs(), 24 * 60 * 60_000);
|
|
1254
|
+
this.logRotateTimer.unref?.();
|
|
1154
1255
|
// Phase 2: Start remaining instances with staggered concurrency
|
|
1155
1256
|
if (others.length > 0) {
|
|
1156
1257
|
await this.startInstancesWithConcurrency(others, topicMode);
|
|
@@ -1497,6 +1598,9 @@ export class FleetManager {
|
|
|
1497
1598
|
this.adapter.on("message", safeHandler(async (msg) => {
|
|
1498
1599
|
await this.handleInboundMessage(msg);
|
|
1499
1600
|
}, this.logger, "adapter.message"));
|
|
1601
|
+
this.adapter.on("reaction", safeHandler(async (r) => {
|
|
1602
|
+
await this.handleInboundReaction(r);
|
|
1603
|
+
}, this.logger, "adapter.reaction"));
|
|
1500
1604
|
this.adapter.on("callback_query", safeHandler(async (data) => {
|
|
1501
1605
|
if (await this.handleClassicBackendSelection(data))
|
|
1502
1606
|
return;
|
|
@@ -1782,6 +1886,9 @@ export class FleetManager {
|
|
|
1782
1886
|
adapter.on("message", safeHandler(async (msg) => {
|
|
1783
1887
|
await this.handleInboundMessage(msg);
|
|
1784
1888
|
}, this.logger, `adapter[${adapterId}].message`));
|
|
1889
|
+
adapter.on("reaction", safeHandler(async (r) => {
|
|
1890
|
+
await this.handleInboundReaction(r);
|
|
1891
|
+
}, this.logger, `adapter[${adapterId}].reaction`));
|
|
1785
1892
|
adapter.on("callback_query", safeHandler(async (data) => {
|
|
1786
1893
|
if (await this.handleClassicBackendSelection(data))
|
|
1787
1894
|
return;
|
|
@@ -2101,6 +2208,9 @@ export class FleetManager {
|
|
|
2101
2208
|
else if (msg.type === "instance_process_state") {
|
|
2102
2209
|
this.cacheInstanceProcessStatus(name, msg.status);
|
|
2103
2210
|
}
|
|
2211
|
+
else if (msg.type === "instance_activity") {
|
|
2212
|
+
this.cacheInstanceActivity(name, msg.activity);
|
|
2213
|
+
}
|
|
2104
2214
|
else if (msg.type === "instance_state" || msg.type === "instance_state_response") {
|
|
2105
2215
|
this.cacheInstanceExecutionState(name, msg);
|
|
2106
2216
|
if (msg.type === "instance_state_response") {
|
|
@@ -2270,6 +2380,54 @@ export class FleetManager {
|
|
|
2270
2380
|
}
|
|
2271
2381
|
return generals[0];
|
|
2272
2382
|
}
|
|
2383
|
+
/**
|
|
2384
|
+
* A user reacted to one of the bot's messages (#408).
|
|
2385
|
+
*
|
|
2386
|
+
* Delivered as a normal inbound so it reuses routing, dedup, the idle gate and
|
|
2387
|
+
* delivery confirmation — nothing new is needed on that path. `message_id`
|
|
2388
|
+
* identifies the bot message that was reacted to, and since the inbound block now
|
|
2389
|
+
* renders message_id, the agent can tell WHICH of its messages this refers to.
|
|
2390
|
+
*
|
|
2391
|
+
* Policy: only the approval emojis wake an instance. A 👍 costing a full agent
|
|
2392
|
+
* turn is acceptable when it means "approved"; every other emoji is chatter and
|
|
2393
|
+
* must not spend a turn, so those are delivered only to an instance that is
|
|
2394
|
+
* already awake (`no_wake`) and dropped for a paused one.
|
|
2395
|
+
*/
|
|
2396
|
+
async handleInboundReaction(r) {
|
|
2397
|
+
const instanceName = this.resolveSlashTarget(r.threadId ?? r.chatId, r.adapterId);
|
|
2398
|
+
if (!instanceName) {
|
|
2399
|
+
this.logger.debug({ emoji: r.emoji, chatId: r.chatId }, "Reaction in an unrouted channel — ignoring");
|
|
2400
|
+
return;
|
|
2401
|
+
}
|
|
2402
|
+
const isApproval = REACTION_APPROVAL_EMOJIS.has(r.emoji);
|
|
2403
|
+
const verb = r.action === "add" ? "reacted" : "removed their reaction";
|
|
2404
|
+
const content = `[reaction:${r.emoji}] ${r.username} ${verb} on your message ${r.messageId}`;
|
|
2405
|
+
this.eventLog?.logActivity("reaction", r.username, `${r.emoji} ${r.action}`, instanceName);
|
|
2406
|
+
try {
|
|
2407
|
+
await this.deliverToInstance(instanceName, {
|
|
2408
|
+
type: "fleet_inbound",
|
|
2409
|
+
content,
|
|
2410
|
+
meta: {
|
|
2411
|
+
chat_id: "", // not a chat message — must not become the reply target
|
|
2412
|
+
thread_id: "",
|
|
2413
|
+
message_id: r.messageId,
|
|
2414
|
+
user: r.username,
|
|
2415
|
+
user_id: r.userId,
|
|
2416
|
+
source: r.source,
|
|
2417
|
+
ts: r.timestamp.toISOString(),
|
|
2418
|
+
request_kind: "update",
|
|
2419
|
+
requires_reply: "false",
|
|
2420
|
+
// Non-approval reactions never wake a paused instance.
|
|
2421
|
+
...(isApproval ? {} : { no_wake: "true" }),
|
|
2422
|
+
},
|
|
2423
|
+
});
|
|
2424
|
+
}
|
|
2425
|
+
catch (err) {
|
|
2426
|
+
// A reaction is a nice-to-have signal; failing to deliver one must not be
|
|
2427
|
+
// noisy. The user can always say it in words.
|
|
2428
|
+
this.logger.debug({ err, instanceName, emoji: r.emoji }, "Reaction delivery failed");
|
|
2429
|
+
}
|
|
2430
|
+
}
|
|
2273
2431
|
async handleInboundMessage(msg) {
|
|
2274
2432
|
const threadId = msg.threadId || undefined;
|
|
2275
2433
|
this.logger.debug({ source: msg.source, chatId: msg.chatId, threadId, userId: msg.userId, isBotMessage: msg.isBotMessage, textLen: (msg.text ?? "").length, text: (msg.text ?? "").slice(0, 80) }, "handleInboundMessage entry");
|
|
@@ -2809,8 +2967,18 @@ export class FleetManager {
|
|
|
2809
2967
|
// Route standard channel tools (reply, react, edit_message, download_attachment)
|
|
2810
2968
|
if (routeToolCall(outAdapter, tool, args, threadId, respond)) {
|
|
2811
2969
|
if (tool === "reply") {
|
|
2812
|
-
//
|
|
2813
|
-
|
|
2970
|
+
// A reply is NOT proof the turn is over: on multi-step work an agent
|
|
2971
|
+
// replies ("starting…") and keeps going for many minutes. Retiring the
|
|
2972
|
+
// button here left the channel looking idle with no way to cancel and no
|
|
2973
|
+
// sign anything was happening (#410). Idle state owns retirement now; if the
|
|
2974
|
+
// instance is still working, move the button below the new reply so it stays
|
|
2975
|
+
// the last thing in the channel.
|
|
2976
|
+
if (this.getInstanceIdle(instanceName)) {
|
|
2977
|
+
this.clearCancelButton(instanceName);
|
|
2978
|
+
}
|
|
2979
|
+
else {
|
|
2980
|
+
void this.sendCancelButton(instanceName);
|
|
2981
|
+
}
|
|
2814
2982
|
this.reactDone(instanceName);
|
|
2815
2983
|
const replyTo = this.lastInboundUser.get(instanceName) ?? "user";
|
|
2816
2984
|
this.logger.info(`${instanceName} → ${replyTo}: ${(args.text ?? "").slice(0, 100)}`);
|
|
@@ -3592,6 +3760,41 @@ export class FleetManager {
|
|
|
3592
3760
|
}
|
|
3593
3761
|
}
|
|
3594
3762
|
/** Drop event/activity rows older than the retention window. Best-effort. */
|
|
3763
|
+
/**
|
|
3764
|
+
* Cap every instance's pipe-pane log, walking the instances **directory** rather
|
|
3765
|
+
* than the config.
|
|
3766
|
+
*
|
|
3767
|
+
* A running instance rotates its own log on each health tick, so the ones that
|
|
3768
|
+
* need this are the ones nothing else looks at:
|
|
3769
|
+
*
|
|
3770
|
+
* - deleted instances, whose directory outlives the config entry. Nothing ever
|
|
3771
|
+
* touched these again. On the machine this was found on, one held 122 MB and
|
|
3772
|
+
* another 74 MB, out of 622 MB of pipe-pane logs in total.
|
|
3773
|
+
* - classic instances, which live in classicChannels, not fleetConfig.instances,
|
|
3774
|
+
* and so were never in the old config-driven loop at all.
|
|
3775
|
+
* - stopped instances, which have no health tick running.
|
|
3776
|
+
*
|
|
3777
|
+
* pipe-pane writes raw TUI output, so a wedged splash screen can emit ANSI frames
|
|
3778
|
+
* at animation rate. Unbounded growth here fills the disk, which takes the whole
|
|
3779
|
+
* fleet down rather than one instance.
|
|
3780
|
+
*/
|
|
3781
|
+
rotateAllInstanceLogs() {
|
|
3782
|
+
const root = join(this.dataDir, "instances");
|
|
3783
|
+
let entries;
|
|
3784
|
+
try {
|
|
3785
|
+
entries = readdirSync(root, { withFileTypes: true });
|
|
3786
|
+
}
|
|
3787
|
+
catch {
|
|
3788
|
+
return; // no instances directory yet
|
|
3789
|
+
}
|
|
3790
|
+
for (const entry of entries) {
|
|
3791
|
+
if (!entry.isDirectory())
|
|
3792
|
+
continue;
|
|
3793
|
+
// rotateLogIfNeeded is already best-effort and returns early on a missing
|
|
3794
|
+
// file, so a directory without a pipe-pane log costs one stat.
|
|
3795
|
+
rotateLogIfNeeded(join(root, entry.name, "output.log"));
|
|
3796
|
+
}
|
|
3797
|
+
}
|
|
3595
3798
|
pruneEventLog() {
|
|
3596
3799
|
try {
|
|
3597
3800
|
this.eventLog?.prune(FleetManager.EVENT_LOG_RETENTION_DAYS);
|
|
@@ -3808,7 +4011,12 @@ export class FleetManager {
|
|
|
3808
4011
|
threadId: sent.threadId ?? threadId,
|
|
3809
4012
|
correlationId,
|
|
3810
4013
|
retryCount: 0,
|
|
4014
|
+
// Elapsed time is measured from when this button was posted — i.e. from
|
|
4015
|
+
// when the work was handed over — not from the pane's working transition,
|
|
4016
|
+
// which resets if the CLI blips idle mid-turn.
|
|
4017
|
+
startedAt: Date.now(),
|
|
3811
4018
|
};
|
|
4019
|
+
this.startProgressTicker(entry);
|
|
3812
4020
|
// Idle-check backstop: every 5min, if the instance is idle, retire the
|
|
3813
4021
|
// button. Covers turns that end without hitting a clear trigger (reply /
|
|
3814
4022
|
// cancel / correlation). Cleared in discardButton when the entry is removed.
|
|
@@ -3829,6 +4037,97 @@ export class FleetManager {
|
|
|
3829
4037
|
this.logger.warn({ err: e.message, instanceName }, "Failed to send cancel button");
|
|
3830
4038
|
}
|
|
3831
4039
|
}
|
|
4040
|
+
/**
|
|
4041
|
+
* The cancel button's text for a given elapsed time.
|
|
4042
|
+
*
|
|
4043
|
+
* Below the threshold it keeps the original wording, so a normal quick answer
|
|
4044
|
+
* looks exactly as it did before. Past it, the button doubles as the live
|
|
4045
|
+
* progress indicator (#409) — the channel showed nothing at all during long work,
|
|
4046
|
+
* and once the agent had replied once there was no sign it was still going.
|
|
4047
|
+
*/
|
|
4048
|
+
static progressText(elapsedMs, activity) {
|
|
4049
|
+
if (elapsedMs < PROGRESS_MIN_ELAPSED_MS)
|
|
4050
|
+
return "👀 處理中…";
|
|
4051
|
+
const totalSeconds = Math.floor(elapsedMs / 1000);
|
|
4052
|
+
const minutes = Math.floor(totalSeconds / 60);
|
|
4053
|
+
const seconds = totalSeconds % 60;
|
|
4054
|
+
const elapsed = minutes >= 60
|
|
4055
|
+
? `${Math.floor(minutes / 60)}h ${minutes % 60}m`
|
|
4056
|
+
: `${minutes}m ${String(seconds).padStart(2, "0")}s`;
|
|
4057
|
+
const detail = FleetManager.sanitizeActivity(activity);
|
|
4058
|
+
return detail
|
|
4059
|
+
? `⏳ 處理中… (已進行 ${elapsed} · ${detail})`
|
|
4060
|
+
: `⏳ 處理中… (已進行 ${elapsed})`;
|
|
4061
|
+
}
|
|
4062
|
+
/**
|
|
4063
|
+
* Make a tool summary safe to paste into a channel message.
|
|
4064
|
+
*
|
|
4065
|
+
* The text is agent-controlled (it is built from tool inputs — file paths,
|
|
4066
|
+
* shell commands), so it gets flattened to one line, capped, and stripped of
|
|
4067
|
+
* the two Discord mass-mention triggers. Neither channel renders it with a
|
|
4068
|
+
* parse mode, so no markup escaping is needed beyond that.
|
|
4069
|
+
*/
|
|
4070
|
+
static sanitizeActivity(activity) {
|
|
4071
|
+
if (!activity)
|
|
4072
|
+
return null;
|
|
4073
|
+
const flat = activity.replace(/\s+/g, " ").replace(/@(everyone|here)/g, "@$1").trim();
|
|
4074
|
+
if (!flat)
|
|
4075
|
+
return null;
|
|
4076
|
+
return flat.length > PROGRESS_ACTIVITY_MAX_CHARS
|
|
4077
|
+
? `${flat.slice(0, PROGRESS_ACTIVITY_MAX_CHARS - 1)}…`
|
|
4078
|
+
: flat;
|
|
4079
|
+
}
|
|
4080
|
+
/**
|
|
4081
|
+
* Remember what an instance is currently doing, for the progress line.
|
|
4082
|
+
*
|
|
4083
|
+
* Best-effort by design: only backends that expose a live activity feed report
|
|
4084
|
+
* anything, and the progress line simply omits the detail for the rest. It is
|
|
4085
|
+
* never used to decide anything — purely what the user is shown.
|
|
4086
|
+
*/
|
|
4087
|
+
cacheInstanceActivity(name, activity) {
|
|
4088
|
+
if (activity)
|
|
4089
|
+
this.instanceActivity.set(name, activity);
|
|
4090
|
+
else
|
|
4091
|
+
this.instanceActivity.delete(name);
|
|
4092
|
+
}
|
|
4093
|
+
/**
|
|
4094
|
+
* Refresh the button's text in place while the instance keeps working.
|
|
4095
|
+
*
|
|
4096
|
+
* Uses `editAlert`, NOT `editMessage`: on Telegram the latter omits reply_markup,
|
|
4097
|
+
* and the Bot API treats that as "clear the keyboard" — so editing with it would
|
|
4098
|
+
* delete the very cancel button this is trying to keep alive.
|
|
4099
|
+
*/
|
|
4100
|
+
startProgressTicker(entry) {
|
|
4101
|
+
entry.progressTimer = setInterval(() => {
|
|
4102
|
+
if (!this.cancelButtons.has(entry.messageId)) {
|
|
4103
|
+
clearInterval(entry.progressTimer);
|
|
4104
|
+
return;
|
|
4105
|
+
}
|
|
4106
|
+
// Idle means the turn ended; the idle-edge handler retires the button.
|
|
4107
|
+
if (this.getInstanceIdle(entry.instanceName))
|
|
4108
|
+
return;
|
|
4109
|
+
const text = FleetManager.progressText(Date.now() - (entry.startedAt ?? Date.now()), this.instanceActivity.get(entry.instanceName));
|
|
4110
|
+
if (text === entry.lastProgressText)
|
|
4111
|
+
return; // nothing changed — skip the API call
|
|
4112
|
+
const adapter = this.getAdapterForInstance(entry.instanceName) ?? this.adapter;
|
|
4113
|
+
if (!adapter?.editAlert)
|
|
4114
|
+
return;
|
|
4115
|
+
entry.lastProgressText = text;
|
|
4116
|
+
adapter.editAlert(entry.chatId, entry.messageId, {
|
|
4117
|
+
type: "cancel",
|
|
4118
|
+
instanceName: entry.instanceName,
|
|
4119
|
+
message: text,
|
|
4120
|
+
choices: [{ id: `cancel:${entry.instanceName}`, label: t("cancel.button") }],
|
|
4121
|
+
}, entry.threadId ? { threadId: entry.threadId } : undefined)
|
|
4122
|
+
.catch(err => {
|
|
4123
|
+
// A failed progress edit must never escalate: the button still works and
|
|
4124
|
+
// the next tick retries. Common causes are a deleted message or a
|
|
4125
|
+
// rate limit.
|
|
4126
|
+
this.logger.debug({ err, instanceName: entry.instanceName }, "Progress edit failed");
|
|
4127
|
+
});
|
|
4128
|
+
}, PROGRESS_UPDATE_INTERVAL_MS);
|
|
4129
|
+
entry.progressTimer.unref?.();
|
|
4130
|
+
}
|
|
3832
4131
|
/** Retire (delete) every cancel button belonging to an instance. */
|
|
3833
4132
|
retireInstanceButtons(instanceName) {
|
|
3834
4133
|
// Snapshot first — retireButton may delete entries from the map on success.
|
|
@@ -3860,6 +4159,8 @@ export class FleetManager {
|
|
|
3860
4159
|
clearTimeout(entry.retryTimer);
|
|
3861
4160
|
if (entry.idleCheckTimer)
|
|
3862
4161
|
clearInterval(entry.idleCheckTimer);
|
|
4162
|
+
if (entry.progressTimer)
|
|
4163
|
+
clearInterval(entry.progressTimer);
|
|
3863
4164
|
this.cancelButtons.delete(entry.messageId);
|
|
3864
4165
|
}
|
|
3865
4166
|
/** Re-attempt a failed button delete up to CANCEL_BTN_MAX_RETRIES times. */
|
|
@@ -5139,6 +5440,22 @@ When users create specialized instances, suggest these configurations:
|
|
|
5139
5440
|
clearInterval(this.eventLogPruneTimer);
|
|
5140
5441
|
this.eventLogPruneTimer = null;
|
|
5141
5442
|
}
|
|
5443
|
+
if (this.logRotateTimer) {
|
|
5444
|
+
clearInterval(this.logRotateTimer);
|
|
5445
|
+
this.logRotateTimer = null;
|
|
5446
|
+
}
|
|
5447
|
+
// Cancel-button timers were never cleared here. The idle-check interval is not
|
|
5448
|
+
// unref'd, so it held the event loop open past shutdown and kept retrying
|
|
5449
|
+
// deletes against an adapter that was already gone.
|
|
5450
|
+
for (const entry of [...this.cancelButtons.values()]) {
|
|
5451
|
+
if (entry.retryTimer)
|
|
5452
|
+
clearTimeout(entry.retryTimer);
|
|
5453
|
+
if (entry.idleCheckTimer)
|
|
5454
|
+
clearInterval(entry.idleCheckTimer);
|
|
5455
|
+
if (entry.progressTimer)
|
|
5456
|
+
clearInterval(entry.progressTimer);
|
|
5457
|
+
}
|
|
5458
|
+
this.cancelButtons.clear();
|
|
5142
5459
|
if (this.topicCleanupTimer) {
|
|
5143
5460
|
clearInterval(this.topicCleanupTimer);
|
|
5144
5461
|
this.topicCleanupTimer = null;
|
|
@@ -5357,6 +5674,15 @@ When users create specialized instances, suggest these configurations:
|
|
|
5357
5674
|
removedRatio,
|
|
5358
5675
|
validationErrors: validation.errors,
|
|
5359
5676
|
}, "Refusing unsafe fleet config reload; running configuration was kept");
|
|
5677
|
+
// Tell the operator. A silently ignored config edit is the most confusing
|
|
5678
|
+
// possible outcome: they change fleet.yaml, send SIGHUP, and nothing happens
|
|
5679
|
+
// with no explanation anywhere they are looking.
|
|
5680
|
+
const why = !validation.valid
|
|
5681
|
+
? `validation failed:\n${validation.errors.map(e => `• ${e.path}: ${e.message}`).join("\n")}`
|
|
5682
|
+
: unsafeEmpty
|
|
5683
|
+
? `it removed every instance (${oldCount} → 0)`
|
|
5684
|
+
: `it removed more than half the instances (${oldCount} → ${newCount})`;
|
|
5685
|
+
this.notifyFleetError(`⚠️ fleet.yaml reload REJECTED — ${why}\nThe previous configuration is still running.`);
|
|
5360
5686
|
return;
|
|
5361
5687
|
}
|
|
5362
5688
|
this.routing.rebuild(this.fleetConfig);
|