@songsid/agend 2.1.1-beta.17 → 2.1.1-beta.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/dist/backend/claude-code.d.ts +37 -0
  2. package/dist/backend/claude-code.js +40 -1
  3. package/dist/backend/claude-code.js.map +1 -1
  4. package/dist/backend/kiro.d.ts +20 -0
  5. package/dist/backend/kiro.js +30 -0
  6. package/dist/backend/kiro.js.map +1 -1
  7. package/dist/backend/types.d.ts +27 -0
  8. package/dist/backend/types.js.map +1 -1
  9. package/dist/channel/adapters/discord.d.ts +15 -0
  10. package/dist/channel/adapters/discord.js +81 -1
  11. package/dist/channel/adapters/discord.js.map +1 -1
  12. package/dist/channel/adapters/telegram.d.ts +21 -0
  13. package/dist/channel/adapters/telegram.js +82 -0
  14. package/dist/channel/adapters/telegram.js.map +1 -1
  15. package/dist/channel/ipc-bridge.d.ts +9 -1
  16. package/dist/channel/ipc-bridge.js +12 -3
  17. package/dist/channel/ipc-bridge.js.map +1 -1
  18. package/dist/channel/types.d.ts +29 -0
  19. package/dist/daemon.d.ts +101 -5
  20. package/dist/daemon.js +304 -50
  21. package/dist/daemon.js.map +1 -1
  22. package/dist/fleet-manager.d.ts +92 -1
  23. package/dist/fleet-manager.js +330 -19
  24. package/dist/fleet-manager.js.map +1 -1
  25. package/dist/hang-detector.d.ts +19 -13
  26. package/dist/hang-detector.js +19 -49
  27. package/dist/hang-detector.js.map +1 -1
  28. package/dist/pane-write-lock.d.ts +48 -0
  29. package/dist/pane-write-lock.js +73 -0
  30. package/dist/pane-write-lock.js.map +1 -0
  31. package/dist/tmux-control.d.ts +48 -1
  32. package/dist/tmux-control.js +79 -6
  33. package/dist/tmux-control.js.map +1 -1
  34. package/package.json +1 -1
  35. package/dist/channel/tool-tracker.d.ts +0 -13
  36. package/dist/channel/tool-tracker.js +0 -58
  37. package/dist/channel/tool-tracker.js.map +0 -1
  38. package/dist/daemon-entry.d.ts +0 -1
  39. package/dist/daemon-entry.js +0 -30
  40. package/dist/daemon-entry.js.map +0 -1
@@ -48,7 +48,7 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
48
48
  private static sighupHandlerInstalled;
49
49
  private children;
50
50
  readonly lifecycle: InstanceLifecycle;
51
- /** @deprecated Use lifecycle.daemons — kept for backward compat */
51
+ /** Live view of lifecycle.daemons — used throughout; not deprecated. */
52
52
  get daemons(): Map<string, import("./daemon.js").Daemon>;
53
53
  fleetConfig: FleetConfig | null;
54
54
  private rawFleetConfig;
@@ -98,6 +98,10 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
98
98
  private instanceIdleWaiters;
99
99
  private lastInboundUser;
100
100
  private cancelButtons;
101
+ /** instanceName → what it is doing right now, when the backend can tell us. */
102
+ private instanceActivity;
103
+ /** instanceName → tail of deliveries waiting for its IPC to come back. */
104
+ private ipcWaitTails;
101
105
  private lastInboundMsg;
102
106
  private topicArchiver;
103
107
  controlClient: TmuxControlClient | null;
@@ -118,6 +122,7 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
118
122
  private healthPortRetried;
119
123
  private updateCheckTimer;
120
124
  private eventLogPruneTimer;
125
+ private logRotateTimer;
121
126
  /** Days of event/activity history to keep. */
122
127
  private static readonly EVENT_LOG_RETENTION_DAYS;
123
128
  private watchdogTimer;
@@ -200,6 +205,25 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
200
205
  private enforceWarmCap;
201
206
  private waitForInstanceIdle;
202
207
  private deliverWithIdleGate;
208
+ /**
209
+ * Hand a payload to an instance's IPC, waiting out a *transient* disconnect.
210
+ *
211
+ * A daemon that is restarting — `/restart`, crash recovery, a model switch —
212
+ * drops its socket for a few seconds. Any message arriving in that window used
213
+ * to fail instantly: the caller logged a warning, put ❌ on the user's message,
214
+ * and the message was gone. The user had to notice the ❌ and retype it. That is
215
+ * the "instance 訊息不容易掉" goal failing on the most predictable event there is.
216
+ *
217
+ * The wait is bounded. If the instance is genuinely down, this still throws and
218
+ * the ❌ still appears — just for a real failure rather than a restart.
219
+ *
220
+ * Ordering is preserved by serialising behind any waiter already queued for this
221
+ * instance, *including* when the socket happens to be up: otherwise a message
222
+ * arriving after the reconnect could overtake one that has been waiting for it.
223
+ */
224
+ private sendWhenConnected;
225
+ /** Poll for the instance's IPC to come back, then send. Throws if it does not. */
226
+ private sendAfterIpcReturns;
203
227
  /** Single delivery facade: wake paused CLIs and serialize non-user work behind idle. */
204
228
  deliverToInstance(instanceName: string, payload: Record<string, unknown>, options?: DeliveryOptions): Promise<void>;
205
229
  /** Fleet admin is an explicit config allowlist entry, not merely an open/paired user. */
@@ -300,6 +324,20 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
300
324
  private restartAdapter;
301
325
  /** Handle inbound message — transcribe voice if present, then route */
302
326
  private findGeneralInstance;
327
+ /**
328
+ * A user reacted to one of the bot's messages (#408).
329
+ *
330
+ * Delivered as a normal inbound so it reuses routing, dedup, the idle gate and
331
+ * delivery confirmation — nothing new is needed on that path. `message_id`
332
+ * identifies the bot message that was reacted to, and since the inbound block now
333
+ * renders message_id, the agent can tell WHICH of its messages this refers to.
334
+ *
335
+ * Policy: only the approval emojis wake an instance. A 👍 costing a full agent
336
+ * turn is acceptable when it means "approved"; every other emoji is chatter and
337
+ * must not spend a turn, so those are delivered only to an instance that is
338
+ * already awake (`no_wake`) and dropped for a paused one.
339
+ */
340
+ private handleInboundReaction;
303
341
  private handleInboundMessage;
304
342
  /** Handle outbound tool calls from a daemon instance */
305
343
  /** Warn (but don't block) when rate limits are high. 30-min debounce per instance. */
@@ -373,6 +411,25 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
373
411
  */
374
412
  private runBackendDoctor;
375
413
  /** Drop event/activity rows older than the retention window. Best-effort. */
414
+ /**
415
+ * Cap every instance's pipe-pane log, walking the instances **directory** rather
416
+ * than the config.
417
+ *
418
+ * A running instance rotates its own log on each health tick, so the ones that
419
+ * need this are the ones nothing else looks at:
420
+ *
421
+ * - deleted instances, whose directory outlives the config entry. Nothing ever
422
+ * touched these again. On the machine this was found on, one held 122 MB and
423
+ * another 74 MB, out of 622 MB of pipe-pane logs in total.
424
+ * - classic instances, which live in classicChannels, not fleetConfig.instances,
425
+ * and so were never in the old config-driven loop at all.
426
+ * - stopped instances, which have no health tick running.
427
+ *
428
+ * pipe-pane writes raw TUI output, so a wedged splash screen can emit ANSI frames
429
+ * at animation rate. Unbounded growth here fills the disk, which takes the whole
430
+ * fleet down rather than one instance.
431
+ */
432
+ private rotateAllInstanceLogs;
376
433
  private pruneEventLog;
377
434
  private openEventLog;
378
435
  /**
@@ -402,6 +459,40 @@ export declare class FleetManager implements FleetContext, LifecycleContext, Arc
402
459
  /** Whether the instance currently has at least one live cancel button. */
403
460
  private hasCancelButton;
404
461
  sendCancelButton(instanceName: string, correlationId?: string): Promise<void>;
462
+ /**
463
+ * The cancel button's text for a given elapsed time.
464
+ *
465
+ * Below the threshold it keeps the original wording, so a normal quick answer
466
+ * looks exactly as it did before. Past it, the button doubles as the live
467
+ * progress indicator (#409) — the channel showed nothing at all during long work,
468
+ * and once the agent had replied once there was no sign it was still going.
469
+ */
470
+ static progressText(elapsedMs: number, activity?: string | null): string;
471
+ /**
472
+ * Make a tool summary safe to paste into a channel message.
473
+ *
474
+ * The text is agent-controlled (it is built from tool inputs — file paths,
475
+ * shell commands), so it gets flattened to one line, capped, and stripped of
476
+ * the two Discord mass-mention triggers. Neither channel renders it with a
477
+ * parse mode, so no markup escaping is needed beyond that.
478
+ */
479
+ private static sanitizeActivity;
480
+ /**
481
+ * Remember what an instance is currently doing, for the progress line.
482
+ *
483
+ * Best-effort by design: only backends that expose a live activity feed report
484
+ * anything, and the progress line simply omits the detail for the rest. It is
485
+ * never used to decide anything — purely what the user is shown.
486
+ */
487
+ private cacheInstanceActivity;
488
+ /**
489
+ * Refresh the button's text in place while the instance keeps working.
490
+ *
491
+ * Uses `editAlert`, NOT `editMessage`: on Telegram the latter omits reply_markup,
492
+ * and the Bot API treats that as "clear the keyboard" — so editing with it would
493
+ * delete the very cancel button this is trying to keep alive.
494
+ */
495
+ private startProgressTicker;
405
496
  /** Retire (delete) every cancel button belonging to an instance. */
406
497
  private retireInstanceButtons;
407
498
  /** Begin retiring one button (delete + bounded retry on failure). Idempotent:
@@ -21,7 +21,7 @@ import { IpcClient } from "./channel/ipc-bridge.js";
21
21
  import { createAdapter } from "./channel/factory.js";
22
22
  import { createBackend } from "./backend/factory.js";
23
23
  import { isModelCompatible } from "./backend/types.js";
24
- import { createLogger } from "./logger.js";
24
+ import { createLogger, rotateLogIfNeeded } from "./logger.js";
25
25
  import { processAttachments } from "./channel/attachment-handler.js";
26
26
  import { routeToolCall } from "./channel/tool-router.js";
27
27
  import { Scheduler } from "./scheduler/index.js";
@@ -84,6 +84,33 @@ const CANCEL_BTN_MAX_RETRIES = 3;
84
84
  * buttons no clear trigger reached (e.g. a scheduled/HTTP turn that never called
85
85
  * reply). 5min (not the old 2s idle-watch) so Thinking isn't misread as idle. */
86
86
  const CANCEL_BTN_IDLE_CHECK_INTERVAL_MS = 5 * 60_000;
87
+ /**
88
+ * How often the cancel button's text is refreshed with elapsed working time.
89
+ *
90
+ * One edit per working instance per interval — at 60s that is trivial for both
91
+ * platforms' rate limits, and it reads as a live counter rather than a stale
92
+ * snapshot. Nothing new is posted, so the channel is never spammed: there is
93
+ * exactly one progress message per turn, and it is the cancel button itself.
94
+ */
95
+ const PROGRESS_UPDATE_INTERVAL_MS = 60_000;
96
+ /** Elapsed time is only shown once work has clearly outlasted a quick answer. */
97
+ const PROGRESS_MIN_ELAPSED_MS = 2 * 60_000;
98
+ /** How much of a tool summary the progress line will show before eliding. */
99
+ const PROGRESS_ACTIVITY_MAX_CHARS = 48;
100
+ /**
101
+ * How long a delivery waits out a disconnected instance IPC before giving up.
102
+ *
103
+ * Sized for a daemon restart (socket close → respawn → CLI ready), which is the
104
+ * event this exists for. Past it the delivery fails loudly as it always did.
105
+ */
106
+ const IPC_RECONNECT_GRACE_MS = 30_000;
107
+ const IPC_RECONNECT_POLL_MS = 250;
108
+ /**
109
+ * Reactions that count as an explicit approve/reject signal, and are therefore worth
110
+ * waking a paused instance for. Everything else is chatter: delivered only if the
111
+ * instance is already awake, never worth a wake-up plus a full agent turn.
112
+ */
113
+ const REACTION_APPROVAL_EMOJIS = new Set(["👍", "👎"]);
87
114
  const CLASSIC_BACKEND_SELECTION_TIMEOUT_MS = 60_000;
88
115
  const CLASSIC_BACKEND_CALLBACK_PREFIX = "classic-backend:";
89
116
  const MODEL_SELECT_CALLBACK_PREFIX = "model-select:";
@@ -94,7 +121,7 @@ export class FleetManager {
94
121
  static sighupHandlerInstalled = false;
95
122
  children = new Map();
96
123
  lifecycle;
97
- /** @deprecated Use lifecycle.daemons — kept for backward compat */
124
+ /** Live view of lifecycle.daemons — used throughout; not deprecated. */
98
125
  get daemons() { return this.lifecycle.daemons; }
99
126
  fleetConfig = null;
100
127
  rawFleetConfig = {};
@@ -152,6 +179,10 @@ export class FleetManager {
152
179
  // reply, on cancel, or when a newer button supersedes it for the same
153
180
  // instance. Per-button tracking means a failed delete never strands a button.
154
181
  cancelButtons = new Map();
182
+ /** instanceName → what it is doing right now, when the backend can tell us. */
183
+ instanceActivity = new Map();
184
+ /** instanceName → tail of deliveries waiting for its IPC to come back. */
185
+ ipcWaitTails = new Map();
155
186
  // Last user message delivered to each instance — used to react ✅ on completion.
156
187
  lastInboundMsg = new Map();
157
188
  topicArchiver;
@@ -178,6 +209,7 @@ export class FleetManager {
178
209
  healthPortRetried = false;
179
210
  updateCheckTimer = null;
180
211
  eventLogPruneTimer = null;
212
+ logRotateTimer = null;
181
213
  /** Days of event/activity history to keep. */
182
214
  static EVENT_LOG_RETENTION_DAYS = 30;
183
215
  watchdogTimer = null;
@@ -487,8 +519,12 @@ export class FleetManager {
487
519
  // warm_cap: a fresh transition into idle may free this instance for eviction,
488
520
  // or (more usefully) reveal that the fleet is now over cap. Only fire on the
489
521
  // edge into idle, not on every idle heartbeat.
490
- if (state === "idle" && previous?.state !== "idle")
522
+ if (state === "idle" && previous?.state !== "idle") {
491
523
  this.enforceWarmCap();
524
+ // The turn is genuinely over — retire the cancel/progress button now rather
525
+ // than waiting for the 5-minute idle backstop to notice.
526
+ this.retireInstanceButtons(name);
527
+ }
492
528
  }
493
529
  cacheInstanceProcessStatus(name, status) {
494
530
  if (status === "running") {
@@ -593,12 +629,70 @@ export class FleetManager {
593
629
  if (!idle) {
594
630
  this.logger.warn({ instanceName, timeoutMs }, "Idle gate timed out; forcing delivery");
595
631
  }
596
- const ipc = this.instanceIpcClients.get(instanceName);
597
- if (!ipc?.connected)
598
- throw new Error(`Instance '${instanceName}' IPC is unavailable`);
599
- ipc.send(payload);
632
+ await this.sendWhenConnected(instanceName, payload);
600
633
  this.lastDeliveryAt.set(instanceName, Date.now());
601
634
  }
635
+ /**
636
+ * Hand a payload to an instance's IPC, waiting out a *transient* disconnect.
637
+ *
638
+ * A daemon that is restarting — `/restart`, crash recovery, a model switch —
639
+ * drops its socket for a few seconds. Any message arriving in that window used
640
+ * to fail instantly: the caller logged a warning, put ❌ on the user's message,
641
+ * and the message was gone. The user had to notice the ❌ and retype it. That is
642
+ * the "instance 訊息不容易掉" goal failing on the most predictable event there is.
643
+ *
644
+ * The wait is bounded. If the instance is genuinely down, this still throws and
645
+ * the ❌ still appears — just for a real failure rather than a restart.
646
+ *
647
+ * Ordering is preserved by serialising behind any waiter already queued for this
648
+ * instance, *including* when the socket happens to be up: otherwise a message
649
+ * arriving after the reconnect could overtake one that has been waiting for it.
650
+ */
651
+ async sendWhenConnected(instanceName, payload) {
652
+ const queued = this.ipcWaitTails.get(instanceName);
653
+ if (!queued) {
654
+ const ipc = this.instanceIpcClients.get(instanceName);
655
+ if (ipc?.connected && ipc.send(payload))
656
+ return;
657
+ }
658
+ const attempt = (queued ?? Promise.resolve())
659
+ .catch(() => { })
660
+ .then(() => this.sendAfterIpcReturns(instanceName, payload));
661
+ // The chain stores a settled-either-way promise so one failed delivery cannot
662
+ // wedge every later one, and so `queued` above is safe to await unguarded.
663
+ const tail = attempt.catch(() => { });
664
+ this.ipcWaitTails.set(instanceName, tail);
665
+ try {
666
+ await attempt;
667
+ }
668
+ finally {
669
+ // Only the last waiter clears the chain; while a queue is still draining the
670
+ // map must keep pointing at it or ordering is lost.
671
+ if (this.ipcWaitTails.get(instanceName) === tail) {
672
+ this.ipcWaitTails.delete(instanceName);
673
+ }
674
+ }
675
+ }
676
+ /** Poll for the instance's IPC to come back, then send. Throws if it does not. */
677
+ async sendAfterIpcReturns(instanceName, payload) {
678
+ const deadline = Date.now() + IPC_RECONNECT_GRACE_MS;
679
+ let warned = false;
680
+ for (;;) {
681
+ // Re-read every round: a reconnect replaces the IpcClient object entirely,
682
+ // so a cached reference would stay dead forever.
683
+ const ipc = this.instanceIpcClients.get(instanceName);
684
+ if (ipc?.connected && ipc.send(payload))
685
+ return;
686
+ if (Date.now() >= deadline) {
687
+ throw new Error(`Instance '${instanceName}' IPC is unavailable`);
688
+ }
689
+ if (!warned) {
690
+ warned = true;
691
+ this.logger.info({ instanceName }, "Instance IPC is down — holding delivery until it reconnects");
692
+ }
693
+ await new Promise(resolve => setTimeout(resolve, IPC_RECONNECT_POLL_MS));
694
+ }
695
+ }
602
696
  /** Single delivery facade: wake paused CLIs and serialize non-user work behind idle. */
603
697
  async deliverToInstance(instanceName, payload, options = {}) {
604
698
  const meta = payload.meta && typeof payload.meta === "object"
@@ -614,10 +708,7 @@ export class FleetManager {
614
708
  await this.lifecycle.wake(instanceName, 30_000);
615
709
  this.enforceWarmCap(instanceName); // woke one → evict a different LRU idle if over cap
616
710
  }
617
- const ipc = this.instanceIpcClients.get(instanceName);
618
- if (!ipc?.connected)
619
- throw new Error(`Instance '${instanceName}' IPC is unavailable`);
620
- ipc.send(payload);
711
+ await this.sendWhenConnected(instanceName, payload);
621
712
  // A cross-instance item arriving before the daemon observes this turn as
622
713
  // working must not trust the stale idle snapshot from before the send.
623
714
  this.lastDeliveryAt.set(instanceName, Date.now());
@@ -863,7 +954,6 @@ export class FleetManager {
863
954
  this.configPath = configPath;
864
955
  this.loadEnvFile();
865
956
  // Rotate fleet.log if oversized (before any logging)
866
- const { rotateLogIfNeeded } = await import("./logger.js");
867
957
  rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
868
958
  const fleet = this.loadConfig(configPath);
869
959
  setLocale(detectLocale(fleet)); // user-facing text language (fleet.yaml defaults.locale / timezone)
@@ -996,11 +1086,10 @@ export class FleetManager {
996
1086
  // Rotate fleet.log daily too (besides the startup size check above), so a
997
1087
  // long-running fleet doesn't accumulate an unbounded log.
998
1088
  rotateLogIfNeeded(join(this.dataDir, "fleet.log"));
999
- // Instance output.log is pipe-pane (TUI ANSI). Daemon health ticks also
1000
- // rotate it; this is the daily safety net for idle/stopped instances.
1001
- for (const name of Object.keys(this.fleetConfig?.instances ?? {})) {
1002
- rotateLogIfNeeded(join(this.dataDir, "instances", name, "output.log"));
1003
- }
1089
+ // Instance output.log is pipe-pane (TUI ANSI). Daemon health ticks rotate a
1090
+ // running instance's own log; this sweep is the safety net for every other
1091
+ // kind. One implementation, so the two cannot cover different sets.
1092
+ this.rotateAllInstanceLogs();
1004
1093
  }, () => {
1005
1094
  const instances = Object.keys(this.fleetConfig?.instances ?? {});
1006
1095
  const costMap = new Map();
@@ -1157,6 +1246,12 @@ export class FleetManager {
1157
1246
  this.pruneEventLog();
1158
1247
  this.eventLogPruneTimer = setInterval(() => this.pruneEventLog(), 24 * 60 * 60_000);
1159
1248
  this.eventLogPruneTimer.unref?.();
1249
+ // Same shape for pipe-pane logs, and for the same reason: the only sweep that
1250
+ // covered them lived inside the daily-summary callback, so it did not run at
1251
+ // all when summaries were off.
1252
+ this.rotateAllInstanceLogs();
1253
+ this.logRotateTimer = setInterval(() => this.rotateAllInstanceLogs(), 24 * 60 * 60_000);
1254
+ this.logRotateTimer.unref?.();
1160
1255
  // Phase 2: Start remaining instances with staggered concurrency
1161
1256
  if (others.length > 0) {
1162
1257
  await this.startInstancesWithConcurrency(others, topicMode);
@@ -1503,6 +1598,9 @@ export class FleetManager {
1503
1598
  this.adapter.on("message", safeHandler(async (msg) => {
1504
1599
  await this.handleInboundMessage(msg);
1505
1600
  }, this.logger, "adapter.message"));
1601
+ this.adapter.on("reaction", safeHandler(async (r) => {
1602
+ await this.handleInboundReaction(r);
1603
+ }, this.logger, "adapter.reaction"));
1506
1604
  this.adapter.on("callback_query", safeHandler(async (data) => {
1507
1605
  if (await this.handleClassicBackendSelection(data))
1508
1606
  return;
@@ -1788,6 +1886,9 @@ export class FleetManager {
1788
1886
  adapter.on("message", safeHandler(async (msg) => {
1789
1887
  await this.handleInboundMessage(msg);
1790
1888
  }, this.logger, `adapter[${adapterId}].message`));
1889
+ adapter.on("reaction", safeHandler(async (r) => {
1890
+ await this.handleInboundReaction(r);
1891
+ }, this.logger, `adapter[${adapterId}].reaction`));
1791
1892
  adapter.on("callback_query", safeHandler(async (data) => {
1792
1893
  if (await this.handleClassicBackendSelection(data))
1793
1894
  return;
@@ -2107,6 +2208,9 @@ export class FleetManager {
2107
2208
  else if (msg.type === "instance_process_state") {
2108
2209
  this.cacheInstanceProcessStatus(name, msg.status);
2109
2210
  }
2211
+ else if (msg.type === "instance_activity") {
2212
+ this.cacheInstanceActivity(name, msg.activity);
2213
+ }
2110
2214
  else if (msg.type === "instance_state" || msg.type === "instance_state_response") {
2111
2215
  this.cacheInstanceExecutionState(name, msg);
2112
2216
  if (msg.type === "instance_state_response") {
@@ -2276,6 +2380,54 @@ export class FleetManager {
2276
2380
  }
2277
2381
  return generals[0];
2278
2382
  }
2383
+ /**
2384
+ * A user reacted to one of the bot's messages (#408).
2385
+ *
2386
+ * Delivered as a normal inbound so it reuses routing, dedup, the idle gate and
2387
+ * delivery confirmation — nothing new is needed on that path. `message_id`
2388
+ * identifies the bot message that was reacted to, and since the inbound block now
2389
+ * renders message_id, the agent can tell WHICH of its messages this refers to.
2390
+ *
2391
+ * Policy: only the approval emojis wake an instance. A 👍 costing a full agent
2392
+ * turn is acceptable when it means "approved"; every other emoji is chatter and
2393
+ * must not spend a turn, so those are delivered only to an instance that is
2394
+ * already awake (`no_wake`) and dropped for a paused one.
2395
+ */
2396
+ async handleInboundReaction(r) {
2397
+ const instanceName = this.resolveSlashTarget(r.threadId ?? r.chatId, r.adapterId);
2398
+ if (!instanceName) {
2399
+ this.logger.debug({ emoji: r.emoji, chatId: r.chatId }, "Reaction in an unrouted channel — ignoring");
2400
+ return;
2401
+ }
2402
+ const isApproval = REACTION_APPROVAL_EMOJIS.has(r.emoji);
2403
+ const verb = r.action === "add" ? "reacted" : "removed their reaction";
2404
+ const content = `[reaction:${r.emoji}] ${r.username} ${verb} on your message ${r.messageId}`;
2405
+ this.eventLog?.logActivity("reaction", r.username, `${r.emoji} ${r.action}`, instanceName);
2406
+ try {
2407
+ await this.deliverToInstance(instanceName, {
2408
+ type: "fleet_inbound",
2409
+ content,
2410
+ meta: {
2411
+ chat_id: "", // not a chat message — must not become the reply target
2412
+ thread_id: "",
2413
+ message_id: r.messageId,
2414
+ user: r.username,
2415
+ user_id: r.userId,
2416
+ source: r.source,
2417
+ ts: r.timestamp.toISOString(),
2418
+ request_kind: "update",
2419
+ requires_reply: "false",
2420
+ // Non-approval reactions never wake a paused instance.
2421
+ ...(isApproval ? {} : { no_wake: "true" }),
2422
+ },
2423
+ });
2424
+ }
2425
+ catch (err) {
2426
+ // A reaction is a nice-to-have signal; failing to deliver one must not be
2427
+ // noisy. The user can always say it in words.
2428
+ this.logger.debug({ err, instanceName, emoji: r.emoji }, "Reaction delivery failed");
2429
+ }
2430
+ }
2279
2431
  async handleInboundMessage(msg) {
2280
2432
  const threadId = msg.threadId || undefined;
2281
2433
  this.logger.debug({ source: msg.source, chatId: msg.chatId, threadId, userId: msg.userId, isBotMessage: msg.isBotMessage, textLen: (msg.text ?? "").length, text: (msg.text ?? "").slice(0, 80) }, "handleInboundMessage entry");
@@ -2815,8 +2967,18 @@ export class FleetManager {
2815
2967
  // Route standard channel tools (reply, react, edit_message, download_attachment)
2816
2968
  if (routeToolCall(outAdapter, tool, args, threadId, respond)) {
2817
2969
  if (tool === "reply") {
2818
- // Agent answered — retire its pending cancel button and mark ✅ done.
2819
- this.clearCancelButton(instanceName);
2970
+ // A reply is NOT proof the turn is over: on multi-step work an agent
2971
+ // replies ("starting…") and keeps going for many minutes. Retiring the
2972
+ // button here left the channel looking idle with no way to cancel and no
2973
+ // sign anything was happening (#410). Idle state owns retirement now; if the
2974
+ // instance is still working, move the button below the new reply so it stays
2975
+ // the last thing in the channel.
2976
+ if (this.getInstanceIdle(instanceName)) {
2977
+ this.clearCancelButton(instanceName);
2978
+ }
2979
+ else {
2980
+ void this.sendCancelButton(instanceName);
2981
+ }
2820
2982
  this.reactDone(instanceName);
2821
2983
  const replyTo = this.lastInboundUser.get(instanceName) ?? "user";
2822
2984
  this.logger.info(`${instanceName} → ${replyTo}: ${(args.text ?? "").slice(0, 100)}`);
@@ -3598,6 +3760,41 @@ export class FleetManager {
3598
3760
  }
3599
3761
  }
3600
3762
  /** Drop event/activity rows older than the retention window. Best-effort. */
3763
+ /**
3764
+ * Cap every instance's pipe-pane log, walking the instances **directory** rather
3765
+ * than the config.
3766
+ *
3767
+ * A running instance rotates its own log on each health tick, so the ones that
3768
+ * need this are the ones nothing else looks at:
3769
+ *
3770
+ * - deleted instances, whose directory outlives the config entry. Nothing ever
3771
+ * touched these again. On the machine this was found on, one held 122 MB and
3772
+ * another 74 MB, out of 622 MB of pipe-pane logs in total.
3773
+ * - classic instances, which live in classicChannels, not fleetConfig.instances,
3774
+ * and so were never in the old config-driven loop at all.
3775
+ * - stopped instances, which have no health tick running.
3776
+ *
3777
+ * pipe-pane writes raw TUI output, so a wedged splash screen can emit ANSI frames
3778
+ * at animation rate. Unbounded growth here fills the disk, which takes the whole
3779
+ * fleet down rather than one instance.
3780
+ */
3781
+ rotateAllInstanceLogs() {
3782
+ const root = join(this.dataDir, "instances");
3783
+ let entries;
3784
+ try {
3785
+ entries = readdirSync(root, { withFileTypes: true });
3786
+ }
3787
+ catch {
3788
+ return; // no instances directory yet
3789
+ }
3790
+ for (const entry of entries) {
3791
+ if (!entry.isDirectory())
3792
+ continue;
3793
+ // rotateLogIfNeeded is already best-effort and returns early on a missing
3794
+ // file, so a directory without a pipe-pane log costs one stat.
3795
+ rotateLogIfNeeded(join(root, entry.name, "output.log"));
3796
+ }
3797
+ }
3601
3798
  pruneEventLog() {
3602
3799
  try {
3603
3800
  this.eventLog?.prune(FleetManager.EVENT_LOG_RETENTION_DAYS);
@@ -3814,7 +4011,12 @@ export class FleetManager {
3814
4011
  threadId: sent.threadId ?? threadId,
3815
4012
  correlationId,
3816
4013
  retryCount: 0,
4014
+ // Elapsed time is measured from when this button was posted — i.e. from
4015
+ // when the work was handed over — not from the pane's working transition,
4016
+ // which resets if the CLI blips idle mid-turn.
4017
+ startedAt: Date.now(),
3817
4018
  };
4019
+ this.startProgressTicker(entry);
3818
4020
  // Idle-check backstop: every 5min, if the instance is idle, retire the
3819
4021
  // button. Covers turns that end without hitting a clear trigger (reply /
3820
4022
  // cancel / correlation). Cleared in discardButton when the entry is removed.
@@ -3835,6 +4037,97 @@ export class FleetManager {
3835
4037
  this.logger.warn({ err: e.message, instanceName }, "Failed to send cancel button");
3836
4038
  }
3837
4039
  }
4040
+ /**
4041
+ * The cancel button's text for a given elapsed time.
4042
+ *
4043
+ * Below the threshold it keeps the original wording, so a normal quick answer
4044
+ * looks exactly as it did before. Past it, the button doubles as the live
4045
+ * progress indicator (#409) — the channel showed nothing at all during long work,
4046
+ * and once the agent had replied once there was no sign it was still going.
4047
+ */
4048
+ static progressText(elapsedMs, activity) {
4049
+ if (elapsedMs < PROGRESS_MIN_ELAPSED_MS)
4050
+ return "👀 處理中…";
4051
+ const totalSeconds = Math.floor(elapsedMs / 1000);
4052
+ const minutes = Math.floor(totalSeconds / 60);
4053
+ const seconds = totalSeconds % 60;
4054
+ const elapsed = minutes >= 60
4055
+ ? `${Math.floor(minutes / 60)}h ${minutes % 60}m`
4056
+ : `${minutes}m ${String(seconds).padStart(2, "0")}s`;
4057
+ const detail = FleetManager.sanitizeActivity(activity);
4058
+ return detail
4059
+ ? `⏳ 處理中… (已進行 ${elapsed} · ${detail})`
4060
+ : `⏳ 處理中… (已進行 ${elapsed})`;
4061
+ }
4062
+ /**
4063
+ * Make a tool summary safe to paste into a channel message.
4064
+ *
4065
+ * The text is agent-controlled (it is built from tool inputs — file paths,
4066
+ * shell commands), so it gets flattened to one line, capped, and stripped of
4067
+ * the two Discord mass-mention triggers. Neither channel renders it with a
4068
+ * parse mode, so no markup escaping is needed beyond that.
4069
+ */
4070
+ static sanitizeActivity(activity) {
4071
+ if (!activity)
4072
+ return null;
4073
+ const flat = activity.replace(/\s+/g, " ").replace(/@(everyone|here)/g, "@​$1").trim();
4074
+ if (!flat)
4075
+ return null;
4076
+ return flat.length > PROGRESS_ACTIVITY_MAX_CHARS
4077
+ ? `${flat.slice(0, PROGRESS_ACTIVITY_MAX_CHARS - 1)}…`
4078
+ : flat;
4079
+ }
4080
+ /**
4081
+ * Remember what an instance is currently doing, for the progress line.
4082
+ *
4083
+ * Best-effort by design: only backends that expose a live activity feed report
4084
+ * anything, and the progress line simply omits the detail for the rest. It is
4085
+ * never used to decide anything — purely what the user is shown.
4086
+ */
4087
+ cacheInstanceActivity(name, activity) {
4088
+ if (activity)
4089
+ this.instanceActivity.set(name, activity);
4090
+ else
4091
+ this.instanceActivity.delete(name);
4092
+ }
4093
+ /**
4094
+ * Refresh the button's text in place while the instance keeps working.
4095
+ *
4096
+ * Uses `editAlert`, NOT `editMessage`: on Telegram the latter omits reply_markup,
4097
+ * and the Bot API treats that as "clear the keyboard" — so editing with it would
4098
+ * delete the very cancel button this is trying to keep alive.
4099
+ */
4100
+ startProgressTicker(entry) {
4101
+ entry.progressTimer = setInterval(() => {
4102
+ if (!this.cancelButtons.has(entry.messageId)) {
4103
+ clearInterval(entry.progressTimer);
4104
+ return;
4105
+ }
4106
+ // Idle means the turn ended; the idle-edge handler retires the button.
4107
+ if (this.getInstanceIdle(entry.instanceName))
4108
+ return;
4109
+ const text = FleetManager.progressText(Date.now() - (entry.startedAt ?? Date.now()), this.instanceActivity.get(entry.instanceName));
4110
+ if (text === entry.lastProgressText)
4111
+ return; // nothing changed — skip the API call
4112
+ const adapter = this.getAdapterForInstance(entry.instanceName) ?? this.adapter;
4113
+ if (!adapter?.editAlert)
4114
+ return;
4115
+ entry.lastProgressText = text;
4116
+ adapter.editAlert(entry.chatId, entry.messageId, {
4117
+ type: "cancel",
4118
+ instanceName: entry.instanceName,
4119
+ message: text,
4120
+ choices: [{ id: `cancel:${entry.instanceName}`, label: t("cancel.button") }],
4121
+ }, entry.threadId ? { threadId: entry.threadId } : undefined)
4122
+ .catch(err => {
4123
+ // A failed progress edit must never escalate: the button still works and
4124
+ // the next tick retries. Common causes are a deleted message or a
4125
+ // rate limit.
4126
+ this.logger.debug({ err, instanceName: entry.instanceName }, "Progress edit failed");
4127
+ });
4128
+ }, PROGRESS_UPDATE_INTERVAL_MS);
4129
+ entry.progressTimer.unref?.();
4130
+ }
3838
4131
  /** Retire (delete) every cancel button belonging to an instance. */
3839
4132
  retireInstanceButtons(instanceName) {
3840
4133
  // Snapshot first — retireButton may delete entries from the map on success.
@@ -3866,6 +4159,8 @@ export class FleetManager {
3866
4159
  clearTimeout(entry.retryTimer);
3867
4160
  if (entry.idleCheckTimer)
3868
4161
  clearInterval(entry.idleCheckTimer);
4162
+ if (entry.progressTimer)
4163
+ clearInterval(entry.progressTimer);
3869
4164
  this.cancelButtons.delete(entry.messageId);
3870
4165
  }
3871
4166
  /** Re-attempt a failed button delete up to CANCEL_BTN_MAX_RETRIES times. */
@@ -5145,6 +5440,22 @@ When users create specialized instances, suggest these configurations:
5145
5440
  clearInterval(this.eventLogPruneTimer);
5146
5441
  this.eventLogPruneTimer = null;
5147
5442
  }
5443
+ if (this.logRotateTimer) {
5444
+ clearInterval(this.logRotateTimer);
5445
+ this.logRotateTimer = null;
5446
+ }
5447
+ // Cancel-button timers were never cleared here. The idle-check interval is not
5448
+ // unref'd, so it held the event loop open past shutdown and kept retrying
5449
+ // deletes against an adapter that was already gone.
5450
+ for (const entry of [...this.cancelButtons.values()]) {
5451
+ if (entry.retryTimer)
5452
+ clearTimeout(entry.retryTimer);
5453
+ if (entry.idleCheckTimer)
5454
+ clearInterval(entry.idleCheckTimer);
5455
+ if (entry.progressTimer)
5456
+ clearInterval(entry.progressTimer);
5457
+ }
5458
+ this.cancelButtons.clear();
5148
5459
  if (this.topicCleanupTimer) {
5149
5460
  clearInterval(this.topicCleanupTimer);
5150
5461
  this.topicCleanupTimer = null;