talon-agent 3.23.1 → 3.24.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.23.1",
3
+ "version": "3.24.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -18,6 +18,10 @@ type ErrorReason =
18
18
  | "bad_request"
19
19
  | "forbidden"
20
20
  | "telegram_api"
21
+ // The user stopped the turn (/stop, talon kill). Not a fault: the
22
+ // frontends deliver nothing for it — the stop acknowledgement already
23
+ // told the user what happened.
24
+ | "stopped"
21
25
  | "unknown";
22
26
 
23
27
  /**
@@ -342,6 +346,7 @@ const FRIENDLY_MESSAGES: Record<ErrorReason, string> = {
342
346
  bad_request: "Something went wrong. Try /reset if this keeps happening.",
343
347
  forbidden: "Permission denied for this action.",
344
348
  telegram_api: "Telegram API error. Try again.",
349
+ stopped: "⏹ Stopped.",
345
350
  unknown: "Something went wrong. Try again or /reset.",
346
351
  };
347
352
 
@@ -24,6 +24,7 @@ import type { ContextManager, ExecuteParams, ExecuteResult } from "../types.js";
24
24
  import { bus } from "../bus/index.js";
25
25
  import { taskTable, type TaskHandle } from "../tasks/index.js";
26
26
  import { log, logDebug, logWarn } from "../../util/log.js";
27
+ import { TalonError } from "../errors.js";
27
28
  import { Loom } from "./loom.js";
28
29
  import { carryTurnEvents } from "./shuttle.js";
29
30
  import type { Thread, ThreadSnapshot } from "./thread.js";
@@ -141,6 +142,16 @@ export class Weaver {
141
142
  return result;
142
143
  } catch (err) {
143
144
  task.fail(err);
145
+ if (lifecycle.killed) {
146
+ // The backend didn't manage a clean interrupt-completion (some
147
+ // SDK versions surface an interrupted turn as an error result).
148
+ // The user asked for this outcome — don't let it unwind as a
149
+ // fault the frontend then reports to the chat.
150
+ throw new TalonError("Turn stopped by user", {
151
+ reason: "stopped",
152
+ cause: err,
153
+ });
154
+ }
144
155
  throw err;
145
156
  } finally {
146
157
  this.activeCount--;
@@ -108,6 +108,7 @@ export async function handleStatus(
108
108
  `**Workspace** ${formatBytes(s.diskBytes)}`,
109
109
  `**Session** ${s.sessionName ? `"${s.sessionName}" ` : ""}${s.sessionId ? "`" + s.sessionId.slice(0, 8) + "...`" : "_(new)_"} · ${s.sessionAge} old`,
110
110
  `**Uptime** ${s.uptime} · ${s.activeSessionCount} active session${s.activeSessionCount === 1 ? "" : "s"}`,
111
+ `**Runtime** ${s.runtime} · ${formatBytes(s.rssBytes)} RSS`,
111
112
  ];
112
113
  await i.editReply(lines.join("\n"));
113
114
  }
@@ -93,6 +93,13 @@ async function flushQueue(chatId: string): Promise<void> {
93
93
  recordMessageProcessed();
94
94
  } catch (err) {
95
95
  const classified = classify(err);
96
+ // A user-initiated /stop is an outcome, not a fault — the stop command
97
+ // already acknowledged it; don't re-report it as an error.
98
+ if (classified.reason === "stopped") {
99
+ log("bot", `[${chatId}] turn stopped by user`);
100
+ recordMessageSettled();
101
+ return;
102
+ }
96
103
  const promptPreview = combinedPrompt.slice(0, 100).replace(/\n/g, " ");
97
104
  logError(
98
105
  "bot",
@@ -88,6 +88,10 @@ export interface SessionStatusData {
88
88
  sessionId: string | undefined;
89
89
  uptime: string;
90
90
  activeSessionCount: number;
91
+ /** Runtime name + version, e.g. "Bun 1.3.9" or "Node 24.4.0". */
92
+ runtime: string;
93
+ /** Daemon resident set size, in bytes. */
94
+ rssBytes: number;
91
95
  }
92
96
 
93
97
  /**
@@ -202,5 +206,9 @@ export async function collectSessionStatus(
202
206
  sessionId: info.sessionId,
203
207
  uptime: formatDuration(process.uptime() * 1000),
204
208
  activeSessionCount: getActiveSessionCount(),
209
+ runtime: process.versions.bun
210
+ ? `Bun ${process.versions.bun}`
211
+ : `Node ${process.versions.node}`,
212
+ rssBytes: process.memoryUsage().rss,
205
213
  };
206
214
  }
@@ -8,6 +8,7 @@
8
8
 
9
9
  import type { Bot } from "grammy";
10
10
  import { respawnSelf } from "../../../util/respawn.js";
11
+ import { isStaleCommand } from "../stale-command.js";
11
12
  import {
12
13
  getRepoRoot,
13
14
  runSelfUpdate,
@@ -154,6 +155,10 @@ export function registerAdminCommands(
154
155
  await ctx.reply("Not authorized.");
155
156
  return;
156
157
  }
158
+ // A restart that predates this process is a redelivery (or an order
159
+ // aimed at a daemon that is already gone) — obeying it would restart
160
+ // us again, and again, on every boot.
161
+ if (isStaleCommand(ctx.message?.date, "/restart")) return;
157
162
  await ctx.reply("♻️ Restarting...");
158
163
  respawnSelf("telegram /restart");
159
164
  });
@@ -169,6 +174,8 @@ export function registerAdminCommands(
169
174
  await ctx.reply("Not authorized.");
170
175
  return;
171
176
  }
177
+ // Same redelivery hazard as /restart — it also ends the process.
178
+ if (isStaleCommand(ctx.message?.date, "/update")) return;
172
179
  const remote = config.update?.remote ?? "origin";
173
180
  const branch = config.update?.branch ?? "main";
174
181
  const sent = await ctx.reply(
@@ -90,6 +90,7 @@ export function registerSessionCommands(
90
90
  `<b>Workspace</b> ${formatBytes(s.diskBytes)}`,
91
91
  `<b>Session</b> ${s.sessionName ? `"${escapeHtml(s.sessionName)}" ` : ""}${s.sessionId ? "<code>" + escapeHtml(s.sessionId.slice(0, 8)) + "...</code>" : "<i>(new)</i>"} · ${s.sessionAge} old`,
92
92
  `<b>Uptime</b> ${s.uptime} · ${s.activeSessionCount} active session${s.activeSessionCount === 1 ? "" : "s"}`,
93
+ `<b>Runtime</b> ${escapeHtml(s.runtime)} · ${formatBytes(s.rssBytes)} RSS`,
93
94
  ];
94
95
  await ctx.reply(lines.join("\n"), { parse_mode: "HTML" });
95
96
  });
@@ -188,6 +188,15 @@ async function flushQueue(chatId: string): Promise<void> {
188
188
  recordMessageProcessed();
189
189
  } catch (err) {
190
190
  const classified = classify(err);
191
+ // A user-initiated /stop is an outcome, not a fault: the stop command
192
+ // already acknowledged it, so delivering the unwound turn's error here
193
+ // would just contradict that with noise.
194
+ if (classified.reason === "stopped") {
195
+ log("bot", `[${chatId}] turn stopped by user`);
196
+ lastHandledMessageIdByChat.set(chatId, last.messageId);
197
+ recordMessageSettled();
198
+ return;
199
+ }
191
200
  const chatType = last.isGroup ? "group" : "DM";
192
201
  const promptPreview = promptWithContext.slice(0, 100).replace(/\n/g, " ");
193
202
  logError(
@@ -23,6 +23,7 @@ import {
23
23
  } from "./commands/index.js";
24
24
  import { setAccessControl } from "./handlers/index.js";
25
25
  import { registerMiddleware } from "./middleware.js";
26
+ import { confirmUpdates } from "./update-offset.js";
26
27
  import { registerCallbacks } from "./callbacks/index.js";
27
28
  import { log, logError } from "../../util/log.js";
28
29
 
@@ -142,6 +143,10 @@ export function createTelegramFrontend(
142
143
  async stop() {
143
144
  try {
144
145
  await bot.stop();
146
+ // grammY advances the update offset on its NEXT poll, which never
147
+ // comes once we are shutting down — so confirm it explicitly or
148
+ // Telegram redelivers the command that triggered this shutdown.
149
+ await confirmUpdates(bot);
145
150
  log("shutdown", "Bot disconnected");
146
151
  } catch (err) {
147
152
  logError("shutdown", "Bot stop error", err);
@@ -10,6 +10,7 @@ import { allowChat, revokeChat } from "./userbot.js";
10
10
  import { registerChat } from "../../core/background/pulse.js";
11
11
  import { log } from "../../util/log.js";
12
12
  import { getSenderName } from "./handlers/index.js";
13
+ import { noteUpdateId } from "./update-offset.js";
13
14
  import { noteInboundThread } from "./topics.js";
14
15
  import { recordJoinRequest } from "./join-requests.js";
15
16
  import { newlyAddedEmojis, recordReactionToBot } from "../../core/soul/taps.js";
@@ -26,6 +27,15 @@ import {
26
27
  } from "./handlers/index.js";
27
28
 
28
29
  export function registerMiddleware(bot: Bot, config: TalonConfig): void {
30
+ // ── Update-offset tracking (every update, before anything else) ──────────
31
+ // Telegram redelivers any update whose id was never confirmed; the
32
+ // shutdown path confirms this one so a process-ending command can't be
33
+ // served twice. See update-offset.ts.
34
+ bot.use((ctx, next) => {
35
+ noteUpdateId(ctx.update.update_id);
36
+ return next();
37
+ });
38
+
29
39
  // ── History capture (runs for ALL messages, before handlers) ─────────────
30
40
  bot.on("message", (ctx, next) => {
31
41
  const chatId = String(ctx.chat.id);
@@ -0,0 +1,46 @@
1
+ /**
2
+ * Staleness guard for process-ending commands.
3
+ *
4
+ * Second layer under `update-offset.ts`. Confirming the offset stops the
5
+ * redelivery that caused the restart loop, but it can't be guaranteed —
6
+ * the confirmation is a network call on the way out, and a SIGKILL, a
7
+ * crash, or a 409 from two pollers skips it entirely. Any command that
8
+ * ends the process is therefore also checked for age: it must be newer
9
+ * than this process, because a `/restart` issued before we booted has
10
+ * either already been served (redelivery) or refers to a daemon that is
11
+ * no longer running.
12
+ *
13
+ * Only self-terminating commands are gated. Ordinary messages sent while
14
+ * the daemon was down are real work and must still be processed.
15
+ */
16
+
17
+ import { logWarn } from "../../util/log.js";
18
+
19
+ /**
20
+ * Grace window for clock skew between Telegram's timestamps and ours,
21
+ * and for a command issued in the seconds around a boot.
22
+ */
23
+ const CLOCK_SKEW_GRACE_MS = 30_000;
24
+
25
+ /** When this process started — the cutoff a fresh command must beat. */
26
+ const PROCESS_START_MS = Date.now();
27
+
28
+ /**
29
+ * True when a command predates this process and must not be re-executed.
30
+ * `messageDate` is Telegram's `message.date` (seconds since epoch).
31
+ */
32
+ export function isStaleCommand(
33
+ messageDate: number | undefined,
34
+ command: string,
35
+ ): boolean {
36
+ if (!messageDate) return false; // no timestamp — treat as live
37
+ const sentAtMs = messageDate * 1000;
38
+ if (sentAtMs >= PROCESS_START_MS - CLOCK_SKEW_GRACE_MS) return false;
39
+ const ageSec = Math.round((Date.now() - sentAtMs) / 1000);
40
+ logWarn(
41
+ "bot",
42
+ `Ignoring stale ${command} from ${ageSec}s ago — it predates this process ` +
43
+ `(a redelivered restart would loop the daemon).`,
44
+ );
45
+ return true;
46
+ }
@@ -0,0 +1,58 @@
1
+ /**
2
+ * Update-offset confirmation — the reason a `/restart` used to run twice.
3
+ *
4
+ * Telegram's getUpdates is at-least-once: an update stays queued until a
5
+ * LATER call passes `offset = update_id + 1`. grammY advances that offset
6
+ * on its next poll, so a command that ends the process — `/restart`,
7
+ * `/update` — exits before the confirmation is ever sent. Telegram then
8
+ * redelivers it to the successor, which restarts, which never confirms
9
+ * either: a boot loop that survives every restart, observed live on
10
+ * 2026-08-22 taking the daemon down four times in a row.
11
+ *
12
+ * The fix is to confirm explicitly before exiting. This module tracks the
13
+ * highest update_id seen and `confirmUpdates` acknowledges it with a
14
+ * zero-timeout getUpdates, which is exactly what grammY's next poll would
15
+ * have done.
16
+ */
17
+
18
+ import type { Bot } from "grammy";
19
+ import { log, logWarn } from "../../util/log.js";
20
+
21
+ let highestUpdateId = 0;
22
+
23
+ /** Record an update as seen. Called for every update grammY dispatches. */
24
+ export function noteUpdateId(updateId: number): void {
25
+ if (updateId > highestUpdateId) highestUpdateId = updateId;
26
+ }
27
+
28
+ /** Highest update id seen this process (0 when none). Test seam. */
29
+ export function lastUpdateId(): number {
30
+ return highestUpdateId;
31
+ }
32
+
33
+ /** Test seam: forget the tracked offset. */
34
+ export function resetUpdateOffset(): void {
35
+ highestUpdateId = 0;
36
+ }
37
+
38
+ /**
39
+ * Tell Telegram every update seen so far is handled, so none of them is
40
+ * redelivered to the next process. Best-effort: this runs on the
41
+ * shutdown path, where a failure must never block exit.
42
+ */
43
+ export async function confirmUpdates(bot: Bot): Promise<void> {
44
+ if (highestUpdateId === 0) return;
45
+ try {
46
+ await bot.api.getUpdates({
47
+ offset: highestUpdateId + 1,
48
+ limit: 1,
49
+ timeout: 0,
50
+ });
51
+ log("shutdown", `Confirmed Telegram updates through ${highestUpdateId}`);
52
+ } catch (err) {
53
+ logWarn(
54
+ "shutdown",
55
+ `Could not confirm Telegram update offset: ${err instanceof Error ? err.message : err}`,
56
+ );
57
+ }
58
+ }