talon-agent 3.15.0 → 3.15.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/package.json +1 -1
  2. package/src/app.ts +24 -10
  3. package/src/backend/kilo/factory.ts +26 -0
  4. package/src/backend/kilo/handler/message.ts +12 -6
  5. package/src/backend/kilo/handler/turn.ts +43 -40
  6. package/src/backend/kilo/index.ts +7 -1
  7. package/src/backend/kilo/server.ts +28 -32
  8. package/src/backend/kilo/sessions.ts +17 -0
  9. package/src/backend/opencode/factory.ts +26 -0
  10. package/src/backend/opencode/handler/message.ts +13 -5
  11. package/src/backend/opencode/handler/turn.ts +41 -32
  12. package/src/backend/opencode/index.ts +7 -1
  13. package/src/backend/opencode/server.ts +27 -16
  14. package/src/backend/opencode/sessions.ts +17 -0
  15. package/src/backend/remote-server/client.ts +1 -1
  16. package/src/backend/remote-server/events.ts +19 -6
  17. package/src/backend/remote-server/index.ts +16 -1
  18. package/src/backend/remote-server/lifecycle.ts +5 -1
  19. package/src/backend/remote-server/mcp.ts +196 -110
  20. package/src/backend/remote-server/one-shot.ts +53 -6
  21. package/src/backend/remote-server/providers.ts +12 -0
  22. package/src/backend/remote-server/session-helpers.ts +69 -0
  23. package/src/backend/remote-server/sessions.ts +67 -7
  24. package/src/backend/remote-server/sse-stream.ts +17 -10
  25. package/src/backend/remote-server/state.ts +6 -0
  26. package/src/backend/remote-server/turn-timeout.ts +71 -0
  27. package/src/core/mcp-hub/index.ts +22 -0
  28. package/src/frontend/telegram/admin.ts +3 -1
  29. package/src/frontend/telegram/callbacks/model.ts +48 -0
  30. package/src/frontend/telegram/commands/info.ts +9 -3
  31. package/src/frontend/telegram/commands/settings.ts +8 -2
  32. package/src/frontend/telegram/handlers/queue.ts +10 -4
  33. package/src/frontend/telegram/helpers/menu.ts +5 -2
  34. package/src/util/respawn.ts +66 -38
@@ -17,7 +17,12 @@ import {
17
17
  import { log, logWarn } from "../../util/log.js";
18
18
  import type { RemoteAgentClient, RemotePermissionRule } from "./client.js";
19
19
  import type { RemoteServerState } from "./state.js";
20
- import { TALON_MCP_SERVER_NAME, getChatMcpServerName } from "./mcp.js";
20
+ import {
21
+ TALON_MCP_SERVER_NAME,
22
+ TALON_PLUGIN_MCP_SERVER_NAME,
23
+ getChatMcpServerName,
24
+ safeMcpNamePart,
25
+ } from "./mcp.js";
21
26
 
22
27
  /**
23
28
  * Build the per-session permission ruleset Talon installs on every fresh
@@ -33,14 +38,13 @@ import { TALON_MCP_SERVER_NAME, getChatMcpServerName } from "./mcp.js";
33
38
  * "No active chat context". Or in the cross-chat case where chat B
34
39
  * IS active, the model in chat A could leak content into chat B.
35
40
  * Deny pattern blocks both. (Visibility is also blocked at the
36
- * MCP-registration layer via `ensureChatMcpServer`; this rule is
37
- * defense in depth.)
41
+ * prompt layer by `buildToolOverrides`; this rule is defense in depth.)
38
42
  *
39
- * 2. Auto-allow built-in tools (`tool *`, `edit *`, `bash *`) so they
43
+ * 2. Auto-allow built-in tools (`tool *`, `edit *`, `bash *`) and
44
+ * workspace paths outside the server process's launch directory so they
40
45
  * don't sit in `permission.asked` waiting for a reply that never
41
- * arrives — Talon's question watchdog only handles `question.*`
42
- * events, not `permission.*`. Without the allow rule the model's
43
- * first `read` call hangs the entire turn.
46
+ * arrives. The permission watchdog is a fallback for new categories;
47
+ * this rule resolves the known path case without a polling round trip.
44
48
  *
45
49
  * Rules are evaluated in order; first match wins. (See upstream's
46
50
  * `PermissionRule` type — `permission` is the rule category, `pattern`
@@ -48,6 +52,7 @@ import { TALON_MCP_SERVER_NAME, getChatMcpServerName } from "./mcp.js";
48
52
  */
49
53
  export function buildPermissionRuleset(chatId: string): RemotePermissionRule[] {
50
54
  const ourServerName = getChatMcpServerName(chatId);
55
+ const ourPluginPrefix = `${TALON_PLUGIN_MCP_SERVER_NAME}-${safeMcpNamePart(chatId, "chat")}-`;
51
56
  return [
52
57
  { permission: "tool", pattern: `${ourServerName}_*`, action: "allow" },
53
58
  {
@@ -55,9 +60,16 @@ export function buildPermissionRuleset(chatId: string): RemotePermissionRule[] {
55
60
  pattern: `${TALON_MCP_SERVER_NAME}-*`,
56
61
  action: "deny",
57
62
  },
63
+ { permission: "tool", pattern: `${ourPluginPrefix}*`, action: "allow" },
64
+ {
65
+ permission: "tool",
66
+ pattern: `${TALON_PLUGIN_MCP_SERVER_NAME}-*`,
67
+ action: "deny",
68
+ },
58
69
  { permission: "tool", pattern: "*", action: "allow" },
59
70
  { permission: "edit", pattern: "*", action: "allow" },
60
71
  { permission: "bash", pattern: "*", action: "allow" },
72
+ { permission: "external_directory", pattern: "*", action: "allow" },
61
73
  ];
62
74
  }
63
75
 
@@ -106,3 +118,51 @@ export async function ensureRemoteSession<TClient extends RemoteAgentClient>(
106
118
 
107
119
  return newId;
108
120
  }
121
+
122
+ /**
123
+ * The per-turn setup a warm-up front-loads. Both remote-server backends
124
+ * expose these under identical signatures; the shape lets the helper stay
125
+ * backend-agnostic without importing either SDK.
126
+ */
127
+ export interface RemoteWarmDeps<TClient extends RemoteAgentClient> {
128
+ ensureServer(): Promise<TClient>;
129
+ ensureSession(client: TClient, chatId: string): Promise<string>;
130
+ ensureChatMcpServer(client: TClient, chatId: string): Promise<string>;
131
+ ensurePluginMcpServers(client: TClient, chatId: string): Promise<string[]>;
132
+ }
133
+
134
+ /**
135
+ * Pre-pay a chat's cold start: spawn the server if it isn't up, create (or
136
+ * resume) the session, and register the chat + plugin MCP servers.
137
+ *
138
+ * This is the remote-server analogue of the Claude backend's `warmSession`,
139
+ * and closes the last `sessions` capability gap between the two families.
140
+ * `performSessionReset` and the native frontend call it right after a reset,
141
+ * so the first turn on a fresh session doesn't serially pay session creation
142
+ * plus a full plugin-MCP registration sweep — the dominant cold-start cost
143
+ * here, since each plugin server is a separate connect.
144
+ *
145
+ * Best-effort by contract: `/reset` has already succeeded by the time this
146
+ * runs, and the same work is idempotent and repeated at the head of every
147
+ * turn. A failure must degrade to a slow first turn, never surface as a
148
+ * failed reset — so everything is caught and logged, not rethrown.
149
+ */
150
+ export async function warmRemoteSession<TClient extends RemoteAgentClient>(
151
+ state: RemoteServerState<TClient>,
152
+ chatId: string,
153
+ deps: RemoteWarmDeps<TClient>,
154
+ ): Promise<void> {
155
+ try {
156
+ const client = await deps.ensureServer();
157
+ await deps.ensureSession(client, chatId);
158
+ await deps.ensureChatMcpServer(client, chatId);
159
+ await deps.ensurePluginMcpServers(client, chatId);
160
+ log("agent", `[${chatId}] Warmed ${state.label} session`);
161
+ } catch (err) {
162
+ logWarn(
163
+ "agent",
164
+ `[${chatId}] ${state.label} warm-up skipped (first turn pays cold start): ` +
165
+ `${err instanceof Error ? err.message : String(err)}`,
166
+ );
167
+ }
168
+ }
@@ -44,17 +44,24 @@ export async function subscribeSseStream(
44
44
  client: SseSubscribableClient,
45
45
  chatId: string,
46
46
  ): Promise<AsyncIterable<unknown> | undefined> {
47
- let result: unknown;
48
- try {
49
- result = await client.global.event();
50
- } catch (err) {
51
- logWarn(
52
- "agent",
53
- `[${chatId}] SSE subscribe failed: ${err instanceof Error ? err.message : String(err)}`,
54
- );
55
- return undefined;
47
+ let lastError: unknown;
48
+ for (let attempt = 1; attempt <= 3; attempt++) {
49
+ try {
50
+ const stream = narrowSseResult(await client.global.event());
51
+ if (stream) return stream;
52
+ lastError = new Error("response did not contain an async event stream");
53
+ } catch (err) {
54
+ lastError = err;
55
+ }
56
+ if (attempt < 3) {
57
+ await new Promise((resolve) => setTimeout(resolve, 150 * attempt));
58
+ }
56
59
  }
57
- return narrowSseResult(result);
60
+ logWarn(
61
+ "agent",
62
+ `[${chatId}] SSE subscribe failed after 3 attempts: ${lastError instanceof Error ? lastError.message : String(lastError)}`,
63
+ );
64
+ return undefined;
58
65
  }
59
66
 
60
67
  /**
@@ -52,6 +52,10 @@ export interface RemoteServerState<TClient extends RemoteAgentClient> {
52
52
  * actual state, so we cache locally instead of trusting the server.
53
53
  */
54
54
  readonly registeredMcpServers: Set<string>;
55
+ /** Exact tool names exposed by each registered MCP server. */
56
+ readonly registeredMcpTools: Map<string, readonly string[]>;
57
+ /** Plugin-name → registered server-name mapping for each chat context. */
58
+ readonly pluginMcpServersByChat: Map<string, Map<string, string>>;
55
59
  }
56
60
 
57
61
  /** Inputs for {@link createRemoteServerState}. */
@@ -84,6 +88,8 @@ export function createRemoteServerState<TClient extends RemoteAgentClient>(
84
88
  serverHandle: null,
85
89
  modelProviderCache: new Map(),
86
90
  registeredMcpServers: new Set(),
91
+ registeredMcpTools: new Map(),
92
+ pluginMcpServersByChat: new Map(),
87
93
  };
88
94
  }
89
95
 
@@ -0,0 +1,71 @@
1
+ /** Bounded turn execution for OpenCode/Kilo's long-lived HTTP servers. */
2
+
3
+ import { logWarn } from "../../util/log.js";
4
+
5
+ const DEFAULT_REMOTE_TURN_TIMEOUT_MS = 10 * 60_000;
6
+
7
+ export function remoteTurnTimeoutMs(): number {
8
+ const configured = Number(process.env.TALON_REMOTE_TURN_TIMEOUT_MS);
9
+ return Number.isFinite(configured) && configured > 0
10
+ ? configured
11
+ : DEFAULT_REMOTE_TURN_TIMEOUT_MS;
12
+ }
13
+
14
+ export class RemoteTurnTimeoutError extends Error {
15
+ constructor(label: string, timeoutMs: number) {
16
+ super(`${label} turn timed out after ${Math.round(timeoutMs / 1000)}s`);
17
+ // Core error classification treats TimeoutError as a retryable transport
18
+ // failure, so the normal fallback-model path remains available.
19
+ this.name = "TimeoutError";
20
+ }
21
+ }
22
+
23
+ export interface RemoteTurnAbortClient {
24
+ session: {
25
+ abort(args: { sessionID: string }): Promise<unknown>;
26
+ };
27
+ }
28
+
29
+ /**
30
+ * Race a prompt+SSE turn against a hard deadline. On expiry, abort upstream
31
+ * generation before rejecting so the provider stops spending tokens and the
32
+ * session can be safely reset by the shared retry path.
33
+ */
34
+ export async function awaitRemoteTurn<T>(
35
+ turn: Promise<T>,
36
+ inputs: {
37
+ client: RemoteTurnAbortClient;
38
+ sessionId: string;
39
+ chatId: string;
40
+ label: string;
41
+ timeoutMs?: number;
42
+ },
43
+ ): Promise<T> {
44
+ const timeoutMs = inputs.timeoutMs ?? remoteTurnTimeoutMs();
45
+ let timer: ReturnType<typeof setTimeout> | null = null;
46
+ try {
47
+ return await Promise.race([
48
+ turn,
49
+ new Promise<never>((_, reject) => {
50
+ timer = setTimeout(() => {
51
+ logWarn(
52
+ "agent",
53
+ `[${inputs.chatId}] ${inputs.label} turn exceeded ${Math.round(timeoutMs / 1000)}s; aborting`,
54
+ );
55
+ inputs.client.session
56
+ .abort({ sessionID: inputs.sessionId })
57
+ .catch((err) =>
58
+ logWarn(
59
+ "agent",
60
+ `[${inputs.chatId}] ${inputs.label} timeout abort failed: ${err instanceof Error ? err.message : String(err)}`,
61
+ ),
62
+ );
63
+ reject(new RemoteTurnTimeoutError(inputs.label, timeoutMs));
64
+ }, timeoutMs);
65
+ timer.unref?.();
66
+ }),
67
+ ]);
68
+ } finally {
69
+ if (timer) clearTimeout(timer);
70
+ }
71
+ }
@@ -104,6 +104,28 @@ export function hubPluginServerNames(only?: string[]): string[] {
104
104
  return Object.keys(getPluginMcpServers("", "hub-enum", only));
105
105
  }
106
106
 
107
+ /**
108
+ * Enumerate one plugin server's tools through the same hub-managed child that
109
+ * serves MCP requests. Remote agent servers do not include dynamically-added
110
+ * MCP tools in their `/experimental/tool/ids` response, so OpenCode/Kilo need
111
+ * this authoritative list to build per-chat visibility overrides.
112
+ */
113
+ export async function listHubPluginToolNames(
114
+ serverName: string,
115
+ chatId: string,
116
+ bridgeUrl: string,
117
+ ): Promise<string[]> {
118
+ const key =
119
+ serverName === "brave-search"
120
+ ? "brave-search"
121
+ : `${serverName}\u0000${chatId}`;
122
+ const child = await acquireChild(key, () =>
123
+ pluginSpec(serverName, chatId, bridgeUrl),
124
+ );
125
+ child.touch();
126
+ return (await child.listTools()).map((tool) => tool.name);
127
+ }
128
+
107
129
  // ── Server construction per session ─────────────────────────────────────────
108
130
 
109
131
  type HubTarget =
@@ -70,7 +70,9 @@ export async function handleAdminCommand(
70
70
  const model = formatModelLabel(
71
71
  getChatSettings(s.chatId).model ?? config.model,
72
72
  );
73
- return `<b>${escapeHtml(title)}</b> <code>${s.chatId}</code>\n ${s.info.turns} turns | ${age} | ${model}`;
73
+ // `model` is a catalog id (OpenRouter/Kilo ids are free-form), so
74
+ // it gets the same escaping the title already had.
75
+ return `<b>${escapeHtml(title)}</b> <code>${s.chatId}</code>\n ${s.info.turns} turns | ${age} | ${escapeHtml(model)}`;
74
76
  });
75
77
  await ctx.reply(
76
78
  `<b>Active chats (${sessions.length})</b>\n\n` + lines.join("\n\n"),
@@ -44,6 +44,47 @@ import {
44
44
  editOrIgnoreSame,
45
45
  type CallbackDeps,
46
46
  } from "./shared.js";
47
+ import { logWarn } from "../../../util/log.js";
48
+
49
+ /**
50
+ * Run the backend-side half of a chat's session handoff.
51
+ *
52
+ * A backend switch already clears Talon's own stores (session row,
53
+ * history, pulse checkpoint), but those are only half the state.
54
+ * `performSessionReset` also drives the backend capability slots, and
55
+ * the switch path skipped them entirely:
56
+ *
57
+ * - the OUTGOING backend keeps in-process per-chat state that no
58
+ * amount of clearing Talon's stores touches. openai-agents holds a
59
+ * `MemorySession` map keyed by chat id, so switching away and back
60
+ * resurrected the old conversation the operator meant to drop.
61
+ * - the INCOMING backend was never warmed, so the first turn after a
62
+ * switch paid the full cold start — on OpenCode/Kilo that is session
63
+ * creation plus a per-plugin MCP registration sweep.
64
+ *
65
+ * The warm is deliberately fire-and-forget: it can take seconds, and the
66
+ * callback still has a toast to answer and a menu to redraw. Telegram
67
+ * expires an unanswered callback query, so blocking here would trade a
68
+ * cold first turn for a visibly stuck button.
69
+ */
70
+ function handOffBackendSession(
71
+ chatId: string,
72
+ previousBackend: ReturnType<typeof resolveBackendForChat>,
73
+ gateway: CallbackDeps["gateway"],
74
+ ): void {
75
+ previousBackend?.sessions?.resetChat?.(chatId);
76
+ const nextBackend = resolveBackendForChat(chatId, gateway);
77
+ // Same instance on a no-op switch — warming it again is harmless
78
+ // (every step is idempotent) but pointless, so skip it.
79
+ if (!nextBackend || nextBackend === previousBackend) return;
80
+ void Promise.resolve(nextBackend.sessions?.warmSession?.(chatId)).catch(
81
+ (err) =>
82
+ logWarn(
83
+ "bot",
84
+ `[${chatId}] warm after backend switch failed: ${err instanceof Error ? err.message : String(err)}`,
85
+ ),
86
+ );
87
+ }
47
88
 
48
89
  export async function handleModelCallback(
49
90
  ctx: Context,
@@ -171,6 +212,10 @@ export async function handleModelCallback(
171
212
  });
172
213
  return;
173
214
  }
215
+ // Resolve the outgoing backend BEFORE rebinding — afterwards this
216
+ // chat already points at the new one and the old in-process state
217
+ // would be unreachable.
218
+ const previousBackend = resolveBackendForChat(cid, gateway);
174
219
  const result = await rebindChat(cid, action.backendId, config);
175
220
  if (!result.ok) {
176
221
  await answerCallbackQuerySafe(ctx, {
@@ -188,6 +233,7 @@ export async function handleModelCallback(
188
233
  resetSession(cid);
189
234
  clearHistory(cid);
190
235
  resetPulseCheckpoint(cid);
236
+ handOffBackendSession(cid, previousBackend, gateway);
191
237
  const label =
192
238
  available.find((b) => b.id === action.backendId)?.label ??
193
239
  action.backendId;
@@ -210,11 +256,13 @@ export async function handleModelCallback(
210
256
  // global chat-role backend. Per-backend model picks are
211
257
  // preserved (modelByBackend stays intact) so reverting and
212
258
  // switching back later still restores prior choices.
259
+ const previousBackend = resolveBackendForChat(cid, gateway);
213
260
  await releaseChat(cid);
214
261
  setChatBackend(cid, undefined);
215
262
  resetSession(cid);
216
263
  clearHistory(cid);
217
264
  resetPulseCheckpoint(cid);
265
+ handOffBackendSession(cid, previousBackend, gateway);
218
266
  // Resolve the now-default backend's model for the toast.
219
267
  const defaultBackend = resolveBackendForChat(cid, gateway);
220
268
  const defaultBackendId = getBackendIdForChat(cid);
@@ -132,11 +132,17 @@ export function registerInfoCommands(bot: Bot): void {
132
132
  return;
133
133
  }
134
134
  const lines = plugins.map((p) => {
135
- const ver = p.plugin.version ? ` v${p.plugin.version}` : "";
136
- const desc = p.plugin.description ? ` — ${p.plugin.description}` : "";
135
+ // Every field here is author-supplied manifest text, not just the
136
+ // name. A description like "R&D tools" or "<beta>" would otherwise
137
+ // reach Telegram as markup and 400 the whole listing, so `/plugins`
138
+ // would look dead rather than show one odd line.
139
+ const ver = p.plugin.version ? ` v${escapeHtml(p.plugin.version)}` : "";
140
+ const desc = p.plugin.description
141
+ ? ` — ${escapeHtml(p.plugin.description)}`
142
+ : "";
137
143
  const mcp = p.plugin.mcpServerPath ? " [MCP]" : "";
138
144
  const fe = p.plugin.frontends?.length
139
- ? ` (${p.plugin.frontends.join(", ")})`
145
+ ? ` (${escapeHtml(p.plugin.frontends.join(", "))})`
140
146
  : "";
141
147
  return `• <b>${escapeHtml(p.plugin.name)}</b>${ver}${mcp}${fe}${desc}`;
142
148
  });
@@ -112,10 +112,16 @@ export function registerSettingsCommands(
112
112
  if (be?.models?.resolveModelInfo) {
113
113
  const resolution = await be.models?.resolveModelInfo(arg);
114
114
  if (resolution.kind !== "exact") {
115
+ // Every backend's formatModelError returns plain text and
116
+ // interpolates the raw query, so escape at this boundary — the
117
+ // same fix as the model menu's status lines. `/model <name>`
118
+ // (which the OpenCode/Kilo hint literally tells you to type)
119
+ // otherwise reaches Telegram as an unsupported `<name>` tag and
120
+ // the whole reply 400s, leaving the command looking dead.
115
121
  const msg =
116
122
  be.models?.formatModelError?.(arg, resolution) ??
117
- `No model matched "${escapeHtml(arg)}".`;
118
- await ctx.reply(msg, { parse_mode: "HTML" });
123
+ `No model matched "${arg}".`;
124
+ await ctx.reply(escapeHtml(msg), { parse_mode: "HTML" });
119
125
  return;
120
126
  }
121
127
  if (!resolution.model.selectable) {
@@ -122,10 +122,16 @@ async function flushQueue(chatId: string): Promise<void> {
122
122
  // Clear hourglass reactions on queued messages now that we're processing
123
123
  for (const msgId of queuedReactionMsgIds) {
124
124
  bot.api.setMessageReaction(numericChatId, msgId, []).catch((err) => {
125
- logWarn(
126
- "bot",
127
- `Failed to clear reaction on msg ${msgId}: ${err instanceof Error ? err.message : err}`,
128
- );
125
+ const detail = err instanceof Error ? err.message : String(err);
126
+ // The id is recorded optimistically — the hourglass `setMessageReaction`
127
+ // above is fire-and-forget, so we queue the clear before knowing the set
128
+ // landed. When it didn't, clearing a reaction that was never there comes
129
+ // back as REACTION_EMPTY. The end state we wanted (no reaction) already
130
+ // holds, so that is a no-op, not a failure. Keeping the optimistic push
131
+ // matters: gating it on the set resolving would race the flush and strand
132
+ // a ⏳ on the message. Anything else is still worth surfacing.
133
+ if (detail.includes("REACTION_EMPTY")) return;
134
+ logWarn("bot", `Failed to clear reaction on msg ${msgId}: ${detail}`);
129
135
  });
130
136
  }
131
137
 
@@ -370,9 +370,12 @@ export function renderModelMenuText(state: ModelMenuState): string {
370
370
  `<i>Use the picker below to choose one — sending a message before picking will be refused.</i>`,
371
371
  );
372
372
  } else {
373
- lines.push(`<b>Model:</b> <code>${state.activeDisplay}</code>`);
373
+ lines.push(`<b>Model:</b> <code>${escapeHtml(state.activeDisplay)}</code>`);
374
374
  }
375
- for (const l of state.statusLines) lines.push(l);
375
+ // Display names and status lines are plain text from backend catalogs —
376
+ // a literal `<name>` in a hint (or a `<` in a model id) is otherwise
377
+ // parsed as an HTML tag and Telegram rejects the whole send with 400.
378
+ for (const l of state.statusLines) lines.push(escapeHtml(l));
376
379
  if (state.freeOnly && state.showFreeToggle) {
377
380
  lines.push("<i>Filtering to free-tier models when browsing.</i>");
378
381
  }
@@ -3,8 +3,7 @@
3
3
  *
4
4
  * Spawns a fresh copy of the current process — same Node binary,
5
5
  * same `execArgv` (preserving the tsx loader so `.ts` entrypoints
6
- * still resolve), same script + user args, same cwd + env — then
7
- * triggers a graceful shutdown of the current process. The new
6
+ * still resolve), same script + user args, same cwd + env. The new
8
7
  * child is detached with stdio:"ignore" so it survives the parent's
9
8
  * exit; calling `unref()` lets the parent exit without waiting on
10
9
  * it.
@@ -16,59 +15,88 @@
16
15
  * debugger. Respawning from our own `process.argv` works regardless
17
16
  * of launch method.
18
17
  *
19
- * Shutdown is done by raising SIGTERM on ourselves rather than
20
- * `process.exit(0)` so the existing graceful-shutdown handler runs
21
- * (`await frontend.stop()`, flush sessions, remove PID file). The
22
- * brief overlap between the new child binding Telegram's long-poll
23
- * and the old one releasing it is benign — grammy retries on 409
24
- * Conflict and the new child takes over within a few seconds.
18
+ * Ordering matters. `respawnSelf()` only *arms* the handoff and
19
+ * raises SIGTERM; the successor is spawned by `spawnSuccessor()` at
20
+ * the tail of graceful shutdown, once the frontends have stopped.
21
+ * Spawning up-front (the previous behaviour) left the successor
22
+ * long-polling `getUpdates` while the outgoing process was still
23
+ * draining in-flight queries — up to DRAIN_TIMEOUT_MS of two live
24
+ * pollers. Telegram answers only one of them and re-delivers the
25
+ * unconfirmed updates to the other, so a restart mid-turn produced
26
+ * a 409 Conflict on the way out and duplicate replies on the way in.
27
+ * Releasing the poll before the successor binds it removes the
28
+ * overlap rather than relying on grammy's 409 retry to paper over it.
25
29
  */
26
30
 
27
31
  import { spawn } from "node:child_process";
28
32
  import { log, logError } from "./log.js";
29
33
 
34
+ let pendingReason: string | null = null;
35
+
30
36
  /**
31
- * Respawn the current process with identical argv + flags, then
32
- * raise SIGTERM on ourselves so the existing graceful-shutdown path
33
- * cleanly stops the frontend, flushes state, and exits.
37
+ * Arm a respawn and raise SIGTERM on ourselves so the existing
38
+ * graceful-shutdown path cleanly stops the frontends, flushes state,
39
+ * and hands off via `spawnSuccessor()`.
34
40
  *
35
41
  * `reason` is logged for operator visibility (e.g. "telegram
36
- * /restart"). The function returns immediately; the actual exit
37
- * happens asynchronously once the child has spawned (or failed).
42
+ * /restart"). The function returns immediately; the successor starts
43
+ * only after shutdown has released the Telegram long-poll.
38
44
  */
39
45
  export function respawnSelf(reason: string): void {
40
46
  log("shutdown", `Respawn requested (${reason})`);
47
+ pendingReason = reason;
48
+ // SIGTERM triggers the graceful-shutdown handler in src/app.ts,
49
+ // which stops the frontends, flushes state, spawns the successor,
50
+ // and calls process.exit(0). Don't exit here directly — that would
51
+ // skip the flush and leave the PID file dangling.
52
+ process.kill(process.pid, "SIGTERM");
53
+ }
41
54
 
42
- const child = spawn(
43
- process.argv[0],
44
- [...process.execArgv, ...process.argv.slice(1)],
45
- {
46
- cwd: process.cwd(),
47
- detached: true,
48
- stdio: "ignore",
49
- env: { ...process.env },
50
- },
51
- );
55
+ /** True when a `/restart` or `/update` armed a handoff. */
56
+ export function respawnRequested(): boolean {
57
+ return pendingReason !== null;
58
+ }
52
59
 
53
- child.once("spawn", () => {
54
- log("shutdown", `Respawn child started (pid ${child.pid}) — exiting self`);
55
- child.unref();
56
- // SIGTERM triggers the existing graceful-shutdown handler in
57
- // src/index.ts, which stops the frontend, flushes state, and
58
- // calls process.exit(0). Don't exit here directly — that would
59
- // skip the flush and leave the PID file dangling.
60
- process.kill(process.pid, "SIGTERM");
61
- });
60
+ /**
61
+ * Spawn the successor process. Called at the end of graceful
62
+ * shutdown, after the frontends have stopped — so the incoming
63
+ * process binds Telegram's long-poll only once this one has let go
64
+ * of it. No-op unless `respawnSelf()` armed a handoff.
65
+ *
66
+ * Never throws: a failed handoff must not prevent this process from
67
+ * exiting. An external supervisor (systemd, pm2, the user's
68
+ * terminal) can pick things up.
69
+ */
70
+ export function spawnSuccessor(): void {
71
+ if (pendingReason === null) return;
72
+ const reason = pendingReason;
73
+ pendingReason = null;
62
74
 
63
- child.once("error", (err) => {
75
+ try {
76
+ const child = spawn(
77
+ process.argv[0],
78
+ [...process.execArgv, ...process.argv.slice(1)],
79
+ {
80
+ cwd: process.cwd(),
81
+ detached: true,
82
+ stdio: "ignore",
83
+ env: { ...process.env },
84
+ },
85
+ );
86
+ child.once("error", (err) => {
87
+ logError(
88
+ "shutdown",
89
+ `Respawn failed; exiting without a successor — restart manually`,
90
+ err,
91
+ );
92
+ });
93
+ child.unref();
94
+ log("shutdown", `Respawn child started (pid ${child.pid}) — ${reason}`);
95
+ } catch (err) {
64
96
  logError(
65
97
  "shutdown",
66
98
  `Respawn failed; exiting without a successor — restart manually`,
67
99
  err,
68
100
  );
69
- // The child never started, so we can't hand off. Still exit so
70
- // any external supervisor (systemd, pm2, the user's terminal)
71
- // can pick things up.
72
- process.kill(process.pid, "SIGTERM");
73
- });
101
+ }
74
102
  }