talon-agent 5.18.2 → 5.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (190) hide show
  1. package/README.md +2 -1
  2. package/package.json +1 -1
  3. package/prompts/system/agent-brief.md +20 -3
  4. package/src/app.ts +13 -0
  5. package/src/backend/claude-sdk/handler.ts +4 -4
  6. package/src/backend/claude-sdk/mcp-ready.ts +16 -2
  7. package/src/backend/claude-sdk/stream.ts +2 -2
  8. package/src/backend/codex/auth.ts +1 -1
  9. package/src/backend/codex/handler/message.ts +11 -11
  10. package/src/backend/codex/init.ts +4 -9
  11. package/src/backend/codex/mcp-config.ts +1 -2
  12. package/src/backend/codex/oauth-incompat.ts +8 -4
  13. package/src/backend/codex/one-shot.ts +1 -1
  14. package/src/backend/openai-agents/builtins.ts +55 -27
  15. package/src/backend/openai-agents/factory.ts +3 -3
  16. package/src/backend/openai-agents/handler/message.ts +3 -5
  17. package/src/backend/openai-agents/mcp-pool.ts +10 -27
  18. package/src/backend/remote-server/chat-turn.ts +6 -6
  19. package/src/backend/remote-server/events.ts +1 -5
  20. package/src/backend/remote-server/index.ts +0 -1
  21. package/src/backend/remote-server/messages.ts +3 -7
  22. package/src/backend/remote-server/one-shot.ts +1 -3
  23. package/src/backend/remote-server/session-helpers.ts +1 -4
  24. package/src/backend/remote-server/sse-stream.ts +8 -9
  25. package/src/backend/runtime/metrics.ts +7 -13
  26. package/src/backend/runtime/sleep.ts +1 -2
  27. package/src/backend/runtime/turn/handle-retry.ts +48 -2
  28. package/src/backend/runtime/turn/handler-to-events.ts +3 -4
  29. package/src/bootstrap.ts +9 -1
  30. package/src/cli/doctor.ts +3 -0
  31. package/src/cli/index.ts +8 -10
  32. package/src/cli/logs.ts +148 -9
  33. package/src/cli/setup.ts +9 -11
  34. package/src/cli/status.ts +29 -0
  35. package/src/core/agent-runtime/README.md +5 -19
  36. package/src/core/agent-runtime/events.ts +3 -39
  37. package/src/core/agent-runtime/model-ref.ts +0 -8
  38. package/src/core/agents/registry.ts +49 -2
  39. package/src/core/auth/expiry-monitor.ts +9 -1
  40. package/src/core/auth/login-flow.ts +9 -1
  41. package/src/core/auth/status.ts +31 -3
  42. package/src/core/background/cron/scheduler.ts +25 -5
  43. package/src/core/background/dream/index.ts +29 -10
  44. package/src/core/background/failure-backoff.ts +30 -0
  45. package/src/core/background/heartbeat/agent.ts +2 -27
  46. package/src/core/background/heartbeat/index.ts +0 -2
  47. package/src/core/background/heartbeat/scheduler.ts +21 -13
  48. package/src/core/background/heartbeat/state.ts +10 -2
  49. package/src/core/background/isolated-agent.ts +6 -2
  50. package/src/core/background/pulse/pulse.ts +9 -0
  51. package/src/core/background/triggers/exit.ts +54 -0
  52. package/src/core/background/triggers/index.ts +1 -3
  53. package/src/core/background/triggers/resume.ts +2 -4
  54. package/src/core/backup/archive/tar.ts +14 -4
  55. package/src/core/backup/restore.ts +21 -13
  56. package/src/core/backup/scheduler.ts +28 -9
  57. package/src/core/backup/snapshot.ts +43 -10
  58. package/src/core/backup/store.ts +7 -19
  59. package/src/core/backup/targets.ts +69 -17
  60. package/src/core/config/index.ts +14 -0
  61. package/src/core/daemon/crash-marker.ts +141 -0
  62. package/src/core/daemon/crash.ts +9 -2
  63. package/src/core/daemon/handoff.ts +15 -0
  64. package/src/core/daemon/health-alerts.ts +297 -0
  65. package/src/core/daemon/log-reader.ts +289 -0
  66. package/src/core/doctor/index.ts +18 -2
  67. package/src/core/doctor/logs.ts +124 -0
  68. package/src/core/doctor/types.ts +1 -1
  69. package/src/core/engine/backend-controller/index.ts +1 -13
  70. package/src/core/engine/backend-router/router.ts +1 -1
  71. package/src/core/engine/dispatcher.ts +55 -2
  72. package/src/core/engine/fault-text.ts +40 -0
  73. package/src/core/engine/gateway-actions/agents/index.ts +3 -2
  74. package/src/core/engine/gateway-actions/agents/report.ts +62 -0
  75. package/src/core/engine/gateway-actions/history.ts +2 -4
  76. package/src/core/engine/gateway-actions/native/exec.ts +13 -16
  77. package/src/core/engine/gateway.ts +60 -1
  78. package/src/core/engine/turn-health.ts +222 -0
  79. package/src/core/errors.ts +2 -2
  80. package/src/core/frontend-runtime/admin-notify.ts +1 -1
  81. package/src/core/frontend-runtime/alerts.ts +130 -0
  82. package/src/core/mcp-hub/children.ts +78 -29
  83. package/src/core/mcp-hub/index.ts +21 -18
  84. package/src/core/mcp-hub/proxy-server.ts +8 -4
  85. package/src/core/mcp-hub/talon-server.ts +5 -12
  86. package/src/core/mesh/credentials/store.ts +16 -1
  87. package/src/core/mesh/devices/service.ts +26 -15
  88. package/src/core/mesh/devices/teleport.ts +14 -2
  89. package/src/core/mesh/links/node-binaries.ts +13 -6
  90. package/src/core/mesh/persist.ts +22 -10
  91. package/src/core/mesh/transfers/device-files.ts +5 -23
  92. package/src/core/mesh/transfers/transfers.ts +16 -3
  93. package/src/core/models/active-model.ts +2 -55
  94. package/src/core/plugin/actions.ts +19 -20
  95. package/src/core/plugin/builtins.ts +80 -90
  96. package/src/core/plugin/index.ts +1 -4
  97. package/src/core/plugin/loader.ts +25 -33
  98. package/src/core/plugin/mcp.ts +3 -5
  99. package/src/core/plugin/registry.ts +19 -35
  100. package/src/core/plugin/types.ts +2 -5
  101. package/src/core/prompt/assemble.ts +15 -3
  102. package/src/core/scripts/lua.ts +6 -2
  103. package/src/core/tasks/table.ts +8 -2
  104. package/src/core/tools/bridge.ts +2 -4
  105. package/src/core/tools/chat/cross-send.ts +1 -1
  106. package/src/core/tools/chat/messaging.ts +1 -1
  107. package/src/core/tools/index.ts +2 -2
  108. package/src/core/tools/mcp-env.ts +2 -59
  109. package/src/core/tools/ops/agents.ts +22 -1
  110. package/src/core/tools/schemas.ts +4 -9
  111. package/src/core/vfs/fusefs.ts +0 -5
  112. package/src/core/vfs/index.ts +9 -2
  113. package/src/core/vfs/mounts/diagnostics.ts +109 -0
  114. package/src/core/vfs/mounts/proc.ts +17 -1
  115. package/src/core/vfs/workspace.ts +7 -3
  116. package/src/core/weaver/shuttle.ts +8 -1
  117. package/src/core/weaver/turn-log.ts +320 -0
  118. package/src/core/weaver/weaver.ts +48 -5
  119. package/src/frontend/discord/actions/index.ts +8 -1
  120. package/src/frontend/discord/diagnostics.ts +82 -6
  121. package/src/frontend/discord/handlers/index.ts +0 -2
  122. package/src/frontend/discord/middleware.ts +7 -13
  123. package/src/frontend/discord/runtime.ts +1 -3
  124. package/src/frontend/health/delivery.ts +115 -0
  125. package/src/frontend/health/outage.ts +116 -0
  126. package/src/frontend/native/bridge/routes/chats.ts +3 -5
  127. package/src/frontend/native/bridge/server.ts +106 -21
  128. package/src/frontend/native/index.ts +1 -1
  129. package/src/frontend/native/media/media.ts +5 -1
  130. package/src/frontend/native/runtime.ts +12 -7
  131. package/src/frontend/native/surface/handlers.ts +1 -1
  132. package/src/frontend/native/surface/memory.ts +1 -1
  133. package/src/frontend/native/surface/models.ts +3 -3
  134. package/src/frontend/native/surface/settings.ts +20 -8
  135. package/src/frontend/native/turn/context.ts +6 -8
  136. package/src/frontend/native/turn/turn-meta.ts +2 -5
  137. package/src/frontend/native/turn/turn.ts +8 -10
  138. package/src/frontend/presentation/format.ts +2 -4
  139. package/src/frontend/presentation/session-status.ts +2 -6
  140. package/src/frontend/teams/actions.ts +8 -1
  141. package/src/frontend/teams/graph.ts +0 -1
  142. package/src/frontend/teams/index.ts +1 -4
  143. package/src/frontend/teams/poll.ts +40 -2
  144. package/src/frontend/teams/runtime.ts +14 -5
  145. package/src/frontend/telegram/actions/index.ts +4 -1
  146. package/src/frontend/telegram/actions/send.ts +8 -0
  147. package/src/frontend/telegram/handlers/context.ts +13 -2
  148. package/src/frontend/telegram/handlers/delivery.ts +12 -9
  149. package/src/frontend/telegram/handlers/index.ts +0 -2
  150. package/src/frontend/telegram/index.ts +35 -9
  151. package/src/frontend/telegram/polling/poll-health.ts +110 -0
  152. package/src/frontend/telegram/userbot.ts +100 -36
  153. package/src/frontend/terminal/builtins/session.ts +2 -2
  154. package/src/frontend/terminal/index.ts +1 -3
  155. package/src/frontend/terminal/renderer.ts +2 -18
  156. package/src/frontend/whatsapp/actions/index.ts +12 -1
  157. package/src/frontend/whatsapp/actions/messaging.ts +7 -2
  158. package/src/frontend/whatsapp/connection/connection.ts +29 -9
  159. package/src/frontend/whatsapp/connection/health.ts +89 -0
  160. package/src/frontend/whatsapp/connection/identity.ts +4 -4
  161. package/src/frontend/whatsapp/runtime.ts +11 -5
  162. package/src/native/blake3.ts +28 -2
  163. package/src/native/fusefs.ts +23 -5
  164. package/src/native/registry.ts +1 -1
  165. package/src/native/warden.ts +33 -5
  166. package/src/plugins/github/index.ts +0 -1
  167. package/src/plugins/mempalace/index.ts +9 -3
  168. package/src/plugins/playwright/index.ts +2 -4
  169. package/src/plugins/playwright/provision.ts +12 -4
  170. package/src/storage/chat-settings.ts +6 -1
  171. package/src/storage/cron.ts +29 -4
  172. package/src/storage/daily-log.ts +43 -47
  173. package/src/storage/db.ts +61 -33
  174. package/src/storage/history.ts +6 -1
  175. package/src/storage/journal.ts +9 -2
  176. package/src/storage/kv.ts +19 -6
  177. package/src/storage/media-index.ts +28 -6
  178. package/src/storage/repositories/chat-settings-repo.ts +8 -2
  179. package/src/storage/repositories/sessions-repo.ts +10 -3
  180. package/src/storage/scripts.ts +24 -13
  181. package/src/storage/sessions.ts +11 -2
  182. package/src/storage/skills.ts +21 -2
  183. package/src/storage/stickers.ts +17 -3
  184. package/src/storage/triggers.ts +8 -3
  185. package/src/storage/turn-meta.ts +25 -7
  186. package/src/util/log.ts +189 -6
  187. package/src/util/logging/turn-scope.ts +85 -0
  188. package/src/util/time.ts +3 -3
  189. package/src/util/watchdog.ts +30 -0
  190. package/src/core/engine/backend-controller/legacy.ts +0 -111
@@ -102,10 +102,8 @@ export function liveTurnEvents(runtime: NativeRuntime): BridgeEvent[] {
102
102
 
103
103
  /**
104
104
  * Safety net: any tool the backend announced but never resolved (a crash can
105
- * eat a tool_result; callback backends historically never emitted one —
106
- * handler-to-events now pairs each tool_call with an immediate synthetic
107
- * result) gets a synthetic result at turn end — a spinner the app opened on
108
- * phase:"call" must always see a phase:"result".
105
+ * eat a tool_result) gets a synthetic result at turn end — a spinner the app
106
+ * opened on phase:"call" must always see a phase:"result".
109
107
  */
110
108
  function flushOpenTools(
111
109
  runtime: NativeRuntime,
@@ -312,14 +310,14 @@ async function runTurn(
312
310
  // clients. Any late tool spinner has already been flushed above.
313
311
  runtime.liveTurns.delete(entry.id);
314
312
  // Flush a queued follow-up as a fresh turn: "once it's done it will
315
- // send". Deferred so we don't re-enter runTurn inside its own finally.
313
+ // send". Started synchronously so the chat never looks idle in between —
314
+ // a /send landing in a deferred gap would start its own turn and run
315
+ // ahead of the message that was queued first.
316
316
  const queued = takeQueued(runtime, entry);
317
317
  if (queued) {
318
- setImmediate(() =>
319
- startTurn(runtime, entry, queued.text, {
320
- attachments: queued.attachments,
321
- }),
322
- );
318
+ startTurn(runtime, entry, queued.text, {
319
+ attachments: queued.attachments,
320
+ });
323
321
  }
324
322
  }
325
323
  }
@@ -1,10 +1,8 @@
1
1
  /**
2
2
  * Frontend-agnostic display formatters shared by all chat frontends.
3
3
  *
4
- * These were previously duplicated per-frontend (the old
5
- * telegram/helpers/format.ts, discord/helpers.ts) and had started to drift —
6
- * parseInterval accepted days on Discord but not Telegram. One definition
7
- * here keeps them in lockstep; frontends import this module directly.
4
+ * One definition here keeps the frontends in lockstep — per-frontend copies
5
+ * drift (parseInterval once accepted days on Discord but not Telegram).
8
6
  */
9
7
 
10
8
  import { resolveModel } from "../../core/models/catalog.js";
@@ -1,12 +1,8 @@
1
1
  /**
2
2
  * Shared /reset and /status logic for chat frontends.
3
3
  *
4
- * Telegram and Discord previously each carried a full copy of the session
5
- * reset sequence and the /status data-gathering pipeline, and the copies had
6
- * drifted (Discord's /status no longer re-fetched the context window for the
7
- * model that actually served the session). All the frontend-agnostic work
8
- * lives here now; frontends only render `SessionStatusData` in their native
9
- * markup.
4
+ * All the frontend-agnostic work lives here so per-frontend copies can't
5
+ * drift; frontends only render `SessionStatusData` in their native markup.
10
6
  */
11
7
 
12
8
  import type { TalonConfig } from "../../core/config/index.js";
@@ -7,6 +7,9 @@ import type { Gateway } from "../../core/engine/gateway.js";
7
7
  import { buildAdaptiveCard, splitTeamsMessage } from "./formatting.js";
8
8
  import { log, logError } from "../../util/log.js";
9
9
  import { proxyFetch } from "./proxy-fetch.js";
10
+ import { createDeliveryTracker, trackDeliveries } from "../health/delivery.js";
11
+
12
+ const delivery = createDeliveryTracker("teams", "Teams", "teams");
10
13
 
11
14
  /**
12
15
  * POST an Adaptive Card to the Power Automate workflow webhook URL.
@@ -32,7 +35,10 @@ export function createTeamsActionHandler(
32
35
  webhookUrl: string,
33
36
  gateway: Gateway,
34
37
  ): FrontendActionHandler {
35
- return async (body, chatId): Promise<ActionResult | null> => {
38
+ const dispatch: FrontendActionHandler = async (
39
+ body,
40
+ chatId,
41
+ ): Promise<ActionResult | null> => {
36
42
  const action = body.action as string;
37
43
 
38
44
  switch (action) {
@@ -100,6 +106,7 @@ export function createTeamsActionHandler(
100
106
  return null;
101
107
  }
102
108
  };
109
+ return trackDeliveries(delivery, dispatch);
103
110
  }
104
111
 
105
112
  export { postToTeams };
@@ -302,7 +302,6 @@ export class GraphClient {
302
302
  * Initialize the Graph client — loads stored tokens or runs device code flow.
303
303
  */
304
304
  export async function initGraphClient(): Promise<GraphClient> {
305
- // Clear old tokens that used channel scopes
306
305
  const stored = loadTokens();
307
306
 
308
307
  if (stored && stored.refreshToken) {
@@ -76,9 +76,7 @@ export function createTeamsFrontend(
76
76
  const graphClient = await initGraphClient();
77
77
  runtime.graphClient = graphClient;
78
78
 
79
- // Get our own user ID (to filter out our own messages)
80
79
  const me = await graphClient.getMe();
81
- runtime.myUserId = me.id;
82
80
  log("teams", `Authenticated as: ${me.displayName} (${me.id})`);
83
81
 
84
82
  const chatId = await resolveChatId(runtime, graphClient, me.id);
@@ -99,8 +97,7 @@ export function createTeamsFrontend(
99
97
 
100
98
  // The receive side is a timer on the runtime, not a loop to sit in:
101
99
  // once the first poll is done the frontend is listening, and
102
- // start() is finished. (It used to park on a promise that never
103
- // resolved, which made the boot end at shutdown.)
100
+ // start() is finished.
104
101
  await startPolling(runtime, chatId);
105
102
  },
106
103
 
@@ -6,6 +6,7 @@
6
6
  */
7
7
 
8
8
  import { log, logError } from "../../util/log.js";
9
+ import { errorText } from "../health/outage.js";
9
10
  import { handleSlashCommand } from "./commands.js";
10
11
  import type { ChatMessage } from "./graph.js";
11
12
  import type { TeamsRuntime } from "./runtime.js";
@@ -51,13 +52,47 @@ async function handleMessage(
51
52
  runTurn(runtime, msg, talonChatId);
52
53
  }
53
54
 
55
+ /**
56
+ * Fetch the newest messages, feeding the poll outage: failures are logged
57
+ * with their attempt number and time down, and the first success after
58
+ * them logs the recovery. Returns null when the fetch failed.
59
+ */
60
+ async function fetchMessages(
61
+ runtime: TeamsRuntime,
62
+ graph: NonNullable<TeamsRuntime["graphClient"]>,
63
+ chatId: string,
64
+ ): Promise<ChatMessage[] | null> {
65
+ let messages: ChatMessage[];
66
+ try {
67
+ messages = await graph.getChatMessages(chatId, 20);
68
+ } catch (err) {
69
+ const { attempt, downMs } = runtime.pollOutage.fail(err);
70
+ logError(
71
+ "teams",
72
+ `Poll error: poll.fail chat=${chatId} attempt=${attempt} down_ms=${downMs} ` +
73
+ `next_poll_ms=${runtime.pollIntervalMs} err=${errorText(err)}`,
74
+ );
75
+ return null;
76
+ }
77
+ const ended = runtime.pollOutage.ok();
78
+ if (ended) {
79
+ log(
80
+ "teams",
81
+ `poll.recovered chat=${chatId} failed_attempts=${ended.attempts} down_ms=${ended.downMs}`,
82
+ );
83
+ }
84
+ return messages;
85
+ }
86
+
54
87
  async function poll(runtime: TeamsRuntime, chatId: string): Promise<void> {
55
88
  if (runtime.polling) return;
56
89
  runtime.polling = true;
90
+ let current: ChatMessage | null = null;
57
91
 
58
92
  try {
59
93
  if (!runtime.graphClient) return;
60
- const messages = await runtime.graphClient.getChatMessages(chatId, 20);
94
+ const messages = await fetchMessages(runtime, runtime.graphClient, chatId);
95
+ if (!messages) return;
61
96
  const newMessages = selectNewMessages(messages, runtime.lastSeenMessageId);
62
97
 
63
98
  if (newMessages.length > 0) {
@@ -69,12 +104,14 @@ async function poll(runtime: TeamsRuntime, chatId: string): Promise<void> {
69
104
  if (!msg.text.trim()) continue;
70
105
  if (msg.edited) continue;
71
106
  if (isBotEcho(runtime, msg)) continue;
107
+ current = msg;
72
108
  await handleMessage(runtime, msg);
73
109
  }
74
110
  } catch (err) {
75
111
  logError(
76
112
  "teams",
77
- `Poll error: ${err instanceof Error ? err.message : err}`,
113
+ `Poll error: message=${current?.id ?? "?"} chat=${current?.chatId ?? chatId} ` +
114
+ `${err instanceof Error ? err.message : err}`,
78
115
  );
79
116
  } finally {
80
117
  runtime.polling = false;
@@ -94,6 +131,7 @@ export async function startPolling(
94
131
  }
95
132
 
96
133
  export function stopPolling(runtime: TeamsRuntime): void {
134
+ runtime.pollOutage.dispose();
97
135
  if (runtime.pollTimer) {
98
136
  clearInterval(runtime.pollTimer);
99
137
  runtime.pollTimer = null;
@@ -1,15 +1,17 @@
1
1
  /**
2
2
  * Teams frontend runtime — the state every module shares.
3
3
  *
4
- * `createTeamsFrontend` used to hold all of this as closure variables, with
5
- * the poll loop and the slash commands nested inside `.start`. It is now one
6
- * explicit object, constructed once, that each module (chat-discovery,
4
+ * One explicit object, constructed once, that each module (chat-discovery,
7
5
  * poll, commands, turn, outbound) takes as its first parameter.
8
6
  */
9
7
 
10
8
  import type { TalonConfig } from "../../core/config/index.js";
11
9
  import type { Gateway } from "../../core/engine/gateway.js";
12
10
  import type { GraphClient } from "./graph.js";
11
+ import { createOutage, type Outage } from "../health/outage.js";
12
+
13
+ /** Graph polling must fail this long before `teams.poll` is raised. */
14
+ const POLL_OUTAGE_MS = 10 * 60_000;
13
15
 
14
16
  export type TeamsRuntime = {
15
17
  readonly config: TalonConfig;
@@ -25,9 +27,10 @@ export type TeamsRuntime = {
25
27
  pollTimer: ReturnType<typeof setInterval> | null;
26
28
  /** Newest message id already handled — the poll loop cuts at it. */
27
29
  lastSeenMessageId: string | null;
28
- myUserId: string | null;
29
30
  /** Re-entrancy guard: a slow poll never overlaps the next tick. */
30
31
  polling: boolean;
32
+ /** Failed Graph fetches, and the `teams.poll` alert they raise. */
33
+ readonly pollOutage: Outage;
31
34
  };
32
35
 
33
36
  export function createTeamsRuntime(
@@ -45,7 +48,13 @@ export function createTeamsRuntime(
45
48
  graphClient: null,
46
49
  pollTimer: null,
47
50
  lastSeenMessageId: null,
48
- myUserId: null,
49
51
  polling: false,
52
+ pollOutage: createOutage({
53
+ key: "teams.poll",
54
+ thresholdMs: POLL_OUTAGE_MS,
55
+ describe: (err, mins) =>
56
+ `Teams polling has failed for ${mins} min: ${err}. Messages are not being received.`,
57
+ recovered: "Teams polling is working again.",
58
+ }),
50
59
  };
51
60
  }
@@ -24,6 +24,8 @@ import { mediaHandlers } from "./media.js";
24
24
  import { chatInfoHandlers } from "./chat-info.js";
25
25
  import { moderationHandlers } from "./moderation/index.js";
26
26
  import type { TelegramActionContext, TelegramActionHandlers } from "./types.js";
27
+ import { trackDeliveries } from "../../health/delivery.js";
28
+ import { telegramDelivery } from "./send.js";
27
29
 
28
30
  export { sendText } from "./send.js";
29
31
 
@@ -58,7 +60,7 @@ export function createTelegramActionHandler(
58
60
  // shutdown — the timers died with the process, the store didn't.
59
61
  restoreScheduledMessages(bot, ctx.scheduledMessages);
60
62
 
61
- return async (
63
+ const dispatch = async (
62
64
  body: Record<string, unknown>,
63
65
  chatId: number,
64
66
  ): Promise<ActionResult | null> => {
@@ -67,4 +69,5 @@ export function createTelegramActionHandler(
67
69
  if (!handler) return null; // not a Telegram action
68
70
  return handler(body, chatId, ctx);
69
71
  };
72
+ return trackDeliveries(telegramDelivery, dispatch);
70
73
  }
@@ -14,8 +14,16 @@ import {
14
14
  noteRichMessageFailure,
15
15
  richMessagesAvailable,
16
16
  } from "./rich-messages.js";
17
+ import { createDeliveryTracker } from "../../health/delivery.js";
17
18
  import { recordOutgoingText } from "./outgoing-log.js";
18
19
 
20
+ /** Reply delivery streaks — shared by the reply actions and text blocks. */
21
+ export const telegramDelivery = createDeliveryTracker(
22
+ "telegram",
23
+ "Telegram",
24
+ "bot",
25
+ );
26
+
19
27
  export function replyParams(
20
28
  body: Record<string, unknown>,
21
29
  ): ReplyParams | undefined {
@@ -147,17 +147,28 @@ export function getForwardContext(msg: {
147
147
  return `[Forwarded from ${from}]\n`;
148
148
  }
149
149
 
150
+ /**
151
+ * Deadline for one whole download (getFile + body). Media handlers run in
152
+ * grammY's update loop, which handles updates one at a time, so a download
153
+ * stalled on a dead socket holds every later message behind it.
154
+ */
155
+ const DOWNLOAD_TIMEOUT_MS = 120_000;
156
+
150
157
  export async function downloadTelegramFile(
151
158
  bot: Bot,
152
159
  config: TalonConfig,
153
160
  fileId: string,
154
161
  fileName: string,
155
162
  ): Promise<string> {
156
- const file = await bot.api.getFile(fileId);
163
+ const signal = AbortSignal.timeout(DOWNLOAD_TIMEOUT_MS);
164
+ const file = await bot.api.getFile(
165
+ fileId,
166
+ signal as Parameters<Bot["api"]["getFile"]>[1],
167
+ );
157
168
  if (!file.file_path) throw new Error("Could not get file path from Telegram");
158
169
 
159
170
  const url = `https://api.telegram.org/file/bot${config.botToken}/${file.file_path}`;
160
- const resp = await fetch(url);
171
+ const resp = await fetch(url, { signal });
161
172
  if (!resp.ok) throw new Error(`Download failed: ${resp.status}`);
162
173
 
163
174
  // Guard against excessively large files (50MB limit)
@@ -10,7 +10,7 @@ import { toolInputToRecord } from "../../../core/agent-runtime/events.js";
10
10
  import { appendDailyLogResponse } from "../../../storage/daily-log.js";
11
11
  import { stripMcpPrefix } from "../../../core/tools/index.js";
12
12
  import { logWarn } from "../../../util/log.js";
13
- import { replyParamsFor, sendText } from "../actions/send.js";
13
+ import { replyParamsFor, sendText, telegramDelivery } from "../actions/send.js";
14
14
  import { ambientThreadId } from "../topics.js";
15
15
  import { isAdminInGroup, trackDmUser } from "./access.js";
16
16
 
@@ -90,10 +90,7 @@ function createStreamCallbacks(
90
90
  state: StreamState,
91
91
  chatTitle?: string,
92
92
  ) {
93
- const onStreamDelta = async (
94
- accumulated: string,
95
- _phase?: "thinking" | "text",
96
- ) => {
93
+ const onStreamDelta = async (accumulated: string) => {
97
94
  // Skip if drafts not supported or not ready
98
95
  if (draftsSupported === false || !state.started || state.editing) return;
99
96
  if (accumulated.length - state.lastSentLength < 40) return;
@@ -122,7 +119,13 @@ function createStreamCallbacks(
122
119
  };
123
120
 
124
121
  const onTextBlock = async (text: string) => {
125
- await sendText(bot, chatId, text, _replyToId);
122
+ try {
123
+ await sendText(bot, chatId, text, _replyToId);
124
+ } catch (err) {
125
+ telegramDelivery.failed(chatId, err);
126
+ throw err;
127
+ }
128
+ telegramDelivery.delivered(chatId);
126
129
  appendDailyLogResponse("Talon", text, { chatTitle });
127
130
  state.lastSentLength = 0;
128
131
  state.sentTextBlock = true;
@@ -223,12 +226,12 @@ export async function processAndReply(
223
226
  textAccum += event.text;
224
227
  // Fire-and-forget: draft edits are throttled + self-mutexed
225
228
  // (`state.editing`), so we must NOT block stream consumption
226
- // on them — same non-awaited semantics the old bridge had.
227
- void onStreamDelta(textAccum, "text");
229
+ // on them.
230
+ void onStreamDelta(textAccum);
228
231
  break;
229
232
  case "reasoning":
230
233
  thinkingAccum += event.text;
231
- void onStreamDelta(thinkingAccum, "thinking");
234
+ void onStreamDelta(thinkingAccum);
232
235
  break;
233
236
  case "assistant_message":
234
237
  // Keep the running total monotonic so a following
@@ -9,8 +9,6 @@
9
9
  * - `delivery` — HTML send, streaming drafts, the agent run pipeline
10
10
  * - `queue` — per-chat debounce queue + per-user rate limiting
11
11
  * - `messages` — per-message-type handlers (text/photo/voice/…/callback)
12
- *
13
- * Re-exports the same public surface the old single-file module exposed.
14
12
  */
15
13
 
16
14
  export {
@@ -6,7 +6,7 @@
6
6
  * core gateway so MCP tool calls route to Telegram API.
7
7
  */
8
8
 
9
- import { Bot, InputFile, API_CONSTANTS } from "grammy";
9
+ import { Bot, GrammyError, InputFile, API_CONSTANTS } from "grammy";
10
10
  import { autoRetry } from "@grammyjs/auto-retry";
11
11
  import { apiThrottler } from "@grammyjs/transformer-throttler";
12
12
  import {
@@ -17,6 +17,7 @@ import type { ContextManager } from "../../core/types.js";
17
17
  import type { Gateway } from "../../core/engine/gateway.js";
18
18
  import { runUntilStopped } from "../../core/frontend-runtime/run-loop.js";
19
19
  import { pollDeadline } from "./polling/poll-deadline.js";
20
+ import { pollHealth } from "./polling/poll-health.js";
20
21
  import { createTelegramActionHandler, sendText } from "./actions/index.js";
21
22
  import { ambientThreadId } from "./topics.js";
22
23
  import { initUserClient, disconnectUserClient } from "./userbot.js";
@@ -48,6 +49,12 @@ export type TelegramFrontend = {
48
49
  stop: () => Promise<void>;
49
50
  };
50
51
 
52
+ /** The slice of a grammY BotError's context the error log reads. */
53
+ type UpdateContext = {
54
+ update?: { update_id?: number };
55
+ chat?: { id?: number };
56
+ };
57
+
51
58
  // ── Access ──────────────────────────────────────────────────────────────────
52
59
 
53
60
  /**
@@ -68,6 +75,30 @@ function applyAccessControl(config: TalonConfig): void {
68
75
  setAllowedGroups(config.allowedGroups);
69
76
  }
70
77
 
78
+ /**
79
+ * The bot's last-resort middleware error handler: log which update in which
80
+ * chat failed, and exit when Telegram says the token itself is bad.
81
+ */
82
+ function onBotError(err: unknown): void {
83
+ const ctx = (err as { ctx?: UpdateContext } | null)?.ctx;
84
+ logError(
85
+ "bot",
86
+ `Unhandled bot error update=${ctx?.update?.update_id ?? "?"} chat=${ctx?.chat?.id ?? "?"}`,
87
+ err,
88
+ );
89
+ // Judge the token by Telegram's error code, never the message text:
90
+ // a handler's ordinary 400 ("message to edit not found", "chat not
91
+ // found") must not take the whole daemon down.
92
+ const cause = (err as { error?: unknown } | null)?.error ?? err;
93
+ if (
94
+ cause instanceof GrammyError &&
95
+ (cause.error_code === 401 || cause.error_code === 404)
96
+ ) {
97
+ logError("bot", "Bot token appears invalid — shutting down");
98
+ process.exit(1);
99
+ }
100
+ }
101
+
71
102
  // ── Factory ─────────────────────────────────────────────────────────────────
72
103
 
73
104
  export function createTelegramFrontend(
@@ -80,6 +111,8 @@ export function createTelegramFrontend(
80
111
  bot.api.config.use(pollDeadline());
81
112
  bot.api.config.use(apiThrottler());
82
113
  bot.api.config.use(autoRetry({ maxRetryAttempts: 3, maxDelaySeconds: 60 }));
114
+ // Outermost: judges each poll by the result grammY finally sees.
115
+ bot.api.config.use(pollHealth());
83
116
 
84
117
  const context: ContextManager = {
85
118
  acquire: (chatId: number, stringId?: string) =>
@@ -149,14 +182,7 @@ export function createTelegramFrontend(
149
182
  },
150
183
 
151
184
  async start() {
152
- bot.catch((err: unknown) => {
153
- const msg = err instanceof Error ? err.message : String(err);
154
- logError("bot", "Unhandled bot error", err);
155
- if (/unauthorized|401|not found|404/i.test(msg)) {
156
- logError("bot", "Bot token appears invalid — shutting down");
157
- process.exit(1);
158
- }
159
- });
185
+ bot.catch(onBotError);
160
186
  // Beyond grammY's defaults: `chat_join_request` feeds the moderation
161
187
  // tool's pending-join cache (inert unless the bot admins an
162
188
  // approval-gated chat).
@@ -0,0 +1,110 @@
1
+ /**
2
+ * Long-poll health — the operator hears when the bot stops receiving.
3
+ *
4
+ * grammY retries a failed getUpdates forever (3s apart, or Telegram's
5
+ * `retry_after`) and logs nothing outside its debug channel, so a bot cut
6
+ * off from api.telegram.org used to look exactly like a quiet day. This
7
+ * transformer watches every getUpdates grammY makes:
8
+ *
9
+ * - each failure is logged with its attempt number, the time down so far,
10
+ * grammY's backoff and the error; failures that persist for
11
+ * `POLL_OUTAGE_MS` raise `telegram.polling`, and the next successful
12
+ * poll resolves it;
13
+ * - 409 Conflict (another process polling the same token) and 401 (token
14
+ * rejected) end grammY's polling loop outright, so they alert at once
15
+ * and critically — nothing will retry them.
16
+ *
17
+ * Installed outermost, so it sees what grammY sees: one result per poll,
18
+ * after the deadline and auto-retry layers have had their turn.
19
+ */
20
+
21
+ import { GrammyError, type Transformer } from "grammy";
22
+ import {
23
+ raiseAlert,
24
+ resolveAlert,
25
+ } from "../../../core/frontend-runtime/alerts.js";
26
+ import { log, logError, logWarn } from "../../../util/log.js";
27
+ import { createOutage, errorText } from "../../health/outage.js";
28
+
29
+ /** Failed polls must persist this long before `telegram.polling` fires. */
30
+ const POLL_OUTAGE_MS = 5 * 60_000;
31
+ /** grammY's pause between failed getUpdates calls (bot.js handlePollingError). */
32
+ const GRAMMY_RETRY_MS = 3_000;
33
+
34
+ function backoffMs(err: unknown): number {
35
+ if (err instanceof GrammyError && err.error_code === 429) {
36
+ return (err.parameters.retry_after ?? 3) * 1000;
37
+ }
38
+ return GRAMMY_RETRY_MS;
39
+ }
40
+
41
+ export function pollHealth(thresholdMs = POLL_OUTAGE_MS): Transformer {
42
+ const outage = createOutage({
43
+ key: "telegram.polling",
44
+ thresholdMs,
45
+ describe: (err, mins) =>
46
+ `Telegram polling has failed for ${mins} min: ${err}. Messages are not being received.`,
47
+ recovered: "Telegram polling is working again.",
48
+ });
49
+ let conflict = false;
50
+
51
+ const onFatal = (err: GrammyError): void => {
52
+ const detail = errorText(err);
53
+ logError(
54
+ "bot",
55
+ `telegram.poll.stopped code=${err.error_code} err=${detail}`,
56
+ );
57
+ if (err.error_code === 409) {
58
+ conflict = true;
59
+ raiseAlert(
60
+ "telegram.conflict",
61
+ `Another process is polling this Telegram bot token (${detail}). ` +
62
+ "Talon has stopped receiving Telegram messages: stop the other instance, then restart Talon.",
63
+ { severity: "critical" },
64
+ );
65
+ } else {
66
+ outage.raiseNow(
67
+ `Telegram rejected the bot token (${detail}). Talon has stopped receiving Telegram messages.`,
68
+ "critical",
69
+ );
70
+ }
71
+ };
72
+
73
+ return async (prev, method, payload, signal) => {
74
+ if (method !== "getUpdates") return prev(method, payload, signal);
75
+ try {
76
+ const res = await prev(method, payload, signal);
77
+ const ended = outage.ok();
78
+ if (ended) {
79
+ log(
80
+ "bot",
81
+ `telegram.poll.recovered failed_attempts=${ended.attempts} down_ms=${ended.downMs}`,
82
+ );
83
+ }
84
+ if (conflict) {
85
+ conflict = false;
86
+ resolveAlert(
87
+ "telegram.conflict",
88
+ "Telegram polling resumed — no other process is polling this bot token.",
89
+ );
90
+ }
91
+ return res;
92
+ } catch (err) {
93
+ // bot.stop() cancelling the in-flight poll is a shutdown, not a fault.
94
+ if (signal?.aborted) throw err;
95
+ if (
96
+ err instanceof GrammyError &&
97
+ (err.error_code === 409 || err.error_code === 401)
98
+ ) {
99
+ onFatal(err);
100
+ throw err;
101
+ }
102
+ const { attempt, downMs } = outage.fail(err);
103
+ logWarn(
104
+ "bot",
105
+ `telegram.poll.fail attempt=${attempt} down_ms=${downMs} backoff_ms=${backoffMs(err)} err=${errorText(err)}`,
106
+ );
107
+ throw err;
108
+ }
109
+ };
110
+ }