talon-agent 3.22.0 → 3.22.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.22.0",
3
+ "version": "3.22.1",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
package/src/app.ts CHANGED
@@ -9,7 +9,7 @@
9
9
  import { getFrontends } from "./util/config.js";
10
10
  import { startUploadCleanup, stopUploadCleanup } from "./util/workspace.js";
11
11
  import { flushDatabase } from "./storage/db.js";
12
- import { getActiveCount } from "./core/engine/dispatcher.js";
12
+ import { getActiveCount, stopAllTurns } from "./core/engine/dispatcher.js";
13
13
  import { startPulseTimer, stopPulseTimer } from "./core/background/pulse.js";
14
14
  import { stopPlanAlerts } from "./core/background/plan-alerts.js";
15
15
  import {
@@ -121,6 +121,7 @@ async function gracefulShutdown(signal: string): Promise<void> {
121
121
  shuttingDown = true;
122
122
  log("shutdown", `${signal} received, shutting down gracefully...`);
123
123
 
124
+ const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
124
125
  const forceTimer = setTimeout(() => {
125
126
  logError("shutdown", "Timeout exceeded, forcing exit");
126
127
  // Hand off even on the forced path. A restart must survive a
@@ -135,12 +136,17 @@ async function gracefulShutdown(signal: string): Promise<void> {
135
136
  }, SHUTDOWN_TIMEOUT_MS);
136
137
  forceTimer.unref();
137
138
 
138
- // Drain in-flight queries: poll instead of a fixed sleep, so an idle
139
- // daemon exits immediately and a busy one gets the full budget.
139
+ // Drain in-flight queries. A turn can legitimately run for minutes, so a
140
+ // drain that only waits can never succeed against one — ask every running
141
+ // turn to abort first, then poll for the aborts to settle so backends can
142
+ // flush partial state before the process exits.
140
143
  if (getActiveCount() > 0) {
144
+ const aborted = stopAllTurns();
141
145
  log(
142
146
  "shutdown",
143
- `Waiting for ${getActiveCount()} in-flight queries to drain...`,
147
+ `Waiting for ${getActiveCount()} in-flight queries to drain` +
148
+ (aborted > 0 ? ` (abort requested for ${aborted})` : "") +
149
+ `...`,
144
150
  );
145
151
  const deadline = Date.now() + DRAIN_TIMEOUT_MS;
146
152
  while (getActiveCount() > 0 && Date.now() < deadline) {
@@ -177,7 +183,11 @@ async function gracefulShutdown(signal: string): Promise<void> {
177
183
  await shutdownStep("pulse timer", stopPulseTimer);
178
184
  await shutdownStep("heartbeat", async () => {
179
185
  stopHeartbeatTimer();
180
- await awaitHeartbeat();
186
+ // Cap the wait by what's left of the force-timer budget (minus margin
187
+ // for the steps below). The default 10s wait plus a full 5s drain used
188
+ // to consume the entire 15s budget, so a slow heartbeat tripped the
189
+ // forced exit even though teardown was proceeding normally.
190
+ await awaitHeartbeat(Math.max(0, deadlineAt - Date.now() - 3_000));
181
191
  });
182
192
  await shutdownStep("cron timer", stopCronTimer);
183
193
  await shutdownStep("plan alerts", stopPlanAlerts);
@@ -67,8 +67,9 @@ import {
67
67
  recordFailedTurnAccounting,
68
68
  recordFlowViolation,
69
69
  formatTurnCache,
70
- exceedsLookbackWindow,
71
- estimateTurnBlocks,
70
+ crossTurnVerdict,
71
+ priorLookbackOverflow,
72
+ noteLookbackRisk,
72
73
  CACHE_LOOKBACK_BLOCKS,
73
74
  } from "../shared/index.js";
74
75
 
@@ -594,16 +595,24 @@ export async function* runChatTurn(
594
595
 
595
596
  // The aggregate `cache=NN%` can't distinguish a turn that reused the
596
597
  // previous turn's prefix from one that re-wrote it — see
597
- // shared/cache-telemetry.ts. Append the cross-turn verdict when the
598
- // provider gave us per-request usage to derive it from.
599
- if (state.cacheStats && exceedsLookbackWindow(state.toolCalls)) {
600
- logWarn(
601
- "agent",
602
- `[${chatId}] turn emitted ~${estimateTurnBlocks(state.toolCalls)} content ` +
603
- `blocks (> ${CACHE_LOOKBACK_BLOCKS} lookback) — the next turn's cache ` +
604
- `breakpoint may not find this turn's prefix`,
605
- );
598
+ // shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
599
+ // so warn when this turn's verdict proves the previous turn's overflow
600
+ // cost a prefix re-write, then record this turn's overflow for the next.
601
+ if (state.cacheStats) {
602
+ const overflow = priorLookbackOverflow(chatId);
603
+ if (
604
+ overflow !== undefined &&
605
+ crossTurnVerdict(state.cacheStats) === "miss"
606
+ ) {
607
+ logWarn(
608
+ "agent",
609
+ `[${chatId}] previous turn emitted ~${overflow} content blocks ` +
610
+ `(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
611
+ `missed — that turn's cache write was likely never read`,
612
+ );
613
+ }
606
614
  }
615
+ noteLookbackRisk(chatId, state.toolCalls);
607
616
 
608
617
  log(
609
618
  "agent",
@@ -29,7 +29,7 @@
29
29
  * registrations run concurrently through the hub; later turns skip them.
30
30
  */
31
31
 
32
- import { log, logWarn } from "../../util/log.js";
32
+ import { log, logDebug, logWarn } from "../../util/log.js";
33
33
  import {
34
34
  talonHubUrl,
35
35
  pluginHubUrl,
@@ -62,6 +62,9 @@ export const TALON_PLUGIN_MCP_SERVER_NAME = "talon-plugin";
62
62
  */
63
63
  const SLOW_MCP_REGISTRATION_MS = 1000;
64
64
 
65
+ /** Servers whose registration failure was already warned about (see below). */
66
+ const warnedMcpRegistrationFailures = new Set<string>();
67
+
65
68
  /**
66
69
  * OpenCode's `/experimental/tool/ids` endpoint omits dynamically registered
67
70
  * MCP tools despite its API description. Synthesize Talon's known MCP ids so
@@ -253,6 +256,7 @@ export async function ensurePluginMcpServers<TClient extends RemoteAgentClient>(
253
256
  state.registeredMcpServers.add(serverName);
254
257
  state.registeredMcpTools.set(serverName, toolNames);
255
258
  byPlugin.set(name, serverName);
259
+ warnedMcpRegistrationFailures.delete(serverName);
256
260
  const ms = Date.now() - startedAt;
257
261
  log(
258
262
  "agent",
@@ -261,7 +265,16 @@ export async function ensurePluginMcpServers<TClient extends RemoteAgentClient>(
261
265
  );
262
266
  return serverName;
263
267
  } catch (err) {
264
- logWarn(
268
+ // Warn once per server, then debug: registration re-runs at the
269
+ // head of every turn, so a dead backing service (an offline
270
+ // browser endpoint, say) would otherwise emit one warning per
271
+ // turn for the whole outage. Success clears the latch so the
272
+ // next outage warns again.
273
+ const level = warnedMcpRegistrationFailures.has(serverName)
274
+ ? logDebug
275
+ : logWarn;
276
+ warnedMcpRegistrationFailures.add(serverName);
277
+ level(
265
278
  "agent",
266
279
  `Plugin MCP registration failed for ${serverName}: ${errMsg(err)}`,
267
280
  );
@@ -154,6 +154,48 @@ export function exceedsLookbackWindow(toolCalls: number): boolean {
154
154
  return estimateTurnBlocks(toolCalls) > CACHE_LOOKBACK_BLOCKS;
155
155
  }
156
156
 
157
+ /**
158
+ * Per-chat estimated block count of the last turn that overflowed the
159
+ * lookback window. Overflow is only a *prediction* of a cache miss — the
160
+ * proof is the NEXT turn's cross-turn verdict, so the overflow is recorded
161
+ * here and the warning waits for that verdict instead of firing on every
162
+ * tool-heavy turn. Bounded like `lastToolSets` below.
163
+ */
164
+ const lookbackOverflows = new Map<string, number>();
165
+
166
+ /**
167
+ * Record whether this turn plausibly overflowed the lookback window, so the
168
+ * next turn can attribute a cross-turn miss to it.
169
+ */
170
+ export function noteLookbackRisk(chatId: string, toolCalls: number): void {
171
+ if (!exceedsLookbackWindow(toolCalls)) {
172
+ lookbackOverflows.delete(chatId);
173
+ return;
174
+ }
175
+ if (
176
+ lookbackOverflows.size >= MAX_TRACKED_CHATS &&
177
+ !lookbackOverflows.has(chatId)
178
+ ) {
179
+ const oldest = lookbackOverflows.keys().next().value;
180
+ if (oldest !== undefined) lookbackOverflows.delete(oldest);
181
+ }
182
+ lookbackOverflows.set(chatId, estimateTurnBlocks(toolCalls));
183
+ }
184
+
185
+ /**
186
+ * Estimated block count of the chat's previous turn IF it overflowed the
187
+ * lookback window, else undefined. Read this before `noteLookbackRisk`
188
+ * records the current turn.
189
+ */
190
+ export function priorLookbackOverflow(chatId: string): number | undefined {
191
+ return lookbackOverflows.get(chatId);
192
+ }
193
+
194
+ /** Drop all recorded overflows (tests / explicit reset). */
195
+ export function resetLookbackRisk(): void {
196
+ lookbackOverflows.clear();
197
+ }
198
+
157
199
  // ── Cacheable minimum ──────────────────────────────────────────────────────
158
200
 
159
201
  /**
@@ -73,8 +73,9 @@ export {
73
73
  // barrel discipline the prompt/ barrel was just trimmed to.
74
74
  export {
75
75
  formatTurnCache,
76
- estimateTurnBlocks,
77
- exceedsLookbackWindow,
76
+ crossTurnVerdict,
77
+ priorLookbackOverflow,
78
+ noteLookbackRisk,
78
79
  CACHE_LOOKBACK_BLOCKS,
79
80
  } from "./cache-telemetry.js";
80
81
 
@@ -66,6 +66,16 @@ export function stopCurrentTurn(chatId: string): KillOutcome {
66
66
  return taskTable.killRunningTurn(chatId);
67
67
  }
68
68
 
69
+ /**
70
+ * Request an abort of every chat's running turn. The shutdown drain calls
71
+ * this before polling `getActiveCount` — a turn can legitimately run for
72
+ * minutes, so a drain that only waits can never succeed against one.
73
+ * Returns the number of kill requests issued.
74
+ */
75
+ export function stopAllTurns(): number {
76
+ return taskTable.killAllRunningTurns();
77
+ }
78
+
69
79
  /**
70
80
  * Execute an AI query with full lifecycle management.
71
81
  * Same-chat queries are serialized (FIFO) to avoid session conflicts.
@@ -33,6 +33,17 @@ type ErrorReason =
33
33
  const USAGE_LIMIT_RE =
34
34
  /you['’]ve hit your .{0,40}limit|you['’]re out of extra usage|claude ai usage limit reached|usage limit reached/i;
35
35
 
36
+ /**
37
+ * Ceiling on how long a failed attempt may have run and still earn the
38
+ * frontend queues' blind retry. `retryable` means "a short pause may clear
39
+ * it", which holds for a 429 or a dropped socket but not for an attempt
40
+ * that already burned minutes before failing (e.g. the 600s remote turn
41
+ * deadline — its rejection is `name: "TimeoutError"`, classified as
42
+ * transient network). Turns serialize per chat, so retrying such an
43
+ * attempt doubles the stall for every message queued behind it.
44
+ */
45
+ export const RETRY_ELAPSED_CAP_MS = 120_000;
46
+
36
47
  // ── TalonError class ────────────────────────────────────────────────────────
37
48
 
38
49
  export class TalonError extends Error {
@@ -62,6 +62,31 @@ type ChildEntry = {
62
62
  const children = new Map<string, ChildEntry>();
63
63
  const inflight = new Map<string, Promise<ChildHandle>>();
64
64
 
65
+ /**
66
+ * Negative cache for spawn failures. A child whose backing service is down
67
+ * (e.g. playwright-tools with its browser endpoint offline) dies at the
68
+ * connect handshake, and without this every single turn re-paid the
69
+ * spawn+handshake (~600ms) and re-logged the failure for the whole outage.
70
+ * Failures back off exponentially; the first attempt after the window
71
+ * clears the entry on success, so recovery costs one turn.
72
+ */
73
+ type SpawnFailure = { at: number; count: number; error: unknown };
74
+ const spawnFailures = new Map<string, SpawnFailure>();
75
+ const FAILURE_BACKOFF_BASE_MS = 30_000;
76
+ const FAILURE_BACKOFF_MAX_MS = 10 * 60_000;
77
+
78
+ function failureBackoffMs(count: number): number {
79
+ return Math.min(
80
+ FAILURE_BACKOFF_BASE_MS * 2 ** (count - 1),
81
+ FAILURE_BACKOFF_MAX_MS,
82
+ );
83
+ }
84
+
85
+ /** Test seam: forget recorded spawn failures. */
86
+ export function resetSpawnFailures(): void {
87
+ spawnFailures.clear();
88
+ }
89
+
65
90
  /** Idle TTL for hub children; tunable for tests / tight deployments. */
66
91
  function idleTtlMs(): number {
67
92
  const raw = Number(process.env.TALON_MCP_HUB_IDLE_MS);
@@ -168,9 +193,28 @@ export function acquireChild(
168
193
  const pending = inflight.get(key);
169
194
  if (pending) return pending;
170
195
 
196
+ const failure = spawnFailures.get(key);
197
+ if (failure && Date.now() - failure.at < failureBackoffMs(failure.count)) {
198
+ return Promise.reject(
199
+ failure.error instanceof Error
200
+ ? failure.error
201
+ : new Error(String(failure.error)),
202
+ );
203
+ }
204
+
171
205
  const promise = (async () => {
172
206
  try {
173
- return await spawnChild(key, spec());
207
+ const handle = await spawnChild(key, spec());
208
+ spawnFailures.delete(key);
209
+ return handle;
210
+ } catch (err) {
211
+ const prior = spawnFailures.get(key);
212
+ spawnFailures.set(key, {
213
+ at: Date.now(),
214
+ count: (prior?.count ?? 0) + 1,
215
+ error: err,
216
+ });
217
+ throw err;
174
218
  } finally {
175
219
  inflight.delete(key);
176
220
  }
@@ -122,6 +122,20 @@ export class TaskTable {
122
122
  return { ok: true };
123
123
  }
124
124
 
125
+ /**
126
+ * Request an abort of every running turn, regardless of chat — the
127
+ * shutdown drain's lever. Returns the number of kill requests issued.
128
+ */
129
+ killAllRunningTurns(): number {
130
+ let killed = 0;
131
+ for (const [id, task] of this.live) {
132
+ if (task.record.kind === "turn" && task.record.state === "running") {
133
+ if (this.kill(id).ok) killed++;
134
+ }
135
+ }
136
+ return killed;
137
+ }
138
+
125
139
  /**
126
140
  * Request an abort for the turn currently running in one chat. Queued turns
127
141
  * are deliberately ignored: `/stop` means "stop what is happening now",
@@ -4,10 +4,15 @@
4
4
  * notification.
5
5
  */
6
6
 
7
- import { classify, friendlyMessage } from "../../../core/errors.js";
7
+ import {
8
+ classify,
9
+ friendlyMessage,
10
+ RETRY_ELAPSED_CAP_MS,
11
+ } from "../../../core/errors.js";
8
12
  import {
9
13
  recordMessageProcessed,
10
14
  recordMessageReceived,
15
+ recordMessageSettled,
11
16
  recordError,
12
17
  } from "../../../util/watchdog.js";
13
18
  import { appendDailyLog } from "../../../storage/daily-log.js";
@@ -82,6 +87,7 @@ async function flushQueue(chatId: string): Promise<void> {
82
87
  chatTitle: last.chatTitle,
83
88
  });
84
89
 
90
+ const startedAt = Date.now();
85
91
  try {
86
92
  await runTurn();
87
93
  recordMessageProcessed();
@@ -94,7 +100,19 @@ async function flushQueue(chatId: string): Promise<void> {
94
100
  );
95
101
  recordError(classified.message);
96
102
 
97
- if (classified.retryable) {
103
+ // Retry once for transient errors — but only when the failed attempt
104
+ // was actually brief. An attempt that already ran for minutes (a remote
105
+ // turn deadline) won't be saved by a 2s pause, and turns serialize per
106
+ // chat, so the blind retry used to double a 10-minute stall for
107
+ // everything queued behind it.
108
+ const attemptMs = Date.now() - startedAt;
109
+ if (classified.retryable && attemptMs >= RETRY_ELAPSED_CAP_MS) {
110
+ log(
111
+ "bot",
112
+ `[${chatId}] Not retrying ${classified.reason}: the attempt already ran ${Math.round(attemptMs / 1000)}s`,
113
+ );
114
+ }
115
+ if (classified.retryable && attemptMs < RETRY_ELAPSED_CAP_MS) {
98
116
  const delayMs = classified.retryAfterMs ?? 2000;
99
117
  log(
100
118
  "bot",
@@ -108,6 +126,7 @@ async function flushQueue(chatId: string): Promise<void> {
108
126
  } catch (retryErr) {
109
127
  const retryClassified = classify(retryErr);
110
128
  logError("bot", `[${chatId}] Retry failed: ${retryClassified.message}`);
129
+ recordMessageSettled();
111
130
  // Error-recovery send must never throw — if the network is fully down
112
131
  // the queue handler would otherwise propagate up and stall future
113
132
  // messages. Best-effort: notify if we can, log + move on otherwise.
@@ -126,6 +145,7 @@ async function flushQueue(chatId: string): Promise<void> {
126
145
  return;
127
146
  }
128
147
  }
148
+ recordMessageSettled();
129
149
  try {
130
150
  await sendChunked(
131
151
  last.channel,
@@ -11,9 +11,31 @@ import { TELEGRAM_MAX_TEXT } from "./types.js";
11
11
 
12
12
  export function replyParams(
13
13
  body: Record<string, unknown>,
14
- ): { message_id: number } | undefined {
15
- const replyTo = toPositiveId(body.reply_to ?? body.reply_to_message_id);
16
- return replyTo !== undefined ? { message_id: replyTo } : undefined;
14
+ ): ReplyParams | undefined {
15
+ return replyParamsFor(
16
+ toPositiveId(body.reply_to ?? body.reply_to_message_id),
17
+ );
18
+ }
19
+
20
+ export type ReplyParams = {
21
+ message_id: number;
22
+ allow_sending_without_reply: true;
23
+ };
24
+
25
+ /**
26
+ * Build `reply_parameters` for an outbound send. Always allows sending
27
+ * without the reply: the target message may have been deleted by the time
28
+ * we send (fast-moving groups, cleanup bots), and without this flag
29
+ * Telegram 400s the whole send — every formatting-level fallback then
30
+ * fails identically and the message is lost. A send that arrives
31
+ * un-linked beats one that never arrives.
32
+ */
33
+ export function replyParamsFor(
34
+ replyTo: number | undefined,
35
+ ): ReplyParams | undefined {
36
+ return replyTo !== undefined && replyTo > 0
37
+ ? { message_id: replyTo, allow_sending_without_reply: true }
38
+ : undefined;
17
39
  }
18
40
 
19
41
  /** Delivery modifiers shared by every outbound send. */
@@ -125,7 +147,7 @@ export async function sendText(
125
147
  chatId,
126
148
  { markdown: text },
127
149
  {
128
- reply_parameters: replyTo ? { message_id: replyTo } : undefined,
150
+ reply_parameters: replyParamsFor(replyTo),
129
151
  reply_markup: replyMarkup,
130
152
  ...opts,
131
153
  },
@@ -140,7 +162,7 @@ export async function sendText(
140
162
  try {
141
163
  const sent = await bot.api.sendMessage(chatId, html, {
142
164
  parse_mode: "HTML",
143
- reply_parameters: replyTo ? { message_id: replyTo } : undefined,
165
+ reply_parameters: replyParamsFor(replyTo),
144
166
  reply_markup: replyMarkup,
145
167
  ...opts,
146
168
  });
@@ -151,7 +173,7 @@ export async function sendText(
151
173
  `Legacy HTML send failed; retrying as plain text (chat=${chatId}): ${err instanceof Error ? err.message : err}`,
152
174
  );
153
175
  const sent = await bot.api.sendMessage(chatId, text, {
154
- reply_parameters: replyTo ? { message_id: replyTo } : undefined,
176
+ reply_parameters: replyParamsFor(replyTo),
155
177
  reply_markup: replyMarkup,
156
178
  ...opts,
157
179
  });
@@ -10,7 +10,7 @@ import { toolInputToRecord } from "../../../core/agent-runtime/events.js";
10
10
  import { appendDailyLogResponse } from "../../../storage/daily-log.js";
11
11
  import { stripMcpPrefix } from "../../../core/tools/index.js";
12
12
  import { logWarn } from "../../../util/log.js";
13
- import { sendText } from "../actions/shared.js";
13
+ import { replyParamsFor, sendText } from "../actions/shared.js";
14
14
  import { ambientThreadId } from "../topics.js";
15
15
  import { trackDmUser } from "./access.js";
16
16
 
@@ -22,7 +22,7 @@ export async function sendHtml(
22
22
  ): Promise<number> {
23
23
  const params = {
24
24
  parse_mode: "HTML" as const,
25
- reply_parameters: replyToId ? { message_id: replyToId } : undefined,
25
+ reply_parameters: replyParamsFor(replyToId),
26
26
  message_thread_id: ambientThreadId(chatId),
27
27
  };
28
28
  try {
@@ -40,7 +40,7 @@ export async function sendHtml(
40
40
  plain = plain.replace(/<[^>]*>/g, "");
41
41
  } while (plain !== prev);
42
42
  const sent = await bot.api.sendMessage(chatId, plain, {
43
- reply_parameters: replyToId ? { message_id: replyToId } : undefined,
43
+ reply_parameters: replyParamsFor(replyToId),
44
44
  message_thread_id: ambientThreadId(chatId),
45
45
  });
46
46
  return sent.message_id;
@@ -9,13 +9,18 @@
9
9
  import type { Bot } from "grammy";
10
10
  import type { TalonConfig } from "../../../util/config.js";
11
11
  import { escapeHtml } from "../formatting.js";
12
- import { classify, friendlyMessage } from "../../../core/errors.js";
12
+ import {
13
+ classify,
14
+ friendlyMessage,
15
+ RETRY_ELAPSED_CAP_MS,
16
+ } from "../../../core/errors.js";
13
17
  import {
14
18
  getRecentHistory,
15
19
  type HistoryMessage,
16
20
  } from "../../../storage/history.js";
17
21
  import {
18
22
  recordMessageProcessed,
23
+ recordMessageSettled,
19
24
  recordMessageReceived,
20
25
  recordError,
21
26
  } from "../../../util/watchdog.js";
@@ -176,6 +181,7 @@ async function flushQueue(chatId: string): Promise<void> {
176
181
  chatTitle: last.chatTitle,
177
182
  });
178
183
 
184
+ const startedAt = Date.now();
179
185
  try {
180
186
  await runTurn();
181
187
  lastHandledMessageIdByChat.set(chatId, last.messageId);
@@ -190,8 +196,20 @@ async function flushQueue(chatId: string): Promise<void> {
190
196
  );
191
197
  recordError(classified.message);
192
198
 
193
- // Retry once for transient errors (rate_limit, overloaded, network)
194
- if (classified.retryable) {
199
+ // Retry once for transient errors (rate_limit, overloaded, network) —
200
+ // but only when the failed attempt was actually brief. An attempt that
201
+ // already ran for minutes (a remote turn deadline, a slow collapse)
202
+ // won't be saved by a 2s pause, and turns serialize per chat, so the
203
+ // blind retry used to double a 10-minute stall for everything queued
204
+ // behind it.
205
+ const attemptMs = Date.now() - startedAt;
206
+ if (classified.retryable && attemptMs >= RETRY_ELAPSED_CAP_MS) {
207
+ log(
208
+ "bot",
209
+ `[${chatId}] Not retrying ${classified.reason}: the attempt already ran ${Math.round(attemptMs / 1000)}s`,
210
+ );
211
+ }
212
+ if (classified.retryable && attemptMs < RETRY_ELAPSED_CAP_MS) {
195
213
  const delayMs = classified.retryAfterMs ?? 2000;
196
214
  log(
197
215
  "bot",
@@ -209,6 +227,7 @@ async function flushQueue(chatId: string): Promise<void> {
209
227
  "bot",
210
228
  `[${chatId}] [${chatType}] Retry failed: ${retryClassified.message}`,
211
229
  );
230
+ recordMessageSettled();
212
231
  await sendHtml(
213
232
  bot,
214
233
  numericChatId,
@@ -219,6 +238,7 @@ async function flushQueue(chatId: string): Promise<void> {
219
238
  }
220
239
  }
221
240
 
241
+ recordMessageSettled();
222
242
  await sendHtml(
223
243
  bot,
224
244
  numericChatId,
@@ -30,6 +30,18 @@ export function recordMessageProcessed(): void {
30
30
  resetStuckWarnBackoff();
31
31
  }
32
32
 
33
+ /**
34
+ * Record that a message's turn settled in failure. The loop is moving —
35
+ * the error was reported and the queue slot freed — so advance the stuck
36
+ * clock without counting a success. Without this, one failed turn leaves
37
+ * `lastProcessedAt` stale forever and the watchdog reports a wedged loop
38
+ * (and an unhealthy bridge) on a bot that is actually idle.
39
+ */
40
+ export function recordMessageSettled(): void {
41
+ lastProcessedAt = Date.now();
42
+ resetStuckWarnBackoff();
43
+ }
44
+
33
45
  /**
34
46
  * Test-only: reset the activity clocks to "just processed, nothing pending"
35
47
  * at the CURRENT Date.now(). Fake-timer suites need this because each test