talon-agent 3.22.0 → 3.22.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/app.ts +15 -5
- package/src/backend/claude-sdk/handler.ts +20 -11
- package/src/backend/remote-server/mcp.ts +15 -2
- package/src/backend/shared/cache-telemetry.ts +42 -0
- package/src/backend/shared/index.ts +3 -2
- package/src/core/engine/dispatcher.ts +10 -0
- package/src/core/errors.ts +11 -0
- package/src/core/mcp-hub/children.ts +45 -1
- package/src/core/tasks/table.ts +14 -0
- package/src/frontend/discord/handlers/queue.ts +22 -2
- package/src/frontend/telegram/actions/shared.ts +28 -6
- package/src/frontend/telegram/handlers/delivery.ts +3 -3
- package/src/frontend/telegram/handlers/queue.ts +23 -3
- package/src/util/watchdog.ts +12 -0
package/package.json
CHANGED
package/src/app.ts
CHANGED
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
import { getFrontends } from "./util/config.js";
|
|
10
10
|
import { startUploadCleanup, stopUploadCleanup } from "./util/workspace.js";
|
|
11
11
|
import { flushDatabase } from "./storage/db.js";
|
|
12
|
-
import { getActiveCount } from "./core/engine/dispatcher.js";
|
|
12
|
+
import { getActiveCount, stopAllTurns } from "./core/engine/dispatcher.js";
|
|
13
13
|
import { startPulseTimer, stopPulseTimer } from "./core/background/pulse.js";
|
|
14
14
|
import { stopPlanAlerts } from "./core/background/plan-alerts.js";
|
|
15
15
|
import {
|
|
@@ -121,6 +121,7 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
121
121
|
shuttingDown = true;
|
|
122
122
|
log("shutdown", `${signal} received, shutting down gracefully...`);
|
|
123
123
|
|
|
124
|
+
const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
|
|
124
125
|
const forceTimer = setTimeout(() => {
|
|
125
126
|
logError("shutdown", "Timeout exceeded, forcing exit");
|
|
126
127
|
// Hand off even on the forced path. A restart must survive a
|
|
@@ -135,12 +136,17 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
135
136
|
}, SHUTDOWN_TIMEOUT_MS);
|
|
136
137
|
forceTimer.unref();
|
|
137
138
|
|
|
138
|
-
// Drain in-flight queries
|
|
139
|
-
//
|
|
139
|
+
// Drain in-flight queries. A turn can legitimately run for minutes, so a
|
|
140
|
+
// drain that only waits can never succeed against one — ask every running
|
|
141
|
+
// turn to abort first, then poll for the aborts to settle so backends can
|
|
142
|
+
// flush partial state before the process exits.
|
|
140
143
|
if (getActiveCount() > 0) {
|
|
144
|
+
const aborted = stopAllTurns();
|
|
141
145
|
log(
|
|
142
146
|
"shutdown",
|
|
143
|
-
`Waiting for ${getActiveCount()} in-flight queries to drain
|
|
147
|
+
`Waiting for ${getActiveCount()} in-flight queries to drain` +
|
|
148
|
+
(aborted > 0 ? ` (abort requested for ${aborted})` : "") +
|
|
149
|
+
`...`,
|
|
144
150
|
);
|
|
145
151
|
const deadline = Date.now() + DRAIN_TIMEOUT_MS;
|
|
146
152
|
while (getActiveCount() > 0 && Date.now() < deadline) {
|
|
@@ -177,7 +183,11 @@ async function gracefulShutdown(signal: string): Promise<void> {
|
|
|
177
183
|
await shutdownStep("pulse timer", stopPulseTimer);
|
|
178
184
|
await shutdownStep("heartbeat", async () => {
|
|
179
185
|
stopHeartbeatTimer();
|
|
180
|
-
|
|
186
|
+
// Cap the wait by what's left of the force-timer budget (minus margin
|
|
187
|
+
// for the steps below). The default 10s wait plus a full 5s drain used
|
|
188
|
+
// to consume the entire 15s budget, so a slow heartbeat tripped the
|
|
189
|
+
// forced exit even though teardown was proceeding normally.
|
|
190
|
+
await awaitHeartbeat(Math.max(0, deadlineAt - Date.now() - 3_000));
|
|
181
191
|
});
|
|
182
192
|
await shutdownStep("cron timer", stopCronTimer);
|
|
183
193
|
await shutdownStep("plan alerts", stopPlanAlerts);
|
|
@@ -67,8 +67,9 @@ import {
|
|
|
67
67
|
recordFailedTurnAccounting,
|
|
68
68
|
recordFlowViolation,
|
|
69
69
|
formatTurnCache,
|
|
70
|
-
|
|
71
|
-
|
|
70
|
+
crossTurnVerdict,
|
|
71
|
+
priorLookbackOverflow,
|
|
72
|
+
noteLookbackRisk,
|
|
72
73
|
CACHE_LOOKBACK_BLOCKS,
|
|
73
74
|
} from "../shared/index.js";
|
|
74
75
|
|
|
@@ -594,16 +595,24 @@ export async function* runChatTurn(
|
|
|
594
595
|
|
|
595
596
|
// The aggregate `cache=NN%` can't distinguish a turn that reused the
|
|
596
597
|
// previous turn's prefix from one that re-wrote it — see
|
|
597
|
-
// shared/cache-telemetry.ts.
|
|
598
|
-
//
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
)
|
|
598
|
+
// shared/cache-telemetry.ts. A lookback overflow only *predicts* a miss,
|
|
599
|
+
// so warn when this turn's verdict proves the previous turn's overflow
|
|
600
|
+
// cost a prefix re-write, then record this turn's overflow for the next.
|
|
601
|
+
if (state.cacheStats) {
|
|
602
|
+
const overflow = priorLookbackOverflow(chatId);
|
|
603
|
+
if (
|
|
604
|
+
overflow !== undefined &&
|
|
605
|
+
crossTurnVerdict(state.cacheStats) === "miss"
|
|
606
|
+
) {
|
|
607
|
+
logWarn(
|
|
608
|
+
"agent",
|
|
609
|
+
`[${chatId}] previous turn emitted ~${overflow} content blocks ` +
|
|
610
|
+
`(> ${CACHE_LOOKBACK_BLOCKS} lookback) and this turn's prefix ` +
|
|
611
|
+
`missed — that turn's cache write was likely never read`,
|
|
612
|
+
);
|
|
613
|
+
}
|
|
606
614
|
}
|
|
615
|
+
noteLookbackRisk(chatId, state.toolCalls);
|
|
607
616
|
|
|
608
617
|
log(
|
|
609
618
|
"agent",
|
|
@@ -29,7 +29,7 @@
|
|
|
29
29
|
* registrations run concurrently through the hub; later turns skip them.
|
|
30
30
|
*/
|
|
31
31
|
|
|
32
|
-
import { log, logWarn } from "../../util/log.js";
|
|
32
|
+
import { log, logDebug, logWarn } from "../../util/log.js";
|
|
33
33
|
import {
|
|
34
34
|
talonHubUrl,
|
|
35
35
|
pluginHubUrl,
|
|
@@ -62,6 +62,9 @@ export const TALON_PLUGIN_MCP_SERVER_NAME = "talon-plugin";
|
|
|
62
62
|
*/
|
|
63
63
|
const SLOW_MCP_REGISTRATION_MS = 1000;
|
|
64
64
|
|
|
65
|
+
/** Servers whose registration failure was already warned about (see below). */
|
|
66
|
+
const warnedMcpRegistrationFailures = new Set<string>();
|
|
67
|
+
|
|
65
68
|
/**
|
|
66
69
|
* OpenCode's `/experimental/tool/ids` endpoint omits dynamically registered
|
|
67
70
|
* MCP tools despite its API description. Synthesize Talon's known MCP ids so
|
|
@@ -253,6 +256,7 @@ export async function ensurePluginMcpServers<TClient extends RemoteAgentClient>(
|
|
|
253
256
|
state.registeredMcpServers.add(serverName);
|
|
254
257
|
state.registeredMcpTools.set(serverName, toolNames);
|
|
255
258
|
byPlugin.set(name, serverName);
|
|
259
|
+
warnedMcpRegistrationFailures.delete(serverName);
|
|
256
260
|
const ms = Date.now() - startedAt;
|
|
257
261
|
log(
|
|
258
262
|
"agent",
|
|
@@ -261,7 +265,16 @@ export async function ensurePluginMcpServers<TClient extends RemoteAgentClient>(
|
|
|
261
265
|
);
|
|
262
266
|
return serverName;
|
|
263
267
|
} catch (err) {
|
|
264
|
-
|
|
268
|
+
// Warn once per server, then debug: registration re-runs at the
|
|
269
|
+
// head of every turn, so a dead backing service (an offline
|
|
270
|
+
// browser endpoint, say) would otherwise emit one warning per
|
|
271
|
+
// turn for the whole outage. Success clears the latch so the
|
|
272
|
+
// next outage warns again.
|
|
273
|
+
const level = warnedMcpRegistrationFailures.has(serverName)
|
|
274
|
+
? logDebug
|
|
275
|
+
: logWarn;
|
|
276
|
+
warnedMcpRegistrationFailures.add(serverName);
|
|
277
|
+
level(
|
|
265
278
|
"agent",
|
|
266
279
|
`Plugin MCP registration failed for ${serverName}: ${errMsg(err)}`,
|
|
267
280
|
);
|
|
@@ -154,6 +154,48 @@ export function exceedsLookbackWindow(toolCalls: number): boolean {
|
|
|
154
154
|
return estimateTurnBlocks(toolCalls) > CACHE_LOOKBACK_BLOCKS;
|
|
155
155
|
}
|
|
156
156
|
|
|
157
|
+
/**
|
|
158
|
+
* Per-chat estimated block count of the last turn that overflowed the
|
|
159
|
+
* lookback window. Overflow is only a *prediction* of a cache miss — the
|
|
160
|
+
* proof is the NEXT turn's cross-turn verdict, so the overflow is recorded
|
|
161
|
+
* here and the warning waits for that verdict instead of firing on every
|
|
162
|
+
* tool-heavy turn. Bounded like `lastToolSets` below.
|
|
163
|
+
*/
|
|
164
|
+
const lookbackOverflows = new Map<string, number>();
|
|
165
|
+
|
|
166
|
+
/**
|
|
167
|
+
* Record whether this turn plausibly overflowed the lookback window, so the
|
|
168
|
+
* next turn can attribute a cross-turn miss to it.
|
|
169
|
+
*/
|
|
170
|
+
export function noteLookbackRisk(chatId: string, toolCalls: number): void {
|
|
171
|
+
if (!exceedsLookbackWindow(toolCalls)) {
|
|
172
|
+
lookbackOverflows.delete(chatId);
|
|
173
|
+
return;
|
|
174
|
+
}
|
|
175
|
+
if (
|
|
176
|
+
lookbackOverflows.size >= MAX_TRACKED_CHATS &&
|
|
177
|
+
!lookbackOverflows.has(chatId)
|
|
178
|
+
) {
|
|
179
|
+
const oldest = lookbackOverflows.keys().next().value;
|
|
180
|
+
if (oldest !== undefined) lookbackOverflows.delete(oldest);
|
|
181
|
+
}
|
|
182
|
+
lookbackOverflows.set(chatId, estimateTurnBlocks(toolCalls));
|
|
183
|
+
}
|
|
184
|
+
|
|
185
|
+
/**
|
|
186
|
+
* Estimated block count of the chat's previous turn IF it overflowed the
|
|
187
|
+
* lookback window, else undefined. Read this before `noteLookbackRisk`
|
|
188
|
+
* records the current turn.
|
|
189
|
+
*/
|
|
190
|
+
export function priorLookbackOverflow(chatId: string): number | undefined {
|
|
191
|
+
return lookbackOverflows.get(chatId);
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
/** Drop all recorded overflows (tests / explicit reset). */
|
|
195
|
+
export function resetLookbackRisk(): void {
|
|
196
|
+
lookbackOverflows.clear();
|
|
197
|
+
}
|
|
198
|
+
|
|
157
199
|
// ── Cacheable minimum ──────────────────────────────────────────────────────
|
|
158
200
|
|
|
159
201
|
/**
|
|
@@ -73,8 +73,9 @@ export {
|
|
|
73
73
|
// barrel discipline the prompt/ barrel was just trimmed to.
|
|
74
74
|
export {
|
|
75
75
|
formatTurnCache,
|
|
76
|
-
|
|
77
|
-
|
|
76
|
+
crossTurnVerdict,
|
|
77
|
+
priorLookbackOverflow,
|
|
78
|
+
noteLookbackRisk,
|
|
78
79
|
CACHE_LOOKBACK_BLOCKS,
|
|
79
80
|
} from "./cache-telemetry.js";
|
|
80
81
|
|
|
@@ -66,6 +66,16 @@ export function stopCurrentTurn(chatId: string): KillOutcome {
|
|
|
66
66
|
return taskTable.killRunningTurn(chatId);
|
|
67
67
|
}
|
|
68
68
|
|
|
69
|
+
/**
|
|
70
|
+
* Request an abort of every chat's running turn. The shutdown drain calls
|
|
71
|
+
* this before polling `getActiveCount` — a turn can legitimately run for
|
|
72
|
+
* minutes, so a drain that only waits can never succeed against one.
|
|
73
|
+
* Returns the number of kill requests issued.
|
|
74
|
+
*/
|
|
75
|
+
export function stopAllTurns(): number {
|
|
76
|
+
return taskTable.killAllRunningTurns();
|
|
77
|
+
}
|
|
78
|
+
|
|
69
79
|
/**
|
|
70
80
|
* Execute an AI query with full lifecycle management.
|
|
71
81
|
* Same-chat queries are serialized (FIFO) to avoid session conflicts.
|
package/src/core/errors.ts
CHANGED
|
@@ -33,6 +33,17 @@ type ErrorReason =
|
|
|
33
33
|
const USAGE_LIMIT_RE =
|
|
34
34
|
/you['’]ve hit your .{0,40}limit|you['’]re out of extra usage|claude ai usage limit reached|usage limit reached/i;
|
|
35
35
|
|
|
36
|
+
/**
|
|
37
|
+
* Ceiling on how long a failed attempt may have run and still earn the
|
|
38
|
+
* frontend queues' blind retry. `retryable` means "a short pause may clear
|
|
39
|
+
* it", which holds for a 429 or a dropped socket but not for an attempt
|
|
40
|
+
* that already burned minutes before failing (e.g. the 600s remote turn
|
|
41
|
+
* deadline — its rejection is `name: "TimeoutError"`, classified as
|
|
42
|
+
* transient network). Turns serialize per chat, so retrying such an
|
|
43
|
+
* attempt doubles the stall for every message queued behind it.
|
|
44
|
+
*/
|
|
45
|
+
export const RETRY_ELAPSED_CAP_MS = 120_000;
|
|
46
|
+
|
|
36
47
|
// ── TalonError class ────────────────────────────────────────────────────────
|
|
37
48
|
|
|
38
49
|
export class TalonError extends Error {
|
|
@@ -62,6 +62,31 @@ type ChildEntry = {
|
|
|
62
62
|
const children = new Map<string, ChildEntry>();
|
|
63
63
|
const inflight = new Map<string, Promise<ChildHandle>>();
|
|
64
64
|
|
|
65
|
+
/**
|
|
66
|
+
* Negative cache for spawn failures. A child whose backing service is down
|
|
67
|
+
* (e.g. playwright-tools with its browser endpoint offline) dies at the
|
|
68
|
+
* connect handshake, and without this every single turn re-paid the
|
|
69
|
+
* spawn+handshake (~600ms) and re-logged the failure for the whole outage.
|
|
70
|
+
* Failures back off exponentially; the first attempt after the window
|
|
71
|
+
* clears the entry on success, so recovery costs one turn.
|
|
72
|
+
*/
|
|
73
|
+
type SpawnFailure = { at: number; count: number; error: unknown };
|
|
74
|
+
const spawnFailures = new Map<string, SpawnFailure>();
|
|
75
|
+
const FAILURE_BACKOFF_BASE_MS = 30_000;
|
|
76
|
+
const FAILURE_BACKOFF_MAX_MS = 10 * 60_000;
|
|
77
|
+
|
|
78
|
+
function failureBackoffMs(count: number): number {
|
|
79
|
+
return Math.min(
|
|
80
|
+
FAILURE_BACKOFF_BASE_MS * 2 ** (count - 1),
|
|
81
|
+
FAILURE_BACKOFF_MAX_MS,
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
/** Test seam: forget recorded spawn failures. */
|
|
86
|
+
export function resetSpawnFailures(): void {
|
|
87
|
+
spawnFailures.clear();
|
|
88
|
+
}
|
|
89
|
+
|
|
65
90
|
/** Idle TTL for hub children; tunable for tests / tight deployments. */
|
|
66
91
|
function idleTtlMs(): number {
|
|
67
92
|
const raw = Number(process.env.TALON_MCP_HUB_IDLE_MS);
|
|
@@ -168,9 +193,28 @@ export function acquireChild(
|
|
|
168
193
|
const pending = inflight.get(key);
|
|
169
194
|
if (pending) return pending;
|
|
170
195
|
|
|
196
|
+
const failure = spawnFailures.get(key);
|
|
197
|
+
if (failure && Date.now() - failure.at < failureBackoffMs(failure.count)) {
|
|
198
|
+
return Promise.reject(
|
|
199
|
+
failure.error instanceof Error
|
|
200
|
+
? failure.error
|
|
201
|
+
: new Error(String(failure.error)),
|
|
202
|
+
);
|
|
203
|
+
}
|
|
204
|
+
|
|
171
205
|
const promise = (async () => {
|
|
172
206
|
try {
|
|
173
|
-
|
|
207
|
+
const handle = await spawnChild(key, spec());
|
|
208
|
+
spawnFailures.delete(key);
|
|
209
|
+
return handle;
|
|
210
|
+
} catch (err) {
|
|
211
|
+
const prior = spawnFailures.get(key);
|
|
212
|
+
spawnFailures.set(key, {
|
|
213
|
+
at: Date.now(),
|
|
214
|
+
count: (prior?.count ?? 0) + 1,
|
|
215
|
+
error: err,
|
|
216
|
+
});
|
|
217
|
+
throw err;
|
|
174
218
|
} finally {
|
|
175
219
|
inflight.delete(key);
|
|
176
220
|
}
|
package/src/core/tasks/table.ts
CHANGED
|
@@ -122,6 +122,20 @@ export class TaskTable {
|
|
|
122
122
|
return { ok: true };
|
|
123
123
|
}
|
|
124
124
|
|
|
125
|
+
/**
|
|
126
|
+
* Request an abort of every running turn, regardless of chat — the
|
|
127
|
+
* shutdown drain's lever. Returns the number of kill requests issued.
|
|
128
|
+
*/
|
|
129
|
+
killAllRunningTurns(): number {
|
|
130
|
+
let killed = 0;
|
|
131
|
+
for (const [id, task] of this.live) {
|
|
132
|
+
if (task.record.kind === "turn" && task.record.state === "running") {
|
|
133
|
+
if (this.kill(id).ok) killed++;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
return killed;
|
|
137
|
+
}
|
|
138
|
+
|
|
125
139
|
/**
|
|
126
140
|
* Request an abort for the turn currently running in one chat. Queued turns
|
|
127
141
|
* are deliberately ignored: `/stop` means "stop what is happening now",
|
|
@@ -4,10 +4,15 @@
|
|
|
4
4
|
* notification.
|
|
5
5
|
*/
|
|
6
6
|
|
|
7
|
-
import {
|
|
7
|
+
import {
|
|
8
|
+
classify,
|
|
9
|
+
friendlyMessage,
|
|
10
|
+
RETRY_ELAPSED_CAP_MS,
|
|
11
|
+
} from "../../../core/errors.js";
|
|
8
12
|
import {
|
|
9
13
|
recordMessageProcessed,
|
|
10
14
|
recordMessageReceived,
|
|
15
|
+
recordMessageSettled,
|
|
11
16
|
recordError,
|
|
12
17
|
} from "../../../util/watchdog.js";
|
|
13
18
|
import { appendDailyLog } from "../../../storage/daily-log.js";
|
|
@@ -82,6 +87,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
82
87
|
chatTitle: last.chatTitle,
|
|
83
88
|
});
|
|
84
89
|
|
|
90
|
+
const startedAt = Date.now();
|
|
85
91
|
try {
|
|
86
92
|
await runTurn();
|
|
87
93
|
recordMessageProcessed();
|
|
@@ -94,7 +100,19 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
94
100
|
);
|
|
95
101
|
recordError(classified.message);
|
|
96
102
|
|
|
97
|
-
|
|
103
|
+
// Retry once for transient errors — but only when the failed attempt
|
|
104
|
+
// was actually brief. An attempt that already ran for minutes (a remote
|
|
105
|
+
// turn deadline) won't be saved by a 2s pause, and turns serialize per
|
|
106
|
+
// chat, so the blind retry used to double a 10-minute stall for
|
|
107
|
+
// everything queued behind it.
|
|
108
|
+
const attemptMs = Date.now() - startedAt;
|
|
109
|
+
if (classified.retryable && attemptMs >= RETRY_ELAPSED_CAP_MS) {
|
|
110
|
+
log(
|
|
111
|
+
"bot",
|
|
112
|
+
`[${chatId}] Not retrying ${classified.reason}: the attempt already ran ${Math.round(attemptMs / 1000)}s`,
|
|
113
|
+
);
|
|
114
|
+
}
|
|
115
|
+
if (classified.retryable && attemptMs < RETRY_ELAPSED_CAP_MS) {
|
|
98
116
|
const delayMs = classified.retryAfterMs ?? 2000;
|
|
99
117
|
log(
|
|
100
118
|
"bot",
|
|
@@ -108,6 +126,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
108
126
|
} catch (retryErr) {
|
|
109
127
|
const retryClassified = classify(retryErr);
|
|
110
128
|
logError("bot", `[${chatId}] Retry failed: ${retryClassified.message}`);
|
|
129
|
+
recordMessageSettled();
|
|
111
130
|
// Error-recovery send must never throw — if the network is fully down
|
|
112
131
|
// the queue handler would otherwise propagate up and stall future
|
|
113
132
|
// messages. Best-effort: notify if we can, log + move on otherwise.
|
|
@@ -126,6 +145,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
126
145
|
return;
|
|
127
146
|
}
|
|
128
147
|
}
|
|
148
|
+
recordMessageSettled();
|
|
129
149
|
try {
|
|
130
150
|
await sendChunked(
|
|
131
151
|
last.channel,
|
|
@@ -11,9 +11,31 @@ import { TELEGRAM_MAX_TEXT } from "./types.js";
|
|
|
11
11
|
|
|
12
12
|
export function replyParams(
|
|
13
13
|
body: Record<string, unknown>,
|
|
14
|
-
):
|
|
15
|
-
|
|
16
|
-
|
|
14
|
+
): ReplyParams | undefined {
|
|
15
|
+
return replyParamsFor(
|
|
16
|
+
toPositiveId(body.reply_to ?? body.reply_to_message_id),
|
|
17
|
+
);
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export type ReplyParams = {
|
|
21
|
+
message_id: number;
|
|
22
|
+
allow_sending_without_reply: true;
|
|
23
|
+
};
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Build `reply_parameters` for an outbound send. Always allows sending
|
|
27
|
+
* without the reply: the target message may have been deleted by the time
|
|
28
|
+
* we send (fast-moving groups, cleanup bots), and without this flag
|
|
29
|
+
* Telegram 400s the whole send — every formatting-level fallback then
|
|
30
|
+
* fails identically and the message is lost. A send that arrives
|
|
31
|
+
* un-linked beats one that never arrives.
|
|
32
|
+
*/
|
|
33
|
+
export function replyParamsFor(
|
|
34
|
+
replyTo: number | undefined,
|
|
35
|
+
): ReplyParams | undefined {
|
|
36
|
+
return replyTo !== undefined && replyTo > 0
|
|
37
|
+
? { message_id: replyTo, allow_sending_without_reply: true }
|
|
38
|
+
: undefined;
|
|
17
39
|
}
|
|
18
40
|
|
|
19
41
|
/** Delivery modifiers shared by every outbound send. */
|
|
@@ -125,7 +147,7 @@ export async function sendText(
|
|
|
125
147
|
chatId,
|
|
126
148
|
{ markdown: text },
|
|
127
149
|
{
|
|
128
|
-
reply_parameters: replyTo
|
|
150
|
+
reply_parameters: replyParamsFor(replyTo),
|
|
129
151
|
reply_markup: replyMarkup,
|
|
130
152
|
...opts,
|
|
131
153
|
},
|
|
@@ -140,7 +162,7 @@ export async function sendText(
|
|
|
140
162
|
try {
|
|
141
163
|
const sent = await bot.api.sendMessage(chatId, html, {
|
|
142
164
|
parse_mode: "HTML",
|
|
143
|
-
reply_parameters: replyTo
|
|
165
|
+
reply_parameters: replyParamsFor(replyTo),
|
|
144
166
|
reply_markup: replyMarkup,
|
|
145
167
|
...opts,
|
|
146
168
|
});
|
|
@@ -151,7 +173,7 @@ export async function sendText(
|
|
|
151
173
|
`Legacy HTML send failed; retrying as plain text (chat=${chatId}): ${err instanceof Error ? err.message : err}`,
|
|
152
174
|
);
|
|
153
175
|
const sent = await bot.api.sendMessage(chatId, text, {
|
|
154
|
-
reply_parameters: replyTo
|
|
176
|
+
reply_parameters: replyParamsFor(replyTo),
|
|
155
177
|
reply_markup: replyMarkup,
|
|
156
178
|
...opts,
|
|
157
179
|
});
|
|
@@ -10,7 +10,7 @@ import { toolInputToRecord } from "../../../core/agent-runtime/events.js";
|
|
|
10
10
|
import { appendDailyLogResponse } from "../../../storage/daily-log.js";
|
|
11
11
|
import { stripMcpPrefix } from "../../../core/tools/index.js";
|
|
12
12
|
import { logWarn } from "../../../util/log.js";
|
|
13
|
-
import { sendText } from "../actions/shared.js";
|
|
13
|
+
import { replyParamsFor, sendText } from "../actions/shared.js";
|
|
14
14
|
import { ambientThreadId } from "../topics.js";
|
|
15
15
|
import { trackDmUser } from "./access.js";
|
|
16
16
|
|
|
@@ -22,7 +22,7 @@ export async function sendHtml(
|
|
|
22
22
|
): Promise<number> {
|
|
23
23
|
const params = {
|
|
24
24
|
parse_mode: "HTML" as const,
|
|
25
|
-
reply_parameters: replyToId
|
|
25
|
+
reply_parameters: replyParamsFor(replyToId),
|
|
26
26
|
message_thread_id: ambientThreadId(chatId),
|
|
27
27
|
};
|
|
28
28
|
try {
|
|
@@ -40,7 +40,7 @@ export async function sendHtml(
|
|
|
40
40
|
plain = plain.replace(/<[^>]*>/g, "");
|
|
41
41
|
} while (plain !== prev);
|
|
42
42
|
const sent = await bot.api.sendMessage(chatId, plain, {
|
|
43
|
-
reply_parameters: replyToId
|
|
43
|
+
reply_parameters: replyParamsFor(replyToId),
|
|
44
44
|
message_thread_id: ambientThreadId(chatId),
|
|
45
45
|
});
|
|
46
46
|
return sent.message_id;
|
|
@@ -9,13 +9,18 @@
|
|
|
9
9
|
import type { Bot } from "grammy";
|
|
10
10
|
import type { TalonConfig } from "../../../util/config.js";
|
|
11
11
|
import { escapeHtml } from "../formatting.js";
|
|
12
|
-
import {
|
|
12
|
+
import {
|
|
13
|
+
classify,
|
|
14
|
+
friendlyMessage,
|
|
15
|
+
RETRY_ELAPSED_CAP_MS,
|
|
16
|
+
} from "../../../core/errors.js";
|
|
13
17
|
import {
|
|
14
18
|
getRecentHistory,
|
|
15
19
|
type HistoryMessage,
|
|
16
20
|
} from "../../../storage/history.js";
|
|
17
21
|
import {
|
|
18
22
|
recordMessageProcessed,
|
|
23
|
+
recordMessageSettled,
|
|
19
24
|
recordMessageReceived,
|
|
20
25
|
recordError,
|
|
21
26
|
} from "../../../util/watchdog.js";
|
|
@@ -176,6 +181,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
176
181
|
chatTitle: last.chatTitle,
|
|
177
182
|
});
|
|
178
183
|
|
|
184
|
+
const startedAt = Date.now();
|
|
179
185
|
try {
|
|
180
186
|
await runTurn();
|
|
181
187
|
lastHandledMessageIdByChat.set(chatId, last.messageId);
|
|
@@ -190,8 +196,20 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
190
196
|
);
|
|
191
197
|
recordError(classified.message);
|
|
192
198
|
|
|
193
|
-
// Retry once for transient errors (rate_limit, overloaded, network)
|
|
194
|
-
|
|
199
|
+
// Retry once for transient errors (rate_limit, overloaded, network) —
|
|
200
|
+
// but only when the failed attempt was actually brief. An attempt that
|
|
201
|
+
// already ran for minutes (a remote turn deadline, a slow collapse)
|
|
202
|
+
// won't be saved by a 2s pause, and turns serialize per chat, so the
|
|
203
|
+
// blind retry used to double a 10-minute stall for everything queued
|
|
204
|
+
// behind it.
|
|
205
|
+
const attemptMs = Date.now() - startedAt;
|
|
206
|
+
if (classified.retryable && attemptMs >= RETRY_ELAPSED_CAP_MS) {
|
|
207
|
+
log(
|
|
208
|
+
"bot",
|
|
209
|
+
`[${chatId}] Not retrying ${classified.reason}: the attempt already ran ${Math.round(attemptMs / 1000)}s`,
|
|
210
|
+
);
|
|
211
|
+
}
|
|
212
|
+
if (classified.retryable && attemptMs < RETRY_ELAPSED_CAP_MS) {
|
|
195
213
|
const delayMs = classified.retryAfterMs ?? 2000;
|
|
196
214
|
log(
|
|
197
215
|
"bot",
|
|
@@ -209,6 +227,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
209
227
|
"bot",
|
|
210
228
|
`[${chatId}] [${chatType}] Retry failed: ${retryClassified.message}`,
|
|
211
229
|
);
|
|
230
|
+
recordMessageSettled();
|
|
212
231
|
await sendHtml(
|
|
213
232
|
bot,
|
|
214
233
|
numericChatId,
|
|
@@ -219,6 +238,7 @@ async function flushQueue(chatId: string): Promise<void> {
|
|
|
219
238
|
}
|
|
220
239
|
}
|
|
221
240
|
|
|
241
|
+
recordMessageSettled();
|
|
222
242
|
await sendHtml(
|
|
223
243
|
bot,
|
|
224
244
|
numericChatId,
|
package/src/util/watchdog.ts
CHANGED
|
@@ -30,6 +30,18 @@ export function recordMessageProcessed(): void {
|
|
|
30
30
|
resetStuckWarnBackoff();
|
|
31
31
|
}
|
|
32
32
|
|
|
33
|
+
/**
|
|
34
|
+
* Record that a message's turn settled in failure. The loop is moving —
|
|
35
|
+
* the error was reported and the queue slot freed — so advance the stuck
|
|
36
|
+
* clock without counting a success. Without this, one failed turn leaves
|
|
37
|
+
* `lastProcessedAt` stale forever and the watchdog reports a wedged loop
|
|
38
|
+
* (and an unhealthy bridge) on a bot that is actually idle.
|
|
39
|
+
*/
|
|
40
|
+
export function recordMessageSettled(): void {
|
|
41
|
+
lastProcessedAt = Date.now();
|
|
42
|
+
resetStuckWarnBackoff();
|
|
43
|
+
}
|
|
44
|
+
|
|
33
45
|
/**
|
|
34
46
|
* Test-only: reset the activity clocks to "just processed, nothing pending"
|
|
35
47
|
* at the CURRENT Date.now(). Fake-timer suites need this because each test
|