talon-agent 3.34.1 → 3.35.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/LICENSE +1 -1
  2. package/README.md +0 -10
  3. package/package.json +3 -3
  4. package/src/backend/codex/auth.ts +40 -1
  5. package/src/backend/codex/handler/message.ts +16 -1
  6. package/src/backend/codex/plan-usage.ts +27 -2
  7. package/src/backend/kilo/handler/message.ts +2 -0
  8. package/src/backend/kilo/server.ts +1 -0
  9. package/src/backend/opencode/handler/message.ts +2 -0
  10. package/src/backend/opencode/server.ts +1 -0
  11. package/src/backend/remote-server/chat-turn.ts +13 -3
  12. package/src/backend/remote-server/lifecycle.ts +37 -0
  13. package/src/backend/remote-server/server-bindings.ts +12 -0
  14. package/src/backend/remote-server/state.ts +8 -0
  15. package/src/backend/remote-server/turn.ts +93 -23
  16. package/src/bootstrap.ts +24 -10
  17. package/src/core/auth/expiry-monitor.ts +89 -0
  18. package/src/core/auth/login-flow.ts +247 -0
  19. package/src/core/auth/status.ts +193 -0
  20. package/src/core/engine/gateway-actions/history.ts +12 -1
  21. package/src/core/errors.ts +1 -1
  22. package/src/core/tools/history.ts +10 -1
  23. package/src/frontend/shared/model-commands.ts +400 -0
  24. package/src/frontend/telegram/auth-panel.ts +205 -0
  25. package/src/frontend/telegram/callbacks/auth.ts +70 -0
  26. package/src/frontend/telegram/callbacks/index.ts +16 -0
  27. package/src/frontend/telegram/callbacks/whatsapp.ts +50 -0
  28. package/src/frontend/telegram/commands/auth.ts +58 -0
  29. package/src/frontend/telegram/commands/definitions.ts +4 -0
  30. package/src/frontend/telegram/commands/index.ts +10 -4
  31. package/src/frontend/telegram/commands/whatsapp-pairing.ts +121 -79
  32. package/src/frontend/whatsapp/actions/chat-info.ts +2 -36
  33. package/src/frontend/whatsapp/actions/history.ts +100 -0
  34. package/src/frontend/whatsapp/actions/index.ts +4 -1
  35. package/src/frontend/whatsapp/commands.ts +371 -0
  36. package/src/frontend/whatsapp/inbound.ts +24 -46
  37. package/src/frontend/whatsapp/index.ts +12 -1
  38. package/src/frontend/whatsapp/media-store.ts +20 -1
  39. package/src/frontend/whatsapp/message-store.ts +101 -24
  40. package/src/storage/history.ts +22 -1
  41. package/src/storage/repositories/history-repo.ts +14 -0
  42. package/src/storage/repositories/whatsapp-messages-repo.ts +101 -0
  43. package/src/storage/sql/history.sql +12 -0
  44. package/src/storage/sql/schema.sql +21 -0
  45. package/src/storage/sql/statements.generated.ts +50 -1
  46. package/src/storage/sql/whatsapp-messages.sql +29 -0
  47. package/src/storage/whatsapp-messages.ts +51 -0
package/LICENSE CHANGED
@@ -1,6 +1,6 @@
1
1
  MIT License
2
2
 
3
- Copyright (c) 2025 Dylan Neve
3
+ Copyright (c) 2026 Dylan Neve
4
4
 
5
5
  Permission is hereby granted, free of charge, to any person obtaining a copy
6
6
  of this software and associated documentation files (the "Software"), to deal
package/README.md CHANGED
@@ -11,7 +11,6 @@
11
11
  [![Backends](https://img.shields.io/badge/backends-Claude_%7C_Kilo_%7C_OpenCode_%7C_Codex_%7C_OpenAI_Agents-D97706)](#backends)
12
12
  [![License: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
13
13
  [![CI](https://github.com/dylanneve1/talon/actions/workflows/ci.yml/badge.svg)](https://github.com/dylanneve1/talon/actions/workflows/ci.yml)
14
- [![Sponsor](https://img.shields.io/badge/Sponsor-%E2%9D%A4-db61a2?logo=githubsponsors&logoColor=white)](https://github.com/sponsors/dylanneve1)
15
14
 
16
15
  Multi-platform agentic AI harness. Runs on **Telegram**, **WhatsApp**, **Discord**, **Microsoft Teams**, the **Terminal**, and a **cross-platform Desktop/Mobile companion app** (Flutter), with a pluggable backend (**Claude Agent SDK**, **Kilo**, **OpenCode**, **Codex**, or **OpenAI Agents**) and full tool access through MCP.
17
16
 
@@ -560,15 +559,6 @@ smoke-tests the CLI and the MCP supervisor from it.
560
559
 
561
560
  ---
562
561
 
563
- ## Support
564
-
565
- Talon is free and MIT-licensed, built and maintained in the open. If it's useful to you, sponsoring helps cover hosting and model costs and funds continued development — and keeps it free for everyone.
566
-
567
- **[❤️ Sponsor Talon on GitHub](https://github.com/sponsors/dylanneve1)**
568
-
569
- Even a one-time tip makes a difference, and every sponsor is appreciated. Starring the repo helps too.
570
-
571
- ---
572
562
 
573
563
  ## License
574
564
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "3.34.1",
3
+ "version": "3.35.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "Dylan Neve",
6
6
  "license": "MIT",
@@ -106,8 +106,8 @@
106
106
  "@grammyjs/transformer-throttler": "^1.2.1",
107
107
  "@kilocode/sdk": "^7.2.22",
108
108
  "@modelcontextprotocol/sdk": "^1.29.0",
109
- "@openai/agents": "^0.17.0",
110
- "@openai/codex-sdk": "^0.153.2",
109
+ "@openai/agents": "^0.18.0",
110
+ "@openai/codex-sdk": "^0.154.0",
111
111
  "@opencode-ai/sdk": "^1.17.4",
112
112
  "@playwright/mcp": "0.0.80",
113
113
  "@types/cross-spawn": "^6.0.6",
@@ -43,6 +43,7 @@
43
43
 
44
44
  import { existsSync, readFileSync } from "node:fs";
45
45
  import { join } from "node:path";
46
+ import { TalonError } from "../../core/errors.js";
46
47
 
47
48
  /** Detected auth mode. */
48
49
  type CodexAuthMode = "api-key" | "chatgpt" | "none";
@@ -321,7 +322,9 @@ export function isChatGptModelMismatchError(message: string): boolean {
321
322
  * - Error text contains `"Codex Exec exited"` (SDK's wrapper);
322
323
  * - Error text contains `"Reading prompt from stdin"` (CLI banner);
323
324
  * - Error text does NOT contain the explicit mismatch phrase (the
324
- * other detector handles that case).
325
+ * other detector handles that case);
326
+ * - Error text does NOT carry the OAuth refresh failure (an expired
327
+ * login exits with the same banner — `isCodexRefreshTokenError`).
325
328
  *
326
329
  * Callers MUST additionally check `authInfo.mode === "chatgpt"` and
327
330
  * that the model isn't already the OAuth default before falling back —
@@ -334,8 +337,44 @@ export function isChatGptModelMismatchError(message: string): boolean {
334
337
  export function isSilentOAuthExitError(message: string): boolean {
335
338
  if (!message) return false;
336
339
  if (isChatGptModelMismatchError(message)) return false;
340
+ if (isCodexRefreshTokenError(message)) return false;
337
341
  return (
338
342
  /Codex\s+Exec\s+exited\s+with\s+code\s+\d+/i.test(message) &&
339
343
  /Reading\s+prompt\s+from\s+stdin/i.test(message)
340
344
  );
341
345
  }
346
+
347
+ /**
348
+ * Detect an expired ChatGPT OAuth login.
349
+ *
350
+ * When the stored refresh token has been invalidated (the user logged
351
+ * out elsewhere, or the session was ended server-side) the Codex CLI
352
+ * prints its startup banner, logs
353
+ *
354
+ * `ERROR codex_login::auth::manager: Failed to refresh token: 401
355
+ * Unauthorized: {"error": {"code": "refresh_token_invalidated", …}}`
356
+ *
357
+ * to stderr, and exits 1 before the prompt reaches the model. The SDK
358
+ * folds that stderr into the same `Codex Exec exited with code 1:
359
+ * Reading prompt from stdin...` wrapper the silent OAuth-incompat exit
360
+ * uses, so without this check it is misread as a model mismatch —
361
+ * resetting the thread and retrying on a fallback model that fails the
362
+ * same way. Nothing but `codex login` fixes it.
363
+ */
364
+ export function isCodexRefreshTokenError(message: string): boolean {
365
+ return /Failed\s+to\s+refresh\s+token|refresh_token_invalidated/i.test(
366
+ message,
367
+ );
368
+ }
369
+
370
+ /**
371
+ * The error surfaced to the user for an expired Codex login. Typed
372
+ * `auth` so the shared retry ladder propagates it untouched — no thread
373
+ * reset, no fallback model — and `friendlyMessage` keeps the detail.
374
+ */
375
+ export function codexLoginExpiredError(cause: unknown): TalonError {
376
+ return new TalonError(
377
+ "Codex login expired — run `codex login` to re-authenticate.",
378
+ { reason: "auth", retryable: false, status: 401, cause },
379
+ );
380
+ }
@@ -52,7 +52,9 @@ import {
52
52
  import { getState } from "../state.js";
53
53
  import { ensureCodex, getCodexAuthInfo } from "../init.js";
54
54
  import {
55
+ codexLoginExpiredError,
55
56
  isChatGptModelMismatchError,
57
+ isCodexRefreshTokenError,
56
58
  isSilentOAuthExitError,
57
59
  } from "../auth.js";
58
60
  import {
@@ -81,6 +83,19 @@ const isTerminatorAbort = (state: StreamState, err: unknown): boolean =>
81
83
  state.turnTerminated &&
82
84
  (errMsg(err) === "AbortError" || /abort/i.test(errMsg(err)));
83
85
 
86
+ /**
87
+ * Swap an expired-login exit for the user-facing auth error before the
88
+ * shared retry ladder classifies it. The raw SDK text is the CLI banner
89
+ * plus a stderr dump (see `isCodexRefreshTokenError`); left alone it
90
+ * reads as an opaque exit-1 and the user never learns that `codex
91
+ * login` is the fix. Any other error passes through unchanged.
92
+ */
93
+ function surfaceLoginExpiry(err: unknown, turnFailedError?: string): unknown {
94
+ return isCodexRefreshTokenError(`${turnFailedError ?? ""} ${errMsg(err)}`)
95
+ ? codexLoginExpiredError(err)
96
+ : err;
97
+ }
98
+
84
99
  /**
85
100
  * One-shot ChatGPT-OAuth model-mismatch recovery.
86
101
  *
@@ -356,7 +371,7 @@ async function recoverCodexFailure(inputs: {
356
371
  if (fallback) return fallback;
357
372
 
358
373
  const decision = await applyRetryDecision({
359
- err,
374
+ err: surfaceLoginExpiry(err, outcome.turnFailedError),
360
375
  chatId,
361
376
  activeModel,
362
377
  retried,
@@ -11,7 +11,7 @@
11
11
  * missing or expired token, or any transport failure.
12
12
  */
13
13
 
14
- import { readFile } from "node:fs/promises";
14
+ import { readFile, stat } from "node:fs/promises";
15
15
  import { homedir } from "node:os";
16
16
  import { join } from "node:path";
17
17
  import { logWarn } from "../../util/log.js";
@@ -26,6 +26,12 @@ const CACHE_TTL_MS = 60_000;
26
26
 
27
27
  let cache: { value: PlanUsage; fetchedAt: number } | undefined;
28
28
  let inFlight: Promise<PlanUsage | undefined> | undefined;
29
+ /**
30
+ * `auth.json` mtime of a token the endpoint rejected with 401. An
31
+ * invalidated login stays invalid until `codex login` rewrites the file,
32
+ * so the poller warns once and stops calling until the mtime moves.
33
+ */
34
+ let rejectedAuthMtimeMs: number | undefined;
29
35
 
30
36
  function authPath(): string {
31
37
  const home = process.env.CODEX_HOME?.trim();
@@ -37,11 +43,18 @@ function authPath(): string {
37
43
  interface CodexAuth {
38
44
  accessToken: string;
39
45
  accountId?: string;
46
+ /** `auth.json` mtime — identifies the login the token came from. */
47
+ mtimeMs: number;
40
48
  }
41
49
 
42
50
  async function readAuth(): Promise<CodexAuth | undefined> {
43
51
  try {
44
- const parsed = JSON.parse(await readFile(authPath(), "utf8")) as {
52
+ const path = authPath();
53
+ const [raw, stats] = await Promise.all([
54
+ readFile(path, "utf8"),
55
+ stat(path),
56
+ ]);
57
+ const parsed = JSON.parse(raw) as {
45
58
  auth_mode?: string;
46
59
  tokens?: { access_token?: string; account_id?: string };
47
60
  };
@@ -52,6 +65,7 @@ async function readAuth(): Promise<CodexAuth | undefined> {
52
65
  ...(parsed.tokens?.account_id
53
66
  ? { accountId: parsed.tokens.account_id }
54
67
  : {}),
68
+ mtimeMs: stats.mtimeMs,
55
69
  };
56
70
  } catch {
57
71
  return undefined;
@@ -123,6 +137,8 @@ export function parseCodexUsage(body: unknown): PlanUsage | undefined {
123
137
  async function load(): Promise<PlanUsage | undefined> {
124
138
  const auth = await readAuth();
125
139
  if (!auth) return undefined;
140
+ if (auth.mtimeMs === rejectedAuthMtimeMs) return undefined;
141
+ rejectedAuthMtimeMs = undefined;
126
142
 
127
143
  try {
128
144
  const res = await fetch(USAGE_ENDPOINT, {
@@ -133,6 +149,15 @@ async function load(): Promise<PlanUsage | undefined> {
133
149
  },
134
150
  signal: AbortSignal.timeout(REQUEST_TIMEOUT_MS),
135
151
  });
152
+ if (res.status === 401) {
153
+ rejectedAuthMtimeMs = auth.mtimeMs;
154
+ logWarn(
155
+ "agent",
156
+ "codex usage: endpoint returned 401 — Codex login expired; " +
157
+ "run `codex login` (not retried until auth.json changes)",
158
+ );
159
+ return undefined;
160
+ }
136
161
  if (!res.ok) {
137
162
  logWarn("agent", `codex usage: endpoint returned ${res.status}`);
138
163
  return undefined;
@@ -7,6 +7,7 @@ import type { QueryParams, QueryResult } from "../../shared/handler-types.js";
7
7
  import { runRemoteChatTurn } from "../../remote-server/chat-turn.js";
8
8
  import {
9
9
  ensureServer,
10
+ trackActiveTurn,
10
11
  ensureSession,
11
12
  ensureChatMcpServer,
12
13
  ensurePluginMcpServers,
@@ -26,6 +27,7 @@ export function handleMessage(params: QueryParams): Promise<QueryResult> {
26
27
  label: "Kilo",
27
28
  getConfig,
28
29
  ensureServer,
30
+ trackActiveTurn,
29
31
  parseModelSelection: parseStoredKiloModelSelection,
30
32
  resolveProviderID,
31
33
  ensureSession,
@@ -82,6 +82,7 @@ export const stopKiloServer = kilo.stop;
82
82
  export const {
83
83
  onServerStop,
84
84
  ensureServer,
85
+ trackActiveTurn,
85
86
  ensureChatMcpServer,
86
87
  ensurePluginMcpServers,
87
88
  buildToolOverrides,
@@ -7,6 +7,7 @@ import type { QueryParams, QueryResult } from "../../shared/handler-types.js";
7
7
  import { runRemoteChatTurn } from "../../remote-server/chat-turn.js";
8
8
  import {
9
9
  ensureServer,
10
+ trackActiveTurn,
10
11
  ensureSession,
11
12
  ensureChatMcpServer,
12
13
  ensurePluginMcpServers,
@@ -26,6 +27,7 @@ export function handleMessage(params: QueryParams): Promise<QueryResult> {
26
27
  label: "OpenCode",
27
28
  getConfig,
28
29
  ensureServer,
30
+ trackActiveTurn,
29
31
  parseModelSelection: parseStoredOpenCodeModelSelection,
30
32
  resolveProviderID,
31
33
  ensureSession,
@@ -64,6 +64,7 @@ export const stopOpenCodeServer = opencode.stop;
64
64
  export const {
65
65
  onServerStop,
66
66
  ensureServer,
67
+ trackActiveTurn,
67
68
  ensureChatMcpServer,
68
69
  ensurePluginMcpServers,
69
70
  buildToolOverrides,
@@ -45,6 +45,7 @@ export type RemoteChatBindings<TClient extends RemoteAgentClient> = Pick<
45
45
  RemoteServerBindings<TClient>,
46
46
  | "getConfig"
47
47
  | "ensureServer"
48
+ | "trackActiveTurn"
48
49
  | "parseModelSelection"
49
50
  | "resolveProviderID"
50
51
  | "ensureSession"
@@ -145,6 +146,11 @@ export async function runRemoteChatTurn<TClient extends RemoteAgentClient>(
145
146
  state.turnTerminated = true;
146
147
  await turnClient.session.abort({ sessionID: sessionId });
147
148
  });
149
+ // Stopping the backend mid-turn (a `/model` swap, shutdown) aborts this
150
+ // controller, which rejects the turn promptly instead of leaving it to
151
+ // the deadline.
152
+ const stopController = new AbortController();
153
+ const untrackTurn = bindings.trackActiveTurn(stopController);
148
154
  const promptStartedAt = Date.now();
149
155
 
150
156
  const setupMs = Date.now() - t0;
@@ -175,6 +181,7 @@ export async function runRemoteChatTurn<TClient extends RemoteAgentClient>(
175
181
  onStreamDelta: undefined,
176
182
  onTextBlock,
177
183
  onToolUse,
184
+ stopSignal: stopController.signal,
178
185
  });
179
186
  promptMs = Date.now() - turnStart;
180
187
  } catch (err) {
@@ -205,6 +212,7 @@ export async function runRemoteChatTurn<TClient extends RemoteAgentClient>(
205
212
  );
206
213
  throw outcome.classified;
207
214
  } finally {
215
+ untrackTurn();
208
216
  unregisterInterrupt();
209
217
  // Note: we deliberately do NOT disconnect the chat MCP server here.
210
218
  // The server is named per-chat so it's safe to keep across turns;
@@ -256,14 +264,16 @@ export async function runRemoteChatTurn<TClient extends RemoteAgentClient>(
256
264
  }
257
265
 
258
266
  /**
259
- * If the SSE loop missed any usage info, fall back to the session
260
- * summary endpoint (which always reflects the final server state).
267
+ * If the SSE loop missed the usage info, fall back to the session summary
268
+ * endpoint (which always reflects the final server state). Best-effort:
269
+ * session summaries can race on cancellation, so a failure leaves the
270
+ * counts at zero.
261
271
  */
262
272
  async function fillUsageFromSummary(
263
273
  oc: RemoteSessionClient,
264
274
  sessionId: string,
265
275
  promptStartedAt: number,
266
- state: StreamState,
276
+ state: ReturnType<typeof createStreamState>,
267
277
  ): Promise<void> {
268
278
  if (
269
279
  state.sdkInputTokens !== 0 ||
@@ -13,11 +13,29 @@
13
13
  * reuse machinery.
14
14
  */
15
15
 
16
+ import { TalonError } from "../../core/errors.js";
16
17
  import { log, logWarn } from "../../util/log.js";
17
18
  import type { RemoteAgentClient } from "./client.js";
18
19
  import type { RemoteServerState } from "./state.js";
19
20
  import { errMsg } from "./state.js";
20
21
 
22
+ /**
23
+ * Rejection for a turn whose server was stopped underneath it (a backend
24
+ * hot-swap or shutdown). Non-retryable on purpose: the chat no longer
25
+ * points at this backend, so a fallback-model retry against it would be
26
+ * wrong — the turn fails cleanly and the next message runs on the new
27
+ * backend.
28
+ */
29
+ export class RemoteServerStoppedError extends TalonError {
30
+ constructor(label: string) {
31
+ super(`${label} server stopped while the turn was in flight`, {
32
+ reason: "unknown",
33
+ retryable: false,
34
+ });
35
+ this.name = "RemoteServerStoppedError";
36
+ }
37
+ }
38
+
21
39
  /**
22
40
  * Inputs passed to {@link ensureRemoteServer}. The backend supplies its
23
41
  * SDK's client/server factories; the shared helper handles the
@@ -180,6 +198,7 @@ export function stopRemoteServer<TClient extends RemoteAgentClient>(
180
198
  extraCleanup?: () => void,
181
199
  ): void {
182
200
  state.clientPromise = null;
201
+ abortActiveTurns(state);
183
202
  state.modelProviderCache.clear();
184
203
  state.registeredMcpServers.clear();
185
204
  state.registeredMcpTools.clear();
@@ -201,3 +220,21 @@ export function stopRemoteServer<TClient extends RemoteAgentClient>(
201
220
  // reused and is intentionally left running for its external owner.
202
221
  state.client = null;
203
222
  }
223
+
224
+ /**
225
+ * Reject every in-flight turn with {@link RemoteServerStoppedError} before
226
+ * the server goes away. The turn's own cleanup unregisters its controller;
227
+ * clearing here only covers turns that never reach their `finally`.
228
+ */
229
+ function abortActiveTurns<TClient extends RemoteAgentClient>(
230
+ state: RemoteServerState<TClient>,
231
+ ): void {
232
+ if (state.activeTurns.size === 0) return;
233
+ log(
234
+ "agent",
235
+ `${state.label} stopping with ${state.activeTurns.size} turn(s) in flight; aborting them`,
236
+ );
237
+ const reason = new RemoteServerStoppedError(state.label);
238
+ for (const controller of state.activeTurns) controller.abort(reason);
239
+ state.activeTurns.clear();
240
+ }
@@ -107,6 +107,12 @@ export interface RemoteServerBindings<TClient extends RemoteAgentClient> {
107
107
  stop(): void;
108
108
  /** Lazily start (or reuse) the local server and return a strict client. */
109
109
  ensureServer(): Promise<TClient>;
110
+ /**
111
+ * Register a chat turn's abort controller for the lifetime of the turn.
112
+ * `stop()` aborts every registered controller so no turn outlives its
113
+ * server. Returns the unregister function; call it when the turn ends.
114
+ */
115
+ trackActiveTurn(controller: AbortController): () => void;
110
116
  ensureChatMcpServer(oc: TClient, chatId: string): Promise<string>;
111
117
  ensurePluginMcpServers(oc: TClient, chatId: string): Promise<string[]>;
112
118
  /**
@@ -193,6 +199,12 @@ export function bindRemoteServer<TClient extends RemoteAgentClient>(
193
199
  });
194
200
  },
195
201
  ensureServer,
202
+ trackActiveTurn(controller) {
203
+ state.activeTurns.add(controller);
204
+ return () => {
205
+ state.activeTurns.delete(controller);
206
+ };
207
+ },
196
208
  ensureChatMcpServer,
197
209
  ensurePluginMcpServers,
198
210
  buildToolOverrides: (oc, chatServerName, pluginServerNames = []) =>
@@ -56,6 +56,13 @@ export interface RemoteServerState<TClient extends RemoteAgentClient> {
56
56
  readonly registeredMcpTools: Map<string, readonly string[]>;
57
57
  /** Plugin-name → registered server-name mapping for each chat context. */
58
58
  readonly pluginMcpServersByChat: Map<string, Map<string, string>>;
59
+ /**
60
+ * One controller per in-flight chat turn. `stopRemoteServer` aborts them
61
+ * so a turn cannot outlive the server it is streaming from — without
62
+ * this, a mid-turn backend swap left the SSE await hanging until the
63
+ * 600s deadline while the question watchdog hammered the dead socket.
64
+ */
65
+ readonly activeTurns: Set<AbortController>;
59
66
  }
60
67
 
61
68
  /** Inputs for {@link createRemoteServerState}. */
@@ -90,6 +97,7 @@ export function createRemoteServerState<TClient extends RemoteAgentClient>(
90
97
  registeredMcpServers: new Set(),
91
98
  registeredMcpTools: new Map(),
92
99
  pluginMcpServersByChat: new Map(),
100
+ activeTurns: new Set(),
93
101
  };
94
102
  }
95
103
 
@@ -69,8 +69,22 @@ export interface RunRemoteTurnInputs {
69
69
  onStreamDelta?: (accumulated: string, phase?: "thinking" | "text") => void;
70
70
  onTextBlock?: (text: string) => Promise<void>;
71
71
  onToolUse?: (toolName: string, input: Record<string, unknown>) => void;
72
+ /**
73
+ * Fires when the backend is stopped underneath the turn (hot-swap or
74
+ * shutdown). Its `reason` is the error the turn rejects with.
75
+ */
76
+ stopSignal?: AbortSignal;
72
77
  }
73
78
 
79
+ /** Poll cadence for the headless question/permission watchdog. */
80
+ const WATCHDOG_INTERVAL_MS = 350;
81
+ /**
82
+ * Consecutive poll failures before the watchdog gives up. A server that
83
+ * refuses three polls in a row is gone; polling on would only log
84
+ * `fetch failed` every 350ms until the turn is torn down.
85
+ */
86
+ const WATCHDOG_MAX_CONSECUTIVE_FAILURES = 3;
87
+
74
88
  /**
75
89
  * Run one turn end-to-end.
76
90
  *
@@ -105,6 +119,7 @@ export async function runRemoteTurn(
105
119
  onStreamDelta,
106
120
  onTextBlock,
107
121
  onToolUse,
122
+ stopSignal,
108
123
  } = inputs;
109
124
  const sessionClient = oc as unknown as RemoteSessionClient;
110
125
 
@@ -112,6 +127,10 @@ export async function runRemoteTurn(
112
127
  // `message.part.updated` events can fire immediately after promptAsync
113
128
  // returns, so the iterator must already be alive.
114
129
  const sseAbort = new AbortController();
130
+ // A backend stop ends the SSE loop and the watchdog the same way a
131
+ // finished turn does; the turn itself rejects via `rejectWhenStopped`.
132
+ const onStop = (): void => sseAbort.abort();
133
+ stopSignal?.addEventListener("abort", onStop, { once: true });
115
134
  const sseDone = subscribeToTurnEvents({
116
135
  label,
117
136
  oc,
@@ -144,35 +163,30 @@ export async function runRemoteTurn(
144
163
  label,
145
164
  ),
146
165
  ]);
147
- const questionWatchdog = (async () => {
148
- while (!sseAbort.signal.aborted) {
149
- try {
150
- await settlePending();
151
- } catch (err) {
152
- logWarn(
153
- "agent",
154
- `[${chatId}] question watchdog failed: ${errMsg(err)}`,
155
- );
156
- }
157
- await sleep(350, sseAbort.signal);
158
- }
159
- })();
166
+ const questionWatchdog = runQuestionWatchdog({
167
+ settlePending,
168
+ signal: sseAbort.signal,
169
+ chatId,
170
+ });
160
171
 
161
172
  try {
162
173
  // Fire and forget — promptAsync returns immediately. The await below
163
174
  // is on the SSE close event. Per-prompt overrides hide sibling chats'
164
175
  // MCP tools while session permissions independently deny execution.
165
176
  await awaitRemoteTurn(
166
- (async () => {
167
- await oc.session.promptAsync({
168
- sessionID: sessionId,
169
- parts: [{ type: "text", text: prompt }],
170
- model: { providerID, modelID },
171
- system: systemPrompt,
172
- ...(toolOverrides ? { tools: toolOverrides } : {}),
173
- });
174
- await sseDone;
175
- })(),
177
+ rejectWhenStopped(
178
+ (async () => {
179
+ await oc.session.promptAsync({
180
+ sessionID: sessionId,
181
+ parts: [{ type: "text", text: prompt }],
182
+ model: { providerID, modelID },
183
+ system: systemPrompt,
184
+ ...(toolOverrides ? { tools: toolOverrides } : {}),
185
+ });
186
+ await sseDone;
187
+ })(),
188
+ stopSignal,
189
+ ),
176
190
  { client: oc, sessionId, chatId, label },
177
191
  );
178
192
 
@@ -214,6 +228,7 @@ export async function runRemoteTurn(
214
228
  }
215
229
  throw err;
216
230
  } finally {
231
+ stopSignal?.removeEventListener("abort", onStop);
217
232
  sseAbort.abort();
218
233
  // A dead SSE socket may ignore the local abort flag until another event
219
234
  // arrives. Bound cleanup so a timed-out turn cannot wedge its caller in
@@ -228,6 +243,61 @@ export async function runRemoteTurn(
228
243
  }
229
244
  }
230
245
 
246
+ /**
247
+ * Race the turn against a backend stop. A stop rejects with the signal's
248
+ * `reason` (a `RemoteServerStoppedError`) the moment it fires, instead of
249
+ * waiting for an SSE socket that will never close on its own.
250
+ */
251
+ function rejectWhenStopped<T>(
252
+ turn: Promise<T>,
253
+ stopSignal: AbortSignal | undefined,
254
+ ): Promise<T> {
255
+ if (!stopSignal) return turn;
256
+ if (stopSignal.aborted) return Promise.reject(stopSignal.reason);
257
+ return new Promise<T>((resolve, reject) => {
258
+ const onAbort = (): void => reject(stopSignal.reason);
259
+ stopSignal.addEventListener("abort", onAbort, { once: true });
260
+ turn.then(resolve, reject).finally(() => {
261
+ stopSignal.removeEventListener("abort", onAbort);
262
+ });
263
+ });
264
+ }
265
+
266
+ /**
267
+ * Poll the pending question/permission lists until the turn's signal
268
+ * fires. Gives up after {@link WATCHDOG_MAX_CONSECUTIVE_FAILURES} failed
269
+ * polls in a row — a server that stopped answering is not coming back
270
+ * within this turn, and the finally-block settle runs once more anyway.
271
+ */
272
+ async function runQuestionWatchdog(inputs: {
273
+ settlePending: () => Promise<unknown>;
274
+ signal: AbortSignal;
275
+ chatId: string;
276
+ }): Promise<void> {
277
+ const { settlePending, signal, chatId } = inputs;
278
+ let consecutiveFailures = 0;
279
+ while (!signal.aborted) {
280
+ try {
281
+ await settlePending();
282
+ consecutiveFailures = 0;
283
+ } catch (err) {
284
+ consecutiveFailures += 1;
285
+ logWarn(
286
+ "agent",
287
+ `[${chatId}] question watchdog failed (${consecutiveFailures}/${WATCHDOG_MAX_CONSECUTIVE_FAILURES}): ${errMsg(err)}`,
288
+ );
289
+ if (consecutiveFailures >= WATCHDOG_MAX_CONSECUTIVE_FAILURES) {
290
+ logWarn(
291
+ "agent",
292
+ `[${chatId}] question watchdog stopped: server unreachable`,
293
+ );
294
+ return;
295
+ }
296
+ }
297
+ await sleep(WATCHDOG_INTERVAL_MS, signal);
298
+ }
299
+ }
300
+
231
301
  interface SubscribeInputs {
232
302
  label: string;
233
303
  oc: RemoteTurnClient;
package/src/bootstrap.ts CHANGED
@@ -28,6 +28,7 @@ import { initPulse, resetPulseTimer } from "./core/background/pulse.js";
28
28
  import { initCron } from "./core/background/cron.js";
29
29
  import { initPlanAlerts } from "./core/background/plan-alerts.js";
30
30
  import { setAdminNotifier } from "./core/notify.js";
31
+ import { startAuthExpiryMonitor } from "./core/auth/expiry-monitor.js";
31
32
  import {
32
33
  initTriggers,
33
34
  resumeAfterRestart as resumeTriggersAfterRestart,
@@ -486,16 +487,7 @@ export async function initBackendAndDispatcher(
486
487
  // first consumer is WhatsApp pairing: codes must travel over a LIVE
487
488
  // frontend, not the dead one's log). Same delivery route as the plan
488
489
  // alerts above.
489
- if (config.adminUserId) {
490
- const adminChatId = config.adminUserId;
491
- setAdminNotifier(async (text: string) =>
492
- resolveFrontendByNumericId(
493
- adminChatId,
494
- String(adminChatId),
495
- frontends,
496
- ).sendMessage(adminChatId, text),
497
- );
498
- }
490
+ wireAdminNotifier(config, frontends);
499
491
 
500
492
  // Soul — initialize the identity kernel singleton from config so the prompt
501
493
  // injection / dream hooks see the right enabled state. Off by default; a
@@ -586,3 +578,25 @@ export async function initBackendAndDispatcher(
586
578
 
587
579
  return { backend };
588
580
  }
581
+
582
+ /**
583
+ * Wire the admin notification seam to the admin's frontend, and start the
584
+ * login-expiry monitor that rides it: the CLIs' "N days to log in again"
585
+ * banner, delivered to the admin instead of a terminal nobody is watching
586
+ * (/auth then completes the sign-in from the chat).
587
+ */
588
+ function wireAdminNotifier(
589
+ config: TalonConfig,
590
+ frontends: Parameters<typeof resolveFrontendByNumericId>[2],
591
+ ): void {
592
+ if (!config.adminUserId) return;
593
+ const adminChatId = config.adminUserId;
594
+ setAdminNotifier(async (text: string) =>
595
+ resolveFrontendByNumericId(
596
+ adminChatId,
597
+ String(adminChatId),
598
+ frontends,
599
+ ).sendMessage(adminChatId, text),
600
+ );
601
+ startAuthExpiryMonitor();
602
+ }