talon-agent 5.26.5 → 5.27.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "talon-agent",
3
- "version": "5.26.5",
3
+ "version": "5.27.0",
4
4
  "description": "Multi-frontend AI agent with full tool access, streaming, cron jobs, and plugin system",
5
5
  "author": "The Falconry",
6
6
  "license": "Apache-2.0",
package/src/app.ts CHANGED
@@ -26,7 +26,11 @@ import {
26
26
  runStartupCatchup,
27
27
  } from "./core/background/cron/scheduler.js";
28
28
  import { shutdownTriggers } from "./core/background/triggers/index.js";
29
- import { shutdownAgents } from "./core/agents/index.js";
29
+ import {
30
+ interruptAgentsForRestart,
31
+ resumeAgentsAfterRestart,
32
+ shutdownAgents,
33
+ } from "./core/agents/index.js";
30
34
  import { stopBackupScheduler } from "./core/backup/index.js";
31
35
  import { pruneSettledTriggers } from "./storage/triggers.js";
32
36
  import { startWatchdog, stopWatchdog } from "./util/watchdog.js";
@@ -340,6 +344,11 @@ async function gracefulShutdown(signal: string): Promise<void> {
340
344
  if (shuttingDown) return;
341
345
  shuttingDown = true;
342
346
  log("shutdown", `${signal} received, shutting down gracefully...`);
347
+ // Park running sub-agents before anything is torn down: stopping the
348
+ // frontends and the backend pool aborts their runs, and an agent that is
349
+ // not parked first would record that abort as its death. Parked agents
350
+ // stay `running` in the store and the next boot resumes them.
351
+ crashStep("sub-agent park", () => interruptAgentsForRestart());
343
352
 
344
353
  const deadlineAt = Date.now() + SHUTDOWN_TIMEOUT_MS;
345
354
  const forceTimer = setTimeout(() => {
@@ -432,8 +441,8 @@ async function gracefulShutdown(signal: string): Promise<void> {
432
441
  triggerPruneTimer = null;
433
442
  });
434
443
  await shutdownStep("triggers", shutdownTriggers);
435
- // Sub-agents are isolated one-shot runs: aborting them is all the daemon
436
- // can do, and their parents are gone with the process anyway.
444
+ // Sub-agents were parked at the top of the shutdown; this aborts whatever
445
+ // is still running so the process can exit. They resume on the next boot.
437
446
  await shutdownStep("sub-agents", shutdownAgents);
438
447
  await shutdownStep("watchdog", stopWatchdog);
439
448
  await shutdownStep("resource sampler", stopResourceSampler);
@@ -528,6 +537,12 @@ async function main(): Promise<void> {
528
537
  // what follows runs while the daemon is alive — not, as it once did,
529
538
  // hours later during shutdown.
530
539
  await bootPhase("frontends start", () => startFrontends(frontends));
540
+ // Sub-agents the previous process left running come back now that their
541
+ // tools (gateway, frontends) and their parents' wake path are up.
542
+ // Fire-and-forget: a slow backend acquisition must not hold the boot.
543
+ resumeAgentsAfterRestart().catch((err) =>
544
+ logError("agents", "resume after restart failed", err),
545
+ );
531
546
  // Phase 0 accounting (docs/ts-migration-plan.md): the boot is over the
532
547
  // moment the frontends are listening, so the totals are folded into the
533
548
  // metrics store here, from the same uptime figure the log line prints.
@@ -62,6 +62,8 @@ const claudeSdkFactory: BackendFactory = {
62
62
 
63
63
  const background: BackgroundRunner = {
64
64
  runOneShotAgent: (p) => host.runOneShot(p),
65
+ // Honours resumeSessionId (SDK `resume`) and reports the session id.
66
+ supportsResume: true,
65
67
  // Not a protocol-table row yet: subprocess eviction reaches into
66
68
  // the SDK's own children, so it belongs to the host — but the
67
69
  // design's message table has no request for it. Tracked as an open
@@ -21,7 +21,7 @@ import { EFFORT_MAP } from "./constants.js";
21
21
  import { buildMcpServers, buildPluginMcpServers } from "./options.js";
22
22
  import { isBackgroundToolContext } from "../../core/agents/context.js";
23
23
  import { warnIfBelowCacheMinimum } from "../runtime/cache/cache-telemetry.js";
24
- import { emitAssistantText } from "../runtime/one-shot-hooks.js";
24
+ import { emitAssistantText, emitSessionId } from "../runtime/one-shot-hooks.js";
25
25
  import {
26
26
  isOtherLiveDaemon,
27
27
  ownerFromEnviron,
@@ -72,6 +72,8 @@ export async function runOneShotAgent(
72
72
  abortController,
73
73
  appendLog,
74
74
  onAssistantText,
75
+ resumeSessionId,
76
+ onSessionId,
75
77
  } = params;
76
78
 
77
79
  // Reasoning effort is opt-in for background runs (config `heartbeatEffort`
@@ -100,7 +102,17 @@ export async function runOneShotAgent(
100
102
  // dream). Same as chat minus `Agent` — nested sub-agent dispatch from
101
103
  // inside an unattended pass complicates lifecycle tracking.
102
104
  tools: [...ALLOWED_TOOLS_BACKGROUND],
105
+ // A sub-agent interrupted by a daemon restart continues its own SDK
106
+ // session: the transcript (brief, tool calls, results) is intact on disk
107
+ // and the prompt below is only the "you were interrupted" note.
108
+ ...(resumeSessionId ? { resume: resumeSessionId } : {}),
103
109
  };
110
+ if (resumeSessionId) {
111
+ log(
112
+ "agent",
113
+ `[${contextLabel}] Claude one-shot resuming session ${resumeSessionId}`,
114
+ );
115
+ }
104
116
 
105
117
  if (reasoningEffort && !thinkingConfig) {
106
118
  // `minimal` / `xhigh` are Codex-side vocabulary with no Claude
@@ -132,7 +144,16 @@ export async function runOneShotAgent(
132
144
  // The final `result` message carries the run's total token usage — the
133
145
  // settlement figure the task table records.
134
146
  let usage: OneShotUsage | undefined;
147
+ let sessionReported: string | undefined;
135
148
  for await (const msg of qi) {
149
+ // Every SDK message carries the session id; report it the first time it
150
+ // is seen (and again only if it ever changes — a resumed session can be
151
+ // forked to a new id).
152
+ const sid = (msg as { session_id?: unknown }).session_id;
153
+ if (typeof sid === "string" && sid && sid !== sessionReported) {
154
+ sessionReported = sid;
155
+ emitSessionId(onSessionId, sid);
156
+ }
136
157
  await formatAndAppendMessage(appendLog, msg, onAssistantText);
137
158
  if (msg.type === "result") {
138
159
  const u = msg.usage;
@@ -60,6 +60,8 @@ const codexFactory: BackendFactory = {
60
60
 
61
61
  const background: BackgroundRunner = {
62
62
  runOneShotAgent: (p) => codexRunOneShotAgent(p),
63
+ // Honours resumeSessionId (resumeThread) and reports the thread id.
64
+ supportsResume: true,
63
65
  // Codex spawns per-turn CLI subprocesses that the SDK reaps on its own
64
66
  // — no explicit orphan-eviction path needed here.
65
67
  };
@@ -20,7 +20,7 @@
20
20
  import type { OneShotAgentParams, OneShotUsage } from "../../core/types.js";
21
21
  import { log, logWarn } from "../../util/log.js";
22
22
  import { appendBackendSuffix } from "../runtime/index.js";
23
- import { emitAssistantText } from "../runtime/one-shot-hooks.js";
23
+ import { emitAssistantText, emitSessionId } from "../runtime/one-shot-hooks.js";
24
24
  import { ensureCodex, getCodexAuthInfo } from "./init.js";
25
25
  import {
26
26
  CODEX_SYSTEM_PROMPT_SUFFIX,
@@ -87,6 +87,8 @@ export async function runOneShotAgent(
87
87
  abortController,
88
88
  appendLog,
89
89
  onAssistantText,
90
+ resumeSessionId,
91
+ onSessionId,
90
92
  } = params;
91
93
 
92
94
  const codex = ensureCodex(contextLabel);
@@ -99,7 +101,10 @@ export async function runOneShotAgent(
99
101
  // Codex SDK doesn't expose `system` on runStreamed — the system
100
102
  // prompt gets prepended to the user prompt for a one-shot, since
101
103
  // there's no thread continuity to worry about.
102
- const inputText = `${finalSystemPrompt}\n\n---\n\n${prompt}`;
104
+ // A resumed thread already carries the system prompt from its first turn.
105
+ const inputText = resumeSessionId
106
+ ? prompt
107
+ : `${finalSystemPrompt}\n\n---\n\n${prompt}`;
103
108
 
104
109
  const resolved = resolveOneShotModel(requestedModel);
105
110
  const activeModel = resolved.model;
@@ -130,12 +135,16 @@ export async function runOneShotAgent(
130
135
  );
131
136
  }
132
137
 
133
- const thread = codex.startThread({
134
- model: activeModel,
135
- skipGitRepoCheck: true,
136
- ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
137
- ...CODEX_THREAD_PERMISSIONS,
138
- });
138
+ const thread = openThread(
139
+ codex,
140
+ {
141
+ model: activeModel,
142
+ skipGitRepoCheck: true,
143
+ ...(modelReasoningEffort ? { modelReasoningEffort } : {}),
144
+ ...CODEX_THREAD_PERMISSIONS,
145
+ },
146
+ params,
147
+ );
139
148
 
140
149
  // The real reason a run failed lives in the stream, not in whatever the
141
150
  // SDK throws afterwards: a model the account can't use yields
@@ -159,19 +168,10 @@ export async function runOneShotAgent(
159
168
  let usage: OneShotUsage | undefined;
160
169
  for await (const event of events) {
161
170
  if (abortController.signal.aborted) break;
171
+ reportThreadStarted(event, onSessionId);
162
172
  await appendCodexEvent(appendLog, event, onAssistantText);
163
173
  failure.observe(event);
164
- if (event.type === "turn.completed") {
165
- const u = (event as { usage?: Record<string, number> }).usage;
166
- if (u) {
167
- usage = {
168
- inputTokens: u.input_tokens ?? 0,
169
- outputTokens: u.output_tokens ?? 0,
170
- cacheRead: u.cached_input_tokens ?? 0,
171
- cacheWrite: 0, // Codex doesn't report cache writes
172
- };
173
- }
174
- }
174
+ usage = turnUsage(event) ?? usage;
175
175
  }
176
176
  // A stream that ends cleanly after `turn.failed` is still a failed run
177
177
  // — returning here is how 26/26 Codex cron runs were stored as "ok".
@@ -202,6 +202,55 @@ export async function runOneShotAgent(
202
202
  }
203
203
  }
204
204
 
205
+ type CodexClient = ReturnType<typeof ensureCodex>;
206
+ type CodexThreadOptions = Parameters<CodexClient["startThread"]>[0];
207
+
208
+ /**
209
+ * Start the run's thread — or, for a sub-agent interrupted by a daemon
210
+ * restart, continue its own thread and re-report the handle.
211
+ */
212
+ function openThread(
213
+ codex: CodexClient,
214
+ options: CodexThreadOptions,
215
+ params: Pick<
216
+ OneShotAgentParams,
217
+ "resumeSessionId" | "onSessionId" | "contextLabel"
218
+ >,
219
+ ): ReturnType<CodexClient["startThread"]> {
220
+ const { resumeSessionId, onSessionId, contextLabel } = params;
221
+ if (!resumeSessionId) return codex.startThread(options);
222
+ log(
223
+ "agent",
224
+ `[${contextLabel}] Codex one-shot resuming thread ${resumeSessionId}`,
225
+ );
226
+ const thread = codex.resumeThread(resumeSessionId, options);
227
+ emitSessionId(onSessionId, resumeSessionId);
228
+ return thread;
229
+ }
230
+
231
+ /** Report the thread id from `thread.started` so a restart can resume it. */
232
+ function reportThreadStarted(
233
+ event: { type: string },
234
+ onSessionId: OneShotAgentParams["onSessionId"],
235
+ ): void {
236
+ if (event.type !== "thread.started") return;
237
+ const threadId = (event as { thread_id?: unknown }).thread_id;
238
+ if (typeof threadId === "string") emitSessionId(onSessionId, threadId);
239
+ }
240
+
241
+ /** The cumulative usage a `turn.completed` event carries, if any. */
242
+ function turnUsage(event: { type: string }): OneShotUsage | undefined {
243
+ if (event.type !== "turn.completed") return undefined;
244
+ const u = (event as { usage?: Record<string, number> }).usage;
245
+ if (!u) return undefined;
246
+ return {
247
+ inputTokens: u.input_tokens ?? 0,
248
+ outputTokens: u.output_tokens ?? 0,
249
+ cacheRead: u.cached_input_tokens ?? 0,
250
+ cacheWrite: 0, // Codex doesn't report cache writes
251
+ };
252
+ }
253
+
205
254
  /**
206
255
  * Record a model as OAuth-incompat when a one-shot failed on it with an
207
256
  * explicit server mismatch, so the next run pre-emptively swaps.
@@ -43,3 +43,27 @@ export function emitAssistantText(
43
43
  );
44
44
  }
45
45
  }
46
+
47
+ /**
48
+ * Report the run's conversation handle (Claude SDK session id, Codex thread
49
+ * id) to the run's optional consumer — the sub-agent runner persists it so a
50
+ * daemon restart can resume the conversation. Same contract as
51
+ * `emitAssistantText`: synchronous, no-op without a hook or an id, and a
52
+ * throwing consumer is logged, never propagated into the run.
53
+ */
54
+ export function emitSessionId(
55
+ onSessionId: OneShotAgentParams["onSessionId"],
56
+ sessionId: string | undefined,
57
+ ): void {
58
+ if (!onSessionId || !sessionId) return;
59
+ try {
60
+ onSessionId(sessionId);
61
+ } catch (err) {
62
+ logWarn(
63
+ "agent",
64
+ `one-shot onSessionId hook threw (ignored): ${
65
+ err instanceof Error ? err.message : String(err)
66
+ }`,
67
+ );
68
+ }
69
+ }
@@ -112,6 +112,13 @@ export interface ChatBackend {
112
112
  */
113
113
  export interface BackgroundRunner {
114
114
  runOneShotAgent(params: OneShotAgentParams): Promise<OneShotUsage | void>;
115
+ /**
116
+ * Whether `runOneShotAgent` honours `resumeSessionId` (and reports the
117
+ * handle through `onSessionId`). A sub-agent interrupted by a daemon
118
+ * restart resumes its conversation on such a backend; on any other it is
119
+ * re-briefed with its previous transcript instead.
120
+ */
121
+ readonly supportsResume?: boolean;
115
122
  evictOrphanSubprocesses?(contextLabel: string): Promise<{
116
123
  found: number;
117
124
  termed: number;
@@ -23,7 +23,9 @@ export {
23
23
  clampTimeout,
24
24
  getAgentCaps,
25
25
  initAgents,
26
+ interruptAgentsForRestart,
26
27
  killAgent,
28
+ resumeAgentsAfterRestart,
27
29
  shutdownAgents,
28
30
  spawnAgent,
29
31
  } from "./runner.js";
@@ -85,6 +85,81 @@ export function buildAgentPrompt(
85
85
  );
86
86
  }
87
87
 
88
+ /** Human time for an interruption stamp. */
89
+ function stamp(at: number): string {
90
+ return new Date(at).toISOString().replace("T", " ").slice(0, 19) + " UTC";
91
+ }
92
+
93
+ /**
94
+ * The note a resumed agent receives when its own backend conversation is
95
+ * continued after a daemon restart. The transcript — brief, tool calls,
96
+ * results — is intact above it, so this only says what happened and what
97
+ * to be careful of.
98
+ */
99
+ export function buildResumePrompt(args: {
100
+ interruptedAt: number;
101
+ elapsedMinutes: number;
102
+ }): string {
103
+ return (
104
+ `[System: You were interrupted by a daemon restart at ` +
105
+ `${stamp(args.interruptedAt)} (about ${args.elapsedMinutes} min into ` +
106
+ `your run). Your conversation so far is intact above — continue the ` +
107
+ `brief from where you left off; do not start over. Anything that was ` +
108
+ `mid-flight when the restart hit (a shell command, a build, a tool call ` +
109
+ `with no result) may not have completed: check the actual state (files, ` +
110
+ `git status, processes) before relying on it or repeating it. Call ` +
111
+ `check_inbox — messages sent while you were down are still there. When ` +
112
+ `you are done, call report_result exactly once.]`
113
+ );
114
+ }
115
+
116
+ /**
117
+ * The activation prompt for an interrupted agent whose backend cannot
118
+ * resume a conversation: the original brief again, plus what the previous
119
+ * attempt did (the tail of its run log) so it picks up rather than redoes.
120
+ */
121
+ export function buildRebriefPrompt(args: {
122
+ brief: string;
123
+ /** Re-append the pre-flight lane instruction the original spawn carried. */
124
+ preflight?: boolean;
125
+ interruptedAt: number;
126
+ elapsedMinutes: number;
127
+ logPath: string;
128
+ logTail: string;
129
+ }): string {
130
+ const tail = args.logTail.trim()
131
+ ? `\n\nTail of the previous attempt's run log (full log: ` +
132
+ `${args.logPath}):\n\n<previous-run-log>\n${args.logTail}\n` +
133
+ `</previous-run-log>`
134
+ : `\n\n(The previous attempt's run log is at ${args.logPath}.)`;
135
+ return (
136
+ `${buildAgentPrompt(args.brief, { preflight: args.preflight === true })}\n\n` +
137
+ `[System: RESUMED AFTER A DAEMON RESTART. You already worked on this ` +
138
+ `brief for about ${args.elapsedMinutes} min before a daemon restart ` +
139
+ `interrupted you at ${stamp(args.interruptedAt)}; this backend could not ` +
140
+ `resume that conversation, so you are starting a new one. Do NOT start ` +
141
+ `over: read what the previous attempt did below, inspect the state it ` +
142
+ `left (files, branches, commits, processes) and continue from there. ` +
143
+ `Call check_inbox — messages sent while you were down are still there.]` +
144
+ tail
145
+ );
146
+ }
147
+
148
+ /** Run-log separator written when a restarted agent resumes. */
149
+ export function agentResumeLogHeader(
150
+ record: AgentRecord,
151
+ model: string,
152
+ interruptedAt: number,
153
+ sessionId?: string,
154
+ ): string {
155
+ return (
156
+ `\n\n---\n\n# resumed after daemon restart — ${new Date().toISOString()}\n` +
157
+ `**Interrupted:** ${new Date(interruptedAt).toISOString()} ` +
158
+ `**Backend:** ${record.backendId} **Model:** ${model} ` +
159
+ `**Mode:** ${sessionId ? `session resume (${sessionId})` : "re-briefed"}\n\n`
160
+ );
161
+ }
162
+
88
163
  /** Header line of a run log. */
89
164
  export function agentLogHeader(record: AgentRecord, model: string): string {
90
165
  return (