talon-agent 5.28.0 → 5.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/package.json +3 -2
  2. package/prompts/system/agent-brief.md +9 -6
  3. package/src/backend/claude-sdk/constants.ts +22 -0
  4. package/src/backend/claude-sdk/models/discovery.ts +3 -0
  5. package/src/backend/claude-sdk/one-shot.ts +3 -1
  6. package/src/backend/claude-sdk/options.ts +7 -1
  7. package/src/core/agents/index.ts +2 -0
  8. package/src/core/agents/prompt.ts +70 -2
  9. package/src/core/agents/registry.ts +47 -5
  10. package/src/core/agents/runner.ts +227 -39
  11. package/src/core/agents/trail.ts +141 -0
  12. package/src/core/agents/types.ts +43 -5
  13. package/src/core/agents/watchdog.ts +70 -0
  14. package/src/core/background/isolated-agent.ts +6 -2
  15. package/src/core/backup/plan.ts +4 -0
  16. package/src/core/config/index.ts +17 -4
  17. package/src/core/engine/gateway-actions/agents/control.ts +9 -2
  18. package/src/core/engine/gateway-actions/agents/preflight.ts +17 -1
  19. package/src/core/engine/gateway-actions/agents/report.ts +81 -24
  20. package/src/core/engine/gateway-actions/index.ts +3 -0
  21. package/src/core/mcp-hub/guest-scope.ts +3 -1
  22. package/src/core/mesh/devices/service.ts +7 -0
  23. package/src/core/mesh/links/bridge-links.ts +20 -0
  24. package/src/core/secrets/actions.ts +18 -0
  25. package/src/core/secrets/drop.ts +176 -0
  26. package/src/core/secrets/index.ts +11 -0
  27. package/src/core/secrets/service.ts +248 -0
  28. package/src/core/secrets/store.ts +100 -0
  29. package/src/core/tools/index.ts +2 -0
  30. package/src/core/tools/ops/agents.ts +30 -12
  31. package/src/core/tools/ops/secrets.ts +36 -0
  32. package/src/core/tools/types.ts +2 -1
  33. package/src/frontend/discord/commands/definitions.ts +21 -0
  34. package/src/frontend/discord/commands/router.ts +3 -0
  35. package/src/frontend/discord/commands/secret.ts +26 -0
  36. package/src/frontend/discord/handlers/messages.ts +6 -4
  37. package/src/frontend/native/bridge/routes/host.ts +10 -0
  38. package/src/frontend/native/bridge/routes/pre-auth.ts +82 -1
  39. package/src/frontend/native/bridge/routes/table.ts +5 -0
  40. package/src/frontend/native/bridge/server.ts +32 -4
  41. package/src/frontend/native/commands/definitions.ts +6 -0
  42. package/src/frontend/native/commands/index.ts +12 -0
  43. package/src/frontend/native/surface/handlers.ts +9 -0
  44. package/src/frontend/telegram/actions/chat-info.ts +22 -5
  45. package/src/frontend/telegram/commands/definitions.ts +4 -0
  46. package/src/frontend/telegram/commands/index.ts +3 -0
  47. package/src/frontend/telegram/commands/secret.ts +24 -0
  48. package/src/frontend/telegram/handlers/messages.ts +7 -5
  49. package/src/frontend/whatsapp/commands.ts +16 -1
  50. package/src/frontend/whatsapp/messages/media-store.ts +4 -2
  51. package/src/storage/media-index.ts +43 -11
  52. package/src/storage/repositories/media-index-repo.ts +12 -1
  53. package/src/storage/sql/media-index.sql +4 -0
  54. package/src/storage/sql/statements.generated.ts +2 -0
  55. package/src/util/log.ts +1 -0
  56. package/src/util/paths.ts +6 -0
@@ -4,8 +4,8 @@
4
4
  *
5
5
  * The shape is the heartbeat / cron-job shape, because a sub-agent *is* one
6
6
  * of those: acquire a backend, resolve a model, open a run log, register a
7
- * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the hard
8
- * timeout → abort → grace → eviction discipline. Nothing here is
7
+ * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the (optional)
8
+ * hard timeout → abort → grace → eviction discipline. Nothing here is
9
9
  * backend-specific, which is the whole point: sub-agents work on Claude,
10
10
  * Codex, Kilo, OpenCode and any future backend with a background capability.
11
11
  *
@@ -65,6 +65,7 @@ import {
65
65
  import { openRunLog } from "../background/run-log.js";
66
66
  import { agentContextLabel } from "./context.js";
67
67
  import {
68
+ deliverMessage,
68
69
  deliverSettlement,
69
70
  initAgentDelivery,
70
71
  type AgentDeliveryDeps,
@@ -77,7 +78,11 @@ import {
77
78
  buildAgentSystemPrompt,
78
79
  buildRebriefPrompt,
79
80
  buildResumePrompt,
81
+ buildStallPing,
82
+ buildStallWarning,
80
83
  } from "./prompt.js";
84
+ import { closeTrail, openTrail, type RunTrail } from "./trail.js";
85
+ import { startWatchdog, type WatchdogHandle } from "./watchdog.js";
81
86
  import * as agentsRepo from "../../storage/agents/repo.js";
82
87
  import type { PersistedAgent } from "../../storage/agents/repo.js";
83
88
  import { agentRegistry } from "./registry.js";
@@ -87,18 +92,32 @@ import type {
87
92
  AgentRecord,
88
93
  AgentSpawnOutcome,
89
94
  AgentSpawnSpec,
95
+ AgentTrail,
90
96
  } from "./types.js";
91
97
 
92
- /** Defaults for `config.agents`, applied when the block is absent. */
98
+ /**
99
+ * Defaults for `config.agents`, applied when the block is absent. No hard
100
+ * timeout: the no-progress watchdog ends a run that has gone quiet, and a
101
+ * run that is still working is left to finish.
102
+ */
93
103
  export const DEFAULT_AGENT_CAPS: AgentCaps = {
94
104
  maxConcurrent: 6,
95
105
  maxDepth: 2,
96
- defaultTimeoutMs: 15 * 60 * 1000,
106
+ stallTimeoutMs: 15 * 60 * 1000,
97
107
  };
98
108
 
99
- /** Floor and ceiling the tool boundary clamps a requested `timeout_s` into. */
109
+ /** Floor the tool boundary clamps a requested `timeout_s` up to. */
100
110
  const MIN_TIMEOUT_MS = 30_000;
101
- const MAX_TIMEOUT_MS = 60 * 60 * 1000;
111
+
112
+ /** Raised to abort a run the no-progress watchdog gave up on. */
113
+ class AgentStalledError extends Error {
114
+ constructor(idleMs: number) {
115
+ super(
116
+ `stalled: no tool call or output for ${Math.round(idleMs / 60_000)} min`,
117
+ );
118
+ this.name = "AgentStalledError";
119
+ }
120
+ }
102
121
 
103
122
  const capsHolder: { caps: AgentCaps } = { caps: DEFAULT_AGENT_CAPS };
104
123
 
@@ -112,7 +131,12 @@ export function initAgents(
112
131
  "agents",
113
132
  `Initialized — maxConcurrent=${capsHolder.caps.maxConcurrent} ` +
114
133
  `maxDepth=${capsHolder.caps.maxDepth} ` +
115
- `timeout=${Math.round(capsHolder.caps.defaultTimeoutMs / 1000)}s`,
134
+ `timeout=${describeTimeout(capsHolder.caps.defaultTimeoutMs)} ` +
135
+ `ceiling=${describeTimeout(capsHolder.caps.maxTimeoutMs)} ` +
136
+ `stall=${describeTimeout(capsHolder.caps.stallTimeoutMs || undefined)}` +
137
+ (capsHolder.caps.allowedBackends?.length
138
+ ? ` allowedBackends=${capsHolder.caps.allowedBackends.join(",")}`
139
+ : ""),
116
140
  );
117
141
  }
118
142
 
@@ -121,14 +145,38 @@ export function getAgentCaps(): AgentCaps {
121
145
  return capsHolder.caps;
122
146
  }
123
147
 
148
+ /** "15m" / "90s" / "none" — for logs and tool text. */
149
+ export function describeTimeout(ms: number | undefined): string {
150
+ if (ms === undefined || !(ms > 0)) return "none";
151
+ return ms % 60_000 === 0 ? `${ms / 60_000}m` : `${Math.round(ms / 1000)}s`;
152
+ }
153
+
154
+ /**
155
+ * The hard timeout a run gets: the requested one, else
156
+ * `agents.defaultTimeoutMs`, either capped by `agents.maxTimeoutMs`; with
157
+ * none of those set, `undefined` — no hard timeout. Applied by the runner,
158
+ * so a resumed run follows the same rule as a fresh one.
159
+ */
160
+ function effectiveTimeout(requestedMs: number | undefined): number | undefined {
161
+ const { defaultTimeoutMs, maxTimeoutMs } = capsHolder.caps;
162
+ const base = requestedMs ?? defaultTimeoutMs;
163
+ if (base === undefined) return maxTimeoutMs;
164
+ return maxTimeoutMs !== undefined ? Math.min(maxTimeoutMs, base) : base;
165
+ }
166
+
124
167
  /**
125
- * Clamp a model-supplied timeout into the supported window, or fall back to
126
- * the configured default. Applied at the tool boundary — `spawnAgent` itself
127
- * honours whatever it is handed, so the runner has one rule and not two.
168
+ * The tool boundary's rule: a model-supplied timeout is floored at 30s,
169
+ * then resolved like any other (`effectiveTimeout`). Returns `undefined`
170
+ * for "no hard timeout".
128
171
  */
129
- export function clampTimeout(requestedMs: number | undefined): number {
130
- if (requestedMs === undefined) return capsHolder.caps.defaultTimeoutMs;
131
- return Math.min(MAX_TIMEOUT_MS, Math.max(MIN_TIMEOUT_MS, requestedMs));
172
+ export function clampTimeout(
173
+ requestedMs: number | undefined,
174
+ ): number | undefined {
175
+ return effectiveTimeout(
176
+ requestedMs !== undefined && Number.isFinite(requestedMs)
177
+ ? Math.max(MIN_TIMEOUT_MS, requestedMs)
178
+ : undefined,
179
+ );
132
180
  }
133
181
 
134
182
  /** The backend an agent inherits when the caller didn't pick one. */
@@ -137,19 +185,37 @@ function inheritedBackendId(parent: AgentParent): string | null {
137
185
  return agentRegistry.get(parent.agentId)?.backendId ?? null;
138
186
  }
139
187
 
188
+ /** Where a spawn lands: backend, the model it inherits, and why. */
189
+ interface SpawnTarget {
190
+ readonly backendId: string | null;
191
+ /** The parent agent's model, inherited by an unpinned child. */
192
+ readonly inheritedModel?: string;
193
+ readonly routing?: string;
194
+ }
195
+
140
196
  /**
141
197
  * Which backend this agent runs on, and why.
142
198
  *
143
199
  * An explicit backend (or model — a model id is backend-specific, so naming
144
- * one pins its backend) is honoured as written. With neither, the run is a
145
- * routing decision: sub-agents are isolated one-shots with no session to
146
- * keep warm, so they are the cheapest work to move onto whichever
147
- * subscription has room.
200
+ * one pins its backend) is honoured as written. A child of another agent
201
+ * with neither inherits its parent's backend *and* model: a tree of agents
202
+ * stays on the backend its root was put on, so a parent that chose (or was
203
+ * told to use) a backend does not see its children wander off to another
204
+ * subscription. A top-level spawn with neither is a routing decision:
205
+ * sub-agents are isolated one-shots with no session to keep warm, so they
206
+ * are the cheapest work to move onto whichever subscription has room.
148
207
  */
149
- async function resolveSpawnBackend(
150
- spec: AgentSpawnSpec,
151
- ): Promise<{ backendId: string | null; routing?: string }> {
208
+ async function resolveSpawnBackend(spec: AgentSpawnSpec): Promise<SpawnTarget> {
152
209
  if (spec.backendId) return { backendId: spec.backendId };
210
+ if (spec.parent.kind === "agent" && !spec.model) {
211
+ const parent = agentRegistry.get(spec.parent.agentId);
212
+ if (parent) {
213
+ return {
214
+ backendId: parent.backendId,
215
+ ...(parent.model ? { inheritedModel: parent.model } : {}),
216
+ };
217
+ }
218
+ }
153
219
  const inherited = inheritedBackendId(spec.parent);
154
220
  if (!inherited) return { backendId: null };
155
221
  const taskClass = taskClassForEffort(spec.reasoningEffort);
@@ -166,12 +232,37 @@ async function resolveSpawnBackend(
166
232
  }
167
233
  : {}),
168
234
  });
235
+ // A routed pick outside the allowlist falls back to the inherited
236
+ // backend rather than refusing a spawn the caller never pinned.
237
+ if (
238
+ decision.routed &&
239
+ !isBackendAllowed(decision.backendId) &&
240
+ isBackendAllowed(inherited)
241
+ ) {
242
+ return { backendId: inherited };
243
+ }
169
244
  return {
170
245
  backendId: decision.backendId,
171
246
  ...(decision.routed ? { routing: decision.reason } : {}),
172
247
  };
173
248
  }
174
249
 
250
+ /** Whether `agents.allowedBackends` (when set) lets an agent run here. */
251
+ function isBackendAllowed(backendId: string): boolean {
252
+ const allowed = capsHolder.caps.allowedBackends;
253
+ return !allowed || allowed.length === 0 || allowed.includes(backendId);
254
+ }
255
+
256
+ /** The tool error for a backend outside `agents.allowedBackends`. */
257
+ function disallowedBackendError(backendId: string): string {
258
+ const allowed = capsHolder.caps.allowedBackends ?? [];
259
+ return (
260
+ `Backend "${backendId}" is not allowed for sub-agents ` +
261
+ `(agents.allowedBackends: ${allowed.join(", ")}). Pass one of those ` +
262
+ `as backend, or leave it unset to inherit.`
263
+ );
264
+ }
265
+
175
266
  /** The chat a run's task belongs to, for `talon ps`. */
176
267
  function taskChatId(parent: AgentParent): string | undefined {
177
268
  if (parent.kind === "chat") return parent.chatId;
@@ -262,6 +353,10 @@ export async function spawnAgent(
262
353
  "Could not resolve a backend for this agent — pass one explicitly.",
263
354
  };
264
355
  }
356
+ if (!isBackendAllowed(backendId)) {
357
+ return { ok: false, error: disallowedBackendError(backendId) };
358
+ }
359
+ const model = spec.model ?? routed.inheritedModel;
265
360
 
266
361
  // Register first: the slot and the depth are claimed synchronously, so two
267
362
  // concurrent spawns can never both slip past maxConcurrent while awaiting
@@ -276,7 +371,7 @@ export async function spawnAgent(
276
371
  ? { reasoningEffort: spec.reasoningEffort }
277
372
  : {}),
278
373
  ...(spec.model ? { requestedModel: spec.model } : {}),
279
- timeoutMs: spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs,
374
+ ...(spec.timeoutMs !== undefined ? { timeoutMs: spec.timeoutMs } : {}),
280
375
  cwd: dirs.workspace,
281
376
  ...(spec.preflight ? { preflight: true } : {}),
282
377
  },
@@ -296,7 +391,7 @@ export async function spawnAgent(
296
391
  };
297
392
  }
298
393
 
299
- const resolved = await resolveRun(acquired.backend, backendId, spec.model);
394
+ const resolved = await resolveRun(acquired.backend, backendId, model);
300
395
  if (!resolved.ok) {
301
396
  agentRegistry.discard(record.id);
302
397
  await acquired.release();
@@ -332,9 +427,10 @@ async function buildRunParams(
332
427
  model: string,
333
428
  abortController: AbortController,
334
429
  capture: { last: string },
430
+ trail: RunTrail,
335
431
  resume?: ResumePlan,
336
432
  ): Promise<OneShotAgentParams> {
337
- const appendLog = await openRunLog(
433
+ const writeLog = await openRunLog(
338
434
  agentLogPath(record.id),
339
435
  resume
340
436
  ? agentResumeLogHeader(
@@ -346,6 +442,10 @@ async function buildRunParams(
346
442
  : agentLogHeader(record, model),
347
443
  );
348
444
  const id = record.id;
445
+ const appendLog = (text: string): Promise<void> => {
446
+ trail.onLog(text);
447
+ return writeLog(text);
448
+ };
349
449
  return {
350
450
  prompt: resume
351
451
  ? resume.prompt
@@ -367,6 +467,7 @@ async function buildRunParams(
367
467
  onAssistantText: (text) => {
368
468
  const trimmed = text.trim();
369
469
  if (trimmed) capture.last = trimmed;
470
+ trail.onAssistantText(text);
370
471
  },
371
472
  // Persisted the moment the backend reports it, so a restart at any
372
473
  // point after the first message can resume the conversation.
@@ -382,11 +483,13 @@ function settleSuccess(
382
483
  task: TaskHandle,
383
484
  lastText: string,
384
485
  usage: TaskUsage | undefined,
486
+ trail: AgentTrail,
385
487
  ): AgentRecord | null {
386
488
  if (agentRegistry.hasReported(id)) {
387
489
  task.succeed(usage);
388
490
  return agentRegistry.settle(id, {
389
491
  state: "done",
492
+ trail,
390
493
  ...(usage ? { usage } : {}),
391
494
  });
392
495
  }
@@ -395,6 +498,7 @@ function settleSuccess(
395
498
  return agentRegistry.settle(id, {
396
499
  state: "done",
397
500
  result: { summary: lastText },
501
+ trail,
398
502
  ...(usage ? { usage } : {}),
399
503
  });
400
504
  }
@@ -404,24 +508,82 @@ function settleSuccess(
404
508
  return agentRegistry.settle(id, {
405
509
  state: "failed",
406
510
  error,
511
+ trail,
407
512
  ...(usage ? { usage } : {}),
408
513
  });
409
514
  }
410
515
 
411
- /** Settle a run that threw: timeout, kill, or a genuine failure. */
516
+ /**
517
+ * Settle a run that threw: timeout, stall, kill, or a genuine failure.
518
+ * `stalled` is the watchdog's own abort reason, checked first because a
519
+ * backend that honours the abort rejects with its own error.
520
+ */
412
521
  function settleFailure(
413
522
  id: string,
414
523
  task: TaskHandle,
415
524
  err: unknown,
525
+ trail: AgentTrail,
526
+ stalled?: AgentStalledError,
416
527
  ): AgentRecord | null {
417
528
  const state =
418
- err instanceof IsolatedAgentTimeoutError
529
+ stalled || err instanceof IsolatedAgentTimeoutError
419
530
  ? "timed_out"
420
531
  : agentRegistry.killRequested(id)
421
532
  ? "killed"
422
533
  : "failed";
423
- task.fail(err);
424
- return agentRegistry.settle(id, { state, error: errText(err) });
534
+ task.fail(stalled ?? err);
535
+ return agentRegistry.settle(id, {
536
+ state,
537
+ error: errText(stalled ?? err),
538
+ trail,
539
+ });
540
+ }
541
+
542
+ /**
543
+ * Start the no-progress watchdog for one run (see `watchdog.ts`). Its kill
544
+ * aborts the run with an `AgentStalledError` recorded in `watch.stalled`,
545
+ * which the settle path reads to classify the run `timed_out`.
546
+ */
547
+ function watchRun(
548
+ id: string,
549
+ trail: RunTrail,
550
+ abortController: AbortController,
551
+ ): { watch: { stalled?: AgentStalledError }; watchdog: WatchdogHandle } {
552
+ const watch: { stalled?: AgentStalledError } = {};
553
+ const watchdog = startWatchdog(capsHolder.caps.stallTimeoutMs, {
554
+ lastActivityAt: () => trail.lastActivityAt,
555
+ pingAgent: (idleMs) => {
556
+ agentRegistry.push(id, {
557
+ from: "watchdog",
558
+ text: buildStallPing(idleMs, capsHolder.caps.stallTimeoutMs),
559
+ at: Date.now(),
560
+ });
561
+ logWarn(
562
+ "agents",
563
+ `${id} quiet for ${Math.round(idleMs / 1000)}s — pinged`,
564
+ );
565
+ },
566
+ warnParent: (idleMs, killInMs) => {
567
+ const current = agentRegistry.get(id);
568
+ if (!current) return;
569
+ void deliverMessage(
570
+ current,
571
+ buildStallWarning(current, idleMs, killInMs),
572
+ ).catch((err: unknown) =>
573
+ logError("agents", `stall warning delivery failed for ${id}`, err),
574
+ );
575
+ },
576
+ kill: (idleMs) => {
577
+ watch.stalled = new AgentStalledError(idleMs);
578
+ logWarn("agents", `${id}: ${watch.stalled.message} — aborting`);
579
+ try {
580
+ abortController.abort(watch.stalled);
581
+ } catch {
582
+ /* the settle path below still runs */
583
+ }
584
+ },
585
+ });
586
+ return { watch, watchdog };
425
587
  }
426
588
 
427
589
  /**
@@ -441,7 +603,9 @@ async function runAgent(
441
603
  const id = record.id;
442
604
  const abortController = new AbortController();
443
605
  const capture = { last: "" };
444
- const timeoutMs = spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs;
606
+ const timeoutMs = effectiveTimeout(spec.timeoutMs);
607
+ const trail = openTrail(id);
608
+ const { watch, watchdog } = watchRun(id, trail, abortController);
445
609
 
446
610
  // Registered as queued, bound, then started — so a kill arriving in the
447
611
  // gap between the task existing and the abort handle being published still
@@ -465,6 +629,7 @@ async function runAgent(
465
629
  model,
466
630
  abortController,
467
631
  capture,
632
+ trail,
468
633
  resume,
469
634
  );
470
635
  if (agentRegistry.isInterrupted(id)) {
@@ -478,12 +643,13 @@ async function runAgent(
478
643
  id,
479
644
  task,
480
645
  new Error("aborted before the run started"),
646
+ trail.snapshot(),
481
647
  );
482
648
  } else {
483
649
  const usage = await runIsolatedAgent({
484
650
  background,
485
651
  params,
486
- timeoutMs,
652
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
487
653
  logCategory: "agents",
488
654
  // Safe to sweep: the context label is unique to this agent, so no
489
655
  // other context's subprocesses share the tag.
@@ -496,18 +662,37 @@ async function runAgent(
496
662
  if (agentRegistry.isInterrupted(id)) {
497
663
  settled = null;
498
664
  } else {
499
- recordBackendRunSuccess(record.backendId);
500
- settled = settleSuccess(id, task, capture.last, usage ?? undefined);
665
+ if (watch.stalled) {
666
+ // The backend swallowed the watchdog's abort and returned.
667
+ settled = settleFailure(
668
+ id,
669
+ task,
670
+ watch.stalled,
671
+ trail.snapshot(),
672
+ watch.stalled,
673
+ );
674
+ } else {
675
+ recordBackendRunSuccess(record.backendId);
676
+ settled = settleSuccess(
677
+ id,
678
+ task,
679
+ capture.last,
680
+ usage ?? undefined,
681
+ trail.snapshot(),
682
+ );
683
+ }
501
684
  }
502
685
  }
503
686
  } catch (err) {
504
687
  if (agentRegistry.isInterrupted(id)) {
505
688
  settled = null;
506
689
  } else {
507
- recordBackendRunFailure(record.backendId, err);
508
- settled = settleFailure(id, task, err);
690
+ if (!watch.stalled) recordBackendRunFailure(record.backendId, err);
691
+ settled = settleFailure(id, task, err, trail.snapshot(), watch.stalled);
509
692
  }
510
693
  } finally {
694
+ watchdog.stop();
695
+ closeTrail(id);
511
696
  await release().catch((err: unknown) =>
512
697
  logError("agents", `failed to release backend for ${id}`, err),
513
698
  );
@@ -526,7 +711,7 @@ async function runAgent(
526
711
  log(
527
712
  "agents",
528
713
  `${id} "${settled.label}" → ${settled.state} ` +
529
- `(${settled.backendId}/${model}, ${timeoutMs}ms cap)`,
714
+ `(${settled.backendId}/${model}, timeout ${describeTimeout(timeoutMs)})`,
530
715
  );
531
716
  reapChildren(settled);
532
717
  await deliverSettlement(settled).catch((err: unknown) =>
@@ -762,9 +947,12 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
762
947
  interruptedAt,
763
948
  };
764
949
 
765
- const budget =
766
- (saved.timeoutMs ?? capsHolder.caps.defaultTimeoutMs) - elapsedMs;
767
- const timeoutMs = Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, budget);
950
+ // An uncapped run stays uncapped; a capped one gets what it had left.
951
+ const cap = effectiveTimeout(saved.timeoutMs);
952
+ const timeoutMs =
953
+ cap === undefined
954
+ ? undefined
955
+ : Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, cap - elapsedMs);
768
956
  const spec: AgentSpawnSpec = {
769
957
  brief: saved.brief,
770
958
  label: saved.label,
@@ -774,7 +962,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
774
962
  ...(record.reasoningEffort
775
963
  ? { reasoningEffort: record.reasoningEffort }
776
964
  : {}),
777
- timeoutMs,
965
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
778
966
  ...(saved.preflight ? { preflight: true } : {}),
779
967
  };
780
968
  agentRegistry.markResumed(record.id);
@@ -782,7 +970,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
782
970
  "agents",
783
971
  `${record.id} "${record.label}" resuming after restart ` +
784
972
  `(${canResume ? `session ${saved.sessionId}` : "re-briefed"}, ` +
785
- `${backendId}/${resolved.model}, ${Math.round(timeoutMs / 1000)}s left, ` +
973
+ `${backendId}/${resolved.model}, timeout ${describeTimeout(timeoutMs)}, ` +
786
974
  `resume #${saved.resumeCount + 1})`,
787
975
  );
788
976
  void runAgent(record, spec, resolved, acquired, plan);
@@ -0,0 +1,141 @@
1
+ /**
2
+ * Trail — what a running sub-agent has been doing, kept so a run that is
3
+ * cut short (killed, timed out, stalled, failed) still hands its parent
4
+ * something to work with.
5
+ *
6
+ * Three bounded lists, all best-effort:
7
+ *
8
+ * - **messages** — its last interim `message_parent` notes.
9
+ * - **notes** — its last assistant texts (progress narration).
10
+ * - **files** — paths it wrote or edited, scraped from the run log the
11
+ * backend writes: Claude-style `**Tool call:** \`Write\`` blocks carrying
12
+ * a `file_path` / `path` / `notebook_path`, and Codex `**File changes:**`
13
+ * lists. A backend that logs neither simply contributes no files.
14
+ *
15
+ * The trail also carries the run's `lastActivityAt` clock, which the
16
+ * no-progress watchdog reads: any log line or assistant text counts.
17
+ *
18
+ * Trails live in memory only, keyed by agent id, for the life of the run.
19
+ */
20
+
21
+ import type { AgentTrail } from "./types.js";
22
+
23
+ const MAX_MESSAGES = 3;
24
+ const MAX_NOTES = 3;
25
+ const MAX_FILES = 50;
26
+ /** One note or message is clipped to this many characters in the trail. */
27
+ const MAX_TEXT_CHARS = 1_500;
28
+
29
+ /** Tool names (bare or MCP-prefixed) that write files. */
30
+ const WRITE_TOOL = /(?:^|__)(?:Write|Edit|MultiEdit|NotebookEdit|write|edit)$/;
31
+ const TOOL_CALL_BLOCK =
32
+ /\*\*(?:MCP )?Tool call:\*\* `([^`]+)`\s*```json\n([\s\S]*?)\n```/g;
33
+ const PATH_KEY = /"(?:file_path|notebook_path|path)":\s*"((?:[^"\\]|\\.)+)"/;
34
+ const FILE_CHANGES_BLOCK = /\*\*File changes:\*\*[^\n]*\n((?:\s+- .*\n?)+)/g;
35
+
36
+ function clip(text: string): string {
37
+ const t = text.trim();
38
+ return t.length > MAX_TEXT_CHARS ? `${t.slice(0, MAX_TEXT_CHARS)}…` : t;
39
+ }
40
+
41
+ function pushBounded(list: string[], item: string, max: number): void {
42
+ list.push(item);
43
+ if (list.length > max) list.splice(0, list.length - max);
44
+ }
45
+
46
+ /** File paths a chunk of run-log text says were written or edited. */
47
+ export function filesFromLogChunk(chunk: string): string[] {
48
+ const out: string[] = [];
49
+ for (const m of chunk.matchAll(TOOL_CALL_BLOCK)) {
50
+ const tool = m[1] ?? "";
51
+ // MCP calls are logged as `server.tool`; normalise to the `__` form.
52
+ if (!WRITE_TOOL.test(tool.replace(/\./g, "__"))) continue;
53
+ const path = PATH_KEY.exec(m[2] ?? "")?.[1];
54
+ if (path) out.push(JSON.parse(`"${path}"`) as string);
55
+ }
56
+ for (const m of chunk.matchAll(FILE_CHANGES_BLOCK)) {
57
+ for (const line of (m[1] ?? "").split("\n")) {
58
+ const path = /^\s+- \S+ (.+)$/.exec(line)?.[1]?.trim();
59
+ if (path && path !== "?") out.push(path);
60
+ }
61
+ }
62
+ return out;
63
+ }
64
+
65
+ /** The live trail of one run. */
66
+ export class RunTrail {
67
+ private readonly messages: string[] = [];
68
+ private readonly notes: string[] = [];
69
+ private readonly files: string[] = [];
70
+ lastActivityAt: number;
71
+
72
+ constructor(now: number = Date.now()) {
73
+ this.lastActivityAt = now;
74
+ }
75
+
76
+ /** Any sign of life — resets the watchdog. */
77
+ touch(now: number = Date.now()): void {
78
+ this.lastActivityAt = now;
79
+ }
80
+
81
+ /** A chunk the backend appended to the run log. */
82
+ onLog(chunk: string): void {
83
+ this.touch();
84
+ for (const file of filesFromLogChunk(chunk)) {
85
+ if (this.files.includes(file)) continue;
86
+ pushBounded(this.files, file, MAX_FILES);
87
+ }
88
+ }
89
+
90
+ onAssistantText(text: string): void {
91
+ this.touch();
92
+ const t = clip(text);
93
+ if (t) pushBounded(this.notes, t, MAX_NOTES);
94
+ }
95
+
96
+ onMessage(text: string): void {
97
+ this.touch();
98
+ const t = clip(text);
99
+ if (t) pushBounded(this.messages, t, MAX_MESSAGES);
100
+ }
101
+
102
+ snapshot(): AgentTrail {
103
+ return {
104
+ messages: [...this.messages],
105
+ notes: [...this.notes],
106
+ files: [...this.files],
107
+ };
108
+ }
109
+ }
110
+
111
+ const trails = new Map<string, RunTrail>();
112
+
113
+ /** Start (or restart, on resume) the trail for a run. */
114
+ export function openTrail(agentId: string): RunTrail {
115
+ const trail = new RunTrail();
116
+ trails.set(agentId, trail);
117
+ return trail;
118
+ }
119
+
120
+ export function getTrail(agentId: string): RunTrail | undefined {
121
+ return trails.get(agentId);
122
+ }
123
+
124
+ export function closeTrail(agentId: string): void {
125
+ trails.delete(agentId);
126
+ }
127
+
128
+ /** Record an interim `message_parent` note against a running agent. */
129
+ export function recordInterimMessage(agentId: string, text: string): void {
130
+ trails.get(agentId)?.onMessage(text);
131
+ }
132
+
133
+ /** Whether a trail snapshot has anything worth showing. */
134
+ export function trailIsEmpty(trail: AgentTrail | undefined): boolean {
135
+ return (
136
+ !trail ||
137
+ (trail.messages.length === 0 &&
138
+ trail.notes.length === 0 &&
139
+ trail.files.length === 0)
140
+ );
141
+ }