talon-agent 5.29.0 → 5.31.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. package/package.json +2 -1
  2. package/prompts/system/agent-brief.md +18 -7
  3. package/src/backend/claude-sdk/constants.ts +22 -0
  4. package/src/backend/claude-sdk/models/discovery.ts +3 -0
  5. package/src/backend/claude-sdk/one-shot.ts +6 -1
  6. package/src/backend/claude-sdk/options.ts +7 -1
  7. package/src/core/agents/index.ts +2 -0
  8. package/src/core/agents/prompt.ts +73 -2
  9. package/src/core/agents/registry.ts +47 -5
  10. package/src/core/agents/runner.ts +178 -31
  11. package/src/core/agents/scratch.ts +80 -0
  12. package/src/core/agents/trail.ts +141 -0
  13. package/src/core/agents/types.ts +29 -3
  14. package/src/core/agents/watchdog.ts +70 -0
  15. package/src/core/background/isolated-agent.ts +6 -2
  16. package/src/core/backup/plan.ts +4 -0
  17. package/src/core/config/index.ts +13 -4
  18. package/src/core/engine/gateway-actions/agents/control.ts +9 -2
  19. package/src/core/engine/gateway-actions/agents/preflight.ts +17 -1
  20. package/src/core/engine/gateway-actions/agents/report.ts +81 -24
  21. package/src/core/engine/gateway-actions/index.ts +3 -0
  22. package/src/core/mcp-hub/guest-scope.ts +3 -1
  23. package/src/core/mesh/devices/service.ts +7 -0
  24. package/src/core/mesh/links/bridge-links.ts +20 -0
  25. package/src/core/secrets/actions.ts +18 -0
  26. package/src/core/secrets/drop.ts +176 -0
  27. package/src/core/secrets/index.ts +11 -0
  28. package/src/core/secrets/service.ts +248 -0
  29. package/src/core/secrets/store.ts +100 -0
  30. package/src/core/tools/index.ts +2 -0
  31. package/src/core/tools/ops/agents.ts +26 -8
  32. package/src/core/tools/ops/secrets.ts +36 -0
  33. package/src/core/tools/types.ts +2 -1
  34. package/src/core/types.ts +7 -0
  35. package/src/frontend/discord/commands/definitions.ts +21 -0
  36. package/src/frontend/discord/commands/router.ts +3 -0
  37. package/src/frontend/discord/commands/secret.ts +26 -0
  38. package/src/frontend/native/bridge/routes/host.ts +10 -0
  39. package/src/frontend/native/bridge/routes/pre-auth.ts +82 -1
  40. package/src/frontend/native/bridge/routes/table.ts +5 -0
  41. package/src/frontend/native/bridge/server.ts +32 -4
  42. package/src/frontend/native/commands/definitions.ts +6 -0
  43. package/src/frontend/native/commands/index.ts +12 -0
  44. package/src/frontend/native/surface/handlers.ts +9 -0
  45. package/src/frontend/telegram/commands/definitions.ts +4 -0
  46. package/src/frontend/telegram/commands/index.ts +3 -0
  47. package/src/frontend/telegram/commands/secret.ts +24 -0
  48. package/src/frontend/whatsapp/commands.ts +16 -1
  49. package/src/util/log.ts +1 -0
  50. package/src/util/paths.ts +6 -0
@@ -4,8 +4,8 @@
4
4
  *
5
5
  * The shape is the heartbeat / cron-job shape, because a sub-agent *is* one
6
6
  * of those: acquire a backend, resolve a model, open a run log, register a
7
- * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the hard
8
- * timeout → abort → grace → eviction discipline. Nothing here is
7
+ * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the (optional)
8
+ * hard timeout → abort → grace → eviction discipline. Nothing here is
9
9
  * backend-specific, which is the whole point: sub-agents work on Claude,
10
10
  * Codex, Kilo, OpenCode and any future backend with a background capability.
11
11
  *
@@ -65,6 +65,7 @@ import {
65
65
  import { openRunLog } from "../background/run-log.js";
66
66
  import { agentContextLabel } from "./context.js";
67
67
  import {
68
+ deliverMessage,
68
69
  deliverSettlement,
69
70
  initAgentDelivery,
70
71
  type AgentDeliveryDeps,
@@ -77,28 +78,47 @@ import {
77
78
  buildAgentSystemPrompt,
78
79
  buildRebriefPrompt,
79
80
  buildResumePrompt,
81
+ buildStallPing,
82
+ buildStallWarning,
80
83
  } from "./prompt.js";
84
+ import { closeTrail, openTrail, type RunTrail } from "./trail.js";
85
+ import { startWatchdog, type WatchdogHandle } from "./watchdog.js";
81
86
  import * as agentsRepo from "../../storage/agents/repo.js";
82
87
  import type { PersistedAgent } from "../../storage/agents/repo.js";
83
88
  import { agentRegistry } from "./registry.js";
89
+ import { closeScratch, openScratch, scratchEnv } from "./scratch.js";
84
90
  import type {
85
91
  AgentCaps,
86
92
  AgentParent,
87
93
  AgentRecord,
88
94
  AgentSpawnOutcome,
89
95
  AgentSpawnSpec,
96
+ AgentTrail,
90
97
  } from "./types.js";
91
98
 
92
- /** Defaults for `config.agents`, applied when the block is absent. */
99
+ /**
100
+ * Defaults for `config.agents`, applied when the block is absent. No hard
101
+ * timeout: the no-progress watchdog ends a run that has gone quiet, and a
102
+ * run that is still working is left to finish.
103
+ */
93
104
  export const DEFAULT_AGENT_CAPS: AgentCaps = {
94
105
  maxConcurrent: 6,
95
106
  maxDepth: 2,
96
- defaultTimeoutMs: 15 * 60 * 1000,
107
+ stallTimeoutMs: 15 * 60 * 1000,
97
108
  };
98
109
 
99
- /** Floor and ceiling the tool boundary clamps a requested `timeout_s` into. */
110
+ /** Floor the tool boundary clamps a requested `timeout_s` up to. */
100
111
  const MIN_TIMEOUT_MS = 30_000;
101
- const MAX_TIMEOUT_MS = 60 * 60 * 1000;
112
+
113
+ /** Raised to abort a run the no-progress watchdog gave up on. */
114
+ class AgentStalledError extends Error {
115
+ constructor(idleMs: number) {
116
+ super(
117
+ `stalled: no tool call or output for ${Math.round(idleMs / 60_000)} min`,
118
+ );
119
+ this.name = "AgentStalledError";
120
+ }
121
+ }
102
122
 
103
123
  const capsHolder: { caps: AgentCaps } = { caps: DEFAULT_AGENT_CAPS };
104
124
 
@@ -112,7 +132,9 @@ export function initAgents(
112
132
  "agents",
113
133
  `Initialized — maxConcurrent=${capsHolder.caps.maxConcurrent} ` +
114
134
  `maxDepth=${capsHolder.caps.maxDepth} ` +
115
- `timeout=${Math.round(capsHolder.caps.defaultTimeoutMs / 1000)}s` +
135
+ `timeout=${describeTimeout(capsHolder.caps.defaultTimeoutMs)} ` +
136
+ `ceiling=${describeTimeout(capsHolder.caps.maxTimeoutMs)} ` +
137
+ `stall=${describeTimeout(capsHolder.caps.stallTimeoutMs || undefined)}` +
116
138
  (capsHolder.caps.allowedBackends?.length
117
139
  ? ` allowedBackends=${capsHolder.caps.allowedBackends.join(",")}`
118
140
  : ""),
@@ -124,14 +146,38 @@ export function getAgentCaps(): AgentCaps {
124
146
  return capsHolder.caps;
125
147
  }
126
148
 
149
+ /** "15m" / "90s" / "none" — for logs and tool text. */
150
+ export function describeTimeout(ms: number | undefined): string {
151
+ if (ms === undefined || !(ms > 0)) return "none";
152
+ return ms % 60_000 === 0 ? `${ms / 60_000}m` : `${Math.round(ms / 1000)}s`;
153
+ }
154
+
127
155
  /**
128
- * Clamp a model-supplied timeout into the supported window, or fall back to
129
- * the configured default. Applied at the tool boundary — `spawnAgent` itself
130
- * honours whatever it is handed, so the runner has one rule and not two.
156
+ * The hard timeout a run gets: the requested one, else
157
+ * `agents.defaultTimeoutMs`, either capped by `agents.maxTimeoutMs`; with
158
+ * none of those set, `undefined` — no hard timeout. Applied by the runner,
159
+ * so a resumed run follows the same rule as a fresh one.
131
160
  */
132
- export function clampTimeout(requestedMs: number | undefined): number {
133
- if (requestedMs === undefined) return capsHolder.caps.defaultTimeoutMs;
134
- return Math.min(MAX_TIMEOUT_MS, Math.max(MIN_TIMEOUT_MS, requestedMs));
161
+ function effectiveTimeout(requestedMs: number | undefined): number | undefined {
162
+ const { defaultTimeoutMs, maxTimeoutMs } = capsHolder.caps;
163
+ const base = requestedMs ?? defaultTimeoutMs;
164
+ if (base === undefined) return maxTimeoutMs;
165
+ return maxTimeoutMs !== undefined ? Math.min(maxTimeoutMs, base) : base;
166
+ }
167
+
168
+ /**
169
+ * The tool boundary's rule: a model-supplied timeout is floored at 30s,
170
+ * then resolved like any other (`effectiveTimeout`). Returns `undefined`
171
+ * for "no hard timeout".
172
+ */
173
+ export function clampTimeout(
174
+ requestedMs: number | undefined,
175
+ ): number | undefined {
176
+ return effectiveTimeout(
177
+ requestedMs !== undefined && Number.isFinite(requestedMs)
178
+ ? Math.max(MIN_TIMEOUT_MS, requestedMs)
179
+ : undefined,
180
+ );
135
181
  }
136
182
 
137
183
  /** The backend an agent inherits when the caller didn't pick one. */
@@ -326,7 +372,7 @@ export async function spawnAgent(
326
372
  ? { reasoningEffort: spec.reasoningEffort }
327
373
  : {}),
328
374
  ...(spec.model ? { requestedModel: spec.model } : {}),
329
- timeoutMs: spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs,
375
+ ...(spec.timeoutMs !== undefined ? { timeoutMs: spec.timeoutMs } : {}),
330
376
  cwd: dirs.workspace,
331
377
  ...(spec.preflight ? { preflight: true } : {}),
332
378
  },
@@ -382,9 +428,11 @@ async function buildRunParams(
382
428
  model: string,
383
429
  abortController: AbortController,
384
430
  capture: { last: string },
431
+ trail: RunTrail,
432
+ scratchDir: string | undefined,
385
433
  resume?: ResumePlan,
386
434
  ): Promise<OneShotAgentParams> {
387
- const appendLog = await openRunLog(
435
+ const writeLog = await openRunLog(
388
436
  agentLogPath(record.id),
389
437
  resume
390
438
  ? agentResumeLogHeader(
@@ -396,6 +444,10 @@ async function buildRunParams(
396
444
  : agentLogHeader(record, model),
397
445
  );
398
446
  const id = record.id;
447
+ const appendLog = (text: string): Promise<void> => {
448
+ trail.onLog(text);
449
+ return writeLog(text);
450
+ };
399
451
  return {
400
452
  prompt: resume
401
453
  ? resume.prompt
@@ -408,7 +460,9 @@ async function buildRunParams(
408
460
  parent: record.parent,
409
461
  depth: record.depth,
410
462
  maxDepth: capsHolder.caps.maxDepth,
463
+ ...(scratchDir ? { scratchDir } : {}),
411
464
  }),
465
+ ...(scratchDir ? { env: scratchEnv(scratchDir) } : {}),
412
466
  workspace: dirs.workspace,
413
467
  model,
414
468
  contextLabel: agentContextLabel(record.id),
@@ -417,6 +471,7 @@ async function buildRunParams(
417
471
  onAssistantText: (text) => {
418
472
  const trimmed = text.trim();
419
473
  if (trimmed) capture.last = trimmed;
474
+ trail.onAssistantText(text);
420
475
  },
421
476
  // Persisted the moment the backend reports it, so a restart at any
422
477
  // point after the first message can resume the conversation.
@@ -432,11 +487,13 @@ function settleSuccess(
432
487
  task: TaskHandle,
433
488
  lastText: string,
434
489
  usage: TaskUsage | undefined,
490
+ trail: AgentTrail,
435
491
  ): AgentRecord | null {
436
492
  if (agentRegistry.hasReported(id)) {
437
493
  task.succeed(usage);
438
494
  return agentRegistry.settle(id, {
439
495
  state: "done",
496
+ trail,
440
497
  ...(usage ? { usage } : {}),
441
498
  });
442
499
  }
@@ -445,6 +502,7 @@ function settleSuccess(
445
502
  return agentRegistry.settle(id, {
446
503
  state: "done",
447
504
  result: { summary: lastText },
505
+ trail,
448
506
  ...(usage ? { usage } : {}),
449
507
  });
450
508
  }
@@ -454,24 +512,82 @@ function settleSuccess(
454
512
  return agentRegistry.settle(id, {
455
513
  state: "failed",
456
514
  error,
515
+ trail,
457
516
  ...(usage ? { usage } : {}),
458
517
  });
459
518
  }
460
519
 
461
- /** Settle a run that threw: timeout, kill, or a genuine failure. */
520
+ /**
521
+ * Settle a run that threw: timeout, stall, kill, or a genuine failure.
522
+ * `stalled` is the watchdog's own abort reason, checked first because a
523
+ * backend that honours the abort rejects with its own error.
524
+ */
462
525
  function settleFailure(
463
526
  id: string,
464
527
  task: TaskHandle,
465
528
  err: unknown,
529
+ trail: AgentTrail,
530
+ stalled?: AgentStalledError,
466
531
  ): AgentRecord | null {
467
532
  const state =
468
- err instanceof IsolatedAgentTimeoutError
533
+ stalled || err instanceof IsolatedAgentTimeoutError
469
534
  ? "timed_out"
470
535
  : agentRegistry.killRequested(id)
471
536
  ? "killed"
472
537
  : "failed";
473
- task.fail(err);
474
- return agentRegistry.settle(id, { state, error: errText(err) });
538
+ task.fail(stalled ?? err);
539
+ return agentRegistry.settle(id, {
540
+ state,
541
+ error: errText(stalled ?? err),
542
+ trail,
543
+ });
544
+ }
545
+
546
+ /**
547
+ * Start the no-progress watchdog for one run (see `watchdog.ts`). Its kill
548
+ * aborts the run with an `AgentStalledError` recorded in `watch.stalled`,
549
+ * which the settle path reads to classify the run `timed_out`.
550
+ */
551
+ function watchRun(
552
+ id: string,
553
+ trail: RunTrail,
554
+ abortController: AbortController,
555
+ ): { watch: { stalled?: AgentStalledError }; watchdog: WatchdogHandle } {
556
+ const watch: { stalled?: AgentStalledError } = {};
557
+ const watchdog = startWatchdog(capsHolder.caps.stallTimeoutMs, {
558
+ lastActivityAt: () => trail.lastActivityAt,
559
+ pingAgent: (idleMs) => {
560
+ agentRegistry.push(id, {
561
+ from: "watchdog",
562
+ text: buildStallPing(idleMs, capsHolder.caps.stallTimeoutMs),
563
+ at: Date.now(),
564
+ });
565
+ logWarn(
566
+ "agents",
567
+ `${id} quiet for ${Math.round(idleMs / 1000)}s — pinged`,
568
+ );
569
+ },
570
+ warnParent: (idleMs, killInMs) => {
571
+ const current = agentRegistry.get(id);
572
+ if (!current) return;
573
+ void deliverMessage(
574
+ current,
575
+ buildStallWarning(current, idleMs, killInMs),
576
+ ).catch((err: unknown) =>
577
+ logError("agents", `stall warning delivery failed for ${id}`, err),
578
+ );
579
+ },
580
+ kill: (idleMs) => {
581
+ watch.stalled = new AgentStalledError(idleMs);
582
+ logWarn("agents", `${id}: ${watch.stalled.message} — aborting`);
583
+ try {
584
+ abortController.abort(watch.stalled);
585
+ } catch {
586
+ /* the settle path below still runs */
587
+ }
588
+ },
589
+ });
590
+ return { watch, watchdog };
475
591
  }
476
592
 
477
593
  /**
@@ -491,7 +607,9 @@ async function runAgent(
491
607
  const id = record.id;
492
608
  const abortController = new AbortController();
493
609
  const capture = { last: "" };
494
- const timeoutMs = spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs;
610
+ const timeoutMs = effectiveTimeout(spec.timeoutMs);
611
+ const trail = openTrail(id);
612
+ const { watch, watchdog } = watchRun(id, trail, abortController);
495
613
 
496
614
  // Registered as queued, bound, then started — so a kill arriving in the
497
615
  // gap between the task existing and the abort handle being published still
@@ -509,12 +627,15 @@ async function runAgent(
509
627
 
510
628
  let settled: AgentRecord | null = null;
511
629
  try {
630
+ const scratchDir = await openScratch(id);
512
631
  const params = await buildRunParams(
513
632
  record,
514
633
  spec,
515
634
  model,
516
635
  abortController,
517
636
  capture,
637
+ trail,
638
+ scratchDir,
518
639
  resume,
519
640
  );
520
641
  if (agentRegistry.isInterrupted(id)) {
@@ -528,12 +649,13 @@ async function runAgent(
528
649
  id,
529
650
  task,
530
651
  new Error("aborted before the run started"),
652
+ trail.snapshot(),
531
653
  );
532
654
  } else {
533
655
  const usage = await runIsolatedAgent({
534
656
  background,
535
657
  params,
536
- timeoutMs,
658
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
537
659
  logCategory: "agents",
538
660
  // Safe to sweep: the context label is unique to this agent, so no
539
661
  // other context's subprocesses share the tag.
@@ -546,18 +668,37 @@ async function runAgent(
546
668
  if (agentRegistry.isInterrupted(id)) {
547
669
  settled = null;
548
670
  } else {
549
- recordBackendRunSuccess(record.backendId);
550
- settled = settleSuccess(id, task, capture.last, usage ?? undefined);
671
+ if (watch.stalled) {
672
+ // The backend swallowed the watchdog's abort and returned.
673
+ settled = settleFailure(
674
+ id,
675
+ task,
676
+ watch.stalled,
677
+ trail.snapshot(),
678
+ watch.stalled,
679
+ );
680
+ } else {
681
+ recordBackendRunSuccess(record.backendId);
682
+ settled = settleSuccess(
683
+ id,
684
+ task,
685
+ capture.last,
686
+ usage ?? undefined,
687
+ trail.snapshot(),
688
+ );
689
+ }
551
690
  }
552
691
  }
553
692
  } catch (err) {
554
693
  if (agentRegistry.isInterrupted(id)) {
555
694
  settled = null;
556
695
  } else {
557
- recordBackendRunFailure(record.backendId, err);
558
- settled = settleFailure(id, task, err);
696
+ if (!watch.stalled) recordBackendRunFailure(record.backendId, err);
697
+ settled = settleFailure(id, task, err, trail.snapshot(), watch.stalled);
559
698
  }
560
699
  } finally {
700
+ watchdog.stop();
701
+ closeTrail(id);
561
702
  await release().catch((err: unknown) =>
562
703
  logError("agents", `failed to release backend for ${id}`, err),
563
704
  );
@@ -573,10 +714,12 @@ async function runAgent(
573
714
  return;
574
715
  }
575
716
  if (!settled) return;
717
+ // Removed on success; kept on failure for whoever picks the work up.
718
+ await closeScratch(id, settled.state);
576
719
  log(
577
720
  "agents",
578
721
  `${id} "${settled.label}" → ${settled.state} ` +
579
- `(${settled.backendId}/${model}, ${timeoutMs}ms cap)`,
722
+ `(${settled.backendId}/${model}, timeout ${describeTimeout(timeoutMs)})`,
580
723
  );
581
724
  reapChildren(settled);
582
725
  await deliverSettlement(settled).catch((err: unknown) =>
@@ -667,6 +810,7 @@ async function settleRestored(
667
810
  ): Promise<void> {
668
811
  const settled = agentRegistry.settle(record.id, patch);
669
812
  if (!settled) return;
813
+ await closeScratch(record.id, settled.state);
670
814
  log(
671
815
  "agents",
672
816
  `${record.id} "${record.label}" → ${settled.state} (after restart)`,
@@ -812,9 +956,12 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
812
956
  interruptedAt,
813
957
  };
814
958
 
815
- const budget =
816
- (saved.timeoutMs ?? capsHolder.caps.defaultTimeoutMs) - elapsedMs;
817
- const timeoutMs = Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, budget);
959
+ // An uncapped run stays uncapped; a capped one gets what it had left.
960
+ const cap = effectiveTimeout(saved.timeoutMs);
961
+ const timeoutMs =
962
+ cap === undefined
963
+ ? undefined
964
+ : Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, cap - elapsedMs);
818
965
  const spec: AgentSpawnSpec = {
819
966
  brief: saved.brief,
820
967
  label: saved.label,
@@ -824,7 +971,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
824
971
  ...(record.reasoningEffort
825
972
  ? { reasoningEffort: record.reasoningEffort }
826
973
  : {}),
827
- timeoutMs,
974
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
828
975
  ...(saved.preflight ? { preflight: true } : {}),
829
976
  };
830
977
  agentRegistry.markResumed(record.id);
@@ -832,7 +979,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
832
979
  "agents",
833
980
  `${record.id} "${record.label}" resuming after restart ` +
834
981
  `(${canResume ? `session ${saved.sessionId}` : "re-briefed"}, ` +
835
- `${backendId}/${resolved.model}, ${Math.round(timeoutMs / 1000)}s left, ` +
982
+ `${backendId}/${resolved.model}, timeout ${describeTimeout(timeoutMs)}, ` +
836
983
  `resume #${saved.resumeCount + 1})`,
837
984
  );
838
985
  void runAgent(record, spec, resolved, acquired, plan);
@@ -0,0 +1,80 @@
1
+ /**
2
+ * Scratch — a private temp directory per sub-agent.
3
+ *
4
+ * Every agent gets `/tmp/talon-agents/<id>/` (under `os.tmpdir()`), created
5
+ * when its run starts and exported as `TMPDIR` / `TMP` / `TEMP` to the
6
+ * backend run (`OneShotAgentParams.env`), so its shells and tools write
7
+ * their temporary files somewhere no other agent is writing. The directory
8
+ * is named in the agent's system prompt too, for backends that cannot take
9
+ * a per-run environment.
10
+ *
11
+ * Lifecycle: kept across a daemon restart (a resumed agent finds its files
12
+ * where it left them), removed when the agent settles `done`, and **kept**
13
+ * on any other terminal state so whoever picks up a failed, killed or
14
+ * timed-out run can inspect what it left behind.
15
+ */
16
+
17
+ import { mkdir, rm } from "node:fs/promises";
18
+ import { tmpdir } from "node:os";
19
+ import { join } from "node:path";
20
+ import { logWarn } from "../../util/log.js";
21
+ import type { AgentState } from "./types.js";
22
+
23
+ /** Root of every agent's scratch dir. Overridable for tests. */
24
+ const root: { dir: string } = { dir: join(tmpdir(), "talon-agents") };
25
+
26
+ export function setScratchRootForTest(dir: string): void {
27
+ root.dir = dir;
28
+ }
29
+
30
+ /** Absolute path of an agent's scratch dir (whether or not it exists). */
31
+ export function agentScratchDir(agentId: string): string {
32
+ return join(root.dir, agentId);
33
+ }
34
+
35
+ /** The env a run with this scratch dir is given. */
36
+ export function scratchEnv(dir: string): Record<string, string> {
37
+ return { TMPDIR: dir, TMP: dir, TEMP: dir };
38
+ }
39
+
40
+ /**
41
+ * Create (or reuse, on resume) the agent's scratch dir. Returns its path,
42
+ * or `undefined` when it could not be created — a run is never refused for
43
+ * want of a temp dir; it just shares the system one.
44
+ */
45
+ export async function openScratch(
46
+ agentId: string,
47
+ ): Promise<string | undefined> {
48
+ const dir = agentScratchDir(agentId);
49
+ try {
50
+ await mkdir(dir, { recursive: true, mode: 0o700 });
51
+ return dir;
52
+ } catch (err) {
53
+ logWarn(
54
+ "agents",
55
+ `could not create scratch dir ${dir}: ${err instanceof Error ? err.message : String(err)}`,
56
+ );
57
+ return undefined;
58
+ }
59
+ }
60
+
61
+ /**
62
+ * Clean up after a settled agent: removed on `done`, kept otherwise.
63
+ * Returns whether the directory was removed. Never throws.
64
+ */
65
+ export async function closeScratch(
66
+ agentId: string,
67
+ state: AgentState,
68
+ ): Promise<boolean> {
69
+ if (state !== "done") return false;
70
+ try {
71
+ await rm(agentScratchDir(agentId), { recursive: true, force: true });
72
+ return true;
73
+ } catch (err) {
74
+ logWarn(
75
+ "agents",
76
+ `could not remove scratch dir for ${agentId}: ${err instanceof Error ? err.message : String(err)}`,
77
+ );
78
+ return false;
79
+ }
80
+ }
@@ -0,0 +1,141 @@
1
+ /**
2
+ * Trail — what a running sub-agent has been doing, kept so a run that is
3
+ * cut short (killed, timed out, stalled, failed) still hands its parent
4
+ * something to work with.
5
+ *
6
+ * Three bounded lists, all best-effort:
7
+ *
8
+ * - **messages** — its last interim `message_parent` notes.
9
+ * - **notes** — its last assistant texts (progress narration).
10
+ * - **files** — paths it wrote or edited, scraped from the run log the
11
+ * backend writes: Claude-style `**Tool call:** \`Write\`` blocks carrying
12
+ * a `file_path` / `path` / `notebook_path`, and Codex `**File changes:**`
13
+ * lists. A backend that logs neither simply contributes no files.
14
+ *
15
+ * The trail also carries the run's `lastActivityAt` clock, which the
16
+ * no-progress watchdog reads: any log line or assistant text counts.
17
+ *
18
+ * Trails live in memory only, keyed by agent id, for the life of the run.
19
+ */
20
+
21
+ import type { AgentTrail } from "./types.js";
22
+
23
+ const MAX_MESSAGES = 3;
24
+ const MAX_NOTES = 3;
25
+ const MAX_FILES = 50;
26
+ /** One note or message is clipped to this many characters in the trail. */
27
+ const MAX_TEXT_CHARS = 1_500;
28
+
29
+ /** Tool names (bare or MCP-prefixed) that write files. */
30
+ const WRITE_TOOL = /(?:^|__)(?:Write|Edit|MultiEdit|NotebookEdit|write|edit)$/;
31
+ const TOOL_CALL_BLOCK =
32
+ /\*\*(?:MCP )?Tool call:\*\* `([^`]+)`\s*```json\n([\s\S]*?)\n```/g;
33
+ const PATH_KEY = /"(?:file_path|notebook_path|path)":\s*"((?:[^"\\]|\\.)+)"/;
34
+ const FILE_CHANGES_BLOCK = /\*\*File changes:\*\*[^\n]*\n((?:\s+- .*\n?)+)/g;
35
+
36
+ function clip(text: string): string {
37
+ const t = text.trim();
38
+ return t.length > MAX_TEXT_CHARS ? `${t.slice(0, MAX_TEXT_CHARS)}…` : t;
39
+ }
40
+
41
+ function pushBounded(list: string[], item: string, max: number): void {
42
+ list.push(item);
43
+ if (list.length > max) list.splice(0, list.length - max);
44
+ }
45
+
46
+ /** File paths a chunk of run-log text says were written or edited. */
47
+ export function filesFromLogChunk(chunk: string): string[] {
48
+ const out: string[] = [];
49
+ for (const m of chunk.matchAll(TOOL_CALL_BLOCK)) {
50
+ const tool = m[1] ?? "";
51
+ // MCP calls are logged as `server.tool`; normalise to the `__` form.
52
+ if (!WRITE_TOOL.test(tool.replace(/\./g, "__"))) continue;
53
+ const path = PATH_KEY.exec(m[2] ?? "")?.[1];
54
+ if (path) out.push(JSON.parse(`"${path}"`) as string);
55
+ }
56
+ for (const m of chunk.matchAll(FILE_CHANGES_BLOCK)) {
57
+ for (const line of (m[1] ?? "").split("\n")) {
58
+ const path = /^\s+- \S+ (.+)$/.exec(line)?.[1]?.trim();
59
+ if (path && path !== "?") out.push(path);
60
+ }
61
+ }
62
+ return out;
63
+ }
64
+
65
+ /** The live trail of one run. */
66
+ export class RunTrail {
67
+ private readonly messages: string[] = [];
68
+ private readonly notes: string[] = [];
69
+ private readonly files: string[] = [];
70
+ lastActivityAt: number;
71
+
72
+ constructor(now: number = Date.now()) {
73
+ this.lastActivityAt = now;
74
+ }
75
+
76
+ /** Any sign of life — resets the watchdog. */
77
+ touch(now: number = Date.now()): void {
78
+ this.lastActivityAt = now;
79
+ }
80
+
81
+ /** A chunk the backend appended to the run log. */
82
+ onLog(chunk: string): void {
83
+ this.touch();
84
+ for (const file of filesFromLogChunk(chunk)) {
85
+ if (this.files.includes(file)) continue;
86
+ pushBounded(this.files, file, MAX_FILES);
87
+ }
88
+ }
89
+
90
+ onAssistantText(text: string): void {
91
+ this.touch();
92
+ const t = clip(text);
93
+ if (t) pushBounded(this.notes, t, MAX_NOTES);
94
+ }
95
+
96
+ onMessage(text: string): void {
97
+ this.touch();
98
+ const t = clip(text);
99
+ if (t) pushBounded(this.messages, t, MAX_MESSAGES);
100
+ }
101
+
102
+ snapshot(): AgentTrail {
103
+ return {
104
+ messages: [...this.messages],
105
+ notes: [...this.notes],
106
+ files: [...this.files],
107
+ };
108
+ }
109
+ }
110
+
111
+ const trails = new Map<string, RunTrail>();
112
+
113
+ /** Start (or restart, on resume) the trail for a run. */
114
+ export function openTrail(agentId: string): RunTrail {
115
+ const trail = new RunTrail();
116
+ trails.set(agentId, trail);
117
+ return trail;
118
+ }
119
+
120
+ export function getTrail(agentId: string): RunTrail | undefined {
121
+ return trails.get(agentId);
122
+ }
123
+
124
+ export function closeTrail(agentId: string): void {
125
+ trails.delete(agentId);
126
+ }
127
+
128
+ /** Record an interim `message_parent` note against a running agent. */
129
+ export function recordInterimMessage(agentId: string, text: string): void {
130
+ trails.get(agentId)?.onMessage(text);
131
+ }
132
+
133
+ /** Whether a trail snapshot has anything worth showing. */
134
+ export function trailIsEmpty(trail: AgentTrail | undefined): boolean {
135
+ return (
136
+ !trail ||
137
+ (trail.messages.length === 0 &&
138
+ trail.notes.length === 0 &&
139
+ trail.files.length === 0)
140
+ );
141
+ }
@@ -87,6 +87,21 @@ export interface AgentRecord {
87
87
  readonly children: readonly string[];
88
88
  /** Messages waiting to be drained by `check_inbox`. */
89
89
  readonly inboxDepth: number;
90
+ /**
91
+ * What the run had been doing when it settled — set on every settlement
92
+ * the runner makes, shown to the parent when the run did not end `done`.
93
+ */
94
+ readonly trail?: AgentTrail;
95
+ }
96
+
97
+ /** A settled run's last interim messages, progress notes and changed files. */
98
+ export interface AgentTrail {
99
+ /** Last `message_parent` notes, oldest first. */
100
+ readonly messages: readonly string[];
101
+ /** Last assistant texts, oldest first. */
102
+ readonly notes: readonly string[];
103
+ /** Files it wrote or edited (best-effort, from the run log). */
104
+ readonly files: readonly string[];
90
105
  }
91
106
 
92
107
  /** What `spawnAgent` is asked for. */
@@ -106,7 +121,11 @@ export interface AgentSpawnSpec {
106
121
  */
107
122
  readonly model?: string;
108
123
  readonly reasoningEffort?: ReasoningEffortLevel;
109
- /** Hard wall-clock cap. Defaults to `agents.defaultTimeoutMs`. */
124
+ /**
125
+ * Hard wall-clock cap. Unset = `agents.defaultTimeoutMs`, and with that
126
+ * unset too, no cap at all — the no-progress watchdog is what ends a run
127
+ * that has gone quiet.
128
+ */
110
129
  readonly timeoutMs?: number;
111
130
  /**
112
131
  * Append the pre-flight lane instruction (run `npm run preflight` before
@@ -137,8 +156,15 @@ export interface AgentCaps {
137
156
  readonly maxConcurrent: number;
138
157
  /** Deepest `depth` an agent may have — 2 means chat → A → B. */
139
158
  readonly maxDepth: number;
140
- /** Default hard timeout for one run. */
141
- readonly defaultTimeoutMs: number;
159
+ /** Hard timeout for a spawn that sets none. Unset = no hard timeout. */
160
+ readonly defaultTimeoutMs?: number;
161
+ /** Ceiling on any run's hard timeout, requested or not. Unset = none. */
162
+ readonly maxTimeoutMs?: number;
163
+ /**
164
+ * No-progress watchdog step N: ping the agent after N ms of silence, warn
165
+ * its parent after 2N, kill it after 3N. 0 disables the watchdog.
166
+ */
167
+ readonly stallTimeoutMs: number;
142
168
  /**
143
169
  * Backends a sub-agent may run on. Unset or empty = any backend with a
144
170
  * background capability. Enforced by `spawnAgent` on the final choice.