talon-agent 5.29.0 → 5.30.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/package.json +2 -1
  2. package/prompts/system/agent-brief.md +9 -6
  3. package/src/backend/claude-sdk/constants.ts +22 -0
  4. package/src/backend/claude-sdk/models/discovery.ts +3 -0
  5. package/src/backend/claude-sdk/one-shot.ts +3 -1
  6. package/src/backend/claude-sdk/options.ts +7 -1
  7. package/src/core/agents/index.ts +2 -0
  8. package/src/core/agents/prompt.ts +70 -2
  9. package/src/core/agents/registry.ts +47 -5
  10. package/src/core/agents/runner.ts +169 -31
  11. package/src/core/agents/trail.ts +141 -0
  12. package/src/core/agents/types.ts +29 -3
  13. package/src/core/agents/watchdog.ts +70 -0
  14. package/src/core/background/isolated-agent.ts +6 -2
  15. package/src/core/backup/plan.ts +4 -0
  16. package/src/core/config/index.ts +13 -4
  17. package/src/core/engine/gateway-actions/agents/control.ts +9 -2
  18. package/src/core/engine/gateway-actions/agents/preflight.ts +17 -1
  19. package/src/core/engine/gateway-actions/agents/report.ts +81 -24
  20. package/src/core/engine/gateway-actions/index.ts +3 -0
  21. package/src/core/mcp-hub/guest-scope.ts +3 -1
  22. package/src/core/mesh/devices/service.ts +7 -0
  23. package/src/core/mesh/links/bridge-links.ts +20 -0
  24. package/src/core/secrets/actions.ts +18 -0
  25. package/src/core/secrets/drop.ts +176 -0
  26. package/src/core/secrets/index.ts +11 -0
  27. package/src/core/secrets/service.ts +248 -0
  28. package/src/core/secrets/store.ts +100 -0
  29. package/src/core/tools/index.ts +2 -0
  30. package/src/core/tools/ops/agents.ts +26 -8
  31. package/src/core/tools/ops/secrets.ts +36 -0
  32. package/src/core/tools/types.ts +2 -1
  33. package/src/frontend/discord/commands/definitions.ts +21 -0
  34. package/src/frontend/discord/commands/router.ts +3 -0
  35. package/src/frontend/discord/commands/secret.ts +26 -0
  36. package/src/frontend/native/bridge/routes/host.ts +10 -0
  37. package/src/frontend/native/bridge/routes/pre-auth.ts +82 -1
  38. package/src/frontend/native/bridge/routes/table.ts +5 -0
  39. package/src/frontend/native/bridge/server.ts +32 -4
  40. package/src/frontend/native/commands/definitions.ts +6 -0
  41. package/src/frontend/native/commands/index.ts +12 -0
  42. package/src/frontend/native/surface/handlers.ts +9 -0
  43. package/src/frontend/telegram/commands/definitions.ts +4 -0
  44. package/src/frontend/telegram/commands/index.ts +3 -0
  45. package/src/frontend/telegram/commands/secret.ts +24 -0
  46. package/src/frontend/whatsapp/commands.ts +16 -1
  47. package/src/util/log.ts +1 -0
  48. package/src/util/paths.ts +6 -0
@@ -4,8 +4,8 @@
4
4
  *
5
5
  * The shape is the heartbeat / cron-job shape, because a sub-agent *is* one
6
6
  * of those: acquire a backend, resolve a model, open a run log, register a
7
- * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the hard
8
- * timeout → abort → grace → eviction discipline. Nothing here is
7
+ * task, and hand `runOneShotAgent` to `runIsolatedAgent` for the (optional)
8
+ * hard timeout → abort → grace → eviction discipline. Nothing here is
9
9
  * backend-specific, which is the whole point: sub-agents work on Claude,
10
10
  * Codex, Kilo, OpenCode and any future backend with a background capability.
11
11
  *
@@ -65,6 +65,7 @@ import {
65
65
  import { openRunLog } from "../background/run-log.js";
66
66
  import { agentContextLabel } from "./context.js";
67
67
  import {
68
+ deliverMessage,
68
69
  deliverSettlement,
69
70
  initAgentDelivery,
70
71
  type AgentDeliveryDeps,
@@ -77,7 +78,11 @@ import {
77
78
  buildAgentSystemPrompt,
78
79
  buildRebriefPrompt,
79
80
  buildResumePrompt,
81
+ buildStallPing,
82
+ buildStallWarning,
80
83
  } from "./prompt.js";
84
+ import { closeTrail, openTrail, type RunTrail } from "./trail.js";
85
+ import { startWatchdog, type WatchdogHandle } from "./watchdog.js";
81
86
  import * as agentsRepo from "../../storage/agents/repo.js";
82
87
  import type { PersistedAgent } from "../../storage/agents/repo.js";
83
88
  import { agentRegistry } from "./registry.js";
@@ -87,18 +92,32 @@ import type {
87
92
  AgentRecord,
88
93
  AgentSpawnOutcome,
89
94
  AgentSpawnSpec,
95
+ AgentTrail,
90
96
  } from "./types.js";
91
97
 
92
- /** Defaults for `config.agents`, applied when the block is absent. */
98
+ /**
99
+ * Defaults for `config.agents`, applied when the block is absent. No hard
100
+ * timeout: the no-progress watchdog ends a run that has gone quiet, and a
101
+ * run that is still working is left to finish.
102
+ */
93
103
  export const DEFAULT_AGENT_CAPS: AgentCaps = {
94
104
  maxConcurrent: 6,
95
105
  maxDepth: 2,
96
- defaultTimeoutMs: 15 * 60 * 1000,
106
+ stallTimeoutMs: 15 * 60 * 1000,
97
107
  };
98
108
 
99
- /** Floor and ceiling the tool boundary clamps a requested `timeout_s` into. */
109
+ /** Floor the tool boundary clamps a requested `timeout_s` up to. */
100
110
  const MIN_TIMEOUT_MS = 30_000;
101
- const MAX_TIMEOUT_MS = 60 * 60 * 1000;
111
+
112
+ /** Raised to abort a run the no-progress watchdog gave up on. */
113
+ class AgentStalledError extends Error {
114
+ constructor(idleMs: number) {
115
+ super(
116
+ `stalled: no tool call or output for ${Math.round(idleMs / 60_000)} min`,
117
+ );
118
+ this.name = "AgentStalledError";
119
+ }
120
+ }
102
121
 
103
122
  const capsHolder: { caps: AgentCaps } = { caps: DEFAULT_AGENT_CAPS };
104
123
 
@@ -112,7 +131,9 @@ export function initAgents(
112
131
  "agents",
113
132
  `Initialized — maxConcurrent=${capsHolder.caps.maxConcurrent} ` +
114
133
  `maxDepth=${capsHolder.caps.maxDepth} ` +
115
- `timeout=${Math.round(capsHolder.caps.defaultTimeoutMs / 1000)}s` +
134
+ `timeout=${describeTimeout(capsHolder.caps.defaultTimeoutMs)} ` +
135
+ `ceiling=${describeTimeout(capsHolder.caps.maxTimeoutMs)} ` +
136
+ `stall=${describeTimeout(capsHolder.caps.stallTimeoutMs || undefined)}` +
116
137
  (capsHolder.caps.allowedBackends?.length
117
138
  ? ` allowedBackends=${capsHolder.caps.allowedBackends.join(",")}`
118
139
  : ""),
@@ -124,14 +145,38 @@ export function getAgentCaps(): AgentCaps {
124
145
  return capsHolder.caps;
125
146
  }
126
147
 
148
+ /** "15m" / "90s" / "none" — for logs and tool text. */
149
+ export function describeTimeout(ms: number | undefined): string {
150
+ if (ms === undefined || !(ms > 0)) return "none";
151
+ return ms % 60_000 === 0 ? `${ms / 60_000}m` : `${Math.round(ms / 1000)}s`;
152
+ }
153
+
127
154
  /**
128
- * Clamp a model-supplied timeout into the supported window, or fall back to
129
- * the configured default. Applied at the tool boundary — `spawnAgent` itself
130
- * honours whatever it is handed, so the runner has one rule and not two.
155
+ * The hard timeout a run gets: the requested one, else
156
+ * `agents.defaultTimeoutMs`, either capped by `agents.maxTimeoutMs`; with
157
+ * none of those set, `undefined` — no hard timeout. Applied by the runner,
158
+ * so a resumed run follows the same rule as a fresh one.
131
159
  */
132
- export function clampTimeout(requestedMs: number | undefined): number {
133
- if (requestedMs === undefined) return capsHolder.caps.defaultTimeoutMs;
134
- return Math.min(MAX_TIMEOUT_MS, Math.max(MIN_TIMEOUT_MS, requestedMs));
160
+ function effectiveTimeout(requestedMs: number | undefined): number | undefined {
161
+ const { defaultTimeoutMs, maxTimeoutMs } = capsHolder.caps;
162
+ const base = requestedMs ?? defaultTimeoutMs;
163
+ if (base === undefined) return maxTimeoutMs;
164
+ return maxTimeoutMs !== undefined ? Math.min(maxTimeoutMs, base) : base;
165
+ }
166
+
167
+ /**
168
+ * The tool boundary's rule: a model-supplied timeout is floored at 30s,
169
+ * then resolved like any other (`effectiveTimeout`). Returns `undefined`
170
+ * for "no hard timeout".
171
+ */
172
+ export function clampTimeout(
173
+ requestedMs: number | undefined,
174
+ ): number | undefined {
175
+ return effectiveTimeout(
176
+ requestedMs !== undefined && Number.isFinite(requestedMs)
177
+ ? Math.max(MIN_TIMEOUT_MS, requestedMs)
178
+ : undefined,
179
+ );
135
180
  }
136
181
 
137
182
  /** The backend an agent inherits when the caller didn't pick one. */
@@ -326,7 +371,7 @@ export async function spawnAgent(
326
371
  ? { reasoningEffort: spec.reasoningEffort }
327
372
  : {}),
328
373
  ...(spec.model ? { requestedModel: spec.model } : {}),
329
- timeoutMs: spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs,
374
+ ...(spec.timeoutMs !== undefined ? { timeoutMs: spec.timeoutMs } : {}),
330
375
  cwd: dirs.workspace,
331
376
  ...(spec.preflight ? { preflight: true } : {}),
332
377
  },
@@ -382,9 +427,10 @@ async function buildRunParams(
382
427
  model: string,
383
428
  abortController: AbortController,
384
429
  capture: { last: string },
430
+ trail: RunTrail,
385
431
  resume?: ResumePlan,
386
432
  ): Promise<OneShotAgentParams> {
387
- const appendLog = await openRunLog(
433
+ const writeLog = await openRunLog(
388
434
  agentLogPath(record.id),
389
435
  resume
390
436
  ? agentResumeLogHeader(
@@ -396,6 +442,10 @@ async function buildRunParams(
396
442
  : agentLogHeader(record, model),
397
443
  );
398
444
  const id = record.id;
445
+ const appendLog = (text: string): Promise<void> => {
446
+ trail.onLog(text);
447
+ return writeLog(text);
448
+ };
399
449
  return {
400
450
  prompt: resume
401
451
  ? resume.prompt
@@ -417,6 +467,7 @@ async function buildRunParams(
417
467
  onAssistantText: (text) => {
418
468
  const trimmed = text.trim();
419
469
  if (trimmed) capture.last = trimmed;
470
+ trail.onAssistantText(text);
420
471
  },
421
472
  // Persisted the moment the backend reports it, so a restart at any
422
473
  // point after the first message can resume the conversation.
@@ -432,11 +483,13 @@ function settleSuccess(
432
483
  task: TaskHandle,
433
484
  lastText: string,
434
485
  usage: TaskUsage | undefined,
486
+ trail: AgentTrail,
435
487
  ): AgentRecord | null {
436
488
  if (agentRegistry.hasReported(id)) {
437
489
  task.succeed(usage);
438
490
  return agentRegistry.settle(id, {
439
491
  state: "done",
492
+ trail,
440
493
  ...(usage ? { usage } : {}),
441
494
  });
442
495
  }
@@ -445,6 +498,7 @@ function settleSuccess(
445
498
  return agentRegistry.settle(id, {
446
499
  state: "done",
447
500
  result: { summary: lastText },
501
+ trail,
448
502
  ...(usage ? { usage } : {}),
449
503
  });
450
504
  }
@@ -454,24 +508,82 @@ function settleSuccess(
454
508
  return agentRegistry.settle(id, {
455
509
  state: "failed",
456
510
  error,
511
+ trail,
457
512
  ...(usage ? { usage } : {}),
458
513
  });
459
514
  }
460
515
 
461
- /** Settle a run that threw: timeout, kill, or a genuine failure. */
516
+ /**
517
+ * Settle a run that threw: timeout, stall, kill, or a genuine failure.
518
+ * `stalled` is the watchdog's own abort reason, checked first because a
519
+ * backend that honours the abort rejects with its own error.
520
+ */
462
521
  function settleFailure(
463
522
  id: string,
464
523
  task: TaskHandle,
465
524
  err: unknown,
525
+ trail: AgentTrail,
526
+ stalled?: AgentStalledError,
466
527
  ): AgentRecord | null {
467
528
  const state =
468
- err instanceof IsolatedAgentTimeoutError
529
+ stalled || err instanceof IsolatedAgentTimeoutError
469
530
  ? "timed_out"
470
531
  : agentRegistry.killRequested(id)
471
532
  ? "killed"
472
533
  : "failed";
473
- task.fail(err);
474
- return agentRegistry.settle(id, { state, error: errText(err) });
534
+ task.fail(stalled ?? err);
535
+ return agentRegistry.settle(id, {
536
+ state,
537
+ error: errText(stalled ?? err),
538
+ trail,
539
+ });
540
+ }
541
+
542
+ /**
543
+ * Start the no-progress watchdog for one run (see `watchdog.ts`). Its kill
544
+ * aborts the run with an `AgentStalledError` recorded in `watch.stalled`,
545
+ * which the settle path reads to classify the run `timed_out`.
546
+ */
547
+ function watchRun(
548
+ id: string,
549
+ trail: RunTrail,
550
+ abortController: AbortController,
551
+ ): { watch: { stalled?: AgentStalledError }; watchdog: WatchdogHandle } {
552
+ const watch: { stalled?: AgentStalledError } = {};
553
+ const watchdog = startWatchdog(capsHolder.caps.stallTimeoutMs, {
554
+ lastActivityAt: () => trail.lastActivityAt,
555
+ pingAgent: (idleMs) => {
556
+ agentRegistry.push(id, {
557
+ from: "watchdog",
558
+ text: buildStallPing(idleMs, capsHolder.caps.stallTimeoutMs),
559
+ at: Date.now(),
560
+ });
561
+ logWarn(
562
+ "agents",
563
+ `${id} quiet for ${Math.round(idleMs / 1000)}s — pinged`,
564
+ );
565
+ },
566
+ warnParent: (idleMs, killInMs) => {
567
+ const current = agentRegistry.get(id);
568
+ if (!current) return;
569
+ void deliverMessage(
570
+ current,
571
+ buildStallWarning(current, idleMs, killInMs),
572
+ ).catch((err: unknown) =>
573
+ logError("agents", `stall warning delivery failed for ${id}`, err),
574
+ );
575
+ },
576
+ kill: (idleMs) => {
577
+ watch.stalled = new AgentStalledError(idleMs);
578
+ logWarn("agents", `${id}: ${watch.stalled.message} — aborting`);
579
+ try {
580
+ abortController.abort(watch.stalled);
581
+ } catch {
582
+ /* the settle path below still runs */
583
+ }
584
+ },
585
+ });
586
+ return { watch, watchdog };
475
587
  }
476
588
 
477
589
  /**
@@ -491,7 +603,9 @@ async function runAgent(
491
603
  const id = record.id;
492
604
  const abortController = new AbortController();
493
605
  const capture = { last: "" };
494
- const timeoutMs = spec.timeoutMs ?? capsHolder.caps.defaultTimeoutMs;
606
+ const timeoutMs = effectiveTimeout(spec.timeoutMs);
607
+ const trail = openTrail(id);
608
+ const { watch, watchdog } = watchRun(id, trail, abortController);
495
609
 
496
610
  // Registered as queued, bound, then started — so a kill arriving in the
497
611
  // gap between the task existing and the abort handle being published still
@@ -515,6 +629,7 @@ async function runAgent(
515
629
  model,
516
630
  abortController,
517
631
  capture,
632
+ trail,
518
633
  resume,
519
634
  );
520
635
  if (agentRegistry.isInterrupted(id)) {
@@ -528,12 +643,13 @@ async function runAgent(
528
643
  id,
529
644
  task,
530
645
  new Error("aborted before the run started"),
646
+ trail.snapshot(),
531
647
  );
532
648
  } else {
533
649
  const usage = await runIsolatedAgent({
534
650
  background,
535
651
  params,
536
- timeoutMs,
652
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
537
653
  logCategory: "agents",
538
654
  // Safe to sweep: the context label is unique to this agent, so no
539
655
  // other context's subprocesses share the tag.
@@ -546,18 +662,37 @@ async function runAgent(
546
662
  if (agentRegistry.isInterrupted(id)) {
547
663
  settled = null;
548
664
  } else {
549
- recordBackendRunSuccess(record.backendId);
550
- settled = settleSuccess(id, task, capture.last, usage ?? undefined);
665
+ if (watch.stalled) {
666
+ // The backend swallowed the watchdog's abort and returned.
667
+ settled = settleFailure(
668
+ id,
669
+ task,
670
+ watch.stalled,
671
+ trail.snapshot(),
672
+ watch.stalled,
673
+ );
674
+ } else {
675
+ recordBackendRunSuccess(record.backendId);
676
+ settled = settleSuccess(
677
+ id,
678
+ task,
679
+ capture.last,
680
+ usage ?? undefined,
681
+ trail.snapshot(),
682
+ );
683
+ }
551
684
  }
552
685
  }
553
686
  } catch (err) {
554
687
  if (agentRegistry.isInterrupted(id)) {
555
688
  settled = null;
556
689
  } else {
557
- recordBackendRunFailure(record.backendId, err);
558
- settled = settleFailure(id, task, err);
690
+ if (!watch.stalled) recordBackendRunFailure(record.backendId, err);
691
+ settled = settleFailure(id, task, err, trail.snapshot(), watch.stalled);
559
692
  }
560
693
  } finally {
694
+ watchdog.stop();
695
+ closeTrail(id);
561
696
  await release().catch((err: unknown) =>
562
697
  logError("agents", `failed to release backend for ${id}`, err),
563
698
  );
@@ -576,7 +711,7 @@ async function runAgent(
576
711
  log(
577
712
  "agents",
578
713
  `${id} "${settled.label}" → ${settled.state} ` +
579
- `(${settled.backendId}/${model}, ${timeoutMs}ms cap)`,
714
+ `(${settled.backendId}/${model}, timeout ${describeTimeout(timeoutMs)})`,
580
715
  );
581
716
  reapChildren(settled);
582
717
  await deliverSettlement(settled).catch((err: unknown) =>
@@ -812,9 +947,12 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
812
947
  interruptedAt,
813
948
  };
814
949
 
815
- const budget =
816
- (saved.timeoutMs ?? capsHolder.caps.defaultTimeoutMs) - elapsedMs;
817
- const timeoutMs = Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, budget);
950
+ // An uncapped run stays uncapped; a capped one gets what it had left.
951
+ const cap = effectiveTimeout(saved.timeoutMs);
952
+ const timeoutMs =
953
+ cap === undefined
954
+ ? undefined
955
+ : Math.max(AGENT_RESUME_MIN_TIMEOUT_MS, cap - elapsedMs);
818
956
  const spec: AgentSpawnSpec = {
819
957
  brief: saved.brief,
820
958
  label: saved.label,
@@ -824,7 +962,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
824
962
  ...(record.reasoningEffort
825
963
  ? { reasoningEffort: record.reasoningEffort }
826
964
  : {}),
827
- timeoutMs,
965
+ ...(timeoutMs !== undefined ? { timeoutMs } : {}),
828
966
  ...(saved.preflight ? { preflight: true } : {}),
829
967
  };
830
968
  agentRegistry.markResumed(record.id);
@@ -832,7 +970,7 @@ async function resumeOne(saved: PersistedAgent, now: number): Promise<void> {
832
970
  "agents",
833
971
  `${record.id} "${record.label}" resuming after restart ` +
834
972
  `(${canResume ? `session ${saved.sessionId}` : "re-briefed"}, ` +
835
- `${backendId}/${resolved.model}, ${Math.round(timeoutMs / 1000)}s left, ` +
973
+ `${backendId}/${resolved.model}, timeout ${describeTimeout(timeoutMs)}, ` +
836
974
  `resume #${saved.resumeCount + 1})`,
837
975
  );
838
976
  void runAgent(record, spec, resolved, acquired, plan);
@@ -0,0 +1,141 @@
1
+ /**
2
+ * Trail — what a running sub-agent has been doing, kept so a run that is
3
+ * cut short (killed, timed out, stalled, failed) still hands its parent
4
+ * something to work with.
5
+ *
6
+ * Three bounded lists, all best-effort:
7
+ *
8
+ * - **messages** — its last interim `message_parent` notes.
9
+ * - **notes** — its last assistant texts (progress narration).
10
+ * - **files** — paths it wrote or edited, scraped from the run log the
11
+ * backend writes: Claude-style `**Tool call:** \`Write\`` blocks carrying
12
+ * a `file_path` / `path` / `notebook_path`, and Codex `**File changes:**`
13
+ * lists. A backend that logs neither simply contributes no files.
14
+ *
15
+ * The trail also carries the run's `lastActivityAt` clock, which the
16
+ * no-progress watchdog reads: any log line or assistant text counts.
17
+ *
18
+ * Trails live in memory only, keyed by agent id, for the life of the run.
19
+ */
20
+
21
+ import type { AgentTrail } from "./types.js";
22
+
23
+ const MAX_MESSAGES = 3;
24
+ const MAX_NOTES = 3;
25
+ const MAX_FILES = 50;
26
+ /** One note or message is clipped to this many characters in the trail. */
27
+ const MAX_TEXT_CHARS = 1_500;
28
+
29
+ /** Tool names (bare or MCP-prefixed) that write files. */
30
+ const WRITE_TOOL = /(?:^|__)(?:Write|Edit|MultiEdit|NotebookEdit|write|edit)$/;
31
+ const TOOL_CALL_BLOCK =
32
+ /\*\*(?:MCP )?Tool call:\*\* `([^`]+)`\s*```json\n([\s\S]*?)\n```/g;
33
+ const PATH_KEY = /"(?:file_path|notebook_path|path)":\s*"((?:[^"\\]|\\.)+)"/;
34
+ const FILE_CHANGES_BLOCK = /\*\*File changes:\*\*[^\n]*\n((?:\s+- .*\n?)+)/g;
35
+
36
+ function clip(text: string): string {
37
+ const t = text.trim();
38
+ return t.length > MAX_TEXT_CHARS ? `${t.slice(0, MAX_TEXT_CHARS)}…` : t;
39
+ }
40
+
41
+ function pushBounded(list: string[], item: string, max: number): void {
42
+ list.push(item);
43
+ if (list.length > max) list.splice(0, list.length - max);
44
+ }
45
+
46
+ /** File paths a chunk of run-log text says were written or edited. */
47
+ export function filesFromLogChunk(chunk: string): string[] {
48
+ const out: string[] = [];
49
+ for (const m of chunk.matchAll(TOOL_CALL_BLOCK)) {
50
+ const tool = m[1] ?? "";
51
+ // MCP calls are logged as `server.tool`; normalise to the `__` form.
52
+ if (!WRITE_TOOL.test(tool.replace(/\./g, "__"))) continue;
53
+ const path = PATH_KEY.exec(m[2] ?? "")?.[1];
54
+ if (path) out.push(JSON.parse(`"${path}"`) as string);
55
+ }
56
+ for (const m of chunk.matchAll(FILE_CHANGES_BLOCK)) {
57
+ for (const line of (m[1] ?? "").split("\n")) {
58
+ const path = /^\s+- \S+ (.+)$/.exec(line)?.[1]?.trim();
59
+ if (path && path !== "?") out.push(path);
60
+ }
61
+ }
62
+ return out;
63
+ }
64
+
65
+ /** The live trail of one run. */
66
+ export class RunTrail {
67
+ private readonly messages: string[] = [];
68
+ private readonly notes: string[] = [];
69
+ private readonly files: string[] = [];
70
+ lastActivityAt: number;
71
+
72
+ constructor(now: number = Date.now()) {
73
+ this.lastActivityAt = now;
74
+ }
75
+
76
+ /** Any sign of life — resets the watchdog. */
77
+ touch(now: number = Date.now()): void {
78
+ this.lastActivityAt = now;
79
+ }
80
+
81
+ /** A chunk the backend appended to the run log. */
82
+ onLog(chunk: string): void {
83
+ this.touch();
84
+ for (const file of filesFromLogChunk(chunk)) {
85
+ if (this.files.includes(file)) continue;
86
+ pushBounded(this.files, file, MAX_FILES);
87
+ }
88
+ }
89
+
90
+ onAssistantText(text: string): void {
91
+ this.touch();
92
+ const t = clip(text);
93
+ if (t) pushBounded(this.notes, t, MAX_NOTES);
94
+ }
95
+
96
+ onMessage(text: string): void {
97
+ this.touch();
98
+ const t = clip(text);
99
+ if (t) pushBounded(this.messages, t, MAX_MESSAGES);
100
+ }
101
+
102
+ snapshot(): AgentTrail {
103
+ return {
104
+ messages: [...this.messages],
105
+ notes: [...this.notes],
106
+ files: [...this.files],
107
+ };
108
+ }
109
+ }
110
+
111
+ const trails = new Map<string, RunTrail>();
112
+
113
+ /** Start (or restart, on resume) the trail for a run. */
114
+ export function openTrail(agentId: string): RunTrail {
115
+ const trail = new RunTrail();
116
+ trails.set(agentId, trail);
117
+ return trail;
118
+ }
119
+
120
+ export function getTrail(agentId: string): RunTrail | undefined {
121
+ return trails.get(agentId);
122
+ }
123
+
124
+ export function closeTrail(agentId: string): void {
125
+ trails.delete(agentId);
126
+ }
127
+
128
+ /** Record an interim `message_parent` note against a running agent. */
129
+ export function recordInterimMessage(agentId: string, text: string): void {
130
+ trails.get(agentId)?.onMessage(text);
131
+ }
132
+
133
+ /** Whether a trail snapshot has anything worth showing. */
134
+ export function trailIsEmpty(trail: AgentTrail | undefined): boolean {
135
+ return (
136
+ !trail ||
137
+ (trail.messages.length === 0 &&
138
+ trail.notes.length === 0 &&
139
+ trail.files.length === 0)
140
+ );
141
+ }
@@ -87,6 +87,21 @@ export interface AgentRecord {
87
87
  readonly children: readonly string[];
88
88
  /** Messages waiting to be drained by `check_inbox`. */
89
89
  readonly inboxDepth: number;
90
+ /**
91
+ * What the run had been doing when it settled — set on every settlement
92
+ * the runner makes, shown to the parent when the run did not end `done`.
93
+ */
94
+ readonly trail?: AgentTrail;
95
+ }
96
+
97
+ /** A settled run's last interim messages, progress notes and changed files. */
98
+ export interface AgentTrail {
99
+ /** Last `message_parent` notes, oldest first. */
100
+ readonly messages: readonly string[];
101
+ /** Last assistant texts, oldest first. */
102
+ readonly notes: readonly string[];
103
+ /** Files it wrote or edited (best-effort, from the run log). */
104
+ readonly files: readonly string[];
90
105
  }
91
106
 
92
107
  /** What `spawnAgent` is asked for. */
@@ -106,7 +121,11 @@ export interface AgentSpawnSpec {
106
121
  */
107
122
  readonly model?: string;
108
123
  readonly reasoningEffort?: ReasoningEffortLevel;
109
- /** Hard wall-clock cap. Defaults to `agents.defaultTimeoutMs`. */
124
+ /**
125
+ * Hard wall-clock cap. Unset = `agents.defaultTimeoutMs`, and with that
126
+ * unset too, no cap at all — the no-progress watchdog is what ends a run
127
+ * that has gone quiet.
128
+ */
110
129
  readonly timeoutMs?: number;
111
130
  /**
112
131
  * Append the pre-flight lane instruction (run `npm run preflight` before
@@ -137,8 +156,15 @@ export interface AgentCaps {
137
156
  readonly maxConcurrent: number;
138
157
  /** Deepest `depth` an agent may have — 2 means chat → A → B. */
139
158
  readonly maxDepth: number;
140
- /** Default hard timeout for one run. */
141
- readonly defaultTimeoutMs: number;
159
+ /** Hard timeout for a spawn that sets none. Unset = no hard timeout. */
160
+ readonly defaultTimeoutMs?: number;
161
+ /** Ceiling on any run's hard timeout, requested or not. Unset = none. */
162
+ readonly maxTimeoutMs?: number;
163
+ /**
164
+ * No-progress watchdog step N: ping the agent after N ms of silence, warn
165
+ * its parent after 2N, kill it after 3N. 0 disables the watchdog.
166
+ */
167
+ readonly stallTimeoutMs: number;
142
168
  /**
143
169
  * Backends a sub-agent may run on. Unset or empty = any backend with a
144
170
  * background capability. Enforced by `spawnAgent` on the final choice.
@@ -0,0 +1,70 @@
1
+ /**
2
+ * No-progress watchdog for a sub-agent run.
3
+ *
4
+ * Replaces the old "everyone gets a 15-minute hard cap" safety net with one
5
+ * that only fires on an agent that has actually gone quiet. With
6
+ * `stallMs = N`, measured from the last sign of life (a run-log line or an
7
+ * assistant text — see `RunTrail`):
8
+ *
9
+ * - after N → the agent is pinged (a message in its own mailbox);
10
+ * - after 2N → its parent is told, so it can intervene or kill early;
11
+ * - after 3N → the run is aborted and settles as `timed_out` ("stalled").
12
+ *
13
+ * Any activity resets the ladder. The watchdog owns one unref'd interval and
14
+ * nothing else; every action is a callback, so it is trivially testable.
15
+ */
16
+
17
+ export interface WatchdogActions {
18
+ /** Last sign of life, epoch ms. */
19
+ readonly lastActivityAt: () => number;
20
+ readonly pingAgent: (idleMs: number) => void;
21
+ readonly warnParent: (idleMs: number, killInMs: number) => void;
22
+ readonly kill: (idleMs: number) => void;
23
+ }
24
+
25
+ export interface WatchdogHandle {
26
+ stop(): void;
27
+ /** Evaluate now — exposed for tests; the interval calls it too. */
28
+ check(now?: number): void;
29
+ }
30
+
31
+ /** Start a watchdog. `stallMs <= 0` returns an inert handle. */
32
+ export function startWatchdog(
33
+ stallMs: number,
34
+ actions: WatchdogActions,
35
+ checkEveryMs: number = Math.max(1_000, Math.min(60_000, stallMs / 4)),
36
+ ): WatchdogHandle {
37
+ if (!(stallMs > 0)) return { stop: () => {}, check: () => {} };
38
+ // 0 = quiet, 1 = pinged, 2 = parent warned, 3 = killed.
39
+ let stage = 0;
40
+ let stageAnchor = actions.lastActivityAt();
41
+
42
+ const check = (now: number = Date.now()): void => {
43
+ if (stage >= 3) return;
44
+ const last = actions.lastActivityAt();
45
+ if (last !== stageAnchor) {
46
+ // Signs of life since the last escalation — start over.
47
+ stageAnchor = last;
48
+ stage = 0;
49
+ }
50
+ const idle = now - last;
51
+ if (stage < 1 && idle >= stallMs) {
52
+ stage = 1;
53
+ actions.pingAgent(idle);
54
+ }
55
+ if (stage < 2 && idle >= 2 * stallMs) {
56
+ stage = 2;
57
+ actions.warnParent(idle, Math.max(0, 3 * stallMs - idle));
58
+ }
59
+ if (stage < 3 && idle >= 3 * stallMs) {
60
+ stage = 3;
61
+ actions.kill(idle);
62
+ stop();
63
+ }
64
+ };
65
+
66
+ const timer = setInterval(() => check(), checkEveryMs);
67
+ timer.unref();
68
+ const stop = (): void => clearInterval(timer);
69
+ return { stop, check };
70
+ }
@@ -37,8 +37,11 @@ export interface IsolatedRunOptions {
37
37
  readonly background: BackgroundRunner;
38
38
  /** Fully-built one-shot params (must carry an `abortController`). */
39
39
  readonly params: OneShotAgentParams;
40
- /** Hard timeout before the run is aborted. */
41
- readonly timeoutMs: number;
40
+ /**
41
+ * Hard timeout before the run is aborted. Unset = none: the run is only
42
+ * ended by its own completion or by an abort of `params.abortController`.
43
+ */
44
+ readonly timeoutMs?: number;
42
45
  /** Bounded grace for the backend to honour the abort (default 30s). */
43
46
  readonly abortGraceMs?: number;
44
47
  /**
@@ -91,6 +94,7 @@ export async function runIsolatedAgent(
91
94
  const agentPromise = background.runOneShotAgent(params);
92
95
 
93
96
  const timeoutPromise = new Promise<never>((_, reject) => {
97
+ if (timeoutMs === undefined) return;
94
98
  timer = setTimeout(() => {
95
99
  timeoutError = new IsolatedAgentTimeoutError(timeoutMs);
96
100
  try {
@@ -75,6 +75,10 @@ export const HOME_INCLUDES: readonly string[] = [
75
75
  "prompts",
76
76
  "data",
77
77
  "keys",
78
+ // Operator secrets (~/.talon/secrets, the secret drop's folder). Held to
79
+ // the same rule as keys/ and workspace/secrets: in the snapshot, so
80
+ // encrypt backups (docs/backup-security.md).
81
+ "secrets",
78
82
  "google",
79
83
  "plugins",
80
84
  "mesh-devices.json",