@arhen/pi-core-subagent 1.3.40 → 1.3.42

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arhen/pi-core-subagent",
3
- "version": "1.3.40",
3
+ "version": "1.3.42",
4
4
  "type": "module",
5
5
  "description": "pi extension: fast in-process subagents with a dependency-graph scheduler (needs edges gate tasks and carry upstream output into dependent prompts), plus background runs, intercom and agent-to-agent mailbox. Leader defines agents inline.",
6
6
  "license": "MIT",
package/src/child.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Child-session tools + watchdog.
2
+ * Child-session tools.
3
3
  *
4
4
  * When `allowIntercom: true`, children get four talk tools:
5
5
  * - ask_parent blocking Q&A with the leader (parent)
@@ -107,49 +107,3 @@ export function createChildTools(taskId: string, handlers: ChildHandlers): ToolD
107
107
  },
108
108
  ];
109
109
  }
110
-
111
- /**
112
- * Watchdog — a subagent is "stalled" when it produces no events for stallMs.
113
- * In-process AgentSessions have no process-exit signal, so we synthesize one
114
- * via event-heartbeat. Every relevant child event must call touch().
115
- */
116
- export interface Watchdog {
117
- touch(): void;
118
- readonly promise: Promise<never>;
119
- dispose(): void;
120
- }
121
-
122
- export function createWatchdog(stallMs: number, label: string): Watchdog {
123
- let lastEventAt = Date.now();
124
- let disposed = false;
125
- let timer: ReturnType<typeof setInterval> | undefined;
126
-
127
- const promise = new Promise<never>((_, reject) => {
128
- const interval = Math.max(1000, Math.min(5000, Math.floor(stallMs / 4)));
129
- timer = setInterval(() => {
130
- if (disposed) return;
131
- if (Date.now() - lastEventAt > stallMs) {
132
- disposed = true;
133
- if (timer) clearInterval(timer);
134
- timer = undefined;
135
- reject(new Error(`${label} stalled: no activity for ${Math.round(stallMs / 1000)}s`));
136
- }
137
- }, interval);
138
- timer.unref?.();
139
- });
140
-
141
- return {
142
- touch() {
143
- lastEventAt = Date.now();
144
- },
145
- promise,
146
- dispose() {
147
- if (disposed) return;
148
- disposed = true;
149
- if (timer) {
150
- clearInterval(timer);
151
- timer = undefined;
152
- }
153
- },
154
- };
155
- }
package/src/format.ts CHANGED
@@ -250,6 +250,11 @@ export function makeSummary(run: RunSnapshot): string {
250
250
  return truncateText(lines.join("\n"));
251
251
  }
252
252
  /** Per-task notice: one task's outcome, small. Full output stays out of parent context. */
253
+ /** Dead on arrival: failed without ever producing assistant text — model/plan/auth/agent-file
254
+ * level, so every respawn with the same config fails identically. */
255
+ export function isStartupFailure(task: TaskSnapshot, kind: string): boolean {
256
+ return kind === "failed" && !task.finalText?.trim();
257
+ }
253
258
  export function makeTaskNotice(run: RunSnapshot, task: TaskSnapshot, kind: string): string {
254
259
  const goal = truncateText(task.task, 120);
255
260
  const detail = task.error ? task.error : truncateText(task.finalText || "(no output)", 200);
@@ -263,7 +268,9 @@ export function makeTaskNotice(run: RunSnapshot, task: TaskSnapshot, kind: strin
263
268
  return [
264
269
  `Task ${task.agent} (${task.id}) ${kind} in run ${run.id}: ${detail}${wt}`,
265
270
  `Goal: ${goal}${src}`,
266
- `Use subagent_result(runId: "${run.id}", taskId: "${task.id}") for full output.`,
271
+ isStartupFailure(task, kind)
272
+ ? "Never started — stop and diagnose before spawning anything else: a config-level error (model, plan, auth, agent file) fails identically on every respawn."
273
+ : `Use subagent_result(runId: "${run.id}", taskId: "${task.id}") for full output.`,
267
274
  ].join("\n");
268
275
  }
269
276
  /** Notification: 3 lines max. Full output stays out of parent context. */
package/src/index.ts CHANGED
@@ -157,7 +157,7 @@ export default function (pi: ExtensionAPI) {
157
157
  // ponytail: this string is billed on every request. No example block — an example
158
158
  // biases the model toward one shape; guidelines + JSON schema describe all of them.
159
159
  description:
160
- "Run isolated subagents (own context, own session). You invent each agent: name, optional system prompt, toolset (read-only default, write:true to edit). Use `agent`+`task` for one, `tasks` for many. `needs` declares dependency edges: a task waits for its needs and receives their outputs prepended to its prompt. If a user agent file in `.agents/agents`, `.claude/agents`, or `.pi/agents` (project dirs, then home) has a `description` matching the spawn goal (name + task), that file is authoritative: body = system prompt, frontmatter `model`/`tools` apply, inline prompt/model/tools ignored. No match → the inline definition stands. Write agents run in an isolated git worktree: on completion the result reports the branch + changed files — review, then merge with `git merge --no-ff <branch>` (merged branches are cleaned automatically). Every run is background: the call returns a runId immediately and completion notifies you. Set autoAwait:true when you need the result before your next step — the call parks until the run finishes and returns runId + final result in one response. allowIntercom:true lets children talk to you and each other.",
160
+ "Run isolated subagents (own context, own session). You invent each agent: name, optional system prompt, toolset (read-only default, write:true to edit). Use `agent`+`task` for one, `tasks` for many. `needs` declares dependency edges: a task waits for its needs and receives their outputs prepended to its prompt. If a user agent file in `.agents/agents`, `.claude/agents`, or `.pi/agents` (project dirs, then home) has a `description` matching the spawn goal (name + task), that file is authoritative: body = system prompt, frontmatter `model`/`tools` apply, inline prompt/model/tools ignored. No match → the inline definition stands. Write agents run in an isolated git worktree: on completion the result reports the branch + changed files — review, then merge with `git merge --no-ff <branch>` (merged branches are cleaned automatically). Every run is background: the call returns a runId immediately and completion notifies you — do NOT park waiting on it. If you have no other work, end your turn; the completion notice wakes you with the results. Set autoAwait:true only when the very next step in the SAME turn consumes the result. allowIntercom:true lets children talk to you and each other.",
161
161
  promptSnippet: "Define and delegate work to specialized subagents.",
162
162
  promptGuidelines: [
163
163
  "Use subagent when independent review, testing, research, or parallel analysis improves quality.",
@@ -167,8 +167,11 @@ export default function (pi: ExtensionAPI) {
167
167
  "End each task with a runnable check, e.g. 'Verify: npx tsc --noEmit && bun test'. A subagent's claim of success is not evidence.",
168
168
  "For write agents (write:true) in a git repo, the child works in an isolated worktree and its changes are committed to a branch — the result reports branch + changed files. Review the diff, then merge with `git merge --no-ff <branch>`; merged branches are cleaned up automatically. Never leave a worktree branch unmerged at the end of the task.",
169
169
  "Define each agent yourself: invented name, focused system prompt, and read-only (default) or write:true. Prefer read-only. A user agent file (`.agents/agents`, `.claude/agents`, `.pi/agents` — project first, then home) whose `description` matches the spawn goal (name + task) takes over: its body is the system prompt, frontmatter `model`/`tools` apply and are validated against the model registry. Matching is by description, not name — name the agent whatever fits the goal.",
170
- "When you need a run's result before your next step, spawn with autoAwait:true — the call returns runId + final result in one response. Otherwise spawn background and settle results (await_subagent / subagent_result) before continuing dependent work.",
171
- "For long multi-task runs, don't autoAwait the whole run: spawn background, then loop await_subagent with short timeoutMs slices (e.g. 20s), processing whichever tasks completed in each slice while the rest keep running. You get incremental results instead of one big wait.",
170
+ "Right after a background spawn, call subagent_status(runId) ONCE before any other work — confirm each task is running (or already progressing), not stuck queued or failed at startup. A child that dies on spawn otherwise stays invisible until far later.",
171
+ "If that first status shows a task failed or never started, fix or respawn immediately; do not move on assuming it runs.",
172
+ "Never block with nothing to do: if you have no work left after spawning, end your turn. Task completion notifies you and wakes a fresh turn with the results — await_subagent/autoAwait in that situation only burns time and tokens.",
173
+ "autoAwait:true only when the same turn must consume the result immediately (e.g. you spawn a reviewer and then must act on its verdict before replying). Otherwise spawn background and read results from the completion notice, or subagent_result when you come back.",
174
+ "await_subagent is for the rare case where you have parallel work of your own and need to sync at a specific point — not the default follow-up to a spawn.",
172
175
  "allowIntercom:true only when a child may need to ask you something.",
173
176
  ],
174
177
  parameters: SubagentParams,
@@ -218,7 +221,7 @@ export default function (pi: ExtensionAPI) {
218
221
  content: [
219
222
  {
220
223
  type: "text",
221
- text: `Background run started: ${details.run.id} (${details.run.mode}, ${details.run.tasks.length} task${details.run.tasks.length > 1 ? "s" : ""}).\nUse subagent_status / subagent_result / await_subagent / reply_subagent / subagent_cancel to interact.`,
224
+ text: `Background run started: ${details.run.id} (${details.run.mode}, ${details.run.tasks.length} task${details.run.tasks.length > 1 ? "s" : ""}).\nNext: call subagent_status("${details.run.id}") now to confirm the tasks actually started before doing anything else.\nAfter that, completion will notify you — if you have no other work, end your turn instead of waiting.\nOther tools: subagent_result / reply_subagent / steer_subagent / subagent_cancel.`,
222
225
  },
223
226
  ],
224
227
  details,
@@ -300,8 +303,11 @@ export default function (pi: ExtensionAPI) {
300
303
  name: "subagent_status",
301
304
  label: "Subagent Status",
302
305
  description:
303
- "Live status of a subagent run (non-blocking): per-task state, plus each child's session file path (JSONL) so you can tail it from outside — e.g. in a terminal multiplexer pane.",
304
- promptSnippet: "Check progress of a subagent run.",
306
+ "Live status of a subagent run (non-blocking): per-task state, plus each child's session file path (JSONL) so you can tail it from outside — e.g. in a terminal multiplexer pane. Call this once right after spawning to verify the children actually started.",
307
+ promptSnippet: "Check progress of a subagent run; use right after spawn as a health check.",
308
+ promptGuidelines: [
309
+ "Health-check every background spawn with one subagent_status(runId) before continuing — catch dead-on-arrival children early instead of at completion time.",
310
+ ],
305
311
  parameters: RunIdParam,
306
312
  async execute(_id, params) {
307
313
  const { runId } = params as { runId: string };
@@ -348,7 +354,10 @@ export default function (pi: ExtensionAPI) {
348
354
  name: "await_subagent",
349
355
  label: "Await Subagent",
350
356
  description:
351
- "Block until a run finishes (or timeoutMs elapses). While parked, child→leader messages (asks, notifies, completions) wake the wait and arrive INSIDE the result — the await doubles as the run's intercom drain, no steering queue involved.",
357
+ "Block until a run finishes (or timeoutMs elapses). Use ONLY when you have work of your own to sync with; if you have nothing else to do, end your turn instead — completion notifies you and wakes a new turn with the results. While parked, child→leader messages (asks, notifies, completions) wake the wait and arrive INSIDE the result.",
358
+ promptGuidelines: [
359
+ "Do not call await_subagent right after spawning with no other work pending — end the turn and let the completion notice wake you.",
360
+ ],
352
361
  parameters: AwaitParam,
353
362
  async execute(_id, params) {
354
363
  const { runId, timeoutMs } = params as { runId: string; timeoutMs?: number };
package/src/manager.ts CHANGED
@@ -17,7 +17,7 @@ import {
17
17
  } from "@earendil-works/pi-coding-agent";
18
18
  import type { TUI } from "@earendil-works/pi-tui";
19
19
  import { resolveAgentFile } from "./agentfile.ts";
20
- import { CHILD_TALK_TOOLS, type ChildHandlers, createChildTools, createWatchdog, type Watchdog } from "./child.ts";
20
+ import { CHILD_TALK_TOOLS, type ChildHandlers, createChildTools } from "./child.ts";
21
21
  import {
22
22
  activitySnippet,
23
23
  describeCall,
@@ -25,6 +25,7 @@ import {
25
25
  isTalking,
26
26
  makeNotice,
27
27
  makeTaskNotice,
28
+ isStartupFailure,
28
29
  SubagentsWidget,
29
30
  truncateText,
30
31
  } from "./format.ts";
@@ -56,19 +57,20 @@ import {
56
57
 
57
58
  export const DEFAULT_CONCURRENCY = 3;
58
59
  export const MAX_CONCURRENCY = 8;
59
- /** No default wall-clock cap: a subagent runs until its task is done, it stalls, or the user aborts. */
60
- /** Hard wall-clock ceiling per child. The stall watchdog is touched by every
61
- * event, so a child stuck in a retry/compaction livelock emits forever and is
62
- * never "stalled" — only a cap that events CANNOT reset bounds that. */
60
+ /**
61
+ * Hard wall-clock ceiling per child — the ONLY liveness bound.
62
+ *
63
+ * There used to be a second one, an event-heartbeat "stall" watchdog. It was
64
+ * deleted: it was armed before the child session even existed, so the only
65
+ * window it could fire in was a slow startup (where firing is always wrong),
66
+ * and once events flowed it could never fire at all. Every observed firing
67
+ * across three releases was a false kill. A wedged child that emits events was
68
+ * always bounded by this cap alone; nothing else changed by removing it.
69
+ */
63
70
  const DEFAULT_RUNTIME_MS = 3_600_000; // 1 h
64
71
  /** "Unlimited" still has a ceiling — an unbounded child pins hasActiveRun() and
65
72
  * its concurrency slot for the life of the session. */
66
73
  const UNLIMITED_RUNTIME_MS = 21_600_000; // 6 h
67
- /** Last-resort hang detector, not a latency budget. A healthy child can be
68
- * silent for minutes (big-context upload, non-streamed reasoning, provider
69
- * retry backoff), so this is deliberately far above any normal quiet window —
70
- * killing a working child is much worse than waiting out a dead one. */
71
- const DEFAULT_STALL_MS = 900_000; // 15 min
72
74
  /** Cap on a child's wait for reply_subagent — an ignored question must not pin the run open forever. */
73
75
  const PARENT_REPLY_TIMEOUT_MS = 600_000; // 10 min
74
76
  /** Intercom messages buffered per park before the followUp path takes over. */
@@ -244,7 +246,7 @@ export class SubagentManager {
244
246
  private pendingReplies = new Map<string, PendingReply>();
245
247
  private liveChildren = new Map<
246
248
  string,
247
- { abort: () => void; dispose: () => void; touchWatchdog: () => void; steer: (message: string) => void }
249
+ { abort: () => void; dispose: () => void; steer: (message: string) => void }
248
250
  >();
249
251
  private mailboxes: Mailbox = createMailbox();
250
252
  /** Live worktrees by `${runId}:${taskId}` — lets cancel drop dirs and keeps
@@ -461,7 +463,16 @@ export class SubagentManager {
461
463
  this.pi.events.emit(type, { type, timestamp: Date.now(), ...payload });
462
464
  }
463
465
 
464
- /** Per-task wake-up: queued follow-up so the parent can interleave responses. */
466
+ /** Only startup failures are forced: they're dead-on-arrival and the leader must
467
+ * notice before it respawns the same broken config. Completed/aborted and even
468
+ * mid-run failures wait for the turn's end (followUp) — they're not urgent and
469
+ * steering every one of them would interrupt the leader mid-tool-call.
470
+ * force = deliver immediately even while streaming (steer). */
471
+ private deliverMode(kind: string, task: TaskSnapshot): "steer" | "followUp" {
472
+ return isStartupFailure(task, kind) ? "steer" : "followUp";
473
+ }
474
+
475
+ /** Per-task wake-up: failures steer in immediately, the rest queue as follow-up. */
465
476
  private notifyTask(run: RunSnapshot, task: TaskSnapshot, kind: "completed" | "failed" | "aborted"): void {
466
477
  const body = makeTaskNotice(run, task, kind);
467
478
  // Parked leader (await_subagent) receives completions through the wait — no queue.
@@ -470,7 +481,7 @@ export class SubagentManager {
470
481
  return;
471
482
  }
472
483
  try {
473
- this.pi.sendUserMessage(body, { deliverAs: "followUp" });
484
+ this.pi.sendUserMessage(body, { deliverAs: this.deliverMode(kind, task) });
474
485
  } catch {
475
486
  /* parent mid-stream; consumers can poll subagent_status */
476
487
  }
@@ -478,8 +489,8 @@ export class SubagentManager {
478
489
  }
479
490
 
480
491
  /** Wake the parent with a 3-line notice. Full text stays out of context.
481
- * deliverAs followUp queues the message if the parent is mid-stream
482
- * (e.g. inside await_subagent) instead of throwing/aborting. */
492
+ * deliverAs queues the message if the parent is mid-stream (e.g. inside
493
+ * await_subagent) instead of throwing/aborting. */
483
494
  private notifyParent(
484
495
  run: RunSnapshot,
485
496
  kind: "completed" | "failed" | "aborted" | "asked",
@@ -598,7 +609,6 @@ export class SubagentManager {
598
609
  private makeChildHandlers(run: RunSnapshot, task: TaskSnapshot, ctx: ExtensionContext): ChildHandlers {
599
610
  return {
600
611
  onAskParent: async (_taskId, question) => {
601
- const key = `${run.id}:${task.id}`;
602
612
  // A tool call already in flight can reach here AFTER the task ended
603
613
  // (abort/timeout/cancel). Reviving it would leave a "running" task in a
604
614
  // finished run — hasActiveRun() then never clears.
@@ -606,7 +616,6 @@ export class SubagentManager {
606
616
  return "(your task has already ended — stop work and return immediately)";
607
617
  }
608
618
  this.updateTask(run, task, { status: "awaiting_parent" }, ctx);
609
- this.liveChildren.get(key)?.touchWatchdog();
610
619
  // While the leader is parked in await_subagent the question rides the wait
611
620
  // (no steering queue, no turn boundary); otherwise it goes out as a notice.
612
621
  // Either way the pending reply entry must exist, or reply_subagent has
@@ -614,24 +623,17 @@ export class SubagentManager {
614
623
  if (!this.collectParked(run.id, { kind: "ask", taskId: task.id, agent: task.agent, text: question })) {
615
624
  this.notifyParent(run, "asked", { taskId: task.id, question });
616
625
  }
617
- // A waiting child is not stalled — keep the watchdog fed until the reply.
618
- // But the wait is BOUNDED: an unanswered question would otherwise keep the
619
- // run non-terminal forever (widget never clears, run never settles).
620
- const keepAlive = setInterval(() => this.liveChildren.get(key)?.touchWatchdog(), 30_000);
621
- try {
622
- const reply = await this.awaitParentReply(run.id, task.id, PARENT_REPLY_TIMEOUT_MS);
623
- // Cancel wins over a reply that arrived in the same tick: never move a
624
- // terminal task back to "running" (that would let a canceled task be
625
- // reported as completed).
626
- if (TERMINAL.includes(task.status)) {
627
- return "(your task was canceled while you waited — stop work and return immediately)";
628
- }
629
- this.updateTask(run, task, { status: "running" }, ctx);
630
- this.liveChildren.get(key)?.touchWatchdog();
631
- return reply;
632
- } finally {
633
- clearInterval(keepAlive);
626
+ // The wait is BOUNDED: an unanswered question would otherwise keep the run
627
+ // non-terminal forever (widget never clears, run never settles).
628
+ const reply = await this.awaitParentReply(run.id, task.id, PARENT_REPLY_TIMEOUT_MS);
629
+ // Cancel wins over a reply that arrived in the same tick: never move a
630
+ // terminal task back to "running" (that would let a canceled task be
631
+ // reported as completed).
632
+ if (TERMINAL.includes(task.status)) {
633
+ return "(your task was canceled while you waited — stop work and return immediately)";
634
634
  }
635
+ this.updateTask(run, task, { status: "running" }, ctx);
636
+ return reply;
635
637
  },
636
638
  onNotifyParent: (_taskId, message, level) => {
637
639
  this.emit("subagent:intercom", { runId: run.id, taskId: task.id, kind: "notify", level, message });
@@ -709,14 +711,8 @@ export class SubagentManager {
709
711
  task: TaskSnapshot,
710
712
  ctx: ExtensionContext,
711
713
  onUpdate: ((partial: any) => void) | undefined,
712
- watchdog: Watchdog,
713
714
  state: ChildEventState,
714
715
  ): void {
715
- // ANY event is proof of life. The old allowlist ignored message_start,
716
- // turn_start/end, compaction and auto-retry, so a child was killed during
717
- // silent-but-healthy windows — context upload, a provider that doesn't
718
- // stream reasoning, retry backoff — and surfaced as "Error: terminated".
719
- watchdog.touch();
720
716
  const active =
721
717
  event.type === "message_update" ||
722
718
  event.type === "message_end" ||
@@ -922,7 +918,6 @@ export class SubagentManager {
922
918
  let unsubscribe: (() => void) | undefined;
923
919
  let timeout: ReturnType<typeof setTimeout> | undefined;
924
920
  let abortListener: (() => void) | undefined;
925
- const watchdog = createWatchdog(DEFAULT_STALL_MS, `Subagent ${task.agent}`);
926
921
  const childState: ChildEventState = {};
927
922
 
928
923
  const key = `${run.id}:${task.id}`;
@@ -981,7 +976,7 @@ export class SubagentManager {
981
976
  });
982
977
 
983
978
  unsubscribe = child.subscribe((event: AgentSessionEvent) =>
984
- this.onChildEvent(event, run, task, ctx, onUpdate, watchdog, childState),
979
+ this.onChildEvent(event, run, task, ctx, onUpdate, childState),
985
980
  );
986
981
 
987
982
  const abortChild = () => {
@@ -1002,8 +997,7 @@ export class SubagentManager {
1002
997
  }
1003
998
  this.liveChildren.set(key, {
1004
999
  abort: () => void child?.abort(),
1005
- dispose: () => watchdog.dispose(),
1006
- touchWatchdog: () => watchdog.touch(),
1000
+ dispose: () => child?.dispose(),
1007
1001
  // Inject a steering message mid-run; queues as steer if the child is streaming.
1008
1002
  steer: (message) =>
1009
1003
  void child?.prompt(message, { streamingBehavior: "steer" }).catch((err) =>
@@ -1014,12 +1008,11 @@ export class SubagentManager {
1014
1008
  });
1015
1009
 
1016
1010
  // The ceiling is the ONLY bound on a child that emits events forever (retry
1017
- // or tool-call livelock): the stall watchdog is touched by every event and
1018
- // cannot fire for one. So auto-limit off RAISES it, never removes it —
1011
+ // or tool-call livelock). So auto-limit off RAISES it, never removes it —
1019
1012
  // removing it reproduced the immortal-child hang.
1020
1013
  const maxRuntimeMs = input.maxRuntimeMs ?? (this.autoLimit ? DEFAULT_RUNTIME_MS : UNLIMITED_RUNTIME_MS);
1021
1014
  const promptPromise = child.prompt(task.task, { source: "extension" });
1022
- const races: Promise<unknown>[] = [promptPromise, childFailurePromise, childEndPromise, watchdog.promise];
1015
+ const races: Promise<unknown>[] = [promptPromise, childFailurePromise, childEndPromise];
1023
1016
  if (maxRuntimeMs > 0) {
1024
1017
  races.push(
1025
1018
  new Promise<never>((_, reject) => {
@@ -1140,7 +1133,6 @@ export class SubagentManager {
1140
1133
  this.pendingReplies.delete(key);
1141
1134
  abortListener?.();
1142
1135
  unsubscribe?.();
1143
- watchdog.dispose();
1144
1136
  if (timeout) clearTimeout(timeout);
1145
1137
  child?.dispose();
1146
1138
  // Failed/aborted: let the aborted child's last writes land (its tools may