@arhen/pi-core-subagent 1.3.41 → 1.3.43

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@arhen/pi-core-subagent",
3
- "version": "1.3.41",
3
+ "version": "1.3.43",
4
4
  "type": "module",
5
5
  "description": "pi extension: fast in-process subagents with a dependency-graph scheduler (needs edges gate tasks and carry upstream output into dependent prompts), plus background runs, intercom and agent-to-agent mailbox. Leader defines agents inline.",
6
6
  "license": "MIT",
package/src/format.ts CHANGED
@@ -242,14 +242,20 @@ export function makeSummary(run: RunSnapshot): string {
242
242
  // Edges are named so the leader can compare what it delegated against what came back.
243
243
  const edge = task.needs?.length ? ` (${task.id}, needs ${task.needs.join(", ")})` : ` (${task.id})`;
244
244
  const fileNote = task.agentFile ? ` [${task.agentFile}]` : "";
245
+ const swap = task.modelNote ? `\nModel: ${task.modelNote}` : "";
245
246
  lines.push(
246
- `\n## ${task.agent}${edge}${fileNote} ${statusIcon(task.status)}${task.error ? `\nError: ${task.error}` : `\n${truncateText(task.finalText || "(no output)")}`}${worktreeLine(task, run.tasks)}`,
247
+ `\n## ${task.agent}${edge}${fileNote} ${statusIcon(task.status)}${swap}${task.error ? `\nError: ${task.error}` : `\n${truncateText(task.finalText || "(no output)")}`}${worktreeLine(task, run.tasks)}`,
247
248
  );
248
249
  }
249
250
  // Ceiling on the WHOLE summary — 16 tasks × 24KB would otherwise flood the parent context.
250
251
  return truncateText(lines.join("\n"));
251
252
  }
252
253
  /** Per-task notice: one task's outcome, small. Full output stays out of parent context. */
254
+ /** Dead on arrival: failed without ever producing assistant text — model/plan/auth/agent-file
255
+ * level, so every respawn with the same config fails identically. */
256
+ export function isStartupFailure(task: TaskSnapshot, kind: string): boolean {
257
+ return kind === "failed" && !task.finalText?.trim();
258
+ }
253
259
  export function makeTaskNotice(run: RunSnapshot, task: TaskSnapshot, kind: string): string {
254
260
  const goal = truncateText(task.task, 120);
255
261
  const detail = task.error ? task.error : truncateText(task.finalText || "(no output)", 200);
@@ -260,10 +266,13 @@ export function makeTaskNotice(run: RunSnapshot, task: TaskSnapshot, kind: strin
260
266
  // 403, wrong persona) that override is the likeliest cause, so it has to be
261
267
  // in the notice, not only in the run summary the leader may never read.
262
268
  const src = task.agentFile ? `\nAgent file: ${task.agentFile}${task.model ? ` (model ${task.model})` : ""}` : "";
269
+ const swap = task.modelNote ? `\nModel: ${task.modelNote}` : "";
263
270
  return [
264
271
  `Task ${task.agent} (${task.id}) ${kind} in run ${run.id}: ${detail}${wt}`,
265
- `Goal: ${goal}${src}`,
266
- `Use subagent_result(runId: "${run.id}", taskId: "${task.id}") for full output.`,
272
+ `Goal: ${goal}${src}${swap}`,
273
+ isStartupFailure(task, kind)
274
+ ? "Never started — stop and diagnose before spawning anything else: a config-level error (model, plan, auth, agent file) fails identically on every respawn."
275
+ : `Use subagent_result(runId: "${run.id}", taskId: "${task.id}") for full output.`,
267
276
  ].join("\n");
268
277
  }
269
278
  /** Notification: 3 lines max. Full output stays out of parent context. */
package/src/index.ts CHANGED
@@ -157,7 +157,7 @@ export default function (pi: ExtensionAPI) {
157
157
  // ponytail: this string is billed on every request. No example block — an example
158
158
  // biases the model toward one shape; guidelines + JSON schema describe all of them.
159
159
  description:
160
- "Run isolated subagents (own context, own session). You invent each agent: name, optional system prompt, toolset (read-only default, write:true to edit). Use `agent`+`task` for one, `tasks` for many. `needs` declares dependency edges: a task waits for its needs and receives their outputs prepended to its prompt. If a user agent file in `.agents/agents`, `.claude/agents`, or `.pi/agents` (project dirs, then home) has a `description` matching the spawn goal (name + task), that file is authoritative: body = system prompt, frontmatter `model`/`tools` apply, inline prompt/model/tools ignored. No match → the inline definition stands. Write agents run in an isolated git worktree: on completion the result reports the branch + changed files — review, then merge with `git merge --no-ff <branch>` (merged branches are cleaned automatically). Every run is background: the call returns a runId immediately and completion notifies you. Set autoAwait:true when you need the result before your next step — the call parks until the run finishes and returns runId + final result in one response. allowIntercom:true lets children talk to you and each other.",
160
+ "Run isolated subagents (own context, own session). You invent each agent: name, optional system prompt, toolset (read-only default, write:true to edit). Use `agent`+`task` for one, `tasks` for many. `needs` declares dependency edges: a task waits for its needs and receives their outputs prepended to its prompt. If a user agent file in `.agents/agents`, `.claude/agents`, or `.pi/agents` (project dirs, then home) has a `description` matching the spawn goal (name + task), that file is authoritative: body = system prompt, frontmatter `model`/`tools` apply, inline prompt/model/tools ignored. No match → the inline definition stands. Write agents run in an isolated git worktree: on completion the result reports the branch + changed files — review, then merge with `git merge --no-ff <branch>` (merged branches are cleaned automatically). Every run is background: the call returns a runId immediately and completion notifies you — do NOT park waiting on it. If you have no other work, end your turn; the completion notice wakes you with the results. Set autoAwait:true only when the very next step in the SAME turn consumes the result. allowIntercom:true lets children talk to you and each other.",
161
161
  promptSnippet: "Define and delegate work to specialized subagents.",
162
162
  promptGuidelines: [
163
163
  "Use subagent when independent review, testing, research, or parallel analysis improves quality.",
@@ -167,8 +167,11 @@ export default function (pi: ExtensionAPI) {
167
167
  "End each task with a runnable check, e.g. 'Verify: npx tsc --noEmit && bun test'. A subagent's claim of success is not evidence.",
168
168
  "For write agents (write:true) in a git repo, the child works in an isolated worktree and its changes are committed to a branch — the result reports branch + changed files. Review the diff, then merge with `git merge --no-ff <branch>`; merged branches are cleaned up automatically. Never leave a worktree branch unmerged at the end of the task.",
169
169
  "Define each agent yourself: invented name, focused system prompt, and read-only (default) or write:true. Prefer read-only. A user agent file (`.agents/agents`, `.claude/agents`, `.pi/agents` — project first, then home) whose `description` matches the spawn goal (name + task) takes over: its body is the system prompt, frontmatter `model`/`tools` apply and are validated against the model registry. Matching is by description, not name — name the agent whatever fits the goal.",
170
- "When you need a run's result before your next step, spawn with autoAwait:true — the call returns runId + final result in one response. Otherwise spawn background and settle results (await_subagent / subagent_result) before continuing dependent work.",
171
- "For long multi-task runs, don't autoAwait the whole run: spawn background, then loop await_subagent with short timeoutMs slices (e.g. 20s), processing whichever tasks completed in each slice while the rest keep running. You get incremental results instead of one big wait.",
170
+ "Right after a background spawn, call subagent_status(runId) ONCE before any other work — confirm each task is running (or already progressing), not stuck queued or failed at startup. A child that dies on spawn otherwise stays invisible until far later.",
171
+ "If that first status shows a task failed or never started, fix or respawn immediately; do not move on assuming it runs.",
172
+ "Never block with nothing to do: if you have no work left after spawning, end your turn. Task completion notifies you and wakes a fresh turn with the results — await_subagent/autoAwait in that situation only burns time and tokens.",
173
+ "autoAwait:true only when the same turn must consume the result immediately (e.g. you spawn a reviewer and then must act on its verdict before replying). Otherwise spawn background and read results from the completion notice, or subagent_result when you come back.",
174
+ "await_subagent is for the rare case where you have parallel work of your own and need to sync at a specific point — not the default follow-up to a spawn.",
172
175
  "allowIntercom:true only when a child may need to ask you something.",
173
176
  ],
174
177
  parameters: SubagentParams,
@@ -218,7 +221,7 @@ export default function (pi: ExtensionAPI) {
218
221
  content: [
219
222
  {
220
223
  type: "text",
221
- text: `Background run started: ${details.run.id} (${details.run.mode}, ${details.run.tasks.length} task${details.run.tasks.length > 1 ? "s" : ""}).\nUse subagent_status / subagent_result / await_subagent / reply_subagent / subagent_cancel to interact.`,
224
+ text: `Background run started: ${details.run.id} (${details.run.mode}, ${details.run.tasks.length} task${details.run.tasks.length > 1 ? "s" : ""}).\nNext: call subagent_status("${details.run.id}") now to confirm the tasks actually started before doing anything else.\nAfter that, completion will notify you — if you have no other work, end your turn instead of waiting.\nOther tools: subagent_result / reply_subagent / steer_subagent / subagent_cancel.`,
222
225
  },
223
226
  ],
224
227
  details,
@@ -300,8 +303,11 @@ export default function (pi: ExtensionAPI) {
300
303
  name: "subagent_status",
301
304
  label: "Subagent Status",
302
305
  description:
303
- "Live status of a subagent run (non-blocking): per-task state, plus each child's session file path (JSONL) so you can tail it from outside — e.g. in a terminal multiplexer pane.",
304
- promptSnippet: "Check progress of a subagent run.",
306
+ "Live status of a subagent run (non-blocking): per-task state, plus each child's session file path (JSONL) so you can tail it from outside — e.g. in a terminal multiplexer pane. Call this once right after spawning to verify the children actually started.",
307
+ promptSnippet: "Check progress of a subagent run; use right after spawn as a health check.",
308
+ promptGuidelines: [
309
+ "Health-check every background spawn with one subagent_status(runId) before continuing — catch dead-on-arrival children early instead of at completion time.",
310
+ ],
305
311
  parameters: RunIdParam,
306
312
  async execute(_id, params) {
307
313
  const { runId } = params as { runId: string };
@@ -348,7 +354,10 @@ export default function (pi: ExtensionAPI) {
348
354
  name: "await_subagent",
349
355
  label: "Await Subagent",
350
356
  description:
351
- "Block until a run finishes (or timeoutMs elapses). While parked, child→leader messages (asks, notifies, completions) wake the wait and arrive INSIDE the result — the await doubles as the run's intercom drain, no steering queue involved.",
357
+ "Block until a run finishes (or timeoutMs elapses). Use ONLY when you have work of your own to sync with; if you have nothing else to do, end your turn instead — completion notifies you and wakes a new turn with the results. While parked, child→leader messages (asks, notifies, completions) wake the wait and arrive INSIDE the result.",
358
+ promptGuidelines: [
359
+ "Do not call await_subagent right after spawning with no other work pending — end the turn and let the completion notice wake you.",
360
+ ],
352
361
  parameters: AwaitParam,
353
362
  async execute(_id, params) {
354
363
  const { runId, timeoutMs } = params as { runId: string; timeoutMs?: number };
package/src/manager.ts CHANGED
@@ -22,6 +22,7 @@ import {
22
22
  activitySnippet,
23
23
  describeCall,
24
24
  getFirstText,
25
+ isStartupFailure,
25
26
  isTalking,
26
27
  makeNotice,
27
28
  makeTaskNotice,
@@ -159,14 +160,26 @@ export function cloneRun(run: RunSnapshot): RunSnapshot {
159
160
  return JSON.parse(JSON.stringify(run)) as RunSnapshot;
160
161
  }
161
162
  /** Resolve a child model from the pi model registry.
162
- * Order: explicit "provider/model-id" or bare id (searched across available
163
- * models) → agent file model → parent's current model (ctx.model) → undefined
164
- * (createAgentSession falls back to settings). */
163
+ * Order: explicit "provider/model-id" or bare id → agent file model → parent's
164
+ * current model (ctx.model) → undefined (createAgentSession falls back to settings).
165
+ *
166
+ * A BARE id is ambiguous: `claude-sonnet-5` exists on anthropic, commandcode,
167
+ * github-copilot and openrouter at once, and an agent file's `model:` is written
168
+ * without a provider. Taking the registry's first match routed children to a
169
+ * provider the session never chose (403 MODEL_NOT_IN_PLAN on every spawn), so the
170
+ * ACTIVE SESSION's provider is searched first — exact id, then its own prefixed
171
+ * form (9router carries `cc/claude-opus-5`, not `claude-opus-5`). */
165
172
  export function resolveChildModel(ctx: ExtensionContext, explicit: string | undefined) {
166
173
  if (!explicit?.trim()) return ctx.model; // inherit the parent's active model
167
174
  const ref = explicit.trim();
168
175
  if (!ctx.modelRegistry) return ctx.model; // no registry to check against (tests, headless)
169
176
  const available = ctx.modelRegistry.getAvailable();
177
+ const sessionProvider = ctx.model?.provider;
178
+ if (sessionProvider && !ref.includes("/")) {
179
+ const own = available.filter((m) => m.provider === sessionProvider);
180
+ const hit = own.find((m) => m.id === ref) ?? own.find((m) => m.id.endsWith(`/${ref}`));
181
+ if (hit) return hit;
182
+ }
170
183
  // Model ids can contain slashes (e.g. 9router/cc/claude-opus-5), so a bare id
171
184
  // match and every provider/id split point must be tried, not just the first.
172
185
  const byId = available.find((m) => m.id === ref);
@@ -178,6 +191,48 @@ export function resolveChildModel(ctx: ExtensionContext, explicit: string | unde
178
191
  throw new Error(`Model not found: ${ref}`);
179
192
  }
180
193
 
194
+ /** One throwaway request against the resolved model. A child that cannot reach its
195
+ * model dies on its FIRST turn with no output, after a worktree and a session have
196
+ * already been built — and an agent file's `model:` is chosen by a file the leader
197
+ * never wrote, so "it resolved" is not evidence it is usable (plan gates, ZDR,
198
+ * dead keys all pass resolution). Returns the error text, or undefined when OK. */
199
+ async function probeModel(
200
+ ctx: ExtensionContext,
201
+ model: Model<Api>,
202
+ signal: AbortSignal | undefined,
203
+ ): Promise<string | undefined> {
204
+ try {
205
+ const reply = await ctx.modelRegistry.complete(
206
+ model,
207
+ { messages: [{ role: "user", content: "ping", timestamp: Date.now() }] },
208
+ { maxTokens: 16, signal },
209
+ );
210
+ return reply.stopReason === "error" ? (reply.errorMessage ?? "provider returned an error") : undefined;
211
+ } catch (err) {
212
+ return err instanceof Error ? err.message : String(err);
213
+ }
214
+ }
215
+
216
+ /** Preflight for the model a child is about to run on. Probes only what is not
217
+ * already proven: the session's own model answered this very turn. On failure the
218
+ * session model is the fallback — the one model known to work right now. */
219
+ export async function ensureUsableModel(
220
+ ctx: ExtensionContext,
221
+ model: Model<Api> | undefined,
222
+ signal: AbortSignal | undefined,
223
+ ): Promise<{ model: Model<Api> | undefined; note?: string }> {
224
+ const session = ctx.model;
225
+ if (!model || !ctx.modelRegistry) return { model };
226
+ if (session && model.provider === session.provider && model.id === session.id) return { model };
227
+ const error = await probeModel(ctx, model, signal);
228
+ if (!error) return { model };
229
+ if (!session) throw new Error(`Model ${model.provider}/${model.id} is unusable: ${error}`);
230
+ return {
231
+ model: session,
232
+ note: `${model.provider}/${model.id} failed preflight (${error}); using session model ${session.provider}/${session.id}`,
233
+ };
234
+ }
235
+
181
236
  /** Extension-registered providers (e.g. 9router) live only in the parent's
182
237
  * in-memory runtime. A child builds its runtime from disk and would lose them,
183
238
  * so replay the parent's registrations before the child resolves auth. */
@@ -462,7 +517,16 @@ export class SubagentManager {
462
517
  this.pi.events.emit(type, { type, timestamp: Date.now(), ...payload });
463
518
  }
464
519
 
465
- /** Per-task wake-up: queued follow-up so the parent can interleave responses. */
520
+ /** Only startup failures are forced: they're dead-on-arrival and the leader must
521
+ * notice before it respawns the same broken config. Completed/aborted and even
522
+ * mid-run failures wait for the turn's end (followUp) — they're not urgent and
523
+ * steering every one of them would interrupt the leader mid-tool-call.
524
+ * force = deliver immediately even while streaming (steer). */
525
+ private deliverMode(kind: string, task: TaskSnapshot): "steer" | "followUp" {
526
+ return isStartupFailure(task, kind) ? "steer" : "followUp";
527
+ }
528
+
529
+ /** Per-task wake-up: failures steer in immediately, the rest queue as follow-up. */
466
530
  private notifyTask(run: RunSnapshot, task: TaskSnapshot, kind: "completed" | "failed" | "aborted"): void {
467
531
  const body = makeTaskNotice(run, task, kind);
468
532
  // Parked leader (await_subagent) receives completions through the wait — no queue.
@@ -471,7 +535,7 @@ export class SubagentManager {
471
535
  return;
472
536
  }
473
537
  try {
474
- this.pi.sendUserMessage(body, { deliverAs: "followUp" });
538
+ this.pi.sendUserMessage(body, { deliverAs: this.deliverMode(kind, task) });
475
539
  } catch {
476
540
  /* parent mid-stream; consumers can poll subagent_status */
477
541
  }
@@ -479,8 +543,8 @@ export class SubagentManager {
479
543
  }
480
544
 
481
545
  /** Wake the parent with a 3-line notice. Full text stays out of context.
482
- * deliverAs followUp queues the message if the parent is mid-stream
483
- * (e.g. inside await_subagent) instead of throwing/aborting. */
546
+ * deliverAs queues the message if the parent is mid-stream (e.g. inside
547
+ * await_subagent) instead of throwing/aborting. */
484
548
  private notifyParent(
485
549
  run: RunSnapshot,
486
550
  kind: "completed" | "failed" | "aborted" | "asked",
@@ -800,10 +864,18 @@ export class SubagentManager {
800
864
  try {
801
865
  model = resolveChildModel(ctx, file?.model ?? input.model);
802
866
  validateThinking(model, thinking);
867
+ // Resolution only proves the id exists. Probe it before anything is built,
868
+ // and fall back to the session's own model when the probe fails.
869
+ const checked = await ensureUsableModel(ctx, model, signal);
870
+ model = checked.model;
871
+ if (checked.note) {
872
+ task.modelNote = checked.note;
873
+ validateThinking(model, thinking);
874
+ }
803
875
  // Record the model ACTUALLY used — an agent file's `model:` overrides the
804
876
  // requested one, and showing the request in the widget hides that entirely
805
877
  // (a 403 then names a model the leader never asked for).
806
- if (model) this.updateTask(run, task, { model: model.id }, ctx, onUpdate);
878
+ if (model) this.updateTask(run, task, { model: model.id, modelNote: checked.note }, ctx, onUpdate);
807
879
  } catch (err) {
808
880
  this.updateTask(
809
881
  run,
package/src/types.ts CHANGED
@@ -36,6 +36,9 @@ export interface TaskSnapshot {
36
36
  finalText?: string;
37
37
  error?: string;
38
38
  model?: string;
39
+ /** Why `model` is not what was requested: preflight failed and the session's
40
+ * model took over. Silent substitution is worse than a slow spawn. */
41
+ modelNote?: string;
39
42
  thinking?: string;
40
43
  tools?: string[];
41
44
  usage: UsageStats;