pi-subagents 0.54.0 → 0.55.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. package/CHANGELOG.md +35 -1
  2. package/agents/reviewer.md +12 -2
  3. package/docs/agents.md +4 -4
  4. package/docs/configuration.md +4 -4
  5. package/docs/extension-api.md +14 -4
  6. package/docs/models.md +24 -3
  7. package/docs/observability.md +14 -3
  8. package/docs/tool-reference.md +21 -6
  9. package/docs/workflows.md +7 -1
  10. package/package.json +1 -1
  11. package/prompts/council.md +9 -0
  12. package/prompts/parallel-review.md +5 -1
  13. package/prompts/review-loop.md +9 -5
  14. package/skills/council-mode/SKILL.md +33 -10
  15. package/skills/pi-subagents/references/constraints-and-recipes.md +16 -1
  16. package/skills/pi-subagents/references/execution-controls.md +11 -4
  17. package/skills/pi-subagents/references/multi-lane-orchestration.md +3 -3
  18. package/skills/pi-subagents/references/prompting-and-roles.md +8 -8
  19. package/src/agents/agent-management.ts +30 -9
  20. package/src/agents/agent-serializer.ts +2 -2
  21. package/src/agents/agents.ts +146 -41
  22. package/src/api/external-job-provider.ts +10 -1
  23. package/src/api/preflight.ts +26 -17
  24. package/src/api/project-panes.ts +2 -0
  25. package/src/extension/index.ts +20 -1
  26. package/src/extension/public-execution.ts +12 -9
  27. package/src/extension/rpc.ts +77 -3
  28. package/src/extension/schemas.ts +2 -1
  29. package/src/extension/tool-description.ts +9 -6
  30. package/src/inspectors/herdr/focus.ts +55 -0
  31. package/src/inspectors/herdr/project-panes.ts +228 -44
  32. package/src/integrations/herdr-status.ts +26 -4
  33. package/src/runs/background/async-execution.ts +103 -9
  34. package/src/runs/background/async-job-tracker.ts +33 -14
  35. package/src/runs/background/async-resume.ts +20 -2
  36. package/src/runs/background/async-status.ts +10 -0
  37. package/src/runs/background/control-channel.ts +98 -10
  38. package/src/runs/background/notify.ts +62 -1
  39. package/src/runs/background/run-status.ts +15 -1
  40. package/src/runs/background/subagent-runner.ts +226 -37
  41. package/src/runs/foreground/async-stop-action.ts +23 -2
  42. package/src/runs/foreground/execution.ts +81 -7
  43. package/src/runs/foreground/subagent-executor.ts +348 -40
  44. package/src/runs/foreground/workflow-detach-reconcile.ts +3 -0
  45. package/src/runs/shared/acceptance.ts +1 -0
  46. package/src/runs/shared/child-identity.ts +36 -0
  47. package/src/runs/shared/completion-guard.ts +50 -1
  48. package/src/runs/shared/external-job-bridge.ts +53 -37
  49. package/src/runs/shared/external-job-runner.ts +126 -23
  50. package/src/runs/shared/model-fallback.ts +10 -0
  51. package/src/runs/shared/orca-progress-tabs.ts +71 -9
  52. package/src/runs/shared/parallel-utils.ts +12 -0
  53. package/src/runs/shared/pi-args.ts +9 -12
  54. package/src/runs/shared/subagent-prompt-runtime.ts +3 -1
  55. package/src/runs/shared/tool-availability.ts +1 -3
  56. package/src/shared/launch-contract.ts +6 -5
  57. package/src/shared/thinking-ceiling.ts +52 -0
  58. package/src/shared/types.ts +59 -2
  59. package/src/tui/fleet-status.ts +66 -13
  60. package/src/tui/render.ts +5 -4
  61. package/src/workflows/scripted-workflow.ts +107 -10
@@ -21,7 +21,13 @@ This file is a detailed reference loaded from `skills/pi-subagents/SKILL.md`.
21
21
  become second decision-makers.
22
22
  - **Respect the fixed authority policy.** `authorityPolicy` is a small `auto` / `confirm` / `forbid` map for supported operational actions. Worktree discard, destructive cleanup, and spawn-budget grants default to confirmation; stop, steer, and schedule creation remain automatic. Use `worktree.discard` with the durable `handoffPath`; confirm-required actions refuse safely without an interactive UI and retained paths include manual Git recovery commands.
23
23
 
24
- Runtime config can change orchestration behavior. `intercomBridge.resultDelivery: false` disables only external acknowledged grouped-result delivery when native parent notifications own completion; supervisor asks/progress stay active, and enabled transport failures are still reported. `asyncByDefault` and `forceTopLevelAsync` affect whether launches detach; `waitTool` can make direct `subagent_wait()` calls return immediately while headless auto-drain remains active, and its effective value is propagated to child runtimes; `globalConcurrencyLimit` bounds concurrent fanout, while a positive `maxSubagentSpawnsPerSession` optionally caps cumulative launches (`0` or unset is unlimited). Status and doctor report the budget; static work preflights declared capacity; only the settled root interactive parent can use `grant-spawn-budget` after native confirmation, with total grants bounded by the original cap. Compaction does not reset usage or grants; `singleRunOutputBaseDir` and `worktreeBaseDir` route outputs and worktrees; `completionBatch` groups async notifications. `artifactDir` is `project` (default), `session`, or `temp` and chooses where subagent artifacts are stored. Set `asyncWidget: false` to hide the above-editor background-run widget when a companion footer or dashboard owns that space (fleet inspector remains available). Per-run `artifacts: false` disables artifact capture for that launch. Async status and result artifacts include `lifecycleArtifactVersion` and fields such as `workflowGraph`, `steps`, `results`, `totalTokens`, `totalCost`, `turnCount`, `toolCount`, and nested `children`. Child protocol failures expose a structured `protocolError`; `protocol_output_limit` means a child emitted a JSONL line above the 16 MiB live-parser cap. Prefer these artifacts and `status` views over scraping terminal output.
24
+ Runtime config can change orchestration behavior. `intercomBridge.resultDelivery: false` disables only external acknowledged grouped-result delivery when native parent notifications own completion; supervisor asks/progress stay active, and enabled transport failures are still reported. `asyncByDefault` and `forceTopLevelAsync` affect whether launches detach; `waitTool` can make direct `subagent_wait()` calls return immediately while headless auto-drain remains active, and its effective value is propagated to child runtimes; `globalConcurrencyLimit` bounds concurrent fanout, while a positive `maxSubagentSpawnsPerSession` optionally caps cumulative launches (`0` or unset is unlimited). Status and doctor report the budget; static work preflights declared capacity; only the settled root interactive parent can use `grant-spawn-budget` after native confirmation, with total grants bounded by the original cap. Compaction does not reset usage or grants; `singleRunOutputBaseDir` and `worktreeBaseDir` route outputs and worktrees; `completionBatch` groups async notifications. `artifactDir` is `session` (default), `project`, or `temp` and chooses where subagent artifacts are stored. Set `asyncWidget: false` to hide the above-editor background-run widget when a companion footer or dashboard owns that space (fleet inspector remains available). Per-run `artifacts: false` disables artifact capture for that launch. Async status and result artifacts include `lifecycleArtifactVersion` and fields such as `workflowGraph`, `steps`, `results`, `totalTokens`, `totalCost`, `turnCount`, `toolCount`, and nested `children`. Child protocol failures expose a structured `protocolError`; `protocol_output_limit` means a child emitted a JSONL line above the 16 MiB live-parser cap. Prefer these artifacts and `status` views over scraping terminal output.
25
+
26
+ ### Keep report artifacts out of the repository root
27
+
28
+ Treat lane reports, review notes, council pass reports, and gate logs as scratch unless the user explicitly asks to keep them. Prefer `output: false` and the aggregate workflow result for short reports. When a later step needs a file, use the runtime-managed output artifact by setting a stable child key plus a relative `output` path such as `plans/deploy.md`; relative child outputs are saved under the run artifact directory, not the project root. Do not put `reports/...`, `*-report.json`, or similar repo-root paths in child task text.
29
+
30
+ For durable evidence, copy only the final summary to session memory, a PR body/comment, a mission artifact, or a user-approved docs path outside the repo. After the PR, issue, or gate reaches a terminal state, delete or move scratch reports from the active worktree before reporting completion. Keep a project `.gitignore` entry for ad-hoc report patterns only as a safety net; it is not the cleanup mechanism.
25
31
 
26
32
  ## Best Practices
27
33
 
@@ -62,6 +68,15 @@ user explicitly requests forked context.
62
68
  Give subagents specific tasks rather than vague mandates.
63
69
  `Review auth.ts for null-check gaps` works better than `Review everything`.
64
70
 
71
+ Before fanout, assign each child a lightweight task profile in the parent prompt:
72
+ work kind, required input, expected output, acceptance check, and context mode.
73
+ Keep the profile prose-only; do not invent runtime fields. Use coarse kinds such
74
+ as `code-write`, `code-read`, `transform`, `summarize`, and `search` only to
75
+ shape the task and choose an existing agent/model setting. If a child task is not
76
+ standalone enough for fresh context, add the missing facts to the prompt, switch
77
+ to forked context, or ask the user. Do not launch vague tasks and rely on
78
+ supervisor round-trips to recover missing context.
79
+
65
80
  ### Escalate decisions upward
66
81
 
67
82
  If a subagent encounters an unapproved product, architecture, scope, merge, release, credential, or authority choice, it should use `contact_supervisor` and wait for the reply instead of deciding alone. Generic `intercom` is external or provider-supplied only. Use it only when external bridge instructions provide an explicit safe target. External checks, receipts, and review bots provide evidence only; they do not grant authority.
@@ -76,7 +76,7 @@ subagent({
76
76
  })
77
77
  ```
78
78
 
79
- Scripts run in a timed worker with only `runs.run`, `runs.all`, `runs.status`, `runs.ref/refs`, `emit`, captured `console`, and standard JavaScript. Pass explicit task text to `runs.run`. Mission-attached workflows also get `await state.get(key)` and `await state.set(key, value)` for durable JSON state shared across workflows on the same mission; `mission: false` workflows have no `state` global. Stable keys are required. Child launches follow ordinary single-agent execution controls. Give each child a distinct decision and output path when reports must outlive the workflow, then consume the aggregate workflow result before opening individual reports.
79
+ Scripts run in a timed worker with only `runs.run`, `runs.all`, `runs.status`, `runs.ref/refs`, `emit`, captured `console`, and standard JavaScript. Pass explicit task text to `runs.run`. Mission-attached workflows also get `await state.get(key)` and `await state.set(key, value)` for durable JSON state shared across workflows on the same mission; `mission: false` workflows have no `state` global. Stable keys are required. Child launches follow ordinary single-agent execution controls. Give each child a distinct decision and output path when reports must outlive the workflow, then consume the aggregate workflow result before opening individual reports. Do not ask children to write `reports/...` or other repo-root scratch paths in task text.
80
80
 
81
81
  If `runs.all` is missing in a running session, reload or update `pi-subagents` before retrying. The current runtime supports `runs.all`; `await Promise.all([runs.run(...)])` is also supported for advanced dynamic fanout.
82
82
 
@@ -116,7 +116,7 @@ subagent({
116
116
  })
117
117
  ```
118
118
 
119
- File-only output mode works for workflowScript child launches. Use distinct absolute or durable output paths when later script steps need stable references. For cross-codebase waves, include the repo slug or lane key in each output path so reports from different repositories cannot collide.
119
+ File-only output mode works for workflowScript child launches. Use relative child output paths for scratch reports so the runtime stores them under the run artifact directory and age-based cleanup can remove them. Use absolute paths only for user-approved durable destinations, such as session memory, a docs folder outside the repo, or a known handoff path. For cross-codebase waves, include the repo slug or lane key in each output path so reports from different repositories cannot collide.
120
120
 
121
121
  For review fanout where the parent continues a local audit:
122
122
 
@@ -129,7 +129,7 @@ const run = subagent({
129
129
  // Continue local inspection, then later call status with the returned id.
130
130
  ```
131
131
 
132
- While children run, the persistent FleetView and the collapsed foreground tool-result card show live per-child detail: resolved model and thinking level, `[fresh]`/`[fork]` context, tool/token/elapsed counters, and current activity. The collapsed running card also prints the configured expand-key hint ("Press … for live detail"); expanding it shows nested children, recent tools, and recent output. Model badges appear once the child's model resolves at first attempt start. `/subagents-fleet` opens the live fleet inspector, which also has per-child controls (`s` steer, `D` stop with confirmation). When optional Herdr 0.7.5+ is available, `H` opens a raw inspector dashboard for the selected active async child; this mirrors artifacts rather than attaching to the headless child. Use it for confusing or long-running active async work when the human wants a dedicated visual pane or FleetView is insufficient, not for routine headless runs.
132
+ While children run, the persistent FleetView and the collapsed foreground tool-result card show live per-child detail: resolved model and thinking level, `[fresh]`/`[fork]` context, tool/token/elapsed counters, and current activity. The collapsed running card also prints the configured expand-key hint ("Press … for live detail"); expanding it shows nested children, recent tools, and recent output. Model badges appear once the child's model resolves at first attempt start. `/subagents-fleet` opens the live fleet inspector, which also has per-child controls (`s` steer, `D` stop with confirmation). When optional Herdr 0.7.5+ is available, `H` opens a raw inspector dashboard for the selected active async child; this mirrors artifacts rather than attaching to the headless child. When optional Orca progress tabs are enabled, Pi creates one passive Orca observer tab for the top-level subagent call. Parallel and chain children share that tab and write child section headers into the mirrored log instead of opening one tab per child. Pi remains authoritative for lifecycle, status, control, artifacts, and results; the Orca tab is display-only. Pi also writes passive display metadata under `.pi/subagents/views/orca/` when possible, so other surfaces can discover the observer without treating it as an owned child run. Use visual panes for confusing or long-running active async work when the human wants a dedicated surface or FleetView is insufficient, not for routine headless runs.
133
133
 
134
134
  Inspect async runs with `subagent({ action: "status", id: "..." })` or `subagent({ action: "status" })` for active runs. Use `subagent({ action: "status", view: "fleet" })` when supervising several active foreground/background runs and `subagent({ action: "status", id: "...", view: "transcript", index: 0 })` when you need the latest child output without digging through artifacts. If a delegated fanout child launches nested runs, the parent status view shows them as a tree and you can target a nested run directly with its nested id.
135
135
 
@@ -137,8 +137,15 @@ Stop a current-session top-level async run with `stop` (or `/subagents-stop`). S
137
137
 
138
138
  ```typescript
139
139
  subagent({ action: "stop", id: "run-id" })
140
+ subagent({ action: "stop", id: "run-id", childId: "child-id" })
140
141
  ```
141
142
 
143
+ Use `childId` only for active async/workflow runs whose status snapshot shows a
144
+ matching child. Observer hosts can listen for the advertised
145
+ `subagent:child-status` event to update child-stop UI quickly, but status
146
+ snapshots remain the source of truth after reconnects or duplicate event
147
+ delivery.
148
+
142
149
  Use `steer` for top-level live async guidance and `resume` after a delegated run pauses or finishes. Routed nested runs retain their existing non-destructive live follow-up path:
143
150
 
144
151
  ```typescript
@@ -314,7 +321,7 @@ Routing rule:
314
321
  - Several projects with independent work: one async `workflowScript` whose child keys include repo slugs and whose child calls set explicit `cwd`; keep publication and merge decisions serial per repo.
315
322
  - Different project, substantial or long-running work: open a project-owned Herdr pane rooted there when a separate visible project session is useful, then give that project Pi session a narrow mission/result contract. Do not model it as ordinary child nesting, and do not expect existing headless runs to move into the pane.
316
323
 
317
- Project panes run a separate Pi session from the target directory. Subagents launched inside that pane use that project's config, agents, skills, files, git state, and mission records. The pane binding lives under `<projectRoot>/.pi/subagents/project-panes/herdr.json`. For ordinary headless delegation to another repo, prefer explicit `cwd` first; reserve project panes for visible or persistent project ownership.
324
+ Project panes run a separate Pi session from the target directory. Subagents launched inside that pane use that project's config, agents, skills, files, git state, and mission records. The pane binding lives under `<projectRoot>/.pi/subagents/project-panes/herdr.json`. When Pi runs inside Herdr, the owning pane reports compact active-work status and title suffixes, and the parent inline status counts opened project panes. Use Herdr itself or `project.status` / `project.close` for pane-level follow-up. For ordinary headless delegation to another repo, prefer explicit `cwd` first; reserve project panes for visible or persistent project ownership.
318
325
 
319
326
  ```typescript
320
327
  subagent({ action: "mission.create", mission: { title: "Ship auth refresh", objective: "Implement and validate refresh handling" } })
@@ -18,9 +18,9 @@ Use one writer per repo/cwd or worktree. Mutation lanes need distinct isolation
18
18
 
19
19
  For Pi extension repositories, keep lane worktrees outside auto-discovered extension directories such as `~/.pi/agent/extensions`. A stale extension worktree there can auto-load duplicate tools and shortcuts. Remove or move it only after its handoff is durable, the worktree is clean, and no run owns it.
20
20
 
21
- Partition fanout by repository, source seam, decision, or review angle. Each run needs a stable key, lane-specific task, and durable output path. Do not launch prompts that differ only by item name or broad file glob.
21
+ Partition fanout by repository, source seam, decision, or review angle. Each run needs a stable key, lane-specific task, and a managed output path when a file is needed. Do not launch prompts that differ only by item name or broad file glob.
22
22
 
23
- Use one async `workflowScript` for a coordinated wave. Use `runs.all` for independent lanes and `runs.run` for dependent lane stages. Give cross-repository runs explicit `cwd` values and lane-qualified outputs. Use `outputMode: "file-only"` when a report must survive the run or feed a later stage.
23
+ Use one async `workflowScript` for a coordinated wave. Use `runs.all` for independent lanes and `runs.run` for dependent lane stages. Give cross-repository runs explicit `cwd` values and lane-qualified outputs. Use `outputMode: "file-only"` when a report must survive the run or feed a later stage. Keep scratch outputs relative so they live under subagent artifacts; use absolute paths only for durable memory, approved docs paths, or final handoff files.
24
24
 
25
25
  ## Keep independent work moving
26
26
 
@@ -32,7 +32,7 @@ After a writer produces a candidate, run the required fresh-context, read-only r
32
32
 
33
33
  ## Handoff, cleanup, and recovery
34
34
 
35
- Use stable lane-qualified paths for reports and review output. A handoff states the lane status, repository and worktree, changed files, validation, open decisions, next action, and artifact or receipt paths.
35
+ Use stable lane-qualified artifact paths for reports and review output. A handoff states the lane status, repository and worktree, changed files, validation, open decisions, next action, and artifact or receipt paths. Copy only the final evidence to memory, a mission record, or a PR/comment, then remove scratch files from the active worktree before closing the lane.
36
36
 
37
37
  Keep a worktree until its handoff is durable, no run owns it, and no later gate needs it. Clean up only inside the recorded authority boundary. If a run stops or needs attention, preserve its worktree and artifacts, record the last known state and recovery owner, then resume that run or create one replacement lane from the handoff. Do not start another writer while worktree ownership is uncertain.
38
38
 
@@ -54,11 +54,11 @@ The prompt templates in `prompts/` encode workflows the parent agent can run on
54
54
 
55
55
  Use Council Mode when the user asks to convene advisors, debate a material decision, cross-examine recommendations, or critique and improve a plan with several model perspectives. This includes requests such as “run a council on this architecture,” “have Sol, Fable, and Kimi critique this plan,” or “get multiple oracles to debate the tradeoffs.” Read `../council-mode/SKILL.md` and follow its bounded parent-supervised protocol instead of launching ad hoc parallel oracle calls.
56
56
 
57
- Council advisors are read-only. User or project `council-*` profiles can pin models such as GPT 5.6 Sol, Fable, or Kimi and define any persistent stance in the profile body. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
57
+ Council advisors are read-only. User or project `council-*` profiles can pin models such as GPT 5.6 Sol, Fable, or Kimi and define any persistent stance in the profile body. Package advisors such as Surf's `gpt-pro` can join the roster only when the `surf-cli` Pi extension is installed and its `surf-oracle` provider is registered; treat them as external runners, omit child `async` for attached results, and do not pass `outputSchema` to them. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
58
58
 
59
59
  ### Parallel review technique
60
60
 
61
- Use this when the user wants adversarial review of a diff, plan, issue, file, or implemented work. Launch fresh-context `reviewer` agents with distinct angles generated from the actual target. Common angles are correctness/regressions, tests/validation, and simplicity/maintainability; adapt for TypeScript, UI, security, docs, or large structural changes. Reviewers should inspect files and diffs directly, return concise evidence-backed findings with file/line references, and avoid edits unless the user explicitly asks for a writer pass. The parent synthesizes fixes worth doing now, optional improvements, and feedback to ignore/defer before applying anything.
61
+ Use this when the user wants adversarial review of a diff, plan, issue, file, or implemented work. Launch fresh-context `reviewer` agents with distinct angles generated from the actual target. Common angles are correctness/regressions, tests/validation, and simplicity/maintainability; adapt for TypeScript, UI, security, docs, or large structural changes. Reviewers should inspect files and diffs directly, return concise evidence-backed findings with file/line references, and avoid edits unless the user explicitly asks for a writer pass. Filter on evidence, not severity: report only concrete current issues caused or made reachable by the target diff, with source proof, a test or repro, or a contract contradiction. Label findings P0/P1/P2 and end with `Merge verdict: BLOCK`, `Merge verdict: OK`, or `Merge verdict: OK with notes`. Use `blockers only` only for final pre-merge re-checks after P1/P2 findings are already captured, or for explicit emergency hotfix lanes where non-blocking findings are intentionally deferred. For targeted follow-up, ask only whether the named finding was resolved, whether the fix introduced a new defect in the fix blast radius, and whether prior P1/P2 notes still stand. For bot or PR-comment triage, classify each comment as VALID, STALE, INVALID, or OUT-OF-POLICY against current HEAD, then assign P0/P1/P2 only to VALID comments. The parent synthesizes fixes worth doing now, optional improvements, and feedback to ignore/defer before applying anything.
62
62
 
63
63
  ### Proactive skill-specialist technique
64
64
 
@@ -88,7 +88,7 @@ subagent({
88
88
 
89
89
  ### Review-loop technique
90
90
 
91
- Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no blockers or fixes worth doing now, remaining feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
91
+ Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no P0 blockers or P1 fixes worth doing now, remaining P2 feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
92
92
 
93
93
  As a conservative orchestration policy, do not pass `turnBudget` or a hard `toolBudget` to an implementation worker, fix worker, reviewer with edit authority, or other mutation-capable child. The default tool budget blocks read/search tools rather than mutation tools, but count limits still do not measure delivery safety. Use a narrow task plus an outer elapsed deadline with enough margin, then request a checkpoint after the current tool returns. The checkpoint should report changed files, build/test state, remaining work, and commit or PR state. An elapsed timeout is not a mutation-safe boundary and must not be used as the checkpoint trigger.
94
94
 
@@ -134,20 +134,20 @@ subagent({
134
134
 
135
135
  // Stage 2: single writer — the only child allowed to edit the active worktree.
136
136
  // Under outputMode "file-only" the awaited .output is the saved-output
137
- // reference, so pass the durable paths declared above to the writer.
137
+ // reference, so pass those managed artifact references to the writer.
138
138
  const worker = await runs.run("apply-fixes", {
139
139
  agent: "worker",
140
140
  phase: "Implementation",
141
141
  label: "Apply accepted fixes",
142
- task: "Apply only the accepted fixes from these planning summaries. You are the sole writer for the active worktree. Run focused validation and report changed files, commands, failures, and remaining issues.\\n\\nDeploy plan: plans/deploy.md\\n\\nScheduler plan: plans/scheduler.md\\n\\nSandbox plan: plans/sandbox.md",
142
+ task: "Apply only the accepted fixes from these planning summaries. You are the sole writer for the active worktree. Run focused validation and report changed files, commands, failures, and remaining issues.\\n\\nDeploy plan: " + plans[0].output + "\\n\\nScheduler plan: " + plans[1].output + "\\n\\nSandbox plan: " + plans[2].output,
143
143
  output: "worker/fixes.md",
144
144
  outputMode: "file-only"
145
145
  });
146
146
 
147
147
  // Stage 3: parallel read-only validation fanout
148
148
  const validations = await runs.all([
149
- { key: "validate-deploy-scheduler", agent: "reviewer", phase: "Validation", label: "Deploy/scheduler validation", task: "Validate the post-worker diff for deploy and scheduler fixes. Start from the worker result: " + worker.output + " (also worker/fixes.md). Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/deploy-scheduler.md", outputMode: "file-only" },
150
- { key: "validate-sandbox", agent: "reviewer", phase: "Validation", label: "Sandbox validation", task: "Validate the post-worker diff for sandbox/security fixes. Start from the worker result: " + worker.output + " (also worker/fixes.md). Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/sandbox.md", outputMode: "file-only" }
149
+ { key: "validate-deploy-scheduler", agent: "reviewer", phase: "Validation", label: "Deploy/scheduler validation", task: "Validate the post-worker diff for deploy and scheduler fixes. Start from the worker result: " + worker.output + ". Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/deploy-scheduler.md", outputMode: "file-only" },
150
+ { key: "validate-sandbox", agent: "reviewer", phase: "Validation", label: "Sandbox validation", task: "Validate the post-worker diff for sandbox/security fixes. Start from the worker result: " + worker.output + ". Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/sandbox.md", outputMode: "file-only" }
151
151
  ]);
152
152
 
153
153
  return { worker: worker.output, validations: validations.map(v => v.output) };
@@ -208,7 +208,7 @@ A strong subagent prompt usually includes:
208
208
  - **Success criteria**: what must be true before the child can finish.
209
209
  - **Hard constraints**: true invariants only, such as no edits for review-only tasks, one writer thread, child must not run subagents unless it is an explicitly assigned `tools: subagent` fanout child, or escalation for unapproved decisions.
210
210
  - **Validation**: targeted checks to run, or the next-best check when validation is impossible.
211
- - **Output**: the expected summary shape, artifact path, or finding format. Use repo-qualified durable output paths for cross-codebase waves.
211
+ - **Output**: the expected summary shape, artifact path, or finding format. Use managed artifact paths for scratch reports; reserve repo-qualified absolute paths for durable handoffs that the user approved.
212
212
  - **Stop rules**: when to ask via `intercom` or `contact_supervisor`, when to stop after enough evidence, and when not to keep searching.
213
213
 
214
214
  Give each role useful discovery anchors. Name source roots, filenames, symbols, types, methods, and paths for scouts. Give workers context files, plans, task paths, and named source seams before asking them to search. Give reviewers changed files, contracts, and any exhaustive-verification target. Tell oracle whether current source behavior, product/policy documents, plans, or inherited decisions are the evidence that matters.
@@ -58,18 +58,23 @@ function parseCsv(value: string): string[] {
58
58
  return [...new Set(value.split(",").map((v) => v.trim()).filter(Boolean))];
59
59
  }
60
60
 
61
- function configObject(config: unknown): { value?: Record<string, unknown>; error?: string } {
61
+ type ConfigObjectResult =
62
+ | { status: "ok"; value: Record<string, unknown> }
63
+ | { status: "missing" }
64
+ | { status: "error"; message: string };
65
+
66
+ function configObject(config: unknown): ConfigObjectResult {
62
67
  let val = config;
63
68
  if (typeof val === "string") {
64
69
  try {
65
70
  val = JSON.parse(val);
66
71
  } catch (error) {
67
72
  const message = error instanceof Error ? error.message : String(error);
68
- return { error: `config must be valid JSON: ${message}` };
73
+ return { status: "error", message: `config must be valid JSON: ${message}` };
69
74
  }
70
75
  }
71
- if (!val || typeof val !== "object" || Array.isArray(val)) return {};
72
- return { value: val as Record<string, unknown> };
76
+ if (!val || typeof val !== "object" || Array.isArray(val)) return { status: "missing" };
77
+ return { status: "ok", value: val as Record<string, unknown> };
73
78
  }
74
79
 
75
80
  function hasKey(obj: Record<string, unknown>, key: string): boolean {
@@ -177,7 +182,8 @@ function isMutableSource(source: AgentSource): source is ManagementScope {
177
182
  function modelWarning(ctx: ManagementContext, model: string | undefined): string | undefined {
178
183
  if (!model) return undefined;
179
184
  const found = ctx.modelRegistry.getAvailable().some((m) => `${m.provider}/${m.id}` === model || m.id === model);
180
- return found ? undefined : `Warning: model '${model}' is not in the current model registry.`;
185
+ if (found) return undefined;
186
+ return `Warning: model '${model}' is not in the current model registry. Run subagent({ action: "models" }) to list valid provider/id selectors, then use the exact provider/id form (bare ids resolve only when unique).`;
181
187
  }
182
188
 
183
189
  function fallbackModelsWarning(ctx: ManagementContext, fallbackModels: string[] | undefined): string | undefined {
@@ -215,7 +221,9 @@ export function editableAgentConfig(agent: AgentConfig): AgentConfig {
215
221
  const base = agent.override?.base;
216
222
  const {
217
223
  override: _override,
224
+ output: _output,
218
225
  outputMode: _outputMode,
226
+ defaultReads: _defaultReads,
219
227
  model: _model,
220
228
  fallbackModels: _fallbackModels,
221
229
  thinking: _thinking,
@@ -243,7 +251,9 @@ export function editableAgentConfig(agent: AgentConfig): AgentConfig {
243
251
 
244
252
  return withDeclaredExtensionPaths({
245
253
  ...editable,
254
+ ...(base.output !== undefined ? { output: base.output } : {}),
246
255
  ...(base.outputMode !== undefined ? { outputMode: base.outputMode } : {}),
256
+ ...(base.defaultReads !== undefined ? { defaultReads: [...base.defaultReads] } : {}),
247
257
  ...(base.model !== undefined ? { model: base.model } : {}),
248
258
  ...(base.fallbackModels !== undefined ? { fallbackModels: [...base.fallbackModels] } : {}),
249
259
  ...(base.thinking !== undefined ? { thinking: base.thinking } : {}),
@@ -832,6 +842,17 @@ function handleModels(params: ManagementParams, ctx: ManagementContext): AgentTo
832
842
  lines.push("");
833
843
  }
834
844
 
845
+ const availableFullIds = availableModels.map((m) => m.fullId).sort();
846
+ if (availableFullIds.length > 0) {
847
+ lines.push("Available models in this session's registry (copy an exact provider/id when passing model):");
848
+ lines.push("");
849
+ const shown = availableFullIds.slice(0, 80);
850
+ for (const fullId of shown) lines.push(` ${fullId}`);
851
+ if (availableFullIds.length > shown.length) lines.push(` ... and ${availableFullIds.length - shown.length} more`);
852
+ lines.push("");
853
+ lines.push("Use an exact provider/id from this list when you pass model; bare ids resolve only when unique in the registry.");
854
+ }
855
+
835
856
  return result(lines.join("\n"));
836
857
  }
837
858
 
@@ -857,9 +878,9 @@ function handleGet(params: ManagementParams, ctx: ManagementContext): AgentToolR
857
878
 
858
879
  export function handleCreate(params: ManagementParams, ctx: ManagementContext): AgentToolResult<Details> {
859
880
  const parsedConfig = configObject(params.config);
860
- if (parsedConfig.error) return result(parsedConfig.error, true);
881
+ if (parsedConfig.status === "error") return result(parsedConfig.message, true);
882
+ if (parsedConfig.status === "missing") return result("config required for create.", true);
861
883
  const cfg = parsedConfig.value;
862
- if (!cfg) return result("config required for create.", true);
863
884
  if (typeof cfg.name !== "string" || !cfg.name.trim()) return result("config.name is required and must be a non-empty string.", true);
864
885
  if (typeof cfg.description !== "string" || !cfg.description.trim()) return result("config.description is required and must be a non-empty string.", true);
865
886
  const name = sanitizeName(cfg.name);
@@ -907,9 +928,9 @@ export function handleCreate(params: ManagementParams, ctx: ManagementContext):
907
928
  export function handleUpdate(params: ManagementParams, ctx: ManagementContext): AgentToolResult<Details> {
908
929
  if (!params.agent) return result("Specify 'agent' for update.", true);
909
930
  const parsedConfig = configObject(params.config);
910
- if (parsedConfig.error) return result(parsedConfig.error, true);
931
+ if (parsedConfig.status === "error") return result(parsedConfig.message, true);
932
+ if (parsedConfig.status === "missing") return result("config required for update.", true);
911
933
  const cfg = parsedConfig.value;
912
- if (!cfg) return result("config required for update.", true);
913
934
  if (hasKey(cfg, "steps")) return result("Durable chain definitions were removed; use workflowScript or /prompt-workflow for repeatable workflows.", true);
914
935
  const warnings: string[] = [];
915
936
  const scopeHint = asDisambiguationScope(params.agentScope);
@@ -113,11 +113,11 @@ export function serializeAgent(config: AgentConfig, options: SerializeAgentOptio
113
113
  lines.push(`subagentOnlyExtensions: ${subagentOnlyExtensionsValue ?? ""}`);
114
114
  }
115
115
 
116
- if (config.output) lines.push(`output: ${config.output}`);
116
+ if (config.output || preserve("output")) lines.push(`output: ${config.output ?? ""}`);
117
117
  if (config.outputMode || preserve("outputMode")) lines.push(`outputMode: ${config.outputMode ?? ""}`);
118
118
 
119
119
  const readsValue = joinComma(config.defaultReads);
120
- if (readsValue) lines.push(`defaultReads: ${readsValue}`);
120
+ if (readsValue || preserve("defaultReads")) lines.push(`defaultReads: ${readsValue ?? ""}`);
121
121
 
122
122
  if (config.defaultProgress) lines.push("defaultProgress: true");
123
123
  if (config.interactive) lines.push("interactive: true");