pi-subagents 0.58.0 → 0.60.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. package/CHANGELOG.md +102 -0
  2. package/docs/agents.md +5 -3
  3. package/docs/configuration.md +4 -4
  4. package/docs/extension-api.md +1 -1
  5. package/docs/models.md +24 -1
  6. package/docs/observability.md +1 -1
  7. package/docs/tool-reference.md +93 -5
  8. package/docs/workflows.md +94 -3
  9. package/package.json +3 -1
  10. package/prompts/review-loop.md +2 -2
  11. package/skills/council-mode/SKILL.md +48 -243
  12. package/skills/council-mode/references/pass-contracts.md +150 -0
  13. package/skills/pi-subagents/SKILL.md +87 -37
  14. package/skills/pi-subagents/references/constraints-and-recipes.md +29 -233
  15. package/skills/pi-subagents/references/execution-controls.md +52 -7
  16. package/skills/pi-subagents/references/management-authoring-rpc.md +3 -4
  17. package/skills/pi-subagents/references/multi-lane-orchestration.md +13 -1
  18. package/skills/pi-subagents/references/prompting-and-roles.md +54 -26
  19. package/skills/pi-subagents/references/review-and-validation.md +73 -0
  20. package/src/agents/agent-management.ts +159 -42
  21. package/src/agents/agent-serializer.ts +4 -2
  22. package/src/agents/agents.ts +67 -26
  23. package/src/agents/runtime-agent-registry.ts +9 -16
  24. package/src/api/background-work.ts +5 -1
  25. package/src/api/delegation.ts +0 -7
  26. package/src/api/preflight.ts +6 -9
  27. package/src/api/shared-types.ts +2 -0
  28. package/src/extension/fanout-child.ts +5 -3
  29. package/src/extension/index.ts +51 -23
  30. package/src/extension/public-execution.ts +16 -2
  31. package/src/extension/schemas.ts +36 -10
  32. package/src/extension/tool-description.ts +24 -5
  33. package/src/intercom/result-intercom.ts +2 -0
  34. package/src/profiles/profiles.ts +5 -6
  35. package/src/runs/background/active-async-capacity.ts +2 -2
  36. package/src/runs/background/async-execution.ts +52 -37
  37. package/src/runs/background/async-job-tracker.ts +17 -12
  38. package/src/runs/background/async-resume.ts +23 -25
  39. package/src/runs/background/async-status-snapshot.ts +23 -261
  40. package/src/runs/background/async-status.ts +110 -8
  41. package/src/runs/background/chain-append.ts +6 -3
  42. package/src/runs/background/chain-root-attachment.ts +60 -8
  43. package/src/runs/background/control-channel.ts +3 -2
  44. package/src/runs/background/fleet-view.ts +21 -11
  45. package/src/runs/background/notify.ts +158 -6
  46. package/src/runs/background/result-files.ts +2 -1
  47. package/src/runs/background/result-watcher.ts +2 -0
  48. package/src/runs/background/resume-guidance.ts +1 -1
  49. package/src/runs/background/retained-children.ts +1 -1
  50. package/src/runs/background/run-status.ts +30 -9
  51. package/src/runs/background/scheduled-runs.ts +86 -7
  52. package/src/runs/background/stale-run-reconciler.ts +10 -4
  53. package/src/runs/background/steering.ts +4 -14
  54. package/src/runs/background/subagent-runner.ts +351 -358
  55. package/src/runs/background/subagent-wait.ts +68 -12
  56. package/src/runs/background/terminal-run-index.ts +1 -1
  57. package/src/runs/background/wait-completions.ts +25 -1
  58. package/src/runs/background/wait-config.ts +23 -9
  59. package/src/runs/background/wait-tool.ts +9 -2
  60. package/src/runs/foreground/async-steering-action.ts +2 -2
  61. package/src/runs/foreground/execution.ts +125 -122
  62. package/src/runs/foreground/foreground-control.ts +3 -0
  63. package/src/runs/foreground/foreground-history.ts +1 -0
  64. package/src/runs/foreground/subagent-executor.ts +576 -240
  65. package/src/runs/foreground/workflow-detach-reconcile.ts +99 -200
  66. package/src/runs/shared/abort-recovery.ts +119 -0
  67. package/src/runs/shared/async-status-projection.ts +597 -0
  68. package/src/runs/shared/background-process-options.ts +9 -0
  69. package/src/runs/shared/child-identity.ts +19 -4
  70. package/src/runs/shared/child-launch-plan.ts +151 -0
  71. package/src/runs/shared/completion-evidence.ts +89 -0
  72. package/src/runs/shared/completion-guard.ts +1 -1
  73. package/src/runs/shared/dynamic-fanout.ts +2 -2
  74. package/src/runs/shared/host-step-status.ts +230 -0
  75. package/src/runs/shared/lane-metadata.ts +105 -0
  76. package/src/runs/shared/mcp-config-sources.ts +42 -6
  77. package/src/runs/shared/mcp-direct-tool-allowlist.ts +93 -143
  78. package/src/runs/shared/mcp-direct-tool-grant.ts +194 -0
  79. package/src/runs/shared/model-fallback.ts +9 -2
  80. package/src/runs/shared/mutation-evidence.ts +52 -3
  81. package/src/runs/shared/nested-events.ts +6 -2
  82. package/src/runs/shared/nested-render.ts +7 -3
  83. package/src/runs/shared/parallel-handoff.ts +419 -7
  84. package/src/runs/shared/parallel-utils.ts +7 -0
  85. package/src/runs/shared/pi-args.ts +67 -3
  86. package/src/runs/shared/single-output.ts +72 -22
  87. package/src/runs/shared/subagent-prompt-runtime.ts +33 -6
  88. package/src/runs/shared/workflow-graph.ts +15 -0
  89. package/src/runs/shared/worktree-cleanup-plan.ts +847 -0
  90. package/src/runs/shared/worktree.ts +18 -0
  91. package/src/shared/child-session-name.ts +46 -0
  92. package/src/shared/extension-context.ts +24 -0
  93. package/src/shared/formatters.ts +5 -2
  94. package/src/shared/launch-contract.ts +1 -1
  95. package/src/shared/settings.ts +9 -103
  96. package/src/shared/types.ts +232 -30
  97. package/src/shared/utils.ts +35 -55
  98. package/src/slash/delegation-adapters.ts +1 -8
  99. package/src/slash/delegation-request.ts +0 -4
  100. package/src/slash/slash-bridge.ts +1 -2
  101. package/src/slash/slash-commands.ts +369 -90
  102. package/src/slash/slash-live-state.ts +22 -11
  103. package/src/tui/fleet-status.ts +134 -75
  104. package/src/tui/fleet.ts +11 -5
  105. package/src/tui/render-helpers.ts +31 -0
  106. package/src/tui/render.ts +895 -121
  107. package/src/watchdog/change-signature.ts +40 -1
  108. package/src/watchdog/turn-delta.ts +1 -1
  109. package/src/workflows/chat-progress.ts +6 -3
  110. package/src/workflows/host-command.ts +235 -0
  111. package/src/workflows/scripted-workflow.ts +503 -38
  112. package/src/workflows/workflow-child-summary.ts +9 -5
  113. package/src/workflows/workflow-preflight.ts +270 -0
  114. package/src/workflows/workflow-receipt.ts +43 -4
  115. package/src/workflows/workflow-settlement.ts +246 -0
  116. package/src/runs/shared/turn-budget.ts +0 -98
@@ -18,7 +18,7 @@ subagent({ action: "list" })
18
18
  subagent({ action: "children.list" })
19
19
  ```
20
20
 
21
- Lists up to the last 10 retained workflow children from this parent session with explicit `resumable` or `not resumable` rows. Resume only rows reported `resumable`. Send a simple follow-up or implementation challenge with `subagent({ action: "resume", id: "<run-id>", message: "..." })`. Continue one inside a workflow with `runs.run(key, { resume: "<run-id>", task: "follow-up" })`; the revived child keeps its stored agent, model, and tool contract. If no resumable child is listed, start a same-role fallback challenge and label it as fallback. `steer` with `mode: "follow_up"` only queues text for the next `resume` when the child has already completed.
21
+ Lists up to the last 10 retained workflow children from this parent session with explicit `resumable` or `not resumable` rows. Resume only rows reported `resumable`. Send a simple follow-up or implementation challenge with `subagent({ action: "resume", id: "<run-id>", message: "..." })`. Continue one inside a workflow with `runs.run(key, { resume: "<run-id>", task: "follow-up" })`; each workflow key identifies one result lane, so use a new stable workflow key for every distinct retained resume pass. Same-key calls are reused only when launch parameters are identical, and incompatible parameters are rejected. The revived child keeps its stored agent, model, and tool contract. If no resumable child is listed, start a same-role fallback challenge and label it as fallback. `steer` with `mode: "follow_up"` only queues text for the next `resume` when the child has already completed.
22
22
 
23
23
  ### Refinement overlays
24
24
 
@@ -41,7 +41,7 @@ subagent({
41
41
  description: "Project-specific implementation helper",
42
42
  systemPrompt: "Your system prompt here.",
43
43
  systemPromptMode: "replace",
44
- model: "openai-codex/gpt-5.4",
44
+ model: "provider/model-id",
45
45
  tools: "read,grep,find,ls,bash"
46
46
  }
47
47
  })
@@ -97,7 +97,7 @@ name: my-agent
97
97
  package: code-analysis
98
98
  description: What this agent does
99
99
  aliases: developer, coder
100
- model: openai-codex/gpt-5.4
100
+ model: provider/model-id
101
101
  thinking: high
102
102
  tools: read, grep, find, ls, bash
103
103
  systemPromptMode: replace
@@ -126,7 +126,6 @@ That is only a starting point. Omit `package` for the traditional unqualified ru
126
126
  - `acceptanceRole`
127
127
  - `async` — single-agent default for background launch (`true`/`false`); explicit tool-call `async` wins
128
128
  - `timeoutMs` — single-agent default run-level max runtime in ms; foreground calls use a 30-minute package default only when neither the call nor agent provides one (tool alias `maxRuntimeMs` is also accepted)
129
- - `turnBudget` — single-agent default `{ maxTurns, graceTurns? }` JSON object
130
129
 
131
130
  `aliases` is an optional comma-separated or block-list set of alternate names for selecting an agent. Aliases resolve to the canonical `name` for execution, status, persistence, and config. Exact canonical names take precedence over aliases, and alias collisions between distinct canonical agents fail as ambiguous. Management create/update accepts a comma-separated string, string array, or `false`/empty string to clear aliases.
132
131
 
@@ -2,6 +2,8 @@
2
2
 
3
3
  Use this reference when several independent tasks need coordinated workers, worktrees, or repositories. It defines lane ownership; use the other pi-subagents references for run controls, prompts, and mission details. The parent remains the final decision-maker.
4
4
 
5
+ Create lanes only when delegation materially improves evidence, independent review, or isolated execution. Do not manufacture parallelism: keep dependent work serial, and only split work when each lane has a distinct decision and useful output.
6
+
5
7
  ## Lane board and authority
6
8
 
7
9
  Before multiple mutation-capable lanes start, record this board in the parent context:
@@ -20,15 +22,25 @@ For Pi extension repositories, keep lane worktrees outside auto-discovered exten
20
22
 
21
23
  Partition fanout by repository, source seam, decision, or review angle. Each run needs a stable key, lane-specific task, and a managed output path when a file is needed. Do not launch prompts that differ only by item name or broad file glob.
22
24
 
25
+ ### Cold-start packets and bounded orchestration audits
26
+
27
+ Every child packet must stand alone: include the goal, exact repository/cwd/ref, authority and edit boundary, relevant context/evidence, success criteria, validation, expected output, and stop/escalation rules. Do not rely on parent history, an issue number, or a broad glob alone. An orchestration audit by a top-reasoning critic model is read-only and returns at most three cited omissions; use high thinking only as an explicit parent/user escalation, never as an autonomous root or a parallel placeholder.
28
+
23
29
  Use one async `workflowScript` for a coordinated wave. Use `runs.all` for independent lanes and `runs.run` for dependent lane stages. Give cross-repository runs explicit `cwd` values and lane-qualified outputs. Use `outputMode: "file-only"` when a report must survive the run or feed a later stage. Keep scratch outputs relative so they live under subagent artifacts; use absolute paths only for durable memory, approved docs paths, or final handoff files.
24
30
 
25
31
  ## Keep independent work moving
26
32
 
27
33
  While one lane waits, run safe independent preparation, validation, or fresh read-only review lanes. Do not block the parent just because a run is active. If no safe lane remains, record the blocker and the event that will reopen work.
28
34
 
35
+ In an ordinary interactive session, completion wakes the parent; after useful
36
+ async lanes are launched or triaged, yield rather than use
37
+ `subagent_wait({ all: true })` as a barrier. “Continue/orchestrate/work until
38
+ done” means keep the board moving while safe immediate work remains. If only
39
+ async lanes are running, record the revisit trigger and yield.
40
+
29
41
  An ordinary coordinated workflow has one mission. Use its durable state, artifacts, run records, and receipts for recovery. Treat a receipt as evidence, not as authority or acceptance.
30
42
 
31
- After a writer produces a candidate, run the required fresh-context, read-only reviewer. The reviewer inspects the exact worktree and returns evidence-backed findings. The parent decides which findings are in scope and whether the lane is ready. Send accepted fixes to that lane's sole writer, then rerun only the affected gate.
43
+ After a writer produces a candidate, run the required fresh-context, read-only reviewer. The reviewer inspects the exact worktree and returns evidence-backed findings. The parent decides which findings are in scope and whether the lane is ready. Use `review-and-validation.md` for finding disposition, validation, and gate-failure triage. Send accepted fixes to that lane's sole writer, then rerun only the affected gate.
32
44
 
33
45
  ## Handoff, cleanup, and recovery
34
46
 
@@ -8,25 +8,26 @@ Parent extensions may register a session-scoped, out-of-band ceiling through `pi
8
8
 
9
9
  ## When to Use
10
10
 
11
- - **Complex work orchestration**: use Fable mode as the default parent-agent loop for complex work. Complex means the task has multiple moving parts, unclear acceptance, cross-cutting code, meaningful user-visible impact, expensive or irreversible validation, broad review surface, or the user asks for orchestration. Lightweight one-off delegation can stay lightweight.
11
+ - **Complex work orchestration**: keep the parent on its ordinary strong default model. Delegate only when another child materially improves evidence, independent review, or isolated execution; omission failures are cheaper than unnecessary commissions. For hard orchestration or root-cause questions, use a top-reasoning model only as a bounded read-only critic/oracle escalation, never as an autonomous root. Complex means the task has multiple moving parts, unclear acceptance, cross-cutting code, meaningful user-visible impact, expensive or irreversible validation, broad review surface, or the user asks for orchestration. Lightweight one-off delegation can stay lightweight.
12
12
  - **Advisory review**: use fresh-context `reviewer` agents for adversarial code review, or fork to `oracle` when inherited decisions and drift matter
13
13
  - **Implementation handoff**: have `oracle` advise, then `worker` implement only after an approved direction
14
14
  - **Recon and planning**: use `scout`, then write a plan when needed
15
15
  - **Parallel exploration**: run multiple non-conflicting tasks concurrently
16
16
  - **Regular skill specialists**: when discovery shows proactive skill subagent suggestions and the current work is broad enough, launch a small fresh-context fanout that asks one subagent per relevant regularly used skill to apply that skill's perspective to the task
17
- - **Long-running work**: launch async/background runs and inspect them later. For mutation-capable work, bound the delivery slice and elapsed runtime, then request checkpoints after active tool work returns. Reserve hard turn and tool-call caps for explicitly read-only children.
17
+ - **Long-running work**: launch async/background runs and inspect them later. For mutation-capable work, bound the delivery slice and elapsed runtime, then request checkpoints after active tool work returns. Reserve hard tool-call caps for explicitly read-only children.
18
18
  - **Subagent control**: watch needs-attention signals and soft-interrupt only when a delegated run is genuinely blocked
19
19
  - **Agent authoring**: create, update, or override project agents. Treat saved chain records as legacy inspection or migration inputs, not as a current authoring target.
20
20
 
21
21
  ## Tool vs Slash Commands
22
22
 
23
- Agents use the `subagent(...)` tool with `workflowScript` for execution, and `action` for management, status, and control. Humans often use the slash-command layer instead:
23
+ Agents use the `subagent(...)` tool for execution, management, status, and control. Direct `{ agent, task }` execution is enough for one bounded child task; use `workflowScript` when the parent needs JavaScript control flow or data-dependent branching, keyed, parallel, sequential, retry, retained-resume, aggregate, or explicit staged-lane behavior (`runs.lanes`). Humans often use the slash-command layer instead:
24
24
 
25
25
  - `/run` — launch a single agent
26
26
  - `workflowScript` — the sole public surface for sequence, parallelism, branching, retries, and aggregation
27
27
  - `/subagents` — interactive admin for inspecting agents and editing model, thinking, or system prompt
28
28
  - `/subagents-stop [run-id]` — stop a current-session top-level async run; opens a selector when no id is given
29
29
  - `/subagents-detach [run-id]` — detach an active foreground single-subagent run without terminating its child
30
+ - `/subagents-steer <run-id> [--child <child-id>] <message>` — steer a live async run (or one child of it) from non-TUI sessions and RPC hosts
30
31
  - `/subagent-cost` — show parent plus child token usage and cost for the session
31
32
  - `/subagents-fleet` — open the live fleet inspector with per-child controls; `Ctrl+Alt+F` opens it during an active foreground turn, `↑↓`/`jk` selects children, `PgUp`/`PgDn` scrolls transcript detail, `s` steers the selected live async child, and `D` stops its top-level async run after confirmation
32
33
  - `/subagents-watchdog` — inspect or configure the opt-in adversarial change watchdog (model, on/off, recommend-model, check)
@@ -50,11 +51,15 @@ Packaged prompt shortcuts are also available for repeatable workflows. Treat the
50
51
 
51
52
  The prompt templates in `prompts/` encode workflows the parent agent can run on demand. If the user provides a URL, issue, PR, plan, local file, screenshot, or freeform target, treat that target as the primary scope: read or fetch it before launching children, then include it explicitly in every child task. For targets outside the parent cwd, include the exact repository, explicit `cwd`, authority boundary, and expected output path in each child task. Do not depend on the parent conversation history when the recipe calls for fresh context.
52
53
 
54
+ ### Commission-risk and cold-start packets
55
+
56
+ Delegate only when the child materially improves evidence, independent review, or isolated execution; do not manufacture parallelism. Every child packet must be cold-start complete: state the goal, exact target/cwd/ref, authority and edit boundary, relevant context/evidence, success criteria, validation, output, and stop/escalation rules. For an orchestration audit by the critic tier, make the child read-only and request at most three omissions, each cited to a file, line, or decision; high thinking is an explicit escalation, not a default.
57
+
53
58
  ### Council Mode technique
54
59
 
55
- Use Council Mode when the user asks to convene advisors, debate a material decision, cross-examine recommendations, or critique and improve a plan with several model perspectives. This includes requests such as “run a council on this architecture,” “have Sol, Fable, and Kimi critique this plan,” or “get multiple oracles to debate the tradeoffs.” Read `../council-mode/SKILL.md` and follow its bounded parent-supervised protocol instead of launching ad hoc parallel oracle calls.
60
+ Use Council Mode when the user asks to convene advisors, debate a material decision, cross-examine recommendations, or critique and improve a plan with several model perspectives. This includes requests such as “run a council on this architecture,” “have the configured advisors critique this plan,” or “get multiple oracles to debate the tradeoffs.” Read `../council-mode/SKILL.md` and follow its bounded parent-supervised protocol instead of launching ad hoc parallel oracle calls.
56
61
 
57
- Council advisors are read-only. User or project `council-*` profiles can pin models such as GPT 5.6 Sol, Fable, or Kimi and define any persistent stance in the profile body. Package advisors such as Surf's `gpt-pro` can join the roster only when the `surf-cli` Pi extension is installed and its `surf-oracle` provider is registered; treat them as external runners, omit child `async` for attached results, and do not pass `outputSchema` to them. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
62
+ Council advisors are read-only. User or project `council-*` profiles choose allowed models and define any persistent stance in the profile body. A top-reasoning advisor remains bounded and read-only; it does not become the root. Package advisors such as Surf's `gpt-pro` can join the roster only when the `surf-cli` Pi extension is installed and its `surf-oracle` provider is registered; treat them as external runners, omit child `async` for attached results, and do not pass `outputSchema` to them. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
58
63
 
59
64
  ### Parallel review technique
60
65
 
@@ -90,7 +95,7 @@ subagent({
90
95
 
91
96
  Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no P0 blockers or P1 fixes worth doing now, remaining P2 feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
92
97
 
93
- As a conservative orchestration policy, do not pass `turnBudget` or a hard `toolBudget` to an implementation worker, fix worker, reviewer with edit authority, or other mutation-capable child. The default tool budget blocks read/search tools rather than mutation tools, but count limits still do not measure delivery safety. Use a narrow task plus an outer elapsed deadline with enough margin, then request a checkpoint after the current tool returns. The checkpoint should report changed files, build/test state, remaining work, and commit or PR state. An elapsed timeout is not a mutation-safe boundary and must not be used as the checkpoint trigger.
98
+ As a conservative orchestration policy, do not pass a hard `toolBudget` to an implementation worker, fix worker, reviewer with edit authority, or other mutation-capable child. The default tool budget blocks read/search tools rather than mutation tools, but count limits still do not measure delivery safety. Use a narrow task plus an outer elapsed deadline with enough margin, then request a checkpoint after the current tool returns. The checkpoint should report changed files, build/test state, remaining work, and commit or PR state. An elapsed timeout is not a mutation-safe boundary and must not be used as the checkpoint trigger.
94
99
 
95
100
  ### Parallel research technique
96
101
 
@@ -110,6 +115,12 @@ Use this after implementation when the user wants cleanup review or when a final
110
115
 
111
116
  Use this when a broad diff has known reviewer findings across several items and the user wants the parent to “orchestrate subagents like a boss.” Keep the active worktree safe with a three-stage `workflowScript`:
112
117
 
118
+ When staged seams are available, a low-tier writer should not receive the
119
+ end-to-end issue. Use `runs.lanes` inside `workflowScript` to keep stages narrow:
120
+ a scout/red test, helper-only change, one render seam, validation, minimality
121
+ challenge, or fresh review. Give the writer only its assigned implementation
122
+ stage; keep sequencing and synthesis with the parent.
123
+
113
124
  1. A parallel read-only planning fanout, one reviewer per issue cluster. Each child inspects the real diff and returns exact files, line refs, proposed fixes, and focused validation. They must not edit.
114
125
  2. One writer worker. It receives the reviewer summaries as the awaited planning results (or their durable output paths) interpolated into its task, plus the parent’s accepted scope, stop rules, and verification contract. It is the only child allowed to edit the active worktree.
115
126
  3. A parallel read-only validation fanout. Validators inspect the worker diff from fresh context with distinct angles, report pass/fail, remaining blockers, and missing verification.
@@ -160,19 +171,19 @@ subagent({
160
171
  Builtin agents load at the lowest priority. Project agents override user agents,
161
172
  and user/project agents override builtins with the same name.
162
173
 
163
- | Agent | Purpose | Model | Typical output / role |
174
+ | Agent | Purpose | Recommended tier | Typical output / role |
164
175
  |-------|---------|-------|------------------------|
165
- | `scout` | Fast codebase recon | inherits default | Writes `context.md` handoff material |
166
- | `worker` | Implementation and approved oracle handoffs | inherits default | Single-writer implementation with decision escalation |
167
- | `reviewer` | Review specialist | inherits default | Default recipes are review-only; tools include edit/write when a fix pass is explicit |
168
- | `researcher` | Web research brief generator | inherits default | Writes `research.md` |
169
- | `delegate` | Lightweight generic delegate | inherits default | No fixed output; generic delegated work |
170
- | `oracle` | Decision-consistency advisory review | inherits default | Advisory review, intercom coordination |
171
- | `advisor` | Claude Code-compatible alias for `oracle` | inherits default | Same advisory role as `oracle` |
176
+ | `scout` | Fast codebase recon | fast worker/scout tier | Writes `context.md` handoff material |
177
+ | `worker` | Implementation and approved oracle handoffs | capable worker tier | Single-writer implementation with decision escalation |
178
+ | `reviewer` | Review specialist | strong reviewer tier; high thinking for serious reviews | Default recipes are review-only; tools include edit/write when a fix pass is explicit |
179
+ | `researcher` | Web research brief generator | inherits configured default | Writes `research.md` |
180
+ | `delegate` | Lightweight generic delegate | inherits configured default | No fixed output; generic delegated work |
181
+ | `oracle` | Decision-consistency advisory review | top-reasoning critic tier, bounded read-only; high thinking escalation only | Advisory review, intercom coordination |
182
+ | `advisor` | Compatibility alias for `oracle` | top-reasoning critic tier, bounded read-only; high thinking escalation only | Same advisory role as `oracle` |
172
183
 
173
184
  Builtin `worker` and `delegate` use strict tool allowlists and do not inherit ambient parent extension tools. To give a child an extension tool, name it in `tools` and load its provider via `extensions`, a path-like `tools` entry, or `subagentOnlyExtensions`. Custom agents without an `extensions` field follow `subagents.defaultExtensions` when set.
174
185
 
175
- Builtin agents inherit the current Pi default model unless a run, user setting, project setting, or `subagents.defaultModel` overrides `model`. Set `subagents.defaultModel` when subagents should use a different default model than the parent session. Override builtin defaults before copying full agent files when a small tweak is enough.
186
+ Builtin agents inherit the current Pi default model unless a run, user setting, project setting, or `subagents.defaultModel` overrides `model`. The table records recommended tier routing, not shipped hard defaults; explicit run, user, or project settings still win. Keep the parent/orchestrator on the ordinary strong default model unless parent/user policy says otherwise. Override builtin defaults before copying full agent files when a small tweak is enough.
176
187
 
177
188
  Set `subagents.defaultThinking` to apply a shared thinking level to builtin, package, user, and project agents whose frontmatter leaves `thinking` unset. Project settings win over user settings; explicit frontmatter (including `thinking: false`), `agentOverrides.<name>.thinking`, and per-run overrides remain more specific. This setting affects child agents only and does not change the parent session's default thinking level.
178
189
 
@@ -187,12 +198,32 @@ Set `subagents.defaultThinking` to apply a shared thinking level to builtin, pac
187
198
  For one run, use inline config:
188
199
 
189
200
  ```text
190
- /run reviewer[model=anthropic/claude-sonnet-4] "Review this diff"
201
+ /run reviewer[model=provider/review-model] "Review this diff"
191
202
  ```
192
203
 
193
204
  For persistent tweaks, edit `subagents.agentOverrides` in user or project settings. User overrides apply everywhere. Project overrides apply only in that repo and win over user overrides. Use `/subagents-models` or `subagent({ action: "models" })` to inspect the live mapping after settings and overrides load.
194
205
 
195
- Model ids do not have to be exact. Separator variations (`claude-haiku-4.5` vs `claude-haiku-4-5`), case (`Claude-Sonnet-4`), and optional trailing date stamps (`claude-haiku-4-5-20251001`) all resolve to the same registry model. Exact `provider/id` wins; a qualified `provider/model` never switches providers. To constrain subagents to a budget or compliance profile, set `subagents.modelScope: { enforce: true, allow: ["anthropic/*", "openai/gpt-5-*"] }` in user or project settings. Out-of-scope models you pass explicitly error and abort; models inherited from frontmatter, `subagents.defaultModel`, agent frontmatter, or the parent session only warn.
206
+ Provider-scoped entries can layer on top of the default override for the active parent session provider. The provider is selected once from the parent model before child model fallback starts, so fallback attempts cannot switch configuration. Within each settings file, the provider entry wins per field; project settings still win over user settings.
207
+
208
+ ```json
209
+ {
210
+ "subagents": {
211
+ "agentOverrides": {
212
+ "worker": { "thinking": "medium" }
213
+ },
214
+ "agentOverridesByProvider": {
215
+ "provider-a": {
216
+ "worker": { "model": "provider-a/fast-worker-model" }
217
+ },
218
+ "provider-b": {
219
+ "worker": { "model": "provider-b/fast-worker-model" }
220
+ }
221
+ }
222
+ }
223
+ }
224
+ ```
225
+
226
+ Model ids do not have to be exact. Separator variations (`fast.worker-v1` vs `fast-worker-v1`), case (`Strong-Review-Model`), and optional trailing date stamps all resolve to the same registry model. Exact `provider/id` wins; a qualified `provider/model` never switches providers. To constrain subagents to a budget or compliance profile, set `subagents.modelScope: { enforce: true, allow: ["approved-provider/*", "second-provider/approved-*"] }` in user or project settings. Out-of-scope models you pass explicitly error and abort; models inherited from frontmatter, `subagents.defaultModel`, agent frontmatter, or the parent session only warn.
196
227
 
197
228
  For model fleets, use the profile commands instead of hand-editing repeated overrides: `/subagents-refresh-provider-models <provider>`, `/subagents-generate-profiles <provider>`, `/subagents-load-profile <name>`, and `/subagents-check-profile <name>`. Profiles live under `~/.pi/agent/profiles/pi-subagents/` and replace only `settings.subagents` when loaded.
198
229
 
@@ -206,7 +237,7 @@ A strong subagent prompt usually includes:
206
237
  - **Authority boundary**: whether the child may read, edit, commit, push, comment, close, merge, publish, or release. Omit or forbid actions that are not approved.
207
238
  - **Context/evidence**: relevant plan paths, files, diffs, decisions, or user constraints already approved.
208
239
  - **Success criteria**: what must be true before the child can finish.
209
- - **Hard constraints**: true invariants only, such as no edits for review-only tasks, one writer thread, child must not run subagents unless it is an explicitly assigned `tools: subagent` fanout child, or escalation for unapproved decisions.
240
+ - **Hard constraints**: true invariants only, such as no edits for review-only tasks, one writer thread, child must not run subagents unless it is explicitly authorized through `tools: subagent` or `allowNestedSubagents: true`, or escalation for unapproved decisions.
210
241
  - **Validation**: targeted checks to run, or the next-best check when validation is impossible.
211
242
  - **Output**: the expected summary shape, artifact path, or finding format. Use managed artifact paths for scratch reports; reserve repo-qualified absolute paths for durable handoffs that the user approved.
212
243
  - **Stop rules**: when to ask via `intercom` or `contact_supervisor`, when to stop after enough evidence, and when not to keep searching.
@@ -228,9 +259,9 @@ Direct settings example:
228
259
  "subagents": {
229
260
  "agentOverrides": {
230
261
  "reviewer": {
231
- "model": "anthropic/claude-sonnet-4",
262
+ "model": "provider/strong-review-model",
232
263
  "thinking": "high",
233
- "fallbackModels": ["openai-codex/gpt-5.6-luna:low"],
264
+ "fallbackModels": ["backup-provider/strong-review-model"],
234
265
  "acceptanceRole": "read-only"
235
266
  }
236
267
  }
@@ -248,14 +279,11 @@ agent with the same name only when you want a substantially different agent.
248
279
 
249
280
  ### Recommended model tiering (optional)
250
281
 
251
- When several providers are available, route agents by task shape instead of one model for everything:
282
+ Keep the parent/orchestrator on the ordinary strong default model because omission failures are cheaper than unnecessary commissions. Route workers and scouts to a fast, capable worker tier, and keep serious reviews on the strong tier at high thinking. Use a top-reasoning model only for bounded, read-only critic/oracle/root-cause audits; critic-tier high thinking is escalation-only and never an autonomous root. Explicit parent/user model policy wins over these recommendations.
252
283
 
253
- 1. **Fast workhorse** — cheapest capable model at low thinking for recon, lookups, and mechanical edits (for example on `scout`).
254
- 2. **Standard well-scoped** — mid-tier model at medium thinking for most delegations: routine multi-file edits, focused reviews, straightforward implementation (for example on `worker`, `reviewer`, `delegate`).
255
- 3. **Deep but bounded** — top reasoning model at high thinking only for hard tasks that arrive with explicit goals and completion criteria; these models loop on vague goals (for example on oracle-style agents).
256
- 4. **Taste and intent** — a model that reads human intent well for ambiguous work: UX/design judgment, product tradeoffs, planning from vague requirements, writing quality.
284
+ Examples are illustrative, not requirements. Map these tiers to concrete models in user/project settings or a profile. A non-OpenAI setup should choose comparable available models by capability.
257
285
 
258
- Routing rule: use tiers 1–3 when the task is well-scoped; use tier 4 when scoping or judging is the task itself. Give tier-4 agents cross-provider `fallbackModels` so subscription usage limits degrade gracefully; fallback triggers automatically on rate-limit and overload errors. Note that forked context over an Anthropic parent transcript with signed thinking blocks forces the child's thinking off, so intent-tier agents work best with fresh context.
286
+ Use `fallbackModels` when a tier has provider quota or availability risk. Prefer fresh context for cross-provider children when inherited provider-specific reasoning blocks would force thinking off.
259
287
 
260
288
  If a provider rejects model IDs with thinking suffixes, use
261
289
  `subagents.disableThinking: true` in user or project settings to clear bundled
@@ -0,0 +1,73 @@
1
+ # Pi Subagents: Review And Validation
2
+
3
+ Generic review and delivery guidance for delegated work. This file does not encode private backlog, merge, or release policy.
4
+
5
+ ## Delivery loop
6
+
7
+ Use the smallest loop that proves the change:
8
+
9
+ 1. Inspect the source, diff, issue, or plan directly.
10
+ 2. Keep one writer for each cwd or worktree.
11
+ 3. Run focused validation that can fail for the changed behavior.
12
+ 4. Use fresh-context read-only review for substantial, risky, public, or hard-to-see changes.
13
+ 5. Apply only accepted findings inside the same writer boundary.
14
+ 6. Re-run affected validation and review only the changed blast radius.
15
+ 7. Inspect the final diff and evidence before parent acceptance.
16
+
17
+ Skip review ceremony for trivial wording, renames, or local-only probes when direct parent inspection is enough.
18
+
19
+ ## Review shape
20
+
21
+ | Situation | Shape |
22
+ | --- | --- |
23
+ | One coherent diff or one risk | one reviewer |
24
+ | Independent risks, such as correctness, tests, security, or UI | parallel reviewers with distinct contracts |
25
+ | Possible over-scope or needless complexity | same-writer challenge before fresh review |
26
+ | Material design tradeoff | council mode |
27
+
28
+ Reviewers are fresh-context by default. Forked reviewers are for parent-history, drift, or prior-decision evidence.
29
+
30
+ ## Finding disposition
31
+
32
+ The parent classifies each finding against current HEAD:
33
+
34
+ - **Valid blocker:** concrete failure, repro, security issue, contract mismatch, or source-proven regression. Fix now.
35
+ - **Valid non-blocker:** real but outside the delivery slice. Record or defer.
36
+ - **Stale:** fixed or absent at the reviewed head. Cite current evidence.
37
+ - **Invalid:** contradicted by source, tests, docs, or user-approved scope. Cite the contradiction.
38
+ - **Out of policy/scope:** needs unapproved product, architecture, authority, release, or public-repo action. Escalate.
39
+ - **Speculative:** no contract, repro, or reachable failure. Do not block.
40
+
41
+ A clean reviewer result is evidence, not publication authority.
42
+
43
+ ## Gate-failure triage
44
+
45
+ When validation fails:
46
+
47
+ 1. Confirm the run belongs to the exact head/ref under judgment.
48
+ 2. Read the focused failing logs first.
49
+ 3. Name the failing test, assertion, contract, or thread.
50
+ 4. Classify cause: current diff, stale test, environment/setup, or existing flake.
51
+ 5. Reproduce locally when practical with the narrowest command.
52
+ 6. Patch forward when the current diff caused it.
53
+ 7. For stale/flaky failures, collect proof before one rerun or residual-risk note.
54
+ 8. Re-run the affected command or exact-head gate after every fix.
55
+
56
+ For bot comments, classify each thread as valid, stale, invalid, or out of policy before assigning severity.
57
+
58
+ ## Final checklist
59
+
60
+ Before reporting delegated work as done, verify the relevant subset:
61
+
62
+ - final diff contains only intended files
63
+ - focused validation covers changed behavior
64
+ - substantial or risky changes have fresh-review evidence
65
+ - accepted findings are fixed and revalidated
66
+ - publication authority exists before push, comment, close, merge, deploy, or release
67
+ - external checks are exact-head when used as evidence
68
+ - handoff is durable before cleanup
69
+ - residual risks, skipped validation, and blocked decisions are explicit
70
+
71
+ ## Public/private boundary
72
+
73
+ For issue/PR backlogs, releases, merge queues, contributor credit, or repo-specific policy, load the matching user/project skill when available. Keep those rules out of this public package until intentionally released.