pi-subagents 0.54.0 → 0.55.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +35 -1
- package/agents/reviewer.md +12 -2
- package/docs/agents.md +4 -4
- package/docs/configuration.md +4 -4
- package/docs/extension-api.md +14 -4
- package/docs/models.md +24 -3
- package/docs/observability.md +14 -3
- package/docs/tool-reference.md +21 -6
- package/docs/workflows.md +7 -1
- package/package.json +1 -1
- package/prompts/council.md +9 -0
- package/prompts/parallel-review.md +5 -1
- package/prompts/review-loop.md +9 -5
- package/skills/council-mode/SKILL.md +33 -10
- package/skills/pi-subagents/references/constraints-and-recipes.md +16 -1
- package/skills/pi-subagents/references/execution-controls.md +11 -4
- package/skills/pi-subagents/references/multi-lane-orchestration.md +3 -3
- package/skills/pi-subagents/references/prompting-and-roles.md +8 -8
- package/src/agents/agent-management.ts +30 -9
- package/src/agents/agent-serializer.ts +2 -2
- package/src/agents/agents.ts +146 -41
- package/src/api/external-job-provider.ts +10 -1
- package/src/api/preflight.ts +26 -17
- package/src/api/project-panes.ts +2 -0
- package/src/extension/index.ts +20 -1
- package/src/extension/public-execution.ts +12 -9
- package/src/extension/rpc.ts +77 -3
- package/src/extension/schemas.ts +2 -1
- package/src/extension/tool-description.ts +9 -6
- package/src/inspectors/herdr/focus.ts +55 -0
- package/src/inspectors/herdr/project-panes.ts +228 -44
- package/src/integrations/herdr-status.ts +26 -4
- package/src/runs/background/async-execution.ts +103 -9
- package/src/runs/background/async-job-tracker.ts +33 -14
- package/src/runs/background/async-resume.ts +20 -2
- package/src/runs/background/async-status.ts +10 -0
- package/src/runs/background/control-channel.ts +98 -10
- package/src/runs/background/notify.ts +62 -1
- package/src/runs/background/run-status.ts +15 -1
- package/src/runs/background/subagent-runner.ts +226 -37
- package/src/runs/foreground/async-stop-action.ts +23 -2
- package/src/runs/foreground/execution.ts +81 -7
- package/src/runs/foreground/subagent-executor.ts +348 -40
- package/src/runs/foreground/workflow-detach-reconcile.ts +3 -0
- package/src/runs/shared/acceptance.ts +1 -0
- package/src/runs/shared/child-identity.ts +36 -0
- package/src/runs/shared/completion-guard.ts +50 -1
- package/src/runs/shared/external-job-bridge.ts +53 -37
- package/src/runs/shared/external-job-runner.ts +126 -23
- package/src/runs/shared/model-fallback.ts +10 -0
- package/src/runs/shared/orca-progress-tabs.ts +71 -9
- package/src/runs/shared/parallel-utils.ts +12 -0
- package/src/runs/shared/pi-args.ts +9 -12
- package/src/runs/shared/subagent-prompt-runtime.ts +3 -1
- package/src/runs/shared/tool-availability.ts +1 -3
- package/src/shared/launch-contract.ts +6 -5
- package/src/shared/thinking-ceiling.ts +52 -0
- package/src/shared/types.ts +59 -2
- package/src/tui/fleet-status.ts +66 -13
- package/src/tui/render.ts +5 -4
- package/src/workflows/scripted-workflow.ts +107 -10
|
@@ -21,7 +21,13 @@ This file is a detailed reference loaded from `skills/pi-subagents/SKILL.md`.
|
|
|
21
21
|
become second decision-makers.
|
|
22
22
|
- **Respect the fixed authority policy.** `authorityPolicy` is a small `auto` / `confirm` / `forbid` map for supported operational actions. Worktree discard, destructive cleanup, and spawn-budget grants default to confirmation; stop, steer, and schedule creation remain automatic. Use `worktree.discard` with the durable `handoffPath`; confirm-required actions refuse safely without an interactive UI and retained paths include manual Git recovery commands.
|
|
23
23
|
|
|
24
|
-
Runtime config can change orchestration behavior. `intercomBridge.resultDelivery: false` disables only external acknowledged grouped-result delivery when native parent notifications own completion; supervisor asks/progress stay active, and enabled transport failures are still reported. `asyncByDefault` and `forceTopLevelAsync` affect whether launches detach; `waitTool` can make direct `subagent_wait()` calls return immediately while headless auto-drain remains active, and its effective value is propagated to child runtimes; `globalConcurrencyLimit` bounds concurrent fanout, while a positive `maxSubagentSpawnsPerSession` optionally caps cumulative launches (`0` or unset is unlimited). Status and doctor report the budget; static work preflights declared capacity; only the settled root interactive parent can use `grant-spawn-budget` after native confirmation, with total grants bounded by the original cap. Compaction does not reset usage or grants; `singleRunOutputBaseDir` and `worktreeBaseDir` route outputs and worktrees; `completionBatch` groups async notifications. `artifactDir` is `
|
|
24
|
+
Runtime config can change orchestration behavior. `intercomBridge.resultDelivery: false` disables only external acknowledged grouped-result delivery when native parent notifications own completion; supervisor asks/progress stay active, and enabled transport failures are still reported. `asyncByDefault` and `forceTopLevelAsync` affect whether launches detach; `waitTool` can make direct `subagent_wait()` calls return immediately while headless auto-drain remains active, and its effective value is propagated to child runtimes; `globalConcurrencyLimit` bounds concurrent fanout, while a positive `maxSubagentSpawnsPerSession` optionally caps cumulative launches (`0` or unset is unlimited). Status and doctor report the budget; static work preflights declared capacity; only the settled root interactive parent can use `grant-spawn-budget` after native confirmation, with total grants bounded by the original cap. Compaction does not reset usage or grants; `singleRunOutputBaseDir` and `worktreeBaseDir` route outputs and worktrees; `completionBatch` groups async notifications. `artifactDir` is `session` (default), `project`, or `temp` and chooses where subagent artifacts are stored. Set `asyncWidget: false` to hide the above-editor background-run widget when a companion footer or dashboard owns that space (fleet inspector remains available). Per-run `artifacts: false` disables artifact capture for that launch. Async status and result artifacts include `lifecycleArtifactVersion` and fields such as `workflowGraph`, `steps`, `results`, `totalTokens`, `totalCost`, `turnCount`, `toolCount`, and nested `children`. Child protocol failures expose a structured `protocolError`; `protocol_output_limit` means a child emitted a JSONL line above the 16 MiB live-parser cap. Prefer these artifacts and `status` views over scraping terminal output.
|
|
25
|
+
|
|
26
|
+
### Keep report artifacts out of the repository root
|
|
27
|
+
|
|
28
|
+
Treat lane reports, review notes, council pass reports, and gate logs as scratch unless the user explicitly asks to keep them. Prefer `output: false` and the aggregate workflow result for short reports. When a later step needs a file, use the runtime-managed output artifact by setting a stable child key plus a relative `output` path such as `plans/deploy.md`; relative child outputs are saved under the run artifact directory, not the project root. Do not put `reports/...`, `*-report.json`, or similar repo-root paths in child task text.
|
|
29
|
+
|
|
30
|
+
For durable evidence, copy only the final summary to session memory, a PR body/comment, a mission artifact, or a user-approved docs path outside the repo. After the PR, issue, or gate reaches a terminal state, delete or move scratch reports from the active worktree before reporting completion. Keep a project `.gitignore` entry for ad-hoc report patterns only as a safety net; it is not the cleanup mechanism.
|
|
25
31
|
|
|
26
32
|
## Best Practices
|
|
27
33
|
|
|
@@ -62,6 +68,15 @@ user explicitly requests forked context.
|
|
|
62
68
|
Give subagents specific tasks rather than vague mandates.
|
|
63
69
|
`Review auth.ts for null-check gaps` works better than `Review everything`.
|
|
64
70
|
|
|
71
|
+
Before fanout, assign each child a lightweight task profile in the parent prompt:
|
|
72
|
+
work kind, required input, expected output, acceptance check, and context mode.
|
|
73
|
+
Keep the profile prose-only; do not invent runtime fields. Use coarse kinds such
|
|
74
|
+
as `code-write`, `code-read`, `transform`, `summarize`, and `search` only to
|
|
75
|
+
shape the task and choose an existing agent/model setting. If a child task is not
|
|
76
|
+
standalone enough for fresh context, add the missing facts to the prompt, switch
|
|
77
|
+
to forked context, or ask the user. Do not launch vague tasks and rely on
|
|
78
|
+
supervisor round-trips to recover missing context.
|
|
79
|
+
|
|
65
80
|
### Escalate decisions upward
|
|
66
81
|
|
|
67
82
|
If a subagent encounters an unapproved product, architecture, scope, merge, release, credential, or authority choice, it should use `contact_supervisor` and wait for the reply instead of deciding alone. Generic `intercom` is external or provider-supplied only. Use it only when external bridge instructions provide an explicit safe target. External checks, receipts, and review bots provide evidence only; they do not grant authority.
|
|
@@ -76,7 +76,7 @@ subagent({
|
|
|
76
76
|
})
|
|
77
77
|
```
|
|
78
78
|
|
|
79
|
-
Scripts run in a timed worker with only `runs.run`, `runs.all`, `runs.status`, `runs.ref/refs`, `emit`, captured `console`, and standard JavaScript. Pass explicit task text to `runs.run`. Mission-attached workflows also get `await state.get(key)` and `await state.set(key, value)` for durable JSON state shared across workflows on the same mission; `mission: false` workflows have no `state` global. Stable keys are required. Child launches follow ordinary single-agent execution controls. Give each child a distinct decision and output path when reports must outlive the workflow, then consume the aggregate workflow result before opening individual reports.
|
|
79
|
+
Scripts run in a timed worker with only `runs.run`, `runs.all`, `runs.status`, `runs.ref/refs`, `emit`, captured `console`, and standard JavaScript. Pass explicit task text to `runs.run`. Mission-attached workflows also get `await state.get(key)` and `await state.set(key, value)` for durable JSON state shared across workflows on the same mission; `mission: false` workflows have no `state` global. Stable keys are required. Child launches follow ordinary single-agent execution controls. Give each child a distinct decision and output path when reports must outlive the workflow, then consume the aggregate workflow result before opening individual reports. Do not ask children to write `reports/...` or other repo-root scratch paths in task text.
|
|
80
80
|
|
|
81
81
|
If `runs.all` is missing in a running session, reload or update `pi-subagents` before retrying. The current runtime supports `runs.all`; `await Promise.all([runs.run(...)])` is also supported for advanced dynamic fanout.
|
|
82
82
|
|
|
@@ -116,7 +116,7 @@ subagent({
|
|
|
116
116
|
})
|
|
117
117
|
```
|
|
118
118
|
|
|
119
|
-
File-only output mode works for workflowScript child launches. Use
|
|
119
|
+
File-only output mode works for workflowScript child launches. Use relative child output paths for scratch reports so the runtime stores them under the run artifact directory and age-based cleanup can remove them. Use absolute paths only for user-approved durable destinations, such as session memory, a docs folder outside the repo, or a known handoff path. For cross-codebase waves, include the repo slug or lane key in each output path so reports from different repositories cannot collide.
|
|
120
120
|
|
|
121
121
|
For review fanout where the parent continues a local audit:
|
|
122
122
|
|
|
@@ -129,7 +129,7 @@ const run = subagent({
|
|
|
129
129
|
// Continue local inspection, then later call status with the returned id.
|
|
130
130
|
```
|
|
131
131
|
|
|
132
|
-
While children run, the persistent FleetView and the collapsed foreground tool-result card show live per-child detail: resolved model and thinking level, `[fresh]`/`[fork]` context, tool/token/elapsed counters, and current activity. The collapsed running card also prints the configured expand-key hint ("Press … for live detail"); expanding it shows nested children, recent tools, and recent output. Model badges appear once the child's model resolves at first attempt start. `/subagents-fleet` opens the live fleet inspector, which also has per-child controls (`s` steer, `D` stop with confirmation). When optional Herdr 0.7.5+ is available, `H` opens a raw inspector dashboard for the selected active async child; this mirrors artifacts rather than attaching to the headless child.
|
|
132
|
+
While children run, the persistent FleetView and the collapsed foreground tool-result card show live per-child detail: resolved model and thinking level, `[fresh]`/`[fork]` context, tool/token/elapsed counters, and current activity. The collapsed running card also prints the configured expand-key hint ("Press … for live detail"); expanding it shows nested children, recent tools, and recent output. Model badges appear once the child's model resolves at first attempt start. `/subagents-fleet` opens the live fleet inspector, which also has per-child controls (`s` steer, `D` stop with confirmation). When optional Herdr 0.7.5+ is available, `H` opens a raw inspector dashboard for the selected active async child; this mirrors artifacts rather than attaching to the headless child. When optional Orca progress tabs are enabled, Pi creates one passive Orca observer tab for the top-level subagent call. Parallel and chain children share that tab and write child section headers into the mirrored log instead of opening one tab per child. Pi remains authoritative for lifecycle, status, control, artifacts, and results; the Orca tab is display-only. Pi also writes passive display metadata under `.pi/subagents/views/orca/` when possible, so other surfaces can discover the observer without treating it as an owned child run. Use visual panes for confusing or long-running active async work when the human wants a dedicated surface or FleetView is insufficient, not for routine headless runs.
|
|
133
133
|
|
|
134
134
|
Inspect async runs with `subagent({ action: "status", id: "..." })` or `subagent({ action: "status" })` for active runs. Use `subagent({ action: "status", view: "fleet" })` when supervising several active foreground/background runs and `subagent({ action: "status", id: "...", view: "transcript", index: 0 })` when you need the latest child output without digging through artifacts. If a delegated fanout child launches nested runs, the parent status view shows them as a tree and you can target a nested run directly with its nested id.
|
|
135
135
|
|
|
@@ -137,8 +137,15 @@ Stop a current-session top-level async run with `stop` (or `/subagents-stop`). S
|
|
|
137
137
|
|
|
138
138
|
```typescript
|
|
139
139
|
subagent({ action: "stop", id: "run-id" })
|
|
140
|
+
subagent({ action: "stop", id: "run-id", childId: "child-id" })
|
|
140
141
|
```
|
|
141
142
|
|
|
143
|
+
Use `childId` only for active async/workflow runs whose status snapshot shows a
|
|
144
|
+
matching child. Observer hosts can listen for the advertised
|
|
145
|
+
`subagent:child-status` event to update child-stop UI quickly, but status
|
|
146
|
+
snapshots remain the source of truth after reconnects or duplicate event
|
|
147
|
+
delivery.
|
|
148
|
+
|
|
142
149
|
Use `steer` for top-level live async guidance and `resume` after a delegated run pauses or finishes. Routed nested runs retain their existing non-destructive live follow-up path:
|
|
143
150
|
|
|
144
151
|
```typescript
|
|
@@ -314,7 +321,7 @@ Routing rule:
|
|
|
314
321
|
- Several projects with independent work: one async `workflowScript` whose child keys include repo slugs and whose child calls set explicit `cwd`; keep publication and merge decisions serial per repo.
|
|
315
322
|
- Different project, substantial or long-running work: open a project-owned Herdr pane rooted there when a separate visible project session is useful, then give that project Pi session a narrow mission/result contract. Do not model it as ordinary child nesting, and do not expect existing headless runs to move into the pane.
|
|
316
323
|
|
|
317
|
-
Project panes run a separate Pi session from the target directory. Subagents launched inside that pane use that project's config, agents, skills, files, git state, and mission records. The pane binding lives under `<projectRoot>/.pi/subagents/project-panes/herdr.json`. For ordinary headless delegation to another repo, prefer explicit `cwd` first; reserve project panes for visible or persistent project ownership.
|
|
324
|
+
Project panes run a separate Pi session from the target directory. Subagents launched inside that pane use that project's config, agents, skills, files, git state, and mission records. The pane binding lives under `<projectRoot>/.pi/subagents/project-panes/herdr.json`. When Pi runs inside Herdr, the owning pane reports compact active-work status and title suffixes, and the parent inline status counts opened project panes. Use Herdr itself or `project.status` / `project.close` for pane-level follow-up. For ordinary headless delegation to another repo, prefer explicit `cwd` first; reserve project panes for visible or persistent project ownership.
|
|
318
325
|
|
|
319
326
|
```typescript
|
|
320
327
|
subagent({ action: "mission.create", mission: { title: "Ship auth refresh", objective: "Implement and validate refresh handling" } })
|
|
@@ -18,9 +18,9 @@ Use one writer per repo/cwd or worktree. Mutation lanes need distinct isolation
|
|
|
18
18
|
|
|
19
19
|
For Pi extension repositories, keep lane worktrees outside auto-discovered extension directories such as `~/.pi/agent/extensions`. A stale extension worktree there can auto-load duplicate tools and shortcuts. Remove or move it only after its handoff is durable, the worktree is clean, and no run owns it.
|
|
20
20
|
|
|
21
|
-
Partition fanout by repository, source seam, decision, or review angle. Each run needs a stable key, lane-specific task, and
|
|
21
|
+
Partition fanout by repository, source seam, decision, or review angle. Each run needs a stable key, lane-specific task, and a managed output path when a file is needed. Do not launch prompts that differ only by item name or broad file glob.
|
|
22
22
|
|
|
23
|
-
Use one async `workflowScript` for a coordinated wave. Use `runs.all` for independent lanes and `runs.run` for dependent lane stages. Give cross-repository runs explicit `cwd` values and lane-qualified outputs. Use `outputMode: "file-only"` when a report must survive the run or feed a later stage.
|
|
23
|
+
Use one async `workflowScript` for a coordinated wave. Use `runs.all` for independent lanes and `runs.run` for dependent lane stages. Give cross-repository runs explicit `cwd` values and lane-qualified outputs. Use `outputMode: "file-only"` when a report must survive the run or feed a later stage. Keep scratch outputs relative so they live under subagent artifacts; use absolute paths only for durable memory, approved docs paths, or final handoff files.
|
|
24
24
|
|
|
25
25
|
## Keep independent work moving
|
|
26
26
|
|
|
@@ -32,7 +32,7 @@ After a writer produces a candidate, run the required fresh-context, read-only r
|
|
|
32
32
|
|
|
33
33
|
## Handoff, cleanup, and recovery
|
|
34
34
|
|
|
35
|
-
Use stable lane-qualified paths for reports and review output. A handoff states the lane status, repository and worktree, changed files, validation, open decisions, next action, and artifact or receipt paths.
|
|
35
|
+
Use stable lane-qualified artifact paths for reports and review output. A handoff states the lane status, repository and worktree, changed files, validation, open decisions, next action, and artifact or receipt paths. Copy only the final evidence to memory, a mission record, or a PR/comment, then remove scratch files from the active worktree before closing the lane.
|
|
36
36
|
|
|
37
37
|
Keep a worktree until its handoff is durable, no run owns it, and no later gate needs it. Clean up only inside the recorded authority boundary. If a run stops or needs attention, preserve its worktree and artifacts, record the last known state and recovery owner, then resume that run or create one replacement lane from the handoff. Do not start another writer while worktree ownership is uncertain.
|
|
38
38
|
|
|
@@ -54,11 +54,11 @@ The prompt templates in `prompts/` encode workflows the parent agent can run on
|
|
|
54
54
|
|
|
55
55
|
Use Council Mode when the user asks to convene advisors, debate a material decision, cross-examine recommendations, or critique and improve a plan with several model perspectives. This includes requests such as “run a council on this architecture,” “have Sol, Fable, and Kimi critique this plan,” or “get multiple oracles to debate the tradeoffs.” Read `../council-mode/SKILL.md` and follow its bounded parent-supervised protocol instead of launching ad hoc parallel oracle calls.
|
|
56
56
|
|
|
57
|
-
Council advisors are read-only. User or project `council-*` profiles can pin models such as GPT 5.6 Sol, Fable, or Kimi and define any persistent stance in the profile body. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
|
|
57
|
+
Council advisors are read-only. User or project `council-*` profiles can pin models such as GPT 5.6 Sol, Fable, or Kimi and define any persistent stance in the profile body. Package advisors such as Surf's `gpt-pro` can join the roster only when the `surf-cli` Pi extension is installed and its `surf-oracle` provider is registered; treat them as external runners, omit child `async` for attached results, and do not pass `outputSchema` to them. The council question and scope provide the decision frame; do not invent per-advisor role labels. The parent collects independent reports, optionally sends curated cross-exam packets, and writes the final memo. Do not treat the council as agent-to-agent chat, implementation authority, or a writer swarm.
|
|
58
58
|
|
|
59
59
|
### Parallel review technique
|
|
60
60
|
|
|
61
|
-
Use this when the user wants adversarial review of a diff, plan, issue, file, or implemented work. Launch fresh-context `reviewer` agents with distinct angles generated from the actual target. Common angles are correctness/regressions, tests/validation, and simplicity/maintainability; adapt for TypeScript, UI, security, docs, or large structural changes. Reviewers should inspect files and diffs directly, return concise evidence-backed findings with file/line references, and avoid edits unless the user explicitly asks for a writer pass. The parent synthesizes fixes worth doing now, optional improvements, and feedback to ignore/defer before applying anything.
|
|
61
|
+
Use this when the user wants adversarial review of a diff, plan, issue, file, or implemented work. Launch fresh-context `reviewer` agents with distinct angles generated from the actual target. Common angles are correctness/regressions, tests/validation, and simplicity/maintainability; adapt for TypeScript, UI, security, docs, or large structural changes. Reviewers should inspect files and diffs directly, return concise evidence-backed findings with file/line references, and avoid edits unless the user explicitly asks for a writer pass. Filter on evidence, not severity: report only concrete current issues caused or made reachable by the target diff, with source proof, a test or repro, or a contract contradiction. Label findings P0/P1/P2 and end with `Merge verdict: BLOCK`, `Merge verdict: OK`, or `Merge verdict: OK with notes`. Use `blockers only` only for final pre-merge re-checks after P1/P2 findings are already captured, or for explicit emergency hotfix lanes where non-blocking findings are intentionally deferred. For targeted follow-up, ask only whether the named finding was resolved, whether the fix introduced a new defect in the fix blast radius, and whether prior P1/P2 notes still stand. For bot or PR-comment triage, classify each comment as VALID, STALE, INVALID, or OUT-OF-POLICY against current HEAD, then assign P0/P1/P2 only to VALID comments. The parent synthesizes fixes worth doing now, optional improvements, and feedback to ignore/defer before applying anything.
|
|
62
62
|
|
|
63
63
|
### Proactive skill-specialist technique
|
|
64
64
|
|
|
@@ -88,7 +88,7 @@ subagent({
|
|
|
88
88
|
|
|
89
89
|
### Review-loop technique
|
|
90
90
|
|
|
91
|
-
Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no blockers or fixes worth doing now, remaining feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
|
|
91
|
+
Use this when the user wants implementation or current diff review to continue until reviewers stop finding fixes worth doing now. Keep the loop in the parent session: one async `worker` implements or fixes, fresh-context `reviewer` agents inspect the actual repo and diff, the parent synthesizes accepted fixes, and one async forked `worker` applies them. The parent can express the sequence up front as an async/background `workflowScript` when the workflow is known, or continue with explicit follow-up workflowScript runs after each async completion. For an initial workflow, pass `async: true` so the main chat is unblocked. Treat an async implementation worker handoff as an intermediate state, not final completion, unless the user explicitly asked for worker-only work, review-only output, or to stop after implementation. Stop when reviewers find no P0 blockers or P1 fixes worth doing now, remaining P2 feedback is optional or deferred, an unapproved product/scope/architecture decision appears, or the max review-round cap is reached. Default to 3 review rounds unless the user sets a different cap. Do not loop for optional polish, and do not let children launch subagents or decide the loop outcome.
|
|
92
92
|
|
|
93
93
|
As a conservative orchestration policy, do not pass `turnBudget` or a hard `toolBudget` to an implementation worker, fix worker, reviewer with edit authority, or other mutation-capable child. The default tool budget blocks read/search tools rather than mutation tools, but count limits still do not measure delivery safety. Use a narrow task plus an outer elapsed deadline with enough margin, then request a checkpoint after the current tool returns. The checkpoint should report changed files, build/test state, remaining work, and commit or PR state. An elapsed timeout is not a mutation-safe boundary and must not be used as the checkpoint trigger.
|
|
94
94
|
|
|
@@ -134,20 +134,20 @@ subagent({
|
|
|
134
134
|
|
|
135
135
|
// Stage 2: single writer — the only child allowed to edit the active worktree.
|
|
136
136
|
// Under outputMode "file-only" the awaited .output is the saved-output
|
|
137
|
-
// reference, so pass
|
|
137
|
+
// reference, so pass those managed artifact references to the writer.
|
|
138
138
|
const worker = await runs.run("apply-fixes", {
|
|
139
139
|
agent: "worker",
|
|
140
140
|
phase: "Implementation",
|
|
141
141
|
label: "Apply accepted fixes",
|
|
142
|
-
task: "Apply only the accepted fixes from these planning summaries. You are the sole writer for the active worktree. Run focused validation and report changed files, commands, failures, and remaining issues.\\n\\nDeploy plan: plans
|
|
142
|
+
task: "Apply only the accepted fixes from these planning summaries. You are the sole writer for the active worktree. Run focused validation and report changed files, commands, failures, and remaining issues.\\n\\nDeploy plan: " + plans[0].output + "\\n\\nScheduler plan: " + plans[1].output + "\\n\\nSandbox plan: " + plans[2].output,
|
|
143
143
|
output: "worker/fixes.md",
|
|
144
144
|
outputMode: "file-only"
|
|
145
145
|
});
|
|
146
146
|
|
|
147
147
|
// Stage 3: parallel read-only validation fanout
|
|
148
148
|
const validations = await runs.all([
|
|
149
|
-
{ key: "validate-deploy-scheduler", agent: "reviewer", phase: "Validation", label: "Deploy/scheduler validation", task: "Validate the post-worker diff for deploy and scheduler fixes. Start from the worker result: " + worker.output + "
|
|
150
|
-
{ key: "validate-sandbox", agent: "reviewer", phase: "Validation", label: "Sandbox validation", task: "Validate the post-worker diff for sandbox/security fixes. Start from the worker result: " + worker.output + "
|
|
149
|
+
{ key: "validate-deploy-scheduler", agent: "reviewer", phase: "Validation", label: "Deploy/scheduler validation", task: "Validate the post-worker diff for deploy and scheduler fixes. Start from the worker result: " + worker.output + ". Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/deploy-scheduler.md", outputMode: "file-only" },
|
|
150
|
+
{ key: "validate-sandbox", agent: "reviewer", phase: "Validation", label: "Sandbox validation", task: "Validate the post-worker diff for sandbox/security fixes. Start from the worker result: " + worker.output + ". Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/sandbox.md", outputMode: "file-only" }
|
|
151
151
|
]);
|
|
152
152
|
|
|
153
153
|
return { worker: worker.output, validations: validations.map(v => v.output) };
|
|
@@ -208,7 +208,7 @@ A strong subagent prompt usually includes:
|
|
|
208
208
|
- **Success criteria**: what must be true before the child can finish.
|
|
209
209
|
- **Hard constraints**: true invariants only, such as no edits for review-only tasks, one writer thread, child must not run subagents unless it is an explicitly assigned `tools: subagent` fanout child, or escalation for unapproved decisions.
|
|
210
210
|
- **Validation**: targeted checks to run, or the next-best check when validation is impossible.
|
|
211
|
-
- **Output**: the expected summary shape, artifact path, or finding format. Use repo-qualified
|
|
211
|
+
- **Output**: the expected summary shape, artifact path, or finding format. Use managed artifact paths for scratch reports; reserve repo-qualified absolute paths for durable handoffs that the user approved.
|
|
212
212
|
- **Stop rules**: when to ask via `intercom` or `contact_supervisor`, when to stop after enough evidence, and when not to keep searching.
|
|
213
213
|
|
|
214
214
|
Give each role useful discovery anchors. Name source roots, filenames, symbols, types, methods, and paths for scouts. Give workers context files, plans, task paths, and named source seams before asking them to search. Give reviewers changed files, contracts, and any exhaustive-verification target. Tell oracle whether current source behavior, product/policy documents, plans, or inherited decisions are the evidence that matters.
|
|
@@ -58,18 +58,23 @@ function parseCsv(value: string): string[] {
|
|
|
58
58
|
return [...new Set(value.split(",").map((v) => v.trim()).filter(Boolean))];
|
|
59
59
|
}
|
|
60
60
|
|
|
61
|
-
|
|
61
|
+
type ConfigObjectResult =
|
|
62
|
+
| { status: "ok"; value: Record<string, unknown> }
|
|
63
|
+
| { status: "missing" }
|
|
64
|
+
| { status: "error"; message: string };
|
|
65
|
+
|
|
66
|
+
function configObject(config: unknown): ConfigObjectResult {
|
|
62
67
|
let val = config;
|
|
63
68
|
if (typeof val === "string") {
|
|
64
69
|
try {
|
|
65
70
|
val = JSON.parse(val);
|
|
66
71
|
} catch (error) {
|
|
67
72
|
const message = error instanceof Error ? error.message : String(error);
|
|
68
|
-
return { error: `config must be valid JSON: ${message}` };
|
|
73
|
+
return { status: "error", message: `config must be valid JSON: ${message}` };
|
|
69
74
|
}
|
|
70
75
|
}
|
|
71
|
-
if (!val || typeof val !== "object" || Array.isArray(val)) return {};
|
|
72
|
-
return { value: val as Record<string, unknown> };
|
|
76
|
+
if (!val || typeof val !== "object" || Array.isArray(val)) return { status: "missing" };
|
|
77
|
+
return { status: "ok", value: val as Record<string, unknown> };
|
|
73
78
|
}
|
|
74
79
|
|
|
75
80
|
function hasKey(obj: Record<string, unknown>, key: string): boolean {
|
|
@@ -177,7 +182,8 @@ function isMutableSource(source: AgentSource): source is ManagementScope {
|
|
|
177
182
|
function modelWarning(ctx: ManagementContext, model: string | undefined): string | undefined {
|
|
178
183
|
if (!model) return undefined;
|
|
179
184
|
const found = ctx.modelRegistry.getAvailable().some((m) => `${m.provider}/${m.id}` === model || m.id === model);
|
|
180
|
-
|
|
185
|
+
if (found) return undefined;
|
|
186
|
+
return `Warning: model '${model}' is not in the current model registry. Run subagent({ action: "models" }) to list valid provider/id selectors, then use the exact provider/id form (bare ids resolve only when unique).`;
|
|
181
187
|
}
|
|
182
188
|
|
|
183
189
|
function fallbackModelsWarning(ctx: ManagementContext, fallbackModels: string[] | undefined): string | undefined {
|
|
@@ -215,7 +221,9 @@ export function editableAgentConfig(agent: AgentConfig): AgentConfig {
|
|
|
215
221
|
const base = agent.override?.base;
|
|
216
222
|
const {
|
|
217
223
|
override: _override,
|
|
224
|
+
output: _output,
|
|
218
225
|
outputMode: _outputMode,
|
|
226
|
+
defaultReads: _defaultReads,
|
|
219
227
|
model: _model,
|
|
220
228
|
fallbackModels: _fallbackModels,
|
|
221
229
|
thinking: _thinking,
|
|
@@ -243,7 +251,9 @@ export function editableAgentConfig(agent: AgentConfig): AgentConfig {
|
|
|
243
251
|
|
|
244
252
|
return withDeclaredExtensionPaths({
|
|
245
253
|
...editable,
|
|
254
|
+
...(base.output !== undefined ? { output: base.output } : {}),
|
|
246
255
|
...(base.outputMode !== undefined ? { outputMode: base.outputMode } : {}),
|
|
256
|
+
...(base.defaultReads !== undefined ? { defaultReads: [...base.defaultReads] } : {}),
|
|
247
257
|
...(base.model !== undefined ? { model: base.model } : {}),
|
|
248
258
|
...(base.fallbackModels !== undefined ? { fallbackModels: [...base.fallbackModels] } : {}),
|
|
249
259
|
...(base.thinking !== undefined ? { thinking: base.thinking } : {}),
|
|
@@ -832,6 +842,17 @@ function handleModels(params: ManagementParams, ctx: ManagementContext): AgentTo
|
|
|
832
842
|
lines.push("");
|
|
833
843
|
}
|
|
834
844
|
|
|
845
|
+
const availableFullIds = availableModels.map((m) => m.fullId).sort();
|
|
846
|
+
if (availableFullIds.length > 0) {
|
|
847
|
+
lines.push("Available models in this session's registry (copy an exact provider/id when passing model):");
|
|
848
|
+
lines.push("");
|
|
849
|
+
const shown = availableFullIds.slice(0, 80);
|
|
850
|
+
for (const fullId of shown) lines.push(` ${fullId}`);
|
|
851
|
+
if (availableFullIds.length > shown.length) lines.push(` ... and ${availableFullIds.length - shown.length} more`);
|
|
852
|
+
lines.push("");
|
|
853
|
+
lines.push("Use an exact provider/id from this list when you pass model; bare ids resolve only when unique in the registry.");
|
|
854
|
+
}
|
|
855
|
+
|
|
835
856
|
return result(lines.join("\n"));
|
|
836
857
|
}
|
|
837
858
|
|
|
@@ -857,9 +878,9 @@ function handleGet(params: ManagementParams, ctx: ManagementContext): AgentToolR
|
|
|
857
878
|
|
|
858
879
|
export function handleCreate(params: ManagementParams, ctx: ManagementContext): AgentToolResult<Details> {
|
|
859
880
|
const parsedConfig = configObject(params.config);
|
|
860
|
-
if (parsedConfig.error) return result(parsedConfig.
|
|
881
|
+
if (parsedConfig.status === "error") return result(parsedConfig.message, true);
|
|
882
|
+
if (parsedConfig.status === "missing") return result("config required for create.", true);
|
|
861
883
|
const cfg = parsedConfig.value;
|
|
862
|
-
if (!cfg) return result("config required for create.", true);
|
|
863
884
|
if (typeof cfg.name !== "string" || !cfg.name.trim()) return result("config.name is required and must be a non-empty string.", true);
|
|
864
885
|
if (typeof cfg.description !== "string" || !cfg.description.trim()) return result("config.description is required and must be a non-empty string.", true);
|
|
865
886
|
const name = sanitizeName(cfg.name);
|
|
@@ -907,9 +928,9 @@ export function handleCreate(params: ManagementParams, ctx: ManagementContext):
|
|
|
907
928
|
export function handleUpdate(params: ManagementParams, ctx: ManagementContext): AgentToolResult<Details> {
|
|
908
929
|
if (!params.agent) return result("Specify 'agent' for update.", true);
|
|
909
930
|
const parsedConfig = configObject(params.config);
|
|
910
|
-
if (parsedConfig.error) return result(parsedConfig.
|
|
931
|
+
if (parsedConfig.status === "error") return result(parsedConfig.message, true);
|
|
932
|
+
if (parsedConfig.status === "missing") return result("config required for update.", true);
|
|
911
933
|
const cfg = parsedConfig.value;
|
|
912
|
-
if (!cfg) return result("config required for update.", true);
|
|
913
934
|
if (hasKey(cfg, "steps")) return result("Durable chain definitions were removed; use workflowScript or /prompt-workflow for repeatable workflows.", true);
|
|
914
935
|
const warnings: string[] = [];
|
|
915
936
|
const scopeHint = asDisambiguationScope(params.agentScope);
|
|
@@ -113,11 +113,11 @@ export function serializeAgent(config: AgentConfig, options: SerializeAgentOptio
|
|
|
113
113
|
lines.push(`subagentOnlyExtensions: ${subagentOnlyExtensionsValue ?? ""}`);
|
|
114
114
|
}
|
|
115
115
|
|
|
116
|
-
if (config.output) lines.push(`output: ${config.output}`);
|
|
116
|
+
if (config.output || preserve("output")) lines.push(`output: ${config.output ?? ""}`);
|
|
117
117
|
if (config.outputMode || preserve("outputMode")) lines.push(`outputMode: ${config.outputMode ?? ""}`);
|
|
118
118
|
|
|
119
119
|
const readsValue = joinComma(config.defaultReads);
|
|
120
|
-
if (readsValue) lines.push(`defaultReads: ${readsValue}`);
|
|
120
|
+
if (readsValue || preserve("defaultReads")) lines.push(`defaultReads: ${readsValue ?? ""}`);
|
|
121
121
|
|
|
122
122
|
if (config.defaultProgress) lines.push("defaultProgress: true");
|
|
123
123
|
if (config.interactive) lines.push("interactive: true");
|