pi-subagents 0.48.0 → 0.50.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +82 -19
- package/agents/oracle.md +7 -5
- package/agents/researcher.md +1 -1
- package/agents/reviewer.md +2 -2
- package/agents/scout.md +1 -1
- package/agents/worker.md +1 -1
- package/docs/agents.md +3 -0
- package/docs/configuration.md +75 -8
- package/docs/extension-api.md +36 -0
- package/docs/missions.md +5 -3
- package/docs/observability.md +42 -2
- package/docs/tool-reference.md +19 -2
- package/docs/workflows.md +2 -2
- package/package.json +1 -1
- package/skills/pi-subagents/references/constraints-and-recipes.md +3 -2
- package/skills/pi-subagents/references/execution-controls.md +5 -4
- package/skills/pi-subagents/references/management-authoring-rpc.md +2 -2
- package/skills/pi-subagents/references/prompting-and-roles.md +31 -15
- package/src/agents/agent-serializer.ts +2 -0
- package/src/agents/agents.ts +37 -12
- package/src/api/external-runs.ts +174 -84
- package/src/api/preflight.ts +13 -7
- package/src/extension/config.ts +53 -0
- package/src/extension/index.ts +62 -9
- package/src/extension/public-execution.ts +32 -4
- package/src/extension/rpc.ts +5 -1
- package/src/extension/schemas.ts +10 -8
- package/src/extension/tool-description.ts +13 -9
- package/src/inspectors/herdr/actions.ts +13 -8
- package/src/inspectors/herdr/inspector-runner.ts +16 -3
- package/src/inspectors/herdr/project-panes.ts +2 -6
- package/src/inspectors/herdr/shell-command.ts +16 -0
- package/src/intercom/intercom-bridge.ts +5 -4
- package/src/intercom/native-supervisor-channel.ts +19 -42
- package/src/missions/goal-driver.ts +3 -1
- package/src/missions/store.ts +8 -3
- package/src/runs/background/active-async-capacity.ts +82 -25
- package/src/runs/background/active-run-index.ts +71 -1
- package/src/runs/background/async-execution.ts +73 -43
- package/src/runs/background/async-job-tracker.ts +5 -0
- package/src/runs/background/async-resume.ts +14 -6
- package/src/runs/background/async-status-snapshot.ts +277 -0
- package/src/runs/background/async-status.ts +8 -3
- package/src/runs/background/chain-root-attachment.ts +2 -2
- package/src/runs/background/completion-replay.ts +11 -1
- package/src/runs/background/fleet-view.ts +21 -6
- package/src/runs/background/result-files.ts +437 -0
- package/src/runs/background/result-watcher.ts +188 -41
- package/src/runs/background/resume-guidance.ts +27 -7
- package/src/runs/background/retained-children.ts +75 -18
- package/src/runs/background/run-id-resolver.ts +30 -24
- package/src/runs/background/run-status.ts +101 -4
- package/src/runs/background/scheduled-runs.ts +54 -28
- package/src/runs/background/stale-run-reconciler.ts +27 -13
- package/src/runs/background/subagent-runner.ts +298 -33
- package/src/runs/background/subagent-wait.ts +2 -0
- package/src/runs/background/wait-completions.ts +5 -2
- package/src/runs/foreground/async-dismiss-action.ts +2 -1
- package/src/runs/foreground/chain-execution.ts +16 -0
- package/src/runs/foreground/execution.ts +219 -15
- package/src/runs/foreground/prompt-audit.ts +4 -3
- package/src/runs/foreground/subagent-executor.ts +328 -43
- package/src/runs/shared/completion-guard.ts +107 -1
- package/src/runs/shared/external-cli-runner.ts +4 -0
- package/src/runs/shared/llm-intent-arbiter.ts +39 -23
- package/src/runs/shared/model-fallback.ts +16 -2
- package/src/runs/shared/nested-events.ts +66 -62
- package/src/runs/shared/orca-progress-tabs.ts +375 -0
- package/src/runs/shared/parallel-utils.ts +2 -0
- package/src/runs/shared/subagent-control.ts +15 -0
- package/src/runs/shared/subagent-prompt-runtime.ts +1 -9
- package/src/runs/shared/subagent-startup-retry.ts +12 -0
- package/src/runs/shared/tool-timeout.ts +93 -0
- package/src/shared/agent-stream-options.ts +5 -0
- package/src/shared/artifacts.ts +2 -6
- package/src/shared/display-text.ts +50 -0
- package/src/shared/node-executable.ts +21 -0
- package/src/shared/types.ts +51 -5
- package/src/slash/slash-commands.ts +34 -25
- package/src/slash/slash-live-state.ts +3 -0
- package/src/tui/fleet-status.ts +160 -45
- package/src/tui/fleet-transcript.ts +1 -48
- package/src/tui/fleet.ts +128 -16
- package/src/tui/render.ts +122 -44
- package/src/watchdog/permission-arbiter.ts +2 -1
- package/src/watchdog/review.ts +4 -3
- package/src/workflows/chat-progress.ts +10 -2
- package/src/workflows/scripted-workflow.ts +272 -76
package/docs/workflows.md
CHANGED
|
@@ -90,12 +90,12 @@ Configure the worktree base directory and setup hook in [configuration.md](confi
|
|
|
90
90
|
|
|
91
91
|
## Supervisor coordination (child asks parent)
|
|
92
92
|
|
|
93
|
-
Child agents can talk back to the parent Pi session without installing `pi-intercom`. `pi-subagents` provides the child-facing `contact_supervisor` tool and the parent-facing `subagent_supervisor({ action: "reply" })` path natively.
|
|
93
|
+
Child agents can talk back to the parent Pi session without installing `pi-intercom`. `pi-subagents` provides the child-facing `contact_supervisor` tool and the parent-facing `subagent_supervisor({ action: "reply" })` path natively. Generic `intercom` remains available only when an explicitly loaded external provider supplies it.
|
|
94
94
|
|
|
95
95
|
Use it for work where the child might need a decision instead of guessing:
|
|
96
96
|
|
|
97
97
|
```text
|
|
98
|
-
Run this implementation in the background. If the worker gets blocked or needs a product decision, have it ask me through
|
|
98
|
+
Run this implementation in the background. If the worker gets blocked or needs a product decision, have it ask me through the supervisor channel.
|
|
99
99
|
```
|
|
100
100
|
|
|
101
101
|
```text
|
package/package.json
CHANGED
|
@@ -61,7 +61,7 @@ Give subagents specific tasks rather than vague mandates.
|
|
|
61
61
|
|
|
62
62
|
### Escalate decisions upward
|
|
63
63
|
|
|
64
|
-
If a subagent encounters an unapproved product, architecture, scope, merge, release, credential, or authority choice, it should use `contact_supervisor` and wait for the reply instead of deciding alone. Generic `intercom` is
|
|
64
|
+
If a subagent encounters an unapproved product, architecture, scope, merge, release, credential, or authority choice, it should use `contact_supervisor` and wait for the reply instead of deciding alone. Generic `intercom` is external or provider-supplied only. Use it only when external bridge instructions provide an explicit safe target. External checks, receipts, and review bots provide evidence only; they do not grant authority.
|
|
65
65
|
|
|
66
66
|
### Intervene only on clear control signals
|
|
67
67
|
|
|
@@ -211,7 +211,8 @@ Use distinct keys, prompts, and output paths. Do not launch parallel writers int
|
|
|
211
211
|
**"Unknown agent"**
|
|
212
212
|
```typescript
|
|
213
213
|
subagent({ action: "list" })
|
|
214
|
-
// Check available agents
|
|
214
|
+
// Check available agents, then confirm scope/precedence. Saved chains are not a
|
|
215
|
+
// public execution surface; author orchestration with workflowScript.
|
|
215
216
|
```
|
|
216
217
|
|
|
217
218
|
**Setup, discovery, or intercom confusion**
|
|
@@ -72,7 +72,7 @@ Scripts run in a timed worker with only `runs.run`, `runs.all`, `runs.status`, `
|
|
|
72
72
|
|
|
73
73
|
For one host-run verification command, pass `gate: "npm test"` on a `runs.run`/`runs.all` item (or at the top level as a workflow default). It is shorthand for verified acceptance with that single command: the runtime executes it on the host, records the result as evidence, and memoizes it per tracked workspace state and effective environment. `gate` cannot be combined with `acceptance`; use explicit `acceptance.verify` for multiple commands or custom criteria.
|
|
74
74
|
|
|
75
|
-
Completed workflow children from this parent session stay addressable as retained children. `subagent({ action: "children.list" })` lists up to the last 10 with run ids
|
|
75
|
+
Completed workflow children from this parent session stay addressable as retained children. `subagent({ action: "children.list" })` lists up to the last 10 with run ids and reports each row as `resumable` or `not resumable` with a reason. Resume only rows reported `resumable`. For a retained-child challenge, use `resume` instead of `steer` when the child is complete. If no retained child is resumable, launch a same-role fallback challenge and label it as fallback. A later workflow continues a resumable child with `runs.run(key, { resume: "<run-id>", task: "follow-up" })`. Inside `workflowScript`, awaiting that call waits for the revived child to finish and returns its completed output and new `runId`; top-level `{ action: "resume" }` remains detached. A follow-up loop can render each task with `await prompts.render(...)`. Assign each returned child result back to the loop variable because every resume can return a new retained `runId`; always resume the latest returned id. `resume` and `agent` are mutually exclusive, the revived child keeps its stored agent/model/tool contract, and `gate` is rejected on retained resume items.
|
|
76
76
|
|
|
77
77
|
### Async/background
|
|
78
78
|
|
|
@@ -287,7 +287,7 @@ Use `mission.update` while work runs to record decisions, artifacts, labels, sum
|
|
|
287
287
|
- **Use `missionId` for follow-up work.** Attach later work to an existing objective with `missionId`; attachment re-marks the mission active. `missionId` and `mission` are mutually exclusive. Explicit attachment fails before launch if the mission is missing, while automatic missions degrade to `details.missionWarning` without blocking the run.
|
|
288
288
|
- **Keep `state` small.** Mission `state` is JSON coordination across workflows on the same mission. Keys use the same format as run keys, values must be JSON, and the whole state file is capped at 256 KiB. Each `set` merges one key under a file lock. Put large content in artifact files and store paths in state. In goal missions, write `state.set("nextReadyAction", "...")` so the next idle-turn notice names the exact ready step.
|
|
289
289
|
- **Use artifacts and receipts as evidence.** Mission-backed launches already record run artifacts such as async `status.json`, `events.jsonl`, child output paths, and handoff manifests. Add `mission.update` artifacts only for extra durable outputs such as `patch`, `review`, or `note` files. Add receipts for external outcomes: `pull_request`, `ci`, `deployment`, or `release`; each receipt needs an absolute URL. Receipts are evidence, not authority to merge, deploy, or release.
|
|
290
|
-
- **
|
|
290
|
+
- **Resolve decisions explicitly.** `mission.update` `decisions` can only add open decisions; `mission.update` itself cannot resolve one — use the `mission.resolve-decision` action (decision `id` plus a non-empty `summary`) to settle and close it. In a goal mission, an unresolved decision becomes the fallback next ready action in each notice. Use decisions sparingly there; record them for escalation and audit, steer goal continuation through `state.nextReadyAction`, and close the mission when the question is settled.
|
|
291
291
|
- **Close missions when done.** `mission.close` takes `missionStatus` `completed`, `failed`, or `cancelled` plus a concise `summary`, and ends any goal loop. Goal notices go only to the owning session and stop silently at `budget-exhausted` without closing or claiming success, so close explicitly. Terminal missions are pruned beyond configured retention, so store durable outputs as artifacts, receipts, and summary before closing.
|
|
292
292
|
|
|
293
293
|
After compaction, restart, or confusing history, recover from durable state first: `mission.list` in the project, `mission.list` with `missionScope: "global"` for the user-local cross-project pointer index, then `mission.show` for the relevant mission. `mission.show` refreshes linked async status when available and returns warnings instead of hiding the mission if a linked status file is temporarily unreadable. Use the linked run ids with normal `status`, `steer`, `resume`, or `stop` actions. Project mission JSON remains authoritative over chat history.
|
|
@@ -305,6 +305,7 @@ subagent({ action: "mission.create", mission: { title: "Ship auth refresh", obje
|
|
|
305
305
|
subagent({ workflowScript: `return runs.run("main", { agent: "worker", task: "Implement the approved plan" })`, missionId: "<mission-id>" })
|
|
306
306
|
subagent({ workflowScript: `return runs.run("main", { agent: "scout", task: "Quickly answer whether this file exists" })`, mission: false })
|
|
307
307
|
subagent({ action: "mission.list", missionScope: "global" })
|
|
308
|
+
subagent({ action: "mission.resolve-decision", missionId: "<mission-id>", id: "<decision-id>", summary: "Settled: ship the v2 API; no schema freeze needed." })
|
|
308
309
|
subagent({ action: "project.open", cwd: "/path/to/other-repo", message: "Own this mission for the project and report back with receipts." })
|
|
309
310
|
subagent({ action: "project.status", cwd: "/path/to/other-repo" })
|
|
310
311
|
subagent({ action: "project.close", cwd: "/path/to/other-repo" })
|
|
@@ -381,7 +382,7 @@ Use `oracle` as a smart-friend escalation when the parent needs help with trajec
|
|
|
381
382
|
|
|
382
383
|
This is separate from optional external completion delivery. Set `intercomBridge.resultDelivery: true` only when an external listener consumes and acknowledges `subagent:result-intercom` grouped results. It does not deliver results by itself, and it does not change native supervisor asks or progress updates.
|
|
383
384
|
|
|
384
|
-
|
|
385
|
+
Generic `intercom` is external or provider-supplied only. Native supervisor coordination injects `contact_supervisor`, not generic `intercom`. Use generic `intercom` only when external bridge instructions provide an explicit safe target. Do not invent a target. Prefer the tool from the injected bridge instructions.
|
|
385
386
|
|
|
386
387
|
Use `contact_supervisor` with `reason: "need_decision"` when:
|
|
387
388
|
- a subagent is blocked on a decision
|
|
@@ -423,6 +424,6 @@ Or inspects unresolved asks first:
|
|
|
423
424
|
subagent_supervisor({ action: "pending" })
|
|
424
425
|
```
|
|
425
426
|
|
|
426
|
-
|
|
427
|
+
Native supervisor coordination does not expose generic `intercom` as a fallback. Use `subagent_supervisor` for parent replies.
|
|
427
428
|
|
|
428
429
|
If intercom messages do not show up, run `subagent({ action: "doctor" })` or `/subagents-doctor`.
|
|
@@ -18,7 +18,7 @@ subagent({ action: "list" })
|
|
|
18
18
|
subagent({ action: "children.list" })
|
|
19
19
|
```
|
|
20
20
|
|
|
21
|
-
Lists up to the last 10
|
|
21
|
+
Lists up to the last 10 retained workflow children from this parent session with explicit `resumable` or `not resumable` rows. Resume only rows reported `resumable`. Send a simple follow-up or implementation challenge with `subagent({ action: "resume", id: "<run-id>", message: "..." })`. Continue one inside a workflow with `runs.run(key, { resume: "<run-id>", task: "follow-up" })`; the revived child keeps its stored agent, model, and tool contract. If no resumable child is listed, start a same-role fallback challenge and label it as fallback. `steer` with `mode: "follow_up"` only queues text for the next `resume` when the child has already completed.
|
|
22
22
|
|
|
23
23
|
### Refinement overlays
|
|
24
24
|
|
|
@@ -158,4 +158,4 @@ Additional user prompt templates can delegate into `pi-subagents` through the na
|
|
|
158
158
|
|
|
159
159
|
Other Pi extensions can call `pi-subagents` through the in-process event bus. The RPC channels are `subagents:rpc:v1:ready`, `subagents:rpc:v1:request`, and per-request replies at `subagents:rpc:v1:reply:<requestId>`. Envelopes use `{ version: 1, requestId, method, params }`, and replies use `{ version: 1, requestId, success, data | error }`. `ping` advertises the exact process-local async completion event as `events.asyncComplete` for RPC-spawn consumers.
|
|
160
160
|
|
|
161
|
-
Methods: `ping`, `status`, `spawn`, `steer`, `interrupt`, `resume`, and `stop`. `ping` capability metadata advertises optional projections: `capabilities.fleetStatus: { version: 1 }` adds bounded current-session `data.fleet` records (opaque reconciliation `key`, resolved `agent`, optional `role`, `model`, `effort`, caller-facing `goal`, `startedAt`, split `{ input, output, total }` tokens, plus `totalActive`/`omitted` overflow counts) to successful `status` replies; `capabilities.launchResolvedExtensions` advertises parent-resolved opaque launch-extension identifiers in status details; `capabilities.runtimeAcknowledgedExtensions` advertises the best-effort child-runtime acknowledgement projection fed by cooperating extensions emitting `subagent:acknowledge-extension`. Foreground `details.results[]` rows carry a stable numeric `index`; correlate children by `(runId, index)` rather than row position. Consumers should read status/result artifacts and RPC projections instead of scraping terminal output and must ignore unknown fields. `spawn` requires `workflowScript`, is async-only, and rejects management actions, `async: false`, or `clarify: true`; it reuses the normal executor, so discovery, validation, session attribution, configured spawn caps, child-safety depth, artifacts, and async status are shared with the `subagent` tool. `status`, acknowledged async `steer`, and `interrupt` map to the normal control actions. RPC steer disables pause-and-revive recovery and advertises `capabilities.nonRecoveringSteer`, preserving the caller's authority over the exact spawned child. `resume` requires a target plus non-empty message and delegates to the package-owned revival path; it may set a caller-owned `file-only` output path but cannot override the persisted child model, tools, budgets, session ownership, or exclusive session lease. `stop` targets running async runs through the existing timeout control channel. `pi.events` is process-local, so separate Pi processes and child subagents need lifecycle artifact files or `pi-intercom` instead.
|
|
161
|
+
Methods: `ping`, `status`, `spawn`, `steer`, `interrupt`, `resume`, and `stop`. `ping` capability metadata advertises optional projections: `capabilities.fleetStatus: { version: 1 }` adds bounded current-session `data.fleet` records (opaque reconciliation `key`, resolved `agent`, optional `role`, `model`, `effort`, caller-facing `goal`, `startedAt`, split `{ input, output, total }` tokens, plus `totalActive`/`omitted` overflow counts) to successful `status` replies; `capabilities.launchResolvedExtensions` advertises parent-resolved opaque launch-extension identifiers in status details; `capabilities.runtimeAcknowledgedExtensions` advertises the best-effort child-runtime acknowledgement projection fed by cooperating extensions emitting `subagent:acknowledge-extension`. Foreground `details.results[]` rows carry a stable numeric `index`; correlate children by `(runId, index)` rather than row position. Consumers should read status/result artifacts and RPC projections instead of scraping terminal output and must ignore unknown fields. `spawn` requires `workflowScript`, is async-only, and rejects management actions, `async: false`, or `clarify: true`; it reuses the normal executor, so discovery, validation, session attribution, configured spawn caps, child-safety depth, artifacts, and async status are shared with the `subagent` tool. `status`, acknowledged async `steer`, and `interrupt` map to the normal control actions. RPC steer disables pause-and-revive recovery and advertises `capabilities.nonRecoveringSteer`, preserving the caller's authority over the exact spawned child. `resume` requires a target plus non-empty message and delegates to the package-owned revival path; it may set a caller-owned `file-only` output path but cannot override the persisted child model, tools, budgets, session ownership, or exclusive session lease. For retained-child workflows, list children first and resume only rows reported `resumable`; otherwise start a same-role fallback challenge and label it as fallback. `stop` targets running async runs through the existing timeout control channel. `pi.events` is process-local, so separate Pi processes and child subagents need lifecycle artifact files or `pi-intercom` instead.
|
|
@@ -101,13 +101,13 @@ Use this after implementation when the user wants cleanup review or when a final
|
|
|
101
101
|
|
|
102
102
|
### Staged fix orchestration technique
|
|
103
103
|
|
|
104
|
-
Use this when a broad diff has known reviewer findings across several items and the user wants the parent to “orchestrate subagents like a boss.” Keep the active worktree safe with a three-stage
|
|
104
|
+
Use this when a broad diff has known reviewer findings across several items and the user wants the parent to “orchestrate subagents like a boss.” Keep the active worktree safe with a three-stage `workflowScript`:
|
|
105
105
|
|
|
106
106
|
1. A parallel read-only planning fanout, one reviewer per issue cluster. Each child inspects the real diff and returns exact files, line refs, proposed fixes, and focused validation. They must not edit.
|
|
107
|
-
2. One writer worker. It receives the reviewer summaries
|
|
107
|
+
2. One writer worker. It receives the reviewer summaries as the awaited planning results (or their durable output paths) interpolated into its task, plus the parent’s accepted scope, stop rules, and verification contract. It is the only child allowed to edit the active worktree.
|
|
108
108
|
3. A parallel read-only validation fanout. Validators inspect the worker diff from fresh context with distinct angles, report pass/fail, remaining blockers, and missing verification.
|
|
109
109
|
|
|
110
|
-
Prefer `async: true`, `context: "fresh"` for reviewers/validators, `outputMode: "file-only"` for large summaries, and per-stage output names that will not collide.
|
|
110
|
+
Prefer `async: true`, `context: "fresh"` for reviewers/validators, `outputMode: "file-only"` for large summaries, and per-stage output names that will not collide. Use stable `runs` keys plus `phase` and `label` on each launch item to make async status readable, and hold each awaited result in an ordinary JavaScript variable when a later step needs that specific result — interpolate it (or the durable output path you declared for that child) into the later task text instead of passing a whole aggregate blob. Use this pattern instead of launching several writer workers into a dirty worktree. Include non-blocking suggestions in the writer prompt only when they are small, safe, and do not expand product scope; otherwise record them as deferred.
|
|
111
111
|
|
|
112
112
|
When one child returns a structured target list, use ordinary JavaScript to validate/filter it and map bounded entries into `runs.all`; do not use the removed chain fanout DSL.
|
|
113
113
|
|
|
@@ -117,18 +117,34 @@ Example shape:
|
|
|
117
117
|
subagent({
|
|
118
118
|
async: true,
|
|
119
119
|
context: "fresh",
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
{ agent: "reviewer", phase: "Planning", label: "
|
|
124
|
-
{ agent: "reviewer", phase: "Planning", label: "
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
120
|
+
workflowScript: `
|
|
121
|
+
// Stage 1: parallel read-only planning fanout (stable keys, one per issue cluster)
|
|
122
|
+
const plans = await runs.all([
|
|
123
|
+
{ key: "deploy-plan", agent: "reviewer", phase: "Planning", label: "Deploy docs", task: "Plan fixes for deploy docs/workflow. Inspect the current diff. Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "plans/deploy.md", outputMode: "file-only" },
|
|
124
|
+
{ key: "scheduler-plan", agent: "reviewer", phase: "Planning", label: "Scheduler contract", task: "Plan fixes for scheduler contract. Inspect the current diff. Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "plans/scheduler.md", outputMode: "file-only" },
|
|
125
|
+
{ key: "sandbox-plan", agent: "reviewer", phase: "Planning", label: "Sandbox/security", task: "Plan fixes for sandbox/security. Inspect the current diff. Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "plans/sandbox.md", outputMode: "file-only" }
|
|
126
|
+
]);
|
|
127
|
+
|
|
128
|
+
// Stage 2: single writer — the only child allowed to edit the active worktree.
|
|
129
|
+
// Under outputMode "file-only" the awaited .output is the saved-output
|
|
130
|
+
// reference, so pass the durable paths declared above to the writer.
|
|
131
|
+
const worker = await runs.run("apply-fixes", {
|
|
132
|
+
agent: "worker",
|
|
133
|
+
phase: "Implementation",
|
|
134
|
+
label: "Apply accepted fixes",
|
|
135
|
+
task: "Apply only the accepted fixes from these planning summaries. You are the sole writer for the active worktree. Run focused validation and report changed files, commands, failures, and remaining issues.\\n\\nDeploy plan: plans/deploy.md\\n\\nScheduler plan: plans/scheduler.md\\n\\nSandbox plan: plans/sandbox.md",
|
|
136
|
+
output: "worker/fixes.md",
|
|
137
|
+
outputMode: "file-only"
|
|
138
|
+
});
|
|
139
|
+
|
|
140
|
+
// Stage 3: parallel read-only validation fanout
|
|
141
|
+
const validations = await runs.all([
|
|
142
|
+
{ key: "validate-deploy-scheduler", agent: "reviewer", phase: "Validation", label: "Deploy/scheduler validation", task: "Validate the post-worker diff for deploy and scheduler fixes. Start from the worker result: " + worker.output + " (also worker/fixes.md). Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/deploy-scheduler.md", outputMode: "file-only" },
|
|
143
|
+
{ key: "validate-sandbox", agent: "reviewer", phase: "Validation", label: "Sandbox validation", task: "Validate the post-worker diff for sandbox/security fixes. Start from the worker result: " + worker.output + " (also worker/fixes.md). Do not modify project/source files; returning findings via the configured output artifact is allowed.", output: "validation/sandbox.md", outputMode: "file-only" }
|
|
144
|
+
]);
|
|
145
|
+
|
|
146
|
+
return { worker: worker.output, validations: validations.map(v => v.output) };
|
|
147
|
+
`
|
|
132
148
|
})
|
|
133
149
|
```
|
|
134
150
|
|
|
@@ -18,6 +18,7 @@ export const KNOWN_FIELDS = new Set([
|
|
|
18
18
|
"defaultContext",
|
|
19
19
|
"async",
|
|
20
20
|
"timeoutMs",
|
|
21
|
+
"toolTimeoutMs",
|
|
21
22
|
"turnBudget",
|
|
22
23
|
"acceptance",
|
|
23
24
|
"acceptanceRole",
|
|
@@ -86,6 +87,7 @@ export function serializeAgent(config: AgentConfig, options: SerializeAgentOptio
|
|
|
86
87
|
}
|
|
87
88
|
if (config.defaultAsync !== undefined || preserve("async")) lines.push(`async: ${config.defaultAsync === undefined ? "" : config.defaultAsync ? "true" : "false"}`);
|
|
88
89
|
if (config.defaultTimeoutMs !== undefined || preserve("timeoutMs")) lines.push(`timeoutMs: ${config.defaultTimeoutMs ?? ""}`);
|
|
90
|
+
if (config.defaultToolTimeoutMs !== undefined || preserve("toolTimeoutMs")) lines.push(`toolTimeoutMs: ${config.defaultToolTimeoutMs ?? ""}`);
|
|
89
91
|
if (config.defaultTurnBudget || preserve("turnBudget")) lines.push(`turnBudget: ${config.defaultTurnBudget ? JSON.stringify(config.defaultTurnBudget) : ""}`);
|
|
90
92
|
if (config.defaultAcceptance !== undefined || preserve("acceptance")) {
|
|
91
93
|
lines.push(`acceptance: ${config.defaultAcceptance === undefined
|
package/src/agents/agents.ts
CHANGED
|
@@ -92,7 +92,7 @@ interface BuiltinAgentOverrideConfig {
|
|
|
92
92
|
disabled?: boolean;
|
|
93
93
|
systemPrompt?: string;
|
|
94
94
|
skills?: string[] | false;
|
|
95
|
-
tools?: string[] | false;
|
|
95
|
+
tools?: string[] | false | "inherit";
|
|
96
96
|
extensions?: string[] | false;
|
|
97
97
|
subagentOnlyExtensions?: string[] | false;
|
|
98
98
|
completionGuard?: boolean;
|
|
@@ -130,6 +130,7 @@ export interface AgentConfig {
|
|
|
130
130
|
defaultContext?: AgentDefaultContext;
|
|
131
131
|
defaultAsync?: boolean;
|
|
132
132
|
defaultTimeoutMs?: number;
|
|
133
|
+
defaultToolTimeoutMs?: number;
|
|
133
134
|
defaultTurnBudget?: TurnBudgetConfig;
|
|
134
135
|
defaultAcceptance?: AcceptanceInput;
|
|
135
136
|
acceptanceRole?: AcceptanceRole;
|
|
@@ -602,7 +603,7 @@ function cloneOverrideValue(override: BuiltinAgentOverrideConfig): BuiltinAgentO
|
|
|
602
603
|
...(override.disabled !== undefined ? { disabled: override.disabled } : {}),
|
|
603
604
|
...(override.systemPrompt !== undefined ? { systemPrompt: override.systemPrompt } : {}),
|
|
604
605
|
...(override.skills !== undefined ? { skills: override.skills === false ? false : [...override.skills] } : {}),
|
|
605
|
-
...(override.tools !== undefined ? { tools: override.tools
|
|
606
|
+
...(override.tools !== undefined ? { tools: Array.isArray(override.tools) ? [...override.tools] : override.tools } : {}),
|
|
606
607
|
...(override.extensions !== undefined ? { extensions: override.extensions === false ? false : [...override.extensions] } : {}),
|
|
607
608
|
...(override.subagentOnlyExtensions !== undefined ? { subagentOnlyExtensions: override.subagentOnlyExtensions === false ? false : [...override.subagentOnlyExtensions] } : {}),
|
|
608
609
|
...(override.completionGuard !== undefined ? { completionGuard: override.completionGuard } : {}),
|
|
@@ -738,6 +739,17 @@ function parseOverrideStringArrayOrFalse(
|
|
|
738
739
|
return items;
|
|
739
740
|
}
|
|
740
741
|
|
|
742
|
+
function parseToolsOverride(
|
|
743
|
+
value: unknown,
|
|
744
|
+
meta: { filePath: string; name: string },
|
|
745
|
+
): BuiltinAgentOverrideConfig["tools"] | undefined {
|
|
746
|
+
if (typeof value === "string" && value.trim() === "inherit") return "inherit";
|
|
747
|
+
if (value === undefined || value === false || Array.isArray(value)) {
|
|
748
|
+
return parseOverrideStringArrayOrFalse(value, { ...meta, field: "tools" });
|
|
749
|
+
}
|
|
750
|
+
throw new Error(`Builtin override '${meta.name}' in '${meta.filePath}' has invalid 'tools'; expected an array of strings, "inherit", or false.`);
|
|
751
|
+
}
|
|
752
|
+
|
|
741
753
|
function parseBuiltinOverrideEntry(
|
|
742
754
|
name: string,
|
|
743
755
|
value: unknown,
|
|
@@ -845,7 +857,7 @@ function parseBuiltinOverrideEntry(
|
|
|
845
857
|
const skills = parseOverrideStringArrayOrFalse(input.skills, { filePath, name, field: "skills" });
|
|
846
858
|
if (skills !== undefined) override.skills = skills;
|
|
847
859
|
|
|
848
|
-
const tools =
|
|
860
|
+
const tools = parseToolsOverride(input.tools, { filePath, name });
|
|
849
861
|
if (tools !== undefined) override.tools = tools;
|
|
850
862
|
|
|
851
863
|
const extensions = parseOverrideStringArrayOrFalse(input.extensions, { filePath, name, field: "extensions" });
|
|
@@ -1004,6 +1016,17 @@ function applySubagentDefaults(
|
|
|
1004
1016
|
);
|
|
1005
1017
|
}
|
|
1006
1018
|
|
|
1019
|
+
function applyToolsOverride(target: AgentConfig, toolsOverride: string[] | false | "inherit"): void {
|
|
1020
|
+
if (toolsOverride === "inherit") {
|
|
1021
|
+
delete target.tools;
|
|
1022
|
+
delete target.mcpDirectTools;
|
|
1023
|
+
return;
|
|
1024
|
+
}
|
|
1025
|
+
const { tools, mcpDirectTools } = splitToolList(toolsOverride === false ? [] : toolsOverride);
|
|
1026
|
+
if (tools === undefined) delete target.tools; else target.tools = tools;
|
|
1027
|
+
if (mcpDirectTools === undefined) delete target.mcpDirectTools; else target.mcpDirectTools = mcpDirectTools;
|
|
1028
|
+
}
|
|
1029
|
+
|
|
1007
1030
|
function applyBuiltinOverride(
|
|
1008
1031
|
agent: AgentConfig,
|
|
1009
1032
|
override: BuiltinAgentOverrideConfig,
|
|
@@ -1029,11 +1052,7 @@ function applyBuiltinOverride(
|
|
|
1029
1052
|
if (override.disabled !== undefined) next.disabled = override.disabled;
|
|
1030
1053
|
if (override.systemPrompt !== undefined) next.systemPrompt = override.systemPrompt;
|
|
1031
1054
|
if (override.skills !== undefined) { if (override.skills === false) delete next.skills; else next.skills = [...override.skills]; }
|
|
1032
|
-
if (override.tools !== undefined)
|
|
1033
|
-
const { tools, mcpDirectTools } = splitToolList(override.tools === false ? [] : override.tools);
|
|
1034
|
-
if (tools === undefined) delete next.tools; else next.tools = tools;
|
|
1035
|
-
if (mcpDirectTools === undefined) delete next.mcpDirectTools; else next.mcpDirectTools = mcpDirectTools;
|
|
1036
|
-
}
|
|
1055
|
+
if (override.tools !== undefined) applyToolsOverride(next, override.tools);
|
|
1037
1056
|
if (override.extensions !== undefined) { if (override.extensions === false) delete next.extensions; else next.extensions = [...override.extensions]; }
|
|
1038
1057
|
if (override.subagentOnlyExtensions !== undefined) { if (override.subagentOnlyExtensions === false) delete next.subagentOnlyExtensions; else next.subagentOnlyExtensions = [...override.subagentOnlyExtensions]; }
|
|
1039
1058
|
if (override.completionGuard !== undefined) next.completionGuard = override.completionGuard;
|
|
@@ -1175,10 +1194,7 @@ function applyCustomAgentOverride(
|
|
|
1175
1194
|
fill("skills", ["skill", "skills"], override.skills === false ? undefined : [...override.skills]);
|
|
1176
1195
|
}
|
|
1177
1196
|
if (override.tools !== undefined && !agentHasFrontmatterField(agent, "tools")) {
|
|
1178
|
-
|
|
1179
|
-
const target = mutable();
|
|
1180
|
-
if (tools === undefined) delete target.tools; else target.tools = tools;
|
|
1181
|
-
if (mcpDirectTools === undefined) delete target.mcpDirectTools; else target.mcpDirectTools = mcpDirectTools;
|
|
1197
|
+
applyToolsOverride(mutable(), override.tools);
|
|
1182
1198
|
anyFilled = true;
|
|
1183
1199
|
}
|
|
1184
1200
|
if (override.extensions !== undefined) {
|
|
@@ -1567,6 +1583,14 @@ function loadAgentsFromDir(dir: string, source: AgentSource): AgentConfig[] {
|
|
|
1567
1583
|
}
|
|
1568
1584
|
defaultTimeoutMs = parsed;
|
|
1569
1585
|
}
|
|
1586
|
+
let defaultToolTimeoutMs: number | undefined;
|
|
1587
|
+
if (frontmatter.toolTimeoutMs !== undefined) {
|
|
1588
|
+
const parsed = Number(frontmatter.toolTimeoutMs);
|
|
1589
|
+
if (!Number.isInteger(parsed) || parsed <= 0 || parsed > 2_147_483_647) {
|
|
1590
|
+
throw new Error(`Agent '${localName}' has invalid toolTimeoutMs frontmatter; expected a positive integer no larger than 2147483647.`);
|
|
1591
|
+
}
|
|
1592
|
+
defaultToolTimeoutMs = parsed;
|
|
1593
|
+
}
|
|
1570
1594
|
let defaultTurnBudget: TurnBudgetConfig | undefined;
|
|
1571
1595
|
if (frontmatter.turnBudget !== undefined && frontmatter.turnBudget.trim()) {
|
|
1572
1596
|
const parsed = JSON.parse(frontmatter.turnBudget) as unknown;
|
|
@@ -1633,6 +1657,7 @@ function loadAgentsFromDir(dir: string, source: AgentSource): AgentConfig[] {
|
|
|
1633
1657
|
...(defaultContext !== undefined ? { defaultContext } : {}),
|
|
1634
1658
|
...(defaultAsync !== undefined ? { defaultAsync } : {}),
|
|
1635
1659
|
...(defaultTimeoutMs !== undefined ? { defaultTimeoutMs } : {}),
|
|
1660
|
+
...(defaultToolTimeoutMs !== undefined ? { defaultToolTimeoutMs } : {}),
|
|
1636
1661
|
...(defaultTurnBudget !== undefined ? { defaultTurnBudget } : {}),
|
|
1637
1662
|
...(defaultAcceptance !== undefined ? { defaultAcceptance } : {}),
|
|
1638
1663
|
...(acceptanceRole !== undefined ? { acceptanceRole } : {}),
|