pi-better-harness 0.13.0 → 0.13.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/node_modules/pi-better-subagents/README.md +1 -0
- package/node_modules/pi-better-subagents/batch.mjs +58 -0
- package/node_modules/pi-better-subagents/docs/usage.md +695 -0
- package/node_modules/pi-better-subagents/index.ts +29 -12
- package/node_modules/pi-better-subagents/package.json +2 -1
- package/package.json +2 -2
|
@@ -12,6 +12,7 @@ Use `pi-better-subagents` when you want Pi to launch independent agent work with
|
|
|
12
12
|
|
|
13
13
|
## Core Features
|
|
14
14
|
- Non-blocking subagent launches, with an optional role and named-agent catalog (`docs/agent-catalog.md`, `docs/agent-catalog-lifecycle.md`).
|
|
15
|
+
- Typed foreground `subagent_spawn_batch` launch receipts for native Pi codemode (Pi 0.99.1+), preserving direct tool calls, per-run callbacks, and batch capacity/catalog guarantees. See [codemode batch launches](docs/usage.md#codemode-batch-launches).
|
|
15
16
|
- Default OS write sandboxing on macOS and Linux.
|
|
16
17
|
- Explicit tool allowlists for child sessions.
|
|
17
18
|
- Durable logs, result retrieval, and [failure observations](docs/failure-observations.md) independent of lifecycle status. A live child handles its own tool errors; the parent is woken only for actionable incidents, and children can classify handled failures with `failure_disposition`.
|
|
@@ -14,6 +14,64 @@ import { assertTimingParams } from "./timing.ts";
|
|
|
14
14
|
|
|
15
15
|
const VALID_CAPACITY_MODES = new Set(["reject", "launch-available"]);
|
|
16
16
|
|
|
17
|
+
/** Public codemode receipt; details alone are not returned to nested callers. */
|
|
18
|
+
export const batchLaunchOutputSchema = {
|
|
19
|
+
type: "object",
|
|
20
|
+
properties: {
|
|
21
|
+
status: { type: "string", enum: ["launched", "partial", "not-launched", "clarification-needed"] },
|
|
22
|
+
batchId: { type: "string" },
|
|
23
|
+
batchName: { type: "string" },
|
|
24
|
+
launched: {
|
|
25
|
+
type: "array",
|
|
26
|
+
items: {
|
|
27
|
+
type: "object",
|
|
28
|
+
properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" }, id: { type: "string" }, modelNote: { type: "string" } },
|
|
29
|
+
required: ["job", "name", "id"],
|
|
30
|
+
additionalProperties: false,
|
|
31
|
+
},
|
|
32
|
+
},
|
|
33
|
+
failed: {
|
|
34
|
+
type: "array",
|
|
35
|
+
items: {
|
|
36
|
+
type: "object",
|
|
37
|
+
properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" }, reason: { type: "string" } },
|
|
38
|
+
required: ["job", "name", "reason"],
|
|
39
|
+
additionalProperties: false,
|
|
40
|
+
},
|
|
41
|
+
},
|
|
42
|
+
skipped: {
|
|
43
|
+
type: "array",
|
|
44
|
+
items: {
|
|
45
|
+
type: "object",
|
|
46
|
+
properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" } },
|
|
47
|
+
required: ["job", "name"],
|
|
48
|
+
additionalProperties: false,
|
|
49
|
+
},
|
|
50
|
+
},
|
|
51
|
+
message: { type: "string" },
|
|
52
|
+
choices: { type: "array", items: { type: "string" } },
|
|
53
|
+
},
|
|
54
|
+
required: ["status", "launched", "failed", "skipped"],
|
|
55
|
+
additionalProperties: false,
|
|
56
|
+
};
|
|
57
|
+
|
|
58
|
+
/** Launch success is not run completion; partial receipts retain every created run. */
|
|
59
|
+
export function batchLaunchResult({ batchId, batchName, launched, skipped, failed }) {
|
|
60
|
+
const receipt = {
|
|
61
|
+
status: launched.length === 0 ? "not-launched" : failed.length || skipped.length ? "partial" : "launched",
|
|
62
|
+
batchId,
|
|
63
|
+
...(batchName !== undefined ? { batchName } : {}),
|
|
64
|
+
launched,
|
|
65
|
+
failed,
|
|
66
|
+
skipped,
|
|
67
|
+
};
|
|
68
|
+
return {
|
|
69
|
+
content: [{ type: "text", text: formatBatchLaunchResponse(receipt) }],
|
|
70
|
+
details: receipt,
|
|
71
|
+
structuredContent: receipt,
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
17
75
|
/**
|
|
18
76
|
* Overlay per-job options on top of shared options. Only defined keys are
|
|
19
77
|
* inherited; booleans keep their false values.
|
|
@@ -0,0 +1,695 @@
|
|
|
1
|
+
# pi-better-subagents
|
|
2
|
+
|
|
3
|
+
A better subagent extension for [pi](https://github.com/earendil-works/pi-coding-agent).
|
|
4
|
+
|
|
5
|
+
Not a clone of Claude Code's subagents — a rethink of what a subagent system
|
|
6
|
+
should be: **autonomous, non-blocking, and safe by default.** You delegate work
|
|
7
|
+
and keep going; each subagent runs on its own in an isolated process, confined to
|
|
8
|
+
what it needs, and reports back when it's done. No blocking waits, no
|
|
9
|
+
back-channel for it to stall on, no unbounded blast radius.
|
|
10
|
+
|
|
11
|
+
```
|
|
12
|
+
launch is the result · completion triggers fetch · the foreground never blocks
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Delegation Mode
|
|
16
|
+
|
|
17
|
+
Set `delegationMode` in this package's `config.json` to `manual`, `adaptive`, or
|
|
18
|
+
`coordinator` (default `adaptive`). This controls foreground delegation guidance,
|
|
19
|
+
not child permissions or catalog definitions. `/subagents` reports the active
|
|
20
|
+
mode; `/subagents mode manual|adaptive|coordinator` overrides it for the current
|
|
21
|
+
session without editing config. Other arguments show usage. `/reload` retains
|
|
22
|
+
this override; new sessions use config, while resuming a session restores that
|
|
23
|
+
session's override.
|
|
24
|
+
|
|
25
|
+
- **Manual:** no proactive delegation, including in plan mode. A user request or
|
|
26
|
+
explicit workflow requirement may still delegate.
|
|
27
|
+
- **Adaptive:** delegate useful, substantial independent work while continuing
|
|
28
|
+
foreground work; keep small or coupled tasks local.
|
|
29
|
+
- **Coordinator:** use `agents_catalog` to discover current roles and delegate
|
|
30
|
+
every nontrivial task owned by an available role, passing the role by its
|
|
31
|
+
short name (`role: "developer"`). The foreground owns
|
|
32
|
+
orchestration, cross-role decisions, unowned or ambiguous work, integration,
|
|
33
|
+
and final verification. Current role descriptions, including custom catalog
|
|
34
|
+
overrides, determine ownership.
|
|
35
|
+
|
|
36
|
+
All modes retain the no-polling rule and require inspection of delegated results
|
|
37
|
+
before final verification. `pi-better-plan` follows the active mode when a plan
|
|
38
|
+
is present; without this extension it uses adaptive guidance.
|
|
39
|
+
|
|
40
|
+
## Principles
|
|
41
|
+
|
|
42
|
+
- **The foreground never blocks.** Launching a subagent *is* the deliverable —
|
|
43
|
+
`subagent_spawn` starts a detached `pi -p` child and returns immediately,
|
|
44
|
+
leaving the session free for the human. When the child finishes, it sends a
|
|
45
|
+
lightweight trigger; the foreground calls `subagent_result` and presents the
|
|
46
|
+
result (as a `followUp`, never cutting into work in progress). The foreground
|
|
47
|
+
is nudged once, at completion — never on a wait/poll loop.
|
|
48
|
+
- **Subagents are autonomous; communication is one-way (parent → child).** The
|
|
49
|
+
parent front-loads everything the child needs into the spawn; the child runs to
|
|
50
|
+
completion and **returns a result**. There is no mid-task child→parent blocking
|
|
51
|
+
call for a subagent to waste wall-clock on — a child missing a piece of info
|
|
52
|
+
resolves it from what it was given, or records it unavailable and returns.
|
|
53
|
+
- **Safe by default.** Every subagent is OS-sandboxed — writes confined to its
|
|
54
|
+
working directory, reads and network open — and scoped to an explicit tool
|
|
55
|
+
allowlist. It can't corrupt the parent, escape its directory, or recurse into
|
|
56
|
+
more subagents without opt-in.
|
|
57
|
+
- **Observable.** A live status widget and on-demand queries show each run's
|
|
58
|
+
elapsed time and token/cost spend.
|
|
59
|
+
|
|
60
|
+
### Observable progress
|
|
61
|
+
|
|
62
|
+
Subagent health uses the shared 60-second quiet and 5-minute stalled defaults,
|
|
63
|
+
while preserving its stricter child-event semantics: active tool calls,
|
|
64
|
+
compaction, and model retry/error phases explain silence rather than becoming
|
|
65
|
+
stale. For a cross-extension override, set `PI_BETTER_STALL_QUIET_MS` and
|
|
66
|
+
`PI_BETTER_STALL_MS` in milliseconds. Existing subagent `config.json` health
|
|
67
|
+
thresholds continue to take precedence for subagent health.
|
|
68
|
+
|
|
69
|
+
## Tools
|
|
70
|
+
|
|
71
|
+
| Tool | Blocks? | What it does |
|
|
72
|
+
|------|---------|--------------|
|
|
73
|
+
| `subagent_spawn` | never | Launch a task in a background subagent; returns a run id at once. Params: `prompt`, `name`, `model`, `tools` (allowlist), `exclude_tools`, `sandbox`, `sandbox_dir`, `callback`, `clean`, `cwd`, `git_clone_workspace`, `approve`, `allow_nested`, and the [timing](#run-timing-deadline-ceiling-stuck) overrides `deadline_minutes`, `grace_minutes`, `max_minutes`, `stuck_minutes`. |
|
|
74
|
+
| `subagent_spawn_batch` | never | Launch several independent subagents at once. Each job becomes a normal run. Params: `batchName`, `shared` (options applied to every job), `jobs[]` (each needs `prompt`; same optional params as `subagent_spawn`), `onCapacity` (`reject` or `launch-available`). |
|
|
75
|
+
| `subagent_list` | never | Compact current-session list (default 10 rows / 1 KiB page). Params: `all` (machine-global / foreign session), `limit` (default 10, max 100), `cursor`, `max_bytes` (max 4 KiB), `status` (`running`, `completed`, `failed`, `killed`, `exited`, durable `orphaned`, `lost`). Incident counts are compact; no spend/tool histories. |
|
|
76
|
+
| `subagent_output` | never | Bounded current-session excerpt (default 1 KiB / 10 lines). Params: `id`, `lines` (`tail_lines` is a deprecated alias), `cursor` (a returned `nextCursor`, `statusCursor`, or `incidentCursor`), `max_bytes` (max 4 KiB; raw 16 KiB default, 64 KiB max), `mode` (`raw` pages retained log bytes), `include` (`["cost"]`, `["tools"]`, or both: opt-in spend and tool-count lines), `all` (foreign-session id). Missing/unreadable logs are gaps, not empty healthy output. |
|
|
77
|
+
| `subagent_result` | never | Finished-run answer for the current session (2048-byte page). Params: `id`, `cursor` (a returned `nextCursor` continues the answer; `statusCursor` / `incidentCursor` also accepted), `max_bytes` (maximum 8192 bytes), `lines` (optional line cap per answer page), `mode` (`raw`), `include` (`["cost"]`, `["tools"]`, or both), `all` (foreign-session id). Failures and exceptional lifecycle facts come before progress; no tool-name histories. TUI folding is display-only. |
|
|
78
|
+
| `subagent_stop` | never | SIGTERM a running run's process group. |
|
|
79
|
+
|
|
80
|
+
## Codemode batch launches
|
|
81
|
+
|
|
82
|
+
On Pi 0.99.1 or newer, enable native codemode alongside the existing direct
|
|
83
|
+
tools with `defaultTools: ["+codemode"]` and `codemode.mode: "on"`. SDK sessions
|
|
84
|
+
must also load Pi's `createCodemodeExtension()`. This support is for the
|
|
85
|
+
foreground coordinator; it does not enable codemode in confined children or
|
|
86
|
+
change sandbox admission.
|
|
87
|
+
|
|
88
|
+
`subagent_spawn_batch` returns the same readable text to direct callers and a
|
|
89
|
+
structured launch receipt to codemode scripts. No text parsing is needed:
|
|
90
|
+
|
|
91
|
+
```javascript
|
|
92
|
+
const receipt = await tools.subagent_spawn_batch({
|
|
93
|
+
batchName: "inspection",
|
|
94
|
+
shared: { tools: "read,bash" },
|
|
95
|
+
jobs: [
|
|
96
|
+
{ role: "explorer", prompt: "Map the request handling path." },
|
|
97
|
+
{ role: "reviewer", prompt: "Review the current changes for regressions." }
|
|
98
|
+
]
|
|
99
|
+
});
|
|
100
|
+
store("inspection-launch", receipt);
|
|
101
|
+
text(receipt);
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
| Field | Meaning |
|
|
105
|
+
|-------|---------|
|
|
106
|
+
| `status` | `launched`: every effective job launched; `partial`: some launched and some failed or were skipped; `not-launched`: none launched; `clarification-needed`: selection needs a human decision and nothing launched. These are launch outcomes, never completion outcomes. |
|
|
107
|
+
| `batchId`, `batchName` | Assigned batch identity and optional label. No batch ID is assigned for clarification. |
|
|
108
|
+
| `launched` | `{ job, name, id, modelNote? }` for each created run. |
|
|
109
|
+
| `failed` | `{ job, name, reason }` for launch failures, including later jobs not attempted after a reject-mode launch failure. |
|
|
110
|
+
| `skipped` | `{ job, name }` for jobs omitted because capacity was full in `launch-available` mode. |
|
|
111
|
+
| `message`, `choices` | Explanation and selection options on a clarification receipt. |
|
|
112
|
+
|
|
113
|
+
`job` is a 1-based position in the effective batch order, after any confirmed
|
|
114
|
+
role split. All three job lists are always present. Invalid input and whole-batch
|
|
115
|
+
capacity rejection still throw, so codemode callers must catch those errors.
|
|
116
|
+
Returned partial or no-launch receipts do not throw; inspect `status` and every
|
|
117
|
+
list instead of treating a fulfilled promise as complete success.
|
|
118
|
+
|
|
119
|
+
Await the launch call and print or retain its receipt before returning from the
|
|
120
|
+
script. Codemode `store()` writes persist only if the script succeeds; launched
|
|
121
|
+
runs are not rolled back if the script later fails, times out, or is interrupted.
|
|
122
|
+
Use the durable run registry to recover IDs when needed, and `subagent_stop` to
|
|
123
|
+
stop a run. Script cancellation is not a stop request. Collect results after
|
|
124
|
+
completion callbacks with `subagent_result`; do not poll or keep a script open
|
|
125
|
+
waiting for completion. Callbacks and capacity/catalog guarantees remain owned
|
|
126
|
+
by the existing harness runtime.
|
|
127
|
+
|
|
128
|
+
## Run timing (deadline, ceiling, stuck)
|
|
129
|
+
|
|
130
|
+
Timeout control belongs to the harness, not to the prompt or the skill that
|
|
131
|
+
launched the run. A time limit written into a prompt is enforced by nothing, and
|
|
132
|
+
a parent that only notices late tends to kill a child that was still making
|
|
133
|
+
progress. Every run gets stall detection by default. Elapsed-time limits are
|
|
134
|
+
opt-in because total runtime does not distinguish a slow agent from a stalled
|
|
135
|
+
one.
|
|
136
|
+
|
|
137
|
+
| Control | Default | What happens |
|
|
138
|
+
|---------|---------|--------------|
|
|
139
|
+
| Soft deadline (`deadline_minutes`) | Off | When configured, the child gets one steering message (delivered through Pi's steer queue after its current tool call finishes): stop starting new work, commit or save what is done, and report what is complete, what is not, and where the work is. The parent gets one wake saying so. |
|
|
140
|
+
| Grace (`grace_minutes`) | 5 | Grace starts when the steering message actually reaches the child, which is after its current tool call. While the message is still waiting behind a running tool call, the deadline stop is held, so a test run that began just before the deadline is not killed mid-run. If the child has not finished by the end of grace, the harness stops it; the ordinary completion callback reports `killed; stopped: deadline`. A child that completes inside grace is not stopped; its surfaces say `deadline: finished in grace`. |
|
|
141
|
+
| Hard ceiling (`max_minutes`) | Off | When configured, the run is stopped at once, without grace or a steer, and reported as `stopped: ceiling`. The ceiling bounds everything, including a tool call that never ends and an orphaned run (child gone, its process group still alive). |
|
|
142
|
+
| Stuck window (`stuck_minutes`) | 10 | No progress (see below) for this long wakes the parent once (`stuck`). It never stops the run. After new progress, a later stuck spell wakes again. |
|
|
143
|
+
|
|
144
|
+
`0` turns a control off. `null` or an omitted value means "inherit": on
|
|
145
|
+
`subagent_spawn` that is the configured default; on a `subagent_spawn_batch` job
|
|
146
|
+
it is the `shared` value, then the configured default. Values are minutes and
|
|
147
|
+
may be fractional. A job's own number (including `0`) wins over `shared`.
|
|
148
|
+
|
|
149
|
+
**Known limits of the deadline hold.**
|
|
150
|
+
|
|
151
|
+
- `max_minutes: 0` turns the ceiling off, and the ceiling is the only control
|
|
152
|
+
that bounds a tool call that never ends. With it off, a steer stuck behind a
|
|
153
|
+
hung tool call is never delivered, the deadline stop stays held, and the run
|
|
154
|
+
is never stopped. Opt into a ceiling when a command must have a wall-clock
|
|
155
|
+
bound.
|
|
156
|
+
- The parent reads at most the last 32 MiB of a child log after `/reload`. If
|
|
157
|
+
the log is larger and the running tool call started before that window, the
|
|
158
|
+
parent cannot see the call, so the hold is lost and grace counts from when
|
|
159
|
+
the steer was sent. The child can be stopped mid-call.
|
|
160
|
+
- The child confirms delivery by matching the steer's exact text in its
|
|
161
|
+
conversation. If another extension rewrites user input (an input-transform
|
|
162
|
+
hook), the text no longer matches, no receipt is written, and grace silently
|
|
163
|
+
counts from when the steer was sent instead of when it arrived.
|
|
164
|
+
|
|
165
|
+
**Progress**, defined simply: any successful tool call that is not an exact
|
|
166
|
+
repeat of an earlier call in the same run. "Exact repeat" means the same tool
|
|
167
|
+
name and the same arguments, with null and absent optional fields treated alike
|
|
168
|
+
(as in #336) and key order ignored. A successful file-mutating tool (`edit`,
|
|
169
|
+
`write`, `multi_edit`, `apply_patch`, `str_replace`, `write_file`, and similarly
|
|
170
|
+
named tools), a `git commit`, and a success directly after a failed call always
|
|
171
|
+
count, even when repeated. Re-reading the same file or re-running the same command with the same
|
|
172
|
+
arguments does not reset the stuck window, so a read-only research child making
|
|
173
|
+
distinct calls is never flagged, while one looping on the same read is. The
|
|
174
|
+
harness remembers up to 4096 distinct calls per run as fixed-size hashes of the
|
|
175
|
+
name and normalized arguments; past that the oldest is forgotten, so a call from
|
|
176
|
+
long ago counts again. Time the child spends waiting on a running tool call does not count toward the
|
|
177
|
+
stuck window, so a child in the middle of a 20-minute test run is waiting, not
|
|
178
|
+
stuck. A command that hangs is bounded only when the caller opts into a deadline
|
|
179
|
+
or ceiling.
|
|
180
|
+
The other stuck signal is the existing escalation: the same operation failing
|
|
181
|
+
three times wakes the parent once as an `Action required` incident
|
|
182
|
+
([failure observations](failure-observations.md)).
|
|
183
|
+
|
|
184
|
+
**Where the reason shows.** `subagent_list` rows, `subagent_output`,
|
|
185
|
+
`subagent_result`, and completion callbacks carry `stopped: deadline`,
|
|
186
|
+
`deadline: wrapping up`, `deadline: finished in grace` (completed after the
|
|
187
|
+
deadline), `deadline: passed` (ended some other way after the deadline, for
|
|
188
|
+
example it crashed or a user stopped it), `stopped: ceiling`, or `stuck`. The spawn response lists the limits in force.
|
|
189
|
+
|
|
190
|
+
**Global defaults.** Precedence is spawn parameter, then environment, then
|
|
191
|
+
`config.json`, then the built-in default:
|
|
192
|
+
|
|
193
|
+
| Setting | Environment | `config.json` |
|
|
194
|
+
|---------|-------------|---------------|
|
|
195
|
+
| Soft deadline | `PI_SUBAGENT_DEADLINE_MINUTES` | `deadlineMinutes` |
|
|
196
|
+
| Grace | `PI_SUBAGENT_GRACE_MINUTES` | `graceMinutes` |
|
|
197
|
+
| Hard ceiling | `PI_SUBAGENT_MAX_MINUTES` | `maxMinutes` |
|
|
198
|
+
| Stuck window | `PI_SUBAGENT_STUCK_MINUTES` | `stuckMinutes` |
|
|
199
|
+
|
|
200
|
+
**Durability and delivery.** The policy and its one-shot markers (steer
|
|
201
|
+
requested, deadline wake, stuck wake, stop reason) are stored in the run's
|
|
202
|
+
metadata at launch, so `/reload` keeps the deadline and never repeats a wake.
|
|
203
|
+
The parent checks timing on its 15-second supervision tick. The steer travels
|
|
204
|
+
through a request file in the run directory, read by a small harness extension
|
|
205
|
+
loaded into every child (it registers no tools and runs no commands). When the
|
|
206
|
+
message enters the child's conversation, the extension writes a receipt file
|
|
207
|
+
next to the request; grace starts at that time. The extension runs in the
|
|
208
|
+
trusted Pi process, and the child's confined tools cannot write the run
|
|
209
|
+
directory, so a child cannot forge a receipt. Without a receipt and with no
|
|
210
|
+
tool call running (for example, a child started without the extension), grace
|
|
211
|
+
runs from when the steer was requested. Wakes follow `callback:false` and session ownership like other callbacks.
|
|
212
|
+
Runs launched before this feature have no timing record and are not timed.
|
|
213
|
+
|
|
214
|
+
Output-control parameters are spelled the same across subagents and background
|
|
215
|
+
tasks: `max_bytes` and `lines`. The older spellings `maxBytes` and `tail_lines`
|
|
216
|
+
are deprecated aliases; both work, and the canonical name wins when both are
|
|
217
|
+
given.
|
|
218
|
+
|
|
219
|
+
## Non-blocking, by construction
|
|
220
|
+
|
|
221
|
+
- **Process isolation.** Each run is a `detached` + `unref`'d `pi -p` process.
|
|
222
|
+
Its context can't clog the parent, its crash can't corrupt parent state, and
|
|
223
|
+
its output is durable in a log file.
|
|
224
|
+
- **Completions trigger one batched turn that fetches durable results.** Ordinary
|
|
225
|
+
terminal callbacks accumulate for 100 ms and share one
|
|
226
|
+
`pi.sendMessage(..., { deliverAs: "followUp", triggerTurn: true })` notification
|
|
227
|
+
with background-task completions from the same Pi host. Each row contains only
|
|
228
|
+
source, id, label, terminal status, and the `subagent_result` lookup. Full
|
|
229
|
+
results and logs stay in durable tools and are never embedded in the callback.
|
|
230
|
+
Set `PI_BETTER_CALLBACK_BATCH_MS` to `0` through `5000` milliseconds to tune
|
|
231
|
+
the accumulation window; invalid values use 100 ms. A failed handoff keeps the
|
|
232
|
+
pending records retryable, and `/reload` recovers records explicitly marked
|
|
233
|
+
pending. `callback:false` sends no model message; read the durable result later
|
|
234
|
+
with `subagent_result`.
|
|
235
|
+
- **The prompt guidelines forbid polling.** The foreground agent is told, in the
|
|
236
|
+
tool guidelines, that spawning is done and it must not loop on `output`/`result`
|
|
237
|
+
or sleep to wait.
|
|
238
|
+
|
|
239
|
+
## Autonomy & safety
|
|
240
|
+
|
|
241
|
+
Every subagent is confined by default, and the confinement is **self-contained** —
|
|
242
|
+
it does not depend on any other extension being installed.
|
|
243
|
+
|
|
244
|
+
- **Task sandbox (default on, macOS and Linux).** Pi startup, authentication,
|
|
245
|
+
provider transport, and session persistence run as trusted runtime operations.
|
|
246
|
+
Admitted task tools run under `sandbox-exec` or Linux Bubblewrap. Project files
|
|
247
|
+
default to Read / write and outside files to Read, including `~/.pi`; runtime
|
|
248
|
+
control files remain protected. Commands use private scratch plus explicit
|
|
249
|
+
runtime exceptions for `/tmp` and the current user's macOS temporary and MDS
|
|
250
|
+
cache directories when Outside is Read or Read/write. Credential and control
|
|
251
|
+
denials retain precedence. See the permission details below.
|
|
252
|
+
- **Tool allowlist.** Confined children admit only verified read/write/edit/bash
|
|
253
|
+
implementations, the guarded `apply_patch`, and the trusted tools ticked in
|
|
254
|
+
`/sandbox`. Other requested tools are reported as unavailable, with the reason.
|
|
255
|
+
- **No runaway recursion.** Confined children cannot spawn nested agents.
|
|
256
|
+
`allow_nested:true` loads nested-agent support only for unconfined children.
|
|
257
|
+
|
|
258
|
+
### Write sandbox
|
|
259
|
+
|
|
260
|
+
The shared `sandbox-core` and `task-sandbox` modules enforce task operations in
|
|
261
|
+
both Main and Subagents. The defaults are Main Off and Subagents On. Subagents
|
|
262
|
+
use Project files = Write & delete and Outside project = **Write**: tasks write
|
|
263
|
+
across home and temp (tool caches such as `~/.gradle` and `~/.npm` just work), but
|
|
264
|
+
outside the workspace they can only remove or rename files in temp, hidden home
|
|
265
|
+
entries, and worktree folders (`.worktrees/`, `*-worktrees/`). Credential files
|
|
266
|
+
default to readable but not writable, so npm, git over SSH, and authenticated
|
|
267
|
+
CLIs can reuse their normal user configuration. This exposes known file-based
|
|
268
|
+
credentials to confined subagents; set Stored credentials to Off to hide them.
|
|
269
|
+
Shell startup files, `~/.pi`, `~/.claude`,
|
|
270
|
+
`~/.agents`, and harness state cannot be written, removed, or renamed. Rename-based
|
|
271
|
+
saves and git commits in a sibling repository fail under Write; set Outside
|
|
272
|
+
project to Write & delete when a task needs that. Linux uses a stricter fallback
|
|
273
|
+
where ordinary home folders are read-only. See
|
|
274
|
+
[pi-better-sandbox](../../pi-better-sandbox/README.md) and
|
|
275
|
+
[ADR 0008](../../../docs/adr/0008-write-without-delete.md). Commands and network
|
|
276
|
+
default On. A subagent's project root is its selected workspace (`cwd` / `sandbox_dir`),
|
|
277
|
+
or its disposable clone. `/sandbox off` changes Main, not Subagents.
|
|
278
|
+
|
|
279
|
+
Pi runs as the trusted runtime so it can take settings/authentication locks,
|
|
280
|
+
connect to its provider, and persist sessions. A mandatory guard loads before
|
|
281
|
+
the task can run. Its immutable launch snapshot controls shell commands and
|
|
282
|
+
kernel-confined file workers. Task access to `~/.pi` follows the outside and
|
|
283
|
+
credential-file rules; there are no writable lock exceptions. Runtime code,
|
|
284
|
+
configuration, and control files remain protected from task writes.
|
|
285
|
+
|
|
286
|
+
Commands Off still permits file tools according to their file permissions.
|
|
287
|
+
Network Off blocks task network access while Pi's provider transport remains
|
|
288
|
+
available. The confined file worker has an 8 MiB file limit and rejects larger
|
|
289
|
+
files explicitly; use confined commands for larger files when commands are On.
|
|
290
|
+
|
|
291
|
+
`read`, `write`, `edit`, and `bash` have verified adapters. Extension tools are
|
|
292
|
+
chosen in `/sandbox` → Subagents · Tools
|
|
293
|
+
([ADR 0009](../../../docs/adr/0009-guarded-and-trusted-subagent-tools.md)):
|
|
294
|
+
|
|
295
|
+
- **Guarded:** `apply_patch`, a harness adapter with the Codex tool's name,
|
|
296
|
+
schema and patch format. Every add, update, move and delete goes through the
|
|
297
|
+
guarded file operations, so the Project files / Outside project levels apply:
|
|
298
|
+
Write refuses deletes and moves outside disposable places, Read refuses
|
|
299
|
+
writes, credential files and protected paths are refused. The whole patch is
|
|
300
|
+
checked first; no file is left half-patched, and a failure part-way reports
|
|
301
|
+
what was and wasn't applied. It is added for a child that may `edit` or
|
|
302
|
+
`write`, or asks for it.
|
|
303
|
+
- **Trusted:** third-party tools you tick (default `web_fetch` and `web_search`).
|
|
304
|
+
Their package is loaded into the child and they run in the child Pi process,
|
|
305
|
+
**outside the file rules**. The child admits one only when both its name and
|
|
306
|
+
its package match the ticked entry, so another package registering the same
|
|
307
|
+
name is refused. A single extension file with no package manifest is loaded
|
|
308
|
+
and admitted by itself, never its directory. Known network tool names
|
|
309
|
+
(`web_fetch`, `web_search`, `firecrawl_scrape`, `firecrawl_extract`, `mcp`, `mcpScript`, `remote_bash`, and any `mcp__*` name) are refused while Network access is Off; this is a name list, not a
|
|
310
|
+
network sandbox. A ticked tool whose package can't be found is refused at launch. Ticked tools
|
|
311
|
+
join the default tool list; an explicit `tools` list still decides.
|
|
312
|
+
|
|
313
|
+
Other tools are disabled in confined children and listed in launch output with
|
|
314
|
+
the reason. The launch line shows what was admitted, for example
|
|
315
|
+
`Runtime: isolated · guarded apply_patch · trusted web_fetch (@juicesharp/rpiv-web-tools)`.
|
|
316
|
+
Provider extensions are trusted runtime code; arbitrary extension tools are not
|
|
317
|
+
admitted merely because their names were requested. Confined children disable project
|
|
318
|
+
runtime configuration and inherited extension discovery. Startup or backend
|
|
319
|
+
failure never falls back to an unconfined child.
|
|
320
|
+
|
|
321
|
+
You still start Pi normally. The subagent package's internal launcher requires
|
|
322
|
+
Pi SDK 0.82.1 or newer; it is not a replacement user-facing Pi command.
|
|
323
|
+
|
|
324
|
+
### Git-mutating subagents and linked worktrees
|
|
325
|
+
|
|
326
|
+
A sandboxed subagent that will mutate Git should set **`git_clone_workspace:true`**
|
|
327
|
+
on `subagent_spawn`. The parent prepares a fresh, self-contained Git clone whose
|
|
328
|
+
`.git/` directory lives **inside the sandbox writable root**, then runs the child
|
|
329
|
+
in that clone.
|
|
330
|
+
|
|
331
|
+
Why this matters: a linked Git worktree (created with `git worktree add`) has a
|
|
332
|
+
`.git` file that points back to administrative state under the main repository,
|
|
333
|
+
typically outside the sandbox directory. A sandbox that only allows writes under
|
|
334
|
+
the worktree directory therefore cannot support normal Git producer operations
|
|
335
|
+
such as fetch, rebase, commit, and push — the child stalls or fails when Git
|
|
336
|
+
tries to write metadata it cannot reach. `git_clone_workspace:true` avoids this
|
|
337
|
+
by cloning the repository with a real `.git/` directory inside the writable root.
|
|
338
|
+
|
|
339
|
+
The clone uses:
|
|
340
|
+
|
|
341
|
+
```
|
|
342
|
+
git clone --reference-if-able <local-reference-repo> --dissociate \
|
|
343
|
+
<remote-url> <sandbox-workspace>
|
|
344
|
+
```
|
|
345
|
+
|
|
346
|
+
`--reference-if-able` borrows local objects from the parent repository during
|
|
347
|
+
setup; `--dissociate` removes the alternates link afterwards, so the clone is
|
|
348
|
+
self-contained and safe to delete. The clone source prefers the source
|
|
349
|
+
workspace's upstream remote URL (`origin` when set) so the disposable
|
|
350
|
+
workspace's `origin` points at the real remote rather than the parent working
|
|
351
|
+
tree — pushes therefore target upstream, not the sandboxed parent. The local
|
|
352
|
+
repository is used only as a reference (and as a content fallback when no
|
|
353
|
+
remote is configured). Source remotes are re-synced after clone. The
|
|
354
|
+
checked-out branch/commit matches the source workspace at spawn time.
|
|
355
|
+
Repo-local Git identity settings from the source (`user.name`, `user.email`,
|
|
356
|
+
and `user.signingkey` when set) are copied into the clone so ordinary commits
|
|
357
|
+
work without reconfiguring identity inside the disposable workspace.
|
|
358
|
+
|
|
359
|
+
If the source workspace is a linked worktree, the clone is prepared from the
|
|
360
|
+
main repository's object database and the requested branch/commit; the child is
|
|
361
|
+
never launched into the structurally broken linked-worktree sandbox. If clone
|
|
362
|
+
preparation fails, the spawn fails fast with a message explaining that the
|
|
363
|
+
linked-worktree Git metadata is outside the sandbox and recommending
|
|
364
|
+
`git_clone_workspace:true`.
|
|
365
|
+
|
|
366
|
+
The mandatory task guard is independent of optional guardrails extensions.
|
|
367
|
+
Adding a package to the tool map does not make its tools safe to execute outside
|
|
368
|
+
the task boundary.
|
|
369
|
+
|
|
370
|
+
## Tool scoping (allowlist)
|
|
371
|
+
|
|
372
|
+
Precedence, highest first: the per-call `tools` param → `config.json`
|
|
373
|
+
`defaultTools` → a built-in default (`read, bash, edit, write, web_search,
|
|
374
|
+
web_fetch`; just `read, bash` in a `clean` child). `exclude_tools` subtracts on
|
|
375
|
+
top.
|
|
376
|
+
|
|
377
|
+
`config.json` (next to the extension) also sets:
|
|
378
|
+
|
|
379
|
+
- `defaultModel` — model for spawns that don't specify one (`null` = inherit the
|
|
380
|
+
foreground model).
|
|
381
|
+
- `delegationMode` — foreground policy (`manual`, `adaptive`, or `coordinator`;
|
|
382
|
+
default `adaptive`). `/subagents mode ...` changes only the current session.
|
|
383
|
+
- `maxConcurrent` — how many subagents may run at once (**default 4**). A spawn
|
|
384
|
+
past the cap is rejected until a running one finishes.
|
|
385
|
+
|
|
386
|
+
## The allowlist also decides what LOADS
|
|
387
|
+
|
|
388
|
+
For confined children, the requested list is first restricted to tools with
|
|
389
|
+
verified task adapters and the trusted tools ticked in `/sandbox`; only the
|
|
390
|
+
ticked tools' packages load. `toolExtensions` still chooses which package to
|
|
391
|
+
load for a tool (an override), but the child admits it only if that is the
|
|
392
|
+
ticked package. Provider extensions can still load as trusted runtime
|
|
393
|
+
dependencies. Nested spawning and
|
|
394
|
+
inherited extension discovery are currently unavailable under confinement.
|
|
395
|
+
|
|
396
|
+
The mapping behavior below applies to unconfined children and to admitted
|
|
397
|
+
runtime dependencies:
|
|
398
|
+
|
|
399
|
+
The `tools` allowlist does double duty: it is both what the child may call **and**
|
|
400
|
+
which extension *code* is loaded into it. A child launches as
|
|
401
|
+
|
|
402
|
+
```
|
|
403
|
+
pi -p --mode json --no-extensions -e <package backing a requested tool> ...
|
|
404
|
+
```
|
|
405
|
+
|
|
406
|
+
so a package that backs no requested tool never loads. With the default
|
|
407
|
+
allowlist (`read, bash, edit, write, web_search, web_fetch`) exactly one package
|
|
408
|
+
loads — the web-tools one — and `web_fetch` works normally.
|
|
409
|
+
|
|
410
|
+
Two maps in `config.json` drive it:
|
|
411
|
+
|
|
412
|
+
- `toolExtensions` — tool name → package(s) providing it. Built-ins (`read`,
|
|
413
|
+
`bash`, `edit`, `write`) need no entry.
|
|
414
|
+
- `providerExtensions` — provider → auth package. Model auth is not tool-shaped:
|
|
415
|
+
`xai/grok-4.5` needs `pi-xai-oauth` loaded whatever tools it was granted.
|
|
416
|
+
|
|
417
|
+
Ask for a tool with no mapping and the spawn still succeeds, but says so at
|
|
418
|
+
launch — the tool simply will not exist in the child.
|
|
419
|
+
|
|
420
|
+
`clean:true` is the narrowest case of the same mechanism: no extensions at all.
|
|
421
|
+
For unconfined children, `allow_nested:true` loads this package into the child;
|
|
422
|
+
without it, nested spawning is unavailable. Confined children disable nesting
|
|
423
|
+
regardless of this flag until a verified adapter exists.
|
|
424
|
+
|
|
425
|
+
`inheritExtensions: true` in `config.json` restores the old load-everything
|
|
426
|
+
behavior. It is **operator-only** — no spawn parameter can reach it, so the child
|
|
427
|
+
model cannot widen its own runtime. It also re-exposes the failure below.
|
|
428
|
+
|
|
429
|
+
### Why: a subagent that loads everything can die mid-turn reporting success
|
|
430
|
+
|
|
431
|
+
Loading every installed package means inheriting their startup side effects. A
|
|
432
|
+
package that replaces builtin `bash` with a `detached` + `unref()` spawn breaks
|
|
433
|
+
`pi -p`: on a parallel `bash` + `read` batch the in-process `read` finishes, the
|
|
434
|
+
unref'd `bash` doesn't hold the event loop, Node drains, and the child **exits 0
|
|
435
|
+
mid-turn** — no `tool_execution_end`, no `agent_end`. Historically, exit 0 was
|
|
436
|
+
indistinguishable from a clean finish, so all 17 observed mid-turn exits were
|
|
437
|
+
reported as ✓ completed. Finalization now requires terminal agent evidence and
|
|
438
|
+
no unmatched tool starts. Lifecycle validation classifies runs as `complete`,
|
|
439
|
+
`incomplete_no_terminal_event`, `incomplete_open_tools`, `failed_exit`, or
|
|
440
|
+
`killed`; incoherent exit-0 streams are recorded as failed with named lifecycle
|
|
441
|
+
diagnostics on `subagent_result` and attention wording on completion callbacks.
|
|
442
|
+
|
|
443
|
+
A tool allowlist alone cannot fix this. `--tools` restricts what the model may
|
|
444
|
+
*call*; the package already overrode builtin `bash` at startup, so the `bash` in
|
|
445
|
+
your allowlist **is** the broken one. Measured with the default 6 tools:
|
|
446
|
+
|
|
447
|
+
| runtime | tool starts / ends | terminal event |
|
|
448
|
+
|---|---|---|
|
|
449
|
+
| all extensions loaded | 2 / 1 | none — exits 0 mid-turn |
|
|
450
|
+
| `--no-extensions -e <web-tools>` | 3 / 3 | `agent_settled`, `web_fetch` OK |
|
|
451
|
+
|
|
452
|
+
A package *denylist* isn't expressible either: pi has only `-e <path>` (add one)
|
|
453
|
+
and `--no-extensions` (all off) — there is no "load all except X" flag. Naming
|
|
454
|
+
what you want is the only mechanism that excludes anything, and it excludes
|
|
455
|
+
future offenders too, with no name to keep updated.
|
|
456
|
+
|
|
457
|
+
### Known incompatibility: `pi-patty-bg-tasks`
|
|
458
|
+
|
|
459
|
+
**`pi-patty-bg-tasks` (tested at 1.1.6) is incompatible with subagents and must
|
|
460
|
+
not be loaded into a child.** It is the package that produced the failure above:
|
|
461
|
+
it replaces builtin `bash` and spawns `detached` + `proc.unref()`
|
|
462
|
+
(`src/spawn.ts`), which in print mode drains the event loop mid-turn. Bisected
|
|
463
|
+
against all 18 installed packages — alone it reproduces; every other package
|
|
464
|
+
alone is fine.
|
|
465
|
+
|
|
466
|
+
The default configuration already excludes it, structurally, because it backs no
|
|
467
|
+
requested tool. You only re-expose it by setting `inheritExtensions: true`, or by
|
|
468
|
+
mapping a tool to it in `toolExtensions`. Don't.
|
|
469
|
+
|
|
470
|
+
A proper fix belongs upstream — preserve builtin `bash` semantics when overriding
|
|
471
|
+
it, keep foreground subprocesses referenced until the tool promise settles, and
|
|
472
|
+
put genuinely detached work behind a separate background-task tool.
|
|
473
|
+
|
|
474
|
+
## Status & cost tracking
|
|
475
|
+
|
|
476
|
+
Driven by the child's `--mode json` usage events:
|
|
477
|
+
|
|
478
|
+
- **Live widget** above the editor while any subagent runs — a spinner per run
|
|
479
|
+
with elapsed time, the current tool, and running token/cost spend, ticking once
|
|
480
|
+
a second. It clears itself when the last run finishes. (TUI/RPC only; silent in
|
|
481
|
+
`-p`/print mode.)
|
|
482
|
+
- **On demand** — `subagent_list`, `subagent_output`, and `subagent_result` are
|
|
483
|
+
bounded current-session pages (see `docs/issue-312-output.md`). They omit
|
|
484
|
+
ordered tool-name sequences and default token/cost lines. Pass
|
|
485
|
+
`include: ["cost"]` to `subagent_output` or `subagent_result` for one spend
|
|
486
|
+
line (`spend: 1.2k tok (↑400 ↓800) · $0.0034`) and `include: ["tools"]` for
|
|
487
|
+
one tool-call line (`tools: 7 calls · distinct: bash, read, edit`). The human
|
|
488
|
+
toast may still include elapsed + spend; the model-facing callback does not.
|
|
489
|
+
- **Folded result display.** In interactive TUI sessions, `subagent_result`
|
|
490
|
+
renders a compact preview by default so long child answers do not flood the
|
|
491
|
+
transcript. Clicking the tool row, or using the row expand action, shows the
|
|
492
|
+
full bounded result. This is a display concern only: the tool still returns
|
|
493
|
+
the complete bounded `content` payload to the model.
|
|
494
|
+
|
|
495
|
+
Spend is summed from each finalized assistant turn's `usage` (so multi-turn
|
|
496
|
+
tool-using runs total correctly), and cost comes straight from the model's
|
|
497
|
+
reported per-request cost.
|
|
498
|
+
|
|
499
|
+
## Subagent navigator (TUI)
|
|
500
|
+
|
|
501
|
+
In an interactive TUI session, a **subagent navigator** lets you inspect and
|
|
502
|
+
organize runs without asking the model to call a tool. The running-subagents
|
|
503
|
+
widget can be focused from an empty input line for quick actions; detail output
|
|
504
|
+
opens in the overlay. Print/RPC modes do not install the navigator; tool access
|
|
505
|
+
is unchanged in every mode.
|
|
506
|
+
|
|
507
|
+
### Open
|
|
508
|
+
|
|
509
|
+
- With the editor **empty** and at least one non-dismissed current-parent run
|
|
510
|
+
still running, press `←` to focus the main-window subagent list above the
|
|
511
|
+
input line.
|
|
512
|
+
- If the editor contains text, `←` keeps normal cursor-left behavior.
|
|
513
|
+
- While running runs exist, the default footer shows `← subagents · N`. The
|
|
514
|
+
live widget also includes a secondary `← to navigate` hint on its title line
|
|
515
|
+
for terminals that do not render the default footer status. The hint clears
|
|
516
|
+
when no non-dismissed current-parent run is still running.
|
|
517
|
+
- While the main-window list is focused, the title hint changes to
|
|
518
|
+
`Enter to view · x to stop`; the selected row is marked with `›`. Press
|
|
519
|
+
`↓` from the bottom row to return to the input line.
|
|
520
|
+
- The Subagents lane pins an informational `main` row above child runs. It
|
|
521
|
+
shows the foreground model, effort, active tool, context tokens, and active
|
|
522
|
+
elapsed time. It is not selectable and can never become an `x` stop target.
|
|
523
|
+
- `↑` moves to the previous row when multiple running rows are shown. `↓`
|
|
524
|
+
moves toward the input line, returning to normal input from the bottom row.
|
|
525
|
+
- `Enter` opens the selected run's live detail view. `x` stops the selected
|
|
526
|
+
running run using shared `subagent_stop` semantics and dismisses it from the
|
|
527
|
+
navigator.
|
|
528
|
+
|
|
529
|
+
### Detail view
|
|
530
|
+
|
|
531
|
+
- Detail uses the same command-sheet treatment, with section rules for the
|
|
532
|
+
inspector groups and command bars at the top and bottom of the view.
|
|
533
|
+
- Tool-call log rows wrap within the detail width instead of truncating. Source
|
|
534
|
+
row boundaries and indentation are retained, with long paths and JSON values
|
|
535
|
+
preferring delimiter breaks before a hard wrap. Header and metadata geometry
|
|
536
|
+
is unchanged.
|
|
537
|
+
- Shows status (colorized), model/effort, elapsed, tools, spend, and parsed
|
|
538
|
+
output, plus sectioned health: process identity/liveness, activity,
|
|
539
|
+
compaction, active tool, model call/error, last log write, thresholds, and
|
|
540
|
+
callback notification timestamps. Compaction, active tool, and model state
|
|
541
|
+
are separate sections. The view refreshes about once per second while open.
|
|
542
|
+
- The transcript shows up to the latest 25 rows by default; `l` switches
|
|
543
|
+
between 25 and 10 rows. The row count is a cap, not a guarantee: the metadata
|
|
544
|
+
lines (provider, id, model, elapsed, tools, spend, pid, pgid) always stay
|
|
545
|
+
visible, and on a short terminal the transcript shows only as many of its
|
|
546
|
+
newest rows as fit below them. `subagent_result` and `subagent_output` page
|
|
547
|
+
sizes are unaffected.
|
|
548
|
+
- Detail opened from the main-window list closes back to the main page; the
|
|
549
|
+
main-window selection remains on the viewed run when it is still visible.
|
|
550
|
+
- `x` arms Stop for a running run, or Dismiss for a terminal run.
|
|
551
|
+
- `esc` closes the detail view and returns to the main page.
|
|
552
|
+
|
|
553
|
+
### Two-press `x` stop/dismiss
|
|
554
|
+
|
|
555
|
+
- First `x` on the selected (list) or viewed (detail) run arms the action for three
|
|
556
|
+
seconds and shows a footer hint: `x again to stop <name>` while running, or
|
|
557
|
+
`x again to dismiss <name>` when terminal.
|
|
558
|
+
- Second `x` within the window, on the **same** run, acts:
|
|
559
|
+
- **Running** — stop the process group (shared `subagent_stop` semantics),
|
|
560
|
+
mark killed, then dismiss from the navigator.
|
|
561
|
+
- **Terminal** — dismiss only; terminal status is not rewritten.
|
|
562
|
+
- Changing selection, leaving the view, closing the overlay, arming timeout,
|
|
563
|
+
reload, and session teardown all disarm close and clear the confirm hint.
|
|
564
|
+
|
|
565
|
+
### Dismissal is navigator-only
|
|
566
|
+
|
|
567
|
+
Dismissed runs leave the navigator list and footer count. Logs, prompt, session
|
|
568
|
+
data, metadata, and id-based tool access stay intact. `subagent_list`,
|
|
569
|
+
`subagent_output`, `subagent_result`, and `subagent_stop` still resolve dismissed
|
|
570
|
+
run ids. Dismissal survives `/reload` as dismissed in the navigator.
|
|
571
|
+
|
|
572
|
+
### Reload and teardown
|
|
573
|
+
|
|
574
|
+
`/reload` and session restart reinstall the empty-editor wrapper without stacking
|
|
575
|
+
duplicate handlers, republish the footer count, and clear any leftover overlay
|
|
576
|
+
timers or close-confirm state. Session shutdown disposes open navigator timers
|
|
577
|
+
and clears navigator footer statuses (TUI only).
|
|
578
|
+
|
|
579
|
+
## Design notes
|
|
580
|
+
|
|
581
|
+
- Runtime lives outside any repo, under `$TMPDIR/pi-better-subagents/`
|
|
582
|
+
(`runs/<id>/` holds `output.log`, `prompt.md`, `meta.json`; `sessions/` holds
|
|
583
|
+
child session state). The `meta.json` sidecar is authoritative, so `list` /
|
|
584
|
+
`output` / `result` survive turns, `/reload`, and pi restarts.
|
|
585
|
+
- The child runs `--mode json`; `subagent_result` / `subagent_output` **parse**
|
|
586
|
+
the event stream and return just the bounded final answer (no tool-name history).
|
|
587
|
+
Non-JSON banner/warning lines fail to parse and are dropped, so the result is
|
|
588
|
+
clean. The prompt is passed as a **positional argument**, never `@file` — some
|
|
589
|
+
models refuse an @-attached file as untrusted content.
|
|
590
|
+
- The child gets **only** the explicit prompt — no silent parent-context bleed.
|
|
591
|
+
- `--approve` is **off by default** (headless runs can't prompt for trust).
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
## Parent-process scoping
|
|
595
|
+
|
|
596
|
+
The live widget, default `subagent_list`, concurrency cap, and `session_start` ticker only include runs this pi process spawned (`spawnPid === process.pid`). The on-disk registry stays machine-global for durability. Default `subagent_list` is current-session, newest first, 10 compact rows / 1 KiB. Pass `limit:N` (max 100) or `max_bytes` (max 4 KiB) for an explicit larger page. Pass `all:true` for a global / foreign-session view. Pass `status:[...]` to filter by effective status: `running`, `completed`, `failed`, `killed`, transient `exited`, or durable `orphaned` / `lost`. Id-based `subagent_result` / `subagent_output` default to the current session; pass `all:true` to read a foreign-session id. Unknown ownership is a gap, not “not found”; when the current session identity cannot be read, no run's ownership is treated as verified and `all:true` is required. List pages are reached with the returned `nextCursor` (every run is reachable, not only the first 100). Runs whose metadata is missing or corrupt are reported as gaps and counted on lists, never as nonexistent.
|
|
597
|
+
|
|
598
|
+
## Supervision health (`orphaned` / `lost`)
|
|
599
|
+
|
|
600
|
+
While a current-parent run is `running` or `orphaned`, a periodic health tick
|
|
601
|
+
reconciles process-group evidence only (see `docs/adr/0002-process-group-only-subagent-health.md`):
|
|
602
|
+
|
|
603
|
+
- **`orphaned`** — direct supervision of the child is broken, but related
|
|
604
|
+
process-group work may still be alive. Non-terminal and non-final; operationally
|
|
605
|
+
unhealthy immediately. The coordinator (when `callback:true`) gets one durable
|
|
606
|
+
ATTENTION follow-up naming `subagent_result` / `subagent_output` / `subagent_stop`
|
|
607
|
+
so it can inspect artifacts and decide whether to wait, stop, or retry. Human
|
|
608
|
+
`ui.notify` still fires when `callback:false`.
|
|
609
|
+
- **`lost`** — no related process remains and no coherent terminal completion was
|
|
610
|
+
observed. Terminal with unknown outcome (not the same as `failed`). Same one-shot
|
|
611
|
+
ATTENTION follow-up + diagnostic `subagent_result` path with best-available
|
|
612
|
+
artifacts.
|
|
613
|
+
|
|
614
|
+
Orphaned/lost callbacks use the same non-interrupting
|
|
615
|
+
`{ deliverAs: "followUp", triggerTurn: true }` mechanics, but bypass the ordinary
|
|
616
|
+
completion accumulation window and retain distinct ATTENTION wording. Per-status
|
|
617
|
+
markers on `meta.json` (`orphanedCallbackSentAt` / `lostCallbackSentAt`) are written
|
|
618
|
+
only after a successful handoff and dedupe across reloads and repeated health ticks.
|
|
619
|
+
Persisted unmarked orphaned/lost states are recovered on the health ticker after
|
|
620
|
+
`/reload` even when process evidence does not produce a fresh transition.
|
|
621
|
+
|
|
622
|
+
Pi's optional `followUpMode: all` remains useful when completion groups arrive
|
|
623
|
+
after an earlier 100 ms batch has already flushed: Pi can consume those queued
|
|
624
|
+
follow-ups in one later agent turn. This extension does not change Pi core or
|
|
625
|
+
Pi's default follow-up mode.
|
|
626
|
+
|
|
627
|
+
### Surfacing health (tools + passive widget)
|
|
628
|
+
|
|
629
|
+
Multi-dimensional observations (`stale`, long tool, compacting / long compaction,
|
|
630
|
+
model error/retry, plus process `orphaned` / `lost`) are computed from durable
|
|
631
|
+
status + child-event evidence and surfaced on existing paths without a parallel
|
|
632
|
+
health model:
|
|
633
|
+
|
|
634
|
+
- **`subagent_list`** — durable `orphaned` / `lost` status brackets; degraded
|
|
635
|
+
compact facts only when actionable. Healthy/quiet rows stay on the compact format.
|
|
636
|
+
- **`subagent_output` / `subagent_result`** — `[health: …]` diagnostics for
|
|
637
|
+
orphaned, lost, and degraded running runs; #65 orphaned/lost result bodies kept.
|
|
638
|
+
- **Passive live widget** — healthy/quiet unchanged; degraded (and orphaned) may
|
|
639
|
+
show a short suffix. Still `setWidget` only — never focusable.
|
|
640
|
+
- **`callback:false`** suppresses coordinator follow-up only; human `ui.notify` and
|
|
641
|
+
TUI/passive visibility remain.
|
|
642
|
+
|
|
643
|
+
## Install
|
|
644
|
+
|
|
645
|
+
Linux sandboxing requires the system `bubblewrap` package. Install it before
|
|
646
|
+
launching sandboxed children:
|
|
647
|
+
|
|
648
|
+
```bash
|
|
649
|
+
# Debian/Ubuntu
|
|
650
|
+
sudo apt-get install bubblewrap
|
|
651
|
+
# Fedora/RHEL
|
|
652
|
+
sudo dnf install bubblewrap
|
|
653
|
+
# Arch Linux
|
|
654
|
+
sudo pacman -S bubblewrap
|
|
655
|
+
```
|
|
656
|
+
|
|
657
|
+
When `bwrap` is absent, explicit sandbox requests fail with an installation hint;
|
|
658
|
+
default-on sandboxing preserves the documented direct-execution fallback. Once
|
|
659
|
+
`bwrap` is selected, a launch failure fails closed rather than retrying the child
|
|
660
|
+
without confinement.
|
|
661
|
+
|
|
662
|
+
Symlink the project into pi's auto-discovered extensions dir:
|
|
663
|
+
|
|
664
|
+
```bash
|
|
665
|
+
ln -sfn "$PWD" ~/.pi/agent/extensions/pi-better-subagents
|
|
666
|
+
```
|
|
667
|
+
|
|
668
|
+
Then `/reload` (or restart pi). It appears as `pi-better-subagents`.
|
|
669
|
+
|
|
670
|
+
> Do **not** add it to `settings.json`'s `extensions` array — a live pi session
|
|
671
|
+
> rewrites that file on its own saves and drops hand-added entries. Auto-discovery
|
|
672
|
+
> via the symlink is stable. Quick throwaway test without installing:
|
|
673
|
+
> `pi -e ./index.ts`.
|
|
674
|
+
|
|
675
|
+
## Tests
|
|
676
|
+
|
|
677
|
+
Unit tests (`node --test tests/*.test.mjs`) cover the pure logic — widget
|
|
678
|
+
rendering, completion delivery, and extension resolution.
|
|
679
|
+
|
|
680
|
+
Real integration smoke tests live in [`tests/`](tests/) — a subagent using
|
|
681
|
+
`web_fetch`, one driving `gh`, env inheritance through the sandbox, and headless
|
|
682
|
+
isolation surviving a parallel `bash` + `read` batch. See
|
|
683
|
+
[`tests/README.md`](tests/README.md).
|
|
684
|
+
|
|
685
|
+
## Roadmap
|
|
686
|
+
|
|
687
|
+
Tracked in [issues](https://github.com/1aboveio/pi-better-subagents/issues).
|
|
688
|
+
Near-term:
|
|
689
|
+
|
|
690
|
+
- Guarantee subagent autonomy — verify/deny any child→parent supervision
|
|
691
|
+
back-channel so a child can never block on the parent ([#1](https://github.com/1aboveio/pi-better-subagents/issues/1)).
|
|
692
|
+
- Make `callback:true` a lightweight trigger instead of embedding the full result
|
|
693
|
+
twice ([#2](https://github.com/1aboveio/pi-better-subagents/issues/2)) — **done**.
|
|
694
|
+
- Named agent-definition files (per-agent system prompt + tool allowlist) and
|
|
695
|
+
chain/parallel orchestration.
|
|
@@ -103,7 +103,8 @@ import {
|
|
|
103
103
|
} from "./health.ts";
|
|
104
104
|
import {
|
|
105
105
|
assignBatchJobNames,
|
|
106
|
-
|
|
106
|
+
batchLaunchOutputSchema,
|
|
107
|
+
batchLaunchResult,
|
|
107
108
|
mergeJobOptions,
|
|
108
109
|
nextBatchId,
|
|
109
110
|
planBatchLaunches,
|
|
@@ -1902,7 +1903,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
1902
1903
|
description:
|
|
1903
1904
|
"Launch several independent background pi subagents at once. Each job becomes a " +
|
|
1904
1905
|
"normal subagent run with its own run id, process, log, and metadata. " +
|
|
1905
|
-
"'shared' options are applied to every job; per-job options override them."
|
|
1906
|
+
"'shared' options are applied to every job; per-job options override them. " +
|
|
1907
|
+
"Codemode callers receive a structured launch receipt: status, batchId when assigned, " +
|
|
1908
|
+
"and launched, failed, skipped arrays with 1-based effective job positions (after any confirmed role split). Inspect all arrays; " +
|
|
1909
|
+
"a returned receipt is not proof that every job launched or completed. " +
|
|
1910
|
+
"Await the launch call and preserve its receipt. Script cancellation does not stop launched runs; use subagent_stop.",
|
|
1911
|
+
outputSchema: batchLaunchOutputSchema,
|
|
1906
1912
|
promptSnippet: "Launch a batch of background subagents at once",
|
|
1907
1913
|
promptGuidelines: [
|
|
1908
1914
|
"When delegation is permitted by the active mode, use subagent_spawn_batch for several independent assigned tasks. It returns immediately with a batch id and one run id per launched job.",
|
|
@@ -1993,7 +1999,17 @@ export default function (pi: ExtensionAPI) {
|
|
|
1993
1999
|
p.jobs.map((job) => mergeJobOptions(p.shared, job) as CatalogJobFields),
|
|
1994
2000
|
{ hasUI: catalogHost.hasUI === true, select: catalogHost.select },
|
|
1995
2001
|
);
|
|
1996
|
-
if (clarified.status === "clarification-needed") return
|
|
2002
|
+
if (clarified.status === "clarification-needed") return {
|
|
2003
|
+
...catalogResult(clarified),
|
|
2004
|
+
structuredContent: {
|
|
2005
|
+
status: "clarification-needed",
|
|
2006
|
+
message: clarified.message,
|
|
2007
|
+
choices: [...clarified.choices],
|
|
2008
|
+
launched: [],
|
|
2009
|
+
failed: [],
|
|
2010
|
+
skipped: [],
|
|
2011
|
+
},
|
|
2012
|
+
};
|
|
1997
2013
|
p.jobs = clarified.jobs as typeof p.jobs;
|
|
1998
2014
|
validateBatchPlan({ shared: undefined, jobs: p.jobs, onCapacity: p.onCapacity, config: cfg });
|
|
1999
2015
|
}
|
|
@@ -2023,9 +2039,9 @@ export default function (pi: ExtensionAPI) {
|
|
|
2023
2039
|
|
|
2024
2040
|
const names = assignBatchJobNames(p.jobs);
|
|
2025
2041
|
const batchId = nextBatchId();
|
|
2026
|
-
const launched: { name: string; id: string; modelNote?: string }[] = [];
|
|
2027
|
-
const failed: { name: string; reason: string }[] = [];
|
|
2028
|
-
const skipped: { name: string }[] = [];
|
|
2042
|
+
const launched: { job: number; name: string; id: string; modelNote?: string }[] = [];
|
|
2043
|
+
const failed: { job: number; name: string; reason: string }[] = [];
|
|
2044
|
+
const skipped: { job: number; name: string }[] = [];
|
|
2029
2045
|
// How many reject-mode reserved slots are still held (not yet committed/released).
|
|
2030
2046
|
let reservedRemaining = launchAvailable ? 0 : p.jobs.length;
|
|
2031
2047
|
|
|
@@ -2039,7 +2055,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2039
2055
|
if (launchAvailable) {
|
|
2040
2056
|
if (!gate.tryReserve(1, maxConcurrent)) {
|
|
2041
2057
|
for (let j = i; j < p.jobs.length; j++) {
|
|
2042
|
-
skipped.push({ name: names[j] });
|
|
2058
|
+
skipped.push({ job: j + 1, name: names[j] });
|
|
2043
2059
|
}
|
|
2044
2060
|
break;
|
|
2045
2061
|
}
|
|
@@ -2064,12 +2080,12 @@ export default function (pi: ExtensionAPI) {
|
|
|
2064
2080
|
const { id, modelNote } = await spawnSubagentRun(ctx, { ...merged, name }, { batchId, batchName: p.batchName });
|
|
2065
2081
|
gate.commit(1);
|
|
2066
2082
|
if (!launchAvailable) reservedRemaining -= 1;
|
|
2067
|
-
launched.push({ name, id, ...(modelNote ? { modelNote } : {}) });
|
|
2083
|
+
launched.push({ job: i + 1, name, id, ...(modelNote ? { modelNote } : {}) });
|
|
2068
2084
|
} catch (err) {
|
|
2069
2085
|
gate.release(1);
|
|
2070
2086
|
if (!launchAvailable) reservedRemaining -= 1;
|
|
2071
2087
|
const reason = err instanceof Error ? err.message : String(err);
|
|
2072
|
-
failed.push({ name, reason });
|
|
2088
|
+
failed.push({ job: i + 1, name, reason });
|
|
2073
2089
|
if (!launchAvailable) {
|
|
2074
2090
|
// reject mode: leave already-launched runs running, release any
|
|
2075
2091
|
// still-held later reservations, and report every later job as failed.
|
|
@@ -2079,13 +2095,14 @@ export default function (pi: ExtensionAPI) {
|
|
|
2079
2095
|
}
|
|
2080
2096
|
for (let j = i + 1; j < p.jobs.length; j++) {
|
|
2081
2097
|
failed.push({
|
|
2098
|
+
job: j + 1,
|
|
2082
2099
|
name: names[j],
|
|
2083
2100
|
reason: "not launched due to earlier job failure in reject mode",
|
|
2084
2101
|
});
|
|
2085
2102
|
}
|
|
2086
|
-
return
|
|
2103
|
+
return batchLaunchResult({
|
|
2087
2104
|
batchId, batchName: p.batchName, launched, skipped, failed,
|
|
2088
|
-
})
|
|
2105
|
+
});
|
|
2089
2106
|
}
|
|
2090
2107
|
// launch-available: failure did not consume a slot — continue so
|
|
2091
2108
|
// later jobs can use remaining capacity (backfill).
|
|
@@ -2098,7 +2115,7 @@ export default function (pi: ExtensionAPI) {
|
|
|
2098
2115
|
reservedRemaining = 0;
|
|
2099
2116
|
}
|
|
2100
2117
|
|
|
2101
|
-
return
|
|
2118
|
+
return batchLaunchResult({ batchId, batchName: p.batchName, launched, skipped, failed });
|
|
2102
2119
|
},
|
|
2103
2120
|
});
|
|
2104
2121
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-better-subagents",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.10.0",
|
|
4
4
|
"description": "Pi extension for detached, sandboxed subagent runs that keep the foreground session free.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -47,6 +47,7 @@
|
|
|
47
47
|
"config.json",
|
|
48
48
|
"README.md",
|
|
49
49
|
"roles/**/*.md",
|
|
50
|
+
"docs/usage.md",
|
|
50
51
|
"docs/agent-catalog.md",
|
|
51
52
|
"docs/agent-catalog-operations.md",
|
|
52
53
|
"docs/agent-model-resolution.md",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-better-harness",
|
|
3
|
-
"version": "0.13.
|
|
3
|
+
"version": "0.13.1",
|
|
4
4
|
"description": "Pi extension bundle for a write sandbox, subagents, background tasks, SSH, goals, and structured plans.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -57,7 +57,7 @@
|
|
|
57
57
|
"pi-better-plan": "0.5.1",
|
|
58
58
|
"pi-better-sandbox": "0.8.0",
|
|
59
59
|
"pi-better-ssh": "0.1.1",
|
|
60
|
-
"pi-better-subagents": "0.
|
|
60
|
+
"pi-better-subagents": "0.10.0",
|
|
61
61
|
"smol-toml": "1.9.0",
|
|
62
62
|
"yaml": "^2.9.1"
|
|
63
63
|
},
|