pi-better-harness 0.13.0 → 0.13.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -12,6 +12,7 @@ Use `pi-better-subagents` when you want Pi to launch independent agent work with
12
12
 
13
13
  ## Core Features
14
14
  - Non-blocking subagent launches, with an optional role and named-agent catalog (`docs/agent-catalog.md`, `docs/agent-catalog-lifecycle.md`).
15
+ - Typed foreground `subagent_spawn_batch` launch receipts for native Pi codemode (Pi 0.99.1+), preserving direct tool calls, per-run callbacks, and batch capacity/catalog guarantees. See [codemode batch launches](docs/usage.md#codemode-batch-launches).
15
16
  - Default OS write sandboxing on macOS and Linux.
16
17
  - Explicit tool allowlists for child sessions.
17
18
  - Durable logs, result retrieval, and [failure observations](docs/failure-observations.md) independent of lifecycle status. A live child handles its own tool errors; the parent is woken only for actionable incidents, and children can classify handled failures with `failure_disposition`.
@@ -14,6 +14,64 @@ import { assertTimingParams } from "./timing.ts";
14
14
 
15
15
  const VALID_CAPACITY_MODES = new Set(["reject", "launch-available"]);
16
16
 
17
+ /** Public codemode receipt; details alone are not returned to nested callers. */
18
+ export const batchLaunchOutputSchema = {
19
+ type: "object",
20
+ properties: {
21
+ status: { type: "string", enum: ["launched", "partial", "not-launched", "clarification-needed"] },
22
+ batchId: { type: "string" },
23
+ batchName: { type: "string" },
24
+ launched: {
25
+ type: "array",
26
+ items: {
27
+ type: "object",
28
+ properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" }, id: { type: "string" }, modelNote: { type: "string" } },
29
+ required: ["job", "name", "id"],
30
+ additionalProperties: false,
31
+ },
32
+ },
33
+ failed: {
34
+ type: "array",
35
+ items: {
36
+ type: "object",
37
+ properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" }, reason: { type: "string" } },
38
+ required: ["job", "name", "reason"],
39
+ additionalProperties: false,
40
+ },
41
+ },
42
+ skipped: {
43
+ type: "array",
44
+ items: {
45
+ type: "object",
46
+ properties: { job: { type: "integer", minimum: 1 }, name: { type: "string" } },
47
+ required: ["job", "name"],
48
+ additionalProperties: false,
49
+ },
50
+ },
51
+ message: { type: "string" },
52
+ choices: { type: "array", items: { type: "string" } },
53
+ },
54
+ required: ["status", "launched", "failed", "skipped"],
55
+ additionalProperties: false,
56
+ };
57
+
58
+ /** Launch success is not run completion; partial receipts retain every created run. */
59
+ export function batchLaunchResult({ batchId, batchName, launched, skipped, failed }) {
60
+ const receipt = {
61
+ status: launched.length === 0 ? "not-launched" : failed.length || skipped.length ? "partial" : "launched",
62
+ batchId,
63
+ ...(batchName !== undefined ? { batchName } : {}),
64
+ launched,
65
+ failed,
66
+ skipped,
67
+ };
68
+ return {
69
+ content: [{ type: "text", text: formatBatchLaunchResponse(receipt) }],
70
+ details: receipt,
71
+ structuredContent: receipt,
72
+ };
73
+ }
74
+
17
75
  /**
18
76
  * Overlay per-job options on top of shared options. Only defined keys are
19
77
  * inherited; booleans keep their false values.
@@ -0,0 +1,695 @@
1
+ # pi-better-subagents
2
+
3
+ A better subagent extension for [pi](https://github.com/earendil-works/pi-coding-agent).
4
+
5
+ Not a clone of Claude Code's subagents — a rethink of what a subagent system
6
+ should be: **autonomous, non-blocking, and safe by default.** You delegate work
7
+ and keep going; each subagent runs on its own in an isolated process, confined to
8
+ what it needs, and reports back when it's done. No blocking waits, no
9
+ back-channel for it to stall on, no unbounded blast radius.
10
+
11
+ ```
12
+ launch is the result · completion triggers fetch · the foreground never blocks
13
+ ```
14
+
15
+ ## Delegation Mode
16
+
17
+ Set `delegationMode` in this package's `config.json` to `manual`, `adaptive`, or
18
+ `coordinator` (default `adaptive`). This controls foreground delegation guidance,
19
+ not child permissions or catalog definitions. `/subagents` reports the active
20
+ mode; `/subagents mode manual|adaptive|coordinator` overrides it for the current
21
+ session without editing config. Other arguments show usage. `/reload` retains
22
+ this override; new sessions use config, while resuming a session restores that
23
+ session's override.
24
+
25
+ - **Manual:** no proactive delegation, including in plan mode. A user request or
26
+ explicit workflow requirement may still delegate.
27
+ - **Adaptive:** delegate useful, substantial independent work while continuing
28
+ foreground work; keep small or coupled tasks local.
29
+ - **Coordinator:** use `agents_catalog` to discover current roles and delegate
30
+ every nontrivial task owned by an available role, passing the role by its
31
+ short name (`role: "developer"`). The foreground owns
32
+ orchestration, cross-role decisions, unowned or ambiguous work, integration,
33
+ and final verification. Current role descriptions, including custom catalog
34
+ overrides, determine ownership.
35
+
36
+ All modes retain the no-polling rule and require inspection of delegated results
37
+ before final verification. `pi-better-plan` follows the active mode when a plan
38
+ is present; without this extension it uses adaptive guidance.
39
+
40
+ ## Principles
41
+
42
+ - **The foreground never blocks.** Launching a subagent *is* the deliverable —
43
+ `subagent_spawn` starts a detached `pi -p` child and returns immediately,
44
+ leaving the session free for the human. When the child finishes, it sends a
45
+ lightweight trigger; the foreground calls `subagent_result` and presents the
46
+ result (as a `followUp`, never cutting into work in progress). The foreground
47
+ is nudged once, at completion — never on a wait/poll loop.
48
+ - **Subagents are autonomous; communication is one-way (parent → child).** The
49
+ parent front-loads everything the child needs into the spawn; the child runs to
50
+ completion and **returns a result**. There is no mid-task child→parent blocking
51
+ call for a subagent to waste wall-clock on — a child missing a piece of info
52
+ resolves it from what it was given, or records it unavailable and returns.
53
+ - **Safe by default.** Every subagent is OS-sandboxed — writes confined to its
54
+ working directory, reads and network open — and scoped to an explicit tool
55
+ allowlist. It can't corrupt the parent, escape its directory, or recurse into
56
+ more subagents without opt-in.
57
+ - **Observable.** A live status widget and on-demand queries show each run's
58
+ elapsed time and token/cost spend.
59
+
60
+ ### Observable progress
61
+
62
+ Subagent health uses the shared 60-second quiet and 5-minute stalled defaults,
63
+ while preserving its stricter child-event semantics: active tool calls,
64
+ compaction, and model retry/error phases explain silence rather than becoming
65
+ stale. For a cross-extension override, set `PI_BETTER_STALL_QUIET_MS` and
66
+ `PI_BETTER_STALL_MS` in milliseconds. Existing subagent `config.json` health
67
+ thresholds continue to take precedence for subagent health.
68
+
69
+ ## Tools
70
+
71
+ | Tool | Blocks? | What it does |
72
+ |------|---------|--------------|
73
+ | `subagent_spawn` | never | Launch a task in a background subagent; returns a run id at once. Params: `prompt`, `name`, `model`, `tools` (allowlist), `exclude_tools`, `sandbox`, `sandbox_dir`, `callback`, `clean`, `cwd`, `git_clone_workspace`, `approve`, `allow_nested`, and the [timing](#run-timing-deadline-ceiling-stuck) overrides `deadline_minutes`, `grace_minutes`, `max_minutes`, `stuck_minutes`. |
74
+ | `subagent_spawn_batch` | never | Launch several independent subagents at once. Each job becomes a normal run. Params: `batchName`, `shared` (options applied to every job), `jobs[]` (each needs `prompt`; same optional params as `subagent_spawn`), `onCapacity` (`reject` or `launch-available`). |
75
+ | `subagent_list` | never | Compact current-session list (default 10 rows / 1 KiB page). Params: `all` (machine-global / foreign session), `limit` (default 10, max 100), `cursor`, `max_bytes` (max 4 KiB), `status` (`running`, `completed`, `failed`, `killed`, `exited`, durable `orphaned`, `lost`). Incident counts are compact; no spend/tool histories. |
76
+ | `subagent_output` | never | Bounded current-session excerpt (default 1 KiB / 10 lines). Params: `id`, `lines` (`tail_lines` is a deprecated alias), `cursor` (a returned `nextCursor`, `statusCursor`, or `incidentCursor`), `max_bytes` (max 4 KiB; raw 16 KiB default, 64 KiB max), `mode` (`raw` pages retained log bytes), `include` (`["cost"]`, `["tools"]`, or both: opt-in spend and tool-count lines), `all` (foreign-session id). Missing/unreadable logs are gaps, not empty healthy output. |
77
+ | `subagent_result` | never | Finished-run answer for the current session (2048-byte page). Params: `id`, `cursor` (a returned `nextCursor` continues the answer; `statusCursor` / `incidentCursor` also accepted), `max_bytes` (maximum 8192 bytes), `lines` (optional line cap per answer page), `mode` (`raw`), `include` (`["cost"]`, `["tools"]`, or both), `all` (foreign-session id). Failures and exceptional lifecycle facts come before progress; no tool-name histories. TUI folding is display-only. |
78
+ | `subagent_stop` | never | SIGTERM a running run's process group. |
79
+
80
+ ## Codemode batch launches
81
+
82
+ On Pi 0.99.1 or newer, enable native codemode alongside the existing direct
83
+ tools with `defaultTools: ["+codemode"]` and `codemode.mode: "on"`. SDK sessions
84
+ must also load Pi's `createCodemodeExtension()`. This support is for the
85
+ foreground coordinator; it does not enable codemode in confined children or
86
+ change sandbox admission.
87
+
88
+ `subagent_spawn_batch` returns the same readable text to direct callers and a
89
+ structured launch receipt to codemode scripts. No text parsing is needed:
90
+
91
+ ```javascript
92
+ const receipt = await tools.subagent_spawn_batch({
93
+ batchName: "inspection",
94
+ shared: { tools: "read,bash" },
95
+ jobs: [
96
+ { role: "explorer", prompt: "Map the request handling path." },
97
+ { role: "reviewer", prompt: "Review the current changes for regressions." }
98
+ ]
99
+ });
100
+ store("inspection-launch", receipt);
101
+ text(receipt);
102
+ ```
103
+
104
+ | Field | Meaning |
105
+ |-------|---------|
106
+ | `status` | `launched`: every effective job launched; `partial`: some launched and some failed or were skipped; `not-launched`: none launched; `clarification-needed`: selection needs a human decision and nothing launched. These are launch outcomes, never completion outcomes. |
107
+ | `batchId`, `batchName` | Assigned batch identity and optional label. No batch ID is assigned for clarification. |
108
+ | `launched` | `{ job, name, id, modelNote? }` for each created run. |
109
+ | `failed` | `{ job, name, reason }` for launch failures, including later jobs not attempted after a reject-mode launch failure. |
110
+ | `skipped` | `{ job, name }` for jobs omitted because capacity was full in `launch-available` mode. |
111
+ | `message`, `choices` | Explanation and selection options on a clarification receipt. |
112
+
113
+ `job` is a 1-based position in the effective batch order, after any confirmed
114
+ role split. All three job lists are always present. Invalid input and whole-batch
115
+ capacity rejection still throw, so codemode callers must catch those errors.
116
+ Returned partial or no-launch receipts do not throw; inspect `status` and every
117
+ list instead of treating a fulfilled promise as complete success.
118
+
119
+ Await the launch call and print or retain its receipt before returning from the
120
+ script. Codemode `store()` writes persist only if the script succeeds; launched
121
+ runs are not rolled back if the script later fails, times out, or is interrupted.
122
+ Use the durable run registry to recover IDs when needed, and `subagent_stop` to
123
+ stop a run. Script cancellation is not a stop request. Collect results after
124
+ completion callbacks with `subagent_result`; do not poll or keep a script open
125
+ waiting for completion. Callbacks and capacity/catalog guarantees remain owned
126
+ by the existing harness runtime.
127
+
128
+ ## Run timing (deadline, ceiling, stuck)
129
+
130
+ Timeout control belongs to the harness, not to the prompt or the skill that
131
+ launched the run. A time limit written into a prompt is enforced by nothing, and
132
+ a parent that only notices late tends to kill a child that was still making
133
+ progress. Every run gets stall detection by default. Elapsed-time limits are
134
+ opt-in because total runtime does not distinguish a slow agent from a stalled
135
+ one.
136
+
137
+ | Control | Default | What happens |
138
+ |---------|---------|--------------|
139
+ | Soft deadline (`deadline_minutes`) | Off | When configured, the child gets one steering message (delivered through Pi's steer queue after its current tool call finishes): stop starting new work, commit or save what is done, and report what is complete, what is not, and where the work is. The parent gets one wake saying so. |
140
+ | Grace (`grace_minutes`) | 5 | Grace starts when the steering message actually reaches the child, which is after its current tool call. While the message is still waiting behind a running tool call, the deadline stop is held, so a test run that began just before the deadline is not killed mid-run. If the child has not finished by the end of grace, the harness stops it; the ordinary completion callback reports `killed; stopped: deadline`. A child that completes inside grace is not stopped; its surfaces say `deadline: finished in grace`. |
141
+ | Hard ceiling (`max_minutes`) | Off | When configured, the run is stopped at once, without grace or a steer, and reported as `stopped: ceiling`. The ceiling bounds everything, including a tool call that never ends and an orphaned run (child gone, its process group still alive). |
142
+ | Stuck window (`stuck_minutes`) | 10 | No progress (see below) for this long wakes the parent once (`stuck`). It never stops the run. After new progress, a later stuck spell wakes again. |
143
+
144
+ `0` turns a control off. `null` or an omitted value means "inherit": on
145
+ `subagent_spawn` that is the configured default; on a `subagent_spawn_batch` job
146
+ it is the `shared` value, then the configured default. Values are minutes and
147
+ may be fractional. A job's own number (including `0`) wins over `shared`.
148
+
149
+ **Known limits of the deadline hold.**
150
+
151
+ - `max_minutes: 0` turns the ceiling off, and the ceiling is the only control
152
+ that bounds a tool call that never ends. With it off, a steer stuck behind a
153
+ hung tool call is never delivered, the deadline stop stays held, and the run
154
+ is never stopped. Opt into a ceiling when a command must have a wall-clock
155
+ bound.
156
+ - The parent reads at most the last 32 MiB of a child log after `/reload`. If
157
+ the log is larger and the running tool call started before that window, the
158
+ parent cannot see the call, so the hold is lost and grace counts from when
159
+ the steer was sent. The child can be stopped mid-call.
160
+ - The child confirms delivery by matching the steer's exact text in its
161
+ conversation. If another extension rewrites user input (an input-transform
162
+ hook), the text no longer matches, no receipt is written, and grace silently
163
+ counts from when the steer was sent instead of when it arrived.
164
+
165
+ **Progress**, defined simply: any successful tool call that is not an exact
166
+ repeat of an earlier call in the same run. "Exact repeat" means the same tool
167
+ name and the same arguments, with null and absent optional fields treated alike
168
+ (as in #336) and key order ignored. A successful file-mutating tool (`edit`,
169
+ `write`, `multi_edit`, `apply_patch`, `str_replace`, `write_file`, and similarly
170
+ named tools), a `git commit`, and a success directly after a failed call always
171
+ count, even when repeated. Re-reading the same file or re-running the same command with the same
172
+ arguments does not reset the stuck window, so a read-only research child making
173
+ distinct calls is never flagged, while one looping on the same read is. The
174
+ harness remembers up to 4096 distinct calls per run as fixed-size hashes of the
175
+ name and normalized arguments; past that the oldest is forgotten, so a call from
176
+ long ago counts again. Time the child spends waiting on a running tool call does not count toward the
177
+ stuck window, so a child in the middle of a 20-minute test run is waiting, not
178
+ stuck. A command that hangs is bounded only when the caller opts into a deadline
179
+ or ceiling.
180
+ The other stuck signal is the existing escalation: the same operation failing
181
+ three times wakes the parent once as an `Action required` incident
182
+ ([failure observations](failure-observations.md)).
183
+
184
+ **Where the reason shows.** `subagent_list` rows, `subagent_output`,
185
+ `subagent_result`, and completion callbacks carry `stopped: deadline`,
186
+ `deadline: wrapping up`, `deadline: finished in grace` (completed after the
187
+ deadline), `deadline: passed` (ended some other way after the deadline, for
188
+ example it crashed or a user stopped it), `stopped: ceiling`, or `stuck`. The spawn response lists the limits in force.
189
+
190
+ **Global defaults.** Precedence is spawn parameter, then environment, then
191
+ `config.json`, then the built-in default:
192
+
193
+ | Setting | Environment | `config.json` |
194
+ |---------|-------------|---------------|
195
+ | Soft deadline | `PI_SUBAGENT_DEADLINE_MINUTES` | `deadlineMinutes` |
196
+ | Grace | `PI_SUBAGENT_GRACE_MINUTES` | `graceMinutes` |
197
+ | Hard ceiling | `PI_SUBAGENT_MAX_MINUTES` | `maxMinutes` |
198
+ | Stuck window | `PI_SUBAGENT_STUCK_MINUTES` | `stuckMinutes` |
199
+
200
+ **Durability and delivery.** The policy and its one-shot markers (steer
201
+ requested, deadline wake, stuck wake, stop reason) are stored in the run's
202
+ metadata at launch, so `/reload` keeps the deadline and never repeats a wake.
203
+ The parent checks timing on its 15-second supervision tick. The steer travels
204
+ through a request file in the run directory, read by a small harness extension
205
+ loaded into every child (it registers no tools and runs no commands). When the
206
+ message enters the child's conversation, the extension writes a receipt file
207
+ next to the request; grace starts at that time. The extension runs in the
208
+ trusted Pi process, and the child's confined tools cannot write the run
209
+ directory, so a child cannot forge a receipt. Without a receipt and with no
210
+ tool call running (for example, a child started without the extension), grace
211
+ runs from when the steer was requested. Wakes follow `callback:false` and session ownership like other callbacks.
212
+ Runs launched before this feature have no timing record and are not timed.
213
+
214
+ Output-control parameters are spelled the same across subagents and background
215
+ tasks: `max_bytes` and `lines`. The older spellings `maxBytes` and `tail_lines`
216
+ are deprecated aliases; both work, and the canonical name wins when both are
217
+ given.
218
+
219
+ ## Non-blocking, by construction
220
+
221
+ - **Process isolation.** Each run is a `detached` + `unref`'d `pi -p` process.
222
+ Its context can't clog the parent, its crash can't corrupt parent state, and
223
+ its output is durable in a log file.
224
+ - **Completions trigger one batched turn that fetches durable results.** Ordinary
225
+ terminal callbacks accumulate for 100 ms and share one
226
+ `pi.sendMessage(..., { deliverAs: "followUp", triggerTurn: true })` notification
227
+ with background-task completions from the same Pi host. Each row contains only
228
+ source, id, label, terminal status, and the `subagent_result` lookup. Full
229
+ results and logs stay in durable tools and are never embedded in the callback.
230
+ Set `PI_BETTER_CALLBACK_BATCH_MS` to `0` through `5000` milliseconds to tune
231
+ the accumulation window; invalid values use 100 ms. A failed handoff keeps the
232
+ pending records retryable, and `/reload` recovers records explicitly marked
233
+ pending. `callback:false` sends no model message; read the durable result later
234
+ with `subagent_result`.
235
+ - **The prompt guidelines forbid polling.** The foreground agent is told, in the
236
+ tool guidelines, that spawning is done and it must not loop on `output`/`result`
237
+ or sleep to wait.
238
+
239
+ ## Autonomy & safety
240
+
241
+ Every subagent is confined by default, and the confinement is **self-contained** —
242
+ it does not depend on any other extension being installed.
243
+
244
+ - **Task sandbox (default on, macOS and Linux).** Pi startup, authentication,
245
+ provider transport, and session persistence run as trusted runtime operations.
246
+ Admitted task tools run under `sandbox-exec` or Linux Bubblewrap. Project files
247
+ default to Read / write and outside files to Read, including `~/.pi`; runtime
248
+ control files remain protected. Commands use private scratch plus explicit
249
+ runtime exceptions for `/tmp` and the current user's macOS temporary and MDS
250
+ cache directories when Outside is Read or Read/write. Credential and control
251
+ denials retain precedence. See the permission details below.
252
+ - **Tool allowlist.** Confined children admit only verified read/write/edit/bash
253
+ implementations, the guarded `apply_patch`, and the trusted tools ticked in
254
+ `/sandbox`. Other requested tools are reported as unavailable, with the reason.
255
+ - **No runaway recursion.** Confined children cannot spawn nested agents.
256
+ `allow_nested:true` loads nested-agent support only for unconfined children.
257
+
258
+ ### Write sandbox
259
+
260
+ The shared `sandbox-core` and `task-sandbox` modules enforce task operations in
261
+ both Main and Subagents. The defaults are Main Off and Subagents On. Subagents
262
+ use Project files = Write & delete and Outside project = **Write**: tasks write
263
+ across home and temp (tool caches such as `~/.gradle` and `~/.npm` just work), but
264
+ outside the workspace they can only remove or rename files in temp, hidden home
265
+ entries, and worktree folders (`.worktrees/`, `*-worktrees/`). Credential files
266
+ default to readable but not writable, so npm, git over SSH, and authenticated
267
+ CLIs can reuse their normal user configuration. This exposes known file-based
268
+ credentials to confined subagents; set Stored credentials to Off to hide them.
269
+ Shell startup files, `~/.pi`, `~/.claude`,
270
+ `~/.agents`, and harness state cannot be written, removed, or renamed. Rename-based
271
+ saves and git commits in a sibling repository fail under Write; set Outside
272
+ project to Write & delete when a task needs that. Linux uses a stricter fallback
273
+ where ordinary home folders are read-only. See
274
+ [pi-better-sandbox](../../pi-better-sandbox/README.md) and
275
+ [ADR 0008](../../../docs/adr/0008-write-without-delete.md). Commands and network
276
+ default On. A subagent's project root is its selected workspace (`cwd` / `sandbox_dir`),
277
+ or its disposable clone. `/sandbox off` changes Main, not Subagents.
278
+
279
+ Pi runs as the trusted runtime so it can take settings/authentication locks,
280
+ connect to its provider, and persist sessions. A mandatory guard loads before
281
+ the task can run. Its immutable launch snapshot controls shell commands and
282
+ kernel-confined file workers. Task access to `~/.pi` follows the outside and
283
+ credential-file rules; there are no writable lock exceptions. Runtime code,
284
+ configuration, and control files remain protected from task writes.
285
+
286
+ Commands Off still permits file tools according to their file permissions.
287
+ Network Off blocks task network access while Pi's provider transport remains
288
+ available. The confined file worker has an 8 MiB file limit and rejects larger
289
+ files explicitly; use confined commands for larger files when commands are On.
290
+
291
+ `read`, `write`, `edit`, and `bash` have verified adapters. Extension tools are
292
+ chosen in `/sandbox` → Subagents · Tools
293
+ ([ADR 0009](../../../docs/adr/0009-guarded-and-trusted-subagent-tools.md)):
294
+
295
+ - **Guarded:** `apply_patch`, a harness adapter with the Codex tool's name,
296
+ schema and patch format. Every add, update, move and delete goes through the
297
+ guarded file operations, so the Project files / Outside project levels apply:
298
+ Write refuses deletes and moves outside disposable places, Read refuses
299
+ writes, credential files and protected paths are refused. The whole patch is
300
+ checked first; no file is left half-patched, and a failure part-way reports
301
+ what was and wasn't applied. It is added for a child that may `edit` or
302
+ `write`, or asks for it.
303
+ - **Trusted:** third-party tools you tick (default `web_fetch` and `web_search`).
304
+ Their package is loaded into the child and they run in the child Pi process,
305
+ **outside the file rules**. The child admits one only when both its name and
306
+ its package match the ticked entry, so another package registering the same
307
+ name is refused. A single extension file with no package manifest is loaded
308
+ and admitted by itself, never its directory. Known network tool names
309
+ (`web_fetch`, `web_search`, `firecrawl_scrape`, `firecrawl_extract`, `mcp`, `mcpScript`, `remote_bash`, and any `mcp__*` name) are refused while Network access is Off; this is a name list, not a
310
+ network sandbox. A ticked tool whose package can't be found is refused at launch. Ticked tools
311
+ join the default tool list; an explicit `tools` list still decides.
312
+
313
+ Other tools are disabled in confined children and listed in launch output with
314
+ the reason. The launch line shows what was admitted, for example
315
+ `Runtime: isolated · guarded apply_patch · trusted web_fetch (@juicesharp/rpiv-web-tools)`.
316
+ Provider extensions are trusted runtime code; arbitrary extension tools are not
317
+ admitted merely because their names were requested. Confined children disable project
318
+ runtime configuration and inherited extension discovery. Startup or backend
319
+ failure never falls back to an unconfined child.
320
+
321
+ You still start Pi normally. The subagent package's internal launcher requires
322
+ Pi SDK 0.82.1 or newer; it is not a replacement user-facing Pi command.
323
+
324
+ ### Git-mutating subagents and linked worktrees
325
+
326
+ A sandboxed subagent that will mutate Git should set **`git_clone_workspace:true`**
327
+ on `subagent_spawn`. The parent prepares a fresh, self-contained Git clone whose
328
+ `.git/` directory lives **inside the sandbox writable root**, then runs the child
329
+ in that clone.
330
+
331
+ Why this matters: a linked Git worktree (created with `git worktree add`) has a
332
+ `.git` file that points back to administrative state under the main repository,
333
+ typically outside the sandbox directory. A sandbox that only allows writes under
334
+ the worktree directory therefore cannot support normal Git producer operations
335
+ such as fetch, rebase, commit, and push — the child stalls or fails when Git
336
+ tries to write metadata it cannot reach. `git_clone_workspace:true` avoids this
337
+ by cloning the repository with a real `.git/` directory inside the writable root.
338
+
339
+ The clone uses:
340
+
341
+ ```
342
+ git clone --reference-if-able <local-reference-repo> --dissociate \
343
+ <remote-url> <sandbox-workspace>
344
+ ```
345
+
346
+ `--reference-if-able` borrows local objects from the parent repository during
347
+ setup; `--dissociate` removes the alternates link afterwards, so the clone is
348
+ self-contained and safe to delete. The clone source prefers the source
349
+ workspace's upstream remote URL (`origin` when set) so the disposable
350
+ workspace's `origin` points at the real remote rather than the parent working
351
+ tree — pushes therefore target upstream, not the sandboxed parent. The local
352
+ repository is used only as a reference (and as a content fallback when no
353
+ remote is configured). Source remotes are re-synced after clone. The
354
+ checked-out branch/commit matches the source workspace at spawn time.
355
+ Repo-local Git identity settings from the source (`user.name`, `user.email`,
356
+ and `user.signingkey` when set) are copied into the clone so ordinary commits
357
+ work without reconfiguring identity inside the disposable workspace.
358
+
359
+ If the source workspace is a linked worktree, the clone is prepared from the
360
+ main repository's object database and the requested branch/commit; the child is
361
+ never launched into the structurally broken linked-worktree sandbox. If clone
362
+ preparation fails, the spawn fails fast with a message explaining that the
363
+ linked-worktree Git metadata is outside the sandbox and recommending
364
+ `git_clone_workspace:true`.
365
+
366
+ The mandatory task guard is independent of optional guardrails extensions.
367
+ Adding a package to the tool map does not make its tools safe to execute outside
368
+ the task boundary.
369
+
370
+ ## Tool scoping (allowlist)
371
+
372
+ Precedence, highest first: the per-call `tools` param → `config.json`
373
+ `defaultTools` → a built-in default (`read, bash, edit, write, web_search,
374
+ web_fetch`; just `read, bash` in a `clean` child). `exclude_tools` subtracts on
375
+ top.
376
+
377
+ `config.json` (next to the extension) also sets:
378
+
379
+ - `defaultModel` — model for spawns that don't specify one (`null` = inherit the
380
+ foreground model).
381
+ - `delegationMode` — foreground policy (`manual`, `adaptive`, or `coordinator`;
382
+ default `adaptive`). `/subagents mode ...` changes only the current session.
383
+ - `maxConcurrent` — how many subagents may run at once (**default 4**). A spawn
384
+ past the cap is rejected until a running one finishes.
385
+
386
+ ## The allowlist also decides what LOADS
387
+
388
+ For confined children, the requested list is first restricted to tools with
389
+ verified task adapters and the trusted tools ticked in `/sandbox`; only the
390
+ ticked tools' packages load. `toolExtensions` still chooses which package to
391
+ load for a tool (an override), but the child admits it only if that is the
392
+ ticked package. Provider extensions can still load as trusted runtime
393
+ dependencies. Nested spawning and
394
+ inherited extension discovery are currently unavailable under confinement.
395
+
396
+ The mapping behavior below applies to unconfined children and to admitted
397
+ runtime dependencies:
398
+
399
+ The `tools` allowlist does double duty: it is both what the child may call **and**
400
+ which extension *code* is loaded into it. A child launches as
401
+
402
+ ```
403
+ pi -p --mode json --no-extensions -e <package backing a requested tool> ...
404
+ ```
405
+
406
+ so a package that backs no requested tool never loads. With the default
407
+ allowlist (`read, bash, edit, write, web_search, web_fetch`) exactly one package
408
+ loads — the web-tools one — and `web_fetch` works normally.
409
+
410
+ Two maps in `config.json` drive it:
411
+
412
+ - `toolExtensions` — tool name → package(s) providing it. Built-ins (`read`,
413
+ `bash`, `edit`, `write`) need no entry.
414
+ - `providerExtensions` — provider → auth package. Model auth is not tool-shaped:
415
+ `xai/grok-4.5` needs `pi-xai-oauth` loaded whatever tools it was granted.
416
+
417
+ Ask for a tool with no mapping and the spawn still succeeds, but says so at
418
+ launch — the tool simply will not exist in the child.
419
+
420
+ `clean:true` is the narrowest case of the same mechanism: no extensions at all.
421
+ For unconfined children, `allow_nested:true` loads this package into the child;
422
+ without it, nested spawning is unavailable. Confined children disable nesting
423
+ regardless of this flag until a verified adapter exists.
424
+
425
+ `inheritExtensions: true` in `config.json` restores the old load-everything
426
+ behavior. It is **operator-only** — no spawn parameter can reach it, so the child
427
+ model cannot widen its own runtime. It also re-exposes the failure below.
428
+
429
+ ### Why: a subagent that loads everything can die mid-turn reporting success
430
+
431
+ Loading every installed package means inheriting their startup side effects. A
432
+ package that replaces builtin `bash` with a `detached` + `unref()` spawn breaks
433
+ `pi -p`: on a parallel `bash` + `read` batch the in-process `read` finishes, the
434
+ unref'd `bash` doesn't hold the event loop, Node drains, and the child **exits 0
435
+ mid-turn** — no `tool_execution_end`, no `agent_end`. Historically, exit 0 was
436
+ indistinguishable from a clean finish, so all 17 observed mid-turn exits were
437
+ reported as ✓ completed. Finalization now requires terminal agent evidence and
438
+ no unmatched tool starts. Lifecycle validation classifies runs as `complete`,
439
+ `incomplete_no_terminal_event`, `incomplete_open_tools`, `failed_exit`, or
440
+ `killed`; incoherent exit-0 streams are recorded as failed with named lifecycle
441
+ diagnostics on `subagent_result` and attention wording on completion callbacks.
442
+
443
+ A tool allowlist alone cannot fix this. `--tools` restricts what the model may
444
+ *call*; the package already overrode builtin `bash` at startup, so the `bash` in
445
+ your allowlist **is** the broken one. Measured with the default 6 tools:
446
+
447
+ | runtime | tool starts / ends | terminal event |
448
+ |---|---|---|
449
+ | all extensions loaded | 2 / 1 | none — exits 0 mid-turn |
450
+ | `--no-extensions -e <web-tools>` | 3 / 3 | `agent_settled`, `web_fetch` OK |
451
+
452
+ A package *denylist* isn't expressible either: pi has only `-e <path>` (add one)
453
+ and `--no-extensions` (all off) — there is no "load all except X" flag. Naming
454
+ what you want is the only mechanism that excludes anything, and it excludes
455
+ future offenders too, with no name to keep updated.
456
+
457
+ ### Known incompatibility: `pi-patty-bg-tasks`
458
+
459
+ **`pi-patty-bg-tasks` (tested at 1.1.6) is incompatible with subagents and must
460
+ not be loaded into a child.** It is the package that produced the failure above:
461
+ it replaces builtin `bash` and spawns `detached` + `proc.unref()`
462
+ (`src/spawn.ts`), which in print mode drains the event loop mid-turn. Bisected
463
+ against all 18 installed packages — alone it reproduces; every other package
464
+ alone is fine.
465
+
466
+ The default configuration already excludes it, structurally, because it backs no
467
+ requested tool. You only re-expose it by setting `inheritExtensions: true`, or by
468
+ mapping a tool to it in `toolExtensions`. Don't.
469
+
470
+ A proper fix belongs upstream — preserve builtin `bash` semantics when overriding
471
+ it, keep foreground subprocesses referenced until the tool promise settles, and
472
+ put genuinely detached work behind a separate background-task tool.
473
+
474
+ ## Status & cost tracking
475
+
476
+ Driven by the child's `--mode json` usage events:
477
+
478
+ - **Live widget** above the editor while any subagent runs — a spinner per run
479
+ with elapsed time, the current tool, and running token/cost spend, ticking once
480
+ a second. It clears itself when the last run finishes. (TUI/RPC only; silent in
481
+ `-p`/print mode.)
482
+ - **On demand** — `subagent_list`, `subagent_output`, and `subagent_result` are
483
+ bounded current-session pages (see `docs/issue-312-output.md`). They omit
484
+ ordered tool-name sequences and default token/cost lines. Pass
485
+ `include: ["cost"]` to `subagent_output` or `subagent_result` for one spend
486
+ line (`spend: 1.2k tok (↑400 ↓800) · $0.0034`) and `include: ["tools"]` for
487
+ one tool-call line (`tools: 7 calls · distinct: bash, read, edit`). The human
488
+ toast may still include elapsed + spend; the model-facing callback does not.
489
+ - **Folded result display.** In interactive TUI sessions, `subagent_result`
490
+ renders a compact preview by default so long child answers do not flood the
491
+ transcript. Clicking the tool row, or using the row expand action, shows the
492
+ full bounded result. This is a display concern only: the tool still returns
493
+ the complete bounded `content` payload to the model.
494
+
495
+ Spend is summed from each finalized assistant turn's `usage` (so multi-turn
496
+ tool-using runs total correctly), and cost comes straight from the model's
497
+ reported per-request cost.
498
+
499
+ ## Subagent navigator (TUI)
500
+
501
+ In an interactive TUI session, a **subagent navigator** lets you inspect and
502
+ organize runs without asking the model to call a tool. The running-subagents
503
+ widget can be focused from an empty input line for quick actions; detail output
504
+ opens in the overlay. Print/RPC modes do not install the navigator; tool access
505
+ is unchanged in every mode.
506
+
507
+ ### Open
508
+
509
+ - With the editor **empty** and at least one non-dismissed current-parent run
510
+ still running, press `←` to focus the main-window subagent list above the
511
+ input line.
512
+ - If the editor contains text, `←` keeps normal cursor-left behavior.
513
+ - While running runs exist, the default footer shows `← subagents · N`. The
514
+ live widget also includes a secondary `← to navigate` hint on its title line
515
+ for terminals that do not render the default footer status. The hint clears
516
+ when no non-dismissed current-parent run is still running.
517
+ - While the main-window list is focused, the title hint changes to
518
+ `Enter to view · x to stop`; the selected row is marked with `›`. Press
519
+ `↓` from the bottom row to return to the input line.
520
+ - The Subagents lane pins an informational `main` row above child runs. It
521
+ shows the foreground model, effort, active tool, context tokens, and active
522
+ elapsed time. It is not selectable and can never become an `x` stop target.
523
+ - `↑` moves to the previous row when multiple running rows are shown. `↓`
524
+ moves toward the input line, returning to normal input from the bottom row.
525
+ - `Enter` opens the selected run's live detail view. `x` stops the selected
526
+ running run using shared `subagent_stop` semantics and dismisses it from the
527
+ navigator.
528
+
529
+ ### Detail view
530
+
531
+ - Detail uses the same command-sheet treatment, with section rules for the
532
+ inspector groups and command bars at the top and bottom of the view.
533
+ - Tool-call log rows wrap within the detail width instead of truncating. Source
534
+ row boundaries and indentation are retained, with long paths and JSON values
535
+ preferring delimiter breaks before a hard wrap. Header and metadata geometry
536
+ is unchanged.
537
+ - Shows status (colorized), model/effort, elapsed, tools, spend, and parsed
538
+ output, plus sectioned health: process identity/liveness, activity,
539
+ compaction, active tool, model call/error, last log write, thresholds, and
540
+ callback notification timestamps. Compaction, active tool, and model state
541
+ are separate sections. The view refreshes about once per second while open.
542
+ - The transcript shows up to the latest 25 rows by default; `l` switches
543
+ between 25 and 10 rows. The row count is a cap, not a guarantee: the metadata
544
+ lines (provider, id, model, elapsed, tools, spend, pid, pgid) always stay
545
+ visible, and on a short terminal the transcript shows only as many of its
546
+ newest rows as fit below them. `subagent_result` and `subagent_output` page
547
+ sizes are unaffected.
548
+ - Detail opened from the main-window list closes back to the main page; the
549
+ main-window selection remains on the viewed run when it is still visible.
550
+ - `x` arms Stop for a running run, or Dismiss for a terminal run.
551
+ - `esc` closes the detail view and returns to the main page.
552
+
553
+ ### Two-press `x` stop/dismiss
554
+
555
+ - First `x` on the selected (list) or viewed (detail) run arms the action for three
556
+ seconds and shows a footer hint: `x again to stop <name>` while running, or
557
+ `x again to dismiss <name>` when terminal.
558
+ - Second `x` within the window, on the **same** run, acts:
559
+ - **Running** — stop the process group (shared `subagent_stop` semantics),
560
+ mark killed, then dismiss from the navigator.
561
+ - **Terminal** — dismiss only; terminal status is not rewritten.
562
+ - Changing selection, leaving the view, closing the overlay, arming timeout,
563
+ reload, and session teardown all disarm close and clear the confirm hint.
564
+
565
+ ### Dismissal is navigator-only
566
+
567
+ Dismissed runs leave the navigator list and footer count. Logs, prompt, session
568
+ data, metadata, and id-based tool access stay intact. `subagent_list`,
569
+ `subagent_output`, `subagent_result`, and `subagent_stop` still resolve dismissed
570
+ run ids. Dismissal survives `/reload` as dismissed in the navigator.
571
+
572
+ ### Reload and teardown
573
+
574
+ `/reload` and session restart reinstall the empty-editor wrapper without stacking
575
+ duplicate handlers, republish the footer count, and clear any leftover overlay
576
+ timers or close-confirm state. Session shutdown disposes open navigator timers
577
+ and clears navigator footer statuses (TUI only).
578
+
579
+ ## Design notes
580
+
581
+ - Runtime lives outside any repo, under `$TMPDIR/pi-better-subagents/`
582
+ (`runs/<id>/` holds `output.log`, `prompt.md`, `meta.json`; `sessions/` holds
583
+ child session state). The `meta.json` sidecar is authoritative, so `list` /
584
+ `output` / `result` survive turns, `/reload`, and pi restarts.
585
+ - The child runs `--mode json`; `subagent_result` / `subagent_output` **parse**
586
+ the event stream and return just the bounded final answer (no tool-name history).
587
+ Non-JSON banner/warning lines fail to parse and are dropped, so the result is
588
+ clean. The prompt is passed as a **positional argument**, never `@file` — some
589
+ models refuse an @-attached file as untrusted content.
590
+ - The child gets **only** the explicit prompt — no silent parent-context bleed.
591
+ - `--approve` is **off by default** (headless runs can't prompt for trust).
592
+
593
+
594
+ ## Parent-process scoping
595
+
596
+ The live widget, default `subagent_list`, concurrency cap, and `session_start` ticker only include runs this pi process spawned (`spawnPid === process.pid`). The on-disk registry stays machine-global for durability. Default `subagent_list` is current-session, newest first, 10 compact rows / 1 KiB. Pass `limit:N` (max 100) or `max_bytes` (max 4 KiB) for an explicit larger page. Pass `all:true` for a global / foreign-session view. Pass `status:[...]` to filter by effective status: `running`, `completed`, `failed`, `killed`, transient `exited`, or durable `orphaned` / `lost`. Id-based `subagent_result` / `subagent_output` default to the current session; pass `all:true` to read a foreign-session id. Unknown ownership is a gap, not “not found”; when the current session identity cannot be read, no run's ownership is treated as verified and `all:true` is required. List pages are reached with the returned `nextCursor` (every run is reachable, not only the first 100). Runs whose metadata is missing or corrupt are reported as gaps and counted on lists, never as nonexistent.
597
+
598
+ ## Supervision health (`orphaned` / `lost`)
599
+
600
+ While a current-parent run is `running` or `orphaned`, a periodic health tick
601
+ reconciles process-group evidence only (see `docs/adr/0002-process-group-only-subagent-health.md`):
602
+
603
+ - **`orphaned`** — direct supervision of the child is broken, but related
604
+ process-group work may still be alive. Non-terminal and non-final; operationally
605
+ unhealthy immediately. The coordinator (when `callback:true`) gets one durable
606
+ ATTENTION follow-up naming `subagent_result` / `subagent_output` / `subagent_stop`
607
+ so it can inspect artifacts and decide whether to wait, stop, or retry. Human
608
+ `ui.notify` still fires when `callback:false`.
609
+ - **`lost`** — no related process remains and no coherent terminal completion was
610
+ observed. Terminal with unknown outcome (not the same as `failed`). Same one-shot
611
+ ATTENTION follow-up + diagnostic `subagent_result` path with best-available
612
+ artifacts.
613
+
614
+ Orphaned/lost callbacks use the same non-interrupting
615
+ `{ deliverAs: "followUp", triggerTurn: true }` mechanics, but bypass the ordinary
616
+ completion accumulation window and retain distinct ATTENTION wording. Per-status
617
+ markers on `meta.json` (`orphanedCallbackSentAt` / `lostCallbackSentAt`) are written
618
+ only after a successful handoff and dedupe across reloads and repeated health ticks.
619
+ Persisted unmarked orphaned/lost states are recovered on the health ticker after
620
+ `/reload` even when process evidence does not produce a fresh transition.
621
+
622
+ Pi's optional `followUpMode: all` remains useful when completion groups arrive
623
+ after an earlier 100 ms batch has already flushed: Pi can consume those queued
624
+ follow-ups in one later agent turn. This extension does not change Pi core or
625
+ Pi's default follow-up mode.
626
+
627
+ ### Surfacing health (tools + passive widget)
628
+
629
+ Multi-dimensional observations (`stale`, long tool, compacting / long compaction,
630
+ model error/retry, plus process `orphaned` / `lost`) are computed from durable
631
+ status + child-event evidence and surfaced on existing paths without a parallel
632
+ health model:
633
+
634
+ - **`subagent_list`** — durable `orphaned` / `lost` status brackets; degraded
635
+ compact facts only when actionable. Healthy/quiet rows stay on the compact format.
636
+ - **`subagent_output` / `subagent_result`** — `[health: …]` diagnostics for
637
+ orphaned, lost, and degraded running runs; #65 orphaned/lost result bodies kept.
638
+ - **Passive live widget** — healthy/quiet unchanged; degraded (and orphaned) may
639
+ show a short suffix. Still `setWidget` only — never focusable.
640
+ - **`callback:false`** suppresses coordinator follow-up only; human `ui.notify` and
641
+ TUI/passive visibility remain.
642
+
643
+ ## Install
644
+
645
+ Linux sandboxing requires the system `bubblewrap` package. Install it before
646
+ launching sandboxed children:
647
+
648
+ ```bash
649
+ # Debian/Ubuntu
650
+ sudo apt-get install bubblewrap
651
+ # Fedora/RHEL
652
+ sudo dnf install bubblewrap
653
+ # Arch Linux
654
+ sudo pacman -S bubblewrap
655
+ ```
656
+
657
+ When `bwrap` is absent, explicit sandbox requests fail with an installation hint;
658
+ default-on sandboxing preserves the documented direct-execution fallback. Once
659
+ `bwrap` is selected, a launch failure fails closed rather than retrying the child
660
+ without confinement.
661
+
662
+ Symlink the project into pi's auto-discovered extensions dir:
663
+
664
+ ```bash
665
+ ln -sfn "$PWD" ~/.pi/agent/extensions/pi-better-subagents
666
+ ```
667
+
668
+ Then `/reload` (or restart pi). It appears as `pi-better-subagents`.
669
+
670
+ > Do **not** add it to `settings.json`'s `extensions` array — a live pi session
671
+ > rewrites that file on its own saves and drops hand-added entries. Auto-discovery
672
+ > via the symlink is stable. Quick throwaway test without installing:
673
+ > `pi -e ./index.ts`.
674
+
675
+ ## Tests
676
+
677
+ Unit tests (`node --test tests/*.test.mjs`) cover the pure logic — widget
678
+ rendering, completion delivery, and extension resolution.
679
+
680
+ Real integration smoke tests live in [`tests/`](tests/) — a subagent using
681
+ `web_fetch`, one driving `gh`, env inheritance through the sandbox, and headless
682
+ isolation surviving a parallel `bash` + `read` batch. See
683
+ [`tests/README.md`](tests/README.md).
684
+
685
+ ## Roadmap
686
+
687
+ Tracked in [issues](https://github.com/1aboveio/pi-better-subagents/issues).
688
+ Near-term:
689
+
690
+ - Guarantee subagent autonomy — verify/deny any child→parent supervision
691
+ back-channel so a child can never block on the parent ([#1](https://github.com/1aboveio/pi-better-subagents/issues/1)).
692
+ - Make `callback:true` a lightweight trigger instead of embedding the full result
693
+ twice ([#2](https://github.com/1aboveio/pi-better-subagents/issues/2)) — **done**.
694
+ - Named agent-definition files (per-agent system prompt + tool allowlist) and
695
+ chain/parallel orchestration.
@@ -103,7 +103,8 @@ import {
103
103
  } from "./health.ts";
104
104
  import {
105
105
  assignBatchJobNames,
106
- formatBatchLaunchResponse,
106
+ batchLaunchOutputSchema,
107
+ batchLaunchResult,
107
108
  mergeJobOptions,
108
109
  nextBatchId,
109
110
  planBatchLaunches,
@@ -1902,7 +1903,12 @@ export default function (pi: ExtensionAPI) {
1902
1903
  description:
1903
1904
  "Launch several independent background pi subagents at once. Each job becomes a " +
1904
1905
  "normal subagent run with its own run id, process, log, and metadata. " +
1905
- "'shared' options are applied to every job; per-job options override them.",
1906
+ "'shared' options are applied to every job; per-job options override them. " +
1907
+ "Codemode callers receive a structured launch receipt: status, batchId when assigned, " +
1908
+ "and launched, failed, skipped arrays with 1-based effective job positions (after any confirmed role split). Inspect all arrays; " +
1909
+ "a returned receipt is not proof that every job launched or completed. " +
1910
+ "Await the launch call and preserve its receipt. Script cancellation does not stop launched runs; use subagent_stop.",
1911
+ outputSchema: batchLaunchOutputSchema,
1906
1912
  promptSnippet: "Launch a batch of background subagents at once",
1907
1913
  promptGuidelines: [
1908
1914
  "When delegation is permitted by the active mode, use subagent_spawn_batch for several independent assigned tasks. It returns immediately with a batch id and one run id per launched job.",
@@ -1993,7 +1999,17 @@ export default function (pi: ExtensionAPI) {
1993
1999
  p.jobs.map((job) => mergeJobOptions(p.shared, job) as CatalogJobFields),
1994
2000
  { hasUI: catalogHost.hasUI === true, select: catalogHost.select },
1995
2001
  );
1996
- if (clarified.status === "clarification-needed") return catalogResult(clarified) as ReturnType<typeof text>;
2002
+ if (clarified.status === "clarification-needed") return {
2003
+ ...catalogResult(clarified),
2004
+ structuredContent: {
2005
+ status: "clarification-needed",
2006
+ message: clarified.message,
2007
+ choices: [...clarified.choices],
2008
+ launched: [],
2009
+ failed: [],
2010
+ skipped: [],
2011
+ },
2012
+ };
1997
2013
  p.jobs = clarified.jobs as typeof p.jobs;
1998
2014
  validateBatchPlan({ shared: undefined, jobs: p.jobs, onCapacity: p.onCapacity, config: cfg });
1999
2015
  }
@@ -2023,9 +2039,9 @@ export default function (pi: ExtensionAPI) {
2023
2039
 
2024
2040
  const names = assignBatchJobNames(p.jobs);
2025
2041
  const batchId = nextBatchId();
2026
- const launched: { name: string; id: string; modelNote?: string }[] = [];
2027
- const failed: { name: string; reason: string }[] = [];
2028
- const skipped: { name: string }[] = [];
2042
+ const launched: { job: number; name: string; id: string; modelNote?: string }[] = [];
2043
+ const failed: { job: number; name: string; reason: string }[] = [];
2044
+ const skipped: { job: number; name: string }[] = [];
2029
2045
  // How many reject-mode reserved slots are still held (not yet committed/released).
2030
2046
  let reservedRemaining = launchAvailable ? 0 : p.jobs.length;
2031
2047
 
@@ -2039,7 +2055,7 @@ export default function (pi: ExtensionAPI) {
2039
2055
  if (launchAvailable) {
2040
2056
  if (!gate.tryReserve(1, maxConcurrent)) {
2041
2057
  for (let j = i; j < p.jobs.length; j++) {
2042
- skipped.push({ name: names[j] });
2058
+ skipped.push({ job: j + 1, name: names[j] });
2043
2059
  }
2044
2060
  break;
2045
2061
  }
@@ -2064,12 +2080,12 @@ export default function (pi: ExtensionAPI) {
2064
2080
  const { id, modelNote } = await spawnSubagentRun(ctx, { ...merged, name }, { batchId, batchName: p.batchName });
2065
2081
  gate.commit(1);
2066
2082
  if (!launchAvailable) reservedRemaining -= 1;
2067
- launched.push({ name, id, ...(modelNote ? { modelNote } : {}) });
2083
+ launched.push({ job: i + 1, name, id, ...(modelNote ? { modelNote } : {}) });
2068
2084
  } catch (err) {
2069
2085
  gate.release(1);
2070
2086
  if (!launchAvailable) reservedRemaining -= 1;
2071
2087
  const reason = err instanceof Error ? err.message : String(err);
2072
- failed.push({ name, reason });
2088
+ failed.push({ job: i + 1, name, reason });
2073
2089
  if (!launchAvailable) {
2074
2090
  // reject mode: leave already-launched runs running, release any
2075
2091
  // still-held later reservations, and report every later job as failed.
@@ -2079,13 +2095,14 @@ export default function (pi: ExtensionAPI) {
2079
2095
  }
2080
2096
  for (let j = i + 1; j < p.jobs.length; j++) {
2081
2097
  failed.push({
2098
+ job: j + 1,
2082
2099
  name: names[j],
2083
2100
  reason: "not launched due to earlier job failure in reject mode",
2084
2101
  });
2085
2102
  }
2086
- return text(formatBatchLaunchResponse({
2103
+ return batchLaunchResult({
2087
2104
  batchId, batchName: p.batchName, launched, skipped, failed,
2088
- }));
2105
+ });
2089
2106
  }
2090
2107
  // launch-available: failure did not consume a slot — continue so
2091
2108
  // later jobs can use remaining capacity (backfill).
@@ -2098,7 +2115,7 @@ export default function (pi: ExtensionAPI) {
2098
2115
  reservedRemaining = 0;
2099
2116
  }
2100
2117
 
2101
- return text(formatBatchLaunchResponse({ batchId, batchName: p.batchName, launched, skipped, failed }));
2118
+ return batchLaunchResult({ batchId, batchName: p.batchName, launched, skipped, failed });
2102
2119
  },
2103
2120
  });
2104
2121
 
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-subagents",
3
- "version": "0.9.3",
3
+ "version": "0.10.0",
4
4
  "description": "Pi extension for detached, sandboxed subagent runs that keep the foreground session free.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -47,6 +47,7 @@
47
47
  "config.json",
48
48
  "README.md",
49
49
  "roles/**/*.md",
50
+ "docs/usage.md",
50
51
  "docs/agent-catalog.md",
51
52
  "docs/agent-catalog-operations.md",
52
53
  "docs/agent-model-resolution.md",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "pi-better-harness",
3
- "version": "0.13.0",
3
+ "version": "0.13.1",
4
4
  "description": "Pi extension bundle for a write sandbox, subagents, background tasks, SSH, goals, and structured plans.",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -57,7 +57,7 @@
57
57
  "pi-better-plan": "0.5.1",
58
58
  "pi-better-sandbox": "0.8.0",
59
59
  "pi-better-ssh": "0.1.1",
60
- "pi-better-subagents": "0.9.3",
60
+ "pi-better-subagents": "0.10.0",
61
61
  "smol-toml": "1.9.0",
62
62
  "yaml": "^2.9.1"
63
63
  },