@fyeeme/pi-review 1.0.0 → 1.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -29,6 +29,9 @@ From the package dir:
29
29
  npm install
30
30
  ```
31
31
 
32
+ This resolves [`@fyeeme/pi-subagent-core`](https://www.npmjs.com/package/@fyeeme/pi-subagent-core)
33
+ (`^0.3.0`, from the npm registry — no sibling-repo layout requirement).
34
+
32
35
  Then point pi at it (e.g. via your extensions config), or symlink into your pi
33
36
  extensions directory.
34
37
 
@@ -63,6 +66,17 @@ This is the deterministic mode selection a pure-prompt skill cannot reproduce
63
66
  `ctx.getContextUsage()`). The decision is announced in the trigger message so
64
67
  it is observable.
65
68
 
69
+ **Apply → verify → revert safety net** (harden-code-simplify): after Phase 2
70
+ applies the cleanups, the handler also injects a verification command detected
71
+ from `package.json` scripts (`check` → `test` → `lint` → `typecheck`). The
72
+ skill snapshots the touched files, applies the fixes, runs that command, and
73
+ on failure auto-reverts per-file (a clean apply runs verify exactly once; only
74
+ a failure escalates to one verify per touched file). The result is reported as
75
+ structured outcomes via `review_report` (`level: "simplify"`), not a free-text
76
+ summary. If no verification command is detectable, fixes are kept but the
77
+ report states no verification was run (verification is opportunistic, never
78
+ blocking).
79
+
66
80
  ### `subagent` tool
67
81
 
68
82
  An LLM-callable tool that spawns one or more real pi subprocesses:
@@ -70,9 +84,20 @@ An LLM-callable tool that spawns one or more real pi subprocesses:
70
84
  | mode | behavior |
71
85
  |---|---|
72
86
  | `single` | run `prompts[0]` once (e.g. an independent verify agent) |
73
- | `parallel` | run all prompts concurrently, capped at 8 (e.g. one finder per angle) |
87
+ | `parallel` | run all prompts concurrently, capped at the ceiling (e.g. one finder per angle) |
74
88
  | `chain` | run sequentially; each later prompt receives prior output |
75
89
 
90
+ **Fan-out guards** (harden-code-simplify, shared with `/code-review`):
91
+
92
+ - **Recursion cap (whitelist-by-default)** — a spawned sub-agent does not
93
+ receive the `subagent` tool in its default toolset, so it cannot recurse. A
94
+ caller opts in by listing `subagent` in the child's `tools` whitelist; set
95
+ `PI_SUBAGENT_MAX_SPAWN_DEPTH` to allow multi-level fan-out up to a hard cap.
96
+ - **Default turn budget** — fan-out agents get a finite default `maxTurns` (25)
97
+ when the caller omits it; an explicit `0` is honored.
98
+ - **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default 8;
99
+ invalid values fall back to the default).
100
+
76
101
  Each sub-agent is a full `pi --mode json -p --no-session` run. Progress streams
77
102
  to the TUI via `onUpdate` as each agent completes. ESC aborts the whole batch
78
103
  (SIGTERM → 5s → SIGKILL per subprocess). Errors are thrown (not returned) so
@@ -85,8 +110,6 @@ pi-review/
85
110
  ├── index.ts factory: registerTool(subagent) + 2 commands
86
111
  ├── skills/ bundled SKILL.md files (code-review, simplify)
87
112
  ├── src/
88
- │ ├── agent/dispatch.ts spawnAgent + mapWithConcurrencyLimit (self-contained copy
89
- │ │ from pi-dynamic-workflows; no external dep beyond node + pi-ai)
90
113
  │ ├── skills.ts bundledSkillPath — resolve this extension's own skills/ dir
91
114
  │ ├── tools/subagent.ts defineTool("subagent") — generic capability layer
92
115
  │ └── commands/
@@ -96,15 +119,18 @@ pi-review/
96
119
  ```
97
120
 
98
121
  The layout is deliberately layered: `src/tools/` is the **generic capability
99
- layer** (subagent tool + dispatch), `src/commands/` is the **entry layer** (one
122
+ layer** (subagent tool), `src/commands/` is the **entry layer** (one
100
123
  file per skill). If a third or fourth skill needs the subagent tool, `src/tools/`
101
124
  can be split into its own `pi-subagent` extension with zero refactor — the code
102
125
  is already separated.
103
126
 
104
- `src/agent/dispatch.ts` is a self-contained copy of the spawn pattern from
105
- `examples/extensions/subagent` and `pi-dynamic-workflows/src/agent/dispatch.ts`
106
- (~150 lines). When pi promotes `spawnAgent` to a public `pi-coding-agent`
107
- export, this file should be deleted in favor of that import.
127
+ The dispatch primitive (`spawnAgent`, `mapWithConcurrencyLimit`,
128
+ `createSpawnRegistry`, `abortAgent`, `getPiInvocation` + types) lives in
129
+ [`pi-subagent-core`](../pi-subagent-core) (npm `@fyeeme/pi-subagent-core`), a
130
+ shared library extracted from the duplicated copies that used to live here and
131
+ in `pi-dynamic-workflows`. When pi promotes `spawnAgent` to a public
132
+ `pi-coding-agent` export, `pi-subagent-core` should be deleted in favor of that
133
+ import.
108
134
 
109
135
  ## Relation to the skills
110
136
 
package/index.ts CHANGED
@@ -4,6 +4,9 @@
4
4
  * Registers:
5
5
  * - the `subagent` tool — general-purpose parallel/sequential sub-agent fan-out
6
6
  * via real pi subprocesses. Shared capability used by both skills below;
7
+ * - the `review_report` tool — structured findings sink for the code-review
8
+ * skill (Pi's counterpart to CC's ReportFindings): renders the Markdown
9
+ * report + writes JSON to <cwd>/.pi/review/ for CI;
7
10
  * - the `/code-review` command — effort-level review via the code-review skill;
8
11
  * - the `/code-simplify` command — cleanup via the simplify skill; the handler
9
12
  * decides parallel vs single-pass from ctx.getContextUsage(), mirroring CC's
@@ -13,16 +16,25 @@
13
16
  * provides the entry commands + the fan-out capability they need.
14
17
  *
15
18
  * Layout (layered so the tool layer can be split into its own extension later):
16
- * src/tools/subagent.ts — generic capability (subagent tool + dispatch)
17
- * src/commands/*.ts — per-skill entry commands
19
+ * src/tools/subagent.ts — generic capability (subagent tool; dispatch from pi-subagent-core)
20
+ * src/tools/review_report.ts — structured findings sink (review_report tool; CC ReportFindings counterpart)
21
+ * src/commands/*.ts — per-skill entry commands
18
22
  */
19
23
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
24
+ import { isFanoutToolAllowed } from "@fyeeme/pi-subagent-core";
20
25
  import { registerCodeReview } from "./src/commands/code-review.ts";
21
26
  import { registerSimplify } from "./src/commands/code-simplify.ts";
22
27
  import { subagentTool } from "./src/tools/subagent.ts";
28
+ import { reviewReportTool } from "./src/tools/review_report.ts";
23
29
 
24
30
  export default function (pi: ExtensionAPI): void {
25
- pi.registerTool(subagentTool);
31
+ // The fan-out tool registers only when recursion is allowed for THIS
32
+ // process (top-level, or a child the spawner explicitly opted in AND that is
33
+ // below the max-depth cap). A default child — spawned without the fan-out
34
+ // tool in its whitelist — loads without it, so it physically cannot recurse.
35
+ // This is the whitelist-by-default recursion guard (harden-code-simplify).
36
+ if (isFanoutToolAllowed()) pi.registerTool(subagentTool);
37
+ pi.registerTool(reviewReportTool);
26
38
  registerCodeReview(pi);
27
39
  registerSimplify(pi);
28
40
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@fyeeme/pi-review",
3
- "version": "1.0.0",
3
+ "version": "1.0.1",
4
4
  "description": "Review & cleanup extension for pi. Registers /code-review and /code-simplify commands plus a general-purpose `subagent` tool that spawns parallel pi subprocesses — providing the real fan-out capability the code-review and simplify skills (bundled under `skills/`) need for their multi-agent flows. The /code-simplify handler uses ctx.getContextUsage() to decide parallel vs single-pass mode deterministically.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -34,16 +34,21 @@
34
34
  "test": "vitest --run",
35
35
  "typecheck": "tsc"
36
36
  },
37
+ "dependencies": {
38
+ "@fyeeme/pi-subagent-core": "^0.3.2"
39
+ },
37
40
  "peerDependencies": {
38
- "@earendil-works/pi-ai": ">=0.77.0",
39
- "@earendil-works/pi-coding-agent": ">=0.77.0",
41
+ "@earendil-works/pi-ai": ">=0.84.1",
42
+ "@earendil-works/pi-coding-agent": ">=0.84.1",
43
+ "@earendil-works/pi-tui": ">=0.84.1",
40
44
  "jiti": ">=2.0.0",
41
45
  "typebox": ">=1.0.0",
42
46
  "typescript": ">=5.0.0"
43
47
  },
44
48
  "devDependencies": {
45
- "@earendil-works/pi-ai": "0.77.0",
46
- "@earendil-works/pi-coding-agent": "0.77.0",
49
+ "@earendil-works/pi-ai": "0.84.1",
50
+ "@earendil-works/pi-coding-agent": "0.84.1",
51
+ "@earendil-works/pi-tui": "0.84.1",
47
52
  "@types/node": "22.19.19",
48
53
  "jiti": "2.7.0",
49
54
  "typebox": "1.1.38",
@@ -41,9 +41,13 @@ description: "Review the current diff for correctness bugs and reuse/simplificat
41
41
  Pi ADAPTATIONS (differ from the CC runtime)
42
42
  ════════════════════════════════════════════════════════════════════════
43
43
  1. Output — CC calls a ReportFindings tool with {level, findings}; Pi
44
- PRINTS a Markdown findings table + details block as text
45
- (no such tool on Pi). [was JSON array; switched for
46
- readability]
44
+ uses this extension's `review_report` tool (the Pi counterpart
45
+ to ReportFindings, verdict/outcome enums aligned to CC
46
+ v2.1.226): it renders the Chinese Markdown report (table +
47
+ details) back to the conversation AND writes a
48
+ machine-readable JSON to <cwd>/.pi/review/ for CI / --fix /
49
+ --comment. If the tool is absent, fall back to printing the
50
+ Markdown as text.
47
51
  2. Fan-out — CC uses the Agent tool; Pi uses the `subagent` tool
48
52
  (mode: parallel), or runs angles sequentially if unavailable.
49
53
  3. Verify — CC uses the Agent tool; Pi uses `subagent` for the
@@ -75,6 +79,8 @@ altitude, and conventions findings when the output cap forces a cut.
75
79
  | high | **recall** — catch every real bug a careful reviewer would; **err on the side of surfacing** | independent agent | more angles | ≤ 10 |
76
80
  | xhigh → max | recall + **gap-hunt** | independent agent | above + 1 fresh gap finder | larger, may include uncertain |
77
81
 
82
+ **max 与 xhigh 结构相同**:fan-out / verify / sweep 完全一致,差别仅在模型 reasoning effort(CC v2.1.226 注释实证:`max → same structure as xhigh (the API reasoning effort differs, not the fan-out)`)。若运行时不支持调节 reasoning effort,max 在结构上退化为 xhigh——不要因档名而期待更多 fan-out。
83
+
78
84
  Each finder surfaces **up to 6 candidate findings** with `file`, `line`, a
79
85
  one-line `summary`, and a concrete `failure_scenario`.
80
86
 
@@ -277,48 +283,38 @@ At **high and below**, skip Phase 3.
277
283
 
278
284
  ## Output
279
285
 
280
- Print the findings as a **Markdown table + a details block** — readable in a
281
- terminal and in rendered Markdown (no JSON, no ReportFindings tool on Pi). Cap =
282
- low's min(files_changed, 4); 8 at medium; 10 at high; larger at xhigh → max.
283
-
284
- **全部用中文输出**:表头、概述、场景一律用中文;`Verdict`、`Category` 作为标识符保留
285
- 英文 token(CONFIRMED / PLAUSIBLE;correctness / reuse …)。
286
-
287
- **1. 表头行** — 单行写明:力度、diff 命令/范围、改动文件数、命中条数、是否真的多智能体
288
- 并发(见下文 Single-pass honesty)。示例:
289
-
290
- > `max` · `git diff HEAD` · 29 个文件 · 3 条发现 · 多智能体(验证 + 查漏)
291
-
292
- **2. 发现汇总表** — 按严重程度从高到低,每条一行:
293
-
294
- | # | 判定 | 类别 | 位置 | 概述 |
295
- |---|------|------|------|------|
296
- | 1 | CONFIRMED | correctness | path/file.ext:123 | 一句话说明这个 bug |
297
- | 2 | PLAUSIBLE | reuse | path/file.ext:45 | … |
298
-
299
- - `判定` — `CONFIRMED` / `PLAUSIBLE` / 留空(未做验证)。
300
- - `类别` — 产生该发现的角度,短横线小写 slug(`correctness`、`simplification`、
301
- `efficiency`、`reuse`、`altitude`、`conventions`,或更具体的如 `test-coverage`)。
302
- - `位置` — `文件:行号`。
303
- - `概述` — 一句话说明(≤ 约 80 字,同时作为紧凑标签)。
304
-
305
- **3. 详情块** — 与表格同序;场景放不进单元格,在这里展开:
306
-
307
- **1. path/file.ext:123 — 类别** *(判定)*
308
- 概述:<一句话>
309
- 场景:<具体的输入/状态 → 错误输出/崩溃;若是清理类发现,写明具体代价——重复了什么、
310
- 浪费了什么、哪里更难维护,或违反了哪条规则>
311
-
312
- If more than `{cap}` survive, keep the `{cap}` most severe (correctness outranks
313
- cleanup/altitude/conventions when cutting). If nothing survives, print the header
314
- line with count 0 and skip the table and details — don't emit an empty table.
315
-
316
- ### Single-pass honesty
317
-
318
- If this review did not actually fan out — low effort, or medium+ where the
319
- `subagent` tool was unavailable so the angles ran sequentially in one context —
320
- state clearly in the header line that this was a single-pass review done without the
321
- multi-agent fan-out, so whoever reads it isn't misled about what actually ran.
286
+ Report the findings via the `review_report` tool (this extension's counterpart
287
+ to CC's `ReportFindings`) — call it **once** with
288
+ `{ level, target, files_changed, fanned_out, findings }`, findings ranked
289
+ most-severe first (empty array if nothing survived verification). The tool
290
+ renders the Chinese Markdown report (table + details) back to the conversation
291
+ AND writes a machine-readable JSON to `<cwd>/.pi/review/` for CI / `--fix` /
292
+ `--comment`. Do **not** also hand-write the Markdown table.
293
+
294
+ Each finding in the array carries: `file`, `line` (optional), `category`
295
+ (`correctness` / `reuse` / `simplification` / `efficiency` / `altitude` /
296
+ `conventions`, or a more specific slug like `test-coverage`), `verdict`
297
+ (`CONFIRMED` / `PLAUSIBLE` / `REFUTED`), `summary` (one line, Chinese), and
298
+ `failure_scenario` (concrete input/state → wrong output/crash; for cleanup
299
+ findings, the concrete cost — Chinese). When re-reporting after applying
300
+ `--fix`, set `outcome` on each finding (`fully_achieved` / `mostly_achieved` /
301
+ `partially_achieved` / `not_achieved` / `unclear_from_transcript` — mirrored
302
+ verbatim from CC's `ReportFindings`, v2.1.226).
303
+
304
+ Cap = low's min(files_changed, 4); 8 at medium; 10 at high; larger at
305
+ xhigh → max. If more than `{cap}` survive, send the `{cap}` most severe
306
+ (correctness outranks cleanup/altitude/conventions when cutting). If nothing
307
+ survives, send an empty `findings` array — the tool prints a zero-count header.
308
+
309
+ **全部用中文**:`summary` 与 `failure_scenario` 一律中文;`verdict`、`category`、
310
+ `outcome` 作为标识符保留英文 token。
311
+
312
+ **`fanned_out` 诚实** — 准确设置:仅当多智能体 fan-out 真的跑起来(subagent
313
+ finder + verify agent)才为 `true`;low effort 或任何单遍/自审降级为 `false`。该
314
+ 字段会出现在报告表头,让读者不被误导(替代旧的 Single-pass honesty 小节)。
315
+
316
+ **降级** — 若 `review_report` 工具未注册(这份 SKILL.md 跑在 pi-review 扩展之外),
317
+ 退回到直接打印 Markdown 表格 + 详情块文本;不要报错。
322
318
 
323
319
  ---
324
320
 
@@ -329,8 +325,13 @@ findings to the working tree instead of stopping at the report: fix each one
329
325
  directly — correctness bugs and reuse/simplification/efficiency cleanups alike.
330
326
  Skip any finding whose fix would change intended behavior, require changes well
331
327
  outside the reviewed diff, or that you judge to be a false positive — note the
332
- skip rather than arguing with it. Finish with a brief summary of what was fixed
333
- and what was skipped.
328
+ skip rather than arguing with it. Then call `review_report` once more to
329
+ re-report, setting `outcome` on each finding (`fully_achieved` / `mostly_achieved`
330
+ / `partially_achieved` / `not_achieved` / `unclear_from_transcript` — skipped
331
+ ones are `not_achieved` or `unclear_from_transcript`). This structured re-report
332
+ replaces the hand-written summary and makes the fix result machine-consumable.
333
+ If `review_report` is unavailable, fall back to a brief text summary of what was
334
+ fixed and what was skipped.
334
335
 
335
336
  ## Posting comments (--comment)
336
337
 
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: simplify
3
- description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that. v2 (from Claude Code CLI v2.1.223) — 4 cleanup agents fan out in parallel when context allows, else a single-pass inline cleanup; either way the fixes are applied to the working tree."
3
+ description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that. v2 (from Claude Code CLI v2.1.223) — 4 cleanup agents fan out in parallel when context allows, else a single-pass inline cleanup; either way the fixes are applied, verified against the project's check command, and auto-reverted on failure, then reported as structured outcomes via review_report."
4
4
  ---
5
5
 
6
6
  <!--
@@ -38,7 +38,9 @@ description: "Review the changed code for reuse, simplification, efficiency, and
38
38
  3. Command — CC: /simplify; Pi: /code-simplify.
39
39
 
40
40
  Prerequisite: the `subagent` tool (provided by the pi-review extension) for
41
- PARALLEL MODE. SINGLE-PASS MODE runs standalone.
41
+ PARALLEL MODE, and the `review_report` tool (same extension) for
42
+ the Phase 2 structured outcome report. SINGLE-PASS MODE runs
43
+ standalone apart from `review_report`.
42
44
  -->
43
45
 
44
46
  You are improving the quality of the changed code, not hunting for bugs. Review
@@ -96,14 +98,9 @@ bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
96
98
  deep enough — prefer generalizing the underlying mechanism over adding special
97
99
  cases.
98
100
 
99
- ## Phase 2 — Apply the fixes
101
+ ## Phase 2 — Apply, verify, and report
100
102
 
101
- Wait for all four agents to complete, dedup findings that point at the same line
102
- or mechanism, and fix each remaining one directly. Skip any finding whose fix
103
- would change intended behavior, require changes well outside the reviewed diff,
104
- or that you judge to be a false positive — note the skip rather than arguing
105
- with it. Finish with a brief summary of what was fixed and what was skipped (or
106
- confirm the code was already clean).
103
+ Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). The parallel fan-out only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: true` in the report since the 4-agent fan-out actually ran.
107
104
 
108
105
  ---
109
106
 
@@ -145,13 +142,106 @@ bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
145
142
  deep enough — prefer generalizing the underlying mechanism over adding special
146
143
  cases.
147
144
 
148
- ## Phase 2 — Apply the fixes
145
+ ## Phase 2 — Apply, verify, and report
149
146
 
150
- Dedup findings that point at the same line or mechanism, and fix each remaining
151
- one directly. Skip any finding whose fix would change intended behavior, require
152
- changes well outside the reviewed diff, or that you judge to be a false positive
153
- — note the skip rather than arguing with it. Finish with a brief summary of what
154
- was fixed and what was skipped (or confirm the code was already clean). State
155
- clearly in your summary that this was a single-pass review done without the
156
- `subagent` tool, not the full 4-agent fan-out, so whoever reads it isn't misled
157
- about what actually ran.
147
+ Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). Single-pass vs parallel only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: false` in the report so a reader is not misled into thinking the 4-agent fan-out ran.
148
+
149
+ ---
150
+
151
+ # Phase 2 — Apply, verify, and report (shared by both modes)
152
+
153
+ Dedup findings that point at the same line or mechanism first. Then apply,
154
+ verify, and report. This safety net is what distinguishes `/code-simplify` from
155
+ a blind cleanup: a finding is only "done" once it is applied AND the project
156
+ still verifies — otherwise it is reverted.
157
+
158
+ ## Step 1 — Snapshot the baseline
159
+
160
+ Before applying any fix, snapshot every file you are about to edit so a failed
161
+ verification can be reverted cleanly. For each touched file, copy its current
162
+ content into a temp dir:
163
+
164
+ ```
165
+ mkdir -p /tmp/pi-simplify-baseline/$(dirname <file>)
166
+ cp <file> /tmp/pi-simplify-baseline/<file>
167
+ ```
168
+
169
+ `$(dirname <file>)` keeps the target's parent dir (e.g. `src/`) inside the
170
+ baseline — a bare `cp <file> /tmp/pi-simplify-baseline/<file>` fails with ENOENT
171
+ for any file in a subdirectory. If a fix CREATES a new file, record its path so
172
+ Step 3a can remove it on rollback (it has no baseline entry).
173
+
174
+ This baseline captures the working-tree state **including** the user's
175
+ uncommitted changes — reverting to it undoes only `/code-simplify`'s fixes,
176
+ never the user's diff. Do **not** use `git checkout` / `git restore` to revert:
177
+ that would discard the user's intended changes too.
178
+
179
+ ## Step 2 — Apply the fixes
180
+
181
+ Apply each surviving finding directly. Skip any finding whose fix would change
182
+ intended behavior, require changes well outside the reviewed diff, or that you
183
+ judge to be a false positive — note the skip (it will be reported as
184
+ `not_achieved`).
185
+
186
+ ## Step 3 — Verify, branching on the result
187
+
188
+ Run the verification command the handler injected in the trigger message (e.g.
189
+ `npm run check`), then branch:
190
+
191
+ - **No verification command was detected** → keep the applied changes, mark each
192
+ applied finding `fully_achieved`, and say in the report that NO verification
193
+ was run. Verification is opportunistic — never block on its absence.
194
+ - **Verification passes** → keep the changes; applied findings are
195
+ `fully_achieved` (or `mostly_achieved` / `partially_achieved` if a fix only
196
+ partly addressed the issue).
197
+ - **Verification fails** → the working tree is verified-broken; go to Step 3a.
198
+
199
+ ### Step 3a — Auto-revert (hybrid granularity, only on failure)
200
+
201
+ 1. Revert ALL touched files from the Step 1 baseline (working-tree parent dirs
202
+ already exist, so copying back is safe):
203
+ ```
204
+ cp /tmp/pi-simplify-baseline/<file> <file>
205
+ ```
206
+ 2. Remove any files the fixes CREATED (they have no baseline entry and would
207
+ otherwise survive the rollback).
208
+ 3. Re-apply ONE file's findings at a time, running the verification command
209
+ after each file. Keep only files whose verification passes; revert any file
210
+ whose verification fails back to its baseline.
211
+ 4. If NO file passes on its own, leave everything reverted and mark every
212
+ finding `not_achieved` — a clean tree is the safe outcome, not a broken one.
213
+
214
+ This caps the cost: the common case (clean apply) runs verification exactly
215
+ once; only a failure escalates to one verification per touched file.
216
+
217
+ ## Step 4 — Report via `review_report`
218
+
219
+ Call the `review_report` tool **once** with `level: "simplify"` and one finding
220
+ entry per cleanup, ranked most-severe first. Each entry carries `file`, `line`
221
+ (optional), `category` (`reuse` / `simplification` / `efficiency` /
222
+ `altitude`), `summary` (one line, Chinese), `failure_scenario` (the concrete
223
+ cost — Chinese), and `outcome`:
224
+
225
+ - `fully_achieved` — applied and verification passed (or no verification command
226
+ existed and the change was kept).
227
+ - `mostly_achieved` / `partially_achieved` — applied but only partly addresses
228
+ the issue.
229
+ - `not_achieved` — skipped, or reverted by the auto-revert in Step 3a.
230
+ - `unclear_from_transcript` — could not determine.
231
+
232
+ Do **not** write a free-text summary as the primary record — the structured
233
+ `review_report` call IS the summary (it renders the report AND writes JSON to
234
+ `<cwd>/.pi/review/` for CI). If `review_report` is unavailable, fall back to a
235
+ brief text summary listing each finding's outcome.
236
+
237
+ Set `fanned_out` honestly in the call: `true` only if the 4-agent fan-out
238
+ (subagent) actually ran; `false` for single-pass. The report header shows this
239
+ so a reader is not misled about what ran.
240
+
241
+ ## Step 5 — Clean up
242
+
243
+ ```
244
+ rm -rf /tmp/pi-simplify-baseline
245
+ ```
246
+
247
+ Remove the baseline snapshots once the report is delivered.
@@ -1,4 +1,6 @@
1
1
  import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
2
+ import * as fs from "node:fs";
3
+ import * as path from "node:path";
2
4
  import { bundledSkillPath } from "../skills.ts";
3
5
 
4
6
  /** Context fraction at which we fall back to single-pass — a Pi-specific heuristic (see decideSimplifyMode). */
@@ -6,6 +8,35 @@ const CONTEXT_NEAR_FULL_THRESHOLD = 0.8;
6
8
 
7
9
  export type SimplifyMode = "parallel" | "single-pass";
8
10
 
11
+ /** Priority order for picking a verification command from package.json scripts. */
12
+ const VERIFY_SCRIPT_PRIORITY = ["check", "test", "lint", "typecheck"] as const;
13
+
14
+ /**
15
+ * Pick the project verification command from a package.json `scripts` map, in
16
+ * priority order (check → test → lint → typecheck). Pure — unit-testable.
17
+ * Returns the runnable command (e.g. `npm run check`) or null when none exists.
18
+ */
19
+ export function detectVerifyCommand(scripts: Record<string, string> | null): string | null {
20
+ if (!scripts) return null;
21
+ for (const key of VERIFY_SCRIPT_PRIORITY) {
22
+ const v = scripts[key];
23
+ if (typeof v === "string" && v.trim() !== "") return `npm run ${key}`;
24
+ }
25
+ return null;
26
+ }
27
+
28
+ /** Read package.json scripts from `cwd`; returns null when absent/unparseable. */
29
+ function readScriptsAt(cwd: string): Record<string, string> | null {
30
+ try {
31
+ const pkg = JSON.parse(fs.readFileSync(path.join(cwd, "package.json"), "utf8")) as {
32
+ scripts?: Record<string, string>;
33
+ };
34
+ return pkg.scripts ?? null;
35
+ } catch {
36
+ return null;
37
+ }
38
+ }
39
+
9
40
  /**
10
41
  * Decide simplify mode deterministically from real context usage + tool availability.
11
42
  * Pure function — unit-testable.
@@ -48,18 +79,20 @@ export function registerSimplify(pi: ExtensionAPI): void {
48
79
  contextWindow: usage?.contextWindow ?? 0,
49
80
  hasSubagent,
50
81
  });
51
- const pct =
52
- usage && usage.tokens != null && usage.contextWindow > 0
53
- ? Math.round((usage.tokens / usage.contextWindow) * 100) + "%"
54
- : "?";
82
+ const pct = usage && usage.percent != null ? `${Math.round(usage.percent)}%` : "?";
55
83
  const bodyLabel = mode === "parallel" ? "PARALLEL MODE" : "SINGLE-PASS MODE";
84
+ const verifyCmd = detectVerifyCommand(readScriptsAt(ctx.cwd));
85
+ const verifyLine = verifyCmd
86
+ ? `Verification command: \`${verifyCmd}\` (detected from package.json scripts). After applying Phase 2 fixes, run it; on failure, follow the skill's auto-revert procedure — never leave the working tree verified-broken.`
87
+ : `No verification command detected in package.json (looked for check/test/lint/typecheck). Apply fixes and report outcomes, but state in the report that no verification was run (verification is opportunistic, never blocking).`;
56
88
  pi.sendUserMessage(
57
89
  `Clean up the changed code now. Target: ${args || "(whole diff)"}.\n\n` +
58
90
  `Handler decided ${mode} mode (context ${pct} full, subagent ${hasSubagent ? "available" : "absent"}). ` +
59
91
  `Load ${bundledSkillPath("simplify/SKILL.md")} via the read tool and follow the ${bodyLabel} body. ` +
60
92
  (mode === "parallel"
61
93
  ? `Use the \`subagent\` tool (mode: parallel) for the 4-agent fan-out.`
62
- : `Work the four angles inline — do not fake fan-out.`),
94
+ : `Work the four angles inline — do not fake fan-out.`) +
95
+ `\n${verifyLine}`,
63
96
  );
64
97
  },
65
98
  });
@@ -0,0 +1,225 @@
1
+ /**
2
+ * src/tools/review_report.ts — the `review_report` LLM tool.
3
+ *
4
+ * Structured findings sink for the code-review skill — Pi's counterpart to
5
+ * CC's native `ReportFindings` tool (verified in CC v2.1.226 binary: "Report
6
+ * code-review findings as a typed list so the host UI can render them"). Pi has
7
+ * no host finding-renderer, so this tool does double duty: it renders a tidy
8
+ * Chinese Markdown report (table + details) back to the conversation AND writes
9
+ * a machine-readable JSON (findings + level + outcome) to
10
+ * `<cwd>/.pi/review/<id>.json` so CI / --fix / --comment can consume it.
11
+ *
12
+ * `verdict` (CONFIRMED/PLAUSIBLE/REFUTED) and `outcome` (5-state) enums follow
13
+ * the CC ReportFindings shape (outcome values copied from the CC binary). The
14
+ * code-review skill drops REFUTED findings before reporting, so that value is
15
+ * accepted by the schema but rarely seen in practice.
16
+ */
17
+ import { defineTool, getMarkdownTheme } from "@earendil-works/pi-coding-agent";
18
+ import { Markdown } from "@earendil-works/pi-tui";
19
+ import { Type } from "typebox";
20
+ import * as fs from "node:fs";
21
+ import * as path from "node:path";
22
+
23
+ // --- enums following the CC ReportFindings shape ----------------------------
24
+
25
+ const Verdict = Type.Union([Type.Literal("CONFIRMED"), Type.Literal("PLAUSIBLE"), Type.Literal("REFUTED")]);
26
+
27
+ /** CC ReportFindings `outcome` 5 档(v2.1.226 二进制实证)。re-report after --fix 时填。 */
28
+ const Outcome = Type.Union([
29
+ Type.Literal("fully_achieved"),
30
+ Type.Literal("mostly_achieved"),
31
+ Type.Literal("partially_achieved"),
32
+ Type.Literal("not_achieved"),
33
+ Type.Literal("unclear_from_transcript"),
34
+ ]);
35
+
36
+ const Level = Type.Union([
37
+ Type.Literal("low"),
38
+ Type.Literal("medium"),
39
+ Type.Literal("high"),
40
+ Type.Literal("xhigh"),
41
+ Type.Literal("max"),
42
+ // simplify reuses this tool for structured apply-outcome reporting
43
+ // (harden-code-simplify). Not a review effort level — carries no verdict.
44
+ Type.Literal("simplify"),
45
+ ]);
46
+
47
+ // --- schema -----------------------------------------------------------------
48
+
49
+ const FindingParams = Type.Object({
50
+ file: Type.String({ description: "相对仓库根的文件路径。" }),
51
+ line: Type.Optional(Type.Number({ description: "行号(1-based)。省略表示文件级。" })),
52
+ category: Type.String({
53
+ description:
54
+ "产生该发现的角度 slug:correctness / reuse / simplification / efficiency / altitude / conventions(或更具体如 test-coverage)。",
55
+ }),
56
+ verdict: Type.Optional(Verdict),
57
+ summary: Type.String({ description: "一句话说明(≤80字),同时作紧凑标签。中文。" }),
58
+ failure_scenario: Type.String({
59
+ description:
60
+ "具体场景:输入/状态 → 错误输出/崩溃;清理类发现写明具体代价(重复/浪费/更难维护/违反哪条规则)。中文。",
61
+ }),
62
+ outcome: Type.Optional(Outcome),
63
+ });
64
+
65
+ const ReviewReportParams = Type.Object({
66
+ level: Level,
67
+ target: Type.Optional(Type.String({ description: "审查目标(diff 命令/范围,或 PR/分支/路径),用于报告表头。" })),
68
+ files_changed: Type.Optional(Type.Number({ description: "改动文件数,用于报告表头。" })),
69
+ fanned_out: Type.Optional(
70
+ Type.Boolean({ description: "是否真的多智能体并发(Single-pass honesty)。false/省略表示单遍自审。" }),
71
+ ),
72
+ findings: Type.Array(FindingParams, {
73
+ description: "已验证、去重、按严重度从高到低排序的发现列表(most-severe first)。空数组表示无发现存活。",
74
+ }),
75
+ });
76
+
77
+ interface ReviewReportDetails {
78
+ level: string;
79
+ findingsCount: number;
80
+ /** 结构化 JSON 落盘路径;落盘失败时为 null(仍返回渲染报告)。 */
81
+ outFile: string | null;
82
+ }
83
+
84
+ // --- render -----------------------------------------------------------------
85
+
86
+ interface FindingInput {
87
+ file: string;
88
+ line?: number;
89
+ category: string;
90
+ verdict?: string;
91
+ summary: string;
92
+ failure_scenario: string;
93
+ outcome?: string;
94
+ }
95
+ interface ReportInput {
96
+ level: string;
97
+ target?: string;
98
+ files_changed?: number;
99
+ fanned_out?: boolean;
100
+ findings: FindingInput[];
101
+ }
102
+
103
+ function fmtLoc(f: { file: string; line?: number }): string {
104
+ return f.line != null ? `${f.file}:${f.line}` : f.file;
105
+ }
106
+
107
+ /** Escape a value for a GFM table cell: backslash-escape pipes and collapse
108
+ * newlines. Free-text fields (summary/category/verdict/loc) are LLM-provided
109
+ * and routinely contain `||`, `|`, regex, or shell pipes that would otherwise
110
+ * split the row into extra columns and break the whole summary table. */
111
+ function escapeCell(v: string): string {
112
+ return v.replace(/\|/g, "\\|").replace(/\r?\n/g, " ");
113
+ }
114
+
115
+ /** 渲染中文 Markdown 报告(表头行 + 汇总表 + 详情块),格式与原 SKILL.md 教的一致。 */
116
+ function renderReport(p: ReportInput): string {
117
+ const lines: string[] = [];
118
+ const fanLabel = p.fanned_out === true ? "多智能体" : p.fanned_out === false ? "单遍自审" : "未标注";
119
+ const targetStr = (p.target ?? "(whole diff)").replace(/`/g, "\\`"); // backtick inside the inline-code cell would close it early
120
+ const filesStr = p.files_changed != null ? `${p.files_changed} 个文件` : "文件数未标注";
121
+ lines.push(`\`${p.level}\` · \`${targetStr}\` · ${filesStr} · ${p.findings.length} 条发现 · ${fanLabel}`);
122
+ lines.push("");
123
+
124
+ if (p.findings.length === 0) {
125
+ lines.push("(无发现存活验证。)");
126
+ return lines.join("\n");
127
+ }
128
+
129
+ lines.push("| # | 判定 | 类别 | 位置 | 概述 |");
130
+ lines.push("|---|------|------|------|------|");
131
+ for (let i = 0; i < p.findings.length; i++) {
132
+ const f = p.findings[i]!;
133
+ lines.push(`| ${i + 1} | ${escapeCell(f.verdict ?? "")} | ${escapeCell(f.category)} | ${escapeCell(fmtLoc(f))} | ${escapeCell(f.summary)} |`);
134
+ }
135
+ lines.push("");
136
+ lines.push("**详情**");
137
+ lines.push("");
138
+ p.findings.forEach((f, i) => {
139
+ const v = f.verdict ? ` *(${f.verdict})*` : "";
140
+ const out = f.outcome ? `\n修复结果:\`${f.outcome}\`` : "";
141
+ lines.push(`**${i + 1}. ${fmtLoc(f)} — ${f.category}**${v}`);
142
+ lines.push(`概述:${f.summary}`);
143
+ lines.push(`场景:${f.failure_scenario}${out}`);
144
+ lines.push("");
145
+ });
146
+ return lines.join("\n").trimEnd();
147
+ }
148
+
149
+ // --- tool -------------------------------------------------------------------
150
+
151
+ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewReportDetails>({
152
+ name: "review_report",
153
+ label: "Report review findings",
154
+ description:
155
+ "Report code-review findings as a typed list — Pi's counterpart to CC's ReportFindings. Use this only when the active code-review instructions tell you to report findings with this tool. Call it once with the verified findings ranked most-severe first (empty array if nothing survived verification) and do not also print the findings as text — the tool renders a tidy Chinese Markdown report back to the conversation AND writes a machine-readable JSON to <cwd>/.pi/review/ for CI / --fix / --comment. When re-reporting after applying fixes, set `outcome` on each finding. 上报结构化 code-review 发现(CC ReportFindings 的 Pi 对等物)。",
156
+ promptSnippet: "review_report — report structured code-review findings (renders Markdown + writes JSON for CI)",
157
+ promptGuidelines: [
158
+ "After verify + dedup, call `review_report` once with { level, findings } (most-severe first; empty array if none survived). Do not also hand-write the Markdown table — this tool renders it.",
159
+ "On re-report after --fix, set each finding's `outcome` (fully_achieved / mostly_achieved / partially_achieved / not_achieved / unclear_from_transcript).",
160
+ "Use this tool only when the code-review skill instructs reporting findings; otherwise follow the active output format.",
161
+ ],
162
+ parameters: ReviewReportParams,
163
+
164
+ async execute(toolCallId, params, _signal, _onUpdate, ctx) {
165
+ const report = renderReport(params);
166
+
167
+ let outFile: string | null = null;
168
+ let writeError: string | null = null;
169
+ const now = new Date();
170
+ try {
171
+ const dir = path.join(ctx.cwd, ".pi", "review");
172
+ await fs.promises.mkdir(dir, { recursive: true });
173
+ const safeId = toolCallId.replace(/[^\w.-]+/g, "_");
174
+ const ts = now.toISOString().replace(/[:.]/g, "-");
175
+ const fp = path.join(dir, `${ts}-${safeId}.json`);
176
+ await fs.promises.writeFile(
177
+ fp,
178
+ JSON.stringify(
179
+ {
180
+ level: params.level,
181
+ target: params.target ?? null,
182
+ filesChanged: params.files_changed ?? null,
183
+ fannedOut: params.fanned_out ?? null,
184
+ generatedAt: now.toISOString(),
185
+ findings: params.findings,
186
+ },
187
+ null,
188
+ 2,
189
+ ),
190
+ { encoding: "utf-8", mode: 0o600 },
191
+ );
192
+ outFile = fp;
193
+ } catch (err) {
194
+ /* 落盘失败不阻塞:仍返回渲染报告,但带上错误信息便于 CI/--fix 排障。 */
195
+ writeError = err instanceof Error ? err.message : String(err);
196
+ }
197
+
198
+ const tail = outFile
199
+ ? `\n\n[结构化发现已写入 \`${outFile}\`]`
200
+ : `\n\n[结构化落盘失败(${writeError ?? "未知原因"}),仅渲染报告]`;
201
+ const details: ReviewReportDetails = {
202
+ level: params.level,
203
+ findingsCount: params.findings.length,
204
+ outFile,
205
+ };
206
+ return {
207
+ content: [{ type: "text" as const, text: report + tail }],
208
+ details,
209
+ };
210
+ },
211
+
212
+ // Render the returned Markdown report through pi's width-aware Markdown component
213
+ // (the same path assistant text takes), not the plain-Text tool-result fallback that
214
+ // renderer-less extension tools get. Without this, the GFM table is shown as raw
215
+ // `|`/`|---|` wrapped to terminal width — no borders, no alignment.
216
+ // tool-execution.ts wraps renderResult in try/catch and falls back to plain Text on
217
+ // throw, so a failure here degrades to the pre-change behavior rather than erroring.
218
+ renderResult(result, _options, _theme, _context) {
219
+ const text = result.content
220
+ .filter((c) => c.type === "text")
221
+ .map((c) => (c.type === "text" ? c.text : ""))
222
+ .join("\n");
223
+ return new Markdown(text, 0, 0, getMarkdownTheme());
224
+ },
225
+ });
@@ -21,14 +21,33 @@ import {
21
21
  abortAgent,
22
22
  createSpawnRegistry,
23
23
  mapWithConcurrencyLimit,
24
+ parsePositiveInt,
24
25
  spawnAgent,
25
26
  type AgentSpawnRegistry,
26
27
  type AgentSpawnOptions,
27
28
  type AgentSpawnResult,
28
- } from "../agent/dispatch.ts";
29
+ } from "@fyeeme/pi-subagent-core";
29
30
 
30
- /** Cap on concurrent subprocesses (matches pi-dynamic-workflows + examples/subagent). */
31
- const MAX_CONCURRENCY = 8;
31
+ /** Default concurrency ceiling when PI_MAX_CONCURRENT_SUBAGENTS is unset/invalid. */
32
+ const DEFAULT_MAX_CONCURRENCY = 8;
33
+
34
+ /**
35
+ * Effective concurrency ceiling, configurable via PI_MAX_CONCURRENT_SUBAGENTS
36
+ * (parity with CC's CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS). Unset, missing, or
37
+ * non-positive/non-integer values fall back to the default. Read at call time
38
+ * so a changed env takes effect without a reload.
39
+ */
40
+ function getMaxConcurrency(): number {
41
+ return parsePositiveInt(process.env.PI_MAX_CONCURRENT_SUBAGENTS) ?? DEFAULT_MAX_CONCURRENCY;
42
+ }
43
+
44
+ /**
45
+ * Default turn budget for a fan-out agent when the caller omits maxTurns. A
46
+ * runaway agent cannot otherwise be bounded. Generous (well above the ~10–15
47
+ * turns the 4-angle finder/verify agents need) so legitimate work is not
48
+ * truncated; callers may override with a smaller or larger explicit value.
49
+ */
50
+ const DEFAULT_FANOUT_MAX_TURNS = 25;
32
51
 
33
52
  // Module-level registry so abortAgent can reach in-flight calls. callIds are
34
53
  // unique per tool call (toolCallId#index), so a single registry is safe.
@@ -45,10 +64,12 @@ const SubagentParams = Type.Object({
45
64
  systemPrompt: Type.Optional(Type.String({ description: "Appended to the sub-agent's system prompt." })),
46
65
  tools: Type.Optional(Type.Array(Type.String(), { description: "Tool whitelist for the sub-agent. Omit for default tools." })),
47
66
  parallelism: Type.Optional(
48
- Type.Number({ description: `Max concurrent agents in parallel mode (default min(prompts.length, ${MAX_CONCURRENCY})).` }),
67
+ Type.Number({
68
+ description: `Max concurrent agents in parallel mode (integer ≥ 1; default min(prompts.length, ceiling)). The ceiling is PI_MAX_CONCURRENT_SUBAGENTS (default ${DEFAULT_MAX_CONCURRENCY}).`,
69
+ }),
49
70
  ),
50
71
  maxTurns: Type.Optional(
51
- Type.Number({ description: "Max assistant turns per sub-agent. When reached, the subprocess is aborted. Omit for unlimited." }),
72
+ Type.Number({ description: "Max assistant turns per sub-agent. When reached, the subprocess is aborted. Omit for the default budget; 0 means abort after the first message." }),
52
73
  ),
53
74
  cwd: Type.Optional(Type.String({ description: "Working directory. Defaults to the session cwd." })),
54
75
  });
@@ -221,13 +242,21 @@ export const subagentTool = defineTool<typeof SubagentParams, SubagentDetails>({
221
242
 
222
243
  const cwd = params.cwd ?? ctx.cwd;
223
244
  const baseSystem = params.systemPrompt;
245
+ // Recursion opt-in: a child may itself spawn sub-agents ONLY when the
246
+ // caller explicitly listed the fan-out tool in the child's whitelist.
247
+ // Default (omitted, or whitelist without it) → the child loads without
248
+ // the subagent tool (see isFanoutToolAllowed in the extension entry).
249
+ const allowChildRecursion = params.tools?.includes("subagent") ?? false;
224
250
  // Per-call-agnostic subset of AgentSpawnOptions; callId/task are added per spawn.
225
251
  const baseOpts: Omit<AgentSpawnOptions, "callId" | "task"> = {
226
252
  cwd,
227
253
  model: params.model,
228
254
  tools: params.tools,
229
255
  signal,
230
- maxTurns: params.maxTurns,
256
+ // Default turn budget applies when omitted; an explicit 0 is honored by
257
+ // spawnAgent (it uses `!= null`, not truthiness) rather than treated as unset.
258
+ maxTurns: params.maxTurns ?? DEFAULT_FANOUT_MAX_TURNS,
259
+ allowChildRecursion,
231
260
  };
232
261
 
233
262
  const partial: SubagentDetails = {
@@ -287,7 +316,12 @@ export const subagentTool = defineTool<typeof SubagentParams, SubagentDetails>({
287
316
  };
288
317
 
289
318
  if (params.mode === "parallel") {
290
- const conc = Math.min(params.parallelism ?? MAX_CONCURRENCY, MAX_CONCURRENCY, prompts.length);
319
+ const ceiling = getMaxConcurrency();
320
+ // Fractional / non-positive parallelism from the model would reach
321
+ // mapWithConcurrencyLimit as `new Array(3.5)` → RangeError. Clamp to a
322
+ // sane integer instead of failing the whole tool call.
323
+ const requested = Math.floor(Math.max(1, params.parallelism ?? ceiling));
324
+ const conc = Math.min(requested, ceiling, prompts.length);
291
325
  await mapWithConcurrencyLimit(prompts, conc, (p, i) => runOne(p, i));
292
326
  } else {
293
327
  // single or chain
@@ -1,353 +0,0 @@
1
- /**
2
- * src/agent/dispatch.ts — agent dispatch via pi subprocess.
3
- *
4
- * Reuses the spawn pattern from examples/extensions/subagent and
5
- * pi-dynamic-workflows/src/agent/dispatch.ts: one `pi --mode json -p
6
- * --no-session` subprocess per agent call, stdout parsed for
7
- * {message_end, tool_result_end} events, AbortSignal → SIGTERM with a
8
- * 5s SIGKILL escalation.
9
- *
10
- * This is a SELF-CONTAINED copy (no pi-dynamic-workflows dependency): the
11
- * workflow-engine retry/skip/lifecycle machinery is stripped, leaving only
12
- * spawnAgent + mapWithConcurrencyLimit + a per-call abort registry. The
13
- * subagent tool builds its single/parallel/chain modes on top of this.
14
- *
15
- * When pi promotes spawnAgent to a public @earendil-works/pi-coding-agent
16
- * export, this file should be deleted in favor of that import.
17
- */
18
- import { spawn, type ChildProcess } from "node:child_process";
19
- import { StringDecoder } from "node:string_decoder";
20
- import * as fs from "node:fs";
21
- import * as os from "node:os";
22
- import * as path from "node:path";
23
- import type { Message } from "@earendil-works/pi-ai";
24
-
25
- /** Stable id for one agent call; the registry key for per-call abort. */
26
- export type AgentCallId = string;
27
-
28
- /** callId → per-call AbortController. */
29
- export type AgentAbortMap = Map<AgentCallId, AbortController>;
30
-
31
- // ---------------------------------------------------------------------------
32
- // Concurrency limiter (ported from examples/extensions/subagent)
33
- // ---------------------------------------------------------------------------
34
-
35
- /**
36
- * Run `fn` over `items` with at most `concurrency` in flight, preserving
37
- * input order in the output array. parallel mode builds on this.
38
- */
39
- export async function mapWithConcurrencyLimit<TIn, TOut>(
40
- items: TIn[],
41
- concurrency: number,
42
- fn: (item: TIn, index: number) => Promise<TOut>,
43
- ): Promise<TOut[]> {
44
- if (items.length === 0) return [];
45
- const limit = Math.max(1, Math.min(concurrency, items.length));
46
- const results: TOut[] = new Array(items.length);
47
- let nextIndex = 0;
48
- // Stop dispatching NEW items once any worker has errored, so a rejection
49
- // doesn't leave sibling workers pulling more items and spawning unawaited
50
- // subprocesses. In-flight calls finish; the failing worker rethrows.
51
- let failed = false;
52
- const workers = new Array(limit).fill(null).map(async () => {
53
- while (!failed) {
54
- const current = nextIndex++;
55
- if (current >= items.length) return;
56
- try {
57
- results[current] = await fn(items[current], current);
58
- } catch (err) {
59
- failed = true;
60
- throw err;
61
- }
62
- }
63
- });
64
- await Promise.all(workers);
65
- return results;
66
- }
67
-
68
- // ---------------------------------------------------------------------------
69
- // pi binary resolution (ported from examples/extensions/subagent)
70
- // ---------------------------------------------------------------------------
71
-
72
- /**
73
- * Resolve the `pi` invocation for the subprocess. Prefers re-entering the
74
- * current script (node <script> / bun <script>); falls back to the `pi`
75
- * binary on PATH when run under a generic runtime.
76
- */
77
- export function getPiInvocation(args: string[]): { command: string; args: string[] } {
78
- const currentScript = process.argv[1];
79
- const isBunVirtualScript = currentScript?.startsWith("/$bunfs/root/");
80
- if (currentScript && !isBunVirtualScript && fs.existsSync(currentScript)) {
81
- return { command: process.execPath, args: [currentScript, ...args] };
82
- }
83
-
84
- const execName = path.basename(process.execPath).toLowerCase();
85
- const isGenericRuntime = /^(node|bun)(\.exe)?$/.test(execName);
86
- if (!isGenericRuntime) {
87
- return { command: process.execPath, args };
88
- }
89
-
90
- return { command: "pi", args };
91
- }
92
-
93
- // ---------------------------------------------------------------------------
94
- // Usage + result
95
- // ---------------------------------------------------------------------------
96
-
97
- export interface AgentUsage {
98
- input: number;
99
- output: number;
100
- cacheRead: number;
101
- cacheWrite: number;
102
- cost: number;
103
- contextTokens: number;
104
- turns: number;
105
- }
106
-
107
- export interface AgentSpawnOptions {
108
- /** Stable id for this call; the registry key for per-call abort. */
109
- readonly callId: AgentCallId;
110
- /** Prompt passed as the final positional arg to `pi -p`. */
111
- readonly task: string;
112
- /** Working directory for the spawned pi process. Defaults to process.cwd(). */
113
- readonly cwd?: string;
114
- /** `--model` override. */
115
- readonly model?: string;
116
- /** `--tools` whitelist (comma-joined). */
117
- readonly tools?: string[];
118
- /** System prompt appended via a temp file (`--append-system-prompt`). */
119
- readonly systemPrompt?: string;
120
- /** Caller-level abort signal; linked to this call's per-call controller. */
121
- readonly signal?: AbortSignal;
122
- /** Max assistant turns. When reached, the subprocess is aborted (SIGTERM). Omit for unlimited. */
123
- readonly maxTurns?: number;
124
- }
125
-
126
- export interface AgentSpawnResult {
127
- callId: AgentCallId;
128
- exitCode: number;
129
- messages: Message[];
130
- stderr: string;
131
- usage: AgentUsage;
132
- model?: string;
133
- stopReason?: string;
134
- errorMessage?: string;
135
- /** True if aborted. exitCode may be null/non-zero. */
136
- aborted: boolean;
137
- /** True if killed because the caller's maxTurns budget was reached. Distinct
138
- * from `aborted` (external cancel): the agent did useful bounded work. */
139
- maxTurnsReached: boolean;
140
- }
141
-
142
- // ---------------------------------------------------------------------------
143
- // Registry — Map<callId, ChildProcess> + per-call AbortController
144
- // ---------------------------------------------------------------------------
145
-
146
- export interface AgentSpawnRegistry {
147
- /** callId → child process. The table that translates abort → SIGTERM on one process. */
148
- readonly processes: Map<AgentCallId, ChildProcess>;
149
- /** callId → per-call controller. */
150
- readonly controllers: AgentAbortMap;
151
- }
152
-
153
- export function createSpawnRegistry(): AgentSpawnRegistry {
154
- return {
155
- processes: new Map(),
156
- controllers: new Map(),
157
- };
158
- }
159
-
160
- /**
161
- * Abort exactly one in-flight call by id. Aborts the call's per-call
162
- * controller; spawnAgent's race-safe listener translates that into
163
- * SIGTERM→SIGKILL on exactly the one subprocess. Returns false if the
164
- * callId is not in flight.
165
- */
166
- export function abortAgent(registry: AgentSpawnRegistry, callId: AgentCallId): boolean {
167
- const controller = registry.controllers.get(callId);
168
- if (!controller) return false;
169
- controller.abort();
170
- return true;
171
- }
172
-
173
- // ---------------------------------------------------------------------------
174
- // Spawn — the dispatch primitive
175
- // ---------------------------------------------------------------------------
176
-
177
- export async function spawnAgent(
178
- registry: AgentSpawnRegistry,
179
- options: AgentSpawnOptions,
180
- ): Promise<AgentSpawnResult> {
181
- const { callId, task, cwd, model, tools, systemPrompt, signal } = options;
182
-
183
- // Per-call controller — the abort entry point.
184
- const controller = new AbortController();
185
- registry.controllers.set(callId, controller);
186
-
187
- // Link the caller-level signal to this call's controller so a run-wide
188
- // abort reaches every in-flight call. Named + removed in finally — otherwise
189
- // a normally-completing call leaks a listener on the parent signal.
190
- const onParentAbort = (): void => controller.abort();
191
- if (signal) {
192
- if (signal.aborted) controller.abort();
193
- else signal.addEventListener("abort", onParentAbort);
194
- }
195
-
196
- const args: string[] = ["--mode", "json", "-p", "--no-session"];
197
- if (model) args.push("--model", model);
198
- if (tools && tools.length > 0) args.push("--tools", tools.join(","));
199
-
200
- let tmpPromptDir: string | null = null;
201
- let tmpPromptPath: string | null = null;
202
-
203
- const result: AgentSpawnResult = {
204
- callId,
205
- exitCode: 0,
206
- messages: [],
207
- stderr: "",
208
- usage: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, cost: 0, contextTokens: 0, turns: 0 },
209
- aborted: false,
210
- maxTurnsReached: false,
211
- };
212
-
213
- try {
214
- if (systemPrompt && systemPrompt.trim()) {
215
- const tmp = await writePromptToTempFile(callId, systemPrompt);
216
- tmpPromptDir = tmp.dir;
217
- tmpPromptPath = tmp.filePath;
218
- args.push("--append-system-prompt", tmpPromptPath);
219
- }
220
-
221
- // The prompt is the final positional arg consumed by `-p`.
222
- args.push(task);
223
-
224
- const exitCode = await new Promise<number>((resolve) => {
225
- const invocation = getPiInvocation(args);
226
- const proc = spawn(invocation.command, invocation.args, {
227
- cwd: cwd ?? process.cwd(),
228
- shell: false,
229
- stdio: ["ignore", "pipe", "pipe"],
230
- });
231
- registry.processes.set(callId, proc);
232
-
233
- let buffer = "";
234
- const decoder = new StringDecoder("utf8");
235
-
236
- const processLine = (line: string) => {
237
- if (!line.trim()) return;
238
- let event: { type: string; message?: Message };
239
- try {
240
- event = JSON.parse(line) as { type: string; message?: Message };
241
- } catch {
242
- return;
243
- }
244
-
245
- if (event.type === "message_end" && event.message) {
246
- const msg = event.message;
247
- result.messages.push(msg);
248
- if (msg.role === "assistant") {
249
- result.usage.turns++;
250
- // Enforce maxTurns: abort the subprocess when the limit is reached.
251
- // killProc handles the SIGTERM→SIGKILL escalation. Use `!= null` so an
252
- // explicit maxTurns: 0 is honored (and marked) rather than treated as "unset".
253
- if (options.maxTurns != null && result.usage.turns >= options.maxTurns) {
254
- result.maxTurnsReached = true;
255
- controller.abort();
256
- }
257
- const usage = msg.usage;
258
- if (usage) {
259
- result.usage.input += usage.input || 0;
260
- result.usage.output += usage.output || 0;
261
- result.usage.cacheRead += usage.cacheRead || 0;
262
- result.usage.cacheWrite += usage.cacheWrite || 0;
263
- result.usage.cost += Number(usage.cost?.total) || 0;
264
- result.usage.contextTokens = usage.totalTokens || 0;
265
- }
266
- if (!result.model && msg.model) result.model = msg.model;
267
- if (msg.stopReason) result.stopReason = msg.stopReason;
268
- if (msg.errorMessage) result.errorMessage = msg.errorMessage;
269
- }
270
- }
271
-
272
- if (event.type === "tool_result_end" && event.message) {
273
- result.messages.push(event.message);
274
- }
275
- };
276
-
277
- proc.stdout.on("data", (data) => {
278
- // StringDecoder buffers incomplete multi-byte UTF-8 sequences across chunk
279
- // boundaries so a CJK char split between two `data` events isn't replaced
280
- // with U+FFFD (which would corrupt the line and silently drop the event).
281
- buffer += decoder.write(data);
282
- const lines = buffer.split("\n");
283
- buffer = lines.pop() || "";
284
- for (const line of lines) processLine(line);
285
- });
286
-
287
- proc.stderr.on("data", (data) => {
288
- result.stderr += data.toString();
289
- });
290
-
291
- proc.on("close", (code) => {
292
- const tail = decoder.end();
293
- if (tail) buffer += tail;
294
- if (buffer.trim()) processLine(buffer);
295
- resolve(code ?? 1); // code===null → signal-killed (OOM/SIGKILL): treat as failure, not silent empty success
296
- });
297
-
298
- proc.on("error", (err) => {
299
- // Surface the spawn error (e.g. ENOENT when `pi` is not on PATH) instead
300
- // of swallowing it.
301
- result.errorMessage = err.message;
302
- result.stderr += err.message;
303
- resolve(1);
304
- });
305
-
306
- // Per-call abort → SIGTERM (SIGKILL after 5s). Race-safe: if the controller
307
- // was already aborted before this listener registered, kill now.
308
- const killProc = () => {
309
- // Late-abort guard: if the proc already exited, don't flip a successful
310
- // result's `aborted` flag.
311
- if (proc.exitCode !== null || proc.signalCode !== null) return;
312
- result.aborted = true;
313
- proc.kill("SIGTERM");
314
- const timer = setTimeout(() => {
315
- // SIGTERM may be ignored — force SIGKILL after the grace period.
316
- proc.kill("SIGKILL");
317
- }, 5000);
318
- // Clear the timer once the proc exits so we don't leak a libuv handle.
319
- proc.once("close", () => clearTimeout(timer));
320
- };
321
- if (controller.signal.aborted) killProc();
322
- else controller.signal.addEventListener("abort", killProc, { once: true });
323
- });
324
-
325
- result.exitCode = exitCode;
326
- return result;
327
- } finally {
328
- // Always release registry slots, the parent-signal listener, and temp files.
329
- registry.processes.delete(callId);
330
- registry.controllers.delete(callId);
331
- if (signal) signal.removeEventListener("abort", onParentAbort);
332
- if (tmpPromptPath)
333
- try {
334
- fs.unlinkSync(tmpPromptPath);
335
- } catch {
336
- /* ignore */
337
- }
338
- if (tmpPromptDir)
339
- try {
340
- fs.rmdirSync(tmpPromptDir);
341
- } catch {
342
- /* ignore */
343
- }
344
- }
345
- }
346
-
347
- async function writePromptToTempFile(callId: string, prompt: string): Promise<{ dir: string; filePath: string }> {
348
- const tmpDir = await fs.promises.mkdtemp(path.join(os.tmpdir(), "pi-cr-agent-"));
349
- const safeName = callId.replace(/[^\w.-]+/g, "_");
350
- const filePath = path.join(tmpDir, `prompt-${safeName}.md`);
351
- await fs.promises.writeFile(filePath, prompt, { encoding: "utf-8", mode: 0o600 });
352
- return { dir: tmpDir, filePath };
353
- }