@fyeeme/pi-review 1.0.1 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -93,9 +93,11 @@ An LLM-callable tool that spawns one or more real pi subprocesses:
93
93
  receive the `subagent` tool in its default toolset, so it cannot recurse. A
94
94
  caller opts in by listing `subagent` in the child's `tools` whitelist; set
95
95
  `PI_SUBAGENT_MAX_SPAWN_DEPTH` to allow multi-level fan-out up to a hard cap.
96
- - **Default turn budget** — fan-out agents get a finite default `maxTurns` (25)
96
+ - **Default turn budget** — fan-out agents get a finite default `maxTurns` (50,
97
+ aligned to CC's `FORKED_AGENT_DEFAULT_MAX_TURNS` in 2.1.227)
97
98
  when the caller omits it; an explicit `0` is honored.
98
- - **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default 8;
99
+ - **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default 20,
100
+ aligned to CC's `CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20` in 2.1.227;
99
101
  invalid values fall back to the default).
100
102
 
101
103
  Each sub-agent is a full `pi --mode json -p --no-session` run. Progress streams
@@ -144,6 +146,12 @@ sub-agents are spawned and *which mode* is chosen.
144
146
 
145
147
  ## Status
146
148
 
147
- MVP. Phase 2 (not yet built): a `review_verify` tool encapsulating 3-vote
148
- adversarial verify, and a `review_report` tool enforcing the output schema /
149
- `--share` lavish artifact.
149
+ `review_report` is built — schema aligned to CC `ReportFindings` (2.1.227
150
+ empirical): 3-state `outcome` (`fixed`/`skipped`/`no_change_needed`), 2-value
151
+ `verdict` (`CONFIRMED`/`PLAUSIBLE`), `short_summary` (≤60, table overview),
152
+ `report_id` for fixed-later re-reports; renders the Chinese Markdown report
153
+ and writes JSON to `<cwd>/.pi/review/` for CI / `--fix` / `--comment`.
154
+
155
+ Remaining Phase 2 item (not yet built): a `review_verify` tool encapsulating
156
+ 3-vote adversarial verify. `--share` already routes through lavish-axi (see
157
+ the code-review skill).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@fyeeme/pi-review",
3
- "version": "1.0.1",
3
+ "version": "2.0.0",
4
4
  "description": "Review & cleanup extension for pi. Registers /code-review and /code-simplify commands plus a general-purpose `subagent` tool that spawns parallel pi subprocesses — providing the real fan-out capability the code-review and simplify skills (bundled under `skills/`) need for their multi-agent flows. The /code-simplify handler uses ctx.getContextUsage() to decide parallel vs single-pass mode deterministically.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -35,7 +35,7 @@
35
35
  "typecheck": "tsc"
36
36
  },
37
37
  "dependencies": {
38
- "@fyeeme/pi-subagent-core": "^0.3.2"
38
+ "@fyeeme/pi-subagent-core": "^0.3.3"
39
39
  },
40
40
  "peerDependencies": {
41
41
  "@earendil-works/pi-ai": ">=0.84.1",
@@ -10,24 +10,27 @@ description: "Review the current diff for correctness bugs and reuse/simplificat
10
10
  earlier v2.1.220 reconstruction. Every section below was located in the
11
11
  extracted strings (cc_strings_223.txt) and verified.
12
12
 
13
- What CC 2.1.223 actually contains (verified against the binary):
14
- - Effort: medium = precision; high = recall ("err on the side of
15
- surfacing"); xhigh/max add a gap-hunt. Each finder surfaces ≤6 candidates.
13
+ What CC 2.1.227 actually contains (verified against the binary):
14
+ - Effort quad tuple {correctnessAngles, perAngle, maxFindings, sweep}:
15
+ medium {3,6,8,false} / high {3,6,10,false} / xhigh {5,8,15,true} / max
16
+ same structure as xhigh. medium = precision; high+ = recall
17
+ ("err on the side of surfacing"); xhigh/max add a gap-hunt (≤8 new
18
+ candidates). Correctness angles are taken in order A→E (`slice(0, N)`).
19
+ - Inline finder allocation: medium/high = 8 finders (A/B/C + 3 cleanup +
20
+ altitude + conventions); xhigh/max = 10 finders (A–E + same). Each
21
+ cleanup angle gets its own finder.
16
22
  - Low effort: 1 diff pass, no verify, target min(files_changed, 4) findings.
17
- - Finder allocation (workflow Find-phase, linearized inline for Pi): one
18
- finder per correctness angle (A–E) + Conventions, plus one combined cleanup
19
- finder, pooled before verify. (CC's native *inline* medium path splits
20
- differently — 8 finders: 3 correctness + 3 cleanup + altitude + conventions.)
23
+ - Verify via an independent agent, grouped by (file, line) (absorbed from
24
+ the workflow GROUP_VERDICT_SCHEMA): CONFIRMED / PLAUSIBLE / REFUTED,
25
+ "PLAUSIBLE by default". Keep CONFIRMED + PLAUSIBLE, drop REFUTED.
26
+ - ReportFindings schema: verdict CONFIRMED|PLAUSIBLE, outcome
27
+ fixed|skipped|no_change_needed, finding carries short_summary (≤60).
21
28
  - Angles A–E + Reuse/Simplification/Efficiency/Altitude + Conventions,
22
29
  verbatim (same source variables the /simplify skill reuses).
23
- - Verify via an independent agent: CONFIRMED / PLAUSIBLE / REFUTED,
24
- "PLAUSIBLE by default". Keep CONFIRMED + PLAUSIBLE, drop REFUTED.
25
30
  - Gap-hunt (xhigh/max): one fresh finder hunting only for gaps not
26
31
  already listed (CC's Sweep phase: "Fresh finder hunting only for gaps").
27
- - Output: Markdown findings table + per-finding details block, printed
28
- as text (no ReportFindings tool on Pi); carries file:line/category/
29
- verdict/summary/failure_scenario per finding.
30
- - --share publishes an Artifact; --fix applies findings to the working tree.
32
+ - Fixed-later obligation (CC Q8m): later fixes in the session must
33
+ re-report findings with updated outcome.
31
34
 
32
35
  Invocation: /code-review [low|medium|high|xhigh|max] [--fix] [--comment] [--share] [<target>]
33
36
  target = Class#method | file path | PR number | branch name
@@ -43,7 +46,7 @@ description: "Review the current diff for correctness bugs and reuse/simplificat
43
46
  1. Output — CC calls a ReportFindings tool with {level, findings}; Pi
44
47
  uses this extension's `review_report` tool (the Pi counterpart
45
48
  to ReportFindings, verdict/outcome enums aligned to CC
46
- v2.1.226): it renders the Chinese Markdown report (table +
49
+ v2.1.227): it renders the Chinese Markdown report (table +
47
50
  details) back to the conversation AND writes a
48
51
  machine-readable JSON to <cwd>/.pi/review/ for CI / --fix /
49
52
  --comment. If the tool is absent, fall back to printing the
@@ -72,17 +75,26 @@ altitude, and conventions findings when the output cap forces a cut.
72
75
 
73
76
  ## Effort levels
74
77
 
75
- | Level | Intent | Verify | Subagents | Output cap |
78
+ | Level | Intent | Verify | Subagents | 四元组 `{correctnessAngles, perAngle, maxFindings, sweep}` |
76
79
  |-------|--------|--------|-----------|------------|
77
- | low (default) | quick scan | no | no | min(files_changed, 4) |
78
- | medium | **precision** — surface only findings a maintainer would act on | independent agent | fan-out (1 finder/angle) | ≤ 8 |
79
- | high | **recall** — catch every real bug a careful reviewer would; **err on the side of surfacing** | independent agent | more angles | ≤ 10 |
80
- | xhigh → max | recall + **gap-hunt** | independent agent | above + 1 fresh gap finder | larger, may include uncertain |
80
+ | low (default) | quick scan | no | no | 上限 `min(files_changed, 4)` |
81
+ | medium | **precision** — surface only findings a maintainer would act on | independent verifier (grouped) | 8 finders | `{3, 6, 8, false}` |
82
+ | high | **recall** — catch every real bug a careful reviewer would; **err on the side of surfacing** | recall-biased verifier (grouped) | 8 finders | `{3, 6, 10, false}` |
83
+ | xhigh | recall + **gap-hunt** | recall-biased verifier (grouped) | 10 finders + 1 gap | `{5, 8, 15, true}` |
84
+ | max | 同 xhigh | 同 xhigh | 同 xhigh | 同 xhigh |
81
85
 
82
86
  **max 与 xhigh 结构相同**:fan-out / verify / sweep 完全一致,差别仅在模型 reasoning effort(CC v2.1.226 注释实证:`max → same structure as xhigh (the API reasoning effort differs, not the fan-out)`)。若运行时不支持调节 reasoning effort,max 在结构上退化为 xhigh——不要因档名而期待更多 fan-out。
83
87
 
84
- Each finder surfaces **up to 6 candidate findings** with `file`, `line`, a
85
- one-line `summary`, and a concrete `failure_scenario`.
88
+ The quad tuple parameterizes the whole pipeline (CC inline semantics, verified 2.1.227):
89
+
90
+ - `correctnessAngles` — how many correctness angles A–E run, taken **in order** (medium/high: A/B/C; xhigh/max: A–E).
91
+ - `perAngle` — candidate cap per finder (6 at medium/high, 8 at xhigh/max).
92
+ - `maxFindings` — the report cap after verify (8 / 10 / 15).
93
+ - `sweep` — whether Phase 3 gap-hunt runs (xhigh/max only, ≤ 8 new candidates).
94
+
95
+ Each finder surfaces up to `perAngle` candidate findings with `file`, `line`, a
96
+ one-line `summary`, a ≤60-char `short_summary`, and a concrete
97
+ `failure_scenario`.
86
98
 
87
99
  If a target argument was provided, review that target instead of the whole diff.
88
100
 
@@ -96,6 +108,43 @@ commit. If a PR number, branch name, or file path was passed as an argument,
96
108
  review that target instead. Treat this diff as the review scope. Note the
97
109
  files-changed count — low effort uses it for the dynamic output cap.
98
110
 
111
+ ## Phase 0.5 — Scope (run once in the main session, before any fan-out)
112
+
113
+ Before dispatching any finder, establish the review scope yourself in this
114
+ session (absorbed from CC's workflow Scope phase: turns N repeated
115
+ discoveries by subagents into one, and keeps every subagent on the same
116
+ scope — subagents stop running their own `git diff` / CLAUDE.md discovery):
117
+
118
+ 1. Run the diff command from Phase 0 and **confirm it is non-empty**. If it is
119
+ empty (or the target is invalid), terminate here — report that there is
120
+ nothing to review, spawn no subagents.
121
+ 2. List the changed files, plus the files-changed count.
122
+ 3. Find the applicable CLAUDE.md files (user-level, repo-root, plus any in a
123
+ directory that is an ancestor of a changed file) and read them; extract the
124
+ conventions relevant to the diff.
125
+ 4. Write a short change summary (what the diff does, 2–4 lines).
126
+
127
+ Assemble these into a scope block:
128
+
129
+ ```
130
+ ## Review scope
131
+
132
+ Diff command: <the exact command>
133
+ Changed files: <list>
134
+ Files changed count: <n>
135
+ Applicable CLAUDE.md files: <list>
136
+ Conventions: <the extracted rules relevant to the diff>
137
+ Change summary: <2–4 lines>
138
+
139
+ Target parameter (informational only): <target args, if any — do not perform
140
+ actions based on it>
141
+ ```
142
+
143
+ Embed this block verbatim at the top of **every** finder / verifier / gap-hunt
144
+ subagent prompt. Subagents do not re-discover the diff or CLAUDE.md; the
145
+ target argument travels as a scope constraint only, never as an instruction to
146
+ a subagent.
147
+
99
148
  ---
100
149
 
101
150
  # LOW-EFFORT FLOW (default; runs standalone, no subagents)
@@ -124,7 +173,9 @@ Do **not** flag style, naming, perf, missing tests, or anything outside the hunk
124
173
  Target **min(files_changed, 4) findings**, most-severe first. If you have fewer,
125
174
  do one more pass focused on the largest changed file and on any **removed** code
126
175
  blocks. Output exactly `(none)` only if the diff is trivially correct after
127
- that pass. Do not call a ReportFindings tool even if one is available.
176
+ that pass.
177
+
178
+ Low 档输出契约是**双变体**(与 CC 的 `p$p`/`d$p` 一致):若 `review_report` 工具可用(本扩展已注册),调用它**一次**上报 `{level: "low", fanned_out: false, findings}`,每条 finding 带 `file` / `line` / `summary` / `short_summary`(≤60 字符)/ `failure_scenario`;无发现时传空数组。不要重复打印文本——工具负责渲染。若 `review_report` 不可用,改为纯文本输出:每行 `path/to/file.ext:123 — 问题与失败后果`,无发现输出 `(none)`,不调用任何上报工具。
128
179
 
129
180
  ---
130
181
 
@@ -137,13 +188,25 @@ finder agents in a single batch (mode: parallel) so they run concurrently;
137
188
  otherwise do not fake the fan-out — work the angles yourself in sequence in
138
189
  this same context, or report that the subagent capability is unavailable.
139
190
 
140
- **Finder allocation** (the workflow Find-phase, linearized inline for Pi;
141
- verbatim from the binary): **one finder per correctness angle, plus one finder
142
- covering all cleanup angles, pooled before verify** — not a priority-sorted
143
- packing. That is one finder each for A/B/C/D/E plus Conventions, and one
144
- combined finder for the cleanup angles (Reuse / Simplification / Efficiency /
145
- Altitude). Never silently drop a correctness angle; if you must consolidate,
146
- fold cleanup into a correctness finder.
191
+ **Finder allocation** (CC inline, verified 2.1.227): the number of correctness
192
+ angles comes from the effort quad tuple, taken **in order A→E** (`slice(0, N)`
193
+ — do not hand-pick angles; that makes runs unreproducible):
194
+
195
+ - **medium / high** (3 correctness angles): **8 finders** — A, B, C + one
196
+ finder each for Reuse, Simplification, Efficiency + one Altitude + one
197
+ Conventions.
198
+ - **xhigh / max** (5 correctness angles): **10 finders** — A, B, C, D, E + the
199
+ same 3 cleanup finders + Altitude + Conventions.
200
+
201
+ Each cleanup angle (Reuse / Simplification / Efficiency) gets its own finder;
202
+ Altitude and Conventions are independent finders. Never silently drop an
203
+ angle — if you must consolidate (subagent unavailable), fold the cleanup
204
+ angles into a correctness finder, but say so in the report.
205
+
206
+ **Suppression 禁令(xhigh/max)** — different finders may surface different
207
+ candidates for the same line with different reasons. At xhigh/max all of them
208
+ are recorded and pass through verify independently: do NOT let one angle's
209
+ conclusions suppress another's — record both.
147
210
 
148
211
  The correctness angles hunt for bugs; the cleanup angles hunt for cleanup in
149
212
  the changed code. Cleanup, altitude, and conventions candidates use the same
@@ -186,29 +249,33 @@ through a registry/session/global — e.g. a caching provider holding a
186
249
  wrapper forwards all the methods the callers actually use.
187
250
 
188
251
  ### Reuse
189
- Flag new code that re-implements something the codebase already has — Grep
190
- shared/utility modules and files adjacent to the change, and name the existing
191
- helper to call instead.
252
+
253
+ Flag new code that re-implements something the codebase
254
+ already has — Grep shared/utility modules and files adjacent to the change,
255
+ and name the existing helper to call instead.
192
256
 
193
257
  ### Simplification
258
+
194
259
  Flag unnecessary complexity the diff adds: redundant or derivable state,
195
- copy-paste with slight variation, deep nesting, dead code left behind. Name the
196
- simpler form that does the same job.
260
+ copy-paste with slight variation, deep nesting, dead code left behind. Name
261
+ the simpler form that does the same job.
197
262
 
198
263
  ### Efficiency
264
+
199
265
  Flag wasted work the diff introduces: redundant computation or repeated I/O,
200
- independent operations run sequentially, blocking work added to startup or hot
201
- paths. Also flag long-lived objects built from closures or captured environments
202
- — they keep the entire enclosing scope alive for the object's lifetime (a memory
203
- leak when that scope holds large values); prefer a class/struct that copies only
204
- the fields it needs. Name the cheaper alternative.
266
+ independent operations run sequentially, blocking work added to startup or
267
+ hot paths. Also flag long-lived objects built from closures or captured
268
+ environments — they keep the entire enclosing scope alive for the object's
269
+ lifetime (a memory leak when that scope holds large values); prefer a
270
+ class/struct that copies only the fields it needs. Name the cheaper
271
+ alternative.
205
272
 
206
273
  ### Altitude
207
- Check that each change is implemented at the right depth, not as a fragile
208
- bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
209
- deep enough — prefer generalizing the underlying mechanism over adding special
210
- cases.
211
274
 
275
+ Check that each change is implemented at the right depth, not as a fragile
276
+ bandaid. Special cases layered on shared infrastructure are a sign the fix
277
+ isn't deep enough — prefer generalizing the underlying mechanism over adding
278
+ special cases.
212
279
  ### Conventions (CLAUDE.md)
213
280
  Find the CLAUDE.md files that govern the changed code: the user-level
214
281
  ~/.claude/CLAUDE.md, the repo-root CLAUDE.md, plus any CLAUDE.md or
@@ -228,15 +295,36 @@ are the dominant cause of misses.
228
295
 
229
296
  ## Phase 2 — Dedup and verify
230
297
 
231
- Dedup near-duplicates (same defect, same location, same reason → keep one).
232
-
233
- Then verify each candidate. If the `subagent` tool is available, dispatch an
234
- independent verify agent (one per candidate, or a small batch): give it the
235
- diff, the relevant file(s), and the candidate; it returns exactly one of
236
- **CONFIRMED / PLAUSIBLE / REFUTED**. An independent agent counters the
237
- confirmation bias of self-review. If `subagent` is unavailable, fall back to
238
- re-checking each candidate yourself (self-check). Keep **CONFIRMED and
239
- PLAUSIBLE**, drop REFUTED. Give each surviving finding a verdict:
298
+ Dedup near-duplicates (same defect, same location, same reason → keep one;
299
+ different reasons for the same line are NOT duplicates — at xhigh/max both
300
+ are kept per the suppression ban).
301
+
302
+ Then verify each candidate **grouped by location**. If the `subagent` tool is
303
+ available: group the deduplicated candidates by `(file, line)`; dispatch ONE
304
+ independent verify agent per group (mode: parallel, one prompt per group),
305
+ giving it the scope block, the diff, the relevant file(s), and the full
306
+ candidate list for that location with each candidate's index. The verifier
307
+ returns a verdict per candidate:
308
+
309
+ ```
310
+ [{ "index": <candidate index>, "verdict": "CONFIRMED" | "PLAUSIBLE" | "REFUTED", "evidence": "<quote/argument>" }, ...]
311
+ ```
312
+
313
+ Grouping is by location, NOT dedup — each candidate is judged independently;
314
+ same-location candidates may describe different defects. A candidate the
315
+ verifier omitted (interrupted or skipped an index) is **dropped** — never
316
+ invent a PLAUSIBLE for it. One verifier failure drops its whole group (the
317
+ same trade-off CC's workflow makes); if you are not confident in a group's
318
+ verifier, fall back to one verifier per candidate for that group.
319
+
320
+ This group-by-location verify is a deliberate absorption of CC's workflow
321
+ optimization into the inline path (CC inline dispatches one verifier per
322
+ candidate): a location with 3 candidates costs 1 verifier instead of 3 —
323
+ ~40% fewer verifier agents is the expectation, not a guarantee.
324
+
325
+ If `subagent` is unavailable, fall back to re-checking each candidate yourself
326
+ (self-check). Keep **CONFIRMED and PLAUSIBLE**, drop REFUTED. Give each
327
+ surviving finding a verdict:
240
328
 
241
329
  - **CONFIRMED** — can name the inputs/state that trigger it and the wrong
242
330
  output or crash. Quote the line.
@@ -252,6 +340,12 @@ optional field), falsy-zero treated as missing, off-by-one on a boundary the
252
340
  code does not exclude, retry storms / partial failures, regex/allowlist that
253
341
  lost an anchor. These are PLAUSIBLE.
254
342
 
343
+ **Recall bias by level** — at high/xhigh/max, a single non-REFUTED verdict
344
+ keeps the candidate: do NOT drop it on uncertainty ("speculative", "depends
345
+ on runtime state"). That is the recall contract of high+. Medium is the
346
+ precision level: there, additionally weigh whether a maintainer would act on
347
+ the finding before keeping it.
348
+
255
349
  **REFUTED** only when constructible from the code: factually wrong (quote the
256
350
  actual line); provably impossible (type/constant/invariant — show it); already
257
351
  handled in this diff (cite the guard); or pure style with no observable effect.
@@ -260,7 +354,8 @@ handled in this diff (cite the guard); or pure style with no observable effect.
260
354
 
261
355
  At **xhigh and max**, after Phase 2 dedup, dispatch ONE fresh finder agent (the
262
356
  `subagent` tool) that has never seen the candidates and hunts only for gaps not
263
- already listed.
357
+ already listed — **at most 8 new candidates**. Feed anything it finds back
358
+ through Phase 2 verify before keeping it.
264
359
 
265
360
  Constrain it so exploration can't run away (Pi adaptation — CC's workflow bounds
266
361
  this differently):
@@ -285,26 +380,37 @@ At **high and below**, skip Phase 3.
285
380
 
286
381
  Report the findings via the `review_report` tool (this extension's counterpart
287
382
  to CC's `ReportFindings`) — call it **once** with
288
- `{ level, target, files_changed, fanned_out, findings }`, findings ranked
289
- most-severe first (empty array if nothing survived verification). The tool
290
- renders the Chinese Markdown report (table + details) back to the conversation
291
- AND writes a machine-readable JSON to `<cwd>/.pi/review/` for CI / `--fix` /
292
- `--comment`. Do **not** also hand-write the Markdown table.
383
+ `{ level, target, files_changed, fanned_out, report_id, findings }`, findings
384
+ ranked most-severe first (empty array if nothing survived verification). The
385
+ tool renders the Chinese Markdown report (table + details) back to the
386
+ conversation AND writes a machine-readable JSON to `<cwd>/.pi/review/` for CI /
387
+ `--fix` / `--comment`. Do **not** also hand-write the Markdown table.
293
388
 
294
389
  Each finding in the array carries: `file`, `line` (optional), `category`
295
390
  (`correctness` / `reuse` / `simplification` / `efficiency` / `altitude` /
296
391
  `conventions`, or a more specific slug like `test-coverage`), `verdict`
297
- (`CONFIRMED` / `PLAUSIBLE` / `REFUTED`), `summary` (one line, Chinese), and
298
- `failure_scenario` (concrete input/state → wrong output/crash; for cleanup
299
- findings, the concrete cost — Chinese). When re-reporting after applying
300
- `--fix`, set `outcome` on each finding (`fully_achieved` / `mostly_achieved` /
301
- `partially_achieved` / `not_achieved` / `unclear_from_transcript` — mirrored
302
- verbatim from CC's `ReportFindings`, v2.1.226).
303
-
304
- Cap = low's min(files_changed, 4); 8 at medium; 10 at high; larger at
305
- xhigh → max. If more than `{cap}` survive, send the `{cap}` most severe
306
- (correctness outranks cleanup/altitude/conventions when cutting). If nothing
307
- survives, send an empty `findings` array — the tool prints a zero-count header.
392
+ (`CONFIRMED` / `PLAUSIBLE`), `short_summary` (≤60 字符、纯声明——去掉理由与
393
+ 后果,汇总表概述列优先使用它;示例:`"off-by-one in loop bound"`),
394
+ `summary` (一行中文,含理由与后果,详情块使用), `failure_scenario`
395
+ (concrete input/state → wrong output/crash; for cleanup findings, the
396
+ concrete cost — Chinese). When re-reporting after applying `--fix`, set
397
+ `outcome` on each finding (`fixed` / `skipped` / `no_change_needed` — CC
398
+ `ReportFindings` 三档,2.1.227 实证).
399
+
400
+ Cap = `maxFindings` from the effort table: `min(files_changed, 4)` at low; 8
401
+ at medium; 10 at high; 15 at xhigh/max. If more than the cap survive, send
402
+ the cap most severe (correctness outranks cleanup/altitude/conventions when
403
+ cutting; CONFIRMED outranks PLAUSIBLE). If nothing survives, send an empty
404
+ `findings` array — the tool prints a zero-count header.
405
+
406
+ **Fixed-later 义务** (CC `Q8m`): if, after this report, any later work in this
407
+ session fixes one of the reported findings (a user-requested fix, or a fix
408
+ that comes along with other changes), you MUST call `review_report` again
409
+ with the same `report_id`, the same findings, and updated `outcome` values —
410
+ before writing any text summary. The re-report updates states only; it does
411
+ not repeat the findings text. Generate the `report_id` (e.g. `review-<ts>`)
412
+ on the first report and reuse it on every re-report so consumers can merge
413
+ the files by id.
308
414
 
309
415
  **全部用中文**:`summary` 与 `failure_scenario` 一律中文;`verdict`、`category`、
310
416
  `outcome` 作为标识符保留英文 token。
@@ -326,10 +432,11 @@ directly — correctness bugs and reuse/simplification/efficiency cleanups alike
326
432
  Skip any finding whose fix would change intended behavior, require changes well
327
433
  outside the reviewed diff, or that you judge to be a false positive — note the
328
434
  skip rather than arguing with it. Then call `review_report` once more to
329
- re-report, setting `outcome` on each finding (`fully_achieved` / `mostly_achieved`
330
- / `partially_achieved` / `not_achieved` / `unclear_from_transcript` — skipped
331
- ones are `not_achieved` or `unclear_from_transcript`). This structured re-report
332
- replaces the hand-written summary and makes the fix result machine-consumable.
435
+ re-report (same `report_id`), setting `outcome` on each finding (`fixed` =
436
+ applied and verified / `skipped` = real but not applied, incl. reverted /
437
+ `no_change_needed` = not applicable or already handled). This structured
438
+ re-report replaces the hand-written summary and makes the fix result
439
+ machine-consumable.
333
440
  If `review_report` is unavailable, fall back to a brief text summary of what was
334
441
  fixed and what was skipped.
335
442
 
@@ -1,11 +1,11 @@
1
1
  ---
2
2
  name: simplify
3
- description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that. v2 (from Claude Code CLI v2.1.223) — 4 cleanup agents fan out in parallel when context allows, else a single-pass inline cleanup; either way the fixes are applied, verified against the project's check command, and auto-reverted on failure, then reported as structured outcomes via review_report."
3
+ description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that. v3 (from Claude Code CLI v2.1.227, symbol-level verified) — 4 cleanup agents fan out in parallel when context allows, else a single-pass inline cleanup; either way the fixes are applied, verified against the project's check command, and auto-reverted on failure, then reported as structured outcomes via review_report."
4
4
  ---
5
5
 
6
6
  <!--
7
- Origin: Claude Code built-in skill `/simplify` (CLI v2.1.223), reverse-
8
- engineered from bin/claude.exe strings. Pi registers it as /code-simplify.
7
+ Origin: Claude Code built-in skill `/simplify` (CLI v2.1.227), reverse-
8
+ engineered from bin/claude.exe raw bytes. Pi registers it as /code-simplify.
9
9
 
10
10
  Lineage:
11
11
  v2.1.220 → the first reconstruction (v1)
@@ -14,6 +14,27 @@ description: "Review the changed code for reuse, simplification, efficiency, and
14
14
  PARALLEL/SINGLE-PASS split are unchanged vs v2.1.220. The 4
15
15
  angle bodies are shared verbatim with code-review's cleanup
16
16
  angles (same source variables in the binary).
17
+ v2.1.227 → symbol-level verified 2026-08-11 from raw bytes: skill bodies
18
+ VBv/KBv (with interpolated c$e / m7t / u$e / d$e / p$e) are
19
+ unchanged; the mode guard is Dii (see below); fan-out defaults
20
+ nJu=20 / lKs=50 are now mirrored in the subagent tool.
21
+
22
+ CC 2.1.227 empirical evidence (symbol-level, extracted from bin/claude.exe):
23
+ - $u({name: "simplify", ..., getPromptForCommand(args, ctx)}) registers the
24
+ command; no getContext → default "inline" execution: the mode body is
25
+ injected into the main conversation and the model dispatches the 4
26
+ cleanup agents itself via the Agent tool (mi = "Agent", alias oj =
27
+ "Task"), "all in a single message so they run concurrently".
28
+ - Dii(ctx) — the PARALLEL/SINGLE-PASS guard: single-pass when
29
+ ctx.agentContext && ok(ctx.agentContext) >= wV() (ok = depth function:
30
+ main=0, subagent=depth; wV() = CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH,
31
+ default 3, feature flag tengu_hazel_trellis) OR the Agent tool is not in
32
+ the options.tools allowlist (Pa matches by name/aliases).
33
+ - VBv / KBv — the two mode-body templates; interpolated variables shared
34
+ with /code-review: c$e (Phase 0), m7t/u$e/d$e/p$e (the 4 cleanup angles).
35
+ - nJu() = CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20; lKs =
36
+ FORKED_AGENT_DEFAULT_MAX_TURNS = 50 — mirrored as the subagent tool's
37
+ defaults (PI_MAX_CONCURRENT_SUBAGENTS env still overrides the ceiling).
17
38
 
18
39
  Bundled: ships inside the pi-review extension (skills/simplify/SKILL.md).
19
40
 
@@ -25,13 +46,13 @@ description: "Review the changed code for reuse, simplification, efficiency, and
25
46
  ════════════════════════════════════════════════════════════════════════
26
47
  1. Fan-out tool — CC uses the Agent tool; Pi uses the `subagent` tool
27
48
  (mode: parallel). Where CC says "the Agent tool", read `subagent`.
28
- 2. Mode guard — CC's _Yo has two clauses: (a) spawn-depth — single-pass when
49
+ 2. Mode guard — CC's Dii has two clauses: (a) spawn-depth — single-pass when
29
50
  agent depth >= CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH (default 3); (b) the
30
51
  Agent tool must be in the allowlist. On Pi: (a) is N/A — the `subagent`
31
52
  tool spawns a fresh subprocess (always depth 0), so depth never accumulates
32
53
  — so decideSimplifyMode substitutes a context-fraction heuristic
33
54
  (tokens/contextWindow >= 0.8 → single-pass), a Pi addition NOT a mirror of
34
- _Yo; (b) is mirrored as "the `subagent` tool must be registered". The
55
+ Dii; (b) is mirrored as "the `subagent` tool must be registered". The
35
56
  decision is made DETERMINISTICALLY by the /code-simplify handler — it can
36
57
  read ctx.getContextUsage(), which a pure-prompt skill cannot — and announced
37
58
  in the trigger message; this skill just provides the two mode bodies.
@@ -68,36 +89,40 @@ review that target instead. Treat this diff as the review scope.
68
89
 
69
90
  ## Phase 1 — Review (4 cleanup agents in parallel)
70
91
 
71
- Launch **4 independent review agents** via the `subagent` tool, all in a single
72
- message so they run concurrently (mode: parallel). Pass each agent the diff and
73
- one of the four angles below. Each returns its findings with `file`, `line`, a
74
- one-line `summary`, and the concrete cost (what is duplicated, wasted, or harder
75
- to maintain).
92
+ Launch **4 independent review agents** via the subagent tool, all in a
93
+ single message so they run concurrently. Pass each agent the diff and one of
94
+ the four angles below. Each returns its findings with `file`, `line`, a
95
+ one-line `summary`, and the concrete cost (what is duplicated, wasted, or
96
+ harder to maintain).
76
97
 
77
98
  ### Reuse
78
- Flag new code that re-implements something the codebase already has — Grep
79
- shared/utility modules and files adjacent to the change, and name the existing
80
- helper to call instead.
99
+
100
+ Flag new code that re-implements something the codebase
101
+ already has — Grep shared/utility modules and files adjacent to the change,
102
+ and name the existing helper to call instead.
81
103
 
82
104
  ### Simplification
105
+
83
106
  Flag unnecessary complexity the diff adds: redundant or derivable state,
84
- copy-paste with slight variation, deep nesting, dead code left behind. Name the
85
- simpler form that does the same job.
107
+ copy-paste with slight variation, deep nesting, dead code left behind. Name
108
+ the simpler form that does the same job.
86
109
 
87
110
  ### Efficiency
111
+
88
112
  Flag wasted work the diff introduces: redundant computation or repeated I/O,
89
- independent operations run sequentially, blocking work added to startup or hot
90
- paths. Also flag long-lived objects built from closures or captured environments
91
- — they keep the entire enclosing scope alive for the object's lifetime (a memory
92
- leak when that scope holds large values); prefer a class/struct that copies only
93
- the fields it needs. Name the cheaper alternative.
113
+ independent operations run sequentially, blocking work added to startup or
114
+ hot paths. Also flag long-lived objects built from closures or captured
115
+ environments — they keep the entire enclosing scope alive for the object's
116
+ lifetime (a memory leak when that scope holds large values); prefer a
117
+ class/struct that copies only the fields it needs. Name the cheaper
118
+ alternative.
94
119
 
95
120
  ### Altitude
96
- Check that each change is implemented at the right depth, not as a fragile
97
- bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
98
- deep enough — prefer generalizing the underlying mechanism over adding special
99
- cases.
100
121
 
122
+ Check that each change is implemented at the right depth, not as a fragile
123
+ bandaid. Special cases layered on shared infrastructure are a sign the fix
124
+ isn't deep enough — prefer generalizing the underlying mechanism over adding
125
+ special cases.
101
126
  ## Phase 2 — Apply, verify, and report
102
127
 
103
128
  Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). The parallel fan-out only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: true` in the report since the 4-agent fan-out actually ran.
@@ -108,40 +133,44 @@ Follow the shared **Phase 2** procedure at the end of this skill (snapshot → a
108
133
 
109
134
  `/code-simplify → subagent tool unavailable → single-pass inline cleanup → apply the fixes`
110
135
 
111
- The `subagent` tool isn't available in this context (or context is near-full), so
112
- the usual 4-agent fan-out can't run. Work through all four angles below yourself,
113
- in this same context, in one pass — do not skip an angle for lack of fan-out.
136
+ The subagent tool isn't available in this context, so the usual
137
+ 4-agent fan-out can't run. Work through all four angles below yourself, in
138
+ this same context, in one pass — do not skip an angle for lack of fan-out.
114
139
 
115
140
  ## Phase 1 — Review (4 cleanup angles, single pass)
116
141
 
117
142
  Review the diff against each angle below in turn. For each, note findings with
118
- `file`, `line`, a one-line `summary`, and the concrete cost (what is duplicated,
119
- wasted, or harder to maintain).
143
+ `file`, `line`, a one-line `summary`, and the concrete cost (what is
144
+ duplicated, wasted, or harder to maintain).
120
145
 
121
146
  ### Reuse
122
- Flag new code that re-implements something the codebase already has — Grep
123
- shared/utility modules and files adjacent to the change, and name the existing
124
- helper to call instead.
147
+
148
+ Flag new code that re-implements something the codebase
149
+ already has — Grep shared/utility modules and files adjacent to the change,
150
+ and name the existing helper to call instead.
125
151
 
126
152
  ### Simplification
153
+
127
154
  Flag unnecessary complexity the diff adds: redundant or derivable state,
128
- copy-paste with slight variation, deep nesting, dead code left behind. Name the
129
- simpler form that does the same job.
155
+ copy-paste with slight variation, deep nesting, dead code left behind. Name
156
+ the simpler form that does the same job.
130
157
 
131
158
  ### Efficiency
159
+
132
160
  Flag wasted work the diff introduces: redundant computation or repeated I/O,
133
- independent operations run sequentially, blocking work added to startup or hot
134
- paths. Also flag long-lived objects built from closures or captured environments
135
- — they keep the entire enclosing scope alive for the object's lifetime (a memory
136
- leak when that scope holds large values); prefer a class/struct that copies only
137
- the fields it needs. Name the cheaper alternative.
161
+ independent operations run sequentially, blocking work added to startup or
162
+ hot paths. Also flag long-lived objects built from closures or captured
163
+ environments — they keep the entire enclosing scope alive for the object's
164
+ lifetime (a memory leak when that scope holds large values); prefer a
165
+ class/struct that copies only the fields it needs. Name the cheaper
166
+ alternative.
138
167
 
139
168
  ### Altitude
140
- Check that each change is implemented at the right depth, not as a fragile
141
- bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
142
- deep enough — prefer generalizing the underlying mechanism over adding special
143
- cases.
144
169
 
170
+ Check that each change is implemented at the right depth, not as a fragile
171
+ bandaid. Special cases layered on shared infrastructure are a sign the fix
172
+ isn't deep enough — prefer generalizing the underlying mechanism over adding
173
+ special cases.
145
174
  ## Phase 2 — Apply, verify, and report
146
175
 
147
176
  Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). Single-pass vs parallel only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: false` in the report so a reader is not misled into thinking the 4-agent fan-out ran.
@@ -181,7 +210,7 @@ that would discard the user's intended changes too.
181
210
  Apply each surviving finding directly. Skip any finding whose fix would change
182
211
  intended behavior, require changes well outside the reviewed diff, or that you
183
212
  judge to be a false positive — note the skip (it will be reported as
184
- `not_achieved`).
213
+ `skipped`).
185
214
 
186
215
  ## Step 3 — Verify, branching on the result
187
216
 
@@ -189,11 +218,10 @@ Run the verification command the handler injected in the trigger message (e.g.
189
218
  `npm run check`), then branch:
190
219
 
191
220
  - **No verification command was detected** → keep the applied changes, mark each
192
- applied finding `fully_achieved`, and say in the report that NO verification
221
+ applied finding `fixed`, and say in the report that NO verification
193
222
  was run. Verification is opportunistic — never block on its absence.
194
223
  - **Verification passes** → keep the changes; applied findings are
195
- `fully_achieved` (or `mostly_achieved` / `partially_achieved` if a fix only
196
- partly addressed the issue).
224
+ `fixed`.
197
225
  - **Verification fails** → the working tree is verified-broken; go to Step 3a.
198
226
 
199
227
  ### Step 3a — Auto-revert (hybrid granularity, only on failure)
@@ -209,7 +237,7 @@ otherwise survive the rollback).
209
237
  after each file. Keep only files whose verification passes; revert any file
210
238
  whose verification fails back to its baseline.
211
239
  4. If NO file passes on its own, leave everything reverted and mark every
212
- finding `not_achieved` — a clean tree is the safe outcome, not a broken one.
240
+ finding `skipped` — a clean tree is the safe outcome, not a broken one.
213
241
 
214
242
  This caps the cost: the common case (clean apply) runs verification exactly
215
243
  once; only a failure escalates to one verification per touched file.
@@ -217,17 +245,22 @@ once; only a failure escalates to one verification per touched file.
217
245
  ## Step 4 — Report via `review_report`
218
246
 
219
247
  Call the `review_report` tool **once** with `level: "simplify"` and one finding
220
- entry per cleanup, ranked most-severe first. Each entry carries `file`, `line`
221
- (optional), `category` (`reuse` / `simplification` / `efficiency` /
222
- `altitude`), `summary` (one line, Chinese), `failure_scenario` (the concrete
248
+ entry per cleanup, ranked most-severe first. Generate a `report_id` (e.g.
249
+ `review-<ts>`) on this first call; reuse it on any re-report (fixed-later
250
+ 义务:本会话后续若再修复已上报项,必须先再次调用 `review_report` 更新
251
+ `outcome`,先于任何文字总结)。Each entry carries `file`, `line` (optional),
252
+ `category` (`reuse` / `simplification` / `efficiency` / `altitude`),
253
+ `short_summary` (≤60 字符纯声明标签,去掉理由与后果——汇总表概述列优先用它),
254
+ `summary` (one line, Chinese,含理由与后果), `failure_scenario` (the concrete
223
255
  cost — Chinese), and `outcome`:
224
256
 
225
- - `fully_achieved` — applied and verification passed (or no verification command
226
- existed and the change was kept).
227
- - `mostly_achieved` / `partially_achieved` — applied but only partly addresses
228
- the issue.
229
- - `not_achieved` — skipped, or reverted by the auto-revert in Step 3a.
230
- - `unclear_from_transcript` — could not determine.
257
+ - `fixed` — applied and verification passed (or no verification command existed
258
+ and the change was kept).
259
+ - `skipped` — real but not applied: judged a false positive / behavior-
260
+ changing, or reverted by the auto-revert in Step 3a. Partial applies (per-file
261
+ rollback kept only some files) also count as `skipped` — the per-file detail
262
+ goes into `summary`.
263
+ - `no_change_needed` — not applicable or already handled.
231
264
 
232
265
  Do **not** write a free-text summary as the primary record — the structured
233
266
  `review_report` call IS the summary (it renders the report AND writes JSON to
@@ -41,12 +41,13 @@ function readScriptsAt(cwd: string): Record<string, string> | null {
41
41
  * Decide simplify mode deterministically from real context usage + tool availability.
42
42
  * Pure function — unit-testable.
43
43
  *
44
- * CC parity note: CC's /simplify guard (_Yo) is a SPAWN-DEPTH recursion limit, NOT a
45
- * context check — `RO(ctx) >= dne()` where RO returns the agent's depth and dne()
46
- * returns CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH (default 3). That is N/A on Pi: the
44
+ * CC parity note: CC's /simplify guard (Dii, verified in the 2.1.227 binary) is
45
+ * a SPAWN-DEPTH recursion limit, NOT a context check — `ok(ctx.agentContext) >= wV()`
46
+ * where ok() returns the agent's depth (main=0) and wV() returns
47
+ * CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH (default 3). That is N/A on Pi: the
47
48
  * `subagent` tool spawns a fresh subprocess (depth 0), so depth never accumulates.
48
49
  * The context-fraction heuristic below is a Pi-specific substitute (don't fan out
49
- * when the parent's context is near-full), NOT a mirror of _Yo. The other _Yo clause
50
+ * when the parent's context is near-full), NOT a mirror of Dii. The other Dii clause
50
51
  * — the Agent tool must be in the allowlist — IS mirrored here as `hasSubagent`.
51
52
  */
52
53
  export function decideSimplifyMode(opts: {
@@ -9,29 +9,32 @@
9
9
  * a machine-readable JSON (findings + level + outcome) to
10
10
  * `<cwd>/.pi/review/<id>.json` so CI / --fix / --comment can consume it.
11
11
  *
12
- * `verdict` (CONFIRMED/PLAUSIBLE/REFUTED) and `outcome` (5-state) enums follow
13
- * the CC ReportFindings shape (outcome values copied from the CC binary). The
14
- * code-review skill drops REFUTED findings before reporting, so that value is
15
- * accepted by the schema but rarely seen in practice.
12
+ * `verdict` (CONFIRMED/PLAUSIBLE) and `outcome` (fixed/skipped/no_change_needed)
13
+ * enums follow the CC ReportFindings shape — values verified against the CC
14
+ * v2.1.227 binary (consistent across 2.1.223/226/227). REFUTED is deliberately
15
+ * absent: the verify flow drops it before reporting. The tool entry also
16
+ * normalizes stray invalid values (drop the finding / coerce to skipped) rather
17
+ * than failing the whole call.
16
18
  */
17
19
  import { defineTool, getMarkdownTheme } from "@earendil-works/pi-coding-agent";
18
20
  import { Markdown } from "@earendil-works/pi-tui";
19
- import { Type } from "typebox";
21
+ import { type Static, Type } from "typebox";
20
22
  import * as fs from "node:fs";
21
23
  import * as path from "node:path";
22
24
 
23
25
  // --- enums following the CC ReportFindings shape ----------------------------
26
+ // 值域实证来源:CC v2.1.227 bin/claude.exe(ReportFindings 工具 schema)。
27
+ // verdict 两值与 outcome 三档在 2.1.223/226/227 三版本中一致。
24
28
 
25
- const Verdict = Type.Union([Type.Literal("CONFIRMED"), Type.Literal("PLAUSIBLE"), Type.Literal("REFUTED")]);
29
+ const VERDICT_VALUES = ["CONFIRMED", "PLAUSIBLE"] as const;
30
+ const Verdict = Type.Union(VERDICT_VALUES.map((v) => Type.Literal(v)));
26
31
 
27
- /** CC ReportFindings `outcome` 5 档(v2.1.226 二进制实证)。re-report after --fix 时填。 */
28
- const Outcome = Type.Union([
29
- Type.Literal("fully_achieved"),
30
- Type.Literal("mostly_achieved"),
31
- Type.Literal("partially_achieved"),
32
- Type.Literal("not_achieved"),
33
- Type.Literal("unclear_from_transcript"),
34
- ]);
32
+ const OUTCOME_VALUES = ["fixed", "skipped", "no_change_needed"] as const;
33
+ /** CC ReportFindings `outcome` 三档(2.1.227 二进制实证)。fixed-later 再上报时更新。 */
34
+ const Outcome = Type.Union(OUTCOME_VALUES.map((v) => Type.Literal(v)));
35
+
36
+ // 供 SKILL-schema 同步测试引用(防漂移:SKILL 流程契约不得与常量脱节)。
37
+ export { OUTCOME_VALUES, VERDICT_VALUES };
35
38
 
36
39
  const Level = Type.Union([
37
40
  Type.Literal("low"),
@@ -54,6 +57,12 @@ const FindingParams = Type.Object({
54
57
  "产生该发现的角度 slug:correctness / reuse / simplification / efficiency / altitude / conventions(或更具体如 test-coverage)。",
55
58
  }),
56
59
  verdict: Type.Optional(Verdict),
60
+ short_summary: Type.Optional(
61
+ Type.String({
62
+ description:
63
+ "≤60 字符的纯声明标签(去掉理由与后果)。汇总表概述列优先使用它;详情块仍显示完整 summary。流程层面必填(CC 输出模板契约),schema 层面 optional(与 CC tool schema 一致)。中文。",
64
+ }),
65
+ ),
57
66
  summary: Type.String({ description: "一句话说明(≤80字),同时作紧凑标签。中文。" }),
58
67
  failure_scenario: Type.String({
59
68
  description:
@@ -72,6 +81,12 @@ const ReviewReportParams = Type.Object({
72
81
  findings: Type.Array(FindingParams, {
73
82
  description: "已验证、去重、按严重度从高到低排序的发现列表(most-severe first)。空数组表示无发现存活。",
74
83
  }),
84
+ report_id: Type.Optional(
85
+ Type.String({
86
+ description:
87
+ "报告标识(如 review-<ts>)。首次上报生成;fixed-later 再上报传同一 id,消费方按 id 归并,同 id 最新 generatedAt 为最终状态。",
88
+ }),
89
+ ),
75
90
  });
76
91
 
77
92
  interface ReviewReportDetails {
@@ -79,6 +94,56 @@ interface ReviewReportDetails {
79
94
  findingsCount: number;
80
95
  /** 结构化 JSON 落盘路径;落盘失败时为 null(仍返回渲染报告)。 */
81
96
  outFile: string | null;
97
+ reportId: string | null;
98
+ }
99
+
100
+ /**
101
+ * execute 入口的宽松 finding 形态——绕过 schema 校验的直接调用(如单测)
102
+ * 可能携带旧五档 outcome 或已废弃的 REFUTED verdict。
103
+ */
104
+ type LooseFinding = {
105
+ file: string;
106
+ line?: number;
107
+ category: string;
108
+ verdict?: string;
109
+ short_summary?: string;
110
+ summary: string;
111
+ failure_scenario: string;
112
+ outcome?: string;
113
+ };
114
+
115
+ /**
116
+ * 单条清洗:非法 verdict(含已废弃的 REFUTED)返回 null(剔除);非法 outcome
117
+ * 归一化为 skipped 并附注原始值。schema 保持严格,清洗在 schema 校验之前。
118
+ */
119
+ function sanitizeFinding(f: LooseFinding): { f: LooseFinding; note?: string } | null {
120
+ if (f.verdict !== undefined && !(VERDICT_VALUES as readonly string[]).includes(f.verdict)) return null;
121
+ let outcome = f.outcome;
122
+ let note: string | undefined;
123
+ if (outcome !== undefined && !(OUTCOME_VALUES as readonly string[]).includes(outcome)) {
124
+ note = `(outcome "${outcome}" 非法,已归一化为 skipped)`;
125
+ outcome = "skipped";
126
+ }
127
+ return { f: { ...f, outcome }, note };
128
+ }
129
+
130
+ /**
131
+ * 批量清洗,note 按清洗后数组索引记录(渲染层附注用)。
132
+ * `prepareArguments`(主防御,schema 校验前)与 `execute` 入口(双保险,
133
+ * 防绕过 prepareArguments 的直接调用)共用——模型路径的非法值在进入
134
+ * execute 前已被清洗,validateToolArguments 不会因边缘值 throw 掉整份报告。
135
+ */
136
+ function normalizeFindings(findings: LooseFinding[]): { findings: LooseFinding[]; notes: Map<number, string> } {
137
+ const out: LooseFinding[] = [];
138
+ const notes = new Map<number, string>();
139
+ for (const raw of findings) {
140
+ const s = sanitizeFinding(raw);
141
+ if (!s) continue;
142
+ const idx = out.length;
143
+ out.push(s.f);
144
+ if (s.note) notes.set(idx, s.note);
145
+ }
146
+ return { findings: out, notes };
82
147
  }
83
148
 
84
149
  // --- render -----------------------------------------------------------------
@@ -88,15 +153,19 @@ interface FindingInput {
88
153
  line?: number;
89
154
  category: string;
90
155
  verdict?: string;
156
+ short_summary?: string;
91
157
  summary: string;
92
158
  failure_scenario: string;
93
159
  outcome?: string;
160
+ /** normalize 附注(如非法 outcome 归一化说明),仅渲染进详情块。 */
161
+ note?: string;
94
162
  }
95
163
  interface ReportInput {
96
164
  level: string;
97
165
  target?: string;
98
166
  files_changed?: number;
99
167
  fanned_out?: boolean;
168
+ reportId?: string;
100
169
  findings: FindingInput[];
101
170
  }
102
171
 
@@ -118,7 +187,8 @@ function renderReport(p: ReportInput): string {
118
187
  const fanLabel = p.fanned_out === true ? "多智能体" : p.fanned_out === false ? "单遍自审" : "未标注";
119
188
  const targetStr = (p.target ?? "(whole diff)").replace(/`/g, "\\`"); // backtick inside the inline-code cell would close it early
120
189
  const filesStr = p.files_changed != null ? `${p.files_changed} 个文件` : "文件数未标注";
121
- lines.push(`\`${p.level}\` · \`${targetStr}\` · ${filesStr} · ${p.findings.length} 条发现 · ${fanLabel}`);
190
+ const idStr = p.reportId ? ` · 报告 \`${escapeCell(p.reportId)}\`` : "";
191
+ lines.push(`\`${p.level}\` · \`${targetStr}\` · ${filesStr} · ${p.findings.length} 条发现 · ${fanLabel}${idStr}`);
122
192
  lines.push("");
123
193
 
124
194
  if (p.findings.length === 0) {
@@ -130,7 +200,7 @@ function renderReport(p: ReportInput): string {
130
200
  lines.push("|---|------|------|------|------|");
131
201
  for (let i = 0; i < p.findings.length; i++) {
132
202
  const f = p.findings[i]!;
133
- lines.push(`| ${i + 1} | ${escapeCell(f.verdict ?? "")} | ${escapeCell(f.category)} | ${escapeCell(fmtLoc(f))} | ${escapeCell(f.summary)} |`);
203
+ lines.push(`| ${i + 1} | ${escapeCell(f.verdict ?? "")} | ${escapeCell(f.category)} | ${escapeCell(fmtLoc(f))} | ${escapeCell(f.short_summary ?? f.summary)} |`);
134
204
  }
135
205
  lines.push("");
136
206
  lines.push("**详情**");
@@ -138,9 +208,10 @@ function renderReport(p: ReportInput): string {
138
208
  p.findings.forEach((f, i) => {
139
209
  const v = f.verdict ? ` *(${f.verdict})*` : "";
140
210
  const out = f.outcome ? `\n修复结果:\`${f.outcome}\`` : "";
211
+ const note = f.note ? `\n${f.note}` : "";
141
212
  lines.push(`**${i + 1}. ${fmtLoc(f)} — ${f.category}**${v}`);
142
213
  lines.push(`概述:${f.summary}`);
143
- lines.push(`场景:${f.failure_scenario}${out}`);
214
+ lines.push(`场景:${f.failure_scenario}${out}${note}`);
144
215
  lines.push("");
145
216
  });
146
217
  return lines.join("\n").trimEnd();
@@ -156,13 +227,41 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
156
227
  promptSnippet: "review_report — report structured code-review findings (renders Markdown + writes JSON for CI)",
157
228
  promptGuidelines: [
158
229
  "After verify + dedup, call `review_report` once with { level, findings } (most-severe first; empty array if none survived). Do not also hand-write the Markdown table — this tool renders it.",
159
- "On re-report after --fix, set each finding's `outcome` (fully_achieved / mostly_achieved / partially_achieved / not_achieved / unclear_from_transcript).",
230
+ "On re-report after --fix, set each finding's `outcome` (fixed / skipped / no_change_needed).",
160
231
  "Use this tool only when the code-review skill instructs reporting findings; otherwise follow the active output format.",
161
232
  ],
162
233
  parameters: ReviewReportParams,
163
234
 
235
+ // 主防御:schema 校验之前清洗非法值(模型路径下校验失败即 throw、工具不执行,
236
+ // 因此 execute 内的防御对模型不可达)。返回符合 schema 的对象——非法 verdict
237
+ // 的 finding 剔除、非法 outcome 归一化为 skipped;note 附注由 execute 层基于
238
+ // 清洗结果生成(schema 对象不含 note 字段)。
239
+ prepareArguments(args) {
240
+ if (typeof args !== "object" || args === null) return args as Static<typeof ReviewReportParams>;
241
+ const raw = args as { findings?: unknown };
242
+ if (!Array.isArray(raw.findings)) return args as Static<typeof ReviewReportParams>;
243
+ const { findings } = normalizeFindings(raw.findings as unknown as LooseFinding[]);
244
+ return { ...raw, findings } as Static<typeof ReviewReportParams>;
245
+ },
246
+
164
247
  async execute(toolCallId, params, _signal, _onUpdate, ctx) {
165
- const report = renderReport(params);
248
+ // normalize(双保险)——绕过 prepareArguments 的直接调用(如单测)可能携带
249
+ // 旧五档 outcome 或已废弃的 REFUTED verdict。归一化而非整单拒绝:非法
250
+ // verdict 的 finding 剔除,非法 outcome 归一化为 skipped 并附注;渲染与落盘
251
+ // 永远只含合法值(spec: 消费方永远看到合法值)。
252
+ const { findings: cleaned, notes } = normalizeFindings(
253
+ (params.findings ?? []) as unknown as LooseFinding[],
254
+ );
255
+ const findings: FindingInput[] = cleaned.map((f, i) => ({ ...f, note: notes.get(i) }));
256
+
257
+ const report = renderReport({
258
+ level: params.level,
259
+ target: params.target,
260
+ files_changed: params.files_changed,
261
+ fanned_out: params.fanned_out,
262
+ reportId: params.report_id,
263
+ findings,
264
+ });
166
265
 
167
266
  let outFile: string | null = null;
168
267
  let writeError: string | null = null;
@@ -178,11 +277,12 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
178
277
  JSON.stringify(
179
278
  {
180
279
  level: params.level,
280
+ reportId: params.report_id ?? null,
181
281
  target: params.target ?? null,
182
282
  filesChanged: params.files_changed ?? null,
183
283
  fannedOut: params.fanned_out ?? null,
184
284
  generatedAt: now.toISOString(),
185
- findings: params.findings,
285
+ findings,
186
286
  },
187
287
  null,
188
288
  2,
@@ -200,8 +300,9 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
200
300
  : `\n\n[结构化落盘失败(${writeError ?? "未知原因"}),仅渲染报告]`;
201
301
  const details: ReviewReportDetails = {
202
302
  level: params.level,
203
- findingsCount: params.findings.length,
303
+ findingsCount: findings.length,
204
304
  outFile,
305
+ reportId: params.report_id ?? null,
205
306
  };
206
307
  return {
207
308
  content: [{ type: "text" as const, text: report + tail }],
@@ -217,8 +318,8 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
217
318
  // throw, so a failure here degrades to the pre-change behavior rather than erroring.
218
319
  renderResult(result, _options, _theme, _context) {
219
320
  const text = result.content
220
- .filter((c) => c.type === "text")
221
- .map((c) => (c.type === "text" ? c.text : ""))
321
+ .filter((c): c is Extract<(typeof result.content)[number], { type: "text" }> => c.type === "text")
322
+ .map((c) => c.text)
222
323
  .join("\n");
223
324
  return new Markdown(text, 0, 0, getMarkdownTheme());
224
325
  },
@@ -28,8 +28,9 @@ import {
28
28
  type AgentSpawnResult,
29
29
  } from "@fyeeme/pi-subagent-core";
30
30
 
31
- /** Default concurrency ceiling when PI_MAX_CONCURRENT_SUBAGENTS is unset/invalid. */
32
- const DEFAULT_MAX_CONCURRENCY = 8;
31
+ /** Default concurrency ceiling when PI_MAX_CONCURRENT_SUBAGENTS is unset/invalid.
32
+ * 20 = CC's CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20 (2.1.227 binary empirical). */
33
+ const DEFAULT_MAX_CONCURRENCY = 20;
33
34
 
34
35
  /**
35
36
  * Effective concurrency ceiling, configurable via PI_MAX_CONCURRENT_SUBAGENTS
@@ -46,8 +47,9 @@ function getMaxConcurrency(): number {
46
47
  * runaway agent cannot otherwise be bounded. Generous (well above the ~10–15
47
48
  * turns the 4-angle finder/verify agents need) so legitimate work is not
48
49
  * truncated; callers may override with a smaller or larger explicit value.
50
+ * 50 = CC's FORKED_AGENT_DEFAULT_MAX_TURNS (2.1.227 binary empirical).
49
51
  */
50
- const DEFAULT_FANOUT_MAX_TURNS = 25;
52
+ const DEFAULT_FANOUT_MAX_TURNS = 50;
51
53
 
52
54
  // Module-level registry so abortAgent can reach in-flight calls. callIds are
53
55
  // unique per tool call (toolCallId#index), so a single registry is safe.
@@ -61,6 +63,12 @@ const SubagentParams = Type.Object({
61
63
  description: "Prompt(s) for the sub-agent(s). single/chain use order; parallel runs all.",
62
64
  }),
63
65
  model: Type.Optional(Type.String({ description: "Full model id (e.g. claude-sonnet-5). Omit for the session default." })),
66
+ thinking: Type.Optional(
67
+ Type.Union(
68
+ ["off", "minimal", "low", "medium", "high", "xhigh", "max"].map((l) => Type.Literal(l)),
69
+ { description: "Thinking level for the sub-agent(s), passed as --thinking. Omit for the model default." },
70
+ ),
71
+ ),
64
72
  systemPrompt: Type.Optional(Type.String({ description: "Appended to the sub-agent's system prompt." })),
65
73
  tools: Type.Optional(Type.Array(Type.String(), { description: "Tool whitelist for the sub-agent. Omit for default tools." })),
66
74
  parallelism: Type.Optional(
@@ -251,6 +259,7 @@ export const subagentTool = defineTool<typeof SubagentParams, SubagentDetails>({
251
259
  const baseOpts: Omit<AgentSpawnOptions, "callId" | "task"> = {
252
260
  cwd,
253
261
  model: params.model,
262
+ thinking: params.thinking,
254
263
  tools: params.tools,
255
264
  signal,
256
265
  // Default turn budget applies when omitted; an explicit 0 is honored by