@fyeeme/pi-review 1.0.1 → 1.0.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +13 -5
- package/package.json +2 -2
- package/skills/code-review/SKILL.md +180 -73
- package/skills/simplify/SKILL.md +90 -57
- package/src/commands/code-simplify.ts +5 -4
- package/src/tools/review_report.ts +124 -23
- package/src/tools/subagent.ts +12 -3
package/README.md
CHANGED
|
@@ -93,9 +93,11 @@ An LLM-callable tool that spawns one or more real pi subprocesses:
|
|
|
93
93
|
receive the `subagent` tool in its default toolset, so it cannot recurse. A
|
|
94
94
|
caller opts in by listing `subagent` in the child's `tools` whitelist; set
|
|
95
95
|
`PI_SUBAGENT_MAX_SPAWN_DEPTH` to allow multi-level fan-out up to a hard cap.
|
|
96
|
-
- **Default turn budget** — fan-out agents get a finite default `maxTurns` (
|
|
96
|
+
- **Default turn budget** — fan-out agents get a finite default `maxTurns` (50,
|
|
97
|
+
aligned to CC's `FORKED_AGENT_DEFAULT_MAX_TURNS` in 2.1.227)
|
|
97
98
|
when the caller omits it; an explicit `0` is honored.
|
|
98
|
-
- **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default
|
|
99
|
+
- **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default 20,
|
|
100
|
+
aligned to CC's `CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20` in 2.1.227;
|
|
99
101
|
invalid values fall back to the default).
|
|
100
102
|
|
|
101
103
|
Each sub-agent is a full `pi --mode json -p --no-session` run. Progress streams
|
|
@@ -144,6 +146,12 @@ sub-agents are spawned and *which mode* is chosen.
|
|
|
144
146
|
|
|
145
147
|
## Status
|
|
146
148
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
149
|
+
`review_report` is built — schema aligned to CC `ReportFindings` (2.1.227
|
|
150
|
+
empirical): 3-state `outcome` (`fixed`/`skipped`/`no_change_needed`), 2-value
|
|
151
|
+
`verdict` (`CONFIRMED`/`PLAUSIBLE`), `short_summary` (≤60, table overview),
|
|
152
|
+
`report_id` for fixed-later re-reports; renders the Chinese Markdown report
|
|
153
|
+
and writes JSON to `<cwd>/.pi/review/` for CI / `--fix` / `--comment`.
|
|
154
|
+
|
|
155
|
+
Remaining Phase 2 item (not yet built): a `review_verify` tool encapsulating
|
|
156
|
+
3-vote adversarial verify. `--share` already routes through lavish-axi (see
|
|
157
|
+
the code-review skill).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@fyeeme/pi-review",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.3",
|
|
4
4
|
"description": "Review & cleanup extension for pi. Registers /code-review and /code-simplify commands plus a general-purpose `subagent` tool that spawns parallel pi subprocesses — providing the real fan-out capability the code-review and simplify skills (bundled under `skills/`) need for their multi-agent flows. The /code-simplify handler uses ctx.getContextUsage() to decide parallel vs single-pass mode deterministically.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -35,7 +35,7 @@
|
|
|
35
35
|
"typecheck": "tsc"
|
|
36
36
|
},
|
|
37
37
|
"dependencies": {
|
|
38
|
-
"@fyeeme/pi-subagent-core": "^0.
|
|
38
|
+
"@fyeeme/pi-subagent-core": "^0.4.0"
|
|
39
39
|
},
|
|
40
40
|
"peerDependencies": {
|
|
41
41
|
"@earendil-works/pi-ai": ">=0.84.1",
|
|
@@ -10,24 +10,27 @@ description: "Review the current diff for correctness bugs and reuse/simplificat
|
|
|
10
10
|
earlier v2.1.220 reconstruction. Every section below was located in the
|
|
11
11
|
extracted strings (cc_strings_223.txt) and verified.
|
|
12
12
|
|
|
13
|
-
What CC 2.1.
|
|
14
|
-
- Effort
|
|
15
|
-
|
|
13
|
+
What CC 2.1.227 actually contains (verified against the binary):
|
|
14
|
+
- Effort quad tuple {correctnessAngles, perAngle, maxFindings, sweep}:
|
|
15
|
+
medium {3,6,8,false} / high {3,6,10,false} / xhigh {5,8,15,true} / max
|
|
16
|
+
same structure as xhigh. medium = precision; high+ = recall
|
|
17
|
+
("err on the side of surfacing"); xhigh/max add a gap-hunt (≤8 new
|
|
18
|
+
candidates). Correctness angles are taken in order A→E (`slice(0, N)`).
|
|
19
|
+
- Inline finder allocation: medium/high = 8 finders (A/B/C + 3 cleanup +
|
|
20
|
+
altitude + conventions); xhigh/max = 10 finders (A–E + same). Each
|
|
21
|
+
cleanup angle gets its own finder.
|
|
16
22
|
- Low effort: 1 diff pass, no verify, target min(files_changed, 4) findings.
|
|
17
|
-
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
23
|
+
- Verify via an independent agent, grouped by (file, line) (absorbed from
|
|
24
|
+
the workflow GROUP_VERDICT_SCHEMA): CONFIRMED / PLAUSIBLE / REFUTED,
|
|
25
|
+
"PLAUSIBLE by default". Keep CONFIRMED + PLAUSIBLE, drop REFUTED.
|
|
26
|
+
- ReportFindings schema: verdict CONFIRMED|PLAUSIBLE, outcome
|
|
27
|
+
fixed|skipped|no_change_needed, finding carries short_summary (≤60).
|
|
21
28
|
- Angles A–E + Reuse/Simplification/Efficiency/Altitude + Conventions,
|
|
22
29
|
verbatim (same source variables the /simplify skill reuses).
|
|
23
|
-
- Verify via an independent agent: CONFIRMED / PLAUSIBLE / REFUTED,
|
|
24
|
-
"PLAUSIBLE by default". Keep CONFIRMED + PLAUSIBLE, drop REFUTED.
|
|
25
30
|
- Gap-hunt (xhigh/max): one fresh finder hunting only for gaps not
|
|
26
31
|
already listed (CC's Sweep phase: "Fresh finder hunting only for gaps").
|
|
27
|
-
-
|
|
28
|
-
|
|
29
|
-
verdict/summary/failure_scenario per finding.
|
|
30
|
-
- --share publishes an Artifact; --fix applies findings to the working tree.
|
|
32
|
+
- Fixed-later obligation (CC Q8m): later fixes in the session must
|
|
33
|
+
re-report findings with updated outcome.
|
|
31
34
|
|
|
32
35
|
Invocation: /code-review [low|medium|high|xhigh|max] [--fix] [--comment] [--share] [<target>]
|
|
33
36
|
target = Class#method | file path | PR number | branch name
|
|
@@ -43,7 +46,7 @@ description: "Review the current diff for correctness bugs and reuse/simplificat
|
|
|
43
46
|
1. Output — CC calls a ReportFindings tool with {level, findings}; Pi
|
|
44
47
|
uses this extension's `review_report` tool (the Pi counterpart
|
|
45
48
|
to ReportFindings, verdict/outcome enums aligned to CC
|
|
46
|
-
v2.1.
|
|
49
|
+
v2.1.227): it renders the Chinese Markdown report (table +
|
|
47
50
|
details) back to the conversation AND writes a
|
|
48
51
|
machine-readable JSON to <cwd>/.pi/review/ for CI / --fix /
|
|
49
52
|
--comment. If the tool is absent, fall back to printing the
|
|
@@ -72,17 +75,26 @@ altitude, and conventions findings when the output cap forces a cut.
|
|
|
72
75
|
|
|
73
76
|
## Effort levels
|
|
74
77
|
|
|
75
|
-
| Level | Intent | Verify | Subagents |
|
|
78
|
+
| Level | Intent | Verify | Subagents | 四元组 `{correctnessAngles, perAngle, maxFindings, sweep}` |
|
|
76
79
|
|-------|--------|--------|-----------|------------|
|
|
77
|
-
| low (default) | quick scan | no | no | min(files_changed, 4) |
|
|
78
|
-
| medium | **precision** — surface only findings a maintainer would act on | independent
|
|
79
|
-
| high | **recall** — catch every real bug a careful reviewer would; **err on the side of surfacing** |
|
|
80
|
-
| xhigh
|
|
80
|
+
| low (default) | quick scan | no | no | 上限 `min(files_changed, 4)` |
|
|
81
|
+
| medium | **precision** — surface only findings a maintainer would act on | independent verifier (grouped) | 8 finders | `{3, 6, 8, false}` |
|
|
82
|
+
| high | **recall** — catch every real bug a careful reviewer would; **err on the side of surfacing** | recall-biased verifier (grouped) | 8 finders | `{3, 6, 10, false}` |
|
|
83
|
+
| xhigh | recall + **gap-hunt** | recall-biased verifier (grouped) | 10 finders + 1 gap | `{5, 8, 15, true}` |
|
|
84
|
+
| max | 同 xhigh | 同 xhigh | 同 xhigh | 同 xhigh |
|
|
81
85
|
|
|
82
86
|
**max 与 xhigh 结构相同**:fan-out / verify / sweep 完全一致,差别仅在模型 reasoning effort(CC v2.1.226 注释实证:`max → same structure as xhigh (the API reasoning effort differs, not the fan-out)`)。若运行时不支持调节 reasoning effort,max 在结构上退化为 xhigh——不要因档名而期待更多 fan-out。
|
|
83
87
|
|
|
84
|
-
|
|
85
|
-
|
|
88
|
+
The quad tuple parameterizes the whole pipeline (CC inline semantics, verified 2.1.227):
|
|
89
|
+
|
|
90
|
+
- `correctnessAngles` — how many correctness angles A–E run, taken **in order** (medium/high: A/B/C; xhigh/max: A–E).
|
|
91
|
+
- `perAngle` — candidate cap per finder (6 at medium/high, 8 at xhigh/max).
|
|
92
|
+
- `maxFindings` — the report cap after verify (8 / 10 / 15).
|
|
93
|
+
- `sweep` — whether Phase 3 gap-hunt runs (xhigh/max only, ≤ 8 new candidates).
|
|
94
|
+
|
|
95
|
+
Each finder surfaces up to `perAngle` candidate findings with `file`, `line`, a
|
|
96
|
+
one-line `summary`, a ≤60-char `short_summary`, and a concrete
|
|
97
|
+
`failure_scenario`.
|
|
86
98
|
|
|
87
99
|
If a target argument was provided, review that target instead of the whole diff.
|
|
88
100
|
|
|
@@ -96,6 +108,43 @@ commit. If a PR number, branch name, or file path was passed as an argument,
|
|
|
96
108
|
review that target instead. Treat this diff as the review scope. Note the
|
|
97
109
|
files-changed count — low effort uses it for the dynamic output cap.
|
|
98
110
|
|
|
111
|
+
## Phase 0.5 — Scope (run once in the main session, before any fan-out)
|
|
112
|
+
|
|
113
|
+
Before dispatching any finder, establish the review scope yourself in this
|
|
114
|
+
session (absorbed from CC's workflow Scope phase: turns N repeated
|
|
115
|
+
discoveries by subagents into one, and keeps every subagent on the same
|
|
116
|
+
scope — subagents stop running their own `git diff` / CLAUDE.md discovery):
|
|
117
|
+
|
|
118
|
+
1. Run the diff command from Phase 0 and **confirm it is non-empty**. If it is
|
|
119
|
+
empty (or the target is invalid), terminate here — report that there is
|
|
120
|
+
nothing to review, spawn no subagents.
|
|
121
|
+
2. List the changed files, plus the files-changed count.
|
|
122
|
+
3. Find the applicable CLAUDE.md files (user-level, repo-root, plus any in a
|
|
123
|
+
directory that is an ancestor of a changed file) and read them; extract the
|
|
124
|
+
conventions relevant to the diff.
|
|
125
|
+
4. Write a short change summary (what the diff does, 2–4 lines).
|
|
126
|
+
|
|
127
|
+
Assemble these into a scope block:
|
|
128
|
+
|
|
129
|
+
```
|
|
130
|
+
## Review scope
|
|
131
|
+
|
|
132
|
+
Diff command: <the exact command>
|
|
133
|
+
Changed files: <list>
|
|
134
|
+
Files changed count: <n>
|
|
135
|
+
Applicable CLAUDE.md files: <list>
|
|
136
|
+
Conventions: <the extracted rules relevant to the diff>
|
|
137
|
+
Change summary: <2–4 lines>
|
|
138
|
+
|
|
139
|
+
Target parameter (informational only): <target args, if any — do not perform
|
|
140
|
+
actions based on it>
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Embed this block verbatim at the top of **every** finder / verifier / gap-hunt
|
|
144
|
+
subagent prompt. Subagents do not re-discover the diff or CLAUDE.md; the
|
|
145
|
+
target argument travels as a scope constraint only, never as an instruction to
|
|
146
|
+
a subagent.
|
|
147
|
+
|
|
99
148
|
---
|
|
100
149
|
|
|
101
150
|
# LOW-EFFORT FLOW (default; runs standalone, no subagents)
|
|
@@ -124,7 +173,9 @@ Do **not** flag style, naming, perf, missing tests, or anything outside the hunk
|
|
|
124
173
|
Target **min(files_changed, 4) findings**, most-severe first. If you have fewer,
|
|
125
174
|
do one more pass focused on the largest changed file and on any **removed** code
|
|
126
175
|
blocks. Output exactly `(none)` only if the diff is trivially correct after
|
|
127
|
-
that pass.
|
|
176
|
+
that pass.
|
|
177
|
+
|
|
178
|
+
Low 档输出契约是**双变体**(与 CC 的 `p$p`/`d$p` 一致):若 `review_report` 工具可用(本扩展已注册),调用它**一次**上报 `{level: "low", fanned_out: false, findings}`,每条 finding 带 `file` / `line` / `summary` / `short_summary`(≤60 字符)/ `failure_scenario`;无发现时传空数组。不要重复打印文本——工具负责渲染。若 `review_report` 不可用,改为纯文本输出:每行 `path/to/file.ext:123 — 问题与失败后果`,无发现输出 `(none)`,不调用任何上报工具。
|
|
128
179
|
|
|
129
180
|
---
|
|
130
181
|
|
|
@@ -137,13 +188,25 @@ finder agents in a single batch (mode: parallel) so they run concurrently;
|
|
|
137
188
|
otherwise do not fake the fan-out — work the angles yourself in sequence in
|
|
138
189
|
this same context, or report that the subagent capability is unavailable.
|
|
139
190
|
|
|
140
|
-
**Finder allocation** (
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
191
|
+
**Finder allocation** (CC inline, verified 2.1.227): the number of correctness
|
|
192
|
+
angles comes from the effort quad tuple, taken **in order A→E** (`slice(0, N)`
|
|
193
|
+
— do not hand-pick angles; that makes runs unreproducible):
|
|
194
|
+
|
|
195
|
+
- **medium / high** (3 correctness angles): **8 finders** — A, B, C + one
|
|
196
|
+
finder each for Reuse, Simplification, Efficiency + one Altitude + one
|
|
197
|
+
Conventions.
|
|
198
|
+
- **xhigh / max** (5 correctness angles): **10 finders** — A, B, C, D, E + the
|
|
199
|
+
same 3 cleanup finders + Altitude + Conventions.
|
|
200
|
+
|
|
201
|
+
Each cleanup angle (Reuse / Simplification / Efficiency) gets its own finder;
|
|
202
|
+
Altitude and Conventions are independent finders. Never silently drop an
|
|
203
|
+
angle — if you must consolidate (subagent unavailable), fold the cleanup
|
|
204
|
+
angles into a correctness finder, but say so in the report.
|
|
205
|
+
|
|
206
|
+
**Suppression 禁令(xhigh/max)** — different finders may surface different
|
|
207
|
+
candidates for the same line with different reasons. At xhigh/max all of them
|
|
208
|
+
are recorded and pass through verify independently: do NOT let one angle's
|
|
209
|
+
conclusions suppress another's — record both.
|
|
147
210
|
|
|
148
211
|
The correctness angles hunt for bugs; the cleanup angles hunt for cleanup in
|
|
149
212
|
the changed code. Cleanup, altitude, and conventions candidates use the same
|
|
@@ -186,29 +249,33 @@ through a registry/session/global — e.g. a caching provider holding a
|
|
|
186
249
|
wrapper forwards all the methods the callers actually use.
|
|
187
250
|
|
|
188
251
|
### Reuse
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
252
|
+
|
|
253
|
+
Flag new code that re-implements something the codebase
|
|
254
|
+
already has — Grep shared/utility modules and files adjacent to the change,
|
|
255
|
+
and name the existing helper to call instead.
|
|
192
256
|
|
|
193
257
|
### Simplification
|
|
258
|
+
|
|
194
259
|
Flag unnecessary complexity the diff adds: redundant or derivable state,
|
|
195
|
-
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
196
|
-
simpler form that does the same job.
|
|
260
|
+
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
261
|
+
the simpler form that does the same job.
|
|
197
262
|
|
|
198
263
|
### Efficiency
|
|
264
|
+
|
|
199
265
|
Flag wasted work the diff introduces: redundant computation or repeated I/O,
|
|
200
|
-
independent operations run sequentially, blocking work added to startup or
|
|
201
|
-
paths. Also flag long-lived objects built from closures or captured
|
|
202
|
-
— they keep the entire enclosing scope alive for the object's
|
|
203
|
-
leak when that scope holds large values); prefer a
|
|
204
|
-
the fields it needs. Name the cheaper
|
|
266
|
+
independent operations run sequentially, blocking work added to startup or
|
|
267
|
+
hot paths. Also flag long-lived objects built from closures or captured
|
|
268
|
+
environments — they keep the entire enclosing scope alive for the object's
|
|
269
|
+
lifetime (a memory leak when that scope holds large values); prefer a
|
|
270
|
+
class/struct that copies only the fields it needs. Name the cheaper
|
|
271
|
+
alternative.
|
|
205
272
|
|
|
206
273
|
### Altitude
|
|
207
|
-
Check that each change is implemented at the right depth, not as a fragile
|
|
208
|
-
bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
|
|
209
|
-
deep enough — prefer generalizing the underlying mechanism over adding special
|
|
210
|
-
cases.
|
|
211
274
|
|
|
275
|
+
Check that each change is implemented at the right depth, not as a fragile
|
|
276
|
+
bandaid. Special cases layered on shared infrastructure are a sign the fix
|
|
277
|
+
isn't deep enough — prefer generalizing the underlying mechanism over adding
|
|
278
|
+
special cases.
|
|
212
279
|
### Conventions (CLAUDE.md)
|
|
213
280
|
Find the CLAUDE.md files that govern the changed code: the user-level
|
|
214
281
|
~/.claude/CLAUDE.md, the repo-root CLAUDE.md, plus any CLAUDE.md or
|
|
@@ -228,15 +295,36 @@ are the dominant cause of misses.
|
|
|
228
295
|
|
|
229
296
|
## Phase 2 — Dedup and verify
|
|
230
297
|
|
|
231
|
-
Dedup near-duplicates (same defect, same location, same reason → keep one
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
298
|
+
Dedup near-duplicates (same defect, same location, same reason → keep one;
|
|
299
|
+
different reasons for the same line are NOT duplicates — at xhigh/max both
|
|
300
|
+
are kept per the suppression ban).
|
|
301
|
+
|
|
302
|
+
Then verify each candidate **grouped by location**. If the `subagent` tool is
|
|
303
|
+
available: group the deduplicated candidates by `(file, line)`; dispatch ONE
|
|
304
|
+
independent verify agent per group (mode: parallel, one prompt per group),
|
|
305
|
+
giving it the scope block, the diff, the relevant file(s), and the full
|
|
306
|
+
candidate list for that location with each candidate's index. The verifier
|
|
307
|
+
returns a verdict per candidate:
|
|
308
|
+
|
|
309
|
+
```
|
|
310
|
+
[{ "index": <candidate index>, "verdict": "CONFIRMED" | "PLAUSIBLE" | "REFUTED", "evidence": "<quote/argument>" }, ...]
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
Grouping is by location, NOT dedup — each candidate is judged independently;
|
|
314
|
+
same-location candidates may describe different defects. A candidate the
|
|
315
|
+
verifier omitted (interrupted or skipped an index) is **dropped** — never
|
|
316
|
+
invent a PLAUSIBLE for it. One verifier failure drops its whole group (the
|
|
317
|
+
same trade-off CC's workflow makes); if you are not confident in a group's
|
|
318
|
+
verifier, fall back to one verifier per candidate for that group.
|
|
319
|
+
|
|
320
|
+
This group-by-location verify is a deliberate absorption of CC's workflow
|
|
321
|
+
optimization into the inline path (CC inline dispatches one verifier per
|
|
322
|
+
candidate): a location with 3 candidates costs 1 verifier instead of 3 —
|
|
323
|
+
~40% fewer verifier agents is the expectation, not a guarantee.
|
|
324
|
+
|
|
325
|
+
If `subagent` is unavailable, fall back to re-checking each candidate yourself
|
|
326
|
+
(self-check). Keep **CONFIRMED and PLAUSIBLE**, drop REFUTED. Give each
|
|
327
|
+
surviving finding a verdict:
|
|
240
328
|
|
|
241
329
|
- **CONFIRMED** — can name the inputs/state that trigger it and the wrong
|
|
242
330
|
output or crash. Quote the line.
|
|
@@ -252,6 +340,12 @@ optional field), falsy-zero treated as missing, off-by-one on a boundary the
|
|
|
252
340
|
code does not exclude, retry storms / partial failures, regex/allowlist that
|
|
253
341
|
lost an anchor. These are PLAUSIBLE.
|
|
254
342
|
|
|
343
|
+
**Recall bias by level** — at high/xhigh/max, a single non-REFUTED verdict
|
|
344
|
+
keeps the candidate: do NOT drop it on uncertainty ("speculative", "depends
|
|
345
|
+
on runtime state"). That is the recall contract of high+. Medium is the
|
|
346
|
+
precision level: there, additionally weigh whether a maintainer would act on
|
|
347
|
+
the finding before keeping it.
|
|
348
|
+
|
|
255
349
|
**REFUTED** only when constructible from the code: factually wrong (quote the
|
|
256
350
|
actual line); provably impossible (type/constant/invariant — show it); already
|
|
257
351
|
handled in this diff (cite the guard); or pure style with no observable effect.
|
|
@@ -260,7 +354,8 @@ handled in this diff (cite the guard); or pure style with no observable effect.
|
|
|
260
354
|
|
|
261
355
|
At **xhigh and max**, after Phase 2 dedup, dispatch ONE fresh finder agent (the
|
|
262
356
|
`subagent` tool) that has never seen the candidates and hunts only for gaps not
|
|
263
|
-
already listed
|
|
357
|
+
already listed — **at most 8 new candidates**. Feed anything it finds back
|
|
358
|
+
through Phase 2 verify before keeping it.
|
|
264
359
|
|
|
265
360
|
Constrain it so exploration can't run away (Pi adaptation — CC's workflow bounds
|
|
266
361
|
this differently):
|
|
@@ -285,26 +380,37 @@ At **high and below**, skip Phase 3.
|
|
|
285
380
|
|
|
286
381
|
Report the findings via the `review_report` tool (this extension's counterpart
|
|
287
382
|
to CC's `ReportFindings`) — call it **once** with
|
|
288
|
-
`{ level, target, files_changed, fanned_out, findings }`, findings
|
|
289
|
-
most-severe first (empty array if nothing survived verification). The
|
|
290
|
-
renders the Chinese Markdown report (table + details) back to the
|
|
291
|
-
AND writes a machine-readable JSON to `<cwd>/.pi/review/` for CI /
|
|
292
|
-
`--comment`. Do **not** also hand-write the Markdown table.
|
|
383
|
+
`{ level, target, files_changed, fanned_out, report_id, findings }`, findings
|
|
384
|
+
ranked most-severe first (empty array if nothing survived verification). The
|
|
385
|
+
tool renders the Chinese Markdown report (table + details) back to the
|
|
386
|
+
conversation AND writes a machine-readable JSON to `<cwd>/.pi/review/` for CI /
|
|
387
|
+
`--fix` / `--comment`. Do **not** also hand-write the Markdown table.
|
|
293
388
|
|
|
294
389
|
Each finding in the array carries: `file`, `line` (optional), `category`
|
|
295
390
|
(`correctness` / `reuse` / `simplification` / `efficiency` / `altitude` /
|
|
296
391
|
`conventions`, or a more specific slug like `test-coverage`), `verdict`
|
|
297
|
-
(`CONFIRMED` / `PLAUSIBLE`
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
392
|
+
(`CONFIRMED` / `PLAUSIBLE`), `short_summary` (≤60 字符、纯声明——去掉理由与
|
|
393
|
+
后果,汇总表概述列优先使用它;示例:`"off-by-one in loop bound"`),
|
|
394
|
+
`summary` (一行中文,含理由与后果,详情块使用), `failure_scenario`
|
|
395
|
+
(concrete input/state → wrong output/crash; for cleanup findings, the
|
|
396
|
+
concrete cost — Chinese). When re-reporting after applying `--fix`, set
|
|
397
|
+
`outcome` on each finding (`fixed` / `skipped` / `no_change_needed` — CC
|
|
398
|
+
`ReportFindings` 三档,2.1.227 实证).
|
|
399
|
+
|
|
400
|
+
Cap = `maxFindings` from the effort table: `min(files_changed, 4)` at low; 8
|
|
401
|
+
at medium; 10 at high; 15 at xhigh/max. If more than the cap survive, send
|
|
402
|
+
the cap most severe (correctness outranks cleanup/altitude/conventions when
|
|
403
|
+
cutting; CONFIRMED outranks PLAUSIBLE). If nothing survives, send an empty
|
|
404
|
+
`findings` array — the tool prints a zero-count header.
|
|
405
|
+
|
|
406
|
+
**Fixed-later 义务** (CC `Q8m`): if, after this report, any later work in this
|
|
407
|
+
session fixes one of the reported findings (a user-requested fix, or a fix
|
|
408
|
+
that comes along with other changes), you MUST call `review_report` again
|
|
409
|
+
with the same `report_id`, the same findings, and updated `outcome` values —
|
|
410
|
+
before writing any text summary. The re-report updates states only; it does
|
|
411
|
+
not repeat the findings text. Generate the `report_id` (e.g. `review-<ts>`)
|
|
412
|
+
on the first report and reuse it on every re-report so consumers can merge
|
|
413
|
+
the files by id.
|
|
308
414
|
|
|
309
415
|
**全部用中文**:`summary` 与 `failure_scenario` 一律中文;`verdict`、`category`、
|
|
310
416
|
`outcome` 作为标识符保留英文 token。
|
|
@@ -326,10 +432,11 @@ directly — correctness bugs and reuse/simplification/efficiency cleanups alike
|
|
|
326
432
|
Skip any finding whose fix would change intended behavior, require changes well
|
|
327
433
|
outside the reviewed diff, or that you judge to be a false positive — note the
|
|
328
434
|
skip rather than arguing with it. Then call `review_report` once more to
|
|
329
|
-
re-report, setting `outcome` on each finding (`
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
replaces the hand-written summary and makes the fix result
|
|
435
|
+
re-report (same `report_id`), setting `outcome` on each finding (`fixed` =
|
|
436
|
+
applied and verified / `skipped` = real but not applied, incl. reverted /
|
|
437
|
+
`no_change_needed` = not applicable or already handled). This structured
|
|
438
|
+
re-report replaces the hand-written summary and makes the fix result
|
|
439
|
+
machine-consumable.
|
|
333
440
|
If `review_report` is unavailable, fall back to a brief text summary of what was
|
|
334
441
|
fixed and what was skipped.
|
|
335
442
|
|
package/skills/simplify/SKILL.md
CHANGED
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: simplify
|
|
3
|
-
description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that.
|
|
3
|
+
description: "Review the changed code for reuse, simplification, efficiency, and altitude cleanups, then apply the fixes. Quality only — it does not hunt for bugs; use /code-review for that. v3 (from Claude Code CLI v2.1.227, symbol-level verified) — 4 cleanup agents fan out in parallel when context allows, else a single-pass inline cleanup; either way the fixes are applied, verified against the project's check command, and auto-reverted on failure, then reported as structured outcomes via review_report."
|
|
4
4
|
---
|
|
5
5
|
|
|
6
6
|
<!--
|
|
7
|
-
Origin: Claude Code built-in skill `/simplify` (CLI v2.1.
|
|
8
|
-
engineered from bin/claude.exe
|
|
7
|
+
Origin: Claude Code built-in skill `/simplify` (CLI v2.1.227), reverse-
|
|
8
|
+
engineered from bin/claude.exe raw bytes. Pi registers it as /code-simplify.
|
|
9
9
|
|
|
10
10
|
Lineage:
|
|
11
11
|
v2.1.220 → the first reconstruction (v1)
|
|
@@ -14,6 +14,27 @@ description: "Review the changed code for reuse, simplification, efficiency, and
|
|
|
14
14
|
PARALLEL/SINGLE-PASS split are unchanged vs v2.1.220. The 4
|
|
15
15
|
angle bodies are shared verbatim with code-review's cleanup
|
|
16
16
|
angles (same source variables in the binary).
|
|
17
|
+
v2.1.227 → symbol-level verified 2026-08-11 from raw bytes: skill bodies
|
|
18
|
+
VBv/KBv (with interpolated c$e / m7t / u$e / d$e / p$e) are
|
|
19
|
+
unchanged; the mode guard is Dii (see below); fan-out defaults
|
|
20
|
+
nJu=20 / lKs=50 are now mirrored in the subagent tool.
|
|
21
|
+
|
|
22
|
+
CC 2.1.227 empirical evidence (symbol-level, extracted from bin/claude.exe):
|
|
23
|
+
- $u({name: "simplify", ..., getPromptForCommand(args, ctx)}) registers the
|
|
24
|
+
command; no getContext → default "inline" execution: the mode body is
|
|
25
|
+
injected into the main conversation and the model dispatches the 4
|
|
26
|
+
cleanup agents itself via the Agent tool (mi = "Agent", alias oj =
|
|
27
|
+
"Task"), "all in a single message so they run concurrently".
|
|
28
|
+
- Dii(ctx) — the PARALLEL/SINGLE-PASS guard: single-pass when
|
|
29
|
+
ctx.agentContext && ok(ctx.agentContext) >= wV() (ok = depth function:
|
|
30
|
+
main=0, subagent=depth; wV() = CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH,
|
|
31
|
+
default 3, feature flag tengu_hazel_trellis) OR the Agent tool is not in
|
|
32
|
+
the options.tools allowlist (Pa matches by name/aliases).
|
|
33
|
+
- VBv / KBv — the two mode-body templates; interpolated variables shared
|
|
34
|
+
with /code-review: c$e (Phase 0), m7t/u$e/d$e/p$e (the 4 cleanup angles).
|
|
35
|
+
- nJu() = CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20; lKs =
|
|
36
|
+
FORKED_AGENT_DEFAULT_MAX_TURNS = 50 — mirrored as the subagent tool's
|
|
37
|
+
defaults (PI_MAX_CONCURRENT_SUBAGENTS env still overrides the ceiling).
|
|
17
38
|
|
|
18
39
|
Bundled: ships inside the pi-review extension (skills/simplify/SKILL.md).
|
|
19
40
|
|
|
@@ -25,13 +46,13 @@ description: "Review the changed code for reuse, simplification, efficiency, and
|
|
|
25
46
|
════════════════════════════════════════════════════════════════════════
|
|
26
47
|
1. Fan-out tool — CC uses the Agent tool; Pi uses the `subagent` tool
|
|
27
48
|
(mode: parallel). Where CC says "the Agent tool", read `subagent`.
|
|
28
|
-
2. Mode guard — CC's
|
|
49
|
+
2. Mode guard — CC's Dii has two clauses: (a) spawn-depth — single-pass when
|
|
29
50
|
agent depth >= CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH (default 3); (b) the
|
|
30
51
|
Agent tool must be in the allowlist. On Pi: (a) is N/A — the `subagent`
|
|
31
52
|
tool spawns a fresh subprocess (always depth 0), so depth never accumulates
|
|
32
53
|
— so decideSimplifyMode substitutes a context-fraction heuristic
|
|
33
54
|
(tokens/contextWindow >= 0.8 → single-pass), a Pi addition NOT a mirror of
|
|
34
|
-
|
|
55
|
+
Dii; (b) is mirrored as "the `subagent` tool must be registered". The
|
|
35
56
|
decision is made DETERMINISTICALLY by the /code-simplify handler — it can
|
|
36
57
|
read ctx.getContextUsage(), which a pure-prompt skill cannot — and announced
|
|
37
58
|
in the trigger message; this skill just provides the two mode bodies.
|
|
@@ -68,36 +89,40 @@ review that target instead. Treat this diff as the review scope.
|
|
|
68
89
|
|
|
69
90
|
## Phase 1 — Review (4 cleanup agents in parallel)
|
|
70
91
|
|
|
71
|
-
Launch **4 independent review agents** via the
|
|
72
|
-
message so they run concurrently
|
|
73
|
-
|
|
74
|
-
one-line `summary`, and the concrete cost (what is duplicated, wasted, or
|
|
75
|
-
to maintain).
|
|
92
|
+
Launch **4 independent review agents** via the subagent tool, all in a
|
|
93
|
+
single message so they run concurrently. Pass each agent the diff and one of
|
|
94
|
+
the four angles below. Each returns its findings with `file`, `line`, a
|
|
95
|
+
one-line `summary`, and the concrete cost (what is duplicated, wasted, or
|
|
96
|
+
harder to maintain).
|
|
76
97
|
|
|
77
98
|
### Reuse
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
|
|
99
|
+
|
|
100
|
+
Flag new code that re-implements something the codebase
|
|
101
|
+
already has — Grep shared/utility modules and files adjacent to the change,
|
|
102
|
+
and name the existing helper to call instead.
|
|
81
103
|
|
|
82
104
|
### Simplification
|
|
105
|
+
|
|
83
106
|
Flag unnecessary complexity the diff adds: redundant or derivable state,
|
|
84
|
-
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
85
|
-
simpler form that does the same job.
|
|
107
|
+
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
108
|
+
the simpler form that does the same job.
|
|
86
109
|
|
|
87
110
|
### Efficiency
|
|
111
|
+
|
|
88
112
|
Flag wasted work the diff introduces: redundant computation or repeated I/O,
|
|
89
|
-
independent operations run sequentially, blocking work added to startup or
|
|
90
|
-
paths. Also flag long-lived objects built from closures or captured
|
|
91
|
-
— they keep the entire enclosing scope alive for the object's
|
|
92
|
-
leak when that scope holds large values); prefer a
|
|
93
|
-
the fields it needs. Name the cheaper
|
|
113
|
+
independent operations run sequentially, blocking work added to startup or
|
|
114
|
+
hot paths. Also flag long-lived objects built from closures or captured
|
|
115
|
+
environments — they keep the entire enclosing scope alive for the object's
|
|
116
|
+
lifetime (a memory leak when that scope holds large values); prefer a
|
|
117
|
+
class/struct that copies only the fields it needs. Name the cheaper
|
|
118
|
+
alternative.
|
|
94
119
|
|
|
95
120
|
### Altitude
|
|
96
|
-
Check that each change is implemented at the right depth, not as a fragile
|
|
97
|
-
bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
|
|
98
|
-
deep enough — prefer generalizing the underlying mechanism over adding special
|
|
99
|
-
cases.
|
|
100
121
|
|
|
122
|
+
Check that each change is implemented at the right depth, not as a fragile
|
|
123
|
+
bandaid. Special cases layered on shared infrastructure are a sign the fix
|
|
124
|
+
isn't deep enough — prefer generalizing the underlying mechanism over adding
|
|
125
|
+
special cases.
|
|
101
126
|
## Phase 2 — Apply, verify, and report
|
|
102
127
|
|
|
103
128
|
Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). The parallel fan-out only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: true` in the report since the 4-agent fan-out actually ran.
|
|
@@ -108,40 +133,44 @@ Follow the shared **Phase 2** procedure at the end of this skill (snapshot → a
|
|
|
108
133
|
|
|
109
134
|
`/code-simplify → subagent tool unavailable → single-pass inline cleanup → apply the fixes`
|
|
110
135
|
|
|
111
|
-
The
|
|
112
|
-
|
|
113
|
-
|
|
136
|
+
The subagent tool isn't available in this context, so the usual
|
|
137
|
+
4-agent fan-out can't run. Work through all four angles below yourself, in
|
|
138
|
+
this same context, in one pass — do not skip an angle for lack of fan-out.
|
|
114
139
|
|
|
115
140
|
## Phase 1 — Review (4 cleanup angles, single pass)
|
|
116
141
|
|
|
117
142
|
Review the diff against each angle below in turn. For each, note findings with
|
|
118
|
-
`file`, `line`, a one-line `summary`, and the concrete cost (what is
|
|
119
|
-
wasted, or harder to maintain).
|
|
143
|
+
`file`, `line`, a one-line `summary`, and the concrete cost (what is
|
|
144
|
+
duplicated, wasted, or harder to maintain).
|
|
120
145
|
|
|
121
146
|
### Reuse
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
147
|
+
|
|
148
|
+
Flag new code that re-implements something the codebase
|
|
149
|
+
already has — Grep shared/utility modules and files adjacent to the change,
|
|
150
|
+
and name the existing helper to call instead.
|
|
125
151
|
|
|
126
152
|
### Simplification
|
|
153
|
+
|
|
127
154
|
Flag unnecessary complexity the diff adds: redundant or derivable state,
|
|
128
|
-
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
129
|
-
simpler form that does the same job.
|
|
155
|
+
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
156
|
+
the simpler form that does the same job.
|
|
130
157
|
|
|
131
158
|
### Efficiency
|
|
159
|
+
|
|
132
160
|
Flag wasted work the diff introduces: redundant computation or repeated I/O,
|
|
133
|
-
independent operations run sequentially, blocking work added to startup or
|
|
134
|
-
paths. Also flag long-lived objects built from closures or captured
|
|
135
|
-
— they keep the entire enclosing scope alive for the object's
|
|
136
|
-
leak when that scope holds large values); prefer a
|
|
137
|
-
the fields it needs. Name the cheaper
|
|
161
|
+
independent operations run sequentially, blocking work added to startup or
|
|
162
|
+
hot paths. Also flag long-lived objects built from closures or captured
|
|
163
|
+
environments — they keep the entire enclosing scope alive for the object's
|
|
164
|
+
lifetime (a memory leak when that scope holds large values); prefer a
|
|
165
|
+
class/struct that copies only the fields it needs. Name the cheaper
|
|
166
|
+
alternative.
|
|
138
167
|
|
|
139
168
|
### Altitude
|
|
140
|
-
Check that each change is implemented at the right depth, not as a fragile
|
|
141
|
-
bandaid. Special cases layered on shared infrastructure are a sign the fix isn't
|
|
142
|
-
deep enough — prefer generalizing the underlying mechanism over adding special
|
|
143
|
-
cases.
|
|
144
169
|
|
|
170
|
+
Check that each change is implemented at the right depth, not as a fragile
|
|
171
|
+
bandaid. Special cases layered on shared infrastructure are a sign the fix
|
|
172
|
+
isn't deep enough — prefer generalizing the underlying mechanism over adding
|
|
173
|
+
special cases.
|
|
145
174
|
## Phase 2 — Apply, verify, and report
|
|
146
175
|
|
|
147
176
|
Follow the shared **Phase 2** procedure at the end of this skill (snapshot → apply → verify → auto-revert on failure → report via `review_report`). Single-pass vs parallel only changes how findings are gathered (Phase 1); applying, verifying, and reporting are identical across modes. Set `fanned_out: false` in the report so a reader is not misled into thinking the 4-agent fan-out ran.
|
|
@@ -181,7 +210,7 @@ that would discard the user's intended changes too.
|
|
|
181
210
|
Apply each surviving finding directly. Skip any finding whose fix would change
|
|
182
211
|
intended behavior, require changes well outside the reviewed diff, or that you
|
|
183
212
|
judge to be a false positive — note the skip (it will be reported as
|
|
184
|
-
`
|
|
213
|
+
`skipped`).
|
|
185
214
|
|
|
186
215
|
## Step 3 — Verify, branching on the result
|
|
187
216
|
|
|
@@ -189,11 +218,10 @@ Run the verification command the handler injected in the trigger message (e.g.
|
|
|
189
218
|
`npm run check`), then branch:
|
|
190
219
|
|
|
191
220
|
- **No verification command was detected** → keep the applied changes, mark each
|
|
192
|
-
applied finding `
|
|
221
|
+
applied finding `fixed`, and say in the report that NO verification
|
|
193
222
|
was run. Verification is opportunistic — never block on its absence.
|
|
194
223
|
- **Verification passes** → keep the changes; applied findings are
|
|
195
|
-
`
|
|
196
|
-
partly addressed the issue).
|
|
224
|
+
`fixed`.
|
|
197
225
|
- **Verification fails** → the working tree is verified-broken; go to Step 3a.
|
|
198
226
|
|
|
199
227
|
### Step 3a — Auto-revert (hybrid granularity, only on failure)
|
|
@@ -209,7 +237,7 @@ otherwise survive the rollback).
|
|
|
209
237
|
after each file. Keep only files whose verification passes; revert any file
|
|
210
238
|
whose verification fails back to its baseline.
|
|
211
239
|
4. If NO file passes on its own, leave everything reverted and mark every
|
|
212
|
-
finding `
|
|
240
|
+
finding `skipped` — a clean tree is the safe outcome, not a broken one.
|
|
213
241
|
|
|
214
242
|
This caps the cost: the common case (clean apply) runs verification exactly
|
|
215
243
|
once; only a failure escalates to one verification per touched file.
|
|
@@ -217,17 +245,22 @@ once; only a failure escalates to one verification per touched file.
|
|
|
217
245
|
## Step 4 — Report via `review_report`
|
|
218
246
|
|
|
219
247
|
Call the `review_report` tool **once** with `level: "simplify"` and one finding
|
|
220
|
-
entry per cleanup, ranked most-severe first.
|
|
221
|
-
|
|
222
|
-
|
|
248
|
+
entry per cleanup, ranked most-severe first. Generate a `report_id` (e.g.
|
|
249
|
+
`review-<ts>`) on this first call; reuse it on any re-report (fixed-later
|
|
250
|
+
义务:本会话后续若再修复已上报项,必须先再次调用 `review_report` 更新
|
|
251
|
+
`outcome`,先于任何文字总结)。Each entry carries `file`, `line` (optional),
|
|
252
|
+
`category` (`reuse` / `simplification` / `efficiency` / `altitude`),
|
|
253
|
+
`short_summary` (≤60 字符纯声明标签,去掉理由与后果——汇总表概述列优先用它),
|
|
254
|
+
`summary` (one line, Chinese,含理由与后果), `failure_scenario` (the concrete
|
|
223
255
|
cost — Chinese), and `outcome`:
|
|
224
256
|
|
|
225
|
-
- `
|
|
226
|
-
|
|
227
|
-
- `
|
|
228
|
-
the
|
|
229
|
-
|
|
230
|
-
|
|
257
|
+
- `fixed` — applied and verification passed (or no verification command existed
|
|
258
|
+
and the change was kept).
|
|
259
|
+
- `skipped` — real but not applied: judged a false positive / behavior-
|
|
260
|
+
changing, or reverted by the auto-revert in Step 3a. Partial applies (per-file
|
|
261
|
+
rollback kept only some files) also count as `skipped` — the per-file detail
|
|
262
|
+
goes into `summary`.
|
|
263
|
+
- `no_change_needed` — not applicable or already handled.
|
|
231
264
|
|
|
232
265
|
Do **not** write a free-text summary as the primary record — the structured
|
|
233
266
|
`review_report` call IS the summary (it renders the report AND writes JSON to
|
|
@@ -41,12 +41,13 @@ function readScriptsAt(cwd: string): Record<string, string> | null {
|
|
|
41
41
|
* Decide simplify mode deterministically from real context usage + tool availability.
|
|
42
42
|
* Pure function — unit-testable.
|
|
43
43
|
*
|
|
44
|
-
* CC parity note: CC's /simplify guard (
|
|
45
|
-
* context check — `
|
|
46
|
-
*
|
|
44
|
+
* CC parity note: CC's /simplify guard (Dii, verified in the 2.1.227 binary) is
|
|
45
|
+
* a SPAWN-DEPTH recursion limit, NOT a context check — `ok(ctx.agentContext) >= wV()`
|
|
46
|
+
* where ok() returns the agent's depth (main=0) and wV() returns
|
|
47
|
+
* CLAUDE_CODE_MAX_SUBAGENT_SPAWN_DEPTH (default 3). That is N/A on Pi: the
|
|
47
48
|
* `subagent` tool spawns a fresh subprocess (depth 0), so depth never accumulates.
|
|
48
49
|
* The context-fraction heuristic below is a Pi-specific substitute (don't fan out
|
|
49
|
-
* when the parent's context is near-full), NOT a mirror of
|
|
50
|
+
* when the parent's context is near-full), NOT a mirror of Dii. The other Dii clause
|
|
50
51
|
* — the Agent tool must be in the allowlist — IS mirrored here as `hasSubagent`.
|
|
51
52
|
*/
|
|
52
53
|
export function decideSimplifyMode(opts: {
|
|
@@ -9,29 +9,32 @@
|
|
|
9
9
|
* a machine-readable JSON (findings + level + outcome) to
|
|
10
10
|
* `<cwd>/.pi/review/<id>.json` so CI / --fix / --comment can consume it.
|
|
11
11
|
*
|
|
12
|
-
* `verdict` (CONFIRMED/PLAUSIBLE
|
|
13
|
-
* the CC ReportFindings shape
|
|
14
|
-
*
|
|
15
|
-
*
|
|
12
|
+
* `verdict` (CONFIRMED/PLAUSIBLE) and `outcome` (fixed/skipped/no_change_needed)
|
|
13
|
+
* enums follow the CC ReportFindings shape — values verified against the CC
|
|
14
|
+
* v2.1.227 binary (consistent across 2.1.223/226/227). REFUTED is deliberately
|
|
15
|
+
* absent: the verify flow drops it before reporting. The tool entry also
|
|
16
|
+
* normalizes stray invalid values (drop the finding / coerce to skipped) rather
|
|
17
|
+
* than failing the whole call.
|
|
16
18
|
*/
|
|
17
19
|
import { defineTool, getMarkdownTheme } from "@earendil-works/pi-coding-agent";
|
|
18
20
|
import { Markdown } from "@earendil-works/pi-tui";
|
|
19
|
-
import { Type } from "typebox";
|
|
21
|
+
import { type Static, Type } from "typebox";
|
|
20
22
|
import * as fs from "node:fs";
|
|
21
23
|
import * as path from "node:path";
|
|
22
24
|
|
|
23
25
|
// --- enums following the CC ReportFindings shape ----------------------------
|
|
26
|
+
// 值域实证来源:CC v2.1.227 bin/claude.exe(ReportFindings 工具 schema)。
|
|
27
|
+
// verdict 两值与 outcome 三档在 2.1.223/226/227 三版本中一致。
|
|
24
28
|
|
|
25
|
-
const
|
|
29
|
+
const VERDICT_VALUES = ["CONFIRMED", "PLAUSIBLE"] as const;
|
|
30
|
+
const Verdict = Type.Union(VERDICT_VALUES.map((v) => Type.Literal(v)));
|
|
26
31
|
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
Type.Literal("unclear_from_transcript"),
|
|
34
|
-
]);
|
|
32
|
+
const OUTCOME_VALUES = ["fixed", "skipped", "no_change_needed"] as const;
|
|
33
|
+
/** CC ReportFindings `outcome` 三档(2.1.227 二进制实证)。fixed-later 再上报时更新。 */
|
|
34
|
+
const Outcome = Type.Union(OUTCOME_VALUES.map((v) => Type.Literal(v)));
|
|
35
|
+
|
|
36
|
+
// 供 SKILL-schema 同步测试引用(防漂移:SKILL 流程契约不得与常量脱节)。
|
|
37
|
+
export { OUTCOME_VALUES, VERDICT_VALUES };
|
|
35
38
|
|
|
36
39
|
const Level = Type.Union([
|
|
37
40
|
Type.Literal("low"),
|
|
@@ -54,6 +57,12 @@ const FindingParams = Type.Object({
|
|
|
54
57
|
"产生该发现的角度 slug:correctness / reuse / simplification / efficiency / altitude / conventions(或更具体如 test-coverage)。",
|
|
55
58
|
}),
|
|
56
59
|
verdict: Type.Optional(Verdict),
|
|
60
|
+
short_summary: Type.Optional(
|
|
61
|
+
Type.String({
|
|
62
|
+
description:
|
|
63
|
+
"≤60 字符的纯声明标签(去掉理由与后果)。汇总表概述列优先使用它;详情块仍显示完整 summary。流程层面必填(CC 输出模板契约),schema 层面 optional(与 CC tool schema 一致)。中文。",
|
|
64
|
+
}),
|
|
65
|
+
),
|
|
57
66
|
summary: Type.String({ description: "一句话说明(≤80字),同时作紧凑标签。中文。" }),
|
|
58
67
|
failure_scenario: Type.String({
|
|
59
68
|
description:
|
|
@@ -72,6 +81,12 @@ const ReviewReportParams = Type.Object({
|
|
|
72
81
|
findings: Type.Array(FindingParams, {
|
|
73
82
|
description: "已验证、去重、按严重度从高到低排序的发现列表(most-severe first)。空数组表示无发现存活。",
|
|
74
83
|
}),
|
|
84
|
+
report_id: Type.Optional(
|
|
85
|
+
Type.String({
|
|
86
|
+
description:
|
|
87
|
+
"报告标识(如 review-<ts>)。首次上报生成;fixed-later 再上报传同一 id,消费方按 id 归并,同 id 最新 generatedAt 为最终状态。",
|
|
88
|
+
}),
|
|
89
|
+
),
|
|
75
90
|
});
|
|
76
91
|
|
|
77
92
|
interface ReviewReportDetails {
|
|
@@ -79,6 +94,56 @@ interface ReviewReportDetails {
|
|
|
79
94
|
findingsCount: number;
|
|
80
95
|
/** 结构化 JSON 落盘路径;落盘失败时为 null(仍返回渲染报告)。 */
|
|
81
96
|
outFile: string | null;
|
|
97
|
+
reportId: string | null;
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* execute 入口的宽松 finding 形态——绕过 schema 校验的直接调用(如单测)
|
|
102
|
+
* 可能携带旧五档 outcome 或已废弃的 REFUTED verdict。
|
|
103
|
+
*/
|
|
104
|
+
type LooseFinding = {
|
|
105
|
+
file: string;
|
|
106
|
+
line?: number;
|
|
107
|
+
category: string;
|
|
108
|
+
verdict?: string;
|
|
109
|
+
short_summary?: string;
|
|
110
|
+
summary: string;
|
|
111
|
+
failure_scenario: string;
|
|
112
|
+
outcome?: string;
|
|
113
|
+
};
|
|
114
|
+
|
|
115
|
+
/**
|
|
116
|
+
* 单条清洗:非法 verdict(含已废弃的 REFUTED)返回 null(剔除);非法 outcome
|
|
117
|
+
* 归一化为 skipped 并附注原始值。schema 保持严格,清洗在 schema 校验之前。
|
|
118
|
+
*/
|
|
119
|
+
function sanitizeFinding(f: LooseFinding): { f: LooseFinding; note?: string } | null {
|
|
120
|
+
if (f.verdict !== undefined && !(VERDICT_VALUES as readonly string[]).includes(f.verdict)) return null;
|
|
121
|
+
let outcome = f.outcome;
|
|
122
|
+
let note: string | undefined;
|
|
123
|
+
if (outcome !== undefined && !(OUTCOME_VALUES as readonly string[]).includes(outcome)) {
|
|
124
|
+
note = `(outcome "${outcome}" 非法,已归一化为 skipped)`;
|
|
125
|
+
outcome = "skipped";
|
|
126
|
+
}
|
|
127
|
+
return { f: { ...f, outcome }, note };
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* 批量清洗,note 按清洗后数组索引记录(渲染层附注用)。
|
|
132
|
+
* `prepareArguments`(主防御,schema 校验前)与 `execute` 入口(双保险,
|
|
133
|
+
* 防绕过 prepareArguments 的直接调用)共用——模型路径的非法值在进入
|
|
134
|
+
* execute 前已被清洗,validateToolArguments 不会因边缘值 throw 掉整份报告。
|
|
135
|
+
*/
|
|
136
|
+
function normalizeFindings(findings: LooseFinding[]): { findings: LooseFinding[]; notes: Map<number, string> } {
|
|
137
|
+
const out: LooseFinding[] = [];
|
|
138
|
+
const notes = new Map<number, string>();
|
|
139
|
+
for (const raw of findings) {
|
|
140
|
+
const s = sanitizeFinding(raw);
|
|
141
|
+
if (!s) continue;
|
|
142
|
+
const idx = out.length;
|
|
143
|
+
out.push(s.f);
|
|
144
|
+
if (s.note) notes.set(idx, s.note);
|
|
145
|
+
}
|
|
146
|
+
return { findings: out, notes };
|
|
82
147
|
}
|
|
83
148
|
|
|
84
149
|
// --- render -----------------------------------------------------------------
|
|
@@ -88,15 +153,19 @@ interface FindingInput {
|
|
|
88
153
|
line?: number;
|
|
89
154
|
category: string;
|
|
90
155
|
verdict?: string;
|
|
156
|
+
short_summary?: string;
|
|
91
157
|
summary: string;
|
|
92
158
|
failure_scenario: string;
|
|
93
159
|
outcome?: string;
|
|
160
|
+
/** normalize 附注(如非法 outcome 归一化说明),仅渲染进详情块。 */
|
|
161
|
+
note?: string;
|
|
94
162
|
}
|
|
95
163
|
interface ReportInput {
|
|
96
164
|
level: string;
|
|
97
165
|
target?: string;
|
|
98
166
|
files_changed?: number;
|
|
99
167
|
fanned_out?: boolean;
|
|
168
|
+
reportId?: string;
|
|
100
169
|
findings: FindingInput[];
|
|
101
170
|
}
|
|
102
171
|
|
|
@@ -118,7 +187,8 @@ function renderReport(p: ReportInput): string {
|
|
|
118
187
|
const fanLabel = p.fanned_out === true ? "多智能体" : p.fanned_out === false ? "单遍自审" : "未标注";
|
|
119
188
|
const targetStr = (p.target ?? "(whole diff)").replace(/`/g, "\\`"); // backtick inside the inline-code cell would close it early
|
|
120
189
|
const filesStr = p.files_changed != null ? `${p.files_changed} 个文件` : "文件数未标注";
|
|
121
|
-
|
|
190
|
+
const idStr = p.reportId ? ` · 报告 \`${escapeCell(p.reportId)}\`` : "";
|
|
191
|
+
lines.push(`\`${p.level}\` · \`${targetStr}\` · ${filesStr} · ${p.findings.length} 条发现 · ${fanLabel}${idStr}`);
|
|
122
192
|
lines.push("");
|
|
123
193
|
|
|
124
194
|
if (p.findings.length === 0) {
|
|
@@ -130,7 +200,7 @@ function renderReport(p: ReportInput): string {
|
|
|
130
200
|
lines.push("|---|------|------|------|------|");
|
|
131
201
|
for (let i = 0; i < p.findings.length; i++) {
|
|
132
202
|
const f = p.findings[i]!;
|
|
133
|
-
lines.push(`| ${i + 1} | ${escapeCell(f.verdict ?? "")} | ${escapeCell(f.category)} | ${escapeCell(fmtLoc(f))} | ${escapeCell(f.summary)} |`);
|
|
203
|
+
lines.push(`| ${i + 1} | ${escapeCell(f.verdict ?? "")} | ${escapeCell(f.category)} | ${escapeCell(fmtLoc(f))} | ${escapeCell(f.short_summary ?? f.summary)} |`);
|
|
134
204
|
}
|
|
135
205
|
lines.push("");
|
|
136
206
|
lines.push("**详情**");
|
|
@@ -138,9 +208,10 @@ function renderReport(p: ReportInput): string {
|
|
|
138
208
|
p.findings.forEach((f, i) => {
|
|
139
209
|
const v = f.verdict ? ` *(${f.verdict})*` : "";
|
|
140
210
|
const out = f.outcome ? `\n修复结果:\`${f.outcome}\`` : "";
|
|
211
|
+
const note = f.note ? `\n${f.note}` : "";
|
|
141
212
|
lines.push(`**${i + 1}. ${fmtLoc(f)} — ${f.category}**${v}`);
|
|
142
213
|
lines.push(`概述:${f.summary}`);
|
|
143
|
-
lines.push(`场景:${f.failure_scenario}${out}`);
|
|
214
|
+
lines.push(`场景:${f.failure_scenario}${out}${note}`);
|
|
144
215
|
lines.push("");
|
|
145
216
|
});
|
|
146
217
|
return lines.join("\n").trimEnd();
|
|
@@ -156,13 +227,41 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
|
|
|
156
227
|
promptSnippet: "review_report — report structured code-review findings (renders Markdown + writes JSON for CI)",
|
|
157
228
|
promptGuidelines: [
|
|
158
229
|
"After verify + dedup, call `review_report` once with { level, findings } (most-severe first; empty array if none survived). Do not also hand-write the Markdown table — this tool renders it.",
|
|
159
|
-
"On re-report after --fix, set each finding's `outcome` (
|
|
230
|
+
"On re-report after --fix, set each finding's `outcome` (fixed / skipped / no_change_needed).",
|
|
160
231
|
"Use this tool only when the code-review skill instructs reporting findings; otherwise follow the active output format.",
|
|
161
232
|
],
|
|
162
233
|
parameters: ReviewReportParams,
|
|
163
234
|
|
|
235
|
+
// 主防御:schema 校验之前清洗非法值(模型路径下校验失败即 throw、工具不执行,
|
|
236
|
+
// 因此 execute 内的防御对模型不可达)。返回符合 schema 的对象——非法 verdict
|
|
237
|
+
// 的 finding 剔除、非法 outcome 归一化为 skipped;note 附注由 execute 层基于
|
|
238
|
+
// 清洗结果生成(schema 对象不含 note 字段)。
|
|
239
|
+
prepareArguments(args) {
|
|
240
|
+
if (typeof args !== "object" || args === null) return args as Static<typeof ReviewReportParams>;
|
|
241
|
+
const raw = args as { findings?: unknown };
|
|
242
|
+
if (!Array.isArray(raw.findings)) return args as Static<typeof ReviewReportParams>;
|
|
243
|
+
const { findings } = normalizeFindings(raw.findings as unknown as LooseFinding[]);
|
|
244
|
+
return { ...raw, findings } as Static<typeof ReviewReportParams>;
|
|
245
|
+
},
|
|
246
|
+
|
|
164
247
|
async execute(toolCallId, params, _signal, _onUpdate, ctx) {
|
|
165
|
-
|
|
248
|
+
// normalize(双保险)——绕过 prepareArguments 的直接调用(如单测)可能携带
|
|
249
|
+
// 旧五档 outcome 或已废弃的 REFUTED verdict。归一化而非整单拒绝:非法
|
|
250
|
+
// verdict 的 finding 剔除,非法 outcome 归一化为 skipped 并附注;渲染与落盘
|
|
251
|
+
// 永远只含合法值(spec: 消费方永远看到合法值)。
|
|
252
|
+
const { findings: cleaned, notes } = normalizeFindings(
|
|
253
|
+
(params.findings ?? []) as unknown as LooseFinding[],
|
|
254
|
+
);
|
|
255
|
+
const findings: FindingInput[] = cleaned.map((f, i) => ({ ...f, note: notes.get(i) }));
|
|
256
|
+
|
|
257
|
+
const report = renderReport({
|
|
258
|
+
level: params.level,
|
|
259
|
+
target: params.target,
|
|
260
|
+
files_changed: params.files_changed,
|
|
261
|
+
fanned_out: params.fanned_out,
|
|
262
|
+
reportId: params.report_id,
|
|
263
|
+
findings,
|
|
264
|
+
});
|
|
166
265
|
|
|
167
266
|
let outFile: string | null = null;
|
|
168
267
|
let writeError: string | null = null;
|
|
@@ -178,11 +277,12 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
|
|
|
178
277
|
JSON.stringify(
|
|
179
278
|
{
|
|
180
279
|
level: params.level,
|
|
280
|
+
reportId: params.report_id ?? null,
|
|
181
281
|
target: params.target ?? null,
|
|
182
282
|
filesChanged: params.files_changed ?? null,
|
|
183
283
|
fannedOut: params.fanned_out ?? null,
|
|
184
284
|
generatedAt: now.toISOString(),
|
|
185
|
-
findings
|
|
285
|
+
findings,
|
|
186
286
|
},
|
|
187
287
|
null,
|
|
188
288
|
2,
|
|
@@ -200,8 +300,9 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
|
|
|
200
300
|
: `\n\n[结构化落盘失败(${writeError ?? "未知原因"}),仅渲染报告]`;
|
|
201
301
|
const details: ReviewReportDetails = {
|
|
202
302
|
level: params.level,
|
|
203
|
-
findingsCount:
|
|
303
|
+
findingsCount: findings.length,
|
|
204
304
|
outFile,
|
|
305
|
+
reportId: params.report_id ?? null,
|
|
205
306
|
};
|
|
206
307
|
return {
|
|
207
308
|
content: [{ type: "text" as const, text: report + tail }],
|
|
@@ -217,8 +318,8 @@ export const reviewReportTool = defineTool<typeof ReviewReportParams, ReviewRepo
|
|
|
217
318
|
// throw, so a failure here degrades to the pre-change behavior rather than erroring.
|
|
218
319
|
renderResult(result, _options, _theme, _context) {
|
|
219
320
|
const text = result.content
|
|
220
|
-
.filter((c) => c.type === "text")
|
|
221
|
-
.map((c) =>
|
|
321
|
+
.filter((c): c is Extract<(typeof result.content)[number], { type: "text" }> => c.type === "text")
|
|
322
|
+
.map((c) => c.text)
|
|
222
323
|
.join("\n");
|
|
223
324
|
return new Markdown(text, 0, 0, getMarkdownTheme());
|
|
224
325
|
},
|
package/src/tools/subagent.ts
CHANGED
|
@@ -28,8 +28,9 @@ import {
|
|
|
28
28
|
type AgentSpawnResult,
|
|
29
29
|
} from "@fyeeme/pi-subagent-core";
|
|
30
30
|
|
|
31
|
-
/** Default concurrency ceiling when PI_MAX_CONCURRENT_SUBAGENTS is unset/invalid.
|
|
32
|
-
|
|
31
|
+
/** Default concurrency ceiling when PI_MAX_CONCURRENT_SUBAGENTS is unset/invalid.
|
|
32
|
+
* 20 = CC's CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20 (2.1.227 binary empirical). */
|
|
33
|
+
const DEFAULT_MAX_CONCURRENCY = 20;
|
|
33
34
|
|
|
34
35
|
/**
|
|
35
36
|
* Effective concurrency ceiling, configurable via PI_MAX_CONCURRENT_SUBAGENTS
|
|
@@ -46,8 +47,9 @@ function getMaxConcurrency(): number {
|
|
|
46
47
|
* runaway agent cannot otherwise be bounded. Generous (well above the ~10–15
|
|
47
48
|
* turns the 4-angle finder/verify agents need) so legitimate work is not
|
|
48
49
|
* truncated; callers may override with a smaller or larger explicit value.
|
|
50
|
+
* 50 = CC's FORKED_AGENT_DEFAULT_MAX_TURNS (2.1.227 binary empirical).
|
|
49
51
|
*/
|
|
50
|
-
const DEFAULT_FANOUT_MAX_TURNS =
|
|
52
|
+
const DEFAULT_FANOUT_MAX_TURNS = 50;
|
|
51
53
|
|
|
52
54
|
// Module-level registry so abortAgent can reach in-flight calls. callIds are
|
|
53
55
|
// unique per tool call (toolCallId#index), so a single registry is safe.
|
|
@@ -61,6 +63,12 @@ const SubagentParams = Type.Object({
|
|
|
61
63
|
description: "Prompt(s) for the sub-agent(s). single/chain use order; parallel runs all.",
|
|
62
64
|
}),
|
|
63
65
|
model: Type.Optional(Type.String({ description: "Full model id (e.g. claude-sonnet-5). Omit for the session default." })),
|
|
66
|
+
thinking: Type.Optional(
|
|
67
|
+
Type.Union(
|
|
68
|
+
["off", "minimal", "low", "medium", "high", "xhigh", "max"].map((l) => Type.Literal(l)),
|
|
69
|
+
{ description: "Thinking level for the sub-agent(s), passed as --thinking. Omit for the model default." },
|
|
70
|
+
),
|
|
71
|
+
),
|
|
64
72
|
systemPrompt: Type.Optional(Type.String({ description: "Appended to the sub-agent's system prompt." })),
|
|
65
73
|
tools: Type.Optional(Type.Array(Type.String(), { description: "Tool whitelist for the sub-agent. Omit for default tools." })),
|
|
66
74
|
parallelism: Type.Optional(
|
|
@@ -251,6 +259,7 @@ export const subagentTool = defineTool<typeof SubagentParams, SubagentDetails>({
|
|
|
251
259
|
const baseOpts: Omit<AgentSpawnOptions, "callId" | "task"> = {
|
|
252
260
|
cwd,
|
|
253
261
|
model: params.model,
|
|
262
|
+
thinking: params.thinking,
|
|
254
263
|
tools: params.tools,
|
|
255
264
|
signal,
|
|
256
265
|
// Default turn budget applies when omitted; an explicit 0 is honored by
|