@fyeeme/pi-review 2.0.0 → 2.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +77 -132
- package/agents/cleaner-altitude.md +18 -0
- package/agents/cleaner-efficiency.md +19 -0
- package/agents/cleaner-reuse.md +16 -0
- package/agents/cleaner-simplification.md +16 -0
- package/agents/finder-conventions.md +23 -0
- package/agents/finder-cross-file.md +21 -0
- package/agents/finder-diff-scan.md +23 -0
- package/agents/finder-language-pitfall.md +21 -0
- package/agents/finder-removed-behavior.md +21 -0
- package/agents/finder-wrapper-proxy.md +23 -0
- package/agents/gap-hunter.md +24 -0
- package/agents/verifier.md +33 -0
- package/index.ts +34 -29
- package/package.json +17 -15
- package/prompts/review.md +23 -0
- package/prompts/simplify.parallel.md +45 -0
- package/prompts/simplify.single.md +23 -0
- package/skills/{code-review → review}/SKILL.md +134 -33
- package/skills/simplify/SKILL.md +98 -41
- package/src/config.ts +103 -0
- package/src/diff.ts +306 -0
- package/src/dispatch.ts +238 -0
- package/src/strategy.ts +76 -0
- package/src/tools/review_report.ts +17 -15
- package/src/commands/code-review.ts +0 -100
- package/src/commands/code-simplify.ts +0 -100
- package/src/tools/subagent.ts +0 -367
package/README.md
CHANGED
|
@@ -1,157 +1,102 @@
|
|
|
1
|
-
# pi-review
|
|
1
|
+
# @fyeeme/pi-review (v2)
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
**2.0.0 — major release** (from 1.1.1), part of the 2.0 extensions family wave:
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
`subagent` tool that spawns parallel pi subprocesses — providing the **real
|
|
7
|
-
fan-out capability** that the `code-review` and `simplify` skills
|
|
8
|
-
(bundled in this package under `skills/`) need for their multi-agent flows.
|
|
5
|
+
- **Sandwich architecture** — skills carry the methodology, `prompts/` carries the orchestration strategy as data (parallel-when guards in frontmatter, phases in the body), `agents/` carries the 12 review roles (finder A–E, cleaner-reuse/simplification/efficiency, altitude, conventions, verifier, gap-hunter), and a thin plugin entry composes the stack.
|
|
9
6
|
|
|
10
|
-
|
|
7
|
+
- **Batteries-included fan-out** — the entry composes [`@fyeeme/pi-subagents`](https://www.npmjs.com/package/@fyeeme/pi-subagents) 2.1.1 from its npm dependency: the `subagent` tool, real `pi` subprocess spawning, and the live agent UI (widget / FleetView / `/agents`) work out of the box.
|
|
11
8
|
|
|
12
|
-
|
|
13
|
-
cleanup angles), but `pi-subagents` does not exist in pi — so the agent
|
|
14
|
-
silently degraded to a sequential self-sweep. This extension ships the actual
|
|
15
|
-
fan-out primitive: an LLM-callable `subagent` tool that spawns real
|
|
16
|
-
`pi --mode json` subprocesses.
|
|
9
|
+
- **`review_report` structured findings sink** — Chinese Markdown rendered back to the conversation plus machine-readable JSON under `<cwd>/.pi/review/` for CI, `--fix` re-reports, and `--comment`.
|
|
17
10
|
|
|
18
|
-
|
|
19
|
-
dispatch (how many agents, parallelism, abort), the **skill** provides the
|
|
20
|
-
review/cleanup semantics. CC's own `/code-review` and `/code-simplify` work the same
|
|
21
|
-
way — one general Agent tool, prompt decides how to use it.
|
|
11
|
+
- **Effort levels with CC-parity semantics** — `/review [low|medium|high|xhigh|max]`: quad tuples `{correctnessAngles, perAngle, maxFindings, sweep}`, grouped-by-location independent verification, and the xhigh/max gap-hunt.
|
|
22
12
|
|
|
23
|
-
|
|
13
|
+
- **`/simplify` dual-mode** — the dispatcher measures context usage and diff size against the declared strategy, then renders either the PARALLEL template (4 cleaner agents via `subagent`) or the SINGLE-PASS one.
|
|
24
14
|
|
|
25
|
-
|
|
26
|
-
From the package dir:
|
|
15
|
+
- Breaking: commands renamed `/code-review` → `/review`, `/code-simplify` → `/simplify`.
|
|
27
16
|
|
|
28
|
-
|
|
29
|
-
|
|
17
|
+
Review & cleanup assets for [pi](https://github.com/earendil-works/pi-mono), in the sandwich shape (skills + prompts + agents on top of a thin plugin entry):
|
|
18
|
+
|
|
19
|
+
```
|
|
20
|
+
skills/ methodology (review, simplify) — registered natively via the `pi` manifest
|
|
21
|
+
prompts/ orchestration strategy as data — parallel-when guards in frontmatter,
|
|
22
|
+
CC-parity phase structure in the body; rendered by the generic dispatcher
|
|
23
|
+
agents/ the review roles as subagent definitions (finder-*, cleaner-*, verifier,
|
|
24
|
+
gap-hunter) invoked via the `subagent` tool of @fyeeme/pi-subagents
|
|
25
|
+
index.ts plugin entry: the review_report structured findings sink + the
|
|
26
|
+
/review and /simplify dispatcher commands
|
|
27
|
+
src/ dispatch.ts (variable gathering, guard evaluation, rendering),
|
|
28
|
+
diff.ts (deterministic diff ladder — unchanged v1 semantics),
|
|
29
|
+
strategy.ts (guard evaluator), tools/review_report.ts
|
|
30
30
|
```
|
|
31
31
|
|
|
32
|
-
|
|
33
|
-
|
|
32
|
+
**Batteries included**: installing this package is enough. Its extension
|
|
33
|
+
factory composes [`@fyeeme/pi-subagents`](../pi-subagents) (the `subagent`
|
|
34
|
+
tool, spawning, and the live agent UI — widget / FleetView / `/agents`) from
|
|
35
|
+
the version-pinned dependency copy, so every fan-out the skills orchestrate
|
|
36
|
+
works out of the box. A standalone pi-subagents install is optional
|
|
37
|
+
(general-purpose scout/planner/reviewer/worker agents) and coexists —
|
|
38
|
+
composition is idempotent.
|
|
39
|
+
|
|
40
|
+
## Commands
|
|
34
41
|
|
|
35
|
-
|
|
36
|
-
|
|
42
|
+
- `/review [low|medium|high|xhigh|max] [--fix] [--comment] [--share] [<pr#>|<branch>|<path>]` — effort-level code review via the review skill. Effort is sticky: an explicit level is remembered; the next bare `/review` reuses it.
|
|
43
|
+
- `/simplify [<target>]` — cleanup of the changed code (reuse/simplification/efficiency/altitude). The dispatcher resolves the diff (upstream merge-base → HEAD worktree → staged → unstaged; submodule-aware), evaluates the strategy declared in `prompts/simplify.parallel.md` frontmatter (context usage < 80%, diff < 400k chars, fan-out available), and renders either the PARALLEL template (Phase 0 visible diff read → `subagent` parallel dispatch of the 4 cleaner agents with `maxTurns: 15` → Phase 2 apply/verify/report) or the SINGLE-PASS template (angles worked inline).
|
|
37
44
|
|
|
38
|
-
|
|
45
|
+
Reports land via the `review_report` tool: Chinese Markdown back to the conversation plus machine-readable JSON under `<cwd>/.pi/review/`.
|
|
39
46
|
|
|
40
|
-
|
|
47
|
+
## Strategy is data
|
|
41
48
|
|
|
42
49
|
```
|
|
43
|
-
/
|
|
50
|
+
# prompts/simplify.parallel.md
|
|
51
|
+
---
|
|
52
|
+
parallel-when:
|
|
53
|
+
context-below: 0.8
|
|
54
|
+
diff-chars-below: 400000
|
|
55
|
+
---
|
|
44
56
|
```
|
|
45
57
|
|
|
46
|
-
|
|
47
|
-
and follow it — using the `subagent` tool for any fan-out / verify / gap-hunt.
|
|
58
|
+
Edit the file, the strategy changes. The dispatcher only executes what the templates declare (the unmeasurable-context and recursion-guard fallbacks stay as code invariants). See `test/dispatch.test.ts` for the anchored semantics.
|
|
48
59
|
|
|
49
|
-
|
|
60
|
+
## Configuration
|
|
50
61
|
|
|
62
|
+
Turn budgets are configurable via a JSON file, following the same two-layer
|
|
63
|
+
pattern as pi-subagents' `pi-subagent.json` (project overrides global):
|
|
64
|
+
|
|
65
|
+
- Global: `<agentDir>/pi-review.json`
|
|
66
|
+
- Project: `<cwd>/.pi/pi-review.json`
|
|
67
|
+
|
|
68
|
+
```jsonc
|
|
69
|
+
// <any layer>/pi-review.json — all keys optional
|
|
70
|
+
{
|
|
71
|
+
"maxTurns": {
|
|
72
|
+
"subagent": 20, // each /review finder-batch subagent call
|
|
73
|
+
"verifier": 15, // each /review Phase 2 verifier call
|
|
74
|
+
"gapHunt": 15, // the /review Phase 3 gap-hunter
|
|
75
|
+
"simplify": 15 // each /simplify PARALLEL cleaner agent
|
|
76
|
+
}
|
|
77
|
+
}
|
|
51
78
|
```
|
|
52
|
-
/code-simplify [<target>]
|
|
53
|
-
```
|
|
54
79
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
(
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
**Apply → verify → revert safety net** (harden-code-simplify): after Phase 2
|
|
70
|
-
applies the cleanups, the handler also injects a verification command detected
|
|
71
|
-
from `package.json` scripts (`check` → `test` → `lint` → `typecheck`). The
|
|
72
|
-
skill snapshots the touched files, applies the fixes, runs that command, and
|
|
73
|
-
on failure auto-reverts per-file (a clean apply runs verify exactly once; only
|
|
74
|
-
a failure escalates to one verify per touched file). The result is reported as
|
|
75
|
-
structured outcomes via `review_report` (`level: "simplify"`), not a free-text
|
|
76
|
-
summary. If no verification command is detectable, fixes are kept but the
|
|
77
|
-
report states no verification was run (verification is opportunistic, never
|
|
78
|
-
blocking).
|
|
79
|
-
|
|
80
|
-
### `subagent` tool
|
|
81
|
-
|
|
82
|
-
An LLM-callable tool that spawns one or more real pi subprocesses:
|
|
83
|
-
|
|
84
|
-
| mode | behavior |
|
|
85
|
-
|---|---|
|
|
86
|
-
| `single` | run `prompts[0]` once (e.g. an independent verify agent) |
|
|
87
|
-
| `parallel` | run all prompts concurrently, capped at the ceiling (e.g. one finder per angle) |
|
|
88
|
-
| `chain` | run sequentially; each later prompt receives prior output |
|
|
89
|
-
|
|
90
|
-
**Fan-out guards** (harden-code-simplify, shared with `/code-review`):
|
|
91
|
-
|
|
92
|
-
- **Recursion cap (whitelist-by-default)** — a spawned sub-agent does not
|
|
93
|
-
receive the `subagent` tool in its default toolset, so it cannot recurse. A
|
|
94
|
-
caller opts in by listing `subagent` in the child's `tools` whitelist; set
|
|
95
|
-
`PI_SUBAGENT_MAX_SPAWN_DEPTH` to allow multi-level fan-out up to a hard cap.
|
|
96
|
-
- **Default turn budget** — fan-out agents get a finite default `maxTurns` (50,
|
|
97
|
-
aligned to CC's `FORKED_AGENT_DEFAULT_MAX_TURNS` in 2.1.227)
|
|
98
|
-
when the caller omits it; an explicit `0` is honored.
|
|
99
|
-
- **Configurable concurrency** — `PI_MAX_CONCURRENT_SUBAGENTS` (default 20,
|
|
100
|
-
aligned to CC's `CLAUDE_CODE_MAX_CONCURRENT_SUBAGENTS ?? 20` in 2.1.227;
|
|
101
|
-
invalid values fall back to the default).
|
|
102
|
-
|
|
103
|
-
Each sub-agent is a full `pi --mode json -p --no-session` run. Progress streams
|
|
104
|
-
to the TUI via `onUpdate` as each agent completes. ESC aborts the whole batch
|
|
105
|
-
(SIGTERM → 5s → SIGKILL per subprocess). Errors are thrown (not returned) so
|
|
106
|
-
the agent loop marks the result `isError`.
|
|
107
|
-
|
|
108
|
-
## Architecture (layered)
|
|
80
|
+
Values must be positive integers; anything else (or an absent file) falls back
|
|
81
|
+
to the built-in defaults — `20` / `15` / `15` / `15`, the numbers the bundled
|
|
82
|
+
prompts and skills were written with — so with no configuration the rendered
|
|
83
|
+
instructions are byte-identical to the pre-config behavior. Files are read at
|
|
84
|
+
command time: an edit takes effect on the next `/review` or `/simplify`
|
|
85
|
+
without a restart. When a budget is configured, the trigger message states it
|
|
86
|
+
and the skills defer to it over their built-in defaults.
|
|
87
|
+
|
|
88
|
+
## Requirements
|
|
89
|
+
|
|
90
|
+
None beyond this package. `@fyeeme/pi-subagents` 2.1.1 is a regular npm dependency (exact-pinned) whose extension factory this entry composes (tool + UI). The four cleaner agents and the finder/verifier/gap-hunter definitions ship with this package, registered via `addAgentDir` at extension load.
|
|
91
|
+
|
|
92
|
+
## Development
|
|
109
93
|
|
|
110
94
|
```
|
|
111
|
-
pi-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
├── src/
|
|
115
|
-
│ ├── skills.ts bundledSkillPath — resolve this extension's own skills/ dir
|
|
116
|
-
│ ├── tools/subagent.ts defineTool("subagent") — generic capability layer
|
|
117
|
-
│ └── commands/
|
|
118
|
-
│ ├── code-review.ts /code-review handler + sticky last-used effort (CC 2.1.223)
|
|
119
|
-
│ └── code-simplify.ts /code-simplify handler + decideSimplifyMode (Jvo guard)
|
|
120
|
-
└── test/ commands unit tests
|
|
95
|
+
npm install --ignore-scripts # @fyeeme/pi-subagents resolves from the npm registry
|
|
96
|
+
npm test # vitest
|
|
97
|
+
npm run typecheck
|
|
121
98
|
```
|
|
122
99
|
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
can be split into its own `pi-subagent` extension with zero refactor — the code
|
|
127
|
-
is already separated.
|
|
128
|
-
|
|
129
|
-
The dispatch primitive (`spawnAgent`, `mapWithConcurrencyLimit`,
|
|
130
|
-
`createSpawnRegistry`, `abortAgent`, `getPiInvocation` + types) lives in
|
|
131
|
-
[`pi-subagent-core`](../pi-subagent-core) (npm `@fyeeme/pi-subagent-core`), a
|
|
132
|
-
shared library extracted from the duplicated copies that used to live here and
|
|
133
|
-
in `pi-dynamic-workflows`. When pi promotes `spawnAgent` to a public
|
|
134
|
-
`pi-coding-agent` export, `pi-subagent-core` should be deleted in favor of that
|
|
135
|
-
import.
|
|
136
|
-
|
|
137
|
-
## Relation to the skills
|
|
138
|
-
|
|
139
|
-
| layer | home | role |
|
|
140
|
-
|---|---|---|
|
|
141
|
-
| review/cleanup semantics (angles, verdicts, mode bodies) | `skills/code-review/` + `skills/simplify/` (bundled in this package) | what to look for |
|
|
142
|
-
| fan-out dispatch + mode decision | this extension (`subagent` tool + command handlers) | how to run sub-agents / which mode |
|
|
143
|
-
|
|
144
|
-
Edit a skill to change *what* it hunts; edit this extension to change *how*
|
|
145
|
-
sub-agents are spawned and *which mode* is chosen.
|
|
146
|
-
|
|
147
|
-
## Status
|
|
148
|
-
|
|
149
|
-
`review_report` is built — schema aligned to CC `ReportFindings` (2.1.227
|
|
150
|
-
empirical): 3-state `outcome` (`fixed`/`skipped`/`no_change_needed`), 2-value
|
|
151
|
-
`verdict` (`CONFIRMED`/`PLAUSIBLE`), `short_summary` (≤60, table overview),
|
|
152
|
-
`report_id` for fixed-later re-reports; renders the Chinese Markdown report
|
|
153
|
-
and writes JSON to `<cwd>/.pi/review/` for CI / `--fix` / `--comment`.
|
|
154
|
-
|
|
155
|
-
Remaining Phase 2 item (not yet built): a `review_verify` tool encapsulating
|
|
156
|
-
3-vote adversarial verify. `--share` already routes through lavish-axi (see
|
|
157
|
-
the code-review skill).
|
|
100
|
+
To test local pi-subagents changes alongside this package, temporarily point the
|
|
101
|
+
dependency back at the sibling checkout (`file:../pi-subagents`) and reinstall;
|
|
102
|
+
restore the pinned registry version before publishing.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cleaner-altitude
|
|
3
|
+
description: Checks each change fixes the root cause at the right depth, not as a fragile bandaid (simplify Altitude angle / review Altitude finder)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are an altitude (right-depth) reviewer. Review the changed code given to
|
|
7
|
+
you for altitude issues.
|
|
8
|
+
|
|
9
|
+
Check that each change fixes the root cause at the right depth rather than
|
|
10
|
+
patching a symptom with a fragile bandaid. Special cases layered on shared
|
|
11
|
+
infrastructure are a sign the fix isn't deep enough — prefer the simpler,
|
|
12
|
+
more general change to the underlying mechanism over adding special cases,
|
|
13
|
+
and name that change.
|
|
14
|
+
|
|
15
|
+
Return your findings as a concise list. For each finding: `file:line` —
|
|
16
|
+
one-line summary — the concrete cost (what is fragile or will not
|
|
17
|
+
generalize), naming the deeper change. Do not propose applying fixes; report
|
|
18
|
+
only. An empty list is a valid answer.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cleaner-efficiency
|
|
3
|
+
description: Flags wasted work the diff introduces (simplify Efficiency angle / review Efficiency finder)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are an efficiency reviewer. Review the changed code given to you for
|
|
7
|
+
efficiency opportunities.
|
|
8
|
+
|
|
9
|
+
Flag wasted work the diff introduces: redundant computation or repeated I/O,
|
|
10
|
+
independent operations run sequentially, blocking work added to startup or
|
|
11
|
+
hot paths. Also flag long-lived objects built from closures or captured
|
|
12
|
+
environments — they keep the entire enclosing scope alive for the object's
|
|
13
|
+
lifetime (a memory leak when that scope holds large values); prefer a
|
|
14
|
+
class/struct that copies only the fields it needs. Name the cheaper
|
|
15
|
+
alternative.
|
|
16
|
+
|
|
17
|
+
Return your findings as a concise list. For each finding: `file:line` —
|
|
18
|
+
one-line summary — the concrete cost. Do not propose applying fixes; report
|
|
19
|
+
only. An empty list is a valid answer.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cleaner-reuse
|
|
3
|
+
description: Flags new code that re-implements something the codebase already has (simplify Reuse angle / review Reuse finder)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a reuse reviewer. Review the changed code given to you for reuse
|
|
7
|
+
cleanup opportunities.
|
|
8
|
+
|
|
9
|
+
Grep shared/utility modules and files adjacent to the change; flag new code
|
|
10
|
+
that re-implements something the codebase already has, and name the existing
|
|
11
|
+
helper to call instead.
|
|
12
|
+
|
|
13
|
+
Return your findings as a concise list. For each finding: `file:line` —
|
|
14
|
+
one-line summary — the concrete cost (what is duplicated, wasted, or harder
|
|
15
|
+
to maintain). Do not propose applying fixes; report only. An empty list is a
|
|
16
|
+
valid answer.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: cleaner-simplification
|
|
3
|
+
description: Flags unnecessary complexity the diff adds (simplify Simplification angle / review Simplification finder)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a simplification reviewer. Review the changed code given to you for
|
|
7
|
+
simplification opportunities.
|
|
8
|
+
|
|
9
|
+
Flag unnecessary complexity the diff adds: redundant or derivable state,
|
|
10
|
+
copy-paste with slight variation, deep nesting, dead code left behind. Name
|
|
11
|
+
the simpler form that does the same job.
|
|
12
|
+
|
|
13
|
+
Return your findings as a concise list. For each finding: `file:line` —
|
|
14
|
+
one-line summary — the concrete cost (what is duplicated, wasted, or harder
|
|
15
|
+
to maintain). Do not propose applying fixes; report only. An empty list is a
|
|
16
|
+
valid answer.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-conventions
|
|
3
|
+
description: "Conventions finder: CLAUDE.md/AGENTS.md rule violations in the changed code"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a conventions finder.
|
|
7
|
+
|
|
8
|
+
Find the CLAUDE.md/AGENTS.md files that govern the changed code you are
|
|
9
|
+
given: the user-level ~/.claude/CLAUDE.md, the repo-root CLAUDE.md, plus any
|
|
10
|
+
CLAUDE.md or CLAUDE.local.md (or AGENTS.md) in a directory that is an
|
|
11
|
+
ancestor of a changed file (a directory's file only applies to files at or
|
|
12
|
+
below it). Read each one that exists, then check the diff for clear
|
|
13
|
+
violations of the rules they state.
|
|
14
|
+
|
|
15
|
+
Only flag a violation when you can quote the exact rule and the exact line
|
|
16
|
+
that breaks it — no style preferences, no vague "spirit of the doc"
|
|
17
|
+
inferences. In the finding, name the doc path and quote the rule so the
|
|
18
|
+
report can cite it. If no doc applies, return an empty array.
|
|
19
|
+
|
|
20
|
+
Your LAST assistant message must be your JSON candidate array — `[]` is a
|
|
21
|
+
valid answer. Each candidate: `{"file", "line"?, "category":
|
|
22
|
+
"conventions", "summary", "failure_scenario"}` (the failure_scenario states
|
|
23
|
+
which quoted rule is broken and the concrete cost).
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-cross-file
|
|
3
|
+
description: "Correctness angle C: cross-file tracer"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a correctness finder (angle C — cross-file tracer).
|
|
7
|
+
|
|
8
|
+
For each function the diff you are given changes, find its callers (grep for
|
|
9
|
+
the symbol) and check whether the change breaks any call site: a new
|
|
10
|
+
precondition, a changed return shape, a new exception, a timing/ordering
|
|
11
|
+
dependency. Also check callees: does a parallel change in the same PR make a
|
|
12
|
+
call unsafe?
|
|
13
|
+
|
|
14
|
+
Spend your tool-call budget on the highest-risk hunks first; when half is
|
|
15
|
+
spent, stop opening new files. Your LAST assistant message must be your JSON
|
|
16
|
+
candidate array — `[]` is a valid answer. Partial output beats none.
|
|
17
|
+
|
|
18
|
+
Each candidate: `{"file", "line"?, "category": "correctness", "summary",
|
|
19
|
+
"failure_scenario"}` — the failure_scenario names a concrete input/state →
|
|
20
|
+
wrong output or crash. Pass every candidate with a nameable failure scenario
|
|
21
|
+
through; do not silently drop half-believed candidates.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-diff-scan
|
|
3
|
+
description: "Correctness angle A: line-by-line diff scan"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a correctness finder (angle A — line-by-line diff scan).
|
|
7
|
+
|
|
8
|
+
Read every hunk in the diff you are given, line by line. Then read the
|
|
9
|
+
enclosing function for each hunk — bugs in unchanged lines of a touched
|
|
10
|
+
function are in scope (the PR re-exposes or fails to fix them). For every
|
|
11
|
+
line ask: what input, state, timing, or platform makes this line wrong?
|
|
12
|
+
Look for inverted/wrong conditions, off-by-one, null/undefined deref,
|
|
13
|
+
missing `await`, falsy-zero checks, wrong-variable copy-paste, error
|
|
14
|
+
swallowed in catch, unescaped regex metachars.
|
|
15
|
+
|
|
16
|
+
Spend your tool-call budget on the highest-risk hunks first; when half is
|
|
17
|
+
spent, stop opening new files. Your LAST assistant message must be your JSON
|
|
18
|
+
candidate array — `[]` is a valid answer. Partial output beats none.
|
|
19
|
+
|
|
20
|
+
Each candidate: `{"file", "line"?, "category": "correctness", "summary",
|
|
21
|
+
"failure_scenario"}` — the failure_scenario names a concrete input/state →
|
|
22
|
+
wrong output or crash. Pass every candidate with a nameable failure scenario
|
|
23
|
+
through; do not silently drop half-believed candidates.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-language-pitfall
|
|
3
|
+
description: "Correctness angle D: language-pitfall specialist"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a correctness finder (angle D — language-pitfall specialist).
|
|
7
|
+
|
|
8
|
+
Scan the diff you are given for the classic pitfalls of its
|
|
9
|
+
language/framework — for example: JS falsy-zero, `==` coercion,
|
|
10
|
+
closure-captured loop var; Python mutable default args, late-binding
|
|
11
|
+
closures; Go nil-map write, range-var capture; SQL injection;
|
|
12
|
+
timezone/DST drift; float equality. Flag any instance the diff introduces.
|
|
13
|
+
|
|
14
|
+
Spend your tool-call budget on the highest-risk hunks first; when half is
|
|
15
|
+
spent, stop opening new files. Your LAST assistant message must be your JSON
|
|
16
|
+
candidate array — `[]` is a valid answer. Partial output beats none.
|
|
17
|
+
|
|
18
|
+
Each candidate: `{"file", "line"?, "category": "correctness", "summary",
|
|
19
|
+
"failure_scenario"}` — the failure_scenario names a concrete input/state →
|
|
20
|
+
wrong output or crash. Pass every candidate with a nameable failure scenario
|
|
21
|
+
through; do not silently drop half-believed candidates.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-removed-behavior
|
|
3
|
+
description: "Correctness angle B: removed-behavior auditor"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a correctness finder (angle B — removed-behavior auditor).
|
|
7
|
+
|
|
8
|
+
For every line the diff you are given DELETES or replaces, name the
|
|
9
|
+
invariant or behavior it enforced, then search the new code for where that
|
|
10
|
+
invariant is re-established. If you can't find it, that's a candidate: a
|
|
11
|
+
removed guard, a dropped error path, a narrowed validation, a deleted test
|
|
12
|
+
that was covering a real case.
|
|
13
|
+
|
|
14
|
+
Spend your tool-call budget on the highest-risk hunks first; when half is
|
|
15
|
+
spent, stop opening new files. Your LAST assistant message must be your JSON
|
|
16
|
+
candidate array — `[]` is a valid answer. Partial output beats none.
|
|
17
|
+
|
|
18
|
+
Each candidate: `{"file", "line"?, "category": "correctness", "summary",
|
|
19
|
+
"failure_scenario"}` — the failure_scenario names a concrete input/state →
|
|
20
|
+
wrong output or crash. Pass every candidate with a nameable failure scenario
|
|
21
|
+
through; do not silently drop half-believed candidates.
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: finder-wrapper-proxy
|
|
3
|
+
description: "Correctness angle E: wrapper/proxy correctness"
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are a correctness finder (angle E — wrapper/proxy correctness).
|
|
7
|
+
|
|
8
|
+
When the diff you are given adds or modifies a type that wraps another
|
|
9
|
+
(cache, proxy, decorator, adapter): check that every method routes to the
|
|
10
|
+
wrapped instance and not back through a registry/session/global — e.g. a
|
|
11
|
+
caching provider holding a `delegate` field that resolves IDs via
|
|
12
|
+
`session.get(...)` instead of `delegate.get(...)` will re-enter the cache or
|
|
13
|
+
recurse. Also check that the wrapper forwards all the methods the callers
|
|
14
|
+
actually use.
|
|
15
|
+
|
|
16
|
+
Spend your tool-call budget on the highest-risk hunks first; when half is
|
|
17
|
+
spent, stop opening new files. Your LAST assistant message must be your JSON
|
|
18
|
+
candidate array — `[]` is a valid answer. Partial output beats none.
|
|
19
|
+
|
|
20
|
+
Each candidate: `{"file", "line"?, "category": "correctness", "summary",
|
|
21
|
+
"failure_scenario"}` — the failure_scenario names a concrete input/state →
|
|
22
|
+
wrong output or crash. Pass every candidate with a nameable failure scenario
|
|
23
|
+
through; do not silently drop half-believed candidates.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: gap-hunter
|
|
3
|
+
description: Fresh finder hunting only for gaps not already in the candidate list (xhigh/max sweep, max 8 new candidates)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are the gap-hunter: a FRESH finder that has never seen the candidates
|
|
7
|
+
before. You hunt ONLY for gaps not already listed. You analyze, you do not
|
|
8
|
+
discover: all context (the diff, enclosing functions, the deduplicated
|
|
9
|
+
finding list, search results) is embedded in the task you are given — do not
|
|
10
|
+
go searching for callers/dependencies yourself.
|
|
11
|
+
|
|
12
|
+
You have ONLY about 3 tool calls, to read the files central to the embedded
|
|
13
|
+
findings. Read them now, then analyze from this message's context. Report
|
|
14
|
+
**at most 8 new candidates** — issues the existing list does NOT already
|
|
15
|
+
cover (a sharper version of a listed issue counts as new). Focus on what the
|
|
16
|
+
first pass tends to miss (CC 2.1.261 sweep list): moved/extracted code that
|
|
17
|
+
dropped a guard or anchor; second-tier footguns (dataclass default evaluated
|
|
18
|
+
once, `hash()` non-determinism, lock-scope shrink, predicate methods with
|
|
19
|
+
side effects); setup/teardown asymmetry in tests; config defaults flipped.
|
|
20
|
+
If nothing new turns up, return an empty sweep — do not pad.
|
|
21
|
+
|
|
22
|
+
Your LAST assistant message must be your JSON candidate array — `[]` is a
|
|
23
|
+
valid answer. Each candidate: `{"file", "line"?, "category",
|
|
24
|
+
"summary", "failure_scenario"}`.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: verifier
|
|
3
|
+
description: Independent verdict agent — judges candidate findings per location group (CONFIRMED/PLAUSIBLE/REFUTED)
|
|
4
|
+
tools: read, grep, find, ls, bash
|
|
5
|
+
---
|
|
6
|
+
You are an independent verifier. You receive the scope, the diff, the
|
|
7
|
+
relevant file(s), and a numbered candidate list for ONE location group. You
|
|
8
|
+
judge each candidate independently — same-location candidates may describe
|
|
9
|
+
different defects.
|
|
10
|
+
|
|
11
|
+
For each candidate, return a verdict:
|
|
12
|
+
|
|
13
|
+
- **CONFIRMED** — you can name the inputs/state that trigger it and the
|
|
14
|
+
wrong output or crash. Quote the line.
|
|
15
|
+
- **PLAUSIBLE** — the mechanism is real, the trigger is uncertain (timing,
|
|
16
|
+
env, config). State what would confirm it. Do NOT refute a candidate for
|
|
17
|
+
being "speculative" or "depends on runtime state" when the state is
|
|
18
|
+
realistic: concurrency races, nil/undefined on a rare-but-reachable path
|
|
19
|
+
(error handler, cold cache, missing optional field), falsy-zero treated
|
|
20
|
+
as missing, off-by-one on a boundary the code does not exclude, retry
|
|
21
|
+
storms / partial failures, regex/allowlist that lost an anchor — these
|
|
22
|
+
are PLAUSIBLE.
|
|
23
|
+
- **REFUTED** only when constructible from the code: factually wrong (quote
|
|
24
|
+
the actual line); provably impossible (type/constant/invariant — show
|
|
25
|
+
it); already handled in this diff (cite the guard); or pure style with no
|
|
26
|
+
observable effect.
|
|
27
|
+
|
|
28
|
+
Your LAST assistant message must be your JSON verdict array, one entry per
|
|
29
|
+
candidate index — never skip an index:
|
|
30
|
+
|
|
31
|
+
```
|
|
32
|
+
[{ "index": <n>, "verdict": "CONFIRMED" | "PLAUSIBLE" | "REFUTED", "evidence": "<quote/argument>" }, ...]
|
|
33
|
+
```
|
package/index.ts
CHANGED
|
@@ -1,40 +1,45 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* pi-review — extension entry.
|
|
2
|
+
* pi-review v2 — extension entry.
|
|
3
3
|
*
|
|
4
|
-
*
|
|
5
|
-
* - the `subagent` tool — general-purpose parallel/sequential sub-agent fan-out
|
|
6
|
-
* via real pi subprocesses. Shared capability used by both skills below;
|
|
7
|
-
* - the `review_report` tool — structured findings sink for the code-review
|
|
8
|
-
* skill (Pi's counterpart to CC's ReportFindings): renders the Markdown
|
|
9
|
-
* report + writes JSON to <cwd>/.pi/review/ for CI;
|
|
10
|
-
* - the `/code-review` command — effort-level review via the code-review skill;
|
|
11
|
-
* - the `/code-simplify` command — cleanup via the simplify skill; the handler
|
|
12
|
-
* decides parallel vs single-pass from ctx.getContextUsage(), mirroring CC's
|
|
13
|
-
* Jvo guard (a deterministic decision a pure-prompt skill cannot reproduce).
|
|
4
|
+
* Sandwich architecture (see openspec change subagent-sandwich-refactor):
|
|
14
5
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
6
|
+
* Skills skills/review, skills/simplify — review methodology,
|
|
7
|
+
* registered natively via the pi manifest (`pi.skills`); they
|
|
8
|
+
* reference capabilities by stable tool/agent names only.
|
|
9
|
+
* Prompts prompts/ — the orchestration strategy as data: parallel-when
|
|
10
|
+
* guards in frontmatter, phases/agents in the body. Rendered by
|
|
11
|
+
* the generic dispatcher (src/dispatch.ts) which gathers the
|
|
12
|
+
* deterministic runtime variables (diff, context usage, sticky
|
|
13
|
+
* effort) and picks the template variant.
|
|
14
|
+
* Agents agents/ — the review angles materialized as subagent definitions
|
|
15
|
+
* (finder-*, cleaner-*, verifier, gap-hunter) invoked via the
|
|
16
|
+
* `subagent` tool.
|
|
17
|
+
* Plugin this entry composes @fyeeme/pi-subagents' extension factory
|
|
18
|
+
* (subagent tool + agent UI + /agents, from the SAME dependency
|
|
19
|
+
* copy this package's imports resolve to — version-pinned, no
|
|
20
|
+
* manifest path wiring and no separate install step), registers
|
|
21
|
+
* this package's agents directory as a discovery source, and adds
|
|
22
|
+
* the `review_report` structured findings sink plus the
|
|
23
|
+
* /review and /simplify dispatcher commands.
|
|
17
24
|
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
25
|
+
* The `subagent` tool registers exactly once per process: if pi-subagents
|
|
26
|
+
* is ALSO installed standalone (or another consumer composes it), the guard
|
|
27
|
+
* in pi-subagents' index.ts keeps ownership single (pi fatal-exits on
|
|
28
|
+
* the same tool name in two extensions' maps).
|
|
22
29
|
*/
|
|
23
30
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
24
|
-
import {
|
|
25
|
-
import {
|
|
26
|
-
import
|
|
27
|
-
import {
|
|
31
|
+
import piSubagents, { addAgentDir } from "@fyeeme/pi-subagents";
|
|
32
|
+
import { fileURLToPath } from "node:url";
|
|
33
|
+
import * as path from "node:path";
|
|
34
|
+
import { registerDispatcher } from "./src/dispatch.ts";
|
|
28
35
|
import { reviewReportTool } from "./src/tools/review_report.ts";
|
|
29
36
|
|
|
30
37
|
export default function (pi: ExtensionAPI): void {
|
|
31
|
-
//
|
|
32
|
-
//
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
if (isFanoutToolAllowed()) pi.registerTool(subagentTool);
|
|
38
|
+
// subagent tool + live agent UI, from this package's pinned dependency
|
|
39
|
+
// copy. <pkg>/index.ts → sibling agents/ dir registers the review roles.
|
|
40
|
+
piSubagents(pi);
|
|
41
|
+
addAgentDir(path.join(path.dirname(fileURLToPath(import.meta.url)), "agents"));
|
|
42
|
+
|
|
37
43
|
pi.registerTool(reviewReportTool);
|
|
38
|
-
|
|
39
|
-
registerSimplify(pi);
|
|
44
|
+
registerDispatcher(pi);
|
|
40
45
|
}
|