@iceinvein/agent-skills 0.21.1 → 0.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/index.json +2 -2
- package/skills/magpie/README.md +3 -3
- package/skills/magpie/SKILL.md +24 -11
- package/skills/magpie/bin/magpie.ts +84 -6
- package/skills/magpie/evals/README.md +15 -0
- package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +63 -22
- package/skills/magpie/evals/post-folds-selection-events/fixture.sh +8 -2
- package/skills/magpie/evals/quality/README.md +141 -0
- package/skills/magpie/evals/quality/__tests__/replay-critic.test.ts +31 -0
- package/skills/magpie/evals/quality/__tests__/score.test.ts +325 -0
- package/skills/magpie/evals/quality/baseline.ts +80 -0
- package/skills/magpie/evals/quality/corpus.ts +259 -0
- package/skills/magpie/evals/quality/replay-critic.ts +277 -0
- package/skills/magpie/evals/quality/replay-full.ts +188 -0
- package/skills/magpie/evals/quality/score.ts +251 -0
- package/skills/magpie/evals/report-ends-the-turn/fixture.sh +5 -1
- package/skills/magpie/references/critic.md +178 -40
- package/skills/magpie/references/peer-review.md +5 -4
- package/skills/magpie/references/scout.md +18 -3
- package/skills/magpie/references/specialists.md +36 -22
- package/skills/magpie/scripts/__tests__/cleanup-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/critic-cmd.test.ts +260 -0
- package/skills/magpie/scripts/__tests__/critic.test.ts +328 -0
- package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +89 -0
- package/skills/magpie/scripts/__tests__/dedupe.test.ts +43 -1
- package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +155 -2
- package/skills/magpie/scripts/__tests__/helper.test.ts +80 -0
- package/skills/magpie/scripts/__tests__/labels.test.ts +152 -0
- package/skills/magpie/scripts/__tests__/open-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/pipeline-e2e.test.ts +1 -0
- package/skills/magpie/scripts/__tests__/post-cmd.test.ts +47 -0
- package/skills/magpie/scripts/__tests__/refresh.test.ts +23 -0
- package/skills/magpie/scripts/__tests__/render-action-bar.test.ts +63 -5
- package/skills/magpie/scripts/__tests__/render-annotation.test.ts +38 -0
- package/skills/magpie/scripts/__tests__/render-cmd.test.ts +48 -0
- package/skills/magpie/scripts/__tests__/render-diff.test.ts +34 -0
- package/skills/magpie/scripts/__tests__/render-findings.test.ts +111 -7
- package/skills/magpie/scripts/__tests__/render-issues-list.test.ts +186 -1
- package/skills/magpie/scripts/__tests__/serve-cmd.test.ts +19 -0
- package/skills/magpie/scripts/__tests__/server.test.ts +76 -0
- package/skills/magpie/scripts/__tests__/skill-lint.test.ts +148 -9
- package/skills/magpie/scripts/__tests__/tests-check.test.ts +57 -0
- package/skills/magpie/scripts/__tests__/types.test.ts +76 -29
- package/skills/magpie/scripts/cleanup-cmd.ts +6 -0
- package/skills/magpie/scripts/critic-cmd.ts +183 -0
- package/skills/magpie/scripts/critic.ts +273 -0
- package/skills/magpie/scripts/dedupe-cmd.ts +14 -4
- package/skills/magpie/scripts/dedupe.ts +41 -0
- package/skills/magpie/scripts/evidence-filter.ts +81 -17
- package/skills/magpie/scripts/helper.js +87 -9
- package/skills/magpie/scripts/labels-cmd.ts +43 -0
- package/skills/magpie/scripts/labels.ts +99 -0
- package/skills/magpie/scripts/open-cmd.ts +10 -3
- package/skills/magpie/scripts/post-cmd.ts +26 -2
- package/skills/magpie/scripts/preview-cmd.ts +4 -0
- package/skills/magpie/scripts/refresh.ts +21 -17
- package/skills/magpie/scripts/render-action-bar.ts +21 -4
- package/skills/magpie/scripts/render-annotation.ts +36 -2
- package/skills/magpie/scripts/render-cmd.ts +21 -2
- package/skills/magpie/scripts/render-diff.ts +6 -2
- package/skills/magpie/scripts/render-findings.ts +21 -3
- package/skills/magpie/scripts/render-issues-list.ts +66 -16
- package/skills/magpie/scripts/server.ts +8 -1
- package/skills/magpie/scripts/tests-check.ts +23 -2
- package/skills/magpie/scripts/types.ts +37 -31
- package/skills/magpie/skill.json +2 -2
- package/skills/magpie/templates/styles.css +83 -0
- package/skills/magpie/tsconfig.json +1 -1
package/package.json
CHANGED
package/skills/index.json
CHANGED
|
@@ -219,9 +219,9 @@
|
|
|
219
219
|
},
|
|
220
220
|
{
|
|
221
221
|
"name": "magpie",
|
|
222
|
-
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture),
|
|
222
|
+
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), drops findings whose quoted evidence is not in the code, dedupes, runs critic subagents that re-check each finding against the worktree and set its risk label, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting or dismissing findings to post via gh, recording the outcome per finding as labels. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
|
|
223
223
|
"type": "prompt",
|
|
224
|
-
"version": "0.
|
|
224
|
+
"version": "0.13.0"
|
|
225
225
|
},
|
|
226
226
|
{
|
|
227
227
|
"name": "migrate",
|
package/skills/magpie/README.md
CHANGED
|
@@ -4,7 +4,7 @@ Interactive Claude Code skill that runs a multi-stage PR review pipeline inside
|
|
|
4
4
|
|
|
5
5
|
## What it does
|
|
6
6
|
|
|
7
|
-
Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture),
|
|
7
|
+
Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture), each quoting an evidence snippet for every anchored finding. Dedupe drops findings whose snippet is not in the file and groups nearby findings from different domains as merge candidates. Critic subagents then re-check each surviving finding against the worktree, keep, drop or merge it, and set its risk label, from which the severity is derived. Peer review runs via `codex exec` (falling back to a Claude second-opinion subagent when codex is unavailable). The interactive HTML report puts the top recommended findings first, lets the user dismiss findings with a reason, and posts the ones they select via `gh`; cleanup records what was posted, dismissed or ignored in `labels.json`.
|
|
8
8
|
|
|
9
9
|
## Requirements
|
|
10
10
|
|
|
@@ -54,7 +54,7 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
|
|
|
54
54
|
## Layout
|
|
55
55
|
|
|
56
56
|
- `SKILL.md` is the agent-facing prompt: the stage walkthrough and nothing else. Installed by the agent-skills CLI.
|
|
57
|
-
- `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract), `specialists.md` (stage 4, the five focus blocks plus the shared output contract), `critic.md` (stage 6), `peer-review.md` (stage 7, including the Claude-fallback preamble). They ship in the bundle and sit next to `SKILL.md` once installed.
|
|
57
|
+
- `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract, including the repository review rules), `specialists.md` (stage 4, the five focus blocks plus the shared output contract), `critic.md` (stage 6, the critic subagent prompt `magpie critic-prompt` fills), `peer-review.md` (stage 7, including the Claude-fallback preamble). They ship in the bundle and sit next to `SKILL.md` once installed.
|
|
58
58
|
- `skill.json` is the agent-skills manifest.
|
|
59
59
|
- `bin/magpie` is the CLI invoked by the agent during stages; symlinked onto PATH by `install.sh`.
|
|
60
60
|
- `scripts/` holds the implementation (server, dedupe, render, setup, cleanup, etc.).
|
|
@@ -66,4 +66,4 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
|
|
|
66
66
|
|
|
67
67
|
## Run directory layout
|
|
68
68
|
|
|
69
|
-
Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `findings.kept.json`, `findings.final.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
|
|
69
|
+
Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `merge-candidates.json`, `critic-prompt.md` and `critic.json` (numbered per batch when there is more than one), `findings.kept.json`, `critic-dropped.json`, `findings.final.json`, `labels.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
|
package/skills/magpie/SKILL.md
CHANGED
|
@@ -148,7 +148,9 @@ If every specialist fails (no findings files written), log
|
|
|
148
148
|
magpie dedupe "$RUN_DIR" [--threshold <0-10>]
|
|
149
149
|
```
|
|
150
150
|
|
|
151
|
-
`magpie dedupe` also
|
|
151
|
+
`magpie dedupe` also checks evidence against the worktree, dropping an anchored finding whose file is missing (`hallucinated-file`), whose line is out of range (`invented-line`), that has no `evidence` snippet (`missing-evidence`), or whose snippet is not near its line (`evidence-not-found`); a snippet found exactly once elsewhere in the file re-anchors it. Both go to `$RUN_DIR/evidence-dropped.json` as `{dropped, reanchored}`, written only when non-empty, so no file is normal. The check is skipped when the worktree is gone (archived run replay).
|
|
152
|
+
|
|
153
|
+
It always writes `$RUN_DIR/merge-candidates.json`: ids from different domains anchored close together in one file, for the critic to merge or keep apart.
|
|
152
154
|
|
|
153
155
|
Each finding gets a derived 0-10 `score` from its risk fields; those below `--threshold` (default 3) are dropped before the critic LLM runs and recorded to `$RUN_DIR/threshold-dropped.json`. Pass `--threshold 0` to keep everything.
|
|
154
156
|
|
|
@@ -156,12 +158,21 @@ Re-render progress.
|
|
|
156
158
|
|
|
157
159
|
### 6. Critic
|
|
158
160
|
|
|
159
|
-
|
|
161
|
+
The critic runs as worktree-reading subagents. The CLI fills its prompt from `references/critic.md`; substitute nothing by hand. Append `{stage: critic, status: running}` to `$RUN_DIR/log.jsonl`, re-render progress, then:
|
|
162
|
+
|
|
163
|
+
```
|
|
164
|
+
magpie critic-prompt "$RUN_DIR"
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
It prints `<prompt path>\t<output path>` per batch of about 30 candidates (a merge group always stays in one batch). Dispatch one subagent (Agent tool, `general-purpose`) per line, all in one message, each one's entire task the verbatim contents of its prompt file. Each writes its output path and returns one summary line. No lines printed means no candidates: go straight to `critic-apply`.
|
|
160
168
|
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
169
|
+
Confirm every output path exists; re-dispatch any batch whose file is missing. On a resume, dispatch only the batches whose output file is missing. Apply the verdicts:
|
|
170
|
+
|
|
171
|
+
```
|
|
172
|
+
magpie critic-apply "$RUN_DIR"
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
It writes `findings.kept.json` and `critic-dropped.json` and logs the critic `done` entry itself; do not append another. On a non-zero exit, stderr names offending ids or a file: delete the output file of each batch whose ids or file it names, re-dispatch only those, and re-run `critic-apply`. If a batch fails twice after re-dispatch, stop and show the user stderr verbatim rather than looping. Re-render progress.
|
|
165
176
|
|
|
166
177
|
### 7. Peer review
|
|
167
178
|
|
|
@@ -169,7 +180,7 @@ Append `{stage: peer-review, status: running}` to `$RUN_DIR/log.jsonl` and re-re
|
|
|
169
180
|
|
|
170
181
|
Build the peer-review prompt first: read `references/peer-review.md`, take the `magpie-peer-review` block from it, and substitute the placeholders listed in that file's `## Substitute before use` preamble.
|
|
171
182
|
|
|
172
|
-
One batch carries up to 40 findings; above that, split them 30 at a time
|
|
183
|
+
One batch carries up to 40 findings; above that, split them 30 at a time.
|
|
173
184
|
Write each batch's prompt, its `<<KEPT_FINDINGS_COMPACT>>` narrowed to that batch, to
|
|
174
185
|
`$RUN_DIR/peer-prompt-<k>.md`, `<k>` counting from 1. **When there is a single batch, drop `-<k>` throughout** (`peer-prompt.md`,
|
|
175
186
|
`peer.out`), which is the common case. Keep the `add` id counter running across batches
|
|
@@ -189,7 +200,7 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
|
|
|
189
200
|
|
|
190
201
|
**Claude path (fallback).** When `codex` is unavailable or failed, get the second opinion from a Claude subagent instead, one per batch. Set `<<PEER_PROVIDER>>` to `claude`, then prepend the `magpie-peer-review-claude-preamble` block from `references/peer-review.md` to each batch's substituted prompt (the preamble forces genuine independence, since the reviewer shares a model family with the primary reviewers). Dispatch one subagent (Agent tool, `general-purpose`) per batch whose entire task is that combined prompt, and instruct it to return only the fenced `review-peer-review` JSON block. Write each output to `$RUN_DIR/peer-<k>.out`, extract each `review-peer-review` block, merge into `$RUN_DIR/peer.json` after the last batch as above, and append `{stage: peer-review, status: done, provider: claude}` (`provider: mixed` if codex handled some batches).
|
|
191
202
|
|
|
192
|
-
**Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
|
|
203
|
+
**Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). A peer `fields.severity` (or `finding.severity` on an `add`) is ignored: severity is derived from `risk.impact`. Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
|
|
193
204
|
|
|
194
205
|
### 8. Report
|
|
195
206
|
|
|
@@ -197,6 +208,8 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
|
|
|
197
208
|
magpie render "$RUN_DIR" findings
|
|
198
209
|
```
|
|
199
210
|
|
|
211
|
+
The report shows the top 10 recommended (`must-fix`/`should-fix`, highest score first) and folds the rest. `--top <n>` is not saved (the stage 9 re-render and `magpie serve`'s auto-refresh show 10 again), so use it only when nothing will re-render.
|
|
212
|
+
|
|
200
213
|
Append `{stage: report, status: done}` to `$RUN_DIR/log.jsonl` and re-render progress (the render CLI does not log this itself, and `magpie status` needs the `done` entry to resume past `report`).
|
|
201
214
|
|
|
202
215
|
Print to the terminal: "Findings ready at <url>. Tick the ones you want and click **Post Selected**, or reply `post` here and I'll post whatever you've ticked."
|
|
@@ -205,9 +218,9 @@ End the turn.
|
|
|
205
218
|
|
|
206
219
|
### 9. Post
|
|
207
220
|
|
|
208
|
-
Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes
|
|
221
|
+
Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes only the top N above the fold; **Select recommended** ticks the same set); the server posts that batch as one GitHub review with inline threads. **Dismiss** on a finding records a reason (`wrong`, `not-worth-it`, `duplicate`, `style`) and drops it from the recommended set. The agent posts only when the user types `post` (optionally `post 1,3,7` for indices), which takes the CLI path below: separate inline comments plus a top-level summary comment. Either path records posted ids in `post-status.json`, so the two cannot double-post the same finding.
|
|
209
222
|
|
|
210
|
-
When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected. (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
|
|
223
|
+
When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected (a later `dismiss` therefore unselects). (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
|
|
211
224
|
|
|
212
225
|
```
|
|
213
226
|
magpie post "$RUN_DIR" --ids id1,id2,id3
|
|
@@ -232,7 +245,7 @@ magpie render "$RUN_DIR" findings
|
|
|
232
245
|
magpie cleanup "$RUN_DIR" --repo "$REPO"
|
|
233
246
|
```
|
|
234
247
|
|
|
235
|
-
The run directory is renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
|
|
248
|
+
Cleanup first writes `labels.json` (each final finding posted, dismissed or ignored, with the post route or dismiss reason; `magpie labels "$RUN_DIR"` writes it on demand). The run directory is then renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
|
|
236
249
|
|
|
237
250
|
The archived `findings.html` is self-contained and auto-switches to read-only "archived" mode when opened, so:
|
|
238
251
|
|
|
@@ -10,10 +10,18 @@ Subcommands:
|
|
|
10
10
|
setup <run-dir> --pr <n> Pre-flight, fetch PR, create worktree
|
|
11
11
|
serve <run-dir-or-id> Start the HTML server (accepts active or archived run id)
|
|
12
12
|
dedupe <run-dir> Merge specialist findings into deduped set
|
|
13
|
+
critic-prompt <run-dir> [--batch-size N]
|
|
14
|
+
Write one critic prompt per batch, print prompt and output paths
|
|
15
|
+
critic-apply <run-dir> [--design-cap N]
|
|
16
|
+
Apply critic verdicts, write findings.kept.json (no cap on
|
|
17
|
+
code-smells + architecture keeps unless --design-cap is given)
|
|
13
18
|
shard <run-dir> [--budget N] [--max-files N]
|
|
14
19
|
Re-split diff.patch into budgeted shards
|
|
15
|
-
render <run-dir> <page>
|
|
16
|
-
|
|
20
|
+
render <run-dir> <page> [--top N]
|
|
21
|
+
Render progress.html or findings.html (findings recommends
|
|
22
|
+
the top N, default 10, and folds the rest)
|
|
23
|
+
labels <run-dir> Write labels.json (posted, dismissed, ignored per finding)
|
|
24
|
+
cleanup <run-dir> Remove worktree, stop server, write labels, archive run
|
|
17
25
|
status <run-dir> Print highest completed stage
|
|
18
26
|
open [id] Open findings.html in your browser (defaults to latest run)
|
|
19
27
|
post <run-dir> --ids a,b Post the given finding ids via gh (rich body + optional summary)
|
|
@@ -92,9 +100,16 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
92
100
|
}
|
|
93
101
|
}
|
|
94
102
|
// Auto-refresh: re-render findings.html with the currently-shipped CSS/JS
|
|
95
|
-
// so serving an old archive picks up new report features.
|
|
103
|
+
// so serving an old archive picks up new report features. A failed refresh
|
|
104
|
+
// leaves the previous page in place, so serve keeps going after reporting why.
|
|
96
105
|
const { refreshFindings } = await import('../scripts/refresh.ts')
|
|
97
|
-
|
|
106
|
+
try {
|
|
107
|
+
await refreshFindings(runDir)
|
|
108
|
+
} catch (err) {
|
|
109
|
+
process.stderr.write(
|
|
110
|
+
`serve: refresh failed, serving the previous page: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
111
|
+
)
|
|
112
|
+
}
|
|
98
113
|
const { runServe } = await import('../scripts/serve-cmd.ts')
|
|
99
114
|
return runServe({ runDir, idleMs, host })
|
|
100
115
|
},
|
|
@@ -118,6 +133,48 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
118
133
|
const { runDedupe } = await import('../scripts/dedupe-cmd.ts')
|
|
119
134
|
return runDedupe(runDir, threshold !== undefined ? { threshold } : {})
|
|
120
135
|
},
|
|
136
|
+
'critic-prompt': async (args) => {
|
|
137
|
+
const runDir = args[0]
|
|
138
|
+
if (!runDir) {
|
|
139
|
+
process.stderr.write('critic-prompt: missing <run-dir> [--batch-size <n>]\n')
|
|
140
|
+
return 2
|
|
141
|
+
}
|
|
142
|
+
const flag = args.indexOf('--batch-size')
|
|
143
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
144
|
+
let batchSize: number | undefined
|
|
145
|
+
if (flag !== -1) {
|
|
146
|
+
const n = Number(raw)
|
|
147
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
148
|
+
process.stderr.write(
|
|
149
|
+
`critic-prompt: invalid --batch-size ${raw} (want a positive integer)\n`,
|
|
150
|
+
)
|
|
151
|
+
return 2
|
|
152
|
+
}
|
|
153
|
+
batchSize = n
|
|
154
|
+
}
|
|
155
|
+
const { runCriticPrompt } = await import('../scripts/critic-cmd.ts')
|
|
156
|
+
return runCriticPrompt(runDir, batchSize !== undefined ? { batchSize } : {})
|
|
157
|
+
},
|
|
158
|
+
'critic-apply': async (args) => {
|
|
159
|
+
const runDir = args[0]
|
|
160
|
+
if (!runDir) {
|
|
161
|
+
process.stderr.write('critic-apply: missing <run-dir> [--design-cap <n>]\n')
|
|
162
|
+
return 2
|
|
163
|
+
}
|
|
164
|
+
const flag = args.indexOf('--design-cap')
|
|
165
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
166
|
+
let designCap: number | undefined
|
|
167
|
+
if (flag !== -1) {
|
|
168
|
+
const n = Number(raw)
|
|
169
|
+
if (!Number.isInteger(n) || n < 0) {
|
|
170
|
+
process.stderr.write(`critic-apply: invalid --design-cap ${raw} (want an integer >= 0)\n`)
|
|
171
|
+
return 2
|
|
172
|
+
}
|
|
173
|
+
designCap = n
|
|
174
|
+
}
|
|
175
|
+
const { runCriticApply } = await import('../scripts/critic-cmd.ts')
|
|
176
|
+
return runCriticApply(runDir, designCap !== undefined ? { designCap } : {})
|
|
177
|
+
},
|
|
121
178
|
shard: async (args) => {
|
|
122
179
|
const runDir = args[0]
|
|
123
180
|
if (!runDir) {
|
|
@@ -163,11 +220,31 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
163
220
|
const runDir = args[0]
|
|
164
221
|
const page = args[1]
|
|
165
222
|
if (!runDir || (page !== 'progress' && page !== 'findings')) {
|
|
166
|
-
process.stderr.write('render: missing <run-dir> <progress|findings
|
|
223
|
+
process.stderr.write('render: missing <run-dir> <progress|findings> [--top <n>]\n')
|
|
167
224
|
return 2
|
|
168
225
|
}
|
|
226
|
+
const flag = args.indexOf('--top')
|
|
227
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
228
|
+
let topN: number | undefined
|
|
229
|
+
if (flag !== -1) {
|
|
230
|
+
const n = Number(raw)
|
|
231
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
232
|
+
process.stderr.write(`render: invalid --top ${raw} (want a positive integer)\n`)
|
|
233
|
+
return 1
|
|
234
|
+
}
|
|
235
|
+
topN = n
|
|
236
|
+
}
|
|
169
237
|
const { runRender } = await import('../scripts/render-cmd.ts')
|
|
170
|
-
return runRender(runDir, page)
|
|
238
|
+
return runRender(runDir, page, topN !== undefined ? { topN } : {})
|
|
239
|
+
},
|
|
240
|
+
labels: async (args) => {
|
|
241
|
+
const runDir = args[0]
|
|
242
|
+
if (!runDir) {
|
|
243
|
+
process.stderr.write('labels: missing <run-dir>\n')
|
|
244
|
+
return 2
|
|
245
|
+
}
|
|
246
|
+
const { runLabels } = await import('../scripts/labels-cmd.ts')
|
|
247
|
+
return runLabels(runDir)
|
|
171
248
|
},
|
|
172
249
|
cleanup: async (args) => {
|
|
173
250
|
const runDir = args[0]
|
|
@@ -248,6 +325,7 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
248
325
|
runDir,
|
|
249
326
|
findingIds,
|
|
250
327
|
dryRun,
|
|
328
|
+
via: 'cli',
|
|
251
329
|
...(includeSummary ? { includeSummary } : {}),
|
|
252
330
|
})
|
|
253
331
|
process.stdout.write(`${JSON.stringify(outcome)}\n`)
|
|
@@ -80,6 +80,21 @@ file is what most of the graders read: "posted exactly these ids", "never ran
|
|
|
80
80
|
invoked, and the call log answers them without depending on how the reply is
|
|
81
81
|
worded.
|
|
82
82
|
|
|
83
|
+
Cases that start past stage 6 hand the agent what the subagent critic would
|
|
84
|
+
have left. `codex-missing-falls-back` writes the whole chain: specialist files
|
|
85
|
+
with an `evidence` snippet per finding and no `severity`, `findings.deduped.json`
|
|
86
|
+
with severity derived from `risk.impact`, `threshold-dropped.json`,
|
|
87
|
+
`merge-candidates.json`, `critic.json` with one verdict per candidate,
|
|
88
|
+
`critic-dropped.json`, and the `findings.kept.json` that `magpie critic-apply`
|
|
89
|
+
produces from them. The log carries the critic `done` entry with the counts
|
|
90
|
+
critic-apply writes. Running the real `magpie dedupe` and `critic-apply` over
|
|
91
|
+
that run directory gives the same `merge-candidates.json`,
|
|
92
|
+
`threshold-dropped.json` and `critic-dropped.json`, and the same kept ids and
|
|
93
|
+
log counts. `findings.deduped.json` and `findings.kept.json` differ only in
|
|
94
|
+
`onChangedLine`, which the fixture leaves out, and in order; no grader reads
|
|
95
|
+
either. Repeat that check after changing either command. The report and post cases start from
|
|
96
|
+
`findings.final.json` and only need the same log entry and the evidence field.
|
|
97
|
+
|
|
83
98
|
`fixture.sh` is duplicated across the cases rather than shared, because
|
|
84
99
|
`context.scaffold_script` reads only from the case's own directory.
|
|
85
100
|
|
|
@@ -111,7 +111,8 @@ cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
|
111
111
|
{"stage":"context","status":"done"}
|
|
112
112
|
{"stage":"specialists","status":"done"}
|
|
113
113
|
{"stage":"dedupe","status":"done"}
|
|
114
|
-
{"stage":"critic","status":"
|
|
114
|
+
{"stage":"critic","status":"running"}
|
|
115
|
+
{"stage":"critic","status":"done","kept":3,"dropped":1,"merged":0,"capped":0}
|
|
115
116
|
LOG
|
|
116
117
|
|
|
117
118
|
# The PR under review, as setup would have left it: the filtered diff, and a
|
|
@@ -187,9 +188,10 @@ export async function load(tenantId: string) {
|
|
|
187
188
|
TS
|
|
188
189
|
|
|
189
190
|
# The findings the run already has: one file per specialist focus, the deduped
|
|
190
|
-
# set derived from them, and the subset
|
|
191
|
-
# the chain holds: nothing is kept that was never deduped,
|
|
192
|
-
# cites a line its hunk carries
|
|
191
|
+
# set derived from them, the critic's verdicts, and the subset critic-apply kept.
|
|
192
|
+
# Generated together so the chain holds: nothing is kept that was never deduped,
|
|
193
|
+
# every finding cites a line its hunk carries, and every evidence snippet is the
|
|
194
|
+
# worktree line it points at.
|
|
193
195
|
python3 - "$RUN_DIR" <<'FINDINGS'
|
|
194
196
|
import json, pathlib, sys
|
|
195
197
|
|
|
@@ -202,9 +204,9 @@ FINDINGS = [
|
|
|
202
204
|
'domain': 'security',
|
|
203
205
|
'file': 'src/settings/cache.ts',
|
|
204
206
|
'line': 4,
|
|
205
|
-
'
|
|
207
|
+
'evidence': 'store.set(tenantId, settings)',
|
|
206
208
|
'risk': {'impact': 'high', 'likelihood': 'likely', 'confidence': 'high', 'action': 'must-fix'},
|
|
207
|
-
'score': 8,
|
|
209
|
+
'score': 8.8,
|
|
208
210
|
'title': 'Tenant settings cache is a process-global Map with no eviction',
|
|
209
211
|
'description': """Observation: put() writes into a module-level Map keyed by tenant id (src/settings/cache.ts:4), with no size bound and no TTL.
|
|
210
212
|
|
|
@@ -218,9 +220,9 @@ Suggested direction: bound the map and give entries a TTL, or key the cache per
|
|
|
218
220
|
'domain': 'bugs',
|
|
219
221
|
'file': 'src/settings/loader.ts',
|
|
220
222
|
'line': 12,
|
|
221
|
-
'
|
|
223
|
+
'evidence': 'const fresh = await fetchSettings(tenantId)',
|
|
222
224
|
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'should-fix'},
|
|
223
|
-
'score':
|
|
225
|
+
'score': 5.1,
|
|
224
226
|
'title': 'Concurrent loads for the same tenant each hit the network',
|
|
225
227
|
'description': """Observation: load() checks the cache, then awaits fetchSettings before writing back (src/settings/loader.ts:12).
|
|
226
228
|
|
|
@@ -234,9 +236,9 @@ Suggested direction: cache the in-flight promise rather than the resolved value.
|
|
|
234
236
|
'domain': 'architecture',
|
|
235
237
|
'file': 'src/settings/loader.ts',
|
|
236
238
|
'line': 10,
|
|
237
|
-
'
|
|
239
|
+
'evidence': 'const hit = get(tenantId)',
|
|
238
240
|
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
239
|
-
'score':
|
|
241
|
+
'score': 4.7,
|
|
240
242
|
'title': 'The loader owns the cache rather than being handed one',
|
|
241
243
|
'description': """Observation: load() calls the cache module's free functions directly (src/settings/loader.ts:10).
|
|
242
244
|
|
|
@@ -250,9 +252,9 @@ Suggested direction: take the cache as a parameter.""",
|
|
|
250
252
|
'domain': 'performance',
|
|
251
253
|
'file': 'src/settings/cache.ts',
|
|
252
254
|
'line': 12,
|
|
253
|
-
'
|
|
255
|
+
'evidence': 'store.clear()',
|
|
254
256
|
'risk': {'impact': 'low', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
255
|
-
'score': 3,
|
|
257
|
+
'score': 3.5,
|
|
256
258
|
'title': 'clear() evicts every tenant, not the one whose settings changed',
|
|
257
259
|
'description': """Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.
|
|
258
260
|
|
|
@@ -266,9 +268,9 @@ Suggested direction: add delete(tenantId) and leave clear() for shutdown.""",
|
|
|
266
268
|
'domain': 'code-smells',
|
|
267
269
|
'file': 'src/settings/cache.ts',
|
|
268
270
|
'line': 8,
|
|
269
|
-
'
|
|
270
|
-
'risk': {'impact': 'low', 'likelihood': '
|
|
271
|
-
'score': 2,
|
|
271
|
+
'evidence': 'return store.get(tenantId)',
|
|
272
|
+
'risk': {'impact': 'low', 'likelihood': 'edge-case', 'confidence': 'medium', 'action': 'optional'},
|
|
273
|
+
'score': 2.3,
|
|
272
274
|
'title': 'get() hands back the stored object, so a caller can mutate the cache',
|
|
273
275
|
'description': """Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).
|
|
274
276
|
|
|
@@ -278,12 +280,22 @@ Suggested direction: freeze the value on put, or return a copy.""",
|
|
|
278
280
|
},
|
|
279
281
|
]
|
|
280
282
|
|
|
281
|
-
#
|
|
283
|
+
# smell-1 scored under the default threshold of 3, so dedupe set it aside and
|
|
284
|
+
# the critic never saw it. Of the four it did see, it kept three and dropped one.
|
|
285
|
+
BELOW_THRESHOLD = {'smell-1'}
|
|
282
286
|
KEPT = {'security-1', 'bugs-1', 'arch-1'}
|
|
283
287
|
|
|
288
|
+
# Severity is derived from risk.impact; specialists do not write it.
|
|
289
|
+
SEVERITY = {'critical': 'blocker', 'high': 'high', 'medium': 'medium', 'low': 'low'}
|
|
290
|
+
|
|
284
291
|
def without(finding, *keys):
|
|
285
292
|
return {k: v for k, v in finding.items() if k not in keys}
|
|
286
293
|
|
|
294
|
+
def derived(finding):
|
|
295
|
+
return {**finding, 'severity': SEVERITY[finding['risk']['impact']]}
|
|
296
|
+
|
|
297
|
+
deduped = [derived(without(f, 'focus')) for f in FINDINGS if f['id'] not in BELOW_THRESHOLD]
|
|
298
|
+
|
|
287
299
|
findings_dir = run / 'findings'
|
|
288
300
|
findings_dir.mkdir(parents=True, exist_ok=True)
|
|
289
301
|
for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
|
|
@@ -291,11 +303,38 @@ for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
|
|
|
291
303
|
(findings_dir / f'{focus}.json').write_text(json.dumps(mine, indent=2) + '\n')
|
|
292
304
|
(findings_dir / 'tests.json').write_text('[]\n')
|
|
293
305
|
|
|
294
|
-
(run / 'findings.deduped.json').write_text(
|
|
295
|
-
|
|
306
|
+
(run / 'findings.deduped.json').write_text(json.dumps(deduped, indent=2) + '\n')
|
|
307
|
+
(run / 'threshold-dropped.json').write_text(
|
|
308
|
+
json.dumps(
|
|
309
|
+
[{'id': f['id'], 'score': f['score'], 'title': f['title']} for f in FINDINGS if f['id'] in BELOW_THRESHOLD],
|
|
310
|
+
indent=2,
|
|
311
|
+
) + '\n'
|
|
312
|
+
)
|
|
313
|
+
# Same file within 8 lines, more than one domain: what dedupe groups for the critic.
|
|
314
|
+
(run / 'merge-candidates.json').write_text(
|
|
315
|
+
json.dumps([['security-1', 'perf-1'], ['arch-1', 'bugs-1']], indent=2) + '\n'
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
CHECKED = {
|
|
319
|
+
'security-1': ['src/settings/cache.ts:1-9', 'src/settings/loader.ts:9-15'],
|
|
320
|
+
'bugs-1': ['src/settings/loader.ts:9-15'],
|
|
321
|
+
'arch-1': ['src/settings/loader.ts:1-15'],
|
|
322
|
+
'perf-1': ['src/settings/cache.ts:11-13'],
|
|
323
|
+
}
|
|
324
|
+
verdicts = []
|
|
325
|
+
for f in deduped:
|
|
326
|
+
if f['id'] in KEPT:
|
|
327
|
+
verdicts.append({'id': f['id'], 'verdict': 'keep', 'reason': 'confirmed in the worktree',
|
|
328
|
+
'risk': f['risk'], 'checked': CHECKED[f['id']]})
|
|
329
|
+
else:
|
|
330
|
+
verdicts.append({'id': f['id'], 'verdict': 'drop', 'reason': 'one flush per settings change is not a measurable cost',
|
|
331
|
+
'checked': CHECKED[f['id']]})
|
|
332
|
+
(run / 'critic.json').write_text(json.dumps(verdicts, indent=2) + '\n')
|
|
333
|
+
(run / 'critic-dropped.json').write_text(
|
|
334
|
+
json.dumps([{'id': v['id'], 'reason': v['reason']} for v in verdicts if v['verdict'] == 'drop'], indent=2) + '\n'
|
|
296
335
|
)
|
|
297
336
|
(run / 'findings.kept.json').write_text(
|
|
298
|
-
json.dumps([
|
|
337
|
+
json.dumps([f for f in deduped if f['id'] in KEPT], indent=2) + '\n'
|
|
299
338
|
)
|
|
300
339
|
FINDINGS
|
|
301
340
|
|
|
@@ -303,8 +342,10 @@ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
|
|
|
303
342
|
|
|
304
343
|
cat > "$RUN_DIR/brief.json" <<'JSON'
|
|
305
344
|
{
|
|
306
|
-
"
|
|
307
|
-
"
|
|
308
|
-
"
|
|
345
|
+
"purpose": "Adds a process-global cache in front of tenant settings loads.",
|
|
346
|
+
"changes": ["cache module gains put and get", "loader reads through the cache"],
|
|
347
|
+
"watchItems": [],
|
|
348
|
+
"unclear": ["how a settings change is meant to invalidate the cache"],
|
|
349
|
+
"reviewRules": []
|
|
309
350
|
}
|
|
310
351
|
JSON
|
|
@@ -142,7 +142,8 @@ cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
|
142
142
|
{"stage":"context","status":"done"}
|
|
143
143
|
{"stage":"specialists","status":"done"}
|
|
144
144
|
{"stage":"dedupe","status":"done"}
|
|
145
|
-
{"stage":"critic","status":"
|
|
145
|
+
{"stage":"critic","status":"running"}
|
|
146
|
+
{"stage":"critic","status":"done","kept":5,"dropped":0,"merged":0,"capped":0}
|
|
146
147
|
{"stage":"peer-review","status":"done","provider":"claude"}
|
|
147
148
|
{"stage":"report","status":"done"}
|
|
148
149
|
LOG
|
|
@@ -225,6 +226,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
225
226
|
"id": "security-1",
|
|
226
227
|
"file": "src/settings/cache.ts",
|
|
227
228
|
"line": 4,
|
|
229
|
+
"evidence": "store.set(tenantId, settings)",
|
|
228
230
|
"severity": "high",
|
|
229
231
|
"risk": { "impact": "high", "likelihood": "likely", "confidence": "high", "action": "must-fix" },
|
|
230
232
|
"domain": "security",
|
|
@@ -235,6 +237,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
235
237
|
"id": "bugs-1",
|
|
236
238
|
"file": "src/settings/loader.ts",
|
|
237
239
|
"line": 12,
|
|
240
|
+
"evidence": "const fresh = await fetchSettings(tenantId)",
|
|
238
241
|
"severity": "medium",
|
|
239
242
|
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "should-fix" },
|
|
240
243
|
"domain": "bugs",
|
|
@@ -245,6 +248,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
245
248
|
"id": "perf-1",
|
|
246
249
|
"file": "src/settings/cache.ts",
|
|
247
250
|
"line": 12,
|
|
251
|
+
"evidence": "store.clear()",
|
|
248
252
|
"severity": "low",
|
|
249
253
|
"risk": { "impact": "low", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
250
254
|
"domain": "performance",
|
|
@@ -255,6 +259,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
255
259
|
"id": "arch-1",
|
|
256
260
|
"file": "src/settings/loader.ts",
|
|
257
261
|
"line": 10,
|
|
262
|
+
"evidence": "const hit = get(tenantId)",
|
|
258
263
|
"severity": "medium",
|
|
259
264
|
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
260
265
|
"domain": "architecture",
|
|
@@ -265,8 +270,9 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
265
270
|
"id": "smell-1",
|
|
266
271
|
"file": "src/settings/cache.ts",
|
|
267
272
|
"line": 8,
|
|
273
|
+
"evidence": "return store.get(tenantId)",
|
|
268
274
|
"severity": "low",
|
|
269
|
-
"risk": { "impact": "low", "likelihood": "
|
|
275
|
+
"risk": { "impact": "low", "likelihood": "edge-case", "confidence": "medium", "action": "optional" },
|
|
270
276
|
"domain": "code-smells",
|
|
271
277
|
"title": "get() hands back the stored object, so a caller can mutate the cache",
|
|
272
278
|
"description": "Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).\n\nWhy it matters: a caller that edits the returned settings edits every later reader's copy.\n\nSuggested direction: freeze the value on put, or return a copy."
|