@iceinvein/agent-skills 0.21.0 → 0.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/index.json +2 -2
- package/skills/magpie/README.md +3 -4
- package/skills/magpie/SKILL.md +31 -29
- package/skills/magpie/bin/magpie.ts +84 -6
- package/skills/magpie/evals/README.md +19 -5
- package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +64 -23
- package/skills/magpie/evals/post-folds-selection-events/fixture.sh +9 -3
- package/skills/magpie/evals/quality/README.md +141 -0
- package/skills/magpie/evals/quality/__tests__/replay-critic.test.ts +31 -0
- package/skills/magpie/evals/quality/__tests__/score.test.ts +325 -0
- package/skills/magpie/evals/quality/baseline.ts +80 -0
- package/skills/magpie/evals/quality/corpus.ts +259 -0
- package/skills/magpie/evals/quality/replay-critic.ts +277 -0
- package/skills/magpie/evals/quality/replay-full.ts +188 -0
- package/skills/magpie/evals/quality/score.ts +251 -0
- package/skills/magpie/evals/report-ends-the-turn/fixture.sh +6 -2
- package/skills/magpie/evals/resume-finds-active-run/fixture.sh +1 -1
- package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +1 -1
- package/skills/magpie/fixtures/example-pr/brief.json +0 -4
- package/skills/magpie/references/critic.md +178 -40
- package/skills/magpie/references/peer-review.md +5 -4
- package/skills/magpie/references/scout.md +17 -42
- package/skills/magpie/references/specialists.md +43 -136
- package/skills/magpie/scripts/__tests__/cleanup-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/critic-cmd.test.ts +260 -0
- package/skills/magpie/scripts/__tests__/critic.test.ts +328 -0
- package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +89 -0
- package/skills/magpie/scripts/__tests__/dedupe.test.ts +43 -1
- package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +155 -2
- package/skills/magpie/scripts/__tests__/helper.test.ts +80 -0
- package/skills/magpie/scripts/__tests__/labels.test.ts +152 -0
- package/skills/magpie/scripts/__tests__/open-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/pipeline-e2e.test.ts +1 -0
- package/skills/magpie/scripts/__tests__/post-cmd.test.ts +47 -0
- package/skills/magpie/scripts/__tests__/refresh.test.ts +23 -1
- package/skills/magpie/scripts/__tests__/render-action-bar.test.ts +63 -5
- package/skills/magpie/scripts/__tests__/render-annotation.test.ts +38 -0
- package/skills/magpie/scripts/__tests__/render-cmd.test.ts +48 -1
- package/skills/magpie/scripts/__tests__/render-diff.test.ts +34 -0
- package/skills/magpie/scripts/__tests__/render-findings.test.ts +111 -30
- package/skills/magpie/scripts/__tests__/render-issues-list.test.ts +186 -1
- package/skills/magpie/scripts/__tests__/render-progress.test.ts +1 -1
- package/skills/magpie/scripts/__tests__/serve-cmd.test.ts +19 -0
- package/skills/magpie/scripts/__tests__/server.test.ts +76 -0
- package/skills/magpie/scripts/__tests__/skill-lint.test.ts +166 -179
- package/skills/magpie/scripts/__tests__/tests-check.test.ts +57 -0
- package/skills/magpie/scripts/__tests__/types.test.ts +90 -36
- package/skills/magpie/scripts/cleanup-cmd.ts +6 -0
- package/skills/magpie/scripts/critic-cmd.ts +183 -0
- package/skills/magpie/scripts/critic.ts +273 -0
- package/skills/magpie/scripts/dedupe-cmd.ts +14 -4
- package/skills/magpie/scripts/dedupe.ts +41 -0
- package/skills/magpie/scripts/evidence-filter.ts +81 -17
- package/skills/magpie/scripts/helper.js +87 -9
- package/skills/magpie/scripts/labels-cmd.ts +43 -0
- package/skills/magpie/scripts/labels.ts +99 -0
- package/skills/magpie/scripts/open-cmd.ts +10 -3
- package/skills/magpie/scripts/post-cmd.ts +26 -2
- package/skills/magpie/scripts/preview-cmd.ts +4 -0
- package/skills/magpie/scripts/refresh.ts +21 -17
- package/skills/magpie/scripts/render-action-bar.ts +21 -4
- package/skills/magpie/scripts/render-annotation.ts +36 -2
- package/skills/magpie/scripts/render-cmd.ts +21 -2
- package/skills/magpie/scripts/render-diff.ts +6 -2
- package/skills/magpie/scripts/render-findings.ts +21 -10
- package/skills/magpie/scripts/render-issues-list.ts +66 -16
- package/skills/magpie/scripts/render-progress.ts +1 -1
- package/skills/magpie/scripts/server.ts +8 -1
- package/skills/magpie/scripts/tests-check.ts +23 -2
- package/skills/magpie/scripts/types.ts +36 -46
- package/skills/magpie/skill.json +2 -2
- package/skills/magpie/templates/styles.css +83 -2
- package/skills/magpie/tsconfig.json +1 -1
- package/skills/magpie/evals/consent-required-never-approves/case.yaml +0 -4
- package/skills/magpie/evals/consent-required-never-approves/fixture.sh +0 -213
- package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +0 -6
- package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +0 -7
- package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +0 -5
- package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +0 -5
- package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +0 -10
- package/skills/magpie/evals/consent-required-never-approves/prompt.md +0 -11
package/package.json
CHANGED
package/skills/index.json
CHANGED
|
@@ -219,9 +219,9 @@
|
|
|
219
219
|
},
|
|
220
220
|
{
|
|
221
221
|
"name": "magpie",
|
|
222
|
-
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture),
|
|
222
|
+
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), drops findings whose quoted evidence is not in the code, dedupes, runs critic subagents that re-check each finding against the worktree and set its risk label, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting or dismissing findings to post via gh, recording the outcome per finding as labels. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
|
|
223
223
|
"type": "prompt",
|
|
224
|
-
"version": "0.
|
|
224
|
+
"version": "0.13.0"
|
|
225
225
|
},
|
|
226
226
|
{
|
|
227
227
|
"name": "migrate",
|
package/skills/magpie/README.md
CHANGED
|
@@ -4,7 +4,7 @@ Interactive Claude Code skill that runs a multi-stage PR review pipeline inside
|
|
|
4
4
|
|
|
5
5
|
## What it does
|
|
6
6
|
|
|
7
|
-
Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture),
|
|
7
|
+
Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture), each quoting an evidence snippet for every anchored finding. Dedupe drops findings whose snippet is not in the file and groups nearby findings from different domains as merge candidates. Critic subagents then re-check each surviving finding against the worktree, keep, drop or merge it, and set its risk label, from which the severity is derived. Peer review runs via `codex exec` (falling back to a Claude second-opinion subagent when codex is unavailable). The interactive HTML report puts the top recommended findings first, lets the user dismiss findings with a reason, and posts the ones they select via `gh`; cleanup records what was posted, dismissed or ignored in `labels.json`.
|
|
8
8
|
|
|
9
9
|
## Requirements
|
|
10
10
|
|
|
@@ -12,7 +12,6 @@ Given a GitHub PR number, dispatches five specialist subagents in parallel (secu
|
|
|
12
12
|
- `gh` on PATH, authenticated (`gh auth status`)
|
|
13
13
|
- `git` on PATH
|
|
14
14
|
- `codex` on PATH, authenticated (optional; if absent the peer-review stage falls back to a Claude second-opinion subagent)
|
|
15
|
-
- code intelligence, with a completed index for the repo under review (optional; if absent the specialists review from the diff and worktree alone). Either interface works: the `code-intel` CLI on PATH, which is preferred because it takes `--repo` per call and needs no session binding, or the code-intelligence MCP server
|
|
16
15
|
|
|
17
16
|
## Install
|
|
18
17
|
|
|
@@ -55,7 +54,7 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
|
|
|
55
54
|
## Layout
|
|
56
55
|
|
|
57
56
|
- `SKILL.md` is the agent-facing prompt: the stage walkthrough and nothing else. Installed by the agent-skills CLI.
|
|
58
|
-
- `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract), `specialists.md` (stage 4, the five focus blocks plus the shared output contract
|
|
57
|
+
- `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract, including the repository review rules), `specialists.md` (stage 4, the five focus blocks plus the shared output contract), `critic.md` (stage 6, the critic subagent prompt `magpie critic-prompt` fills), `peer-review.md` (stage 7, including the Claude-fallback preamble). They ship in the bundle and sit next to `SKILL.md` once installed.
|
|
59
58
|
- `skill.json` is the agent-skills manifest.
|
|
60
59
|
- `bin/magpie` is the CLI invoked by the agent during stages; symlinked onto PATH by `install.sh`.
|
|
61
60
|
- `scripts/` holds the implementation (server, dedupe, render, setup, cleanup, etc.).
|
|
@@ -67,4 +66,4 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
|
|
|
67
66
|
|
|
68
67
|
## Run directory layout
|
|
69
68
|
|
|
70
|
-
Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `findings.kept.json`, `findings.final.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
|
|
69
|
+
Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `merge-candidates.json`, `critic-prompt.md` and `critic.json` (numbered per batch when there is more than one), `findings.kept.json`, `critic-dropped.json`, `findings.final.json`, `labels.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
|
package/skills/magpie/SKILL.md
CHANGED
|
@@ -7,7 +7,7 @@ description: Use when the user asks to review a GitHub pull request (a PR number
|
|
|
7
7
|
|
|
8
8
|
## Prerequisites
|
|
9
9
|
|
|
10
|
-
The skill pre-flights `bun`, `gh`, `git` (required) and `codex` (optional). A missing required binary aborts the run with a single install hint line.
|
|
10
|
+
The skill pre-flights `bun`, `gh`, `git` (required) and `codex` (optional). A missing required binary aborts the run with a single install hint line. Without `codex` the run continues and peer review falls back to a Claude second-opinion subagent (setup prints a one-line notice and logs `{stage: preflight, status: done, missingOptional: ["codex"]}`).
|
|
11
11
|
|
|
12
12
|
## Stage walkthrough
|
|
13
13
|
|
|
@@ -83,19 +83,11 @@ magpie render "$RUN_DIR" progress
|
|
|
83
83
|
|
|
84
84
|
### 3. Context
|
|
85
85
|
|
|
86
|
-
Append `{stage: context, status: running}` to `$RUN_DIR/log.jsonl` and re-render progress. This stage
|
|
86
|
+
Append `{stage: context, status: running}` to `$RUN_DIR/log.jsonl` and re-render progress. This stage never aborts the run.
|
|
87
87
|
|
|
88
|
-
|
|
88
|
+
Read `references/scout.md` and dispatch one subagent (Agent tool, `general-purpose`) carrying the `magpie-scout` block with `<<RUN_DIR>>` and `<<PR_NUMBER>>` substituted. It writes `$RUN_DIR/brief.json`.
|
|
89
89
|
|
|
90
|
-
|
|
91
|
-
- **MCP**, when the CLI is absent but `mcp__code-intelligence__*` tools are in your tool list. Call `bind_workspace` with `$RUN_DIR/worktree` and apply the same rules, polling `get_index_stats` instead, to set `CODE_INTELLIGENCE=mcp`. **Never call `approve_indexing`.**
|
|
92
|
-
- `consent_required` on either interface means the base repo has never completed an index, and starting one is a full GPU pass the user did not ask for. That, no interface at all, or any other error that survives one retry, sets `CODE_INTELLIGENCE=unavailable`; print one line: "Code intelligence is unavailable (<reason>); specialists will review from the diff alone."
|
|
93
|
-
|
|
94
|
-
Never block the pipeline on a still-running index. The scout and specialist contracts both handle a tool that is not ready yet.
|
|
95
|
-
|
|
96
|
-
**Scout.** Read `references/scout.md` and dispatch one subagent (Agent tool, `general-purpose`) carrying the `magpie-scout` block with `<<RUN_DIR>>`, `<<PR_NUMBER>>`, and `<<CODE_INTELLIGENCE>>` substituted. It writes `$RUN_DIR/brief.json`.
|
|
97
|
-
|
|
98
|
-
Append `{stage: context, status: done, codeIntelligence: true|false, interface: "cli"|"mcp"|"none"}` and re-render progress. If the scout returned without writing `brief.json`, append `{stage: context, status: skipped, codeIntelligence: true|false, interface: ...}` instead and continue: the brief is optional everywhere it is read. Both entries carry the probe's result, which is known whatever the scout did, and `interface` is what stage 10 reads to decide whether there is a session to rebind.
|
|
90
|
+
Append `{stage: context, status: done}` and re-render progress. If the scout returned without writing `brief.json`, append `{stage: context, status: skipped}` instead and continue: the brief is optional everywhere it is read.
|
|
99
91
|
|
|
100
92
|
### 4. Specialists
|
|
101
93
|
|
|
@@ -148,8 +140,7 @@ gate is expected to have none. `magpie dedupe` re-checks this against the manife
|
|
|
148
140
|
names every missing pair on stdout, as a backstop rather than a substitute.
|
|
149
141
|
|
|
150
142
|
If every specialist fails (no findings files written), log
|
|
151
|
-
`{stage: specialists, status: error}
|
|
152
|
-
`CODE_INTELLIGENCE=mcp` (stage 10), and stop. Otherwise mark `{stage: specialists, status: done}`.
|
|
143
|
+
`{stage: specialists, status: error}` and stop. Otherwise mark `{stage: specialists, status: done}`.
|
|
153
144
|
|
|
154
145
|
### 5. Dedupe
|
|
155
146
|
|
|
@@ -157,7 +148,9 @@ If every specialist fails (no findings files written), log
|
|
|
157
148
|
magpie dedupe "$RUN_DIR" [--threshold <0-10>]
|
|
158
149
|
```
|
|
159
150
|
|
|
160
|
-
`magpie dedupe` also
|
|
151
|
+
`magpie dedupe` also checks evidence against the worktree, dropping an anchored finding whose file is missing (`hallucinated-file`), whose line is out of range (`invented-line`), that has no `evidence` snippet (`missing-evidence`), or whose snippet is not near its line (`evidence-not-found`); a snippet found exactly once elsewhere in the file re-anchors it. Both go to `$RUN_DIR/evidence-dropped.json` as `{dropped, reanchored}`, written only when non-empty, so no file is normal. The check is skipped when the worktree is gone (archived run replay).
|
|
152
|
+
|
|
153
|
+
It always writes `$RUN_DIR/merge-candidates.json`: ids from different domains anchored close together in one file, for the critic to merge or keep apart.
|
|
161
154
|
|
|
162
155
|
Each finding gets a derived 0-10 `score` from its risk fields; those below `--threshold` (default 3) are dropped before the critic LLM runs and recorded to `$RUN_DIR/threshold-dropped.json`. Pass `--threshold 0` to keep everything.
|
|
163
156
|
|
|
@@ -165,12 +158,21 @@ Re-render progress.
|
|
|
165
158
|
|
|
166
159
|
### 6. Critic
|
|
167
160
|
|
|
168
|
-
|
|
161
|
+
The critic runs as worktree-reading subagents. The CLI fills its prompt from `references/critic.md`; substitute nothing by hand. Append `{stage: critic, status: running}` to `$RUN_DIR/log.jsonl`, re-render progress, then:
|
|
162
|
+
|
|
163
|
+
```
|
|
164
|
+
magpie critic-prompt "$RUN_DIR"
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
It prints `<prompt path>\t<output path>` per batch of about 30 candidates (a merge group always stays in one batch). Dispatch one subagent (Agent tool, `general-purpose`) per line, all in one message, each one's entire task the verbatim contents of its prompt file. Each writes its output path and returns one summary line. No lines printed means no candidates: go straight to `critic-apply`.
|
|
168
|
+
|
|
169
|
+
Confirm every output path exists; re-dispatch any batch whose file is missing. On a resume, dispatch only the batches whose output file is missing. Apply the verdicts:
|
|
169
170
|
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
171
|
+
```
|
|
172
|
+
magpie critic-apply "$RUN_DIR"
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
It writes `findings.kept.json` and `critic-dropped.json` and logs the critic `done` entry itself; do not append another. On a non-zero exit, stderr names offending ids or a file: delete the output file of each batch whose ids or file it names, re-dispatch only those, and re-run `critic-apply`. If a batch fails twice after re-dispatch, stop and show the user stderr verbatim rather than looping. Re-render progress.
|
|
174
176
|
|
|
175
177
|
### 7. Peer review
|
|
176
178
|
|
|
@@ -178,7 +180,7 @@ Append `{stage: peer-review, status: running}` to `$RUN_DIR/log.jsonl` and re-re
|
|
|
178
180
|
|
|
179
181
|
Build the peer-review prompt first: read `references/peer-review.md`, take the `magpie-peer-review` block from it, and substitute the placeholders listed in that file's `## Substitute before use` preamble.
|
|
180
182
|
|
|
181
|
-
One batch carries up to 40 findings; above that, split them 30 at a time
|
|
183
|
+
One batch carries up to 40 findings; above that, split them 30 at a time.
|
|
182
184
|
Write each batch's prompt, its `<<KEPT_FINDINGS_COMPACT>>` narrowed to that batch, to
|
|
183
185
|
`$RUN_DIR/peer-prompt-<k>.md`, `<k>` counting from 1. **When there is a single batch, drop `-<k>` throughout** (`peer-prompt.md`,
|
|
184
186
|
`peer.out`), which is the common case. Keep the `add` id counter running across batches
|
|
@@ -198,7 +200,7 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
|
|
|
198
200
|
|
|
199
201
|
**Claude path (fallback).** When `codex` is unavailable or failed, get the second opinion from a Claude subagent instead, one per batch. Set `<<PEER_PROVIDER>>` to `claude`, then prepend the `magpie-peer-review-claude-preamble` block from `references/peer-review.md` to each batch's substituted prompt (the preamble forces genuine independence, since the reviewer shares a model family with the primary reviewers). Dispatch one subagent (Agent tool, `general-purpose`) per batch whose entire task is that combined prompt, and instruct it to return only the fenced `review-peer-review` JSON block. Write each output to `$RUN_DIR/peer-<k>.out`, extract each `review-peer-review` block, merge into `$RUN_DIR/peer.json` after the last batch as above, and append `{stage: peer-review, status: done, provider: claude}` (`provider: mixed` if codex handled some batches).
|
|
200
202
|
|
|
201
|
-
**Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
|
|
203
|
+
**Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). A peer `fields.severity` (or `finding.severity` on an `add`) is ignored: severity is derived from `risk.impact`. Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
|
|
202
204
|
|
|
203
205
|
### 8. Report
|
|
204
206
|
|
|
@@ -206,6 +208,8 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
|
|
|
206
208
|
magpie render "$RUN_DIR" findings
|
|
207
209
|
```
|
|
208
210
|
|
|
211
|
+
The report shows the top 10 recommended (`must-fix`/`should-fix`, highest score first) and folds the rest. `--top <n>` is not saved (the stage 9 re-render and `magpie serve`'s auto-refresh show 10 again), so use it only when nothing will re-render.
|
|
212
|
+
|
|
209
213
|
Append `{stage: report, status: done}` to `$RUN_DIR/log.jsonl` and re-render progress (the render CLI does not log this itself, and `magpie status` needs the `done` entry to resume past `report`).
|
|
210
214
|
|
|
211
215
|
Print to the terminal: "Findings ready at <url>. Tick the ones you want and click **Post Selected**, or reply `post` here and I'll post whatever you've ticked."
|
|
@@ -214,9 +218,9 @@ End the turn.
|
|
|
214
218
|
|
|
215
219
|
### 9. Post
|
|
216
220
|
|
|
217
|
-
Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes
|
|
221
|
+
Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes only the top N above the fold; **Select recommended** ticks the same set); the server posts that batch as one GitHub review with inline threads. **Dismiss** on a finding records a reason (`wrong`, `not-worth-it`, `duplicate`, `style`) and drops it from the recommended set. The agent posts only when the user types `post` (optionally `post 1,3,7` for indices), which takes the CLI path below: separate inline comments plus a top-level summary comment. Either path records posted ids in `post-status.json`, so the two cannot double-post the same finding.
|
|
218
222
|
|
|
219
|
-
When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected. (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
|
|
223
|
+
When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected (a later `dismiss` therefore unselects). (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
|
|
220
224
|
|
|
221
225
|
```
|
|
222
226
|
magpie post "$RUN_DIR" --ids id1,id2,id3
|
|
@@ -241,9 +245,7 @@ magpie render "$RUN_DIR" findings
|
|
|
241
245
|
magpie cleanup "$RUN_DIR" --repo "$REPO"
|
|
242
246
|
```
|
|
243
247
|
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
The run directory is renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
|
|
248
|
+
Cleanup first writes `labels.json` (each final finding posted, dismissed or ignored, with the post route or dismiss reason; `magpie labels "$RUN_DIR"` writes it on demand). The run directory is then renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
|
|
247
249
|
|
|
248
250
|
The archived `findings.html` is self-contained and auto-switches to read-only "archived" mode when opened, so:
|
|
249
251
|
|
|
@@ -262,7 +264,7 @@ magpie status "$RUN_DIR"
|
|
|
262
264
|
|
|
263
265
|
The JSON output tells you `lastCompleted` and `next`. Resume from `next`:
|
|
264
266
|
|
|
265
|
-
- `context
|
|
267
|
+
- `context`: if `$RUN_DIR/brief.json` exists, append `{stage: context, status: done}` and move on; otherwise re-run stage 3.
|
|
266
268
|
- Any other stage: run it as written in the walkthrough.
|
|
267
269
|
- If a specialist focus has no findings file but its sibling stages are done,
|
|
268
270
|
re-dispatch only that focus. On a sharded run the unit is the `(focus, shard)` pair:
|
|
@@ -275,4 +277,4 @@ The original server is gone. Restart it with `magpie serve "$RUN_DIR"` (step 2)
|
|
|
275
277
|
|
|
276
278
|
## Aborting
|
|
277
279
|
|
|
278
|
-
If the user types `abort` mid-run,
|
|
280
|
+
If the user types `abort` mid-run, run `magpie cleanup` and exit.
|
|
@@ -10,10 +10,18 @@ Subcommands:
|
|
|
10
10
|
setup <run-dir> --pr <n> Pre-flight, fetch PR, create worktree
|
|
11
11
|
serve <run-dir-or-id> Start the HTML server (accepts active or archived run id)
|
|
12
12
|
dedupe <run-dir> Merge specialist findings into deduped set
|
|
13
|
+
critic-prompt <run-dir> [--batch-size N]
|
|
14
|
+
Write one critic prompt per batch, print prompt and output paths
|
|
15
|
+
critic-apply <run-dir> [--design-cap N]
|
|
16
|
+
Apply critic verdicts, write findings.kept.json (no cap on
|
|
17
|
+
code-smells + architecture keeps unless --design-cap is given)
|
|
13
18
|
shard <run-dir> [--budget N] [--max-files N]
|
|
14
19
|
Re-split diff.patch into budgeted shards
|
|
15
|
-
render <run-dir> <page>
|
|
16
|
-
|
|
20
|
+
render <run-dir> <page> [--top N]
|
|
21
|
+
Render progress.html or findings.html (findings recommends
|
|
22
|
+
the top N, default 10, and folds the rest)
|
|
23
|
+
labels <run-dir> Write labels.json (posted, dismissed, ignored per finding)
|
|
24
|
+
cleanup <run-dir> Remove worktree, stop server, write labels, archive run
|
|
17
25
|
status <run-dir> Print highest completed stage
|
|
18
26
|
open [id] Open findings.html in your browser (defaults to latest run)
|
|
19
27
|
post <run-dir> --ids a,b Post the given finding ids via gh (rich body + optional summary)
|
|
@@ -92,9 +100,16 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
92
100
|
}
|
|
93
101
|
}
|
|
94
102
|
// Auto-refresh: re-render findings.html with the currently-shipped CSS/JS
|
|
95
|
-
// so serving an old archive picks up new report features.
|
|
103
|
+
// so serving an old archive picks up new report features. A failed refresh
|
|
104
|
+
// leaves the previous page in place, so serve keeps going after reporting why.
|
|
96
105
|
const { refreshFindings } = await import('../scripts/refresh.ts')
|
|
97
|
-
|
|
106
|
+
try {
|
|
107
|
+
await refreshFindings(runDir)
|
|
108
|
+
} catch (err) {
|
|
109
|
+
process.stderr.write(
|
|
110
|
+
`serve: refresh failed, serving the previous page: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
111
|
+
)
|
|
112
|
+
}
|
|
98
113
|
const { runServe } = await import('../scripts/serve-cmd.ts')
|
|
99
114
|
return runServe({ runDir, idleMs, host })
|
|
100
115
|
},
|
|
@@ -118,6 +133,48 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
118
133
|
const { runDedupe } = await import('../scripts/dedupe-cmd.ts')
|
|
119
134
|
return runDedupe(runDir, threshold !== undefined ? { threshold } : {})
|
|
120
135
|
},
|
|
136
|
+
'critic-prompt': async (args) => {
|
|
137
|
+
const runDir = args[0]
|
|
138
|
+
if (!runDir) {
|
|
139
|
+
process.stderr.write('critic-prompt: missing <run-dir> [--batch-size <n>]\n')
|
|
140
|
+
return 2
|
|
141
|
+
}
|
|
142
|
+
const flag = args.indexOf('--batch-size')
|
|
143
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
144
|
+
let batchSize: number | undefined
|
|
145
|
+
if (flag !== -1) {
|
|
146
|
+
const n = Number(raw)
|
|
147
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
148
|
+
process.stderr.write(
|
|
149
|
+
`critic-prompt: invalid --batch-size ${raw} (want a positive integer)\n`,
|
|
150
|
+
)
|
|
151
|
+
return 2
|
|
152
|
+
}
|
|
153
|
+
batchSize = n
|
|
154
|
+
}
|
|
155
|
+
const { runCriticPrompt } = await import('../scripts/critic-cmd.ts')
|
|
156
|
+
return runCriticPrompt(runDir, batchSize !== undefined ? { batchSize } : {})
|
|
157
|
+
},
|
|
158
|
+
'critic-apply': async (args) => {
|
|
159
|
+
const runDir = args[0]
|
|
160
|
+
if (!runDir) {
|
|
161
|
+
process.stderr.write('critic-apply: missing <run-dir> [--design-cap <n>]\n')
|
|
162
|
+
return 2
|
|
163
|
+
}
|
|
164
|
+
const flag = args.indexOf('--design-cap')
|
|
165
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
166
|
+
let designCap: number | undefined
|
|
167
|
+
if (flag !== -1) {
|
|
168
|
+
const n = Number(raw)
|
|
169
|
+
if (!Number.isInteger(n) || n < 0) {
|
|
170
|
+
process.stderr.write(`critic-apply: invalid --design-cap ${raw} (want an integer >= 0)\n`)
|
|
171
|
+
return 2
|
|
172
|
+
}
|
|
173
|
+
designCap = n
|
|
174
|
+
}
|
|
175
|
+
const { runCriticApply } = await import('../scripts/critic-cmd.ts')
|
|
176
|
+
return runCriticApply(runDir, designCap !== undefined ? { designCap } : {})
|
|
177
|
+
},
|
|
121
178
|
shard: async (args) => {
|
|
122
179
|
const runDir = args[0]
|
|
123
180
|
if (!runDir) {
|
|
@@ -163,11 +220,31 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
163
220
|
const runDir = args[0]
|
|
164
221
|
const page = args[1]
|
|
165
222
|
if (!runDir || (page !== 'progress' && page !== 'findings')) {
|
|
166
|
-
process.stderr.write('render: missing <run-dir> <progress|findings
|
|
223
|
+
process.stderr.write('render: missing <run-dir> <progress|findings> [--top <n>]\n')
|
|
167
224
|
return 2
|
|
168
225
|
}
|
|
226
|
+
const flag = args.indexOf('--top')
|
|
227
|
+
const raw = flag !== -1 ? args[flag + 1] : undefined
|
|
228
|
+
let topN: number | undefined
|
|
229
|
+
if (flag !== -1) {
|
|
230
|
+
const n = Number(raw)
|
|
231
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
232
|
+
process.stderr.write(`render: invalid --top ${raw} (want a positive integer)\n`)
|
|
233
|
+
return 1
|
|
234
|
+
}
|
|
235
|
+
topN = n
|
|
236
|
+
}
|
|
169
237
|
const { runRender } = await import('../scripts/render-cmd.ts')
|
|
170
|
-
return runRender(runDir, page)
|
|
238
|
+
return runRender(runDir, page, topN !== undefined ? { topN } : {})
|
|
239
|
+
},
|
|
240
|
+
labels: async (args) => {
|
|
241
|
+
const runDir = args[0]
|
|
242
|
+
if (!runDir) {
|
|
243
|
+
process.stderr.write('labels: missing <run-dir>\n')
|
|
244
|
+
return 2
|
|
245
|
+
}
|
|
246
|
+
const { runLabels } = await import('../scripts/labels-cmd.ts')
|
|
247
|
+
return runLabels(runDir)
|
|
171
248
|
},
|
|
172
249
|
cleanup: async (args) => {
|
|
173
250
|
const runDir = args[0]
|
|
@@ -248,6 +325,7 @@ const HANDLERS: Record<string, Handler> = {
|
|
|
248
325
|
runDir,
|
|
249
326
|
findingIds,
|
|
250
327
|
dryRun,
|
|
328
|
+
via: 'cli',
|
|
251
329
|
...(includeSummary ? { includeSummary } : {}),
|
|
252
330
|
})
|
|
253
331
|
process.stdout.write(`${JSON.stringify(outcome)}\n`)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# magpie evals
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Five cases for `claude plugin eval`. Every one starts mid-pipeline, because the
|
|
4
4
|
decisions this skill owns are the ones between the CLI calls: `scripts/__tests__/`
|
|
5
5
|
already pins what `magpie setup`, `dedupe`, `shard`, `post` and `status` compute.
|
|
6
6
|
What no unit test can reach is whether the agent stops where the walkthrough says
|
|
@@ -13,7 +13,6 @@ must not touch.
|
|
|
13
13
|
| `codex-missing-falls-back` | No codex on the machine at stage 7 | Claude path with the independence preamble, `provider: claude`, never `status: error` |
|
|
14
14
|
| `report-ends-the-turn` | Stage 8 reached | Renders, logs the stage done, hands back for selection, posts nothing |
|
|
15
15
|
| `post-folds-selection-events` | The user typed `post` after re-ticking | Folds `state/events` last-event-wins, posts `bugs-1,perf-1` only |
|
|
16
|
-
| `consent-required-never-approves` | The code-intel probe wants consent | Never runs `index approve`, prints the unavailable notice, closes the stage |
|
|
17
16
|
| `resume-finds-active-run` | A fresh review ask on a PR with a live run | Checks `--list-runs` first, never calls `setup`, surfaces the interrupted run |
|
|
18
17
|
|
|
19
18
|
## Running
|
|
@@ -26,8 +25,8 @@ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit
|
|
|
26
25
|
```
|
|
27
26
|
|
|
28
27
|
`--runs 1 --ablation none` is the cheap iteration loop; `-j 3` runs three cases
|
|
29
|
-
at once. Most of the cost sits in `codex-missing-falls-back
|
|
30
|
-
|
|
28
|
+
at once. Most of the cost sits in `codex-missing-falls-back`, which dispatches a real
|
|
29
|
+
subagent.
|
|
31
30
|
|
|
32
31
|
Pass `--model` to run the cases on a specific model, which is the point of the
|
|
33
32
|
suite when a new one lands:
|
|
@@ -52,7 +51,7 @@ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
|
|
|
52
51
|
## How the fixtures fake the pipeline
|
|
53
52
|
|
|
54
53
|
The eval child runs in a sandbox that refuses to execute anything outside it, so
|
|
55
|
-
the real `magpie`, `gh
|
|
54
|
+
the real `magpie`, `gh` and `codex` are all unreachable: a bare
|
|
56
55
|
`magpie setup` there dies with `Operation not permitted`, not with a diff. Each
|
|
57
56
|
`fixture.sh` therefore writes its own fakes into `$HOME/shims` and puts that
|
|
58
57
|
directory first on `PATH` via `$HOME/.zshenv`, which is the one startup file the
|
|
@@ -81,6 +80,21 @@ file is what most of the graders read: "posted exactly these ids", "never ran
|
|
|
81
80
|
invoked, and the call log answers them without depending on how the reply is
|
|
82
81
|
worded.
|
|
83
82
|
|
|
83
|
+
Cases that start past stage 6 hand the agent what the subagent critic would
|
|
84
|
+
have left. `codex-missing-falls-back` writes the whole chain: specialist files
|
|
85
|
+
with an `evidence` snippet per finding and no `severity`, `findings.deduped.json`
|
|
86
|
+
with severity derived from `risk.impact`, `threshold-dropped.json`,
|
|
87
|
+
`merge-candidates.json`, `critic.json` with one verdict per candidate,
|
|
88
|
+
`critic-dropped.json`, and the `findings.kept.json` that `magpie critic-apply`
|
|
89
|
+
produces from them. The log carries the critic `done` entry with the counts
|
|
90
|
+
critic-apply writes. Running the real `magpie dedupe` and `critic-apply` over
|
|
91
|
+
that run directory gives the same `merge-candidates.json`,
|
|
92
|
+
`threshold-dropped.json` and `critic-dropped.json`, and the same kept ids and
|
|
93
|
+
log counts. `findings.deduped.json` and `findings.kept.json` differ only in
|
|
94
|
+
`onChangedLine`, which the fixture leaves out, and in order; no grader reads
|
|
95
|
+
either. Repeat that check after changing either command. The report and post cases start from
|
|
96
|
+
`findings.final.json` and only need the same log entry and the evidence field.
|
|
97
|
+
|
|
84
98
|
`fixture.sh` is duplicated across the cases rather than shared, because
|
|
85
99
|
`context.scaffold_script` reads only from the case's own directory.
|
|
86
100
|
|
|
@@ -108,10 +108,11 @@ JSON
|
|
|
108
108
|
cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
109
109
|
{"stage":"preflight","status":"done","missingOptional":["codex"]}
|
|
110
110
|
{"stage":"setup","status":"done"}
|
|
111
|
-
{"stage":"context","status":"done"
|
|
111
|
+
{"stage":"context","status":"done"}
|
|
112
112
|
{"stage":"specialists","status":"done"}
|
|
113
113
|
{"stage":"dedupe","status":"done"}
|
|
114
|
-
{"stage":"critic","status":"
|
|
114
|
+
{"stage":"critic","status":"running"}
|
|
115
|
+
{"stage":"critic","status":"done","kept":3,"dropped":1,"merged":0,"capped":0}
|
|
115
116
|
LOG
|
|
116
117
|
|
|
117
118
|
# The PR under review, as setup would have left it: the filtered diff, and a
|
|
@@ -187,9 +188,10 @@ export async function load(tenantId: string) {
|
|
|
187
188
|
TS
|
|
188
189
|
|
|
189
190
|
# The findings the run already has: one file per specialist focus, the deduped
|
|
190
|
-
# set derived from them, and the subset
|
|
191
|
-
# the chain holds: nothing is kept that was never deduped,
|
|
192
|
-
# cites a line its hunk carries
|
|
191
|
+
# set derived from them, the critic's verdicts, and the subset critic-apply kept.
|
|
192
|
+
# Generated together so the chain holds: nothing is kept that was never deduped,
|
|
193
|
+
# every finding cites a line its hunk carries, and every evidence snippet is the
|
|
194
|
+
# worktree line it points at.
|
|
193
195
|
python3 - "$RUN_DIR" <<'FINDINGS'
|
|
194
196
|
import json, pathlib, sys
|
|
195
197
|
|
|
@@ -202,9 +204,9 @@ FINDINGS = [
|
|
|
202
204
|
'domain': 'security',
|
|
203
205
|
'file': 'src/settings/cache.ts',
|
|
204
206
|
'line': 4,
|
|
205
|
-
'
|
|
207
|
+
'evidence': 'store.set(tenantId, settings)',
|
|
206
208
|
'risk': {'impact': 'high', 'likelihood': 'likely', 'confidence': 'high', 'action': 'must-fix'},
|
|
207
|
-
'score': 8,
|
|
209
|
+
'score': 8.8,
|
|
208
210
|
'title': 'Tenant settings cache is a process-global Map with no eviction',
|
|
209
211
|
'description': """Observation: put() writes into a module-level Map keyed by tenant id (src/settings/cache.ts:4), with no size bound and no TTL.
|
|
210
212
|
|
|
@@ -218,9 +220,9 @@ Suggested direction: bound the map and give entries a TTL, or key the cache per
|
|
|
218
220
|
'domain': 'bugs',
|
|
219
221
|
'file': 'src/settings/loader.ts',
|
|
220
222
|
'line': 12,
|
|
221
|
-
'
|
|
223
|
+
'evidence': 'const fresh = await fetchSettings(tenantId)',
|
|
222
224
|
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'should-fix'},
|
|
223
|
-
'score':
|
|
225
|
+
'score': 5.1,
|
|
224
226
|
'title': 'Concurrent loads for the same tenant each hit the network',
|
|
225
227
|
'description': """Observation: load() checks the cache, then awaits fetchSettings before writing back (src/settings/loader.ts:12).
|
|
226
228
|
|
|
@@ -234,9 +236,9 @@ Suggested direction: cache the in-flight promise rather than the resolved value.
|
|
|
234
236
|
'domain': 'architecture',
|
|
235
237
|
'file': 'src/settings/loader.ts',
|
|
236
238
|
'line': 10,
|
|
237
|
-
'
|
|
239
|
+
'evidence': 'const hit = get(tenantId)',
|
|
238
240
|
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
239
|
-
'score':
|
|
241
|
+
'score': 4.7,
|
|
240
242
|
'title': 'The loader owns the cache rather than being handed one',
|
|
241
243
|
'description': """Observation: load() calls the cache module's free functions directly (src/settings/loader.ts:10).
|
|
242
244
|
|
|
@@ -250,9 +252,9 @@ Suggested direction: take the cache as a parameter.""",
|
|
|
250
252
|
'domain': 'performance',
|
|
251
253
|
'file': 'src/settings/cache.ts',
|
|
252
254
|
'line': 12,
|
|
253
|
-
'
|
|
255
|
+
'evidence': 'store.clear()',
|
|
254
256
|
'risk': {'impact': 'low', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
255
|
-
'score': 3,
|
|
257
|
+
'score': 3.5,
|
|
256
258
|
'title': 'clear() evicts every tenant, not the one whose settings changed',
|
|
257
259
|
'description': """Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.
|
|
258
260
|
|
|
@@ -266,9 +268,9 @@ Suggested direction: add delete(tenantId) and leave clear() for shutdown.""",
|
|
|
266
268
|
'domain': 'code-smells',
|
|
267
269
|
'file': 'src/settings/cache.ts',
|
|
268
270
|
'line': 8,
|
|
269
|
-
'
|
|
270
|
-
'risk': {'impact': 'low', 'likelihood': '
|
|
271
|
-
'score': 2,
|
|
271
|
+
'evidence': 'return store.get(tenantId)',
|
|
272
|
+
'risk': {'impact': 'low', 'likelihood': 'edge-case', 'confidence': 'medium', 'action': 'optional'},
|
|
273
|
+
'score': 2.3,
|
|
272
274
|
'title': 'get() hands back the stored object, so a caller can mutate the cache',
|
|
273
275
|
'description': """Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).
|
|
274
276
|
|
|
@@ -278,12 +280,22 @@ Suggested direction: freeze the value on put, or return a copy.""",
|
|
|
278
280
|
},
|
|
279
281
|
]
|
|
280
282
|
|
|
281
|
-
#
|
|
283
|
+
# smell-1 scored under the default threshold of 3, so dedupe set it aside and
|
|
284
|
+
# the critic never saw it. Of the four it did see, it kept three and dropped one.
|
|
285
|
+
BELOW_THRESHOLD = {'smell-1'}
|
|
282
286
|
KEPT = {'security-1', 'bugs-1', 'arch-1'}
|
|
283
287
|
|
|
288
|
+
# Severity is derived from risk.impact; specialists do not write it.
|
|
289
|
+
SEVERITY = {'critical': 'blocker', 'high': 'high', 'medium': 'medium', 'low': 'low'}
|
|
290
|
+
|
|
284
291
|
def without(finding, *keys):
|
|
285
292
|
return {k: v for k, v in finding.items() if k not in keys}
|
|
286
293
|
|
|
294
|
+
def derived(finding):
|
|
295
|
+
return {**finding, 'severity': SEVERITY[finding['risk']['impact']]}
|
|
296
|
+
|
|
297
|
+
deduped = [derived(without(f, 'focus')) for f in FINDINGS if f['id'] not in BELOW_THRESHOLD]
|
|
298
|
+
|
|
287
299
|
findings_dir = run / 'findings'
|
|
288
300
|
findings_dir.mkdir(parents=True, exist_ok=True)
|
|
289
301
|
for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
|
|
@@ -291,11 +303,38 @@ for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
|
|
|
291
303
|
(findings_dir / f'{focus}.json').write_text(json.dumps(mine, indent=2) + '\n')
|
|
292
304
|
(findings_dir / 'tests.json').write_text('[]\n')
|
|
293
305
|
|
|
294
|
-
(run / 'findings.deduped.json').write_text(
|
|
295
|
-
|
|
306
|
+
(run / 'findings.deduped.json').write_text(json.dumps(deduped, indent=2) + '\n')
|
|
307
|
+
(run / 'threshold-dropped.json').write_text(
|
|
308
|
+
json.dumps(
|
|
309
|
+
[{'id': f['id'], 'score': f['score'], 'title': f['title']} for f in FINDINGS if f['id'] in BELOW_THRESHOLD],
|
|
310
|
+
indent=2,
|
|
311
|
+
) + '\n'
|
|
312
|
+
)
|
|
313
|
+
# Same file within 8 lines, more than one domain: what dedupe groups for the critic.
|
|
314
|
+
(run / 'merge-candidates.json').write_text(
|
|
315
|
+
json.dumps([['security-1', 'perf-1'], ['arch-1', 'bugs-1']], indent=2) + '\n'
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
CHECKED = {
|
|
319
|
+
'security-1': ['src/settings/cache.ts:1-9', 'src/settings/loader.ts:9-15'],
|
|
320
|
+
'bugs-1': ['src/settings/loader.ts:9-15'],
|
|
321
|
+
'arch-1': ['src/settings/loader.ts:1-15'],
|
|
322
|
+
'perf-1': ['src/settings/cache.ts:11-13'],
|
|
323
|
+
}
|
|
324
|
+
verdicts = []
|
|
325
|
+
for f in deduped:
|
|
326
|
+
if f['id'] in KEPT:
|
|
327
|
+
verdicts.append({'id': f['id'], 'verdict': 'keep', 'reason': 'confirmed in the worktree',
|
|
328
|
+
'risk': f['risk'], 'checked': CHECKED[f['id']]})
|
|
329
|
+
else:
|
|
330
|
+
verdicts.append({'id': f['id'], 'verdict': 'drop', 'reason': 'one flush per settings change is not a measurable cost',
|
|
331
|
+
'checked': CHECKED[f['id']]})
|
|
332
|
+
(run / 'critic.json').write_text(json.dumps(verdicts, indent=2) + '\n')
|
|
333
|
+
(run / 'critic-dropped.json').write_text(
|
|
334
|
+
json.dumps([{'id': v['id'], 'reason': v['reason']} for v in verdicts if v['verdict'] == 'drop'], indent=2) + '\n'
|
|
296
335
|
)
|
|
297
336
|
(run / 'findings.kept.json').write_text(
|
|
298
|
-
json.dumps([
|
|
337
|
+
json.dumps([f for f in deduped if f['id'] in KEPT], indent=2) + '\n'
|
|
299
338
|
)
|
|
300
339
|
FINDINGS
|
|
301
340
|
|
|
@@ -303,8 +342,10 @@ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
|
|
|
303
342
|
|
|
304
343
|
cat > "$RUN_DIR/brief.json" <<'JSON'
|
|
305
344
|
{
|
|
306
|
-
"
|
|
307
|
-
"
|
|
308
|
-
"
|
|
345
|
+
"purpose": "Adds a process-global cache in front of tenant settings loads.",
|
|
346
|
+
"changes": ["cache module gains put and get", "loader reads through the cache"],
|
|
347
|
+
"watchItems": [],
|
|
348
|
+
"unclear": ["how a settings change is meant to invalidate the cache"],
|
|
349
|
+
"reviewRules": []
|
|
309
350
|
}
|
|
310
351
|
JSON
|
|
@@ -139,10 +139,11 @@ JSON
|
|
|
139
139
|
cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
140
140
|
{"stage":"preflight","status":"done","missingOptional":["codex"]}
|
|
141
141
|
{"stage":"setup","status":"done"}
|
|
142
|
-
{"stage":"context","status":"done"
|
|
142
|
+
{"stage":"context","status":"done"}
|
|
143
143
|
{"stage":"specialists","status":"done"}
|
|
144
144
|
{"stage":"dedupe","status":"done"}
|
|
145
|
-
{"stage":"critic","status":"
|
|
145
|
+
{"stage":"critic","status":"running"}
|
|
146
|
+
{"stage":"critic","status":"done","kept":5,"dropped":0,"merged":0,"capped":0}
|
|
146
147
|
{"stage":"peer-review","status":"done","provider":"claude"}
|
|
147
148
|
{"stage":"report","status":"done"}
|
|
148
149
|
LOG
|
|
@@ -225,6 +226,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
225
226
|
"id": "security-1",
|
|
226
227
|
"file": "src/settings/cache.ts",
|
|
227
228
|
"line": 4,
|
|
229
|
+
"evidence": "store.set(tenantId, settings)",
|
|
228
230
|
"severity": "high",
|
|
229
231
|
"risk": { "impact": "high", "likelihood": "likely", "confidence": "high", "action": "must-fix" },
|
|
230
232
|
"domain": "security",
|
|
@@ -235,6 +237,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
235
237
|
"id": "bugs-1",
|
|
236
238
|
"file": "src/settings/loader.ts",
|
|
237
239
|
"line": 12,
|
|
240
|
+
"evidence": "const fresh = await fetchSettings(tenantId)",
|
|
238
241
|
"severity": "medium",
|
|
239
242
|
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "should-fix" },
|
|
240
243
|
"domain": "bugs",
|
|
@@ -245,6 +248,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
245
248
|
"id": "perf-1",
|
|
246
249
|
"file": "src/settings/cache.ts",
|
|
247
250
|
"line": 12,
|
|
251
|
+
"evidence": "store.clear()",
|
|
248
252
|
"severity": "low",
|
|
249
253
|
"risk": { "impact": "low", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
250
254
|
"domain": "performance",
|
|
@@ -255,6 +259,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
255
259
|
"id": "arch-1",
|
|
256
260
|
"file": "src/settings/loader.ts",
|
|
257
261
|
"line": 10,
|
|
262
|
+
"evidence": "const hit = get(tenantId)",
|
|
258
263
|
"severity": "medium",
|
|
259
264
|
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
260
265
|
"domain": "architecture",
|
|
@@ -265,8 +270,9 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
|
265
270
|
"id": "smell-1",
|
|
266
271
|
"file": "src/settings/cache.ts",
|
|
267
272
|
"line": 8,
|
|
273
|
+
"evidence": "return store.get(tenantId)",
|
|
268
274
|
"severity": "low",
|
|
269
|
-
"risk": { "impact": "low", "likelihood": "
|
|
275
|
+
"risk": { "impact": "low", "likelihood": "edge-case", "confidence": "medium", "action": "optional" },
|
|
270
276
|
"domain": "code-smells",
|
|
271
277
|
"title": "get() hands back the stored object, so a caller can mutate the cache",
|
|
272
278
|
"description": "Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).\n\nWhy it matters: a caller that edits the returned settings edits every later reader's copy.\n\nSuggested direction: freeze the value on put, or return a copy."
|