@iceinvein/agent-skills 0.21.0 → 0.21.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/package.json +1 -1
  2. package/skills/index.json +2 -2
  3. package/skills/magpie/README.md +3 -4
  4. package/skills/magpie/SKILL.md +31 -29
  5. package/skills/magpie/bin/magpie.ts +84 -6
  6. package/skills/magpie/evals/README.md +19 -5
  7. package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +64 -23
  8. package/skills/magpie/evals/post-folds-selection-events/fixture.sh +9 -3
  9. package/skills/magpie/evals/quality/README.md +141 -0
  10. package/skills/magpie/evals/quality/__tests__/replay-critic.test.ts +31 -0
  11. package/skills/magpie/evals/quality/__tests__/score.test.ts +325 -0
  12. package/skills/magpie/evals/quality/baseline.ts +80 -0
  13. package/skills/magpie/evals/quality/corpus.ts +259 -0
  14. package/skills/magpie/evals/quality/replay-critic.ts +277 -0
  15. package/skills/magpie/evals/quality/replay-full.ts +188 -0
  16. package/skills/magpie/evals/quality/score.ts +251 -0
  17. package/skills/magpie/evals/report-ends-the-turn/fixture.sh +6 -2
  18. package/skills/magpie/evals/resume-finds-active-run/fixture.sh +1 -1
  19. package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +1 -1
  20. package/skills/magpie/fixtures/example-pr/brief.json +0 -4
  21. package/skills/magpie/references/critic.md +178 -40
  22. package/skills/magpie/references/peer-review.md +5 -4
  23. package/skills/magpie/references/scout.md +17 -42
  24. package/skills/magpie/references/specialists.md +43 -136
  25. package/skills/magpie/scripts/__tests__/cleanup-cmd.test.ts +28 -0
  26. package/skills/magpie/scripts/__tests__/critic-cmd.test.ts +260 -0
  27. package/skills/magpie/scripts/__tests__/critic.test.ts +328 -0
  28. package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +89 -0
  29. package/skills/magpie/scripts/__tests__/dedupe.test.ts +43 -1
  30. package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +155 -2
  31. package/skills/magpie/scripts/__tests__/helper.test.ts +80 -0
  32. package/skills/magpie/scripts/__tests__/labels.test.ts +152 -0
  33. package/skills/magpie/scripts/__tests__/open-cmd.test.ts +28 -0
  34. package/skills/magpie/scripts/__tests__/pipeline-e2e.test.ts +1 -0
  35. package/skills/magpie/scripts/__tests__/post-cmd.test.ts +47 -0
  36. package/skills/magpie/scripts/__tests__/refresh.test.ts +23 -1
  37. package/skills/magpie/scripts/__tests__/render-action-bar.test.ts +63 -5
  38. package/skills/magpie/scripts/__tests__/render-annotation.test.ts +38 -0
  39. package/skills/magpie/scripts/__tests__/render-cmd.test.ts +48 -1
  40. package/skills/magpie/scripts/__tests__/render-diff.test.ts +34 -0
  41. package/skills/magpie/scripts/__tests__/render-findings.test.ts +111 -30
  42. package/skills/magpie/scripts/__tests__/render-issues-list.test.ts +186 -1
  43. package/skills/magpie/scripts/__tests__/render-progress.test.ts +1 -1
  44. package/skills/magpie/scripts/__tests__/serve-cmd.test.ts +19 -0
  45. package/skills/magpie/scripts/__tests__/server.test.ts +76 -0
  46. package/skills/magpie/scripts/__tests__/skill-lint.test.ts +166 -179
  47. package/skills/magpie/scripts/__tests__/tests-check.test.ts +57 -0
  48. package/skills/magpie/scripts/__tests__/types.test.ts +90 -36
  49. package/skills/magpie/scripts/cleanup-cmd.ts +6 -0
  50. package/skills/magpie/scripts/critic-cmd.ts +183 -0
  51. package/skills/magpie/scripts/critic.ts +273 -0
  52. package/skills/magpie/scripts/dedupe-cmd.ts +14 -4
  53. package/skills/magpie/scripts/dedupe.ts +41 -0
  54. package/skills/magpie/scripts/evidence-filter.ts +81 -17
  55. package/skills/magpie/scripts/helper.js +87 -9
  56. package/skills/magpie/scripts/labels-cmd.ts +43 -0
  57. package/skills/magpie/scripts/labels.ts +99 -0
  58. package/skills/magpie/scripts/open-cmd.ts +10 -3
  59. package/skills/magpie/scripts/post-cmd.ts +26 -2
  60. package/skills/magpie/scripts/preview-cmd.ts +4 -0
  61. package/skills/magpie/scripts/refresh.ts +21 -17
  62. package/skills/magpie/scripts/render-action-bar.ts +21 -4
  63. package/skills/magpie/scripts/render-annotation.ts +36 -2
  64. package/skills/magpie/scripts/render-cmd.ts +21 -2
  65. package/skills/magpie/scripts/render-diff.ts +6 -2
  66. package/skills/magpie/scripts/render-findings.ts +21 -10
  67. package/skills/magpie/scripts/render-issues-list.ts +66 -16
  68. package/skills/magpie/scripts/render-progress.ts +1 -1
  69. package/skills/magpie/scripts/server.ts +8 -1
  70. package/skills/magpie/scripts/tests-check.ts +23 -2
  71. package/skills/magpie/scripts/types.ts +36 -46
  72. package/skills/magpie/skill.json +2 -2
  73. package/skills/magpie/templates/styles.css +83 -2
  74. package/skills/magpie/tsconfig.json +1 -1
  75. package/skills/magpie/evals/consent-required-never-approves/case.yaml +0 -4
  76. package/skills/magpie/evals/consent-required-never-approves/fixture.sh +0 -213
  77. package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +0 -6
  78. package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +0 -7
  79. package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +0 -5
  80. package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +0 -5
  81. package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +0 -10
  82. package/skills/magpie/evals/consent-required-never-approves/prompt.md +0 -11
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iceinvein/agent-skills",
3
- "version": "0.21.0",
3
+ "version": "0.21.2",
4
4
  "description": "Install agent skills into AI coding tools",
5
5
  "author": "iceinvein",
6
6
  "license": "MIT",
package/skills/index.json CHANGED
@@ -219,9 +219,9 @@
219
219
  },
220
220
  {
221
221
  "name": "magpie",
222
- "description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), dedupes findings, applies a critic rubric, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting findings to post via gh. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
222
+ "description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), drops findings whose quoted evidence is not in the code, dedupes, runs critic subagents that re-check each finding against the worktree and set its risk label, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting or dismissing findings to post via gh, recording the outcome per finding as labels. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
223
223
  "type": "prompt",
224
- "version": "0.11.1"
224
+ "version": "0.13.0"
225
225
  },
226
226
  {
227
227
  "name": "migrate",
@@ -4,7 +4,7 @@ Interactive Claude Code skill that runs a multi-stage PR review pipeline inside
4
4
 
5
5
  ## What it does
6
6
 
7
- Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture), dedupes their findings, applies a critic rubric, peer-reviews via `codex exec` (falling back to a Claude second-opinion subagent when codex is unavailable), serves an interactive HTML report, and posts the findings the user selects via `gh`.
7
+ Given a GitHub PR number, dispatches five specialist subagents in parallel (security, bugs, performance, code-smells, architecture), each quoting an evidence snippet for every anchored finding. Dedupe drops findings whose snippet is not in the file and groups nearby findings from different domains as merge candidates. Critic subagents then re-check each surviving finding against the worktree, keep, drop or merge it, and set its risk label, from which the severity is derived. Peer review runs via `codex exec` (falling back to a Claude second-opinion subagent when codex is unavailable). The interactive HTML report puts the top recommended findings first, lets the user dismiss findings with a reason, and posts the ones they select via `gh`; cleanup records what was posted, dismissed or ignored in `labels.json`.
8
8
 
9
9
  ## Requirements
10
10
 
@@ -12,7 +12,6 @@ Given a GitHub PR number, dispatches five specialist subagents in parallel (secu
12
12
  - `gh` on PATH, authenticated (`gh auth status`)
13
13
  - `git` on PATH
14
14
  - `codex` on PATH, authenticated (optional; if absent the peer-review stage falls back to a Claude second-opinion subagent)
15
- - code intelligence, with a completed index for the repo under review (optional; if absent the specialists review from the diff and worktree alone). Either interface works: the `code-intel` CLI on PATH, which is preferred because it takes `--repo` per call and needs no session binding, or the code-intelligence MCP server
16
15
 
17
16
  ## Install
18
17
 
@@ -55,7 +54,7 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
55
54
  ## Layout
56
55
 
57
56
  - `SKILL.md` is the agent-facing prompt: the stage walkthrough and nothing else. Installed by the agent-skills CLI.
58
- - `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract), `specialists.md` (stage 4, the five focus blocks plus the shared output contract and the codebase-intelligence block), `critic.md` (stage 6), `peer-review.md` (stage 7, including the Claude-fallback preamble). They ship in the bundle and sit next to `SKILL.md` once installed.
57
+ - `references/` holds the prompt bodies the walkthrough loads on demand, one file per stage that needs one: `scout.md` (stage 3, the PR-brief prompt and the `brief.json` contract, including the repository review rules), `specialists.md` (stage 4, the five focus blocks plus the shared output contract), `critic.md` (stage 6, the critic subagent prompt `magpie critic-prompt` fills), `peer-review.md` (stage 7, including the Claude-fallback preamble). They ship in the bundle and sit next to `SKILL.md` once installed.
59
58
  - `skill.json` is the agent-skills manifest.
60
59
  - `bin/magpie` is the CLI invoked by the agent during stages; symlinked onto PATH by `install.sh`.
61
60
  - `scripts/` holds the implementation (server, dedupe, render, setup, cleanup, etc.).
@@ -67,4 +66,4 @@ The fixture lives at `fixtures/example-pr/` (pr.json + findings.final.json + pos
67
66
 
68
67
  ## Run directory layout
69
68
 
70
- Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `findings.kept.json`, `findings.final.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
69
+ Each invocation creates `~/.magpie/pr-<n>-<ts>/` with `pr.json`, `diff.patch`, `findings/`, `findings.deduped.json`, `merge-candidates.json`, `critic-prompt.md` and `critic.json` (numbered per batch when there is more than one), `findings.kept.json`, `critic-dropped.json`, `findings.final.json`, `labels.json`, `screen/`, `state/`, `log.jsonl`. On completion the directory is renamed to `<run-dir>.archived-<timestamp>` rather than deleted, so logs survive for postmortem.
@@ -7,7 +7,7 @@ description: Use when the user asks to review a GitHub pull request (a PR number
7
7
 
8
8
  ## Prerequisites
9
9
 
10
- The skill pre-flights `bun`, `gh`, `git` (required) and `codex` (optional). A missing required binary aborts the run with a single install hint line. Code intelligence is not pre-flighted: stage 3 probes for it, over either the `code-intel` CLI or the code-intelligence MCP server, and the run continues without it. Without `codex` the run continues and peer review falls back to a Claude second-opinion subagent (setup prints a one-line notice and logs `{stage: preflight, status: done, missingOptional: ["codex"]}`).
10
+ The skill pre-flights `bun`, `gh`, `git` (required) and `codex` (optional). A missing required binary aborts the run with a single install hint line. Without `codex` the run continues and peer review falls back to a Claude second-opinion subagent (setup prints a one-line notice and logs `{stage: preflight, status: done, missingOptional: ["codex"]}`).
11
11
 
12
12
  ## Stage walkthrough
13
13
 
@@ -83,19 +83,11 @@ magpie render "$RUN_DIR" progress
83
83
 
84
84
  ### 3. Context
85
85
 
86
- Append `{stage: context, status: running}` to `$RUN_DIR/log.jsonl` and re-render progress. This stage has two steps and never aborts the run.
86
+ Append `{stage: context, status: running}` to `$RUN_DIR/log.jsonl` and re-render progress. This stage never aborts the run.
87
87
 
88
- **Probe.** Code intelligence reaches the same on-device daemon through two interfaces. Prefer the `code-intel` CLI: it takes `--repo` on every call, so it holds no session binding to leak past cleanup and the specialists can query in parallel without clobbering each other's workspace. `$RUN_DIR/worktree` is a linked git worktree either way, so an already-indexed base repo seeds its index instead of re-indexing.
88
+ Read `references/scout.md` and dispatch one subagent (Agent tool, `general-purpose`) carrying the `magpie-scout` block with `<<RUN_DIR>>` and `<<PR_NUMBER>>` substituted. It writes `$RUN_DIR/brief.json`.
89
89
 
90
- - **CLI**, when `command -v code-intel` succeeds. Run `code-intel index status --repo "$RUN_DIR/worktree" --json` and read `.status`: `ok` sets `CODE_INTELLIGENCE=cli`. `indexing_started` or `indexing_in_progress` means the seed took, so poll the same command every 5s for at most 60s and set `CODE_INTELLIGENCE=cli` either way. Exit 3 is a stopped daemon: run `code-intel start`, re-probe once. **Never run `code-intel index approve`.**
91
- - **MCP**, when the CLI is absent but `mcp__code-intelligence__*` tools are in your tool list. Call `bind_workspace` with `$RUN_DIR/worktree` and apply the same rules, polling `get_index_stats` instead, to set `CODE_INTELLIGENCE=mcp`. **Never call `approve_indexing`.**
92
- - `consent_required` on either interface means the base repo has never completed an index, and starting one is a full GPU pass the user did not ask for. That, no interface at all, or any other error that survives one retry, sets `CODE_INTELLIGENCE=unavailable`; print one line: "Code intelligence is unavailable (<reason>); specialists will review from the diff alone."
93
-
94
- Never block the pipeline on a still-running index. The scout and specialist contracts both handle a tool that is not ready yet.
95
-
96
- **Scout.** Read `references/scout.md` and dispatch one subagent (Agent tool, `general-purpose`) carrying the `magpie-scout` block with `<<RUN_DIR>>`, `<<PR_NUMBER>>`, and `<<CODE_INTELLIGENCE>>` substituted. It writes `$RUN_DIR/brief.json`.
97
-
98
- Append `{stage: context, status: done, codeIntelligence: true|false, interface: "cli"|"mcp"|"none"}` and re-render progress. If the scout returned without writing `brief.json`, append `{stage: context, status: skipped, codeIntelligence: true|false, interface: ...}` instead and continue: the brief is optional everywhere it is read. Both entries carry the probe's result, which is known whatever the scout did, and `interface` is what stage 10 reads to decide whether there is a session to rebind.
90
+ Append `{stage: context, status: done}` and re-render progress. If the scout returned without writing `brief.json`, append `{stage: context, status: skipped}` instead and continue: the brief is optional everywhere it is read.
99
91
 
100
92
  ### 4. Specialists
101
93
 
@@ -148,8 +140,7 @@ gate is expected to have none. `magpie dedupe` re-checks this against the manife
148
140
  names every missing pair on stdout, as a backstop rather than a substitute.
149
141
 
150
142
  If every specialist fails (no findings files written), log
151
- `{stage: specialists, status: error}`, rebind code intelligence to `$REPO` if
152
- `CODE_INTELLIGENCE=mcp` (stage 10), and stop. Otherwise mark `{stage: specialists, status: done}`.
143
+ `{stage: specialists, status: error}` and stop. Otherwise mark `{stage: specialists, status: done}`.
153
144
 
154
145
  ### 5. Dedupe
155
146
 
@@ -157,7 +148,9 @@ If every specialist fails (no findings files written), log
157
148
  magpie dedupe "$RUN_DIR" [--threshold <0-10>]
158
149
  ```
159
150
 
160
- `magpie dedupe` also runs a deterministic evidence check against the worktree: findings whose `file` is missing or whose `line` is out of range are dropped, logged, and recorded to `$RUN_DIR/evidence-dropped.json`. The check is skipped when the worktree is gone (archived run replay).
151
+ `magpie dedupe` also checks evidence against the worktree, dropping an anchored finding whose file is missing (`hallucinated-file`), whose line is out of range (`invented-line`), that has no `evidence` snippet (`missing-evidence`), or whose snippet is not near its line (`evidence-not-found`); a snippet found exactly once elsewhere in the file re-anchors it. Both go to `$RUN_DIR/evidence-dropped.json` as `{dropped, reanchored}`, written only when non-empty, so no file is normal. The check is skipped when the worktree is gone (archived run replay).
152
+
153
+ It always writes `$RUN_DIR/merge-candidates.json`: ids from different domains anchored close together in one file, for the critic to merge or keep apart.
161
154
 
162
155
  Each finding gets a derived 0-10 `score` from its risk fields; those below `--threshold` (default 3) are dropped before the critic LLM runs and recorded to `$RUN_DIR/threshold-dropped.json`. Pass `--threshold 0` to keep everything.
163
156
 
@@ -165,12 +158,21 @@ Re-render progress.
165
158
 
166
159
  ### 6. Critic
167
160
 
168
- Read `references/critic.md` and `$RUN_DIR/findings.deduped.json`. Substitute both placeholders in the critic rubric (the compact candidate list including each finding's `onChangedLine`, and the `<<DIFF_EXCERPT>>` hunks for the referenced files), then apply the rubric verbatim (one verdict per finding). Write the kept subset to `$RUN_DIR/findings.kept.json`. Append `{stage: critic, status: done}` and re-render progress.
161
+ The critic runs as worktree-reading subagents. The CLI fills its prompt from `references/critic.md`; substitute nothing by hand. Append `{stage: critic, status: running}` to `$RUN_DIR/log.jsonl`, re-render progress, then:
162
+
163
+ ```
164
+ magpie critic-prompt "$RUN_DIR"
165
+ ```
166
+
167
+ It prints `<prompt path>\t<output path>` per batch of about 30 candidates (a merge group always stays in one batch). Dispatch one subagent (Agent tool, `general-purpose`) per line, all in one message, each one's entire task the verbatim contents of its prompt file. Each writes its output path and returns one summary line. No lines printed means no candidates: go straight to `critic-apply`.
168
+
169
+ Confirm every output path exists; re-dispatch any batch whose file is missing. On a resume, dispatch only the batches whose output file is missing. Apply the verdicts:
169
170
 
170
- When `findings.deduped.json` holds more than 40 findings, run the rubric in batches of
171
- 30 rather than one prompt: a sharded run can produce more candidates than fit alongside
172
- their diff excerpts. Apply the same rubric verbatim per batch and concatenate the kept
173
- subsets into `findings.kept.json`.
171
+ ```
172
+ magpie critic-apply "$RUN_DIR"
173
+ ```
174
+
175
+ It writes `findings.kept.json` and `critic-dropped.json` and logs the critic `done` entry itself; do not append another. On a non-zero exit, stderr names offending ids or a file: delete the output file of each batch whose ids or file it names, re-dispatch only those, and re-run `critic-apply`. If a batch fails twice after re-dispatch, stop and show the user stderr verbatim rather than looping. Re-render progress.
174
176
 
175
177
  ### 7. Peer review
176
178
 
@@ -178,7 +180,7 @@ Append `{stage: peer-review, status: running}` to `$RUN_DIR/log.jsonl` and re-re
178
180
 
179
181
  Build the peer-review prompt first: read `references/peer-review.md`, take the `magpie-peer-review` block from it, and substitute the placeholders listed in that file's `## Substitute before use` preamble.
180
182
 
181
- One batch carries up to 40 findings; above that, split them 30 at a time, as in stage 6.
183
+ One batch carries up to 40 findings; above that, split them 30 at a time.
182
184
  Write each batch's prompt, its `<<KEPT_FINDINGS_COMPACT>>` narrowed to that batch, to
183
185
  `$RUN_DIR/peer-prompt-<k>.md`, `<k>` counting from 1. **When there is a single batch, drop `-<k>` throughout** (`peer-prompt.md`,
184
186
  `peer.out`), which is the common case. Keep the `add` id counter running across batches
@@ -198,7 +200,7 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
198
200
 
199
201
  **Claude path (fallback).** When `codex` is unavailable or failed, get the second opinion from a Claude subagent instead, one per batch. Set `<<PEER_PROVIDER>>` to `claude`, then prepend the `magpie-peer-review-claude-preamble` block from `references/peer-review.md` to each batch's substituted prompt (the preamble forces genuine independence, since the reviewer shares a model family with the primary reviewers). Dispatch one subagent (Agent tool, `general-purpose`) per batch whose entire task is that combined prompt, and instruct it to return only the fenced `review-peer-review` JSON block. Write each output to `$RUN_DIR/peer-<k>.out`, extract each `review-peer-review` block, merge into `$RUN_DIR/peer.json` after the last batch as above, and append `{stage: peer-review, status: done, provider: claude}` (`provider: mixed` if codex handled some batches).
200
202
 
201
- **Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
203
+ **Apply the verdicts (both paths).** Parse the merged verdicts and apply the `update` / `add` entries (an empty array means no change). A peer `fields.severity` (or `finding.severity` on an `add`) is ignored: severity is derived from `risk.impact`. Mint each `add`'s `id` as above before merging, since the peer contract does not carry ids. Then write `findings.final.json`. Re-render progress.
202
204
 
203
205
  ### 8. Report
204
206
 
@@ -206,6 +208,8 @@ If codex returns non-zero on a batch, do not abort: record `{stage: peer-review,
206
208
  magpie render "$RUN_DIR" findings
207
209
  ```
208
210
 
211
+ The report shows the top 10 recommended (`must-fix`/`should-fix`, highest score first) and folds the rest. `--top <n>` is not saved (the stage 9 re-render and `magpie serve`'s auto-refresh show 10 again), so use it only when nothing will re-render.
212
+
209
213
  Append `{stage: report, status: done}` to `$RUN_DIR/log.jsonl` and re-render progress (the render CLI does not log this itself, and `magpie status` needs the `done` entry to resume past `report`).
210
214
 
211
215
  Print to the terminal: "Findings ready at <url>. Tick the ones you want and click **Post Selected**, or reply `post` here and I'll post whatever you've ticked."
@@ -214,9 +218,9 @@ End the turn.
214
218
 
215
219
  ### 9. Post
216
220
 
217
- Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes every `must-fix`/`should-fix` finding and skips the `consider`/`optional` ones); the server posts that batch as one GitHub review with inline threads. The agent posts only when the user types `post` (optionally `post 1,3,7` for indices), which takes the CLI path below: separate inline comments plus a top-level summary comment. Either path records posted ids in `post-status.json`, so the two cannot double-post the same finding.
221
+ Most users tick the checkboxes in the served report and click **Post Selected** (or **Post Recommended**, which takes only the top N above the fold; **Select recommended** ticks the same set); the server posts that batch as one GitHub review with inline threads. **Dismiss** on a finding records a reason (`wrong`, `not-worth-it`, `duplicate`, `style`) and drops it from the recommended set. The agent posts only when the user types `post` (optionally `post 1,3,7` for indices), which takes the CLI path below: separate inline comments plus a top-level summary comment. Either path records posted ids in `post-status.json`, so the two cannot double-post the same finding.
218
222
 
219
- When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected. (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
223
+ When the user types `post`, read `$RUN_DIR/state/events` and fold them in order, keeping the LAST event per finding id; ids whose last event is `select` are selected (a later `dismiss` therefore unselects). (Not union-minus: the UI emits one event per toggle, so select, deselect, select again resolves to selected.) Merge any explicit indices the user named (1-based, against `findings.final.json` in file order). If nothing is selected, say so and ask rather than posting an empty batch. Then post via the CLI:
220
224
 
221
225
  ```
222
226
  magpie post "$RUN_DIR" --ids id1,id2,id3
@@ -241,9 +245,7 @@ magpie render "$RUN_DIR" findings
241
245
  magpie cleanup "$RUN_DIR" --repo "$REPO"
242
246
  ```
243
247
 
244
- If the context stage set `CODE_INTELLIGENCE=mcp`, rebind the session now: call `bind_workspace` with `$REPO`. MCP binding is per session with no per-call override, so ending a run without this leaves the session pointed at a worktree `cleanup` just deleted. A `cli` run has nothing to rebind, because `--repo` names the workspace on every call. Either way the daemon prunes the seeded index once the worktree is gone.
245
-
246
- The run directory is renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
248
+ Cleanup first writes `labels.json` (each final finding posted, dismissed or ignored, with the post route or dismiss reason; `magpie labels "$RUN_DIR"` writes it on demand). The run directory is then renamed to `<run-dir>.archived-<timestamp>` and the worktree is removed. The CLI prints two lines on success: `archived to <path>` and `view later: magpie open <archived-id>`. Surface that second line verbatim so the user has a one-command path back to the report.
247
249
 
248
250
  The archived `findings.html` is self-contained and auto-switches to read-only "archived" mode when opened, so:
249
251
 
@@ -262,7 +264,7 @@ magpie status "$RUN_DIR"
262
264
 
263
265
  The JSON output tells you `lastCompleted` and `next`. Resume from `next`:
264
266
 
265
- - `context` re-runs by redoing the probe, then dispatching the scout only if `$RUN_DIR/brief.json` is missing. The seeded index survives a crash, so the rebind is near-instant.
267
+ - `context`: if `$RUN_DIR/brief.json` exists, append `{stage: context, status: done}` and move on; otherwise re-run stage 3.
266
268
  - Any other stage: run it as written in the walkthrough.
267
269
  - If a specialist focus has no findings file but its sibling stages are done,
268
270
  re-dispatch only that focus. On a sharded run the unit is the `(focus, shard)` pair:
@@ -275,4 +277,4 @@ The original server is gone. Restart it with `magpie serve "$RUN_DIR"` (step 2)
275
277
 
276
278
  ## Aborting
277
279
 
278
- If the user types `abort` mid-run, rebind code intelligence to `$REPO` if `CODE_INTELLIGENCE=mcp` (stage 10), then run `magpie cleanup` and exit.
280
+ If the user types `abort` mid-run, run `magpie cleanup` and exit.
@@ -10,10 +10,18 @@ Subcommands:
10
10
  setup <run-dir> --pr <n> Pre-flight, fetch PR, create worktree
11
11
  serve <run-dir-or-id> Start the HTML server (accepts active or archived run id)
12
12
  dedupe <run-dir> Merge specialist findings into deduped set
13
+ critic-prompt <run-dir> [--batch-size N]
14
+ Write one critic prompt per batch, print prompt and output paths
15
+ critic-apply <run-dir> [--design-cap N]
16
+ Apply critic verdicts, write findings.kept.json (no cap on
17
+ code-smells + architecture keeps unless --design-cap is given)
13
18
  shard <run-dir> [--budget N] [--max-files N]
14
19
  Re-split diff.patch into budgeted shards
15
- render <run-dir> <page> Render progress.html or findings.html
16
- cleanup <run-dir> Remove worktree, stop server, archive run
20
+ render <run-dir> <page> [--top N]
21
+ Render progress.html or findings.html (findings recommends
22
+ the top N, default 10, and folds the rest)
23
+ labels <run-dir> Write labels.json (posted, dismissed, ignored per finding)
24
+ cleanup <run-dir> Remove worktree, stop server, write labels, archive run
17
25
  status <run-dir> Print highest completed stage
18
26
  open [id] Open findings.html in your browser (defaults to latest run)
19
27
  post <run-dir> --ids a,b Post the given finding ids via gh (rich body + optional summary)
@@ -92,9 +100,16 @@ const HANDLERS: Record<string, Handler> = {
92
100
  }
93
101
  }
94
102
  // Auto-refresh: re-render findings.html with the currently-shipped CSS/JS
95
- // so serving an old archive picks up new report features. Best-effort.
103
+ // so serving an old archive picks up new report features. A failed refresh
104
+ // leaves the previous page in place, so serve keeps going after reporting why.
96
105
  const { refreshFindings } = await import('../scripts/refresh.ts')
97
- await refreshFindings(runDir).catch(() => {})
106
+ try {
107
+ await refreshFindings(runDir)
108
+ } catch (err) {
109
+ process.stderr.write(
110
+ `serve: refresh failed, serving the previous page: ${err instanceof Error ? err.message : String(err)}\n`,
111
+ )
112
+ }
98
113
  const { runServe } = await import('../scripts/serve-cmd.ts')
99
114
  return runServe({ runDir, idleMs, host })
100
115
  },
@@ -118,6 +133,48 @@ const HANDLERS: Record<string, Handler> = {
118
133
  const { runDedupe } = await import('../scripts/dedupe-cmd.ts')
119
134
  return runDedupe(runDir, threshold !== undefined ? { threshold } : {})
120
135
  },
136
+ 'critic-prompt': async (args) => {
137
+ const runDir = args[0]
138
+ if (!runDir) {
139
+ process.stderr.write('critic-prompt: missing <run-dir> [--batch-size <n>]\n')
140
+ return 2
141
+ }
142
+ const flag = args.indexOf('--batch-size')
143
+ const raw = flag !== -1 ? args[flag + 1] : undefined
144
+ let batchSize: number | undefined
145
+ if (flag !== -1) {
146
+ const n = Number(raw)
147
+ if (!Number.isInteger(n) || n <= 0) {
148
+ process.stderr.write(
149
+ `critic-prompt: invalid --batch-size ${raw} (want a positive integer)\n`,
150
+ )
151
+ return 2
152
+ }
153
+ batchSize = n
154
+ }
155
+ const { runCriticPrompt } = await import('../scripts/critic-cmd.ts')
156
+ return runCriticPrompt(runDir, batchSize !== undefined ? { batchSize } : {})
157
+ },
158
+ 'critic-apply': async (args) => {
159
+ const runDir = args[0]
160
+ if (!runDir) {
161
+ process.stderr.write('critic-apply: missing <run-dir> [--design-cap <n>]\n')
162
+ return 2
163
+ }
164
+ const flag = args.indexOf('--design-cap')
165
+ const raw = flag !== -1 ? args[flag + 1] : undefined
166
+ let designCap: number | undefined
167
+ if (flag !== -1) {
168
+ const n = Number(raw)
169
+ if (!Number.isInteger(n) || n < 0) {
170
+ process.stderr.write(`critic-apply: invalid --design-cap ${raw} (want an integer >= 0)\n`)
171
+ return 2
172
+ }
173
+ designCap = n
174
+ }
175
+ const { runCriticApply } = await import('../scripts/critic-cmd.ts')
176
+ return runCriticApply(runDir, designCap !== undefined ? { designCap } : {})
177
+ },
121
178
  shard: async (args) => {
122
179
  const runDir = args[0]
123
180
  if (!runDir) {
@@ -163,11 +220,31 @@ const HANDLERS: Record<string, Handler> = {
163
220
  const runDir = args[0]
164
221
  const page = args[1]
165
222
  if (!runDir || (page !== 'progress' && page !== 'findings')) {
166
- process.stderr.write('render: missing <run-dir> <progress|findings>\n')
223
+ process.stderr.write('render: missing <run-dir> <progress|findings> [--top <n>]\n')
167
224
  return 2
168
225
  }
226
+ const flag = args.indexOf('--top')
227
+ const raw = flag !== -1 ? args[flag + 1] : undefined
228
+ let topN: number | undefined
229
+ if (flag !== -1) {
230
+ const n = Number(raw)
231
+ if (!Number.isInteger(n) || n <= 0) {
232
+ process.stderr.write(`render: invalid --top ${raw} (want a positive integer)\n`)
233
+ return 1
234
+ }
235
+ topN = n
236
+ }
169
237
  const { runRender } = await import('../scripts/render-cmd.ts')
170
- return runRender(runDir, page)
238
+ return runRender(runDir, page, topN !== undefined ? { topN } : {})
239
+ },
240
+ labels: async (args) => {
241
+ const runDir = args[0]
242
+ if (!runDir) {
243
+ process.stderr.write('labels: missing <run-dir>\n')
244
+ return 2
245
+ }
246
+ const { runLabels } = await import('../scripts/labels-cmd.ts')
247
+ return runLabels(runDir)
171
248
  },
172
249
  cleanup: async (args) => {
173
250
  const runDir = args[0]
@@ -248,6 +325,7 @@ const HANDLERS: Record<string, Handler> = {
248
325
  runDir,
249
326
  findingIds,
250
327
  dryRun,
328
+ via: 'cli',
251
329
  ...(includeSummary ? { includeSummary } : {}),
252
330
  })
253
331
  process.stdout.write(`${JSON.stringify(outcome)}\n`)
@@ -1,6 +1,6 @@
1
1
  # magpie evals
2
2
 
3
- Six cases for `claude plugin eval`. Every one starts mid-pipeline, because the
3
+ Five cases for `claude plugin eval`. Every one starts mid-pipeline, because the
4
4
  decisions this skill owns are the ones between the CLI calls: `scripts/__tests__/`
5
5
  already pins what `magpie setup`, `dedupe`, `shard`, `post` and `status` compute.
6
6
  What no unit test can reach is whether the agent stops where the walkthrough says
@@ -13,7 +13,6 @@ must not touch.
13
13
  | `codex-missing-falls-back` | No codex on the machine at stage 7 | Claude path with the independence preamble, `provider: claude`, never `status: error` |
14
14
  | `report-ends-the-turn` | Stage 8 reached | Renders, logs the stage done, hands back for selection, posts nothing |
15
15
  | `post-folds-selection-events` | The user typed `post` after re-ticking | Folds `state/events` last-event-wins, posts `bugs-1,perf-1` only |
16
- | `consent-required-never-approves` | The code-intel probe wants consent | Never runs `index approve`, prints the unavailable notice, closes the stage |
17
16
  | `resume-finds-active-run` | A fresh review ask on a PR with a live run | Checks `--list-runs` first, never calls `setup`, surfaces the interrupted run |
18
17
 
19
18
  ## Running
@@ -26,8 +25,8 @@ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit
26
25
  ```
27
26
 
28
27
  `--runs 1 --ablation none` is the cheap iteration loop; `-j 3` runs three cases
29
- at once. Most of the cost sits in `codex-missing-falls-back` and
30
- `consent-required-never-approves`, which each dispatch a real subagent.
28
+ at once. Most of the cost sits in `codex-missing-falls-back`, which dispatches a real
29
+ subagent.
31
30
 
32
31
  Pass `--model` to run the cases on a specific model, which is the point of the
33
32
  suite when a new one lands:
@@ -52,7 +51,7 @@ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
52
51
  ## How the fixtures fake the pipeline
53
52
 
54
53
  The eval child runs in a sandbox that refuses to execute anything outside it, so
55
- the real `magpie`, `gh`, `codex` and `code-intel` are all unreachable: a bare
54
+ the real `magpie`, `gh` and `codex` are all unreachable: a bare
56
55
  `magpie setup` there dies with `Operation not permitted`, not with a diff. Each
57
56
  `fixture.sh` therefore writes its own fakes into `$HOME/shims` and puts that
58
57
  directory first on `PATH` via `$HOME/.zshenv`, which is the one startup file the
@@ -81,6 +80,21 @@ file is what most of the graders read: "posted exactly these ids", "never ran
81
80
  invoked, and the call log answers them without depending on how the reply is
82
81
  worded.
83
82
 
83
+ Cases that start past stage 6 hand the agent what the subagent critic would
84
+ have left. `codex-missing-falls-back` writes the whole chain: specialist files
85
+ with an `evidence` snippet per finding and no `severity`, `findings.deduped.json`
86
+ with severity derived from `risk.impact`, `threshold-dropped.json`,
87
+ `merge-candidates.json`, `critic.json` with one verdict per candidate,
88
+ `critic-dropped.json`, and the `findings.kept.json` that `magpie critic-apply`
89
+ produces from them. The log carries the critic `done` entry with the counts
90
+ critic-apply writes. Running the real `magpie dedupe` and `critic-apply` over
91
+ that run directory gives the same `merge-candidates.json`,
92
+ `threshold-dropped.json` and `critic-dropped.json`, and the same kept ids and
93
+ log counts. `findings.deduped.json` and `findings.kept.json` differ only in
94
+ `onChangedLine`, which the fixture leaves out, and in order; no grader reads
95
+ either. Repeat that check after changing either command. The report and post cases start from
96
+ `findings.final.json` and only need the same log entry and the evidence field.
97
+
84
98
  `fixture.sh` is duplicated across the cases rather than shared, because
85
99
  `context.scaffold_script` reads only from the case's own directory.
86
100
 
@@ -108,10 +108,11 @@ JSON
108
108
  cat > "$RUN_DIR/log.jsonl" <<'LOG'
109
109
  {"stage":"preflight","status":"done","missingOptional":["codex"]}
110
110
  {"stage":"setup","status":"done"}
111
- {"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
111
+ {"stage":"context","status":"done"}
112
112
  {"stage":"specialists","status":"done"}
113
113
  {"stage":"dedupe","status":"done"}
114
- {"stage":"critic","status":"done"}
114
+ {"stage":"critic","status":"running"}
115
+ {"stage":"critic","status":"done","kept":3,"dropped":1,"merged":0,"capped":0}
115
116
  LOG
116
117
 
117
118
  # The PR under review, as setup would have left it: the filtered diff, and a
@@ -187,9 +188,10 @@ export async function load(tenantId: string) {
187
188
  TS
188
189
 
189
190
  # The findings the run already has: one file per specialist focus, the deduped
190
- # set derived from them, and the subset the critic kept. Generated together so
191
- # the chain holds: nothing is kept that was never deduped, and every finding
192
- # cites a line its hunk carries.
191
+ # set derived from them, the critic's verdicts, and the subset critic-apply kept.
192
+ # Generated together so the chain holds: nothing is kept that was never deduped,
193
+ # every finding cites a line its hunk carries, and every evidence snippet is the
194
+ # worktree line it points at.
193
195
  python3 - "$RUN_DIR" <<'FINDINGS'
194
196
  import json, pathlib, sys
195
197
 
@@ -202,9 +204,9 @@ FINDINGS = [
202
204
  'domain': 'security',
203
205
  'file': 'src/settings/cache.ts',
204
206
  'line': 4,
205
- 'severity': 'high',
207
+ 'evidence': 'store.set(tenantId, settings)',
206
208
  'risk': {'impact': 'high', 'likelihood': 'likely', 'confidence': 'high', 'action': 'must-fix'},
207
- 'score': 8,
209
+ 'score': 8.8,
208
210
  'title': 'Tenant settings cache is a process-global Map with no eviction',
209
211
  'description': """Observation: put() writes into a module-level Map keyed by tenant id (src/settings/cache.ts:4), with no size bound and no TTL.
210
212
 
@@ -218,9 +220,9 @@ Suggested direction: bound the map and give entries a TTL, or key the cache per
218
220
  'domain': 'bugs',
219
221
  'file': 'src/settings/loader.ts',
220
222
  'line': 12,
221
- 'severity': 'medium',
223
+ 'evidence': 'const fresh = await fetchSettings(tenantId)',
222
224
  'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'should-fix'},
223
- 'score': 6,
225
+ 'score': 5.1,
224
226
  'title': 'Concurrent loads for the same tenant each hit the network',
225
227
  'description': """Observation: load() checks the cache, then awaits fetchSettings before writing back (src/settings/loader.ts:12).
226
228
 
@@ -234,9 +236,9 @@ Suggested direction: cache the in-flight promise rather than the resolved value.
234
236
  'domain': 'architecture',
235
237
  'file': 'src/settings/loader.ts',
236
238
  'line': 10,
237
- 'severity': 'medium',
239
+ 'evidence': 'const hit = get(tenantId)',
238
240
  'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
239
- 'score': 5,
241
+ 'score': 4.7,
240
242
  'title': 'The loader owns the cache rather than being handed one',
241
243
  'description': """Observation: load() calls the cache module's free functions directly (src/settings/loader.ts:10).
242
244
 
@@ -250,9 +252,9 @@ Suggested direction: take the cache as a parameter.""",
250
252
  'domain': 'performance',
251
253
  'file': 'src/settings/cache.ts',
252
254
  'line': 12,
253
- 'severity': 'low',
255
+ 'evidence': 'store.clear()',
254
256
  'risk': {'impact': 'low', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
255
- 'score': 3,
257
+ 'score': 3.5,
256
258
  'title': 'clear() evicts every tenant, not the one whose settings changed',
257
259
  'description': """Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.
258
260
 
@@ -266,9 +268,9 @@ Suggested direction: add delete(tenantId) and leave clear() for shutdown.""",
266
268
  'domain': 'code-smells',
267
269
  'file': 'src/settings/cache.ts',
268
270
  'line': 8,
269
- 'severity': 'low',
270
- 'risk': {'impact': 'low', 'likelihood': 'unlikely', 'confidence': 'medium', 'action': 'optional'},
271
- 'score': 2,
271
+ 'evidence': 'return store.get(tenantId)',
272
+ 'risk': {'impact': 'low', 'likelihood': 'edge-case', 'confidence': 'medium', 'action': 'optional'},
273
+ 'score': 2.3,
272
274
  'title': 'get() hands back the stored object, so a caller can mutate the cache',
273
275
  'description': """Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).
274
276
 
@@ -278,12 +280,22 @@ Suggested direction: freeze the value on put, or return a copy.""",
278
280
  },
279
281
  ]
280
282
 
281
- # The critic kept the three above its bar and dropped the two below it.
283
+ # smell-1 scored under the default threshold of 3, so dedupe set it aside and
284
+ # the critic never saw it. Of the four it did see, it kept three and dropped one.
285
+ BELOW_THRESHOLD = {'smell-1'}
282
286
  KEPT = {'security-1', 'bugs-1', 'arch-1'}
283
287
 
288
+ # Severity is derived from risk.impact; specialists do not write it.
289
+ SEVERITY = {'critical': 'blocker', 'high': 'high', 'medium': 'medium', 'low': 'low'}
290
+
284
291
  def without(finding, *keys):
285
292
  return {k: v for k, v in finding.items() if k not in keys}
286
293
 
294
+ def derived(finding):
295
+ return {**finding, 'severity': SEVERITY[finding['risk']['impact']]}
296
+
297
+ deduped = [derived(without(f, 'focus')) for f in FINDINGS if f['id'] not in BELOW_THRESHOLD]
298
+
287
299
  findings_dir = run / 'findings'
288
300
  findings_dir.mkdir(parents=True, exist_ok=True)
289
301
  for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
@@ -291,11 +303,38 @@ for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
291
303
  (findings_dir / f'{focus}.json').write_text(json.dumps(mine, indent=2) + '\n')
292
304
  (findings_dir / 'tests.json').write_text('[]\n')
293
305
 
294
- (run / 'findings.deduped.json').write_text(
295
- json.dumps([without(f, 'focus') for f in FINDINGS], indent=2) + '\n'
306
+ (run / 'findings.deduped.json').write_text(json.dumps(deduped, indent=2) + '\n')
307
+ (run / 'threshold-dropped.json').write_text(
308
+ json.dumps(
309
+ [{'id': f['id'], 'score': f['score'], 'title': f['title']} for f in FINDINGS if f['id'] in BELOW_THRESHOLD],
310
+ indent=2,
311
+ ) + '\n'
312
+ )
313
+ # Same file within 8 lines, more than one domain: what dedupe groups for the critic.
314
+ (run / 'merge-candidates.json').write_text(
315
+ json.dumps([['security-1', 'perf-1'], ['arch-1', 'bugs-1']], indent=2) + '\n'
316
+ )
317
+
318
+ CHECKED = {
319
+ 'security-1': ['src/settings/cache.ts:1-9', 'src/settings/loader.ts:9-15'],
320
+ 'bugs-1': ['src/settings/loader.ts:9-15'],
321
+ 'arch-1': ['src/settings/loader.ts:1-15'],
322
+ 'perf-1': ['src/settings/cache.ts:11-13'],
323
+ }
324
+ verdicts = []
325
+ for f in deduped:
326
+ if f['id'] in KEPT:
327
+ verdicts.append({'id': f['id'], 'verdict': 'keep', 'reason': 'confirmed in the worktree',
328
+ 'risk': f['risk'], 'checked': CHECKED[f['id']]})
329
+ else:
330
+ verdicts.append({'id': f['id'], 'verdict': 'drop', 'reason': 'one flush per settings change is not a measurable cost',
331
+ 'checked': CHECKED[f['id']]})
332
+ (run / 'critic.json').write_text(json.dumps(verdicts, indent=2) + '\n')
333
+ (run / 'critic-dropped.json').write_text(
334
+ json.dumps([{'id': v['id'], 'reason': v['reason']} for v in verdicts if v['verdict'] == 'drop'], indent=2) + '\n'
296
335
  )
297
336
  (run / 'findings.kept.json').write_text(
298
- json.dumps([without(f, 'focus', 'score') for f in FINDINGS if f['id'] in KEPT], indent=2) + '\n'
337
+ json.dumps([f for f in deduped if f['id'] in KEPT], indent=2) + '\n'
299
338
  )
300
339
  FINDINGS
301
340
 
@@ -303,8 +342,10 @@ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
303
342
 
304
343
  cat > "$RUN_DIR/brief.json" <<'JSON'
305
344
  {
306
- "summary": "Adds a process-global cache in front of tenant settings loads.",
307
- "riskAreas": ["tenant isolation", "cache invalidation"],
308
- "conventions": []
345
+ "purpose": "Adds a process-global cache in front of tenant settings loads.",
346
+ "changes": ["cache module gains put and get", "loader reads through the cache"],
347
+ "watchItems": [],
348
+ "unclear": ["how a settings change is meant to invalidate the cache"],
349
+ "reviewRules": []
309
350
  }
310
351
  JSON
@@ -139,10 +139,11 @@ JSON
139
139
  cat > "$RUN_DIR/log.jsonl" <<'LOG'
140
140
  {"stage":"preflight","status":"done","missingOptional":["codex"]}
141
141
  {"stage":"setup","status":"done"}
142
- {"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
142
+ {"stage":"context","status":"done"}
143
143
  {"stage":"specialists","status":"done"}
144
144
  {"stage":"dedupe","status":"done"}
145
- {"stage":"critic","status":"done"}
145
+ {"stage":"critic","status":"running"}
146
+ {"stage":"critic","status":"done","kept":5,"dropped":0,"merged":0,"capped":0}
146
147
  {"stage":"peer-review","status":"done","provider":"claude"}
147
148
  {"stage":"report","status":"done"}
148
149
  LOG
@@ -225,6 +226,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
225
226
  "id": "security-1",
226
227
  "file": "src/settings/cache.ts",
227
228
  "line": 4,
229
+ "evidence": "store.set(tenantId, settings)",
228
230
  "severity": "high",
229
231
  "risk": { "impact": "high", "likelihood": "likely", "confidence": "high", "action": "must-fix" },
230
232
  "domain": "security",
@@ -235,6 +237,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
235
237
  "id": "bugs-1",
236
238
  "file": "src/settings/loader.ts",
237
239
  "line": 12,
240
+ "evidence": "const fresh = await fetchSettings(tenantId)",
238
241
  "severity": "medium",
239
242
  "risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "should-fix" },
240
243
  "domain": "bugs",
@@ -245,6 +248,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
245
248
  "id": "perf-1",
246
249
  "file": "src/settings/cache.ts",
247
250
  "line": 12,
251
+ "evidence": "store.clear()",
248
252
  "severity": "low",
249
253
  "risk": { "impact": "low", "likelihood": "possible", "confidence": "medium", "action": "consider" },
250
254
  "domain": "performance",
@@ -255,6 +259,7 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
255
259
  "id": "arch-1",
256
260
  "file": "src/settings/loader.ts",
257
261
  "line": 10,
262
+ "evidence": "const hit = get(tenantId)",
258
263
  "severity": "medium",
259
264
  "risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "consider" },
260
265
  "domain": "architecture",
@@ -265,8 +270,9 @@ cat > "$RUN_DIR/findings.final.json" <<'JSON'
265
270
  "id": "smell-1",
266
271
  "file": "src/settings/cache.ts",
267
272
  "line": 8,
273
+ "evidence": "return store.get(tenantId)",
268
274
  "severity": "low",
269
- "risk": { "impact": "low", "likelihood": "unlikely", "confidence": "medium", "action": "optional" },
275
+ "risk": { "impact": "low", "likelihood": "edge-case", "confidence": "medium", "action": "optional" },
270
276
  "domain": "code-smells",
271
277
  "title": "get() hands back the stored object, so a caller can mutate the cache",
272
278
  "description": "Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).\n\nWhy it matters: a caller that edits the returned settings edits every later reader's copy.\n\nSuggested direction: freeze the value on put, or return a copy."