@iceinvein/agent-skills 0.20.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +6 -2
- package/package.json +3 -2
- package/skills/index.json +2 -2
- package/skills/magpie/evals/README.md +151 -0
- package/skills/magpie/evals/codex-missing-falls-back/case.yaml +4 -0
- package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +310 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/final-findings-were-written.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/kept-findings-were-reachable.md +5 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/missing-codex-is-not-an-error.md +7 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/peer-prompt-carries-the-preamble.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/provider-logged-as-claude.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/codex-missing-falls-back/prompt.md +11 -0
- package/skills/magpie/evals/consent-required-never-approves/case.yaml +4 -0
- package/skills/magpie/evals/consent-required-never-approves/fixture.sh +213 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +6 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +7 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +5 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +10 -0
- package/skills/magpie/evals/consent-required-never-approves/prompt.md +11 -0
- package/skills/magpie/evals/post-folds-selection-events/case.yaml +4 -0
- package/skills/magpie/evals/post-folds-selection-events/fixture.sh +290 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/deselected-finding-was-left-alone.md +7 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/events-were-reachable.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/post-stage-logged-done.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/posts-the-last-event-selection.md +6 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/report-was-re-rendered-after-posting.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/prompt.md +11 -0
- package/skills/magpie/evals/report-ends-the-turn/case.yaml +4 -0
- package/skills/magpie/evals/report-ends-the-turn/fixture.sh +246 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/findings-report-was-rendered.md +6 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/findings-were-reachable.md +5 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/hands-back-for-selection.md +10 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/nothing-was-posted.md +7 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/report-stage-logged-done.md +6 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/report-ends-the-turn/prompt.md +11 -0
- package/skills/magpie/evals/resume-finds-active-run/case.yaml +4 -0
- package/skills/magpie/evals/resume-finds-active-run/fixture.sh +234 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/existing-runs-were-checked.md +5 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/no-fresh-run-was-started.md +7 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/surfaces-the-interrupted-run.md +10 -0
- package/skills/magpie/evals/resume-finds-active-run/prompt.md +11 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/case.yaml +4 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +215 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/asks-before-dispatching.md +16 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/manifest-was-read.md +5 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-findings-were-written.md +6 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-specialist-was-dispatched.md +7 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/prompt.md +11 -0
- package/skills/magpie/skill.json +1 -1
- package/skills/sluice/SKILL.md +20 -11
- package/skills/sluice/agents/sluice-implementer-low.md +31 -0
- package/skills/sluice/evals/README.md +218 -12
- package/skills/sluice/evals/announcement-reaches-the-ledger/case.yaml +4 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/fixture.sh +112 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/escalates-fast-to-main.md +12 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/verbose-flag-implemented.md +5 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/prompt.md +11 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/fixture.sh +9 -1
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +5 -4
- package/skills/sluice/evals/deep-plan-asks-the-fork/case.yaml +4 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/fixture.sh +108 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/announces-deep-channel.md +12 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/asks-one-fork-question.md +20 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-design-before-the-answer.md +8 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-implementation-yet.md +10 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +26 -26
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +1 -1
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/fixture.sh +216 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/hands-back-not-a-decision-list.md +23 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/nothing-blocked.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/record-names-the-prefix.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-fans-out/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-fans-out/fixture.sh +250 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/disjoint-tasks-in-one-message.md +15 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/implementers-dispatched.md +6 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-fans-out/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +26 -26
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +4 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/fixture.sh +310 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/hands-back-finished-work.md +19 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/stop-hook-never-refused.md +9 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/fixture.sh +310 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/hands-back-after-the-last-result.md +23 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/suite-was-run.md +9 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/t2-dispatched-with-label.md +6 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/prompt.md +11 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +6 -1
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +6 -1
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/case.yaml +4 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/fixture.sh +149 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/help-test-lists-verbose.md +8 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/verbose-registered-in-flags-table.md +9 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/prompt.md +11 -0
- package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +5 -4
- package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +7 -2
- package/skills/sluice/references/deep-channel.md +107 -59
- package/skills/sluice/references/meter.md +8 -0
- package/skills/sluice/references/status.md +15 -8
- package/skills/sluice/scripts/plan.sh +31 -18
- package/skills/sluice/scripts/run-stats.sh +19 -3
- package/skills/sluice/scripts/status.sh +42 -14
- package/skills/sluice/scripts/stop-guard.sh +24 -11
- package/skills/sluice/skill.json +8 -2
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +0 -105
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +0 -300
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +0 -122
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +0 -324
package/dist/cli/index.js
CHANGED
|
@@ -666,6 +666,8 @@ var claudeAdapter = {
|
|
|
666
666
|
for (const [relPath, content] of files) {
|
|
667
667
|
if (relPath === promptPath)
|
|
668
668
|
continue;
|
|
669
|
+
if (config.supporting && Object.hasOwn(config.supporting, relPath))
|
|
670
|
+
continue;
|
|
669
671
|
const targetRel = join2(config.bundleRoot, relPath);
|
|
670
672
|
const targetPath = join2(cwd, targetRel);
|
|
671
673
|
mkdirSync(dirname(targetPath), { recursive: true });
|
|
@@ -747,9 +749,11 @@ var claudeAdapter = {
|
|
|
747
749
|
unlinkSync(fullPath);
|
|
748
750
|
}
|
|
749
751
|
} catch {}
|
|
752
|
+
const stops = [join2(cwd, ".claude"), cwd];
|
|
753
|
+
if (config.bundleRoot)
|
|
754
|
+
stops.push(join2(cwd, config.bundleRoot, ".."));
|
|
750
755
|
let dir = dirname(fullPath);
|
|
751
|
-
|
|
752
|
-
while (dir !== stopAt && dir !== "/" && dir.startsWith(cwd)) {
|
|
756
|
+
while (!stops.includes(dir) && dir.startsWith(`${cwd}/`)) {
|
|
753
757
|
try {
|
|
754
758
|
rmdirSync(dir);
|
|
755
759
|
} catch {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@iceinvein/agent-skills",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.21.0",
|
|
4
4
|
"description": "Install agent skills into AI coding tools",
|
|
5
5
|
"author": "iceinvein",
|
|
6
6
|
"license": "MIT",
|
|
@@ -10,7 +10,8 @@
|
|
|
10
10
|
},
|
|
11
11
|
"files": [
|
|
12
12
|
"dist",
|
|
13
|
-
"skills"
|
|
13
|
+
"skills",
|
|
14
|
+
"!skills/*/evals/results"
|
|
14
15
|
],
|
|
15
16
|
"scripts": {
|
|
16
17
|
"build": "bun build src/cli/index.ts --outdir dist/cli --target bun",
|
package/skills/index.json
CHANGED
|
@@ -221,7 +221,7 @@
|
|
|
221
221
|
"name": "magpie",
|
|
222
222
|
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), dedupes findings, applies a critic rubric, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting findings to post via gh. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
|
|
223
223
|
"type": "prompt",
|
|
224
|
-
"version": "0.11.
|
|
224
|
+
"version": "0.11.1"
|
|
225
225
|
},
|
|
226
226
|
{
|
|
227
227
|
"name": "migrate",
|
|
@@ -283,7 +283,7 @@
|
|
|
283
283
|
"name": "sluice",
|
|
284
284
|
"description": "Routes work by change shape into four channels (bypass, fast, main, deep) and applies only the rules each channel needs, so a one-line fix does not pay the cost of a multi-subsystem build. Carries seven rules as one-liners in the router and the full treatment in references read only on friction. Checks the finished plan with plan.sh validate rather than trusting it to memory, seeds the run state from it, keeps a deep run's task breakdown in .sluice/run.json so a statusline segment, one status command and a SessionStart hook can answer where the run is (the hook prints a live run at every session start, compaction included), and closes each run with a ledger read out of the session transcript: elapsed, tools, tokens, and what each dispatched agent cost where the transcript recorded it. Claude Code only; stands down where the superpowers pipeline governs the repo.",
|
|
285
285
|
"type": "prompt",
|
|
286
|
-
"version": "0.
|
|
286
|
+
"version": "0.23.0"
|
|
287
287
|
},
|
|
288
288
|
{
|
|
289
289
|
"name": "temporal-coupling-detector",
|
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
# magpie evals
|
|
2
|
+
|
|
3
|
+
Six cases for `claude plugin eval`. Every one starts mid-pipeline, because the
|
|
4
|
+
decisions this skill owns are the ones between the CLI calls: `scripts/__tests__/`
|
|
5
|
+
already pins what `magpie setup`, `dedupe`, `shard`, `post` and `status` compute.
|
|
6
|
+
What no unit test can reach is whether the agent stops where the walkthrough says
|
|
7
|
+
stop, falls back where it says fall back, and keeps its hands off the things it
|
|
8
|
+
must not touch.
|
|
9
|
+
|
|
10
|
+
| Case | Signal under test | What it pins |
|
|
11
|
+
|---|---|---|
|
|
12
|
+
| `shard-gate-stops-and-asks` | A resume at stage 4 with seven shards | Stops before dispatching, names 7 shards and 35 subagents, offers all three options |
|
|
13
|
+
| `codex-missing-falls-back` | No codex on the machine at stage 7 | Claude path with the independence preamble, `provider: claude`, never `status: error` |
|
|
14
|
+
| `report-ends-the-turn` | Stage 8 reached | Renders, logs the stage done, hands back for selection, posts nothing |
|
|
15
|
+
| `post-folds-selection-events` | The user typed `post` after re-ticking | Folds `state/events` last-event-wins, posts `bugs-1,perf-1` only |
|
|
16
|
+
| `consent-required-never-approves` | The code-intel probe wants consent | Never runs `index approve`, prints the unavailable notice, closes the stage |
|
|
17
|
+
| `resume-finds-active-run` | A fresh review ask on a PR with a live run | Checks `--list-runs` first, never calls `setup`, surfaces the interrupted run |
|
|
18
|
+
|
|
19
|
+
## Running
|
|
20
|
+
|
|
21
|
+
Every case scaffolds a run directory and then writes into it, so they all need
|
|
22
|
+
the scaffold flag and a tool grant. From the repo root:
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
`--runs 1 --ablation none` is the cheap iteration loop; `-j 3` runs three cases
|
|
29
|
+
at once. Most of the cost sits in `codex-missing-falls-back` and
|
|
30
|
+
`consent-required-never-approves`, which each dispatch a real subagent.
|
|
31
|
+
|
|
32
|
+
Pass `--model` to run the cases on a specific model, which is the point of the
|
|
33
|
+
suite when a new one lands:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
|
|
37
|
+
--model claude-opus-5-5
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
A full single-run pass costs about $1.90 and takes five minutes on Claude Opus
|
|
41
|
+
5.5, against about $3.60 and eight minutes on Claude Opus 5. `--judge-model`
|
|
42
|
+
is separate and defaults to haiku; leave it alone when comparing models, or the
|
|
43
|
+
judge moves at the same time as the thing being judged.
|
|
44
|
+
|
|
45
|
+
To gate CI, pick a floor and let a miss fail the job:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
|
|
49
|
+
--trust-plugin --threshold 0.8
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## How the fixtures fake the pipeline
|
|
53
|
+
|
|
54
|
+
The eval child runs in a sandbox that refuses to execute anything outside it, so
|
|
55
|
+
the real `magpie`, `gh`, `codex` and `code-intel` are all unreachable: a bare
|
|
56
|
+
`magpie setup` there dies with `Operation not permitted`, not with a diff. Each
|
|
57
|
+
`fixture.sh` therefore writes its own fakes into `$HOME/shims` and puts that
|
|
58
|
+
directory first on `PATH` via `$HOME/.zshenv`, which is the one startup file the
|
|
59
|
+
child's Bash tool reads. The same `PATH` drops every real binary directory, so
|
|
60
|
+
`codex` is genuinely absent in the fallback case rather than merely unused, and
|
|
61
|
+
nothing in a case can reach the network or a real PR.
|
|
62
|
+
|
|
63
|
+
Three things follow from the sandbox, and cases are written around them:
|
|
64
|
+
|
|
65
|
+
- **No listening sockets.** `magpie serve` writes the `server-info` the
|
|
66
|
+
walkthrough reads and exits; the page behind that URL is never reachable. No
|
|
67
|
+
case pins anything that needs the browser surface.
|
|
68
|
+
- **Run directories live under the workspace**, at `runs/<run id>`, not under
|
|
69
|
+
`~/.magpie`. File graders refuse to follow a link out of the workspace, and
|
|
70
|
+
`magpie --list-runs` is what names the path a resume uses anyway, so the fake
|
|
71
|
+
reports the workspace path.
|
|
72
|
+
- **Fakes compute rather than answer.** `magpie status` reads `log.jsonl` with
|
|
73
|
+
the same stage ladder as `scripts/status-cmd.ts`, so a stage the agent logs
|
|
74
|
+
moves `next` exactly as the real CLI would. A canned answer went stale the
|
|
75
|
+
moment the agent appended to the log, and the agent noticed and spent a
|
|
76
|
+
paragraph on it.
|
|
77
|
+
|
|
78
|
+
Each fake also appends its argv to `.magpie-calls.log` in the workspace. That
|
|
79
|
+
file is what most of the graders read: "posted exactly these ids", "never ran
|
|
80
|
+
`index approve`", "never called `setup`" are all claims about what the run
|
|
81
|
+
invoked, and the call log answers them without depending on how the reply is
|
|
82
|
+
worded.
|
|
83
|
+
|
|
84
|
+
`fixture.sh` is duplicated across the cases rather than shared, because
|
|
85
|
+
`context.scaffold_script` reads only from the case's own directory.
|
|
86
|
+
|
|
87
|
+
## Grader notes
|
|
88
|
+
|
|
89
|
+
`tool_used: Skill` graders are excluded from the score in a two-arm run and
|
|
90
|
+
reported as pass/fail indicators, because they can never pass without the
|
|
91
|
+
plugin. They are there to tell you whether a score came from magpie or from the
|
|
92
|
+
model's own habits.
|
|
93
|
+
|
|
94
|
+
**Every `llm` grader carries `focus: last_message`.** Without it the judge is
|
|
95
|
+
handed a window of the whole trace, and in a case that reads a 40,000-line diff
|
|
96
|
+
the handback falls outside that window: the shard-gate rubric voted FAIL nine
|
|
97
|
+
times out of nine on replies that laid the gate out correctly. That flap is what
|
|
98
|
+
a missing `focus` looks like, not a rubric that needs loosening.
|
|
99
|
+
|
|
100
|
+
`file_exists` with `exists: true` asks whether the *run* created a file, not
|
|
101
|
+
whether one is there: a fixture file fails it. Assertions about fixture content
|
|
102
|
+
use a `regex` grader with a file target instead, and every case carries one such
|
|
103
|
+
grader over a file the fixture wrote, so a fixture that failed to scaffold shows
|
|
104
|
+
up as a failure rather than as a vacuous pass on the `exists: false` graders.
|
|
105
|
+
|
|
106
|
+
Graders will not follow a link out of the workspace, which is why the run
|
|
107
|
+
directories are where they are.
|
|
108
|
+
|
|
109
|
+
## What the first passes turned up
|
|
110
|
+
|
|
111
|
+
Five of the six defects the early runs surfaced were in the fixtures, and the
|
|
112
|
+
agent found them by reading the state it was handed:
|
|
113
|
+
|
|
114
|
+
- `findings.deduped.json` was `[]` while `findings.kept.json` held three
|
|
115
|
+
findings. The run stopped and refused to peer-review them, correctly: the
|
|
116
|
+
critic keeps a subset of the deduped set, so that state cannot happen. The
|
|
117
|
+
findings chain is generated from one list now, per focus, deduped and kept
|
|
118
|
+
together.
|
|
119
|
+
- The shard manifest advertised 5,400 lines a shard over 212-byte patch stubs.
|
|
120
|
+
The manifest is derived from the patches the fixture writes now, and they are
|
|
121
|
+
sized so a seven-way split is what the default 6,000-line budget gives.
|
|
122
|
+
- Diff paths, worktree paths and hunk headers disagreed with each other, in both
|
|
123
|
+
the small-PR cases and the sharded one. The diff, the worktree and the line
|
|
124
|
+
each finding cites are one block now, and the hunk headers count the lines
|
|
125
|
+
they carry.
|
|
126
|
+
- `magpie serve` promised a URL the sandbox will not let anything bind. No case
|
|
127
|
+
pins the browser surface, and the rubric accepts a reply that reports the
|
|
128
|
+
server as unreachable.
|
|
129
|
+
- `magpie status` answered from a canned string, so it went stale the moment the
|
|
130
|
+
agent logged a stage and the agent spent a paragraph on the discrepancy. It
|
|
131
|
+
reads `log.jsonl` now, with the ladder from `scripts/status-cmd.ts`.
|
|
132
|
+
|
|
133
|
+
## Verification status
|
|
134
|
+
|
|
135
|
+
The suite has been run end to end four times (`--runs 1 --ablation none`)
|
|
136
|
+
against the committed fixtures, scoring 1.00 every time: twice on Claude Opus 5
|
|
137
|
+
and twice on Claude Opus 5.5. `shard-gate-stops-and-asks` was run three more
|
|
138
|
+
times on its own at `--runs 3` after the `focus` fix, passing 9/9 judge votes.
|
|
139
|
+
Every `llm` rubric has been tuned against real transcripts rather than written
|
|
140
|
+
blind.
|
|
141
|
+
|
|
142
|
+
Claude Opus 5.5 ran the suite at roughly half the cost and two thirds of the
|
|
143
|
+
wall clock of Claude Opus 5, with no case needing a rubric or fixture change.
|
|
144
|
+
Its default effort is `medium` where Claude Opus 5's is `high`, so a rubric that
|
|
145
|
+
starts failing there is worth reading as a real behavioural difference before
|
|
146
|
+
it is loosened.
|
|
147
|
+
|
|
148
|
+
Not yet known: no case has been run with the no-plugin baseline arm, so the
|
|
149
|
+
ablation delta is unmeasured, and no case other than the shard gate has been
|
|
150
|
+
repeated within one invocation, so the per-case variance is only bounded by
|
|
151
|
+
those two clean passes.
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# A run parked after the critic stage on a machine with no codex. Stage 7 still
|
|
3
|
+
# has to happen: the Claude second opinion stands in, with the preamble that
|
|
4
|
+
# buys back the independence a same-family reviewer loses.
|
|
5
|
+
#
|
|
6
|
+
# The run directory sits under the workspace rather than ~/.magpie because file
|
|
7
|
+
# graders refuse to follow a link out of the workspace. `magpie --list-runs` is
|
|
8
|
+
# what names the path a resume uses, so the shim reports this one.
|
|
9
|
+
set -euo pipefail
|
|
10
|
+
|
|
11
|
+
RUN_ID="pr-1337-1789600000"
|
|
12
|
+
RUN_DIR="$PWD/runs/$RUN_ID"
|
|
13
|
+
CALLS="$PWD/.magpie-calls.log"
|
|
14
|
+
|
|
15
|
+
mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$HOME/shims"
|
|
16
|
+
: > "$CALLS"
|
|
17
|
+
|
|
18
|
+
cat > "$HOME/shim-config" <<EOF
|
|
19
|
+
RUN_ID="$RUN_ID"
|
|
20
|
+
RUN_DIR="$RUN_DIR"
|
|
21
|
+
CALLS="$CALLS"
|
|
22
|
+
PORT=4599
|
|
23
|
+
EOF
|
|
24
|
+
|
|
25
|
+
# The real magpie is outside the eval sandbox and cannot be executed from
|
|
26
|
+
# inside it, so the child gets a fake rather than exit 126. The PATH here also
|
|
27
|
+
# leaves out the real codex, which is the condition under test.
|
|
28
|
+
mkdir -p "$HOME/tmp"
|
|
29
|
+
cat > "$HOME/.zshenv" <<'RC'
|
|
30
|
+
export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
|
|
31
|
+
# /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
|
|
32
|
+
# trying to create its xcrun cache in a directory the sandbox blocks.
|
|
33
|
+
export TMPDIR="$HOME/tmp"
|
|
34
|
+
RC
|
|
35
|
+
|
|
36
|
+
cat > "$HOME/shims/magpie" <<'SHIM'
|
|
37
|
+
#!/usr/bin/env bash
|
|
38
|
+
. "$HOME/shim-config"
|
|
39
|
+
echo "magpie $*" >> "$CALLS"
|
|
40
|
+
case "${1:-}" in
|
|
41
|
+
--list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
|
|
42
|
+
status)
|
|
43
|
+
python3 - "${2:-$RUN_DIR}" <<'STATUS'
|
|
44
|
+
import json, pathlib, sys
|
|
45
|
+
|
|
46
|
+
ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
|
|
47
|
+
last, error = None, None
|
|
48
|
+
for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
|
|
49
|
+
if not line.strip():
|
|
50
|
+
continue
|
|
51
|
+
try:
|
|
52
|
+
entry = json.loads(line)
|
|
53
|
+
except ValueError:
|
|
54
|
+
continue
|
|
55
|
+
if entry.get('status') == 'error':
|
|
56
|
+
error = entry.get('stage')
|
|
57
|
+
break
|
|
58
|
+
if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
|
|
59
|
+
last = entry['stage']
|
|
60
|
+
index = ORDER.index(last) + 1 if last else 0
|
|
61
|
+
print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
|
|
62
|
+
STATUS
|
|
63
|
+
;;
|
|
64
|
+
serve)
|
|
65
|
+
mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
|
|
66
|
+
# The eval sandbox refuses listening sockets, so no fake can hold a port
|
|
67
|
+
# open: this writes the server-info the walkthrough reads and exits. The
|
|
68
|
+
# page is never reachable in a case, so no case pins the browser surface.
|
|
69
|
+
echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
|
|
70
|
+
echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
|
|
71
|
+
;;
|
|
72
|
+
render)
|
|
73
|
+
mkdir -p "$RUN_DIR/screen"
|
|
74
|
+
python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
|
|
75
|
+
import json, pathlib, sys
|
|
76
|
+
|
|
77
|
+
run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
|
|
78
|
+
findings = run / 'findings.final.json'
|
|
79
|
+
rows = ''
|
|
80
|
+
if screen == 'findings' and findings.exists():
|
|
81
|
+
for finding in json.loads(findings.read_text()):
|
|
82
|
+
rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
|
|
83
|
+
buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
|
|
84
|
+
(run / 'screen').mkdir(exist_ok=True)
|
|
85
|
+
(run / 'screen' / f'{screen}.html').write_text(
|
|
86
|
+
f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
|
|
87
|
+
)
|
|
88
|
+
RENDER
|
|
89
|
+
echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
|
|
90
|
+
;;
|
|
91
|
+
*) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
|
|
92
|
+
esac
|
|
93
|
+
SHIM
|
|
94
|
+
chmod +x "$HOME/shims/magpie"
|
|
95
|
+
|
|
96
|
+
cat > "$RUN_DIR/pr.json" <<'JSON'
|
|
97
|
+
{
|
|
98
|
+
"number": 1337,
|
|
99
|
+
"title": "Cache tenant settings in the request path",
|
|
100
|
+
"author": { "login": "asha-platform" },
|
|
101
|
+
"headRefName": "feat/tenant-settings-cache",
|
|
102
|
+
"baseRefName": "main",
|
|
103
|
+
"headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
|
|
104
|
+
"url": "https://github.com/example/repo/pull/1337"
|
|
105
|
+
}
|
|
106
|
+
JSON
|
|
107
|
+
|
|
108
|
+
cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
109
|
+
{"stage":"preflight","status":"done","missingOptional":["codex"]}
|
|
110
|
+
{"stage":"setup","status":"done"}
|
|
111
|
+
{"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
|
|
112
|
+
{"stage":"specialists","status":"done"}
|
|
113
|
+
{"stage":"dedupe","status":"done"}
|
|
114
|
+
{"stage":"critic","status":"done"}
|
|
115
|
+
LOG
|
|
116
|
+
|
|
117
|
+
# The PR under review, as setup would have left it: the filtered diff, and a
|
|
118
|
+
# worktree holding the head state the diff produces. The hunk headers count the
|
|
119
|
+
# lines they carry, and every finding below cites a line inside a hunk, so
|
|
120
|
+
# nothing here contradicts anything else.
|
|
121
|
+
cat > "$RUN_DIR/diff.patch" <<'PATCH'
|
|
122
|
+
diff --git a/src/settings/cache.ts b/src/settings/cache.ts
|
|
123
|
+
--- a/src/settings/cache.ts
|
|
124
|
+
+++ b/src/settings/cache.ts
|
|
125
|
+
@@ -1,5 +1,13 @@
|
|
126
|
+
const store = new Map<string, Settings>()
|
|
127
|
+
|
|
128
|
+
+export function put(tenantId: string, settings: Settings) {
|
|
129
|
+
+ store.set(tenantId, settings)
|
|
130
|
+
+}
|
|
131
|
+
+
|
|
132
|
+
+export function get(tenantId: string): Settings | undefined {
|
|
133
|
+
+ return store.get(tenantId)
|
|
134
|
+
+}
|
|
135
|
+
+
|
|
136
|
+
export function clear() {
|
|
137
|
+
store.clear()
|
|
138
|
+
}
|
|
139
|
+
diff --git a/src/settings/loader.ts b/src/settings/loader.ts
|
|
140
|
+
--- a/src/settings/loader.ts
|
|
141
|
+
+++ b/src/settings/loader.ts
|
|
142
|
+
@@ -9,3 +9,7 @@
|
|
143
|
+
export async function load(tenantId: string) {
|
|
144
|
+
- return fetchSettings(tenantId)
|
|
145
|
+
+ const hit = get(tenantId)
|
|
146
|
+
+ if (hit) return hit
|
|
147
|
+
+ const fresh = await fetchSettings(tenantId)
|
|
148
|
+
+ put(tenantId, fresh)
|
|
149
|
+
+ return fresh
|
|
150
|
+
}
|
|
151
|
+
PATCH
|
|
152
|
+
|
|
153
|
+
mkdir -p "$RUN_DIR/worktree/src/settings"
|
|
154
|
+
|
|
155
|
+
cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
|
|
156
|
+
const store = new Map<string, Settings>()
|
|
157
|
+
|
|
158
|
+
export function put(tenantId: string, settings: Settings) {
|
|
159
|
+
store.set(tenantId, settings)
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
export function get(tenantId: string): Settings | undefined {
|
|
163
|
+
return store.get(tenantId)
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
export function clear() {
|
|
167
|
+
store.clear()
|
|
168
|
+
}
|
|
169
|
+
TS
|
|
170
|
+
|
|
171
|
+
cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
|
|
172
|
+
import { get, put } from './cache'
|
|
173
|
+
|
|
174
|
+
type Settings = { theme: string }
|
|
175
|
+
|
|
176
|
+
async function fetchSettings(tenantId: string): Promise<Settings> {
|
|
177
|
+
return { theme: 'default' }
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
export async function load(tenantId: string) {
|
|
181
|
+
const hit = get(tenantId)
|
|
182
|
+
if (hit) return hit
|
|
183
|
+
const fresh = await fetchSettings(tenantId)
|
|
184
|
+
put(tenantId, fresh)
|
|
185
|
+
return fresh
|
|
186
|
+
}
|
|
187
|
+
TS
|
|
188
|
+
|
|
189
|
+
# The findings the run already has: one file per specialist focus, the deduped
|
|
190
|
+
# set derived from them, and the subset the critic kept. Generated together so
|
|
191
|
+
# the chain holds: nothing is kept that was never deduped, and every finding
|
|
192
|
+
# cites a line its hunk carries.
|
|
193
|
+
python3 - "$RUN_DIR" <<'FINDINGS'
|
|
194
|
+
import json, pathlib, sys
|
|
195
|
+
|
|
196
|
+
run = pathlib.Path(sys.argv[1])
|
|
197
|
+
|
|
198
|
+
FINDINGS = [
|
|
199
|
+
{
|
|
200
|
+
'id': 'security-1',
|
|
201
|
+
'focus': 'security',
|
|
202
|
+
'domain': 'security',
|
|
203
|
+
'file': 'src/settings/cache.ts',
|
|
204
|
+
'line': 4,
|
|
205
|
+
'severity': 'high',
|
|
206
|
+
'risk': {'impact': 'high', 'likelihood': 'likely', 'confidence': 'high', 'action': 'must-fix'},
|
|
207
|
+
'score': 8,
|
|
208
|
+
'title': 'Tenant settings cache is a process-global Map with no eviction',
|
|
209
|
+
'description': """Observation: put() writes into a module-level Map keyed by tenant id (src/settings/cache.ts:4), with no size bound and no TTL.
|
|
210
|
+
|
|
211
|
+
Why it matters: a long-lived process accumulates every tenant it has served, and a settings change never reaches the cached copy.
|
|
212
|
+
|
|
213
|
+
Suggested direction: bound the map and give entries a TTL, or key the cache per request.""",
|
|
214
|
+
},
|
|
215
|
+
{
|
|
216
|
+
'id': 'bugs-1',
|
|
217
|
+
'focus': 'bugs',
|
|
218
|
+
'domain': 'bugs',
|
|
219
|
+
'file': 'src/settings/loader.ts',
|
|
220
|
+
'line': 12,
|
|
221
|
+
'severity': 'medium',
|
|
222
|
+
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'should-fix'},
|
|
223
|
+
'score': 6,
|
|
224
|
+
'title': 'Concurrent loads for the same tenant each hit the network',
|
|
225
|
+
'description': """Observation: load() checks the cache, then awaits fetchSettings before writing back (src/settings/loader.ts:12).
|
|
226
|
+
|
|
227
|
+
Why it matters: N concurrent first requests for one tenant produce N fetches.
|
|
228
|
+
|
|
229
|
+
Suggested direction: cache the in-flight promise rather than the resolved value.""",
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
'id': 'arch-1',
|
|
233
|
+
'focus': 'architecture',
|
|
234
|
+
'domain': 'architecture',
|
|
235
|
+
'file': 'src/settings/loader.ts',
|
|
236
|
+
'line': 10,
|
|
237
|
+
'severity': 'medium',
|
|
238
|
+
'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
239
|
+
'score': 5,
|
|
240
|
+
'title': 'The loader owns the cache rather than being handed one',
|
|
241
|
+
'description': """Observation: load() calls the cache module's free functions directly (src/settings/loader.ts:10).
|
|
242
|
+
|
|
243
|
+
Why it matters: no caller can swap the policy, and the loader cannot be tested without the module-global store.
|
|
244
|
+
|
|
245
|
+
Suggested direction: take the cache as a parameter.""",
|
|
246
|
+
},
|
|
247
|
+
{
|
|
248
|
+
'id': 'perf-1',
|
|
249
|
+
'focus': 'performance',
|
|
250
|
+
'domain': 'performance',
|
|
251
|
+
'file': 'src/settings/cache.ts',
|
|
252
|
+
'line': 12,
|
|
253
|
+
'severity': 'low',
|
|
254
|
+
'risk': {'impact': 'low', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
|
|
255
|
+
'score': 3,
|
|
256
|
+
'title': 'clear() evicts every tenant, not the one whose settings changed',
|
|
257
|
+
'description': """Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.
|
|
258
|
+
|
|
259
|
+
Why it matters: one tenant's change flushes the entry for every tenant.
|
|
260
|
+
|
|
261
|
+
Suggested direction: add delete(tenantId) and leave clear() for shutdown.""",
|
|
262
|
+
},
|
|
263
|
+
{
|
|
264
|
+
'id': 'smell-1',
|
|
265
|
+
'focus': 'code-smells',
|
|
266
|
+
'domain': 'code-smells',
|
|
267
|
+
'file': 'src/settings/cache.ts',
|
|
268
|
+
'line': 8,
|
|
269
|
+
'severity': 'low',
|
|
270
|
+
'risk': {'impact': 'low', 'likelihood': 'unlikely', 'confidence': 'medium', 'action': 'optional'},
|
|
271
|
+
'score': 2,
|
|
272
|
+
'title': 'get() hands back the stored object, so a caller can mutate the cache',
|
|
273
|
+
'description': """Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).
|
|
274
|
+
|
|
275
|
+
Why it matters: a caller that edits the returned settings edits every later reader's copy.
|
|
276
|
+
|
|
277
|
+
Suggested direction: freeze the value on put, or return a copy.""",
|
|
278
|
+
},
|
|
279
|
+
]
|
|
280
|
+
|
|
281
|
+
# The critic kept the three above its bar and dropped the two below it.
|
|
282
|
+
KEPT = {'security-1', 'bugs-1', 'arch-1'}
|
|
283
|
+
|
|
284
|
+
def without(finding, *keys):
|
|
285
|
+
return {k: v for k, v in finding.items() if k not in keys}
|
|
286
|
+
|
|
287
|
+
findings_dir = run / 'findings'
|
|
288
|
+
findings_dir.mkdir(parents=True, exist_ok=True)
|
|
289
|
+
for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
|
|
290
|
+
mine = [without(f, 'focus', 'score') for f in FINDINGS if f['focus'] == focus]
|
|
291
|
+
(findings_dir / f'{focus}.json').write_text(json.dumps(mine, indent=2) + '\n')
|
|
292
|
+
(findings_dir / 'tests.json').write_text('[]\n')
|
|
293
|
+
|
|
294
|
+
(run / 'findings.deduped.json').write_text(
|
|
295
|
+
json.dumps([without(f, 'focus') for f in FINDINGS], indent=2) + '\n'
|
|
296
|
+
)
|
|
297
|
+
(run / 'findings.kept.json').write_text(
|
|
298
|
+
json.dumps([without(f, 'focus', 'score') for f in FINDINGS if f['id'] in KEPT], indent=2) + '\n'
|
|
299
|
+
)
|
|
300
|
+
FINDINGS
|
|
301
|
+
|
|
302
|
+
echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
|
|
303
|
+
|
|
304
|
+
cat > "$RUN_DIR/brief.json" <<'JSON'
|
|
305
|
+
{
|
|
306
|
+
"summary": "Adds a process-global cache in front of tenant settings loads.",
|
|
307
|
+
"riskAreas": ["tenant isolation", "cache invalidation"],
|
|
308
|
+
"conventions": []
|
|
309
|
+
}
|
|
310
|
+
JSON
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: codex-missing-falls-back
|
|
3
|
+
description: With codex off the machine, peer review still runs. A Claude subagent stands in, carrying the independence preamble, and the stage never logs an error.
|
|
4
|
+
tags: [magpie, peer-review, fallback, scaffold]
|
|
5
|
+
max_turns: 40
|
|
6
|
+
timeout_seconds: 1200
|
|
7
|
+
allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash, Task]
|
|
8
|
+
expected_outcome: Peer review runs on the Claude path with the preamble prepended, logs provider claude and no error, and findings.final.json is written.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
The magpie run on PR 1337 is parked just after the critic stage. Carry on with it.
|