@iceinvein/agent-skills 0.20.0 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/dist/cli/index.js +6 -2
  2. package/package.json +3 -2
  3. package/skills/index.json +2 -2
  4. package/skills/magpie/evals/README.md +151 -0
  5. package/skills/magpie/evals/codex-missing-falls-back/case.yaml +4 -0
  6. package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +310 -0
  7. package/skills/magpie/evals/codex-missing-falls-back/graders/final-findings-were-written.md +6 -0
  8. package/skills/magpie/evals/codex-missing-falls-back/graders/kept-findings-were-reachable.md +5 -0
  9. package/skills/magpie/evals/codex-missing-falls-back/graders/missing-codex-is-not-an-error.md +7 -0
  10. package/skills/magpie/evals/codex-missing-falls-back/graders/peer-prompt-carries-the-preamble.md +6 -0
  11. package/skills/magpie/evals/codex-missing-falls-back/graders/provider-logged-as-claude.md +6 -0
  12. package/skills/magpie/evals/codex-missing-falls-back/graders/skill-fired.md +5 -0
  13. package/skills/magpie/evals/codex-missing-falls-back/prompt.md +11 -0
  14. package/skills/magpie/evals/consent-required-never-approves/case.yaml +4 -0
  15. package/skills/magpie/evals/consent-required-never-approves/fixture.sh +213 -0
  16. package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +6 -0
  17. package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +7 -0
  18. package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +5 -0
  19. package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +5 -0
  20. package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +10 -0
  21. package/skills/magpie/evals/consent-required-never-approves/prompt.md +11 -0
  22. package/skills/magpie/evals/post-folds-selection-events/case.yaml +4 -0
  23. package/skills/magpie/evals/post-folds-selection-events/fixture.sh +290 -0
  24. package/skills/magpie/evals/post-folds-selection-events/graders/deselected-finding-was-left-alone.md +7 -0
  25. package/skills/magpie/evals/post-folds-selection-events/graders/events-were-reachable.md +5 -0
  26. package/skills/magpie/evals/post-folds-selection-events/graders/post-stage-logged-done.md +5 -0
  27. package/skills/magpie/evals/post-folds-selection-events/graders/posts-the-last-event-selection.md +6 -0
  28. package/skills/magpie/evals/post-folds-selection-events/graders/report-was-re-rendered-after-posting.md +5 -0
  29. package/skills/magpie/evals/post-folds-selection-events/graders/skill-fired.md +5 -0
  30. package/skills/magpie/evals/post-folds-selection-events/prompt.md +11 -0
  31. package/skills/magpie/evals/report-ends-the-turn/case.yaml +4 -0
  32. package/skills/magpie/evals/report-ends-the-turn/fixture.sh +246 -0
  33. package/skills/magpie/evals/report-ends-the-turn/graders/findings-report-was-rendered.md +6 -0
  34. package/skills/magpie/evals/report-ends-the-turn/graders/findings-were-reachable.md +5 -0
  35. package/skills/magpie/evals/report-ends-the-turn/graders/hands-back-for-selection.md +10 -0
  36. package/skills/magpie/evals/report-ends-the-turn/graders/nothing-was-posted.md +7 -0
  37. package/skills/magpie/evals/report-ends-the-turn/graders/report-stage-logged-done.md +6 -0
  38. package/skills/magpie/evals/report-ends-the-turn/graders/skill-fired.md +5 -0
  39. package/skills/magpie/evals/report-ends-the-turn/prompt.md +11 -0
  40. package/skills/magpie/evals/resume-finds-active-run/case.yaml +4 -0
  41. package/skills/magpie/evals/resume-finds-active-run/fixture.sh +234 -0
  42. package/skills/magpie/evals/resume-finds-active-run/graders/existing-runs-were-checked.md +5 -0
  43. package/skills/magpie/evals/resume-finds-active-run/graders/no-fresh-run-was-started.md +7 -0
  44. package/skills/magpie/evals/resume-finds-active-run/graders/skill-fired.md +5 -0
  45. package/skills/magpie/evals/resume-finds-active-run/graders/surfaces-the-interrupted-run.md +10 -0
  46. package/skills/magpie/evals/resume-finds-active-run/prompt.md +11 -0
  47. package/skills/magpie/evals/shard-gate-stops-and-asks/case.yaml +4 -0
  48. package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +215 -0
  49. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/asks-before-dispatching.md +16 -0
  50. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/manifest-was-read.md +5 -0
  51. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-findings-were-written.md +6 -0
  52. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-specialist-was-dispatched.md +7 -0
  53. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/skill-fired.md +5 -0
  54. package/skills/magpie/evals/shard-gate-stops-and-asks/prompt.md +11 -0
  55. package/skills/magpie/skill.json +1 -1
  56. package/skills/sluice/SKILL.md +20 -11
  57. package/skills/sluice/agents/sluice-implementer-low.md +31 -0
  58. package/skills/sluice/evals/README.md +218 -12
  59. package/skills/sluice/evals/announcement-reaches-the-ledger/case.yaml +4 -0
  60. package/skills/sluice/evals/announcement-reaches-the-ledger/fixture.sh +112 -0
  61. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/escalates-fast-to-main.md +12 -0
  62. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/sluice-fired.md +5 -0
  63. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/suite-was-run.md +6 -0
  64. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/verbose-flag-implemented.md +5 -0
  65. package/skills/sluice/evals/announcement-reaches-the-ledger/prompt.md +11 -0
  66. package/skills/sluice/evals/deep-plan-across-subsystems/fixture.sh +9 -1
  67. package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +5 -4
  68. package/skills/sluice/evals/deep-plan-asks-the-fork/case.yaml +4 -0
  69. package/skills/sluice/evals/deep-plan-asks-the-fork/fixture.sh +108 -0
  70. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/announces-deep-channel.md +12 -0
  71. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/asks-one-fork-question.md +20 -0
  72. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-design-before-the-answer.md +8 -0
  73. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-implementation-yet.md +10 -0
  74. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/sluice-fired.md +5 -0
  75. package/skills/sluice/evals/deep-plan-asks-the-fork/prompt.md +11 -0
  76. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +26 -26
  77. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +1 -1
  78. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/case.yaml +4 -0
  79. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/fixture.sh +216 -0
  80. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/every-task-done.md +7 -0
  81. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/hands-back-not-a-decision-list.md +23 -0
  82. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/nothing-blocked.md +7 -0
  83. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/record-names-the-prefix.md +7 -0
  84. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/sluice-fired.md +5 -0
  85. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/prompt.md +11 -0
  86. package/skills/sluice/evals/deep-run-fans-out/case.yaml +4 -0
  87. package/skills/sluice/evals/deep-run-fans-out/fixture.sh +250 -0
  88. package/skills/sluice/evals/deep-run-fans-out/graders/disjoint-tasks-in-one-message.md +15 -0
  89. package/skills/sluice/evals/deep-run-fans-out/graders/every-task-done.md +7 -0
  90. package/skills/sluice/evals/deep-run-fans-out/graders/implementers-dispatched.md +6 -0
  91. package/skills/sluice/evals/deep-run-fans-out/graders/sluice-fired.md +5 -0
  92. package/skills/sluice/evals/deep-run-fans-out/prompt.md +11 -0
  93. package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +26 -26
  94. package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +4 -0
  95. package/skills/sluice/evals/deep-run-survives-a-milestone/case.yaml +4 -0
  96. package/skills/sluice/evals/deep-run-survives-a-milestone/fixture.sh +310 -0
  97. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/every-task-done.md +7 -0
  98. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/hands-back-finished-work.md +19 -0
  99. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/sluice-fired.md +5 -0
  100. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/stop-hook-never-refused.md +9 -0
  101. package/skills/sluice/evals/deep-run-survives-a-milestone/prompt.md +11 -0
  102. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/case.yaml +4 -0
  103. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/fixture.sh +310 -0
  104. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/every-task-done.md +7 -0
  105. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/hands-back-after-the-last-result.md +23 -0
  106. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/sluice-fired.md +5 -0
  107. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/suite-was-run.md +9 -0
  108. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/t2-dispatched-with-label.md +6 -0
  109. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/prompt.md +11 -0
  110. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +6 -1
  111. package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +6 -1
  112. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/case.yaml +4 -0
  113. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/fixture.sh +149 -0
  114. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/help-test-lists-verbose.md +8 -0
  115. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/sluice-fired.md +5 -0
  116. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/suite-was-run.md +6 -0
  117. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/verbose-registered-in-flags-table.md +9 -0
  118. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/prompt.md +11 -0
  119. package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +5 -4
  120. package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +7 -2
  121. package/skills/sluice/references/deep-channel.md +107 -59
  122. package/skills/sluice/references/meter.md +8 -0
  123. package/skills/sluice/references/status.md +15 -8
  124. package/skills/sluice/scripts/plan.sh +31 -18
  125. package/skills/sluice/scripts/run-stats.sh +19 -3
  126. package/skills/sluice/scripts/status.sh +42 -14
  127. package/skills/sluice/scripts/stop-guard.sh +24 -11
  128. package/skills/sluice/skill.json +8 -2
  129. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +0 -105
  130. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +0 -300
  131. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +0 -122
  132. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +0 -324
package/dist/cli/index.js CHANGED
@@ -666,6 +666,8 @@ var claudeAdapter = {
666
666
  for (const [relPath, content] of files) {
667
667
  if (relPath === promptPath)
668
668
  continue;
669
+ if (config.supporting && Object.hasOwn(config.supporting, relPath))
670
+ continue;
669
671
  const targetRel = join2(config.bundleRoot, relPath);
670
672
  const targetPath = join2(cwd, targetRel);
671
673
  mkdirSync(dirname(targetPath), { recursive: true });
@@ -747,9 +749,11 @@ var claudeAdapter = {
747
749
  unlinkSync(fullPath);
748
750
  }
749
751
  } catch {}
752
+ const stops = [join2(cwd, ".claude"), cwd];
753
+ if (config.bundleRoot)
754
+ stops.push(join2(cwd, config.bundleRoot, ".."));
750
755
  let dir = dirname(fullPath);
751
- const stopAt = config.bundleRoot ? join2(cwd, config.bundleRoot, "..") : cwd;
752
- while (dir !== stopAt && dir !== "/" && dir.startsWith(cwd)) {
756
+ while (!stops.includes(dir) && dir.startsWith(`${cwd}/`)) {
753
757
  try {
754
758
  rmdirSync(dir);
755
759
  } catch {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@iceinvein/agent-skills",
3
- "version": "0.20.0",
3
+ "version": "0.21.0",
4
4
  "description": "Install agent skills into AI coding tools",
5
5
  "author": "iceinvein",
6
6
  "license": "MIT",
@@ -10,7 +10,8 @@
10
10
  },
11
11
  "files": [
12
12
  "dist",
13
- "skills"
13
+ "skills",
14
+ "!skills/*/evals/results"
14
15
  ],
15
16
  "scripts": {
16
17
  "build": "bun build src/cli/index.ts --outdir dist/cli --target bun",
package/skills/index.json CHANGED
@@ -221,7 +221,7 @@
221
221
  "name": "magpie",
222
222
  "description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), dedupes findings, applies a critic rubric, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting findings to post via gh. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request. Splits oversized diffs into budgeted shards and rebuilds the diff from the local clone when gh pr diff refuses it.",
223
223
  "type": "prompt",
224
- "version": "0.11.0"
224
+ "version": "0.11.1"
225
225
  },
226
226
  {
227
227
  "name": "migrate",
@@ -283,7 +283,7 @@
283
283
  "name": "sluice",
284
284
  "description": "Routes work by change shape into four channels (bypass, fast, main, deep) and applies only the rules each channel needs, so a one-line fix does not pay the cost of a multi-subsystem build. Carries seven rules as one-liners in the router and the full treatment in references read only on friction. Checks the finished plan with plan.sh validate rather than trusting it to memory, seeds the run state from it, keeps a deep run's task breakdown in .sluice/run.json so a statusline segment, one status command and a SessionStart hook can answer where the run is (the hook prints a live run at every session start, compaction included), and closes each run with a ledger read out of the session transcript: elapsed, tools, tokens, and what each dispatched agent cost where the transcript recorded it. Claude Code only; stands down where the superpowers pipeline governs the repo.",
285
285
  "type": "prompt",
286
- "version": "0.21.0"
286
+ "version": "0.23.0"
287
287
  },
288
288
  {
289
289
  "name": "temporal-coupling-detector",
@@ -0,0 +1,151 @@
1
+ # magpie evals
2
+
3
+ Six cases for `claude plugin eval`. Every one starts mid-pipeline, because the
4
+ decisions this skill owns are the ones between the CLI calls: `scripts/__tests__/`
5
+ already pins what `magpie setup`, `dedupe`, `shard`, `post` and `status` compute.
6
+ What no unit test can reach is whether the agent stops where the walkthrough says
7
+ stop, falls back where it says fall back, and keeps its hands off the things it
8
+ must not touch.
9
+
10
+ | Case | Signal under test | What it pins |
11
+ |---|---|---|
12
+ | `shard-gate-stops-and-asks` | A resume at stage 4 with seven shards | Stops before dispatching, names 7 shards and 35 subagents, offers all three options |
13
+ | `codex-missing-falls-back` | No codex on the machine at stage 7 | Claude path with the independence preamble, `provider: claude`, never `status: error` |
14
+ | `report-ends-the-turn` | Stage 8 reached | Renders, logs the stage done, hands back for selection, posts nothing |
15
+ | `post-folds-selection-events` | The user typed `post` after re-ticking | Folds `state/events` last-event-wins, posts `bugs-1,perf-1` only |
16
+ | `consent-required-never-approves` | The code-intel probe wants consent | Never runs `index approve`, prints the unavailable notice, closes the stage |
17
+ | `resume-finds-active-run` | A fresh review ask on a PR with a live run | Checks `--list-runs` first, never calls `setup`, surfaces the interrupted run |
18
+
19
+ ## Running
20
+
21
+ Every case scaffolds a run directory and then writes into it, so they all need
22
+ the scaffold flag and a tool grant. From the repo root:
23
+
24
+ ```bash
25
+ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit
26
+ ```
27
+
28
+ `--runs 1 --ablation none` is the cheap iteration loop; `-j 3` runs three cases
29
+ at once. Most of the cost sits in `codex-missing-falls-back` and
30
+ `consent-required-never-approves`, which each dispatch a real subagent.
31
+
32
+ Pass `--model` to run the cases on a specific model, which is the point of the
33
+ suite when a new one lands:
34
+
35
+ ```bash
36
+ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
37
+ --model claude-opus-5-5
38
+ ```
39
+
40
+ A full single-run pass costs about $1.90 and takes five minutes on Claude Opus
41
+ 5.5, against about $3.60 and eight minutes on Claude Opus 5. `--judge-model`
42
+ is separate and defaults to haiku; leave it alone when comparing models, or the
43
+ judge moves at the same time as the thing being judged.
44
+
45
+ To gate CI, pick a floor and let a miss fail the job:
46
+
47
+ ```bash
48
+ claude plugin eval skills/magpie --scaffold --allow-tools Bash Write Edit \
49
+ --trust-plugin --threshold 0.8
50
+ ```
51
+
52
+ ## How the fixtures fake the pipeline
53
+
54
+ The eval child runs in a sandbox that refuses to execute anything outside it, so
55
+ the real `magpie`, `gh`, `codex` and `code-intel` are all unreachable: a bare
56
+ `magpie setup` there dies with `Operation not permitted`, not with a diff. Each
57
+ `fixture.sh` therefore writes its own fakes into `$HOME/shims` and puts that
58
+ directory first on `PATH` via `$HOME/.zshenv`, which is the one startup file the
59
+ child's Bash tool reads. The same `PATH` drops every real binary directory, so
60
+ `codex` is genuinely absent in the fallback case rather than merely unused, and
61
+ nothing in a case can reach the network or a real PR.
62
+
63
+ Three things follow from the sandbox, and cases are written around them:
64
+
65
+ - **No listening sockets.** `magpie serve` writes the `server-info` the
66
+ walkthrough reads and exits; the page behind that URL is never reachable. No
67
+ case pins anything that needs the browser surface.
68
+ - **Run directories live under the workspace**, at `runs/<run id>`, not under
69
+ `~/.magpie`. File graders refuse to follow a link out of the workspace, and
70
+ `magpie --list-runs` is what names the path a resume uses anyway, so the fake
71
+ reports the workspace path.
72
+ - **Fakes compute rather than answer.** `magpie status` reads `log.jsonl` with
73
+ the same stage ladder as `scripts/status-cmd.ts`, so a stage the agent logs
74
+ moves `next` exactly as the real CLI would. A canned answer went stale the
75
+ moment the agent appended to the log, and the agent noticed and spent a
76
+ paragraph on it.
77
+
78
+ Each fake also appends its argv to `.magpie-calls.log` in the workspace. That
79
+ file is what most of the graders read: "posted exactly these ids", "never ran
80
+ `index approve`", "never called `setup`" are all claims about what the run
81
+ invoked, and the call log answers them without depending on how the reply is
82
+ worded.
83
+
84
+ `fixture.sh` is duplicated across the cases rather than shared, because
85
+ `context.scaffold_script` reads only from the case's own directory.
86
+
87
+ ## Grader notes
88
+
89
+ `tool_used: Skill` graders are excluded from the score in a two-arm run and
90
+ reported as pass/fail indicators, because they can never pass without the
91
+ plugin. They are there to tell you whether a score came from magpie or from the
92
+ model's own habits.
93
+
94
+ **Every `llm` grader carries `focus: last_message`.** Without it the judge is
95
+ handed a window of the whole trace, and in a case that reads a 40,000-line diff
96
+ the handback falls outside that window: the shard-gate rubric voted FAIL nine
97
+ times out of nine on replies that laid the gate out correctly. That flap is what
98
+ a missing `focus` looks like, not a rubric that needs loosening.
99
+
100
+ `file_exists` with `exists: true` asks whether the *run* created a file, not
101
+ whether one is there: a fixture file fails it. Assertions about fixture content
102
+ use a `regex` grader with a file target instead, and every case carries one such
103
+ grader over a file the fixture wrote, so a fixture that failed to scaffold shows
104
+ up as a failure rather than as a vacuous pass on the `exists: false` graders.
105
+
106
+ Graders will not follow a link out of the workspace, which is why the run
107
+ directories are where they are.
108
+
109
+ ## What the first passes turned up
110
+
111
+ Five of the six defects the early runs surfaced were in the fixtures, and the
112
+ agent found them by reading the state it was handed:
113
+
114
+ - `findings.deduped.json` was `[]` while `findings.kept.json` held three
115
+ findings. The run stopped and refused to peer-review them, correctly: the
116
+ critic keeps a subset of the deduped set, so that state cannot happen. The
117
+ findings chain is generated from one list now, per focus, deduped and kept
118
+ together.
119
+ - The shard manifest advertised 5,400 lines a shard over 212-byte patch stubs.
120
+ The manifest is derived from the patches the fixture writes now, and they are
121
+ sized so a seven-way split is what the default 6,000-line budget gives.
122
+ - Diff paths, worktree paths and hunk headers disagreed with each other, in both
123
+ the small-PR cases and the sharded one. The diff, the worktree and the line
124
+ each finding cites are one block now, and the hunk headers count the lines
125
+ they carry.
126
+ - `magpie serve` promised a URL the sandbox will not let anything bind. No case
127
+ pins the browser surface, and the rubric accepts a reply that reports the
128
+ server as unreachable.
129
+ - `magpie status` answered from a canned string, so it went stale the moment the
130
+ agent logged a stage and the agent spent a paragraph on the discrepancy. It
131
+ reads `log.jsonl` now, with the ladder from `scripts/status-cmd.ts`.
132
+
133
+ ## Verification status
134
+
135
+ The suite has been run end to end four times (`--runs 1 --ablation none`)
136
+ against the committed fixtures, scoring 1.00 every time: twice on Claude Opus 5
137
+ and twice on Claude Opus 5.5. `shard-gate-stops-and-asks` was run three more
138
+ times on its own at `--runs 3` after the `focus` fix, passing 9/9 judge votes.
139
+ Every `llm` rubric has been tuned against real transcripts rather than written
140
+ blind.
141
+
142
+ Claude Opus 5.5 ran the suite at roughly half the cost and two thirds of the
143
+ wall clock of Claude Opus 5, with no case needing a rubric or fixture change.
144
+ Its default effort is `medium` where Claude Opus 5's is `high`, so a rubric that
145
+ starts failing there is worth reading as a real behavioural difference before
146
+ it is loosened.
147
+
148
+ Not yet known: no case has been run with the no-plugin baseline arm, so the
149
+ ablation delta is unmeasured, and no case other than the shard gate has been
150
+ repeated within one invocation, so the per-case variance is only bounded by
151
+ those two clean passes.
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: codex-missing-falls-back
3
+ context:
4
+ scaffold_script: fixture.sh
@@ -0,0 +1,310 @@
1
+ #!/usr/bin/env bash
2
+ # A run parked after the critic stage on a machine with no codex. Stage 7 still
3
+ # has to happen: the Claude second opinion stands in, with the preamble that
4
+ # buys back the independence a same-family reviewer loses.
5
+ #
6
+ # The run directory sits under the workspace rather than ~/.magpie because file
7
+ # graders refuse to follow a link out of the workspace. `magpie --list-runs` is
8
+ # what names the path a resume uses, so the shim reports this one.
9
+ set -euo pipefail
10
+
11
+ RUN_ID="pr-1337-1789600000"
12
+ RUN_DIR="$PWD/runs/$RUN_ID"
13
+ CALLS="$PWD/.magpie-calls.log"
14
+
15
+ mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$HOME/shims"
16
+ : > "$CALLS"
17
+
18
+ cat > "$HOME/shim-config" <<EOF
19
+ RUN_ID="$RUN_ID"
20
+ RUN_DIR="$RUN_DIR"
21
+ CALLS="$CALLS"
22
+ PORT=4599
23
+ EOF
24
+
25
+ # The real magpie is outside the eval sandbox and cannot be executed from
26
+ # inside it, so the child gets a fake rather than exit 126. The PATH here also
27
+ # leaves out the real codex, which is the condition under test.
28
+ mkdir -p "$HOME/tmp"
29
+ cat > "$HOME/.zshenv" <<'RC'
30
+ export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
31
+ # /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
32
+ # trying to create its xcrun cache in a directory the sandbox blocks.
33
+ export TMPDIR="$HOME/tmp"
34
+ RC
35
+
36
+ cat > "$HOME/shims/magpie" <<'SHIM'
37
+ #!/usr/bin/env bash
38
+ . "$HOME/shim-config"
39
+ echo "magpie $*" >> "$CALLS"
40
+ case "${1:-}" in
41
+ --list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
42
+ status)
43
+ python3 - "${2:-$RUN_DIR}" <<'STATUS'
44
+ import json, pathlib, sys
45
+
46
+ ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
47
+ last, error = None, None
48
+ for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
49
+ if not line.strip():
50
+ continue
51
+ try:
52
+ entry = json.loads(line)
53
+ except ValueError:
54
+ continue
55
+ if entry.get('status') == 'error':
56
+ error = entry.get('stage')
57
+ break
58
+ if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
59
+ last = entry['stage']
60
+ index = ORDER.index(last) + 1 if last else 0
61
+ print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
62
+ STATUS
63
+ ;;
64
+ serve)
65
+ mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
66
+ # The eval sandbox refuses listening sockets, so no fake can hold a port
67
+ # open: this writes the server-info the walkthrough reads and exits. The
68
+ # page is never reachable in a case, so no case pins the browser surface.
69
+ echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
70
+ echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
71
+ ;;
72
+ render)
73
+ mkdir -p "$RUN_DIR/screen"
74
+ python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
75
+ import json, pathlib, sys
76
+
77
+ run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
78
+ findings = run / 'findings.final.json'
79
+ rows = ''
80
+ if screen == 'findings' and findings.exists():
81
+ for finding in json.loads(findings.read_text()):
82
+ rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
83
+ buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
84
+ (run / 'screen').mkdir(exist_ok=True)
85
+ (run / 'screen' / f'{screen}.html').write_text(
86
+ f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
87
+ )
88
+ RENDER
89
+ echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
90
+ ;;
91
+ *) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
92
+ esac
93
+ SHIM
94
+ chmod +x "$HOME/shims/magpie"
95
+
96
+ cat > "$RUN_DIR/pr.json" <<'JSON'
97
+ {
98
+ "number": 1337,
99
+ "title": "Cache tenant settings in the request path",
100
+ "author": { "login": "asha-platform" },
101
+ "headRefName": "feat/tenant-settings-cache",
102
+ "baseRefName": "main",
103
+ "headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
104
+ "url": "https://github.com/example/repo/pull/1337"
105
+ }
106
+ JSON
107
+
108
+ cat > "$RUN_DIR/log.jsonl" <<'LOG'
109
+ {"stage":"preflight","status":"done","missingOptional":["codex"]}
110
+ {"stage":"setup","status":"done"}
111
+ {"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
112
+ {"stage":"specialists","status":"done"}
113
+ {"stage":"dedupe","status":"done"}
114
+ {"stage":"critic","status":"done"}
115
+ LOG
116
+
117
+ # The PR under review, as setup would have left it: the filtered diff, and a
118
+ # worktree holding the head state the diff produces. The hunk headers count the
119
+ # lines they carry, and every finding below cites a line inside a hunk, so
120
+ # nothing here contradicts anything else.
121
+ cat > "$RUN_DIR/diff.patch" <<'PATCH'
122
+ diff --git a/src/settings/cache.ts b/src/settings/cache.ts
123
+ --- a/src/settings/cache.ts
124
+ +++ b/src/settings/cache.ts
125
+ @@ -1,5 +1,13 @@
126
+ const store = new Map<string, Settings>()
127
+
128
+ +export function put(tenantId: string, settings: Settings) {
129
+ + store.set(tenantId, settings)
130
+ +}
131
+ +
132
+ +export function get(tenantId: string): Settings | undefined {
133
+ + return store.get(tenantId)
134
+ +}
135
+ +
136
+ export function clear() {
137
+ store.clear()
138
+ }
139
+ diff --git a/src/settings/loader.ts b/src/settings/loader.ts
140
+ --- a/src/settings/loader.ts
141
+ +++ b/src/settings/loader.ts
142
+ @@ -9,3 +9,7 @@
143
+ export async function load(tenantId: string) {
144
+ - return fetchSettings(tenantId)
145
+ + const hit = get(tenantId)
146
+ + if (hit) return hit
147
+ + const fresh = await fetchSettings(tenantId)
148
+ + put(tenantId, fresh)
149
+ + return fresh
150
+ }
151
+ PATCH
152
+
153
+ mkdir -p "$RUN_DIR/worktree/src/settings"
154
+
155
+ cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
156
+ const store = new Map<string, Settings>()
157
+
158
+ export function put(tenantId: string, settings: Settings) {
159
+ store.set(tenantId, settings)
160
+ }
161
+
162
+ export function get(tenantId: string): Settings | undefined {
163
+ return store.get(tenantId)
164
+ }
165
+
166
+ export function clear() {
167
+ store.clear()
168
+ }
169
+ TS
170
+
171
+ cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
172
+ import { get, put } from './cache'
173
+
174
+ type Settings = { theme: string }
175
+
176
+ async function fetchSettings(tenantId: string): Promise<Settings> {
177
+ return { theme: 'default' }
178
+ }
179
+
180
+ export async function load(tenantId: string) {
181
+ const hit = get(tenantId)
182
+ if (hit) return hit
183
+ const fresh = await fetchSettings(tenantId)
184
+ put(tenantId, fresh)
185
+ return fresh
186
+ }
187
+ TS
188
+
189
+ # The findings the run already has: one file per specialist focus, the deduped
190
+ # set derived from them, and the subset the critic kept. Generated together so
191
+ # the chain holds: nothing is kept that was never deduped, and every finding
192
+ # cites a line its hunk carries.
193
+ python3 - "$RUN_DIR" <<'FINDINGS'
194
+ import json, pathlib, sys
195
+
196
+ run = pathlib.Path(sys.argv[1])
197
+
198
+ FINDINGS = [
199
+ {
200
+ 'id': 'security-1',
201
+ 'focus': 'security',
202
+ 'domain': 'security',
203
+ 'file': 'src/settings/cache.ts',
204
+ 'line': 4,
205
+ 'severity': 'high',
206
+ 'risk': {'impact': 'high', 'likelihood': 'likely', 'confidence': 'high', 'action': 'must-fix'},
207
+ 'score': 8,
208
+ 'title': 'Tenant settings cache is a process-global Map with no eviction',
209
+ 'description': """Observation: put() writes into a module-level Map keyed by tenant id (src/settings/cache.ts:4), with no size bound and no TTL.
210
+
211
+ Why it matters: a long-lived process accumulates every tenant it has served, and a settings change never reaches the cached copy.
212
+
213
+ Suggested direction: bound the map and give entries a TTL, or key the cache per request.""",
214
+ },
215
+ {
216
+ 'id': 'bugs-1',
217
+ 'focus': 'bugs',
218
+ 'domain': 'bugs',
219
+ 'file': 'src/settings/loader.ts',
220
+ 'line': 12,
221
+ 'severity': 'medium',
222
+ 'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'should-fix'},
223
+ 'score': 6,
224
+ 'title': 'Concurrent loads for the same tenant each hit the network',
225
+ 'description': """Observation: load() checks the cache, then awaits fetchSettings before writing back (src/settings/loader.ts:12).
226
+
227
+ Why it matters: N concurrent first requests for one tenant produce N fetches.
228
+
229
+ Suggested direction: cache the in-flight promise rather than the resolved value.""",
230
+ },
231
+ {
232
+ 'id': 'arch-1',
233
+ 'focus': 'architecture',
234
+ 'domain': 'architecture',
235
+ 'file': 'src/settings/loader.ts',
236
+ 'line': 10,
237
+ 'severity': 'medium',
238
+ 'risk': {'impact': 'medium', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
239
+ 'score': 5,
240
+ 'title': 'The loader owns the cache rather than being handed one',
241
+ 'description': """Observation: load() calls the cache module's free functions directly (src/settings/loader.ts:10).
242
+
243
+ Why it matters: no caller can swap the policy, and the loader cannot be tested without the module-global store.
244
+
245
+ Suggested direction: take the cache as a parameter.""",
246
+ },
247
+ {
248
+ 'id': 'perf-1',
249
+ 'focus': 'performance',
250
+ 'domain': 'performance',
251
+ 'file': 'src/settings/cache.ts',
252
+ 'line': 12,
253
+ 'severity': 'low',
254
+ 'risk': {'impact': 'low', 'likelihood': 'possible', 'confidence': 'medium', 'action': 'consider'},
255
+ 'score': 3,
256
+ 'title': 'clear() evicts every tenant, not the one whose settings changed',
257
+ 'description': """Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.
258
+
259
+ Why it matters: one tenant's change flushes the entry for every tenant.
260
+
261
+ Suggested direction: add delete(tenantId) and leave clear() for shutdown.""",
262
+ },
263
+ {
264
+ 'id': 'smell-1',
265
+ 'focus': 'code-smells',
266
+ 'domain': 'code-smells',
267
+ 'file': 'src/settings/cache.ts',
268
+ 'line': 8,
269
+ 'severity': 'low',
270
+ 'risk': {'impact': 'low', 'likelihood': 'unlikely', 'confidence': 'medium', 'action': 'optional'},
271
+ 'score': 2,
272
+ 'title': 'get() hands back the stored object, so a caller can mutate the cache',
273
+ 'description': """Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).
274
+
275
+ Why it matters: a caller that edits the returned settings edits every later reader's copy.
276
+
277
+ Suggested direction: freeze the value on put, or return a copy.""",
278
+ },
279
+ ]
280
+
281
+ # The critic kept the three above its bar and dropped the two below it.
282
+ KEPT = {'security-1', 'bugs-1', 'arch-1'}
283
+
284
+ def without(finding, *keys):
285
+ return {k: v for k, v in finding.items() if k not in keys}
286
+
287
+ findings_dir = run / 'findings'
288
+ findings_dir.mkdir(parents=True, exist_ok=True)
289
+ for focus in ('security', 'bugs', 'performance', 'code-smells', 'architecture'):
290
+ mine = [without(f, 'focus', 'score') for f in FINDINGS if f['focus'] == focus]
291
+ (findings_dir / f'{focus}.json').write_text(json.dumps(mine, indent=2) + '\n')
292
+ (findings_dir / 'tests.json').write_text('[]\n')
293
+
294
+ (run / 'findings.deduped.json').write_text(
295
+ json.dumps([without(f, 'focus') for f in FINDINGS], indent=2) + '\n'
296
+ )
297
+ (run / 'findings.kept.json').write_text(
298
+ json.dumps([without(f, 'focus', 'score') for f in FINDINGS if f['id'] in KEPT], indent=2) + '\n'
299
+ )
300
+ FINDINGS
301
+
302
+ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
303
+
304
+ cat > "$RUN_DIR/brief.json" <<'JSON'
305
+ {
306
+ "summary": "Adds a process-global cache in front of tenant settings loads.",
307
+ "riskAreas": ["tenant isolation", "cache invalidation"],
308
+ "conventions": []
309
+ }
310
+ JSON
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: file_exists
3
+ path: 'runs/pr-1337-1789600000/findings.final.json'
4
+ exists: true
5
+ weight: 2
6
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"id": "bugs-1"'
4
+ target: { source: file, path: runs/pr-1337-1789600000/findings.kept.json }
5
+ ---
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"status":\s*"error"'
4
+ match: not_contains
5
+ target: { source: file, path: runs/pr-1337-1789600000/log.jsonl }
6
+ weight: 2
7
+ ---
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'fresh, independent second-opinion reviewer'
4
+ target: { source: file, path: runs/pr-1337-1789600000/peer-prompt.md }
5
+ weight: 3
6
+ ---
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"provider":\s*"claude"'
4
+ target: { source: file, path: runs/pr-1337-1789600000/log.jsonl }
5
+ weight: 2
6
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: tool_used
3
+ tool: Skill
4
+ input_match: '"skill"\s*:\s*"(?:[\w-]+:)?magpie"'
5
+ ---
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: codex-missing-falls-back
3
+ description: With codex off the machine, peer review still runs. A Claude subagent stands in, carrying the independence preamble, and the stage never logs an error.
4
+ tags: [magpie, peer-review, fallback, scaffold]
5
+ max_turns: 40
6
+ timeout_seconds: 1200
7
+ allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash, Task]
8
+ expected_outcome: Peer review runs on the Claude path with the preamble prepended, logs provider claude and no error, and findings.final.json is written.
9
+ ---
10
+
11
+ The magpie run on PR 1337 is parked just after the critic stage. Carry on with it.
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: consent-required-never-approves
3
+ context:
4
+ scaffold_script: fixture.sh