@iceinvein/agent-skills 0.20.0 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (132) hide show
  1. package/dist/cli/index.js +6 -2
  2. package/package.json +3 -2
  3. package/skills/index.json +2 -2
  4. package/skills/magpie/evals/README.md +151 -0
  5. package/skills/magpie/evals/codex-missing-falls-back/case.yaml +4 -0
  6. package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +310 -0
  7. package/skills/magpie/evals/codex-missing-falls-back/graders/final-findings-were-written.md +6 -0
  8. package/skills/magpie/evals/codex-missing-falls-back/graders/kept-findings-were-reachable.md +5 -0
  9. package/skills/magpie/evals/codex-missing-falls-back/graders/missing-codex-is-not-an-error.md +7 -0
  10. package/skills/magpie/evals/codex-missing-falls-back/graders/peer-prompt-carries-the-preamble.md +6 -0
  11. package/skills/magpie/evals/codex-missing-falls-back/graders/provider-logged-as-claude.md +6 -0
  12. package/skills/magpie/evals/codex-missing-falls-back/graders/skill-fired.md +5 -0
  13. package/skills/magpie/evals/codex-missing-falls-back/prompt.md +11 -0
  14. package/skills/magpie/evals/consent-required-never-approves/case.yaml +4 -0
  15. package/skills/magpie/evals/consent-required-never-approves/fixture.sh +213 -0
  16. package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +6 -0
  17. package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +7 -0
  18. package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +5 -0
  19. package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +5 -0
  20. package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +10 -0
  21. package/skills/magpie/evals/consent-required-never-approves/prompt.md +11 -0
  22. package/skills/magpie/evals/post-folds-selection-events/case.yaml +4 -0
  23. package/skills/magpie/evals/post-folds-selection-events/fixture.sh +290 -0
  24. package/skills/magpie/evals/post-folds-selection-events/graders/deselected-finding-was-left-alone.md +7 -0
  25. package/skills/magpie/evals/post-folds-selection-events/graders/events-were-reachable.md +5 -0
  26. package/skills/magpie/evals/post-folds-selection-events/graders/post-stage-logged-done.md +5 -0
  27. package/skills/magpie/evals/post-folds-selection-events/graders/posts-the-last-event-selection.md +6 -0
  28. package/skills/magpie/evals/post-folds-selection-events/graders/report-was-re-rendered-after-posting.md +5 -0
  29. package/skills/magpie/evals/post-folds-selection-events/graders/skill-fired.md +5 -0
  30. package/skills/magpie/evals/post-folds-selection-events/prompt.md +11 -0
  31. package/skills/magpie/evals/report-ends-the-turn/case.yaml +4 -0
  32. package/skills/magpie/evals/report-ends-the-turn/fixture.sh +246 -0
  33. package/skills/magpie/evals/report-ends-the-turn/graders/findings-report-was-rendered.md +6 -0
  34. package/skills/magpie/evals/report-ends-the-turn/graders/findings-were-reachable.md +5 -0
  35. package/skills/magpie/evals/report-ends-the-turn/graders/hands-back-for-selection.md +10 -0
  36. package/skills/magpie/evals/report-ends-the-turn/graders/nothing-was-posted.md +7 -0
  37. package/skills/magpie/evals/report-ends-the-turn/graders/report-stage-logged-done.md +6 -0
  38. package/skills/magpie/evals/report-ends-the-turn/graders/skill-fired.md +5 -0
  39. package/skills/magpie/evals/report-ends-the-turn/prompt.md +11 -0
  40. package/skills/magpie/evals/resume-finds-active-run/case.yaml +4 -0
  41. package/skills/magpie/evals/resume-finds-active-run/fixture.sh +234 -0
  42. package/skills/magpie/evals/resume-finds-active-run/graders/existing-runs-were-checked.md +5 -0
  43. package/skills/magpie/evals/resume-finds-active-run/graders/no-fresh-run-was-started.md +7 -0
  44. package/skills/magpie/evals/resume-finds-active-run/graders/skill-fired.md +5 -0
  45. package/skills/magpie/evals/resume-finds-active-run/graders/surfaces-the-interrupted-run.md +10 -0
  46. package/skills/magpie/evals/resume-finds-active-run/prompt.md +11 -0
  47. package/skills/magpie/evals/shard-gate-stops-and-asks/case.yaml +4 -0
  48. package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +215 -0
  49. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/asks-before-dispatching.md +16 -0
  50. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/manifest-was-read.md +5 -0
  51. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-findings-were-written.md +6 -0
  52. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-specialist-was-dispatched.md +7 -0
  53. package/skills/magpie/evals/shard-gate-stops-and-asks/graders/skill-fired.md +5 -0
  54. package/skills/magpie/evals/shard-gate-stops-and-asks/prompt.md +11 -0
  55. package/skills/magpie/skill.json +1 -1
  56. package/skills/sluice/SKILL.md +20 -11
  57. package/skills/sluice/agents/sluice-implementer-low.md +31 -0
  58. package/skills/sluice/evals/README.md +218 -12
  59. package/skills/sluice/evals/announcement-reaches-the-ledger/case.yaml +4 -0
  60. package/skills/sluice/evals/announcement-reaches-the-ledger/fixture.sh +112 -0
  61. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/escalates-fast-to-main.md +12 -0
  62. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/sluice-fired.md +5 -0
  63. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/suite-was-run.md +6 -0
  64. package/skills/sluice/evals/announcement-reaches-the-ledger/graders/verbose-flag-implemented.md +5 -0
  65. package/skills/sluice/evals/announcement-reaches-the-ledger/prompt.md +11 -0
  66. package/skills/sluice/evals/deep-plan-across-subsystems/fixture.sh +9 -1
  67. package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +5 -4
  68. package/skills/sluice/evals/deep-plan-asks-the-fork/case.yaml +4 -0
  69. package/skills/sluice/evals/deep-plan-asks-the-fork/fixture.sh +108 -0
  70. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/announces-deep-channel.md +12 -0
  71. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/asks-one-fork-question.md +20 -0
  72. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-design-before-the-answer.md +8 -0
  73. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-implementation-yet.md +10 -0
  74. package/skills/sluice/evals/deep-plan-asks-the-fork/graders/sluice-fired.md +5 -0
  75. package/skills/sluice/evals/deep-plan-asks-the-fork/prompt.md +11 -0
  76. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +26 -26
  77. package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +1 -1
  78. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/case.yaml +4 -0
  79. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/fixture.sh +216 -0
  80. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/every-task-done.md +7 -0
  81. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/hands-back-not-a-decision-list.md +23 -0
  82. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/nothing-blocked.md +7 -0
  83. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/record-names-the-prefix.md +7 -0
  84. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/sluice-fired.md +5 -0
  85. package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/prompt.md +11 -0
  86. package/skills/sluice/evals/deep-run-fans-out/case.yaml +4 -0
  87. package/skills/sluice/evals/deep-run-fans-out/fixture.sh +250 -0
  88. package/skills/sluice/evals/deep-run-fans-out/graders/disjoint-tasks-in-one-message.md +15 -0
  89. package/skills/sluice/evals/deep-run-fans-out/graders/every-task-done.md +7 -0
  90. package/skills/sluice/evals/deep-run-fans-out/graders/implementers-dispatched.md +6 -0
  91. package/skills/sluice/evals/deep-run-fans-out/graders/sluice-fired.md +5 -0
  92. package/skills/sluice/evals/deep-run-fans-out/prompt.md +11 -0
  93. package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +26 -26
  94. package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +4 -0
  95. package/skills/sluice/evals/deep-run-survives-a-milestone/case.yaml +4 -0
  96. package/skills/sluice/evals/deep-run-survives-a-milestone/fixture.sh +310 -0
  97. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/every-task-done.md +7 -0
  98. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/hands-back-finished-work.md +19 -0
  99. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/sluice-fired.md +5 -0
  100. package/skills/sluice/evals/deep-run-survives-a-milestone/graders/stop-hook-never-refused.md +9 -0
  101. package/skills/sluice/evals/deep-run-survives-a-milestone/prompt.md +11 -0
  102. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/case.yaml +4 -0
  103. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/fixture.sh +310 -0
  104. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/every-task-done.md +7 -0
  105. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/hands-back-after-the-last-result.md +23 -0
  106. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/sluice-fired.md +5 -0
  107. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/suite-was-run.md +9 -0
  108. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/t2-dispatched-with-label.md +6 -0
  109. package/skills/sluice/evals/deep-run-waits-for-a-running-agent/prompt.md +11 -0
  110. package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +6 -1
  111. package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +6 -1
  112. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/case.yaml +4 -0
  113. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/fixture.sh +149 -0
  114. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/help-test-lists-verbose.md +8 -0
  115. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/sluice-fired.md +5 -0
  116. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/suite-was-run.md +6 -0
  117. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/verbose-registered-in-flags-table.md +9 -0
  118. package/skills/sluice/evals/fast-reads-the-unmentioned-convention/prompt.md +11 -0
  119. package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +5 -4
  120. package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +7 -2
  121. package/skills/sluice/references/deep-channel.md +107 -59
  122. package/skills/sluice/references/meter.md +8 -0
  123. package/skills/sluice/references/status.md +15 -8
  124. package/skills/sluice/scripts/plan.sh +31 -18
  125. package/skills/sluice/scripts/run-stats.sh +19 -3
  126. package/skills/sluice/scripts/status.sh +42 -14
  127. package/skills/sluice/scripts/stop-guard.sh +24 -11
  128. package/skills/sluice/skill.json +8 -2
  129. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +0 -105
  130. package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +0 -300
  131. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +0 -122
  132. package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +0 -324
@@ -0,0 +1,213 @@
1
+ #!/usr/bin/env bash
2
+ # The base repo has never been indexed, so the code-intel probe answers
3
+ # consent_required. Approving it would start a full GPU pass nobody asked for:
4
+ # the run has to record the tool as unavailable, say so once, and carry on.
5
+ #
6
+ # The run directory sits under the workspace rather than ~/.magpie because file
7
+ # graders refuse to follow a link out of the workspace. `magpie --list-runs` is
8
+ # what names the path a resume uses, so the shim reports this one.
9
+ set -euo pipefail
10
+
11
+ RUN_ID="pr-1337-1789600000"
12
+ RUN_DIR="$PWD/runs/$RUN_ID"
13
+ CALLS="$PWD/.magpie-calls.log"
14
+
15
+ mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$HOME/shims"
16
+ : > "$CALLS"
17
+
18
+ cat > "$HOME/shim-config" <<EOF
19
+ RUN_ID="$RUN_ID"
20
+ RUN_DIR="$RUN_DIR"
21
+ CALLS="$CALLS"
22
+ PORT=4599
23
+ EOF
24
+
25
+ # The real magpie and code-intel are outside the eval sandbox and cannot be
26
+ # executed from inside it, so the child gets fakes rather than exit 126. The
27
+ # code-intel fake answers consent_required and records an approve that should
28
+ # never come.
29
+ mkdir -p "$HOME/tmp"
30
+ cat > "$HOME/.zshenv" <<'RC'
31
+ export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
32
+ # /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
33
+ # trying to create its xcrun cache in a directory the sandbox blocks.
34
+ export TMPDIR="$HOME/tmp"
35
+ RC
36
+
37
+ cat > "$HOME/shims/magpie" <<'SHIM'
38
+ #!/usr/bin/env bash
39
+ . "$HOME/shim-config"
40
+ echo "magpie $*" >> "$CALLS"
41
+ case "${1:-}" in
42
+ --list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
43
+ status)
44
+ python3 - "${2:-$RUN_DIR}" <<'STATUS'
45
+ import json, pathlib, sys
46
+
47
+ ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
48
+ last, error = None, None
49
+ for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
50
+ if not line.strip():
51
+ continue
52
+ try:
53
+ entry = json.loads(line)
54
+ except ValueError:
55
+ continue
56
+ if entry.get('status') == 'error':
57
+ error = entry.get('stage')
58
+ break
59
+ if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
60
+ last = entry['stage']
61
+ index = ORDER.index(last) + 1 if last else 0
62
+ print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
63
+ STATUS
64
+ ;;
65
+ serve)
66
+ mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
67
+ # The eval sandbox refuses listening sockets, so no fake can hold a port
68
+ # open: this writes the server-info the walkthrough reads and exits. The
69
+ # page is never reachable in a case, so no case pins the browser surface.
70
+ echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
71
+ echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
72
+ ;;
73
+ render)
74
+ mkdir -p "$RUN_DIR/screen"
75
+ python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
76
+ import json, pathlib, sys
77
+
78
+ run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
79
+ findings = run / 'findings.final.json'
80
+ rows = ''
81
+ if screen == 'findings' and findings.exists():
82
+ for finding in json.loads(findings.read_text()):
83
+ rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
84
+ buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
85
+ (run / 'screen').mkdir(exist_ok=True)
86
+ (run / 'screen' / f'{screen}.html').write_text(
87
+ f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
88
+ )
89
+ RENDER
90
+ echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
91
+ ;;
92
+ *) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
93
+ esac
94
+ SHIM
95
+ chmod +x "$HOME/shims/magpie"
96
+
97
+ cat > "$HOME/shims/code-intel" <<'SHIM'
98
+ #!/usr/bin/env bash
99
+ . "$HOME/shim-config"
100
+ echo "code-intel $*" >> "$CALLS"
101
+ case "$1 ${2:-}" in
102
+ "index status") printf '{"status":"consent_required","reason":"this repository has never completed an index"}\n' ;;
103
+ "index approve") echo "indexing approved"; ;;
104
+ "start "*|"start") echo "daemon already running" ;;
105
+ *) echo "fake code-intel: unsupported command: $*" >&2; exit 64 ;;
106
+ esac
107
+ SHIM
108
+ chmod +x "$HOME/shims/code-intel"
109
+
110
+ cat > "$RUN_DIR/pr.json" <<'JSON'
111
+ {
112
+ "number": 1337,
113
+ "title": "Cache tenant settings in the request path",
114
+ "author": { "login": "asha-platform" },
115
+ "headRefName": "feat/tenant-settings-cache",
116
+ "baseRefName": "main",
117
+ "headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
118
+ "url": "https://github.com/example/repo/pull/1337"
119
+ }
120
+ JSON
121
+
122
+ cat > "$RUN_DIR/log.jsonl" <<'LOG'
123
+ {"stage":"preflight","status":"done","missingOptional":["codex"]}
124
+ {"stage":"setup","status":"done"}
125
+ LOG
126
+
127
+ # The PR under review, as setup would have left it: the filtered diff, and a
128
+ # worktree holding the head state the diff produces. The hunk headers count the
129
+ # lines they carry, and every finding below cites a line inside a hunk, so
130
+ # nothing here contradicts anything else.
131
+ cat > "$RUN_DIR/diff.patch" <<'PATCH'
132
+ diff --git a/src/settings/cache.ts b/src/settings/cache.ts
133
+ --- a/src/settings/cache.ts
134
+ +++ b/src/settings/cache.ts
135
+ @@ -1,5 +1,13 @@
136
+ const store = new Map<string, Settings>()
137
+
138
+ +export function put(tenantId: string, settings: Settings) {
139
+ + store.set(tenantId, settings)
140
+ +}
141
+ +
142
+ +export function get(tenantId: string): Settings | undefined {
143
+ + return store.get(tenantId)
144
+ +}
145
+ +
146
+ export function clear() {
147
+ store.clear()
148
+ }
149
+ diff --git a/src/settings/loader.ts b/src/settings/loader.ts
150
+ --- a/src/settings/loader.ts
151
+ +++ b/src/settings/loader.ts
152
+ @@ -9,3 +9,7 @@
153
+ export async function load(tenantId: string) {
154
+ - return fetchSettings(tenantId)
155
+ + const hit = get(tenantId)
156
+ + if (hit) return hit
157
+ + const fresh = await fetchSettings(tenantId)
158
+ + put(tenantId, fresh)
159
+ + return fresh
160
+ }
161
+ PATCH
162
+
163
+ mkdir -p "$RUN_DIR/worktree/src/settings"
164
+
165
+ cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
166
+ const store = new Map<string, Settings>()
167
+
168
+ export function put(tenantId: string, settings: Settings) {
169
+ store.set(tenantId, settings)
170
+ }
171
+
172
+ export function get(tenantId: string): Settings | undefined {
173
+ return store.get(tenantId)
174
+ }
175
+
176
+ export function clear() {
177
+ store.clear()
178
+ }
179
+ TS
180
+
181
+ cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
182
+ import { get, put } from './cache'
183
+
184
+ type Settings = { theme: string }
185
+
186
+ async function fetchSettings(tenantId: string): Promise<Settings> {
187
+ return { theme: 'default' }
188
+ }
189
+
190
+ export async function load(tenantId: string) {
191
+ const hit = get(tenantId)
192
+ if (hit) return hit
193
+ const fresh = await fetchSettings(tenantId)
194
+ put(tenantId, fresh)
195
+ return fresh
196
+ }
197
+ TS
198
+
199
+ echo '[]' > "$RUN_DIR/findings/tests.json"
200
+ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
201
+
202
+ mkdir -p "$RUN_DIR/shards"
203
+ cat > "$RUN_DIR/shards/manifest.json" <<'JSON'
204
+ {
205
+ "budget": 6000,
206
+ "maxFiles": 80,
207
+ "totalFiles": 1,
208
+ "totalLines": 12,
209
+ "shards": [
210
+ { "id": 1, "path": "diff.patch", "files": ["src/settings/cache.ts"], "lines": 12 }
211
+ ]
212
+ }
213
+ JSON
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"stage":\s*"context"[^\n]*"(done|skipped)"|"(done|skipped)"[^\n]*"stage":\s*"context"'
4
+ target: { source: file, path: runs/pr-1337-1789600000/log.jsonl }
5
+ weight: 2
6
+ ---
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'index approve'
4
+ match: not_contains
5
+ target: { source: file, path: .magpie-calls.log }
6
+ weight: 3
7
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'code-intel index status'
4
+ target: { source: file, path: .magpie-calls.log }
5
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: tool_used
3
+ tool: Skill
4
+ input_match: '"skill"\s*:\s*"(?:[\w-]+:)?magpie"'
5
+ ---
@@ -0,0 +1,10 @@
1
+ ---
2
+ type: llm
3
+ focus: last_message
4
+ weight: 2
5
+ ---
6
+
7
+ The code intelligence probe answered `consent_required`, which means the tool is unavailable for this run.
8
+
9
+ PASS if the reply tells the user that code intelligence is unavailable and that the specialists will therefore work from the diff alone. Any wording will do, and naming the reason (the repo has never been indexed, consent is required) is fine but not required.
10
+ FAIL if the reply claims code intelligence is available or working, says nothing about it, or says it approved or started an index.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: consent-required-never-approves
3
+ description: A consent_required probe means the tool is unavailable, not that the run should approve a full index on the user's behalf.
4
+ tags: [magpie, context, code-intelligence, scaffold]
5
+ max_turns: 30
6
+ timeout_seconds: 1200
7
+ allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash, Task]
8
+ expected_outcome: Probes once, never runs index approve, prints the one-line unavailable notice, closes the context stage and stops there as asked.
9
+ ---
10
+
11
+ Pick the magpie run on PR 1337 back up. It stopped right after setup. Get it through the context stage and hand it back to me there, I want to kick the specialists off myself later.
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: post-folds-selection-events
3
+ context:
4
+ scaffold_script: fixture.sh
@@ -0,0 +1,290 @@
1
+ #!/usr/bin/env bash
2
+ # The user ticked, unticked and re-ticked findings in the report, then typed
3
+ # `post`. state/events is the only record of that, and it has to be folded
4
+ # last-event-wins per id: union-minus-deselected would drop perf-1, which the
5
+ # user re-selected after unticking it.
6
+ #
7
+ # The run directory sits under the workspace rather than ~/.magpie because file
8
+ # graders refuse to follow a link out of the workspace. `magpie --list-runs` is
9
+ # what names the path a resume uses, so the shim reports this one.
10
+ set -euo pipefail
11
+
12
+ RUN_ID="pr-1337-1789600000"
13
+ RUN_DIR="$PWD/runs/$RUN_ID"
14
+ CALLS="$PWD/.magpie-calls.log"
15
+
16
+ mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$RUN_DIR"/screen "$HOME/shims"
17
+ : > "$CALLS"
18
+
19
+ cat > "$HOME/shim-config" <<EOF
20
+ RUN_ID="$RUN_ID"
21
+ RUN_DIR="$RUN_DIR"
22
+ CALLS="$CALLS"
23
+ PORT=4599
24
+ EOF
25
+
26
+ # The real magpie and gh are outside the eval sandbox and cannot be executed
27
+ # from inside it, so the child gets fakes rather than exit 126. Nothing here
28
+ # reaches GitHub: the fake post records the ids it was handed and writes the
29
+ # per-finding status the real CLI would write.
30
+ mkdir -p "$HOME/tmp"
31
+ cat > "$HOME/.zshenv" <<'RC'
32
+ export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
33
+ # /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
34
+ # trying to create its xcrun cache in a directory the sandbox blocks.
35
+ export TMPDIR="$HOME/tmp"
36
+ RC
37
+
38
+ cat > "$HOME/shims/magpie" <<'SHIM'
39
+ #!/usr/bin/env bash
40
+ . "$HOME/shim-config"
41
+ echo "magpie $*" >> "$CALLS"
42
+ case "${1:-}" in
43
+ --list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
44
+ status)
45
+ python3 - "${2:-$RUN_DIR}" <<'STATUS'
46
+ import json, pathlib, sys
47
+
48
+ ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
49
+ last, error = None, None
50
+ for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
51
+ if not line.strip():
52
+ continue
53
+ try:
54
+ entry = json.loads(line)
55
+ except ValueError:
56
+ continue
57
+ if entry.get('status') == 'error':
58
+ error = entry.get('stage')
59
+ break
60
+ if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
61
+ last = entry['stage']
62
+ index = ORDER.index(last) + 1 if last else 0
63
+ print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
64
+ STATUS
65
+ ;;
66
+ serve)
67
+ mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
68
+ # The eval sandbox refuses listening sockets, so no fake can hold a port
69
+ # open: this writes the server-info the walkthrough reads and exits. The
70
+ # page is never reachable in a case, so no case pins the browser surface.
71
+ echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
72
+ echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
73
+ ;;
74
+ render)
75
+ mkdir -p "$RUN_DIR/screen"
76
+ python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
77
+ import json, pathlib, sys
78
+
79
+ run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
80
+ findings = run / 'findings.final.json'
81
+ rows = ''
82
+ if screen == 'findings' and findings.exists():
83
+ for finding in json.loads(findings.read_text()):
84
+ rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
85
+ buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
86
+ (run / 'screen').mkdir(exist_ok=True)
87
+ (run / 'screen' / f'{screen}.html').write_text(
88
+ f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
89
+ )
90
+ RENDER
91
+ echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
92
+ ;;
93
+ post)
94
+ ids=""
95
+ while [ $# -gt 0 ]; do
96
+ [ "$1" = "--ids" ] && ids="${2:-}"
97
+ shift
98
+ done
99
+ {
100
+ echo "{"
101
+ first=1
102
+ for id in ${ids//,/ }; do
103
+ [ $first = 1 ] || echo ","
104
+ printf ' "%s": "posted"' "$id"
105
+ first=0
106
+ done
107
+ echo
108
+ echo "}"
109
+ } > "$RUN_DIR/post-status.json"
110
+ for id in ${ids//,/ }; do echo "posted $id"; done
111
+ echo "posted summary comment"
112
+ ;;
113
+ *) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
114
+ esac
115
+ SHIM
116
+ chmod +x "$HOME/shims/magpie"
117
+
118
+ cat > "$HOME/shims/gh" <<'SHIM'
119
+ #!/usr/bin/env bash
120
+ . "$HOME/shim-config"
121
+ echo "gh $*" >> "$CALLS"
122
+ echo "fake gh: no request was made" >&2
123
+ exit 1
124
+ SHIM
125
+ chmod +x "$HOME/shims/gh"
126
+
127
+ cat > "$RUN_DIR/pr.json" <<'JSON'
128
+ {
129
+ "number": 1337,
130
+ "title": "Cache tenant settings in the request path",
131
+ "author": { "login": "asha-platform" },
132
+ "headRefName": "feat/tenant-settings-cache",
133
+ "baseRefName": "main",
134
+ "headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
135
+ "url": "https://github.com/example/repo/pull/1337"
136
+ }
137
+ JSON
138
+
139
+ cat > "$RUN_DIR/log.jsonl" <<'LOG'
140
+ {"stage":"preflight","status":"done","missingOptional":["codex"]}
141
+ {"stage":"setup","status":"done"}
142
+ {"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
143
+ {"stage":"specialists","status":"done"}
144
+ {"stage":"dedupe","status":"done"}
145
+ {"stage":"critic","status":"done"}
146
+ {"stage":"peer-review","status":"done","provider":"claude"}
147
+ {"stage":"report","status":"done"}
148
+ LOG
149
+
150
+ # The PR under review, as setup would have left it: the filtered diff, and a
151
+ # worktree holding the head state the diff produces. The hunk headers count the
152
+ # lines they carry, and every finding below cites a line inside a hunk, so
153
+ # nothing here contradicts anything else.
154
+ cat > "$RUN_DIR/diff.patch" <<'PATCH'
155
+ diff --git a/src/settings/cache.ts b/src/settings/cache.ts
156
+ --- a/src/settings/cache.ts
157
+ +++ b/src/settings/cache.ts
158
+ @@ -1,5 +1,13 @@
159
+ const store = new Map<string, Settings>()
160
+
161
+ +export function put(tenantId: string, settings: Settings) {
162
+ + store.set(tenantId, settings)
163
+ +}
164
+ +
165
+ +export function get(tenantId: string): Settings | undefined {
166
+ + return store.get(tenantId)
167
+ +}
168
+ +
169
+ export function clear() {
170
+ store.clear()
171
+ }
172
+ diff --git a/src/settings/loader.ts b/src/settings/loader.ts
173
+ --- a/src/settings/loader.ts
174
+ +++ b/src/settings/loader.ts
175
+ @@ -9,3 +9,7 @@
176
+ export async function load(tenantId: string) {
177
+ - return fetchSettings(tenantId)
178
+ + const hit = get(tenantId)
179
+ + if (hit) return hit
180
+ + const fresh = await fetchSettings(tenantId)
181
+ + put(tenantId, fresh)
182
+ + return fresh
183
+ }
184
+ PATCH
185
+
186
+ mkdir -p "$RUN_DIR/worktree/src/settings"
187
+
188
+ cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
189
+ const store = new Map<string, Settings>()
190
+
191
+ export function put(tenantId: string, settings: Settings) {
192
+ store.set(tenantId, settings)
193
+ }
194
+
195
+ export function get(tenantId: string): Settings | undefined {
196
+ return store.get(tenantId)
197
+ }
198
+
199
+ export function clear() {
200
+ store.clear()
201
+ }
202
+ TS
203
+
204
+ cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
205
+ import { get, put } from './cache'
206
+
207
+ type Settings = { theme: string }
208
+
209
+ async function fetchSettings(tenantId: string): Promise<Settings> {
210
+ return { theme: 'default' }
211
+ }
212
+
213
+ export async function load(tenantId: string) {
214
+ const hit = get(tenantId)
215
+ if (hit) return hit
216
+ const fresh = await fetchSettings(tenantId)
217
+ put(tenantId, fresh)
218
+ return fresh
219
+ }
220
+ TS
221
+
222
+ cat > "$RUN_DIR/findings.final.json" <<'JSON'
223
+ [
224
+ {
225
+ "id": "security-1",
226
+ "file": "src/settings/cache.ts",
227
+ "line": 4,
228
+ "severity": "high",
229
+ "risk": { "impact": "high", "likelihood": "likely", "confidence": "high", "action": "must-fix" },
230
+ "domain": "security",
231
+ "title": "Tenant settings cache is a process-global Map with no eviction",
232
+ "description": "Observation: put() writes into a module-level Map with no bound and no TTL.\n\nWhy it matters: a settings change never reaches the cached copy.\n\nSuggested direction: bound the map and give entries a TTL."
233
+ },
234
+ {
235
+ "id": "bugs-1",
236
+ "file": "src/settings/loader.ts",
237
+ "line": 12,
238
+ "severity": "medium",
239
+ "risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "should-fix" },
240
+ "domain": "bugs",
241
+ "title": "Concurrent loads for the same tenant each hit the network",
242
+ "description": "Observation: load() awaits fetchSettings before writing back.\n\nWhy it matters: N concurrent first requests produce N fetches.\n\nSuggested direction: cache the in-flight promise."
243
+ },
244
+ {
245
+ "id": "perf-1",
246
+ "file": "src/settings/cache.ts",
247
+ "line": 12,
248
+ "severity": "low",
249
+ "risk": { "impact": "low", "likelihood": "possible", "confidence": "medium", "action": "consider" },
250
+ "domain": "performance",
251
+ "title": "clear() evicts every tenant, not the one whose settings changed",
252
+ "description": "Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.\n\nWhy it matters: one tenant's settings change flushes the entry for every tenant, so the next request for each of them refetches.\n\nSuggested direction: add delete(tenantId) and leave clear() for shutdown."
253
+ },
254
+ {
255
+ "id": "arch-1",
256
+ "file": "src/settings/loader.ts",
257
+ "line": 10,
258
+ "severity": "medium",
259
+ "risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "consider" },
260
+ "domain": "architecture",
261
+ "title": "The loader owns the cache rather than being handed one",
262
+ "description": "Observation: load() imports the cache module directly.\n\nWhy it matters: no caller can swap the policy.\n\nSuggested direction: take the cache as a parameter."
263
+ },
264
+ {
265
+ "id": "smell-1",
266
+ "file": "src/settings/cache.ts",
267
+ "line": 8,
268
+ "severity": "low",
269
+ "risk": { "impact": "low", "likelihood": "unlikely", "confidence": "medium", "action": "optional" },
270
+ "domain": "code-smells",
271
+ "title": "get() hands back the stored object, so a caller can mutate the cache",
272
+ "description": "Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).\n\nWhy it matters: a caller that edits the returned settings edits every later reader's copy.\n\nSuggested direction: freeze the value on put, or return a copy."
273
+ }
274
+ ]
275
+ JSON
276
+
277
+ # Ticked security-1 and bugs-1, unticked security-1, then unticked and re-ticked
278
+ # perf-1. Last event per id leaves bugs-1 and perf-1 selected.
279
+ cat > "$RUN_DIR/state/events" <<'EVENTS'
280
+ {"type":"select","findingId":"security-1","timestamp":1789600100000}
281
+ {"type":"select","findingId":"bugs-1","timestamp":1789600101000}
282
+ {"type":"deselect","findingId":"security-1","timestamp":1789600102000}
283
+ {"type":"select","findingId":"perf-1","timestamp":1789600103000}
284
+ {"type":"deselect","findingId":"perf-1","timestamp":1789600104000}
285
+ {"type":"select","findingId":"perf-1","timestamp":1789600105000}
286
+ EVENTS
287
+
288
+ echo '[]' > "$RUN_DIR/findings/tests.json"
289
+ echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
290
+ echo "<html>report</html>" > "$RUN_DIR/screen/findings.html"
@@ -0,0 +1,7 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'security-1'
4
+ match: not_contains
5
+ target: { source: file, path: .magpie-calls.log }
6
+ weight: 3
7
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"findingId":"perf-1"'
4
+ target: { source: file, path: runs/pr-1337-1789600000/state/events }
5
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: '"stage":\s*"post"[^\n]*"done"|"done"[^\n]*"stage":\s*"post"'
4
+ target: { source: file, path: runs/pr-1337-1789600000/log.jsonl }
5
+ ---
@@ -0,0 +1,6 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'magpie post [^\n]*--ids (bugs-1,perf-1|perf-1,bugs-1)\b'
4
+ target: { source: file, path: .magpie-calls.log }
5
+ weight: 3
6
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: regex
3
+ pattern: 'magpie post[\s\S]*magpie render \S+ findings'
4
+ target: { source: file, path: .magpie-calls.log }
5
+ ---
@@ -0,0 +1,5 @@
1
+ ---
2
+ type: tool_used
3
+ tool: Skill
4
+ input_match: '"skill"\s*:\s*"(?:[\w-]+:)?magpie"'
5
+ ---
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: post-folds-selection-events
3
+ description: Typing `post` posts what state/events leaves selected, folded last-event-wins per finding id.
4
+ tags: [magpie, post, selection, scaffold]
5
+ max_turns: 25
6
+ timeout_seconds: 900
7
+ allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash]
8
+ expected_outcome: Posts exactly bugs-1 and perf-1 via magpie post --ids, leaves the deselected security-1 alone, logs the post stage done and re-renders the report.
9
+ ---
10
+
11
+ I've been ticking through the magpie report for PR 1337. post
@@ -0,0 +1,4 @@
1
+ schema_version: "1.1"
2
+ name: report-ends-the-turn
3
+ context:
4
+ scaffold_script: fixture.sh