@iceinvein/agent-skills 0.19.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +6 -2
- package/package.json +3 -2
- package/skills/index.json +2 -2
- package/skills/magpie/evals/README.md +151 -0
- package/skills/magpie/evals/codex-missing-falls-back/case.yaml +4 -0
- package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +310 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/final-findings-were-written.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/kept-findings-were-reachable.md +5 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/missing-codex-is-not-an-error.md +7 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/peer-prompt-carries-the-preamble.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/provider-logged-as-claude.md +6 -0
- package/skills/magpie/evals/codex-missing-falls-back/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/codex-missing-falls-back/prompt.md +11 -0
- package/skills/magpie/evals/consent-required-never-approves/case.yaml +4 -0
- package/skills/magpie/evals/consent-required-never-approves/fixture.sh +213 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +6 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +7 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +5 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +10 -0
- package/skills/magpie/evals/consent-required-never-approves/prompt.md +11 -0
- package/skills/magpie/evals/post-folds-selection-events/case.yaml +4 -0
- package/skills/magpie/evals/post-folds-selection-events/fixture.sh +290 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/deselected-finding-was-left-alone.md +7 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/events-were-reachable.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/post-stage-logged-done.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/posts-the-last-event-selection.md +6 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/report-was-re-rendered-after-posting.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/post-folds-selection-events/prompt.md +11 -0
- package/skills/magpie/evals/report-ends-the-turn/case.yaml +4 -0
- package/skills/magpie/evals/report-ends-the-turn/fixture.sh +246 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/findings-report-was-rendered.md +6 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/findings-were-reachable.md +5 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/hands-back-for-selection.md +10 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/nothing-was-posted.md +7 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/report-stage-logged-done.md +6 -0
- package/skills/magpie/evals/report-ends-the-turn/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/report-ends-the-turn/prompt.md +11 -0
- package/skills/magpie/evals/resume-finds-active-run/case.yaml +4 -0
- package/skills/magpie/evals/resume-finds-active-run/fixture.sh +234 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/existing-runs-were-checked.md +5 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/no-fresh-run-was-started.md +7 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/resume-finds-active-run/graders/surfaces-the-interrupted-run.md +10 -0
- package/skills/magpie/evals/resume-finds-active-run/prompt.md +11 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/case.yaml +4 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +215 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/asks-before-dispatching.md +16 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/manifest-was-read.md +5 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-findings-were-written.md +6 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/no-specialist-was-dispatched.md +7 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/graders/skill-fired.md +5 -0
- package/skills/magpie/evals/shard-gate-stops-and-asks/prompt.md +11 -0
- package/skills/magpie/skill.json +1 -1
- package/skills/sluice/SKILL.md +24 -12
- package/skills/sluice/agents/sluice-implementer-low.md +31 -0
- package/skills/sluice/evals/README.md +246 -19
- package/skills/sluice/evals/announcement-reaches-the-ledger/case.yaml +4 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/fixture.sh +112 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/escalates-fast-to-main.md +12 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/graders/verbose-flag-implemented.md +5 -0
- package/skills/sluice/evals/announcement-reaches-the-ledger/prompt.md +11 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/case.yaml +4 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/fixture.sh +113 -0
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/announces-deep-channel.md +6 -1
- package/skills/sluice/evals/deep-plan-across-subsystems/graders/no-implementation-yet.md +7 -3
- package/skills/sluice/evals/deep-plan-asks-the-fork/case.yaml +4 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/fixture.sh +108 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/announces-deep-channel.md +12 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/asks-one-fork-question.md +20 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-design-before-the-answer.md +8 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/no-implementation-yet.md +10 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-plan-asks-the-fork/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/fixture.sh +26 -26
- package/skills/sluice/evals/deep-run-blocks-on-a-real-decision/prompt.md +1 -1
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/fixture.sh +216 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/hands-back-not-a-decision-list.md +23 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/nothing-blocked.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/record-names-the-prefix.md +7 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-decides-a-non-blocking-choice/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-fans-out/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-fans-out/fixture.sh +250 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/disjoint-tasks-in-one-message.md +15 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/implementers-dispatched.md +6 -0
- package/skills/sluice/evals/deep-run-fans-out/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-fans-out/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-finishes-every-task/fixture.sh +26 -26
- package/skills/sluice/evals/deep-run-finishes-every-task/graders/did-not-check-in-between-tasks.md +4 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/fixture.sh +310 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/hands-back-finished-work.md +19 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/graders/stop-hook-never-refused.md +9 -0
- package/skills/sluice/evals/deep-run-survives-a-milestone/prompt.md +11 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/case.yaml +4 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/fixture.sh +310 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/every-task-done.md +7 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/hands-back-after-the-last-result.md +23 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/suite-was-run.md +9 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/graders/t2-dispatched-with-label.md +6 -0
- package/skills/sluice/evals/deep-run-waits-for-a-running-agent/prompt.md +11 -0
- package/skills/sluice/evals/explicit-instruction-collapses-to-fast/graders/announces-fast-channel.md +6 -1
- package/skills/sluice/evals/fast-flag-on-existing-command/graders/announces-fast-channel.md +6 -1
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/case.yaml +4 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/fixture.sh +149 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/help-test-lists-verbose.md +8 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/sluice-fired.md +5 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/suite-was-run.md +6 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/graders/verbose-registered-in-flags-table.md +9 -0
- package/skills/sluice/evals/fast-reads-the-unmentioned-convention/prompt.md +11 -0
- package/skills/sluice/evals/main-new-interface/graders/announces-main-channel.md +6 -1
- package/skills/sluice/evals/main-new-interface/graders/shape-agreed-before-building.md +15 -7
- package/skills/sluice/evals/main-new-interface/prompt.md +2 -2
- package/skills/sluice/references/deep-channel.md +124 -65
- package/skills/sluice/references/meter.md +8 -0
- package/skills/sluice/references/status.md +60 -19
- package/skills/sluice/scripts/plan.sh +31 -18
- package/skills/sluice/scripts/run-stats.sh +19 -3
- package/skills/sluice/scripts/session-start.sh +1 -1
- package/skills/sluice/scripts/status.sh +138 -18
- package/skills/sluice/scripts/statusline.sh +6 -1
- package/skills/sluice/scripts/stop-guard.sh +66 -22
- package/skills/sluice/skill.json +8 -2
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/aggregate-result.json +0 -105
- package/skills/sluice/evals/results/2026-09-20T01-51-28-540Z/report.html +0 -300
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/aggregate-result.json +0 -122
- package/skills/sluice/evals/results/2026-09-20T01-52-06-287Z/report.html +0 -324
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# The base repo has never been indexed, so the code-intel probe answers
|
|
3
|
+
# consent_required. Approving it would start a full GPU pass nobody asked for:
|
|
4
|
+
# the run has to record the tool as unavailable, say so once, and carry on.
|
|
5
|
+
#
|
|
6
|
+
# The run directory sits under the workspace rather than ~/.magpie because file
|
|
7
|
+
# graders refuse to follow a link out of the workspace. `magpie --list-runs` is
|
|
8
|
+
# what names the path a resume uses, so the shim reports this one.
|
|
9
|
+
set -euo pipefail
|
|
10
|
+
|
|
11
|
+
RUN_ID="pr-1337-1789600000"
|
|
12
|
+
RUN_DIR="$PWD/runs/$RUN_ID"
|
|
13
|
+
CALLS="$PWD/.magpie-calls.log"
|
|
14
|
+
|
|
15
|
+
mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$HOME/shims"
|
|
16
|
+
: > "$CALLS"
|
|
17
|
+
|
|
18
|
+
cat > "$HOME/shim-config" <<EOF
|
|
19
|
+
RUN_ID="$RUN_ID"
|
|
20
|
+
RUN_DIR="$RUN_DIR"
|
|
21
|
+
CALLS="$CALLS"
|
|
22
|
+
PORT=4599
|
|
23
|
+
EOF
|
|
24
|
+
|
|
25
|
+
# The real magpie and code-intel are outside the eval sandbox and cannot be
|
|
26
|
+
# executed from inside it, so the child gets fakes rather than exit 126. The
|
|
27
|
+
# code-intel fake answers consent_required and records an approve that should
|
|
28
|
+
# never come.
|
|
29
|
+
mkdir -p "$HOME/tmp"
|
|
30
|
+
cat > "$HOME/.zshenv" <<'RC'
|
|
31
|
+
export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
|
|
32
|
+
# /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
|
|
33
|
+
# trying to create its xcrun cache in a directory the sandbox blocks.
|
|
34
|
+
export TMPDIR="$HOME/tmp"
|
|
35
|
+
RC
|
|
36
|
+
|
|
37
|
+
cat > "$HOME/shims/magpie" <<'SHIM'
|
|
38
|
+
#!/usr/bin/env bash
|
|
39
|
+
. "$HOME/shim-config"
|
|
40
|
+
echo "magpie $*" >> "$CALLS"
|
|
41
|
+
case "${1:-}" in
|
|
42
|
+
--list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
|
|
43
|
+
status)
|
|
44
|
+
python3 - "${2:-$RUN_DIR}" <<'STATUS'
|
|
45
|
+
import json, pathlib, sys
|
|
46
|
+
|
|
47
|
+
ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
|
|
48
|
+
last, error = None, None
|
|
49
|
+
for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
|
|
50
|
+
if not line.strip():
|
|
51
|
+
continue
|
|
52
|
+
try:
|
|
53
|
+
entry = json.loads(line)
|
|
54
|
+
except ValueError:
|
|
55
|
+
continue
|
|
56
|
+
if entry.get('status') == 'error':
|
|
57
|
+
error = entry.get('stage')
|
|
58
|
+
break
|
|
59
|
+
if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
|
|
60
|
+
last = entry['stage']
|
|
61
|
+
index = ORDER.index(last) + 1 if last else 0
|
|
62
|
+
print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
|
|
63
|
+
STATUS
|
|
64
|
+
;;
|
|
65
|
+
serve)
|
|
66
|
+
mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
|
|
67
|
+
# The eval sandbox refuses listening sockets, so no fake can hold a port
|
|
68
|
+
# open: this writes the server-info the walkthrough reads and exits. The
|
|
69
|
+
# page is never reachable in a case, so no case pins the browser surface.
|
|
70
|
+
echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
|
|
71
|
+
echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
|
|
72
|
+
;;
|
|
73
|
+
render)
|
|
74
|
+
mkdir -p "$RUN_DIR/screen"
|
|
75
|
+
python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
|
|
76
|
+
import json, pathlib, sys
|
|
77
|
+
|
|
78
|
+
run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
|
|
79
|
+
findings = run / 'findings.final.json'
|
|
80
|
+
rows = ''
|
|
81
|
+
if screen == 'findings' and findings.exists():
|
|
82
|
+
for finding in json.loads(findings.read_text()):
|
|
83
|
+
rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
|
|
84
|
+
buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
|
|
85
|
+
(run / 'screen').mkdir(exist_ok=True)
|
|
86
|
+
(run / 'screen' / f'{screen}.html').write_text(
|
|
87
|
+
f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
|
|
88
|
+
)
|
|
89
|
+
RENDER
|
|
90
|
+
echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
|
|
91
|
+
;;
|
|
92
|
+
*) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
|
|
93
|
+
esac
|
|
94
|
+
SHIM
|
|
95
|
+
chmod +x "$HOME/shims/magpie"
|
|
96
|
+
|
|
97
|
+
cat > "$HOME/shims/code-intel" <<'SHIM'
|
|
98
|
+
#!/usr/bin/env bash
|
|
99
|
+
. "$HOME/shim-config"
|
|
100
|
+
echo "code-intel $*" >> "$CALLS"
|
|
101
|
+
case "$1 ${2:-}" in
|
|
102
|
+
"index status") printf '{"status":"consent_required","reason":"this repository has never completed an index"}\n' ;;
|
|
103
|
+
"index approve") echo "indexing approved"; ;;
|
|
104
|
+
"start "*|"start") echo "daemon already running" ;;
|
|
105
|
+
*) echo "fake code-intel: unsupported command: $*" >&2; exit 64 ;;
|
|
106
|
+
esac
|
|
107
|
+
SHIM
|
|
108
|
+
chmod +x "$HOME/shims/code-intel"
|
|
109
|
+
|
|
110
|
+
cat > "$RUN_DIR/pr.json" <<'JSON'
|
|
111
|
+
{
|
|
112
|
+
"number": 1337,
|
|
113
|
+
"title": "Cache tenant settings in the request path",
|
|
114
|
+
"author": { "login": "asha-platform" },
|
|
115
|
+
"headRefName": "feat/tenant-settings-cache",
|
|
116
|
+
"baseRefName": "main",
|
|
117
|
+
"headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
|
|
118
|
+
"url": "https://github.com/example/repo/pull/1337"
|
|
119
|
+
}
|
|
120
|
+
JSON
|
|
121
|
+
|
|
122
|
+
cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
123
|
+
{"stage":"preflight","status":"done","missingOptional":["codex"]}
|
|
124
|
+
{"stage":"setup","status":"done"}
|
|
125
|
+
LOG
|
|
126
|
+
|
|
127
|
+
# The PR under review, as setup would have left it: the filtered diff, and a
|
|
128
|
+
# worktree holding the head state the diff produces. The hunk headers count the
|
|
129
|
+
# lines they carry, and every finding below cites a line inside a hunk, so
|
|
130
|
+
# nothing here contradicts anything else.
|
|
131
|
+
cat > "$RUN_DIR/diff.patch" <<'PATCH'
|
|
132
|
+
diff --git a/src/settings/cache.ts b/src/settings/cache.ts
|
|
133
|
+
--- a/src/settings/cache.ts
|
|
134
|
+
+++ b/src/settings/cache.ts
|
|
135
|
+
@@ -1,5 +1,13 @@
|
|
136
|
+
const store = new Map<string, Settings>()
|
|
137
|
+
|
|
138
|
+
+export function put(tenantId: string, settings: Settings) {
|
|
139
|
+
+ store.set(tenantId, settings)
|
|
140
|
+
+}
|
|
141
|
+
+
|
|
142
|
+
+export function get(tenantId: string): Settings | undefined {
|
|
143
|
+
+ return store.get(tenantId)
|
|
144
|
+
+}
|
|
145
|
+
+
|
|
146
|
+
export function clear() {
|
|
147
|
+
store.clear()
|
|
148
|
+
}
|
|
149
|
+
diff --git a/src/settings/loader.ts b/src/settings/loader.ts
|
|
150
|
+
--- a/src/settings/loader.ts
|
|
151
|
+
+++ b/src/settings/loader.ts
|
|
152
|
+
@@ -9,3 +9,7 @@
|
|
153
|
+
export async function load(tenantId: string) {
|
|
154
|
+
- return fetchSettings(tenantId)
|
|
155
|
+
+ const hit = get(tenantId)
|
|
156
|
+
+ if (hit) return hit
|
|
157
|
+
+ const fresh = await fetchSettings(tenantId)
|
|
158
|
+
+ put(tenantId, fresh)
|
|
159
|
+
+ return fresh
|
|
160
|
+
}
|
|
161
|
+
PATCH
|
|
162
|
+
|
|
163
|
+
mkdir -p "$RUN_DIR/worktree/src/settings"
|
|
164
|
+
|
|
165
|
+
cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
|
|
166
|
+
const store = new Map<string, Settings>()
|
|
167
|
+
|
|
168
|
+
export function put(tenantId: string, settings: Settings) {
|
|
169
|
+
store.set(tenantId, settings)
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
export function get(tenantId: string): Settings | undefined {
|
|
173
|
+
return store.get(tenantId)
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export function clear() {
|
|
177
|
+
store.clear()
|
|
178
|
+
}
|
|
179
|
+
TS
|
|
180
|
+
|
|
181
|
+
cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
|
|
182
|
+
import { get, put } from './cache'
|
|
183
|
+
|
|
184
|
+
type Settings = { theme: string }
|
|
185
|
+
|
|
186
|
+
async function fetchSettings(tenantId: string): Promise<Settings> {
|
|
187
|
+
return { theme: 'default' }
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
export async function load(tenantId: string) {
|
|
191
|
+
const hit = get(tenantId)
|
|
192
|
+
if (hit) return hit
|
|
193
|
+
const fresh = await fetchSettings(tenantId)
|
|
194
|
+
put(tenantId, fresh)
|
|
195
|
+
return fresh
|
|
196
|
+
}
|
|
197
|
+
TS
|
|
198
|
+
|
|
199
|
+
echo '[]' > "$RUN_DIR/findings/tests.json"
|
|
200
|
+
echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
|
|
201
|
+
|
|
202
|
+
mkdir -p "$RUN_DIR/shards"
|
|
203
|
+
cat > "$RUN_DIR/shards/manifest.json" <<'JSON'
|
|
204
|
+
{
|
|
205
|
+
"budget": 6000,
|
|
206
|
+
"maxFiles": 80,
|
|
207
|
+
"totalFiles": 1,
|
|
208
|
+
"totalLines": 12,
|
|
209
|
+
"shards": [
|
|
210
|
+
{ "id": 1, "path": "diff.patch", "files": ["src/settings/cache.ts"], "lines": 12 }
|
|
211
|
+
]
|
|
212
|
+
}
|
|
213
|
+
JSON
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
---
|
|
2
|
+
type: llm
|
|
3
|
+
focus: last_message
|
|
4
|
+
weight: 2
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
The code intelligence probe answered `consent_required`, which means the tool is unavailable for this run.
|
|
8
|
+
|
|
9
|
+
PASS if the reply tells the user that code intelligence is unavailable and that the specialists will therefore work from the diff alone. Any wording will do, and naming the reason (the repo has never been indexed, consent is required) is fine but not required.
|
|
10
|
+
FAIL if the reply claims code intelligence is available or working, says nothing about it, or says it approved or started an index.
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: consent-required-never-approves
|
|
3
|
+
description: A consent_required probe means the tool is unavailable, not that the run should approve a full index on the user's behalf.
|
|
4
|
+
tags: [magpie, context, code-intelligence, scaffold]
|
|
5
|
+
max_turns: 30
|
|
6
|
+
timeout_seconds: 1200
|
|
7
|
+
allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash, Task]
|
|
8
|
+
expected_outcome: Probes once, never runs index approve, prints the one-line unavailable notice, closes the context stage and stops there as asked.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
Pick the magpie run on PR 1337 back up. It stopped right after setup. Get it through the context stage and hand it back to me there, I want to kick the specialists off myself later.
|
|
@@ -0,0 +1,290 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# The user ticked, unticked and re-ticked findings in the report, then typed
|
|
3
|
+
# `post`. state/events is the only record of that, and it has to be folded
|
|
4
|
+
# last-event-wins per id: union-minus-deselected would drop perf-1, which the
|
|
5
|
+
# user re-selected after unticking it.
|
|
6
|
+
#
|
|
7
|
+
# The run directory sits under the workspace rather than ~/.magpie because file
|
|
8
|
+
# graders refuse to follow a link out of the workspace. `magpie --list-runs` is
|
|
9
|
+
# what names the path a resume uses, so the shim reports this one.
|
|
10
|
+
set -euo pipefail
|
|
11
|
+
|
|
12
|
+
RUN_ID="pr-1337-1789600000"
|
|
13
|
+
RUN_DIR="$PWD/runs/$RUN_ID"
|
|
14
|
+
CALLS="$PWD/.magpie-calls.log"
|
|
15
|
+
|
|
16
|
+
mkdir -p "$RUN_DIR"/findings "$RUN_DIR"/state "$RUN_DIR"/screen "$HOME/shims"
|
|
17
|
+
: > "$CALLS"
|
|
18
|
+
|
|
19
|
+
cat > "$HOME/shim-config" <<EOF
|
|
20
|
+
RUN_ID="$RUN_ID"
|
|
21
|
+
RUN_DIR="$RUN_DIR"
|
|
22
|
+
CALLS="$CALLS"
|
|
23
|
+
PORT=4599
|
|
24
|
+
EOF
|
|
25
|
+
|
|
26
|
+
# The real magpie and gh are outside the eval sandbox and cannot be executed
|
|
27
|
+
# from inside it, so the child gets fakes rather than exit 126. Nothing here
|
|
28
|
+
# reaches GitHub: the fake post records the ids it was handed and writes the
|
|
29
|
+
# per-finding status the real CLI would write.
|
|
30
|
+
mkdir -p "$HOME/tmp"
|
|
31
|
+
cat > "$HOME/.zshenv" <<'RC'
|
|
32
|
+
export PATH="$HOME/shims:/usr/bin:/bin:/usr/sbin:/sbin"
|
|
33
|
+
# /usr/bin/python3 is the Xcode shim, and without a writable TMPDIR it fails
|
|
34
|
+
# trying to create its xcrun cache in a directory the sandbox blocks.
|
|
35
|
+
export TMPDIR="$HOME/tmp"
|
|
36
|
+
RC
|
|
37
|
+
|
|
38
|
+
cat > "$HOME/shims/magpie" <<'SHIM'
|
|
39
|
+
#!/usr/bin/env bash
|
|
40
|
+
. "$HOME/shim-config"
|
|
41
|
+
echo "magpie $*" >> "$CALLS"
|
|
42
|
+
case "${1:-}" in
|
|
43
|
+
--list-runs) printf '%s\tactive\t%s\n' "$RUN_ID" "$RUN_DIR" ;;
|
|
44
|
+
status)
|
|
45
|
+
python3 - "${2:-$RUN_DIR}" <<'STATUS'
|
|
46
|
+
import json, pathlib, sys
|
|
47
|
+
|
|
48
|
+
ORDER = ['setup', 'context', 'specialists', 'dedupe', 'critic', 'peer-review', 'report', 'post']
|
|
49
|
+
last, error = None, None
|
|
50
|
+
for line in (pathlib.Path(sys.argv[1]) / 'log.jsonl').read_text().splitlines():
|
|
51
|
+
if not line.strip():
|
|
52
|
+
continue
|
|
53
|
+
try:
|
|
54
|
+
entry = json.loads(line)
|
|
55
|
+
except ValueError:
|
|
56
|
+
continue
|
|
57
|
+
if entry.get('status') == 'error':
|
|
58
|
+
error = entry.get('stage')
|
|
59
|
+
break
|
|
60
|
+
if entry.get('status') in ('done', 'skipped') and entry.get('stage') in ORDER:
|
|
61
|
+
last = entry['stage']
|
|
62
|
+
index = ORDER.index(last) + 1 if last else 0
|
|
63
|
+
print(json.dumps({'lastCompleted': last, 'next': ORDER[index] if index < len(ORDER) else 'cleanup', 'error': error}))
|
|
64
|
+
STATUS
|
|
65
|
+
;;
|
|
66
|
+
serve)
|
|
67
|
+
mkdir -p "$RUN_DIR/screen" "$RUN_DIR/state"
|
|
68
|
+
# The eval sandbox refuses listening sockets, so no fake can hold a port
|
|
69
|
+
# open: this writes the server-info the walkthrough reads and exits. The
|
|
70
|
+
# page is never reachable in a case, so no case pins the browser surface.
|
|
71
|
+
echo "http://127.0.0.1:$PORT" > "$RUN_DIR/state/server-info"
|
|
72
|
+
echo "serving $RUN_DIR on http://127.0.0.1:$PORT"
|
|
73
|
+
;;
|
|
74
|
+
render)
|
|
75
|
+
mkdir -p "$RUN_DIR/screen"
|
|
76
|
+
python3 - "${2:-$RUN_DIR}" "${3:-progress}" <<'RENDER'
|
|
77
|
+
import json, pathlib, sys
|
|
78
|
+
|
|
79
|
+
run, screen = pathlib.Path(sys.argv[1]), sys.argv[2]
|
|
80
|
+
findings = run / 'findings.final.json'
|
|
81
|
+
rows = ''
|
|
82
|
+
if screen == 'findings' and findings.exists():
|
|
83
|
+
for finding in json.loads(findings.read_text()):
|
|
84
|
+
rows += f'<li><input type="checkbox" data-finding-id="{finding["id"]}"> {finding["id"]}: {finding["title"]}</li>'
|
|
85
|
+
buttons = '<button>Post Selected</button><button>Post Recommended</button>' if rows else ''
|
|
86
|
+
(run / 'screen').mkdir(exist_ok=True)
|
|
87
|
+
(run / 'screen' / f'{screen}.html').write_text(
|
|
88
|
+
f'<html><body><h1>magpie {screen}</h1><ul>{rows}</ul>{buttons}</body></html>'
|
|
89
|
+
)
|
|
90
|
+
RENDER
|
|
91
|
+
echo "rendered ${3:-progress} -> $RUN_DIR/screen/${3:-progress}.html"
|
|
92
|
+
;;
|
|
93
|
+
post)
|
|
94
|
+
ids=""
|
|
95
|
+
while [ $# -gt 0 ]; do
|
|
96
|
+
[ "$1" = "--ids" ] && ids="${2:-}"
|
|
97
|
+
shift
|
|
98
|
+
done
|
|
99
|
+
{
|
|
100
|
+
echo "{"
|
|
101
|
+
first=1
|
|
102
|
+
for id in ${ids//,/ }; do
|
|
103
|
+
[ $first = 1 ] || echo ","
|
|
104
|
+
printf ' "%s": "posted"' "$id"
|
|
105
|
+
first=0
|
|
106
|
+
done
|
|
107
|
+
echo
|
|
108
|
+
echo "}"
|
|
109
|
+
} > "$RUN_DIR/post-status.json"
|
|
110
|
+
for id in ${ids//,/ }; do echo "posted $id"; done
|
|
111
|
+
echo "posted summary comment"
|
|
112
|
+
;;
|
|
113
|
+
*) echo "fake magpie: unsupported subcommand: $*" >&2; exit 64 ;;
|
|
114
|
+
esac
|
|
115
|
+
SHIM
|
|
116
|
+
chmod +x "$HOME/shims/magpie"
|
|
117
|
+
|
|
118
|
+
cat > "$HOME/shims/gh" <<'SHIM'
|
|
119
|
+
#!/usr/bin/env bash
|
|
120
|
+
. "$HOME/shim-config"
|
|
121
|
+
echo "gh $*" >> "$CALLS"
|
|
122
|
+
echo "fake gh: no request was made" >&2
|
|
123
|
+
exit 1
|
|
124
|
+
SHIM
|
|
125
|
+
chmod +x "$HOME/shims/gh"
|
|
126
|
+
|
|
127
|
+
cat > "$RUN_DIR/pr.json" <<'JSON'
|
|
128
|
+
{
|
|
129
|
+
"number": 1337,
|
|
130
|
+
"title": "Cache tenant settings in the request path",
|
|
131
|
+
"author": { "login": "asha-platform" },
|
|
132
|
+
"headRefName": "feat/tenant-settings-cache",
|
|
133
|
+
"baseRefName": "main",
|
|
134
|
+
"headRefOid": "9f3a8c0211dbb5fe7a82a2c1b08e0a45c2d1ee01",
|
|
135
|
+
"url": "https://github.com/example/repo/pull/1337"
|
|
136
|
+
}
|
|
137
|
+
JSON
|
|
138
|
+
|
|
139
|
+
cat > "$RUN_DIR/log.jsonl" <<'LOG'
|
|
140
|
+
{"stage":"preflight","status":"done","missingOptional":["codex"]}
|
|
141
|
+
{"stage":"setup","status":"done"}
|
|
142
|
+
{"stage":"context","status":"done","codeIntelligence":false,"interface":"none"}
|
|
143
|
+
{"stage":"specialists","status":"done"}
|
|
144
|
+
{"stage":"dedupe","status":"done"}
|
|
145
|
+
{"stage":"critic","status":"done"}
|
|
146
|
+
{"stage":"peer-review","status":"done","provider":"claude"}
|
|
147
|
+
{"stage":"report","status":"done"}
|
|
148
|
+
LOG
|
|
149
|
+
|
|
150
|
+
# The PR under review, as setup would have left it: the filtered diff, and a
|
|
151
|
+
# worktree holding the head state the diff produces. The hunk headers count the
|
|
152
|
+
# lines they carry, and every finding below cites a line inside a hunk, so
|
|
153
|
+
# nothing here contradicts anything else.
|
|
154
|
+
cat > "$RUN_DIR/diff.patch" <<'PATCH'
|
|
155
|
+
diff --git a/src/settings/cache.ts b/src/settings/cache.ts
|
|
156
|
+
--- a/src/settings/cache.ts
|
|
157
|
+
+++ b/src/settings/cache.ts
|
|
158
|
+
@@ -1,5 +1,13 @@
|
|
159
|
+
const store = new Map<string, Settings>()
|
|
160
|
+
|
|
161
|
+
+export function put(tenantId: string, settings: Settings) {
|
|
162
|
+
+ store.set(tenantId, settings)
|
|
163
|
+
+}
|
|
164
|
+
+
|
|
165
|
+
+export function get(tenantId: string): Settings | undefined {
|
|
166
|
+
+ return store.get(tenantId)
|
|
167
|
+
+}
|
|
168
|
+
+
|
|
169
|
+
export function clear() {
|
|
170
|
+
store.clear()
|
|
171
|
+
}
|
|
172
|
+
diff --git a/src/settings/loader.ts b/src/settings/loader.ts
|
|
173
|
+
--- a/src/settings/loader.ts
|
|
174
|
+
+++ b/src/settings/loader.ts
|
|
175
|
+
@@ -9,3 +9,7 @@
|
|
176
|
+
export async function load(tenantId: string) {
|
|
177
|
+
- return fetchSettings(tenantId)
|
|
178
|
+
+ const hit = get(tenantId)
|
|
179
|
+
+ if (hit) return hit
|
|
180
|
+
+ const fresh = await fetchSettings(tenantId)
|
|
181
|
+
+ put(tenantId, fresh)
|
|
182
|
+
+ return fresh
|
|
183
|
+
}
|
|
184
|
+
PATCH
|
|
185
|
+
|
|
186
|
+
mkdir -p "$RUN_DIR/worktree/src/settings"
|
|
187
|
+
|
|
188
|
+
cat > "$RUN_DIR/worktree/src/settings/cache.ts" <<'TS'
|
|
189
|
+
const store = new Map<string, Settings>()
|
|
190
|
+
|
|
191
|
+
export function put(tenantId: string, settings: Settings) {
|
|
192
|
+
store.set(tenantId, settings)
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
export function get(tenantId: string): Settings | undefined {
|
|
196
|
+
return store.get(tenantId)
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
export function clear() {
|
|
200
|
+
store.clear()
|
|
201
|
+
}
|
|
202
|
+
TS
|
|
203
|
+
|
|
204
|
+
cat > "$RUN_DIR/worktree/src/settings/loader.ts" <<'TS'
|
|
205
|
+
import { get, put } from './cache'
|
|
206
|
+
|
|
207
|
+
type Settings = { theme: string }
|
|
208
|
+
|
|
209
|
+
async function fetchSettings(tenantId: string): Promise<Settings> {
|
|
210
|
+
return { theme: 'default' }
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
export async function load(tenantId: string) {
|
|
214
|
+
const hit = get(tenantId)
|
|
215
|
+
if (hit) return hit
|
|
216
|
+
const fresh = await fetchSettings(tenantId)
|
|
217
|
+
put(tenantId, fresh)
|
|
218
|
+
return fresh
|
|
219
|
+
}
|
|
220
|
+
TS
|
|
221
|
+
|
|
222
|
+
cat > "$RUN_DIR/findings.final.json" <<'JSON'
|
|
223
|
+
[
|
|
224
|
+
{
|
|
225
|
+
"id": "security-1",
|
|
226
|
+
"file": "src/settings/cache.ts",
|
|
227
|
+
"line": 4,
|
|
228
|
+
"severity": "high",
|
|
229
|
+
"risk": { "impact": "high", "likelihood": "likely", "confidence": "high", "action": "must-fix" },
|
|
230
|
+
"domain": "security",
|
|
231
|
+
"title": "Tenant settings cache is a process-global Map with no eviction",
|
|
232
|
+
"description": "Observation: put() writes into a module-level Map with no bound and no TTL.\n\nWhy it matters: a settings change never reaches the cached copy.\n\nSuggested direction: bound the map and give entries a TTL."
|
|
233
|
+
},
|
|
234
|
+
{
|
|
235
|
+
"id": "bugs-1",
|
|
236
|
+
"file": "src/settings/loader.ts",
|
|
237
|
+
"line": 12,
|
|
238
|
+
"severity": "medium",
|
|
239
|
+
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "should-fix" },
|
|
240
|
+
"domain": "bugs",
|
|
241
|
+
"title": "Concurrent loads for the same tenant each hit the network",
|
|
242
|
+
"description": "Observation: load() awaits fetchSettings before writing back.\n\nWhy it matters: N concurrent first requests produce N fetches.\n\nSuggested direction: cache the in-flight promise."
|
|
243
|
+
},
|
|
244
|
+
{
|
|
245
|
+
"id": "perf-1",
|
|
246
|
+
"file": "src/settings/cache.ts",
|
|
247
|
+
"line": 12,
|
|
248
|
+
"severity": "low",
|
|
249
|
+
"risk": { "impact": "low", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
250
|
+
"domain": "performance",
|
|
251
|
+
"title": "clear() evicts every tenant, not the one whose settings changed",
|
|
252
|
+
"description": "Observation: clear() calls store.clear() (src/settings/cache.ts:12) and is the only invalidation the module offers.\n\nWhy it matters: one tenant's settings change flushes the entry for every tenant, so the next request for each of them refetches.\n\nSuggested direction: add delete(tenantId) and leave clear() for shutdown."
|
|
253
|
+
},
|
|
254
|
+
{
|
|
255
|
+
"id": "arch-1",
|
|
256
|
+
"file": "src/settings/loader.ts",
|
|
257
|
+
"line": 10,
|
|
258
|
+
"severity": "medium",
|
|
259
|
+
"risk": { "impact": "medium", "likelihood": "possible", "confidence": "medium", "action": "consider" },
|
|
260
|
+
"domain": "architecture",
|
|
261
|
+
"title": "The loader owns the cache rather than being handed one",
|
|
262
|
+
"description": "Observation: load() imports the cache module directly.\n\nWhy it matters: no caller can swap the policy.\n\nSuggested direction: take the cache as a parameter."
|
|
263
|
+
},
|
|
264
|
+
{
|
|
265
|
+
"id": "smell-1",
|
|
266
|
+
"file": "src/settings/cache.ts",
|
|
267
|
+
"line": 8,
|
|
268
|
+
"severity": "low",
|
|
269
|
+
"risk": { "impact": "low", "likelihood": "unlikely", "confidence": "medium", "action": "optional" },
|
|
270
|
+
"domain": "code-smells",
|
|
271
|
+
"title": "get() hands back the stored object, so a caller can mutate the cache",
|
|
272
|
+
"description": "Observation: get() returns store.get(tenantId) directly (src/settings/cache.ts:8).\n\nWhy it matters: a caller that edits the returned settings edits every later reader's copy.\n\nSuggested direction: freeze the value on put, or return a copy."
|
|
273
|
+
}
|
|
274
|
+
]
|
|
275
|
+
JSON
|
|
276
|
+
|
|
277
|
+
# Ticked security-1 and bugs-1, unticked security-1, then unticked and re-ticked
|
|
278
|
+
# perf-1. Last event per id leaves bugs-1 and perf-1 selected.
|
|
279
|
+
cat > "$RUN_DIR/state/events" <<'EVENTS'
|
|
280
|
+
{"type":"select","findingId":"security-1","timestamp":1789600100000}
|
|
281
|
+
{"type":"select","findingId":"bugs-1","timestamp":1789600101000}
|
|
282
|
+
{"type":"deselect","findingId":"security-1","timestamp":1789600102000}
|
|
283
|
+
{"type":"select","findingId":"perf-1","timestamp":1789600103000}
|
|
284
|
+
{"type":"deselect","findingId":"perf-1","timestamp":1789600104000}
|
|
285
|
+
{"type":"select","findingId":"perf-1","timestamp":1789600105000}
|
|
286
|
+
EVENTS
|
|
287
|
+
|
|
288
|
+
echo '[]' > "$RUN_DIR/findings/tests.json"
|
|
289
|
+
echo "http://127.0.0.1:4599" > "$RUN_DIR/state/server-info"
|
|
290
|
+
echo "<html>report</html>" > "$RUN_DIR/screen/findings.html"
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: post-folds-selection-events
|
|
3
|
+
description: Typing `post` posts what state/events leaves selected, folded last-event-wins per finding id.
|
|
4
|
+
tags: [magpie, post, selection, scaffold]
|
|
5
|
+
max_turns: 25
|
|
6
|
+
timeout_seconds: 900
|
|
7
|
+
allowed_tools: [Read, Glob, Grep, Skill, Write, Edit, Bash]
|
|
8
|
+
expected_outcome: Posts exactly bugs-1 and perf-1 via magpie post --ids, leaves the deselected security-1 alone, logs the post stage done and re-renders the report.
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
I've been ticking through the magpie report for PR 1337. post
|