@iceinvein/agent-skills 0.1.38 → 0.1.39
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/index.json +1 -1
- package/skills/magpie/README.md +2 -1
- package/skills/magpie/SKILL.md +29 -530
- package/skills/magpie/package.json +1 -1
- package/skills/magpie/references/critic.md +58 -0
- package/skills/magpie/references/peer-review.md +84 -0
- package/skills/magpie/references/specialists.md +391 -0
- package/skills/magpie/scripts/__tests__/helper.test.ts +40 -0
- package/skills/magpie/scripts/__tests__/skill-lint.test.ts +116 -28
- package/skills/magpie/scripts/__tests__/status-cmd.test.ts +13 -0
- package/skills/magpie/scripts/helper.js +24 -13
- package/skills/magpie/scripts/status-cmd.ts +10 -1
- package/skills/magpie/skill.json +2 -1
|
@@ -2,10 +2,35 @@ import { expect, test } from 'bun:test'
|
|
|
2
2
|
import { readFile } from 'node:fs/promises'
|
|
3
3
|
|
|
4
4
|
const SKILL = new URL('../../SKILL.md', import.meta.url).pathname
|
|
5
|
+
const SKILL_JSON = new URL('../../skill.json', import.meta.url).pathname
|
|
6
|
+
const ref = (name: string) => new URL(`../../references/${name}`, import.meta.url).pathname
|
|
5
7
|
const FOCUSES = ['security', 'bugs', 'performance', 'code-smells', 'architecture'] as const
|
|
6
8
|
|
|
7
|
-
test('SKILL.md
|
|
9
|
+
test('every references/ path SKILL.md cites exists on disk', async () => {
|
|
8
10
|
const text = await readFile(SKILL, 'utf8')
|
|
11
|
+
const cited = [...text.matchAll(/references\/[a-z0-9-]+\.md/g)].map((m) => m[0])
|
|
12
|
+
expect(cited.length).toBeGreaterThan(0)
|
|
13
|
+
for (const rel of new Set(cited)) {
|
|
14
|
+
const body = await readFile(new URL(`../../${rel}`, import.meta.url).pathname, 'utf8')
|
|
15
|
+
expect(body.length).toBeGreaterThan(0)
|
|
16
|
+
}
|
|
17
|
+
})
|
|
18
|
+
|
|
19
|
+
test('references/ ships in the install bundle', async () => {
|
|
20
|
+
// resolveBundlePaths silently drops include entries that match nothing, so a
|
|
21
|
+
// missing entry here would install a SKILL.md whose prompts are all 404s.
|
|
22
|
+
const manifest = JSON.parse(await readFile(SKILL_JSON, 'utf8')) as {
|
|
23
|
+
bundle: { include: string[] }
|
|
24
|
+
}
|
|
25
|
+
const covers = (p: string) =>
|
|
26
|
+
manifest.bundle.include.some((inc) => inc === p || p.startsWith(`${inc}/`))
|
|
27
|
+
for (const name of ['specialists.md', 'critic.md', 'peer-review.md']) {
|
|
28
|
+
expect(covers(`references/${name}`)).toBe(true)
|
|
29
|
+
}
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
test('references/specialists.md has a block for every focus', async () => {
|
|
33
|
+
const text = await readFile(ref('specialists.md'), 'utf8')
|
|
9
34
|
for (const focus of FOCUSES) {
|
|
10
35
|
const tag = `magpie-specialist-${focus}`
|
|
11
36
|
const fence = `\`\`\`${tag}`
|
|
@@ -13,23 +38,69 @@ test('SKILL.md has a specialist block for every focus', async () => {
|
|
|
13
38
|
const start = text.indexOf(fence)
|
|
14
39
|
const end = text.indexOf('```', start + tag.length + 3)
|
|
15
40
|
const block = text.slice(start, end)
|
|
16
|
-
expect(block).toContain('
|
|
41
|
+
expect(block).toContain('Output Contract')
|
|
17
42
|
expect(block).toContain(focus)
|
|
18
43
|
}
|
|
19
44
|
})
|
|
20
45
|
|
|
21
|
-
test('
|
|
22
|
-
const text = await readFile(
|
|
23
|
-
// The
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
expect(
|
|
46
|
+
test('references/specialists.md carries the output contract next to the blocks', async () => {
|
|
47
|
+
const text = await readFile(ref('specialists.md'), 'utf8')
|
|
48
|
+
// The contract has to travel with the prompts: the orchestrator assembles
|
|
49
|
+
// both into one subagent prompt from this single file.
|
|
50
|
+
const contract = text.slice(0, text.indexOf('```magpie-specialist-'))
|
|
51
|
+
expect(contract).toContain('## Output Contract')
|
|
52
|
+
expect(contract).toMatch(/findings\/<focus>\.json/)
|
|
53
|
+
expect(contract).toMatch(/Write findings to/i)
|
|
54
|
+
// Severity, impact, likelihood, confidence, action enums must all be listed.
|
|
55
|
+
expect(contract).toMatch(/"blocker".*"high".*"medium".*"low"/)
|
|
56
|
+
expect(contract).toMatch(/"critical".*"high".*"medium".*"low"/)
|
|
57
|
+
expect(contract).toMatch(/"likely".*"possible".*"edge-case".*"unknown"/)
|
|
58
|
+
expect(contract).toMatch(/"must-fix".*"should-fix".*"consider".*"optional"/)
|
|
59
|
+
expect(contract).toMatch(/"impact"[\s\S]*"likelihood"[\s\S]*"confidence"[\s\S]*"action"/)
|
|
60
|
+
expect(contract).toMatch(/"body"[\s\S]*"startLine"[\s\S]*"endLine"/)
|
|
61
|
+
// Anti-patterns flagged explicitly so subagents don't repeat the JSON-shape mistakes.
|
|
62
|
+
expect(contract).toMatch(/NOT "lines"/)
|
|
63
|
+
expect(contract).toMatch(/NOT "recommendation"/)
|
|
27
64
|
})
|
|
28
65
|
|
|
29
|
-
test('
|
|
30
|
-
const text = await readFile(
|
|
66
|
+
test('references/critic.md holds the rubric and both placeholders', async () => {
|
|
67
|
+
const text = await readFile(ref('critic.md'), 'utf8')
|
|
31
68
|
expect(text).toContain('```magpie-critic')
|
|
69
|
+
expect(text).toContain('<<DEDUPED_FINDINGS_COMPACT>>')
|
|
70
|
+
expect(text).toContain('<<DIFF_EXCERPT>>')
|
|
71
|
+
expect(text).toContain('review-critic')
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
test('references/peer-review.md holds the prompt and the Claude preamble', async () => {
|
|
75
|
+
const text = await readFile(ref('peer-review.md'), 'utf8')
|
|
32
76
|
expect(text).toContain('```magpie-peer-review')
|
|
77
|
+
expect(text).toContain('```magpie-peer-review-claude-preamble')
|
|
78
|
+
expect(text).toContain('review-peer-review')
|
|
79
|
+
for (const ph of ['<<PRIMARY_PROVIDER>>', '<<PEER_PROVIDER>>', '<<KEPT_FINDINGS_COMPACT>>']) {
|
|
80
|
+
expect(text).toContain(ph)
|
|
81
|
+
}
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
test('SKILL.md sends each stage to the reference file it needs', async () => {
|
|
85
|
+
const text = await readFile(SKILL, 'utf8')
|
|
86
|
+
const section = (heading: string) => {
|
|
87
|
+
const start = text.indexOf(heading)
|
|
88
|
+
expect(start).toBeGreaterThan(-1)
|
|
89
|
+
const next = text.indexOf('\n### ', start + heading.length)
|
|
90
|
+
return text.slice(start, next === -1 ? undefined : next)
|
|
91
|
+
}
|
|
92
|
+
expect(section('### 3. Specialists')).toContain('references/specialists.md')
|
|
93
|
+
expect(section('### 5. Critic')).toContain('references/critic.md')
|
|
94
|
+
expect(section('### 6. Peer review')).toContain('references/peer-review.md')
|
|
95
|
+
})
|
|
96
|
+
|
|
97
|
+
test('SKILL.md no longer inlines the prompt bodies it moved out', async () => {
|
|
98
|
+
const text = await readFile(SKILL, 'utf8')
|
|
99
|
+
for (const tag of ['magpie-specialist-', 'magpie-critic', 'magpie-peer-review']) {
|
|
100
|
+
expect(text).not.toContain(`\`\`\`${tag}`)
|
|
101
|
+
}
|
|
102
|
+
// The walkthrough is the always-read part; keep it small enough to be cheap.
|
|
103
|
+
expect(text.split(/\s+/).length).toBeLessThan(2600)
|
|
33
104
|
})
|
|
34
105
|
|
|
35
106
|
test('styles.css declares a prefers-color-scheme:dark block that overrides core tokens', async () => {
|
|
@@ -104,25 +175,42 @@ test('severity chips and submit button no longer hardcode `color: white` (must f
|
|
|
104
175
|
}
|
|
105
176
|
})
|
|
106
177
|
|
|
107
|
-
test('SKILL.md
|
|
178
|
+
test('SKILL.md checks for a resumable run before minting a new run id', async () => {
|
|
108
179
|
const text = await readFile(SKILL, 'utf8')
|
|
109
|
-
//
|
|
110
|
-
//
|
|
111
|
-
const
|
|
112
|
-
expect(
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
//
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
expect(
|
|
125
|
-
|
|
180
|
+
// §0 always computed a fresh `pr-<n>-<epoch>` id, so the resume section
|
|
181
|
+
// (which keyed off that brand-new directory) could never fire.
|
|
182
|
+
const setupSection = text.slice(0, text.indexOf('### 1. Setup'))
|
|
183
|
+
expect(setupSection).toContain('magpie --list-runs')
|
|
184
|
+
expect(setupSection).toMatch(/active/i)
|
|
185
|
+
})
|
|
186
|
+
|
|
187
|
+
test('SKILL.md does not gate resume on state/server-info', async () => {
|
|
188
|
+
const text = await readFile(SKILL, 'utf8')
|
|
189
|
+
const start = text.indexOf('## Resuming a crashed run')
|
|
190
|
+
expect(start).toBeGreaterThan(-1)
|
|
191
|
+
const section = text.slice(start, text.indexOf('## Aborting', start))
|
|
192
|
+
// server-info is deleted when the server idles out, so a resumable run
|
|
193
|
+
// fails that check. log.jsonl is the durable signal.
|
|
194
|
+
expect(section).toContain('log.jsonl')
|
|
195
|
+
expect(section).not.toMatch(/if .*server-info.* exists/i)
|
|
196
|
+
// Resuming must restart the server; the old one is gone.
|
|
197
|
+
expect(section).toContain('magpie serve')
|
|
198
|
+
// `context` is a no-op stage: say so, or the agent stalls on it.
|
|
199
|
+
expect(section).toContain('context')
|
|
200
|
+
})
|
|
201
|
+
|
|
202
|
+
test('SKILL.md names the report buttons that actually exist', async () => {
|
|
203
|
+
const text = await readFile(SKILL, 'utf8')
|
|
204
|
+
const actionBar = await readFile(
|
|
205
|
+
new URL('../render-action-bar.ts', import.meta.url).pathname,
|
|
206
|
+
'utf8',
|
|
207
|
+
)
|
|
208
|
+
for (const label of ['Post Selected', 'Post Recommended']) {
|
|
209
|
+
expect(actionBar).toContain(label)
|
|
210
|
+
expect(text).toContain(label)
|
|
211
|
+
}
|
|
212
|
+
// The old label was never rendered anywhere.
|
|
213
|
+
expect(text).not.toContain('Post to PR')
|
|
126
214
|
})
|
|
127
215
|
|
|
128
216
|
test('SKILL.md has the stage walkthrough', async () => {
|
|
@@ -31,6 +31,19 @@ test('empty log reports nothing completed', async () => {
|
|
|
31
31
|
expect(result.next).toBe('setup')
|
|
32
32
|
})
|
|
33
33
|
|
|
34
|
+
test('a skipped stage advances the resume pointer past it', async () => {
|
|
35
|
+
// `context` has no work in the pipeline; SKILL.md tells the orchestrator to
|
|
36
|
+
// log it as skipped. If `skipped` did not advance the pointer, a resume
|
|
37
|
+
// would be sent back to a stage that has no step to run.
|
|
38
|
+
await writeFile(
|
|
39
|
+
join(runDir, 'log.jsonl'),
|
|
40
|
+
`${JSON.stringify({ stage: 'setup', status: 'done' })}\n${JSON.stringify({ stage: 'context', status: 'skipped' })}\n`,
|
|
41
|
+
)
|
|
42
|
+
const result = await runStatus(runDir)
|
|
43
|
+
expect(result.lastCompleted).toBe('context')
|
|
44
|
+
expect(result.next).toBe('specialists')
|
|
45
|
+
})
|
|
46
|
+
|
|
34
47
|
test('error stage halts progression', async () => {
|
|
35
48
|
await writeFile(
|
|
36
49
|
join(runDir, 'log.jsonl'),
|
|
@@ -352,6 +352,23 @@
|
|
|
352
352
|
}
|
|
353
353
|
}
|
|
354
354
|
|
|
355
|
+
function isSelected(id) {
|
|
356
|
+
return findCheckboxes(id).some((cb) => cb.checked && !cb.disabled)
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
// Assigning `cb.checked` in script fires no change event, so any programmatic
|
|
360
|
+
// selection has to emit its own /events record. `state/events` is the only
|
|
361
|
+
// channel the orchestrator can read when the user types `post` in the
|
|
362
|
+
// terminal instead of using the in-page buttons; a silent set drops the
|
|
363
|
+
// finding from that post.
|
|
364
|
+
function setCheckedAndNotify(id, checked) {
|
|
365
|
+
const before = isSelected(id)
|
|
366
|
+
setChecked(id, checked)
|
|
367
|
+
const after = isSelected(id)
|
|
368
|
+
if (after === before) return
|
|
369
|
+
post({ type: after ? 'select' : 'deselect', findingId: id, timestamp: Date.now() })
|
|
370
|
+
}
|
|
371
|
+
|
|
355
372
|
// ---------------------------------------------------------------------------
|
|
356
373
|
// Bulk selection
|
|
357
374
|
// ---------------------------------------------------------------------------
|
|
@@ -367,8 +384,8 @@
|
|
|
367
384
|
if (el.getAttribute('data-posted') === 'true') continue
|
|
368
385
|
ids.add(el.getAttribute('data-finding-id'))
|
|
369
386
|
}
|
|
370
|
-
for (const id of ids)
|
|
371
|
-
|
|
387
|
+
for (const id of ids) setCheckedAndNotify(id, true)
|
|
388
|
+
recountSelected()
|
|
372
389
|
}
|
|
373
390
|
|
|
374
391
|
function handleSelectRecommended() {
|
|
@@ -378,8 +395,8 @@
|
|
|
378
395
|
if (el.getAttribute('data-posted') === 'true') continue
|
|
379
396
|
ids.add(el.getAttribute('data-finding-id'))
|
|
380
397
|
}
|
|
381
|
-
for (const id of ids)
|
|
382
|
-
|
|
398
|
+
for (const id of ids) setCheckedAndNotify(id, true)
|
|
399
|
+
recountSelected()
|
|
383
400
|
}
|
|
384
401
|
|
|
385
402
|
// ---------------------------------------------------------------------------
|
|
@@ -399,14 +416,8 @@
|
|
|
399
416
|
if (cbs.length === 0) return
|
|
400
417
|
const cb = cbs[0]
|
|
401
418
|
if (cb.disabled) return
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
post({
|
|
405
|
-
type: next ? 'select' : 'deselect',
|
|
406
|
-
findingId: id,
|
|
407
|
-
timestamp: Date.now(),
|
|
408
|
-
})
|
|
409
|
-
updateSelectedCount()
|
|
419
|
+
setCheckedAndNotify(id, !cb.checked)
|
|
420
|
+
recountSelected()
|
|
410
421
|
}
|
|
411
422
|
|
|
412
423
|
// ---------------------------------------------------------------------------
|
|
@@ -535,7 +546,7 @@
|
|
|
535
546
|
if (cb.disabled) continue
|
|
536
547
|
cb.checked = target.checked
|
|
537
548
|
}
|
|
538
|
-
|
|
549
|
+
recountSelected()
|
|
539
550
|
post({
|
|
540
551
|
type: target.checked ? 'select' : 'deselect',
|
|
541
552
|
findingId: id,
|
|
@@ -13,6 +13,12 @@ const ORDER = [
|
|
|
13
13
|
] as const
|
|
14
14
|
|
|
15
15
|
export type StatusResult = {
|
|
16
|
+
/**
|
|
17
|
+
* Highest stage the log says is behind us. A stage logged `skipped` counts:
|
|
18
|
+
* `context` has no work in the pipeline and is always logged that way, so
|
|
19
|
+
* treating it as unfinished would send a resume back to a stage that has no
|
|
20
|
+
* step to run.
|
|
21
|
+
*/
|
|
16
22
|
lastCompleted: (typeof ORDER)[number] | null
|
|
17
23
|
next: (typeof ORDER)[number] | 'cleanup'
|
|
18
24
|
error: string | null
|
|
@@ -35,7 +41,10 @@ export async function runStatus(runDir: string): Promise<StatusResult> {
|
|
|
35
41
|
error = stage
|
|
36
42
|
break
|
|
37
43
|
}
|
|
38
|
-
if (
|
|
44
|
+
if (
|
|
45
|
+
(status === 'done' || status === 'skipped') &&
|
|
46
|
+
(ORDER as readonly string[]).includes(stage)
|
|
47
|
+
) {
|
|
39
48
|
lastCompleted = stage as StatusResult['lastCompleted']
|
|
40
49
|
}
|
|
41
50
|
}
|
package/skills/magpie/skill.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "magpie",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.8.0",
|
|
4
4
|
"description": "Interactive PR review pipeline. Runs five parallel specialist subagents (security, bugs, performance, code-smells, architecture), dedupes findings, applies a critic rubric, peer-reviews via codex exec (falling back to a Claude second opinion when codex is unavailable), and serves an interactive HTML report for selecting findings to post via gh. Bundles a Bun CLI installed onto PATH via the skill's postinstall step. Use when the user asks to review a GitHub pull request.",
|
|
5
5
|
"author": "iceinvein",
|
|
6
6
|
"type": "prompt",
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
"bundle": {
|
|
14
14
|
"include": [
|
|
15
15
|
"bin",
|
|
16
|
+
"references",
|
|
16
17
|
"scripts",
|
|
17
18
|
"templates",
|
|
18
19
|
"fixtures/example-pr",
|