@iceinvein/agent-skills 0.1.28 → 0.1.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli/index.js +42 -12
- package/package.json +1 -1
- package/skills/index.json +1 -1
- package/skills/magpie/SKILL.md +18 -37
- package/skills/magpie/bin/magpie.ts +13 -2
- package/skills/magpie/fixtures/fake-gh-with-lockfile.sh +52 -0
- package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +45 -0
- package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +83 -0
- package/skills/magpie/scripts/__tests__/incremental.test.ts +87 -0
- package/skills/magpie/scripts/__tests__/path-filter.test.ts +161 -0
- package/skills/magpie/scripts/__tests__/post-cmd.test.ts +108 -8
- package/skills/magpie/scripts/__tests__/score.test.ts +51 -0
- package/skills/magpie/scripts/__tests__/server.test.ts +4 -1
- package/skills/magpie/scripts/__tests__/setup-cmd.test.ts +104 -1
- package/skills/magpie/scripts/__tests__/tests-check.test.ts +94 -0
- package/skills/magpie/scripts/__tests__/types.test.ts +9 -2
- package/skills/magpie/scripts/dedupe-cmd.ts +40 -3
- package/skills/magpie/scripts/evidence-filter.ts +73 -0
- package/skills/magpie/scripts/incremental.ts +75 -0
- package/skills/magpie/scripts/path-filter.ts +150 -0
- package/skills/magpie/scripts/post-cmd.ts +133 -9
- package/skills/magpie/scripts/score.ts +39 -0
- package/skills/magpie/scripts/setup-cmd.ts +73 -2
- package/skills/magpie/scripts/tests-check.ts +118 -0
- package/skills/magpie/scripts/types.ts +11 -1
- package/skills/magpie/skill.json +1 -1
|
@@ -3,6 +3,7 @@ import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
|
|
|
3
3
|
import { tmpdir } from 'node:os'
|
|
4
4
|
import { join } from 'node:path'
|
|
5
5
|
import {
|
|
6
|
+
computeEffortScore,
|
|
6
7
|
formatConversationBody,
|
|
7
8
|
formatInlineBody,
|
|
8
9
|
formatPostBody,
|
|
@@ -158,8 +159,8 @@ test('runPost skips ids already marked as posted', async () => {
|
|
|
158
159
|
await seedRunDir()
|
|
159
160
|
await writeFile(join(runDir, 'post-status.json'), JSON.stringify({ 'sec-1': 'posted' }))
|
|
160
161
|
const outcome = await runPost({ runDir, findingIds: ['sec-1', 'bug-1'], dryRun: true })
|
|
161
|
-
expect(outcome.results
|
|
162
|
-
expect(outcome.results
|
|
162
|
+
expect(outcome.results.find((r) => r.id === 'sec-1')?.status).toBe('already-posted')
|
|
163
|
+
expect(outcome.results.find((r) => r.id === 'bug-1')?.status).toBe('posted')
|
|
163
164
|
})
|
|
164
165
|
|
|
165
166
|
test('runPost surfaces gh failure per finding (here: gh binary missing)', async () => {
|
|
@@ -169,6 +170,7 @@ test('runPost surfaces gh failure per finding (here: gh binary missing)', async
|
|
|
169
170
|
runDir,
|
|
170
171
|
findingIds: ['sec-1'],
|
|
171
172
|
ghBin: '/does/not/exist/gh-binary',
|
|
173
|
+
includeSummary: 'never',
|
|
172
174
|
})
|
|
173
175
|
expect(outcome.ok).toBe(true) // overall request succeeds; per-id reflects gh's failure
|
|
174
176
|
expect(outcome.results[0]?.status).toBe('failed')
|
|
@@ -204,8 +206,9 @@ test('runPost falls back to a top-level PR comment when GitHub rejects the inlin
|
|
|
204
206
|
|
|
205
207
|
const outcome = await runPost({ runDir, findingIds: ['sec-1'], ghBin: fakeGh })
|
|
206
208
|
expect(outcome.ok).toBe(true)
|
|
207
|
-
|
|
208
|
-
expect(
|
|
209
|
+
const sec1 = outcome.results.find((r) => r.id === 'sec-1')
|
|
210
|
+
expect(sec1?.status).toBe('posted')
|
|
211
|
+
expect(sec1?.message).toMatch(/posted as PR comment/i)
|
|
209
212
|
const status = JSON.parse(await readFile(join(runDir, 'post-status.json'), 'utf8'))
|
|
210
213
|
expect(status['sec-1']).toBe('posted')
|
|
211
214
|
const log = await readFile(join(runDir, 'log.jsonl'), 'utf8')
|
|
@@ -226,8 +229,9 @@ test('runPost reports a combined failure when both inline and fallback gh calls
|
|
|
226
229
|
)
|
|
227
230
|
await Bun.spawn(['chmod', '+x', fakeGh]).exited
|
|
228
231
|
const outcome = await runPost({ runDir, findingIds: ['sec-1'], ghBin: fakeGh })
|
|
229
|
-
|
|
230
|
-
expect(
|
|
232
|
+
const sec1 = outcome.results.find((r) => r.id === 'sec-1')
|
|
233
|
+
expect(sec1?.status).toBe('failed')
|
|
234
|
+
expect(sec1?.message).toMatch(/inline.*fallback also failed/i)
|
|
231
235
|
})
|
|
232
236
|
|
|
233
237
|
test('formatReviewSummaryBody renders verdict, Needs Attention and Risk breakdown', () => {
|
|
@@ -265,7 +269,82 @@ test('formatReviewSummaryBody handles the empty case', () => {
|
|
|
265
269
|
expect(body).not.toContain('Risk breakdown')
|
|
266
270
|
})
|
|
267
271
|
|
|
268
|
-
test('
|
|
272
|
+
test('formatReviewSummaryBody includes effort score when files are provided', () => {
|
|
273
|
+
const body = formatReviewSummaryBody([], {
|
|
274
|
+
files: [{ path: 'src/a.ts', additions: 5, deletions: 0 }],
|
|
275
|
+
})
|
|
276
|
+
expect(body).toContain('Review effort')
|
|
277
|
+
expect(body).toMatch(/\d\/5/)
|
|
278
|
+
})
|
|
279
|
+
|
|
280
|
+
test('formatReviewSummaryBody renders walkthrough table sorted by churn', () => {
|
|
281
|
+
const body = formatReviewSummaryBody([], {
|
|
282
|
+
files: [
|
|
283
|
+
{ path: 'small.ts', additions: 1, deletions: 0 },
|
|
284
|
+
{ path: 'big.ts', additions: 200, deletions: 100 },
|
|
285
|
+
],
|
|
286
|
+
})
|
|
287
|
+
expect(body).toContain('Files changed')
|
|
288
|
+
expect(body).toContain('big.ts')
|
|
289
|
+
const bigIdx = body.indexOf('big.ts')
|
|
290
|
+
const smallIdx = body.indexOf('small.ts')
|
|
291
|
+
expect(bigIdx).toBeLessThan(smallIdx)
|
|
292
|
+
})
|
|
293
|
+
|
|
294
|
+
test('formatReviewSummaryBody surfaces risk hotspots when 2+ findings per file', () => {
|
|
295
|
+
const findings = [
|
|
296
|
+
{
|
|
297
|
+
...findingA,
|
|
298
|
+
id: 'h1',
|
|
299
|
+
file: 'src/hot.ts',
|
|
300
|
+
line: 5,
|
|
301
|
+
},
|
|
302
|
+
{
|
|
303
|
+
...findingA,
|
|
304
|
+
id: 'h2',
|
|
305
|
+
file: 'src/hot.ts',
|
|
306
|
+
line: 10,
|
|
307
|
+
},
|
|
308
|
+
{
|
|
309
|
+
...findingA,
|
|
310
|
+
id: 'h3',
|
|
311
|
+
file: 'src/cool.ts',
|
|
312
|
+
line: 1,
|
|
313
|
+
},
|
|
314
|
+
] as Parameters<typeof formatReviewSummaryBody>[0]
|
|
315
|
+
const body = formatReviewSummaryBody(findings)
|
|
316
|
+
expect(body).toContain('Risk hotspots')
|
|
317
|
+
expect(body).toContain('src/hot.ts')
|
|
318
|
+
expect(body).not.toContain('- `src/cool.ts`')
|
|
319
|
+
})
|
|
320
|
+
|
|
321
|
+
test('formatReviewSummaryBody renders incremental trailer with new commits', () => {
|
|
322
|
+
const body = formatReviewSummaryBody([], {
|
|
323
|
+
incremental: { previousSha: 'abcdef0123456', sameSha: false },
|
|
324
|
+
})
|
|
325
|
+
expect(body).toContain('Incremental review since')
|
|
326
|
+
expect(body).toContain('abcdef0')
|
|
327
|
+
})
|
|
328
|
+
|
|
329
|
+
test('formatReviewSummaryBody renders re-review trailer when sameSha', () => {
|
|
330
|
+
const body = formatReviewSummaryBody([], {
|
|
331
|
+
incremental: { previousSha: 'abcdef0123456', sameSha: true },
|
|
332
|
+
})
|
|
333
|
+
expect(body).toContain('Re-review')
|
|
334
|
+
})
|
|
335
|
+
|
|
336
|
+
test('computeEffortScore: small PR = 1, huge PR = 5', () => {
|
|
337
|
+
expect(computeEffortScore([])).toBe(1)
|
|
338
|
+
expect(computeEffortScore([{ path: 'a', additions: 5, deletions: 5 }])).toBe(1)
|
|
339
|
+
const huge = Array.from({ length: 25 }, (_, i) => ({
|
|
340
|
+
path: `f${i}.ts`,
|
|
341
|
+
additions: 100,
|
|
342
|
+
deletions: 0,
|
|
343
|
+
}))
|
|
344
|
+
expect(computeEffortScore(huge)).toBe(5)
|
|
345
|
+
})
|
|
346
|
+
|
|
347
|
+
test('runPost auto-posts the review summary on any non-empty batch', async () => {
|
|
269
348
|
await seedRunDir()
|
|
270
349
|
const outcome = await runPost({
|
|
271
350
|
runDir,
|
|
@@ -280,16 +359,37 @@ test('runPost auto-posts the review summary when batch has 2+ pending findings',
|
|
|
280
359
|
expect(status.__summary__).toBe('posted')
|
|
281
360
|
})
|
|
282
361
|
|
|
283
|
-
test('runPost
|
|
362
|
+
test('runPost auto-posts the summary even on single-finding batches', async () => {
|
|
284
363
|
await seedRunDir()
|
|
285
364
|
const outcome = await runPost({
|
|
286
365
|
runDir,
|
|
287
366
|
findingIds: ['sec-1'],
|
|
288
367
|
dryRun: true,
|
|
289
368
|
})
|
|
369
|
+
expect(outcome.results.find((r) => r.id === '__summary__')?.status).toBe('posted')
|
|
370
|
+
})
|
|
371
|
+
|
|
372
|
+
test('runPost auto-skips the summary when the batch has zero findings', async () => {
|
|
373
|
+
await seedRunDir()
|
|
374
|
+
const outcome = await runPost({
|
|
375
|
+
runDir,
|
|
376
|
+
findingIds: ['nope-id'],
|
|
377
|
+
dryRun: true,
|
|
378
|
+
})
|
|
290
379
|
expect(outcome.results.find((r) => r.id === '__summary__')).toBeUndefined()
|
|
291
380
|
})
|
|
292
381
|
|
|
382
|
+
test('runPost with includeSummary: always posts the summary even on empty batches', async () => {
|
|
383
|
+
await seedRunDir()
|
|
384
|
+
const outcome = await runPost({
|
|
385
|
+
runDir,
|
|
386
|
+
findingIds: [],
|
|
387
|
+
dryRun: true,
|
|
388
|
+
includeSummary: 'always',
|
|
389
|
+
})
|
|
390
|
+
expect(outcome.results.find((r) => r.id === '__summary__')?.status).toBe('posted')
|
|
391
|
+
})
|
|
392
|
+
|
|
293
393
|
test('runPost forces the summary when includeSummary: always', async () => {
|
|
294
394
|
await seedRunDir()
|
|
295
395
|
const outcome = await runPost({
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
import { expect, test } from 'bun:test'
|
|
2
|
+
import { DEFAULT_THRESHOLD, scoreRisk } from '../score.ts'
|
|
3
|
+
import type { Risk } from '../types.ts'
|
|
4
|
+
|
|
5
|
+
const risk = (over: Partial<Risk> = {}): Risk => ({
|
|
6
|
+
impact: 'medium',
|
|
7
|
+
likelihood: 'possible',
|
|
8
|
+
confidence: 'medium',
|
|
9
|
+
action: 'should-fix',
|
|
10
|
+
...over,
|
|
11
|
+
})
|
|
12
|
+
|
|
13
|
+
test('scoreRisk: maxed-out risk hits ~10', () => {
|
|
14
|
+
const s = scoreRisk({
|
|
15
|
+
impact: 'critical',
|
|
16
|
+
likelihood: 'likely',
|
|
17
|
+
confidence: 'high',
|
|
18
|
+
action: 'must-fix',
|
|
19
|
+
})
|
|
20
|
+
expect(s).toBeCloseTo(10, 1)
|
|
21
|
+
})
|
|
22
|
+
|
|
23
|
+
test('scoreRisk: floor risk is near 1', () => {
|
|
24
|
+
const s = scoreRisk({
|
|
25
|
+
impact: 'low',
|
|
26
|
+
likelihood: 'edge-case',
|
|
27
|
+
confidence: 'low',
|
|
28
|
+
action: 'optional',
|
|
29
|
+
})
|
|
30
|
+
expect(s).toBeGreaterThanOrEqual(1)
|
|
31
|
+
expect(s).toBeLessThan(2.5)
|
|
32
|
+
})
|
|
33
|
+
|
|
34
|
+
test('scoreRisk: impact dominates over action', () => {
|
|
35
|
+
const highImpact = scoreRisk(risk({ impact: 'critical', action: 'optional' }))
|
|
36
|
+
const lowImpact = scoreRisk(risk({ impact: 'low', action: 'must-fix' }))
|
|
37
|
+
expect(highImpact).toBeGreaterThan(lowImpact)
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
test('scoreRisk: confidence pulls weight', () => {
|
|
41
|
+
const highConf = scoreRisk(risk({ confidence: 'high' }))
|
|
42
|
+
const lowConf = scoreRisk(risk({ confidence: 'low' }))
|
|
43
|
+
expect(highConf - lowConf).toBeGreaterThan(0.5)
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
test('DEFAULT_THRESHOLD is sensible (drops floor, keeps medium)', () => {
|
|
47
|
+
expect(DEFAULT_THRESHOLD).toBeGreaterThan(0)
|
|
48
|
+
expect(DEFAULT_THRESHOLD).toBeLessThan(5)
|
|
49
|
+
const medium = scoreRisk(risk())
|
|
50
|
+
expect(medium).toBeGreaterThanOrEqual(DEFAULT_THRESHOLD)
|
|
51
|
+
})
|
|
@@ -126,7 +126,10 @@ test('POST /post wires through to runPost (dry-run) and returns results', async
|
|
|
126
126
|
}
|
|
127
127
|
expect(body.ok).toBe(true)
|
|
128
128
|
expect(body.target).toEqual({ repo: 'o/r', number: 7 })
|
|
129
|
-
expect(body.results
|
|
129
|
+
expect(body.results.find((r) => r.id === 'sec-1')).toMatchObject({
|
|
130
|
+
id: 'sec-1',
|
|
131
|
+
status: 'posted',
|
|
132
|
+
})
|
|
130
133
|
})
|
|
131
134
|
|
|
132
135
|
test('writes server-info on start with url and port', async () => {
|
|
@@ -1,10 +1,12 @@
|
|
|
1
1
|
import { afterEach, beforeEach, expect, test } from 'bun:test'
|
|
2
|
-
import { mkdtemp, readdir, rm, writeFile } from 'node:fs/promises'
|
|
2
|
+
import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises'
|
|
3
3
|
import { tmpdir } from 'node:os'
|
|
4
4
|
import { join } from 'node:path'
|
|
5
5
|
import { runSetup } from '../setup-cmd.ts'
|
|
6
6
|
|
|
7
7
|
const FAKE_GH = new URL('../../fixtures/fake-gh.sh', import.meta.url).pathname
|
|
8
|
+
const FAKE_GH_WITH_LOCKFILE = new URL('../../fixtures/fake-gh-with-lockfile.sh', import.meta.url)
|
|
9
|
+
.pathname
|
|
8
10
|
const FALSE_BIN = '/usr/bin/false'
|
|
9
11
|
|
|
10
12
|
let repo: string
|
|
@@ -64,6 +66,107 @@ test('runSetup with missing dep returns non-zero and cleans up', async () => {
|
|
|
64
66
|
expect(contents.filter((c) => c !== 'log.jsonl')).toHaveLength(0)
|
|
65
67
|
})
|
|
66
68
|
|
|
69
|
+
test('runSetup filters excluded files by default and preserves raw diff', async () => {
|
|
70
|
+
const exit = await runSetup({
|
|
71
|
+
runDir,
|
|
72
|
+
prNumber: 1234,
|
|
73
|
+
repoPath: repo,
|
|
74
|
+
deps: { bun: 'bun', gh: FAKE_GH_WITH_LOCKFILE, codex: 'echo', git: 'git' },
|
|
75
|
+
})
|
|
76
|
+
expect(exit).toBe(0)
|
|
77
|
+
const contents = await readdir(runDir)
|
|
78
|
+
expect(contents).toContain('diff.patch')
|
|
79
|
+
expect(contents).toContain('diff.full.patch')
|
|
80
|
+
expect(contents).toContain('excluded-files.json')
|
|
81
|
+
const filtered = await readFile(join(runDir, 'diff.patch'), 'utf8')
|
|
82
|
+
expect(filtered).toContain('src/a.ts')
|
|
83
|
+
expect(filtered).not.toContain('bun.lock')
|
|
84
|
+
expect(filtered).not.toContain('dist/x.js')
|
|
85
|
+
const excluded = JSON.parse(
|
|
86
|
+
await readFile(join(runDir, 'excluded-files.json'), 'utf8'),
|
|
87
|
+
) as Array<{
|
|
88
|
+
path: string
|
|
89
|
+
pattern: string
|
|
90
|
+
}>
|
|
91
|
+
expect(excluded.map((e) => e.path).sort()).toEqual(['bun.lock', 'dist/x.js'])
|
|
92
|
+
})
|
|
93
|
+
|
|
94
|
+
test('runSetup with no excludable files does not write filter sidecars', async () => {
|
|
95
|
+
const exit = await runSetup({
|
|
96
|
+
runDir,
|
|
97
|
+
prNumber: 1234,
|
|
98
|
+
repoPath: repo,
|
|
99
|
+
deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
|
|
100
|
+
})
|
|
101
|
+
expect(exit).toBe(0)
|
|
102
|
+
const contents = await readdir(runDir)
|
|
103
|
+
expect(contents).not.toContain('diff.full.patch')
|
|
104
|
+
expect(contents).not.toContain('excluded-files.json')
|
|
105
|
+
})
|
|
106
|
+
|
|
107
|
+
test('runSetup honors .magpie.json useDefaults=false', async () => {
|
|
108
|
+
await writeFile(join(repo, '.magpie.json'), JSON.stringify({ useDefaults: false, exclude: [] }))
|
|
109
|
+
const exit = await runSetup({
|
|
110
|
+
runDir,
|
|
111
|
+
prNumber: 1234,
|
|
112
|
+
repoPath: repo,
|
|
113
|
+
deps: { bun: 'bun', gh: FAKE_GH_WITH_LOCKFILE, codex: 'echo', git: 'git' },
|
|
114
|
+
})
|
|
115
|
+
expect(exit).toBe(0)
|
|
116
|
+
const filtered = await readFile(join(runDir, 'diff.patch'), 'utf8')
|
|
117
|
+
expect(filtered).toContain('bun.lock')
|
|
118
|
+
expect(filtered).toContain('dist/x.js')
|
|
119
|
+
})
|
|
120
|
+
|
|
121
|
+
test('runSetup writes incremental.json when a prior run for the same PR exists', async () => {
|
|
122
|
+
const magpieHome = await mkdtemp(join(tmpdir(), 'magpie-home-incr-'))
|
|
123
|
+
await sh(magpieHome, 'mkdir', '-p', 'pr-1234-100')
|
|
124
|
+
await writeFile(
|
|
125
|
+
join(magpieHome, 'pr-1234-100', 'pr.json'),
|
|
126
|
+
JSON.stringify({ headRefOid: 'oldsha', number: 1234 }),
|
|
127
|
+
)
|
|
128
|
+
const originalHome = process.env.MAGPIE_HOME
|
|
129
|
+
process.env.MAGPIE_HOME = magpieHome
|
|
130
|
+
try {
|
|
131
|
+
const exit = await runSetup({
|
|
132
|
+
runDir,
|
|
133
|
+
prNumber: 1234,
|
|
134
|
+
repoPath: repo,
|
|
135
|
+
deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
|
|
136
|
+
})
|
|
137
|
+
expect(exit).toBe(0)
|
|
138
|
+
const incremental = JSON.parse(await readFile(join(runDir, 'incremental.json'), 'utf8'))
|
|
139
|
+
expect(incremental.previousRunId).toBe('pr-1234-100')
|
|
140
|
+
expect(incremental.previousSha).toBe('oldsha')
|
|
141
|
+
expect(incremental.sameSha).toBe(false)
|
|
142
|
+
} finally {
|
|
143
|
+
if (originalHome === undefined) delete process.env.MAGPIE_HOME
|
|
144
|
+
else process.env.MAGPIE_HOME = originalHome
|
|
145
|
+
await rm(magpieHome, { recursive: true, force: true })
|
|
146
|
+
}
|
|
147
|
+
})
|
|
148
|
+
|
|
149
|
+
test('runSetup omits incremental.json when no prior run exists', async () => {
|
|
150
|
+
const magpieHome = await mkdtemp(join(tmpdir(), 'magpie-home-incr-empty-'))
|
|
151
|
+
const originalHome = process.env.MAGPIE_HOME
|
|
152
|
+
process.env.MAGPIE_HOME = magpieHome
|
|
153
|
+
try {
|
|
154
|
+
const exit = await runSetup({
|
|
155
|
+
runDir,
|
|
156
|
+
prNumber: 1234,
|
|
157
|
+
repoPath: repo,
|
|
158
|
+
deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
|
|
159
|
+
})
|
|
160
|
+
expect(exit).toBe(0)
|
|
161
|
+
const contents = await readdir(runDir)
|
|
162
|
+
expect(contents).not.toContain('incremental.json')
|
|
163
|
+
} finally {
|
|
164
|
+
if (originalHome === undefined) delete process.env.MAGPIE_HOME
|
|
165
|
+
else process.env.MAGPIE_HOME = originalHome
|
|
166
|
+
await rm(magpieHome, { recursive: true, force: true })
|
|
167
|
+
}
|
|
168
|
+
})
|
|
169
|
+
|
|
67
170
|
test('runSetup with failing gh removes any partial state', async () => {
|
|
68
171
|
const exit = await runSetup({
|
|
69
172
|
runDir,
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
import { expect, test } from 'bun:test'
|
|
2
|
+
import { detectMissingTests, isTestFile } from '../tests-check.ts'
|
|
3
|
+
|
|
4
|
+
const sourceDiff = (path: string, added: number): string => {
|
|
5
|
+
const lines = [
|
|
6
|
+
`diff --git a/${path} b/${path}`,
|
|
7
|
+
'index 1..2 100644',
|
|
8
|
+
`--- a/${path}`,
|
|
9
|
+
`+++ b/${path}`,
|
|
10
|
+
`@@ -1,1 +1,${added + 1} @@`,
|
|
11
|
+
' const existing = 1',
|
|
12
|
+
]
|
|
13
|
+
for (let i = 0; i < added; i++) {
|
|
14
|
+
lines.push(`+export function newFn${i}() { return ${i} }`)
|
|
15
|
+
}
|
|
16
|
+
return `${lines.join('\n')}\n`
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const testDiff = (path: string): string =>
|
|
20
|
+
[
|
|
21
|
+
`diff --git a/${path} b/${path}`,
|
|
22
|
+
'index 1..2 100644',
|
|
23
|
+
`--- a/${path}`,
|
|
24
|
+
`+++ b/${path}`,
|
|
25
|
+
'@@ -1,1 +1,2 @@',
|
|
26
|
+
' import { x } from "./x"',
|
|
27
|
+
'+test("x", () => expect(x()).toBe(1))',
|
|
28
|
+
'',
|
|
29
|
+
].join('\n')
|
|
30
|
+
|
|
31
|
+
test('isTestFile: common patterns', () => {
|
|
32
|
+
expect(isTestFile('src/a.test.ts')).toBe(true)
|
|
33
|
+
expect(isTestFile('src/a.spec.ts')).toBe(true)
|
|
34
|
+
expect(isTestFile('src/__tests__/a.ts')).toBe(true)
|
|
35
|
+
expect(isTestFile('tests/foo.ts')).toBe(true)
|
|
36
|
+
expect(isTestFile('app/test_foo.py')).toBe(true)
|
|
37
|
+
expect(isTestFile('cmd/foo_test.go')).toBe(true)
|
|
38
|
+
expect(isTestFile('UserTest.java')).toBe(true)
|
|
39
|
+
expect(isTestFile('src/a.ts')).toBe(false)
|
|
40
|
+
expect(isTestFile('docs/a.md')).toBe(false)
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
test('detectMissingTests: flags source files when no test file in diff', () => {
|
|
44
|
+
const findings = detectMissingTests(sourceDiff('src/a.ts', 12))
|
|
45
|
+
expect(findings).toHaveLength(1)
|
|
46
|
+
expect(findings[0]?.file).toBe('src/a.ts')
|
|
47
|
+
expect(findings[0]?.domain).toBe('tests')
|
|
48
|
+
expect(findings[0]?.severity).toBe('medium')
|
|
49
|
+
expect(findings[0]?.line).toBeNull()
|
|
50
|
+
})
|
|
51
|
+
|
|
52
|
+
test('detectMissingTests: empty when a test file is present anywhere', () => {
|
|
53
|
+
const diff = sourceDiff('src/a.ts', 20) + testDiff('src/a.test.ts')
|
|
54
|
+
const findings = detectMissingTests(diff)
|
|
55
|
+
expect(findings).toHaveLength(0)
|
|
56
|
+
})
|
|
57
|
+
|
|
58
|
+
test('detectMissingTests: respects minAddedLines threshold', () => {
|
|
59
|
+
const findings = detectMissingTests(sourceDiff('src/a.ts', 5))
|
|
60
|
+
expect(findings).toHaveLength(0)
|
|
61
|
+
})
|
|
62
|
+
|
|
63
|
+
test('detectMissingTests: skips non-source files', () => {
|
|
64
|
+
const diff = sourceDiff('README.md', 30)
|
|
65
|
+
expect(detectMissingTests(diff)).toHaveLength(0)
|
|
66
|
+
})
|
|
67
|
+
|
|
68
|
+
test('detectMissingTests: emits one finding per source file', () => {
|
|
69
|
+
const diff = sourceDiff('src/a.ts', 12) + sourceDiff('src/b.ts', 15)
|
|
70
|
+
const findings = detectMissingTests(diff)
|
|
71
|
+
expect(findings.map((f) => f.file).sort()).toEqual(['src/a.ts', 'src/b.ts'])
|
|
72
|
+
})
|
|
73
|
+
|
|
74
|
+
test('detectMissingTests: empty diff yields nothing', () => {
|
|
75
|
+
expect(detectMissingTests('')).toEqual([])
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
test('detectMissingTests: ignores comment-only and import-only additions', () => {
|
|
79
|
+
const diff = [
|
|
80
|
+
'diff --git a/src/a.ts b/src/a.ts',
|
|
81
|
+
'index 1..2 100644',
|
|
82
|
+
'--- a/src/a.ts',
|
|
83
|
+
'+++ b/src/a.ts',
|
|
84
|
+
'@@ -1,1 +1,15 @@',
|
|
85
|
+
' const x = 1',
|
|
86
|
+
'+// just a comment',
|
|
87
|
+
'+import { foo } from "./foo"',
|
|
88
|
+
'+import { bar } from "./bar"',
|
|
89
|
+
'+// another comment',
|
|
90
|
+
'+# python-style comment',
|
|
91
|
+
'',
|
|
92
|
+
].join('\n')
|
|
93
|
+
expect(detectMissingTests(diff)).toHaveLength(0)
|
|
94
|
+
})
|
|
@@ -2,8 +2,15 @@ import { describe, expect, test } from 'bun:test'
|
|
|
2
2
|
import type { FocusId, PrFileEntry, ReviewFinding } from '../types.ts'
|
|
3
3
|
import { FOCUS_IDS, isSuggestion, parseFinding } from '../types.ts'
|
|
4
4
|
|
|
5
|
-
test('FOCUS_IDS contains the
|
|
6
|
-
expect(FOCUS_IDS).toEqual([
|
|
5
|
+
test('FOCUS_IDS contains the LLM focuses plus the deterministic tests domain', () => {
|
|
6
|
+
expect(FOCUS_IDS).toEqual([
|
|
7
|
+
'security',
|
|
8
|
+
'bugs',
|
|
9
|
+
'performance',
|
|
10
|
+
'code-smells',
|
|
11
|
+
'architecture',
|
|
12
|
+
'tests',
|
|
13
|
+
])
|
|
7
14
|
})
|
|
8
15
|
|
|
9
16
|
test('parseFinding accepts a minimal finding', () => {
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
import { appendFile, readdir, readFile, writeFile } from 'node:fs/promises'
|
|
2
2
|
import { join } from 'node:path'
|
|
3
3
|
import { deduplicateFindings } from './dedupe.ts'
|
|
4
|
+
import { verifyEvidence } from './evidence-filter.ts'
|
|
5
|
+
import { DEFAULT_THRESHOLD, scoreRisk } from './score.ts'
|
|
4
6
|
import { FOCUS_IDS, parseFinding, type ReviewFinding } from './types.ts'
|
|
5
7
|
|
|
6
8
|
async function logLine(runDir: string, entry: Record<string, unknown>): Promise<void> {
|
|
@@ -8,7 +10,13 @@ async function logLine(runDir: string, entry: Record<string, unknown>): Promise<
|
|
|
8
10
|
await appendFile(join(runDir, 'log.jsonl'), line)
|
|
9
11
|
}
|
|
10
12
|
|
|
11
|
-
export
|
|
13
|
+
export type RunDedupeOptions = {
|
|
14
|
+
/** 0-10 importance score below which findings are dropped pre-critic. */
|
|
15
|
+
threshold?: number
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
export async function runDedupe(runDir: string, options: RunDedupeOptions = {}): Promise<number> {
|
|
19
|
+
const threshold = options.threshold ?? DEFAULT_THRESHOLD
|
|
12
20
|
const findingsDir = join(runDir, 'findings')
|
|
13
21
|
const collected: ReviewFinding[] = []
|
|
14
22
|
let files: string[]
|
|
@@ -67,12 +75,41 @@ export async function runDedupe(runDir: string): Promise<number> {
|
|
|
67
75
|
}
|
|
68
76
|
|
|
69
77
|
const deduped = deduplicateFindings(collected)
|
|
70
|
-
|
|
78
|
+
const scored = deduped.map((f) => ({ ...f, score: scoreRisk(f.risk) }))
|
|
79
|
+
const evidence = await verifyEvidence(scored, join(runDir, 'worktree'))
|
|
80
|
+
const aboveThreshold = evidence.kept.filter((f) => (f.score ?? 0) >= threshold)
|
|
81
|
+
const belowThreshold = evidence.kept.filter((f) => (f.score ?? 0) < threshold)
|
|
82
|
+
await writeFile(
|
|
83
|
+
join(runDir, 'findings.deduped.json'),
|
|
84
|
+
`${JSON.stringify(aboveThreshold, null, 2)}\n`,
|
|
85
|
+
)
|
|
86
|
+
if (evidence.dropped.length > 0) {
|
|
87
|
+
await writeFile(
|
|
88
|
+
join(runDir, 'evidence-dropped.json'),
|
|
89
|
+
`${JSON.stringify(evidence.dropped, null, 2)}\n`,
|
|
90
|
+
)
|
|
91
|
+
}
|
|
92
|
+
if (belowThreshold.length > 0) {
|
|
93
|
+
await writeFile(
|
|
94
|
+
join(runDir, 'threshold-dropped.json'),
|
|
95
|
+
`${JSON.stringify(
|
|
96
|
+
belowThreshold.map((f) => ({ id: f.id, score: f.score, title: f.title })),
|
|
97
|
+
null,
|
|
98
|
+
2,
|
|
99
|
+
)}\n`,
|
|
100
|
+
)
|
|
101
|
+
}
|
|
71
102
|
await logLine(runDir, {
|
|
72
103
|
stage: 'dedupe',
|
|
73
104
|
status: 'done',
|
|
74
105
|
input: collected.length,
|
|
75
|
-
output:
|
|
106
|
+
output: aboveThreshold.length,
|
|
107
|
+
threshold,
|
|
108
|
+
threshold_dropped: belowThreshold.length,
|
|
109
|
+
evidence: {
|
|
110
|
+
skipped: evidence.skipped,
|
|
111
|
+
dropped: evidence.dropped.length,
|
|
112
|
+
},
|
|
76
113
|
})
|
|
77
114
|
return 0
|
|
78
115
|
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
import { existsSync, statSync } from 'node:fs'
|
|
2
|
+
import { readFile } from 'node:fs/promises'
|
|
3
|
+
import { join } from 'node:path'
|
|
4
|
+
import type { ReviewFinding } from './types.ts'
|
|
5
|
+
|
|
6
|
+
export type EvidenceDrop = {
|
|
7
|
+
id: string
|
|
8
|
+
reason: 'hallucinated-file' | 'invented-line'
|
|
9
|
+
file: string
|
|
10
|
+
line: number | null
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export type VerifyEvidenceResult = {
|
|
14
|
+
kept: ReviewFinding[]
|
|
15
|
+
dropped: EvidenceDrop[]
|
|
16
|
+
/** Set when the worktree itself was not available; verification was skipped. */
|
|
17
|
+
skipped: boolean
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
function isReadableFile(path: string): boolean {
|
|
21
|
+
if (!existsSync(path)) return false
|
|
22
|
+
try {
|
|
23
|
+
return statSync(path).isFile()
|
|
24
|
+
} catch {
|
|
25
|
+
return false
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
async function lineCount(path: string): Promise<number> {
|
|
30
|
+
const text = await readFile(path, 'utf8')
|
|
31
|
+
if (text.length === 0) return 0
|
|
32
|
+
let count = 1
|
|
33
|
+
for (let i = 0; i < text.length; i++) {
|
|
34
|
+
if (text.charCodeAt(i) === 10) count += 1
|
|
35
|
+
}
|
|
36
|
+
if (text.charCodeAt(text.length - 1) === 10) count -= 1
|
|
37
|
+
return count
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export async function verifyEvidence(
|
|
41
|
+
findings: ReviewFinding[],
|
|
42
|
+
worktreePath: string,
|
|
43
|
+
): Promise<VerifyEvidenceResult> {
|
|
44
|
+
if (!existsSync(worktreePath)) {
|
|
45
|
+
return { kept: findings, dropped: [], skipped: true }
|
|
46
|
+
}
|
|
47
|
+
const kept: ReviewFinding[] = []
|
|
48
|
+
const dropped: EvidenceDrop[] = []
|
|
49
|
+
const lineCounts = new Map<string, number>()
|
|
50
|
+
|
|
51
|
+
for (const f of findings) {
|
|
52
|
+
if (f.line === null || !f.file) {
|
|
53
|
+
kept.push(f)
|
|
54
|
+
continue
|
|
55
|
+
}
|
|
56
|
+
const abs = join(worktreePath, f.file)
|
|
57
|
+
if (!isReadableFile(abs)) {
|
|
58
|
+
dropped.push({ id: f.id, reason: 'hallucinated-file', file: f.file, line: f.line })
|
|
59
|
+
continue
|
|
60
|
+
}
|
|
61
|
+
let count = lineCounts.get(abs)
|
|
62
|
+
if (count === undefined) {
|
|
63
|
+
count = await lineCount(abs)
|
|
64
|
+
lineCounts.set(abs, count)
|
|
65
|
+
}
|
|
66
|
+
if (f.line < 1 || f.line > count) {
|
|
67
|
+
dropped.push({ id: f.id, reason: 'invented-line', file: f.file, line: f.line })
|
|
68
|
+
continue
|
|
69
|
+
}
|
|
70
|
+
kept.push(f)
|
|
71
|
+
}
|
|
72
|
+
return { kept, dropped, skipped: false }
|
|
73
|
+
}
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { readdir, readFile, stat } from 'node:fs/promises'
|
|
2
|
+
import { homedir } from 'node:os'
|
|
3
|
+
import { basename, join } from 'node:path'
|
|
4
|
+
|
|
5
|
+
export type IncrementalContext = {
|
|
6
|
+
previousRunId: string
|
|
7
|
+
previousSha: string
|
|
8
|
+
currentSha: string
|
|
9
|
+
/** True when the head SHA is unchanged from the previous run (no new commits). */
|
|
10
|
+
sameSha: boolean
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
const RUN_PATTERN = /^pr-(\d+)-(\d+)(?:\.archived-\d+)?$/
|
|
14
|
+
|
|
15
|
+
function reviewHome(): string {
|
|
16
|
+
return process.env.MAGPIE_HOME ?? join(homedir(), '.magpie')
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* Find the most recent prior run (active or archived) for the same PR number,
|
|
21
|
+
* excluding the current run. Returns null when no prior run exists. The current
|
|
22
|
+
* run is identified by basename match so we don't accidentally pick it up.
|
|
23
|
+
*/
|
|
24
|
+
export async function findPreviousRun(
|
|
25
|
+
prNumber: number,
|
|
26
|
+
currentRunDir: string,
|
|
27
|
+
home?: string,
|
|
28
|
+
): Promise<{ runId: string; runDir: string; prJson: Record<string, unknown> } | null> {
|
|
29
|
+
const root = home ?? reviewHome()
|
|
30
|
+
let entries: string[]
|
|
31
|
+
try {
|
|
32
|
+
entries = await readdir(root)
|
|
33
|
+
} catch {
|
|
34
|
+
return null
|
|
35
|
+
}
|
|
36
|
+
const currentId = basename(currentRunDir)
|
|
37
|
+
const candidates: Array<{ id: string; ts: number }> = []
|
|
38
|
+
for (const id of entries) {
|
|
39
|
+
if (id === currentId) continue
|
|
40
|
+
const m = id.match(RUN_PATTERN)
|
|
41
|
+
if (!m) continue
|
|
42
|
+
if (Number(m[1]) !== prNumber) continue
|
|
43
|
+
candidates.push({ id, ts: Number(m[2]) })
|
|
44
|
+
}
|
|
45
|
+
candidates.sort((a, b) => b.ts - a.ts)
|
|
46
|
+
for (const c of candidates) {
|
|
47
|
+
const runDir = join(root, c.id)
|
|
48
|
+
const s = await stat(runDir).catch(() => null)
|
|
49
|
+
if (!s?.isDirectory()) continue
|
|
50
|
+
const prJsonPath = join(runDir, 'pr.json')
|
|
51
|
+
try {
|
|
52
|
+
const text = await readFile(prJsonPath, 'utf8')
|
|
53
|
+
return { runId: c.id, runDir, prJson: JSON.parse(text) }
|
|
54
|
+
} catch {
|
|
55
|
+
// Missing or malformed pr.json; try the next candidate.
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
return null
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export function buildIncrementalContext(
|
|
62
|
+
current: Record<string, unknown>,
|
|
63
|
+
previous: { runId: string; prJson: Record<string, unknown> },
|
|
64
|
+
): IncrementalContext | null {
|
|
65
|
+
const currentSha = typeof current.headRefOid === 'string' ? current.headRefOid : ''
|
|
66
|
+
const previousSha =
|
|
67
|
+
typeof previous.prJson.headRefOid === 'string' ? previous.prJson.headRefOid : ''
|
|
68
|
+
if (!currentSha || !previousSha) return null
|
|
69
|
+
return {
|
|
70
|
+
previousRunId: previous.runId,
|
|
71
|
+
previousSha,
|
|
72
|
+
currentSha,
|
|
73
|
+
sameSha: previousSha === currentSha,
|
|
74
|
+
}
|
|
75
|
+
}
|