@iceinvein/agent-skills 0.1.28 → 0.1.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import { mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
3
3
  import { tmpdir } from 'node:os'
4
4
  import { join } from 'node:path'
5
5
  import {
6
+ computeEffortScore,
6
7
  formatConversationBody,
7
8
  formatInlineBody,
8
9
  formatPostBody,
@@ -158,8 +159,8 @@ test('runPost skips ids already marked as posted', async () => {
158
159
  await seedRunDir()
159
160
  await writeFile(join(runDir, 'post-status.json'), JSON.stringify({ 'sec-1': 'posted' }))
160
161
  const outcome = await runPost({ runDir, findingIds: ['sec-1', 'bug-1'], dryRun: true })
161
- expect(outcome.results[0]?.status).toBe('already-posted')
162
- expect(outcome.results[1]?.status).toBe('posted')
162
+ expect(outcome.results.find((r) => r.id === 'sec-1')?.status).toBe('already-posted')
163
+ expect(outcome.results.find((r) => r.id === 'bug-1')?.status).toBe('posted')
163
164
  })
164
165
 
165
166
  test('runPost surfaces gh failure per finding (here: gh binary missing)', async () => {
@@ -169,6 +170,7 @@ test('runPost surfaces gh failure per finding (here: gh binary missing)', async
169
170
  runDir,
170
171
  findingIds: ['sec-1'],
171
172
  ghBin: '/does/not/exist/gh-binary',
173
+ includeSummary: 'never',
172
174
  })
173
175
  expect(outcome.ok).toBe(true) // overall request succeeds; per-id reflects gh's failure
174
176
  expect(outcome.results[0]?.status).toBe('failed')
@@ -204,8 +206,9 @@ test('runPost falls back to a top-level PR comment when GitHub rejects the inlin
204
206
 
205
207
  const outcome = await runPost({ runDir, findingIds: ['sec-1'], ghBin: fakeGh })
206
208
  expect(outcome.ok).toBe(true)
207
- expect(outcome.results[0]?.status).toBe('posted')
208
- expect(outcome.results[0]?.message).toMatch(/posted as PR comment/i)
209
+ const sec1 = outcome.results.find((r) => r.id === 'sec-1')
210
+ expect(sec1?.status).toBe('posted')
211
+ expect(sec1?.message).toMatch(/posted as PR comment/i)
209
212
  const status = JSON.parse(await readFile(join(runDir, 'post-status.json'), 'utf8'))
210
213
  expect(status['sec-1']).toBe('posted')
211
214
  const log = await readFile(join(runDir, 'log.jsonl'), 'utf8')
@@ -226,8 +229,9 @@ test('runPost reports a combined failure when both inline and fallback gh calls
226
229
  )
227
230
  await Bun.spawn(['chmod', '+x', fakeGh]).exited
228
231
  const outcome = await runPost({ runDir, findingIds: ['sec-1'], ghBin: fakeGh })
229
- expect(outcome.results[0]?.status).toBe('failed')
230
- expect(outcome.results[0]?.message).toMatch(/inline.*fallback also failed/i)
232
+ const sec1 = outcome.results.find((r) => r.id === 'sec-1')
233
+ expect(sec1?.status).toBe('failed')
234
+ expect(sec1?.message).toMatch(/inline.*fallback also failed/i)
231
235
  })
232
236
 
233
237
  test('formatReviewSummaryBody renders verdict, Needs Attention and Risk breakdown', () => {
@@ -265,7 +269,82 @@ test('formatReviewSummaryBody handles the empty case', () => {
265
269
  expect(body).not.toContain('Risk breakdown')
266
270
  })
267
271
 
268
- test('runPost auto-posts the review summary when batch has 2+ pending findings', async () => {
272
+ test('formatReviewSummaryBody includes effort score when files are provided', () => {
273
+ const body = formatReviewSummaryBody([], {
274
+ files: [{ path: 'src/a.ts', additions: 5, deletions: 0 }],
275
+ })
276
+ expect(body).toContain('Review effort')
277
+ expect(body).toMatch(/\d\/5/)
278
+ })
279
+
280
+ test('formatReviewSummaryBody renders walkthrough table sorted by churn', () => {
281
+ const body = formatReviewSummaryBody([], {
282
+ files: [
283
+ { path: 'small.ts', additions: 1, deletions: 0 },
284
+ { path: 'big.ts', additions: 200, deletions: 100 },
285
+ ],
286
+ })
287
+ expect(body).toContain('Files changed')
288
+ expect(body).toContain('big.ts')
289
+ const bigIdx = body.indexOf('big.ts')
290
+ const smallIdx = body.indexOf('small.ts')
291
+ expect(bigIdx).toBeLessThan(smallIdx)
292
+ })
293
+
294
+ test('formatReviewSummaryBody surfaces risk hotspots when 2+ findings per file', () => {
295
+ const findings = [
296
+ {
297
+ ...findingA,
298
+ id: 'h1',
299
+ file: 'src/hot.ts',
300
+ line: 5,
301
+ },
302
+ {
303
+ ...findingA,
304
+ id: 'h2',
305
+ file: 'src/hot.ts',
306
+ line: 10,
307
+ },
308
+ {
309
+ ...findingA,
310
+ id: 'h3',
311
+ file: 'src/cool.ts',
312
+ line: 1,
313
+ },
314
+ ] as Parameters<typeof formatReviewSummaryBody>[0]
315
+ const body = formatReviewSummaryBody(findings)
316
+ expect(body).toContain('Risk hotspots')
317
+ expect(body).toContain('src/hot.ts')
318
+ expect(body).not.toContain('- `src/cool.ts`')
319
+ })
320
+
321
+ test('formatReviewSummaryBody renders incremental trailer with new commits', () => {
322
+ const body = formatReviewSummaryBody([], {
323
+ incremental: { previousSha: 'abcdef0123456', sameSha: false },
324
+ })
325
+ expect(body).toContain('Incremental review since')
326
+ expect(body).toContain('abcdef0')
327
+ })
328
+
329
+ test('formatReviewSummaryBody renders re-review trailer when sameSha', () => {
330
+ const body = formatReviewSummaryBody([], {
331
+ incremental: { previousSha: 'abcdef0123456', sameSha: true },
332
+ })
333
+ expect(body).toContain('Re-review')
334
+ })
335
+
336
+ test('computeEffortScore: small PR = 1, huge PR = 5', () => {
337
+ expect(computeEffortScore([])).toBe(1)
338
+ expect(computeEffortScore([{ path: 'a', additions: 5, deletions: 5 }])).toBe(1)
339
+ const huge = Array.from({ length: 25 }, (_, i) => ({
340
+ path: `f${i}.ts`,
341
+ additions: 100,
342
+ deletions: 0,
343
+ }))
344
+ expect(computeEffortScore(huge)).toBe(5)
345
+ })
346
+
347
+ test('runPost auto-posts the review summary on any non-empty batch', async () => {
269
348
  await seedRunDir()
270
349
  const outcome = await runPost({
271
350
  runDir,
@@ -280,16 +359,37 @@ test('runPost auto-posts the review summary when batch has 2+ pending findings',
280
359
  expect(status.__summary__).toBe('posted')
281
360
  })
282
361
 
283
- test('runPost skips the summary on single-finding batches in auto mode', async () => {
362
+ test('runPost auto-posts the summary even on single-finding batches', async () => {
284
363
  await seedRunDir()
285
364
  const outcome = await runPost({
286
365
  runDir,
287
366
  findingIds: ['sec-1'],
288
367
  dryRun: true,
289
368
  })
369
+ expect(outcome.results.find((r) => r.id === '__summary__')?.status).toBe('posted')
370
+ })
371
+
372
+ test('runPost auto-skips the summary when the batch has zero findings', async () => {
373
+ await seedRunDir()
374
+ const outcome = await runPost({
375
+ runDir,
376
+ findingIds: ['nope-id'],
377
+ dryRun: true,
378
+ })
290
379
  expect(outcome.results.find((r) => r.id === '__summary__')).toBeUndefined()
291
380
  })
292
381
 
382
+ test('runPost with includeSummary: always posts the summary even on empty batches', async () => {
383
+ await seedRunDir()
384
+ const outcome = await runPost({
385
+ runDir,
386
+ findingIds: [],
387
+ dryRun: true,
388
+ includeSummary: 'always',
389
+ })
390
+ expect(outcome.results.find((r) => r.id === '__summary__')?.status).toBe('posted')
391
+ })
392
+
293
393
  test('runPost forces the summary when includeSummary: always', async () => {
294
394
  await seedRunDir()
295
395
  const outcome = await runPost({
@@ -0,0 +1,51 @@
1
+ import { expect, test } from 'bun:test'
2
+ import { DEFAULT_THRESHOLD, scoreRisk } from '../score.ts'
3
+ import type { Risk } from '../types.ts'
4
+
5
+ const risk = (over: Partial<Risk> = {}): Risk => ({
6
+ impact: 'medium',
7
+ likelihood: 'possible',
8
+ confidence: 'medium',
9
+ action: 'should-fix',
10
+ ...over,
11
+ })
12
+
13
+ test('scoreRisk: maxed-out risk hits ~10', () => {
14
+ const s = scoreRisk({
15
+ impact: 'critical',
16
+ likelihood: 'likely',
17
+ confidence: 'high',
18
+ action: 'must-fix',
19
+ })
20
+ expect(s).toBeCloseTo(10, 1)
21
+ })
22
+
23
+ test('scoreRisk: floor risk is near 1', () => {
24
+ const s = scoreRisk({
25
+ impact: 'low',
26
+ likelihood: 'edge-case',
27
+ confidence: 'low',
28
+ action: 'optional',
29
+ })
30
+ expect(s).toBeGreaterThanOrEqual(1)
31
+ expect(s).toBeLessThan(2.5)
32
+ })
33
+
34
+ test('scoreRisk: impact dominates over action', () => {
35
+ const highImpact = scoreRisk(risk({ impact: 'critical', action: 'optional' }))
36
+ const lowImpact = scoreRisk(risk({ impact: 'low', action: 'must-fix' }))
37
+ expect(highImpact).toBeGreaterThan(lowImpact)
38
+ })
39
+
40
+ test('scoreRisk: confidence pulls weight', () => {
41
+ const highConf = scoreRisk(risk({ confidence: 'high' }))
42
+ const lowConf = scoreRisk(risk({ confidence: 'low' }))
43
+ expect(highConf - lowConf).toBeGreaterThan(0.5)
44
+ })
45
+
46
+ test('DEFAULT_THRESHOLD is sensible (drops floor, keeps medium)', () => {
47
+ expect(DEFAULT_THRESHOLD).toBeGreaterThan(0)
48
+ expect(DEFAULT_THRESHOLD).toBeLessThan(5)
49
+ const medium = scoreRisk(risk())
50
+ expect(medium).toBeGreaterThanOrEqual(DEFAULT_THRESHOLD)
51
+ })
@@ -126,7 +126,10 @@ test('POST /post wires through to runPost (dry-run) and returns results', async
126
126
  }
127
127
  expect(body.ok).toBe(true)
128
128
  expect(body.target).toEqual({ repo: 'o/r', number: 7 })
129
- expect(body.results[0]).toMatchObject({ id: 'sec-1', status: 'posted' })
129
+ expect(body.results.find((r) => r.id === 'sec-1')).toMatchObject({
130
+ id: 'sec-1',
131
+ status: 'posted',
132
+ })
130
133
  })
131
134
 
132
135
  test('writes server-info on start with url and port', async () => {
@@ -1,10 +1,12 @@
1
1
  import { afterEach, beforeEach, expect, test } from 'bun:test'
2
- import { mkdtemp, readdir, rm, writeFile } from 'node:fs/promises'
2
+ import { mkdtemp, readdir, readFile, rm, writeFile } from 'node:fs/promises'
3
3
  import { tmpdir } from 'node:os'
4
4
  import { join } from 'node:path'
5
5
  import { runSetup } from '../setup-cmd.ts'
6
6
 
7
7
  const FAKE_GH = new URL('../../fixtures/fake-gh.sh', import.meta.url).pathname
8
+ const FAKE_GH_WITH_LOCKFILE = new URL('../../fixtures/fake-gh-with-lockfile.sh', import.meta.url)
9
+ .pathname
8
10
  const FALSE_BIN = '/usr/bin/false'
9
11
 
10
12
  let repo: string
@@ -64,6 +66,107 @@ test('runSetup with missing dep returns non-zero and cleans up', async () => {
64
66
  expect(contents.filter((c) => c !== 'log.jsonl')).toHaveLength(0)
65
67
  })
66
68
 
69
+ test('runSetup filters excluded files by default and preserves raw diff', async () => {
70
+ const exit = await runSetup({
71
+ runDir,
72
+ prNumber: 1234,
73
+ repoPath: repo,
74
+ deps: { bun: 'bun', gh: FAKE_GH_WITH_LOCKFILE, codex: 'echo', git: 'git' },
75
+ })
76
+ expect(exit).toBe(0)
77
+ const contents = await readdir(runDir)
78
+ expect(contents).toContain('diff.patch')
79
+ expect(contents).toContain('diff.full.patch')
80
+ expect(contents).toContain('excluded-files.json')
81
+ const filtered = await readFile(join(runDir, 'diff.patch'), 'utf8')
82
+ expect(filtered).toContain('src/a.ts')
83
+ expect(filtered).not.toContain('bun.lock')
84
+ expect(filtered).not.toContain('dist/x.js')
85
+ const excluded = JSON.parse(
86
+ await readFile(join(runDir, 'excluded-files.json'), 'utf8'),
87
+ ) as Array<{
88
+ path: string
89
+ pattern: string
90
+ }>
91
+ expect(excluded.map((e) => e.path).sort()).toEqual(['bun.lock', 'dist/x.js'])
92
+ })
93
+
94
+ test('runSetup with no excludable files does not write filter sidecars', async () => {
95
+ const exit = await runSetup({
96
+ runDir,
97
+ prNumber: 1234,
98
+ repoPath: repo,
99
+ deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
100
+ })
101
+ expect(exit).toBe(0)
102
+ const contents = await readdir(runDir)
103
+ expect(contents).not.toContain('diff.full.patch')
104
+ expect(contents).not.toContain('excluded-files.json')
105
+ })
106
+
107
+ test('runSetup honors .magpie.json useDefaults=false', async () => {
108
+ await writeFile(join(repo, '.magpie.json'), JSON.stringify({ useDefaults: false, exclude: [] }))
109
+ const exit = await runSetup({
110
+ runDir,
111
+ prNumber: 1234,
112
+ repoPath: repo,
113
+ deps: { bun: 'bun', gh: FAKE_GH_WITH_LOCKFILE, codex: 'echo', git: 'git' },
114
+ })
115
+ expect(exit).toBe(0)
116
+ const filtered = await readFile(join(runDir, 'diff.patch'), 'utf8')
117
+ expect(filtered).toContain('bun.lock')
118
+ expect(filtered).toContain('dist/x.js')
119
+ })
120
+
121
+ test('runSetup writes incremental.json when a prior run for the same PR exists', async () => {
122
+ const magpieHome = await mkdtemp(join(tmpdir(), 'magpie-home-incr-'))
123
+ await sh(magpieHome, 'mkdir', '-p', 'pr-1234-100')
124
+ await writeFile(
125
+ join(magpieHome, 'pr-1234-100', 'pr.json'),
126
+ JSON.stringify({ headRefOid: 'oldsha', number: 1234 }),
127
+ )
128
+ const originalHome = process.env.MAGPIE_HOME
129
+ process.env.MAGPIE_HOME = magpieHome
130
+ try {
131
+ const exit = await runSetup({
132
+ runDir,
133
+ prNumber: 1234,
134
+ repoPath: repo,
135
+ deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
136
+ })
137
+ expect(exit).toBe(0)
138
+ const incremental = JSON.parse(await readFile(join(runDir, 'incremental.json'), 'utf8'))
139
+ expect(incremental.previousRunId).toBe('pr-1234-100')
140
+ expect(incremental.previousSha).toBe('oldsha')
141
+ expect(incremental.sameSha).toBe(false)
142
+ } finally {
143
+ if (originalHome === undefined) delete process.env.MAGPIE_HOME
144
+ else process.env.MAGPIE_HOME = originalHome
145
+ await rm(magpieHome, { recursive: true, force: true })
146
+ }
147
+ })
148
+
149
+ test('runSetup omits incremental.json when no prior run exists', async () => {
150
+ const magpieHome = await mkdtemp(join(tmpdir(), 'magpie-home-incr-empty-'))
151
+ const originalHome = process.env.MAGPIE_HOME
152
+ process.env.MAGPIE_HOME = magpieHome
153
+ try {
154
+ const exit = await runSetup({
155
+ runDir,
156
+ prNumber: 1234,
157
+ repoPath: repo,
158
+ deps: { bun: 'bun', gh: FAKE_GH, codex: 'echo', git: 'git' },
159
+ })
160
+ expect(exit).toBe(0)
161
+ const contents = await readdir(runDir)
162
+ expect(contents).not.toContain('incremental.json')
163
+ } finally {
164
+ if (originalHome === undefined) delete process.env.MAGPIE_HOME
165
+ else process.env.MAGPIE_HOME = originalHome
166
+ await rm(magpieHome, { recursive: true, force: true })
167
+ }
168
+ })
169
+
67
170
  test('runSetup with failing gh removes any partial state', async () => {
68
171
  const exit = await runSetup({
69
172
  runDir,
@@ -0,0 +1,94 @@
1
+ import { expect, test } from 'bun:test'
2
+ import { detectMissingTests, isTestFile } from '../tests-check.ts'
3
+
4
+ const sourceDiff = (path: string, added: number): string => {
5
+ const lines = [
6
+ `diff --git a/${path} b/${path}`,
7
+ 'index 1..2 100644',
8
+ `--- a/${path}`,
9
+ `+++ b/${path}`,
10
+ `@@ -1,1 +1,${added + 1} @@`,
11
+ ' const existing = 1',
12
+ ]
13
+ for (let i = 0; i < added; i++) {
14
+ lines.push(`+export function newFn${i}() { return ${i} }`)
15
+ }
16
+ return `${lines.join('\n')}\n`
17
+ }
18
+
19
+ const testDiff = (path: string): string =>
20
+ [
21
+ `diff --git a/${path} b/${path}`,
22
+ 'index 1..2 100644',
23
+ `--- a/${path}`,
24
+ `+++ b/${path}`,
25
+ '@@ -1,1 +1,2 @@',
26
+ ' import { x } from "./x"',
27
+ '+test("x", () => expect(x()).toBe(1))',
28
+ '',
29
+ ].join('\n')
30
+
31
+ test('isTestFile: common patterns', () => {
32
+ expect(isTestFile('src/a.test.ts')).toBe(true)
33
+ expect(isTestFile('src/a.spec.ts')).toBe(true)
34
+ expect(isTestFile('src/__tests__/a.ts')).toBe(true)
35
+ expect(isTestFile('tests/foo.ts')).toBe(true)
36
+ expect(isTestFile('app/test_foo.py')).toBe(true)
37
+ expect(isTestFile('cmd/foo_test.go')).toBe(true)
38
+ expect(isTestFile('UserTest.java')).toBe(true)
39
+ expect(isTestFile('src/a.ts')).toBe(false)
40
+ expect(isTestFile('docs/a.md')).toBe(false)
41
+ })
42
+
43
+ test('detectMissingTests: flags source files when no test file in diff', () => {
44
+ const findings = detectMissingTests(sourceDiff('src/a.ts', 12))
45
+ expect(findings).toHaveLength(1)
46
+ expect(findings[0]?.file).toBe('src/a.ts')
47
+ expect(findings[0]?.domain).toBe('tests')
48
+ expect(findings[0]?.severity).toBe('medium')
49
+ expect(findings[0]?.line).toBeNull()
50
+ })
51
+
52
+ test('detectMissingTests: empty when a test file is present anywhere', () => {
53
+ const diff = sourceDiff('src/a.ts', 20) + testDiff('src/a.test.ts')
54
+ const findings = detectMissingTests(diff)
55
+ expect(findings).toHaveLength(0)
56
+ })
57
+
58
+ test('detectMissingTests: respects minAddedLines threshold', () => {
59
+ const findings = detectMissingTests(sourceDiff('src/a.ts', 5))
60
+ expect(findings).toHaveLength(0)
61
+ })
62
+
63
+ test('detectMissingTests: skips non-source files', () => {
64
+ const diff = sourceDiff('README.md', 30)
65
+ expect(detectMissingTests(diff)).toHaveLength(0)
66
+ })
67
+
68
+ test('detectMissingTests: emits one finding per source file', () => {
69
+ const diff = sourceDiff('src/a.ts', 12) + sourceDiff('src/b.ts', 15)
70
+ const findings = detectMissingTests(diff)
71
+ expect(findings.map((f) => f.file).sort()).toEqual(['src/a.ts', 'src/b.ts'])
72
+ })
73
+
74
+ test('detectMissingTests: empty diff yields nothing', () => {
75
+ expect(detectMissingTests('')).toEqual([])
76
+ })
77
+
78
+ test('detectMissingTests: ignores comment-only and import-only additions', () => {
79
+ const diff = [
80
+ 'diff --git a/src/a.ts b/src/a.ts',
81
+ 'index 1..2 100644',
82
+ '--- a/src/a.ts',
83
+ '+++ b/src/a.ts',
84
+ '@@ -1,1 +1,15 @@',
85
+ ' const x = 1',
86
+ '+// just a comment',
87
+ '+import { foo } from "./foo"',
88
+ '+import { bar } from "./bar"',
89
+ '+// another comment',
90
+ '+# python-style comment',
91
+ '',
92
+ ].join('\n')
93
+ expect(detectMissingTests(diff)).toHaveLength(0)
94
+ })
@@ -2,8 +2,15 @@ import { describe, expect, test } from 'bun:test'
2
2
  import type { FocusId, PrFileEntry, ReviewFinding } from '../types.ts'
3
3
  import { FOCUS_IDS, isSuggestion, parseFinding } from '../types.ts'
4
4
 
5
- test('FOCUS_IDS contains the five default focuses', () => {
6
- expect(FOCUS_IDS).toEqual(['security', 'bugs', 'performance', 'code-smells', 'architecture'])
5
+ test('FOCUS_IDS contains the LLM focuses plus the deterministic tests domain', () => {
6
+ expect(FOCUS_IDS).toEqual([
7
+ 'security',
8
+ 'bugs',
9
+ 'performance',
10
+ 'code-smells',
11
+ 'architecture',
12
+ 'tests',
13
+ ])
7
14
  })
8
15
 
9
16
  test('parseFinding accepts a minimal finding', () => {
@@ -1,6 +1,8 @@
1
1
  import { appendFile, readdir, readFile, writeFile } from 'node:fs/promises'
2
2
  import { join } from 'node:path'
3
3
  import { deduplicateFindings } from './dedupe.ts'
4
+ import { verifyEvidence } from './evidence-filter.ts'
5
+ import { DEFAULT_THRESHOLD, scoreRisk } from './score.ts'
4
6
  import { FOCUS_IDS, parseFinding, type ReviewFinding } from './types.ts'
5
7
 
6
8
  async function logLine(runDir: string, entry: Record<string, unknown>): Promise<void> {
@@ -8,7 +10,13 @@ async function logLine(runDir: string, entry: Record<string, unknown>): Promise<
8
10
  await appendFile(join(runDir, 'log.jsonl'), line)
9
11
  }
10
12
 
11
- export async function runDedupe(runDir: string): Promise<number> {
13
+ export type RunDedupeOptions = {
14
+ /** 0-10 importance score below which findings are dropped pre-critic. */
15
+ threshold?: number
16
+ }
17
+
18
+ export async function runDedupe(runDir: string, options: RunDedupeOptions = {}): Promise<number> {
19
+ const threshold = options.threshold ?? DEFAULT_THRESHOLD
12
20
  const findingsDir = join(runDir, 'findings')
13
21
  const collected: ReviewFinding[] = []
14
22
  let files: string[]
@@ -67,12 +75,41 @@ export async function runDedupe(runDir: string): Promise<number> {
67
75
  }
68
76
 
69
77
  const deduped = deduplicateFindings(collected)
70
- await writeFile(join(runDir, 'findings.deduped.json'), `${JSON.stringify(deduped, null, 2)}\n`)
78
+ const scored = deduped.map((f) => ({ ...f, score: scoreRisk(f.risk) }))
79
+ const evidence = await verifyEvidence(scored, join(runDir, 'worktree'))
80
+ const aboveThreshold = evidence.kept.filter((f) => (f.score ?? 0) >= threshold)
81
+ const belowThreshold = evidence.kept.filter((f) => (f.score ?? 0) < threshold)
82
+ await writeFile(
83
+ join(runDir, 'findings.deduped.json'),
84
+ `${JSON.stringify(aboveThreshold, null, 2)}\n`,
85
+ )
86
+ if (evidence.dropped.length > 0) {
87
+ await writeFile(
88
+ join(runDir, 'evidence-dropped.json'),
89
+ `${JSON.stringify(evidence.dropped, null, 2)}\n`,
90
+ )
91
+ }
92
+ if (belowThreshold.length > 0) {
93
+ await writeFile(
94
+ join(runDir, 'threshold-dropped.json'),
95
+ `${JSON.stringify(
96
+ belowThreshold.map((f) => ({ id: f.id, score: f.score, title: f.title })),
97
+ null,
98
+ 2,
99
+ )}\n`,
100
+ )
101
+ }
71
102
  await logLine(runDir, {
72
103
  stage: 'dedupe',
73
104
  status: 'done',
74
105
  input: collected.length,
75
- output: deduped.length,
106
+ output: aboveThreshold.length,
107
+ threshold,
108
+ threshold_dropped: belowThreshold.length,
109
+ evidence: {
110
+ skipped: evidence.skipped,
111
+ dropped: evidence.dropped.length,
112
+ },
76
113
  })
77
114
  return 0
78
115
  }
@@ -0,0 +1,73 @@
1
+ import { existsSync, statSync } from 'node:fs'
2
+ import { readFile } from 'node:fs/promises'
3
+ import { join } from 'node:path'
4
+ import type { ReviewFinding } from './types.ts'
5
+
6
+ export type EvidenceDrop = {
7
+ id: string
8
+ reason: 'hallucinated-file' | 'invented-line'
9
+ file: string
10
+ line: number | null
11
+ }
12
+
13
+ export type VerifyEvidenceResult = {
14
+ kept: ReviewFinding[]
15
+ dropped: EvidenceDrop[]
16
+ /** Set when the worktree itself was not available; verification was skipped. */
17
+ skipped: boolean
18
+ }
19
+
20
+ function isReadableFile(path: string): boolean {
21
+ if (!existsSync(path)) return false
22
+ try {
23
+ return statSync(path).isFile()
24
+ } catch {
25
+ return false
26
+ }
27
+ }
28
+
29
+ async function lineCount(path: string): Promise<number> {
30
+ const text = await readFile(path, 'utf8')
31
+ if (text.length === 0) return 0
32
+ let count = 1
33
+ for (let i = 0; i < text.length; i++) {
34
+ if (text.charCodeAt(i) === 10) count += 1
35
+ }
36
+ if (text.charCodeAt(text.length - 1) === 10) count -= 1
37
+ return count
38
+ }
39
+
40
+ export async function verifyEvidence(
41
+ findings: ReviewFinding[],
42
+ worktreePath: string,
43
+ ): Promise<VerifyEvidenceResult> {
44
+ if (!existsSync(worktreePath)) {
45
+ return { kept: findings, dropped: [], skipped: true }
46
+ }
47
+ const kept: ReviewFinding[] = []
48
+ const dropped: EvidenceDrop[] = []
49
+ const lineCounts = new Map<string, number>()
50
+
51
+ for (const f of findings) {
52
+ if (f.line === null || !f.file) {
53
+ kept.push(f)
54
+ continue
55
+ }
56
+ const abs = join(worktreePath, f.file)
57
+ if (!isReadableFile(abs)) {
58
+ dropped.push({ id: f.id, reason: 'hallucinated-file', file: f.file, line: f.line })
59
+ continue
60
+ }
61
+ let count = lineCounts.get(abs)
62
+ if (count === undefined) {
63
+ count = await lineCount(abs)
64
+ lineCounts.set(abs, count)
65
+ }
66
+ if (f.line < 1 || f.line > count) {
67
+ dropped.push({ id: f.id, reason: 'invented-line', file: f.file, line: f.line })
68
+ continue
69
+ }
70
+ kept.push(f)
71
+ }
72
+ return { kept, dropped, skipped: false }
73
+ }
@@ -0,0 +1,75 @@
1
+ import { readdir, readFile, stat } from 'node:fs/promises'
2
+ import { homedir } from 'node:os'
3
+ import { basename, join } from 'node:path'
4
+
5
+ export type IncrementalContext = {
6
+ previousRunId: string
7
+ previousSha: string
8
+ currentSha: string
9
+ /** True when the head SHA is unchanged from the previous run (no new commits). */
10
+ sameSha: boolean
11
+ }
12
+
13
+ const RUN_PATTERN = /^pr-(\d+)-(\d+)(?:\.archived-\d+)?$/
14
+
15
+ function reviewHome(): string {
16
+ return process.env.MAGPIE_HOME ?? join(homedir(), '.magpie')
17
+ }
18
+
19
+ /**
20
+ * Find the most recent prior run (active or archived) for the same PR number,
21
+ * excluding the current run. Returns null when no prior run exists. The current
22
+ * run is identified by basename match so we don't accidentally pick it up.
23
+ */
24
+ export async function findPreviousRun(
25
+ prNumber: number,
26
+ currentRunDir: string,
27
+ home?: string,
28
+ ): Promise<{ runId: string; runDir: string; prJson: Record<string, unknown> } | null> {
29
+ const root = home ?? reviewHome()
30
+ let entries: string[]
31
+ try {
32
+ entries = await readdir(root)
33
+ } catch {
34
+ return null
35
+ }
36
+ const currentId = basename(currentRunDir)
37
+ const candidates: Array<{ id: string; ts: number }> = []
38
+ for (const id of entries) {
39
+ if (id === currentId) continue
40
+ const m = id.match(RUN_PATTERN)
41
+ if (!m) continue
42
+ if (Number(m[1]) !== prNumber) continue
43
+ candidates.push({ id, ts: Number(m[2]) })
44
+ }
45
+ candidates.sort((a, b) => b.ts - a.ts)
46
+ for (const c of candidates) {
47
+ const runDir = join(root, c.id)
48
+ const s = await stat(runDir).catch(() => null)
49
+ if (!s?.isDirectory()) continue
50
+ const prJsonPath = join(runDir, 'pr.json')
51
+ try {
52
+ const text = await readFile(prJsonPath, 'utf8')
53
+ return { runId: c.id, runDir, prJson: JSON.parse(text) }
54
+ } catch {
55
+ // Missing or malformed pr.json; try the next candidate.
56
+ }
57
+ }
58
+ return null
59
+ }
60
+
61
+ export function buildIncrementalContext(
62
+ current: Record<string, unknown>,
63
+ previous: { runId: string; prJson: Record<string, unknown> },
64
+ ): IncrementalContext | null {
65
+ const currentSha = typeof current.headRefOid === 'string' ? current.headRefOid : ''
66
+ const previousSha =
67
+ typeof previous.prJson.headRefOid === 'string' ? previous.prJson.headRefOid : ''
68
+ if (!currentSha || !previousSha) return null
69
+ return {
70
+ previousRunId: previous.runId,
71
+ previousSha,
72
+ currentSha,
73
+ sameSha: previousSha === currentSha,
74
+ }
75
+ }