@iceinvein/agent-skills 0.21.0 → 0.21.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/package.json +1 -1
  2. package/skills/index.json +2 -2
  3. package/skills/magpie/README.md +3 -4
  4. package/skills/magpie/SKILL.md +31 -29
  5. package/skills/magpie/bin/magpie.ts +84 -6
  6. package/skills/magpie/evals/README.md +19 -5
  7. package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +64 -23
  8. package/skills/magpie/evals/post-folds-selection-events/fixture.sh +9 -3
  9. package/skills/magpie/evals/quality/README.md +141 -0
  10. package/skills/magpie/evals/quality/__tests__/replay-critic.test.ts +31 -0
  11. package/skills/magpie/evals/quality/__tests__/score.test.ts +325 -0
  12. package/skills/magpie/evals/quality/baseline.ts +80 -0
  13. package/skills/magpie/evals/quality/corpus.ts +259 -0
  14. package/skills/magpie/evals/quality/replay-critic.ts +277 -0
  15. package/skills/magpie/evals/quality/replay-full.ts +188 -0
  16. package/skills/magpie/evals/quality/score.ts +251 -0
  17. package/skills/magpie/evals/report-ends-the-turn/fixture.sh +6 -2
  18. package/skills/magpie/evals/resume-finds-active-run/fixture.sh +1 -1
  19. package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +1 -1
  20. package/skills/magpie/fixtures/example-pr/brief.json +0 -4
  21. package/skills/magpie/references/critic.md +178 -40
  22. package/skills/magpie/references/peer-review.md +5 -4
  23. package/skills/magpie/references/scout.md +17 -42
  24. package/skills/magpie/references/specialists.md +43 -136
  25. package/skills/magpie/scripts/__tests__/cleanup-cmd.test.ts +28 -0
  26. package/skills/magpie/scripts/__tests__/critic-cmd.test.ts +260 -0
  27. package/skills/magpie/scripts/__tests__/critic.test.ts +328 -0
  28. package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +89 -0
  29. package/skills/magpie/scripts/__tests__/dedupe.test.ts +43 -1
  30. package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +155 -2
  31. package/skills/magpie/scripts/__tests__/helper.test.ts +80 -0
  32. package/skills/magpie/scripts/__tests__/labels.test.ts +152 -0
  33. package/skills/magpie/scripts/__tests__/open-cmd.test.ts +28 -0
  34. package/skills/magpie/scripts/__tests__/pipeline-e2e.test.ts +1 -0
  35. package/skills/magpie/scripts/__tests__/post-cmd.test.ts +47 -0
  36. package/skills/magpie/scripts/__tests__/refresh.test.ts +23 -1
  37. package/skills/magpie/scripts/__tests__/render-action-bar.test.ts +63 -5
  38. package/skills/magpie/scripts/__tests__/render-annotation.test.ts +38 -0
  39. package/skills/magpie/scripts/__tests__/render-cmd.test.ts +48 -1
  40. package/skills/magpie/scripts/__tests__/render-diff.test.ts +34 -0
  41. package/skills/magpie/scripts/__tests__/render-findings.test.ts +111 -30
  42. package/skills/magpie/scripts/__tests__/render-issues-list.test.ts +186 -1
  43. package/skills/magpie/scripts/__tests__/render-progress.test.ts +1 -1
  44. package/skills/magpie/scripts/__tests__/serve-cmd.test.ts +19 -0
  45. package/skills/magpie/scripts/__tests__/server.test.ts +76 -0
  46. package/skills/magpie/scripts/__tests__/skill-lint.test.ts +166 -179
  47. package/skills/magpie/scripts/__tests__/tests-check.test.ts +57 -0
  48. package/skills/magpie/scripts/__tests__/types.test.ts +90 -36
  49. package/skills/magpie/scripts/cleanup-cmd.ts +6 -0
  50. package/skills/magpie/scripts/critic-cmd.ts +183 -0
  51. package/skills/magpie/scripts/critic.ts +273 -0
  52. package/skills/magpie/scripts/dedupe-cmd.ts +14 -4
  53. package/skills/magpie/scripts/dedupe.ts +41 -0
  54. package/skills/magpie/scripts/evidence-filter.ts +81 -17
  55. package/skills/magpie/scripts/helper.js +87 -9
  56. package/skills/magpie/scripts/labels-cmd.ts +43 -0
  57. package/skills/magpie/scripts/labels.ts +99 -0
  58. package/skills/magpie/scripts/open-cmd.ts +10 -3
  59. package/skills/magpie/scripts/post-cmd.ts +26 -2
  60. package/skills/magpie/scripts/preview-cmd.ts +4 -0
  61. package/skills/magpie/scripts/refresh.ts +21 -17
  62. package/skills/magpie/scripts/render-action-bar.ts +21 -4
  63. package/skills/magpie/scripts/render-annotation.ts +36 -2
  64. package/skills/magpie/scripts/render-cmd.ts +21 -2
  65. package/skills/magpie/scripts/render-diff.ts +6 -2
  66. package/skills/magpie/scripts/render-findings.ts +21 -10
  67. package/skills/magpie/scripts/render-issues-list.ts +66 -16
  68. package/skills/magpie/scripts/render-progress.ts +1 -1
  69. package/skills/magpie/scripts/server.ts +8 -1
  70. package/skills/magpie/scripts/tests-check.ts +23 -2
  71. package/skills/magpie/scripts/types.ts +36 -46
  72. package/skills/magpie/skill.json +2 -2
  73. package/skills/magpie/templates/styles.css +83 -2
  74. package/skills/magpie/tsconfig.json +1 -1
  75. package/skills/magpie/evals/consent-required-never-approves/case.yaml +0 -4
  76. package/skills/magpie/evals/consent-required-never-approves/fixture.sh +0 -213
  77. package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +0 -6
  78. package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +0 -7
  79. package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +0 -5
  80. package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +0 -5
  81. package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +0 -10
  82. package/skills/magpie/evals/consent-required-never-approves/prompt.md +0 -11
@@ -0,0 +1,141 @@
1
+ # Quality eval
2
+
3
+ Measures how well magpie picks findings, using what reviewers actually did with past runs
4
+ as labels: a final finding is `posted`, `dismissed` (with a reason) or `ignored`. The
5
+ other evals in `evals/` check that the skill follows its walkthrough; this one checks
6
+ whether the findings it keeps are the ones a reviewer wanted.
7
+
8
+ Run every command from `skills/magpie`.
9
+
10
+ ## The corpus
11
+
12
+ ```
13
+ bun evals/quality/corpus.ts
14
+ ```
15
+
16
+ Copies every run under `~/.magpie` (or `$MAGPIE_HOME`; active and `.archived-` alike,
17
+ `preview-*` skipped)
18
+ that has both `post-status.json` and `findings.final.json` into
19
+ `~/.magpie/corpus/<runId>/`: `pr.json`, `diff.patch`, `findings.deduped.json`,
20
+ `findings.kept.json`, `findings.final.json`, `merge-candidates.json` and `brief.json`,
21
+ each when present. It then writes `labels.json` by folding `post-status.json`,
22
+ `state/events` and `log.jsonl`, and prints one line of counts per run. Rebuilding replaces
23
+ each run's corpus directory.
24
+
25
+ The corpus holds client code, so it never lives in the repo: the script exits 1 when the
26
+ corpus dir resolves inside the git repository that holds it. `--out <dir>` points it
27
+ elsewhere, for tests and dry runs; every script below takes the same flag.
28
+
29
+ Older runs predate merge candidates, briefs and the dismiss UI. A missing optional file is
30
+ read as absent; a file that is present but does not parse fails the build, naming it.
31
+
32
+ ## Metrics
33
+
34
+ `score.ts` exports `scoreSelection`, which scores a set of kept ids against the labels:
35
+
36
+ - `kept`: findings in the selection.
37
+ - `posted`: posted labels on findings in the candidate pool the selection was made from
38
+ (`findings.deduped.json` for the critic). Peer-review additions reach
39
+ `findings.final.json` without passing the critic, so they are not held against it.
40
+ - `postedKept`: kept findings that were posted.
41
+ - `precision`: `postedKept` over kept findings that have a label. Every id in
42
+ `findings.final.json` has one; a kept id that peer review removed before the report is
43
+ labelled `ignored`, since the reviewer never saw it. A kept id with no label at all (a
44
+ replay keeping a candidate the original critic dropped) is left out of the denominator.
45
+ Null when nothing labelled was kept.
46
+ - `recall`: `postedKept` over `posted`. Null when nothing was posted.
47
+ - `dismissedKept`: kept findings dismissed, by reason (`wrong`, `not-worth-it`,
48
+ `duplicate`, `style`).
49
+ - `byDomain`: `kept` and `postedKept` per specialist domain (`unassigned` when a finding
50
+ has none).
51
+ - `byVia`: posted kept findings by post route (`recommended`, `selected`, `one`, `cli`;
52
+ `unrecorded` for runs that logged no route).
53
+
54
+ `matchFindings` pairs a replay's freshly generated findings with labelled ones: same file,
55
+ line within 5, title Dice of at least 0.4 on `tokenize(title)`. Pairs are taken best Dice
56
+ first, each labelled finding matched at most once.
57
+
58
+ ## Confounds
59
+
60
+ Read every number here with these in mind.
61
+
62
+ - **Post Recommended.** The report's "Post recommended" button posts the top findings in
63
+ one click. A finding posted that way was not individually judged, so `posted` partly
64
+ measures rank, not worth. `byVia` separates those posts out where the run logged a
65
+ route.
66
+ - **Ignored is not rejected.** Most runs predate the dismiss UI, so a finding the reviewer
67
+ disliked usually reads as `ignored`, not `dismissed`. Precision treats `ignored` as a
68
+ miss, which is harsh on any finding the reviewer simply did not get to.
69
+ - **Repeated PRs.** Several runs review the same PR at different heads (three runs of
70
+ PR 50 today). Their findings overlap heavily, and a reviewer who posted a finding on one
71
+ run tends not to post it again on the next, which drags the later run's precision down.
72
+ Pooled totals weight those PRs once per run.
73
+ - **Baseline recall is 1 by construction.** The reviewer only ever saw kept findings, so
74
+ every posted finding in the pool was kept. Recall only means something for a replay,
75
+ where a different critic can drop a finding the reviewer posted.
76
+
77
+ ## Baseline
78
+
79
+ ```
80
+ bun evals/quality/baseline.ts
81
+ ```
82
+
83
+ Scores each run's original `findings.kept.json` against its labels, prints a table with a
84
+ pooled `TOTAL` row, and writes `~/.magpie/corpus/results/<ts>-baseline.json`. No model
85
+ calls; free to run.
86
+
87
+ ## Critic replay
88
+
89
+ ```
90
+ bun evals/quality/replay-critic.ts --corpus <runId> --repo <path-to-local-clone> [--design-cap <n>]
91
+ ```
92
+
93
+ Re-runs stage 6 alone on one corpus run:
94
+
95
+ 1. `git -C <repo> worktree add --detach <scratch>/worktree <headRefOid>`. The PR head must
96
+ already be in the clone (`git fetch origin pull/<n>/head`); a missing sha exits 1.
97
+ 2. Copies the corpus files into a scratch run dir under the system temp dir.
98
+ 3. `magpie critic-prompt`, then one `claude -p --allowedTools Read,Grep,Glob,Write
99
+ --output-format json --add-dir <scratch>` per batch, prompt on stdin, from the
100
+ worktree. Batches run in parallel. A non-zero exit, an `is_error` result or a missing
101
+ output file fails the replay.
102
+ 4. `magpie critic-apply` (with `--design-cap <n>` when given, so a cap can be compared
103
+ against the default of 3), then `scoreSelection` over the new `findings.kept.json`,
104
+ written with the run's baseline score to `results/<ts>-critic[-cap<n>]-<runId>.json`.
105
+ The result also keeps every critic verdict, `critic-dropped.json`, and
106
+ `postedDropped`: each posted candidate the replay did not keep, with the critic's
107
+ reason, `design-cap`, or its merge target. Those lines are printed too, so a recall
108
+ loss can be pinned on the critic or the cap. `unlabelledKept` counts kept findings the
109
+ original critic had dropped: nobody ever judged them, so precision leaves them out
110
+ and the count says how much of the selection precision does not cover.
111
+ 5. Removes the worktree and scratch dir whatever happened.
112
+
113
+ **Cost:** one Claude session per batch of up to 30 candidates; the corpus runs have 17 to
114
+ 84 deduped candidates, so 1 to 3 sessions each. Each session reads the candidates' code in
115
+ the worktree, so cost grows with candidate count. The results file records the
116
+ `total_cost_usd` Claude reports, summed over batches. There is no spend cap on this tier.
117
+
118
+ ## Full replay
119
+
120
+ ```
121
+ bun evals/quality/replay-full.ts --corpus <runId> --repo <path-to-local-clone> [--max-cost-usd N]
122
+ ```
123
+
124
+ Re-runs stages 3 to 6 (context, specialists, dedupe, critic) on the run's `diff.patch`:
125
+
126
+ 1. Scaffolds a scratch run dir: worktree at the PR head as above, `pr.json` and
127
+ `diff.patch` copied, the deterministic tests finding and a `setup` done line written in
128
+ place of `magpie setup` (which needs the live PR), then `magpie shard`.
129
+ 2. Runs one `claude -p --allowedTools Bash,Read,Grep,Glob,Write,Edit,Agent
130
+ --max-budget-usd <N> --output-format json` from the worktree, telling it to follow this
131
+ checkout's `SKILL.md`, resume from `magpie status`, and stop once `findings.kept.json`
132
+ is written. Serve, peer review, report, post and cleanup are skipped.
133
+ 3. Matches the new kept findings to `findings.final.json` with `matchFindings` and scores
134
+ them; unmatched findings count as `unlabelled` and stay out of precision. Writes
135
+ `results/<ts>-full-<runId>.json`.
136
+
137
+ **Cost:** a whole review: a scout, five specialists per shard, and the critic batches.
138
+ `--max-cost-usd` (default 10) is passed to Claude as `--max-budget-usd`, the CLI's own
139
+ spend cap; `claude -p` has no `--max-cost-usd` flag. Hitting the cap stops the session,
140
+ and the replay then fails without scoring, on Claude's error result or on the missing
141
+ `findings.kept.json`.
@@ -0,0 +1,31 @@
1
+ import { expect, test } from 'bun:test'
2
+ import { claudeFailureMessage, worktreeRemoveCommand } from '../replay-critic.ts'
3
+
4
+ test('a failed claude run names its label and the result subtype from stdout', () => {
5
+ const stdout = JSON.stringify({ type: 'result', is_error: true, subtype: 'error_max_turns' })
6
+ expect(claudeFailureMessage('critic batch 2', 1, stdout, 'boom\n')).toBe(
7
+ 'critic batch 2: claude -p exit 1; is_error true, subtype error_max_turns; stderr: boom',
8
+ )
9
+ })
10
+
11
+ test('a failed claude run with non-JSON stdout carries the raw stdout', () => {
12
+ expect(claudeFailureMessage('full replay', 137, 'Killed\n', '')).toBe(
13
+ 'full replay: claude -p exit 137; stdout: Killed; stderr: ',
14
+ )
15
+ })
16
+
17
+ test('a worktree that was never added has no removal command', () => {
18
+ expect(worktreeRemoveCommand('/repo', '/scratch/worktree', false)).toBeNull()
19
+ })
20
+
21
+ test('an added worktree is removed with a forced git worktree remove', () => {
22
+ expect(worktreeRemoveCommand('/repo', '/scratch/worktree', true)).toEqual([
23
+ 'git',
24
+ '-C',
25
+ '/repo',
26
+ 'worktree',
27
+ 'remove',
28
+ '--force',
29
+ '/scratch/worktree',
30
+ ])
31
+ })
@@ -0,0 +1,325 @@
1
+ import { afterEach, beforeEach, expect, test } from 'bun:test'
2
+ import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
3
+ import { tmpdir } from 'node:os'
4
+ import { join } from 'node:path'
5
+ import type { FindingLabel } from '../../../scripts/labels.ts'
6
+ import type { ReviewFinding } from '../../../scripts/types.ts'
7
+ import { assertOutsideRepo, buildCorpusRun, listSourceRuns, loadCorpusRun } from '../corpus.ts'
8
+ import {
9
+ actionableIds,
10
+ explainPostedDrops,
11
+ matchFindings,
12
+ parseClaudeResult,
13
+ scoreReplay,
14
+ scoreSelection,
15
+ } from '../score.ts'
16
+
17
+ function finding(over: Partial<ReviewFinding> & { id: string }): ReviewFinding {
18
+ return {
19
+ file: 'src/orders.ts',
20
+ line: 10,
21
+ severity: 'medium',
22
+ risk: { impact: 'medium', likelihood: 'possible', confidence: 'medium', action: 'should-fix' },
23
+ title: 'Something is off',
24
+ description: 'Details.',
25
+ domain: 'bugs',
26
+ ...over,
27
+ }
28
+ }
29
+
30
+ // 2 posted kept, 1 posted dropped, 1 dismissed `wrong` kept, 2 ignored kept.
31
+ const fixtureFindings: ReviewFinding[] = [
32
+ finding({ id: 'p1', domain: 'bugs' }),
33
+ finding({ id: 'p2', domain: 'security' }),
34
+ finding({ id: 'p3', domain: 'bugs' }),
35
+ finding({ id: 'd1', domain: 'bugs' }),
36
+ finding({ id: 'i1', domain: 'code-smells' }),
37
+ finding({ id: 'i2', domain: 'code-smells' }),
38
+ ]
39
+ const fixtureLabels: FindingLabel[] = [
40
+ { id: 'p1', label: 'posted', via: 'recommended' },
41
+ { id: 'p2', label: 'posted', via: 'selected' },
42
+ { id: 'p3', label: 'posted', via: 'selected' },
43
+ { id: 'd1', label: 'dismissed', reason: 'wrong' },
44
+ { id: 'i1', label: 'ignored' },
45
+ { id: 'i2', label: 'ignored' },
46
+ ]
47
+ const fixtureKept = ['p1', 'p2', 'd1', 'i1', 'i2']
48
+
49
+ function scoreFixture() {
50
+ return scoreSelection({ labels: fixtureLabels, keptIds: fixtureKept, findings: fixtureFindings })
51
+ }
52
+
53
+ test('precision is posted kept over labelled kept', () => {
54
+ expect(scoreFixture().precision).toBe(2 / 5)
55
+ })
56
+
57
+ test('recall is posted kept over all posted', () => {
58
+ expect(scoreFixture().recall).toBe(2 / 3)
59
+ })
60
+
61
+ test('kept dismissals are counted by reason', () => {
62
+ expect(scoreFixture().dismissedKept).toEqual({ wrong: 1 })
63
+ })
64
+
65
+ test('kept and posted-kept counts are split by domain', () => {
66
+ expect(scoreFixture().byDomain).toEqual({
67
+ bugs: { kept: 2, postedKept: 1 },
68
+ security: { kept: 1, postedKept: 1 },
69
+ 'code-smells': { kept: 2, postedKept: 0 },
70
+ })
71
+ })
72
+
73
+ test('posted kept findings are counted by post route', () => {
74
+ expect(scoreFixture().byVia).toEqual({ recommended: 1, selected: 1 })
75
+ })
76
+
77
+ test('totals count kept, posted and posted kept', () => {
78
+ const s = scoreFixture()
79
+ expect([s.kept, s.posted, s.postedKept]).toEqual([5, 3, 2])
80
+ })
81
+
82
+ test('zero kept findings gives a null precision', () => {
83
+ const s = scoreSelection({ labels: fixtureLabels, keptIds: [], findings: fixtureFindings })
84
+ expect(s.precision).toBeNull()
85
+ })
86
+
87
+ test('a kept id with no label stays out of the precision denominator', () => {
88
+ const findings = [...fixtureFindings, finding({ id: 'new-1' })]
89
+ const s = scoreSelection({
90
+ labels: fixtureLabels,
91
+ keptIds: [...fixtureKept, 'new-1'],
92
+ findings,
93
+ })
94
+ expect([s.kept, s.precision]).toEqual([6, 2 / 5])
95
+ })
96
+
97
+ test('kept ids with no label are counted as unlabelled', () => {
98
+ const findings = [...fixtureFindings, finding({ id: 'new-1' }), finding({ id: 'new-2' })]
99
+ const s = scoreSelection({
100
+ labels: fixtureLabels,
101
+ keptIds: [...fixtureKept, 'new-1', 'new-2'],
102
+ findings,
103
+ })
104
+ expect(s.unlabelledKept).toBe(2)
105
+ })
106
+
107
+ test('a posted finding the critic dropped is explained by its drop reason', () => {
108
+ const drops = explainPostedDrops({
109
+ labels: fixtureLabels,
110
+ candidateIds: fixtureFindings.map((f) => f.id),
111
+ keptIds: fixtureKept,
112
+ dropped: [{ id: 'p3', reason: 'design-cap' }],
113
+ verdicts: [],
114
+ })
115
+ expect(drops).toEqual([{ id: 'p3', reason: 'design-cap' }])
116
+ })
117
+
118
+ test('a posted finding merged into another is explained by its merge target', () => {
119
+ const drops = explainPostedDrops({
120
+ labels: fixtureLabels,
121
+ candidateIds: fixtureFindings.map((f) => f.id),
122
+ keptIds: fixtureKept,
123
+ dropped: [],
124
+ verdicts: [{ id: 'p3', verdict: 'merge', mergeInto: 'p1' }],
125
+ })
126
+ expect(drops).toEqual([{ id: 'p3', reason: 'merged into p1' }])
127
+ })
128
+
129
+ test('a posted finding missing with no drop or merge record is an error naming it', () => {
130
+ expect(() =>
131
+ explainPostedDrops({
132
+ labels: fixtureLabels,
133
+ candidateIds: fixtureFindings.map((f) => f.id),
134
+ keptIds: fixtureKept,
135
+ dropped: [],
136
+ verdicts: [],
137
+ }),
138
+ ).toThrow('p3')
139
+ })
140
+
141
+ test('a posted finding outside the candidate pool is not reported as dropped', () => {
142
+ const drops = explainPostedDrops({
143
+ labels: [...fixtureLabels, { id: 'peer-1', label: 'posted' }],
144
+ candidateIds: fixtureFindings.map((f) => f.id),
145
+ keptIds: fixtureKept,
146
+ dropped: [{ id: 'p3', reason: 'speculative' }],
147
+ verdicts: [],
148
+ })
149
+ expect(drops).toEqual([{ id: 'p3', reason: 'speculative' }])
150
+ })
151
+
152
+ test('actionable ids keep must-fix and should-fix findings and leave out suggestions', () => {
153
+ const risk = (action: ReviewFinding['risk']['action']) => ({
154
+ impact: 'medium' as const,
155
+ likelihood: 'possible' as const,
156
+ confidence: 'high' as const,
157
+ action,
158
+ })
159
+ const kept = [
160
+ finding({ id: 'm', risk: risk('must-fix') }),
161
+ finding({ id: 's', risk: risk('should-fix') }),
162
+ finding({ id: 'c', risk: risk('consider') }),
163
+ finding({ id: 'o', risk: risk('optional') }),
164
+ ]
165
+ expect(actionableIds(kept)).toEqual(['m', 's'])
166
+ })
167
+
168
+ test('a kept id missing from the findings is an error naming it', () => {
169
+ expect(() =>
170
+ scoreSelection({ labels: fixtureLabels, keptIds: ['ghost'], findings: fixtureFindings }),
171
+ ).toThrow('ghost')
172
+ })
173
+
174
+ const labelled = finding({
175
+ id: 'L',
176
+ file: 'src/orders.ts',
177
+ line: 10,
178
+ title: 'Unbounded retry loop in fetchOrders handler',
179
+ })
180
+
181
+ test('a 4-line offset with a similar title matches', () => {
182
+ const moved = finding({
183
+ id: 'N',
184
+ line: 14,
185
+ title: 'Unbounded retry loop when fetchOrders fails',
186
+ })
187
+ expect(matchFindings([moved], [labelled])).toEqual(new Map([['N', 'L']]))
188
+ })
189
+
190
+ test('a 6-line offset does not match', () => {
191
+ const moved = finding({ id: 'N', line: 16, title: labelled.title })
192
+ expect(matchFindings([moved], [labelled]).size).toBe(0)
193
+ })
194
+
195
+ test('a different file does not match', () => {
196
+ const elsewhere = finding({ id: 'N', file: 'src/billing.ts', title: labelled.title })
197
+ expect(matchFindings([elsewhere], [labelled]).size).toBe(0)
198
+ })
199
+
200
+ test('two candidates for one labelled finding resolve to the higher dice', () => {
201
+ const weaker = finding({ id: 'B', line: 11, title: 'Retry loop missing backoff' })
202
+ const stronger = finding({ id: 'A', line: 12, title: labelled.title })
203
+ expect(matchFindings([weaker, stronger], [labelled])).toEqual(new Map([['A', 'L']]))
204
+ })
205
+
206
+ test('a replayed finding that matches a posted one scores as posted kept', () => {
207
+ const moved = finding({ id: 'bugs-9', line: 12, title: labelled.title })
208
+ const { score } = scoreReplay({
209
+ labels: [{ id: 'L', label: 'posted' }],
210
+ labelled: [labelled],
211
+ newKept: [moved],
212
+ })
213
+ expect([score.postedKept, score.precision]).toEqual([1, 1])
214
+ })
215
+
216
+ test('an unmatched replayed finding is unlabelled even when its id repeats a labelled id', () => {
217
+ const unrelated = finding({ id: 'L', file: 'src/billing.ts', title: 'Totally different' })
218
+ const { score, unlabelled } = scoreReplay({
219
+ labels: [{ id: 'L', label: 'posted' }],
220
+ labelled: [labelled],
221
+ newKept: [unrelated],
222
+ })
223
+ expect([unlabelled, score.postedKept, score.precision]).toEqual([1, 0, null])
224
+ })
225
+
226
+ let root: string
227
+
228
+ beforeEach(async () => {
229
+ root = await mkdtemp(join(tmpdir(), 'magpie-quality-'))
230
+ })
231
+
232
+ afterEach(async () => {
233
+ await rm(root, { recursive: true, force: true })
234
+ })
235
+
236
+ async function writeRun(dir: string, files: Record<string, string>): Promise<void> {
237
+ await mkdir(dir, { recursive: true })
238
+ for (const [name, content] of Object.entries(files)) {
239
+ await mkdir(join(dir, name, '..'), { recursive: true })
240
+ await writeFile(join(dir, name), content)
241
+ }
242
+ }
243
+
244
+ const postedRun = {
245
+ 'post-status.json': '{"a":"posted"}',
246
+ 'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
247
+ }
248
+
249
+ test('an archived run with post status and final findings is a source run', async () => {
250
+ await writeRun(join(root, 'pr-50-1.archived-2'), postedRun)
251
+ expect(await listSourceRuns(root)).toEqual(['pr-50-1.archived-2'])
252
+ })
253
+
254
+ test('a preview run is not a source run', async () => {
255
+ await writeRun(join(root, 'preview-123'), postedRun)
256
+ expect(await listSourceRuns(root)).toEqual([])
257
+ })
258
+
259
+ test('a run that never posted is not a source run', async () => {
260
+ await writeRun(join(root, 'pr-7-1'), {
261
+ 'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
262
+ })
263
+ expect(await listSourceRuns(root)).toEqual([])
264
+ })
265
+
266
+ test('a corpus dir inside the repo is refused', () => {
267
+ expect(() => assertOutsideRepo('/work/repo/corpus', '/work/repo')).toThrow('/work/repo')
268
+ })
269
+
270
+ test('a corpus dir beside the repo with a shared name prefix is allowed', () => {
271
+ expect(() => assertOutsideRepo('/work/repo-corpus', '/work/repo')).not.toThrow()
272
+ })
273
+
274
+ test('a kept id absent from final findings is labelled ignored in the corpus', async () => {
275
+ const src = join(root, 'pr-9-1')
276
+ await writeRun(src, {
277
+ ...postedRun,
278
+ 'findings.kept.json': JSON.stringify([finding({ id: 'a' }), finding({ id: 'b' })]),
279
+ })
280
+ await buildCorpusRun(src, join(root, 'corpus', 'pr-9-1'))
281
+ const labels = JSON.parse(await readFile(join(root, 'corpus', 'pr-9-1', 'labels.json'), 'utf8'))
282
+ expect(labels).toEqual([
283
+ { id: 'a', label: 'posted' },
284
+ { id: 'b', label: 'ignored' },
285
+ ])
286
+ })
287
+
288
+ test('a corpus run without merge candidates loads them as absent', async () => {
289
+ const dir = join(root, 'pr-9-1')
290
+ await writeRun(dir, {
291
+ 'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
292
+ 'labels.json': JSON.stringify([{ id: 'a', label: 'ignored' }]),
293
+ })
294
+ expect((await loadCorpusRun(dir)).mergeCandidates).toBeNull()
295
+ })
296
+
297
+ test('a corpus file that fails to parse is an error naming the file', async () => {
298
+ const dir = join(root, 'pr-9-1')
299
+ await writeRun(dir, {
300
+ 'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
301
+ 'labels.json': JSON.stringify([{ id: 'a', label: 'ignored' }]),
302
+ 'merge-candidates.json': '{not json',
303
+ })
304
+ await expect(loadCorpusRun(dir)).rejects.toThrow(join(dir, 'merge-candidates.json'))
305
+ })
306
+
307
+ test('a claude result marked as an error is an error naming the step', () => {
308
+ const stdout = JSON.stringify({
309
+ type: 'result',
310
+ subtype: 'error_max_budget_usd',
311
+ is_error: true,
312
+ result: 'budget exceeded',
313
+ })
314
+ expect(() => parseClaudeResult(stdout, 'critic batch 2')).toThrow('critic batch 2')
315
+ })
316
+
317
+ test('a successful claude result reports its cost', () => {
318
+ const stdout = JSON.stringify({
319
+ type: 'result',
320
+ subtype: 'success',
321
+ is_error: false,
322
+ total_cost_usd: 0.42,
323
+ })
324
+ expect(parseClaudeResult(stdout, 'critic batch 1')).toEqual({ costUsd: 0.42 })
325
+ })
@@ -0,0 +1,80 @@
1
+ #!/usr/bin/env bun
2
+ // Scores each corpus run's original critic selection (findings.kept.json)
3
+ // against its labels: the number a replayed critic has to beat.
4
+
5
+ import { mkdir, writeFile } from 'node:fs/promises'
6
+ import { join } from 'node:path'
7
+ import { type CorpusRun, listCorpusRuns, loadCorpusRun, resolveCorpusRoot } from './corpus.ts'
8
+ import {
9
+ formatRatio,
10
+ parseFlags,
11
+ poolRuns,
12
+ type QualityScore,
13
+ resultTimestamp,
14
+ scoreSelection,
15
+ } from './score.ts'
16
+
17
+ /** The critic chose from findings.deduped.json, so that is the pool recall counts against. */
18
+ export function originalSelection(run: CorpusRun) {
19
+ if (!run.kept) throw new Error(`${run.dir} has no findings.kept.json`)
20
+ if (!run.deduped) throw new Error(`${run.dir} has no findings.deduped.json`)
21
+ return {
22
+ runId: run.runId,
23
+ labels: run.labels,
24
+ keptIds: run.kept.map((f) => f.id),
25
+ findings: run.deduped,
26
+ }
27
+ }
28
+
29
+ const DISMISS_COLUMNS = ['wrong', 'not-worth-it', 'duplicate', 'style']
30
+
31
+ function row(name: string, s: QualityScore): string[] {
32
+ return [
33
+ name,
34
+ String(s.kept),
35
+ String(s.posted),
36
+ String(s.postedKept),
37
+ formatRatio(s.precision),
38
+ formatRatio(s.recall),
39
+ ...DISMISS_COLUMNS.map((r) => String(s.dismissedKept[r] ?? 0)),
40
+ ]
41
+ }
42
+
43
+ function renderTable(rows: string[][]): string {
44
+ const header = ['run', 'kept', 'posted', 'postedKept', 'precision', 'recall', ...DISMISS_COLUMNS]
45
+ const all = [header, ...rows]
46
+ const widths = header.map((_, i) => Math.max(...all.map((r) => (r[i] ?? '').length)))
47
+ return all.map((r) => r.map((cell, i) => cell.padEnd(widths[i] ?? 0)).join(' ')).join('\n')
48
+ }
49
+
50
+ async function main(argv: string[]): Promise<number> {
51
+ const corpusRoot = resolveCorpusRoot(parseFlags(argv, ['out']))
52
+ const selections = []
53
+ for (const runId of await listCorpusRuns(corpusRoot)) {
54
+ selections.push(originalSelection(await loadCorpusRun(join(corpusRoot, runId))))
55
+ }
56
+ if (selections.length === 0) throw new Error(`no corpus runs in ${corpusRoot}; run corpus.ts`)
57
+
58
+ const runs = selections.map((s) => ({ runId: s.runId, score: scoreSelection(s) }))
59
+ const total = scoreSelection(poolRuns(selections))
60
+ process.stdout.write(
61
+ `${renderTable([...runs.map((r) => row(r.runId, r.score)), row('TOTAL', total)])}\n`,
62
+ )
63
+
64
+ const resultsDir = join(corpusRoot, 'results')
65
+ await mkdir(resultsDir, { recursive: true })
66
+ const outPath = join(resultsDir, `${resultTimestamp(new Date())}-baseline.json`)
67
+ await writeFile(outPath, `${JSON.stringify({ tier: 'baseline', runs, total }, null, 2)}\n`)
68
+ process.stdout.write(`wrote ${outPath}\n`)
69
+ return 0
70
+ }
71
+
72
+ if (import.meta.main) {
73
+ main(process.argv.slice(2)).then(
74
+ (code) => process.exit(code),
75
+ (err) => {
76
+ process.stderr.write(`baseline: ${err instanceof Error ? err.message : String(err)}\n`)
77
+ process.exit(1)
78
+ },
79
+ )
80
+ }