@iceinvein/agent-skills 0.21.0 → 0.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/skills/index.json +2 -2
- package/skills/magpie/README.md +3 -4
- package/skills/magpie/SKILL.md +31 -29
- package/skills/magpie/bin/magpie.ts +84 -6
- package/skills/magpie/evals/README.md +19 -5
- package/skills/magpie/evals/codex-missing-falls-back/fixture.sh +64 -23
- package/skills/magpie/evals/post-folds-selection-events/fixture.sh +9 -3
- package/skills/magpie/evals/quality/README.md +141 -0
- package/skills/magpie/evals/quality/__tests__/replay-critic.test.ts +31 -0
- package/skills/magpie/evals/quality/__tests__/score.test.ts +325 -0
- package/skills/magpie/evals/quality/baseline.ts +80 -0
- package/skills/magpie/evals/quality/corpus.ts +259 -0
- package/skills/magpie/evals/quality/replay-critic.ts +277 -0
- package/skills/magpie/evals/quality/replay-full.ts +188 -0
- package/skills/magpie/evals/quality/score.ts +251 -0
- package/skills/magpie/evals/report-ends-the-turn/fixture.sh +6 -2
- package/skills/magpie/evals/resume-finds-active-run/fixture.sh +1 -1
- package/skills/magpie/evals/shard-gate-stops-and-asks/fixture.sh +1 -1
- package/skills/magpie/fixtures/example-pr/brief.json +0 -4
- package/skills/magpie/references/critic.md +178 -40
- package/skills/magpie/references/peer-review.md +5 -4
- package/skills/magpie/references/scout.md +17 -42
- package/skills/magpie/references/specialists.md +43 -136
- package/skills/magpie/scripts/__tests__/cleanup-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/critic-cmd.test.ts +260 -0
- package/skills/magpie/scripts/__tests__/critic.test.ts +328 -0
- package/skills/magpie/scripts/__tests__/dedupe-cmd.test.ts +89 -0
- package/skills/magpie/scripts/__tests__/dedupe.test.ts +43 -1
- package/skills/magpie/scripts/__tests__/evidence-filter.test.ts +155 -2
- package/skills/magpie/scripts/__tests__/helper.test.ts +80 -0
- package/skills/magpie/scripts/__tests__/labels.test.ts +152 -0
- package/skills/magpie/scripts/__tests__/open-cmd.test.ts +28 -0
- package/skills/magpie/scripts/__tests__/pipeline-e2e.test.ts +1 -0
- package/skills/magpie/scripts/__tests__/post-cmd.test.ts +47 -0
- package/skills/magpie/scripts/__tests__/refresh.test.ts +23 -1
- package/skills/magpie/scripts/__tests__/render-action-bar.test.ts +63 -5
- package/skills/magpie/scripts/__tests__/render-annotation.test.ts +38 -0
- package/skills/magpie/scripts/__tests__/render-cmd.test.ts +48 -1
- package/skills/magpie/scripts/__tests__/render-diff.test.ts +34 -0
- package/skills/magpie/scripts/__tests__/render-findings.test.ts +111 -30
- package/skills/magpie/scripts/__tests__/render-issues-list.test.ts +186 -1
- package/skills/magpie/scripts/__tests__/render-progress.test.ts +1 -1
- package/skills/magpie/scripts/__tests__/serve-cmd.test.ts +19 -0
- package/skills/magpie/scripts/__tests__/server.test.ts +76 -0
- package/skills/magpie/scripts/__tests__/skill-lint.test.ts +166 -179
- package/skills/magpie/scripts/__tests__/tests-check.test.ts +57 -0
- package/skills/magpie/scripts/__tests__/types.test.ts +90 -36
- package/skills/magpie/scripts/cleanup-cmd.ts +6 -0
- package/skills/magpie/scripts/critic-cmd.ts +183 -0
- package/skills/magpie/scripts/critic.ts +273 -0
- package/skills/magpie/scripts/dedupe-cmd.ts +14 -4
- package/skills/magpie/scripts/dedupe.ts +41 -0
- package/skills/magpie/scripts/evidence-filter.ts +81 -17
- package/skills/magpie/scripts/helper.js +87 -9
- package/skills/magpie/scripts/labels-cmd.ts +43 -0
- package/skills/magpie/scripts/labels.ts +99 -0
- package/skills/magpie/scripts/open-cmd.ts +10 -3
- package/skills/magpie/scripts/post-cmd.ts +26 -2
- package/skills/magpie/scripts/preview-cmd.ts +4 -0
- package/skills/magpie/scripts/refresh.ts +21 -17
- package/skills/magpie/scripts/render-action-bar.ts +21 -4
- package/skills/magpie/scripts/render-annotation.ts +36 -2
- package/skills/magpie/scripts/render-cmd.ts +21 -2
- package/skills/magpie/scripts/render-diff.ts +6 -2
- package/skills/magpie/scripts/render-findings.ts +21 -10
- package/skills/magpie/scripts/render-issues-list.ts +66 -16
- package/skills/magpie/scripts/render-progress.ts +1 -1
- package/skills/magpie/scripts/server.ts +8 -1
- package/skills/magpie/scripts/tests-check.ts +23 -2
- package/skills/magpie/scripts/types.ts +36 -46
- package/skills/magpie/skill.json +2 -2
- package/skills/magpie/templates/styles.css +83 -2
- package/skills/magpie/tsconfig.json +1 -1
- package/skills/magpie/evals/consent-required-never-approves/case.yaml +0 -4
- package/skills/magpie/evals/consent-required-never-approves/fixture.sh +0 -213
- package/skills/magpie/evals/consent-required-never-approves/graders/context-stage-was-closed.md +0 -6
- package/skills/magpie/evals/consent-required-never-approves/graders/indexing-was-never-approved.md +0 -7
- package/skills/magpie/evals/consent-required-never-approves/graders/probe-was-run.md +0 -5
- package/skills/magpie/evals/consent-required-never-approves/graders/skill-fired.md +0 -5
- package/skills/magpie/evals/consent-required-never-approves/graders/user-was-told-it-is-unavailable.md +0 -10
- package/skills/magpie/evals/consent-required-never-approves/prompt.md +0 -11
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# Quality eval
|
|
2
|
+
|
|
3
|
+
Measures how well magpie picks findings, using what reviewers actually did with past runs
|
|
4
|
+
as labels: a final finding is `posted`, `dismissed` (with a reason) or `ignored`. The
|
|
5
|
+
other evals in `evals/` check that the skill follows its walkthrough; this one checks
|
|
6
|
+
whether the findings it keeps are the ones a reviewer wanted.
|
|
7
|
+
|
|
8
|
+
Run every command from `skills/magpie`.
|
|
9
|
+
|
|
10
|
+
## The corpus
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
bun evals/quality/corpus.ts
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Copies every run under `~/.magpie` (or `$MAGPIE_HOME`; active and `.archived-` alike,
|
|
17
|
+
`preview-*` skipped)
|
|
18
|
+
that has both `post-status.json` and `findings.final.json` into
|
|
19
|
+
`~/.magpie/corpus/<runId>/`: `pr.json`, `diff.patch`, `findings.deduped.json`,
|
|
20
|
+
`findings.kept.json`, `findings.final.json`, `merge-candidates.json` and `brief.json`,
|
|
21
|
+
each when present. It then writes `labels.json` by folding `post-status.json`,
|
|
22
|
+
`state/events` and `log.jsonl`, and prints one line of counts per run. Rebuilding replaces
|
|
23
|
+
each run's corpus directory.
|
|
24
|
+
|
|
25
|
+
The corpus holds client code, so it never lives in the repo: the script exits 1 when the
|
|
26
|
+
corpus dir resolves inside the git repository that holds it. `--out <dir>` points it
|
|
27
|
+
elsewhere, for tests and dry runs; every script below takes the same flag.
|
|
28
|
+
|
|
29
|
+
Older runs predate merge candidates, briefs and the dismiss UI. A missing optional file is
|
|
30
|
+
read as absent; a file that is present but does not parse fails the build, naming it.
|
|
31
|
+
|
|
32
|
+
## Metrics
|
|
33
|
+
|
|
34
|
+
`score.ts` exports `scoreSelection`, which scores a set of kept ids against the labels:
|
|
35
|
+
|
|
36
|
+
- `kept`: findings in the selection.
|
|
37
|
+
- `posted`: posted labels on findings in the candidate pool the selection was made from
|
|
38
|
+
(`findings.deduped.json` for the critic). Peer-review additions reach
|
|
39
|
+
`findings.final.json` without passing the critic, so they are not held against it.
|
|
40
|
+
- `postedKept`: kept findings that were posted.
|
|
41
|
+
- `precision`: `postedKept` over kept findings that have a label. Every id in
|
|
42
|
+
`findings.final.json` has one; a kept id that peer review removed before the report is
|
|
43
|
+
labelled `ignored`, since the reviewer never saw it. A kept id with no label at all (a
|
|
44
|
+
replay keeping a candidate the original critic dropped) is left out of the denominator.
|
|
45
|
+
Null when nothing labelled was kept.
|
|
46
|
+
- `recall`: `postedKept` over `posted`. Null when nothing was posted.
|
|
47
|
+
- `dismissedKept`: kept findings dismissed, by reason (`wrong`, `not-worth-it`,
|
|
48
|
+
`duplicate`, `style`).
|
|
49
|
+
- `byDomain`: `kept` and `postedKept` per specialist domain (`unassigned` when a finding
|
|
50
|
+
has none).
|
|
51
|
+
- `byVia`: posted kept findings by post route (`recommended`, `selected`, `one`, `cli`;
|
|
52
|
+
`unrecorded` for runs that logged no route).
|
|
53
|
+
|
|
54
|
+
`matchFindings` pairs a replay's freshly generated findings with labelled ones: same file,
|
|
55
|
+
line within 5, title Dice of at least 0.4 on `tokenize(title)`. Pairs are taken best Dice
|
|
56
|
+
first, each labelled finding matched at most once.
|
|
57
|
+
|
|
58
|
+
## Confounds
|
|
59
|
+
|
|
60
|
+
Read every number here with these in mind.
|
|
61
|
+
|
|
62
|
+
- **Post Recommended.** The report's "Post recommended" button posts the top findings in
|
|
63
|
+
one click. A finding posted that way was not individually judged, so `posted` partly
|
|
64
|
+
measures rank, not worth. `byVia` separates those posts out where the run logged a
|
|
65
|
+
route.
|
|
66
|
+
- **Ignored is not rejected.** Most runs predate the dismiss UI, so a finding the reviewer
|
|
67
|
+
disliked usually reads as `ignored`, not `dismissed`. Precision treats `ignored` as a
|
|
68
|
+
miss, which is harsh on any finding the reviewer simply did not get to.
|
|
69
|
+
- **Repeated PRs.** Several runs review the same PR at different heads (three runs of
|
|
70
|
+
PR 50 today). Their findings overlap heavily, and a reviewer who posted a finding on one
|
|
71
|
+
run tends not to post it again on the next, which drags the later run's precision down.
|
|
72
|
+
Pooled totals weight those PRs once per run.
|
|
73
|
+
- **Baseline recall is 1 by construction.** The reviewer only ever saw kept findings, so
|
|
74
|
+
every posted finding in the pool was kept. Recall only means something for a replay,
|
|
75
|
+
where a different critic can drop a finding the reviewer posted.
|
|
76
|
+
|
|
77
|
+
## Baseline
|
|
78
|
+
|
|
79
|
+
```
|
|
80
|
+
bun evals/quality/baseline.ts
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Scores each run's original `findings.kept.json` against its labels, prints a table with a
|
|
84
|
+
pooled `TOTAL` row, and writes `~/.magpie/corpus/results/<ts>-baseline.json`. No model
|
|
85
|
+
calls; free to run.
|
|
86
|
+
|
|
87
|
+
## Critic replay
|
|
88
|
+
|
|
89
|
+
```
|
|
90
|
+
bun evals/quality/replay-critic.ts --corpus <runId> --repo <path-to-local-clone> [--design-cap <n>]
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Re-runs stage 6 alone on one corpus run:
|
|
94
|
+
|
|
95
|
+
1. `git -C <repo> worktree add --detach <scratch>/worktree <headRefOid>`. The PR head must
|
|
96
|
+
already be in the clone (`git fetch origin pull/<n>/head`); a missing sha exits 1.
|
|
97
|
+
2. Copies the corpus files into a scratch run dir under the system temp dir.
|
|
98
|
+
3. `magpie critic-prompt`, then one `claude -p --allowedTools Read,Grep,Glob,Write
|
|
99
|
+
--output-format json --add-dir <scratch>` per batch, prompt on stdin, from the
|
|
100
|
+
worktree. Batches run in parallel. A non-zero exit, an `is_error` result or a missing
|
|
101
|
+
output file fails the replay.
|
|
102
|
+
4. `magpie critic-apply` (with `--design-cap <n>` when given, so a cap can be compared
|
|
103
|
+
against the default of 3), then `scoreSelection` over the new `findings.kept.json`,
|
|
104
|
+
written with the run's baseline score to `results/<ts>-critic[-cap<n>]-<runId>.json`.
|
|
105
|
+
The result also keeps every critic verdict, `critic-dropped.json`, and
|
|
106
|
+
`postedDropped`: each posted candidate the replay did not keep, with the critic's
|
|
107
|
+
reason, `design-cap`, or its merge target. Those lines are printed too, so a recall
|
|
108
|
+
loss can be pinned on the critic or the cap. `unlabelledKept` counts kept findings the
|
|
109
|
+
original critic had dropped: nobody ever judged them, so precision leaves them out
|
|
110
|
+
and the count says how much of the selection precision does not cover.
|
|
111
|
+
5. Removes the worktree and scratch dir whatever happened.
|
|
112
|
+
|
|
113
|
+
**Cost:** one Claude session per batch of up to 30 candidates; the corpus runs have 17 to
|
|
114
|
+
84 deduped candidates, so 1 to 3 sessions each. Each session reads the candidates' code in
|
|
115
|
+
the worktree, so cost grows with candidate count. The results file records the
|
|
116
|
+
`total_cost_usd` Claude reports, summed over batches. There is no spend cap on this tier.
|
|
117
|
+
|
|
118
|
+
## Full replay
|
|
119
|
+
|
|
120
|
+
```
|
|
121
|
+
bun evals/quality/replay-full.ts --corpus <runId> --repo <path-to-local-clone> [--max-cost-usd N]
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Re-runs stages 3 to 6 (context, specialists, dedupe, critic) on the run's `diff.patch`:
|
|
125
|
+
|
|
126
|
+
1. Scaffolds a scratch run dir: worktree at the PR head as above, `pr.json` and
|
|
127
|
+
`diff.patch` copied, the deterministic tests finding and a `setup` done line written in
|
|
128
|
+
place of `magpie setup` (which needs the live PR), then `magpie shard`.
|
|
129
|
+
2. Runs one `claude -p --allowedTools Bash,Read,Grep,Glob,Write,Edit,Agent
|
|
130
|
+
--max-budget-usd <N> --output-format json` from the worktree, telling it to follow this
|
|
131
|
+
checkout's `SKILL.md`, resume from `magpie status`, and stop once `findings.kept.json`
|
|
132
|
+
is written. Serve, peer review, report, post and cleanup are skipped.
|
|
133
|
+
3. Matches the new kept findings to `findings.final.json` with `matchFindings` and scores
|
|
134
|
+
them; unmatched findings count as `unlabelled` and stay out of precision. Writes
|
|
135
|
+
`results/<ts>-full-<runId>.json`.
|
|
136
|
+
|
|
137
|
+
**Cost:** a whole review: a scout, five specialists per shard, and the critic batches.
|
|
138
|
+
`--max-cost-usd` (default 10) is passed to Claude as `--max-budget-usd`, the CLI's own
|
|
139
|
+
spend cap; `claude -p` has no `--max-cost-usd` flag. Hitting the cap stops the session,
|
|
140
|
+
and the replay then fails without scoring, on Claude's error result or on the missing
|
|
141
|
+
`findings.kept.json`.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { expect, test } from 'bun:test'
|
|
2
|
+
import { claudeFailureMessage, worktreeRemoveCommand } from '../replay-critic.ts'
|
|
3
|
+
|
|
4
|
+
test('a failed claude run names its label and the result subtype from stdout', () => {
|
|
5
|
+
const stdout = JSON.stringify({ type: 'result', is_error: true, subtype: 'error_max_turns' })
|
|
6
|
+
expect(claudeFailureMessage('critic batch 2', 1, stdout, 'boom\n')).toBe(
|
|
7
|
+
'critic batch 2: claude -p exit 1; is_error true, subtype error_max_turns; stderr: boom',
|
|
8
|
+
)
|
|
9
|
+
})
|
|
10
|
+
|
|
11
|
+
test('a failed claude run with non-JSON stdout carries the raw stdout', () => {
|
|
12
|
+
expect(claudeFailureMessage('full replay', 137, 'Killed\n', '')).toBe(
|
|
13
|
+
'full replay: claude -p exit 137; stdout: Killed; stderr: ',
|
|
14
|
+
)
|
|
15
|
+
})
|
|
16
|
+
|
|
17
|
+
test('a worktree that was never added has no removal command', () => {
|
|
18
|
+
expect(worktreeRemoveCommand('/repo', '/scratch/worktree', false)).toBeNull()
|
|
19
|
+
})
|
|
20
|
+
|
|
21
|
+
test('an added worktree is removed with a forced git worktree remove', () => {
|
|
22
|
+
expect(worktreeRemoveCommand('/repo', '/scratch/worktree', true)).toEqual([
|
|
23
|
+
'git',
|
|
24
|
+
'-C',
|
|
25
|
+
'/repo',
|
|
26
|
+
'worktree',
|
|
27
|
+
'remove',
|
|
28
|
+
'--force',
|
|
29
|
+
'/scratch/worktree',
|
|
30
|
+
])
|
|
31
|
+
})
|
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
import { afterEach, beforeEach, expect, test } from 'bun:test'
|
|
2
|
+
import { mkdir, mkdtemp, readFile, rm, writeFile } from 'node:fs/promises'
|
|
3
|
+
import { tmpdir } from 'node:os'
|
|
4
|
+
import { join } from 'node:path'
|
|
5
|
+
import type { FindingLabel } from '../../../scripts/labels.ts'
|
|
6
|
+
import type { ReviewFinding } from '../../../scripts/types.ts'
|
|
7
|
+
import { assertOutsideRepo, buildCorpusRun, listSourceRuns, loadCorpusRun } from '../corpus.ts'
|
|
8
|
+
import {
|
|
9
|
+
actionableIds,
|
|
10
|
+
explainPostedDrops,
|
|
11
|
+
matchFindings,
|
|
12
|
+
parseClaudeResult,
|
|
13
|
+
scoreReplay,
|
|
14
|
+
scoreSelection,
|
|
15
|
+
} from '../score.ts'
|
|
16
|
+
|
|
17
|
+
function finding(over: Partial<ReviewFinding> & { id: string }): ReviewFinding {
|
|
18
|
+
return {
|
|
19
|
+
file: 'src/orders.ts',
|
|
20
|
+
line: 10,
|
|
21
|
+
severity: 'medium',
|
|
22
|
+
risk: { impact: 'medium', likelihood: 'possible', confidence: 'medium', action: 'should-fix' },
|
|
23
|
+
title: 'Something is off',
|
|
24
|
+
description: 'Details.',
|
|
25
|
+
domain: 'bugs',
|
|
26
|
+
...over,
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
// 2 posted kept, 1 posted dropped, 1 dismissed `wrong` kept, 2 ignored kept.
|
|
31
|
+
const fixtureFindings: ReviewFinding[] = [
|
|
32
|
+
finding({ id: 'p1', domain: 'bugs' }),
|
|
33
|
+
finding({ id: 'p2', domain: 'security' }),
|
|
34
|
+
finding({ id: 'p3', domain: 'bugs' }),
|
|
35
|
+
finding({ id: 'd1', domain: 'bugs' }),
|
|
36
|
+
finding({ id: 'i1', domain: 'code-smells' }),
|
|
37
|
+
finding({ id: 'i2', domain: 'code-smells' }),
|
|
38
|
+
]
|
|
39
|
+
const fixtureLabels: FindingLabel[] = [
|
|
40
|
+
{ id: 'p1', label: 'posted', via: 'recommended' },
|
|
41
|
+
{ id: 'p2', label: 'posted', via: 'selected' },
|
|
42
|
+
{ id: 'p3', label: 'posted', via: 'selected' },
|
|
43
|
+
{ id: 'd1', label: 'dismissed', reason: 'wrong' },
|
|
44
|
+
{ id: 'i1', label: 'ignored' },
|
|
45
|
+
{ id: 'i2', label: 'ignored' },
|
|
46
|
+
]
|
|
47
|
+
const fixtureKept = ['p1', 'p2', 'd1', 'i1', 'i2']
|
|
48
|
+
|
|
49
|
+
function scoreFixture() {
|
|
50
|
+
return scoreSelection({ labels: fixtureLabels, keptIds: fixtureKept, findings: fixtureFindings })
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
test('precision is posted kept over labelled kept', () => {
|
|
54
|
+
expect(scoreFixture().precision).toBe(2 / 5)
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
test('recall is posted kept over all posted', () => {
|
|
58
|
+
expect(scoreFixture().recall).toBe(2 / 3)
|
|
59
|
+
})
|
|
60
|
+
|
|
61
|
+
test('kept dismissals are counted by reason', () => {
|
|
62
|
+
expect(scoreFixture().dismissedKept).toEqual({ wrong: 1 })
|
|
63
|
+
})
|
|
64
|
+
|
|
65
|
+
test('kept and posted-kept counts are split by domain', () => {
|
|
66
|
+
expect(scoreFixture().byDomain).toEqual({
|
|
67
|
+
bugs: { kept: 2, postedKept: 1 },
|
|
68
|
+
security: { kept: 1, postedKept: 1 },
|
|
69
|
+
'code-smells': { kept: 2, postedKept: 0 },
|
|
70
|
+
})
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
test('posted kept findings are counted by post route', () => {
|
|
74
|
+
expect(scoreFixture().byVia).toEqual({ recommended: 1, selected: 1 })
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
test('totals count kept, posted and posted kept', () => {
|
|
78
|
+
const s = scoreFixture()
|
|
79
|
+
expect([s.kept, s.posted, s.postedKept]).toEqual([5, 3, 2])
|
|
80
|
+
})
|
|
81
|
+
|
|
82
|
+
test('zero kept findings gives a null precision', () => {
|
|
83
|
+
const s = scoreSelection({ labels: fixtureLabels, keptIds: [], findings: fixtureFindings })
|
|
84
|
+
expect(s.precision).toBeNull()
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
test('a kept id with no label stays out of the precision denominator', () => {
|
|
88
|
+
const findings = [...fixtureFindings, finding({ id: 'new-1' })]
|
|
89
|
+
const s = scoreSelection({
|
|
90
|
+
labels: fixtureLabels,
|
|
91
|
+
keptIds: [...fixtureKept, 'new-1'],
|
|
92
|
+
findings,
|
|
93
|
+
})
|
|
94
|
+
expect([s.kept, s.precision]).toEqual([6, 2 / 5])
|
|
95
|
+
})
|
|
96
|
+
|
|
97
|
+
test('kept ids with no label are counted as unlabelled', () => {
|
|
98
|
+
const findings = [...fixtureFindings, finding({ id: 'new-1' }), finding({ id: 'new-2' })]
|
|
99
|
+
const s = scoreSelection({
|
|
100
|
+
labels: fixtureLabels,
|
|
101
|
+
keptIds: [...fixtureKept, 'new-1', 'new-2'],
|
|
102
|
+
findings,
|
|
103
|
+
})
|
|
104
|
+
expect(s.unlabelledKept).toBe(2)
|
|
105
|
+
})
|
|
106
|
+
|
|
107
|
+
test('a posted finding the critic dropped is explained by its drop reason', () => {
|
|
108
|
+
const drops = explainPostedDrops({
|
|
109
|
+
labels: fixtureLabels,
|
|
110
|
+
candidateIds: fixtureFindings.map((f) => f.id),
|
|
111
|
+
keptIds: fixtureKept,
|
|
112
|
+
dropped: [{ id: 'p3', reason: 'design-cap' }],
|
|
113
|
+
verdicts: [],
|
|
114
|
+
})
|
|
115
|
+
expect(drops).toEqual([{ id: 'p3', reason: 'design-cap' }])
|
|
116
|
+
})
|
|
117
|
+
|
|
118
|
+
test('a posted finding merged into another is explained by its merge target', () => {
|
|
119
|
+
const drops = explainPostedDrops({
|
|
120
|
+
labels: fixtureLabels,
|
|
121
|
+
candidateIds: fixtureFindings.map((f) => f.id),
|
|
122
|
+
keptIds: fixtureKept,
|
|
123
|
+
dropped: [],
|
|
124
|
+
verdicts: [{ id: 'p3', verdict: 'merge', mergeInto: 'p1' }],
|
|
125
|
+
})
|
|
126
|
+
expect(drops).toEqual([{ id: 'p3', reason: 'merged into p1' }])
|
|
127
|
+
})
|
|
128
|
+
|
|
129
|
+
test('a posted finding missing with no drop or merge record is an error naming it', () => {
|
|
130
|
+
expect(() =>
|
|
131
|
+
explainPostedDrops({
|
|
132
|
+
labels: fixtureLabels,
|
|
133
|
+
candidateIds: fixtureFindings.map((f) => f.id),
|
|
134
|
+
keptIds: fixtureKept,
|
|
135
|
+
dropped: [],
|
|
136
|
+
verdicts: [],
|
|
137
|
+
}),
|
|
138
|
+
).toThrow('p3')
|
|
139
|
+
})
|
|
140
|
+
|
|
141
|
+
test('a posted finding outside the candidate pool is not reported as dropped', () => {
|
|
142
|
+
const drops = explainPostedDrops({
|
|
143
|
+
labels: [...fixtureLabels, { id: 'peer-1', label: 'posted' }],
|
|
144
|
+
candidateIds: fixtureFindings.map((f) => f.id),
|
|
145
|
+
keptIds: fixtureKept,
|
|
146
|
+
dropped: [{ id: 'p3', reason: 'speculative' }],
|
|
147
|
+
verdicts: [],
|
|
148
|
+
})
|
|
149
|
+
expect(drops).toEqual([{ id: 'p3', reason: 'speculative' }])
|
|
150
|
+
})
|
|
151
|
+
|
|
152
|
+
test('actionable ids keep must-fix and should-fix findings and leave out suggestions', () => {
|
|
153
|
+
const risk = (action: ReviewFinding['risk']['action']) => ({
|
|
154
|
+
impact: 'medium' as const,
|
|
155
|
+
likelihood: 'possible' as const,
|
|
156
|
+
confidence: 'high' as const,
|
|
157
|
+
action,
|
|
158
|
+
})
|
|
159
|
+
const kept = [
|
|
160
|
+
finding({ id: 'm', risk: risk('must-fix') }),
|
|
161
|
+
finding({ id: 's', risk: risk('should-fix') }),
|
|
162
|
+
finding({ id: 'c', risk: risk('consider') }),
|
|
163
|
+
finding({ id: 'o', risk: risk('optional') }),
|
|
164
|
+
]
|
|
165
|
+
expect(actionableIds(kept)).toEqual(['m', 's'])
|
|
166
|
+
})
|
|
167
|
+
|
|
168
|
+
test('a kept id missing from the findings is an error naming it', () => {
|
|
169
|
+
expect(() =>
|
|
170
|
+
scoreSelection({ labels: fixtureLabels, keptIds: ['ghost'], findings: fixtureFindings }),
|
|
171
|
+
).toThrow('ghost')
|
|
172
|
+
})
|
|
173
|
+
|
|
174
|
+
const labelled = finding({
|
|
175
|
+
id: 'L',
|
|
176
|
+
file: 'src/orders.ts',
|
|
177
|
+
line: 10,
|
|
178
|
+
title: 'Unbounded retry loop in fetchOrders handler',
|
|
179
|
+
})
|
|
180
|
+
|
|
181
|
+
test('a 4-line offset with a similar title matches', () => {
|
|
182
|
+
const moved = finding({
|
|
183
|
+
id: 'N',
|
|
184
|
+
line: 14,
|
|
185
|
+
title: 'Unbounded retry loop when fetchOrders fails',
|
|
186
|
+
})
|
|
187
|
+
expect(matchFindings([moved], [labelled])).toEqual(new Map([['N', 'L']]))
|
|
188
|
+
})
|
|
189
|
+
|
|
190
|
+
test('a 6-line offset does not match', () => {
|
|
191
|
+
const moved = finding({ id: 'N', line: 16, title: labelled.title })
|
|
192
|
+
expect(matchFindings([moved], [labelled]).size).toBe(0)
|
|
193
|
+
})
|
|
194
|
+
|
|
195
|
+
test('a different file does not match', () => {
|
|
196
|
+
const elsewhere = finding({ id: 'N', file: 'src/billing.ts', title: labelled.title })
|
|
197
|
+
expect(matchFindings([elsewhere], [labelled]).size).toBe(0)
|
|
198
|
+
})
|
|
199
|
+
|
|
200
|
+
test('two candidates for one labelled finding resolve to the higher dice', () => {
|
|
201
|
+
const weaker = finding({ id: 'B', line: 11, title: 'Retry loop missing backoff' })
|
|
202
|
+
const stronger = finding({ id: 'A', line: 12, title: labelled.title })
|
|
203
|
+
expect(matchFindings([weaker, stronger], [labelled])).toEqual(new Map([['A', 'L']]))
|
|
204
|
+
})
|
|
205
|
+
|
|
206
|
+
test('a replayed finding that matches a posted one scores as posted kept', () => {
|
|
207
|
+
const moved = finding({ id: 'bugs-9', line: 12, title: labelled.title })
|
|
208
|
+
const { score } = scoreReplay({
|
|
209
|
+
labels: [{ id: 'L', label: 'posted' }],
|
|
210
|
+
labelled: [labelled],
|
|
211
|
+
newKept: [moved],
|
|
212
|
+
})
|
|
213
|
+
expect([score.postedKept, score.precision]).toEqual([1, 1])
|
|
214
|
+
})
|
|
215
|
+
|
|
216
|
+
test('an unmatched replayed finding is unlabelled even when its id repeats a labelled id', () => {
|
|
217
|
+
const unrelated = finding({ id: 'L', file: 'src/billing.ts', title: 'Totally different' })
|
|
218
|
+
const { score, unlabelled } = scoreReplay({
|
|
219
|
+
labels: [{ id: 'L', label: 'posted' }],
|
|
220
|
+
labelled: [labelled],
|
|
221
|
+
newKept: [unrelated],
|
|
222
|
+
})
|
|
223
|
+
expect([unlabelled, score.postedKept, score.precision]).toEqual([1, 0, null])
|
|
224
|
+
})
|
|
225
|
+
|
|
226
|
+
let root: string
|
|
227
|
+
|
|
228
|
+
beforeEach(async () => {
|
|
229
|
+
root = await mkdtemp(join(tmpdir(), 'magpie-quality-'))
|
|
230
|
+
})
|
|
231
|
+
|
|
232
|
+
afterEach(async () => {
|
|
233
|
+
await rm(root, { recursive: true, force: true })
|
|
234
|
+
})
|
|
235
|
+
|
|
236
|
+
async function writeRun(dir: string, files: Record<string, string>): Promise<void> {
|
|
237
|
+
await mkdir(dir, { recursive: true })
|
|
238
|
+
for (const [name, content] of Object.entries(files)) {
|
|
239
|
+
await mkdir(join(dir, name, '..'), { recursive: true })
|
|
240
|
+
await writeFile(join(dir, name), content)
|
|
241
|
+
}
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
const postedRun = {
|
|
245
|
+
'post-status.json': '{"a":"posted"}',
|
|
246
|
+
'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
test('an archived run with post status and final findings is a source run', async () => {
|
|
250
|
+
await writeRun(join(root, 'pr-50-1.archived-2'), postedRun)
|
|
251
|
+
expect(await listSourceRuns(root)).toEqual(['pr-50-1.archived-2'])
|
|
252
|
+
})
|
|
253
|
+
|
|
254
|
+
test('a preview run is not a source run', async () => {
|
|
255
|
+
await writeRun(join(root, 'preview-123'), postedRun)
|
|
256
|
+
expect(await listSourceRuns(root)).toEqual([])
|
|
257
|
+
})
|
|
258
|
+
|
|
259
|
+
test('a run that never posted is not a source run', async () => {
|
|
260
|
+
await writeRun(join(root, 'pr-7-1'), {
|
|
261
|
+
'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
|
|
262
|
+
})
|
|
263
|
+
expect(await listSourceRuns(root)).toEqual([])
|
|
264
|
+
})
|
|
265
|
+
|
|
266
|
+
test('a corpus dir inside the repo is refused', () => {
|
|
267
|
+
expect(() => assertOutsideRepo('/work/repo/corpus', '/work/repo')).toThrow('/work/repo')
|
|
268
|
+
})
|
|
269
|
+
|
|
270
|
+
test('a corpus dir beside the repo with a shared name prefix is allowed', () => {
|
|
271
|
+
expect(() => assertOutsideRepo('/work/repo-corpus', '/work/repo')).not.toThrow()
|
|
272
|
+
})
|
|
273
|
+
|
|
274
|
+
test('a kept id absent from final findings is labelled ignored in the corpus', async () => {
|
|
275
|
+
const src = join(root, 'pr-9-1')
|
|
276
|
+
await writeRun(src, {
|
|
277
|
+
...postedRun,
|
|
278
|
+
'findings.kept.json': JSON.stringify([finding({ id: 'a' }), finding({ id: 'b' })]),
|
|
279
|
+
})
|
|
280
|
+
await buildCorpusRun(src, join(root, 'corpus', 'pr-9-1'))
|
|
281
|
+
const labels = JSON.parse(await readFile(join(root, 'corpus', 'pr-9-1', 'labels.json'), 'utf8'))
|
|
282
|
+
expect(labels).toEqual([
|
|
283
|
+
{ id: 'a', label: 'posted' },
|
|
284
|
+
{ id: 'b', label: 'ignored' },
|
|
285
|
+
])
|
|
286
|
+
})
|
|
287
|
+
|
|
288
|
+
test('a corpus run without merge candidates loads them as absent', async () => {
|
|
289
|
+
const dir = join(root, 'pr-9-1')
|
|
290
|
+
await writeRun(dir, {
|
|
291
|
+
'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
|
|
292
|
+
'labels.json': JSON.stringify([{ id: 'a', label: 'ignored' }]),
|
|
293
|
+
})
|
|
294
|
+
expect((await loadCorpusRun(dir)).mergeCandidates).toBeNull()
|
|
295
|
+
})
|
|
296
|
+
|
|
297
|
+
test('a corpus file that fails to parse is an error naming the file', async () => {
|
|
298
|
+
const dir = join(root, 'pr-9-1')
|
|
299
|
+
await writeRun(dir, {
|
|
300
|
+
'findings.final.json': JSON.stringify([finding({ id: 'a' })]),
|
|
301
|
+
'labels.json': JSON.stringify([{ id: 'a', label: 'ignored' }]),
|
|
302
|
+
'merge-candidates.json': '{not json',
|
|
303
|
+
})
|
|
304
|
+
await expect(loadCorpusRun(dir)).rejects.toThrow(join(dir, 'merge-candidates.json'))
|
|
305
|
+
})
|
|
306
|
+
|
|
307
|
+
test('a claude result marked as an error is an error naming the step', () => {
|
|
308
|
+
const stdout = JSON.stringify({
|
|
309
|
+
type: 'result',
|
|
310
|
+
subtype: 'error_max_budget_usd',
|
|
311
|
+
is_error: true,
|
|
312
|
+
result: 'budget exceeded',
|
|
313
|
+
})
|
|
314
|
+
expect(() => parseClaudeResult(stdout, 'critic batch 2')).toThrow('critic batch 2')
|
|
315
|
+
})
|
|
316
|
+
|
|
317
|
+
test('a successful claude result reports its cost', () => {
|
|
318
|
+
const stdout = JSON.stringify({
|
|
319
|
+
type: 'result',
|
|
320
|
+
subtype: 'success',
|
|
321
|
+
is_error: false,
|
|
322
|
+
total_cost_usd: 0.42,
|
|
323
|
+
})
|
|
324
|
+
expect(parseClaudeResult(stdout, 'critic batch 1')).toEqual({ costUsd: 0.42 })
|
|
325
|
+
})
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
#!/usr/bin/env bun
|
|
2
|
+
// Scores each corpus run's original critic selection (findings.kept.json)
|
|
3
|
+
// against its labels: the number a replayed critic has to beat.
|
|
4
|
+
|
|
5
|
+
import { mkdir, writeFile } from 'node:fs/promises'
|
|
6
|
+
import { join } from 'node:path'
|
|
7
|
+
import { type CorpusRun, listCorpusRuns, loadCorpusRun, resolveCorpusRoot } from './corpus.ts'
|
|
8
|
+
import {
|
|
9
|
+
formatRatio,
|
|
10
|
+
parseFlags,
|
|
11
|
+
poolRuns,
|
|
12
|
+
type QualityScore,
|
|
13
|
+
resultTimestamp,
|
|
14
|
+
scoreSelection,
|
|
15
|
+
} from './score.ts'
|
|
16
|
+
|
|
17
|
+
/** The critic chose from findings.deduped.json, so that is the pool recall counts against. */
|
|
18
|
+
export function originalSelection(run: CorpusRun) {
|
|
19
|
+
if (!run.kept) throw new Error(`${run.dir} has no findings.kept.json`)
|
|
20
|
+
if (!run.deduped) throw new Error(`${run.dir} has no findings.deduped.json`)
|
|
21
|
+
return {
|
|
22
|
+
runId: run.runId,
|
|
23
|
+
labels: run.labels,
|
|
24
|
+
keptIds: run.kept.map((f) => f.id),
|
|
25
|
+
findings: run.deduped,
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
const DISMISS_COLUMNS = ['wrong', 'not-worth-it', 'duplicate', 'style']
|
|
30
|
+
|
|
31
|
+
function row(name: string, s: QualityScore): string[] {
|
|
32
|
+
return [
|
|
33
|
+
name,
|
|
34
|
+
String(s.kept),
|
|
35
|
+
String(s.posted),
|
|
36
|
+
String(s.postedKept),
|
|
37
|
+
formatRatio(s.precision),
|
|
38
|
+
formatRatio(s.recall),
|
|
39
|
+
...DISMISS_COLUMNS.map((r) => String(s.dismissedKept[r] ?? 0)),
|
|
40
|
+
]
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function renderTable(rows: string[][]): string {
|
|
44
|
+
const header = ['run', 'kept', 'posted', 'postedKept', 'precision', 'recall', ...DISMISS_COLUMNS]
|
|
45
|
+
const all = [header, ...rows]
|
|
46
|
+
const widths = header.map((_, i) => Math.max(...all.map((r) => (r[i] ?? '').length)))
|
|
47
|
+
return all.map((r) => r.map((cell, i) => cell.padEnd(widths[i] ?? 0)).join(' ')).join('\n')
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
async function main(argv: string[]): Promise<number> {
|
|
51
|
+
const corpusRoot = resolveCorpusRoot(parseFlags(argv, ['out']))
|
|
52
|
+
const selections = []
|
|
53
|
+
for (const runId of await listCorpusRuns(corpusRoot)) {
|
|
54
|
+
selections.push(originalSelection(await loadCorpusRun(join(corpusRoot, runId))))
|
|
55
|
+
}
|
|
56
|
+
if (selections.length === 0) throw new Error(`no corpus runs in ${corpusRoot}; run corpus.ts`)
|
|
57
|
+
|
|
58
|
+
const runs = selections.map((s) => ({ runId: s.runId, score: scoreSelection(s) }))
|
|
59
|
+
const total = scoreSelection(poolRuns(selections))
|
|
60
|
+
process.stdout.write(
|
|
61
|
+
`${renderTable([...runs.map((r) => row(r.runId, r.score)), row('TOTAL', total)])}\n`,
|
|
62
|
+
)
|
|
63
|
+
|
|
64
|
+
const resultsDir = join(corpusRoot, 'results')
|
|
65
|
+
await mkdir(resultsDir, { recursive: true })
|
|
66
|
+
const outPath = join(resultsDir, `${resultTimestamp(new Date())}-baseline.json`)
|
|
67
|
+
await writeFile(outPath, `${JSON.stringify({ tier: 'baseline', runs, total }, null, 2)}\n`)
|
|
68
|
+
process.stdout.write(`wrote ${outPath}\n`)
|
|
69
|
+
return 0
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
if (import.meta.main) {
|
|
73
|
+
main(process.argv.slice(2)).then(
|
|
74
|
+
(code) => process.exit(code),
|
|
75
|
+
(err) => {
|
|
76
|
+
process.stderr.write(`baseline: ${err instanceof Error ? err.message : String(err)}\n`)
|
|
77
|
+
process.exit(1)
|
|
78
|
+
},
|
|
79
|
+
)
|
|
80
|
+
}
|