control-arm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +82 -0
- package/DESIGN.md +161 -0
- package/LICENSE +21 -0
- package/README.md +164 -0
- package/action.yml +95 -0
- package/bin/ca.mjs +245 -0
- package/fixtures/build.mjs +178 -0
- package/package.json +12 -0
- package/scripts/gh-api.mjs +90 -0
- package/scripts-analyze.mjs +86 -0
- package/scripts-recompute.mjs +53 -0
- package/src/assertions.mjs +350 -0
- package/src/html-report.mjs +125 -0
- package/src/identity.mjs +105 -0
- package/src/markdown-report.mjs +81 -0
- package/src/report.mjs +183 -0
- package/src/runner-json.mjs +127 -0
- package/src/runner.mjs +83 -0
- package/src/select-runner.mjs +58 -0
- package/src/tap.mjs +83 -0
- package/src/verdict.mjs +168 -0
- package/src/verify.mjs +322 -0
- package/src/worktree.mjs +197 -0
- package/test/assertions.test.mjs +130 -0
- package/test/fixtures.test.mjs +41 -0
- package/test/runner-json.test.mjs +69 -0
- package/test/select-runner.test.mjs +78 -0
- package/test/verdict.test.mjs +113 -0
package/src/verdict.mjs
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The decision. Pure functions only — no git, no filesystem, no subprocesses.
|
|
3
|
+
*
|
|
4
|
+
* This file is deliberately the only place a verdict is decided, and deliberately
|
|
5
|
+
* has no I/O, because it is the part that has to be exhaustively tested. Everything
|
|
6
|
+
* else in this tool is plumbing that produces its two inputs.
|
|
7
|
+
*
|
|
8
|
+
* THE FIVE VERDICTS
|
|
9
|
+
*
|
|
10
|
+
* CAUGHT the case ran on the broken code and DISAGREED with it (assertion)
|
|
11
|
+
* NON-DISCRIMINATING the case ran and was fine with it (case level — a fact, not a fault)
|
|
12
|
+
* BLIND NO case in the commit discriminated (commit level — a judgement)
|
|
13
|
+
* INCONCLUSIVE the case did not run, or we cannot prove what it ran against
|
|
14
|
+
* FLAKY repeated runs disagreed with each other
|
|
15
|
+
* SKIPPED there was nothing to judge
|
|
16
|
+
*
|
|
17
|
+
* WHY EXIT CODE IS NOT THE SIGNAL. A test that "fails" at the parent commit with
|
|
18
|
+
* `SyntaxError: does not provide an export named 'TITLE_SEPARATOR'` — because the fix
|
|
19
|
+
* ADDED that export — has a non-zero exit and has told you nothing: it never ran. Read
|
|
20
|
+
* exit code only and you report CAUGHT, and the tool's headline number is inflated
|
|
21
|
+
* garbage. Observed on a real fix commit whose new export the parent did not have.
|
|
22
|
+
*
|
|
23
|
+
* WHY A FALSE `BLIND` IS THE WORST OUTPUT. CAUGHT and INCONCLUSIVE are both survivable
|
|
24
|
+
* — one is good news, the other is an honest shrug. BLIND accuses an engineer of having
|
|
25
|
+
* written a test that cannot fail. Get that wrong and nobody trusts the tool again. So
|
|
26
|
+
* every ambiguity resolves AWAY from BLIND, never toward it.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
export const CAUGHT = 'CAUGHT';
|
|
30
|
+
/**
|
|
31
|
+
* CASE level. The case ran on the broken code and was fine with it.
|
|
32
|
+
*
|
|
33
|
+
* NOT a synonym for "bad test", and renamed from BLIND on 2026-09-23 after the first real
|
|
34
|
+
* run said this about a commit whose fix added a separator class:
|
|
35
|
+
*
|
|
36
|
+
* ✗ BLIND CONTROL ARM: the old pattern really does split one phrase in two
|
|
37
|
+
* ✗ BLIND the prefix rule survives the wider separator class
|
|
38
|
+
* ✗ BLIND the labels the separator change must not touch
|
|
39
|
+
*
|
|
40
|
+
* All three are REGRESSION GUARDS. They are supposed to be green on both arms — that is
|
|
41
|
+
* their entire job. Calling them blind is a false accusation of a careful engineer, which
|
|
42
|
+
* is the one output that destroys trust in this tool.
|
|
43
|
+
*
|
|
44
|
+
* Red/green cannot distinguish "meant to catch this bug and failed" from "meant to stay
|
|
45
|
+
* green". That is INTENT, and the tool cannot read it. So the case level states the FACT
|
|
46
|
+
* (it did not discriminate) and the judgement is made only where it is safe: at the
|
|
47
|
+
* commit, where a fix in which NOTHING discriminates shipped no test that could catch its
|
|
48
|
+
* own bug, whatever each case was for.
|
|
49
|
+
*/
|
|
50
|
+
export const NON_DISCRIMINATING = 'NON-DISCRIMINATING';
|
|
51
|
+
/** COMMIT level only: every case ran, and not one of them discriminated. */
|
|
52
|
+
export const BLIND = 'BLIND';
|
|
53
|
+
export const INCONCLUSIVE = 'INCONCLUSIVE';
|
|
54
|
+
export const FLAKY = 'FLAKY';
|
|
55
|
+
export const SKIPPED = 'SKIPPED';
|
|
56
|
+
|
|
57
|
+
/**
|
|
58
|
+
* Did this failure come from the test DISAGREEING, or from it being unable to run?
|
|
59
|
+
*
|
|
60
|
+
* node:test TAP puts `name: 'AssertionError'` / `code: 'ERR_ASSERTION'` on the first and
|
|
61
|
+
* a real error name (ReferenceError, SyntaxError, TypeError) on the second. Other runners
|
|
62
|
+
* differ; `runners/*.mjs` maps them onto this same shape, which is why this predicate
|
|
63
|
+
* takes a normalised case and not raw output.
|
|
64
|
+
*/
|
|
65
|
+
export function isDisagreement(caseResult) {
|
|
66
|
+
if (!caseResult || caseResult.status !== 'fail') return false;
|
|
67
|
+
if (caseResult.code === 'ERR_ASSERTION') return true;
|
|
68
|
+
if (caseResult.errorName === 'AssertionError') return true;
|
|
69
|
+
// Assertion libraries that are not node:assert. Narrow on purpose: an unknown error
|
|
70
|
+
// name must fall through to INCONCLUSIVE, not be guessed into CAUGHT.
|
|
71
|
+
return ['JestAssertionError', 'ExpectationFailed', 'AssertionFailedError'].includes(caseResult.errorName);
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/**
|
|
75
|
+
* One case, one run. `armA` is the case on the FIXED code, `armB` the same case
|
|
76
|
+
* transplanted onto the BROKEN (parent) code.
|
|
77
|
+
*
|
|
78
|
+
* `identity` is the proof that armB actually loaded the parent's modules. It is a hard
|
|
79
|
+
* gate before anything else, because without it a BLIND is indistinguishable from a
|
|
80
|
+
* harness that silently resolved the fixed code — which is exactly what happened on the
|
|
81
|
+
* first real run of this tool.
|
|
82
|
+
*/
|
|
83
|
+
export function classify({ armA, armB, identity }) {
|
|
84
|
+
if (identity && identity.proven === false) {
|
|
85
|
+
return { verdict: INCONCLUSIVE, reason: `module identity unproven: ${identity.reason}` };
|
|
86
|
+
}
|
|
87
|
+
if (!armA) return { verdict: SKIPPED, reason: 'case not present on the fix' };
|
|
88
|
+
|
|
89
|
+
// A case the runner SKIPPED tells us nothing and is not a failure of anything. It gets
|
|
90
|
+
// its own bucket rather than inflating INCONCLUSIVE — 257 of 1,185 cases in the first
|
|
91
|
+
// 300-commit audit were `skip`, almost all DB-gated tests with no TEST_DATABASE_URL.
|
|
92
|
+
// Folding those into "could not be judged" hides the fact that they are judgeable, by
|
|
93
|
+
// anyone who runs the audit with a database.
|
|
94
|
+
if (armA.status === 'skip') {
|
|
95
|
+
return { verdict: SKIPPED, reason: 'the runner skipped this case on the fix (gated on an env var or a service?)' };
|
|
96
|
+
}
|
|
97
|
+
// The fix must be green, or there is no "before and after" to compare. A red case on
|
|
98
|
+
// the fix means the commit does not stand on its own — not that the test is bad.
|
|
99
|
+
if (armA.status !== 'pass') {
|
|
100
|
+
return { verdict: INCONCLUSIVE, reason: `case is not green on the fix (${armA.errorName || armA.status})` };
|
|
101
|
+
}
|
|
102
|
+
// The case exists on the fix but never reported on the parent: the file failed to
|
|
103
|
+
// load, or the runner died before reaching it.
|
|
104
|
+
if (!armB) {
|
|
105
|
+
return { verdict: INCONCLUSIVE, reason: 'case did not report on the parent (file failed to load?)' };
|
|
106
|
+
}
|
|
107
|
+
if (armB.status === 'pass') {
|
|
108
|
+
return { verdict: NON_DISCRIMINATING, reason: 'green on the broken code (a regression guard looks identical here)' };
|
|
109
|
+
}
|
|
110
|
+
if (isDisagreement(armB)) {
|
|
111
|
+
return { verdict: CAUGHT, reason: armB.message || 'assertion failed on the broken code' };
|
|
112
|
+
}
|
|
113
|
+
return {
|
|
114
|
+
verdict: INCONCLUSIVE,
|
|
115
|
+
reason: `case errored rather than failed on the parent (${armB.errorName || armB.code || 'unknown'}) — it never ran`,
|
|
116
|
+
};
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* N runs of the same case. Disagreement between runs is its own verdict: a case that is
|
|
121
|
+
* CAUGHT twice and BLIND once has told you nothing about the fix and something important
|
|
122
|
+
* about the test.
|
|
123
|
+
*
|
|
124
|
+
* INCONCLUSIVE does not outvote a real result — an infra hiccup on run 2 should not erase
|
|
125
|
+
* a clean CAUGHT from runs 1 and 3 — but a CAUGHT/BLIND split is FLAKY, always.
|
|
126
|
+
*/
|
|
127
|
+
export function reduceRuns(results) {
|
|
128
|
+
if (results.length === 0) return { verdict: SKIPPED, reason: 'no runs' };
|
|
129
|
+
const decided = results.filter(r => r.verdict === CAUGHT || r.verdict === NON_DISCRIMINATING);
|
|
130
|
+
if (decided.length === 0) return results[0];
|
|
131
|
+
|
|
132
|
+
const distinct = new Set(decided.map(r => r.verdict));
|
|
133
|
+
if (distinct.size > 1) {
|
|
134
|
+
const tally = decided.map(r => r.verdict).join(', ');
|
|
135
|
+
return { verdict: FLAKY, reason: `runs disagreed: ${tally}` };
|
|
136
|
+
}
|
|
137
|
+
if (decided.length < results.length) {
|
|
138
|
+
const d = decided[0];
|
|
139
|
+
return { ...d, reason: `${d.reason} (${results.length - decided.length} of ${results.length} runs inconclusive)` };
|
|
140
|
+
}
|
|
141
|
+
return decided[0];
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
/**
|
|
145
|
+
* Commit-level roll-up. A commit is CAUGHT if ANY of its cases discriminates.
|
|
146
|
+
*
|
|
147
|
+
* BLIND requires that EVERY case actually RAN. A commit with one skipped case and three
|
|
148
|
+
* non-discriminating ones is INCONCLUSIVE, not BLIND — the skipped case might have been
|
|
149
|
+
* the discriminating one, and nothing here can know.
|
|
150
|
+
*
|
|
151
|
+
* Earned on a commit fixing a null-limit bug in a metering module. All four
|
|
152
|
+
* tests it shipped are `{ skip: SKIP }`, gated on TEST_DATABASE_URL, which the audit host
|
|
153
|
+
* did not set. The older cases in the same file ran and did not discriminate, so an
|
|
154
|
+
* earlier version of this function called the commit BLIND — on the strength of the tests
|
|
155
|
+
* that were NOT written for the bug, while the four that were sat unexecuted. Run the
|
|
156
|
+
* audit with a database and the same commit may well be CAUGHT.
|
|
157
|
+
*
|
|
158
|
+
* Same principle as everywhere else here: ambiguity resolves AWAY from BLIND.
|
|
159
|
+
*/
|
|
160
|
+
export function rollUp(caseVerdicts) {
|
|
161
|
+
if (caseVerdicts.length === 0) return SKIPPED;
|
|
162
|
+
const has = v => caseVerdicts.some(c => c.verdict === v);
|
|
163
|
+
if (has(CAUGHT)) return CAUGHT;
|
|
164
|
+
if (has(FLAKY)) return FLAKY;
|
|
165
|
+
if (has(INCONCLUSIVE) || has(SKIPPED)) return INCONCLUSIVE;
|
|
166
|
+
if (caseVerdicts.every(c => c.verdict === NON_DISCRIMINATING)) return BLIND;
|
|
167
|
+
return INCONCLUSIVE;
|
|
168
|
+
}
|
package/src/verify.mjs
ADDED
|
@@ -0,0 +1,322 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Orchestration: two arms, one commit.
|
|
3
|
+
*
|
|
4
|
+
* ARM A worktree @ FIX + the fix's own test -> must be GREEN, or there is no
|
|
5
|
+
* before/after to compare
|
|
6
|
+
* ARM B worktree @ PARENT + the fix's test TRANSPLANTED onto the old source
|
|
7
|
+
*
|
|
8
|
+
* The transplant is the whole idea. You do NOT run the parent's tests — the parent does
|
|
9
|
+
* not have this test. You take the new test and ask it about the old code.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import { git, ensureWorktree, linkDependencies, linkEnvFiles, transplant } from './worktree.mjs';
|
|
14
|
+
import { proveIdentity } from './identity.mjs';
|
|
15
|
+
import { nodeTest } from './runner.mjs';
|
|
16
|
+
import { analyseCase } from './assertions.mjs';
|
|
17
|
+
import { vitest, jest } from './runner-json.mjs';
|
|
18
|
+
import { selectRunner } from './select-runner.mjs';
|
|
19
|
+
|
|
20
|
+
const RUNNERS = { node: nodeTest, vitest, jest };
|
|
21
|
+
import { classify, reduceRuns, rollUp, isDisagreement, BLIND, NON_DISCRIMINATING, SKIPPED, INCONCLUSIVE } from './verdict.mjs';
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* Documentation, and nothing else, is "not source".
|
|
25
|
+
*
|
|
26
|
+
* This used to be an ALLOWLIST of `.js/.ts/.tsx/.mjs`, and it made the tool decline the
|
|
27
|
+
* one kind of change it should be best at judging. A real PR wired a mutation-testing
|
|
28
|
+
* gate: it changed a tool config and a CI workflow, and shipped a test asserting that
|
|
29
|
+
* wiring. Zero JS changed, so the commit read as "test-only — no source change to be
|
|
30
|
+
* blind to" and got no verdict.
|
|
31
|
+
*
|
|
32
|
+
* That is backwards. A guard that is installed but wired to nothing is the dominant defect
|
|
33
|
+
* class in many repos — three instances found in one afternoon — and the test proving a
|
|
34
|
+
* gate is armed is precisely a test that should be shown to fail without the wiring.
|
|
35
|
+
* Config, CI YAML and shell ARE the source for those.
|
|
36
|
+
*
|
|
37
|
+
* So: deny-list the docs, accept the rest. A commit that changes only Markdown genuinely
|
|
38
|
+
* has nothing to be blind to; everything else might.
|
|
39
|
+
*/
|
|
40
|
+
const DOC_RE = /\.(md|mdx|txt|rst|adoc)$/i;
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* Files no unit test can exercise, however good the suite is.
|
|
44
|
+
*
|
|
45
|
+
* A `fix(ios):` commit repairing an aborted CocoaPods install came back as a still-open
|
|
46
|
+
* BLIND. It is not a finding. A dependency-resolution failure is caught by a BUILD, and
|
|
47
|
+
* asking whether a unit test would have caught it is a category error. Reporting it as a gap teaches the reader that the tool does not know
|
|
48
|
+
* what a test is for.
|
|
49
|
+
*
|
|
50
|
+
* Deliberately narrow: only manifests and project files for toolchains that build rather
|
|
51
|
+
* than run. A .json or .yml can absolutely be under test — a CI-wiring test asserts a
|
|
52
|
+
* workflow file — so those are NOT here.
|
|
53
|
+
*/
|
|
54
|
+
const BUILD_ONLY_RE = /(^|\/)(Podfile(\.lock)?|Gemfile(\.lock)?|Cartfile.*|package-lock\.json|yarn\.lock|pnpm-lock\.yaml|.*\.pbxproj|.*\.xcworkspacedata|.*\.xcscheme|.*\.gradle(\.kts)?|gradle\.properties|.*\.plist|.*\.podspec|.*\.lock)$/i;
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* KNOWN LIMIT, stated rather than papered over: build-time JAVASCRIPT is not detectable
|
|
58
|
+
* by filename. One such commit changed `apps/mobile/plugins/withModularHeaders.js` — an
|
|
59
|
+
* Expo config plugin that runs at BUILD time, spelled exactly like runtime source. Webpack/vite/rollup configs and codegen
|
|
60
|
+
* scripts are the same shape.
|
|
61
|
+
*
|
|
62
|
+
* A heuristic wide enough to catch those (path contains "plugins", "scripts", "config")
|
|
63
|
+
* would also exclude real application code, and a FALSE EXCLUSION is worse than a false
|
|
64
|
+
* inclusion here: it hides a finding silently, where a category error is at least visible
|
|
65
|
+
* and arguable. So these reach a verdict and need human triage. The audit's subject
|
|
66
|
+
* filter catches most of them in practice, because they are usually scoped fix(ios),
|
|
67
|
+
* fix(build) or similar.
|
|
68
|
+
*/
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* What counts as a test file.
|
|
72
|
+
*
|
|
73
|
+
* TWO independent signals, because projects pick one or the other and a tool that demands
|
|
74
|
+
* both measures nothing:
|
|
75
|
+
*
|
|
76
|
+
* 1. a `.test.` / `.spec.` suffix anywhere (most application repos)
|
|
77
|
+
* 2. living under a test directory (undici, node core, most library repos)
|
|
78
|
+
*
|
|
79
|
+
* This used to require BOTH — a file under `tests/` AND a `.test.` suffix. Run against
|
|
80
|
+
* nodejs/undici, whose tests are `test/client-request.js` with no suffix, the tool
|
|
81
|
+
* reported "284 commits matched, 0 ship both a test and a source change" and produced an
|
|
82
|
+
* entirely empty audit. Not a wrong answer: NO answer, on a repo with 261 fix commits
|
|
83
|
+
* that ship tests.
|
|
84
|
+
*
|
|
85
|
+
* Found the first time it was pointed at a codebase its author did not write, which is
|
|
86
|
+
* the whole argument for doing that before believing any number it prints.
|
|
87
|
+
*/
|
|
88
|
+
const TEST_DIR_RE = /(^|\/)(tests?|__tests__|spec|specs)\//i;
|
|
89
|
+
const TEST_SUFFIX_RE = /\.(test|spec)\.(m?[jt]sx?)$/i;
|
|
90
|
+
const CODE_RE = /\.(m?[jt]sx?)$/i;
|
|
91
|
+
const TEST_RE = (f) => CODE_RE.test(f) && (TEST_SUFFIX_RE.test(f) || TEST_DIR_RE.test(f));
|
|
92
|
+
|
|
93
|
+
/**
|
|
94
|
+
* IS THIS A REPAIR, OR IS IT NEW CODE? — and why the answer changes what CAUGHT means.
|
|
95
|
+
*
|
|
96
|
+
* On a BUG FIX the base is the broken code, so a test that fails there genuinely
|
|
97
|
+
* discriminates: it would have caught the bug. That is the case this tool was built for.
|
|
98
|
+
*
|
|
99
|
+
* On a FEATURE the base is code where the thing does not exist yet. Essentially ANY test
|
|
100
|
+
* touching the new code fails there — the import is missing, the field is absent, a count
|
|
101
|
+
* is zero. `expected 0 to be greater than 0` is a real AssertionError, correctly
|
|
102
|
+
* classified, and says nothing whatever about whether the test is well aimed. A test
|
|
103
|
+
* asserting `expect(1).toBe(1)` in the same file would NOT have been CAUGHT; essentially
|
|
104
|
+
* any test reading the new field would. The signal comes from the field's existence, not
|
|
105
|
+
* from the test's design.
|
|
106
|
+
*
|
|
107
|
+
* This is the mirror of the INCONCLUSIVE rule. That one says a red test that never RAN
|
|
108
|
+
* proves nothing; this one says a red test that ran against ABSENT CODE proves nearly as
|
|
109
|
+
* little. Counting them together lets the headline drift upward for free on a
|
|
110
|
+
* feature-heavy week, and a number that drifts for free is one people stop reading.
|
|
111
|
+
*
|
|
112
|
+
* MEASURED, because both signals are proxies and only one is any good:
|
|
113
|
+
* conventional prefix 1,213 fix / 988 feat in one real corpus — used consistently
|
|
114
|
+
* source additions-only 9 of 28 feat commits, but also 1 of 35 fix commits —
|
|
115
|
+
* specific, not sensitive; a supporting hint, never the verdict
|
|
116
|
+
*
|
|
117
|
+
* So this labels the CLAIM and never silently reclassifies a verdict. CAUGHT on a feature
|
|
118
|
+
* is still CAUGHT — it just does not get to say "this would have caught the bug", because
|
|
119
|
+
* there was no bug.
|
|
120
|
+
*/
|
|
121
|
+
export function commitKind(subject, sourceAddedOnly) {
|
|
122
|
+
const m = String(subject).match(/^\s*([a-z]+)\s*(\([^)]*\))?\s*!?:/i);
|
|
123
|
+
const prefix = m ? m[1].toLowerCase() : null;
|
|
124
|
+
if (prefix === 'fix' || prefix === 'bug' || prefix === 'perf') {
|
|
125
|
+
// A fix that only ADDED source is worth flagging: its test may be reading
|
|
126
|
+
// something that simply was not there, exactly like a feature's.
|
|
127
|
+
return { kind: 'fix', newCode: !!sourceAddedOnly, prefix };
|
|
128
|
+
}
|
|
129
|
+
if (prefix === 'feat' || prefix === 'feature') return { kind: 'feature', newCode: true, prefix };
|
|
130
|
+
if (prefix) return { kind: 'other', newCode: !!sourceAddedOnly, prefix };
|
|
131
|
+
return { kind: 'unknown', newCode: !!sourceAddedOnly, prefix: null };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/**
|
|
135
|
+
* `withDiffStat` is OFF by default, and that default is load-bearing.
|
|
136
|
+
*
|
|
137
|
+
* The additions-only signal needs `git show --numstat` — one subprocess per commit. The
|
|
138
|
+
* audit calls commitInfo on EVERY candidate just to test eligibility (1,178 of them on one
|
|
139
|
+
* real corpus), so computing it unconditionally pushed the selection pass past 70 minutes
|
|
140
|
+
* BEFORE a single commit was judged. Measured, not estimated: that run was killed at 72.
|
|
141
|
+
*
|
|
142
|
+
* Only the commits actually verified need it, so verifyCommit asks and the eligibility
|
|
143
|
+
* pass does not. A regression introduced by the fix for issue #3, and caught by noticing a
|
|
144
|
+
* run sit in "selecting commits" for over an hour rather than by any test.
|
|
145
|
+
*/
|
|
146
|
+
export async function commitInfo(repo, sha, against = null, withDiffStat = false) {
|
|
147
|
+
const out = await git(repo, ['show', '--no-patch', '--format=%H%n%s%n%ad', '--date=short', sha]);
|
|
148
|
+
const [full, subject, date] = out.trim().split('\n');
|
|
149
|
+
// With a base, the changed set is the WHOLE branch, not just the tip commit — a PR's
|
|
150
|
+
// test may have arrived in commit 1 and its source change in commit 3.
|
|
151
|
+
const files = against
|
|
152
|
+
? (await git(repo, ['diff', '--name-only', `${(await git(repo, ['merge-base', against, sha])).trim()}...${sha}`])).trim().split('\n').filter(Boolean)
|
|
153
|
+
: (await git(repo, ['show', '--name-only', '--format=', sha])).trim().split('\n').filter(Boolean);
|
|
154
|
+
// Deletions in non-test source: a repair usually changes lines, new code only adds.
|
|
155
|
+
const numstat = !withDiffStat ? '' : against
|
|
156
|
+
? await git(repo, ['diff', '--numstat', `${(await git(repo, ['merge-base', against, sha])).trim()}...${sha}`])
|
|
157
|
+
: await git(repo, ['show', '--numstat', '--format=', sha]);
|
|
158
|
+
let srcDeletions = 0;
|
|
159
|
+
for (const line of numstat.trim().split('\n')) {
|
|
160
|
+
const [, del, file] = line.split(/\t/).length === 3 ? ['', ...line.split(/\t/).slice(1)] : [];
|
|
161
|
+
const parts = line.split(/\t/);
|
|
162
|
+
if (parts.length !== 3) continue;
|
|
163
|
+
const [, d, f] = parts;
|
|
164
|
+
if (TEST_RE(f) || DOC_RE.test(f)) continue;
|
|
165
|
+
if (/^\d+$/.test(d)) srcDeletions += Number(d);
|
|
166
|
+
}
|
|
167
|
+
const kindInfo = commitKind(subject, srcDeletions === 0);
|
|
168
|
+
|
|
169
|
+
return {
|
|
170
|
+
sha: full, short: full.slice(0, 8), subject, date,
|
|
171
|
+
...kindInfo, srcDeletions,
|
|
172
|
+
files,
|
|
173
|
+
testFiles: files.filter(f => TEST_RE(f)),
|
|
174
|
+
sourceFiles: files.filter(f => !TEST_RE(f) && !DOC_RE.test(f)),
|
|
175
|
+
};
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
/**
|
|
179
|
+
* `against` turns this from a commit check into a PR check, and the distinction matters.
|
|
180
|
+
*
|
|
181
|
+
* By default the "broken" side is the commit's own parent — right for auditing history,
|
|
182
|
+
* where each fix is judged against the bug it fixed.
|
|
183
|
+
*
|
|
184
|
+
* A PR is different. A real PR landed a follow-up commit whose parent ALREADY contained
|
|
185
|
+
* the change under test — so commit-vs-parent compares the branch to itself and every
|
|
186
|
+
* case reads non-discriminating, and a sound PR looks unproven. The question a PR gate asks is
|
|
187
|
+
* "does this test fail WITHOUT THIS BRANCH", so the base is the MERGE BASE of the branch
|
|
188
|
+
* and its target, never the target's tip: develop moves, and diffing against a moved tip
|
|
189
|
+
* drags in everyone else's changes and attributes them here.
|
|
190
|
+
*/
|
|
191
|
+
export async function verifyCommit({ repo, workDir, sha, against = null, runs = 1, timeoutMs = 120_000, onStep = () => {} }) {
|
|
192
|
+
const info = await commitInfo(repo, sha, against, true);
|
|
193
|
+
const result = { ...info, cases: [], verdict: SKIPPED, note: null };
|
|
194
|
+
|
|
195
|
+
if (info.testFiles.length === 0) { result.note = 'no test file in the commit'; return result; }
|
|
196
|
+
if (info.sourceFiles.length === 0) { result.note = 'test-only commit — no source change to be blind to'; return result; }
|
|
197
|
+
if (info.sourceFiles.every(f => BUILD_ONLY_RE.test(f))) {
|
|
198
|
+
result.note = 'build-only change (lockfiles / project files) — a build catches this, not a unit test';
|
|
199
|
+
return result;
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const parent = against
|
|
203
|
+
? (await git(repo, ['merge-base', against, sha])).trim()
|
|
204
|
+
: (await git(repo, ['rev-parse', `${sha}^`])).trim();
|
|
205
|
+
|
|
206
|
+
// --- ARM A -------------------------------------------------------------------------
|
|
207
|
+
onStep('arm A');
|
|
208
|
+
const fixDir = await ensureWorktree(repo, workDir, 'fix', sha);
|
|
209
|
+
await linkDependencies(repo, fixDir);
|
|
210
|
+
await linkEnvFiles(repo, fixDir);
|
|
211
|
+
|
|
212
|
+
// --- ARM B -------------------------------------------------------------------------
|
|
213
|
+
onStep('arm B');
|
|
214
|
+
const parentDir = await ensureWorktree(repo, workDir, 'parent', parent);
|
|
215
|
+
await linkDependencies(repo, parentDir);
|
|
216
|
+
await linkEnvFiles(repo, parentDir);
|
|
217
|
+
|
|
218
|
+
const perFile = [];
|
|
219
|
+
for (const rel of info.testFiles) {
|
|
220
|
+
// Chosen from the FIX worktree: the parent may predate the config file entirely,
|
|
221
|
+
// and the question is which runner the test was written for.
|
|
222
|
+
const { flavour, pkgDir } = await selectRunner(fixDir, rel);
|
|
223
|
+
const runner = RUNNERS[flavour];
|
|
224
|
+
const opts = { relTestPath: rel, pkgDir, timeoutMs };
|
|
225
|
+
|
|
226
|
+
const a = await runner.execute({ worktreeDir: fixDir, ...opts });
|
|
227
|
+
if (!a.ok) { perFile.push({ file: rel, runner: flavour, pkgDir, skip: `arm A did not run (${flavour}): ${a.loadFailure}` }); continue; }
|
|
228
|
+
|
|
229
|
+
const dest = await transplant(repo, sha, rel, parentDir);
|
|
230
|
+
const identity = await proveIdentity({ worktreeRoot: parentDir, repoRoot: repo, testFilePath: dest });
|
|
231
|
+
|
|
232
|
+
const runsOut = [];
|
|
233
|
+
for (let i = 0; i < runs; i++) {
|
|
234
|
+
const b = await runner.execute({ worktreeDir: parentDir, ...opts });
|
|
235
|
+
runsOut.push({ b, identity });
|
|
236
|
+
}
|
|
237
|
+
perFile.push({ file: rel, runner: flavour, pkgDir, armA: a, runs: runsOut, identity });
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
// --- verdicts ----------------------------------------------------------------------
|
|
241
|
+
// The fix's own copy of each test file, read ONCE per file rather than once per case:
|
|
242
|
+
// the assertion analysis needs the source, and a 40-case file would otherwise re-read
|
|
243
|
+
// it forty times.
|
|
244
|
+
const sourceOf = new Map();
|
|
245
|
+
for (const f of perFile) {
|
|
246
|
+
if (f.skip) continue;
|
|
247
|
+
try { sourceOf.set(f.file, await git(repo, ['show', `${sha}:${f.file}`])); } catch { /* unreadable */ }
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
for (const f of perFile) {
|
|
251
|
+
if (f.skip) { result.cases.push({ file: f.file, name: '(file)', verdict: INCONCLUSIVE, reason: f.skip }); continue; }
|
|
252
|
+
for (const aCase of f.armA.cases) {
|
|
253
|
+
const perRun = f.runs.map(({ b, identity }) => {
|
|
254
|
+
if (!b.ok) return { verdict: INCONCLUSIVE, reason: `did not run on the parent: ${b.loadFailure}` };
|
|
255
|
+
const bCase = b.cases.find(c => c.name === aCase.name) || null;
|
|
256
|
+
return classify({ armA: aCase, armB: bCase, identity });
|
|
257
|
+
});
|
|
258
|
+
const final = reduceRuns(perRun);
|
|
259
|
+
// WHY is this case weak — attached only where the answer is useful. A CAUGHT
|
|
260
|
+
// case needs no explanation (it did its job) and an INCONCLUSIVE one already
|
|
261
|
+
// carries its reason, so the note goes on the cases a reader would otherwise
|
|
262
|
+
// have to open the file to understand.
|
|
263
|
+
let why = null;
|
|
264
|
+
const src = sourceOf.get(f.file);
|
|
265
|
+
if (src && final.verdict === NON_DISCRIMINATING) {
|
|
266
|
+
const a = analyseCase(src, aCase.name);
|
|
267
|
+
if (a.verdict === 'weak' && a.findings.length) why = a.findings[0].note;
|
|
268
|
+
else if (a.verdict === 'suspect' && a.findings.length) why = `${a.findings[0].note} — but it also asserts an exact value, so it may still be sound`;
|
|
269
|
+
else if (a.verdict === 'strong') why = 'asserts an exact expected value — most likely a deliberate regression guard';
|
|
270
|
+
}
|
|
271
|
+
result.cases.push({ file: f.file, name: aCase.name, ...final, ...(why ? { why } : {}) });
|
|
272
|
+
}
|
|
273
|
+
}
|
|
274
|
+
result.verdict = rollUp(result.cases);
|
|
275
|
+
|
|
276
|
+
// --- ARM C: is this BLIND finding still open? ---------------------------------------
|
|
277
|
+
if (result.verdict === BLIND) {
|
|
278
|
+
result.stillOpen = await armC({ repo, parentDir, perFile, timeoutMs });
|
|
279
|
+
}
|
|
280
|
+
return result;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
/**
|
|
284
|
+
* ARM C — the current test against the historical bug.
|
|
285
|
+
*
|
|
286
|
+
* WHY THIS EXISTS, and it is the most important caveat in the tool.
|
|
287
|
+
*
|
|
288
|
+
* Arms A and B answer "did the test SHIPPED WITH THIS COMMIT catch its own bug". That is a
|
|
289
|
+
* fact about the past and it stays true forever. It does NOT mean there is a gap today:
|
|
290
|
+
* the repo may have repaired it since, and a BLIND verdict reported as an open defect is a
|
|
291
|
+
* redundant ticket handed to a colleague.
|
|
292
|
+
*
|
|
293
|
+
* Earned by doing exactly that. A commit shipped a property-based test whose loop stepped
|
|
294
|
+
* over the very date its own comment named as the bug, so it could not fail on the code it
|
|
295
|
+
* was written for. True — and the one-character fix was recommended to a colleague who was
|
|
296
|
+
* about to open a PR for it. The stride had already been corrected three months earlier,
|
|
297
|
+
* by MUTATION TESTING, which deleted the loop body and saw nothing fail. A different
|
|
298
|
+
* instrument had found the same defect from the opposite direction.
|
|
299
|
+
*
|
|
300
|
+
* So: take the CURRENT version of the test file, put it on the parent's broken code, and
|
|
301
|
+
* run it. If it fails now, the gap was repaired and the finding is history, not a ticket.
|
|
302
|
+
*/
|
|
303
|
+
async function armC({ repo, parentDir, perFile, timeoutMs }) {
|
|
304
|
+
const files = perFile.filter(f => !f.skip);
|
|
305
|
+
if (files.length === 0) return { status: 'unknown', reason: 'no runnable test file' };
|
|
306
|
+
|
|
307
|
+
for (const f of files) {
|
|
308
|
+
let dest;
|
|
309
|
+
try { dest = await transplant(repo, 'HEAD', f.file, parentDir); }
|
|
310
|
+
catch { return { status: 'unknown', reason: `${f.file} does not exist at HEAD (renamed or deleted)` }; }
|
|
311
|
+
|
|
312
|
+
const runner = RUNNERS[f.runner] || nodeTest;
|
|
313
|
+
const r = await runner.execute({ worktreeDir: parentDir, relTestPath: f.file, pkgDir: f.pkgDir, timeoutMs });
|
|
314
|
+
if (!r.ok) return { status: 'unknown', reason: `current test could not run on the parent: ${r.loadFailure}` };
|
|
315
|
+
|
|
316
|
+
const killer = r.cases.find(c => isDisagreement(c));
|
|
317
|
+
if (killer) {
|
|
318
|
+
return { status: 'repaired', reason: `the CURRENT "${killer.name}" fails on this bug — the gap was closed after this commit`, by: killer.name, file: f.file };
|
|
319
|
+
}
|
|
320
|
+
}
|
|
321
|
+
return { status: 'open', reason: 'even the CURRENT tests are green on this bug — still unguarded today' };
|
|
322
|
+
}
|