control-arm 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,168 @@
1
+ /**
2
+ * The decision. Pure functions only — no git, no filesystem, no subprocesses.
3
+ *
4
+ * This file is deliberately the only place a verdict is decided, and deliberately
5
+ * has no I/O, because it is the part that has to be exhaustively tested. Everything
6
+ * else in this tool is plumbing that produces its two inputs.
7
+ *
8
+ * THE FIVE VERDICTS
9
+ *
10
+ * CAUGHT the case ran on the broken code and DISAGREED with it (assertion)
11
+ * NON-DISCRIMINATING the case ran and was fine with it (case level — a fact, not a fault)
12
+ * BLIND NO case in the commit discriminated (commit level — a judgement)
13
+ * INCONCLUSIVE the case did not run, or we cannot prove what it ran against
14
+ * FLAKY repeated runs disagreed with each other
15
+ * SKIPPED there was nothing to judge
16
+ *
17
+ * WHY EXIT CODE IS NOT THE SIGNAL. A test that "fails" at the parent commit with
18
+ * `SyntaxError: does not provide an export named 'TITLE_SEPARATOR'` — because the fix
19
+ * ADDED that export — has a non-zero exit and has told you nothing: it never ran. Read
20
+ * exit code only and you report CAUGHT, and the tool's headline number is inflated
21
+ * garbage. Observed on a real fix commit whose new export the parent did not have.
22
+ *
23
+ * WHY A FALSE `BLIND` IS THE WORST OUTPUT. CAUGHT and INCONCLUSIVE are both survivable
24
+ * — one is good news, the other is an honest shrug. BLIND accuses an engineer of having
25
+ * written a test that cannot fail. Get that wrong and nobody trusts the tool again. So
26
+ * every ambiguity resolves AWAY from BLIND, never toward it.
27
+ */
28
+
29
+ export const CAUGHT = 'CAUGHT';
30
+ /**
31
+ * CASE level. The case ran on the broken code and was fine with it.
32
+ *
33
+ * NOT a synonym for "bad test", and renamed from BLIND on 2026-09-23 after the first real
34
+ * run said this about a commit whose fix added a separator class:
35
+ *
36
+ * ✗ BLIND CONTROL ARM: the old pattern really does split one phrase in two
37
+ * ✗ BLIND the prefix rule survives the wider separator class
38
+ * ✗ BLIND the labels the separator change must not touch
39
+ *
40
+ * All three are REGRESSION GUARDS. They are supposed to be green on both arms — that is
41
+ * their entire job. Calling them blind is a false accusation of a careful engineer, which
42
+ * is the one output that destroys trust in this tool.
43
+ *
44
+ * Red/green cannot distinguish "meant to catch this bug and failed" from "meant to stay
45
+ * green". That is INTENT, and the tool cannot read it. So the case level states the FACT
46
+ * (it did not discriminate) and the judgement is made only where it is safe: at the
47
+ * commit, where a fix in which NOTHING discriminates shipped no test that could catch its
48
+ * own bug, whatever each case was for.
49
+ */
50
+ export const NON_DISCRIMINATING = 'NON-DISCRIMINATING';
51
+ /** COMMIT level only: every case ran, and not one of them discriminated. */
52
+ export const BLIND = 'BLIND';
53
+ export const INCONCLUSIVE = 'INCONCLUSIVE';
54
+ export const FLAKY = 'FLAKY';
55
+ export const SKIPPED = 'SKIPPED';
56
+
57
+ /**
58
+ * Did this failure come from the test DISAGREEING, or from it being unable to run?
59
+ *
60
+ * node:test TAP puts `name: 'AssertionError'` / `code: 'ERR_ASSERTION'` on the first and
61
+ * a real error name (ReferenceError, SyntaxError, TypeError) on the second. Other runners
62
+ * differ; `runners/*.mjs` maps them onto this same shape, which is why this predicate
63
+ * takes a normalised case and not raw output.
64
+ */
65
+ export function isDisagreement(caseResult) {
66
+ if (!caseResult || caseResult.status !== 'fail') return false;
67
+ if (caseResult.code === 'ERR_ASSERTION') return true;
68
+ if (caseResult.errorName === 'AssertionError') return true;
69
+ // Assertion libraries that are not node:assert. Narrow on purpose: an unknown error
70
+ // name must fall through to INCONCLUSIVE, not be guessed into CAUGHT.
71
+ return ['JestAssertionError', 'ExpectationFailed', 'AssertionFailedError'].includes(caseResult.errorName);
72
+ }
73
+
74
+ /**
75
+ * One case, one run. `armA` is the case on the FIXED code, `armB` the same case
76
+ * transplanted onto the BROKEN (parent) code.
77
+ *
78
+ * `identity` is the proof that armB actually loaded the parent's modules. It is a hard
79
+ * gate before anything else, because without it a BLIND is indistinguishable from a
80
+ * harness that silently resolved the fixed code — which is exactly what happened on the
81
+ * first real run of this tool.
82
+ */
83
+ export function classify({ armA, armB, identity }) {
84
+ if (identity && identity.proven === false) {
85
+ return { verdict: INCONCLUSIVE, reason: `module identity unproven: ${identity.reason}` };
86
+ }
87
+ if (!armA) return { verdict: SKIPPED, reason: 'case not present on the fix' };
88
+
89
+ // A case the runner SKIPPED tells us nothing and is not a failure of anything. It gets
90
+ // its own bucket rather than inflating INCONCLUSIVE — 257 of 1,185 cases in the first
91
+ // 300-commit audit were `skip`, almost all DB-gated tests with no TEST_DATABASE_URL.
92
+ // Folding those into "could not be judged" hides the fact that they are judgeable, by
93
+ // anyone who runs the audit with a database.
94
+ if (armA.status === 'skip') {
95
+ return { verdict: SKIPPED, reason: 'the runner skipped this case on the fix (gated on an env var or a service?)' };
96
+ }
97
+ // The fix must be green, or there is no "before and after" to compare. A red case on
98
+ // the fix means the commit does not stand on its own — not that the test is bad.
99
+ if (armA.status !== 'pass') {
100
+ return { verdict: INCONCLUSIVE, reason: `case is not green on the fix (${armA.errorName || armA.status})` };
101
+ }
102
+ // The case exists on the fix but never reported on the parent: the file failed to
103
+ // load, or the runner died before reaching it.
104
+ if (!armB) {
105
+ return { verdict: INCONCLUSIVE, reason: 'case did not report on the parent (file failed to load?)' };
106
+ }
107
+ if (armB.status === 'pass') {
108
+ return { verdict: NON_DISCRIMINATING, reason: 'green on the broken code (a regression guard looks identical here)' };
109
+ }
110
+ if (isDisagreement(armB)) {
111
+ return { verdict: CAUGHT, reason: armB.message || 'assertion failed on the broken code' };
112
+ }
113
+ return {
114
+ verdict: INCONCLUSIVE,
115
+ reason: `case errored rather than failed on the parent (${armB.errorName || armB.code || 'unknown'}) — it never ran`,
116
+ };
117
+ }
118
+
119
+ /**
120
+ * N runs of the same case. Disagreement between runs is its own verdict: a case that is
121
+ * CAUGHT twice and BLIND once has told you nothing about the fix and something important
122
+ * about the test.
123
+ *
124
+ * INCONCLUSIVE does not outvote a real result — an infra hiccup on run 2 should not erase
125
+ * a clean CAUGHT from runs 1 and 3 — but a CAUGHT/BLIND split is FLAKY, always.
126
+ */
127
+ export function reduceRuns(results) {
128
+ if (results.length === 0) return { verdict: SKIPPED, reason: 'no runs' };
129
+ const decided = results.filter(r => r.verdict === CAUGHT || r.verdict === NON_DISCRIMINATING);
130
+ if (decided.length === 0) return results[0];
131
+
132
+ const distinct = new Set(decided.map(r => r.verdict));
133
+ if (distinct.size > 1) {
134
+ const tally = decided.map(r => r.verdict).join(', ');
135
+ return { verdict: FLAKY, reason: `runs disagreed: ${tally}` };
136
+ }
137
+ if (decided.length < results.length) {
138
+ const d = decided[0];
139
+ return { ...d, reason: `${d.reason} (${results.length - decided.length} of ${results.length} runs inconclusive)` };
140
+ }
141
+ return decided[0];
142
+ }
143
+
144
+ /**
145
+ * Commit-level roll-up. A commit is CAUGHT if ANY of its cases discriminates.
146
+ *
147
+ * BLIND requires that EVERY case actually RAN. A commit with one skipped case and three
148
+ * non-discriminating ones is INCONCLUSIVE, not BLIND — the skipped case might have been
149
+ * the discriminating one, and nothing here can know.
150
+ *
151
+ * Earned on a commit fixing a null-limit bug in a metering module. All four
152
+ * tests it shipped are `{ skip: SKIP }`, gated on TEST_DATABASE_URL, which the audit host
153
+ * did not set. The older cases in the same file ran and did not discriminate, so an
154
+ * earlier version of this function called the commit BLIND — on the strength of the tests
155
+ * that were NOT written for the bug, while the four that were sat unexecuted. Run the
156
+ * audit with a database and the same commit may well be CAUGHT.
157
+ *
158
+ * Same principle as everywhere else here: ambiguity resolves AWAY from BLIND.
159
+ */
160
+ export function rollUp(caseVerdicts) {
161
+ if (caseVerdicts.length === 0) return SKIPPED;
162
+ const has = v => caseVerdicts.some(c => c.verdict === v);
163
+ if (has(CAUGHT)) return CAUGHT;
164
+ if (has(FLAKY)) return FLAKY;
165
+ if (has(INCONCLUSIVE) || has(SKIPPED)) return INCONCLUSIVE;
166
+ if (caseVerdicts.every(c => c.verdict === NON_DISCRIMINATING)) return BLIND;
167
+ return INCONCLUSIVE;
168
+ }
package/src/verify.mjs ADDED
@@ -0,0 +1,322 @@
1
+ /**
2
+ * Orchestration: two arms, one commit.
3
+ *
4
+ * ARM A worktree @ FIX + the fix's own test -> must be GREEN, or there is no
5
+ * before/after to compare
6
+ * ARM B worktree @ PARENT + the fix's test TRANSPLANTED onto the old source
7
+ *
8
+ * The transplant is the whole idea. You do NOT run the parent's tests — the parent does
9
+ * not have this test. You take the new test and ask it about the old code.
10
+ */
11
+
12
+ import path from 'node:path';
13
+ import { git, ensureWorktree, linkDependencies, linkEnvFiles, transplant } from './worktree.mjs';
14
+ import { proveIdentity } from './identity.mjs';
15
+ import { nodeTest } from './runner.mjs';
16
+ import { analyseCase } from './assertions.mjs';
17
+ import { vitest, jest } from './runner-json.mjs';
18
+ import { selectRunner } from './select-runner.mjs';
19
+
20
+ const RUNNERS = { node: nodeTest, vitest, jest };
21
+ import { classify, reduceRuns, rollUp, isDisagreement, BLIND, NON_DISCRIMINATING, SKIPPED, INCONCLUSIVE } from './verdict.mjs';
22
+
23
+ /**
24
+ * Documentation, and nothing else, is "not source".
25
+ *
26
+ * This used to be an ALLOWLIST of `.js/.ts/.tsx/.mjs`, and it made the tool decline the
27
+ * one kind of change it should be best at judging. A real PR wired a mutation-testing
28
+ * gate: it changed a tool config and a CI workflow, and shipped a test asserting that
29
+ * wiring. Zero JS changed, so the commit read as "test-only — no source change to be
30
+ * blind to" and got no verdict.
31
+ *
32
+ * That is backwards. A guard that is installed but wired to nothing is the dominant defect
33
+ * class in many repos — three instances found in one afternoon — and the test proving a
34
+ * gate is armed is precisely a test that should be shown to fail without the wiring.
35
+ * Config, CI YAML and shell ARE the source for those.
36
+ *
37
+ * So: deny-list the docs, accept the rest. A commit that changes only Markdown genuinely
38
+ * has nothing to be blind to; everything else might.
39
+ */
40
+ const DOC_RE = /\.(md|mdx|txt|rst|adoc)$/i;
41
+
42
+ /**
43
+ * Files no unit test can exercise, however good the suite is.
44
+ *
45
+ * A `fix(ios):` commit repairing an aborted CocoaPods install came back as a still-open
46
+ * BLIND. It is not a finding. A dependency-resolution failure is caught by a BUILD, and
47
+ * asking whether a unit test would have caught it is a category error. Reporting it as a gap teaches the reader that the tool does not know
48
+ * what a test is for.
49
+ *
50
+ * Deliberately narrow: only manifests and project files for toolchains that build rather
51
+ * than run. A .json or .yml can absolutely be under test — a CI-wiring test asserts a
52
+ * workflow file — so those are NOT here.
53
+ */
54
+ const BUILD_ONLY_RE = /(^|\/)(Podfile(\.lock)?|Gemfile(\.lock)?|Cartfile.*|package-lock\.json|yarn\.lock|pnpm-lock\.yaml|.*\.pbxproj|.*\.xcworkspacedata|.*\.xcscheme|.*\.gradle(\.kts)?|gradle\.properties|.*\.plist|.*\.podspec|.*\.lock)$/i;
55
+
56
+ /**
57
+ * KNOWN LIMIT, stated rather than papered over: build-time JAVASCRIPT is not detectable
58
+ * by filename. One such commit changed `apps/mobile/plugins/withModularHeaders.js` — an
59
+ * Expo config plugin that runs at BUILD time, spelled exactly like runtime source. Webpack/vite/rollup configs and codegen
60
+ * scripts are the same shape.
61
+ *
62
+ * A heuristic wide enough to catch those (path contains "plugins", "scripts", "config")
63
+ * would also exclude real application code, and a FALSE EXCLUSION is worse than a false
64
+ * inclusion here: it hides a finding silently, where a category error is at least visible
65
+ * and arguable. So these reach a verdict and need human triage. The audit's subject
66
+ * filter catches most of them in practice, because they are usually scoped fix(ios),
67
+ * fix(build) or similar.
68
+ */
69
+
70
+ /**
71
+ * What counts as a test file.
72
+ *
73
+ * TWO independent signals, because projects pick one or the other and a tool that demands
74
+ * both measures nothing:
75
+ *
76
+ * 1. a `.test.` / `.spec.` suffix anywhere (most application repos)
77
+ * 2. living under a test directory (undici, node core, most library repos)
78
+ *
79
+ * This used to require BOTH — a file under `tests/` AND a `.test.` suffix. Run against
80
+ * nodejs/undici, whose tests are `test/client-request.js` with no suffix, the tool
81
+ * reported "284 commits matched, 0 ship both a test and a source change" and produced an
82
+ * entirely empty audit. Not a wrong answer: NO answer, on a repo with 261 fix commits
83
+ * that ship tests.
84
+ *
85
+ * Found the first time it was pointed at a codebase its author did not write, which is
86
+ * the whole argument for doing that before believing any number it prints.
87
+ */
88
+ const TEST_DIR_RE = /(^|\/)(tests?|__tests__|spec|specs)\//i;
89
+ const TEST_SUFFIX_RE = /\.(test|spec)\.(m?[jt]sx?)$/i;
90
+ const CODE_RE = /\.(m?[jt]sx?)$/i;
91
+ const TEST_RE = (f) => CODE_RE.test(f) && (TEST_SUFFIX_RE.test(f) || TEST_DIR_RE.test(f));
92
+
93
+ /**
94
+ * IS THIS A REPAIR, OR IS IT NEW CODE? — and why the answer changes what CAUGHT means.
95
+ *
96
+ * On a BUG FIX the base is the broken code, so a test that fails there genuinely
97
+ * discriminates: it would have caught the bug. That is the case this tool was built for.
98
+ *
99
+ * On a FEATURE the base is code where the thing does not exist yet. Essentially ANY test
100
+ * touching the new code fails there — the import is missing, the field is absent, a count
101
+ * is zero. `expected 0 to be greater than 0` is a real AssertionError, correctly
102
+ * classified, and says nothing whatever about whether the test is well aimed. A test
103
+ * asserting `expect(1).toBe(1)` in the same file would NOT have been CAUGHT; essentially
104
+ * any test reading the new field would. The signal comes from the field's existence, not
105
+ * from the test's design.
106
+ *
107
+ * This is the mirror of the INCONCLUSIVE rule. That one says a red test that never RAN
108
+ * proves nothing; this one says a red test that ran against ABSENT CODE proves nearly as
109
+ * little. Counting them together lets the headline drift upward for free on a
110
+ * feature-heavy week, and a number that drifts for free is one people stop reading.
111
+ *
112
+ * MEASURED, because both signals are proxies and only one is any good:
113
+ * conventional prefix 1,213 fix / 988 feat in one real corpus — used consistently
114
+ * source additions-only 9 of 28 feat commits, but also 1 of 35 fix commits —
115
+ * specific, not sensitive; a supporting hint, never the verdict
116
+ *
117
+ * So this labels the CLAIM and never silently reclassifies a verdict. CAUGHT on a feature
118
+ * is still CAUGHT — it just does not get to say "this would have caught the bug", because
119
+ * there was no bug.
120
+ */
121
+ export function commitKind(subject, sourceAddedOnly) {
122
+ const m = String(subject).match(/^\s*([a-z]+)\s*(\([^)]*\))?\s*!?:/i);
123
+ const prefix = m ? m[1].toLowerCase() : null;
124
+ if (prefix === 'fix' || prefix === 'bug' || prefix === 'perf') {
125
+ // A fix that only ADDED source is worth flagging: its test may be reading
126
+ // something that simply was not there, exactly like a feature's.
127
+ return { kind: 'fix', newCode: !!sourceAddedOnly, prefix };
128
+ }
129
+ if (prefix === 'feat' || prefix === 'feature') return { kind: 'feature', newCode: true, prefix };
130
+ if (prefix) return { kind: 'other', newCode: !!sourceAddedOnly, prefix };
131
+ return { kind: 'unknown', newCode: !!sourceAddedOnly, prefix: null };
132
+ }
133
+
134
+ /**
135
+ * `withDiffStat` is OFF by default, and that default is load-bearing.
136
+ *
137
+ * The additions-only signal needs `git show --numstat` — one subprocess per commit. The
138
+ * audit calls commitInfo on EVERY candidate just to test eligibility (1,178 of them on one
139
+ * real corpus), so computing it unconditionally pushed the selection pass past 70 minutes
140
+ * BEFORE a single commit was judged. Measured, not estimated: that run was killed at 72.
141
+ *
142
+ * Only the commits actually verified need it, so verifyCommit asks and the eligibility
143
+ * pass does not. A regression introduced by the fix for issue #3, and caught by noticing a
144
+ * run sit in "selecting commits" for over an hour rather than by any test.
145
+ */
146
+ export async function commitInfo(repo, sha, against = null, withDiffStat = false) {
147
+ const out = await git(repo, ['show', '--no-patch', '--format=%H%n%s%n%ad', '--date=short', sha]);
148
+ const [full, subject, date] = out.trim().split('\n');
149
+ // With a base, the changed set is the WHOLE branch, not just the tip commit — a PR's
150
+ // test may have arrived in commit 1 and its source change in commit 3.
151
+ const files = against
152
+ ? (await git(repo, ['diff', '--name-only', `${(await git(repo, ['merge-base', against, sha])).trim()}...${sha}`])).trim().split('\n').filter(Boolean)
153
+ : (await git(repo, ['show', '--name-only', '--format=', sha])).trim().split('\n').filter(Boolean);
154
+ // Deletions in non-test source: a repair usually changes lines, new code only adds.
155
+ const numstat = !withDiffStat ? '' : against
156
+ ? await git(repo, ['diff', '--numstat', `${(await git(repo, ['merge-base', against, sha])).trim()}...${sha}`])
157
+ : await git(repo, ['show', '--numstat', '--format=', sha]);
158
+ let srcDeletions = 0;
159
+ for (const line of numstat.trim().split('\n')) {
160
+ const [, del, file] = line.split(/\t/).length === 3 ? ['', ...line.split(/\t/).slice(1)] : [];
161
+ const parts = line.split(/\t/);
162
+ if (parts.length !== 3) continue;
163
+ const [, d, f] = parts;
164
+ if (TEST_RE(f) || DOC_RE.test(f)) continue;
165
+ if (/^\d+$/.test(d)) srcDeletions += Number(d);
166
+ }
167
+ const kindInfo = commitKind(subject, srcDeletions === 0);
168
+
169
+ return {
170
+ sha: full, short: full.slice(0, 8), subject, date,
171
+ ...kindInfo, srcDeletions,
172
+ files,
173
+ testFiles: files.filter(f => TEST_RE(f)),
174
+ sourceFiles: files.filter(f => !TEST_RE(f) && !DOC_RE.test(f)),
175
+ };
176
+ }
177
+
178
+ /**
179
+ * `against` turns this from a commit check into a PR check, and the distinction matters.
180
+ *
181
+ * By default the "broken" side is the commit's own parent — right for auditing history,
182
+ * where each fix is judged against the bug it fixed.
183
+ *
184
+ * A PR is different. A real PR landed a follow-up commit whose parent ALREADY contained
185
+ * the change under test — so commit-vs-parent compares the branch to itself and every
186
+ * case reads non-discriminating, and a sound PR looks unproven. The question a PR gate asks is
187
+ * "does this test fail WITHOUT THIS BRANCH", so the base is the MERGE BASE of the branch
188
+ * and its target, never the target's tip: develop moves, and diffing against a moved tip
189
+ * drags in everyone else's changes and attributes them here.
190
+ */
191
+ export async function verifyCommit({ repo, workDir, sha, against = null, runs = 1, timeoutMs = 120_000, onStep = () => {} }) {
192
+ const info = await commitInfo(repo, sha, against, true);
193
+ const result = { ...info, cases: [], verdict: SKIPPED, note: null };
194
+
195
+ if (info.testFiles.length === 0) { result.note = 'no test file in the commit'; return result; }
196
+ if (info.sourceFiles.length === 0) { result.note = 'test-only commit — no source change to be blind to'; return result; }
197
+ if (info.sourceFiles.every(f => BUILD_ONLY_RE.test(f))) {
198
+ result.note = 'build-only change (lockfiles / project files) — a build catches this, not a unit test';
199
+ return result;
200
+ }
201
+
202
+ const parent = against
203
+ ? (await git(repo, ['merge-base', against, sha])).trim()
204
+ : (await git(repo, ['rev-parse', `${sha}^`])).trim();
205
+
206
+ // --- ARM A -------------------------------------------------------------------------
207
+ onStep('arm A');
208
+ const fixDir = await ensureWorktree(repo, workDir, 'fix', sha);
209
+ await linkDependencies(repo, fixDir);
210
+ await linkEnvFiles(repo, fixDir);
211
+
212
+ // --- ARM B -------------------------------------------------------------------------
213
+ onStep('arm B');
214
+ const parentDir = await ensureWorktree(repo, workDir, 'parent', parent);
215
+ await linkDependencies(repo, parentDir);
216
+ await linkEnvFiles(repo, parentDir);
217
+
218
+ const perFile = [];
219
+ for (const rel of info.testFiles) {
220
+ // Chosen from the FIX worktree: the parent may predate the config file entirely,
221
+ // and the question is which runner the test was written for.
222
+ const { flavour, pkgDir } = await selectRunner(fixDir, rel);
223
+ const runner = RUNNERS[flavour];
224
+ const opts = { relTestPath: rel, pkgDir, timeoutMs };
225
+
226
+ const a = await runner.execute({ worktreeDir: fixDir, ...opts });
227
+ if (!a.ok) { perFile.push({ file: rel, runner: flavour, pkgDir, skip: `arm A did not run (${flavour}): ${a.loadFailure}` }); continue; }
228
+
229
+ const dest = await transplant(repo, sha, rel, parentDir);
230
+ const identity = await proveIdentity({ worktreeRoot: parentDir, repoRoot: repo, testFilePath: dest });
231
+
232
+ const runsOut = [];
233
+ for (let i = 0; i < runs; i++) {
234
+ const b = await runner.execute({ worktreeDir: parentDir, ...opts });
235
+ runsOut.push({ b, identity });
236
+ }
237
+ perFile.push({ file: rel, runner: flavour, pkgDir, armA: a, runs: runsOut, identity });
238
+ }
239
+
240
+ // --- verdicts ----------------------------------------------------------------------
241
+ // The fix's own copy of each test file, read ONCE per file rather than once per case:
242
+ // the assertion analysis needs the source, and a 40-case file would otherwise re-read
243
+ // it forty times.
244
+ const sourceOf = new Map();
245
+ for (const f of perFile) {
246
+ if (f.skip) continue;
247
+ try { sourceOf.set(f.file, await git(repo, ['show', `${sha}:${f.file}`])); } catch { /* unreadable */ }
248
+ }
249
+
250
+ for (const f of perFile) {
251
+ if (f.skip) { result.cases.push({ file: f.file, name: '(file)', verdict: INCONCLUSIVE, reason: f.skip }); continue; }
252
+ for (const aCase of f.armA.cases) {
253
+ const perRun = f.runs.map(({ b, identity }) => {
254
+ if (!b.ok) return { verdict: INCONCLUSIVE, reason: `did not run on the parent: ${b.loadFailure}` };
255
+ const bCase = b.cases.find(c => c.name === aCase.name) || null;
256
+ return classify({ armA: aCase, armB: bCase, identity });
257
+ });
258
+ const final = reduceRuns(perRun);
259
+ // WHY is this case weak — attached only where the answer is useful. A CAUGHT
260
+ // case needs no explanation (it did its job) and an INCONCLUSIVE one already
261
+ // carries its reason, so the note goes on the cases a reader would otherwise
262
+ // have to open the file to understand.
263
+ let why = null;
264
+ const src = sourceOf.get(f.file);
265
+ if (src && final.verdict === NON_DISCRIMINATING) {
266
+ const a = analyseCase(src, aCase.name);
267
+ if (a.verdict === 'weak' && a.findings.length) why = a.findings[0].note;
268
+ else if (a.verdict === 'suspect' && a.findings.length) why = `${a.findings[0].note} — but it also asserts an exact value, so it may still be sound`;
269
+ else if (a.verdict === 'strong') why = 'asserts an exact expected value — most likely a deliberate regression guard';
270
+ }
271
+ result.cases.push({ file: f.file, name: aCase.name, ...final, ...(why ? { why } : {}) });
272
+ }
273
+ }
274
+ result.verdict = rollUp(result.cases);
275
+
276
+ // --- ARM C: is this BLIND finding still open? ---------------------------------------
277
+ if (result.verdict === BLIND) {
278
+ result.stillOpen = await armC({ repo, parentDir, perFile, timeoutMs });
279
+ }
280
+ return result;
281
+ }
282
+
283
+ /**
284
+ * ARM C — the current test against the historical bug.
285
+ *
286
+ * WHY THIS EXISTS, and it is the most important caveat in the tool.
287
+ *
288
+ * Arms A and B answer "did the test SHIPPED WITH THIS COMMIT catch its own bug". That is a
289
+ * fact about the past and it stays true forever. It does NOT mean there is a gap today:
290
+ * the repo may have repaired it since, and a BLIND verdict reported as an open defect is a
291
+ * redundant ticket handed to a colleague.
292
+ *
293
+ * Earned by doing exactly that. A commit shipped a property-based test whose loop stepped
294
+ * over the very date its own comment named as the bug, so it could not fail on the code it
295
+ * was written for. True — and the one-character fix was recommended to a colleague who was
296
+ * about to open a PR for it. The stride had already been corrected three months earlier,
297
+ * by MUTATION TESTING, which deleted the loop body and saw nothing fail. A different
298
+ * instrument had found the same defect from the opposite direction.
299
+ *
300
+ * So: take the CURRENT version of the test file, put it on the parent's broken code, and
301
+ * run it. If it fails now, the gap was repaired and the finding is history, not a ticket.
302
+ */
303
+ async function armC({ repo, parentDir, perFile, timeoutMs }) {
304
+ const files = perFile.filter(f => !f.skip);
305
+ if (files.length === 0) return { status: 'unknown', reason: 'no runnable test file' };
306
+
307
+ for (const f of files) {
308
+ let dest;
309
+ try { dest = await transplant(repo, 'HEAD', f.file, parentDir); }
310
+ catch { return { status: 'unknown', reason: `${f.file} does not exist at HEAD (renamed or deleted)` }; }
311
+
312
+ const runner = RUNNERS[f.runner] || nodeTest;
313
+ const r = await runner.execute({ worktreeDir: parentDir, relTestPath: f.file, pkgDir: f.pkgDir, timeoutMs });
314
+ if (!r.ok) return { status: 'unknown', reason: `current test could not run on the parent: ${r.loadFailure}` };
315
+
316
+ const killer = r.cases.find(c => isDisagreement(c));
317
+ if (killer) {
318
+ return { status: 'repaired', reason: `the CURRENT "${killer.name}" fails on this bug — the gap was closed after this commit`, by: killer.name, file: f.file };
319
+ }
320
+ }
321
+ return { status: 'open', reason: 'even the CURRENT tests are green on this bug — still unguarded today' };
322
+ }