control-arm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +82 -0
- package/DESIGN.md +161 -0
- package/LICENSE +21 -0
- package/README.md +164 -0
- package/action.yml +95 -0
- package/bin/ca.mjs +245 -0
- package/fixtures/build.mjs +178 -0
- package/package.json +12 -0
- package/scripts/gh-api.mjs +90 -0
- package/scripts-analyze.mjs +86 -0
- package/scripts-recompute.mjs +53 -0
- package/src/assertions.mjs +350 -0
- package/src/html-report.mjs +125 -0
- package/src/identity.mjs +105 -0
- package/src/markdown-report.mjs +81 -0
- package/src/report.mjs +183 -0
- package/src/runner-json.mjs +127 -0
- package/src/runner.mjs +83 -0
- package/src/select-runner.mjs +58 -0
- package/src/tap.mjs +83 -0
- package/src/verdict.mjs +168 -0
- package/src/verify.mjs +322 -0
- package/src/worktree.mjs +197 -0
- package/test/assertions.test.mjs +130 -0
- package/test/fixtures.test.mjs +41 -0
- package/test/runner-json.test.mjs +69 -0
- package/test/select-runner.test.mjs +78 -0
- package/test/verdict.test.mjs +113 -0
package/src/worktree.mjs
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Worktree + dependency plumbing.
|
|
3
|
+
*
|
|
4
|
+
* TWO HARD RULES.
|
|
5
|
+
*
|
|
6
|
+
* 1. NEVER touch the user's working tree. No `git stash`, no `git checkout` in their
|
|
7
|
+
* clone, no writes outside the work dir. A validation tool that can lose somebody's
|
|
8
|
+
* uncommitted work is dead on arrival, and this one runs over hundreds of commits
|
|
9
|
+
* unattended. Everything happens in detached worktrees under `.ca-work/`.
|
|
10
|
+
*
|
|
11
|
+
* 2. Worktrees are REUSED across commits. `git worktree add` copies the whole tree
|
|
12
|
+
* (~3,000 files in a mid-sized repo, ~20s); checking out a new commit inside an existing
|
|
13
|
+
* worktree only touches what differs. Over 300 commits that is the difference between
|
|
14
|
+
* a coffee and an afternoon.
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import { execFile } from 'node:child_process';
|
|
18
|
+
import { promisify } from 'node:util';
|
|
19
|
+
import { mkdir, rm, symlink, lstat, readlink, readdir, writeFile } from 'node:fs/promises';
|
|
20
|
+
import path from 'node:path';
|
|
21
|
+
|
|
22
|
+
const exec = promisify(execFile);
|
|
23
|
+
|
|
24
|
+
export async function git(repo, args, opts = {}) {
|
|
25
|
+
const { stdout } = await exec('git', ['-C', repo, ...args], { maxBuffer: 64 << 20, ...opts });
|
|
26
|
+
return stdout;
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
export async function ensureWorktree(repo, workDir, name, sha) {
|
|
30
|
+
const dir = path.join(workDir, name);
|
|
31
|
+
let exists = false;
|
|
32
|
+
try { await lstat(path.join(dir, '.git')); exists = true; } catch { /* first use */ }
|
|
33
|
+
|
|
34
|
+
if (!exists) {
|
|
35
|
+
await mkdir(workDir, { recursive: true });
|
|
36
|
+
await rm(dir, { recursive: true, force: true });
|
|
37
|
+
await git(repo, ['worktree', 'add', '--detach', '--force', dir, sha]);
|
|
38
|
+
} else {
|
|
39
|
+
// Reset hard, not checkout: a previous run transplanted a test file in here, and
|
|
40
|
+
// a dirty tree would make the next checkout fail or, worse, silently keep it.
|
|
41
|
+
await git(dir, ['checkout', '--detach', '--force', sha]);
|
|
42
|
+
await git(dir, ['reset', '--hard', sha]);
|
|
43
|
+
await git(dir, ['clean', '-fdq', '-e', 'node_modules']);
|
|
44
|
+
}
|
|
45
|
+
return dir;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
/**
|
|
49
|
+
* Build a node_modules inside the worktree.
|
|
50
|
+
*
|
|
51
|
+
* Third-party packages are symlinked straight to the target repo's copies — they are
|
|
52
|
+
* version-pinned content and sharing them saves an install per commit. WORKSPACE packages
|
|
53
|
+
* are NOT: those are symlinks into the repo's own source, and following them is exactly
|
|
54
|
+
* how arm B ends up loading today's code. Those get re-pointed at the worktree.
|
|
55
|
+
*
|
|
56
|
+
* `identity.mjs` verifies the result rather than trusting it, because this function is a
|
|
57
|
+
* heuristic over somebody else's directory layout and will eventually be wrong.
|
|
58
|
+
*/
|
|
59
|
+
/**
|
|
60
|
+
* Every node_modules directory in the repo, as paths relative to the root.
|
|
61
|
+
*
|
|
62
|
+
* npm workspaces HOIST most packages to the root — but not all. Anything with a native
|
|
63
|
+
* or version-conflicting dependency stays in its own workspace, and there is no way to
|
|
64
|
+
* know which from the outside.
|
|
65
|
+
*
|
|
66
|
+
* Earned in practice: a React Native workspace held 25 entries including `jest-expo`.
|
|
67
|
+
* Mirroring only the root gave every mobile commit
|
|
68
|
+
*
|
|
69
|
+
* ● Validation Error: Preset jest-expo not found.
|
|
70
|
+
*
|
|
71
|
+
* which the tool reported as INCONCLUSIVE for 25 cases. The fail-safe held — nothing was
|
|
72
|
+
* mislabelled BLIND — but a whole surface silently went unmeasured, which is its own kind
|
|
73
|
+
* of wrong answer.
|
|
74
|
+
*
|
|
75
|
+
* Discovered rather than configured: a hardcoded workspace list would be right for this
|
|
76
|
+
* repo and wrong for the next one.
|
|
77
|
+
*/
|
|
78
|
+
async function findNodeModulesDirs(repoRoot, maxDepth = 3) {
|
|
79
|
+
const found = [];
|
|
80
|
+
const walk = async (rel, depth) => {
|
|
81
|
+
const abs = path.join(repoRoot, rel);
|
|
82
|
+
let entries;
|
|
83
|
+
try { entries = await readdir(abs, { withFileTypes: true }); } catch { return; }
|
|
84
|
+
if (entries.some(e => e.name === 'node_modules')) found.push(rel);
|
|
85
|
+
if (depth >= maxDepth) return;
|
|
86
|
+
for (const e of entries) {
|
|
87
|
+
if (!e.isDirectory()) continue;
|
|
88
|
+
// Never descend INTO node_modules — its own nested copies belong to it.
|
|
89
|
+
if (e.name === 'node_modules' || e.name.startsWith('.')) continue;
|
|
90
|
+
await walk(path.join(rel, e.name), depth + 1);
|
|
91
|
+
}
|
|
92
|
+
};
|
|
93
|
+
await walk('', 0);
|
|
94
|
+
return found;
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export async function linkDependencies(repoRoot, worktreeDir, { force = false } = {}) {
|
|
98
|
+
const dirs = await findNodeModulesDirs(repoRoot);
|
|
99
|
+
let total = { linked: 0, repointed: 0, trees: 0 };
|
|
100
|
+
for (const rel of dirs) {
|
|
101
|
+
const r = await linkOneTree(path.join(repoRoot, rel), path.join(worktreeDir, rel), repoRoot, worktreeDir, force);
|
|
102
|
+
total.linked += r.linked; total.repointed += r.repointed; total.trees += r.linked ? 1 : 0;
|
|
103
|
+
}
|
|
104
|
+
return total;
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
async function linkOneTree(srcParent, dstParent, repoRoot, worktreeDir, force) {
|
|
108
|
+
const src = path.join(srcParent, 'node_modules');
|
|
109
|
+
const dst = path.join(dstParent, 'node_modules');
|
|
110
|
+
try { await lstat(src); } catch { return { linked: 0, repointed: 0, note: 'no node_modules here' }; }
|
|
111
|
+
|
|
112
|
+
// Worktrees are reused across commits and node_modules is excluded from `git clean`,
|
|
113
|
+
// so this only has to run once per worktree. Rebuilding ~1,300 symlinks per commit
|
|
114
|
+
// dominated the audit's runtime and changed nothing.
|
|
115
|
+
if (!force) { try { await lstat(dst); return { linked: 0, repointed: 0, note: 'reused' }; } catch { /* build it */ } }
|
|
116
|
+
|
|
117
|
+
await rm(dst, { recursive: true, force: true });
|
|
118
|
+
await mkdir(dst, { recursive: true });
|
|
119
|
+
|
|
120
|
+
let linked = 0, repointed = 0;
|
|
121
|
+
const repoReal = path.resolve(repoRoot);
|
|
122
|
+
|
|
123
|
+
const link = async (relEntry) => {
|
|
124
|
+
const from = path.join(src, relEntry);
|
|
125
|
+
const to = path.join(dst, relEntry);
|
|
126
|
+
let target = from;
|
|
127
|
+
try {
|
|
128
|
+
const st = await lstat(from);
|
|
129
|
+
if (st.isSymbolicLink()) {
|
|
130
|
+
const raw = await readlink(from);
|
|
131
|
+
const abs = path.resolve(path.dirname(from), raw);
|
|
132
|
+
if (abs.startsWith(repoReal + path.sep)) {
|
|
133
|
+
// A workspace package. Point it at the SAME relative path inside the worktree.
|
|
134
|
+
target = path.join(worktreeDir, path.relative(repoReal, abs));
|
|
135
|
+
repointed++;
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
} catch { return; }
|
|
139
|
+
await mkdir(path.dirname(to), { recursive: true });
|
|
140
|
+
await symlink(target, to).catch(() => {});
|
|
141
|
+
linked++;
|
|
142
|
+
};
|
|
143
|
+
|
|
144
|
+
for (const entry of await readdir(src)) {
|
|
145
|
+
if (entry.startsWith('@')) {
|
|
146
|
+
for (const scoped of await readdir(path.join(src, entry)).catch(() => [])) {
|
|
147
|
+
await link(path.join(entry, scoped));
|
|
148
|
+
}
|
|
149
|
+
} else {
|
|
150
|
+
await link(entry);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
return { linked, repointed };
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
/**
|
|
157
|
+
* Link the repo's gitignored env files into the worktree.
|
|
158
|
+
*
|
|
159
|
+
* `git worktree add` checks out TRACKED files only, so `.env` — gitignored by design —
|
|
160
|
+
* is absent. Every DB-backed test then sees no TEST_DATABASE_URL and skips, which is why
|
|
161
|
+
* 955 of 5,956 cases in a 300-commit audit reported SKIPPED even with a database
|
|
162
|
+
* running. The tool was measuring a subset of the suite and correctly saying so, but the
|
|
163
|
+
* subset was an artifact of the harness rather than of the repo.
|
|
164
|
+
*
|
|
165
|
+
* SYMLINKED, never copied. The file holds live credentials; a copy would leave a second
|
|
166
|
+
* one on disk in a temp directory whose cleanup is not guaranteed. A symlink gives the
|
|
167
|
+
* child process the same bytes and leaves nothing behind when the worktree is removed.
|
|
168
|
+
*/
|
|
169
|
+
export async function linkEnvFiles(repoRoot, worktreeDir) {
|
|
170
|
+
const linked = [];
|
|
171
|
+
for (const name of ['.env', '.env.local', '.env.test', '.env.test.local']) {
|
|
172
|
+
const src = path.join(repoRoot, name);
|
|
173
|
+
try { await lstat(src); } catch { continue; }
|
|
174
|
+
const dst = path.join(worktreeDir, name);
|
|
175
|
+
await rm(dst, { force: true }).catch(() => {});
|
|
176
|
+
await symlink(src, dst).catch(() => {});
|
|
177
|
+
linked.push(name);
|
|
178
|
+
}
|
|
179
|
+
return linked;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/** Put the fix's version of a test file onto the parent tree. The transplant. */
|
|
183
|
+
export async function transplant(repo, sha, relPath, worktreeDir) {
|
|
184
|
+
const content = await git(repo, ['show', `${sha}:${relPath}`]);
|
|
185
|
+
const dest = path.join(worktreeDir, relPath);
|
|
186
|
+
await mkdir(path.dirname(dest), { recursive: true });
|
|
187
|
+
await writeFile(dest, content);
|
|
188
|
+
return dest;
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
export async function removeWorktrees(repo, workDir) {
|
|
192
|
+
for (const name of ['fix', 'parent']) {
|
|
193
|
+
await git(repo, ['worktree', 'remove', '--force', path.join(workDir, name)]).catch(() => {});
|
|
194
|
+
}
|
|
195
|
+
await git(repo, ['worktree', 'prune']).catch(() => {});
|
|
196
|
+
await rm(workDir, { recursive: true, force: true }).catch(() => {});
|
|
197
|
+
}
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Case extraction, pinned against the four shapes that actually broke it.
|
|
3
|
+
*
|
|
4
|
+
* Each of these was measured on a real corpus of 400 test cases, and each one alone
|
|
5
|
+
* accounted for a double-digit share of "could not analyse". Together they took the
|
|
6
|
+
* unresolvable pile from 64% to 1.3%. None would be caught by a suite that only fed the
|
|
7
|
+
* analyser well-formed input, which is why they are here with the messy input attached.
|
|
8
|
+
*/
|
|
9
|
+
import { test } from 'node:test';
|
|
10
|
+
import assert from 'node:assert/strict';
|
|
11
|
+
import { extractCase, analyseCase } from '../src/assertions.mjs';
|
|
12
|
+
|
|
13
|
+
test('nested suites: the runner reports the CONCATENATED name, the source holds the inner one', () => {
|
|
14
|
+
const src = `
|
|
15
|
+
describe('auction client', () => {
|
|
16
|
+
it('AUCWEB-001 calls GET /auction/:vin', () => { assert.equal(a, 1); });
|
|
17
|
+
});`;
|
|
18
|
+
// Reported by the runner as describe-title + space + it-title.
|
|
19
|
+
const body = extractCase(src, 'auction client AUCWEB-001 calls GET /auction/:vin');
|
|
20
|
+
assert.ok(body, 'progressive reduction must find the inner title');
|
|
21
|
+
assert.match(body, /assert\.equal\(a, 1\)/);
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
test('an APOSTROPHE in the title: source holds a backslash the reported name does not', () => {
|
|
25
|
+
const src = `it('METER-038: usage on the user\\'s LOCAL today is counted', () => { assert.equal(n, 3); });`;
|
|
26
|
+
const body = extractCase(src, "METER-038: usage on the user's LOCAL today is counted");
|
|
27
|
+
assert.ok(body, "an escaped quote inside the literal must still match the unescaped reported name");
|
|
28
|
+
assert.match(body, /assert\.equal\(n, 3\)/);
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
test('JSX: a closing tag is NOT a regex literal', () => {
|
|
32
|
+
// `</div>` puts a slash after `<`. The usual "slash after a non-expression starts a
|
|
33
|
+
// regex" heuristic swallows the rest of the component, and the body is never found.
|
|
34
|
+
const src = `
|
|
35
|
+
it('hero renders', () => {
|
|
36
|
+
render(<div className="x">{name}</div>);
|
|
37
|
+
expect(screen.getByText('hi')).toBeTruthy();
|
|
38
|
+
});`;
|
|
39
|
+
const body = extractCase(src, 'hero renders');
|
|
40
|
+
assert.ok(body, 'a JSX body must brace-match cleanly');
|
|
41
|
+
assert.match(body, /getByText/);
|
|
42
|
+
});
|
|
43
|
+
|
|
44
|
+
test('braces inside strings and regexes do not unbalance the scan', () => {
|
|
45
|
+
const src = `
|
|
46
|
+
it('handles braces', () => {
|
|
47
|
+
const s = '{';
|
|
48
|
+
const re = /[{}]/;
|
|
49
|
+
assert.equal(f(s, re), '}');
|
|
50
|
+
});`;
|
|
51
|
+
const body = extractCase(src, 'handles braces');
|
|
52
|
+
assert.ok(body, 'a brace inside a string or a character class is not structure');
|
|
53
|
+
assert.match(body, /assert\.equal\(f\(s, re\)/);
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test('template-literal titles: every generated case resolves to the shared body', () => {
|
|
57
|
+
const src = `
|
|
58
|
+
for (const rel of PAGES) {
|
|
59
|
+
it(\`\${rel}: no internal repo paths\`, () => { expect(md).not.toMatch(BAD); });
|
|
60
|
+
}`;
|
|
61
|
+
const a = extractCase(src, 'app/page.tsx: no internal repo paths');
|
|
62
|
+
const b = extractCase(src, 'app/pricing/page.tsx: no internal repo paths');
|
|
63
|
+
assert.ok(a && b, 'a parameterised title must match by pattern, not by equality');
|
|
64
|
+
assert.equal(a, b, 'all generated cases share one body — that is correct, not a bug');
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
test('a wrong body is worse than none: a short remainder is refused', () => {
|
|
68
|
+
const src = `
|
|
69
|
+
it('alpha returns null', () => { assert.equal(x, null); });
|
|
70
|
+
it('beta returns null', () => { assert.equal(y, 0); });`;
|
|
71
|
+
// "returns null" alone is ambiguous between the two — reduction must not take it.
|
|
72
|
+
const body = extractCase(src, 'some suite that does not exist returns null');
|
|
73
|
+
assert.equal(body, null, 'an ambiguous short suffix must not resolve to an arbitrary case');
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
test('CONTROL ARM: naive equality-only extraction fails every one of these', () => {
|
|
77
|
+
// The implementation this replaced. Asserted to still get them wrong.
|
|
78
|
+
const naive = (src, name) => {
|
|
79
|
+
const esc = name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
80
|
+
return new RegExp(`\\b(?:test|it)\\s*\\(\\s*(['"\`])${esc}\\1`).test(src);
|
|
81
|
+
};
|
|
82
|
+
assert.equal(naive(`it('a b c', () => {});`, 'suite a b c'), false); // nested
|
|
83
|
+
assert.equal(naive(`it('u\\'s v', () => {});`, "u's v"), false); // apostrophe
|
|
84
|
+
assert.equal(naive('it(`${r}: x`, () => {});', 'p.tsx: x'), false); // template
|
|
85
|
+
});
|
|
86
|
+
|
|
87
|
+
test('the analyser reports UNKNOWN rather than a clean bill when it cannot tell', () => {
|
|
88
|
+
const r = analyseCase(`it('odd', () => { somethingEntirelyUnrecognised(); });`, 'odd');
|
|
89
|
+
assert.notEqual(r.verdict, 'strong', 'an unrecognised body must never read as strong');
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
test('a short helper name must not be accused of stubbing the unit under test', () => {
|
|
93
|
+
// FOUND ON A REAL RUN. The stub check substring-matched the name against every import
|
|
94
|
+
// path, so a two-letter helper `at` matched '@acme/core' — "acme" contains "at" — and
|
|
95
|
+
// every case in that file was reported as testing a stub. A wrong explanation is worse
|
|
96
|
+
// than none: it sends the reader to look at code that is fine.
|
|
97
|
+
const src = `
|
|
98
|
+
import { computeTier } from '@acme/core/tiers.js';
|
|
99
|
+
const at = (arr, i) => arr[i];
|
|
100
|
+
it('TIER-001: the tier changes the money', () => {
|
|
101
|
+
assert.equal(at(computeTier(x), 0), 1200);
|
|
102
|
+
});`;
|
|
103
|
+
const r = analyseCase(src, 'TIER-001: the tier changes the money');
|
|
104
|
+
assert.ok(!r.findings.some(f => f.id === 'stubbed-subject'),
|
|
105
|
+
'a local helper whose name merely appears inside an import path is not a stub');
|
|
106
|
+
});
|
|
107
|
+
|
|
108
|
+
test('a REAL stub is still caught — the tightening must not blind the check', () => {
|
|
109
|
+
const src = `
|
|
110
|
+
import { SLA_HOURS } from '../src/priority.mjs';
|
|
111
|
+
const priority = () => 1;
|
|
112
|
+
it('p is 1', () => { assert.equal(priority('HIGH-PRIORITY'), 1); });`;
|
|
113
|
+
const r = analyseCase(src, 'p is 1');
|
|
114
|
+
assert.ok(r.findings.some(f => f.id === 'stubbed-subject'),
|
|
115
|
+
'a stub matching the imported module basename must still be reported');
|
|
116
|
+
assert.equal(r.verdict, 'weak');
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
test('a declined commit produces NO pr comment — silence, not an accusation', async () => {
|
|
120
|
+
// It briefly posted "SKIPPED · No case in this branch fails without the change" onto a
|
|
121
|
+
// workflow-only PR. Technically true and completely misleading: that PR has no tests,
|
|
122
|
+
// so of course none discriminate. A comment reading as an accusation on a PR doing
|
|
123
|
+
// nothing wrong is worse than no comment.
|
|
124
|
+
const { prComment } = await import('../src/markdown-report.mjs');
|
|
125
|
+
const declined = { short: 'abc12345', verdict: 'SKIPPED', cases: [], note: 'no test file in the commit' };
|
|
126
|
+
assert.equal(prComment(declined), '', 'a noted (declined) result must render as empty');
|
|
127
|
+
|
|
128
|
+
const real = { short: 'abc12345', verdict: 'CAUGHT', cases: [{ verdict: 'CAUGHT', name: 'x', reason: 'y' }] };
|
|
129
|
+
assert.notEqual(prComment(real), '', 'a real verdict must still render');
|
|
130
|
+
});
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The tool's control arm. Eight repos where the right answer is known by construction.
|
|
3
|
+
*
|
|
4
|
+
* A green run here is the only reason to believe a verdict this tool prints about
|
|
5
|
+
* anybody else's code. It is also the thing that would catch the failure mode that
|
|
6
|
+
* matters most — a FALSE BLIND — because 01 and 07 must never come back BLIND.
|
|
7
|
+
*/
|
|
8
|
+
import { test } from 'node:test';
|
|
9
|
+
import assert from 'node:assert/strict';
|
|
10
|
+
import path from 'node:path';
|
|
11
|
+
import { buildFixtures } from '../fixtures/build.mjs';
|
|
12
|
+
import { verifyCommit } from '../src/verify.mjs';
|
|
13
|
+
import { removeWorktrees } from '../src/worktree.mjs';
|
|
14
|
+
|
|
15
|
+
const fixtures = buildFixtures();
|
|
16
|
+
|
|
17
|
+
for (const [name, f] of Object.entries(fixtures)) {
|
|
18
|
+
test(`${name} -> ${f.expect} (${f.why})`, async () => {
|
|
19
|
+
const workDir = path.join(f.dir, '.ca-work');
|
|
20
|
+
try {
|
|
21
|
+
const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
|
|
22
|
+
assert.equal(r.verdict, f.expect,
|
|
23
|
+
`expected ${f.expect}, got ${r.verdict}\n` +
|
|
24
|
+
r.cases.map(c => ` ${c.verdict} ${c.name} — ${c.reason}`).join('\n') +
|
|
25
|
+
(r.note ? `\n note: ${r.note}` : ''));
|
|
26
|
+
} finally {
|
|
27
|
+
await removeWorktrees(f.dir, workDir);
|
|
28
|
+
}
|
|
29
|
+
});
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
test('a FALSE BLIND is the failure that matters: no discriminating fixture may read BLIND', async () => {
|
|
33
|
+
for (const name of ['01-caught-value', '07-caught-mixed']) {
|
|
34
|
+
const f = fixtures[name];
|
|
35
|
+
const workDir = path.join(f.dir, '.ca-work-fb');
|
|
36
|
+
try {
|
|
37
|
+
const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
|
|
38
|
+
assert.notEqual(r.verdict, 'BLIND', `${name} was accused of being blind — this is the output that destroys trust in the tool`);
|
|
39
|
+
} finally { await removeWorktrees(f.dir, workDir); }
|
|
40
|
+
}
|
|
41
|
+
});
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The vitest/jest adapter, at the level where it can actually be wrong.
|
|
3
|
+
*
|
|
4
|
+
* WHAT THIS DOES AND DOES NOT COVER — stated because the gap matters.
|
|
5
|
+
*
|
|
6
|
+
* The node:test path has 9 end-to-end fixtures: real git repos, both arms, known verdict.
|
|
7
|
+
* These two runners do NOT, because a fixture would have to `npm install` vitest or jest
|
|
8
|
+
* per repo — slow, network-dependent, and version-drifting. The full two-arm path for
|
|
9
|
+
* each is verified on ONE real commit apiece — a vitest component commit and a jest
|
|
10
|
+
* React Native commit — which is a smoke test, not a control arm.
|
|
11
|
+
*
|
|
12
|
+
* So this file covers the part that is BOTH untested end-to-end AND most likely to be
|
|
13
|
+
* wrong: the text-matching in `classifyMessages`. node:test hands over `ERR_ASSERTION`;
|
|
14
|
+
* these two give formatted strings, so the adapter has to read prose. Every string below
|
|
15
|
+
* is a real shape emitted by vitest or jest, not an invented one.
|
|
16
|
+
*/
|
|
17
|
+
import { test } from 'node:test';
|
|
18
|
+
import assert from 'node:assert/strict';
|
|
19
|
+
import { classifyMessages } from '../src/runner-json.mjs';
|
|
20
|
+
|
|
21
|
+
const isDisagreement = m => classifyMessages(m).code === 'ERR_ASSERTION';
|
|
22
|
+
const isCannotRun = m => classifyMessages(m).code === 'ERR_TEST_FAILURE';
|
|
23
|
+
const isUnknown = m => classifyMessages(m).code === null;
|
|
24
|
+
|
|
25
|
+
test('vitest/chai disagreement is read as an assertion', () => {
|
|
26
|
+
assert.ok(isDisagreement(["AssertionError: expected 'Urgent' to deeply equal 'Normal'"]));
|
|
27
|
+
assert.ok(isDisagreement(['AssertionError: expected 40 to be 65 // Object.is equality']));
|
|
28
|
+
});
|
|
29
|
+
|
|
30
|
+
test('jest/expect disagreement is read as an assertion', () => {
|
|
31
|
+
assert.ok(isDisagreement(['expect(received).toBe(expected) // Object.is equality\n\nExpected: "Normal"\nReceived: "Urgent"']));
|
|
32
|
+
assert.ok(isDisagreement(['Error: expect(received).toEqual(expected)\n\n- Expected\n+ Received']));
|
|
33
|
+
});
|
|
34
|
+
|
|
35
|
+
test('a module that could not load is NOT an assertion', () => {
|
|
36
|
+
assert.ok(isCannotRun(["Error: Cannot find module '../src/newHelper' from 'lib/__tests__/x.test.tsx'"]));
|
|
37
|
+
assert.ok(isCannotRun(["SyntaxError: The requested module './bucket.js' does not provide an export named 'SEP'"]));
|
|
38
|
+
assert.ok(isCannotRun(['Error: Failed to resolve import "./notYet" from "lib/x.test.tsx". Does the file exist?']));
|
|
39
|
+
assert.ok(isCannotRun(['TypeError [ERR_UNKNOWN_FILE_EXTENSION]: Unknown file extension ".tsx"']));
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
test('CANNOT-RUN WINS when both shapes appear — the ordering that matters', () => {
|
|
43
|
+
// A failed module load frequently ALSO prints an expect() frame from the stack. If
|
|
44
|
+
// the disagreement patterns were checked first, this would read CAUGHT, and a commit
|
|
45
|
+
// whose test never ran would be credited with catching its bug.
|
|
46
|
+
const both = ['SyntaxError: does not provide an export named \'SEP\'\n at expect(received).toBe(expected)\n expected \'a\' to be \'b\''];
|
|
47
|
+
assert.ok(isCannotRun(both), 'a load failure that also prints an expect() frame must not read as a disagreement');
|
|
48
|
+
assert.notEqual(classifyMessages(both).code, 'ERR_ASSERTION');
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
test('an UNRECOGNISED message is unknown — never guessed into an assertion', () => {
|
|
52
|
+
// This is the design: an unrecognised shape costs COVERAGE (it reads INCONCLUSIVE),
|
|
53
|
+
// it cannot manufacture a finding. A drifting regex must fail safe, not fail loud.
|
|
54
|
+
assert.ok(isUnknown(['Something entirely unexpected happened in a custom matcher']));
|
|
55
|
+
assert.ok(isUnknown([]));
|
|
56
|
+
assert.ok(isUnknown(['']));
|
|
57
|
+
assert.notEqual(classifyMessages(['whatever']).code, 'ERR_ASSERTION');
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
test('CONTROL ARM: a naive "any failure is a disagreement" classifier gets these wrong', () => {
|
|
61
|
+
// TEST-001 — run the broken implementation and assert it still reproduces the bug.
|
|
62
|
+
const naive = msgs => (msgs && msgs.length ? 'ERR_ASSERTION' : null);
|
|
63
|
+
const loadFail = ["Error: Cannot find module '../src/newHelper'"];
|
|
64
|
+
assert.equal(naive(loadFail), 'ERR_ASSERTION'); // the naive one says CAUGHT
|
|
65
|
+
assert.equal(classifyMessages(loadFail).code, 'ERR_TEST_FAILURE'); // ours says it never ran
|
|
66
|
+
const weird = ['custom matcher blew up'];
|
|
67
|
+
assert.equal(naive(weird), 'ERR_ASSERTION');
|
|
68
|
+
assert.equal(classifyMessages(weird).code, null);
|
|
69
|
+
});
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Runner selection, over synthetic trees.
|
|
3
|
+
*
|
|
4
|
+
* Picking the WRONG runner is a silent failure: vitest run from the repo root finds no
|
|
5
|
+
* config and reports "No test files found", which reads INCONCLUSIVE — a whole surface
|
|
6
|
+
* disappears without ever saying why. So the walk-up has to be pinned.
|
|
7
|
+
*/
|
|
8
|
+
import { test } from 'node:test';
|
|
9
|
+
import assert from 'node:assert/strict';
|
|
10
|
+
import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
|
11
|
+
import { tmpdir } from 'node:os';
|
|
12
|
+
import path from 'node:path';
|
|
13
|
+
import { selectRunner } from '../src/select-runner.mjs';
|
|
14
|
+
|
|
15
|
+
function tree(spec) {
|
|
16
|
+
const root = mkdtempSync(path.join(tmpdir(), 'ca-sel-'));
|
|
17
|
+
for (const [rel, content] of Object.entries(spec)) {
|
|
18
|
+
const p = path.join(root, rel);
|
|
19
|
+
mkdirSync(path.dirname(p), { recursive: true });
|
|
20
|
+
writeFileSync(p, content);
|
|
21
|
+
}
|
|
22
|
+
return root;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
test('a vitest config in the workspace wins, and names that workspace as cwd', async () => {
|
|
26
|
+
const root = tree({
|
|
27
|
+
'package.json': '{"workspaces":["apps/*"]}',
|
|
28
|
+
'apps/web/vitest.config.ts': 'export default {}',
|
|
29
|
+
'apps/web/lib/__tests__/x.test.tsx': '',
|
|
30
|
+
});
|
|
31
|
+
try {
|
|
32
|
+
const r = await selectRunner(root, 'apps/web/lib/__tests__/x.test.tsx');
|
|
33
|
+
assert.equal(r.flavour, 'vitest');
|
|
34
|
+
assert.equal(r.pkgDir, 'apps/web', 'must run FROM apps/web — vitest resolves config relative to cwd');
|
|
35
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
test('a jest config in the workspace wins', async () => {
|
|
39
|
+
const root = tree({ 'package.json': '{}', 'apps/mobile/jest.config.js': '', 'apps/mobile/__tests__/y.test.ts': '' });
|
|
40
|
+
try {
|
|
41
|
+
const r = await selectRunner(root, 'apps/mobile/__tests__/y.test.ts');
|
|
42
|
+
assert.equal(r.flavour, 'jest');
|
|
43
|
+
assert.equal(r.pkgDir, 'apps/mobile');
|
|
44
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test('root tests fall to node:test when nothing else claims them', async () => {
|
|
48
|
+
const root = tree({ 'package.json': '{"scripts":{"test":"node --test tests/*.test.js"}}', 'tests/a.test.js': '' });
|
|
49
|
+
try {
|
|
50
|
+
const r = await selectRunner(root, 'tests/a.test.js');
|
|
51
|
+
assert.equal(r.flavour, 'node');
|
|
52
|
+
assert.equal(r.pkgDir, '');
|
|
53
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
54
|
+
});
|
|
55
|
+
|
|
56
|
+
test('a repo that declares NOTHING falls through to node:test, not to a guess', async () => {
|
|
57
|
+
// The safe default: node:test either works, or reports a load failure, which reads
|
|
58
|
+
// INCONCLUSIVE. Guessing vitest here would produce "No test files found" — the same
|
|
59
|
+
// silent disappearance this test exists to prevent.
|
|
60
|
+
const root = tree({ 'tests/a.test.js': '' });
|
|
61
|
+
try {
|
|
62
|
+
assert.equal((await selectRunner(root, 'tests/a.test.js')).flavour, 'node');
|
|
63
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
64
|
+
});
|
|
65
|
+
|
|
66
|
+
test('a workspace config beats a root test script — the nearest owner wins', async () => {
|
|
67
|
+
// The root script often delegates (`npm run test --workspaces`), while the config
|
|
68
|
+
// that actually governs the file sits in the package. Nearest wins, walking up.
|
|
69
|
+
const root = tree({
|
|
70
|
+
'package.json': '{"scripts":{"test":"node --test"}}',
|
|
71
|
+
'apps/web/vitest.config.ts': '',
|
|
72
|
+
'apps/web/x.test.tsx': '',
|
|
73
|
+
});
|
|
74
|
+
try {
|
|
75
|
+
const r = await selectRunner(root, 'apps/web/x.test.tsx');
|
|
76
|
+
assert.equal(r.flavour, 'vitest');
|
|
77
|
+
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
78
|
+
});
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The decision logic, exhaustively, with no git and no subprocesses.
|
|
3
|
+
*
|
|
4
|
+
* Every case here is a shape observed in a real run, not an invented one.
|
|
5
|
+
*/
|
|
6
|
+
import { test } from 'node:test';
|
|
7
|
+
import assert from 'node:assert/strict';
|
|
8
|
+
import { classify, reduceRuns, rollUp, isDisagreement, CAUGHT, BLIND, NON_DISCRIMINATING, INCONCLUSIVE, FLAKY, SKIPPED } from '../src/verdict.mjs';
|
|
9
|
+
|
|
10
|
+
const pass = { status: 'pass' };
|
|
11
|
+
const assertFail = { status: 'fail', code: 'ERR_ASSERTION', errorName: 'AssertionError', message: "expected 1, got 2" };
|
|
12
|
+
const errFail = { status: 'fail', code: 'ERR_TEST_FAILURE', errorName: 'SyntaxError', message: "does not provide an export named 'TITLE_SEPARATOR'" };
|
|
13
|
+
|
|
14
|
+
test('CAUGHT: green on the fix, assertion-red on the parent', () => {
|
|
15
|
+
assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
|
|
16
|
+
});
|
|
17
|
+
|
|
18
|
+
test('NON-DISCRIMINATING: green on the fix AND green on the parent', () => {
|
|
19
|
+
assert.equal(classify({ armA: pass, armB: pass }).verdict, NON_DISCRIMINATING);
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
test('a case green on both arms is never called BLIND — a regression guard is indistinguishable', () => {
|
|
23
|
+
// Observed in practice: 3 of one commit's 4 cases were deliberate regression guards.
|
|
24
|
+
assert.notEqual(classify({ armA: pass, armB: pass }).verdict, BLIND);
|
|
25
|
+
});
|
|
26
|
+
|
|
27
|
+
test('a non-assertion failure on the parent is INCONCLUSIVE, never CAUGHT — the test never ran', () => {
|
|
28
|
+
const r = classify({ armA: pass, armB: errFail });
|
|
29
|
+
assert.equal(r.verdict, INCONCLUSIVE);
|
|
30
|
+
assert.match(r.reason, /never ran/);
|
|
31
|
+
});
|
|
32
|
+
|
|
33
|
+
test('exit code alone would have gotten that wrong', () => {
|
|
34
|
+
// Both of these are "the process exited non-zero". Only one is a verdict.
|
|
35
|
+
assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
|
|
36
|
+
assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE);
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
test('unproven module identity withholds the verdict even when arm B passed', () => {
|
|
40
|
+
// THE false-BLIND guard. Arm B green + identity unproven is precisely the symlink
|
|
41
|
+
// trap that made this tool report a false BLIND before it had an identity gate.
|
|
42
|
+
const r = classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'resolved OUTSIDE the worktree' } });
|
|
43
|
+
assert.equal(r.verdict, INCONCLUSIVE);
|
|
44
|
+
assert.notEqual(r.verdict, BLIND);
|
|
45
|
+
});
|
|
46
|
+
|
|
47
|
+
test('a case that is red on the fix is INCONCLUSIVE — the commit does not stand up', () => {
|
|
48
|
+
assert.equal(classify({ armA: assertFail, armB: assertFail }).verdict, INCONCLUSIVE);
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
test('a case missing from the parent run is INCONCLUSIVE, not BLIND', () => {
|
|
52
|
+
assert.equal(classify({ armA: pass, armB: null }).verdict, INCONCLUSIVE);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
test('isDisagreement is narrow: an unknown error name is not guessed into CAUGHT', () => {
|
|
56
|
+
assert.equal(isDisagreement({ status: 'fail', errorName: 'WeirdCustomError' }), false);
|
|
57
|
+
assert.equal(isDisagreement({ status: 'fail', errorName: 'JestAssertionError' }), true);
|
|
58
|
+
});
|
|
59
|
+
|
|
60
|
+
test('FLAKY: runs that disagree between CAUGHT and BLIND', () => {
|
|
61
|
+
const r = reduceRuns([{ verdict: CAUGHT }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]);
|
|
62
|
+
assert.equal(r.verdict, FLAKY);
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
test('an INCONCLUSIVE run does not erase a decided one, but is disclosed', () => {
|
|
66
|
+
const r = reduceRuns([{ verdict: CAUGHT, reason: 'assertion' }, { verdict: INCONCLUSIVE, reason: 'timeout' }, { verdict: CAUGHT, reason: 'assertion' }]);
|
|
67
|
+
assert.equal(r.verdict, CAUGHT);
|
|
68
|
+
assert.match(r.reason, /1 of 3 runs inconclusive/);
|
|
69
|
+
});
|
|
70
|
+
|
|
71
|
+
test('rollUp: one discriminating case carries the commit, guards and all', () => {
|
|
72
|
+
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]), CAUGHT);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
test('rollUp: a SKIPPED case blocks BLIND — it might have been the discriminating one', () => {
|
|
76
|
+
// Observed: all four tests written FOR a bug were { skip: SKIP } on a host with no
|
|
77
|
+
// TEST_DATABASE_URL, while older cases in the same file ran and did not discriminate.
|
|
78
|
+
// Calling that blind judges the commit on the tests NOT written for it.
|
|
79
|
+
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
|
|
80
|
+
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
test('rollUp: BLIND only when every case ran and NOT ONE discriminated', () => {
|
|
84
|
+
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }]), BLIND);
|
|
85
|
+
// one case could not be judged -> the commit cannot be called blind
|
|
86
|
+
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: INCONCLUSIVE }]), INCONCLUSIVE);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test('CONTROL ARM: a verdict engine that only read exit codes fails these', () => {
|
|
90
|
+
// TEST-001 — the old, naive implementation, executed here, asserted to still be wrong.
|
|
91
|
+
const naive = ({ armB }) => (armB && armB.status === 'fail' ? CAUGHT : BLIND);
|
|
92
|
+
assert.equal(naive({ armB: errFail }), CAUGHT); // it says CAUGHT...
|
|
93
|
+
assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE); // ...we say no
|
|
94
|
+
assert.equal(naive({ armB: pass }), BLIND); // and calls every regression guard blind // and it cannot see
|
|
95
|
+
assert.equal(classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'x' } }).verdict, INCONCLUSIVE);
|
|
96
|
+
});
|
|
97
|
+
|
|
98
|
+
test('a CAUGHT on new code is still CAUGHT — the verdict does not lie, the CLAIM narrows', async () => {
|
|
99
|
+
// Issue #3. On a feature the base lacks the code entirely, so essentially any test
|
|
100
|
+
// touching it fails there — `expected 0 to be greater than 0` is a real AssertionError
|
|
101
|
+
// that says nothing about whether the test is well aimed. Reclassifying it would be
|
|
102
|
+
// dishonest (it IS a disagreement); counting it as evidence would be worse.
|
|
103
|
+
const { commitKind } = await import('../src/verify.mjs');
|
|
104
|
+
|
|
105
|
+
assert.equal(commitKind('feat(web): add seoTitle', false).newCode, true,
|
|
106
|
+
'a feature is new code whatever its diff looks like');
|
|
107
|
+
assert.equal(commitKind('fix(core): a hyphen decided a title', false).newCode, false,
|
|
108
|
+
'a fix that changes lines is a repair, and its CAUGHT is strong evidence');
|
|
109
|
+
assert.equal(commitKind('fix(api): add a missing export', true).newCode, true,
|
|
110
|
+
'a fix whose source diff only ADDS may still be reading something that was absent');
|
|
111
|
+
assert.equal(commitKind('no conventional prefix at all', false).kind, 'unknown',
|
|
112
|
+
'an unrecognised subject must not be guessed into fix or feature');
|
|
113
|
+
});
|