control-arm 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.github/workflows/ci.yml +82 -0
- package/DESIGN.md +161 -0
- package/LICENSE +21 -0
- package/README.md +164 -0
- package/action.yml +95 -0
- package/bin/ca.mjs +245 -0
- package/fixtures/build.mjs +178 -0
- package/package.json +12 -0
- package/scripts/gh-api.mjs +90 -0
- package/scripts-analyze.mjs +86 -0
- package/scripts-recompute.mjs +53 -0
- package/src/assertions.mjs +350 -0
- package/src/html-report.mjs +125 -0
- package/src/identity.mjs +105 -0
- package/src/markdown-report.mjs +81 -0
- package/src/report.mjs +183 -0
- package/src/runner-json.mjs +127 -0
- package/src/runner.mjs +83 -0
- package/src/select-runner.mjs +58 -0
- package/src/tap.mjs +83 -0
- package/src/verdict.mjs +168 -0
- package/src/verify.mjs +322 -0
- package/src/worktree.mjs +197 -0
- package/test/assertions.test.mjs +130 -0
- package/test/fixtures.test.mjs +41 -0
- package/test/runner-json.test.mjs +69 -0
- package/test/select-runner.test.mjs +78 -0
- package/test/verdict.test.mjs +113 -0
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The PR-comment format: the same findings, compressed to what fits above the fold.
|
|
3
|
+
*
|
|
4
|
+
* A reviewer reads this inline on a pull request, so it is per-CASE and not per-commit:
|
|
5
|
+
* the useful sentence is "these two cases cannot pass without your change", and the
|
|
6
|
+
* useful warning is "this one errored rather than asserted, so it proves nothing".
|
|
7
|
+
*/
|
|
8
|
+
export function prComment(r, { repoName = '.' } = {}) {
|
|
9
|
+
// A DECLINED commit says NOTHING, and an empty string is how it says it.
|
|
10
|
+
//
|
|
11
|
+
// `r.note` is set when the tool refused to judge — a test-only commit, a build-only
|
|
12
|
+
// change, no test at all. There is no finding to report and no accusation to make, so
|
|
13
|
+
// the caller writes an empty file and the comment step skips.
|
|
14
|
+
//
|
|
15
|
+
// It did not, briefly, and posted "### control-arm — SKIPPED · No case in this branch
|
|
16
|
+
// fails without the change" onto a workflow-only PR. Technically true and completely
|
|
17
|
+
// misleading: that PR has no tests, so of course none of them discriminate. A comment
|
|
18
|
+
// that reads as an accusation on a PR doing nothing wrong is worse than no comment.
|
|
19
|
+
//
|
|
20
|
+
// The decision belongs here, where the note lives, and not in a shell scraping stdout
|
|
21
|
+
// for a VERDICT line that a declined run never prints.
|
|
22
|
+
if (r.note) return '';
|
|
23
|
+
const G = { CAUGHT: '✅', 'NON-DISCRIMINATING': '⚪️', INCONCLUSIVE: '⚠️', FLAKY: '🔁', SKIPPED: '⏭' };
|
|
24
|
+
const disc = r.cases.filter(c => c.verdict === 'CAUGHT');
|
|
25
|
+
const inc = r.cases.filter(c => c.verdict === 'INCONCLUSIVE');
|
|
26
|
+
const L = [];
|
|
27
|
+
const head = r.verdict !== 'CAUGHT' ? r.verdict
|
|
28
|
+
: r.newCode ? 'new code — these tests cannot be judged this way'
|
|
29
|
+
: 'this branch is proven';
|
|
30
|
+
L.push(`### \`control-arm\` — ${head}`);
|
|
31
|
+
L.push('');
|
|
32
|
+
if (disc.length && r.newCode) {
|
|
33
|
+
// The claim CAUGHT is allowed to make depends on whether there was a bug. On new
|
|
34
|
+
// code the base lacks the thing entirely, so essentially any test touching it
|
|
35
|
+
// fails there. Saying "cannot pass without this change" would be true and
|
|
36
|
+
// worthless — it restates that the code is new.
|
|
37
|
+
L.push(`**${disc.length} of ${r.cases.length} tests fail without this change** — but this ${r.kind === 'feature' ? 'is a feature' : 'commit only adds code'}, so the base does not have it at all.`);
|
|
38
|
+
L.push('');
|
|
39
|
+
L.push(`That is expected, and it is weak evidence: on new code almost any test that reads the new thing fails on the base, whether or not it is well aimed. A test asserting \`1 === 1\` in the same file would NOT show up here; one that merely reads the new field would.`);
|
|
40
|
+
} else L.push(disc.length
|
|
41
|
+
? `**${disc.length} of ${r.cases.length} tests here genuinely catch this change.** Run against the code as it was before this PR, they fail; with the change, they pass.`
|
|
42
|
+
: `**No test here fails without this change.** Every test in this PR is already green on the code it is meant to fix — so none of them would have caught it.`);
|
|
43
|
+
L.push('');
|
|
44
|
+
// PLAIN WORDS IN THE TABLE HEAD. "case" means "test" to almost nobody outside
|
|
45
|
+
// testing, and "the base" is version-control jargon used as a bare column header.
|
|
46
|
+
// The reader of a PR comment is usually the author, mid-review, not a specialist.
|
|
47
|
+
L.push('| | test | what it did on the code WITHOUT this change |');
|
|
48
|
+
L.push('|---|---|---|');
|
|
49
|
+
for (const c of r.cases) {
|
|
50
|
+
const why = c.verdict === 'CAUGHT'
|
|
51
|
+
? `**failed** — ${(c.reason || '').replace(/\s+/g, ' ').slice(0, 84)}`
|
|
52
|
+
: c.verdict === 'INCONCLUSIVE'
|
|
53
|
+
? `_could not run there — ${(c.reason || '').replace(/\s+/g, ' ').slice(0, 74)}_`
|
|
54
|
+
: c.verdict === 'SKIPPED' ? '_skipped by the test runner_'
|
|
55
|
+
: c.verdict === 'FLAKY' ? '_different answers on repeated runs — trust neither_'
|
|
56
|
+
: '_passed too — so it is not what catches this bug. Often deliberate: a test that guards something else._';
|
|
57
|
+
L.push(`| ${G[c.verdict] || '·'} | ${c.name.replace(/\|/g, '\\|').slice(0, 96)} | ${why.replace(/\|/g, '\\|')} |`);
|
|
58
|
+
}
|
|
59
|
+
L.push('');
|
|
60
|
+
if (inc.length) {
|
|
61
|
+
// The single most important idea in the tool, and previously its most opaque
|
|
62
|
+
// sentence. Say the mechanism, not the category.
|
|
63
|
+
L.push(`> ⚠️ **${inc.length} test${inc.length > 1 ? 's' : ''} could not run at all** on the older code — ${inc.length > 1 ? 'they' : 'it'} crashed instead of disagreeing (a missing import, usually).`);
|
|
64
|
+
L.push('>');
|
|
65
|
+
L.push('> That still shows up red, which looks like proof but is not: a test that never executed cannot tell you whether it would have caught anything.');
|
|
66
|
+
}
|
|
67
|
+
L.push('');
|
|
68
|
+
L.push('<details><summary>What this check does</summary>');
|
|
69
|
+
L.push('');
|
|
70
|
+
L.push('It takes the tests in this PR and runs them against the code **as it was before your change**.');
|
|
71
|
+
L.push('');
|
|
72
|
+
L.push('- ✅ **failed there** — the test genuinely catches this. It would have gone red on the old code.');
|
|
73
|
+
L.push('- ⚪️ **passed there too** — this test is not what catches this change. That is often correct: a test guarding something else *should* stay green.');
|
|
74
|
+
L.push('- ⚠️ **could not run there** — it crashed rather than disagreeing, so it proves nothing either way.');
|
|
75
|
+
L.push('');
|
|
76
|
+
L.push('A green test suite cannot tell these apart. That is the whole point of the check.');
|
|
77
|
+
L.push('</details>');
|
|
78
|
+
L.push('');
|
|
79
|
+
L.push(`<sub>\`ca verify ${r.short} --against <base> --repo ${repoName}\` · verified it loaded the old code, not the new</sub>`);
|
|
80
|
+
return L.join('\n');
|
|
81
|
+
}
|
package/src/report.mjs
ADDED
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Output. Deliberately separate from every decision: `verdict.mjs` returns data, this
|
|
3
|
+
* turns data into text, and nothing here may change an outcome.
|
|
4
|
+
*
|
|
5
|
+
* The house rule this obeys: never round a verdict UP. INCONCLUSIVE is printed as
|
|
6
|
+
* INCONCLUSIVE and counted in its own column, because a tool that quietly folds "could
|
|
7
|
+
* not run" into "caught it" is reporting a number nobody can act on.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
import { CAUGHT, BLIND, NON_DISCRIMINATING, INCONCLUSIVE, FLAKY, SKIPPED } from './verdict.mjs';
|
|
11
|
+
|
|
12
|
+
export const MARK = { ok: '✓', no: '✗', hm: '?' };
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* One glyph per verdict, and they carry meaning rather than decorate.
|
|
16
|
+
*
|
|
17
|
+
* Single-width characters on purpose: an emoji is two columns wide in most terminals and
|
|
18
|
+
* silently breaks every aligned column after it. Markdown output (the PR comment, the
|
|
19
|
+
* issue body) uses emoji instead, where width does not matter and colour does.
|
|
20
|
+
*
|
|
21
|
+
* ⚠ replaced ? for INCONCLUSIVE. A question mark reads as "unknown, probably fine"; the
|
|
22
|
+
* verdict actually means "I could not run this, so do not count it either way", which is
|
|
23
|
+
* a thing the reader must notice rather than skim past.
|
|
24
|
+
*/
|
|
25
|
+
const GLYPH = {
|
|
26
|
+
[CAUGHT]: '✓', [BLIND]: '✗', [NON_DISCRIMINATING]: '–', [INCONCLUSIVE]: '⚠', [FLAKY]: '~', [SKIPPED]: '·',
|
|
27
|
+
};
|
|
28
|
+
const ORDER = [CAUGHT, BLIND, FLAKY, INCONCLUSIVE, SKIPPED];
|
|
29
|
+
const CASE_ORDER = [CAUGHT, NON_DISCRIMINATING, FLAKY, INCONCLUSIVE, SKIPPED];
|
|
30
|
+
|
|
31
|
+
export function renderVerify(r) {
|
|
32
|
+
const L = [];
|
|
33
|
+
L.push('');
|
|
34
|
+
L.push(` ${r.short} ${r.subject}`);
|
|
35
|
+
L.push(` ${r.date} · ${r.testFiles.length} test file(s) · ${r.sourceFiles.length} source file(s) changed`);
|
|
36
|
+
if (r.note) { L.push(` ${MARK.hm} ${r.note}`); L.push(''); return L.join('\n'); }
|
|
37
|
+
|
|
38
|
+
const idn = r.cases.find(c => /identity unproven/.test(c.reason || ''));
|
|
39
|
+
L.push(idn ? ` ${MARK.no} module identity NOT proven — every verdict below is withheld`
|
|
40
|
+
: ` ${MARK.ok} module identity verified (arm B resolved inside the parent worktree)`);
|
|
41
|
+
L.push('');
|
|
42
|
+
// The legend belongs HERE more than on the audit summary. `verify` is the command
|
|
43
|
+
// someone runs first, and "– no-discrim." means nothing to a reader seeing it once.
|
|
44
|
+
L.push(' LEGEND ✓ fails without the fix (this test works) – green either way (a guard looks like this)');
|
|
45
|
+
L.push(' ⚠ could not run on the old code ~ flaky · skipped by the runner');
|
|
46
|
+
L.push('');
|
|
47
|
+
|
|
48
|
+
const byFile = new Map();
|
|
49
|
+
for (const c of r.cases) { if (!byFile.has(c.file)) byFile.set(c.file, []); byFile.get(c.file).push(c); }
|
|
50
|
+
for (const [file, cases] of byFile) {
|
|
51
|
+
L.push(` ${file}`);
|
|
52
|
+
for (const c of cases) {
|
|
53
|
+
const label = c.verdict === CAUGHT ? 'CATCHES IT '
|
|
54
|
+
: c.verdict === NON_DISCRIMINATING ? 'green either way'
|
|
55
|
+
: c.verdict === FLAKY ? 'FLAKY '
|
|
56
|
+
: c.verdict === SKIPPED ? 'skipped '
|
|
57
|
+
: 'COULD NOT RUN ';
|
|
58
|
+
L.push(` ${GLYPH[c.verdict]} ${label} ${c.name}`);
|
|
59
|
+
// The assertion note, where there is one, says something the generic reason
|
|
60
|
+
// cannot: WHY this case could not have caught the bug. Prefer it.
|
|
61
|
+
const detail = c.why || c.reason;
|
|
62
|
+
if (detail && c.verdict !== SKIPPED) L.push(` └ ${String(detail).replace(/\s+/g, ' ').slice(0, 160)}`);
|
|
63
|
+
}
|
|
64
|
+
L.push('');
|
|
65
|
+
}
|
|
66
|
+
const n = v => r.cases.filter(c => c.verdict === v).length;
|
|
67
|
+
if (r.stillOpen) {
|
|
68
|
+
const m = { repaired: '↻ REPAIRED SINCE', open: '‼ STILL OPEN TODAY', unknown: '⚠ cannot tell' }[r.stillOpen.status];
|
|
69
|
+
L.push(` ${m} — ${r.stillOpen.reason}`);
|
|
70
|
+
L.push('');
|
|
71
|
+
}
|
|
72
|
+
// WHAT CAUGHT IS ALLOWED TO CLAIM depends on whether there was a bug to catch.
|
|
73
|
+
// On new code the base lacks the thing entirely, so a failing test there is expected
|
|
74
|
+
// rather than evidence. Same verdict, a much weaker sentence.
|
|
75
|
+
const plain = r.verdict === CAUGHT && r.newCode
|
|
76
|
+
? `${n(CAUGHT)} test${n(CAUGHT) === 1 ? ' fails' : 's fail'} on the base — but this commit ADDS code, so the base lacks it entirely. That is expected, not evidence the tests are well aimed.`
|
|
77
|
+
: r.verdict === CAUGHT
|
|
78
|
+
? `${n(CAUGHT)} test${n(CAUGHT) === 1 ? '' : 's'} here would have caught this bug. The rest are guards or could not run.`
|
|
79
|
+
: r.verdict === BLIND
|
|
80
|
+
? `NOTHING here would have caught this bug — every test is green on the broken code.`
|
|
81
|
+
: r.verdict === FLAKY ? `A test gave different answers on repeated runs. Do not trust either.`
|
|
82
|
+
: `Could not judge this commit. See the reason above — it is not a pass or a fail.`;
|
|
83
|
+
const qualifier = r.verdict === CAUGHT && r.newCode
|
|
84
|
+
? ` (${r.kind === 'feature' ? 'a feature' : 'adds code, no source deletions'} — weak evidence)` : '';
|
|
85
|
+
L.push(` VERDICT ${GLYPH[r.verdict] ?? ''} ${r.verdict}${qualifier}`);
|
|
86
|
+
L.push(` ${plain}`);
|
|
87
|
+
L.push(` ${n(CAUGHT)} catch it · ${n(NON_DISCRIMINATING)} green either way · ${n(FLAKY)} flaky · ${n(INCONCLUSIVE)} could not run`);
|
|
88
|
+
L.push('');
|
|
89
|
+
return L.join('\n');
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
function bar(count, total, width = 26) {
|
|
93
|
+
if (!total) return ' '.repeat(width);
|
|
94
|
+
const filled = Math.round((count / total) * width);
|
|
95
|
+
return '█'.repeat(filled) + '░'.repeat(width - filled);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
export function renderAudit(results, meta) {
|
|
99
|
+
const L = [];
|
|
100
|
+
const commitTally = Object.fromEntries(ORDER.map(v => [v, results.filter(r => r.verdict === v).length]));
|
|
101
|
+
const allCases = results.flatMap(r => r.cases);
|
|
102
|
+
const caseTally = Object.fromEntries([...ORDER, NON_DISCRIMINATING].map(v => [v, allCases.filter(c => c.verdict === v).length]));
|
|
103
|
+
|
|
104
|
+
L.push('');
|
|
105
|
+
L.push(' ┌─ ca audit ──────────────────────────────────────────────────────────────┐');
|
|
106
|
+
L.push(` since ${meta.since} · ${meta.matched} commits matched · ${meta.eligible} judgeable · ${meta.n} drawn at random (seed ${meta.seed})`);
|
|
107
|
+
L.push(` ${(meta.seconds / 60).toFixed(1)} min · ${(meta.seconds / Math.max(meta.n, 1)).toFixed(1)}s per commit`);
|
|
108
|
+
L.push(' └─────────────────────────────────────────────────────────────────────────┘');
|
|
109
|
+
L.push('');
|
|
110
|
+
L.push(' LEGEND ✓ caught ✗ blind – ran, did not discriminate ⚠ could not run ~ flaky');
|
|
111
|
+
L.push('');
|
|
112
|
+
L.push(' BY COMMIT (a commit is CAUGHT if ANY of its cases discriminates)');
|
|
113
|
+
for (const v of ORDER) {
|
|
114
|
+
const c = commitTally[v];
|
|
115
|
+
if (!c && v === SKIPPED) continue;
|
|
116
|
+
L.push(` ${(GLYPH[v] + ' ' + v).padEnd(16)} ${bar(c, meta.n)} ${String(c).padStart(4)} ${((c / Math.max(meta.n, 1)) * 100).toFixed(1).padStart(5)}%`);
|
|
117
|
+
}
|
|
118
|
+
L.push('');
|
|
119
|
+
L.push(` BY CASE (${allCases.length} individual test cases — a regression guard reads as non-discriminating, which is correct for it)`);
|
|
120
|
+
for (const v of CASE_ORDER) {
|
|
121
|
+
const c = caseTally[v];
|
|
122
|
+
if (!c && v === SKIPPED) continue;
|
|
123
|
+
L.push(` ${(GLYPH[v] + ' ' + v).padEnd(16)} ${bar(c, allCases.length)} ${String(c).padStart(4)} ${((c / Math.max(allCases.length, 1)) * 100).toFixed(1).padStart(5)}%`);
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const strongCaught = results.filter(r => r.verdict === CAUGHT && !r.newCode).length;
|
|
127
|
+
const decided = strongCaught + commitTally[BLIND];
|
|
128
|
+
L.push('');
|
|
129
|
+
// NEW CODE IS NOT EVIDENCE. A feature's base lacks the thing entirely, so essentially
|
|
130
|
+
// any test touching it fails there. Counting those in the headline lets the number
|
|
131
|
+
// drift upward for free on a feature-heavy sample.
|
|
132
|
+
const newCodeCaught = results.filter(r => r.verdict === CAUGHT && r.newCode).length;
|
|
133
|
+
if (newCodeCaught) {
|
|
134
|
+
L.push(` ${newCodeCaught} CAUGHT commit(s) only ADD code — the base lacks it entirely, so a failing`);
|
|
135
|
+
L.push(' test there is expected rather than evidence. Excluded from the ratio below.');
|
|
136
|
+
L.push('');
|
|
137
|
+
}
|
|
138
|
+
L.push(' THE NUMBER THAT MATTERS');
|
|
139
|
+
L.push(` Of ${decided} commits where the instrument could answer at all,`);
|
|
140
|
+
L.push(` ${strongCaught} shipped a test that would have caught the bug — ${decided ? ((strongCaught / decided) * 100).toFixed(1) : '—'}%`);
|
|
141
|
+
L.push(` ${commitTally[BLIND]} shipped a test that was green on the broken code — ${decided ? ((commitTally[BLIND] / decided) * 100).toFixed(1) : '—'}%`);
|
|
142
|
+
L.push(` ${commitTally[INCONCLUSIVE]} could not be judged, and are excluded from that ratio rather than assumed.`);
|
|
143
|
+
L.push('');
|
|
144
|
+
|
|
145
|
+
const blind = results.filter(r => r.verdict === BLIND);
|
|
146
|
+
if (blind.length) {
|
|
147
|
+
// ARM C SPLIT — the only line in this report that is a TICKET rather than a fact.
|
|
148
|
+
// A raw BLIND count is close to useless as a headline: on a 300-commit run it was
|
|
149
|
+
// 13, of which 3 were already repaired and 2 unjudgeable. Reporting the count
|
|
150
|
+
// alone hands somebody thirteen tickets, eight of which waste their afternoon.
|
|
151
|
+
const open = blind.filter(r => r.stillOpen?.status === 'open');
|
|
152
|
+
const repaired = blind.filter(r => r.stillOpen?.status === 'repaired');
|
|
153
|
+
const cannot = blind.filter(r => !r.stillOpen || r.stillOpen.status === 'unknown');
|
|
154
|
+
|
|
155
|
+
L.push(' STILL OPEN TODAY — the only rows that are work');
|
|
156
|
+
if (open.length) for (const r of open) L.push(` ‼ ${r.short} ${r.date} ${r.subject.slice(0, 82)}`);
|
|
157
|
+
else L.push(' (none — every blind finding was repaired later or cannot be judged)');
|
|
158
|
+
L.push('');
|
|
159
|
+
if (repaired.length) {
|
|
160
|
+
L.push(` ↻ REPAIRED SINCE — true of the commit, already fixed in the tree (${repaired.length})`);
|
|
161
|
+
for (const r of repaired.slice(0, 12)) L.push(` ${r.short} ${r.subject.slice(0, 80)}`);
|
|
162
|
+
L.push('');
|
|
163
|
+
}
|
|
164
|
+
if (cannot.length) {
|
|
165
|
+
L.push(` ⚠ CANNOT TELL — the test file no longer exists, or will not run at HEAD (${cannot.length})`);
|
|
166
|
+
for (const r of cannot.slice(0, 12)) L.push(` ${r.short} ${r.subject.slice(0, 80)}`);
|
|
167
|
+
L.push('');
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
const why = new Map();
|
|
171
|
+
for (const c of allCases.filter(c => c.verdict === INCONCLUSIVE)) {
|
|
172
|
+
const key = String(c.reason || '').replace(/\s+/g, ' ')
|
|
173
|
+
.replace(/'[^']*'/g, "'…'").replace(/["`][^"`]*["`]/g, '…')
|
|
174
|
+
.replace(/\d+/g, 'N').slice(0, 92);
|
|
175
|
+
why.set(key, (why.get(key) || 0) + 1);
|
|
176
|
+
}
|
|
177
|
+
if (why.size) {
|
|
178
|
+
L.push(' WHY INCONCLUSIVE (the honest denominator — not folded into the ratio above)');
|
|
179
|
+
for (const [k, v] of [...why].sort((a, b) => b[1] - a[1]).slice(0, 12)) L.push(` ${String(v).padStart(4)} ${k}`);
|
|
180
|
+
L.push('');
|
|
181
|
+
}
|
|
182
|
+
return L.join('\n');
|
|
183
|
+
}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* vitest / jest runner.
|
|
3
|
+
*
|
|
4
|
+
* Both emit the SAME JSON shape (jest's, which vitest adopted), so one adapter covers
|
|
5
|
+
* `apps/web` (145 fix commits) and `apps/mobile` (101) — 246 of the 267 commits the
|
|
6
|
+
* node:test runner had to decline.
|
|
7
|
+
*
|
|
8
|
+
* { testResults: [ { assertionResults: [ { fullName, status, failureMessages[] } ] } ] }
|
|
9
|
+
*
|
|
10
|
+
* THE HARD PART IS THE SAME ONE AS EVERYWHERE ELSE: telling a test that DISAGREED from a
|
|
11
|
+
* test that could not RUN. node:test hands that over as `code: 'ERR_ASSERTION'`. These two
|
|
12
|
+
* do not — there is only `failureMessages`, an array of formatted strings. So the adapter
|
|
13
|
+
* has to read the message, which is exactly the kind of text-matching this tool tells
|
|
14
|
+
* other people not to do.
|
|
15
|
+
*
|
|
16
|
+
* It is acceptable here for one reason: the fallback is INCONCLUSIVE, not CAUGHT. An
|
|
17
|
+
* unrecognised message means "I could not tell", which withholds the verdict. A regex that
|
|
18
|
+
* drifts costs coverage; it cannot manufacture a finding. That asymmetry is the whole
|
|
19
|
+
* design and it is why the patterns below are deliberately narrow.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { spawn } from 'node:child_process';
|
|
23
|
+
import { readFile, rm } from 'node:fs/promises';
|
|
24
|
+
import path from 'node:path';
|
|
25
|
+
|
|
26
|
+
/** An assertion library disagreeing, as vitest/chai and jest/expect phrase it. */
|
|
27
|
+
const DISAGREEMENT = [
|
|
28
|
+
/^AssertionError\b/m,
|
|
29
|
+
/^JestAssertionError\b/m,
|
|
30
|
+
/\bexpected .* to (?:be|equal|deeply equal|eql|contain|match|have)\b/,
|
|
31
|
+
/^\s*expect\(.*\)\..*\n/m, // jest's "expect(received).toBe(expected)" header
|
|
32
|
+
/\bError: expect\(/,
|
|
33
|
+
];
|
|
34
|
+
/** Could-not-run, which must beat the patterns above if both appear. */
|
|
35
|
+
const CANNOT_RUN = [
|
|
36
|
+
/\b(?:SyntaxError|ReferenceError)\b/,
|
|
37
|
+
/Cannot find (?:module|package)/i,
|
|
38
|
+
/ERR_MODULE_NOT_FOUND/,
|
|
39
|
+
/Failed to (?:load|resolve) (?:url|import)/i,
|
|
40
|
+
/does not provide an export named/,
|
|
41
|
+
/Unknown file extension/,
|
|
42
|
+
/No test (?:files )?found/i,
|
|
43
|
+
];
|
|
44
|
+
|
|
45
|
+
export function classifyMessages(msgs) {
|
|
46
|
+
const text = (msgs || []).join('\n');
|
|
47
|
+
if (!text.trim()) return { errorName: 'Unknown', code: null, message: null };
|
|
48
|
+
// Order matters: a module that failed to load often ALSO prints an expect() frame.
|
|
49
|
+
for (const re of CANNOT_RUN) if (re.test(text)) {
|
|
50
|
+
return { errorName: (text.match(/\b(SyntaxError|ReferenceError|TypeError|Error)\b/) || [, 'LoadError'])[1], code: 'ERR_TEST_FAILURE', message: text.split('\n')[0].slice(0, 200) };
|
|
51
|
+
}
|
|
52
|
+
for (const re of DISAGREEMENT) if (re.test(text)) {
|
|
53
|
+
return { errorName: 'AssertionError', code: 'ERR_ASSERTION', message: text.split('\n').find(l => l.trim()) ?.slice(0, 200) ?? null };
|
|
54
|
+
}
|
|
55
|
+
// Unrecognised: say so. INCONCLUSIVE is the honest answer, and it is the safe one.
|
|
56
|
+
return { errorName: 'Unrecognised', code: null, message: text.split('\n')[0].slice(0, 200) };
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
function run(cmd, args, { cwd, timeoutMs, env }) {
|
|
60
|
+
return new Promise((resolve) => {
|
|
61
|
+
const child = spawn(cmd, args, { cwd, env, shell: false });
|
|
62
|
+
let stdout = '', stderr = '', killed = false;
|
|
63
|
+
const t = setTimeout(() => { killed = true; child.kill('SIGKILL'); }, timeoutMs);
|
|
64
|
+
child.stdout.on('data', d => { stdout += d; });
|
|
65
|
+
child.stderr.on('data', d => { stderr += d; });
|
|
66
|
+
child.on('close', code => { clearTimeout(t); resolve({ code, stdout, stderr, killed }); });
|
|
67
|
+
child.on('error', e => { clearTimeout(t); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
68
|
+
});
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
function childEnv() {
|
|
72
|
+
const env = { ...process.env, NO_COLOR: '1', CI: '1', FORCE_COLOR: '0' };
|
|
73
|
+
for (const k of Object.keys(env)) if (k.startsWith('NODE_TEST')) delete env[k];
|
|
74
|
+
delete env.NODE_OPTIONS;
|
|
75
|
+
return env;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* `flavour` is 'vitest' or 'jest'. `pkgDir` is the workspace the test belongs to
|
|
80
|
+
* (apps/web), NOT the repo root — both tools resolve their config relative to cwd.
|
|
81
|
+
*/
|
|
82
|
+
export function jsonRunner(flavour) {
|
|
83
|
+
return {
|
|
84
|
+
name: flavour,
|
|
85
|
+
async execute({ worktreeDir, relTestPath, pkgDir, timeoutMs = 180_000 }) {
|
|
86
|
+
const cwd = path.join(worktreeDir, pkgDir || '.');
|
|
87
|
+
const rel = pkgDir ? path.relative(pkgDir, relTestPath) : relTestPath;
|
|
88
|
+
const outFile = path.join(cwd, `.ca-${flavour}-${process.pid}.json`);
|
|
89
|
+
const args = flavour === 'vitest'
|
|
90
|
+
? ['vitest', 'run', rel, '--reporter=json', '--outputFile', outFile]
|
|
91
|
+
: ['jest', rel, '--json', '--outputFile', outFile, '--ci'];
|
|
92
|
+
|
|
93
|
+
const r = await run('npx', ['--no-install', ...args], { cwd, timeoutMs, env: childEnv() });
|
|
94
|
+
if (r.killed) { await rm(outFile, { force: true }); return { ok: false, loadFailure: 'timeout', cases: [], raw: r }; }
|
|
95
|
+
|
|
96
|
+
let report;
|
|
97
|
+
try { report = JSON.parse(await readFile(outFile, 'utf8')); }
|
|
98
|
+
catch {
|
|
99
|
+
const why = (r.stderr + r.stdout).match(/\b(Cannot find module[^\n]*|SyntaxError[^\n]*|ERR_MODULE_NOT_FOUND[^\n]*|No test files found[^\n]*|Failed to (?:load|resolve)[^\n]*)/);
|
|
100
|
+
await rm(outFile, { force: true });
|
|
101
|
+
return { ok: false, loadFailure: why ? why[1].slice(0, 220) : `${flavour} produced no JSON report (exit ${r.code})`, cases: [], raw: r };
|
|
102
|
+
}
|
|
103
|
+
await rm(outFile, { force: true });
|
|
104
|
+
|
|
105
|
+
const cases = [];
|
|
106
|
+
for (const file of report.testResults || []) {
|
|
107
|
+
// A file that failed to compile has no assertionResults, only a message.
|
|
108
|
+
if ((file.assertionResults || []).length === 0 && file.message) {
|
|
109
|
+
return { ok: false, loadFailure: String(file.message).split('\n')[0].slice(0, 220), cases: [], raw: r };
|
|
110
|
+
}
|
|
111
|
+
for (const a of file.assertionResults || []) {
|
|
112
|
+
const status = a.status === 'passed' ? 'pass' : a.status === 'failed' ? 'fail' : 'skip';
|
|
113
|
+
cases.push({
|
|
114
|
+
name: a.fullName || a.title,
|
|
115
|
+
status,
|
|
116
|
+
...(status === 'fail' ? classifyMessages(a.failureMessages) : { errorName: null, code: null, message: null }),
|
|
117
|
+
});
|
|
118
|
+
}
|
|
119
|
+
}
|
|
120
|
+
if (cases.length === 0) return { ok: false, loadFailure: 'no cases reported', cases: [], raw: r };
|
|
121
|
+
return { ok: true, cases, raw: r };
|
|
122
|
+
},
|
|
123
|
+
};
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
export const vitest = jsonRunner('vitest');
|
|
127
|
+
export const jest = jsonRunner('jest');
|
package/src/runner.mjs
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Run a test file and return normalised cases.
|
|
3
|
+
*
|
|
4
|
+
* One runner today (node:test). The seam is deliberate: `verdict.mjs` never sees runner
|
|
5
|
+
* output, only the normalised shape, so adding vitest/jest/pytest means adding a file
|
|
6
|
+
* here and changing no decision logic.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { spawn } from 'node:child_process';
|
|
10
|
+
import { parseTap, isFileLevelFailure } from './tap.mjs';
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* The child's environment.
|
|
14
|
+
*
|
|
15
|
+
* `NODE_TEST_*` MUST be stripped. When `ca`'s own suite runs under `node --test`, the
|
|
16
|
+
* runner exports NODE_TEST_CONTEXT into every child; a nested `node --test` sees it,
|
|
17
|
+
* decides it is a subprocess of a parent runner, and switches from TAP to the v8
|
|
18
|
+
* serialiser. The TAP parser then finds no cases and every fixture comes back
|
|
19
|
+
* INCONCLUSIVE — which is exactly how this tool failed its own control arm on the first
|
|
20
|
+
* green run, 2026-09-23. It works perfectly from a shell and lies inside a test.
|
|
21
|
+
*
|
|
22
|
+
* Any tool that shells out to a test runner from inside a test runner has this bug.
|
|
23
|
+
*/
|
|
24
|
+
function childEnv() {
|
|
25
|
+
const env = { ...process.env, NO_COLOR: '1', CI: '1', FORCE_COLOR: '0' };
|
|
26
|
+
for (const k of Object.keys(env)) if (k.startsWith('NODE_TEST')) delete env[k];
|
|
27
|
+
delete env.NODE_OPTIONS; // a --require/--import from the parent would load into arm B
|
|
28
|
+
return env;
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
function run(cmd, args, { cwd, timeoutMs }) {
|
|
32
|
+
return new Promise((resolve) => {
|
|
33
|
+
const child = spawn(cmd, args, { cwd, env: childEnv() });
|
|
34
|
+
let stdout = '', stderr = '', killed = false;
|
|
35
|
+
const timer = setTimeout(() => { killed = true; child.kill('SIGKILL'); }, timeoutMs);
|
|
36
|
+
child.stdout.on('data', d => { stdout += d; });
|
|
37
|
+
child.stderr.on('data', d => { stderr += d; });
|
|
38
|
+
child.on('close', (code) => { clearTimeout(timer); resolve({ code, stdout, stderr, killed }); });
|
|
39
|
+
child.on('error', (e) => { clearTimeout(timer); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
40
|
+
});
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
export const nodeTest = {
|
|
44
|
+
name: 'node:test',
|
|
45
|
+
matches: (repoRoot, pkg) => !pkg?.scripts?.test?.includes('vitest') || true,
|
|
46
|
+
async execute({ worktreeDir, relTestPath, timeoutMs = 120_000 }) {
|
|
47
|
+
const r = await run(process.execPath, ['--test', '--test-reporter=tap', relTestPath],
|
|
48
|
+
{ cwd: worktreeDir, timeoutMs });
|
|
49
|
+
|
|
50
|
+
if (r.killed) return { ok: false, loadFailure: 'timeout', cases: [], raw: r };
|
|
51
|
+
|
|
52
|
+
const cases = parseTap(r.stdout);
|
|
53
|
+
// Drop the FILE-level TAP row (node:test emits one named after the path) without
|
|
54
|
+
// touching real cases.
|
|
55
|
+
//
|
|
56
|
+
// FALSE BLIND — the first thing a random hand-audit caught. This filter used to
|
|
57
|
+
// be `!c.name.includes('/')`, to drop path-shaped rows. A test named
|
|
58
|
+
// label parsing / separator handling
|
|
59
|
+
// contains a slash, so it was silently dropped from BOTH arms. It was the only
|
|
60
|
+
// discriminating case in its commit, which then came back BLIND — the tool
|
|
61
|
+
// accusing a correct test of being decoration, which is the one output that ends
|
|
62
|
+
// trust in it.
|
|
63
|
+
//
|
|
64
|
+
// Match the path we were GIVEN instead of guessing from the shape of a name.
|
|
65
|
+
const isFileRow = (name) => name === relTestPath || name.endsWith('/' + relTestPath) || name.endsWith(relTestPath.split('/').pop());
|
|
66
|
+
const real = cases.filter(c => !isFileRow(c.name));
|
|
67
|
+
|
|
68
|
+
// No individual cases at all, or a single file-path-shaped failure: the file never
|
|
69
|
+
// loaded. This is the SyntaxError/module-not-found path and it must NOT be handed
|
|
70
|
+
// to the verdict engine as "every case failed".
|
|
71
|
+
if (real.length === 0) {
|
|
72
|
+
const why = (r.stderr + r.stdout).match(/\b(SyntaxError|ReferenceError|TypeError|Error \[ERR_MODULE_NOT_FOUND\]|ERR_MODULE_NOT_FOUND|Cannot find (?:module|package))\b[^\n]*/);
|
|
73
|
+
return {
|
|
74
|
+
ok: false,
|
|
75
|
+
loadFailure: why ? why[0].slice(0, 220) : (cases.length ? (cases[0].message || 'file-level failure') : 'no cases reported'),
|
|
76
|
+
cases: [], raw: r,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
return { ok: true, cases: real, raw: r };
|
|
80
|
+
},
|
|
81
|
+
};
|
|
82
|
+
|
|
83
|
+
export const RUNNERS = [nodeTest];
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Which runner owns this test file, and from which directory must it be run?
|
|
3
|
+
*
|
|
4
|
+
* A monorepo has several: `tests/**` may be node:test at the repo root while
|
|
5
|
+
* `apps/web/**` is vitest run from apps/web and `apps/mobile/**` is jest run from
|
|
6
|
+
* apps/mobile. vitest and jest both resolve their CONFIG relative to cwd, so running
|
|
7
|
+
* them from the repo root silently picks up the wrong project or none at all.
|
|
8
|
+
*
|
|
9
|
+
* Detection is CONFIG-BASED, walking up from the test file, rather than a hardcoded path
|
|
10
|
+
* map — the map would be right for this repo and wrong for the next one. A repo that
|
|
11
|
+
* declares nothing falls through to node:test, which is the safe default: it either works
|
|
12
|
+
* or it reports a load failure, and a load failure reads INCONCLUSIVE.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import { readFile, access } from 'node:fs/promises';
|
|
16
|
+
import path from 'node:path';
|
|
17
|
+
|
|
18
|
+
const CONFIGS = [
|
|
19
|
+
{ flavour: 'vitest', files: ['vitest.config.ts', 'vitest.config.js', 'vitest.config.mjs', 'vite.config.ts', 'vite.config.js'] },
|
|
20
|
+
{ flavour: 'jest', files: ['jest.config.js', 'jest.config.ts', 'jest.config.mjs', 'jest.config.json'] },
|
|
21
|
+
];
|
|
22
|
+
|
|
23
|
+
const exists = async p => { try { await access(p); return true; } catch { return false; } };
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* @returns {{flavour: 'node'|'vitest'|'jest', pkgDir: string}} pkgDir is relative to the
|
|
27
|
+
* worktree root and is the cwd the runner must be invoked from.
|
|
28
|
+
*/
|
|
29
|
+
export async function selectRunner(worktreeRoot, relTestPath) {
|
|
30
|
+
let dir = path.dirname(relTestPath);
|
|
31
|
+
|
|
32
|
+
while (true) {
|
|
33
|
+
const abs = path.join(worktreeRoot, dir);
|
|
34
|
+
|
|
35
|
+
for (const { flavour, files } of CONFIGS) {
|
|
36
|
+
for (const f of files) {
|
|
37
|
+
if (await exists(path.join(abs, f))) return { flavour, pkgDir: dir === '.' ? '' : dir };
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
// A package.json whose `test` script names a runner is the second signal. Checked
|
|
42
|
+
// AFTER config files because a workspace root often has a `test` script that
|
|
43
|
+
// delegates, while the config file sits in the package that actually owns the tests.
|
|
44
|
+
const pkgPath = path.join(abs, 'package.json');
|
|
45
|
+
if (await exists(pkgPath)) {
|
|
46
|
+
try {
|
|
47
|
+
const pkg = JSON.parse(await readFile(pkgPath, 'utf8'));
|
|
48
|
+
const script = pkg.scripts?.test || '';
|
|
49
|
+
if (/\bvitest\b/.test(script)) return { flavour: 'vitest', pkgDir: dir === '.' ? '' : dir };
|
|
50
|
+
if (/\bjest\b/.test(script)) return { flavour: 'jest', pkgDir: dir === '.' ? '' : dir };
|
|
51
|
+
if (/node\s+--test|\bnode:test\b/.test(script)) return { flavour: 'node', pkgDir: dir === '.' ? '' : dir };
|
|
52
|
+
} catch { /* unparseable package.json is not a signal */ }
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
if (dir === '.' || dir === '' || dir === path.sep) return { flavour: 'node', pkgDir: '' };
|
|
56
|
+
dir = path.dirname(dir);
|
|
57
|
+
}
|
|
58
|
+
}
|
package/src/tap.mjs
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Parse node:test TAP 13 into normalised case results.
|
|
3
|
+
*
|
|
4
|
+
* Normalised shape, which `verdict.mjs` is written against and other runners must also
|
|
5
|
+
* produce: { name, status: 'pass'|'fail', errorName, code, message }
|
|
6
|
+
*
|
|
7
|
+
* We take the TAP reporter rather than the spec reporter because the YAML block carries
|
|
8
|
+
* `code:` and `name:`, which is the ONLY thing separating "the test disagreed" from "the
|
|
9
|
+
* test could not run". The human-readable reporters drop it.
|
|
10
|
+
*/
|
|
11
|
+
|
|
12
|
+
const OK = /^\s*ok\s+\d+\s+-\s+(.*)$/;
|
|
13
|
+
const NOT_OK = /^\s*not ok\s+\d+\s+-\s+(.*)$/;
|
|
14
|
+
|
|
15
|
+
export function parseTap(stdout) {
|
|
16
|
+
const lines = stdout.split('\n');
|
|
17
|
+
const cases = [];
|
|
18
|
+
let current = null;
|
|
19
|
+
// TAP folds a multi-line error into a `|-` block of indented lines. We used to record
|
|
20
|
+
// the literal string '<multiline>' for those, which threw away the actual assertion
|
|
21
|
+
// message — and the assertion message is the single most useful thing in the output.
|
|
22
|
+
// It surfaced as "<multiline>" in PR comments where the real text belonged.
|
|
23
|
+
let collecting = null;
|
|
24
|
+
|
|
25
|
+
for (const line of lines) {
|
|
26
|
+
const ok = line.match(OK);
|
|
27
|
+
const notOk = line.match(NOT_OK);
|
|
28
|
+
if (ok || notOk) {
|
|
29
|
+
// Trailing TAP directives are not part of the name.
|
|
30
|
+
const raw = (ok || notOk)[1];
|
|
31
|
+
const name = raw.replace(/\s*#\s*(SKIP|TODO).*$/i, '').trim();
|
|
32
|
+
const skipped = /#\s*SKIP/i.test(raw);
|
|
33
|
+
current = { name, status: skipped ? 'skip' : (ok ? 'pass' : 'fail'), errorName: null, code: null, message: null };
|
|
34
|
+
cases.push(current);
|
|
35
|
+
continue;
|
|
36
|
+
}
|
|
37
|
+
if (collecting !== null) {
|
|
38
|
+
// Indented continuation lines belong to the block; anything at or above the
|
|
39
|
+
// opening indent ends it.
|
|
40
|
+
const indent = line.match(/^\s*/)[0].length;
|
|
41
|
+
if (line.trim() && indent > collecting.indent) {
|
|
42
|
+
if (!collecting.done) {
|
|
43
|
+
const t = line.trim();
|
|
44
|
+
if (t) { collecting.target.message = t.slice(0, 220); collecting.done = true; }
|
|
45
|
+
}
|
|
46
|
+
continue;
|
|
47
|
+
}
|
|
48
|
+
collecting = null;
|
|
49
|
+
}
|
|
50
|
+
if (!current || current.status !== 'fail') continue;
|
|
51
|
+
// Flat scalars inside the YAML block. Deliberately not a YAML parser: we want four
|
|
52
|
+
// fields and a malformed block must degrade to "unknown", which reads INCONCLUSIVE.
|
|
53
|
+
let m;
|
|
54
|
+
if ((m = line.match(/^\s*code:\s*'?([^'\n]+)'?\s*$/))) current.code = m[1].trim();
|
|
55
|
+
else if ((m = line.match(/^\s*name:\s*'?([^'\n]+)'?\s*$/))) current.errorName = m[1].trim();
|
|
56
|
+
// `expected:` and `actual:` are the ONLY fields that say WHAT the disagreement
|
|
57
|
+
// was. "Expected values to be strictly equal:" is a category, not a finding — the
|
|
58
|
+
// reader wants "expected 1, got 2", and only these two carry it.
|
|
59
|
+
else if ((m = line.match(/^\s*expected:\s*(.+?)\s*$/))) current.expected = m[1];
|
|
60
|
+
else if ((m = line.match(/^\s*actual:\s*(.+?)\s*$/))) current.actual = m[1];
|
|
61
|
+
else if ((m = line.match(/^\s*error:\s*'([^']*)'\s*$/))) current.message = m[1].trim();
|
|
62
|
+
else if ((m = line.match(/^(\s*)error:\s*\|-\s*$/))) {
|
|
63
|
+
collecting = { indent: m[1].length, target: current, done: false };
|
|
64
|
+
}
|
|
65
|
+
}
|
|
66
|
+
for (const c of cases) {
|
|
67
|
+
if (c.expected !== undefined && c.actual !== undefined) {
|
|
68
|
+
c.message = `expected ${c.expected}, got ${c.actual}`;
|
|
69
|
+
}
|
|
70
|
+
delete c.expected; delete c.actual;
|
|
71
|
+
}
|
|
72
|
+
return cases;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* A whole-file load failure. node:test reports this as a `not ok` for the FILE path with
|
|
77
|
+
* no individual cases, which must not be mistaken for every case failing by assertion.
|
|
78
|
+
*/
|
|
79
|
+
export function isFileLevelFailure(cases, testFile) {
|
|
80
|
+
if (cases.length !== 1) return false;
|
|
81
|
+
const only = cases[0];
|
|
82
|
+
return only.status === 'fail' && (only.name.includes('/') || only.name.endsWith('.mjs') || only.name.endsWith('.js') || only.name === testFile);
|
|
83
|
+
}
|