control-arm 1.0.0 → 1.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/DESIGN.md +37 -21
- package/README.md +205 -24
- package/bin/ca.mjs +53 -6
- package/package.json +11 -4
- package/src/assertions.mjs +37 -1
- package/src/children.mjs +55 -0
- package/src/cli-spec.mjs +50 -0
- package/src/contract.mjs +193 -0
- package/src/identity.mjs +18 -5
- package/src/markdown-report.mjs +5 -1
- package/src/report.mjs +14 -1
- package/src/runner-json.mjs +47 -21
- package/src/runner.mjs +21 -8
- package/src/sample-warning.mjs +69 -0
- package/src/select-runner.mjs +7 -0
- package/src/tap.mjs +0 -10
- package/src/verify.mjs +397 -9
- package/src/worktree.mjs +11 -1
package/src/cli-spec.mjs
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The CLI's surface, as data.
|
|
3
|
+
*
|
|
4
|
+
* It lives here rather than inline in `bin/ca.mjs` so a test can read the REAL table
|
|
5
|
+
* instead of a copy of it. A copy drifts, and a test asserting against a copy passes
|
|
6
|
+
* while the README documents a command the binary has never heard of.
|
|
7
|
+
*
|
|
8
|
+
* That is not hypothetical. `ca probe` was the first command in this README from the
|
|
9
|
+
* day it was written. It does not exist. It survived a full rewrite, a review pass, a
|
|
10
|
+
* GitHub release, a Marketplace listing and an npm publish, and reached the npm front
|
|
11
|
+
* page — because nothing in a 48-test suite knew the README existed.
|
|
12
|
+
*/
|
|
13
|
+
|
|
14
|
+
/** Subcommands. `bin/ca.mjs` builds its dispatch table from these names. */
|
|
15
|
+
export const COMMANDS = Object.freeze({
|
|
16
|
+
doctor: 'can this repo be measured at all?',
|
|
17
|
+
verify: 'one commit, per-case verdicts',
|
|
18
|
+
audit: 'sample N fix commits and report the CAUGHT rate',
|
|
19
|
+
issues: 'file a ticket per open BLIND commit',
|
|
20
|
+
});
|
|
21
|
+
|
|
22
|
+
export const COMMAND_NAMES = Object.freeze(Object.keys(COMMANDS));
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Every flag the binary reads. `takesValue` matters to the README check: a valued flag
|
|
26
|
+
* documented without a value is a copy-paste that silently swallows the next argument.
|
|
27
|
+
*/
|
|
28
|
+
export const FLAGS = Object.freeze({
|
|
29
|
+
repo: { takesValue: true, help: 'the repository to measure (default: cwd)' },
|
|
30
|
+
against: { takesValue: true, help: 'compare against a base ref — turns verify into a PR check' },
|
|
31
|
+
runs: { takesValue: true, help: 'repeat each case N times to expose flakes' },
|
|
32
|
+
timeout: { takesValue: true, help: 'per-run timeout in ms' },
|
|
33
|
+
work: { takesValue: true, help: 'where to put the throwaway worktrees' },
|
|
34
|
+
n: { takesValue: true, help: 'how many fix commits to sample' },
|
|
35
|
+
since: { takesValue: true, help: "how far back to look, e.g. '6 years' (default: 12 months)" },
|
|
36
|
+
grep: { takesValue: true, help: 'commit-subject filter' },
|
|
37
|
+
seed: { takesValue: true, help: 'make the random draw reproducible' },
|
|
38
|
+
out: { takesValue: true, help: 'write the audit to a CSV' },
|
|
39
|
+
html: { takesValue: true, help: 'write an HTML report' },
|
|
40
|
+
json: { takesValue: false, help: 'findings-contract v1 envelope on stdout; audit additionally accepts `--json <path>` to write its summary to a file' },
|
|
41
|
+
'fail-on-blind': { takesValue: false, help: 'exit 1 when a commit is BLIND (off by default — BLIND is a warning, not a blocker)' },
|
|
42
|
+
keep: { takesValue: false, help: 'leave the worktrees behind for inspection' },
|
|
43
|
+
'pr-comment': { takesValue: false, help: 'print the PR comment markdown and nothing else' },
|
|
44
|
+
apply: { takesValue: false, help: 'actually file the issues, rather than printing them' },
|
|
45
|
+
'include-test-fixes': { takesValue: false, help: 'do not skip commits whose subject is fix(test)' },
|
|
46
|
+
});
|
|
47
|
+
|
|
48
|
+
export const FLAG_NAMES = Object.freeze(Object.keys(FLAGS));
|
|
49
|
+
|
|
50
|
+
export const BIN = 'ca';
|
package/src/contract.mjs
ADDED
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The findings contract, v1 — see the findings-contract SPEC.
|
|
3
|
+
*
|
|
4
|
+
* This is the second tool to conform, and the clauses it was chosen to test are C-005
|
|
5
|
+
* (`skipped`) and C-006 (`confidence`), because this tool already refuses to answer.
|
|
6
|
+
* `INCONCLUSIVE` has always been the rule that a red test which never RAN proves nothing;
|
|
7
|
+
* `skipped` is that rule with a schema.
|
|
8
|
+
*
|
|
9
|
+
* The mapping that matters, and it is not one-to-one:
|
|
10
|
+
*
|
|
11
|
+
* commit CAUGHT -> no finding. Nothing is wrong.
|
|
12
|
+
* commit BLIND -> a finding, severity `warn`
|
|
13
|
+
* commit INCONCLUSIVE -> NOT a finding. It goes to `skipped`, with the reason.
|
|
14
|
+
* case NON-DISCRIM. -> never a finding. It is a fact about a guard, not a fault.
|
|
15
|
+
* case FLAKY -> a finding. Runs disagreed, which is a real defect.
|
|
16
|
+
*
|
|
17
|
+
* `BLIND` is `warn` and not `blocker` for the same reason `unreached` has no blockers:
|
|
18
|
+
* C-004 asks whether a build fails NOW, and a test that cannot catch a bug breaks
|
|
19
|
+
* nothing today. It is the absence of protection. `--fail-on-blind` remains this tool's
|
|
20
|
+
* own opinion, named separately, exactly as the spec now requires.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { CAUGHT, BLIND, INCONCLUSIVE, FLAKY, SKIPPED } from './verdict.mjs';
|
|
24
|
+
import { createRequire } from 'node:module';
|
|
25
|
+
|
|
26
|
+
// Read the real version rather than hardcoding it: a hardcoded '1.1.0' drifted the moment
|
|
27
|
+
// the next release bumped package.json, and the contract advertises a version to consumers.
|
|
28
|
+
const { version } = createRequire(import.meta.url)('../package.json');
|
|
29
|
+
|
|
30
|
+
export const BLOCKER = 'blocker', WARN = 'warn', INFO = 'info';
|
|
31
|
+
|
|
32
|
+
const BLIND_COMMIT = 'CA-001';
|
|
33
|
+
const FLAKY_CASE = 'CA-002';
|
|
34
|
+
const WEAK_ON_NEW = 'CA-003';
|
|
35
|
+
const DOCTOR_BLOCK = 'CA-010';
|
|
36
|
+
const NOT_JUDGED = 'CA-900';
|
|
37
|
+
|
|
38
|
+
const envelope = (findings, skipped, target, extra = {}) => {
|
|
39
|
+
const n = s => findings.filter(f => f.severity === s).length;
|
|
40
|
+
return {
|
|
41
|
+
contract: 1,
|
|
42
|
+
tool: 'control-arm',
|
|
43
|
+
version,
|
|
44
|
+
ran_at: new Date().toISOString(),
|
|
45
|
+
target,
|
|
46
|
+
findings: findings.sort((a, b) =>
|
|
47
|
+
({ blocker: 0, warn: 1, info: 2 })[a.severity] - ({ blocker: 0, warn: 1, info: 2 })[b.severity]),
|
|
48
|
+
skipped,
|
|
49
|
+
summary: { blocker: n(BLOCKER), warn: n(WARN), info: n(INFO), skipped: skipped.length, ...extra },
|
|
50
|
+
};
|
|
51
|
+
};
|
|
52
|
+
|
|
53
|
+
/**
|
|
54
|
+
* One commit's result -> findings and skips. Shared by `verify` and `audit`, so the two
|
|
55
|
+
* cannot drift into describing the same verdict differently.
|
|
56
|
+
*/
|
|
57
|
+
export function commitRows(r, repo) {
|
|
58
|
+
const findings = [], skipped = [];
|
|
59
|
+
|
|
60
|
+
// The tool declined before it began — no test file, not a fix, nothing to replay.
|
|
61
|
+
if (r.note) {
|
|
62
|
+
skipped.push({
|
|
63
|
+
id: NOT_JUDGED,
|
|
64
|
+
reason: `${r.short} ${r.subject?.slice(0, 60) ?? ''} — ${r.note}`,
|
|
65
|
+
requires: 'a commit that ships both a test and a source change',
|
|
66
|
+
});
|
|
67
|
+
return { findings, skipped };
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
const disc = (r.cases ?? []).filter(c => c.verdict === CAUGHT);
|
|
71
|
+
const inc = (r.cases ?? []).filter(c => c.verdict === INCONCLUSIVE);
|
|
72
|
+
const flaky = (r.cases ?? []).filter(c => c.verdict === FLAKY);
|
|
73
|
+
|
|
74
|
+
// C-005. Every case that could not be judged is named, with its own reason — not
|
|
75
|
+
// summed into a count. The reason is the useful part: "the import is missing on the
|
|
76
|
+
// parent" and "the test is not green on the fix either" need different fixes.
|
|
77
|
+
for (const c of inc) {
|
|
78
|
+
skipped.push({
|
|
79
|
+
id: NOT_JUDGED,
|
|
80
|
+
reason: `${r.short} · ${c.name} — ${c.reason ?? 'could not be judged'}`,
|
|
81
|
+
requires: 'a test that loads and passes on the fix, and loads on the parent',
|
|
82
|
+
});
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
for (const c of flaky) {
|
|
86
|
+
findings.push({
|
|
87
|
+
id: FLAKY_CASE,
|
|
88
|
+
severity: WARN,
|
|
89
|
+
title: `"${c.name}" gave different answers on repeated runs`,
|
|
90
|
+
observed: { what: 'the same case run more than once against the same parent',
|
|
91
|
+
where: `${c.file} · ${r.short}`,
|
|
92
|
+
value: c.reason ?? 'runs disagreed' },
|
|
93
|
+
next: 'A flaky test cannot prove anything about this commit, and it will not prove anything about the next one either. Fix the flake before trusting either verdict.',
|
|
94
|
+
confidence: 'observed',
|
|
95
|
+
fingerprint: `${FLAKY_CASE}:${c.file}:${c.name}`,
|
|
96
|
+
});
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if (r.verdict === BLIND) {
|
|
100
|
+
// C-006. Arm C is what separates a real gap from an artefact, and its answer is
|
|
101
|
+
// exactly the confidence distinction: `open` was checked against HEAD and holds;
|
|
102
|
+
// `unknown` could not be checked and must not be asserted.
|
|
103
|
+
const arm = r.stillOpen ?? null; // armC, attached by verifyCommit
|
|
104
|
+
if (arm?.status === 'repaired') {
|
|
105
|
+
findings.push({
|
|
106
|
+
id: BLIND_COMMIT, severity: INFO,
|
|
107
|
+
title: `${r.short} shipped with no test that catches it — but the gap was closed later`,
|
|
108
|
+
observed: { what: 'a commit whose own tests all pass on the broken code',
|
|
109
|
+
where: r.short, value: arm.reason },
|
|
110
|
+
next: 'Nothing to do. Recorded because the history is worth knowing, not because it is open.',
|
|
111
|
+
confidence: 'observed',
|
|
112
|
+
fingerprint: `${BLIND_COMMIT}:${r.sha}`,
|
|
113
|
+
});
|
|
114
|
+
} else if (arm?.status === 'unknown') {
|
|
115
|
+
skipped.push({
|
|
116
|
+
id: NOT_JUDGED,
|
|
117
|
+
reason: `${r.short} looks BLIND, but it could not be confirmed against HEAD — ${arm.reason}`,
|
|
118
|
+
requires: 'the test file to still exist at HEAD',
|
|
119
|
+
});
|
|
120
|
+
} else {
|
|
121
|
+
findings.push({
|
|
122
|
+
id: BLIND_COMMIT, severity: WARN,
|
|
123
|
+
title: `No test in ${r.short} fails without the change`,
|
|
124
|
+
observed: {
|
|
125
|
+
what: `${(r.cases ?? []).length} case(s) replayed against the parent commit`,
|
|
126
|
+
where: `${r.short} — ${r.subject?.slice(0, 70) ?? ''}`,
|
|
127
|
+
value: 'every case ran on the broken code and passed',
|
|
128
|
+
},
|
|
129
|
+
next: 'This fix shipped without a test that would have caught the bug. Add one that fails on the parent, or confirm the existing tests are regression guards and were never meant to.',
|
|
130
|
+
confidence: 'observed',
|
|
131
|
+
fingerprint: `${BLIND_COMMIT}:${r.sha}`,
|
|
132
|
+
});
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
// C-006 in its plainest form: a CAUGHT on new code is a real result reached against
|
|
137
|
+
// absent code, so the claim is weaker than the word suggests. Say so rather than let
|
|
138
|
+
// the headline drift upward for free on a feature-heavy week.
|
|
139
|
+
if (r.verdict === CAUGHT && r.newCode && disc.length) {
|
|
140
|
+
findings.push({
|
|
141
|
+
id: WEAK_ON_NEW, severity: INFO,
|
|
142
|
+
title: `${r.short} is CAUGHT, but on code that did not exist before`,
|
|
143
|
+
observed: { what: `${disc.length} case(s) that fail on the parent`,
|
|
144
|
+
where: r.short,
|
|
145
|
+
value: r.kind === 'feature' ? 'conventional prefix says feature' : 'this commit only added source lines' },
|
|
146
|
+
next: 'Almost any test reading a new field fails on a base that lacks it, well aimed or not. Read this as "the code is new", not as "the test is good".',
|
|
147
|
+
confidence: 'inferred',
|
|
148
|
+
fingerprint: `${WEAK_ON_NEW}:${r.sha}`,
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
return { findings, skipped };
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
export function verifyEnvelope(r, { repo, sha }) {
|
|
156
|
+
const { findings, skipped } = commitRows(r, repo);
|
|
157
|
+
return envelope(findings, skipped, { kind: 'commit', id: repo, ref: r.sha ?? sha },
|
|
158
|
+
{ verdict: r.verdict ?? SKIPPED, cases: (r.cases ?? []).length });
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
export function auditEnvelope(results, { repo, since, n, seed }) {
|
|
162
|
+
const findings = [], skipped = [];
|
|
163
|
+
for (const r of results) {
|
|
164
|
+
const rows = commitRows(r, repo);
|
|
165
|
+
findings.push(...rows.findings);
|
|
166
|
+
skipped.push(...rows.skipped);
|
|
167
|
+
}
|
|
168
|
+
const answerable = results.filter(r => r.verdict === CAUGHT || r.verdict === BLIND).length;
|
|
169
|
+
const caught = results.filter(r => r.verdict === CAUGHT).length;
|
|
170
|
+
return envelope(findings, skipped, { kind: 'repo', id: repo, ref: null }, {
|
|
171
|
+
commits_judged: results.length,
|
|
172
|
+
answerable,
|
|
173
|
+
caught,
|
|
174
|
+
// C-006 as a number: a rate over a handful of commits is not a rate, so the
|
|
175
|
+
// denominator ships beside it and a consumer can refuse to divide.
|
|
176
|
+
caught_rate: answerable ? +(caught / answerable).toFixed(3) : null,
|
|
177
|
+
since, sample: n, seed,
|
|
178
|
+
});
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
export function doctorEnvelope(checks, { repo }) {
|
|
182
|
+
const findings = checks.filter(c => !c.ok).map(c => ({
|
|
183
|
+
id: DOCTOR_BLOCK,
|
|
184
|
+
severity: BLOCKER,
|
|
185
|
+
title: `Cannot measure this repository: ${c.name}`,
|
|
186
|
+
observed: { what: c.name, where: repo, value: c.detail },
|
|
187
|
+
next: c.detail,
|
|
188
|
+
confidence: 'observed',
|
|
189
|
+
fingerprint: `${DOCTOR_BLOCK}:${c.name}`,
|
|
190
|
+
}));
|
|
191
|
+
return envelope(findings, [], { kind: 'repo', id: repo, ref: null },
|
|
192
|
+
{ checks: checks.length, passed: checks.filter(c => c.ok).length });
|
|
193
|
+
}
|
package/src/identity.mjs
CHANGED
|
@@ -53,12 +53,21 @@ export async function proveIdentity({ worktreeRoot, repoRoot, testFilePath, time
|
|
|
53
53
|
if (specs.length === 0) return { proven: true, resolved: {}, note: 'no external imports' };
|
|
54
54
|
|
|
55
55
|
const probe = `
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
56
|
+
// import.meta.resolve (sync, single-arg) landed in Node 20.6.0. On an older runtime
|
|
57
|
+
// it is undefined, and calling it used to be caught per-specifier as
|
|
58
|
+
// 'UNRESOLVED:undefined' — which the gate below reads as "genuinely missing", so every
|
|
59
|
+
// verdict would have been "proven" without resolving anything. Fail loudly instead: a
|
|
60
|
+
// withheld verdict is the safe answer; a silent no-op proof is how a false BLIND ships.
|
|
61
|
+
if (typeof import.meta.resolve !== 'function') {
|
|
62
|
+
console.log(JSON.stringify({ __no_resolve__: true }));
|
|
63
|
+
} else {
|
|
64
|
+
const out = {};
|
|
65
|
+
for (const s of ${JSON.stringify(specs)}) {
|
|
66
|
+
try { out[s] = await import.meta.resolve(s); }
|
|
67
|
+
catch (e) { out[s] = 'UNRESOLVED:' + e.code; }
|
|
68
|
+
}
|
|
69
|
+
console.log(JSON.stringify(out));
|
|
60
70
|
}
|
|
61
|
-
console.log(JSON.stringify(out));
|
|
62
71
|
`;
|
|
63
72
|
let stdout;
|
|
64
73
|
try {
|
|
@@ -73,6 +82,10 @@ export async function proveIdentity({ worktreeRoot, repoRoot, testFilePath, time
|
|
|
73
82
|
try { resolved = JSON.parse(stdout.trim().split('\n').pop()); }
|
|
74
83
|
catch { return { proven: false, reason: 'resolution probe produced no JSON' }; }
|
|
75
84
|
|
|
85
|
+
if (resolved?.__no_resolve__) {
|
|
86
|
+
return { proven: false, reason: 'this runtime cannot resolve import specifiers (import.meta.resolve needs Node 20.6+) — verdict withheld rather than trusted' };
|
|
87
|
+
}
|
|
88
|
+
|
|
76
89
|
const rootUrl = new URL('file://' + path.resolve(worktreeRoot) + '/').href;
|
|
77
90
|
// Third-party packages are DELIBERATELY shared with the target repo's node_modules —
|
|
78
91
|
// they are version-pinned content, not the source under test, and installing them per
|
package/src/markdown-report.mjs
CHANGED
|
@@ -24,9 +24,13 @@ export function prComment(r, { repoName = '.' } = {}) {
|
|
|
24
24
|
const disc = r.cases.filter(c => c.verdict === 'CAUGHT');
|
|
25
25
|
const inc = r.cases.filter(c => c.verdict === 'INCONCLUSIVE');
|
|
26
26
|
const L = [];
|
|
27
|
+
// STATE THE FACT, do not grade the branch. "this branch is proven" was the one
|
|
28
|
+
// heading in this tool that claimed more than it had measured: what was observed is
|
|
29
|
+
// that N tests fail on the base, which is a count, not a verdict on the branch.
|
|
30
|
+
const n = disc.length;
|
|
27
31
|
const head = r.verdict !== 'CAUGHT' ? r.verdict
|
|
28
32
|
: r.newCode ? 'new code — these tests cannot be judged this way'
|
|
29
|
-
: '
|
|
33
|
+
: `${n} test${n === 1 ? '' : 's'} here fail${n === 1 ? 's' : ''} without this change`;
|
|
30
34
|
L.push(`### \`control-arm\` — ${head}`);
|
|
31
35
|
L.push('');
|
|
32
36
|
if (disc.length && r.newCode) {
|
package/src/report.mjs
CHANGED
|
@@ -33,6 +33,13 @@ export function renderVerify(r) {
|
|
|
33
33
|
L.push('');
|
|
34
34
|
L.push(` ${r.short} ${r.subject}`);
|
|
35
35
|
L.push(` ${r.date} · ${r.testFiles.length} test file(s) · ${r.sourceFiles.length} source file(s) changed`);
|
|
36
|
+
// Say it when arm B is not purely the parent. The verdict still describes the parent's
|
|
37
|
+
// BEHAVIOUR — only files the commit added are carried over, and nothing at the parent
|
|
38
|
+
// could depend on those — but a reader is entitled to know the tree was not untouched.
|
|
39
|
+
if ((r.transplantedAdded || []).length) {
|
|
40
|
+
const n = r.transplantedAdded.length;
|
|
41
|
+
L.push(` arm B also carries ${n} file(s) this commit ADDED, or the test could not load: ${r.transplantedAdded.slice(0, 3).join(', ')}${n > 3 ? ` +${n - 3} more` : ''}`);
|
|
42
|
+
}
|
|
36
43
|
if (r.note) { L.push(` ${MARK.hm} ${r.note}`); L.push(''); return L.join('\n'); }
|
|
37
44
|
|
|
38
45
|
const idn = r.cases.find(c => /identity unproven/.test(c.reason || ''));
|
|
@@ -65,7 +72,7 @@ export function renderVerify(r) {
|
|
|
65
72
|
}
|
|
66
73
|
const n = v => r.cases.filter(c => c.verdict === v).length;
|
|
67
74
|
if (r.stillOpen) {
|
|
68
|
-
const m = { repaired: '↻ REPAIRED SINCE', open: '‼ STILL OPEN TODAY', unknown: '⚠ cannot tell' }[r.stillOpen.status];
|
|
75
|
+
const m = { repaired: '↻ REPAIRED SINCE', 'repaired-elsewhere': '↻ LIKELY REPAIRED (elsewhere — confirm)', open: '‼ STILL OPEN TODAY', unknown: '⚠ cannot tell' }[r.stillOpen.status];
|
|
69
76
|
L.push(` ${m} — ${r.stillOpen.reason}`);
|
|
70
77
|
L.push('');
|
|
71
78
|
}
|
|
@@ -150,6 +157,7 @@ export function renderAudit(results, meta) {
|
|
|
150
157
|
// alone hands somebody thirteen tickets, eight of which waste their afternoon.
|
|
151
158
|
const open = blind.filter(r => r.stillOpen?.status === 'open');
|
|
152
159
|
const repaired = blind.filter(r => r.stillOpen?.status === 'repaired');
|
|
160
|
+
const repairedElsewhere = blind.filter(r => r.stillOpen?.status === 'repaired-elsewhere');
|
|
153
161
|
const cannot = blind.filter(r => !r.stillOpen || r.stillOpen.status === 'unknown');
|
|
154
162
|
|
|
155
163
|
L.push(' STILL OPEN TODAY — the only rows that are work');
|
|
@@ -158,6 +166,11 @@ export function renderAudit(results, meta) {
|
|
|
158
166
|
L.push('');
|
|
159
167
|
if (repaired.length) {
|
|
160
168
|
L.push(` ↻ REPAIRED SINCE — true of the commit, already fixed in the tree (${repaired.length})`);
|
|
169
|
+
}
|
|
170
|
+
if (repairedElsewhere.length) {
|
|
171
|
+
L.push('');
|
|
172
|
+
L.push(` ↻ LIKELY REPAIRED — a later test in ANOTHER file fails on this bug; confirm before closing (${repairedElsewhere.length})`);
|
|
173
|
+
for (const r of repairedElsewhere) L.push(` ${r.short} ${r.subject.slice(0, 78)}`);
|
|
161
174
|
for (const r of repaired.slice(0, 12)) L.push(` ${r.short} ${r.subject.slice(0, 80)}`);
|
|
162
175
|
L.push('');
|
|
163
176
|
}
|
package/src/runner-json.mjs
CHANGED
|
@@ -20,6 +20,7 @@
|
|
|
20
20
|
*/
|
|
21
21
|
|
|
22
22
|
import { spawn } from 'node:child_process';
|
|
23
|
+
import { register, unregister, killGroup, armReaper } from './children.mjs';
|
|
23
24
|
import { readFile, rm } from 'node:fs/promises';
|
|
24
25
|
import path from 'node:path';
|
|
25
26
|
|
|
@@ -56,15 +57,30 @@ export function classifyMessages(msgs) {
|
|
|
56
57
|
return { errorName: 'Unrecognised', code: null, message: text.split('\n')[0].slice(0, 200) };
|
|
57
58
|
}
|
|
58
59
|
|
|
60
|
+
/**
|
|
61
|
+
* Kill the process GROUP, not the child.
|
|
62
|
+
*
|
|
63
|
+
* `node --test` spawns a worker per test file, and vitest/jest fork too. Killing only the
|
|
64
|
+
* process we spawned orphans every one of them: a machine running these audits was found
|
|
65
|
+
* carrying ten stray `node --test` processes, seven of them TWO DAYS old, each holding a
|
|
66
|
+
* worktree and file descriptors open. They sat at 0% CPU, which is why nothing noticed.
|
|
67
|
+
*
|
|
68
|
+
* `detached: true` makes the child a group leader, so `process.kill(-pid)` reaches the
|
|
69
|
+
* whole tree. The group is swept on normal close as well, because a test that leaks a
|
|
70
|
+
* server of its own exits cleanly and leaves it running.
|
|
71
|
+
*/
|
|
59
72
|
function run(cmd, args, { cwd, timeoutMs, env }) {
|
|
60
73
|
return new Promise((resolve) => {
|
|
61
|
-
const child = spawn(cmd, args, { cwd, env, shell: false });
|
|
74
|
+
const child = spawn(cmd, args, { cwd, env, shell: false, detached: true });
|
|
75
|
+
armReaper();
|
|
76
|
+
register(child.pid);
|
|
77
|
+
const killTree = () => { unregister(child.pid); killGroup(child.pid); };
|
|
62
78
|
let stdout = '', stderr = '', killed = false;
|
|
63
|
-
const t = setTimeout(() => { killed = true;
|
|
79
|
+
const t = setTimeout(() => { killed = true; killTree(); }, timeoutMs);
|
|
64
80
|
child.stdout.on('data', d => { stdout += d; });
|
|
65
81
|
child.stderr.on('data', d => { stderr += d; });
|
|
66
|
-
child.on('close', code => { clearTimeout(t); resolve({ code, stdout, stderr, killed }); });
|
|
67
|
-
child.on('error', e => { clearTimeout(t); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
82
|
+
child.on('close', code => { clearTimeout(t); killTree(); resolve({ code, stdout, stderr, killed }); });
|
|
83
|
+
child.on('error', e => { clearTimeout(t); killTree(); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
68
84
|
});
|
|
69
85
|
}
|
|
70
86
|
|
|
@@ -75,6 +91,32 @@ function childEnv() {
|
|
|
75
91
|
return env;
|
|
76
92
|
}
|
|
77
93
|
|
|
94
|
+
/**
|
|
95
|
+
* Parse the jest-shaped JSON report into normalised cases. Pure, so it can be tested
|
|
96
|
+
* against captured real runner output without installing either runner — the gap the
|
|
97
|
+
* two-arm path cannot cover is at least closed for the parsing, which is the part most
|
|
98
|
+
* likely to be wrong.
|
|
99
|
+
*/
|
|
100
|
+
export function parseReport(report) {
|
|
101
|
+
const cases = [];
|
|
102
|
+
for (const file of report.testResults || []) {
|
|
103
|
+
// A file that failed to compile has no assertionResults, only a message.
|
|
104
|
+
if ((file.assertionResults || []).length === 0 && file.message) {
|
|
105
|
+
return { ok: false, loadFailure: String(file.message).split('\n')[0].slice(0, 220), cases: [] };
|
|
106
|
+
}
|
|
107
|
+
for (const a of file.assertionResults || []) {
|
|
108
|
+
const status = a.status === 'passed' ? 'pass' : a.status === 'failed' ? 'fail' : 'skip';
|
|
109
|
+
cases.push({
|
|
110
|
+
name: a.fullName || a.title,
|
|
111
|
+
status,
|
|
112
|
+
...(status === 'fail' ? classifyMessages(a.failureMessages) : { errorName: null, code: null, message: null }),
|
|
113
|
+
});
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
if (cases.length === 0) return { ok: false, loadFailure: 'no cases reported', cases: [] };
|
|
117
|
+
return { ok: true, cases };
|
|
118
|
+
}
|
|
119
|
+
|
|
78
120
|
/**
|
|
79
121
|
* `flavour` is 'vitest' or 'jest'. `pkgDir` is the workspace the test belongs to
|
|
80
122
|
* (apps/web), NOT the repo root — both tools resolve their config relative to cwd.
|
|
@@ -102,23 +144,7 @@ export function jsonRunner(flavour) {
|
|
|
102
144
|
}
|
|
103
145
|
await rm(outFile, { force: true });
|
|
104
146
|
|
|
105
|
-
|
|
106
|
-
for (const file of report.testResults || []) {
|
|
107
|
-
// A file that failed to compile has no assertionResults, only a message.
|
|
108
|
-
if ((file.assertionResults || []).length === 0 && file.message) {
|
|
109
|
-
return { ok: false, loadFailure: String(file.message).split('\n')[0].slice(0, 220), cases: [], raw: r };
|
|
110
|
-
}
|
|
111
|
-
for (const a of file.assertionResults || []) {
|
|
112
|
-
const status = a.status === 'passed' ? 'pass' : a.status === 'failed' ? 'fail' : 'skip';
|
|
113
|
-
cases.push({
|
|
114
|
-
name: a.fullName || a.title,
|
|
115
|
-
status,
|
|
116
|
-
...(status === 'fail' ? classifyMessages(a.failureMessages) : { errorName: null, code: null, message: null }),
|
|
117
|
-
});
|
|
118
|
-
}
|
|
119
|
-
}
|
|
120
|
-
if (cases.length === 0) return { ok: false, loadFailure: 'no cases reported', cases: [], raw: r };
|
|
121
|
-
return { ok: true, cases, raw: r };
|
|
147
|
+
return { ...parseReport(report), raw: r };
|
|
122
148
|
},
|
|
123
149
|
};
|
|
124
150
|
}
|
package/src/runner.mjs
CHANGED
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import { spawn } from 'node:child_process';
|
|
10
|
-
import {
|
|
10
|
+
import { register, unregister, killGroup, armReaper } from './children.mjs';
|
|
11
|
+
import { parseTap } from './tap.mjs';
|
|
11
12
|
|
|
12
13
|
/**
|
|
13
14
|
* The child's environment.
|
|
@@ -28,21 +29,35 @@ function childEnv() {
|
|
|
28
29
|
return env;
|
|
29
30
|
}
|
|
30
31
|
|
|
32
|
+
/**
|
|
33
|
+
* Kill the process GROUP, not the child.
|
|
34
|
+
*
|
|
35
|
+
* `node --test` spawns a worker per test file, and vitest/jest fork too. Killing only the
|
|
36
|
+
* process we spawned orphans every one of them: a machine running these audits was found
|
|
37
|
+
* carrying ten stray `node --test` processes, seven of them TWO DAYS old, each holding a
|
|
38
|
+
* worktree and file descriptors open. They sat at 0% CPU, which is why nothing noticed.
|
|
39
|
+
*
|
|
40
|
+
* `detached: true` makes the child a group leader, so `process.kill(-pid)` reaches the
|
|
41
|
+
* whole tree. The group is swept on normal close as well, because a test that leaks a
|
|
42
|
+
* server of its own exits cleanly and leaves it running.
|
|
43
|
+
*/
|
|
31
44
|
function run(cmd, args, { cwd, timeoutMs }) {
|
|
32
45
|
return new Promise((resolve) => {
|
|
33
|
-
const child = spawn(cmd, args, { cwd, env: childEnv() });
|
|
46
|
+
const child = spawn(cmd, args, { cwd, env: childEnv(), detached: true });
|
|
47
|
+
armReaper();
|
|
48
|
+
register(child.pid);
|
|
49
|
+
const killTree = () => { unregister(child.pid); killGroup(child.pid); };
|
|
34
50
|
let stdout = '', stderr = '', killed = false;
|
|
35
|
-
const timer = setTimeout(() => { killed = true;
|
|
51
|
+
const timer = setTimeout(() => { killed = true; killTree(); }, timeoutMs);
|
|
36
52
|
child.stdout.on('data', d => { stdout += d; });
|
|
37
53
|
child.stderr.on('data', d => { stderr += d; });
|
|
38
|
-
child.on('close', (code) => { clearTimeout(timer); resolve({ code, stdout, stderr, killed }); });
|
|
39
|
-
child.on('error', (e) => { clearTimeout(timer); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
54
|
+
child.on('close', (code) => { clearTimeout(timer); killTree(); resolve({ code, stdout, stderr, killed }); });
|
|
55
|
+
child.on('error', (e) => { clearTimeout(timer); killTree(); resolve({ code: -1, stdout, stderr: String(e), killed }); });
|
|
40
56
|
});
|
|
41
57
|
}
|
|
42
58
|
|
|
43
59
|
export const nodeTest = {
|
|
44
60
|
name: 'node:test',
|
|
45
|
-
matches: (repoRoot, pkg) => !pkg?.scripts?.test?.includes('vitest') || true,
|
|
46
61
|
async execute({ worktreeDir, relTestPath, timeoutMs = 120_000 }) {
|
|
47
62
|
const r = await run(process.execPath, ['--test', '--test-reporter=tap', relTestPath],
|
|
48
63
|
{ cwd: worktreeDir, timeoutMs });
|
|
@@ -79,5 +94,3 @@ export const nodeTest = {
|
|
|
79
94
|
return { ok: true, cases: real, raw: r };
|
|
80
95
|
},
|
|
81
96
|
};
|
|
82
|
-
|
|
83
|
-
export const RUNNERS = [nodeTest];
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Say so when the draw came up short. (#10)
|
|
3
|
+
*
|
|
4
|
+
* `audit --n 40` on iamkun/dayjs judged FIVE commits and printed a confident 100%.
|
|
5
|
+
* Nothing was broken: the default window is twelve months, dayjs ships few `fix:`
|
|
6
|
+
* commits that also touch a test, and the pool was simply smaller than the request.
|
|
7
|
+
* The window is echoed in the header, so it was visible — but a printed default is not
|
|
8
|
+
* a signal, and a 100% over five commits reads exactly like a 100% over five hundred.
|
|
9
|
+
*
|
|
10
|
+
* What exposed it was that two different `--n` values and two different seeds gave
|
|
11
|
+
* BYTE-IDENTICAL output. A rate that does not move when you change the sample size is
|
|
12
|
+
* not a rate.
|
|
13
|
+
*
|
|
14
|
+
* Deliberately no magic threshold. The condition is the one thing that needs no
|
|
15
|
+
* judgement call: you asked for N and did not get N. Where the pool ran out decides
|
|
16
|
+
* which cause gets named.
|
|
17
|
+
*/
|
|
18
|
+
|
|
19
|
+
/**
|
|
20
|
+
* @param {number} requested --n
|
|
21
|
+
* @param {number} matched commits whose subject matched --grep, within --since
|
|
22
|
+
* @param {number} eligible of those, the ones shipping both a test and a source change
|
|
23
|
+
* @param {number} drawn what was actually sampled
|
|
24
|
+
* @param {string} since the window in force
|
|
25
|
+
* @param {boolean} sinceWasExplicit did the user pass --since themselves?
|
|
26
|
+
* @returns {string|null} a warning to print, or null when the draw was full
|
|
27
|
+
*/
|
|
28
|
+
export function sampleWarning({ requested, matched, eligible, drawn, since, sinceWasExplicit }) {
|
|
29
|
+
if (drawn >= requested) return null;
|
|
30
|
+
|
|
31
|
+
const plural = (n, word) => `${n} ${word}${n === 1 ? '' : 's'}`;
|
|
32
|
+
const lines = [` asked for ${plural(requested, 'commit')}, drew ${drawn}.`];
|
|
33
|
+
|
|
34
|
+
if (matched === 0) {
|
|
35
|
+
lines.push(` Nothing matched the subject filter within '${since}'.`);
|
|
36
|
+
} else if (eligible < requested && matched >= requested) {
|
|
37
|
+
// The window held enough commits; the test+source pre-filter is what trimmed it.
|
|
38
|
+
lines.push(` ${matched} matched the filter but only ${eligible} ship both a test and a`);
|
|
39
|
+
lines.push(` source change, and only those can be judged.`);
|
|
40
|
+
} else {
|
|
41
|
+
lines.push(` Only ${plural(matched, 'commit')} matched the subject filter within '${since}'.`);
|
|
42
|
+
if (!sinceWasExplicit)
|
|
43
|
+
lines.push(` That is the DEFAULT window, not a choice — widen it with --since '6 years'.`);
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
lines.push('');
|
|
47
|
+
lines.push(` A rate over ${drawn} commit${drawn === 1 ? '' : 's'} is not a rate. Read the counts, not the percentage.`);
|
|
48
|
+
return lines.join('\n');
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* A seed makes the draw reproducible only over a FIXED pool. `--since '4 months'` is a
|
|
53
|
+
* MOVING window: run it again tomorrow and the pool has shifted, so the same seed draws a
|
|
54
|
+
* different sample and the headline moves for no reason anyone can see.
|
|
55
|
+
*
|
|
56
|
+
* Measured 2026-09-26: the same command, same seed, hours apart, went from
|
|
57
|
+
* `890 matched / 653 judgeable` to `888 / 652` — and from 85.7% to 100%, because two old
|
|
58
|
+
* commits fell out of the window and the shuffle landed elsewhere. I spent twenty minutes
|
|
59
|
+
* suspecting my own change had broken determinism. It had not; the clock had moved.
|
|
60
|
+
*
|
|
61
|
+
* With an absolute date the same command twice gives byte-identical pools and results.
|
|
62
|
+
*/
|
|
63
|
+
export function relativeWindowWarning(since, seedWasExplicit) {
|
|
64
|
+
if (!seedWasExplicit) return null;
|
|
65
|
+
if (/^\d{4}-\d{2}-\d{2}/.test(String(since).trim())) return null; // absolute: fine
|
|
66
|
+
return ` --seed makes the draw reproducible only over a fixed pool, and '${since}' is a
|
|
67
|
+
MOVING window — the same seed will draw a different sample tomorrow.
|
|
68
|
+
For a number you can compare over time, pass an absolute date: --since '2026-06-01'.`;
|
|
69
|
+
}
|
package/src/select-runner.mjs
CHANGED
|
@@ -18,6 +18,12 @@ import path from 'node:path';
|
|
|
18
18
|
const CONFIGS = [
|
|
19
19
|
{ flavour: 'vitest', files: ['vitest.config.ts', 'vitest.config.js', 'vitest.config.mjs', 'vite.config.ts', 'vite.config.js'] },
|
|
20
20
|
{ flavour: 'jest', files: ['jest.config.js', 'jest.config.ts', 'jest.config.mjs', 'jest.config.json'] },
|
|
21
|
+
// Playwright is detected precisely so it can be DECLINED by name. There is no
|
|
22
|
+
// Playwright runner here, and falling through to node:test made a Playwright spec
|
|
23
|
+
// report `arm A did not run (node): test failed` — which reads as "your test is
|
|
24
|
+
// broken" when the truth is "I used the wrong tool and should have said so".
|
|
25
|
+
// Found by a user pointing this at a real Playwright suite.
|
|
26
|
+
{ flavour: 'playwright', files: ['playwright.config.ts', 'playwright.config.js', 'playwright.config.mjs'] },
|
|
21
27
|
];
|
|
22
28
|
|
|
23
29
|
const exists = async p => { try { await access(p); return true; } catch { return false; } };
|
|
@@ -49,6 +55,7 @@ export async function selectRunner(worktreeRoot, relTestPath) {
|
|
|
49
55
|
if (/\bvitest\b/.test(script)) return { flavour: 'vitest', pkgDir: dir === '.' ? '' : dir };
|
|
50
56
|
if (/\bjest\b/.test(script)) return { flavour: 'jest', pkgDir: dir === '.' ? '' : dir };
|
|
51
57
|
if (/node\s+--test|\bnode:test\b/.test(script)) return { flavour: 'node', pkgDir: dir === '.' ? '' : dir };
|
|
58
|
+
if (/\bplaywright\s+test\b/.test(script)) return { flavour: 'playwright', pkgDir: dir === '.' ? '' : dir };
|
|
52
59
|
} catch { /* unparseable package.json is not a signal */ }
|
|
53
60
|
}
|
|
54
61
|
|
package/src/tap.mjs
CHANGED
|
@@ -71,13 +71,3 @@ export function parseTap(stdout) {
|
|
|
71
71
|
}
|
|
72
72
|
return cases;
|
|
73
73
|
}
|
|
74
|
-
|
|
75
|
-
/**
|
|
76
|
-
* A whole-file load failure. node:test reports this as a `not ok` for the FILE path with
|
|
77
|
-
* no individual cases, which must not be mistaken for every case failing by assertion.
|
|
78
|
-
*/
|
|
79
|
-
export function isFileLevelFailure(cases, testFile) {
|
|
80
|
-
if (cases.length !== 1) return false;
|
|
81
|
-
const only = cases[0];
|
|
82
|
-
return only.status === 'fail' && (only.name.includes('/') || only.name.endsWith('.mjs') || only.name.endsWith('.js') || only.name === testFile);
|
|
83
|
-
}
|