control-arm 0.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,55 @@
1
+ /**
2
+ * Reap the test processes when this process goes away.
3
+ *
4
+ * A machine running these audits was found carrying ten stray `node --test` processes,
5
+ * seven of them TWO DAYS old, each holding a worktree and file descriptors open. They sat
6
+ * at 0% CPU, which is why nothing noticed: the tool looked idle rather than leaky.
7
+ *
8
+ * The cause is NOT the timeout — that path already killed what it spawned. It is the audit
9
+ * being killed itself. Observed directly:
10
+ *
11
+ * before 1242399 ppid 1242397 node --test hangs.test.mjs
12
+ * (kill the parent)
13
+ * after 1242399 ppid 2099 node --test hangs.test.mjs <- survived
14
+ *
15
+ * An audit runs for the better part of an hour, so it gets interrupted often — and each
16
+ * interruption stranded whatever was mid-run. `detached: true` alone makes this WORSE,
17
+ * because a detached child is meant to outlive its parent. So the spawn stays detached (to
18
+ * get a killable process GROUP for runners that fork workers) and every live group is
19
+ * registered here, to be swept when this process ends however it ends.
20
+ */
21
+ const live = new Set();
22
+
23
+ /** Kill a process group, tolerating one that has already gone. */
24
+ export function killGroup(pid) {
25
+ try { process.kill(-pid, 'SIGKILL'); } catch { /* already gone, or never grouped */ }
26
+ }
27
+
28
+ export function register(pid) { live.add(pid); }
29
+ export function unregister(pid) { live.delete(pid); }
30
+ export function liveCount() { return live.size; }
31
+
32
+ export function reapAll() {
33
+ for (const pid of live) killGroup(pid);
34
+ live.clear();
35
+ }
36
+
37
+ let armed = false;
38
+ /**
39
+ * Idempotent, and deliberately not installed at import time: importing a module should not
40
+ * change how the process handles signals. The runners call this the first time they spawn.
41
+ */
42
+ export function armReaper() {
43
+ if (armed) return;
44
+ armed = true;
45
+ process.on('exit', reapAll);
46
+ for (const sig of ['SIGINT', 'SIGTERM', 'SIGHUP']) {
47
+ process.on(sig, () => {
48
+ reapAll();
49
+ // Re-raise with the handler removed, so the exit code still says "signalled"
50
+ // rather than pretending this was a clean exit.
51
+ process.removeAllListeners(sig);
52
+ process.kill(process.pid, sig);
53
+ });
54
+ }
55
+ }
@@ -0,0 +1,50 @@
1
+ /**
2
+ * The CLI's surface, as data.
3
+ *
4
+ * It lives here rather than inline in `bin/ca.mjs` so a test can read the REAL table
5
+ * instead of a copy of it. A copy drifts, and a test asserting against a copy passes
6
+ * while the README documents a command the binary has never heard of.
7
+ *
8
+ * That is not hypothetical. `ca probe` was the first command in this README from the
9
+ * day it was written. It does not exist. It survived a full rewrite, a review pass, a
10
+ * GitHub release, a Marketplace listing and an npm publish, and reached the npm front
11
+ * page — because nothing in a 48-test suite knew the README existed.
12
+ */
13
+
14
+ /** Subcommands. `bin/ca.mjs` builds its dispatch table from these names. */
15
+ export const COMMANDS = Object.freeze({
16
+ doctor: 'can this repo be measured at all?',
17
+ verify: 'one commit, per-case verdicts',
18
+ audit: 'sample N fix commits and report the CAUGHT rate',
19
+ issues: 'file a ticket per open BLIND commit',
20
+ });
21
+
22
+ export const COMMAND_NAMES = Object.freeze(Object.keys(COMMANDS));
23
+
24
+ /**
25
+ * Every flag the binary reads. `takesValue` matters to the README check: a valued flag
26
+ * documented without a value is a copy-paste that silently swallows the next argument.
27
+ */
28
+ export const FLAGS = Object.freeze({
29
+ repo: { takesValue: true, help: 'the repository to measure (default: cwd)' },
30
+ against: { takesValue: true, help: 'compare against a base ref — turns verify into a PR check' },
31
+ runs: { takesValue: true, help: 'repeat each case N times to expose flakes' },
32
+ timeout: { takesValue: true, help: 'per-run timeout in ms' },
33
+ work: { takesValue: true, help: 'where to put the throwaway worktrees' },
34
+ n: { takesValue: true, help: 'how many fix commits to sample' },
35
+ since: { takesValue: true, help: "how far back to look, e.g. '6 years' (default: 12 months)" },
36
+ grep: { takesValue: true, help: 'commit-subject filter' },
37
+ seed: { takesValue: true, help: 'make the random draw reproducible' },
38
+ out: { takesValue: true, help: 'write the audit to a CSV' },
39
+ html: { takesValue: true, help: 'write an HTML report' },
40
+ json: { takesValue: false, help: 'findings-contract v1 envelope on stdout; audit additionally accepts `--json <path>` to write its summary to a file' },
41
+ 'fail-on-blind': { takesValue: false, help: 'exit 1 when a commit is BLIND (off by default — BLIND is a warning, not a blocker)' },
42
+ keep: { takesValue: false, help: 'leave the worktrees behind for inspection' },
43
+ 'pr-comment': { takesValue: false, help: 'print the PR comment markdown and nothing else' },
44
+ apply: { takesValue: false, help: 'actually file the issues, rather than printing them' },
45
+ 'include-test-fixes': { takesValue: false, help: 'do not skip commits whose subject is fix(test)' },
46
+ });
47
+
48
+ export const FLAG_NAMES = Object.freeze(Object.keys(FLAGS));
49
+
50
+ export const BIN = 'ca';
@@ -0,0 +1,193 @@
1
+ /**
2
+ * The findings contract, v1 — see the findings-contract SPEC.
3
+ *
4
+ * This is the second tool to conform, and the clauses it was chosen to test are C-005
5
+ * (`skipped`) and C-006 (`confidence`), because this tool already refuses to answer.
6
+ * `INCONCLUSIVE` has always been the rule that a red test which never RAN proves nothing;
7
+ * `skipped` is that rule with a schema.
8
+ *
9
+ * The mapping that matters, and it is not one-to-one:
10
+ *
11
+ * commit CAUGHT -> no finding. Nothing is wrong.
12
+ * commit BLIND -> a finding, severity `warn`
13
+ * commit INCONCLUSIVE -> NOT a finding. It goes to `skipped`, with the reason.
14
+ * case NON-DISCRIM. -> never a finding. It is a fact about a guard, not a fault.
15
+ * case FLAKY -> a finding. Runs disagreed, which is a real defect.
16
+ *
17
+ * `BLIND` is `warn` and not `blocker` for the same reason `unreached` has no blockers:
18
+ * C-004 asks whether a build fails NOW, and a test that cannot catch a bug breaks
19
+ * nothing today. It is the absence of protection. `--fail-on-blind` remains this tool's
20
+ * own opinion, named separately, exactly as the spec now requires.
21
+ */
22
+
23
+ import { CAUGHT, BLIND, INCONCLUSIVE, FLAKY, SKIPPED } from './verdict.mjs';
24
+ import { createRequire } from 'node:module';
25
+
26
+ // Read the real version rather than hardcoding it: a hardcoded '1.1.0' drifted the moment
27
+ // the next release bumped package.json, and the contract advertises a version to consumers.
28
+ const { version } = createRequire(import.meta.url)('../package.json');
29
+
30
+ export const BLOCKER = 'blocker', WARN = 'warn', INFO = 'info';
31
+
32
+ const BLIND_COMMIT = 'CA-001';
33
+ const FLAKY_CASE = 'CA-002';
34
+ const WEAK_ON_NEW = 'CA-003';
35
+ const DOCTOR_BLOCK = 'CA-010';
36
+ const NOT_JUDGED = 'CA-900';
37
+
38
+ const envelope = (findings, skipped, target, extra = {}) => {
39
+ const n = s => findings.filter(f => f.severity === s).length;
40
+ return {
41
+ contract: 1,
42
+ tool: 'control-arm',
43
+ version,
44
+ ran_at: new Date().toISOString(),
45
+ target,
46
+ findings: findings.sort((a, b) =>
47
+ ({ blocker: 0, warn: 1, info: 2 })[a.severity] - ({ blocker: 0, warn: 1, info: 2 })[b.severity]),
48
+ skipped,
49
+ summary: { blocker: n(BLOCKER), warn: n(WARN), info: n(INFO), skipped: skipped.length, ...extra },
50
+ };
51
+ };
52
+
53
+ /**
54
+ * One commit's result -> findings and skips. Shared by `verify` and `audit`, so the two
55
+ * cannot drift into describing the same verdict differently.
56
+ */
57
+ export function commitRows(r, repo) {
58
+ const findings = [], skipped = [];
59
+
60
+ // The tool declined before it began — no test file, not a fix, nothing to replay.
61
+ if (r.note) {
62
+ skipped.push({
63
+ id: NOT_JUDGED,
64
+ reason: `${r.short} ${r.subject?.slice(0, 60) ?? ''} — ${r.note}`,
65
+ requires: 'a commit that ships both a test and a source change',
66
+ });
67
+ return { findings, skipped };
68
+ }
69
+
70
+ const disc = (r.cases ?? []).filter(c => c.verdict === CAUGHT);
71
+ const inc = (r.cases ?? []).filter(c => c.verdict === INCONCLUSIVE);
72
+ const flaky = (r.cases ?? []).filter(c => c.verdict === FLAKY);
73
+
74
+ // C-005. Every case that could not be judged is named, with its own reason — not
75
+ // summed into a count. The reason is the useful part: "the import is missing on the
76
+ // parent" and "the test is not green on the fix either" need different fixes.
77
+ for (const c of inc) {
78
+ skipped.push({
79
+ id: NOT_JUDGED,
80
+ reason: `${r.short} · ${c.name} — ${c.reason ?? 'could not be judged'}`,
81
+ requires: 'a test that loads and passes on the fix, and loads on the parent',
82
+ });
83
+ }
84
+
85
+ for (const c of flaky) {
86
+ findings.push({
87
+ id: FLAKY_CASE,
88
+ severity: WARN,
89
+ title: `"${c.name}" gave different answers on repeated runs`,
90
+ observed: { what: 'the same case run more than once against the same parent',
91
+ where: `${c.file} · ${r.short}`,
92
+ value: c.reason ?? 'runs disagreed' },
93
+ next: 'A flaky test cannot prove anything about this commit, and it will not prove anything about the next one either. Fix the flake before trusting either verdict.',
94
+ confidence: 'observed',
95
+ fingerprint: `${FLAKY_CASE}:${c.file}:${c.name}`,
96
+ });
97
+ }
98
+
99
+ if (r.verdict === BLIND) {
100
+ // C-006. Arm C is what separates a real gap from an artefact, and its answer is
101
+ // exactly the confidence distinction: `open` was checked against HEAD and holds;
102
+ // `unknown` could not be checked and must not be asserted.
103
+ const arm = r.stillOpen ?? null; // armC, attached by verifyCommit
104
+ if (arm?.status === 'repaired') {
105
+ findings.push({
106
+ id: BLIND_COMMIT, severity: INFO,
107
+ title: `${r.short} shipped with no test that catches it — but the gap was closed later`,
108
+ observed: { what: 'a commit whose own tests all pass on the broken code',
109
+ where: r.short, value: arm.reason },
110
+ next: 'Nothing to do. Recorded because the history is worth knowing, not because it is open.',
111
+ confidence: 'observed',
112
+ fingerprint: `${BLIND_COMMIT}:${r.sha}`,
113
+ });
114
+ } else if (arm?.status === 'unknown') {
115
+ skipped.push({
116
+ id: NOT_JUDGED,
117
+ reason: `${r.short} looks BLIND, but it could not be confirmed against HEAD — ${arm.reason}`,
118
+ requires: 'the test file to still exist at HEAD',
119
+ });
120
+ } else {
121
+ findings.push({
122
+ id: BLIND_COMMIT, severity: WARN,
123
+ title: `No test in ${r.short} fails without the change`,
124
+ observed: {
125
+ what: `${(r.cases ?? []).length} case(s) replayed against the parent commit`,
126
+ where: `${r.short} — ${r.subject?.slice(0, 70) ?? ''}`,
127
+ value: 'every case ran on the broken code and passed',
128
+ },
129
+ next: 'This fix shipped without a test that would have caught the bug. Add one that fails on the parent, or confirm the existing tests are regression guards and were never meant to.',
130
+ confidence: 'observed',
131
+ fingerprint: `${BLIND_COMMIT}:${r.sha}`,
132
+ });
133
+ }
134
+ }
135
+
136
+ // C-006 in its plainest form: a CAUGHT on new code is a real result reached against
137
+ // absent code, so the claim is weaker than the word suggests. Say so rather than let
138
+ // the headline drift upward for free on a feature-heavy week.
139
+ if (r.verdict === CAUGHT && r.newCode && disc.length) {
140
+ findings.push({
141
+ id: WEAK_ON_NEW, severity: INFO,
142
+ title: `${r.short} is CAUGHT, but on code that did not exist before`,
143
+ observed: { what: `${disc.length} case(s) that fail on the parent`,
144
+ where: r.short,
145
+ value: r.kind === 'feature' ? 'conventional prefix says feature' : 'this commit only added source lines' },
146
+ next: 'Almost any test reading a new field fails on a base that lacks it, well aimed or not. Read this as "the code is new", not as "the test is good".',
147
+ confidence: 'inferred',
148
+ fingerprint: `${WEAK_ON_NEW}:${r.sha}`,
149
+ });
150
+ }
151
+
152
+ return { findings, skipped };
153
+ }
154
+
155
+ export function verifyEnvelope(r, { repo, sha }) {
156
+ const { findings, skipped } = commitRows(r, repo);
157
+ return envelope(findings, skipped, { kind: 'commit', id: repo, ref: r.sha ?? sha },
158
+ { verdict: r.verdict ?? SKIPPED, cases: (r.cases ?? []).length });
159
+ }
160
+
161
+ export function auditEnvelope(results, { repo, since, n, seed }) {
162
+ const findings = [], skipped = [];
163
+ for (const r of results) {
164
+ const rows = commitRows(r, repo);
165
+ findings.push(...rows.findings);
166
+ skipped.push(...rows.skipped);
167
+ }
168
+ const answerable = results.filter(r => r.verdict === CAUGHT || r.verdict === BLIND).length;
169
+ const caught = results.filter(r => r.verdict === CAUGHT).length;
170
+ return envelope(findings, skipped, { kind: 'repo', id: repo, ref: null }, {
171
+ commits_judged: results.length,
172
+ answerable,
173
+ caught,
174
+ // C-006 as a number: a rate over a handful of commits is not a rate, so the
175
+ // denominator ships beside it and a consumer can refuse to divide.
176
+ caught_rate: answerable ? +(caught / answerable).toFixed(3) : null,
177
+ since, sample: n, seed,
178
+ });
179
+ }
180
+
181
+ export function doctorEnvelope(checks, { repo }) {
182
+ const findings = checks.filter(c => !c.ok).map(c => ({
183
+ id: DOCTOR_BLOCK,
184
+ severity: BLOCKER,
185
+ title: `Cannot measure this repository: ${c.name}`,
186
+ observed: { what: c.name, where: repo, value: c.detail },
187
+ next: c.detail,
188
+ confidence: 'observed',
189
+ fingerprint: `${DOCTOR_BLOCK}:${c.name}`,
190
+ }));
191
+ return envelope(findings, [], { kind: 'repo', id: repo, ref: null },
192
+ { checks: checks.length, passed: checks.filter(c => c.ok).length });
193
+ }
package/src/identity.mjs CHANGED
@@ -53,12 +53,21 @@ export async function proveIdentity({ worktreeRoot, repoRoot, testFilePath, time
53
53
  if (specs.length === 0) return { proven: true, resolved: {}, note: 'no external imports' };
54
54
 
55
55
  const probe = `
56
- const out = {};
57
- for (const s of ${JSON.stringify(specs)}) {
58
- try { out[s] = await import.meta.resolve(s); }
59
- catch (e) { out[s] = 'UNRESOLVED:' + e.code; }
56
+ // import.meta.resolve (sync, single-arg) landed in Node 20.6.0. On an older runtime
57
+ // it is undefined, and calling it used to be caught per-specifier as
58
+ // 'UNRESOLVED:undefined' — which the gate below reads as "genuinely missing", so every
59
+ // verdict would have been "proven" without resolving anything. Fail loudly instead: a
60
+ // withheld verdict is the safe answer; a silent no-op proof is how a false BLIND ships.
61
+ if (typeof import.meta.resolve !== 'function') {
62
+ console.log(JSON.stringify({ __no_resolve__: true }));
63
+ } else {
64
+ const out = {};
65
+ for (const s of ${JSON.stringify(specs)}) {
66
+ try { out[s] = await import.meta.resolve(s); }
67
+ catch (e) { out[s] = 'UNRESOLVED:' + e.code; }
68
+ }
69
+ console.log(JSON.stringify(out));
60
70
  }
61
- console.log(JSON.stringify(out));
62
71
  `;
63
72
  let stdout;
64
73
  try {
@@ -73,6 +82,10 @@ export async function proveIdentity({ worktreeRoot, repoRoot, testFilePath, time
73
82
  try { resolved = JSON.parse(stdout.trim().split('\n').pop()); }
74
83
  catch { return { proven: false, reason: 'resolution probe produced no JSON' }; }
75
84
 
85
+ if (resolved?.__no_resolve__) {
86
+ return { proven: false, reason: 'this runtime cannot resolve import specifiers (import.meta.resolve needs Node 20.6+) — verdict withheld rather than trusted' };
87
+ }
88
+
76
89
  const rootUrl = new URL('file://' + path.resolve(worktreeRoot) + '/').href;
77
90
  // Third-party packages are DELIBERATELY shared with the target repo's node_modules —
78
91
  // they are version-pinned content, not the source under test, and installing them per
@@ -24,9 +24,13 @@ export function prComment(r, { repoName = '.' } = {}) {
24
24
  const disc = r.cases.filter(c => c.verdict === 'CAUGHT');
25
25
  const inc = r.cases.filter(c => c.verdict === 'INCONCLUSIVE');
26
26
  const L = [];
27
+ // STATE THE FACT, do not grade the branch. "this branch is proven" was the one
28
+ // heading in this tool that claimed more than it had measured: what was observed is
29
+ // that N tests fail on the base, which is a count, not a verdict on the branch.
30
+ const n = disc.length;
27
31
  const head = r.verdict !== 'CAUGHT' ? r.verdict
28
32
  : r.newCode ? 'new code — these tests cannot be judged this way'
29
- : 'this branch is proven';
33
+ : `${n} test${n === 1 ? '' : 's'} here fail${n === 1 ? 's' : ''} without this change`;
30
34
  L.push(`### \`control-arm\` — ${head}`);
31
35
  L.push('');
32
36
  if (disc.length && r.newCode) {
package/src/report.mjs CHANGED
@@ -33,6 +33,13 @@ export function renderVerify(r) {
33
33
  L.push('');
34
34
  L.push(` ${r.short} ${r.subject}`);
35
35
  L.push(` ${r.date} · ${r.testFiles.length} test file(s) · ${r.sourceFiles.length} source file(s) changed`);
36
+ // Say it when arm B is not purely the parent. The verdict still describes the parent's
37
+ // BEHAVIOUR — only files the commit added are carried over, and nothing at the parent
38
+ // could depend on those — but a reader is entitled to know the tree was not untouched.
39
+ if ((r.transplantedAdded || []).length) {
40
+ const n = r.transplantedAdded.length;
41
+ L.push(` arm B also carries ${n} file(s) this commit ADDED, or the test could not load: ${r.transplantedAdded.slice(0, 3).join(', ')}${n > 3 ? ` +${n - 3} more` : ''}`);
42
+ }
36
43
  if (r.note) { L.push(` ${MARK.hm} ${r.note}`); L.push(''); return L.join('\n'); }
37
44
 
38
45
  const idn = r.cases.find(c => /identity unproven/.test(c.reason || ''));
@@ -65,7 +72,7 @@ export function renderVerify(r) {
65
72
  }
66
73
  const n = v => r.cases.filter(c => c.verdict === v).length;
67
74
  if (r.stillOpen) {
68
- const m = { repaired: '↻ REPAIRED SINCE', open: '‼ STILL OPEN TODAY', unknown: '⚠ cannot tell' }[r.stillOpen.status];
75
+ const m = { repaired: '↻ REPAIRED SINCE', 'repaired-elsewhere': '↻ LIKELY REPAIRED (elsewhere — confirm)', open: '‼ STILL OPEN TODAY', unknown: '⚠ cannot tell' }[r.stillOpen.status];
69
76
  L.push(` ${m} — ${r.stillOpen.reason}`);
70
77
  L.push('');
71
78
  }
@@ -150,6 +157,7 @@ export function renderAudit(results, meta) {
150
157
  // alone hands somebody thirteen tickets, eight of which waste their afternoon.
151
158
  const open = blind.filter(r => r.stillOpen?.status === 'open');
152
159
  const repaired = blind.filter(r => r.stillOpen?.status === 'repaired');
160
+ const repairedElsewhere = blind.filter(r => r.stillOpen?.status === 'repaired-elsewhere');
153
161
  const cannot = blind.filter(r => !r.stillOpen || r.stillOpen.status === 'unknown');
154
162
 
155
163
  L.push(' STILL OPEN TODAY — the only rows that are work');
@@ -158,6 +166,11 @@ export function renderAudit(results, meta) {
158
166
  L.push('');
159
167
  if (repaired.length) {
160
168
  L.push(` ↻ REPAIRED SINCE — true of the commit, already fixed in the tree (${repaired.length})`);
169
+ }
170
+ if (repairedElsewhere.length) {
171
+ L.push('');
172
+ L.push(` ↻ LIKELY REPAIRED — a later test in ANOTHER file fails on this bug; confirm before closing (${repairedElsewhere.length})`);
173
+ for (const r of repairedElsewhere) L.push(` ${r.short} ${r.subject.slice(0, 78)}`);
161
174
  for (const r of repaired.slice(0, 12)) L.push(` ${r.short} ${r.subject.slice(0, 80)}`);
162
175
  L.push('');
163
176
  }
@@ -20,6 +20,7 @@
20
20
  */
21
21
 
22
22
  import { spawn } from 'node:child_process';
23
+ import { register, unregister, killGroup, armReaper } from './children.mjs';
23
24
  import { readFile, rm } from 'node:fs/promises';
24
25
  import path from 'node:path';
25
26
 
@@ -56,15 +57,30 @@ export function classifyMessages(msgs) {
56
57
  return { errorName: 'Unrecognised', code: null, message: text.split('\n')[0].slice(0, 200) };
57
58
  }
58
59
 
60
+ /**
61
+ * Kill the process GROUP, not the child.
62
+ *
63
+ * `node --test` spawns a worker per test file, and vitest/jest fork too. Killing only the
64
+ * process we spawned orphans every one of them: a machine running these audits was found
65
+ * carrying ten stray `node --test` processes, seven of them TWO DAYS old, each holding a
66
+ * worktree and file descriptors open. They sat at 0% CPU, which is why nothing noticed.
67
+ *
68
+ * `detached: true` makes the child a group leader, so `process.kill(-pid)` reaches the
69
+ * whole tree. The group is swept on normal close as well, because a test that leaks a
70
+ * server of its own exits cleanly and leaves it running.
71
+ */
59
72
  function run(cmd, args, { cwd, timeoutMs, env }) {
60
73
  return new Promise((resolve) => {
61
- const child = spawn(cmd, args, { cwd, env, shell: false });
74
+ const child = spawn(cmd, args, { cwd, env, shell: false, detached: true });
75
+ armReaper();
76
+ register(child.pid);
77
+ const killTree = () => { unregister(child.pid); killGroup(child.pid); };
62
78
  let stdout = '', stderr = '', killed = false;
63
- const t = setTimeout(() => { killed = true; child.kill('SIGKILL'); }, timeoutMs);
79
+ const t = setTimeout(() => { killed = true; killTree(); }, timeoutMs);
64
80
  child.stdout.on('data', d => { stdout += d; });
65
81
  child.stderr.on('data', d => { stderr += d; });
66
- child.on('close', code => { clearTimeout(t); resolve({ code, stdout, stderr, killed }); });
67
- child.on('error', e => { clearTimeout(t); resolve({ code: -1, stdout, stderr: String(e), killed }); });
82
+ child.on('close', code => { clearTimeout(t); killTree(); resolve({ code, stdout, stderr, killed }); });
83
+ child.on('error', e => { clearTimeout(t); killTree(); resolve({ code: -1, stdout, stderr: String(e), killed }); });
68
84
  });
69
85
  }
70
86
 
@@ -75,6 +91,32 @@ function childEnv() {
75
91
  return env;
76
92
  }
77
93
 
94
+ /**
95
+ * Parse the jest-shaped JSON report into normalised cases. Pure, so it can be tested
96
+ * against captured real runner output without installing either runner — the gap the
97
+ * two-arm path cannot cover is at least closed for the parsing, which is the part most
98
+ * likely to be wrong.
99
+ */
100
+ export function parseReport(report) {
101
+ const cases = [];
102
+ for (const file of report.testResults || []) {
103
+ // A file that failed to compile has no assertionResults, only a message.
104
+ if ((file.assertionResults || []).length === 0 && file.message) {
105
+ return { ok: false, loadFailure: String(file.message).split('\n')[0].slice(0, 220), cases: [] };
106
+ }
107
+ for (const a of file.assertionResults || []) {
108
+ const status = a.status === 'passed' ? 'pass' : a.status === 'failed' ? 'fail' : 'skip';
109
+ cases.push({
110
+ name: a.fullName || a.title,
111
+ status,
112
+ ...(status === 'fail' ? classifyMessages(a.failureMessages) : { errorName: null, code: null, message: null }),
113
+ });
114
+ }
115
+ }
116
+ if (cases.length === 0) return { ok: false, loadFailure: 'no cases reported', cases: [] };
117
+ return { ok: true, cases };
118
+ }
119
+
78
120
  /**
79
121
  * `flavour` is 'vitest' or 'jest'. `pkgDir` is the workspace the test belongs to
80
122
  * (apps/web), NOT the repo root — both tools resolve their config relative to cwd.
@@ -102,23 +144,7 @@ export function jsonRunner(flavour) {
102
144
  }
103
145
  await rm(outFile, { force: true });
104
146
 
105
- const cases = [];
106
- for (const file of report.testResults || []) {
107
- // A file that failed to compile has no assertionResults, only a message.
108
- if ((file.assertionResults || []).length === 0 && file.message) {
109
- return { ok: false, loadFailure: String(file.message).split('\n')[0].slice(0, 220), cases: [], raw: r };
110
- }
111
- for (const a of file.assertionResults || []) {
112
- const status = a.status === 'passed' ? 'pass' : a.status === 'failed' ? 'fail' : 'skip';
113
- cases.push({
114
- name: a.fullName || a.title,
115
- status,
116
- ...(status === 'fail' ? classifyMessages(a.failureMessages) : { errorName: null, code: null, message: null }),
117
- });
118
- }
119
- }
120
- if (cases.length === 0) return { ok: false, loadFailure: 'no cases reported', cases: [], raw: r };
121
- return { ok: true, cases, raw: r };
147
+ return { ...parseReport(report), raw: r };
122
148
  },
123
149
  };
124
150
  }
package/src/runner.mjs CHANGED
@@ -7,7 +7,8 @@
7
7
  */
8
8
 
9
9
  import { spawn } from 'node:child_process';
10
- import { parseTap, isFileLevelFailure } from './tap.mjs';
10
+ import { register, unregister, killGroup, armReaper } from './children.mjs';
11
+ import { parseTap } from './tap.mjs';
11
12
 
12
13
  /**
13
14
  * The child's environment.
@@ -28,21 +29,35 @@ function childEnv() {
28
29
  return env;
29
30
  }
30
31
 
32
+ /**
33
+ * Kill the process GROUP, not the child.
34
+ *
35
+ * `node --test` spawns a worker per test file, and vitest/jest fork too. Killing only the
36
+ * process we spawned orphans every one of them: a machine running these audits was found
37
+ * carrying ten stray `node --test` processes, seven of them TWO DAYS old, each holding a
38
+ * worktree and file descriptors open. They sat at 0% CPU, which is why nothing noticed.
39
+ *
40
+ * `detached: true` makes the child a group leader, so `process.kill(-pid)` reaches the
41
+ * whole tree. The group is swept on normal close as well, because a test that leaks a
42
+ * server of its own exits cleanly and leaves it running.
43
+ */
31
44
  function run(cmd, args, { cwd, timeoutMs }) {
32
45
  return new Promise((resolve) => {
33
- const child = spawn(cmd, args, { cwd, env: childEnv() });
46
+ const child = spawn(cmd, args, { cwd, env: childEnv(), detached: true });
47
+ armReaper();
48
+ register(child.pid);
49
+ const killTree = () => { unregister(child.pid); killGroup(child.pid); };
34
50
  let stdout = '', stderr = '', killed = false;
35
- const timer = setTimeout(() => { killed = true; child.kill('SIGKILL'); }, timeoutMs);
51
+ const timer = setTimeout(() => { killed = true; killTree(); }, timeoutMs);
36
52
  child.stdout.on('data', d => { stdout += d; });
37
53
  child.stderr.on('data', d => { stderr += d; });
38
- child.on('close', (code) => { clearTimeout(timer); resolve({ code, stdout, stderr, killed }); });
39
- child.on('error', (e) => { clearTimeout(timer); resolve({ code: -1, stdout, stderr: String(e), killed }); });
54
+ child.on('close', (code) => { clearTimeout(timer); killTree(); resolve({ code, stdout, stderr, killed }); });
55
+ child.on('error', (e) => { clearTimeout(timer); killTree(); resolve({ code: -1, stdout, stderr: String(e), killed }); });
40
56
  });
41
57
  }
42
58
 
43
59
  export const nodeTest = {
44
60
  name: 'node:test',
45
- matches: (repoRoot, pkg) => !pkg?.scripts?.test?.includes('vitest') || true,
46
61
  async execute({ worktreeDir, relTestPath, timeoutMs = 120_000 }) {
47
62
  const r = await run(process.execPath, ['--test', '--test-reporter=tap', relTestPath],
48
63
  { cwd: worktreeDir, timeoutMs });
@@ -79,5 +94,3 @@ export const nodeTest = {
79
94
  return { ok: true, cases: real, raw: r };
80
95
  },
81
96
  };
82
-
83
- export const RUNNERS = [nodeTest];
@@ -0,0 +1,69 @@
1
+ /**
2
+ * Say so when the draw came up short. (#10)
3
+ *
4
+ * `audit --n 40` on iamkun/dayjs judged FIVE commits and printed a confident 100%.
5
+ * Nothing was broken: the default window is twelve months, dayjs ships few `fix:`
6
+ * commits that also touch a test, and the pool was simply smaller than the request.
7
+ * The window is echoed in the header, so it was visible — but a printed default is not
8
+ * a signal, and a 100% over five commits reads exactly like a 100% over five hundred.
9
+ *
10
+ * What exposed it was that two different `--n` values and two different seeds gave
11
+ * BYTE-IDENTICAL output. A rate that does not move when you change the sample size is
12
+ * not a rate.
13
+ *
14
+ * Deliberately no magic threshold. The condition is the one thing that needs no
15
+ * judgement call: you asked for N and did not get N. Where the pool ran out decides
16
+ * which cause gets named.
17
+ */
18
+
19
+ /**
20
+ * @param {number} requested --n
21
+ * @param {number} matched commits whose subject matched --grep, within --since
22
+ * @param {number} eligible of those, the ones shipping both a test and a source change
23
+ * @param {number} drawn what was actually sampled
24
+ * @param {string} since the window in force
25
+ * @param {boolean} sinceWasExplicit did the user pass --since themselves?
26
+ * @returns {string|null} a warning to print, or null when the draw was full
27
+ */
28
+ export function sampleWarning({ requested, matched, eligible, drawn, since, sinceWasExplicit }) {
29
+ if (drawn >= requested) return null;
30
+
31
+ const plural = (n, word) => `${n} ${word}${n === 1 ? '' : 's'}`;
32
+ const lines = [` asked for ${plural(requested, 'commit')}, drew ${drawn}.`];
33
+
34
+ if (matched === 0) {
35
+ lines.push(` Nothing matched the subject filter within '${since}'.`);
36
+ } else if (eligible < requested && matched >= requested) {
37
+ // The window held enough commits; the test+source pre-filter is what trimmed it.
38
+ lines.push(` ${matched} matched the filter but only ${eligible} ship both a test and a`);
39
+ lines.push(` source change, and only those can be judged.`);
40
+ } else {
41
+ lines.push(` Only ${plural(matched, 'commit')} matched the subject filter within '${since}'.`);
42
+ if (!sinceWasExplicit)
43
+ lines.push(` That is the DEFAULT window, not a choice — widen it with --since '6 years'.`);
44
+ }
45
+
46
+ lines.push('');
47
+ lines.push(` A rate over ${drawn} commit${drawn === 1 ? '' : 's'} is not a rate. Read the counts, not the percentage.`);
48
+ return lines.join('\n');
49
+ }
50
+
51
+ /**
52
+ * A seed makes the draw reproducible only over a FIXED pool. `--since '4 months'` is a
53
+ * MOVING window: run it again tomorrow and the pool has shifted, so the same seed draws a
54
+ * different sample and the headline moves for no reason anyone can see.
55
+ *
56
+ * Measured 2026-09-26: the same command, same seed, hours apart, went from
57
+ * `890 matched / 653 judgeable` to `888 / 652` — and from 85.7% to 100%, because two old
58
+ * commits fell out of the window and the shuffle landed elsewhere. I spent twenty minutes
59
+ * suspecting my own change had broken determinism. It had not; the clock had moved.
60
+ *
61
+ * With an absolute date the same command twice gives byte-identical pools and results.
62
+ */
63
+ export function relativeWindowWarning(since, seedWasExplicit) {
64
+ if (!seedWasExplicit) return null;
65
+ if (/^\d{4}-\d{2}-\d{2}/.test(String(since).trim())) return null; // absolute: fine
66
+ return ` --seed makes the draw reproducible only over a fixed pool, and '${since}' is a
67
+ MOVING window — the same seed will draw a different sample tomorrow.
68
+ For a number you can compare over time, pass an absolute date: --since '2026-06-01'.`;
69
+ }
@@ -18,6 +18,12 @@ import path from 'node:path';
18
18
  const CONFIGS = [
19
19
  { flavour: 'vitest', files: ['vitest.config.ts', 'vitest.config.js', 'vitest.config.mjs', 'vite.config.ts', 'vite.config.js'] },
20
20
  { flavour: 'jest', files: ['jest.config.js', 'jest.config.ts', 'jest.config.mjs', 'jest.config.json'] },
21
+ // Playwright is detected precisely so it can be DECLINED by name. There is no
22
+ // Playwright runner here, and falling through to node:test made a Playwright spec
23
+ // report `arm A did not run (node): test failed` — which reads as "your test is
24
+ // broken" when the truth is "I used the wrong tool and should have said so".
25
+ // Found by a user pointing this at a real Playwright suite.
26
+ { flavour: 'playwright', files: ['playwright.config.ts', 'playwright.config.js', 'playwright.config.mjs'] },
21
27
  ];
22
28
 
23
29
  const exists = async p => { try { await access(p); return true; } catch { return false; } };
@@ -49,6 +55,7 @@ export async function selectRunner(worktreeRoot, relTestPath) {
49
55
  if (/\bvitest\b/.test(script)) return { flavour: 'vitest', pkgDir: dir === '.' ? '' : dir };
50
56
  if (/\bjest\b/.test(script)) return { flavour: 'jest', pkgDir: dir === '.' ? '' : dir };
51
57
  if (/node\s+--test|\bnode:test\b/.test(script)) return { flavour: 'node', pkgDir: dir === '.' ? '' : dir };
58
+ if (/\bplaywright\s+test\b/.test(script)) return { flavour: 'playwright', pkgDir: dir === '.' ? '' : dir };
52
59
  } catch { /* unparseable package.json is not a signal */ }
53
60
  }
54
61
 
package/src/tap.mjs CHANGED
@@ -71,13 +71,3 @@ export function parseTap(stdout) {
71
71
  }
72
72
  return cases;
73
73
  }
74
-
75
- /**
76
- * A whole-file load failure. node:test reports this as a `not ok` for the FILE path with
77
- * no individual cases, which must not be mistaken for every case failing by assertion.
78
- */
79
- export function isFileLevelFailure(cases, testFile) {
80
- if (cases.length !== 1) return false;
81
- const only = cases[0];
82
- return only.status === 'fail' && (only.name.includes('/') || only.name.endsWith('.mjs') || only.name.endsWith('.js') || only.name === testFile);
83
- }