control-arm 0.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/action.yml DELETED
@@ -1,95 +0,0 @@
1
- # A composite action, not a Docker one: the tool is plain Node with no dependencies, so a
2
- # container would add a minute of build time to a check that otherwise takes seconds.
3
- name: 'control-arm'
4
- description: 'Prove the tests in this PR fail without it'
5
- inputs:
6
- base:
7
- description: >-
8
- Branch this PR targets. The comparison uses the MERGE BASE with it, never its tip.
9
- Defaults to `main`; set it explicitly if your project integrates somewhere else
10
- (`develop`, `trunk`, a release branch).
11
- required: false
12
- default: 'main'
13
- comment:
14
- description: 'Post the result as a PR comment. Needs pull-requests:write and GH_TOKEN.'
15
- required: false
16
- default: 'true'
17
- fail-on-blind:
18
- description: >-
19
- Fail the check when NO case in the PR discriminates. Default false: a PR can
20
- legitimately ship only regression guards, and a gate that fires on those gets
21
- switched off within a week.
22
- required: false
23
- default: 'false'
24
- timeout-ms:
25
- required: false
26
- default: '180000'
27
- outputs:
28
- verdict:
29
- description: 'CAUGHT | BLIND | INCONCLUSIVE | FLAKY | SKIPPED'
30
- value: ${{ steps.run.outputs.verdict }}
31
- runs:
32
- using: composite
33
- steps:
34
- - id: run
35
- shell: bash
36
- env:
37
- CA_BASE: ${{ inputs.base }}
38
- CA_TIMEOUT: ${{ inputs.timeout-ms }}
39
- run: |
40
- set -uo pipefail
41
- # The merge base needs both histories. A shallow checkout has neither.
42
- git -C "$GITHUB_WORKSPACE" fetch --no-tags --depth=200 origin "$CA_BASE" 2>/dev/null || true
43
- TIP="$(git -C "$GITHUB_WORKSPACE" rev-parse HEAD)"
44
-
45
- OUT=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
46
- --against "origin/$CA_BASE" \
47
- --repo "$GITHUB_WORKSPACE" \
48
- --timeout "$CA_TIMEOUT" 2>&1) || true
49
- echo "$OUT"
50
-
51
- # `|| true` is load-bearing. GitHub runs bash steps with `set -e` by default, and
52
- # `set -uo pipefail` above does not unset it. A grep that finds nothing exits 1,
53
- # pipefail propagates it, and the step dies BEFORE the :-INCONCLUSIVE fallback can
54
- # run. That is exactly what happens on a PR the tool declines — a workflow-only
55
- # change with no test — so the one case designed to be a non-event failed the check.
56
- VERDICT=$(printf '%s' "$OUT" | grep -oE 'VERDICT +[A-Z]+' | awk '{print $2}' | head -1 || true)
57
- VERDICT="${VERDICT:-INCONCLUSIVE}"
58
- echo "verdict=$VERDICT" >> "$GITHUB_OUTPUT"
59
-
60
- # `--pr-comment` returns EMPTY for a commit the tool declined to judge, and the
61
- # comment step skips on an empty file. The decision is the tool's, not a shell's:
62
- # scraping stdout for a VERDICT line a declined run never prints is what put
63
- # "SKIPPED · no case fails without the change" onto a PR that simply has no tests.
64
- MD=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
65
- --against "origin/$CA_BASE" --pr-comment \
66
- --repo "$GITHUB_WORKSPACE" --timeout "$CA_TIMEOUT" 2>/dev/null) || true
67
- printf '%s' "$MD" > "$RUNNER_TEMP/ca-comment.md"
68
- [ -s "$RUNNER_TEMP/ca-comment.md" ] || echo "declined — nothing to post: $(printf '%s' "$OUT" | tail -1)"
69
-
70
- - shell: bash
71
- if: inputs.comment == 'true' && github.event_name == 'pull_request'
72
- env:
73
- GITHUB_TOKEN: ${{ github.token }}
74
- run: |
75
- # node, not `gh`. The CLI is not installed on every runner — a self-hosted
76
- # bare-metal box failed here with `gh: command not found` after every other step
77
- # had passed, so the tool ran, reached the right verdict, and could not say so.
78
- node "${{ github.action_path }}/scripts/gh-api.mjs" upsert-pr-comment \
79
- "${{ github.repository }}" \
80
- "${{ github.event.pull_request.number }}" \
81
- "$RUNNER_TEMP/ca-comment.md" \
82
- '### `control-arm`'
83
-
84
- - shell: bash
85
- if: inputs.fail-on-blind == 'true' && steps.run.outputs.verdict == 'BLIND'
86
- run: |
87
- echo "::error::No test in this PR fails without the change."
88
- exit 1
89
-
90
- # Required by the GitHub Marketplace listing. `rewind` is not decoration: it is
91
- # literally what this does — wind the code back to before the fix, then run the
92
- # new test against it.
93
- branding:
94
- icon: 'rewind'
95
- color: 'orange'
@@ -1,178 +0,0 @@
1
- /**
2
- * Fixture repos with KNOWN answers — this tool's own control arm.
3
- *
4
- * The tool's whole claim is "a test that cannot fail is decoration". A tool that asserts
5
- * that about other people's tests, while its own suite only checks that it doesn't crash,
6
- * is the same defect one level up. So: tiny git repos where the right verdict is known by
7
- * construction, and `test/fixtures.test.mjs` asserts the tool returns it.
8
- *
9
- * If `ca` cannot tell 02-blind-direction from 01-caught-value, it does not ship.
10
- *
11
- * THE BUG THEY ALL SHARE. `priority()` maps a label to a number. The broken version joins
12
- * its words with `\s*`, which matches whitespace and nothing else, so it reads
13
- * "HIGH PRIORITY" but not "HIGH-PRIORITY" — one separator, a different answer. The fixed
14
- * version accepts any run of real separators.
15
- *
16
- * Deliberately a boring, universal domain: every issue tracker has priority labels, and
17
- * the fixtures should not require knowing anybody's product to read. Only the TEST differs
18
- * between fixtures; the bug is identical in all of them, so a verdict can only come from
19
- * the test's quality.
20
- */
21
-
22
- import { execFileSync } from 'node:child_process';
23
- import { mkdirSync, writeFileSync, rmSync } from 'node:fs';
24
- import path from 'node:path';
25
- import { fileURLToPath } from 'node:url';
26
-
27
- const ROOT = path.join(path.dirname(fileURLToPath(import.meta.url)), '.build');
28
-
29
- const BROKEN = `export function priority(label) {
30
- const s = String(label).toLowerCase();
31
- if (/high\\s*priority/.test(s)) return 1; // \\s* matches whitespace and nothing else
32
- if (/low\\s*priority/.test(s)) return 3;
33
- return 2;
34
- }
35
- export const SLA_HOURS = { 1: 4, 2: 24, 3: 72 };
36
- `;
37
- const FIXED = BROKEN
38
- .replace('/high\\s*priority/', '/high[-_\\s.]*priority/')
39
- .replace('/low\\s*priority/', '/low[-_\\s.]*priority/');
40
-
41
- const FIXTURES = {
42
- '01-caught-value': {
43
- expect: 'CAUGHT',
44
- why: 'asserts the exact priority for the hyphenated spelling',
45
- test: `import { test } from 'node:test';
46
- import assert from 'node:assert/strict';
47
- import { priority, SLA_HOURS } from '../src/priority.mjs';
48
- test('a hyphenated HIGH-PRIORITY label is priority 1, four-hour SLA', () => {
49
- assert.equal(priority('HIGH-PRIORITY'), 1);
50
- assert.equal(SLA_HOURS[priority('HIGH-PRIORITY')], 4);
51
- });`,
52
- },
53
- '02-blind-direction': {
54
- expect: 'BLIND',
55
- why: 'asserts a DIRECTION (>0) that the wrong answer also satisfies',
56
- test: `import { test } from 'node:test';
57
- import assert from 'node:assert/strict';
58
- import { priority, SLA_HOURS } from '../src/priority.mjs';
59
- test('hyphenated labels are handled', () => {
60
- const p = priority('HIGH-PRIORITY');
61
- assert.ok(p, 'a priority comes back');
62
- assert.ok(SLA_HOURS[p] > 0, 'it has an SLA');
63
- assert.notEqual(p, undefined);
64
- });`,
65
- },
66
- '03-blind-sourcetext': {
67
- expect: 'BLIND',
68
- why: 'greps the source instead of executing it',
69
- test: `import { test } from 'node:test';
70
- import assert from 'node:assert/strict';
71
- import { readFileSync } from 'node:fs';
72
- test('the separator class is tolerant', () => {
73
- const src = readFileSync(new URL('../src/priority.mjs', import.meta.url), 'utf8');
74
- assert.ok(src.includes('priority'), 'the rule mentions priority');
75
- assert.ok(/high/.test(src));
76
- });`,
77
- },
78
- '04-blind-overmock': {
79
- expect: 'BLIND',
80
- why: 'mocks the unit under test, so the real function never runs',
81
- test: `import { test } from 'node:test';
82
- import assert from 'node:assert/strict';
83
- import { SLA_HOURS } from '../src/priority.mjs';
84
- const priority = () => 1; // "stubbed for speed"
85
- test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
86
- assert.equal(priority('HIGH-PRIORITY'), 1);
87
- assert.equal(SLA_HOURS[1], 4);
88
- });`,
89
- },
90
- '05-inconclusive-newexport': {
91
- expect: 'INCONCLUSIVE',
92
- why: 'imports a symbol the fix added — cannot even load at the parent',
93
- fixedExtra: `export const SEPARATORS = /[-_\\s.]*/;\n`,
94
- test: `import { test } from 'node:test';
95
- import assert from 'node:assert/strict';
96
- import { SEPARATORS } from '../src/priority.mjs';
97
- test('the separator class is shared', () => {
98
- assert.equal(SEPARATORS.source, '[-_\\\\s.]*');
99
- });`,
100
- },
101
- '06-inconclusive-armA-red': {
102
- expect: 'INCONCLUSIVE',
103
- why: 'the case is not green on the fix either — the commit does not stand up',
104
- test: `import { test } from 'node:test';
105
- import assert from 'node:assert/strict';
106
- import { priority } from '../src/priority.mjs';
107
- test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
108
- assert.equal(priority('HIGH-PRIORITY'), 99);
109
- });`,
110
- },
111
- '07-caught-mixed': {
112
- expect: 'CAUGHT',
113
- why: 'one discriminating case plus two regression guards green on both arms',
114
- test: `import { test } from 'node:test';
115
- import assert from 'node:assert/strict';
116
- import { priority } from '../src/priority.mjs';
117
- test('DISCRIMINATES: the hyphenated spelling is priority 1', () => {
118
- assert.equal(priority('HIGH-PRIORITY'), 1);
119
- });
120
- test('GUARD: the spaced spelling is still priority 1', () => {
121
- assert.equal(priority('HIGH PRIORITY'), 1);
122
- });
123
- test('GUARD: an unlabelled ticket is still the default priority 2', () => {
124
- assert.equal(priority('needs triage'), 2);
125
- });`,
126
- },
127
- '09-caught-slash-in-name': {
128
- expect: 'CAUGHT',
129
- why: 'the discriminating case has a SLASH in its name — it used to be silently dropped',
130
- test: `import { test } from 'node:test';
131
- import assert from 'node:assert/strict';
132
- import { priority } from '../src/priority.mjs';
133
- test('label parsing / separator handling', () => {
134
- assert.equal(priority('HIGH-PRIORITY'), 1);
135
- });`,
136
- },
137
- '08-skipped-notest': {
138
- expect: 'SKIPPED',
139
- why: 'the fix shipped no test at all',
140
- test: null,
141
- },
142
- };
143
-
144
- function sh(cwd, args) { execFileSync('git', args, { cwd, stdio: 'pipe' }); }
145
-
146
- export function buildFixtures() {
147
- rmSync(ROOT, { recursive: true, force: true });
148
- mkdirSync(ROOT, { recursive: true });
149
- const built = {};
150
-
151
- for (const [name, spec] of Object.entries(FIXTURES)) {
152
- const dir = path.join(ROOT, name);
153
- mkdirSync(path.join(dir, 'src'), { recursive: true });
154
- mkdirSync(path.join(dir, 'tests'), { recursive: true });
155
- sh(dir, ['init', '-q']);
156
- sh(dir, ['config', 'user.email', 'ca@fixture']);
157
- sh(dir, ['config', 'user.name', 'ca fixture']);
158
- writeFileSync(path.join(dir, 'package.json'), JSON.stringify({ name, type: 'module', private: true }, null, 2));
159
-
160
- // --- parent: the bug, and whatever tests existed before (none) ---
161
- writeFileSync(path.join(dir, 'src/priority.mjs'), BROKEN);
162
- sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'feat: priority labels']);
163
-
164
- // --- fix: source repaired, test added ---
165
- writeFileSync(path.join(dir, 'src/priority.mjs'), FIXED + (spec.fixedExtra || ''));
166
- if (spec.test) writeFileSync(path.join(dir, 'tests/priority.test.mjs'), spec.test + '\n');
167
- sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'fix: a hyphen made a HIGH-PRIORITY ticket read as normal']);
168
-
169
- built[name] = { dir, sha: execFileSync('git', ['-C', dir, 'rev-parse', 'HEAD']).toString().trim(), ...spec };
170
- }
171
- return built;
172
- }
173
-
174
- if (import.meta.url === `file://${process.argv[1]}`) {
175
- const b = buildFixtures();
176
- for (const [n, f] of Object.entries(b)) console.log(` ${f.expect.padEnd(13)} ${n.padEnd(26)} ${f.why}`);
177
- console.log(`\n ${Object.keys(b).length} fixtures in ${ROOT}\n`);
178
- }
@@ -1,90 +0,0 @@
1
- #!/usr/bin/env node
2
- /**
3
- * The few GitHub API calls these workflows need, over plain fetch.
4
- *
5
- * WHY NOT THE `gh` CLI. It is not installed on every runner. Measured: a self-hosted
6
- * bare-metal runner failed with `gh: command not found` (exit 127) after every other step
7
- * had passed — the tool ran, reached the right verdict, and then could not say so.
8
- * GitHub-hosted runners ship `gh`; a self-hosted one ships whatever was installed on it,
9
- * and a workflow that assumes otherwise works until it lands on the wrong machine.
10
- *
11
- * Node is already a hard requirement here — the tool is written in it — so this adds no
12
- * dependency at all. Usage:
13
- *
14
- * gh-api.mjs upsert-pr-comment <repo> <pr> <body-file> <marker>
15
- * gh-api.mjs upsert-issue <repo> <title> <body-file> <search> [label]
16
- *
17
- * Reads GITHUB_TOKEN / GH_TOKEN from the environment. Prints what it did, and exits
18
- * non-zero only when the API refuses — never merely because there was nothing to do.
19
- */
20
-
21
- const TOKEN = process.env.GITHUB_TOKEN || process.env.GH_TOKEN;
22
- const API = process.env.GITHUB_API_URL || 'https://api.github.com';
23
-
24
- async function gh(path, init = {}) {
25
- const res = await fetch(`${API}${path}`, {
26
- ...init,
27
- headers: {
28
- authorization: `Bearer ${TOKEN}`,
29
- accept: 'application/vnd.github+json',
30
- 'content-type': 'application/json',
31
- 'x-github-api-version': '2022-11-28',
32
- ...(init.headers || {}),
33
- },
34
- });
35
- if (!res.ok) {
36
- const text = await res.text();
37
- throw new Error(`${init.method || 'GET'} ${path} → ${res.status} ${text.slice(0, 300)}`);
38
- }
39
- return res.status === 204 ? null : res.json();
40
- }
41
-
42
- const [cmd, ...args] = process.argv.slice(2);
43
- const read = async (f) => (await import('node:fs/promises')).readFile(f, 'utf8');
44
-
45
- try {
46
- if (!TOKEN) throw new Error('no GITHUB_TOKEN / GH_TOKEN in the environment');
47
-
48
- if (cmd === 'upsert-pr-comment') {
49
- const [repo, pr, bodyFile, marker] = args;
50
- const body = await read(bodyFile);
51
- if (!body.trim()) { console.log('nothing to post'); process.exit(0); }
52
- // ONE COMMENT PER PR, edited in place. A new comment per push turns a useful signal
53
- // into noise by the third revision, and the marker is how we find ours again.
54
- const comments = await gh(`/repos/${repo}/issues/${pr}/comments?per_page=100`);
55
- const mine = comments.find(c => (c.body || '').startsWith(marker));
56
- if (mine) {
57
- await gh(`/repos/${repo}/issues/comments/${mine.id}`, { method: 'PATCH', body: JSON.stringify({ body }) });
58
- console.log(`updated comment ${mine.id}`);
59
- } else {
60
- const made = await gh(`/repos/${repo}/issues/${pr}/comments`, { method: 'POST', body: JSON.stringify({ body }) });
61
- console.log(`posted comment ${made.id}`);
62
- }
63
- } else if (cmd === 'upsert-issue') {
64
- const [repo, title, bodyFile, search, label] = args;
65
- const body = await read(bodyFile);
66
- if (label) {
67
- // Create the label if absent. `labels` on an issue with an unknown label is
68
- // rejected outright, so this cannot be left to chance — but a failure here is
69
- // a warning, not a reason to drop the finding on the floor.
70
- try {
71
- await gh(`/repos/${repo}/labels`, { method: 'POST', body: JSON.stringify({ name: label, color: '0E8A16', description: 'Test-gap findings from control-arm' }) });
72
- } catch (e) { if (!/already_exists|422/.test(e.message)) console.log(`::warning::could not create label ${label}: ${e.message}`); }
73
- }
74
- const found = await gh(`/search/issues?q=${encodeURIComponent(`repo:${repo} is:issue is:open ${search}`)}`);
75
- const hit = found.items?.[0];
76
- if (hit) {
77
- await gh(`/repos/${repo}/issues/${hit.number}`, { method: 'PATCH', body: JSON.stringify({ title, body }) });
78
- console.log(`refreshed issue #${hit.number}`);
79
- } else {
80
- const made = await gh(`/repos/${repo}/issues`, { method: 'POST', body: JSON.stringify({ title, body, ...(label ? { labels: [label] } : {}) }) });
81
- console.log(`filed issue #${made.number}`);
82
- }
83
- } else {
84
- console.error('usage: gh-api.mjs upsert-pr-comment|upsert-issue ...');
85
- process.exit(2);
86
- }
87
- } catch (e) {
88
- console.error(`::error::${e.message}`);
89
- process.exit(1);
90
- }
@@ -1,86 +0,0 @@
1
- /**
2
- * Segment the audit CSV by which runner the test file belongs to.
3
- *
4
- * `ca` ships one runner (node:test). A monorepo's fix commits touch vitest and jest test
5
- * files too, and those come back INCONCLUSIVE for a reason that says nothing about the
6
- * test — the tool simply cannot execute it. Reporting one blended ratio over both would
7
- * be the same defect the tool exists to catch: a number fitted to a population it does
8
- * not describe.
9
- */
10
- import { readFileSync } from 'node:fs';
11
-
12
- const rows = [];
13
- const raw = readFileSync(process.argv[2], 'utf8').split('\n').filter(Boolean);
14
- const hdr = raw.shift().split(',');
15
- for (const line of raw) {
16
- // naive CSV with quoted fields
17
- const f = []; let cur = '', q = false;
18
- for (let i = 0; i < line.length; i++) {
19
- const c = line[i];
20
- if (q) { if (c === '"' && line[i + 1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
21
- else if (c === '"') q = true;
22
- else if (c === ',') { f.push(cur); cur = ''; }
23
- else cur += c;
24
- }
25
- f.push(cur);
26
- rows.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
27
- }
28
-
29
- const runnerOf = (file) => {
30
- if (!file) return 'none';
31
- if (file.startsWith('tests/')) return 'node:test (supported)';
32
- if (file.startsWith('apps/web/')) return 'vitest (unsupported)';
33
- if (file.startsWith('apps/mobile/')) return 'jest (unsupported)';
34
- if (file.startsWith('apps/e2e/')) return 'playwright (unsupported)';
35
- if (file.startsWith('packages/')) return 'node:test (supported)';
36
- return 'other';
37
- };
38
-
39
- const commits = new Map();
40
- for (const r of rows) {
41
- if (!commits.has(r.sha)) commits.set(r.sha, { sha: r.sha, date: r.date, subject: r.subject, verdict: r.commit_verdict, cases: [] });
42
- commits.get(r.sha).cases.push(r);
43
- }
44
-
45
- // A commit belongs to the runner of its test files; mixed commits are called out.
46
- const seg = new Map();
47
- for (const c of commits.values()) {
48
- const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
49
- const key = rs.length === 0 ? 'no test file' : rs.length === 1 ? rs[0] : 'mixed';
50
- if (!seg.has(key)) seg.set(key, []);
51
- seg.get(key).push(c);
52
- }
53
-
54
- const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
55
- console.log('\n COMMITS BY RUNNER SEGMENT\n');
56
- console.log(' ' + 'segment'.padEnd(26) + V.map(v => v.slice(0, 6).padStart(7)).join('') + ' n');
57
- for (const [k, list] of [...seg].sort((a, b) => b[1].length - a[1].length)) {
58
- const t = V.map(v => String(list.filter(c => c.verdict === v).length).padStart(7)).join('');
59
- console.log(' ' + k.padEnd(26) + t + String(list.length).padStart(5));
60
- }
61
-
62
- const supported = seg.get('node:test (supported)') || [];
63
- const dec = supported.filter(c => c.verdict === 'CAUGHT' || c.verdict === 'BLIND');
64
- console.log(`\n SUPPORTED SEGMENT ONLY (node:test)\n`);
65
- console.log(` ${supported.length} commits · ${dec.length} the instrument could answer`);
66
- if (dec.length) {
67
- const caught = dec.filter(c => c.verdict === 'CAUGHT').length;
68
- console.log(` CAUGHT ${caught}/${dec.length} = ${(caught / dec.length * 100).toFixed(1)}% BLIND ${dec.length - caught}/${dec.length} = ${((dec.length - caught) / dec.length * 100).toFixed(1)}%`);
69
- }
70
- const blind = supported.filter(c => c.verdict === 'BLIND');
71
- if (blind.length) {
72
- console.log(`\n BLIND COMMITS IN THE SUPPORTED SEGMENT (hand-audit these)\n`);
73
- for (const c of blind) console.log(` ${c.sha.slice(0, 8)} ${c.date} ${c.subject.slice(0, 90)}`);
74
- }
75
- const why = new Map();
76
- for (const c of supported.filter(c => c.verdict === 'INCONCLUSIVE')) {
77
- for (const cs of c.cases.filter(x => x.case_verdict === 'INCONCLUSIVE')) {
78
- const k = cs.reason.replace(/'[^']*'/g, "'…'").replace(/\d+/g, 'N').slice(0, 88);
79
- why.set(k, (why.get(k) || 0) + 1);
80
- }
81
- }
82
- if (why.size) {
83
- console.log(`\n WHY INCONCLUSIVE, supported segment only\n`);
84
- for (const [k, v] of [...why].sort((a, b) => b[1] - a[1]).slice(0, 14)) console.log(` ${String(v).padStart(4)} ${k}`);
85
- }
86
- console.log('');
@@ -1,53 +0,0 @@
1
- /**
2
- * Re-roll commit verdicts from a saved audit CSV using the CURRENT rollUp.
3
- *
4
- * Per-case verdicts are what the two arms measured; the commit verdict is a pure function
5
- * of them. So a change to the roll-up rule does not need the 7-minute run again — and
6
- * re-running would also change the sample, which is the wrong thing to do when comparing
7
- * a rule change.
8
- */
9
- import { readFileSync } from 'node:fs';
10
- import { rollUp } from './src/verdict.mjs';
11
-
12
- function parseCsv(text) {
13
- const out = []; const lines = text.split('\n').filter(Boolean); const hdr = lines.shift().split(',');
14
- for (const line of lines) {
15
- const f = []; let cur = '', q = false;
16
- for (let i = 0; i < line.length; i++) { const c = line[i];
17
- if (q) { if (c === '"' && line[i+1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
18
- else if (c === '"') q = true; else if (c === ',') { f.push(cur); cur = ''; } else cur += c; }
19
- f.push(cur); out.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
20
- }
21
- return out;
22
- }
23
- const runnerOf = f => !f ? 'none'
24
- : f.startsWith('tests/') || f.startsWith('packages/') ? 'node:test'
25
- : f.startsWith('apps/web/') ? 'vitest' : f.startsWith('apps/mobile/') ? 'jest'
26
- : f.startsWith('apps/e2e/') ? 'playwright' : 'other';
27
-
28
- const rows = parseCsv(readFileSync(process.argv[2], 'utf8'));
29
- const commits = new Map();
30
- for (const r of rows) {
31
- if (!commits.has(r.sha)) commits.set(r.sha, { ...r, cases: [] });
32
- if (r.case) commits.get(r.sha).cases.push(r);
33
- }
34
- const list = [...commits.values()].map(c => {
35
- const nv = c.cases.length ? rollUp(c.cases.map(x => ({ verdict: x.case_verdict }))) : c.commit_verdict;
36
- const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
37
- return { ...c, was: c.commit_verdict, now: nv, seg: rs.length === 1 ? rs[0] : rs.length ? 'mixed' : 'none' };
38
- });
39
-
40
- const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
41
- const changed = list.filter(c => c.was !== c.now);
42
- console.log(`\n ${list.length} commits · ${changed.length} changed verdict under the corrected roll-up\n`);
43
- for (const c of changed) console.log(` ${c.was.padEnd(13)} -> ${c.now.padEnd(13)} ${c.sha.slice(0,8)} ${c.subject.slice(0,70)}`);
44
-
45
- const node = list.filter(c => c.seg === 'node:test' || c.seg === 'mixed');
46
- const dec = node.filter(c => c.now === 'CAUGHT' || c.now === 'BLIND');
47
- const caught = dec.filter(c => c.now === 'CAUGHT').length;
48
- console.log(`\n SUPPORTED SEGMENT (node:test, incl. mixed): ${node.length} commits`);
49
- for (const v of V) { const n = node.filter(c => c.now === v).length; if (n) console.log(` ${v.padEnd(14)} ${String(n).padStart(4)} ${(n/node.length*100).toFixed(1)}%`); }
50
- console.log(`\n ANSWERABLE: ${dec.length} CAUGHT ${caught} (${(caught/dec.length*100).toFixed(1)}%) BLIND ${dec.length-caught} (${((dec.length-caught)/dec.length*100).toFixed(1)}%)\n`);
51
- console.log(' BLIND after correction:\n');
52
- for (const c of list.filter(c => c.now === 'BLIND')) console.log(` ${c.sha.slice(0,8)} ${c.date} ${c.subject.slice(0,84)}`);
53
- console.log('');
@@ -1,130 +0,0 @@
1
- /**
2
- * Case extraction, pinned against the four shapes that actually broke it.
3
- *
4
- * Each of these was measured on a real corpus of 400 test cases, and each one alone
5
- * accounted for a double-digit share of "could not analyse". Together they took the
6
- * unresolvable pile from 64% to 1.3%. None would be caught by a suite that only fed the
7
- * analyser well-formed input, which is why they are here with the messy input attached.
8
- */
9
- import { test } from 'node:test';
10
- import assert from 'node:assert/strict';
11
- import { extractCase, analyseCase } from '../src/assertions.mjs';
12
-
13
- test('nested suites: the runner reports the CONCATENATED name, the source holds the inner one', () => {
14
- const src = `
15
- describe('auction client', () => {
16
- it('AUCWEB-001 calls GET /auction/:vin', () => { assert.equal(a, 1); });
17
- });`;
18
- // Reported by the runner as describe-title + space + it-title.
19
- const body = extractCase(src, 'auction client AUCWEB-001 calls GET /auction/:vin');
20
- assert.ok(body, 'progressive reduction must find the inner title');
21
- assert.match(body, /assert\.equal\(a, 1\)/);
22
- });
23
-
24
- test('an APOSTROPHE in the title: source holds a backslash the reported name does not', () => {
25
- const src = `it('METER-038: usage on the user\\'s LOCAL today is counted', () => { assert.equal(n, 3); });`;
26
- const body = extractCase(src, "METER-038: usage on the user's LOCAL today is counted");
27
- assert.ok(body, "an escaped quote inside the literal must still match the unescaped reported name");
28
- assert.match(body, /assert\.equal\(n, 3\)/);
29
- });
30
-
31
- test('JSX: a closing tag is NOT a regex literal', () => {
32
- // `</div>` puts a slash after `<`. The usual "slash after a non-expression starts a
33
- // regex" heuristic swallows the rest of the component, and the body is never found.
34
- const src = `
35
- it('hero renders', () => {
36
- render(<div className="x">{name}</div>);
37
- expect(screen.getByText('hi')).toBeTruthy();
38
- });`;
39
- const body = extractCase(src, 'hero renders');
40
- assert.ok(body, 'a JSX body must brace-match cleanly');
41
- assert.match(body, /getByText/);
42
- });
43
-
44
- test('braces inside strings and regexes do not unbalance the scan', () => {
45
- const src = `
46
- it('handles braces', () => {
47
- const s = '{';
48
- const re = /[{}]/;
49
- assert.equal(f(s, re), '}');
50
- });`;
51
- const body = extractCase(src, 'handles braces');
52
- assert.ok(body, 'a brace inside a string or a character class is not structure');
53
- assert.match(body, /assert\.equal\(f\(s, re\)/);
54
- });
55
-
56
- test('template-literal titles: every generated case resolves to the shared body', () => {
57
- const src = `
58
- for (const rel of PAGES) {
59
- it(\`\${rel}: no internal repo paths\`, () => { expect(md).not.toMatch(BAD); });
60
- }`;
61
- const a = extractCase(src, 'app/page.tsx: no internal repo paths');
62
- const b = extractCase(src, 'app/pricing/page.tsx: no internal repo paths');
63
- assert.ok(a && b, 'a parameterised title must match by pattern, not by equality');
64
- assert.equal(a, b, 'all generated cases share one body — that is correct, not a bug');
65
- });
66
-
67
- test('a wrong body is worse than none: a short remainder is refused', () => {
68
- const src = `
69
- it('alpha returns null', () => { assert.equal(x, null); });
70
- it('beta returns null', () => { assert.equal(y, 0); });`;
71
- // "returns null" alone is ambiguous between the two — reduction must not take it.
72
- const body = extractCase(src, 'some suite that does not exist returns null');
73
- assert.equal(body, null, 'an ambiguous short suffix must not resolve to an arbitrary case');
74
- });
75
-
76
- test('CONTROL ARM: naive equality-only extraction fails every one of these', () => {
77
- // The implementation this replaced. Asserted to still get them wrong.
78
- const naive = (src, name) => {
79
- const esc = name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
80
- return new RegExp(`\\b(?:test|it)\\s*\\(\\s*(['"\`])${esc}\\1`).test(src);
81
- };
82
- assert.equal(naive(`it('a b c', () => {});`, 'suite a b c'), false); // nested
83
- assert.equal(naive(`it('u\\'s v', () => {});`, "u's v"), false); // apostrophe
84
- assert.equal(naive('it(`${r}: x`, () => {});', 'p.tsx: x'), false); // template
85
- });
86
-
87
- test('the analyser reports UNKNOWN rather than a clean bill when it cannot tell', () => {
88
- const r = analyseCase(`it('odd', () => { somethingEntirelyUnrecognised(); });`, 'odd');
89
- assert.notEqual(r.verdict, 'strong', 'an unrecognised body must never read as strong');
90
- });
91
-
92
- test('a short helper name must not be accused of stubbing the unit under test', () => {
93
- // FOUND ON A REAL RUN. The stub check substring-matched the name against every import
94
- // path, so a two-letter helper `at` matched '@acme/core' — "acme" contains "at" — and
95
- // every case in that file was reported as testing a stub. A wrong explanation is worse
96
- // than none: it sends the reader to look at code that is fine.
97
- const src = `
98
- import { computeTier } from '@acme/core/tiers.js';
99
- const at = (arr, i) => arr[i];
100
- it('TIER-001: the tier changes the money', () => {
101
- assert.equal(at(computeTier(x), 0), 1200);
102
- });`;
103
- const r = analyseCase(src, 'TIER-001: the tier changes the money');
104
- assert.ok(!r.findings.some(f => f.id === 'stubbed-subject'),
105
- 'a local helper whose name merely appears inside an import path is not a stub');
106
- });
107
-
108
- test('a REAL stub is still caught — the tightening must not blind the check', () => {
109
- const src = `
110
- import { SLA_HOURS } from '../src/priority.mjs';
111
- const priority = () => 1;
112
- it('p is 1', () => { assert.equal(priority('HIGH-PRIORITY'), 1); });`;
113
- const r = analyseCase(src, 'p is 1');
114
- assert.ok(r.findings.some(f => f.id === 'stubbed-subject'),
115
- 'a stub matching the imported module basename must still be reported');
116
- assert.equal(r.verdict, 'weak');
117
- });
118
-
119
- test('a declined commit produces NO pr comment — silence, not an accusation', async () => {
120
- // It briefly posted "SKIPPED · No case in this branch fails without the change" onto a
121
- // workflow-only PR. Technically true and completely misleading: that PR has no tests,
122
- // so of course none discriminate. A comment reading as an accusation on a PR doing
123
- // nothing wrong is worse than no comment.
124
- const { prComment } = await import('../src/markdown-report.mjs');
125
- const declined = { short: 'abc12345', verdict: 'SKIPPED', cases: [], note: 'no test file in the commit' };
126
- assert.equal(prComment(declined), '', 'a noted (declined) result must render as empty');
127
-
128
- const real = { short: 'abc12345', verdict: 'CAUGHT', cases: [{ verdict: 'CAUGHT', name: 'x', reason: 'y' }] };
129
- assert.notEqual(prComment(real), '', 'a real verdict must still render');
130
- });