control-arm 0.1.0 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -46,7 +46,7 @@ Node 18+. No other dependencies.
46
46
 
47
47
  ```bash
48
48
  # Can this repo be measured at all?
49
- ca probe
49
+ ca doctor
50
50
 
51
51
  # Judge one commit — a fix, with the test that shipped alongside it
52
52
  ca verify <sha>
@@ -62,6 +62,12 @@ ca verify HEAD --against origin/main
62
62
  name: control-arm
63
63
  on: pull_request
64
64
 
65
+ # `comment: true` posts with the default GITHUB_TOKEN, which is read-only in most
66
+ # repos. Without this block the run succeeds and the comment silently never appears.
67
+ permissions:
68
+ contents: read
69
+ pull-requests: write
70
+
65
71
  jobs:
66
72
  verify:
67
73
  runs-on: ubuntu-latest
package/package.json CHANGED
@@ -1,9 +1,32 @@
1
1
  {
2
2
  "name": "control-arm",
3
- "version": "0.1.0",
3
+ "version": "1.0.0",
4
4
  "description": "Does a test actually fail on the code it was written to catch?",
5
5
  "type": "module",
6
6
  "bin": { "ca": "./bin/ca.mjs" },
7
+ "files": [
8
+ "bin",
9
+ "src",
10
+ "README.md",
11
+ "DESIGN.md",
12
+ "LICENSE"
13
+ ],
14
+ "engines": { "node": ">=18" },
15
+ "keywords": [
16
+ "testing",
17
+ "test-quality",
18
+ "mutation-testing",
19
+ "regression-testing",
20
+ "continuous-integration",
21
+ "github-actions",
22
+ "code-quality"
23
+ ],
24
+ "repository": {
25
+ "type": "git",
26
+ "url": "git+https://github.com/mekanhan/control-arm.git"
27
+ },
28
+ "homepage": "https://github.com/mekanhan/control-arm#readme",
29
+ "bugs": { "url": "https://github.com/mekanhan/control-arm/issues" },
7
30
  "scripts": {
8
31
  "test": "node --test test/*.test.mjs",
9
32
  "fixtures": "node fixtures/build.mjs"
@@ -1,82 +0,0 @@
1
- # The tool that asks whether your tests can fail, running its own.
2
- #
3
- # It shipped for a day without this. That is the defect it exists to find, one level up:
4
- # a suite that is never executed by anything but its author's terminal is not a gate, and
5
- # "46/46 green" meant "green on one laptop, when I remembered".
6
- #
7
- # GitHub-hosted runners on purpose: control-arm has no dependencies and must work on a
8
- # stock box. Pinning it to a self-hosted fleet would hide exactly the assumptions —
9
- # a preinstalled binary, a warm cache, a particular git version — that break for the
10
- # first stranger who clones it.
11
- name: CI
12
-
13
- on:
14
- push:
15
- branches: [master, main]
16
- pull_request:
17
- workflow_dispatch:
18
-
19
- concurrency:
20
- group: ci-${{ github.ref }}
21
- cancel-in-progress: true
22
-
23
- jobs:
24
- test:
25
- name: node ${{ matrix.node }}
26
- runs-on: ubuntu-latest
27
- strategy:
28
- fail-fast: false
29
- matrix:
30
- # 20 is the oldest LTS with a stable node:test reporter API, which the TAP parser
31
- # depends on. 24 is what it is developed against. A break in either is worth knowing.
32
- node: ['20', '22', '24']
33
- steps:
34
- - uses: actions/checkout@v4
35
- with:
36
- # FULL HISTORY, not the default shallow clone. The dogfood step verifies the tool
37
- # against one of its OWN past commits, and `git show 1845c1d3` on a depth-1
38
- # checkout fails with "unknown revision". Caught by this workflow's first run,
39
- # which is the argument for having it.
40
- fetch-depth: 0
41
- - uses: actions/setup-node@v4
42
- with:
43
- node-version: ${{ matrix.node }}
44
-
45
- - name: Unit + fixtures
46
- run: node --test --test-concurrency=1 test/*.test.mjs
47
-
48
- # DOGFOOD. The fixtures prove the verdicts on repos built to have known answers.
49
- # This proves the whole two-arm machinery works on a REAL history with real commits,
50
- # worktrees and module resolution — which is where every bug so far has come from.
51
- - name: Judge its own history
52
- run: |
53
- set -euo pipefail
54
- git config --global user.email ci@control-arm
55
- git config --global user.name ci
56
-
57
- OUT=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --timeout 120000)
58
- echo "$OUT"
59
-
60
- # That commit added the rule "a SKIPPED case blocks BLIND", with a test written
61
- # for it. If the tool cannot still see that test discriminate, the tool is broken
62
- # — regardless of what its unit suite says.
63
- # Match the VERDICT WORD, not the whole line. This grep was 'VERDICT CAUGHT'
64
- # and broke the moment a glyph was added between them — a cosmetic change that
65
- # failed a correctness gate, which trains people to edit the gate rather than
66
- # believe it. Anchor on what the check is actually about.
67
- echo "$OUT" | grep -qE '^ *VERDICT .*\bCAUGHT\b' \
68
- || { echo "::error::control-arm no longer judges its own fix correctly"; exit 1; }
69
- # And it must be the STRONG claim: 1845c1d3 is a repair, so a "weak evidence"
70
- # qualifier here would mean the new-code heuristic has started misfiring on fixes.
71
- echo "$OUT" | grep -q 'weak evidence' \
72
- && { echo "::error::a repair was labelled new code — the kind heuristic is wrong"; exit 1; }
73
- echo "$OUT" | grep -q 'module identity verified' \
74
- || { echo "::error::module identity was not proven — a verdict here is not trustworthy"; exit 1; }
75
-
76
- - name: Determinism
77
- run: |
78
- set -euo pipefail
79
- A=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --work /tmp/d1 --timeout 120000 | tail -14)
80
- B=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --work /tmp/d2 --timeout 120000 | tail -14)
81
- [ "$A" = "$B" ] || { echo "::error::two runs disagreed — the tool is not deterministic"; exit 1; }
82
- echo "two independent runs are byte-identical"
package/action.yml DELETED
@@ -1,95 +0,0 @@
1
- # A composite action, not a Docker one: the tool is plain Node with no dependencies, so a
2
- # container would add a minute of build time to a check that otherwise takes seconds.
3
- name: 'control-arm'
4
- description: 'Prove the tests in this PR fail without it'
5
- inputs:
6
- base:
7
- description: >-
8
- Branch this PR targets. The comparison uses the MERGE BASE with it, never its tip.
9
- Defaults to `main`; set it explicitly if your project integrates somewhere else
10
- (`develop`, `trunk`, a release branch).
11
- required: false
12
- default: 'main'
13
- comment:
14
- description: 'Post the result as a PR comment. Needs pull-requests:write and GH_TOKEN.'
15
- required: false
16
- default: 'true'
17
- fail-on-blind:
18
- description: >-
19
- Fail the check when NO case in the PR discriminates. Default false: a PR can
20
- legitimately ship only regression guards, and a gate that fires on those gets
21
- switched off within a week.
22
- required: false
23
- default: 'false'
24
- timeout-ms:
25
- required: false
26
- default: '180000'
27
- outputs:
28
- verdict:
29
- description: 'CAUGHT | BLIND | INCONCLUSIVE | FLAKY | SKIPPED'
30
- value: ${{ steps.run.outputs.verdict }}
31
- runs:
32
- using: composite
33
- steps:
34
- - id: run
35
- shell: bash
36
- env:
37
- CA_BASE: ${{ inputs.base }}
38
- CA_TIMEOUT: ${{ inputs.timeout-ms }}
39
- run: |
40
- set -uo pipefail
41
- # The merge base needs both histories. A shallow checkout has neither.
42
- git -C "$GITHUB_WORKSPACE" fetch --no-tags --depth=200 origin "$CA_BASE" 2>/dev/null || true
43
- TIP="$(git -C "$GITHUB_WORKSPACE" rev-parse HEAD)"
44
-
45
- OUT=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
46
- --against "origin/$CA_BASE" \
47
- --repo "$GITHUB_WORKSPACE" \
48
- --timeout "$CA_TIMEOUT" 2>&1) || true
49
- echo "$OUT"
50
-
51
- # `|| true` is load-bearing. GitHub runs bash steps with `set -e` by default, and
52
- # `set -uo pipefail` above does not unset it. A grep that finds nothing exits 1,
53
- # pipefail propagates it, and the step dies BEFORE the :-INCONCLUSIVE fallback can
54
- # run. That is exactly what happens on a PR the tool declines — a workflow-only
55
- # change with no test — so the one case designed to be a non-event failed the check.
56
- VERDICT=$(printf '%s' "$OUT" | grep -oE 'VERDICT +[A-Z]+' | awk '{print $2}' | head -1 || true)
57
- VERDICT="${VERDICT:-INCONCLUSIVE}"
58
- echo "verdict=$VERDICT" >> "$GITHUB_OUTPUT"
59
-
60
- # `--pr-comment` returns EMPTY for a commit the tool declined to judge, and the
61
- # comment step skips on an empty file. The decision is the tool's, not a shell's:
62
- # scraping stdout for a VERDICT line a declined run never prints is what put
63
- # "SKIPPED · no case fails without the change" onto a PR that simply has no tests.
64
- MD=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
65
- --against "origin/$CA_BASE" --pr-comment \
66
- --repo "$GITHUB_WORKSPACE" --timeout "$CA_TIMEOUT" 2>/dev/null) || true
67
- printf '%s' "$MD" > "$RUNNER_TEMP/ca-comment.md"
68
- [ -s "$RUNNER_TEMP/ca-comment.md" ] || echo "declined — nothing to post: $(printf '%s' "$OUT" | tail -1)"
69
-
70
- - shell: bash
71
- if: inputs.comment == 'true' && github.event_name == 'pull_request'
72
- env:
73
- GITHUB_TOKEN: ${{ github.token }}
74
- run: |
75
- # node, not `gh`. The CLI is not installed on every runner — a self-hosted
76
- # bare-metal box failed here with `gh: command not found` after every other step
77
- # had passed, so the tool ran, reached the right verdict, and could not say so.
78
- node "${{ github.action_path }}/scripts/gh-api.mjs" upsert-pr-comment \
79
- "${{ github.repository }}" \
80
- "${{ github.event.pull_request.number }}" \
81
- "$RUNNER_TEMP/ca-comment.md" \
82
- '### `control-arm`'
83
-
84
- - shell: bash
85
- if: inputs.fail-on-blind == 'true' && steps.run.outputs.verdict == 'BLIND'
86
- run: |
87
- echo "::error::No test in this PR fails without the change."
88
- exit 1
89
-
90
- # Required by the GitHub Marketplace listing. `rewind` is not decoration: it is
91
- # literally what this does — wind the code back to before the fix, then run the
92
- # new test against it.
93
- branding:
94
- icon: 'rewind'
95
- color: 'orange'
@@ -1,178 +0,0 @@
1
- /**
2
- * Fixture repos with KNOWN answers — this tool's own control arm.
3
- *
4
- * The tool's whole claim is "a test that cannot fail is decoration". A tool that asserts
5
- * that about other people's tests, while its own suite only checks that it doesn't crash,
6
- * is the same defect one level up. So: tiny git repos where the right verdict is known by
7
- * construction, and `test/fixtures.test.mjs` asserts the tool returns it.
8
- *
9
- * If `ca` cannot tell 02-blind-direction from 01-caught-value, it does not ship.
10
- *
11
- * THE BUG THEY ALL SHARE. `priority()` maps a label to a number. The broken version joins
12
- * its words with `\s*`, which matches whitespace and nothing else, so it reads
13
- * "HIGH PRIORITY" but not "HIGH-PRIORITY" — one separator, a different answer. The fixed
14
- * version accepts any run of real separators.
15
- *
16
- * Deliberately a boring, universal domain: every issue tracker has priority labels, and
17
- * the fixtures should not require knowing anybody's product to read. Only the TEST differs
18
- * between fixtures; the bug is identical in all of them, so a verdict can only come from
19
- * the test's quality.
20
- */
21
-
22
- import { execFileSync } from 'node:child_process';
23
- import { mkdirSync, writeFileSync, rmSync } from 'node:fs';
24
- import path from 'node:path';
25
- import { fileURLToPath } from 'node:url';
26
-
27
- const ROOT = path.join(path.dirname(fileURLToPath(import.meta.url)), '.build');
28
-
29
- const BROKEN = `export function priority(label) {
30
- const s = String(label).toLowerCase();
31
- if (/high\\s*priority/.test(s)) return 1; // \\s* matches whitespace and nothing else
32
- if (/low\\s*priority/.test(s)) return 3;
33
- return 2;
34
- }
35
- export const SLA_HOURS = { 1: 4, 2: 24, 3: 72 };
36
- `;
37
- const FIXED = BROKEN
38
- .replace('/high\\s*priority/', '/high[-_\\s.]*priority/')
39
- .replace('/low\\s*priority/', '/low[-_\\s.]*priority/');
40
-
41
- const FIXTURES = {
42
- '01-caught-value': {
43
- expect: 'CAUGHT',
44
- why: 'asserts the exact priority for the hyphenated spelling',
45
- test: `import { test } from 'node:test';
46
- import assert from 'node:assert/strict';
47
- import { priority, SLA_HOURS } from '../src/priority.mjs';
48
- test('a hyphenated HIGH-PRIORITY label is priority 1, four-hour SLA', () => {
49
- assert.equal(priority('HIGH-PRIORITY'), 1);
50
- assert.equal(SLA_HOURS[priority('HIGH-PRIORITY')], 4);
51
- });`,
52
- },
53
- '02-blind-direction': {
54
- expect: 'BLIND',
55
- why: 'asserts a DIRECTION (>0) that the wrong answer also satisfies',
56
- test: `import { test } from 'node:test';
57
- import assert from 'node:assert/strict';
58
- import { priority, SLA_HOURS } from '../src/priority.mjs';
59
- test('hyphenated labels are handled', () => {
60
- const p = priority('HIGH-PRIORITY');
61
- assert.ok(p, 'a priority comes back');
62
- assert.ok(SLA_HOURS[p] > 0, 'it has an SLA');
63
- assert.notEqual(p, undefined);
64
- });`,
65
- },
66
- '03-blind-sourcetext': {
67
- expect: 'BLIND',
68
- why: 'greps the source instead of executing it',
69
- test: `import { test } from 'node:test';
70
- import assert from 'node:assert/strict';
71
- import { readFileSync } from 'node:fs';
72
- test('the separator class is tolerant', () => {
73
- const src = readFileSync(new URL('../src/priority.mjs', import.meta.url), 'utf8');
74
- assert.ok(src.includes('priority'), 'the rule mentions priority');
75
- assert.ok(/high/.test(src));
76
- });`,
77
- },
78
- '04-blind-overmock': {
79
- expect: 'BLIND',
80
- why: 'mocks the unit under test, so the real function never runs',
81
- test: `import { test } from 'node:test';
82
- import assert from 'node:assert/strict';
83
- import { SLA_HOURS } from '../src/priority.mjs';
84
- const priority = () => 1; // "stubbed for speed"
85
- test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
86
- assert.equal(priority('HIGH-PRIORITY'), 1);
87
- assert.equal(SLA_HOURS[1], 4);
88
- });`,
89
- },
90
- '05-inconclusive-newexport': {
91
- expect: 'INCONCLUSIVE',
92
- why: 'imports a symbol the fix added — cannot even load at the parent',
93
- fixedExtra: `export const SEPARATORS = /[-_\\s.]*/;\n`,
94
- test: `import { test } from 'node:test';
95
- import assert from 'node:assert/strict';
96
- import { SEPARATORS } from '../src/priority.mjs';
97
- test('the separator class is shared', () => {
98
- assert.equal(SEPARATORS.source, '[-_\\\\s.]*');
99
- });`,
100
- },
101
- '06-inconclusive-armA-red': {
102
- expect: 'INCONCLUSIVE',
103
- why: 'the case is not green on the fix either — the commit does not stand up',
104
- test: `import { test } from 'node:test';
105
- import assert from 'node:assert/strict';
106
- import { priority } from '../src/priority.mjs';
107
- test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
108
- assert.equal(priority('HIGH-PRIORITY'), 99);
109
- });`,
110
- },
111
- '07-caught-mixed': {
112
- expect: 'CAUGHT',
113
- why: 'one discriminating case plus two regression guards green on both arms',
114
- test: `import { test } from 'node:test';
115
- import assert from 'node:assert/strict';
116
- import { priority } from '../src/priority.mjs';
117
- test('DISCRIMINATES: the hyphenated spelling is priority 1', () => {
118
- assert.equal(priority('HIGH-PRIORITY'), 1);
119
- });
120
- test('GUARD: the spaced spelling is still priority 1', () => {
121
- assert.equal(priority('HIGH PRIORITY'), 1);
122
- });
123
- test('GUARD: an unlabelled ticket is still the default priority 2', () => {
124
- assert.equal(priority('needs triage'), 2);
125
- });`,
126
- },
127
- '09-caught-slash-in-name': {
128
- expect: 'CAUGHT',
129
- why: 'the discriminating case has a SLASH in its name — it used to be silently dropped',
130
- test: `import { test } from 'node:test';
131
- import assert from 'node:assert/strict';
132
- import { priority } from '../src/priority.mjs';
133
- test('label parsing / separator handling', () => {
134
- assert.equal(priority('HIGH-PRIORITY'), 1);
135
- });`,
136
- },
137
- '08-skipped-notest': {
138
- expect: 'SKIPPED',
139
- why: 'the fix shipped no test at all',
140
- test: null,
141
- },
142
- };
143
-
144
- function sh(cwd, args) { execFileSync('git', args, { cwd, stdio: 'pipe' }); }
145
-
146
- export function buildFixtures() {
147
- rmSync(ROOT, { recursive: true, force: true });
148
- mkdirSync(ROOT, { recursive: true });
149
- const built = {};
150
-
151
- for (const [name, spec] of Object.entries(FIXTURES)) {
152
- const dir = path.join(ROOT, name);
153
- mkdirSync(path.join(dir, 'src'), { recursive: true });
154
- mkdirSync(path.join(dir, 'tests'), { recursive: true });
155
- sh(dir, ['init', '-q']);
156
- sh(dir, ['config', 'user.email', 'ca@fixture']);
157
- sh(dir, ['config', 'user.name', 'ca fixture']);
158
- writeFileSync(path.join(dir, 'package.json'), JSON.stringify({ name, type: 'module', private: true }, null, 2));
159
-
160
- // --- parent: the bug, and whatever tests existed before (none) ---
161
- writeFileSync(path.join(dir, 'src/priority.mjs'), BROKEN);
162
- sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'feat: priority labels']);
163
-
164
- // --- fix: source repaired, test added ---
165
- writeFileSync(path.join(dir, 'src/priority.mjs'), FIXED + (spec.fixedExtra || ''));
166
- if (spec.test) writeFileSync(path.join(dir, 'tests/priority.test.mjs'), spec.test + '\n');
167
- sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'fix: a hyphen made a HIGH-PRIORITY ticket read as normal']);
168
-
169
- built[name] = { dir, sha: execFileSync('git', ['-C', dir, 'rev-parse', 'HEAD']).toString().trim(), ...spec };
170
- }
171
- return built;
172
- }
173
-
174
- if (import.meta.url === `file://${process.argv[1]}`) {
175
- const b = buildFixtures();
176
- for (const [n, f] of Object.entries(b)) console.log(` ${f.expect.padEnd(13)} ${n.padEnd(26)} ${f.why}`);
177
- console.log(`\n ${Object.keys(b).length} fixtures in ${ROOT}\n`);
178
- }
@@ -1,90 +0,0 @@
1
- #!/usr/bin/env node
2
- /**
3
- * The few GitHub API calls these workflows need, over plain fetch.
4
- *
5
- * WHY NOT THE `gh` CLI. It is not installed on every runner. Measured: a self-hosted
6
- * bare-metal runner failed with `gh: command not found` (exit 127) after every other step
7
- * had passed — the tool ran, reached the right verdict, and then could not say so.
8
- * GitHub-hosted runners ship `gh`; a self-hosted one ships whatever was installed on it,
9
- * and a workflow that assumes otherwise works until it lands on the wrong machine.
10
- *
11
- * Node is already a hard requirement here — the tool is written in it — so this adds no
12
- * dependency at all. Usage:
13
- *
14
- * gh-api.mjs upsert-pr-comment <repo> <pr> <body-file> <marker>
15
- * gh-api.mjs upsert-issue <repo> <title> <body-file> <search> [label]
16
- *
17
- * Reads GITHUB_TOKEN / GH_TOKEN from the environment. Prints what it did, and exits
18
- * non-zero only when the API refuses — never merely because there was nothing to do.
19
- */
20
-
21
- const TOKEN = process.env.GITHUB_TOKEN || process.env.GH_TOKEN;
22
- const API = process.env.GITHUB_API_URL || 'https://api.github.com';
23
-
24
- async function gh(path, init = {}) {
25
- const res = await fetch(`${API}${path}`, {
26
- ...init,
27
- headers: {
28
- authorization: `Bearer ${TOKEN}`,
29
- accept: 'application/vnd.github+json',
30
- 'content-type': 'application/json',
31
- 'x-github-api-version': '2022-11-28',
32
- ...(init.headers || {}),
33
- },
34
- });
35
- if (!res.ok) {
36
- const text = await res.text();
37
- throw new Error(`${init.method || 'GET'} ${path} → ${res.status} ${text.slice(0, 300)}`);
38
- }
39
- return res.status === 204 ? null : res.json();
40
- }
41
-
42
- const [cmd, ...args] = process.argv.slice(2);
43
- const read = async (f) => (await import('node:fs/promises')).readFile(f, 'utf8');
44
-
45
- try {
46
- if (!TOKEN) throw new Error('no GITHUB_TOKEN / GH_TOKEN in the environment');
47
-
48
- if (cmd === 'upsert-pr-comment') {
49
- const [repo, pr, bodyFile, marker] = args;
50
- const body = await read(bodyFile);
51
- if (!body.trim()) { console.log('nothing to post'); process.exit(0); }
52
- // ONE COMMENT PER PR, edited in place. A new comment per push turns a useful signal
53
- // into noise by the third revision, and the marker is how we find ours again.
54
- const comments = await gh(`/repos/${repo}/issues/${pr}/comments?per_page=100`);
55
- const mine = comments.find(c => (c.body || '').startsWith(marker));
56
- if (mine) {
57
- await gh(`/repos/${repo}/issues/comments/${mine.id}`, { method: 'PATCH', body: JSON.stringify({ body }) });
58
- console.log(`updated comment ${mine.id}`);
59
- } else {
60
- const made = await gh(`/repos/${repo}/issues/${pr}/comments`, { method: 'POST', body: JSON.stringify({ body }) });
61
- console.log(`posted comment ${made.id}`);
62
- }
63
- } else if (cmd === 'upsert-issue') {
64
- const [repo, title, bodyFile, search, label] = args;
65
- const body = await read(bodyFile);
66
- if (label) {
67
- // Create the label if absent. `labels` on an issue with an unknown label is
68
- // rejected outright, so this cannot be left to chance — but a failure here is
69
- // a warning, not a reason to drop the finding on the floor.
70
- try {
71
- await gh(`/repos/${repo}/labels`, { method: 'POST', body: JSON.stringify({ name: label, color: '0E8A16', description: 'Test-gap findings from control-arm' }) });
72
- } catch (e) { if (!/already_exists|422/.test(e.message)) console.log(`::warning::could not create label ${label}: ${e.message}`); }
73
- }
74
- const found = await gh(`/search/issues?q=${encodeURIComponent(`repo:${repo} is:issue is:open ${search}`)}`);
75
- const hit = found.items?.[0];
76
- if (hit) {
77
- await gh(`/repos/${repo}/issues/${hit.number}`, { method: 'PATCH', body: JSON.stringify({ title, body }) });
78
- console.log(`refreshed issue #${hit.number}`);
79
- } else {
80
- const made = await gh(`/repos/${repo}/issues`, { method: 'POST', body: JSON.stringify({ title, body, ...(label ? { labels: [label] } : {}) }) });
81
- console.log(`filed issue #${made.number}`);
82
- }
83
- } else {
84
- console.error('usage: gh-api.mjs upsert-pr-comment|upsert-issue ...');
85
- process.exit(2);
86
- }
87
- } catch (e) {
88
- console.error(`::error::${e.message}`);
89
- process.exit(1);
90
- }
@@ -1,86 +0,0 @@
1
- /**
2
- * Segment the audit CSV by which runner the test file belongs to.
3
- *
4
- * `ca` ships one runner (node:test). A monorepo's fix commits touch vitest and jest test
5
- * files too, and those come back INCONCLUSIVE for a reason that says nothing about the
6
- * test — the tool simply cannot execute it. Reporting one blended ratio over both would
7
- * be the same defect the tool exists to catch: a number fitted to a population it does
8
- * not describe.
9
- */
10
- import { readFileSync } from 'node:fs';
11
-
12
- const rows = [];
13
- const raw = readFileSync(process.argv[2], 'utf8').split('\n').filter(Boolean);
14
- const hdr = raw.shift().split(',');
15
- for (const line of raw) {
16
- // naive CSV with quoted fields
17
- const f = []; let cur = '', q = false;
18
- for (let i = 0; i < line.length; i++) {
19
- const c = line[i];
20
- if (q) { if (c === '"' && line[i + 1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
21
- else if (c === '"') q = true;
22
- else if (c === ',') { f.push(cur); cur = ''; }
23
- else cur += c;
24
- }
25
- f.push(cur);
26
- rows.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
27
- }
28
-
29
- const runnerOf = (file) => {
30
- if (!file) return 'none';
31
- if (file.startsWith('tests/')) return 'node:test (supported)';
32
- if (file.startsWith('apps/web/')) return 'vitest (unsupported)';
33
- if (file.startsWith('apps/mobile/')) return 'jest (unsupported)';
34
- if (file.startsWith('apps/e2e/')) return 'playwright (unsupported)';
35
- if (file.startsWith('packages/')) return 'node:test (supported)';
36
- return 'other';
37
- };
38
-
39
- const commits = new Map();
40
- for (const r of rows) {
41
- if (!commits.has(r.sha)) commits.set(r.sha, { sha: r.sha, date: r.date, subject: r.subject, verdict: r.commit_verdict, cases: [] });
42
- commits.get(r.sha).cases.push(r);
43
- }
44
-
45
- // A commit belongs to the runner of its test files; mixed commits are called out.
46
- const seg = new Map();
47
- for (const c of commits.values()) {
48
- const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
49
- const key = rs.length === 0 ? 'no test file' : rs.length === 1 ? rs[0] : 'mixed';
50
- if (!seg.has(key)) seg.set(key, []);
51
- seg.get(key).push(c);
52
- }
53
-
54
- const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
55
- console.log('\n COMMITS BY RUNNER SEGMENT\n');
56
- console.log(' ' + 'segment'.padEnd(26) + V.map(v => v.slice(0, 6).padStart(7)).join('') + ' n');
57
- for (const [k, list] of [...seg].sort((a, b) => b[1].length - a[1].length)) {
58
- const t = V.map(v => String(list.filter(c => c.verdict === v).length).padStart(7)).join('');
59
- console.log(' ' + k.padEnd(26) + t + String(list.length).padStart(5));
60
- }
61
-
62
- const supported = seg.get('node:test (supported)') || [];
63
- const dec = supported.filter(c => c.verdict === 'CAUGHT' || c.verdict === 'BLIND');
64
- console.log(`\n SUPPORTED SEGMENT ONLY (node:test)\n`);
65
- console.log(` ${supported.length} commits · ${dec.length} the instrument could answer`);
66
- if (dec.length) {
67
- const caught = dec.filter(c => c.verdict === 'CAUGHT').length;
68
- console.log(` CAUGHT ${caught}/${dec.length} = ${(caught / dec.length * 100).toFixed(1)}% BLIND ${dec.length - caught}/${dec.length} = ${((dec.length - caught) / dec.length * 100).toFixed(1)}%`);
69
- }
70
- const blind = supported.filter(c => c.verdict === 'BLIND');
71
- if (blind.length) {
72
- console.log(`\n BLIND COMMITS IN THE SUPPORTED SEGMENT (hand-audit these)\n`);
73
- for (const c of blind) console.log(` ${c.sha.slice(0, 8)} ${c.date} ${c.subject.slice(0, 90)}`);
74
- }
75
- const why = new Map();
76
- for (const c of supported.filter(c => c.verdict === 'INCONCLUSIVE')) {
77
- for (const cs of c.cases.filter(x => x.case_verdict === 'INCONCLUSIVE')) {
78
- const k = cs.reason.replace(/'[^']*'/g, "'…'").replace(/\d+/g, 'N').slice(0, 88);
79
- why.set(k, (why.get(k) || 0) + 1);
80
- }
81
- }
82
- if (why.size) {
83
- console.log(`\n WHY INCONCLUSIVE, supported segment only\n`);
84
- for (const [k, v] of [...why].sort((a, b) => b[1] - a[1]).slice(0, 14)) console.log(` ${String(v).padStart(4)} ${k}`);
85
- }
86
- console.log('');
@@ -1,53 +0,0 @@
1
- /**
2
- * Re-roll commit verdicts from a saved audit CSV using the CURRENT rollUp.
3
- *
4
- * Per-case verdicts are what the two arms measured; the commit verdict is a pure function
5
- * of them. So a change to the roll-up rule does not need the 7-minute run again — and
6
- * re-running would also change the sample, which is the wrong thing to do when comparing
7
- * a rule change.
8
- */
9
- import { readFileSync } from 'node:fs';
10
- import { rollUp } from './src/verdict.mjs';
11
-
12
- function parseCsv(text) {
13
- const out = []; const lines = text.split('\n').filter(Boolean); const hdr = lines.shift().split(',');
14
- for (const line of lines) {
15
- const f = []; let cur = '', q = false;
16
- for (let i = 0; i < line.length; i++) { const c = line[i];
17
- if (q) { if (c === '"' && line[i+1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
18
- else if (c === '"') q = true; else if (c === ',') { f.push(cur); cur = ''; } else cur += c; }
19
- f.push(cur); out.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
20
- }
21
- return out;
22
- }
23
- const runnerOf = f => !f ? 'none'
24
- : f.startsWith('tests/') || f.startsWith('packages/') ? 'node:test'
25
- : f.startsWith('apps/web/') ? 'vitest' : f.startsWith('apps/mobile/') ? 'jest'
26
- : f.startsWith('apps/e2e/') ? 'playwright' : 'other';
27
-
28
- const rows = parseCsv(readFileSync(process.argv[2], 'utf8'));
29
- const commits = new Map();
30
- for (const r of rows) {
31
- if (!commits.has(r.sha)) commits.set(r.sha, { ...r, cases: [] });
32
- if (r.case) commits.get(r.sha).cases.push(r);
33
- }
34
- const list = [...commits.values()].map(c => {
35
- const nv = c.cases.length ? rollUp(c.cases.map(x => ({ verdict: x.case_verdict }))) : c.commit_verdict;
36
- const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
37
- return { ...c, was: c.commit_verdict, now: nv, seg: rs.length === 1 ? rs[0] : rs.length ? 'mixed' : 'none' };
38
- });
39
-
40
- const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
41
- const changed = list.filter(c => c.was !== c.now);
42
- console.log(`\n ${list.length} commits · ${changed.length} changed verdict under the corrected roll-up\n`);
43
- for (const c of changed) console.log(` ${c.was.padEnd(13)} -> ${c.now.padEnd(13)} ${c.sha.slice(0,8)} ${c.subject.slice(0,70)}`);
44
-
45
- const node = list.filter(c => c.seg === 'node:test' || c.seg === 'mixed');
46
- const dec = node.filter(c => c.now === 'CAUGHT' || c.now === 'BLIND');
47
- const caught = dec.filter(c => c.now === 'CAUGHT').length;
48
- console.log(`\n SUPPORTED SEGMENT (node:test, incl. mixed): ${node.length} commits`);
49
- for (const v of V) { const n = node.filter(c => c.now === v).length; if (n) console.log(` ${v.padEnd(14)} ${String(n).padStart(4)} ${(n/node.length*100).toFixed(1)}%`); }
50
- console.log(`\n ANSWERABLE: ${dec.length} CAUGHT ${caught} (${(caught/dec.length*100).toFixed(1)}%) BLIND ${dec.length-caught} (${((dec.length-caught)/dec.length*100).toFixed(1)}%)\n`);
51
- console.log(' BLIND after correction:\n');
52
- for (const c of list.filter(c => c.now === 'BLIND')) console.log(` ${c.sha.slice(0,8)} ${c.date} ${c.subject.slice(0,84)}`);
53
- console.log('');
@@ -1,130 +0,0 @@
1
- /**
2
- * Case extraction, pinned against the four shapes that actually broke it.
3
- *
4
- * Each of these was measured on a real corpus of 400 test cases, and each one alone
5
- * accounted for a double-digit share of "could not analyse". Together they took the
6
- * unresolvable pile from 64% to 1.3%. None would be caught by a suite that only fed the
7
- * analyser well-formed input, which is why they are here with the messy input attached.
8
- */
9
- import { test } from 'node:test';
10
- import assert from 'node:assert/strict';
11
- import { extractCase, analyseCase } from '../src/assertions.mjs';
12
-
13
- test('nested suites: the runner reports the CONCATENATED name, the source holds the inner one', () => {
14
- const src = `
15
- describe('auction client', () => {
16
- it('AUCWEB-001 calls GET /auction/:vin', () => { assert.equal(a, 1); });
17
- });`;
18
- // Reported by the runner as describe-title + space + it-title.
19
- const body = extractCase(src, 'auction client AUCWEB-001 calls GET /auction/:vin');
20
- assert.ok(body, 'progressive reduction must find the inner title');
21
- assert.match(body, /assert\.equal\(a, 1\)/);
22
- });
23
-
24
- test('an APOSTROPHE in the title: source holds a backslash the reported name does not', () => {
25
- const src = `it('METER-038: usage on the user\\'s LOCAL today is counted', () => { assert.equal(n, 3); });`;
26
- const body = extractCase(src, "METER-038: usage on the user's LOCAL today is counted");
27
- assert.ok(body, "an escaped quote inside the literal must still match the unescaped reported name");
28
- assert.match(body, /assert\.equal\(n, 3\)/);
29
- });
30
-
31
- test('JSX: a closing tag is NOT a regex literal', () => {
32
- // `</div>` puts a slash after `<`. The usual "slash after a non-expression starts a
33
- // regex" heuristic swallows the rest of the component, and the body is never found.
34
- const src = `
35
- it('hero renders', () => {
36
- render(<div className="x">{name}</div>);
37
- expect(screen.getByText('hi')).toBeTruthy();
38
- });`;
39
- const body = extractCase(src, 'hero renders');
40
- assert.ok(body, 'a JSX body must brace-match cleanly');
41
- assert.match(body, /getByText/);
42
- });
43
-
44
- test('braces inside strings and regexes do not unbalance the scan', () => {
45
- const src = `
46
- it('handles braces', () => {
47
- const s = '{';
48
- const re = /[{}]/;
49
- assert.equal(f(s, re), '}');
50
- });`;
51
- const body = extractCase(src, 'handles braces');
52
- assert.ok(body, 'a brace inside a string or a character class is not structure');
53
- assert.match(body, /assert\.equal\(f\(s, re\)/);
54
- });
55
-
56
- test('template-literal titles: every generated case resolves to the shared body', () => {
57
- const src = `
58
- for (const rel of PAGES) {
59
- it(\`\${rel}: no internal repo paths\`, () => { expect(md).not.toMatch(BAD); });
60
- }`;
61
- const a = extractCase(src, 'app/page.tsx: no internal repo paths');
62
- const b = extractCase(src, 'app/pricing/page.tsx: no internal repo paths');
63
- assert.ok(a && b, 'a parameterised title must match by pattern, not by equality');
64
- assert.equal(a, b, 'all generated cases share one body — that is correct, not a bug');
65
- });
66
-
67
- test('a wrong body is worse than none: a short remainder is refused', () => {
68
- const src = `
69
- it('alpha returns null', () => { assert.equal(x, null); });
70
- it('beta returns null', () => { assert.equal(y, 0); });`;
71
- // "returns null" alone is ambiguous between the two — reduction must not take it.
72
- const body = extractCase(src, 'some suite that does not exist returns null');
73
- assert.equal(body, null, 'an ambiguous short suffix must not resolve to an arbitrary case');
74
- });
75
-
76
- test('CONTROL ARM: naive equality-only extraction fails every one of these', () => {
77
- // The implementation this replaced. Asserted to still get them wrong.
78
- const naive = (src, name) => {
79
- const esc = name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
80
- return new RegExp(`\\b(?:test|it)\\s*\\(\\s*(['"\`])${esc}\\1`).test(src);
81
- };
82
- assert.equal(naive(`it('a b c', () => {});`, 'suite a b c'), false); // nested
83
- assert.equal(naive(`it('u\\'s v', () => {});`, "u's v"), false); // apostrophe
84
- assert.equal(naive('it(`${r}: x`, () => {});', 'p.tsx: x'), false); // template
85
- });
86
-
87
- test('the analyser reports UNKNOWN rather than a clean bill when it cannot tell', () => {
88
- const r = analyseCase(`it('odd', () => { somethingEntirelyUnrecognised(); });`, 'odd');
89
- assert.notEqual(r.verdict, 'strong', 'an unrecognised body must never read as strong');
90
- });
91
-
92
- test('a short helper name must not be accused of stubbing the unit under test', () => {
93
- // FOUND ON A REAL RUN. The stub check substring-matched the name against every import
94
- // path, so a two-letter helper `at` matched '@acme/core' — "acme" contains "at" — and
95
- // every case in that file was reported as testing a stub. A wrong explanation is worse
96
- // than none: it sends the reader to look at code that is fine.
97
- const src = `
98
- import { computeTier } from '@acme/core/tiers.js';
99
- const at = (arr, i) => arr[i];
100
- it('TIER-001: the tier changes the money', () => {
101
- assert.equal(at(computeTier(x), 0), 1200);
102
- });`;
103
- const r = analyseCase(src, 'TIER-001: the tier changes the money');
104
- assert.ok(!r.findings.some(f => f.id === 'stubbed-subject'),
105
- 'a local helper whose name merely appears inside an import path is not a stub');
106
- });
107
-
108
- test('a REAL stub is still caught — the tightening must not blind the check', () => {
109
- const src = `
110
- import { SLA_HOURS } from '../src/priority.mjs';
111
- const priority = () => 1;
112
- it('p is 1', () => { assert.equal(priority('HIGH-PRIORITY'), 1); });`;
113
- const r = analyseCase(src, 'p is 1');
114
- assert.ok(r.findings.some(f => f.id === 'stubbed-subject'),
115
- 'a stub matching the imported module basename must still be reported');
116
- assert.equal(r.verdict, 'weak');
117
- });
118
-
119
- test('a declined commit produces NO pr comment — silence, not an accusation', async () => {
120
- // It briefly posted "SKIPPED · No case in this branch fails without the change" onto a
121
- // workflow-only PR. Technically true and completely misleading: that PR has no tests,
122
- // so of course none discriminate. A comment reading as an accusation on a PR doing
123
- // nothing wrong is worse than no comment.
124
- const { prComment } = await import('../src/markdown-report.mjs');
125
- const declined = { short: 'abc12345', verdict: 'SKIPPED', cases: [], note: 'no test file in the commit' };
126
- assert.equal(prComment(declined), '', 'a noted (declined) result must render as empty');
127
-
128
- const real = { short: 'abc12345', verdict: 'CAUGHT', cases: [{ verdict: 'CAUGHT', name: 'x', reason: 'y' }] };
129
- assert.notEqual(prComment(real), '', 'a real verdict must still render');
130
- });
@@ -1,41 +0,0 @@
1
- /**
2
- * The tool's control arm. Eight repos where the right answer is known by construction.
3
- *
4
- * A green run here is the only reason to believe a verdict this tool prints about
5
- * anybody else's code. It is also the thing that would catch the failure mode that
6
- * matters most — a FALSE BLIND — because 01 and 07 must never come back BLIND.
7
- */
8
- import { test } from 'node:test';
9
- import assert from 'node:assert/strict';
10
- import path from 'node:path';
11
- import { buildFixtures } from '../fixtures/build.mjs';
12
- import { verifyCommit } from '../src/verify.mjs';
13
- import { removeWorktrees } from '../src/worktree.mjs';
14
-
15
- const fixtures = buildFixtures();
16
-
17
- for (const [name, f] of Object.entries(fixtures)) {
18
- test(`${name} -> ${f.expect} (${f.why})`, async () => {
19
- const workDir = path.join(f.dir, '.ca-work');
20
- try {
21
- const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
22
- assert.equal(r.verdict, f.expect,
23
- `expected ${f.expect}, got ${r.verdict}\n` +
24
- r.cases.map(c => ` ${c.verdict} ${c.name} — ${c.reason}`).join('\n') +
25
- (r.note ? `\n note: ${r.note}` : ''));
26
- } finally {
27
- await removeWorktrees(f.dir, workDir);
28
- }
29
- });
30
- }
31
-
32
- test('a FALSE BLIND is the failure that matters: no discriminating fixture may read BLIND', async () => {
33
- for (const name of ['01-caught-value', '07-caught-mixed']) {
34
- const f = fixtures[name];
35
- const workDir = path.join(f.dir, '.ca-work-fb');
36
- try {
37
- const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
38
- assert.notEqual(r.verdict, 'BLIND', `${name} was accused of being blind — this is the output that destroys trust in the tool`);
39
- } finally { await removeWorktrees(f.dir, workDir); }
40
- }
41
- });
@@ -1,69 +0,0 @@
1
- /**
2
- * The vitest/jest adapter, at the level where it can actually be wrong.
3
- *
4
- * WHAT THIS DOES AND DOES NOT COVER — stated because the gap matters.
5
- *
6
- * The node:test path has 9 end-to-end fixtures: real git repos, both arms, known verdict.
7
- * These two runners do NOT, because a fixture would have to `npm install` vitest or jest
8
- * per repo — slow, network-dependent, and version-drifting. The full two-arm path for
9
- * each is verified on ONE real commit apiece — a vitest component commit and a jest
10
- * React Native commit — which is a smoke test, not a control arm.
11
- *
12
- * So this file covers the part that is BOTH untested end-to-end AND most likely to be
13
- * wrong: the text-matching in `classifyMessages`. node:test hands over `ERR_ASSERTION`;
14
- * these two give formatted strings, so the adapter has to read prose. Every string below
15
- * is a real shape emitted by vitest or jest, not an invented one.
16
- */
17
- import { test } from 'node:test';
18
- import assert from 'node:assert/strict';
19
- import { classifyMessages } from '../src/runner-json.mjs';
20
-
21
- const isDisagreement = m => classifyMessages(m).code === 'ERR_ASSERTION';
22
- const isCannotRun = m => classifyMessages(m).code === 'ERR_TEST_FAILURE';
23
- const isUnknown = m => classifyMessages(m).code === null;
24
-
25
- test('vitest/chai disagreement is read as an assertion', () => {
26
- assert.ok(isDisagreement(["AssertionError: expected 'Urgent' to deeply equal 'Normal'"]));
27
- assert.ok(isDisagreement(['AssertionError: expected 40 to be 65 // Object.is equality']));
28
- });
29
-
30
- test('jest/expect disagreement is read as an assertion', () => {
31
- assert.ok(isDisagreement(['expect(received).toBe(expected) // Object.is equality\n\nExpected: "Normal"\nReceived: "Urgent"']));
32
- assert.ok(isDisagreement(['Error: expect(received).toEqual(expected)\n\n- Expected\n+ Received']));
33
- });
34
-
35
- test('a module that could not load is NOT an assertion', () => {
36
- assert.ok(isCannotRun(["Error: Cannot find module '../src/newHelper' from 'lib/__tests__/x.test.tsx'"]));
37
- assert.ok(isCannotRun(["SyntaxError: The requested module './bucket.js' does not provide an export named 'SEP'"]));
38
- assert.ok(isCannotRun(['Error: Failed to resolve import "./notYet" from "lib/x.test.tsx". Does the file exist?']));
39
- assert.ok(isCannotRun(['TypeError [ERR_UNKNOWN_FILE_EXTENSION]: Unknown file extension ".tsx"']));
40
- });
41
-
42
- test('CANNOT-RUN WINS when both shapes appear — the ordering that matters', () => {
43
- // A failed module load frequently ALSO prints an expect() frame from the stack. If
44
- // the disagreement patterns were checked first, this would read CAUGHT, and a commit
45
- // whose test never ran would be credited with catching its bug.
46
- const both = ['SyntaxError: does not provide an export named \'SEP\'\n at expect(received).toBe(expected)\n expected \'a\' to be \'b\''];
47
- assert.ok(isCannotRun(both), 'a load failure that also prints an expect() frame must not read as a disagreement');
48
- assert.notEqual(classifyMessages(both).code, 'ERR_ASSERTION');
49
- });
50
-
51
- test('an UNRECOGNISED message is unknown — never guessed into an assertion', () => {
52
- // This is the design: an unrecognised shape costs COVERAGE (it reads INCONCLUSIVE),
53
- // it cannot manufacture a finding. A drifting regex must fail safe, not fail loud.
54
- assert.ok(isUnknown(['Something entirely unexpected happened in a custom matcher']));
55
- assert.ok(isUnknown([]));
56
- assert.ok(isUnknown(['']));
57
- assert.notEqual(classifyMessages(['whatever']).code, 'ERR_ASSERTION');
58
- });
59
-
60
- test('CONTROL ARM: a naive "any failure is a disagreement" classifier gets these wrong', () => {
61
- // TEST-001 — run the broken implementation and assert it still reproduces the bug.
62
- const naive = msgs => (msgs && msgs.length ? 'ERR_ASSERTION' : null);
63
- const loadFail = ["Error: Cannot find module '../src/newHelper'"];
64
- assert.equal(naive(loadFail), 'ERR_ASSERTION'); // the naive one says CAUGHT
65
- assert.equal(classifyMessages(loadFail).code, 'ERR_TEST_FAILURE'); // ours says it never ran
66
- const weird = ['custom matcher blew up'];
67
- assert.equal(naive(weird), 'ERR_ASSERTION');
68
- assert.equal(classifyMessages(weird).code, null);
69
- });
@@ -1,78 +0,0 @@
1
- /**
2
- * Runner selection, over synthetic trees.
3
- *
4
- * Picking the WRONG runner is a silent failure: vitest run from the repo root finds no
5
- * config and reports "No test files found", which reads INCONCLUSIVE — a whole surface
6
- * disappears without ever saying why. So the walk-up has to be pinned.
7
- */
8
- import { test } from 'node:test';
9
- import assert from 'node:assert/strict';
10
- import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
11
- import { tmpdir } from 'node:os';
12
- import path from 'node:path';
13
- import { selectRunner } from '../src/select-runner.mjs';
14
-
15
- function tree(spec) {
16
- const root = mkdtempSync(path.join(tmpdir(), 'ca-sel-'));
17
- for (const [rel, content] of Object.entries(spec)) {
18
- const p = path.join(root, rel);
19
- mkdirSync(path.dirname(p), { recursive: true });
20
- writeFileSync(p, content);
21
- }
22
- return root;
23
- }
24
-
25
- test('a vitest config in the workspace wins, and names that workspace as cwd', async () => {
26
- const root = tree({
27
- 'package.json': '{"workspaces":["apps/*"]}',
28
- 'apps/web/vitest.config.ts': 'export default {}',
29
- 'apps/web/lib/__tests__/x.test.tsx': '',
30
- });
31
- try {
32
- const r = await selectRunner(root, 'apps/web/lib/__tests__/x.test.tsx');
33
- assert.equal(r.flavour, 'vitest');
34
- assert.equal(r.pkgDir, 'apps/web', 'must run FROM apps/web — vitest resolves config relative to cwd');
35
- } finally { rmSync(root, { recursive: true, force: true }); }
36
- });
37
-
38
- test('a jest config in the workspace wins', async () => {
39
- const root = tree({ 'package.json': '{}', 'apps/mobile/jest.config.js': '', 'apps/mobile/__tests__/y.test.ts': '' });
40
- try {
41
- const r = await selectRunner(root, 'apps/mobile/__tests__/y.test.ts');
42
- assert.equal(r.flavour, 'jest');
43
- assert.equal(r.pkgDir, 'apps/mobile');
44
- } finally { rmSync(root, { recursive: true, force: true }); }
45
- });
46
-
47
- test('root tests fall to node:test when nothing else claims them', async () => {
48
- const root = tree({ 'package.json': '{"scripts":{"test":"node --test tests/*.test.js"}}', 'tests/a.test.js': '' });
49
- try {
50
- const r = await selectRunner(root, 'tests/a.test.js');
51
- assert.equal(r.flavour, 'node');
52
- assert.equal(r.pkgDir, '');
53
- } finally { rmSync(root, { recursive: true, force: true }); }
54
- });
55
-
56
- test('a repo that declares NOTHING falls through to node:test, not to a guess', async () => {
57
- // The safe default: node:test either works, or reports a load failure, which reads
58
- // INCONCLUSIVE. Guessing vitest here would produce "No test files found" — the same
59
- // silent disappearance this test exists to prevent.
60
- const root = tree({ 'tests/a.test.js': '' });
61
- try {
62
- assert.equal((await selectRunner(root, 'tests/a.test.js')).flavour, 'node');
63
- } finally { rmSync(root, { recursive: true, force: true }); }
64
- });
65
-
66
- test('a workspace config beats a root test script — the nearest owner wins', async () => {
67
- // The root script often delegates (`npm run test --workspaces`), while the config
68
- // that actually governs the file sits in the package. Nearest wins, walking up.
69
- const root = tree({
70
- 'package.json': '{"scripts":{"test":"node --test"}}',
71
- 'apps/web/vitest.config.ts': '',
72
- 'apps/web/x.test.tsx': '',
73
- });
74
- try {
75
- const r = await selectRunner(root, 'apps/web/x.test.tsx');
76
- assert.equal(r.flavour, 'vitest');
77
- } finally { rmSync(root, { recursive: true, force: true }); }
78
- });
@@ -1,113 +0,0 @@
1
- /**
2
- * The decision logic, exhaustively, with no git and no subprocesses.
3
- *
4
- * Every case here is a shape observed in a real run, not an invented one.
5
- */
6
- import { test } from 'node:test';
7
- import assert from 'node:assert/strict';
8
- import { classify, reduceRuns, rollUp, isDisagreement, CAUGHT, BLIND, NON_DISCRIMINATING, INCONCLUSIVE, FLAKY, SKIPPED } from '../src/verdict.mjs';
9
-
10
- const pass = { status: 'pass' };
11
- const assertFail = { status: 'fail', code: 'ERR_ASSERTION', errorName: 'AssertionError', message: "expected 1, got 2" };
12
- const errFail = { status: 'fail', code: 'ERR_TEST_FAILURE', errorName: 'SyntaxError', message: "does not provide an export named 'TITLE_SEPARATOR'" };
13
-
14
- test('CAUGHT: green on the fix, assertion-red on the parent', () => {
15
- assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
16
- });
17
-
18
- test('NON-DISCRIMINATING: green on the fix AND green on the parent', () => {
19
- assert.equal(classify({ armA: pass, armB: pass }).verdict, NON_DISCRIMINATING);
20
- });
21
-
22
- test('a case green on both arms is never called BLIND — a regression guard is indistinguishable', () => {
23
- // Observed in practice: 3 of one commit's 4 cases were deliberate regression guards.
24
- assert.notEqual(classify({ armA: pass, armB: pass }).verdict, BLIND);
25
- });
26
-
27
- test('a non-assertion failure on the parent is INCONCLUSIVE, never CAUGHT — the test never ran', () => {
28
- const r = classify({ armA: pass, armB: errFail });
29
- assert.equal(r.verdict, INCONCLUSIVE);
30
- assert.match(r.reason, /never ran/);
31
- });
32
-
33
- test('exit code alone would have gotten that wrong', () => {
34
- // Both of these are "the process exited non-zero". Only one is a verdict.
35
- assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
36
- assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE);
37
- });
38
-
39
- test('unproven module identity withholds the verdict even when arm B passed', () => {
40
- // THE false-BLIND guard. Arm B green + identity unproven is precisely the symlink
41
- // trap that made this tool report a false BLIND before it had an identity gate.
42
- const r = classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'resolved OUTSIDE the worktree' } });
43
- assert.equal(r.verdict, INCONCLUSIVE);
44
- assert.notEqual(r.verdict, BLIND);
45
- });
46
-
47
- test('a case that is red on the fix is INCONCLUSIVE — the commit does not stand up', () => {
48
- assert.equal(classify({ armA: assertFail, armB: assertFail }).verdict, INCONCLUSIVE);
49
- });
50
-
51
- test('a case missing from the parent run is INCONCLUSIVE, not BLIND', () => {
52
- assert.equal(classify({ armA: pass, armB: null }).verdict, INCONCLUSIVE);
53
- });
54
-
55
- test('isDisagreement is narrow: an unknown error name is not guessed into CAUGHT', () => {
56
- assert.equal(isDisagreement({ status: 'fail', errorName: 'WeirdCustomError' }), false);
57
- assert.equal(isDisagreement({ status: 'fail', errorName: 'JestAssertionError' }), true);
58
- });
59
-
60
- test('FLAKY: runs that disagree between CAUGHT and BLIND', () => {
61
- const r = reduceRuns([{ verdict: CAUGHT }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]);
62
- assert.equal(r.verdict, FLAKY);
63
- });
64
-
65
- test('an INCONCLUSIVE run does not erase a decided one, but is disclosed', () => {
66
- const r = reduceRuns([{ verdict: CAUGHT, reason: 'assertion' }, { verdict: INCONCLUSIVE, reason: 'timeout' }, { verdict: CAUGHT, reason: 'assertion' }]);
67
- assert.equal(r.verdict, CAUGHT);
68
- assert.match(r.reason, /1 of 3 runs inconclusive/);
69
- });
70
-
71
- test('rollUp: one discriminating case carries the commit, guards and all', () => {
72
- assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]), CAUGHT);
73
- });
74
-
75
- test('rollUp: a SKIPPED case blocks BLIND — it might have been the discriminating one', () => {
76
- // Observed: all four tests written FOR a bug were { skip: SKIP } on a host with no
77
- // TEST_DATABASE_URL, while older cases in the same file ran and did not discriminate.
78
- // Calling that blind judges the commit on the tests NOT written for it.
79
- assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
80
- assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
81
- });
82
-
83
- test('rollUp: BLIND only when every case ran and NOT ONE discriminated', () => {
84
- assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }]), BLIND);
85
- // one case could not be judged -> the commit cannot be called blind
86
- assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: INCONCLUSIVE }]), INCONCLUSIVE);
87
- });
88
-
89
- test('CONTROL ARM: a verdict engine that only read exit codes fails these', () => {
90
- // TEST-001 — the old, naive implementation, executed here, asserted to still be wrong.
91
- const naive = ({ armB }) => (armB && armB.status === 'fail' ? CAUGHT : BLIND);
92
- assert.equal(naive({ armB: errFail }), CAUGHT); // it says CAUGHT...
93
- assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE); // ...we say no
94
- assert.equal(naive({ armB: pass }), BLIND); // and calls every regression guard blind // and it cannot see
95
- assert.equal(classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'x' } }).verdict, INCONCLUSIVE);
96
- });
97
-
98
- test('a CAUGHT on new code is still CAUGHT — the verdict does not lie, the CLAIM narrows', async () => {
99
- // Issue #3. On a feature the base lacks the code entirely, so essentially any test
100
- // touching it fails there — `expected 0 to be greater than 0` is a real AssertionError
101
- // that says nothing about whether the test is well aimed. Reclassifying it would be
102
- // dishonest (it IS a disagreement); counting it as evidence would be worse.
103
- const { commitKind } = await import('../src/verify.mjs');
104
-
105
- assert.equal(commitKind('feat(web): add seoTitle', false).newCode, true,
106
- 'a feature is new code whatever its diff looks like');
107
- assert.equal(commitKind('fix(core): a hyphen decided a title', false).newCode, false,
108
- 'a fix that changes lines is a repair, and its CAUGHT is strong evidence');
109
- assert.equal(commitKind('fix(api): add a missing export', true).newCode, true,
110
- 'a fix whose source diff only ADDS may still be reading something that was absent');
111
- assert.equal(commitKind('no conventional prefix at all', false).kind, 'unknown',
112
- 'an unrecognised subject must not be guessed into fix or feature');
113
- });