control-arm 0.1.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +7 -1
- package/package.json +24 -1
- package/.github/workflows/ci.yml +0 -82
- package/action.yml +0 -95
- package/fixtures/build.mjs +0 -178
- package/scripts/gh-api.mjs +0 -90
- package/scripts-analyze.mjs +0 -86
- package/scripts-recompute.mjs +0 -53
- package/test/assertions.test.mjs +0 -130
- package/test/fixtures.test.mjs +0 -41
- package/test/runner-json.test.mjs +0 -69
- package/test/select-runner.test.mjs +0 -78
- package/test/verdict.test.mjs +0 -113
package/README.md
CHANGED
|
@@ -46,7 +46,7 @@ Node 18+. No other dependencies.
|
|
|
46
46
|
|
|
47
47
|
```bash
|
|
48
48
|
# Can this repo be measured at all?
|
|
49
|
-
ca
|
|
49
|
+
ca doctor
|
|
50
50
|
|
|
51
51
|
# Judge one commit — a fix, with the test that shipped alongside it
|
|
52
52
|
ca verify <sha>
|
|
@@ -62,6 +62,12 @@ ca verify HEAD --against origin/main
|
|
|
62
62
|
name: control-arm
|
|
63
63
|
on: pull_request
|
|
64
64
|
|
|
65
|
+
# `comment: true` posts with the default GITHUB_TOKEN, which is read-only in most
|
|
66
|
+
# repos. Without this block the run succeeds and the comment silently never appears.
|
|
67
|
+
permissions:
|
|
68
|
+
contents: read
|
|
69
|
+
pull-requests: write
|
|
70
|
+
|
|
65
71
|
jobs:
|
|
66
72
|
verify:
|
|
67
73
|
runs-on: ubuntu-latest
|
package/package.json
CHANGED
|
@@ -1,9 +1,32 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "control-arm",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "1.0.0",
|
|
4
4
|
"description": "Does a test actually fail on the code it was written to catch?",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": { "ca": "./bin/ca.mjs" },
|
|
7
|
+
"files": [
|
|
8
|
+
"bin",
|
|
9
|
+
"src",
|
|
10
|
+
"README.md",
|
|
11
|
+
"DESIGN.md",
|
|
12
|
+
"LICENSE"
|
|
13
|
+
],
|
|
14
|
+
"engines": { "node": ">=18" },
|
|
15
|
+
"keywords": [
|
|
16
|
+
"testing",
|
|
17
|
+
"test-quality",
|
|
18
|
+
"mutation-testing",
|
|
19
|
+
"regression-testing",
|
|
20
|
+
"continuous-integration",
|
|
21
|
+
"github-actions",
|
|
22
|
+
"code-quality"
|
|
23
|
+
],
|
|
24
|
+
"repository": {
|
|
25
|
+
"type": "git",
|
|
26
|
+
"url": "git+https://github.com/mekanhan/control-arm.git"
|
|
27
|
+
},
|
|
28
|
+
"homepage": "https://github.com/mekanhan/control-arm#readme",
|
|
29
|
+
"bugs": { "url": "https://github.com/mekanhan/control-arm/issues" },
|
|
7
30
|
"scripts": {
|
|
8
31
|
"test": "node --test test/*.test.mjs",
|
|
9
32
|
"fixtures": "node fixtures/build.mjs"
|
package/.github/workflows/ci.yml
DELETED
|
@@ -1,82 +0,0 @@
|
|
|
1
|
-
# The tool that asks whether your tests can fail, running its own.
|
|
2
|
-
#
|
|
3
|
-
# It shipped for a day without this. That is the defect it exists to find, one level up:
|
|
4
|
-
# a suite that is never executed by anything but its author's terminal is not a gate, and
|
|
5
|
-
# "46/46 green" meant "green on one laptop, when I remembered".
|
|
6
|
-
#
|
|
7
|
-
# GitHub-hosted runners on purpose: control-arm has no dependencies and must work on a
|
|
8
|
-
# stock box. Pinning it to a self-hosted fleet would hide exactly the assumptions —
|
|
9
|
-
# a preinstalled binary, a warm cache, a particular git version — that break for the
|
|
10
|
-
# first stranger who clones it.
|
|
11
|
-
name: CI
|
|
12
|
-
|
|
13
|
-
on:
|
|
14
|
-
push:
|
|
15
|
-
branches: [master, main]
|
|
16
|
-
pull_request:
|
|
17
|
-
workflow_dispatch:
|
|
18
|
-
|
|
19
|
-
concurrency:
|
|
20
|
-
group: ci-${{ github.ref }}
|
|
21
|
-
cancel-in-progress: true
|
|
22
|
-
|
|
23
|
-
jobs:
|
|
24
|
-
test:
|
|
25
|
-
name: node ${{ matrix.node }}
|
|
26
|
-
runs-on: ubuntu-latest
|
|
27
|
-
strategy:
|
|
28
|
-
fail-fast: false
|
|
29
|
-
matrix:
|
|
30
|
-
# 20 is the oldest LTS with a stable node:test reporter API, which the TAP parser
|
|
31
|
-
# depends on. 24 is what it is developed against. A break in either is worth knowing.
|
|
32
|
-
node: ['20', '22', '24']
|
|
33
|
-
steps:
|
|
34
|
-
- uses: actions/checkout@v4
|
|
35
|
-
with:
|
|
36
|
-
# FULL HISTORY, not the default shallow clone. The dogfood step verifies the tool
|
|
37
|
-
# against one of its OWN past commits, and `git show 1845c1d3` on a depth-1
|
|
38
|
-
# checkout fails with "unknown revision". Caught by this workflow's first run,
|
|
39
|
-
# which is the argument for having it.
|
|
40
|
-
fetch-depth: 0
|
|
41
|
-
- uses: actions/setup-node@v4
|
|
42
|
-
with:
|
|
43
|
-
node-version: ${{ matrix.node }}
|
|
44
|
-
|
|
45
|
-
- name: Unit + fixtures
|
|
46
|
-
run: node --test --test-concurrency=1 test/*.test.mjs
|
|
47
|
-
|
|
48
|
-
# DOGFOOD. The fixtures prove the verdicts on repos built to have known answers.
|
|
49
|
-
# This proves the whole two-arm machinery works on a REAL history with real commits,
|
|
50
|
-
# worktrees and module resolution — which is where every bug so far has come from.
|
|
51
|
-
- name: Judge its own history
|
|
52
|
-
run: |
|
|
53
|
-
set -euo pipefail
|
|
54
|
-
git config --global user.email ci@control-arm
|
|
55
|
-
git config --global user.name ci
|
|
56
|
-
|
|
57
|
-
OUT=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --timeout 120000)
|
|
58
|
-
echo "$OUT"
|
|
59
|
-
|
|
60
|
-
# That commit added the rule "a SKIPPED case blocks BLIND", with a test written
|
|
61
|
-
# for it. If the tool cannot still see that test discriminate, the tool is broken
|
|
62
|
-
# — regardless of what its unit suite says.
|
|
63
|
-
# Match the VERDICT WORD, not the whole line. This grep was 'VERDICT CAUGHT'
|
|
64
|
-
# and broke the moment a glyph was added between them — a cosmetic change that
|
|
65
|
-
# failed a correctness gate, which trains people to edit the gate rather than
|
|
66
|
-
# believe it. Anchor on what the check is actually about.
|
|
67
|
-
echo "$OUT" | grep -qE '^ *VERDICT .*\bCAUGHT\b' \
|
|
68
|
-
|| { echo "::error::control-arm no longer judges its own fix correctly"; exit 1; }
|
|
69
|
-
# And it must be the STRONG claim: 1845c1d3 is a repair, so a "weak evidence"
|
|
70
|
-
# qualifier here would mean the new-code heuristic has started misfiring on fixes.
|
|
71
|
-
echo "$OUT" | grep -q 'weak evidence' \
|
|
72
|
-
&& { echo "::error::a repair was labelled new code — the kind heuristic is wrong"; exit 1; }
|
|
73
|
-
echo "$OUT" | grep -q 'module identity verified' \
|
|
74
|
-
|| { echo "::error::module identity was not proven — a verdict here is not trustworthy"; exit 1; }
|
|
75
|
-
|
|
76
|
-
- name: Determinism
|
|
77
|
-
run: |
|
|
78
|
-
set -euo pipefail
|
|
79
|
-
A=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --work /tmp/d1 --timeout 120000 | tail -14)
|
|
80
|
-
B=$(node bin/ca.mjs verify 1845c1d3 --repo "$GITHUB_WORKSPACE" --work /tmp/d2 --timeout 120000 | tail -14)
|
|
81
|
-
[ "$A" = "$B" ] || { echo "::error::two runs disagreed — the tool is not deterministic"; exit 1; }
|
|
82
|
-
echo "two independent runs are byte-identical"
|
package/action.yml
DELETED
|
@@ -1,95 +0,0 @@
|
|
|
1
|
-
# A composite action, not a Docker one: the tool is plain Node with no dependencies, so a
|
|
2
|
-
# container would add a minute of build time to a check that otherwise takes seconds.
|
|
3
|
-
name: 'control-arm'
|
|
4
|
-
description: 'Prove the tests in this PR fail without it'
|
|
5
|
-
inputs:
|
|
6
|
-
base:
|
|
7
|
-
description: >-
|
|
8
|
-
Branch this PR targets. The comparison uses the MERGE BASE with it, never its tip.
|
|
9
|
-
Defaults to `main`; set it explicitly if your project integrates somewhere else
|
|
10
|
-
(`develop`, `trunk`, a release branch).
|
|
11
|
-
required: false
|
|
12
|
-
default: 'main'
|
|
13
|
-
comment:
|
|
14
|
-
description: 'Post the result as a PR comment. Needs pull-requests:write and GH_TOKEN.'
|
|
15
|
-
required: false
|
|
16
|
-
default: 'true'
|
|
17
|
-
fail-on-blind:
|
|
18
|
-
description: >-
|
|
19
|
-
Fail the check when NO case in the PR discriminates. Default false: a PR can
|
|
20
|
-
legitimately ship only regression guards, and a gate that fires on those gets
|
|
21
|
-
switched off within a week.
|
|
22
|
-
required: false
|
|
23
|
-
default: 'false'
|
|
24
|
-
timeout-ms:
|
|
25
|
-
required: false
|
|
26
|
-
default: '180000'
|
|
27
|
-
outputs:
|
|
28
|
-
verdict:
|
|
29
|
-
description: 'CAUGHT | BLIND | INCONCLUSIVE | FLAKY | SKIPPED'
|
|
30
|
-
value: ${{ steps.run.outputs.verdict }}
|
|
31
|
-
runs:
|
|
32
|
-
using: composite
|
|
33
|
-
steps:
|
|
34
|
-
- id: run
|
|
35
|
-
shell: bash
|
|
36
|
-
env:
|
|
37
|
-
CA_BASE: ${{ inputs.base }}
|
|
38
|
-
CA_TIMEOUT: ${{ inputs.timeout-ms }}
|
|
39
|
-
run: |
|
|
40
|
-
set -uo pipefail
|
|
41
|
-
# The merge base needs both histories. A shallow checkout has neither.
|
|
42
|
-
git -C "$GITHUB_WORKSPACE" fetch --no-tags --depth=200 origin "$CA_BASE" 2>/dev/null || true
|
|
43
|
-
TIP="$(git -C "$GITHUB_WORKSPACE" rev-parse HEAD)"
|
|
44
|
-
|
|
45
|
-
OUT=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
|
|
46
|
-
--against "origin/$CA_BASE" \
|
|
47
|
-
--repo "$GITHUB_WORKSPACE" \
|
|
48
|
-
--timeout "$CA_TIMEOUT" 2>&1) || true
|
|
49
|
-
echo "$OUT"
|
|
50
|
-
|
|
51
|
-
# `|| true` is load-bearing. GitHub runs bash steps with `set -e` by default, and
|
|
52
|
-
# `set -uo pipefail` above does not unset it. A grep that finds nothing exits 1,
|
|
53
|
-
# pipefail propagates it, and the step dies BEFORE the :-INCONCLUSIVE fallback can
|
|
54
|
-
# run. That is exactly what happens on a PR the tool declines — a workflow-only
|
|
55
|
-
# change with no test — so the one case designed to be a non-event failed the check.
|
|
56
|
-
VERDICT=$(printf '%s' "$OUT" | grep -oE 'VERDICT +[A-Z]+' | awk '{print $2}' | head -1 || true)
|
|
57
|
-
VERDICT="${VERDICT:-INCONCLUSIVE}"
|
|
58
|
-
echo "verdict=$VERDICT" >> "$GITHUB_OUTPUT"
|
|
59
|
-
|
|
60
|
-
# `--pr-comment` returns EMPTY for a commit the tool declined to judge, and the
|
|
61
|
-
# comment step skips on an empty file. The decision is the tool's, not a shell's:
|
|
62
|
-
# scraping stdout for a VERDICT line a declined run never prints is what put
|
|
63
|
-
# "SKIPPED · no case fails without the change" onto a PR that simply has no tests.
|
|
64
|
-
MD=$(node "${{ github.action_path }}/bin/ca.mjs" verify "$TIP" \
|
|
65
|
-
--against "origin/$CA_BASE" --pr-comment \
|
|
66
|
-
--repo "$GITHUB_WORKSPACE" --timeout "$CA_TIMEOUT" 2>/dev/null) || true
|
|
67
|
-
printf '%s' "$MD" > "$RUNNER_TEMP/ca-comment.md"
|
|
68
|
-
[ -s "$RUNNER_TEMP/ca-comment.md" ] || echo "declined — nothing to post: $(printf '%s' "$OUT" | tail -1)"
|
|
69
|
-
|
|
70
|
-
- shell: bash
|
|
71
|
-
if: inputs.comment == 'true' && github.event_name == 'pull_request'
|
|
72
|
-
env:
|
|
73
|
-
GITHUB_TOKEN: ${{ github.token }}
|
|
74
|
-
run: |
|
|
75
|
-
# node, not `gh`. The CLI is not installed on every runner — a self-hosted
|
|
76
|
-
# bare-metal box failed here with `gh: command not found` after every other step
|
|
77
|
-
# had passed, so the tool ran, reached the right verdict, and could not say so.
|
|
78
|
-
node "${{ github.action_path }}/scripts/gh-api.mjs" upsert-pr-comment \
|
|
79
|
-
"${{ github.repository }}" \
|
|
80
|
-
"${{ github.event.pull_request.number }}" \
|
|
81
|
-
"$RUNNER_TEMP/ca-comment.md" \
|
|
82
|
-
'### `control-arm`'
|
|
83
|
-
|
|
84
|
-
- shell: bash
|
|
85
|
-
if: inputs.fail-on-blind == 'true' && steps.run.outputs.verdict == 'BLIND'
|
|
86
|
-
run: |
|
|
87
|
-
echo "::error::No test in this PR fails without the change."
|
|
88
|
-
exit 1
|
|
89
|
-
|
|
90
|
-
# Required by the GitHub Marketplace listing. `rewind` is not decoration: it is
|
|
91
|
-
# literally what this does — wind the code back to before the fix, then run the
|
|
92
|
-
# new test against it.
|
|
93
|
-
branding:
|
|
94
|
-
icon: 'rewind'
|
|
95
|
-
color: 'orange'
|
package/fixtures/build.mjs
DELETED
|
@@ -1,178 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Fixture repos with KNOWN answers — this tool's own control arm.
|
|
3
|
-
*
|
|
4
|
-
* The tool's whole claim is "a test that cannot fail is decoration". A tool that asserts
|
|
5
|
-
* that about other people's tests, while its own suite only checks that it doesn't crash,
|
|
6
|
-
* is the same defect one level up. So: tiny git repos where the right verdict is known by
|
|
7
|
-
* construction, and `test/fixtures.test.mjs` asserts the tool returns it.
|
|
8
|
-
*
|
|
9
|
-
* If `ca` cannot tell 02-blind-direction from 01-caught-value, it does not ship.
|
|
10
|
-
*
|
|
11
|
-
* THE BUG THEY ALL SHARE. `priority()` maps a label to a number. The broken version joins
|
|
12
|
-
* its words with `\s*`, which matches whitespace and nothing else, so it reads
|
|
13
|
-
* "HIGH PRIORITY" but not "HIGH-PRIORITY" — one separator, a different answer. The fixed
|
|
14
|
-
* version accepts any run of real separators.
|
|
15
|
-
*
|
|
16
|
-
* Deliberately a boring, universal domain: every issue tracker has priority labels, and
|
|
17
|
-
* the fixtures should not require knowing anybody's product to read. Only the TEST differs
|
|
18
|
-
* between fixtures; the bug is identical in all of them, so a verdict can only come from
|
|
19
|
-
* the test's quality.
|
|
20
|
-
*/
|
|
21
|
-
|
|
22
|
-
import { execFileSync } from 'node:child_process';
|
|
23
|
-
import { mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
|
24
|
-
import path from 'node:path';
|
|
25
|
-
import { fileURLToPath } from 'node:url';
|
|
26
|
-
|
|
27
|
-
const ROOT = path.join(path.dirname(fileURLToPath(import.meta.url)), '.build');
|
|
28
|
-
|
|
29
|
-
const BROKEN = `export function priority(label) {
|
|
30
|
-
const s = String(label).toLowerCase();
|
|
31
|
-
if (/high\\s*priority/.test(s)) return 1; // \\s* matches whitespace and nothing else
|
|
32
|
-
if (/low\\s*priority/.test(s)) return 3;
|
|
33
|
-
return 2;
|
|
34
|
-
}
|
|
35
|
-
export const SLA_HOURS = { 1: 4, 2: 24, 3: 72 };
|
|
36
|
-
`;
|
|
37
|
-
const FIXED = BROKEN
|
|
38
|
-
.replace('/high\\s*priority/', '/high[-_\\s.]*priority/')
|
|
39
|
-
.replace('/low\\s*priority/', '/low[-_\\s.]*priority/');
|
|
40
|
-
|
|
41
|
-
const FIXTURES = {
|
|
42
|
-
'01-caught-value': {
|
|
43
|
-
expect: 'CAUGHT',
|
|
44
|
-
why: 'asserts the exact priority for the hyphenated spelling',
|
|
45
|
-
test: `import { test } from 'node:test';
|
|
46
|
-
import assert from 'node:assert/strict';
|
|
47
|
-
import { priority, SLA_HOURS } from '../src/priority.mjs';
|
|
48
|
-
test('a hyphenated HIGH-PRIORITY label is priority 1, four-hour SLA', () => {
|
|
49
|
-
assert.equal(priority('HIGH-PRIORITY'), 1);
|
|
50
|
-
assert.equal(SLA_HOURS[priority('HIGH-PRIORITY')], 4);
|
|
51
|
-
});`,
|
|
52
|
-
},
|
|
53
|
-
'02-blind-direction': {
|
|
54
|
-
expect: 'BLIND',
|
|
55
|
-
why: 'asserts a DIRECTION (>0) that the wrong answer also satisfies',
|
|
56
|
-
test: `import { test } from 'node:test';
|
|
57
|
-
import assert from 'node:assert/strict';
|
|
58
|
-
import { priority, SLA_HOURS } from '../src/priority.mjs';
|
|
59
|
-
test('hyphenated labels are handled', () => {
|
|
60
|
-
const p = priority('HIGH-PRIORITY');
|
|
61
|
-
assert.ok(p, 'a priority comes back');
|
|
62
|
-
assert.ok(SLA_HOURS[p] > 0, 'it has an SLA');
|
|
63
|
-
assert.notEqual(p, undefined);
|
|
64
|
-
});`,
|
|
65
|
-
},
|
|
66
|
-
'03-blind-sourcetext': {
|
|
67
|
-
expect: 'BLIND',
|
|
68
|
-
why: 'greps the source instead of executing it',
|
|
69
|
-
test: `import { test } from 'node:test';
|
|
70
|
-
import assert from 'node:assert/strict';
|
|
71
|
-
import { readFileSync } from 'node:fs';
|
|
72
|
-
test('the separator class is tolerant', () => {
|
|
73
|
-
const src = readFileSync(new URL('../src/priority.mjs', import.meta.url), 'utf8');
|
|
74
|
-
assert.ok(src.includes('priority'), 'the rule mentions priority');
|
|
75
|
-
assert.ok(/high/.test(src));
|
|
76
|
-
});`,
|
|
77
|
-
},
|
|
78
|
-
'04-blind-overmock': {
|
|
79
|
-
expect: 'BLIND',
|
|
80
|
-
why: 'mocks the unit under test, so the real function never runs',
|
|
81
|
-
test: `import { test } from 'node:test';
|
|
82
|
-
import assert from 'node:assert/strict';
|
|
83
|
-
import { SLA_HOURS } from '../src/priority.mjs';
|
|
84
|
-
const priority = () => 1; // "stubbed for speed"
|
|
85
|
-
test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
|
|
86
|
-
assert.equal(priority('HIGH-PRIORITY'), 1);
|
|
87
|
-
assert.equal(SLA_HOURS[1], 4);
|
|
88
|
-
});`,
|
|
89
|
-
},
|
|
90
|
-
'05-inconclusive-newexport': {
|
|
91
|
-
expect: 'INCONCLUSIVE',
|
|
92
|
-
why: 'imports a symbol the fix added — cannot even load at the parent',
|
|
93
|
-
fixedExtra: `export const SEPARATORS = /[-_\\s.]*/;\n`,
|
|
94
|
-
test: `import { test } from 'node:test';
|
|
95
|
-
import assert from 'node:assert/strict';
|
|
96
|
-
import { SEPARATORS } from '../src/priority.mjs';
|
|
97
|
-
test('the separator class is shared', () => {
|
|
98
|
-
assert.equal(SEPARATORS.source, '[-_\\\\s.]*');
|
|
99
|
-
});`,
|
|
100
|
-
},
|
|
101
|
-
'06-inconclusive-armA-red': {
|
|
102
|
-
expect: 'INCONCLUSIVE',
|
|
103
|
-
why: 'the case is not green on the fix either — the commit does not stand up',
|
|
104
|
-
test: `import { test } from 'node:test';
|
|
105
|
-
import assert from 'node:assert/strict';
|
|
106
|
-
import { priority } from '../src/priority.mjs';
|
|
107
|
-
test('a hyphenated HIGH-PRIORITY label is priority 1', () => {
|
|
108
|
-
assert.equal(priority('HIGH-PRIORITY'), 99);
|
|
109
|
-
});`,
|
|
110
|
-
},
|
|
111
|
-
'07-caught-mixed': {
|
|
112
|
-
expect: 'CAUGHT',
|
|
113
|
-
why: 'one discriminating case plus two regression guards green on both arms',
|
|
114
|
-
test: `import { test } from 'node:test';
|
|
115
|
-
import assert from 'node:assert/strict';
|
|
116
|
-
import { priority } from '../src/priority.mjs';
|
|
117
|
-
test('DISCRIMINATES: the hyphenated spelling is priority 1', () => {
|
|
118
|
-
assert.equal(priority('HIGH-PRIORITY'), 1);
|
|
119
|
-
});
|
|
120
|
-
test('GUARD: the spaced spelling is still priority 1', () => {
|
|
121
|
-
assert.equal(priority('HIGH PRIORITY'), 1);
|
|
122
|
-
});
|
|
123
|
-
test('GUARD: an unlabelled ticket is still the default priority 2', () => {
|
|
124
|
-
assert.equal(priority('needs triage'), 2);
|
|
125
|
-
});`,
|
|
126
|
-
},
|
|
127
|
-
'09-caught-slash-in-name': {
|
|
128
|
-
expect: 'CAUGHT',
|
|
129
|
-
why: 'the discriminating case has a SLASH in its name — it used to be silently dropped',
|
|
130
|
-
test: `import { test } from 'node:test';
|
|
131
|
-
import assert from 'node:assert/strict';
|
|
132
|
-
import { priority } from '../src/priority.mjs';
|
|
133
|
-
test('label parsing / separator handling', () => {
|
|
134
|
-
assert.equal(priority('HIGH-PRIORITY'), 1);
|
|
135
|
-
});`,
|
|
136
|
-
},
|
|
137
|
-
'08-skipped-notest': {
|
|
138
|
-
expect: 'SKIPPED',
|
|
139
|
-
why: 'the fix shipped no test at all',
|
|
140
|
-
test: null,
|
|
141
|
-
},
|
|
142
|
-
};
|
|
143
|
-
|
|
144
|
-
function sh(cwd, args) { execFileSync('git', args, { cwd, stdio: 'pipe' }); }
|
|
145
|
-
|
|
146
|
-
export function buildFixtures() {
|
|
147
|
-
rmSync(ROOT, { recursive: true, force: true });
|
|
148
|
-
mkdirSync(ROOT, { recursive: true });
|
|
149
|
-
const built = {};
|
|
150
|
-
|
|
151
|
-
for (const [name, spec] of Object.entries(FIXTURES)) {
|
|
152
|
-
const dir = path.join(ROOT, name);
|
|
153
|
-
mkdirSync(path.join(dir, 'src'), { recursive: true });
|
|
154
|
-
mkdirSync(path.join(dir, 'tests'), { recursive: true });
|
|
155
|
-
sh(dir, ['init', '-q']);
|
|
156
|
-
sh(dir, ['config', 'user.email', 'ca@fixture']);
|
|
157
|
-
sh(dir, ['config', 'user.name', 'ca fixture']);
|
|
158
|
-
writeFileSync(path.join(dir, 'package.json'), JSON.stringify({ name, type: 'module', private: true }, null, 2));
|
|
159
|
-
|
|
160
|
-
// --- parent: the bug, and whatever tests existed before (none) ---
|
|
161
|
-
writeFileSync(path.join(dir, 'src/priority.mjs'), BROKEN);
|
|
162
|
-
sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'feat: priority labels']);
|
|
163
|
-
|
|
164
|
-
// --- fix: source repaired, test added ---
|
|
165
|
-
writeFileSync(path.join(dir, 'src/priority.mjs'), FIXED + (spec.fixedExtra || ''));
|
|
166
|
-
if (spec.test) writeFileSync(path.join(dir, 'tests/priority.test.mjs'), spec.test + '\n');
|
|
167
|
-
sh(dir, ['add', '-A']); sh(dir, ['commit', '-qm', 'fix: a hyphen made a HIGH-PRIORITY ticket read as normal']);
|
|
168
|
-
|
|
169
|
-
built[name] = { dir, sha: execFileSync('git', ['-C', dir, 'rev-parse', 'HEAD']).toString().trim(), ...spec };
|
|
170
|
-
}
|
|
171
|
-
return built;
|
|
172
|
-
}
|
|
173
|
-
|
|
174
|
-
if (import.meta.url === `file://${process.argv[1]}`) {
|
|
175
|
-
const b = buildFixtures();
|
|
176
|
-
for (const [n, f] of Object.entries(b)) console.log(` ${f.expect.padEnd(13)} ${n.padEnd(26)} ${f.why}`);
|
|
177
|
-
console.log(`\n ${Object.keys(b).length} fixtures in ${ROOT}\n`);
|
|
178
|
-
}
|
package/scripts/gh-api.mjs
DELETED
|
@@ -1,90 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
/**
|
|
3
|
-
* The few GitHub API calls these workflows need, over plain fetch.
|
|
4
|
-
*
|
|
5
|
-
* WHY NOT THE `gh` CLI. It is not installed on every runner. Measured: a self-hosted
|
|
6
|
-
* bare-metal runner failed with `gh: command not found` (exit 127) after every other step
|
|
7
|
-
* had passed — the tool ran, reached the right verdict, and then could not say so.
|
|
8
|
-
* GitHub-hosted runners ship `gh`; a self-hosted one ships whatever was installed on it,
|
|
9
|
-
* and a workflow that assumes otherwise works until it lands on the wrong machine.
|
|
10
|
-
*
|
|
11
|
-
* Node is already a hard requirement here — the tool is written in it — so this adds no
|
|
12
|
-
* dependency at all. Usage:
|
|
13
|
-
*
|
|
14
|
-
* gh-api.mjs upsert-pr-comment <repo> <pr> <body-file> <marker>
|
|
15
|
-
* gh-api.mjs upsert-issue <repo> <title> <body-file> <search> [label]
|
|
16
|
-
*
|
|
17
|
-
* Reads GITHUB_TOKEN / GH_TOKEN from the environment. Prints what it did, and exits
|
|
18
|
-
* non-zero only when the API refuses — never merely because there was nothing to do.
|
|
19
|
-
*/
|
|
20
|
-
|
|
21
|
-
const TOKEN = process.env.GITHUB_TOKEN || process.env.GH_TOKEN;
|
|
22
|
-
const API = process.env.GITHUB_API_URL || 'https://api.github.com';
|
|
23
|
-
|
|
24
|
-
async function gh(path, init = {}) {
|
|
25
|
-
const res = await fetch(`${API}${path}`, {
|
|
26
|
-
...init,
|
|
27
|
-
headers: {
|
|
28
|
-
authorization: `Bearer ${TOKEN}`,
|
|
29
|
-
accept: 'application/vnd.github+json',
|
|
30
|
-
'content-type': 'application/json',
|
|
31
|
-
'x-github-api-version': '2022-11-28',
|
|
32
|
-
...(init.headers || {}),
|
|
33
|
-
},
|
|
34
|
-
});
|
|
35
|
-
if (!res.ok) {
|
|
36
|
-
const text = await res.text();
|
|
37
|
-
throw new Error(`${init.method || 'GET'} ${path} → ${res.status} ${text.slice(0, 300)}`);
|
|
38
|
-
}
|
|
39
|
-
return res.status === 204 ? null : res.json();
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
const [cmd, ...args] = process.argv.slice(2);
|
|
43
|
-
const read = async (f) => (await import('node:fs/promises')).readFile(f, 'utf8');
|
|
44
|
-
|
|
45
|
-
try {
|
|
46
|
-
if (!TOKEN) throw new Error('no GITHUB_TOKEN / GH_TOKEN in the environment');
|
|
47
|
-
|
|
48
|
-
if (cmd === 'upsert-pr-comment') {
|
|
49
|
-
const [repo, pr, bodyFile, marker] = args;
|
|
50
|
-
const body = await read(bodyFile);
|
|
51
|
-
if (!body.trim()) { console.log('nothing to post'); process.exit(0); }
|
|
52
|
-
// ONE COMMENT PER PR, edited in place. A new comment per push turns a useful signal
|
|
53
|
-
// into noise by the third revision, and the marker is how we find ours again.
|
|
54
|
-
const comments = await gh(`/repos/${repo}/issues/${pr}/comments?per_page=100`);
|
|
55
|
-
const mine = comments.find(c => (c.body || '').startsWith(marker));
|
|
56
|
-
if (mine) {
|
|
57
|
-
await gh(`/repos/${repo}/issues/comments/${mine.id}`, { method: 'PATCH', body: JSON.stringify({ body }) });
|
|
58
|
-
console.log(`updated comment ${mine.id}`);
|
|
59
|
-
} else {
|
|
60
|
-
const made = await gh(`/repos/${repo}/issues/${pr}/comments`, { method: 'POST', body: JSON.stringify({ body }) });
|
|
61
|
-
console.log(`posted comment ${made.id}`);
|
|
62
|
-
}
|
|
63
|
-
} else if (cmd === 'upsert-issue') {
|
|
64
|
-
const [repo, title, bodyFile, search, label] = args;
|
|
65
|
-
const body = await read(bodyFile);
|
|
66
|
-
if (label) {
|
|
67
|
-
// Create the label if absent. `labels` on an issue with an unknown label is
|
|
68
|
-
// rejected outright, so this cannot be left to chance — but a failure here is
|
|
69
|
-
// a warning, not a reason to drop the finding on the floor.
|
|
70
|
-
try {
|
|
71
|
-
await gh(`/repos/${repo}/labels`, { method: 'POST', body: JSON.stringify({ name: label, color: '0E8A16', description: 'Test-gap findings from control-arm' }) });
|
|
72
|
-
} catch (e) { if (!/already_exists|422/.test(e.message)) console.log(`::warning::could not create label ${label}: ${e.message}`); }
|
|
73
|
-
}
|
|
74
|
-
const found = await gh(`/search/issues?q=${encodeURIComponent(`repo:${repo} is:issue is:open ${search}`)}`);
|
|
75
|
-
const hit = found.items?.[0];
|
|
76
|
-
if (hit) {
|
|
77
|
-
await gh(`/repos/${repo}/issues/${hit.number}`, { method: 'PATCH', body: JSON.stringify({ title, body }) });
|
|
78
|
-
console.log(`refreshed issue #${hit.number}`);
|
|
79
|
-
} else {
|
|
80
|
-
const made = await gh(`/repos/${repo}/issues`, { method: 'POST', body: JSON.stringify({ title, body, ...(label ? { labels: [label] } : {}) }) });
|
|
81
|
-
console.log(`filed issue #${made.number}`);
|
|
82
|
-
}
|
|
83
|
-
} else {
|
|
84
|
-
console.error('usage: gh-api.mjs upsert-pr-comment|upsert-issue ...');
|
|
85
|
-
process.exit(2);
|
|
86
|
-
}
|
|
87
|
-
} catch (e) {
|
|
88
|
-
console.error(`::error::${e.message}`);
|
|
89
|
-
process.exit(1);
|
|
90
|
-
}
|
package/scripts-analyze.mjs
DELETED
|
@@ -1,86 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Segment the audit CSV by which runner the test file belongs to.
|
|
3
|
-
*
|
|
4
|
-
* `ca` ships one runner (node:test). A monorepo's fix commits touch vitest and jest test
|
|
5
|
-
* files too, and those come back INCONCLUSIVE for a reason that says nothing about the
|
|
6
|
-
* test — the tool simply cannot execute it. Reporting one blended ratio over both would
|
|
7
|
-
* be the same defect the tool exists to catch: a number fitted to a population it does
|
|
8
|
-
* not describe.
|
|
9
|
-
*/
|
|
10
|
-
import { readFileSync } from 'node:fs';
|
|
11
|
-
|
|
12
|
-
const rows = [];
|
|
13
|
-
const raw = readFileSync(process.argv[2], 'utf8').split('\n').filter(Boolean);
|
|
14
|
-
const hdr = raw.shift().split(',');
|
|
15
|
-
for (const line of raw) {
|
|
16
|
-
// naive CSV with quoted fields
|
|
17
|
-
const f = []; let cur = '', q = false;
|
|
18
|
-
for (let i = 0; i < line.length; i++) {
|
|
19
|
-
const c = line[i];
|
|
20
|
-
if (q) { if (c === '"' && line[i + 1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
|
|
21
|
-
else if (c === '"') q = true;
|
|
22
|
-
else if (c === ',') { f.push(cur); cur = ''; }
|
|
23
|
-
else cur += c;
|
|
24
|
-
}
|
|
25
|
-
f.push(cur);
|
|
26
|
-
rows.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
const runnerOf = (file) => {
|
|
30
|
-
if (!file) return 'none';
|
|
31
|
-
if (file.startsWith('tests/')) return 'node:test (supported)';
|
|
32
|
-
if (file.startsWith('apps/web/')) return 'vitest (unsupported)';
|
|
33
|
-
if (file.startsWith('apps/mobile/')) return 'jest (unsupported)';
|
|
34
|
-
if (file.startsWith('apps/e2e/')) return 'playwright (unsupported)';
|
|
35
|
-
if (file.startsWith('packages/')) return 'node:test (supported)';
|
|
36
|
-
return 'other';
|
|
37
|
-
};
|
|
38
|
-
|
|
39
|
-
const commits = new Map();
|
|
40
|
-
for (const r of rows) {
|
|
41
|
-
if (!commits.has(r.sha)) commits.set(r.sha, { sha: r.sha, date: r.date, subject: r.subject, verdict: r.commit_verdict, cases: [] });
|
|
42
|
-
commits.get(r.sha).cases.push(r);
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
// A commit belongs to the runner of its test files; mixed commits are called out.
|
|
46
|
-
const seg = new Map();
|
|
47
|
-
for (const c of commits.values()) {
|
|
48
|
-
const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
|
|
49
|
-
const key = rs.length === 0 ? 'no test file' : rs.length === 1 ? rs[0] : 'mixed';
|
|
50
|
-
if (!seg.has(key)) seg.set(key, []);
|
|
51
|
-
seg.get(key).push(c);
|
|
52
|
-
}
|
|
53
|
-
|
|
54
|
-
const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
|
|
55
|
-
console.log('\n COMMITS BY RUNNER SEGMENT\n');
|
|
56
|
-
console.log(' ' + 'segment'.padEnd(26) + V.map(v => v.slice(0, 6).padStart(7)).join('') + ' n');
|
|
57
|
-
for (const [k, list] of [...seg].sort((a, b) => b[1].length - a[1].length)) {
|
|
58
|
-
const t = V.map(v => String(list.filter(c => c.verdict === v).length).padStart(7)).join('');
|
|
59
|
-
console.log(' ' + k.padEnd(26) + t + String(list.length).padStart(5));
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
const supported = seg.get('node:test (supported)') || [];
|
|
63
|
-
const dec = supported.filter(c => c.verdict === 'CAUGHT' || c.verdict === 'BLIND');
|
|
64
|
-
console.log(`\n SUPPORTED SEGMENT ONLY (node:test)\n`);
|
|
65
|
-
console.log(` ${supported.length} commits · ${dec.length} the instrument could answer`);
|
|
66
|
-
if (dec.length) {
|
|
67
|
-
const caught = dec.filter(c => c.verdict === 'CAUGHT').length;
|
|
68
|
-
console.log(` CAUGHT ${caught}/${dec.length} = ${(caught / dec.length * 100).toFixed(1)}% BLIND ${dec.length - caught}/${dec.length} = ${((dec.length - caught) / dec.length * 100).toFixed(1)}%`);
|
|
69
|
-
}
|
|
70
|
-
const blind = supported.filter(c => c.verdict === 'BLIND');
|
|
71
|
-
if (blind.length) {
|
|
72
|
-
console.log(`\n BLIND COMMITS IN THE SUPPORTED SEGMENT (hand-audit these)\n`);
|
|
73
|
-
for (const c of blind) console.log(` ${c.sha.slice(0, 8)} ${c.date} ${c.subject.slice(0, 90)}`);
|
|
74
|
-
}
|
|
75
|
-
const why = new Map();
|
|
76
|
-
for (const c of supported.filter(c => c.verdict === 'INCONCLUSIVE')) {
|
|
77
|
-
for (const cs of c.cases.filter(x => x.case_verdict === 'INCONCLUSIVE')) {
|
|
78
|
-
const k = cs.reason.replace(/'[^']*'/g, "'…'").replace(/\d+/g, 'N').slice(0, 88);
|
|
79
|
-
why.set(k, (why.get(k) || 0) + 1);
|
|
80
|
-
}
|
|
81
|
-
}
|
|
82
|
-
if (why.size) {
|
|
83
|
-
console.log(`\n WHY INCONCLUSIVE, supported segment only\n`);
|
|
84
|
-
for (const [k, v] of [...why].sort((a, b) => b[1] - a[1]).slice(0, 14)) console.log(` ${String(v).padStart(4)} ${k}`);
|
|
85
|
-
}
|
|
86
|
-
console.log('');
|
package/scripts-recompute.mjs
DELETED
|
@@ -1,53 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Re-roll commit verdicts from a saved audit CSV using the CURRENT rollUp.
|
|
3
|
-
*
|
|
4
|
-
* Per-case verdicts are what the two arms measured; the commit verdict is a pure function
|
|
5
|
-
* of them. So a change to the roll-up rule does not need the 7-minute run again — and
|
|
6
|
-
* re-running would also change the sample, which is the wrong thing to do when comparing
|
|
7
|
-
* a rule change.
|
|
8
|
-
*/
|
|
9
|
-
import { readFileSync } from 'node:fs';
|
|
10
|
-
import { rollUp } from './src/verdict.mjs';
|
|
11
|
-
|
|
12
|
-
function parseCsv(text) {
|
|
13
|
-
const out = []; const lines = text.split('\n').filter(Boolean); const hdr = lines.shift().split(',');
|
|
14
|
-
for (const line of lines) {
|
|
15
|
-
const f = []; let cur = '', q = false;
|
|
16
|
-
for (let i = 0; i < line.length; i++) { const c = line[i];
|
|
17
|
-
if (q) { if (c === '"' && line[i+1] === '"') { cur += '"'; i++; } else if (c === '"') q = false; else cur += c; }
|
|
18
|
-
else if (c === '"') q = true; else if (c === ',') { f.push(cur); cur = ''; } else cur += c; }
|
|
19
|
-
f.push(cur); out.push(Object.fromEntries(hdr.map((h, i) => [h, f[i]])));
|
|
20
|
-
}
|
|
21
|
-
return out;
|
|
22
|
-
}
|
|
23
|
-
const runnerOf = f => !f ? 'none'
|
|
24
|
-
: f.startsWith('tests/') || f.startsWith('packages/') ? 'node:test'
|
|
25
|
-
: f.startsWith('apps/web/') ? 'vitest' : f.startsWith('apps/mobile/') ? 'jest'
|
|
26
|
-
: f.startsWith('apps/e2e/') ? 'playwright' : 'other';
|
|
27
|
-
|
|
28
|
-
const rows = parseCsv(readFileSync(process.argv[2], 'utf8'));
|
|
29
|
-
const commits = new Map();
|
|
30
|
-
for (const r of rows) {
|
|
31
|
-
if (!commits.has(r.sha)) commits.set(r.sha, { ...r, cases: [] });
|
|
32
|
-
if (r.case) commits.get(r.sha).cases.push(r);
|
|
33
|
-
}
|
|
34
|
-
const list = [...commits.values()].map(c => {
|
|
35
|
-
const nv = c.cases.length ? rollUp(c.cases.map(x => ({ verdict: x.case_verdict }))) : c.commit_verdict;
|
|
36
|
-
const rs = [...new Set(c.cases.map(x => runnerOf(x.file)).filter(x => x !== 'none'))];
|
|
37
|
-
return { ...c, was: c.commit_verdict, now: nv, seg: rs.length === 1 ? rs[0] : rs.length ? 'mixed' : 'none' };
|
|
38
|
-
});
|
|
39
|
-
|
|
40
|
-
const V = ['CAUGHT', 'BLIND', 'FLAKY', 'INCONCLUSIVE', 'SKIPPED'];
|
|
41
|
-
const changed = list.filter(c => c.was !== c.now);
|
|
42
|
-
console.log(`\n ${list.length} commits · ${changed.length} changed verdict under the corrected roll-up\n`);
|
|
43
|
-
for (const c of changed) console.log(` ${c.was.padEnd(13)} -> ${c.now.padEnd(13)} ${c.sha.slice(0,8)} ${c.subject.slice(0,70)}`);
|
|
44
|
-
|
|
45
|
-
const node = list.filter(c => c.seg === 'node:test' || c.seg === 'mixed');
|
|
46
|
-
const dec = node.filter(c => c.now === 'CAUGHT' || c.now === 'BLIND');
|
|
47
|
-
const caught = dec.filter(c => c.now === 'CAUGHT').length;
|
|
48
|
-
console.log(`\n SUPPORTED SEGMENT (node:test, incl. mixed): ${node.length} commits`);
|
|
49
|
-
for (const v of V) { const n = node.filter(c => c.now === v).length; if (n) console.log(` ${v.padEnd(14)} ${String(n).padStart(4)} ${(n/node.length*100).toFixed(1)}%`); }
|
|
50
|
-
console.log(`\n ANSWERABLE: ${dec.length} CAUGHT ${caught} (${(caught/dec.length*100).toFixed(1)}%) BLIND ${dec.length-caught} (${((dec.length-caught)/dec.length*100).toFixed(1)}%)\n`);
|
|
51
|
-
console.log(' BLIND after correction:\n');
|
|
52
|
-
for (const c of list.filter(c => c.now === 'BLIND')) console.log(` ${c.sha.slice(0,8)} ${c.date} ${c.subject.slice(0,84)}`);
|
|
53
|
-
console.log('');
|
package/test/assertions.test.mjs
DELETED
|
@@ -1,130 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Case extraction, pinned against the four shapes that actually broke it.
|
|
3
|
-
*
|
|
4
|
-
* Each of these was measured on a real corpus of 400 test cases, and each one alone
|
|
5
|
-
* accounted for a double-digit share of "could not analyse". Together they took the
|
|
6
|
-
* unresolvable pile from 64% to 1.3%. None would be caught by a suite that only fed the
|
|
7
|
-
* analyser well-formed input, which is why they are here with the messy input attached.
|
|
8
|
-
*/
|
|
9
|
-
import { test } from 'node:test';
|
|
10
|
-
import assert from 'node:assert/strict';
|
|
11
|
-
import { extractCase, analyseCase } from '../src/assertions.mjs';
|
|
12
|
-
|
|
13
|
-
test('nested suites: the runner reports the CONCATENATED name, the source holds the inner one', () => {
|
|
14
|
-
const src = `
|
|
15
|
-
describe('auction client', () => {
|
|
16
|
-
it('AUCWEB-001 calls GET /auction/:vin', () => { assert.equal(a, 1); });
|
|
17
|
-
});`;
|
|
18
|
-
// Reported by the runner as describe-title + space + it-title.
|
|
19
|
-
const body = extractCase(src, 'auction client AUCWEB-001 calls GET /auction/:vin');
|
|
20
|
-
assert.ok(body, 'progressive reduction must find the inner title');
|
|
21
|
-
assert.match(body, /assert\.equal\(a, 1\)/);
|
|
22
|
-
});
|
|
23
|
-
|
|
24
|
-
test('an APOSTROPHE in the title: source holds a backslash the reported name does not', () => {
|
|
25
|
-
const src = `it('METER-038: usage on the user\\'s LOCAL today is counted', () => { assert.equal(n, 3); });`;
|
|
26
|
-
const body = extractCase(src, "METER-038: usage on the user's LOCAL today is counted");
|
|
27
|
-
assert.ok(body, "an escaped quote inside the literal must still match the unescaped reported name");
|
|
28
|
-
assert.match(body, /assert\.equal\(n, 3\)/);
|
|
29
|
-
});
|
|
30
|
-
|
|
31
|
-
test('JSX: a closing tag is NOT a regex literal', () => {
|
|
32
|
-
// `</div>` puts a slash after `<`. The usual "slash after a non-expression starts a
|
|
33
|
-
// regex" heuristic swallows the rest of the component, and the body is never found.
|
|
34
|
-
const src = `
|
|
35
|
-
it('hero renders', () => {
|
|
36
|
-
render(<div className="x">{name}</div>);
|
|
37
|
-
expect(screen.getByText('hi')).toBeTruthy();
|
|
38
|
-
});`;
|
|
39
|
-
const body = extractCase(src, 'hero renders');
|
|
40
|
-
assert.ok(body, 'a JSX body must brace-match cleanly');
|
|
41
|
-
assert.match(body, /getByText/);
|
|
42
|
-
});
|
|
43
|
-
|
|
44
|
-
test('braces inside strings and regexes do not unbalance the scan', () => {
|
|
45
|
-
const src = `
|
|
46
|
-
it('handles braces', () => {
|
|
47
|
-
const s = '{';
|
|
48
|
-
const re = /[{}]/;
|
|
49
|
-
assert.equal(f(s, re), '}');
|
|
50
|
-
});`;
|
|
51
|
-
const body = extractCase(src, 'handles braces');
|
|
52
|
-
assert.ok(body, 'a brace inside a string or a character class is not structure');
|
|
53
|
-
assert.match(body, /assert\.equal\(f\(s, re\)/);
|
|
54
|
-
});
|
|
55
|
-
|
|
56
|
-
test('template-literal titles: every generated case resolves to the shared body', () => {
|
|
57
|
-
const src = `
|
|
58
|
-
for (const rel of PAGES) {
|
|
59
|
-
it(\`\${rel}: no internal repo paths\`, () => { expect(md).not.toMatch(BAD); });
|
|
60
|
-
}`;
|
|
61
|
-
const a = extractCase(src, 'app/page.tsx: no internal repo paths');
|
|
62
|
-
const b = extractCase(src, 'app/pricing/page.tsx: no internal repo paths');
|
|
63
|
-
assert.ok(a && b, 'a parameterised title must match by pattern, not by equality');
|
|
64
|
-
assert.equal(a, b, 'all generated cases share one body — that is correct, not a bug');
|
|
65
|
-
});
|
|
66
|
-
|
|
67
|
-
test('a wrong body is worse than none: a short remainder is refused', () => {
|
|
68
|
-
const src = `
|
|
69
|
-
it('alpha returns null', () => { assert.equal(x, null); });
|
|
70
|
-
it('beta returns null', () => { assert.equal(y, 0); });`;
|
|
71
|
-
// "returns null" alone is ambiguous between the two — reduction must not take it.
|
|
72
|
-
const body = extractCase(src, 'some suite that does not exist returns null');
|
|
73
|
-
assert.equal(body, null, 'an ambiguous short suffix must not resolve to an arbitrary case');
|
|
74
|
-
});
|
|
75
|
-
|
|
76
|
-
test('CONTROL ARM: naive equality-only extraction fails every one of these', () => {
|
|
77
|
-
// The implementation this replaced. Asserted to still get them wrong.
|
|
78
|
-
const naive = (src, name) => {
|
|
79
|
-
const esc = name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
80
|
-
return new RegExp(`\\b(?:test|it)\\s*\\(\\s*(['"\`])${esc}\\1`).test(src);
|
|
81
|
-
};
|
|
82
|
-
assert.equal(naive(`it('a b c', () => {});`, 'suite a b c'), false); // nested
|
|
83
|
-
assert.equal(naive(`it('u\\'s v', () => {});`, "u's v"), false); // apostrophe
|
|
84
|
-
assert.equal(naive('it(`${r}: x`, () => {});', 'p.tsx: x'), false); // template
|
|
85
|
-
});
|
|
86
|
-
|
|
87
|
-
test('the analyser reports UNKNOWN rather than a clean bill when it cannot tell', () => {
|
|
88
|
-
const r = analyseCase(`it('odd', () => { somethingEntirelyUnrecognised(); });`, 'odd');
|
|
89
|
-
assert.notEqual(r.verdict, 'strong', 'an unrecognised body must never read as strong');
|
|
90
|
-
});
|
|
91
|
-
|
|
92
|
-
test('a short helper name must not be accused of stubbing the unit under test', () => {
|
|
93
|
-
// FOUND ON A REAL RUN. The stub check substring-matched the name against every import
|
|
94
|
-
// path, so a two-letter helper `at` matched '@acme/core' — "acme" contains "at" — and
|
|
95
|
-
// every case in that file was reported as testing a stub. A wrong explanation is worse
|
|
96
|
-
// than none: it sends the reader to look at code that is fine.
|
|
97
|
-
const src = `
|
|
98
|
-
import { computeTier } from '@acme/core/tiers.js';
|
|
99
|
-
const at = (arr, i) => arr[i];
|
|
100
|
-
it('TIER-001: the tier changes the money', () => {
|
|
101
|
-
assert.equal(at(computeTier(x), 0), 1200);
|
|
102
|
-
});`;
|
|
103
|
-
const r = analyseCase(src, 'TIER-001: the tier changes the money');
|
|
104
|
-
assert.ok(!r.findings.some(f => f.id === 'stubbed-subject'),
|
|
105
|
-
'a local helper whose name merely appears inside an import path is not a stub');
|
|
106
|
-
});
|
|
107
|
-
|
|
108
|
-
test('a REAL stub is still caught — the tightening must not blind the check', () => {
|
|
109
|
-
const src = `
|
|
110
|
-
import { SLA_HOURS } from '../src/priority.mjs';
|
|
111
|
-
const priority = () => 1;
|
|
112
|
-
it('p is 1', () => { assert.equal(priority('HIGH-PRIORITY'), 1); });`;
|
|
113
|
-
const r = analyseCase(src, 'p is 1');
|
|
114
|
-
assert.ok(r.findings.some(f => f.id === 'stubbed-subject'),
|
|
115
|
-
'a stub matching the imported module basename must still be reported');
|
|
116
|
-
assert.equal(r.verdict, 'weak');
|
|
117
|
-
});
|
|
118
|
-
|
|
119
|
-
test('a declined commit produces NO pr comment — silence, not an accusation', async () => {
|
|
120
|
-
// It briefly posted "SKIPPED · No case in this branch fails without the change" onto a
|
|
121
|
-
// workflow-only PR. Technically true and completely misleading: that PR has no tests,
|
|
122
|
-
// so of course none discriminate. A comment reading as an accusation on a PR doing
|
|
123
|
-
// nothing wrong is worse than no comment.
|
|
124
|
-
const { prComment } = await import('../src/markdown-report.mjs');
|
|
125
|
-
const declined = { short: 'abc12345', verdict: 'SKIPPED', cases: [], note: 'no test file in the commit' };
|
|
126
|
-
assert.equal(prComment(declined), '', 'a noted (declined) result must render as empty');
|
|
127
|
-
|
|
128
|
-
const real = { short: 'abc12345', verdict: 'CAUGHT', cases: [{ verdict: 'CAUGHT', name: 'x', reason: 'y' }] };
|
|
129
|
-
assert.notEqual(prComment(real), '', 'a real verdict must still render');
|
|
130
|
-
});
|
package/test/fixtures.test.mjs
DELETED
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The tool's control arm. Eight repos where the right answer is known by construction.
|
|
3
|
-
*
|
|
4
|
-
* A green run here is the only reason to believe a verdict this tool prints about
|
|
5
|
-
* anybody else's code. It is also the thing that would catch the failure mode that
|
|
6
|
-
* matters most — a FALSE BLIND — because 01 and 07 must never come back BLIND.
|
|
7
|
-
*/
|
|
8
|
-
import { test } from 'node:test';
|
|
9
|
-
import assert from 'node:assert/strict';
|
|
10
|
-
import path from 'node:path';
|
|
11
|
-
import { buildFixtures } from '../fixtures/build.mjs';
|
|
12
|
-
import { verifyCommit } from '../src/verify.mjs';
|
|
13
|
-
import { removeWorktrees } from '../src/worktree.mjs';
|
|
14
|
-
|
|
15
|
-
const fixtures = buildFixtures();
|
|
16
|
-
|
|
17
|
-
for (const [name, f] of Object.entries(fixtures)) {
|
|
18
|
-
test(`${name} -> ${f.expect} (${f.why})`, async () => {
|
|
19
|
-
const workDir = path.join(f.dir, '.ca-work');
|
|
20
|
-
try {
|
|
21
|
-
const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
|
|
22
|
-
assert.equal(r.verdict, f.expect,
|
|
23
|
-
`expected ${f.expect}, got ${r.verdict}\n` +
|
|
24
|
-
r.cases.map(c => ` ${c.verdict} ${c.name} — ${c.reason}`).join('\n') +
|
|
25
|
-
(r.note ? `\n note: ${r.note}` : ''));
|
|
26
|
-
} finally {
|
|
27
|
-
await removeWorktrees(f.dir, workDir);
|
|
28
|
-
}
|
|
29
|
-
});
|
|
30
|
-
}
|
|
31
|
-
|
|
32
|
-
test('a FALSE BLIND is the failure that matters: no discriminating fixture may read BLIND', async () => {
|
|
33
|
-
for (const name of ['01-caught-value', '07-caught-mixed']) {
|
|
34
|
-
const f = fixtures[name];
|
|
35
|
-
const workDir = path.join(f.dir, '.ca-work-fb');
|
|
36
|
-
try {
|
|
37
|
-
const r = await verifyCommit({ repo: f.dir, workDir, sha: f.sha, runs: 1, timeoutMs: 30_000 });
|
|
38
|
-
assert.notEqual(r.verdict, 'BLIND', `${name} was accused of being blind — this is the output that destroys trust in the tool`);
|
|
39
|
-
} finally { await removeWorktrees(f.dir, workDir); }
|
|
40
|
-
}
|
|
41
|
-
});
|
|
@@ -1,69 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The vitest/jest adapter, at the level where it can actually be wrong.
|
|
3
|
-
*
|
|
4
|
-
* WHAT THIS DOES AND DOES NOT COVER — stated because the gap matters.
|
|
5
|
-
*
|
|
6
|
-
* The node:test path has 9 end-to-end fixtures: real git repos, both arms, known verdict.
|
|
7
|
-
* These two runners do NOT, because a fixture would have to `npm install` vitest or jest
|
|
8
|
-
* per repo — slow, network-dependent, and version-drifting. The full two-arm path for
|
|
9
|
-
* each is verified on ONE real commit apiece — a vitest component commit and a jest
|
|
10
|
-
* React Native commit — which is a smoke test, not a control arm.
|
|
11
|
-
*
|
|
12
|
-
* So this file covers the part that is BOTH untested end-to-end AND most likely to be
|
|
13
|
-
* wrong: the text-matching in `classifyMessages`. node:test hands over `ERR_ASSERTION`;
|
|
14
|
-
* these two give formatted strings, so the adapter has to read prose. Every string below
|
|
15
|
-
* is a real shape emitted by vitest or jest, not an invented one.
|
|
16
|
-
*/
|
|
17
|
-
import { test } from 'node:test';
|
|
18
|
-
import assert from 'node:assert/strict';
|
|
19
|
-
import { classifyMessages } from '../src/runner-json.mjs';
|
|
20
|
-
|
|
21
|
-
const isDisagreement = m => classifyMessages(m).code === 'ERR_ASSERTION';
|
|
22
|
-
const isCannotRun = m => classifyMessages(m).code === 'ERR_TEST_FAILURE';
|
|
23
|
-
const isUnknown = m => classifyMessages(m).code === null;
|
|
24
|
-
|
|
25
|
-
test('vitest/chai disagreement is read as an assertion', () => {
|
|
26
|
-
assert.ok(isDisagreement(["AssertionError: expected 'Urgent' to deeply equal 'Normal'"]));
|
|
27
|
-
assert.ok(isDisagreement(['AssertionError: expected 40 to be 65 // Object.is equality']));
|
|
28
|
-
});
|
|
29
|
-
|
|
30
|
-
test('jest/expect disagreement is read as an assertion', () => {
|
|
31
|
-
assert.ok(isDisagreement(['expect(received).toBe(expected) // Object.is equality\n\nExpected: "Normal"\nReceived: "Urgent"']));
|
|
32
|
-
assert.ok(isDisagreement(['Error: expect(received).toEqual(expected)\n\n- Expected\n+ Received']));
|
|
33
|
-
});
|
|
34
|
-
|
|
35
|
-
test('a module that could not load is NOT an assertion', () => {
|
|
36
|
-
assert.ok(isCannotRun(["Error: Cannot find module '../src/newHelper' from 'lib/__tests__/x.test.tsx'"]));
|
|
37
|
-
assert.ok(isCannotRun(["SyntaxError: The requested module './bucket.js' does not provide an export named 'SEP'"]));
|
|
38
|
-
assert.ok(isCannotRun(['Error: Failed to resolve import "./notYet" from "lib/x.test.tsx". Does the file exist?']));
|
|
39
|
-
assert.ok(isCannotRun(['TypeError [ERR_UNKNOWN_FILE_EXTENSION]: Unknown file extension ".tsx"']));
|
|
40
|
-
});
|
|
41
|
-
|
|
42
|
-
test('CANNOT-RUN WINS when both shapes appear — the ordering that matters', () => {
|
|
43
|
-
// A failed module load frequently ALSO prints an expect() frame from the stack. If
|
|
44
|
-
// the disagreement patterns were checked first, this would read CAUGHT, and a commit
|
|
45
|
-
// whose test never ran would be credited with catching its bug.
|
|
46
|
-
const both = ['SyntaxError: does not provide an export named \'SEP\'\n at expect(received).toBe(expected)\n expected \'a\' to be \'b\''];
|
|
47
|
-
assert.ok(isCannotRun(both), 'a load failure that also prints an expect() frame must not read as a disagreement');
|
|
48
|
-
assert.notEqual(classifyMessages(both).code, 'ERR_ASSERTION');
|
|
49
|
-
});
|
|
50
|
-
|
|
51
|
-
test('an UNRECOGNISED message is unknown — never guessed into an assertion', () => {
|
|
52
|
-
// This is the design: an unrecognised shape costs COVERAGE (it reads INCONCLUSIVE),
|
|
53
|
-
// it cannot manufacture a finding. A drifting regex must fail safe, not fail loud.
|
|
54
|
-
assert.ok(isUnknown(['Something entirely unexpected happened in a custom matcher']));
|
|
55
|
-
assert.ok(isUnknown([]));
|
|
56
|
-
assert.ok(isUnknown(['']));
|
|
57
|
-
assert.notEqual(classifyMessages(['whatever']).code, 'ERR_ASSERTION');
|
|
58
|
-
});
|
|
59
|
-
|
|
60
|
-
test('CONTROL ARM: a naive "any failure is a disagreement" classifier gets these wrong', () => {
|
|
61
|
-
// TEST-001 — run the broken implementation and assert it still reproduces the bug.
|
|
62
|
-
const naive = msgs => (msgs && msgs.length ? 'ERR_ASSERTION' : null);
|
|
63
|
-
const loadFail = ["Error: Cannot find module '../src/newHelper'"];
|
|
64
|
-
assert.equal(naive(loadFail), 'ERR_ASSERTION'); // the naive one says CAUGHT
|
|
65
|
-
assert.equal(classifyMessages(loadFail).code, 'ERR_TEST_FAILURE'); // ours says it never ran
|
|
66
|
-
const weird = ['custom matcher blew up'];
|
|
67
|
-
assert.equal(naive(weird), 'ERR_ASSERTION');
|
|
68
|
-
assert.equal(classifyMessages(weird).code, null);
|
|
69
|
-
});
|
|
@@ -1,78 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Runner selection, over synthetic trees.
|
|
3
|
-
*
|
|
4
|
-
* Picking the WRONG runner is a silent failure: vitest run from the repo root finds no
|
|
5
|
-
* config and reports "No test files found", which reads INCONCLUSIVE — a whole surface
|
|
6
|
-
* disappears without ever saying why. So the walk-up has to be pinned.
|
|
7
|
-
*/
|
|
8
|
-
import { test } from 'node:test';
|
|
9
|
-
import assert from 'node:assert/strict';
|
|
10
|
-
import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';
|
|
11
|
-
import { tmpdir } from 'node:os';
|
|
12
|
-
import path from 'node:path';
|
|
13
|
-
import { selectRunner } from '../src/select-runner.mjs';
|
|
14
|
-
|
|
15
|
-
function tree(spec) {
|
|
16
|
-
const root = mkdtempSync(path.join(tmpdir(), 'ca-sel-'));
|
|
17
|
-
for (const [rel, content] of Object.entries(spec)) {
|
|
18
|
-
const p = path.join(root, rel);
|
|
19
|
-
mkdirSync(path.dirname(p), { recursive: true });
|
|
20
|
-
writeFileSync(p, content);
|
|
21
|
-
}
|
|
22
|
-
return root;
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
test('a vitest config in the workspace wins, and names that workspace as cwd', async () => {
|
|
26
|
-
const root = tree({
|
|
27
|
-
'package.json': '{"workspaces":["apps/*"]}',
|
|
28
|
-
'apps/web/vitest.config.ts': 'export default {}',
|
|
29
|
-
'apps/web/lib/__tests__/x.test.tsx': '',
|
|
30
|
-
});
|
|
31
|
-
try {
|
|
32
|
-
const r = await selectRunner(root, 'apps/web/lib/__tests__/x.test.tsx');
|
|
33
|
-
assert.equal(r.flavour, 'vitest');
|
|
34
|
-
assert.equal(r.pkgDir, 'apps/web', 'must run FROM apps/web — vitest resolves config relative to cwd');
|
|
35
|
-
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
36
|
-
});
|
|
37
|
-
|
|
38
|
-
test('a jest config in the workspace wins', async () => {
|
|
39
|
-
const root = tree({ 'package.json': '{}', 'apps/mobile/jest.config.js': '', 'apps/mobile/__tests__/y.test.ts': '' });
|
|
40
|
-
try {
|
|
41
|
-
const r = await selectRunner(root, 'apps/mobile/__tests__/y.test.ts');
|
|
42
|
-
assert.equal(r.flavour, 'jest');
|
|
43
|
-
assert.equal(r.pkgDir, 'apps/mobile');
|
|
44
|
-
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
45
|
-
});
|
|
46
|
-
|
|
47
|
-
test('root tests fall to node:test when nothing else claims them', async () => {
|
|
48
|
-
const root = tree({ 'package.json': '{"scripts":{"test":"node --test tests/*.test.js"}}', 'tests/a.test.js': '' });
|
|
49
|
-
try {
|
|
50
|
-
const r = await selectRunner(root, 'tests/a.test.js');
|
|
51
|
-
assert.equal(r.flavour, 'node');
|
|
52
|
-
assert.equal(r.pkgDir, '');
|
|
53
|
-
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
54
|
-
});
|
|
55
|
-
|
|
56
|
-
test('a repo that declares NOTHING falls through to node:test, not to a guess', async () => {
|
|
57
|
-
// The safe default: node:test either works, or reports a load failure, which reads
|
|
58
|
-
// INCONCLUSIVE. Guessing vitest here would produce "No test files found" — the same
|
|
59
|
-
// silent disappearance this test exists to prevent.
|
|
60
|
-
const root = tree({ 'tests/a.test.js': '' });
|
|
61
|
-
try {
|
|
62
|
-
assert.equal((await selectRunner(root, 'tests/a.test.js')).flavour, 'node');
|
|
63
|
-
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
64
|
-
});
|
|
65
|
-
|
|
66
|
-
test('a workspace config beats a root test script — the nearest owner wins', async () => {
|
|
67
|
-
// The root script often delegates (`npm run test --workspaces`), while the config
|
|
68
|
-
// that actually governs the file sits in the package. Nearest wins, walking up.
|
|
69
|
-
const root = tree({
|
|
70
|
-
'package.json': '{"scripts":{"test":"node --test"}}',
|
|
71
|
-
'apps/web/vitest.config.ts': '',
|
|
72
|
-
'apps/web/x.test.tsx': '',
|
|
73
|
-
});
|
|
74
|
-
try {
|
|
75
|
-
const r = await selectRunner(root, 'apps/web/x.test.tsx');
|
|
76
|
-
assert.equal(r.flavour, 'vitest');
|
|
77
|
-
} finally { rmSync(root, { recursive: true, force: true }); }
|
|
78
|
-
});
|
package/test/verdict.test.mjs
DELETED
|
@@ -1,113 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The decision logic, exhaustively, with no git and no subprocesses.
|
|
3
|
-
*
|
|
4
|
-
* Every case here is a shape observed in a real run, not an invented one.
|
|
5
|
-
*/
|
|
6
|
-
import { test } from 'node:test';
|
|
7
|
-
import assert from 'node:assert/strict';
|
|
8
|
-
import { classify, reduceRuns, rollUp, isDisagreement, CAUGHT, BLIND, NON_DISCRIMINATING, INCONCLUSIVE, FLAKY, SKIPPED } from '../src/verdict.mjs';
|
|
9
|
-
|
|
10
|
-
const pass = { status: 'pass' };
|
|
11
|
-
const assertFail = { status: 'fail', code: 'ERR_ASSERTION', errorName: 'AssertionError', message: "expected 1, got 2" };
|
|
12
|
-
const errFail = { status: 'fail', code: 'ERR_TEST_FAILURE', errorName: 'SyntaxError', message: "does not provide an export named 'TITLE_SEPARATOR'" };
|
|
13
|
-
|
|
14
|
-
test('CAUGHT: green on the fix, assertion-red on the parent', () => {
|
|
15
|
-
assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
|
|
16
|
-
});
|
|
17
|
-
|
|
18
|
-
test('NON-DISCRIMINATING: green on the fix AND green on the parent', () => {
|
|
19
|
-
assert.equal(classify({ armA: pass, armB: pass }).verdict, NON_DISCRIMINATING);
|
|
20
|
-
});
|
|
21
|
-
|
|
22
|
-
test('a case green on both arms is never called BLIND — a regression guard is indistinguishable', () => {
|
|
23
|
-
// Observed in practice: 3 of one commit's 4 cases were deliberate regression guards.
|
|
24
|
-
assert.notEqual(classify({ armA: pass, armB: pass }).verdict, BLIND);
|
|
25
|
-
});
|
|
26
|
-
|
|
27
|
-
test('a non-assertion failure on the parent is INCONCLUSIVE, never CAUGHT — the test never ran', () => {
|
|
28
|
-
const r = classify({ armA: pass, armB: errFail });
|
|
29
|
-
assert.equal(r.verdict, INCONCLUSIVE);
|
|
30
|
-
assert.match(r.reason, /never ran/);
|
|
31
|
-
});
|
|
32
|
-
|
|
33
|
-
test('exit code alone would have gotten that wrong', () => {
|
|
34
|
-
// Both of these are "the process exited non-zero". Only one is a verdict.
|
|
35
|
-
assert.equal(classify({ armA: pass, armB: assertFail }).verdict, CAUGHT);
|
|
36
|
-
assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE);
|
|
37
|
-
});
|
|
38
|
-
|
|
39
|
-
test('unproven module identity withholds the verdict even when arm B passed', () => {
|
|
40
|
-
// THE false-BLIND guard. Arm B green + identity unproven is precisely the symlink
|
|
41
|
-
// trap that made this tool report a false BLIND before it had an identity gate.
|
|
42
|
-
const r = classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'resolved OUTSIDE the worktree' } });
|
|
43
|
-
assert.equal(r.verdict, INCONCLUSIVE);
|
|
44
|
-
assert.notEqual(r.verdict, BLIND);
|
|
45
|
-
});
|
|
46
|
-
|
|
47
|
-
test('a case that is red on the fix is INCONCLUSIVE — the commit does not stand up', () => {
|
|
48
|
-
assert.equal(classify({ armA: assertFail, armB: assertFail }).verdict, INCONCLUSIVE);
|
|
49
|
-
});
|
|
50
|
-
|
|
51
|
-
test('a case missing from the parent run is INCONCLUSIVE, not BLIND', () => {
|
|
52
|
-
assert.equal(classify({ armA: pass, armB: null }).verdict, INCONCLUSIVE);
|
|
53
|
-
});
|
|
54
|
-
|
|
55
|
-
test('isDisagreement is narrow: an unknown error name is not guessed into CAUGHT', () => {
|
|
56
|
-
assert.equal(isDisagreement({ status: 'fail', errorName: 'WeirdCustomError' }), false);
|
|
57
|
-
assert.equal(isDisagreement({ status: 'fail', errorName: 'JestAssertionError' }), true);
|
|
58
|
-
});
|
|
59
|
-
|
|
60
|
-
test('FLAKY: runs that disagree between CAUGHT and BLIND', () => {
|
|
61
|
-
const r = reduceRuns([{ verdict: CAUGHT }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]);
|
|
62
|
-
assert.equal(r.verdict, FLAKY);
|
|
63
|
-
});
|
|
64
|
-
|
|
65
|
-
test('an INCONCLUSIVE run does not erase a decided one, but is disclosed', () => {
|
|
66
|
-
const r = reduceRuns([{ verdict: CAUGHT, reason: 'assertion' }, { verdict: INCONCLUSIVE, reason: 'timeout' }, { verdict: CAUGHT, reason: 'assertion' }]);
|
|
67
|
-
assert.equal(r.verdict, CAUGHT);
|
|
68
|
-
assert.match(r.reason, /1 of 3 runs inconclusive/);
|
|
69
|
-
});
|
|
70
|
-
|
|
71
|
-
test('rollUp: one discriminating case carries the commit, guards and all', () => {
|
|
72
|
-
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: CAUGHT }]), CAUGHT);
|
|
73
|
-
});
|
|
74
|
-
|
|
75
|
-
test('rollUp: a SKIPPED case blocks BLIND — it might have been the discriminating one', () => {
|
|
76
|
-
// Observed: all four tests written FOR a bug were { skip: SKIP } on a host with no
|
|
77
|
-
// TEST_DATABASE_URL, while older cases in the same file ran and did not discriminate.
|
|
78
|
-
// Calling that blind judges the commit on the tests NOT written for it.
|
|
79
|
-
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
|
|
80
|
-
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }, { verdict: SKIPPED }]), INCONCLUSIVE);
|
|
81
|
-
});
|
|
82
|
-
|
|
83
|
-
test('rollUp: BLIND only when every case ran and NOT ONE discriminated', () => {
|
|
84
|
-
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: NON_DISCRIMINATING }]), BLIND);
|
|
85
|
-
// one case could not be judged -> the commit cannot be called blind
|
|
86
|
-
assert.equal(rollUp([{ verdict: NON_DISCRIMINATING }, { verdict: INCONCLUSIVE }]), INCONCLUSIVE);
|
|
87
|
-
});
|
|
88
|
-
|
|
89
|
-
test('CONTROL ARM: a verdict engine that only read exit codes fails these', () => {
|
|
90
|
-
// TEST-001 — the old, naive implementation, executed here, asserted to still be wrong.
|
|
91
|
-
const naive = ({ armB }) => (armB && armB.status === 'fail' ? CAUGHT : BLIND);
|
|
92
|
-
assert.equal(naive({ armB: errFail }), CAUGHT); // it says CAUGHT...
|
|
93
|
-
assert.equal(classify({ armA: pass, armB: errFail }).verdict, INCONCLUSIVE); // ...we say no
|
|
94
|
-
assert.equal(naive({ armB: pass }), BLIND); // and calls every regression guard blind // and it cannot see
|
|
95
|
-
assert.equal(classify({ armA: pass, armB: pass, identity: { proven: false, reason: 'x' } }).verdict, INCONCLUSIVE);
|
|
96
|
-
});
|
|
97
|
-
|
|
98
|
-
test('a CAUGHT on new code is still CAUGHT — the verdict does not lie, the CLAIM narrows', async () => {
|
|
99
|
-
// Issue #3. On a feature the base lacks the code entirely, so essentially any test
|
|
100
|
-
// touching it fails there — `expected 0 to be greater than 0` is a real AssertionError
|
|
101
|
-
// that says nothing about whether the test is well aimed. Reclassifying it would be
|
|
102
|
-
// dishonest (it IS a disagreement); counting it as evidence would be worse.
|
|
103
|
-
const { commitKind } = await import('../src/verify.mjs');
|
|
104
|
-
|
|
105
|
-
assert.equal(commitKind('feat(web): add seoTitle', false).newCode, true,
|
|
106
|
-
'a feature is new code whatever its diff looks like');
|
|
107
|
-
assert.equal(commitKind('fix(core): a hyphen decided a title', false).newCode, false,
|
|
108
|
-
'a fix that changes lines is a repair, and its CAUGHT is strong evidence');
|
|
109
|
-
assert.equal(commitKind('fix(api): add a missing export', true).newCode, true,
|
|
110
|
-
'a fix whose source diff only ADDS may still be reading something that was absent');
|
|
111
|
-
assert.equal(commitKind('no conventional prefix at all', false).kind, 'unknown',
|
|
112
|
-
'an unrecognised subject must not be guessed into fix or feature');
|
|
113
|
-
});
|