jules-orchestrator-kit 0.60.0 → 0.64.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +4 -2
- package/scripts/guard-reach-check.mjs +226 -0
- package/scripts/package-integrity-check.mjs +251 -0
- package/scripts/release.mjs +33 -0
- package/src/assertions.mjs +16 -0
- package/src/config.mjs +68 -0
- package/src/coverage.mjs +17 -2
- package/src/engine.mjs +33 -2
- package/src/evidence.mjs +38 -1
- package/src/guard-policy.mjs +481 -0
- package/src/ops/test-collection.mjs +149 -0
- package/src/security.mjs +216 -10
- package/src/stack-detector.mjs +113 -10
- package/src/task-optimizer.mjs +7 -3
package/src/coverage.mjs
CHANGED
|
@@ -300,12 +300,27 @@ export function calculateDiffCoverage(coverageByFile, diffStr = "", options = {}
|
|
|
300
300
|
}
|
|
301
301
|
}
|
|
302
302
|
|
|
303
|
-
|
|
304
|
-
|
|
303
|
+
// 100% of nothing is not 100%.
|
|
304
|
+
//
|
|
305
|
+
// The denominator counts only the added lines V8 actually mapped, and V8
|
|
306
|
+
// maps nothing outside Node. So a Python diff adding three executable lines
|
|
307
|
+
// measured zero of them and was reported as `score: 100` — the best possible
|
|
308
|
+
// number, produced by a measurement that never happened, on 20-odd of the 25
|
|
309
|
+
// stacks this kit claims to support. `mutation.mjs` had the identical bug and
|
|
310
|
+
// was fixed in v0.57.0; this is the same shape one module over.
|
|
311
|
+
//
|
|
312
|
+
// `ok` stays true because nothing failed to be covered, and a gate that
|
|
313
|
+
// blocks every non-Node diff gets switched off. What changes is the claim:
|
|
314
|
+
// `scored: false` and a reason, instead of a number nobody measured.
|
|
315
|
+
const scored = totalLines > 0;
|
|
316
|
+
const score = scored ? Math.round((coveredLines / totalLines) * 10000) / 100 : null;
|
|
317
|
+
const ok = scored ? score >= minCoverage : true;
|
|
305
318
|
|
|
306
319
|
return {
|
|
307
320
|
ok,
|
|
308
321
|
score,
|
|
322
|
+
scored,
|
|
323
|
+
...(scored ? {} : { reason: "No added executable lines were measurable — V8 coverage only observes code Node itself ran, so nothing was scored." }),
|
|
309
324
|
minCoverage,
|
|
310
325
|
totalLines,
|
|
311
326
|
coveredLines,
|
package/src/engine.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { loadConfig, parseYaml, normalizeScope } from "./config.mjs";
|
|
2
2
|
import { isTestPath } from "./test-paths.mjs";
|
|
3
|
+
import { checkCollectionFloor } from "./ops/test-collection.mjs";
|
|
3
4
|
import { checkScope, scanDiff, scanBinaryPayloads, redactSecrets } from "./security.mjs";
|
|
4
5
|
import { changedFiles, diffBytes, diffText, binaryDiffEntries, symlinkChanges, showFromOrigin, runCmd } from "./git.mjs";
|
|
5
6
|
import { createProvider, ProviderRateLimitError, ProviderUnavailableError } from "./provider.mjs";
|
|
@@ -521,7 +522,25 @@ export async function gate(opts = {}) {
|
|
|
521
522
|
const ranNoVerification = !executionRecords.some((r) => r && r.kind !== "assert");
|
|
522
523
|
const missingOracle = verificationRequired && ranNoVerification;
|
|
523
524
|
|
|
524
|
-
|
|
525
|
+
// A command that ran is not the same as a command that tested something.
|
|
526
|
+
//
|
|
527
|
+
// `missingOracle` above catches "no stage executed". It cannot catch the
|
|
528
|
+
// case where a stage executed, exited 0, and collected zero tests — which
|
|
529
|
+
// several runners report as success by design: `go test ./...` prints
|
|
530
|
+
// "[no test files]" and exits 0, jest has --passWithNoTests, and
|
|
531
|
+
// `npm test --workspaces` is green when the one package the diff touched
|
|
532
|
+
// has no suite. A repository could invert a function, add an untested one
|
|
533
|
+
// and collect five green phases, verified against nothing at all.
|
|
534
|
+
//
|
|
535
|
+
// The count is read out of the runner's own summary and only a *stated*
|
|
536
|
+
// zero counts. An unrecognised runner yields null and passes: failing on
|
|
537
|
+
// "I could not tell" would break every runner not on the list.
|
|
538
|
+
const collectionFloor = verificationRequired
|
|
539
|
+
? checkCollectionFloor(testResult, { minTests: trustedVerify.minTests })
|
|
540
|
+
: { ok: true, count: null, runner: null, reason: null };
|
|
541
|
+
const emptySuite = !collectionFloor.ok;
|
|
542
|
+
|
|
543
|
+
const verifyOk = !failingCmd && !testTampered && !missingOracle && !emptySuite;
|
|
525
544
|
|
|
526
545
|
// What actually broke. Without this the verify phase reported `ok: false` and
|
|
527
546
|
// nothing else — not the stage, not the exit code, not a line of output — so
|
|
@@ -568,7 +587,18 @@ export async function gate(opts = {}) {
|
|
|
568
587
|
"The gate approves a change because verification passed. Zero stages executed is not a pass.",
|
|
569
588
|
],
|
|
570
589
|
}
|
|
571
|
-
:
|
|
590
|
+
: emptySuite
|
|
591
|
+
? {
|
|
592
|
+
stageId: "empty-suite",
|
|
593
|
+
command: testResult?.command || null,
|
|
594
|
+
exitCode: 0,
|
|
595
|
+
stdout: "",
|
|
596
|
+
stderr: collectionFloor.reason,
|
|
597
|
+
diagnostics: [
|
|
598
|
+
"An exit code of 0 from a runner that collected no tests is not evidence about this change.",
|
|
599
|
+
],
|
|
600
|
+
}
|
|
601
|
+
: null;
|
|
572
602
|
|
|
573
603
|
// Generate & persist Evidence Manifest
|
|
574
604
|
const evidenceManifest = generateEvidenceManifest(root, {
|
|
@@ -583,6 +613,7 @@ export async function gate(opts = {}) {
|
|
|
583
613
|
failedStage: failingCmd?.stageId || failingCmd?.phase || null,
|
|
584
614
|
diagnostics: failingCmd?.diagnostics?.length ? failingCmd.diagnostics : (failingCmd?.stderr ? [failingCmd.stderr] : []),
|
|
585
615
|
metrics: failingCmd?.metrics || {},
|
|
616
|
+
collection: collectionFloor,
|
|
586
617
|
ok: verifyOk,
|
|
587
618
|
});
|
|
588
619
|
if (testTampered) {
|
package/src/evidence.mjs
CHANGED
|
@@ -129,7 +129,20 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
129
129
|
.filter((p) => existsSync(join(root, p)) && statSync(join(root, p)).isFile())
|
|
130
130
|
.sort();
|
|
131
131
|
} else {
|
|
132
|
-
|
|
132
|
+
// A sixth spelling of "where do the tests live?", and the one that got
|
|
133
|
+
// missed when the other five were unified behind `isTestPath`: this is a
|
|
134
|
+
// list of *directory names at the repository root*, not a predicate. Go
|
|
135
|
+
// puts its tests beside the code (`internal/calc/calc_test.go`), and every
|
|
136
|
+
// monorepo puts them under `packages/*/test/`. Neither is under a
|
|
137
|
+
// root-level `test/`, so the walk found nothing, `fileCount` was 0, and
|
|
138
|
+
// `strictTestLock` — which requires `fileCount > 0` — switched itself off
|
|
139
|
+
// without saying so. The tree hash then became the SHA-256 of the empty
|
|
140
|
+
// string, and the evidence manifest attested to it.
|
|
141
|
+
//
|
|
142
|
+
// The named directories stay as a fast path; when they yield nothing, walk
|
|
143
|
+
// the repository and let the shared predicate decide. The walk already
|
|
144
|
+
// skips node_modules, vendor, target and the build caches.
|
|
145
|
+
const targetDirs = options.directories || ["test", "tests", "__tests__", "spec", "specs", "src"];
|
|
133
146
|
for (const dirName of targetDirs) {
|
|
134
147
|
const dirPath = join(root, dirName);
|
|
135
148
|
if (existsSync(dirPath)) {
|
|
@@ -137,6 +150,11 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
137
150
|
fileList.push(...found);
|
|
138
151
|
}
|
|
139
152
|
}
|
|
153
|
+
if (options.testOnly && !fileList.some((f) => isTestPath(f))) {
|
|
154
|
+
for (const f of findFilesRecursively(root, root)) {
|
|
155
|
+
if (isTestPath(f)) fileList.push(f);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
140
158
|
|
|
141
159
|
// Plenty of projects keep `app.test.mjs` or `index.js` beside package.json
|
|
142
160
|
// rather than under one of the directories above, and those files were
|
|
@@ -201,6 +219,7 @@ export function computeEvidenceHash(manifest) {
|
|
|
201
219
|
intent: manifest.intent,
|
|
202
220
|
provenance: manifest.provenance,
|
|
203
221
|
testIntegrity: manifest.testIntegrity,
|
|
222
|
+
...(manifest.verification ? { verification: manifest.verification } : {}),
|
|
204
223
|
...(manifest.sourceIntegrity ? { sourceIntegrity: manifest.sourceIntegrity } : {}),
|
|
205
224
|
executionRecords: manifest.executionRecords,
|
|
206
225
|
securityChecks: manifest.securityChecks,
|
|
@@ -342,6 +361,24 @@ export function generateEvidenceManifest(root = process.cwd(), options = {}) {
|
|
|
342
361
|
maxDiffKb: options.maxDiffKb || 75,
|
|
343
362
|
protectedScopeOk: options.protectedScopeOk ?? true,
|
|
344
363
|
},
|
|
364
|
+
// How many tests the runner said it collected, and whether it said at all.
|
|
365
|
+
//
|
|
366
|
+
// The collection floor fails a *stated* zero and lets an unstated count
|
|
367
|
+
// pass, because failing on "I could not tell" would break every runner not
|
|
368
|
+
// on the list. That is the right call for the verdict and the wrong thing
|
|
369
|
+
// to leave out of the record: a manifest that says nothing here reads as
|
|
370
|
+
// though a suite ran. `counted: false` is the honest shape for a run where
|
|
371
|
+
// the number was never observable — a quiet runner (`cargo test --quiet`,
|
|
372
|
+
// `pytest -q`) suppresses the very line the floor reads.
|
|
373
|
+
...(options.collection
|
|
374
|
+
? {
|
|
375
|
+
verification: {
|
|
376
|
+
testsCollected: options.collection.count,
|
|
377
|
+
counted: options.collection.count !== null,
|
|
378
|
+
runner: options.collection.runner,
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
: {}),
|
|
345
382
|
...(diagnostics.length > 0 ? { diagnostics } : {}),
|
|
346
383
|
...(Object.keys(metrics).length > 0 ? { metrics } : {}),
|
|
347
384
|
};
|
|
@@ -0,0 +1,481 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The policy this kit claims to enforce, written by hand.
|
|
3
|
+
*
|
|
4
|
+
* Every entry here is derived from what the tool *advertises* — the stacks
|
|
5
|
+
* `detectStack` declares, the layouts each ecosystem actually uses — and never
|
|
6
|
+
* from the regexes, path lists or registries that implement the checks. That
|
|
7
|
+
* separation is the whole point: a contract generated from the implementation
|
|
8
|
+
* makes the implementation its own oracle, and an implementation that is its
|
|
9
|
+
* own oracle cannot be wrong.
|
|
10
|
+
*
|
|
11
|
+
* This is what the substring bug in the file classifier cost. `isTestFile`
|
|
12
|
+
* matched `/test/`, which does not occur in `tests/test_calc.py`, so the entire
|
|
13
|
+
* tamper guard was off for the standard pytest, Rust and RSpec layouts — and
|
|
14
|
+
* every mechanism that should have caught it (a large suite, a doc-sync gate,
|
|
15
|
+
* a nine-way CI matrix, two cold reviews, a blocking release) was sampling the
|
|
16
|
+
* same distribution the implementation was written from. Nine runs of
|
|
17
|
+
* `test/foo.test.js` do not explore `tests/test_calc.py`.
|
|
18
|
+
*
|
|
19
|
+
* Adding a stack to `detectStack` is not finished until it has a row here.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Paths the policy says are test files, and near-misses it says are not.
|
|
24
|
+
*
|
|
25
|
+
* The near-misses matter as much as the hits: a predicate that answers "yes"
|
|
26
|
+
* to everything also has no denominator.
|
|
27
|
+
*/
|
|
28
|
+
export const TEST_PATH_CASES = [
|
|
29
|
+
// Node / JavaScript
|
|
30
|
+
{ path: "test/calc.test.js", expected: true, why: "node, conventional" },
|
|
31
|
+
{ path: "src/calc.spec.ts", expected: true, why: "co-located spec" },
|
|
32
|
+
{ path: "src/__tests__/calc.js", expected: true, why: "jest convention" },
|
|
33
|
+
// Python
|
|
34
|
+
{ path: "tests/test_calc.py", expected: true, why: "pytest, repository root — the reported gap" },
|
|
35
|
+
{ path: "test/test_calc.py", expected: true, why: "pytest, singular directory" },
|
|
36
|
+
{ path: "backend/tests/test_api.py", expected: true, why: "pytest, nested" },
|
|
37
|
+
// Go
|
|
38
|
+
{ path: "internal/calc/calc_test.go", expected: true, why: "go, co-located with the code" },
|
|
39
|
+
{ path: "cmd/api/main_test.go", expected: true, why: "go, command package" },
|
|
40
|
+
// Rust
|
|
41
|
+
{ path: "tests/integration.rs", expected: true, why: "rust integration tests" },
|
|
42
|
+
// Ruby
|
|
43
|
+
{ path: "spec/models/user_spec.rb", expected: true, why: "rspec" },
|
|
44
|
+
{ path: "test/user_test.rb", expected: true, why: "minitest" },
|
|
45
|
+
// JVM / .NET / PHP
|
|
46
|
+
{ path: "src/test/java/com/x/CalcTest.java", expected: true, why: "maven layout" },
|
|
47
|
+
{ path: "tests/Unit/CalcTest.php", expected: true, why: "phpunit" },
|
|
48
|
+
// Solidity
|
|
49
|
+
{ path: "test/Token.t.sol", expected: true, why: "foundry" },
|
|
50
|
+
// Monorepo position
|
|
51
|
+
{ path: "packages/api/test/handler.test.js", expected: true, why: "monorepo package" },
|
|
52
|
+
|
|
53
|
+
// Near-misses: segment match, not substring
|
|
54
|
+
{ path: "latest/build.js", expected: false, why: "`latest` is not `test`" },
|
|
55
|
+
{ path: "contest/result.js", expected: false, why: "`contest` is not `test`" },
|
|
56
|
+
{ path: "lib/myspec/render.js", expected: false, why: "`myspec` is not `spec`" },
|
|
57
|
+
{ path: "src/attestation.js", expected: false, why: "contains 'test', is not one" },
|
|
58
|
+
{ path: "src/index.js", expected: false, why: "ordinary source" },
|
|
59
|
+
{ path: "docs/testing-guide.md", expected: false, why: "documentation about testing" },
|
|
60
|
+
];
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Known-bad diffs, one per guard rule, each with the finding it MUST produce.
|
|
64
|
+
*
|
|
65
|
+
* A canary that comes back clean is not a passing test — it is proof that the
|
|
66
|
+
* rule it names has stopped being reachable. `layout` is deliberately varied
|
|
67
|
+
* across ecosystems so that a canary set cannot be green by only ever handing
|
|
68
|
+
* the guard the layout the implementation was written from.
|
|
69
|
+
*/
|
|
70
|
+
export const TAMPER_CANARIES = [
|
|
71
|
+
{
|
|
72
|
+
id: "skip-injection/node",
|
|
73
|
+
file: "test/calc.test.js",
|
|
74
|
+
removed: [],
|
|
75
|
+
added: ['it.skip("adds", () => { assert.equal(add(1,2), 3); });'],
|
|
76
|
+
expect: "TEST_SKIP_INJECTION",
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
id: "vacuous/node",
|
|
80
|
+
file: "test/calc.test.js",
|
|
81
|
+
removed: ["assert.equal(add(1, 2), 3);"],
|
|
82
|
+
added: ["assert.ok(true);"],
|
|
83
|
+
expect: "VACUOUS_ASSERTION",
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
id: "commented/node",
|
|
87
|
+
file: "test/calc.test.js",
|
|
88
|
+
removed: ["assert.equal(add(1, 2), 3);"],
|
|
89
|
+
added: ["// assert.equal(add(1, 2), 3);"],
|
|
90
|
+
expect: "COMMENTED_ASSERTION",
|
|
91
|
+
},
|
|
92
|
+
{
|
|
93
|
+
id: "removal/node",
|
|
94
|
+
file: "test/calc.test.js",
|
|
95
|
+
removed: ["assert.equal(add(1, 2), 3);", "assert.equal(add(2, 2), 4);"],
|
|
96
|
+
added: [],
|
|
97
|
+
expect: "ASSERTION_REMOVAL",
|
|
98
|
+
},
|
|
99
|
+
{
|
|
100
|
+
id: "weakening/node",
|
|
101
|
+
file: "test/calc.test.js",
|
|
102
|
+
removed: ["assert.strictEqual(add(1, 2), 3);"],
|
|
103
|
+
added: ["assert.ok(add(1, 2) !== undefined);"],
|
|
104
|
+
expect: "ASSERTION_WEAKENED",
|
|
105
|
+
},
|
|
106
|
+
{
|
|
107
|
+
id: "expectation/node",
|
|
108
|
+
file: "test/calc.test.js",
|
|
109
|
+
removed: ["assert.equal(add(1, 2), 3);"],
|
|
110
|
+
added: ["assert.equal(add(1, 2), -1);"],
|
|
111
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
112
|
+
},
|
|
113
|
+
// The same attacks, in the layouts the classifier used to be blind to.
|
|
114
|
+
// These are the canaries that were silent when the classifier matched a
|
|
115
|
+
// substring instead of a path segment.
|
|
116
|
+
{
|
|
117
|
+
id: "expectation/pytest-root",
|
|
118
|
+
file: "tests/test_calc.py",
|
|
119
|
+
removed: [" self.assertEqual(add(1, 2), 3)"],
|
|
120
|
+
added: [" self.assertEqual(add(1, 2), -1)"],
|
|
121
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
id: "skip-injection/pytest-root",
|
|
125
|
+
file: "tests/test_calc.py",
|
|
126
|
+
removed: [],
|
|
127
|
+
added: ["@pytest.mark.skip", "def test_add():"],
|
|
128
|
+
expect: "TEST_SKIP_INJECTION",
|
|
129
|
+
},
|
|
130
|
+
{
|
|
131
|
+
id: "expectation/rust",
|
|
132
|
+
file: "tests/integration.rs",
|
|
133
|
+
removed: [" assert_eq!(add(1, 2), 3);"],
|
|
134
|
+
added: [" assert_eq!(add(1, 2), -1);"],
|
|
135
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
136
|
+
},
|
|
137
|
+
{
|
|
138
|
+
id: "expectation/go-colocated",
|
|
139
|
+
file: "internal/calc/calc_test.go",
|
|
140
|
+
removed: ['\t\tt.Errorf("got %d want %d", got, 3)'],
|
|
141
|
+
added: ['\t\tt.Errorf("got %d want %d", got, 999)'],
|
|
142
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
id: "removal/monorepo",
|
|
146
|
+
file: "packages/api/test/handler.test.js",
|
|
147
|
+
removed: ["expect(handler(req)).toBe(200);", "expect(handler(bad)).toBe(400);"],
|
|
148
|
+
added: [],
|
|
149
|
+
expect: "ASSERTION_REMOVAL",
|
|
150
|
+
},
|
|
151
|
+
// Dialects the guard was measured silent on. `assertEqual` matched only
|
|
152
|
+
// because the pattern's optional dot and case-insensitive flag happened to
|
|
153
|
+
// line up; `assertEquals` — one letter longer — did not, and neither did
|
|
154
|
+
// RSpec's `.to eq(`, PHPUnit's `$this->assertSame`, Minitest's
|
|
155
|
+
// `assert_equal` or XCTest's `XCTAssertEqual`. Every one of them returned
|
|
156
|
+
// PASS with a non-zero denominator, which is the exact shape this file
|
|
157
|
+
// exists to reject, one level down inside the mechanism built to catch it.
|
|
158
|
+
{
|
|
159
|
+
id: "expectation/junit",
|
|
160
|
+
file: "src/test/java/com/x/CalcTest.java",
|
|
161
|
+
removed: [" assertEquals(3, calc.add(1, 2));"],
|
|
162
|
+
added: [" assertEquals(-1, calc.add(1, 2));"],
|
|
163
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
id: "expectation/rspec",
|
|
167
|
+
file: "spec/models/user_spec.rb",
|
|
168
|
+
removed: [" expect(user.age).to eq(30)"],
|
|
169
|
+
added: [" expect(user.age).to eq(-1)"],
|
|
170
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
171
|
+
},
|
|
172
|
+
{
|
|
173
|
+
id: "expectation/phpunit",
|
|
174
|
+
file: "tests/Unit/CalcTest.php",
|
|
175
|
+
removed: [" $this->assertSame(3, $c->add(1, 2));"],
|
|
176
|
+
added: [" $this->assertSame(-1, $c->add(1, 2));"],
|
|
177
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
178
|
+
},
|
|
179
|
+
{
|
|
180
|
+
id: "expectation/minitest",
|
|
181
|
+
file: "test/user_test.rb",
|
|
182
|
+
removed: [" assert_equal(3, add(1, 2))"],
|
|
183
|
+
added: [" assert_equal(-1, add(1, 2))"],
|
|
184
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
185
|
+
},
|
|
186
|
+
{
|
|
187
|
+
id: "expectation/xctest",
|
|
188
|
+
file: "Tests/CalcTests/CalcTests.swift",
|
|
189
|
+
removed: [" XCTAssertEqual(add(1, 2), 3)"],
|
|
190
|
+
added: [" XCTAssertEqual(add(1, 2), -1)"],
|
|
191
|
+
expect: "ASSERTION_EXPECTATION_CHANGED",
|
|
192
|
+
},
|
|
193
|
+
{
|
|
194
|
+
id: "weakening/junit",
|
|
195
|
+
file: "src/test/java/com/x/CalcTest.java",
|
|
196
|
+
removed: [" assertEquals(3, calc.add(1, 2));"],
|
|
197
|
+
added: [" assertTrue(calc.add(1, 2) != null);"],
|
|
198
|
+
expect: "ASSERTION_WEAKENED",
|
|
199
|
+
},
|
|
200
|
+
{
|
|
201
|
+
id: "removal/phpunit",
|
|
202
|
+
file: "tests/Unit/CalcTest.php",
|
|
203
|
+
removed: [" $this->assertSame(3, $c->add(1, 2));", " $this->assertSame(4, $c->add(2, 2));"],
|
|
204
|
+
added: [],
|
|
205
|
+
expect: "ASSERTION_REMOVAL",
|
|
206
|
+
},
|
|
207
|
+
{
|
|
208
|
+
id: "vacuous/xctest",
|
|
209
|
+
file: "Tests/CalcTests/CalcTests.swift",
|
|
210
|
+
removed: [" XCTAssertEqual(add(1, 2), 3)"],
|
|
211
|
+
added: [" XCTAssertTrue(true)"],
|
|
212
|
+
expect: "VACUOUS_ASSERTION",
|
|
213
|
+
},
|
|
214
|
+
// Skip injection is the same blindness in the other five ecosystems: a
|
|
215
|
+
// suite that never runs cannot fail, and `@Disabled` is as effective as
|
|
216
|
+
// `it.skip` at making that happen.
|
|
217
|
+
{
|
|
218
|
+
id: "skip-injection/junit",
|
|
219
|
+
file: "src/test/java/com/x/CalcTest.java",
|
|
220
|
+
removed: [],
|
|
221
|
+
added: [" @Disabled(\"flaky\")", " void addsTwoNumbers() {"],
|
|
222
|
+
expect: "TEST_SKIP_INJECTION",
|
|
223
|
+
},
|
|
224
|
+
{
|
|
225
|
+
id: "skip-injection/rspec",
|
|
226
|
+
file: "spec/models/user_spec.rb",
|
|
227
|
+
removed: [],
|
|
228
|
+
added: [" xit \"computes the age\" do"],
|
|
229
|
+
expect: "TEST_SKIP_INJECTION",
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
id: "skip-injection/phpunit",
|
|
233
|
+
file: "tests/Unit/CalcTest.php",
|
|
234
|
+
removed: [],
|
|
235
|
+
added: [" $this->markTestSkipped(\"later\");"],
|
|
236
|
+
expect: "TEST_SKIP_INJECTION",
|
|
237
|
+
},
|
|
238
|
+
{
|
|
239
|
+
id: "skip-injection/xctest",
|
|
240
|
+
file: "Tests/CalcTests/CalcTests.swift",
|
|
241
|
+
removed: [],
|
|
242
|
+
added: [" throw XCTSkip(\"not now\")"],
|
|
243
|
+
expect: "TEST_SKIP_INJECTION",
|
|
244
|
+
},
|
|
245
|
+
];
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Mutants of the applicability predicate.
|
|
250
|
+
*
|
|
251
|
+
* Each one must kill at least one canary. A mutant that survives means no
|
|
252
|
+
* canary ever required the guard to *activate* — the suite would stay green if
|
|
253
|
+
* the guard silently stopped looking, which is precisely the defect.
|
|
254
|
+
*
|
|
255
|
+
* These are hand-written rather than generated: the original bug was not an
|
|
256
|
+
* untested branch, it was a branch nobody wrote, and no mutation operator
|
|
257
|
+
* invents the case the code never handled.
|
|
258
|
+
*/
|
|
259
|
+
export const PREDICATE_MUTANTS = [
|
|
260
|
+
{ id: "alwaysFalse", fn: () => false, why: "the guard looks at nothing" },
|
|
261
|
+
{ id: "rootBlind", fn: (p) => String(p).includes("/test/"), why: "the original substring bug" },
|
|
262
|
+
{ id: "nodeOnly", fn: (p) => /\.(test|spec)\.[jt]sx?$/.test(String(p)), why: "only the layout the code was written from" },
|
|
263
|
+
{ id: "caseSensitive", fn: (p) => String(p) === String(p).toLowerCase() && /(^|\/)tests?\//.test(String(p)), why: "case and separator drift" },
|
|
264
|
+
];
|
|
265
|
+
|
|
266
|
+
/** Runner outputs that state zero collected tests, per ecosystem. */
|
|
267
|
+
export const EMPTY_RUN_CANARIES = [
|
|
268
|
+
{ id: "pytest", output: "collected 0 items\n\nno tests ran in 0.01s" },
|
|
269
|
+
{ id: "jest", output: "No tests found, exiting with code 0" },
|
|
270
|
+
{ id: "vitest", output: "No test files found, exiting with code 0" },
|
|
271
|
+
{ id: "cargo", output: "running 0 tests\ntest result: ok. 0 passed" },
|
|
272
|
+
{ id: "mocha", output: " 0 passing (1ms)" },
|
|
273
|
+
{ id: "go", output: "? example.com/app\t[no test files]" },
|
|
274
|
+
{ id: "surefire", output: "Tests run: 0, Failures: 0, Errors: 0, Skipped: 0" },
|
|
275
|
+
{ id: "gradle", output: "> Task :test NO-SOURCE" },
|
|
276
|
+
{ id: "phpunit", output: "No tests executed!" },
|
|
277
|
+
{ id: "rspec", output: "0 examples, 0 failures" },
|
|
278
|
+
{ id: "dotnet", output: "Total tests: 0" },
|
|
279
|
+
{ id: "xctest", output: "Executed 0 tests" },
|
|
280
|
+
{ id: "ctest", output: "No tests were found!!!" },
|
|
281
|
+
];
|
|
282
|
+
|
|
283
|
+
/** Paths the policy says an agent must never modify without an override. */
|
|
284
|
+
export const SCOPE_CANARIES = [
|
|
285
|
+
{ path: ".github/workflows/ci.yml", rule: "deny", why: "runs with repo credentials" },
|
|
286
|
+
{ path: ".gitlab-ci.yml", rule: "deny", why: "same, another forge" },
|
|
287
|
+
{ path: "Jenkinsfile", rule: "deny", why: "same, another forge" },
|
|
288
|
+
{ path: ".envrc", rule: "deny", why: "direnv executes it on cd" },
|
|
289
|
+
{ path: "package-lock.json", rule: "protect", why: "decides which code installs" },
|
|
290
|
+
{ path: "Cargo.lock", rule: "protect", why: "same, another ecosystem" },
|
|
291
|
+
{ path: "conftest.py", rule: "protect", why: "runs before every pytest collection" },
|
|
292
|
+
{ path: "jest.config.js", rule: "protect", why: "decides which tests run" },
|
|
293
|
+
{ path: "CODEOWNERS", rule: "protect", why: "decides who must approve" },
|
|
294
|
+
];
|
|
295
|
+
|
|
296
|
+
/**
|
|
297
|
+
* Edits that must produce no finding at all.
|
|
298
|
+
*
|
|
299
|
+
* A guard that answers "yes" to everything has no more discrimination than
|
|
300
|
+
* one that answers "no" to everything, and it is worse in practice: the
|
|
301
|
+
* operator learns to pass the override without reading it, and the day it
|
|
302
|
+
* reports something real, nobody looks. Every entry here is an edit an
|
|
303
|
+
* honest agent makes constantly.
|
|
304
|
+
*
|
|
305
|
+
* The first two are not hypothetical. `//` begins with a division sign, so a
|
|
306
|
+
* comment line matched the operator-continuation test and folded itself into
|
|
307
|
+
* the assertion above it — which meant *adding* an assertion next to a
|
|
308
|
+
* comment was reported as rewriting an expectation. Python was immune,
|
|
309
|
+
* because `#` is not an operator, so the fixtures this project was written
|
|
310
|
+
* from never showed it.
|
|
311
|
+
*/
|
|
312
|
+
export const INNOCENT_EDITS = [
|
|
313
|
+
{
|
|
314
|
+
id: "add-assertion-beside-comment/node",
|
|
315
|
+
file: "test/calc.test.js",
|
|
316
|
+
context: "// arithmetic",
|
|
317
|
+
removed: [" assert.equal(add(1, 2), 3);"],
|
|
318
|
+
added: [" assert.equal(add(1, 2), 3);", " assert.equal(add(2, 2), 4);"],
|
|
319
|
+
why: "adding a test is the behaviour the gate exists to encourage",
|
|
320
|
+
},
|
|
321
|
+
{
|
|
322
|
+
id: "add-assertion-beside-comment/rust",
|
|
323
|
+
file: "tests/integration.rs",
|
|
324
|
+
context: "// arithmetic",
|
|
325
|
+
removed: [" assert!(add(1, 2) == 3);"],
|
|
326
|
+
added: [" assert!(add(1, 2) == 3);", " assert!(add(2, 2) == 4);"],
|
|
327
|
+
why: "same edit, same comment syntax, another ecosystem",
|
|
328
|
+
},
|
|
329
|
+
{
|
|
330
|
+
id: "reorder-assertions/pytest",
|
|
331
|
+
file: "tests/test_calc.py",
|
|
332
|
+
context: "# arithmetic",
|
|
333
|
+
removed: [" assert a() == 1", " assert b() == 2"],
|
|
334
|
+
added: [" assert b() == 2", " assert a() == 1"],
|
|
335
|
+
why: "moving an assertion changes nothing it checks",
|
|
336
|
+
},
|
|
337
|
+
{
|
|
338
|
+
id: "reindent/pytest",
|
|
339
|
+
file: "tests/test_calc.py",
|
|
340
|
+
context: "# arithmetic",
|
|
341
|
+
removed: [" assert add(1, 2) == 3"],
|
|
342
|
+
added: [" assert add(1, 2) == 3"],
|
|
343
|
+
why: "a formatter run must not read as tampering",
|
|
344
|
+
},
|
|
345
|
+
{
|
|
346
|
+
id: "reword-message/rspec",
|
|
347
|
+
file: "spec/models/user_spec.rb",
|
|
348
|
+
context: "# age",
|
|
349
|
+
removed: [' expect(u.age).to eq(30), "wrong"'],
|
|
350
|
+
added: [' expect(u.age).to eq(30), "unexpected age"'],
|
|
351
|
+
why: "RSpec writes the message outside the call, where an argument check cannot see it",
|
|
352
|
+
},
|
|
353
|
+
{
|
|
354
|
+
id: "reword-message/minitest",
|
|
355
|
+
file: "test/user_test.rb",
|
|
356
|
+
context: "# age",
|
|
357
|
+
removed: [' assert_equal 3, add(1, 2), "wrong"'],
|
|
358
|
+
added: [' assert_equal 3, add(1, 2), "unexpected sum"'],
|
|
359
|
+
why: "same, with the message in argument position and no parentheses",
|
|
360
|
+
},
|
|
361
|
+
{
|
|
362
|
+
id: "reword-message/node",
|
|
363
|
+
file: "test/calc.test.js",
|
|
364
|
+
context: "// arithmetic",
|
|
365
|
+
removed: [' assert.equal(add(1, 2), 3, "wrong");'],
|
|
366
|
+
added: [' assert.equal(add(1, 2), 3, "unexpected sum");'],
|
|
367
|
+
why: "rewording a failure message says nothing about what is checked",
|
|
368
|
+
},
|
|
369
|
+
{
|
|
370
|
+
id: "rename-test/junit",
|
|
371
|
+
file: "src/test/java/com/x/CalcTest.java",
|
|
372
|
+
context: "// arithmetic",
|
|
373
|
+
removed: [" void addsNumbers() {"],
|
|
374
|
+
added: [" void addsTwoNumbers() {"],
|
|
375
|
+
why: "a test name is not an expectation",
|
|
376
|
+
},
|
|
377
|
+
{
|
|
378
|
+
id: "change-import/node",
|
|
379
|
+
file: "test/calc.test.js",
|
|
380
|
+
context: "// setup",
|
|
381
|
+
removed: ['const calc = require("./calc");'],
|
|
382
|
+
added: ['const calc = require("../src/calc");'],
|
|
383
|
+
why: "`require(` is an import, not a claim — the loose net must not read it as one",
|
|
384
|
+
},
|
|
385
|
+
{
|
|
386
|
+
id: "rename-helper/node",
|
|
387
|
+
file: "test/calc.test.js",
|
|
388
|
+
context: "// setup",
|
|
389
|
+
removed: [" const shouldRetry = false;"],
|
|
390
|
+
added: [" const shouldRetry = true;"],
|
|
391
|
+
why: "`shouldRetry` and `expected` are identifiers; reading them as assertions makes the dialect warning worthless",
|
|
392
|
+
},
|
|
393
|
+
];
|
|
394
|
+
|
|
395
|
+
/**
|
|
396
|
+
* Dialects the guard genuinely cannot parse, which it must say out loud.
|
|
397
|
+
*
|
|
398
|
+
* This is the case the whole denominator exists for. A JUnit diff used to
|
|
399
|
+
* return `PASS` with `inputsSeen: 1` while not one assertion in it had been
|
|
400
|
+
* recognised — a verdict indistinguishable from a clean Node suite. Coverage
|
|
401
|
+
* will always end somewhere; what must never happen again is that the edge
|
|
402
|
+
* is silent.
|
|
403
|
+
*/
|
|
404
|
+
export const UNREADABLE_DIALECTS = [
|
|
405
|
+
{
|
|
406
|
+
id: "hspec",
|
|
407
|
+
file: "tests/CalcSpec.hs",
|
|
408
|
+
context: "-- arithmetic",
|
|
409
|
+
removed: [" calc `shouldBe` 3"],
|
|
410
|
+
added: [" calc `shouldBe` (-1)"],
|
|
411
|
+
why: "an infix assertion with no parentheses anywhere near it",
|
|
412
|
+
},
|
|
413
|
+
{
|
|
414
|
+
id: "googletest",
|
|
415
|
+
file: "tests/calc_test.cc",
|
|
416
|
+
context: "// arithmetic",
|
|
417
|
+
removed: [" EXPECT_EQ(add(1, 2), 3);"],
|
|
418
|
+
added: [" EXPECT_EQ(add(1, 2), -1);"],
|
|
419
|
+
why: "a macro dialect the pattern list does not cover",
|
|
420
|
+
},
|
|
421
|
+
];
|
|
422
|
+
|
|
423
|
+
/**
|
|
424
|
+
* Import forms the package-integrity extractor must find.
|
|
425
|
+
*
|
|
426
|
+
* The first pass of that check reported "every relative import resolves" on
|
|
427
|
+
* a package whose newest script could not start: its matcher was written as
|
|
428
|
+
* `[^;\n]*?from`, and the import it needed to see spanned several lines. The
|
|
429
|
+
* check was confidently green about a file that threw ERR_MODULE_NOT_FOUND
|
|
430
|
+
* on load — the same failure the tool exists to prevent, committed by the
|
|
431
|
+
* tool's own integrity check.
|
|
432
|
+
*
|
|
433
|
+
* Every entry is a source fragment and the specifiers it must yield.
|
|
434
|
+
*/
|
|
435
|
+
export const IMPORT_EXTRACTION_CASES = [
|
|
436
|
+
{ id: "single-line named", src: 'import { a, b } from "./one.mjs";', expect: ["./one.mjs"] },
|
|
437
|
+
{
|
|
438
|
+
id: "multi-line named",
|
|
439
|
+
src: 'import {\n a,\n b,\n} from "./two.mjs";',
|
|
440
|
+
expect: ["./two.mjs"],
|
|
441
|
+
why: "the form the first version of the check could not see",
|
|
442
|
+
},
|
|
443
|
+
{ id: "default", src: 'import three from "./three.mjs";', expect: ["./three.mjs"] },
|
|
444
|
+
{ id: "namespace", src: 'import * as four from "./four.mjs";', expect: ["./four.mjs"] },
|
|
445
|
+
{ id: "side-effect only", src: 'import "./five.mjs";', expect: ["./five.mjs"] },
|
|
446
|
+
{ id: "re-export", src: 'export { six } from "./six.mjs";', expect: ["./six.mjs"] },
|
|
447
|
+
{ id: "re-export all", src: 'export * from "./seven.mjs";', expect: ["./seven.mjs"] },
|
|
448
|
+
{ id: "dynamic", src: 'const m = await import("./eight.mjs");', expect: ["./eight.mjs"] },
|
|
449
|
+
{ id: "require", src: 'const nine = require("./nine.js");', expect: ["./nine.js"] },
|
|
450
|
+
{ id: "single quotes", src: "import ten from './ten.mjs';", expect: ["./ten.mjs"] },
|
|
451
|
+
{
|
|
452
|
+
id: "bare specifiers are not files",
|
|
453
|
+
src: 'import { readFileSync } from "node:fs";\nimport x from "some-package";',
|
|
454
|
+
expect: ["node:fs", "some-package"],
|
|
455
|
+
why: "found, then ignored by the resolver — never resolved against the tarball",
|
|
456
|
+
},
|
|
457
|
+
];
|
|
458
|
+
|
|
459
|
+
// Cases that exercise the mask rather than the matcher. Appended separately
|
|
460
|
+
// because each one is a fixture *about* fixtures: the check has to tell a
|
|
461
|
+
// module reference from a picture of one.
|
|
462
|
+
IMPORT_EXTRACTION_CASES.push(
|
|
463
|
+
{
|
|
464
|
+
id: "regex holding a quote, then a real import",
|
|
465
|
+
src: 'const q = /["\']/;\nimport x from "./after-regex.mjs";',
|
|
466
|
+
expect: ["./after-regex.mjs"],
|
|
467
|
+
why: "the phantom string this file's own subject matter opens",
|
|
468
|
+
},
|
|
469
|
+
{
|
|
470
|
+
id: "an import quoted inside a string",
|
|
471
|
+
src: 'const example = \'import a from "./not-real.mjs";\';',
|
|
472
|
+
expect: [],
|
|
473
|
+
why: "a picture of an import is not an import",
|
|
474
|
+
},
|
|
475
|
+
{
|
|
476
|
+
id: "an import written in a comment",
|
|
477
|
+
src: '// import a from "./commented.mjs";\nconst x = 1;',
|
|
478
|
+
expect: [],
|
|
479
|
+
why: "same, in the other place examples live",
|
|
480
|
+
}
|
|
481
|
+
);
|