jules-orchestrator-kit 0.59.0 → 0.63.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/bin/agentctl.mjs +6 -0
- package/package.json +2 -1
- package/scripts/guard-reach-check.mjs +176 -0
- package/scripts/release.mjs +17 -0
- package/src/config.mjs +68 -0
- package/src/coverage.mjs +17 -2
- package/src/engine.mjs +38 -3
- package/src/evidence.mjs +38 -1
- package/src/ops/test-collection.mjs +149 -0
- package/src/security.mjs +266 -20
- package/src/stack-detector.mjs +113 -10
- package/src/task-optimizer.mjs +7 -3
package/README.md
CHANGED
|
@@ -204,7 +204,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
204
204
|
* **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
|
|
205
205
|
* **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
|
|
206
206
|
* **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
|
|
207
|
-
* **Verified Test Suite:** Tested with **
|
|
207
|
+
* **Verified Test Suite:** Tested with **1015 unit tests across 145 suites passing in < 15.0s**.
|
|
208
208
|
|
|
209
209
|
<br/>
|
|
210
210
|
|
|
@@ -237,7 +237,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
237
237
|
| `doctor` | `agentctl doctor [--probe] [--json]` | Diagnostic check runner. `--probe` additionally starts the configured provider's CLI to confirm it answers, rather than only finding it on `PATH`. | `0` (Healthy), `1` (Failures) |
|
|
238
238
|
| `queue` | `agentctl queue [--dag] [--concurrency <n>] [--dry-run] [--json]` | Consumes and executes task envelopes in `.agent/jules-queue/` with Kahn's DAG dependency resolution. Non-task files (manifests, `README.md`) are skipped, and `--dry-run` previews without moving anything. | `0` (Complete) |
|
|
239
239
|
| `swarm` | `agentctl swarm [--json]` | Runs parallel multi-agent swarm across worker slots with PID liveness detection. | `0` (Complete) |
|
|
240
|
-
| `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--allow-protected] [--allow-test-
|
|
240
|
+
| `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--allow-protected] [--allow-test-change <kind>] [--json] [--json-report <path>]` | Runs security, secret scanning, rules budget audit, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `1` (Budget/Arg), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
|
|
241
241
|
| `mutate` / `mutation` | `agentctl mutate [--min-score <n>] [--max-mutants <n>] [--cmd <testCmd>] [--json]` | Runs zero-dependency diff mutation testing harness on changed hunks with operator inversion and safety rollback. | `0` (Passed), `1` (Score Low) |
|
|
242
242
|
| `coverage` | `agentctl coverage [--min <pct>] [--cmd <testCmd>] [--base <ref>] [--json]` | Runs native zero-dependency V8 diff coverage check against added diff lines. | `0` (Passed), `1` (Low Coverage) |
|
|
243
243
|
| `probe` / `stability` | `agentctl probe [--repeat <n>] [--min <passRate>] [--cmd <testCmd>] [--json]` | Probes test suite flakiness across N consecutive iterations with oscillation detection. | `0` (Passed), `1` (Flaky) |
|
package/bin/agentctl.mjs
CHANGED
|
@@ -388,7 +388,12 @@ async function main() {
|
|
|
388
388
|
// of spec, which necessarily rewrites what a test expects, hit a
|
|
389
389
|
// CRITICAL finding at exit 6 with no documented way past it. A guard
|
|
390
390
|
// with no override is not a guard, it is an outage.
|
|
391
|
+
//
|
|
392
|
+
// This one is the blunt form and turns off all six checks. Prefer
|
|
393
|
+
// `--allow-test-change <kind>`: answering one finding should not
|
|
394
|
+
// silence five other checks nobody looked at.
|
|
391
395
|
"allow-test-modifications": { type: "boolean" },
|
|
396
|
+
"allow-test-change": { type: "string", multiple: true },
|
|
392
397
|
json: { type: "boolean", short: "j" },
|
|
393
398
|
"json-report": { type: "string" },
|
|
394
399
|
"dry-run": { type: "boolean", short: "d" },
|
|
@@ -409,6 +414,7 @@ async function main() {
|
|
|
409
414
|
fix: values.fix,
|
|
410
415
|
allowProtected: values["allow-protected"],
|
|
411
416
|
allowTestModifications: values["allow-test-modifications"],
|
|
417
|
+
allowTestChanges: values["allow-test-change"],
|
|
412
418
|
jsonReport: values["json-report"],
|
|
413
419
|
});
|
|
414
420
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jules-orchestrator-kit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.63.0",
|
|
4
4
|
"description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -55,6 +55,7 @@
|
|
|
55
55
|
"jules:rules-lint": "node scripts/rules-lint.mjs",
|
|
56
56
|
"jules:doc-sync": "node scripts/doc-sync-check.mjs",
|
|
57
57
|
"release": "node scripts/release.mjs",
|
|
58
|
+
"guard-reach": "node scripts/guard-reach-check.mjs",
|
|
58
59
|
"lint": "eslint ."
|
|
59
60
|
},
|
|
60
61
|
"engines": {
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Activation coverage: proof that every blocking check can still be made red.
|
|
5
|
+
*
|
|
6
|
+
* A defect that turns a check off cannot be found by the check it turns off.
|
|
7
|
+
* That is not a hypothetical — `isTestFile` matched the substring `/test/`,
|
|
8
|
+
* which does not occur in `tests/test_calc.py`, so the entire tamper guard was
|
|
9
|
+
* silent for the standard pytest, Rust and RSpec layouts. Every mechanism that
|
|
10
|
+
* should have caught it was working exactly as designed:
|
|
11
|
+
*
|
|
12
|
+
* - the unit suite sampled the same distribution the implementation was
|
|
13
|
+
* written from, so its fixtures re-confirmed the dialect it already knew;
|
|
14
|
+
* - the doc-sync gate compares counts and versions, and a guard that guards
|
|
15
|
+
* nothing still contributes passing tests;
|
|
16
|
+
* - the nine-way CI matrix varies OS and Node version — dimensions
|
|
17
|
+
* orthogonal to the defect. Nine runs of `test/foo.test.js` never explore
|
|
18
|
+
* `tests/test_calc.py`;
|
|
19
|
+
* - cold review reads the code against its stated intent, and here the code
|
|
20
|
+
* and the intent agreed. The eye supplies the leading slash;
|
|
21
|
+
* - the release gate is a conjunction over those four, and a signal that
|
|
22
|
+
* silently goes absent contributes `true`.
|
|
23
|
+
*
|
|
24
|
+
* The common property: `ok: true` from a check that examined nothing is
|
|
25
|
+
* byte-identical to `ok: true` from a check that examined everything. There is
|
|
26
|
+
* no denominator. This script supplies one.
|
|
27
|
+
*
|
|
28
|
+
* Three steps, all in-process, no dependencies, well under a second:
|
|
29
|
+
*
|
|
30
|
+
* 1. POLICY — the hand-written witness table in test/fixtures/guard-policy.mjs
|
|
31
|
+
* must hold. It is derived from what the tool advertises, never
|
|
32
|
+
* from the regexes that implement it.
|
|
33
|
+
* 2. CANARIES — every known-bad input must produce the finding it names. A
|
|
34
|
+
* canary that comes back clean is not a pass; it is proof that
|
|
35
|
+
* the rule stopped being reachable.
|
|
36
|
+
* 3. MUTANTS — each hand-written mutant of the applicability predicate must
|
|
37
|
+
* kill at least one canary. A surviving mutant means no canary
|
|
38
|
+
* ever required the guard to activate, so the suite would stay
|
|
39
|
+
* green if it silently stopped looking.
|
|
40
|
+
*
|
|
41
|
+
* Usage: node scripts/guard-reach-check.mjs [--json]
|
|
42
|
+
* Exit codes: 0 = every guard reachable, 1 = a guard has gone silent.
|
|
43
|
+
*/
|
|
44
|
+
|
|
45
|
+
import { checkTestTampering, checkScope } from "../src/security.mjs";
|
|
46
|
+
import { isTestPath } from "../src/test-paths.mjs";
|
|
47
|
+
import { normalizeScope } from "../src/config.mjs";
|
|
48
|
+
import { parseCollectedTests } from "../src/ops/test-collection.mjs";
|
|
49
|
+
import {
|
|
50
|
+
TEST_PATH_CASES,
|
|
51
|
+
TAMPER_CANARIES,
|
|
52
|
+
PREDICATE_MUTANTS,
|
|
53
|
+
EMPTY_RUN_CANARIES,
|
|
54
|
+
SCOPE_CANARIES,
|
|
55
|
+
} from "../test/fixtures/guard-policy.mjs";
|
|
56
|
+
|
|
57
|
+
/** Build a unified diff for one canary. */
|
|
58
|
+
function canaryDiff(c) {
|
|
59
|
+
const lines = [`--- a/${c.file}`, `+++ b/${c.file}`, "@@ -1,20 +1,20 @@", " // context"];
|
|
60
|
+
for (const l of c.removed) lines.push(`-${l}`);
|
|
61
|
+
for (const l of c.added) lines.push(`+${l}`);
|
|
62
|
+
lines.push(" // context");
|
|
63
|
+
return lines.join("\n");
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const failures = [];
|
|
67
|
+
const checks = [];
|
|
68
|
+
const add = (name, ok, detail) => {
|
|
69
|
+
checks.push({ name, ok, detail });
|
|
70
|
+
if (!ok) failures.push(`${name}: ${detail}`);
|
|
71
|
+
};
|
|
72
|
+
|
|
73
|
+
// --- 1. Policy contract -----------------------------------------------------
|
|
74
|
+
{
|
|
75
|
+
const wrong = TEST_PATH_CASES.filter((c) => isTestPath(c.path) !== c.expected);
|
|
76
|
+
add(
|
|
77
|
+
"policy: test-path domain",
|
|
78
|
+
wrong.length === 0,
|
|
79
|
+
wrong.length
|
|
80
|
+
? wrong.map((c) => `${c.path} → ${isTestPath(c.path)}, policy says ${c.expected} (${c.why})`).join("; ")
|
|
81
|
+
: `${TEST_PATH_CASES.length} witnesses hold`
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
{
|
|
86
|
+
const scope = normalizeScope({ deny: [], allow: [], protect: [] });
|
|
87
|
+
const wrong = [];
|
|
88
|
+
for (const c of SCOPE_CANARIES) {
|
|
89
|
+
const res = checkScope([c.path], scope);
|
|
90
|
+
const rule = res.ok ? "none" : res.violations[0].rule;
|
|
91
|
+
if (rule !== c.rule) wrong.push(`${c.path} → ${rule}, policy says ${c.rule} (${c.why})`);
|
|
92
|
+
}
|
|
93
|
+
add("policy: scope tiers", wrong.length === 0, wrong.length ? wrong.join("; ") : `${SCOPE_CANARIES.length} paths tiered as declared`);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
{
|
|
97
|
+
const missed = EMPTY_RUN_CANARIES.filter((c) => parseCollectedTests(c.output, "").count !== 0);
|
|
98
|
+
add(
|
|
99
|
+
"policy: empty-run detection",
|
|
100
|
+
missed.length === 0,
|
|
101
|
+
missed.length ? `${missed.map((m) => m.id).join(", ")} report zero tests in a spelling the floor cannot read` : `${EMPTY_RUN_CANARIES.length} runners recognised`
|
|
102
|
+
);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// --- 2. Canaries ------------------------------------------------------------
|
|
106
|
+
const canaryResults = new Map();
|
|
107
|
+
{
|
|
108
|
+
const silent = [];
|
|
109
|
+
const noDenominator = [];
|
|
110
|
+
for (const c of TAMPER_CANARIES) {
|
|
111
|
+
const res = checkTestTampering(canaryDiff(c));
|
|
112
|
+
const hit = (res.violations || []).some((v) => v.type === c.expect);
|
|
113
|
+
canaryResults.set(c.id, hit);
|
|
114
|
+
if (!hit) silent.push(`${c.id} expected ${c.expect}, got ${JSON.stringify((res.violations || []).map((v) => v.type))}`);
|
|
115
|
+
// A finding with no denominator is the shape this script exists to reject.
|
|
116
|
+
if (hit && !(res.inputsSeen > 0)) noDenominator.push(c.id);
|
|
117
|
+
}
|
|
118
|
+
add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length} canaries red as required`);
|
|
119
|
+
add("canaries: every finding carries a denominator", noDenominator.length === 0, noDenominator.length ? noDenominator.join(", ") : "inputsSeen > 0 on every hit");
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// --- 3. Predicate mutants ---------------------------------------------------
|
|
123
|
+
{
|
|
124
|
+
const survivors = [];
|
|
125
|
+
for (const mutant of PREDICATE_MUTANTS) {
|
|
126
|
+
let killed = false;
|
|
127
|
+
for (const c of TAMPER_CANARIES) {
|
|
128
|
+
// Only canaries the healthy predicate catches can kill a mutant.
|
|
129
|
+
if (!canaryResults.get(c.id)) continue;
|
|
130
|
+
const res = checkTestTampering(canaryDiff(c), { isTestPath: mutant.fn });
|
|
131
|
+
if (!(res.violations || []).some((v) => v.type === c.expect)) {
|
|
132
|
+
killed = true;
|
|
133
|
+
break;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
if (!killed) survivors.push(`${mutant.id} (${mutant.why})`);
|
|
137
|
+
}
|
|
138
|
+
add(
|
|
139
|
+
"mutants: blinding the predicate breaks a canary",
|
|
140
|
+
survivors.length === 0,
|
|
141
|
+
survivors.length
|
|
142
|
+
? `survived: ${survivors.join(", ")} — no canary required the guard to activate`
|
|
143
|
+
: `${PREDICATE_MUTANTS.length} mutants killed`
|
|
144
|
+
);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// --- Report -----------------------------------------------------------------
|
|
148
|
+
const json = process.argv.includes("--json");
|
|
149
|
+
const activated = [...canaryResults.values()].filter(Boolean).length;
|
|
150
|
+
|
|
151
|
+
if (json) {
|
|
152
|
+
console.log(
|
|
153
|
+
JSON.stringify(
|
|
154
|
+
{
|
|
155
|
+
ok: failures.length === 0,
|
|
156
|
+
checks,
|
|
157
|
+
activationCoverage: { canaries: canaryResults.size, activated },
|
|
158
|
+
},
|
|
159
|
+
null,
|
|
160
|
+
2
|
|
161
|
+
)
|
|
162
|
+
);
|
|
163
|
+
} else {
|
|
164
|
+
console.log("\n🎯 Guard Reach Check (activation coverage)");
|
|
165
|
+
console.log("-------------------------------------------------------");
|
|
166
|
+
for (const c of checks) console.log(` ${c.ok ? "✅" : "❌"} ${c.name.padEnd(46)} ${c.detail}`);
|
|
167
|
+
console.log("-------------------------------------------------------");
|
|
168
|
+
console.log(` canaries activated: ${activated}/${canaryResults.size}`);
|
|
169
|
+
console.log(
|
|
170
|
+
failures.length === 0
|
|
171
|
+
? "✅ Every blocking guard can still be made red.\n"
|
|
172
|
+
: `\n❌ ${failures.length} guard(s) may have gone silent. A check that cannot be made red is not a check.\n`
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
process.exit(failures.length === 0 ? 0 : 1);
|
package/scripts/release.mjs
CHANGED
|
@@ -41,6 +41,23 @@ try {
|
|
|
41
41
|
process.exit(1);
|
|
42
42
|
}
|
|
43
43
|
|
|
44
|
+
// 1a. Activation coverage (blocking).
|
|
45
|
+
//
|
|
46
|
+
// Step 1 proved the suite is green. Green is only evidence if the guards were
|
|
47
|
+
// switched on: a check that silently stopped applying contributes passing
|
|
48
|
+
// tests and a zero exit code exactly like one that ran. This asks the question
|
|
49
|
+
// the suite cannot — can every blocking guard still be made red? — and it runs
|
|
50
|
+
// before the doc-sync gate because a silent guard makes every later signal
|
|
51
|
+
// meaningless.
|
|
52
|
+
console.log("1a. Verifying every blocking guard can still be made red...");
|
|
53
|
+
try {
|
|
54
|
+
execSync("node scripts/guard-reach-check.mjs", { cwd: root, stdio: "inherit" });
|
|
55
|
+
console.log("");
|
|
56
|
+
} catch (_) {
|
|
57
|
+
console.error("\n❌ Release Aborted: a guard has gone silent. See the failing rows above.");
|
|
58
|
+
process.exit(1);
|
|
59
|
+
}
|
|
60
|
+
|
|
44
61
|
// 1b. Documentation / version consistency gate (blocking).
|
|
45
62
|
console.log("1b. Verifying documentation is in sync with package.json & test suite...");
|
|
46
63
|
{
|
package/src/config.mjs
CHANGED
|
@@ -54,6 +54,9 @@ const CI_DEFINITIONS = [
|
|
|
54
54
|
"appveyor.yml",
|
|
55
55
|
".teamcity/**",
|
|
56
56
|
".githooks/**",
|
|
57
|
+
"buildspec.yml",
|
|
58
|
+
"**/buildspec.yml",
|
|
59
|
+
".buildspec/**",
|
|
57
60
|
];
|
|
58
61
|
|
|
59
62
|
export const BUILTIN_DENY = [
|
|
@@ -64,6 +67,24 @@ export const BUILTIN_DENY = [
|
|
|
64
67
|
"**/*.key",
|
|
65
68
|
"**/id_rsa*",
|
|
66
69
|
".agent/jules-queue/**",
|
|
70
|
+
|
|
71
|
+
// Shell that runs on `cd`, and credentials in plaintext. Same class as a CI
|
|
72
|
+
// definition: code or secrets that take effect before anyone reviews them.
|
|
73
|
+
"**/.envrc",
|
|
74
|
+
"**/.git-credentials",
|
|
75
|
+
"**/.aws/**",
|
|
76
|
+
"**/.ssh/**",
|
|
77
|
+
"**/.kube/**",
|
|
78
|
+
"**/kubeconfig*",
|
|
79
|
+
"**/.docker/config.json",
|
|
80
|
+
"**/*.p12",
|
|
81
|
+
"**/*.pfx",
|
|
82
|
+
"**/*.p8",
|
|
83
|
+
"**/id_ed25519*",
|
|
84
|
+
"**/credentials.json",
|
|
85
|
+
"**/service-account*.json",
|
|
86
|
+
"**/*.tfstate",
|
|
87
|
+
"**/*.tfstate.*",
|
|
67
88
|
...CI_DEFINITIONS,
|
|
68
89
|
];
|
|
69
90
|
|
|
@@ -110,6 +131,53 @@ export const BUILTIN_PROTECT = [
|
|
|
110
131
|
"Dockerfile",
|
|
111
132
|
"**/Dockerfile",
|
|
112
133
|
|
|
134
|
+
// Lockfiles decide which code actually *runs*. `package.json` was protected
|
|
135
|
+
// and `package-lock.json` was not, so an agent could change a resolved URL
|
|
136
|
+
// or an integrity hash — swapping the code that gets installed — without
|
|
137
|
+
// touching a single declared dependency, and the gate said nothing. The
|
|
138
|
+
// entropy scanner is deliberately blind to lockfiles too (they are full of
|
|
139
|
+
// hashes), so the change was invisible twice over. `BUILTIN_RESTRICTED` in
|
|
140
|
+
// risk.mjs already knew these mattered; only the risk tier consumed it,
|
|
141
|
+
// never checkScope.
|
|
142
|
+
"package-lock.json",
|
|
143
|
+
"**/package-lock.json",
|
|
144
|
+
"pnpm-lock.yaml",
|
|
145
|
+
"**/pnpm-lock.yaml",
|
|
146
|
+
"yarn.lock",
|
|
147
|
+
"**/yarn.lock",
|
|
148
|
+
"bun.lockb",
|
|
149
|
+
"bun.lock",
|
|
150
|
+
"Cargo.lock",
|
|
151
|
+
"**/Cargo.lock",
|
|
152
|
+
"go.sum",
|
|
153
|
+
"**/go.sum",
|
|
154
|
+
"poetry.lock",
|
|
155
|
+
"uv.lock",
|
|
156
|
+
"Pipfile.lock",
|
|
157
|
+
"Gemfile.lock",
|
|
158
|
+
"composer.lock",
|
|
159
|
+
"gradle.lockfile",
|
|
160
|
+
"npm-shrinkwrap.json",
|
|
161
|
+
"mix.lock",
|
|
162
|
+
"pubspec.lock",
|
|
163
|
+
"Podfile.lock",
|
|
164
|
+
"Package.resolved",
|
|
165
|
+
".terraform.lock.hcl",
|
|
166
|
+
|
|
167
|
+
// Toolchain pins choose the compiler that runs the whole suite.
|
|
168
|
+
".nvmrc",
|
|
169
|
+
".tool-versions",
|
|
170
|
+
".mise.toml",
|
|
171
|
+
"rust-toolchain",
|
|
172
|
+
"rust-toolchain.toml",
|
|
173
|
+
"**/gradle/wrapper/gradle-wrapper.properties",
|
|
174
|
+
"**/gradle/wrapper/gradle-wrapper.jar",
|
|
175
|
+
|
|
176
|
+
// Who has to approve a change, and what runs on every developer's machine.
|
|
177
|
+
"CODEOWNERS",
|
|
178
|
+
"docs/CODEOWNERS",
|
|
179
|
+
".pre-commit-config.yaml",
|
|
180
|
+
|
|
113
181
|
// Test-runner configuration decides which tests run and what counts as a
|
|
114
182
|
// pass. Rewriting it is the cheapest way to make a suite green without
|
|
115
183
|
// touching a single assertion — `--passWithNoTests`, an added ignore
|
package/src/coverage.mjs
CHANGED
|
@@ -300,12 +300,27 @@ export function calculateDiffCoverage(coverageByFile, diffStr = "", options = {}
|
|
|
300
300
|
}
|
|
301
301
|
}
|
|
302
302
|
|
|
303
|
-
|
|
304
|
-
|
|
303
|
+
// 100% of nothing is not 100%.
|
|
304
|
+
//
|
|
305
|
+
// The denominator counts only the added lines V8 actually mapped, and V8
|
|
306
|
+
// maps nothing outside Node. So a Python diff adding three executable lines
|
|
307
|
+
// measured zero of them and was reported as `score: 100` — the best possible
|
|
308
|
+
// number, produced by a measurement that never happened, on 20-odd of the 25
|
|
309
|
+
// stacks this kit claims to support. `mutation.mjs` had the identical bug and
|
|
310
|
+
// was fixed in v0.57.0; this is the same shape one module over.
|
|
311
|
+
//
|
|
312
|
+
// `ok` stays true because nothing failed to be covered, and a gate that
|
|
313
|
+
// blocks every non-Node diff gets switched off. What changes is the claim:
|
|
314
|
+
// `scored: false` and a reason, instead of a number nobody measured.
|
|
315
|
+
const scored = totalLines > 0;
|
|
316
|
+
const score = scored ? Math.round((coveredLines / totalLines) * 10000) / 100 : null;
|
|
317
|
+
const ok = scored ? score >= minCoverage : true;
|
|
305
318
|
|
|
306
319
|
return {
|
|
307
320
|
ok,
|
|
308
321
|
score,
|
|
322
|
+
scored,
|
|
323
|
+
...(scored ? {} : { reason: "No added executable lines were measurable — V8 coverage only observes code Node itself ran, so nothing was scored." }),
|
|
309
324
|
minCoverage,
|
|
310
325
|
totalLines,
|
|
311
326
|
coveredLines,
|
package/src/engine.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { loadConfig, parseYaml, normalizeScope } from "./config.mjs";
|
|
2
2
|
import { isTestPath } from "./test-paths.mjs";
|
|
3
|
+
import { checkCollectionFloor } from "./ops/test-collection.mjs";
|
|
3
4
|
import { checkScope, scanDiff, scanBinaryPayloads, redactSecrets } from "./security.mjs";
|
|
4
5
|
import { changedFiles, diffBytes, diffText, binaryDiffEntries, symlinkChanges, showFromOrigin, runCmd } from "./git.mjs";
|
|
5
6
|
import { createProvider, ProviderRateLimitError, ProviderUnavailableError } from "./provider.mjs";
|
|
@@ -275,7 +276,11 @@ export async function gate(opts = {}) {
|
|
|
275
276
|
}
|
|
276
277
|
|
|
277
278
|
// Phase 3: Diff Secret Scanner & Security Checks
|
|
278
|
-
const secretResult = scanDiff(diffStr, {
|
|
279
|
+
const secretResult = scanDiff(diffStr, {
|
|
280
|
+
root,
|
|
281
|
+
allowTestModifications: opts.allowTestModifications === true,
|
|
282
|
+
allowTestChanges: opts.allowTestChanges,
|
|
283
|
+
});
|
|
279
284
|
// A binary file reaches the scanner as one summary line, so its contents were
|
|
280
285
|
// never looked at — a NUL byte in front of a token was enough to hide it.
|
|
281
286
|
// Inspect those files directly and fold the verdict in.
|
|
@@ -517,7 +522,25 @@ export async function gate(opts = {}) {
|
|
|
517
522
|
const ranNoVerification = !executionRecords.some((r) => r && r.kind !== "assert");
|
|
518
523
|
const missingOracle = verificationRequired && ranNoVerification;
|
|
519
524
|
|
|
520
|
-
|
|
525
|
+
// A command that ran is not the same as a command that tested something.
|
|
526
|
+
//
|
|
527
|
+
// `missingOracle` above catches "no stage executed". It cannot catch the
|
|
528
|
+
// case where a stage executed, exited 0, and collected zero tests — which
|
|
529
|
+
// several runners report as success by design: `go test ./...` prints
|
|
530
|
+
// "[no test files]" and exits 0, jest has --passWithNoTests, and
|
|
531
|
+
// `npm test --workspaces` is green when the one package the diff touched
|
|
532
|
+
// has no suite. A repository could invert a function, add an untested one
|
|
533
|
+
// and collect five green phases, verified against nothing at all.
|
|
534
|
+
//
|
|
535
|
+
// The count is read out of the runner's own summary and only a *stated*
|
|
536
|
+
// zero counts. An unrecognised runner yields null and passes: failing on
|
|
537
|
+
// "I could not tell" would break every runner not on the list.
|
|
538
|
+
const collectionFloor = verificationRequired
|
|
539
|
+
? checkCollectionFloor(testResult, { minTests: trustedVerify.minTests })
|
|
540
|
+
: { ok: true, count: null, runner: null, reason: null };
|
|
541
|
+
const emptySuite = !collectionFloor.ok;
|
|
542
|
+
|
|
543
|
+
const verifyOk = !failingCmd && !testTampered && !missingOracle && !emptySuite;
|
|
521
544
|
|
|
522
545
|
// What actually broke. Without this the verify phase reported `ok: false` and
|
|
523
546
|
// nothing else — not the stage, not the exit code, not a line of output — so
|
|
@@ -564,7 +587,18 @@ export async function gate(opts = {}) {
|
|
|
564
587
|
"The gate approves a change because verification passed. Zero stages executed is not a pass.",
|
|
565
588
|
],
|
|
566
589
|
}
|
|
567
|
-
:
|
|
590
|
+
: emptySuite
|
|
591
|
+
? {
|
|
592
|
+
stageId: "empty-suite",
|
|
593
|
+
command: testResult?.command || null,
|
|
594
|
+
exitCode: 0,
|
|
595
|
+
stdout: "",
|
|
596
|
+
stderr: collectionFloor.reason,
|
|
597
|
+
diagnostics: [
|
|
598
|
+
"An exit code of 0 from a runner that collected no tests is not evidence about this change.",
|
|
599
|
+
],
|
|
600
|
+
}
|
|
601
|
+
: null;
|
|
568
602
|
|
|
569
603
|
// Generate & persist Evidence Manifest
|
|
570
604
|
const evidenceManifest = generateEvidenceManifest(root, {
|
|
@@ -579,6 +613,7 @@ export async function gate(opts = {}) {
|
|
|
579
613
|
failedStage: failingCmd?.stageId || failingCmd?.phase || null,
|
|
580
614
|
diagnostics: failingCmd?.diagnostics?.length ? failingCmd.diagnostics : (failingCmd?.stderr ? [failingCmd.stderr] : []),
|
|
581
615
|
metrics: failingCmd?.metrics || {},
|
|
616
|
+
collection: collectionFloor,
|
|
582
617
|
ok: verifyOk,
|
|
583
618
|
});
|
|
584
619
|
if (testTampered) {
|
package/src/evidence.mjs
CHANGED
|
@@ -129,7 +129,20 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
129
129
|
.filter((p) => existsSync(join(root, p)) && statSync(join(root, p)).isFile())
|
|
130
130
|
.sort();
|
|
131
131
|
} else {
|
|
132
|
-
|
|
132
|
+
// A sixth spelling of "where do the tests live?", and the one that got
|
|
133
|
+
// missed when the other five were unified behind `isTestPath`: this is a
|
|
134
|
+
// list of *directory names at the repository root*, not a predicate. Go
|
|
135
|
+
// puts its tests beside the code (`internal/calc/calc_test.go`), and every
|
|
136
|
+
// monorepo puts them under `packages/*/test/`. Neither is under a
|
|
137
|
+
// root-level `test/`, so the walk found nothing, `fileCount` was 0, and
|
|
138
|
+
// `strictTestLock` — which requires `fileCount > 0` — switched itself off
|
|
139
|
+
// without saying so. The tree hash then became the SHA-256 of the empty
|
|
140
|
+
// string, and the evidence manifest attested to it.
|
|
141
|
+
//
|
|
142
|
+
// The named directories stay as a fast path; when they yield nothing, walk
|
|
143
|
+
// the repository and let the shared predicate decide. The walk already
|
|
144
|
+
// skips node_modules, vendor, target and the build caches.
|
|
145
|
+
const targetDirs = options.directories || ["test", "tests", "__tests__", "spec", "specs", "src"];
|
|
133
146
|
for (const dirName of targetDirs) {
|
|
134
147
|
const dirPath = join(root, dirName);
|
|
135
148
|
if (existsSync(dirPath)) {
|
|
@@ -137,6 +150,11 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
137
150
|
fileList.push(...found);
|
|
138
151
|
}
|
|
139
152
|
}
|
|
153
|
+
if (options.testOnly && !fileList.some((f) => isTestPath(f))) {
|
|
154
|
+
for (const f of findFilesRecursively(root, root)) {
|
|
155
|
+
if (isTestPath(f)) fileList.push(f);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
140
158
|
|
|
141
159
|
// Plenty of projects keep `app.test.mjs` or `index.js` beside package.json
|
|
142
160
|
// rather than under one of the directories above, and those files were
|
|
@@ -201,6 +219,7 @@ export function computeEvidenceHash(manifest) {
|
|
|
201
219
|
intent: manifest.intent,
|
|
202
220
|
provenance: manifest.provenance,
|
|
203
221
|
testIntegrity: manifest.testIntegrity,
|
|
222
|
+
...(manifest.verification ? { verification: manifest.verification } : {}),
|
|
204
223
|
...(manifest.sourceIntegrity ? { sourceIntegrity: manifest.sourceIntegrity } : {}),
|
|
205
224
|
executionRecords: manifest.executionRecords,
|
|
206
225
|
securityChecks: manifest.securityChecks,
|
|
@@ -342,6 +361,24 @@ export function generateEvidenceManifest(root = process.cwd(), options = {}) {
|
|
|
342
361
|
maxDiffKb: options.maxDiffKb || 75,
|
|
343
362
|
protectedScopeOk: options.protectedScopeOk ?? true,
|
|
344
363
|
},
|
|
364
|
+
// How many tests the runner said it collected, and whether it said at all.
|
|
365
|
+
//
|
|
366
|
+
// The collection floor fails a *stated* zero and lets an unstated count
|
|
367
|
+
// pass, because failing on "I could not tell" would break every runner not
|
|
368
|
+
// on the list. That is the right call for the verdict and the wrong thing
|
|
369
|
+
// to leave out of the record: a manifest that says nothing here reads as
|
|
370
|
+
// though a suite ran. `counted: false` is the honest shape for a run where
|
|
371
|
+
// the number was never observable — a quiet runner (`cargo test --quiet`,
|
|
372
|
+
// `pytest -q`) suppresses the very line the floor reads.
|
|
373
|
+
...(options.collection
|
|
374
|
+
? {
|
|
375
|
+
verification: {
|
|
376
|
+
testsCollected: options.collection.count,
|
|
377
|
+
counted: options.collection.count !== null,
|
|
378
|
+
runner: options.collection.runner,
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
: {}),
|
|
345
382
|
...(diagnostics.length > 0 ? { diagnostics } : {}),
|
|
346
383
|
...(Object.keys(metrics).length > 0 ? { metrics } : {}),
|
|
347
384
|
};
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How many tests the runner actually collected, read out of its own output.
|
|
3
|
+
*
|
|
4
|
+
* The gate's oracle is one number: the exit code of the verification command.
|
|
5
|
+
* That number cannot distinguish "every test passed" from "there were no
|
|
6
|
+
* tests". Several runners report the second case as success, by design:
|
|
7
|
+
*
|
|
8
|
+
* go test ./... → "? example.com/app [no test files]", exit 0
|
|
9
|
+
* jest --passWithNoTests → "No tests found, exiting with code 0"
|
|
10
|
+
* npm test --workspaces → exit 0 when the changed package has no suite
|
|
11
|
+
* pytest --exitfirst on a path that matches nothing, in some configurations
|
|
12
|
+
*
|
|
13
|
+
* So a repository could invert a function, add an untested one, and collect
|
|
14
|
+
* five green phases — verified against nothing. `verify.required: false` is
|
|
15
|
+
* the switch for a repository that genuinely has no oracle; silently passing
|
|
16
|
+
* is not.
|
|
17
|
+
*
|
|
18
|
+
* The parsing is deliberately one-sided. A count is only returned when the
|
|
19
|
+
* runner stated one in a form recognised here; an unrecognised runner yields
|
|
20
|
+
* `null`, and null is not a failure. Failing on "I could not tell" would break
|
|
21
|
+
* every runner not on this list, which is most of them.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Patterns that state a test count, per runner family.
|
|
26
|
+
*
|
|
27
|
+
* Each entry captures a single number. The first pattern that matches wins,
|
|
28
|
+
* so the more specific summaries come first.
|
|
29
|
+
*/
|
|
30
|
+
const COUNT_PATTERNS = [
|
|
31
|
+
// node:test — spec reporter ("ℹ tests 940") and tap ("# tests 940")
|
|
32
|
+
{ name: "node:test", re: /^[^\n]*?(?:ℹ|#)\s*tests\s+(\d+)\s*$/m },
|
|
33
|
+
// pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
|
|
34
|
+
{ name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
|
|
35
|
+
{ name: "pytest", re: /=+\s*(\d+)\s+passed/m },
|
|
36
|
+
// cargo — "running 7 tests"
|
|
37
|
+
{ name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
|
|
38
|
+
// jest / vitest — "Tests: 12 passed, 12 total"
|
|
39
|
+
{ name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
|
|
40
|
+
// mocha — "12 passing"
|
|
41
|
+
{ name: "mocha", re: /^\s*(\d+)\s+passing/m },
|
|
42
|
+
// Maven / Surefire — "Tests run: 12, Failures: 0"
|
|
43
|
+
{ name: "surefire", re: /\bTests run:\s*(\d+)/i },
|
|
44
|
+
// PHPUnit — "OK (12 tests, 30 assertions)"
|
|
45
|
+
{ name: "phpunit", re: /\bOK\s*\((\d+)\s+tests?/i },
|
|
46
|
+
// RSpec / ExUnit — "12 examples, 0 failures" / "12 tests, 0 failures"
|
|
47
|
+
{ name: "rspec", re: /^\s*(\d+)\s+examples?,\s*\d+\s+failures?/m },
|
|
48
|
+
{ name: "exunit", re: /^\s*(\d+)\s+tests?,\s*\d+\s+failures?/m },
|
|
49
|
+
// dotnet test — "Total tests: 12" / "Passed! - Failed: 0, Passed: 12"
|
|
50
|
+
{ name: "dotnet", re: /\bTotal(?:\s+tests)?:\s*(\d+)/i },
|
|
51
|
+
// swift test / XCTest — "Executed 12 tests"
|
|
52
|
+
{ name: "xctest", re: /\bExecuted\s+(\d+)\s+tests?/i },
|
|
53
|
+
];
|
|
54
|
+
|
|
55
|
+
/** Per-test lines, which `go test` only prints under -v. */
|
|
56
|
+
const GO_PER_TEST = /^\s*--- (?:PASS|FAIL|SKIP):/gm;
|
|
57
|
+
|
|
58
|
+
/** Phrases that state, in so many words, that nothing was collected. */
|
|
59
|
+
const EXPLICIT_ZERO = [
|
|
60
|
+
{ name: "pytest", re: /\bno tests ran\b/i },
|
|
61
|
+
{ name: "pytest", re: /^\s*collected\s+0\s+items?/m },
|
|
62
|
+
{ name: "jest", re: /\bNo tests found\b/i },
|
|
63
|
+
{ name: "vitest", re: /\bNo test files found\b/i },
|
|
64
|
+
{ name: "mocha", re: /^\s*0\s+passing/m },
|
|
65
|
+
{ name: "cargo", re: /^\s*running\s+0\s+tests?\s*$/m },
|
|
66
|
+
{ name: "phpunit", re: /\bNo tests executed!/i },
|
|
67
|
+
{ name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
|
|
68
|
+
{ name: "ctest", re: /\bNo tests were found\b/i },
|
|
69
|
+
{ name: "flutter", re: /\bNo tests ran\.?/i },
|
|
70
|
+
];
|
|
71
|
+
|
|
72
|
+
/** Go prints this per package that has no test files at all. */
|
|
73
|
+
const GO_NO_TEST_FILES = /\[no test files\]/;
|
|
74
|
+
/** Any sign that a Go package did run tests. */
|
|
75
|
+
const GO_RAN_SOMETHING = /^(?:ok|FAIL|---\s+(?:PASS|FAIL|SKIP)):?\s/m;
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Read a collected-test count out of a runner's output.
|
|
79
|
+
*
|
|
80
|
+
* @param {string} [stdout]
|
|
81
|
+
* @param {string} [stderr]
|
|
82
|
+
* @returns {{ count: number|null, runner: string|null }}
|
|
83
|
+
* `count` is null when no recognised runner stated one — which is not a
|
|
84
|
+
* finding, only an absence of evidence.
|
|
85
|
+
*/
|
|
86
|
+
export function parseCollectedTests(stdout = "", stderr = "") {
|
|
87
|
+
const text = `${stdout || ""}\n${stderr || ""}`;
|
|
88
|
+
if (!text.trim()) return { count: null, runner: null };
|
|
89
|
+
|
|
90
|
+
for (const rule of EXPLICIT_ZERO) {
|
|
91
|
+
if (rule.re.test(text)) return { count: 0, runner: rule.name };
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// Go states absence per package rather than as a count, so it needs its own
|
|
95
|
+
// pass before the generic patterns.
|
|
96
|
+
if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
|
|
97
|
+
// Only a run where *no* package did anything is a zero: a monorepo where
|
|
98
|
+
// one package has no tests and three do is a normal, healthy repository.
|
|
99
|
+
if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };
|
|
100
|
+
// Something ran. `--- PASS:` lines are per-test but appear only under -v,
|
|
101
|
+
// so their absence means the count was not stated — not that it was zero.
|
|
102
|
+
// Reporting zero here would have failed every ordinary `go test ./...`.
|
|
103
|
+
const perTest = text.match(GO_PER_TEST);
|
|
104
|
+
return { count: perTest && perTest.length > 0 ? perTest.length : null, runner: "go" };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
for (const rule of COUNT_PATTERNS) {
|
|
108
|
+
const m = rule.re.exec(text);
|
|
109
|
+
if (!m) continue;
|
|
110
|
+
const n = Number(m[1]);
|
|
111
|
+
if (Number.isFinite(n)) return { count: n, runner: rule.name };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
return { count: null, runner: null };
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Decide whether a passing verification command actually verified anything.
|
|
119
|
+
*
|
|
120
|
+
* @param {object} testResult - the test stage's result ({ ok, stdout, stderr, command })
|
|
121
|
+
* @param {object} [opts]
|
|
122
|
+
* @param {number} [opts.minTests=1] - the floor, from `verify.minTests`.
|
|
123
|
+
* @returns {{ ok: boolean, count: number|null, runner: string|null, reason: string|null }}
|
|
124
|
+
*/
|
|
125
|
+
export function checkCollectionFloor(testResult, opts = {}) {
|
|
126
|
+
const minTests = Number.isFinite(opts.minTests) ? opts.minTests : 1;
|
|
127
|
+
if (minTests <= 0) return { ok: true, count: null, runner: null, reason: null };
|
|
128
|
+
// Only a *passing* command can lie about this. A failing one already fails.
|
|
129
|
+
if (!testResult || testResult.ok !== true) {
|
|
130
|
+
return { ok: true, count: null, runner: null, reason: null };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
|
|
134
|
+
if (count === null || count >= minTests) {
|
|
135
|
+
return { ok: true, count, runner, reason: null };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
ok: false,
|
|
140
|
+
count,
|
|
141
|
+
runner,
|
|
142
|
+
reason:
|
|
143
|
+
`The verification command exited 0 without running any tests` +
|
|
144
|
+
(runner ? ` (${runner} reported ${count})` : "") +
|
|
145
|
+
`, so this change was approved against nothing. ` +
|
|
146
|
+
`Point verify.test at a suite that covers this repository, lower the floor with verify.minTests, ` +
|
|
147
|
+
`or — if this repository intentionally uses only the scope and secret phases — set verify.required: false.`,
|
|
148
|
+
};
|
|
149
|
+
}
|
package/src/security.mjs
CHANGED
|
@@ -1138,8 +1138,11 @@ function locateFindingLine(lines, type, file = null) {
|
|
|
1138
1138
|
// assertion — identical once every literal is blanked out — with different
|
|
1139
1139
|
// values. That does not distinguish an attack from a deliberate change of
|
|
1140
1140
|
// spec; nothing can, from a diff alone. This reports rather than decides,
|
|
1141
|
-
// and `--allow-test-
|
|
1142
|
-
// the correct one.
|
|
1141
|
+
// and `--allow-test-change expectation` is the answer when the new
|
|
1142
|
+
// expectation is the correct one. Narrow on purpose: the blunt
|
|
1143
|
+
// `--allow-test-modifications` turns off the other five checks too, and a
|
|
1144
|
+
// check that can only be answered by disabling its neighbours ends up
|
|
1145
|
+
// disabling its neighbours.
|
|
1143
1146
|
|
|
1144
1147
|
// An assertion that states a *specific* expected value. Counting assertions
|
|
1145
1148
|
// alone let a test be gutted while looking untouched: swapping
|
|
@@ -1627,6 +1630,135 @@ const hasLiteralPlaceholder = (shape) =>
|
|
|
1627
1630
|
const collapseWhitespace = (s) => s.replace(/\s+/g, " ").trim();
|
|
1628
1631
|
const shorten = (s) => (s.length > 160 ? `${s.slice(0, 157)}…` : s);
|
|
1629
1632
|
|
|
1633
|
+
/**
|
|
1634
|
+
* Split the argument list of the outermost assertion call in `clean`.
|
|
1635
|
+
*
|
|
1636
|
+
* Comments are already stripped by the caller, so only string state has to be
|
|
1637
|
+
* tracked. Returns null whenever the shape is not confidently understood — a
|
|
1638
|
+
* truncated fragment, an unbalanced hunk, a quoting form not handled here —
|
|
1639
|
+
* because every caller uses this to *suppress* a finding, and failing to
|
|
1640
|
+
* understand a statement must never become a reason to stay quiet about it.
|
|
1641
|
+
*
|
|
1642
|
+
* @param {string} clean - comment-stripped statement text
|
|
1643
|
+
* @param {string} lang
|
|
1644
|
+
* @returns {string[] | null} top-level arguments, trimmed
|
|
1645
|
+
*/
|
|
1646
|
+
function splitAssertionArgs(clean, lang) {
|
|
1647
|
+
SPECIFIC_ASSERTION.lastIndex = 0;
|
|
1648
|
+
const m = SPECIFIC_ASSERTION.exec(clean);
|
|
1649
|
+
if (!m) return null;
|
|
1650
|
+
|
|
1651
|
+
let i = m.index + m[0].length; // just past the opening paren
|
|
1652
|
+
let depth = 1;
|
|
1653
|
+
let quote = null;
|
|
1654
|
+
let triple = false;
|
|
1655
|
+
const args = [];
|
|
1656
|
+
let start = i;
|
|
1657
|
+
|
|
1658
|
+
while (i < clean.length) {
|
|
1659
|
+
const c = clean[i];
|
|
1660
|
+
|
|
1661
|
+
if (quote !== null) {
|
|
1662
|
+
if (c === "\\") { i += 2; continue; }
|
|
1663
|
+
if (triple && c === quote && clean[i + 1] === quote && clean[i + 2] === quote) {
|
|
1664
|
+
quote = null; triple = false; i += 3; continue;
|
|
1665
|
+
}
|
|
1666
|
+
if (!triple && c === quote) { quote = null; i += 1; continue; }
|
|
1667
|
+
i += 1;
|
|
1668
|
+
continue;
|
|
1669
|
+
}
|
|
1670
|
+
|
|
1671
|
+
if (c === '"' || c === "'" || c === "`") {
|
|
1672
|
+
if (lang === "python" && clean[i + 1] === c && clean[i + 2] === c) {
|
|
1673
|
+
quote = c; triple = true; i += 3; continue;
|
|
1674
|
+
}
|
|
1675
|
+
quote = c; i += 1; continue;
|
|
1676
|
+
}
|
|
1677
|
+
|
|
1678
|
+
if (c === "(" || c === "[" || c === "{") { depth += 1; i += 1; continue; }
|
|
1679
|
+
if (c === ")" || c === "]" || c === "}") {
|
|
1680
|
+
depth -= 1;
|
|
1681
|
+
if (depth === 0) {
|
|
1682
|
+
args.push(clean.slice(start, i).trim());
|
|
1683
|
+
return args;
|
|
1684
|
+
}
|
|
1685
|
+
i += 1;
|
|
1686
|
+
continue;
|
|
1687
|
+
}
|
|
1688
|
+
if (c === "," && depth === 1) {
|
|
1689
|
+
args.push(clean.slice(start, i).trim());
|
|
1690
|
+
start = i + 1;
|
|
1691
|
+
i += 1;
|
|
1692
|
+
continue;
|
|
1693
|
+
}
|
|
1694
|
+
i += 1;
|
|
1695
|
+
}
|
|
1696
|
+
return null; // never closed: an unbalanced fragment, so no suppression
|
|
1697
|
+
}
|
|
1698
|
+
|
|
1699
|
+
/** One plain string literal and nothing else. */
|
|
1700
|
+
const PURE_STRING_LITERAL = new RegExp(
|
|
1701
|
+
[
|
|
1702
|
+
"^'(?:\\\\.|[^'\\\\])*'$",
|
|
1703
|
+
'^"(?:\\\\.|[^"\\\\])*"$',
|
|
1704
|
+
"^`(?:\\\\.|[^`\\\\])*`$",
|
|
1705
|
+
'^"""[\\s\\S]*"""$',
|
|
1706
|
+
"^'''[\\s\\S]*'''$",
|
|
1707
|
+
].join("|")
|
|
1708
|
+
);
|
|
1709
|
+
|
|
1710
|
+
function isPureStringLiteral(arg) {
|
|
1711
|
+
if (!arg) return false;
|
|
1712
|
+
return PURE_STRING_LITERAL.test(arg.trim());
|
|
1713
|
+
}
|
|
1714
|
+
|
|
1715
|
+
/**
|
|
1716
|
+
* Argument positions that carry a message for a human rather than an expected
|
|
1717
|
+
* value.
|
|
1718
|
+
*
|
|
1719
|
+
* Trailing, for `assert.equal(got, want, "message")` and
|
|
1720
|
+
* `assert_eq!(a, b, "message")`; leading, for Go's
|
|
1721
|
+
* `t.Errorf("got %d want %d", got, want)`. Two arguments is the classic
|
|
1722
|
+
* `(actual, expected)` shape, so a string in last position *there* is the
|
|
1723
|
+
* expected value: `assert.equal(name, "Alice")` must still be judged when
|
|
1724
|
+
* "Alice" becomes "Bob".
|
|
1725
|
+
*/
|
|
1726
|
+
function messageArgIndices(args) {
|
|
1727
|
+
const idx = new Set();
|
|
1728
|
+
if (args.length >= 3 && isPureStringLiteral(args[args.length - 1])) idx.add(args.length - 1);
|
|
1729
|
+
if (args.length >= 2 && isPureStringLiteral(args[0])) idx.add(0);
|
|
1730
|
+
return idx;
|
|
1731
|
+
}
|
|
1732
|
+
|
|
1733
|
+
/**
|
|
1734
|
+
* True when two assertions differ only in text written to be read by a person.
|
|
1735
|
+
*
|
|
1736
|
+
* Rewording the message on a failing assertion is among the most common edits
|
|
1737
|
+
* any test file receives, and it says nothing whatsoever about what the suite
|
|
1738
|
+
* checks. But a message is a literal, so blanking literals made the two
|
|
1739
|
+
* statements the same shape and the pairing reported a rewritten expectation
|
|
1740
|
+
* every time somebody improved the wording of a failure. Firing on that is
|
|
1741
|
+
* how an operator learns to pass the override without reading it.
|
|
1742
|
+
*/
|
|
1743
|
+
function differsOnlyInMessage(cleanRemoved, cleanAdded, lang) {
|
|
1744
|
+
const a = splitAssertionArgs(cleanRemoved, lang);
|
|
1745
|
+
const b = splitAssertionArgs(cleanAdded, lang);
|
|
1746
|
+
if (!a || !b || a.length !== b.length || a.length === 0) return false;
|
|
1747
|
+
|
|
1748
|
+
const msgIdx = messageArgIndices(a);
|
|
1749
|
+
if (msgIdx.size === 0) return false;
|
|
1750
|
+
|
|
1751
|
+
let sawDifference = false;
|
|
1752
|
+
for (let i = 0; i < a.length; i++) {
|
|
1753
|
+
if (a[i].replace(/\s+/g, "") === b[i].replace(/\s+/g, "")) continue;
|
|
1754
|
+
// A difference outside a message position, or in a position that stopped
|
|
1755
|
+
// being a plain string, is a real change.
|
|
1756
|
+
if (!msgIdx.has(i) || !isPureStringLiteral(b[i])) return false;
|
|
1757
|
+
sawDifference = true;
|
|
1758
|
+
}
|
|
1759
|
+
return sawDifference;
|
|
1760
|
+
}
|
|
1761
|
+
|
|
1630
1762
|
/**
|
|
1631
1763
|
* Pair rewritten expectations across the removed and added images of every
|
|
1632
1764
|
* hunk of one file, and report each pair.
|
|
@@ -1664,14 +1796,14 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1664
1796
|
if (s.removedLines.length === 0) continue;
|
|
1665
1797
|
const clean = stripComments(s.text, lang);
|
|
1666
1798
|
if (!isSpecificAssertion(clean)) continue;
|
|
1667
|
-
oldCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
|
|
1799
|
+
oldCands.push({ s, clean, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
|
|
1668
1800
|
}
|
|
1669
1801
|
const newCands = [];
|
|
1670
1802
|
for (const s of newStmts) {
|
|
1671
1803
|
if (s.addedLines.length === 0) continue;
|
|
1672
1804
|
const clean = stripComments(s.text, lang);
|
|
1673
1805
|
if (!isSpecificAssertion(clean)) continue;
|
|
1674
|
-
newCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
|
|
1806
|
+
newCands.push({ s, clean, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
|
|
1675
1807
|
}
|
|
1676
1808
|
|
|
1677
1809
|
// The t-th removed candidate of a shape pairs with the t-th added
|
|
@@ -1698,11 +1830,32 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1698
1830
|
const pairs = [];
|
|
1699
1831
|
for (const [shape, olds] of oldByShape) {
|
|
1700
1832
|
const news = newByShape.get(shape) || [];
|
|
1701
|
-
|
|
1833
|
+
|
|
1834
|
+
// Cancel the assertions that are byte-identical on both sides before
|
|
1835
|
+
// aligning anything.
|
|
1836
|
+
//
|
|
1837
|
+
// Reordering two assertions removes both and adds both back unchanged.
|
|
1838
|
+
// Positional alignment then matched the first removed against the first
|
|
1839
|
+
// added — a different assertion — and reported two rewritten
|
|
1840
|
+
// expectations for an edit that changed no expected value at all. The
|
|
1841
|
+
// same happened to an assertion that simply moved within its block.
|
|
1842
|
+
// What is present unchanged on both sides did not change; only the
|
|
1843
|
+
// residue can have been rewritten.
|
|
1844
|
+
const survivingNew = news.slice();
|
|
1845
|
+
const survivingOld = [];
|
|
1846
|
+
for (const o of olds) {
|
|
1847
|
+
const twin = survivingNew.findIndex((n) => n.canon === o.canon);
|
|
1848
|
+
if (twin === -1) survivingOld.push(o);
|
|
1849
|
+
else survivingNew.splice(twin, 1);
|
|
1850
|
+
}
|
|
1851
|
+
|
|
1852
|
+
const k = Math.min(survivingOld.length, survivingNew.length);
|
|
1702
1853
|
for (let t = 0; t < k; t++) {
|
|
1703
|
-
|
|
1704
|
-
|
|
1705
|
-
|
|
1854
|
+
const r = survivingOld[t];
|
|
1855
|
+
const a = survivingNew[t];
|
|
1856
|
+
if (r.canon === a.canon) continue;
|
|
1857
|
+
if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
|
|
1858
|
+
pairs.push({ r: r.s, a: a.s });
|
|
1706
1859
|
}
|
|
1707
1860
|
}
|
|
1708
1861
|
|
|
@@ -1717,9 +1870,11 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1717
1870
|
const sr = blankLiterals(stripComments(r.text, lang));
|
|
1718
1871
|
const sa = blankLiterals(stripComments(a.text, lang));
|
|
1719
1872
|
if (sr === sa && hasLiteralPlaceholder(sr)) {
|
|
1720
|
-
const
|
|
1721
|
-
const
|
|
1722
|
-
if (
|
|
1873
|
+
const clr = stripComments(r.text, lang);
|
|
1874
|
+
const cla = stripComments(a.text, lang);
|
|
1875
|
+
if (clr.replace(/\s+/g, "") !== cla.replace(/\s+/g, "") && !differsOnlyInMessage(clr, cla, lang)) {
|
|
1876
|
+
pairs.push({ r, a });
|
|
1877
|
+
}
|
|
1723
1878
|
}
|
|
1724
1879
|
}
|
|
1725
1880
|
}
|
|
@@ -1736,9 +1891,10 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1736
1891
|
`"${shorten(collapseWhitespace(p.r.text))}" became "${shorten(collapseWhitespace(p.a.text))}". ` +
|
|
1737
1892
|
`A deliberately changed spec looks identical to a test bent to match broken ` +
|
|
1738
1893
|
`output, and a diff alone cannot tell the two apart, so this is flagged for ` +
|
|
1739
|
-
`review rather than assumed. If the new expectation is correct,
|
|
1740
|
-
|
|
1741
|
-
`and
|
|
1894
|
+
`review rather than assumed. If the new expectation is the correct one, ` +
|
|
1895
|
+
`re-run with --allow-test-change expectation — which allows exactly this ` +
|
|
1896
|
+
`check and leaves the skip, vacuous, commented, removal and weakening ` +
|
|
1897
|
+
`checks doing their job.`,
|
|
1742
1898
|
});
|
|
1743
1899
|
|
|
1744
1900
|
// Both sides are accounted for here, so they must not also feed the
|
|
@@ -1771,17 +1927,85 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
|
|
|
1771
1927
|
return allPairs;
|
|
1772
1928
|
}
|
|
1773
1929
|
|
|
1930
|
+
/**
|
|
1931
|
+
* The tamper checks, by the name an operator uses to allow one of them.
|
|
1932
|
+
*
|
|
1933
|
+
* There was one override for all six, and it was a switch marked "off". A
|
|
1934
|
+
* deliberate change of spec rewrites what a test expects, which is
|
|
1935
|
+
* indistinguishable from bending a test to match broken output — so the honest
|
|
1936
|
+
* answer to that finding is sometimes an override. But reaching for it also
|
|
1937
|
+
* silenced injected `.skip()`, `expect(true).toBe(true)`, commented-out
|
|
1938
|
+
* assertions and outright deletions, none of which the operator had looked at.
|
|
1939
|
+
* The check with the highest firing rate therefore set the ceiling for every
|
|
1940
|
+
* other check in the bundle: the more useful this one became, the more often
|
|
1941
|
+
* it would be used to turn the others off.
|
|
1942
|
+
*/
|
|
1943
|
+
export const TAMPER_KINDS = new Map([
|
|
1944
|
+
["TEST_SKIP_INJECTION", "skip"],
|
|
1945
|
+
["VACUOUS_ASSERTION", "vacuous"],
|
|
1946
|
+
["COMMENTED_ASSERTION", "commented"],
|
|
1947
|
+
["ASSERTION_REMOVAL", "removal"],
|
|
1948
|
+
["ASSERTION_WEAKENED", "weakening"],
|
|
1949
|
+
["ASSERTION_EXPECTATION_CHANGED", "expectation"],
|
|
1950
|
+
]);
|
|
1951
|
+
|
|
1952
|
+
/** Every kind name, for CLI validation and help text. */
|
|
1953
|
+
export const TAMPER_KIND_NAMES = Object.freeze([...new Set(TAMPER_KINDS.values())].sort());
|
|
1954
|
+
|
|
1955
|
+
/**
|
|
1956
|
+
* Which tamper checks this run is allowed to stay quiet about.
|
|
1957
|
+
*
|
|
1958
|
+
* @param {object} options
|
|
1959
|
+
* @param {boolean} [options.allowTestModifications] - the blunt form: all of them.
|
|
1960
|
+
* @param {string|string[]} [options.allowTestChanges] - kind names, comma-separated or an array.
|
|
1961
|
+
* @returns {{ all: boolean, kinds: Set<string>, unknown: string[] }}
|
|
1962
|
+
*/
|
|
1963
|
+
export function resolveAllowedTamperKinds(options = {}) {
|
|
1964
|
+
if (options.allowTestModifications === true) {
|
|
1965
|
+
return { all: true, kinds: new Set(TAMPER_KIND_NAMES), unknown: [] };
|
|
1966
|
+
}
|
|
1967
|
+
const raw = options.allowTestChanges;
|
|
1968
|
+
const list = (Array.isArray(raw) ? raw : [raw])
|
|
1969
|
+
.flatMap((v) => String(v == null ? "" : v).split(","))
|
|
1970
|
+
.map((v) => v.trim().toLowerCase())
|
|
1971
|
+
.filter(Boolean);
|
|
1972
|
+
|
|
1973
|
+
if (list.includes("all")) {
|
|
1974
|
+
return { all: true, kinds: new Set(TAMPER_KIND_NAMES), unknown: [] };
|
|
1975
|
+
}
|
|
1976
|
+
const kinds = new Set();
|
|
1977
|
+
const unknown = [];
|
|
1978
|
+
for (const name of list) {
|
|
1979
|
+
if (TAMPER_KIND_NAMES.includes(name)) kinds.add(name);
|
|
1980
|
+
else unknown.push(name);
|
|
1981
|
+
}
|
|
1982
|
+
return { all: false, kinds, unknown };
|
|
1983
|
+
}
|
|
1984
|
+
|
|
1774
1985
|
/**
|
|
1775
1986
|
* Detects test file assertion tampering, weakening, or test skips.
|
|
1776
1987
|
*
|
|
1777
1988
|
* @param {string} diffOrText - Unified git diff
|
|
1778
1989
|
* @param {Object} [options]
|
|
1779
1990
|
* @param {boolean} [options.allowTestModifications=false]
|
|
1780
|
-
* @returns {{ ok: boolean, violations: Array<
|
|
1991
|
+
* @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, status: "PASS"|"FAIL"|"NOT_APPLICABLE" }}
|
|
1992
|
+
* `status` distinguishes "checked and clean" from "nothing was checked";
|
|
1993
|
+
* `ok: true` alone cannot, and that ambiguity is the defect class this
|
|
1994
|
+
* field exists to make visible.
|
|
1781
1995
|
*/
|
|
1782
1996
|
export function checkTestTampering(diffOrText = "", options = {}) {
|
|
1783
|
-
if (!diffOrText || typeof diffOrText !== "string")
|
|
1784
|
-
|
|
1997
|
+
if (!diffOrText || typeof diffOrText !== "string") {
|
|
1998
|
+
return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "empty diff" };
|
|
1999
|
+
}
|
|
2000
|
+
const allowed = resolveAllowedTamperKinds(options);
|
|
2001
|
+
if (allowed.all) {
|
|
2002
|
+
return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "all kinds allowed" };
|
|
2003
|
+
}
|
|
2004
|
+
|
|
2005
|
+
// Which predicate decides what this guard even looks at. Injectable so the
|
|
2006
|
+
// meta-check can mutate it: a canary that still passes when the predicate is
|
|
2007
|
+
// replaced by `() => false` was never requiring this guard to activate.
|
|
2008
|
+
const isTestPath_ = typeof options.isTestPath === "function" ? options.isTestPath : isTestPath;
|
|
1785
2009
|
|
|
1786
2010
|
const violations = [];
|
|
1787
2011
|
const lines = diffOrText.split("\n");
|
|
@@ -1790,7 +2014,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1790
2014
|
let currentOldLineNo = null;
|
|
1791
2015
|
let currentNewLineNo = null;
|
|
1792
2016
|
|
|
1793
|
-
const isTestFile =
|
|
2017
|
+
const isTestFile = isTestPath_;
|
|
1794
2018
|
|
|
1795
2019
|
const SKIP_INJECTIONS = [
|
|
1796
2020
|
{ pattern: /\b(?:it|test|describe|context)\.skip\s*\(/i, desc: "Injected test skip (.skip())" },
|
|
@@ -1987,9 +2211,31 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
1987
2211
|
}
|
|
1988
2212
|
}
|
|
1989
2213
|
|
|
2214
|
+
// A kind the operator has already looked at and accepted is dropped here
|
|
2215
|
+
// rather than never being computed, so the reasoning above stays one code
|
|
2216
|
+
// path regardless of what any given run allows.
|
|
2217
|
+
const reported =
|
|
2218
|
+
allowed.kinds.size === 0
|
|
2219
|
+
? violations
|
|
2220
|
+
: violations.filter((v) => !allowed.kinds.has(TAMPER_KINDS.get(v.type)));
|
|
2221
|
+
|
|
2222
|
+
// What was examined, not only what was found.
|
|
2223
|
+
//
|
|
2224
|
+
// `ok: true` from a guard that looked at nothing is byte-identical to
|
|
2225
|
+
// `ok: true` from a guard that looked at everything and approved it. That
|
|
2226
|
+
// ambiguity is how a substring bug in the file classifier switched this
|
|
2227
|
+
// entire guard off for the standard pytest, Rust and RSpec layouts while
|
|
2228
|
+
// every signal stayed green. A verdict without a denominator is not a
|
|
2229
|
+
// verdict, so `inputsSeen` reports the number of test files this run
|
|
2230
|
+
// actually reasoned about, and `status` distinguishes "nothing to check"
|
|
2231
|
+
// from "checked and clean".
|
|
2232
|
+
const inputsSeen = fileAssertions.size;
|
|
2233
|
+
|
|
1990
2234
|
return {
|
|
1991
|
-
ok:
|
|
1992
|
-
violations,
|
|
2235
|
+
ok: reported.length === 0,
|
|
2236
|
+
violations: reported,
|
|
2237
|
+
inputsSeen,
|
|
2238
|
+
status: reported.length > 0 ? "FAIL" : inputsSeen > 0 ? "PASS" : "NOT_APPLICABLE",
|
|
1993
2239
|
};
|
|
1994
2240
|
}
|
|
1995
2241
|
|
package/src/stack-detector.mjs
CHANGED
|
@@ -39,7 +39,65 @@ export function pytestCmd(env = process.env) {
|
|
|
39
39
|
}
|
|
40
40
|
|
|
41
41
|
/**
|
|
42
|
-
*
|
|
42
|
+
* Does this Makefile declare a `test` target?
|
|
43
|
+
*
|
|
44
|
+
* Read rather than assumed: the presence of the file says nothing about
|
|
45
|
+
* whether `make test` will run.
|
|
46
|
+
*/
|
|
47
|
+
function makefileHasTestTarget(root) {
|
|
48
|
+
try {
|
|
49
|
+
const text = readFileSync(join(root, "Makefile"), "utf-8");
|
|
50
|
+
return /^\.PHONY:.*\btest\b/m.test(text) || /^test\s*:/m.test(text);
|
|
51
|
+
} catch (_) {
|
|
52
|
+
return false;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* A declared test script that runs no tests and exits 0.
|
|
58
|
+
*
|
|
59
|
+
* This is the single most dangerous input the gate can receive, because every
|
|
60
|
+
* downstream check reads "a command ran and passed". `bootstrapZeroTestRepo`
|
|
61
|
+
* called `"test": "echo 'no tests yet' && exit 0"` an
|
|
62
|
+
* EXISTING_VERIFICATION_ORACLE — it asked whether the field was set, never
|
|
63
|
+
* what was in it.
|
|
64
|
+
*
|
|
65
|
+
* npm's own default (`echo "Error: no test specified" && exit 1`) is not a
|
|
66
|
+
* placeholder by this definition, and correctly so: it exits non-zero, which
|
|
67
|
+
* fails loudly rather than certifying nothing.
|
|
68
|
+
*/
|
|
69
|
+
export function isPlaceholderTestScript(cmd) {
|
|
70
|
+
if (typeof cmd !== "string") return false;
|
|
71
|
+
const trimmed = cmd.trim();
|
|
72
|
+
if (!trimmed) return true;
|
|
73
|
+
// Drop the announcements; what matters is what the shell is left doing.
|
|
74
|
+
const remainder = trimmed
|
|
75
|
+
.split(/&&|;/)
|
|
76
|
+
.map((part) => part.trim())
|
|
77
|
+
.filter((part) => part && !/^(?:echo|printf|:)\b/.test(part));
|
|
78
|
+
if (remainder.length === 0) return true;
|
|
79
|
+
return remainder.every((part) => /^(?:exit\s+0|true|:)$/.test(part));
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Generate the fallback verification oracle for a JS/generic repo with no tests.
|
|
84
|
+
*
|
|
85
|
+
* What this used to write could not fail:
|
|
86
|
+
*
|
|
87
|
+
* assert.ok(fs.existsSync(process.cwd()));
|
|
88
|
+
* assert.ok(fs.readdirSync(process.cwd()).length > 0);
|
|
89
|
+
*
|
|
90
|
+
* Both hold for every repository and every change, so the generated "oracle"
|
|
91
|
+
* was green against arbitrary broken code — and worse, it *silenced* the
|
|
92
|
+
* `missingOracle` guard in engine.mjs, which fires only when no command ran at
|
|
93
|
+
* all. A repository that honestly had no oracle was converted into one that
|
|
94
|
+
* claimed to have one. That is the tool writing its own blindness to disk.
|
|
95
|
+
*
|
|
96
|
+
* The other stacks already get a real static gate at this point — `tsc
|
|
97
|
+
* --noEmit`, `cargo check`, `go vet`, `compileall` — each of which fails on a
|
|
98
|
+
* real class of defect. This is the JavaScript equivalent: every source file
|
|
99
|
+
* must parse. It proves the code compiles, not that it works, and the caller
|
|
100
|
+
* says so; but a syntax error fails it, which is one more than before.
|
|
43
101
|
*/
|
|
44
102
|
export function generateSmokeTestScript(root = process.cwd()) {
|
|
45
103
|
const agentDir = join(root, ".agent");
|
|
@@ -48,15 +106,45 @@ export function generateSmokeTestScript(root = process.cwd()) {
|
|
|
48
106
|
} catch (_) {}
|
|
49
107
|
|
|
50
108
|
const smokePath = join(agentDir, "smoke.test.mjs");
|
|
51
|
-
const content = `// Auto-generated zero-dependency
|
|
109
|
+
const content = `// Auto-generated zero-dependency parse gate (.agent/smoke.test.mjs)
|
|
110
|
+
//
|
|
111
|
+
// Written by \`agentctl bootstrap\` for a repository that had no test suite.
|
|
112
|
+
// It proves that every source file still parses. It does NOT prove the code
|
|
113
|
+
// is correct — replace it with real tests as soon as there are any.
|
|
52
114
|
import { test } from "node:test";
|
|
53
115
|
import assert from "node:assert/strict";
|
|
54
|
-
import
|
|
116
|
+
import { readdirSync, statSync } from "node:fs";
|
|
117
|
+
import { join, extname } from "node:path";
|
|
118
|
+
import { spawnSync } from "node:child_process";
|
|
119
|
+
|
|
120
|
+
const SKIP = new Set([".git", "node_modules", "vendor", "dist", "build", "coverage", ".venv", "venv", ".next", ".agent"]);
|
|
121
|
+
const SOURCE = new Set([".js", ".mjs", ".cjs"]);
|
|
122
|
+
|
|
123
|
+
function sources(dir, acc = [], depth = 0) {
|
|
124
|
+
if (depth > 8) return acc;
|
|
125
|
+
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
|
126
|
+
if (entry.name.startsWith(".") && entry.name !== ".agent") continue;
|
|
127
|
+
if (SKIP.has(entry.name)) continue;
|
|
128
|
+
const full = join(dir, entry.name);
|
|
129
|
+
if (entry.isDirectory()) sources(full, acc, depth + 1);
|
|
130
|
+
else if (SOURCE.has(extname(entry.name))) acc.push(full);
|
|
131
|
+
}
|
|
132
|
+
return acc;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
test("every source file parses", () => {
|
|
136
|
+
const files = sources(process.cwd());
|
|
55
137
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
138
|
+
// A gate with nothing to check is not a passing gate. Reporting success
|
|
139
|
+
// over an empty file list is exactly the vacuous oracle this replaced.
|
|
140
|
+
assert.ok(files.length > 0, "No JavaScript sources found to verify — this is not an oracle. Set verify.test in .agent/config.yml.");
|
|
141
|
+
|
|
142
|
+
const broken = [];
|
|
143
|
+
for (const file of files) {
|
|
144
|
+
const res = spawnSync(process.execPath, ["--check", file], { encoding: "utf-8" });
|
|
145
|
+
if (res.status !== 0) broken.push(\`\${file}: \${(res.stderr || "").trim().split("\\n")[0]}\`);
|
|
146
|
+
}
|
|
147
|
+
assert.deepEqual(broken, [], \`\${broken.length} file(s) failed to parse\`);
|
|
60
148
|
});
|
|
61
149
|
`;
|
|
62
150
|
writeFileSync(smokePath, content, "utf-8");
|
|
@@ -173,7 +261,15 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
173
261
|
if (existsSync(join(projectRoot, "Package.swift"))) {
|
|
174
262
|
return { ...container, stack: "swift", testCmd: "swift test", buildCmd: "swift build", triggerFile: "Package.swift" };
|
|
175
263
|
}
|
|
176
|
-
|
|
264
|
+
// `app.json` is not a manifest, it is an Expo/React Native *configuration*
|
|
265
|
+
// file, and the name is generic enough that unrelated projects use it. Its
|
|
266
|
+
// test command is `npm test`, so without a package.json beside it the
|
|
267
|
+
// detector was claiming a Node stack for a repository that has no Node in
|
|
268
|
+
// it: `Cargo.toml` + `app.json` was measured as `react-native` / `npm test`.
|
|
269
|
+
if (
|
|
270
|
+
(existsSync(join(projectRoot, "app.json")) && existsSync(join(projectRoot, "package.json"))) ||
|
|
271
|
+
existsSync(join(projectRoot, "react-native.config.js"))
|
|
272
|
+
) {
|
|
177
273
|
const triggerFile = existsSync(join(projectRoot, "app.json")) ? "app.json" : "react-native.config.js";
|
|
178
274
|
return { ...container, stack: "react-native", testCmd: "npm test", buildCmd: "npx react-native bundle --platform android --dev false --entry-file index.js --bundle-output android/main.jsbundle", triggerFile };
|
|
179
275
|
}
|
|
@@ -188,7 +284,14 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
188
284
|
if (existsSync(join(projectRoot, "go.mod"))) {
|
|
189
285
|
return { ...container, stack: "go", testCmd: "go test ./...", buildCmd: "go build ./...", triggerFile: "go.mod" };
|
|
190
286
|
}
|
|
191
|
-
|
|
287
|
+
// A Makefile is only an oracle if it declares the target we are about to
|
|
288
|
+
// run. `make test` on a Makefile with only a `build:` target exits 2 with
|
|
289
|
+
// "No rule to make target 'test'" — measured on a repository whose
|
|
290
|
+
// package.json declared a perfectly good `vitest run`, because the Makefile
|
|
291
|
+
// was checked first and the presence of the *file* was the whole test. A
|
|
292
|
+
// hard red on day one is how a user learns the gate is broken and turns it
|
|
293
|
+
// off, so the file must earn the claim.
|
|
294
|
+
if (existsSync(join(projectRoot, "Makefile")) && makefileHasTestTarget(projectRoot)) {
|
|
192
295
|
return { ...container, stack: "make", testCmd: "make test", buildCmd: "make build", triggerFile: "Makefile" };
|
|
193
296
|
}
|
|
194
297
|
|
|
@@ -715,7 +818,7 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
|
|
|
715
818
|
if (existsSync(join(root, "package.json"))) {
|
|
716
819
|
try {
|
|
717
820
|
const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
|
|
718
|
-
hasPkgTest = Boolean(pkg?.scripts?.test);
|
|
821
|
+
hasPkgTest = Boolean(pkg?.scripts?.test) && !isPlaceholderTestScript(pkg.scripts.test);
|
|
719
822
|
} catch (_) {}
|
|
720
823
|
}
|
|
721
824
|
|
package/src/task-optimizer.mjs
CHANGED
|
@@ -265,9 +265,13 @@ export function scorePromptFalsifiability(promptText, options = {}) {
|
|
|
265
265
|
let isTrivial = false;
|
|
266
266
|
|
|
267
267
|
if (!verifyCmd) {
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
268
|
+
// `detectStackOracles` returns an object, not an array: `.length` was
|
|
269
|
+
// undefined, so this branch never ran and every task without an explicit
|
|
270
|
+
// --verify-cmd was scored MISSING_ORACLE even in a repository with a
|
|
271
|
+
// working suite.
|
|
272
|
+
const detected = detectStackOracles(rootDir);
|
|
273
|
+
if (detected?.candidates?.testCmd) {
|
|
274
|
+
verifyCmd = detected.candidates.testCmd;
|
|
271
275
|
autoDetected = true;
|
|
272
276
|
}
|
|
273
277
|
}
|