jules-orchestrator-kit 0.60.0 → 0.63.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/package.json +2 -1
- package/scripts/guard-reach-check.mjs +176 -0
- package/scripts/release.mjs +17 -0
- package/src/config.mjs +68 -0
- package/src/coverage.mjs +17 -2
- package/src/engine.mjs +33 -2
- package/src/evidence.mjs +38 -1
- package/src/ops/test-collection.mjs +149 -0
- package/src/security.mjs +30 -4
- package/src/stack-detector.mjs +113 -10
- package/src/task-optimizer.mjs +7 -3
package/README.md
CHANGED
|
@@ -204,7 +204,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
|
|
|
204
204
|
* **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
|
|
205
205
|
* **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
|
|
206
206
|
* **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
|
|
207
|
-
* **Verified Test Suite:** Tested with **
|
|
207
|
+
* **Verified Test Suite:** Tested with **1015 unit tests across 145 suites passing in < 15.0s**.
|
|
208
208
|
|
|
209
209
|
<br/>
|
|
210
210
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "jules-orchestrator-kit",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.63.0",
|
|
4
4
|
"description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
|
|
5
5
|
"repository": {
|
|
6
6
|
"type": "git",
|
|
@@ -55,6 +55,7 @@
|
|
|
55
55
|
"jules:rules-lint": "node scripts/rules-lint.mjs",
|
|
56
56
|
"jules:doc-sync": "node scripts/doc-sync-check.mjs",
|
|
57
57
|
"release": "node scripts/release.mjs",
|
|
58
|
+
"guard-reach": "node scripts/guard-reach-check.mjs",
|
|
58
59
|
"lint": "eslint ."
|
|
59
60
|
},
|
|
60
61
|
"engines": {
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* Activation coverage: proof that every blocking check can still be made red.
|
|
5
|
+
*
|
|
6
|
+
* A defect that turns a check off cannot be found by the check it turns off.
|
|
7
|
+
* That is not a hypothetical — `isTestFile` matched the substring `/test/`,
|
|
8
|
+
* which does not occur in `tests/test_calc.py`, so the entire tamper guard was
|
|
9
|
+
* silent for the standard pytest, Rust and RSpec layouts. Every mechanism that
|
|
10
|
+
* should have caught it was working exactly as designed:
|
|
11
|
+
*
|
|
12
|
+
* - the unit suite sampled the same distribution the implementation was
|
|
13
|
+
* written from, so its fixtures re-confirmed the dialect it already knew;
|
|
14
|
+
* - the doc-sync gate compares counts and versions, and a guard that guards
|
|
15
|
+
* nothing still contributes passing tests;
|
|
16
|
+
* - the nine-way CI matrix varies OS and Node version — dimensions
|
|
17
|
+
* orthogonal to the defect. Nine runs of `test/foo.test.js` never explore
|
|
18
|
+
* `tests/test_calc.py`;
|
|
19
|
+
* - cold review reads the code against its stated intent, and here the code
|
|
20
|
+
* and the intent agreed. The eye supplies the leading slash;
|
|
21
|
+
* - the release gate is a conjunction over those four, and a signal that
|
|
22
|
+
* silently goes absent contributes `true`.
|
|
23
|
+
*
|
|
24
|
+
* The common property: `ok: true` from a check that examined nothing is
|
|
25
|
+
* byte-identical to `ok: true` from a check that examined everything. There is
|
|
26
|
+
* no denominator. This script supplies one.
|
|
27
|
+
*
|
|
28
|
+
* Three steps, all in-process, no dependencies, well under a second:
|
|
29
|
+
*
|
|
30
|
+
* 1. POLICY — the hand-written witness table in test/fixtures/guard-policy.mjs
|
|
31
|
+
* must hold. It is derived from what the tool advertises, never
|
|
32
|
+
* from the regexes that implement it.
|
|
33
|
+
* 2. CANARIES — every known-bad input must produce the finding it names. A
|
|
34
|
+
* canary that comes back clean is not a pass; it is proof that
|
|
35
|
+
* the rule stopped being reachable.
|
|
36
|
+
* 3. MUTANTS — each hand-written mutant of the applicability predicate must
|
|
37
|
+
* kill at least one canary. A surviving mutant means no canary
|
|
38
|
+
* ever required the guard to activate, so the suite would stay
|
|
39
|
+
* green if it silently stopped looking.
|
|
40
|
+
*
|
|
41
|
+
* Usage: node scripts/guard-reach-check.mjs [--json]
|
|
42
|
+
* Exit codes: 0 = every guard reachable, 1 = a guard has gone silent.
|
|
43
|
+
*/
|
|
44
|
+
|
|
45
|
+
import { checkTestTampering, checkScope } from "../src/security.mjs";
|
|
46
|
+
import { isTestPath } from "../src/test-paths.mjs";
|
|
47
|
+
import { normalizeScope } from "../src/config.mjs";
|
|
48
|
+
import { parseCollectedTests } from "../src/ops/test-collection.mjs";
|
|
49
|
+
import {
|
|
50
|
+
TEST_PATH_CASES,
|
|
51
|
+
TAMPER_CANARIES,
|
|
52
|
+
PREDICATE_MUTANTS,
|
|
53
|
+
EMPTY_RUN_CANARIES,
|
|
54
|
+
SCOPE_CANARIES,
|
|
55
|
+
} from "../test/fixtures/guard-policy.mjs";
|
|
56
|
+
|
|
57
|
+
/** Build a unified diff for one canary. */
|
|
58
|
+
function canaryDiff(c) {
|
|
59
|
+
const lines = [`--- a/${c.file}`, `+++ b/${c.file}`, "@@ -1,20 +1,20 @@", " // context"];
|
|
60
|
+
for (const l of c.removed) lines.push(`-${l}`);
|
|
61
|
+
for (const l of c.added) lines.push(`+${l}`);
|
|
62
|
+
lines.push(" // context");
|
|
63
|
+
return lines.join("\n");
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const failures = [];
|
|
67
|
+
const checks = [];
|
|
68
|
+
const add = (name, ok, detail) => {
|
|
69
|
+
checks.push({ name, ok, detail });
|
|
70
|
+
if (!ok) failures.push(`${name}: ${detail}`);
|
|
71
|
+
};
|
|
72
|
+
|
|
73
|
+
// --- 1. Policy contract -----------------------------------------------------
|
|
74
|
+
{
|
|
75
|
+
const wrong = TEST_PATH_CASES.filter((c) => isTestPath(c.path) !== c.expected);
|
|
76
|
+
add(
|
|
77
|
+
"policy: test-path domain",
|
|
78
|
+
wrong.length === 0,
|
|
79
|
+
wrong.length
|
|
80
|
+
? wrong.map((c) => `${c.path} → ${isTestPath(c.path)}, policy says ${c.expected} (${c.why})`).join("; ")
|
|
81
|
+
: `${TEST_PATH_CASES.length} witnesses hold`
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
{
|
|
86
|
+
const scope = normalizeScope({ deny: [], allow: [], protect: [] });
|
|
87
|
+
const wrong = [];
|
|
88
|
+
for (const c of SCOPE_CANARIES) {
|
|
89
|
+
const res = checkScope([c.path], scope);
|
|
90
|
+
const rule = res.ok ? "none" : res.violations[0].rule;
|
|
91
|
+
if (rule !== c.rule) wrong.push(`${c.path} → ${rule}, policy says ${c.rule} (${c.why})`);
|
|
92
|
+
}
|
|
93
|
+
add("policy: scope tiers", wrong.length === 0, wrong.length ? wrong.join("; ") : `${SCOPE_CANARIES.length} paths tiered as declared`);
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
{
|
|
97
|
+
const missed = EMPTY_RUN_CANARIES.filter((c) => parseCollectedTests(c.output, "").count !== 0);
|
|
98
|
+
add(
|
|
99
|
+
"policy: empty-run detection",
|
|
100
|
+
missed.length === 0,
|
|
101
|
+
missed.length ? `${missed.map((m) => m.id).join(", ")} report zero tests in a spelling the floor cannot read` : `${EMPTY_RUN_CANARIES.length} runners recognised`
|
|
102
|
+
);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
// --- 2. Canaries ------------------------------------------------------------
|
|
106
|
+
const canaryResults = new Map();
|
|
107
|
+
{
|
|
108
|
+
const silent = [];
|
|
109
|
+
const noDenominator = [];
|
|
110
|
+
for (const c of TAMPER_CANARIES) {
|
|
111
|
+
const res = checkTestTampering(canaryDiff(c));
|
|
112
|
+
const hit = (res.violations || []).some((v) => v.type === c.expect);
|
|
113
|
+
canaryResults.set(c.id, hit);
|
|
114
|
+
if (!hit) silent.push(`${c.id} expected ${c.expect}, got ${JSON.stringify((res.violations || []).map((v) => v.type))}`);
|
|
115
|
+
// A finding with no denominator is the shape this script exists to reject.
|
|
116
|
+
if (hit && !(res.inputsSeen > 0)) noDenominator.push(c.id);
|
|
117
|
+
}
|
|
118
|
+
add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length} canaries red as required`);
|
|
119
|
+
add("canaries: every finding carries a denominator", noDenominator.length === 0, noDenominator.length ? noDenominator.join(", ") : "inputsSeen > 0 on every hit");
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
// --- 3. Predicate mutants ---------------------------------------------------
|
|
123
|
+
{
|
|
124
|
+
const survivors = [];
|
|
125
|
+
for (const mutant of PREDICATE_MUTANTS) {
|
|
126
|
+
let killed = false;
|
|
127
|
+
for (const c of TAMPER_CANARIES) {
|
|
128
|
+
// Only canaries the healthy predicate catches can kill a mutant.
|
|
129
|
+
if (!canaryResults.get(c.id)) continue;
|
|
130
|
+
const res = checkTestTampering(canaryDiff(c), { isTestPath: mutant.fn });
|
|
131
|
+
if (!(res.violations || []).some((v) => v.type === c.expect)) {
|
|
132
|
+
killed = true;
|
|
133
|
+
break;
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
if (!killed) survivors.push(`${mutant.id} (${mutant.why})`);
|
|
137
|
+
}
|
|
138
|
+
add(
|
|
139
|
+
"mutants: blinding the predicate breaks a canary",
|
|
140
|
+
survivors.length === 0,
|
|
141
|
+
survivors.length
|
|
142
|
+
? `survived: ${survivors.join(", ")} — no canary required the guard to activate`
|
|
143
|
+
: `${PREDICATE_MUTANTS.length} mutants killed`
|
|
144
|
+
);
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
// --- Report -----------------------------------------------------------------
|
|
148
|
+
const json = process.argv.includes("--json");
|
|
149
|
+
const activated = [...canaryResults.values()].filter(Boolean).length;
|
|
150
|
+
|
|
151
|
+
if (json) {
|
|
152
|
+
console.log(
|
|
153
|
+
JSON.stringify(
|
|
154
|
+
{
|
|
155
|
+
ok: failures.length === 0,
|
|
156
|
+
checks,
|
|
157
|
+
activationCoverage: { canaries: canaryResults.size, activated },
|
|
158
|
+
},
|
|
159
|
+
null,
|
|
160
|
+
2
|
|
161
|
+
)
|
|
162
|
+
);
|
|
163
|
+
} else {
|
|
164
|
+
console.log("\n🎯 Guard Reach Check (activation coverage)");
|
|
165
|
+
console.log("-------------------------------------------------------");
|
|
166
|
+
for (const c of checks) console.log(` ${c.ok ? "✅" : "❌"} ${c.name.padEnd(46)} ${c.detail}`);
|
|
167
|
+
console.log("-------------------------------------------------------");
|
|
168
|
+
console.log(` canaries activated: ${activated}/${canaryResults.size}`);
|
|
169
|
+
console.log(
|
|
170
|
+
failures.length === 0
|
|
171
|
+
? "✅ Every blocking guard can still be made red.\n"
|
|
172
|
+
: `\n❌ ${failures.length} guard(s) may have gone silent. A check that cannot be made red is not a check.\n`
|
|
173
|
+
);
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
process.exit(failures.length === 0 ? 0 : 1);
|
package/scripts/release.mjs
CHANGED
|
@@ -41,6 +41,23 @@ try {
|
|
|
41
41
|
process.exit(1);
|
|
42
42
|
}
|
|
43
43
|
|
|
44
|
+
// 1a. Activation coverage (blocking).
|
|
45
|
+
//
|
|
46
|
+
// Step 1 proved the suite is green. Green is only evidence if the guards were
|
|
47
|
+
// switched on: a check that silently stopped applying contributes passing
|
|
48
|
+
// tests and a zero exit code exactly like one that ran. This asks the question
|
|
49
|
+
// the suite cannot — can every blocking guard still be made red? — and it runs
|
|
50
|
+
// before the doc-sync gate because a silent guard makes every later signal
|
|
51
|
+
// meaningless.
|
|
52
|
+
console.log("1a. Verifying every blocking guard can still be made red...");
|
|
53
|
+
try {
|
|
54
|
+
execSync("node scripts/guard-reach-check.mjs", { cwd: root, stdio: "inherit" });
|
|
55
|
+
console.log("");
|
|
56
|
+
} catch (_) {
|
|
57
|
+
console.error("\n❌ Release Aborted: a guard has gone silent. See the failing rows above.");
|
|
58
|
+
process.exit(1);
|
|
59
|
+
}
|
|
60
|
+
|
|
44
61
|
// 1b. Documentation / version consistency gate (blocking).
|
|
45
62
|
console.log("1b. Verifying documentation is in sync with package.json & test suite...");
|
|
46
63
|
{
|
package/src/config.mjs
CHANGED
|
@@ -54,6 +54,9 @@ const CI_DEFINITIONS = [
|
|
|
54
54
|
"appveyor.yml",
|
|
55
55
|
".teamcity/**",
|
|
56
56
|
".githooks/**",
|
|
57
|
+
"buildspec.yml",
|
|
58
|
+
"**/buildspec.yml",
|
|
59
|
+
".buildspec/**",
|
|
57
60
|
];
|
|
58
61
|
|
|
59
62
|
export const BUILTIN_DENY = [
|
|
@@ -64,6 +67,24 @@ export const BUILTIN_DENY = [
|
|
|
64
67
|
"**/*.key",
|
|
65
68
|
"**/id_rsa*",
|
|
66
69
|
".agent/jules-queue/**",
|
|
70
|
+
|
|
71
|
+
// Shell that runs on `cd`, and credentials in plaintext. Same class as a CI
|
|
72
|
+
// definition: code or secrets that take effect before anyone reviews them.
|
|
73
|
+
"**/.envrc",
|
|
74
|
+
"**/.git-credentials",
|
|
75
|
+
"**/.aws/**",
|
|
76
|
+
"**/.ssh/**",
|
|
77
|
+
"**/.kube/**",
|
|
78
|
+
"**/kubeconfig*",
|
|
79
|
+
"**/.docker/config.json",
|
|
80
|
+
"**/*.p12",
|
|
81
|
+
"**/*.pfx",
|
|
82
|
+
"**/*.p8",
|
|
83
|
+
"**/id_ed25519*",
|
|
84
|
+
"**/credentials.json",
|
|
85
|
+
"**/service-account*.json",
|
|
86
|
+
"**/*.tfstate",
|
|
87
|
+
"**/*.tfstate.*",
|
|
67
88
|
...CI_DEFINITIONS,
|
|
68
89
|
];
|
|
69
90
|
|
|
@@ -110,6 +131,53 @@ export const BUILTIN_PROTECT = [
|
|
|
110
131
|
"Dockerfile",
|
|
111
132
|
"**/Dockerfile",
|
|
112
133
|
|
|
134
|
+
// Lockfiles decide which code actually *runs*. `package.json` was protected
|
|
135
|
+
// and `package-lock.json` was not, so an agent could change a resolved URL
|
|
136
|
+
// or an integrity hash — swapping the code that gets installed — without
|
|
137
|
+
// touching a single declared dependency, and the gate said nothing. The
|
|
138
|
+
// entropy scanner is deliberately blind to lockfiles too (they are full of
|
|
139
|
+
// hashes), so the change was invisible twice over. `BUILTIN_RESTRICTED` in
|
|
140
|
+
// risk.mjs already knew these mattered; only the risk tier consumed it,
|
|
141
|
+
// never checkScope.
|
|
142
|
+
"package-lock.json",
|
|
143
|
+
"**/package-lock.json",
|
|
144
|
+
"pnpm-lock.yaml",
|
|
145
|
+
"**/pnpm-lock.yaml",
|
|
146
|
+
"yarn.lock",
|
|
147
|
+
"**/yarn.lock",
|
|
148
|
+
"bun.lockb",
|
|
149
|
+
"bun.lock",
|
|
150
|
+
"Cargo.lock",
|
|
151
|
+
"**/Cargo.lock",
|
|
152
|
+
"go.sum",
|
|
153
|
+
"**/go.sum",
|
|
154
|
+
"poetry.lock",
|
|
155
|
+
"uv.lock",
|
|
156
|
+
"Pipfile.lock",
|
|
157
|
+
"Gemfile.lock",
|
|
158
|
+
"composer.lock",
|
|
159
|
+
"gradle.lockfile",
|
|
160
|
+
"npm-shrinkwrap.json",
|
|
161
|
+
"mix.lock",
|
|
162
|
+
"pubspec.lock",
|
|
163
|
+
"Podfile.lock",
|
|
164
|
+
"Package.resolved",
|
|
165
|
+
".terraform.lock.hcl",
|
|
166
|
+
|
|
167
|
+
// Toolchain pins choose the compiler that runs the whole suite.
|
|
168
|
+
".nvmrc",
|
|
169
|
+
".tool-versions",
|
|
170
|
+
".mise.toml",
|
|
171
|
+
"rust-toolchain",
|
|
172
|
+
"rust-toolchain.toml",
|
|
173
|
+
"**/gradle/wrapper/gradle-wrapper.properties",
|
|
174
|
+
"**/gradle/wrapper/gradle-wrapper.jar",
|
|
175
|
+
|
|
176
|
+
// Who has to approve a change, and what runs on every developer's machine.
|
|
177
|
+
"CODEOWNERS",
|
|
178
|
+
"docs/CODEOWNERS",
|
|
179
|
+
".pre-commit-config.yaml",
|
|
180
|
+
|
|
113
181
|
// Test-runner configuration decides which tests run and what counts as a
|
|
114
182
|
// pass. Rewriting it is the cheapest way to make a suite green without
|
|
115
183
|
// touching a single assertion — `--passWithNoTests`, an added ignore
|
package/src/coverage.mjs
CHANGED
|
@@ -300,12 +300,27 @@ export function calculateDiffCoverage(coverageByFile, diffStr = "", options = {}
|
|
|
300
300
|
}
|
|
301
301
|
}
|
|
302
302
|
|
|
303
|
-
|
|
304
|
-
|
|
303
|
+
// 100% of nothing is not 100%.
|
|
304
|
+
//
|
|
305
|
+
// The denominator counts only the added lines V8 actually mapped, and V8
|
|
306
|
+
// maps nothing outside Node. So a Python diff adding three executable lines
|
|
307
|
+
// measured zero of them and was reported as `score: 100` — the best possible
|
|
308
|
+
// number, produced by a measurement that never happened, on 20-odd of the 25
|
|
309
|
+
// stacks this kit claims to support. `mutation.mjs` had the identical bug and
|
|
310
|
+
// was fixed in v0.57.0; this is the same shape one module over.
|
|
311
|
+
//
|
|
312
|
+
// `ok` stays true because nothing failed to be covered, and a gate that
|
|
313
|
+
// blocks every non-Node diff gets switched off. What changes is the claim:
|
|
314
|
+
// `scored: false` and a reason, instead of a number nobody measured.
|
|
315
|
+
const scored = totalLines > 0;
|
|
316
|
+
const score = scored ? Math.round((coveredLines / totalLines) * 10000) / 100 : null;
|
|
317
|
+
const ok = scored ? score >= minCoverage : true;
|
|
305
318
|
|
|
306
319
|
return {
|
|
307
320
|
ok,
|
|
308
321
|
score,
|
|
322
|
+
scored,
|
|
323
|
+
...(scored ? {} : { reason: "No added executable lines were measurable — V8 coverage only observes code Node itself ran, so nothing was scored." }),
|
|
309
324
|
minCoverage,
|
|
310
325
|
totalLines,
|
|
311
326
|
coveredLines,
|
package/src/engine.mjs
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { loadConfig, parseYaml, normalizeScope } from "./config.mjs";
|
|
2
2
|
import { isTestPath } from "./test-paths.mjs";
|
|
3
|
+
import { checkCollectionFloor } from "./ops/test-collection.mjs";
|
|
3
4
|
import { checkScope, scanDiff, scanBinaryPayloads, redactSecrets } from "./security.mjs";
|
|
4
5
|
import { changedFiles, diffBytes, diffText, binaryDiffEntries, symlinkChanges, showFromOrigin, runCmd } from "./git.mjs";
|
|
5
6
|
import { createProvider, ProviderRateLimitError, ProviderUnavailableError } from "./provider.mjs";
|
|
@@ -521,7 +522,25 @@ export async function gate(opts = {}) {
|
|
|
521
522
|
const ranNoVerification = !executionRecords.some((r) => r && r.kind !== "assert");
|
|
522
523
|
const missingOracle = verificationRequired && ranNoVerification;
|
|
523
524
|
|
|
524
|
-
|
|
525
|
+
// A command that ran is not the same as a command that tested something.
|
|
526
|
+
//
|
|
527
|
+
// `missingOracle` above catches "no stage executed". It cannot catch the
|
|
528
|
+
// case where a stage executed, exited 0, and collected zero tests — which
|
|
529
|
+
// several runners report as success by design: `go test ./...` prints
|
|
530
|
+
// "[no test files]" and exits 0, jest has --passWithNoTests, and
|
|
531
|
+
// `npm test --workspaces` is green when the one package the diff touched
|
|
532
|
+
// has no suite. A repository could invert a function, add an untested one
|
|
533
|
+
// and collect five green phases, verified against nothing at all.
|
|
534
|
+
//
|
|
535
|
+
// The count is read out of the runner's own summary and only a *stated*
|
|
536
|
+
// zero counts. An unrecognised runner yields null and passes: failing on
|
|
537
|
+
// "I could not tell" would break every runner not on the list.
|
|
538
|
+
const collectionFloor = verificationRequired
|
|
539
|
+
? checkCollectionFloor(testResult, { minTests: trustedVerify.minTests })
|
|
540
|
+
: { ok: true, count: null, runner: null, reason: null };
|
|
541
|
+
const emptySuite = !collectionFloor.ok;
|
|
542
|
+
|
|
543
|
+
const verifyOk = !failingCmd && !testTampered && !missingOracle && !emptySuite;
|
|
525
544
|
|
|
526
545
|
// What actually broke. Without this the verify phase reported `ok: false` and
|
|
527
546
|
// nothing else — not the stage, not the exit code, not a line of output — so
|
|
@@ -568,7 +587,18 @@ export async function gate(opts = {}) {
|
|
|
568
587
|
"The gate approves a change because verification passed. Zero stages executed is not a pass.",
|
|
569
588
|
],
|
|
570
589
|
}
|
|
571
|
-
:
|
|
590
|
+
: emptySuite
|
|
591
|
+
? {
|
|
592
|
+
stageId: "empty-suite",
|
|
593
|
+
command: testResult?.command || null,
|
|
594
|
+
exitCode: 0,
|
|
595
|
+
stdout: "",
|
|
596
|
+
stderr: collectionFloor.reason,
|
|
597
|
+
diagnostics: [
|
|
598
|
+
"An exit code of 0 from a runner that collected no tests is not evidence about this change.",
|
|
599
|
+
],
|
|
600
|
+
}
|
|
601
|
+
: null;
|
|
572
602
|
|
|
573
603
|
// Generate & persist Evidence Manifest
|
|
574
604
|
const evidenceManifest = generateEvidenceManifest(root, {
|
|
@@ -583,6 +613,7 @@ export async function gate(opts = {}) {
|
|
|
583
613
|
failedStage: failingCmd?.stageId || failingCmd?.phase || null,
|
|
584
614
|
diagnostics: failingCmd?.diagnostics?.length ? failingCmd.diagnostics : (failingCmd?.stderr ? [failingCmd.stderr] : []),
|
|
585
615
|
metrics: failingCmd?.metrics || {},
|
|
616
|
+
collection: collectionFloor,
|
|
586
617
|
ok: verifyOk,
|
|
587
618
|
});
|
|
588
619
|
if (testTampered) {
|
package/src/evidence.mjs
CHANGED
|
@@ -129,7 +129,20 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
129
129
|
.filter((p) => existsSync(join(root, p)) && statSync(join(root, p)).isFile())
|
|
130
130
|
.sort();
|
|
131
131
|
} else {
|
|
132
|
-
|
|
132
|
+
// A sixth spelling of "where do the tests live?", and the one that got
|
|
133
|
+
// missed when the other five were unified behind `isTestPath`: this is a
|
|
134
|
+
// list of *directory names at the repository root*, not a predicate. Go
|
|
135
|
+
// puts its tests beside the code (`internal/calc/calc_test.go`), and every
|
|
136
|
+
// monorepo puts them under `packages/*/test/`. Neither is under a
|
|
137
|
+
// root-level `test/`, so the walk found nothing, `fileCount` was 0, and
|
|
138
|
+
// `strictTestLock` — which requires `fileCount > 0` — switched itself off
|
|
139
|
+
// without saying so. The tree hash then became the SHA-256 of the empty
|
|
140
|
+
// string, and the evidence manifest attested to it.
|
|
141
|
+
//
|
|
142
|
+
// The named directories stay as a fast path; when they yield nothing, walk
|
|
143
|
+
// the repository and let the shared predicate decide. The walk already
|
|
144
|
+
// skips node_modules, vendor, target and the build caches.
|
|
145
|
+
const targetDirs = options.directories || ["test", "tests", "__tests__", "spec", "specs", "src"];
|
|
133
146
|
for (const dirName of targetDirs) {
|
|
134
147
|
const dirPath = join(root, dirName);
|
|
135
148
|
if (existsSync(dirPath)) {
|
|
@@ -137,6 +150,11 @@ export function computeDirectoryHash(root, options = {}) {
|
|
|
137
150
|
fileList.push(...found);
|
|
138
151
|
}
|
|
139
152
|
}
|
|
153
|
+
if (options.testOnly && !fileList.some((f) => isTestPath(f))) {
|
|
154
|
+
for (const f of findFilesRecursively(root, root)) {
|
|
155
|
+
if (isTestPath(f)) fileList.push(f);
|
|
156
|
+
}
|
|
157
|
+
}
|
|
140
158
|
|
|
141
159
|
// Plenty of projects keep `app.test.mjs` or `index.js` beside package.json
|
|
142
160
|
// rather than under one of the directories above, and those files were
|
|
@@ -201,6 +219,7 @@ export function computeEvidenceHash(manifest) {
|
|
|
201
219
|
intent: manifest.intent,
|
|
202
220
|
provenance: manifest.provenance,
|
|
203
221
|
testIntegrity: manifest.testIntegrity,
|
|
222
|
+
...(manifest.verification ? { verification: manifest.verification } : {}),
|
|
204
223
|
...(manifest.sourceIntegrity ? { sourceIntegrity: manifest.sourceIntegrity } : {}),
|
|
205
224
|
executionRecords: manifest.executionRecords,
|
|
206
225
|
securityChecks: manifest.securityChecks,
|
|
@@ -342,6 +361,24 @@ export function generateEvidenceManifest(root = process.cwd(), options = {}) {
|
|
|
342
361
|
maxDiffKb: options.maxDiffKb || 75,
|
|
343
362
|
protectedScopeOk: options.protectedScopeOk ?? true,
|
|
344
363
|
},
|
|
364
|
+
// How many tests the runner said it collected, and whether it said at all.
|
|
365
|
+
//
|
|
366
|
+
// The collection floor fails a *stated* zero and lets an unstated count
|
|
367
|
+
// pass, because failing on "I could not tell" would break every runner not
|
|
368
|
+
// on the list. That is the right call for the verdict and the wrong thing
|
|
369
|
+
// to leave out of the record: a manifest that says nothing here reads as
|
|
370
|
+
// though a suite ran. `counted: false` is the honest shape for a run where
|
|
371
|
+
// the number was never observable — a quiet runner (`cargo test --quiet`,
|
|
372
|
+
// `pytest -q`) suppresses the very line the floor reads.
|
|
373
|
+
...(options.collection
|
|
374
|
+
? {
|
|
375
|
+
verification: {
|
|
376
|
+
testsCollected: options.collection.count,
|
|
377
|
+
counted: options.collection.count !== null,
|
|
378
|
+
runner: options.collection.runner,
|
|
379
|
+
},
|
|
380
|
+
}
|
|
381
|
+
: {}),
|
|
345
382
|
...(diagnostics.length > 0 ? { diagnostics } : {}),
|
|
346
383
|
...(Object.keys(metrics).length > 0 ? { metrics } : {}),
|
|
347
384
|
};
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* How many tests the runner actually collected, read out of its own output.
|
|
3
|
+
*
|
|
4
|
+
* The gate's oracle is one number: the exit code of the verification command.
|
|
5
|
+
* That number cannot distinguish "every test passed" from "there were no
|
|
6
|
+
* tests". Several runners report the second case as success, by design:
|
|
7
|
+
*
|
|
8
|
+
* go test ./... → "? example.com/app [no test files]", exit 0
|
|
9
|
+
* jest --passWithNoTests → "No tests found, exiting with code 0"
|
|
10
|
+
* npm test --workspaces → exit 0 when the changed package has no suite
|
|
11
|
+
* pytest --exitfirst on a path that matches nothing, in some configurations
|
|
12
|
+
*
|
|
13
|
+
* So a repository could invert a function, add an untested one, and collect
|
|
14
|
+
* five green phases — verified against nothing. `verify.required: false` is
|
|
15
|
+
* the switch for a repository that genuinely has no oracle; silently passing
|
|
16
|
+
* is not.
|
|
17
|
+
*
|
|
18
|
+
* The parsing is deliberately one-sided. A count is only returned when the
|
|
19
|
+
* runner stated one in a form recognised here; an unrecognised runner yields
|
|
20
|
+
* `null`, and null is not a failure. Failing on "I could not tell" would break
|
|
21
|
+
* every runner not on this list, which is most of them.
|
|
22
|
+
*/
|
|
23
|
+
|
|
24
|
+
/**
|
|
25
|
+
* Patterns that state a test count, per runner family.
|
|
26
|
+
*
|
|
27
|
+
* Each entry captures a single number. The first pattern that matches wins,
|
|
28
|
+
* so the more specific summaries come first.
|
|
29
|
+
*/
|
|
30
|
+
const COUNT_PATTERNS = [
|
|
31
|
+
// node:test — spec reporter ("ℹ tests 940") and tap ("# tests 940")
|
|
32
|
+
{ name: "node:test", re: /^[^\n]*?(?:ℹ|#)\s*tests\s+(\d+)\s*$/m },
|
|
33
|
+
// pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
|
|
34
|
+
{ name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
|
|
35
|
+
{ name: "pytest", re: /=+\s*(\d+)\s+passed/m },
|
|
36
|
+
// cargo — "running 7 tests"
|
|
37
|
+
{ name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
|
|
38
|
+
// jest / vitest — "Tests: 12 passed, 12 total"
|
|
39
|
+
{ name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
|
|
40
|
+
// mocha — "12 passing"
|
|
41
|
+
{ name: "mocha", re: /^\s*(\d+)\s+passing/m },
|
|
42
|
+
// Maven / Surefire — "Tests run: 12, Failures: 0"
|
|
43
|
+
{ name: "surefire", re: /\bTests run:\s*(\d+)/i },
|
|
44
|
+
// PHPUnit — "OK (12 tests, 30 assertions)"
|
|
45
|
+
{ name: "phpunit", re: /\bOK\s*\((\d+)\s+tests?/i },
|
|
46
|
+
// RSpec / ExUnit — "12 examples, 0 failures" / "12 tests, 0 failures"
|
|
47
|
+
{ name: "rspec", re: /^\s*(\d+)\s+examples?,\s*\d+\s+failures?/m },
|
|
48
|
+
{ name: "exunit", re: /^\s*(\d+)\s+tests?,\s*\d+\s+failures?/m },
|
|
49
|
+
// dotnet test — "Total tests: 12" / "Passed! - Failed: 0, Passed: 12"
|
|
50
|
+
{ name: "dotnet", re: /\bTotal(?:\s+tests)?:\s*(\d+)/i },
|
|
51
|
+
// swift test / XCTest — "Executed 12 tests"
|
|
52
|
+
{ name: "xctest", re: /\bExecuted\s+(\d+)\s+tests?/i },
|
|
53
|
+
];
|
|
54
|
+
|
|
55
|
+
/** Per-test lines, which `go test` only prints under -v. */
|
|
56
|
+
const GO_PER_TEST = /^\s*--- (?:PASS|FAIL|SKIP):/gm;
|
|
57
|
+
|
|
58
|
+
/** Phrases that state, in so many words, that nothing was collected. */
|
|
59
|
+
const EXPLICIT_ZERO = [
|
|
60
|
+
{ name: "pytest", re: /\bno tests ran\b/i },
|
|
61
|
+
{ name: "pytest", re: /^\s*collected\s+0\s+items?/m },
|
|
62
|
+
{ name: "jest", re: /\bNo tests found\b/i },
|
|
63
|
+
{ name: "vitest", re: /\bNo test files found\b/i },
|
|
64
|
+
{ name: "mocha", re: /^\s*0\s+passing/m },
|
|
65
|
+
{ name: "cargo", re: /^\s*running\s+0\s+tests?\s*$/m },
|
|
66
|
+
{ name: "phpunit", re: /\bNo tests executed!/i },
|
|
67
|
+
{ name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
|
|
68
|
+
{ name: "ctest", re: /\bNo tests were found\b/i },
|
|
69
|
+
{ name: "flutter", re: /\bNo tests ran\.?/i },
|
|
70
|
+
];
|
|
71
|
+
|
|
72
|
+
/** Go prints this per package that has no test files at all. */
|
|
73
|
+
const GO_NO_TEST_FILES = /\[no test files\]/;
|
|
74
|
+
/** Any sign that a Go package did run tests. */
|
|
75
|
+
const GO_RAN_SOMETHING = /^(?:ok|FAIL|---\s+(?:PASS|FAIL|SKIP)):?\s/m;
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Read a collected-test count out of a runner's output.
|
|
79
|
+
*
|
|
80
|
+
* @param {string} [stdout]
|
|
81
|
+
* @param {string} [stderr]
|
|
82
|
+
* @returns {{ count: number|null, runner: string|null }}
|
|
83
|
+
* `count` is null when no recognised runner stated one — which is not a
|
|
84
|
+
* finding, only an absence of evidence.
|
|
85
|
+
*/
|
|
86
|
+
export function parseCollectedTests(stdout = "", stderr = "") {
|
|
87
|
+
const text = `${stdout || ""}\n${stderr || ""}`;
|
|
88
|
+
if (!text.trim()) return { count: null, runner: null };
|
|
89
|
+
|
|
90
|
+
for (const rule of EXPLICIT_ZERO) {
|
|
91
|
+
if (rule.re.test(text)) return { count: 0, runner: rule.name };
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
// Go states absence per package rather than as a count, so it needs its own
|
|
95
|
+
// pass before the generic patterns.
|
|
96
|
+
if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
|
|
97
|
+
// Only a run where *no* package did anything is a zero: a monorepo where
|
|
98
|
+
// one package has no tests and three do is a normal, healthy repository.
|
|
99
|
+
if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };
|
|
100
|
+
// Something ran. `--- PASS:` lines are per-test but appear only under -v,
|
|
101
|
+
// so their absence means the count was not stated — not that it was zero.
|
|
102
|
+
// Reporting zero here would have failed every ordinary `go test ./...`.
|
|
103
|
+
const perTest = text.match(GO_PER_TEST);
|
|
104
|
+
return { count: perTest && perTest.length > 0 ? perTest.length : null, runner: "go" };
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
for (const rule of COUNT_PATTERNS) {
|
|
108
|
+
const m = rule.re.exec(text);
|
|
109
|
+
if (!m) continue;
|
|
110
|
+
const n = Number(m[1]);
|
|
111
|
+
if (Number.isFinite(n)) return { count: n, runner: rule.name };
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
return { count: null, runner: null };
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/**
|
|
118
|
+
* Decide whether a passing verification command actually verified anything.
|
|
119
|
+
*
|
|
120
|
+
* @param {object} testResult - the test stage's result ({ ok, stdout, stderr, command })
|
|
121
|
+
* @param {object} [opts]
|
|
122
|
+
* @param {number} [opts.minTests=1] - the floor, from `verify.minTests`.
|
|
123
|
+
* @returns {{ ok: boolean, count: number|null, runner: string|null, reason: string|null }}
|
|
124
|
+
*/
|
|
125
|
+
export function checkCollectionFloor(testResult, opts = {}) {
|
|
126
|
+
const minTests = Number.isFinite(opts.minTests) ? opts.minTests : 1;
|
|
127
|
+
if (minTests <= 0) return { ok: true, count: null, runner: null, reason: null };
|
|
128
|
+
// Only a *passing* command can lie about this. A failing one already fails.
|
|
129
|
+
if (!testResult || testResult.ok !== true) {
|
|
130
|
+
return { ok: true, count: null, runner: null, reason: null };
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
|
|
134
|
+
if (count === null || count >= minTests) {
|
|
135
|
+
return { ok: true, count, runner, reason: null };
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
ok: false,
|
|
140
|
+
count,
|
|
141
|
+
runner,
|
|
142
|
+
reason:
|
|
143
|
+
`The verification command exited 0 without running any tests` +
|
|
144
|
+
(runner ? ` (${runner} reported ${count})` : "") +
|
|
145
|
+
`, so this change was approved against nothing. ` +
|
|
146
|
+
`Point verify.test at a suite that covers this repository, lower the floor with verify.minTests, ` +
|
|
147
|
+
`or — if this repository intentionally uses only the scope and secret phases — set verify.required: false.`,
|
|
148
|
+
};
|
|
149
|
+
}
|
package/src/security.mjs
CHANGED
|
@@ -1988,12 +1988,24 @@ export function resolveAllowedTamperKinds(options = {}) {
|
|
|
1988
1988
|
* @param {string} diffOrText - Unified git diff
|
|
1989
1989
|
* @param {Object} [options]
|
|
1990
1990
|
* @param {boolean} [options.allowTestModifications=false]
|
|
1991
|
-
* @returns {{ ok: boolean, violations: Array<
|
|
1991
|
+
* @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, status: "PASS"|"FAIL"|"NOT_APPLICABLE" }}
|
|
1992
|
+
* `status` distinguishes "checked and clean" from "nothing was checked";
|
|
1993
|
+
* `ok: true` alone cannot, and that ambiguity is the defect class this
|
|
1994
|
+
* field exists to make visible.
|
|
1992
1995
|
*/
|
|
1993
1996
|
export function checkTestTampering(diffOrText = "", options = {}) {
|
|
1994
|
-
if (!diffOrText || typeof diffOrText !== "string")
|
|
1997
|
+
if (!diffOrText || typeof diffOrText !== "string") {
|
|
1998
|
+
return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "empty diff" };
|
|
1999
|
+
}
|
|
1995
2000
|
const allowed = resolveAllowedTamperKinds(options);
|
|
1996
|
-
if (allowed.all)
|
|
2001
|
+
if (allowed.all) {
|
|
2002
|
+
return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "all kinds allowed" };
|
|
2003
|
+
}
|
|
2004
|
+
|
|
2005
|
+
// Which predicate decides what this guard even looks at. Injectable so the
|
|
2006
|
+
// meta-check can mutate it: a canary that still passes when the predicate is
|
|
2007
|
+
// replaced by `() => false` was never requiring this guard to activate.
|
|
2008
|
+
const isTestPath_ = typeof options.isTestPath === "function" ? options.isTestPath : isTestPath;
|
|
1997
2009
|
|
|
1998
2010
|
const violations = [];
|
|
1999
2011
|
const lines = diffOrText.split("\n");
|
|
@@ -2002,7 +2014,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2002
2014
|
let currentOldLineNo = null;
|
|
2003
2015
|
let currentNewLineNo = null;
|
|
2004
2016
|
|
|
2005
|
-
const isTestFile =
|
|
2017
|
+
const isTestFile = isTestPath_;
|
|
2006
2018
|
|
|
2007
2019
|
const SKIP_INJECTIONS = [
|
|
2008
2020
|
{ pattern: /\b(?:it|test|describe|context)\.skip\s*\(/i, desc: "Injected test skip (.skip())" },
|
|
@@ -2207,9 +2219,23 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2207
2219
|
? violations
|
|
2208
2220
|
: violations.filter((v) => !allowed.kinds.has(TAMPER_KINDS.get(v.type)));
|
|
2209
2221
|
|
|
2222
|
+
// What was examined, not only what was found.
|
|
2223
|
+
//
|
|
2224
|
+
// `ok: true` from a guard that looked at nothing is byte-identical to
|
|
2225
|
+
// `ok: true` from a guard that looked at everything and approved it. That
|
|
2226
|
+
// ambiguity is how a substring bug in the file classifier switched this
|
|
2227
|
+
// entire guard off for the standard pytest, Rust and RSpec layouts while
|
|
2228
|
+
// every signal stayed green. A verdict without a denominator is not a
|
|
2229
|
+
// verdict, so `inputsSeen` reports the number of test files this run
|
|
2230
|
+
// actually reasoned about, and `status` distinguishes "nothing to check"
|
|
2231
|
+
// from "checked and clean".
|
|
2232
|
+
const inputsSeen = fileAssertions.size;
|
|
2233
|
+
|
|
2210
2234
|
return {
|
|
2211
2235
|
ok: reported.length === 0,
|
|
2212
2236
|
violations: reported,
|
|
2237
|
+
inputsSeen,
|
|
2238
|
+
status: reported.length > 0 ? "FAIL" : inputsSeen > 0 ? "PASS" : "NOT_APPLICABLE",
|
|
2213
2239
|
};
|
|
2214
2240
|
}
|
|
2215
2241
|
|
package/src/stack-detector.mjs
CHANGED
|
@@ -39,7 +39,65 @@ export function pytestCmd(env = process.env) {
|
|
|
39
39
|
}
|
|
40
40
|
|
|
41
41
|
/**
|
|
42
|
-
*
|
|
42
|
+
* Does this Makefile declare a `test` target?
|
|
43
|
+
*
|
|
44
|
+
* Read rather than assumed: the presence of the file says nothing about
|
|
45
|
+
* whether `make test` will run.
|
|
46
|
+
*/
|
|
47
|
+
function makefileHasTestTarget(root) {
|
|
48
|
+
try {
|
|
49
|
+
const text = readFileSync(join(root, "Makefile"), "utf-8");
|
|
50
|
+
return /^\.PHONY:.*\btest\b/m.test(text) || /^test\s*:/m.test(text);
|
|
51
|
+
} catch (_) {
|
|
52
|
+
return false;
|
|
53
|
+
}
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* A declared test script that runs no tests and exits 0.
|
|
58
|
+
*
|
|
59
|
+
* This is the single most dangerous input the gate can receive, because every
|
|
60
|
+
* downstream check reads "a command ran and passed". `bootstrapZeroTestRepo`
|
|
61
|
+
* called `"test": "echo 'no tests yet' && exit 0"` an
|
|
62
|
+
* EXISTING_VERIFICATION_ORACLE — it asked whether the field was set, never
|
|
63
|
+
* what was in it.
|
|
64
|
+
*
|
|
65
|
+
* npm's own default (`echo "Error: no test specified" && exit 1`) is not a
|
|
66
|
+
* placeholder by this definition, and correctly so: it exits non-zero, which
|
|
67
|
+
* fails loudly rather than certifying nothing.
|
|
68
|
+
*/
|
|
69
|
+
export function isPlaceholderTestScript(cmd) {
|
|
70
|
+
if (typeof cmd !== "string") return false;
|
|
71
|
+
const trimmed = cmd.trim();
|
|
72
|
+
if (!trimmed) return true;
|
|
73
|
+
// Drop the announcements; what matters is what the shell is left doing.
|
|
74
|
+
const remainder = trimmed
|
|
75
|
+
.split(/&&|;/)
|
|
76
|
+
.map((part) => part.trim())
|
|
77
|
+
.filter((part) => part && !/^(?:echo|printf|:)\b/.test(part));
|
|
78
|
+
if (remainder.length === 0) return true;
|
|
79
|
+
return remainder.every((part) => /^(?:exit\s+0|true|:)$/.test(part));
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Generate the fallback verification oracle for a JS/generic repo with no tests.
|
|
84
|
+
*
|
|
85
|
+
* What this used to write could not fail:
|
|
86
|
+
*
|
|
87
|
+
* assert.ok(fs.existsSync(process.cwd()));
|
|
88
|
+
* assert.ok(fs.readdirSync(process.cwd()).length > 0);
|
|
89
|
+
*
|
|
90
|
+
* Both hold for every repository and every change, so the generated "oracle"
|
|
91
|
+
* was green against arbitrary broken code — and worse, it *silenced* the
|
|
92
|
+
* `missingOracle` guard in engine.mjs, which fires only when no command ran at
|
|
93
|
+
* all. A repository that honestly had no oracle was converted into one that
|
|
94
|
+
* claimed to have one. That is the tool writing its own blindness to disk.
|
|
95
|
+
*
|
|
96
|
+
* The other stacks already get a real static gate at this point — `tsc
|
|
97
|
+
* --noEmit`, `cargo check`, `go vet`, `compileall` — each of which fails on a
|
|
98
|
+
* real class of defect. This is the JavaScript equivalent: every source file
|
|
99
|
+
* must parse. It proves the code compiles, not that it works, and the caller
|
|
100
|
+
* says so; but a syntax error fails it, which is one more than before.
|
|
43
101
|
*/
|
|
44
102
|
export function generateSmokeTestScript(root = process.cwd()) {
|
|
45
103
|
const agentDir = join(root, ".agent");
|
|
@@ -48,15 +106,45 @@ export function generateSmokeTestScript(root = process.cwd()) {
|
|
|
48
106
|
} catch (_) {}
|
|
49
107
|
|
|
50
108
|
const smokePath = join(agentDir, "smoke.test.mjs");
|
|
51
|
-
const content = `// Auto-generated zero-dependency
|
|
109
|
+
const content = `// Auto-generated zero-dependency parse gate (.agent/smoke.test.mjs)
|
|
110
|
+
//
|
|
111
|
+
// Written by \`agentctl bootstrap\` for a repository that had no test suite.
|
|
112
|
+
// It proves that every source file still parses. It does NOT prove the code
|
|
113
|
+
// is correct — replace it with real tests as soon as there are any.
|
|
52
114
|
import { test } from "node:test";
|
|
53
115
|
import assert from "node:assert/strict";
|
|
54
|
-
import
|
|
116
|
+
import { readdirSync, statSync } from "node:fs";
|
|
117
|
+
import { join, extname } from "node:path";
|
|
118
|
+
import { spawnSync } from "node:child_process";
|
|
119
|
+
|
|
120
|
+
const SKIP = new Set([".git", "node_modules", "vendor", "dist", "build", "coverage", ".venv", "venv", ".next", ".agent"]);
|
|
121
|
+
const SOURCE = new Set([".js", ".mjs", ".cjs"]);
|
|
122
|
+
|
|
123
|
+
function sources(dir, acc = [], depth = 0) {
|
|
124
|
+
if (depth > 8) return acc;
|
|
125
|
+
for (const entry of readdirSync(dir, { withFileTypes: true })) {
|
|
126
|
+
if (entry.name.startsWith(".") && entry.name !== ".agent") continue;
|
|
127
|
+
if (SKIP.has(entry.name)) continue;
|
|
128
|
+
const full = join(dir, entry.name);
|
|
129
|
+
if (entry.isDirectory()) sources(full, acc, depth + 1);
|
|
130
|
+
else if (SOURCE.has(extname(entry.name))) acc.push(full);
|
|
131
|
+
}
|
|
132
|
+
return acc;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
test("every source file parses", () => {
|
|
136
|
+
const files = sources(process.cwd());
|
|
55
137
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
138
|
+
// A gate with nothing to check is not a passing gate. Reporting success
|
|
139
|
+
// over an empty file list is exactly the vacuous oracle this replaced.
|
|
140
|
+
assert.ok(files.length > 0, "No JavaScript sources found to verify — this is not an oracle. Set verify.test in .agent/config.yml.");
|
|
141
|
+
|
|
142
|
+
const broken = [];
|
|
143
|
+
for (const file of files) {
|
|
144
|
+
const res = spawnSync(process.execPath, ["--check", file], { encoding: "utf-8" });
|
|
145
|
+
if (res.status !== 0) broken.push(\`\${file}: \${(res.stderr || "").trim().split("\\n")[0]}\`);
|
|
146
|
+
}
|
|
147
|
+
assert.deepEqual(broken, [], \`\${broken.length} file(s) failed to parse\`);
|
|
60
148
|
});
|
|
61
149
|
`;
|
|
62
150
|
writeFileSync(smokePath, content, "utf-8");
|
|
@@ -173,7 +261,15 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
173
261
|
if (existsSync(join(projectRoot, "Package.swift"))) {
|
|
174
262
|
return { ...container, stack: "swift", testCmd: "swift test", buildCmd: "swift build", triggerFile: "Package.swift" };
|
|
175
263
|
}
|
|
176
|
-
|
|
264
|
+
// `app.json` is not a manifest, it is an Expo/React Native *configuration*
|
|
265
|
+
// file, and the name is generic enough that unrelated projects use it. Its
|
|
266
|
+
// test command is `npm test`, so without a package.json beside it the
|
|
267
|
+
// detector was claiming a Node stack for a repository that has no Node in
|
|
268
|
+
// it: `Cargo.toml` + `app.json` was measured as `react-native` / `npm test`.
|
|
269
|
+
if (
|
|
270
|
+
(existsSync(join(projectRoot, "app.json")) && existsSync(join(projectRoot, "package.json"))) ||
|
|
271
|
+
existsSync(join(projectRoot, "react-native.config.js"))
|
|
272
|
+
) {
|
|
177
273
|
const triggerFile = existsSync(join(projectRoot, "app.json")) ? "app.json" : "react-native.config.js";
|
|
178
274
|
return { ...container, stack: "react-native", testCmd: "npm test", buildCmd: "npx react-native bundle --platform android --dev false --entry-file index.js --bundle-output android/main.jsbundle", triggerFile };
|
|
179
275
|
}
|
|
@@ -188,7 +284,14 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
|
188
284
|
if (existsSync(join(projectRoot, "go.mod"))) {
|
|
189
285
|
return { ...container, stack: "go", testCmd: "go test ./...", buildCmd: "go build ./...", triggerFile: "go.mod" };
|
|
190
286
|
}
|
|
191
|
-
|
|
287
|
+
// A Makefile is only an oracle if it declares the target we are about to
|
|
288
|
+
// run. `make test` on a Makefile with only a `build:` target exits 2 with
|
|
289
|
+
// "No rule to make target 'test'" — measured on a repository whose
|
|
290
|
+
// package.json declared a perfectly good `vitest run`, because the Makefile
|
|
291
|
+
// was checked first and the presence of the *file* was the whole test. A
|
|
292
|
+
// hard red on day one is how a user learns the gate is broken and turns it
|
|
293
|
+
// off, so the file must earn the claim.
|
|
294
|
+
if (existsSync(join(projectRoot, "Makefile")) && makefileHasTestTarget(projectRoot)) {
|
|
192
295
|
return { ...container, stack: "make", testCmd: "make test", buildCmd: "make build", triggerFile: "Makefile" };
|
|
193
296
|
}
|
|
194
297
|
|
|
@@ -715,7 +818,7 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
|
|
|
715
818
|
if (existsSync(join(root, "package.json"))) {
|
|
716
819
|
try {
|
|
717
820
|
const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
|
|
718
|
-
hasPkgTest = Boolean(pkg?.scripts?.test);
|
|
821
|
+
hasPkgTest = Boolean(pkg?.scripts?.test) && !isPlaceholderTestScript(pkg.scripts.test);
|
|
719
822
|
} catch (_) {}
|
|
720
823
|
}
|
|
721
824
|
|
package/src/task-optimizer.mjs
CHANGED
|
@@ -265,9 +265,13 @@ export function scorePromptFalsifiability(promptText, options = {}) {
|
|
|
265
265
|
let isTrivial = false;
|
|
266
266
|
|
|
267
267
|
if (!verifyCmd) {
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
268
|
+
// `detectStackOracles` returns an object, not an array: `.length` was
|
|
269
|
+
// undefined, so this branch never ran and every task without an explicit
|
|
270
|
+
// --verify-cmd was scored MISSING_ORACLE even in a repository with a
|
|
271
|
+
// working suite.
|
|
272
|
+
const detected = detectStackOracles(rootDir);
|
|
273
|
+
if (detected?.candidates?.testCmd) {
|
|
274
|
+
verifyCmd = detected.candidates.testCmd;
|
|
271
275
|
autoDetected = true;
|
|
272
276
|
}
|
|
273
277
|
}
|