jules-orchestrator-kit 0.59.0 → 0.63.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -204,7 +204,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
204
204
  * **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
205
205
  * **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
206
206
  * **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
207
- * **Verified Test Suite:** Tested with **940 unit tests across 130 suites passing in < 15.0s**.
207
+ * **Verified Test Suite:** Tested with **1015 unit tests across 145 suites passing in < 15.0s**.
208
208
 
209
209
  <br/>
210
210
 
@@ -237,7 +237,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
237
237
  | `doctor` | `agentctl doctor [--probe] [--json]` | Diagnostic check runner. `--probe` additionally starts the configured provider's CLI to confirm it answers, rather than only finding it on `PATH`. | `0` (Healthy), `1` (Failures) |
238
238
  | `queue` | `agentctl queue [--dag] [--concurrency <n>] [--dry-run] [--json]` | Consumes and executes task envelopes in `.agent/jules-queue/` with Kahn's DAG dependency resolution. Non-task files (manifests, `README.md`) are skipped, and `--dry-run` previews without moving anything. | `0` (Complete) |
239
239
  | `swarm` | `agentctl swarm [--json]` | Runs parallel multi-agent swarm across worker slots with PID liveness detection. | `0` (Complete) |
240
- | `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--allow-protected] [--allow-test-modifications] [--json] [--json-report <path>]` | Runs security, secret scanning, rules budget audit, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `1` (Budget/Arg), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
240
+ | `check` / `gate` / `audit`| `agentctl check [--mode working-tree] [--fix] [--allow-protected] [--allow-test-change <kind>] [--json] [--json-report <path>]` | Runs security, secret scanning, rules budget audit, and tiered verification gates (with declarative assertion support) against working tree or branch. | `0` (Approved), `1` (Budget/Arg), `3` (Scope), `4` (Verify), `5` (Diff >75K), `6` (Secret), `8` (Flaky) |
241
241
  | `mutate` / `mutation` | `agentctl mutate [--min-score <n>] [--max-mutants <n>] [--cmd <testCmd>] [--json]` | Runs zero-dependency diff mutation testing harness on changed hunks with operator inversion and safety rollback. | `0` (Passed), `1` (Score Low) |
242
242
  | `coverage` | `agentctl coverage [--min <pct>] [--cmd <testCmd>] [--base <ref>] [--json]` | Runs native zero-dependency V8 diff coverage check against added diff lines. | `0` (Passed), `1` (Low Coverage) |
243
243
  | `probe` / `stability` | `agentctl probe [--repeat <n>] [--min <passRate>] [--cmd <testCmd>] [--json]` | Probes test suite flakiness across N consecutive iterations with oscillation detection. | `0` (Passed), `1` (Flaky) |
package/bin/agentctl.mjs CHANGED
@@ -388,7 +388,12 @@ async function main() {
388
388
  // of spec, which necessarily rewrites what a test expects, hit a
389
389
  // CRITICAL finding at exit 6 with no documented way past it. A guard
390
390
  // with no override is not a guard, it is an outage.
391
+ //
392
+ // This one is the blunt form and turns off all six checks. Prefer
393
+ // `--allow-test-change <kind>`: answering one finding should not
394
+ // silence five other checks nobody looked at.
391
395
  "allow-test-modifications": { type: "boolean" },
396
+ "allow-test-change": { type: "string", multiple: true },
392
397
  json: { type: "boolean", short: "j" },
393
398
  "json-report": { type: "string" },
394
399
  "dry-run": { type: "boolean", short: "d" },
@@ -409,6 +414,7 @@ async function main() {
409
414
  fix: values.fix,
410
415
  allowProtected: values["allow-protected"],
411
416
  allowTestModifications: values["allow-test-modifications"],
417
+ allowTestChanges: values["allow-test-change"],
412
418
  jsonReport: values["json-report"],
413
419
  });
414
420
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.59.0",
3
+ "version": "0.63.0",
4
4
  "description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
5
5
  "repository": {
6
6
  "type": "git",
@@ -55,6 +55,7 @@
55
55
  "jules:rules-lint": "node scripts/rules-lint.mjs",
56
56
  "jules:doc-sync": "node scripts/doc-sync-check.mjs",
57
57
  "release": "node scripts/release.mjs",
58
+ "guard-reach": "node scripts/guard-reach-check.mjs",
58
59
  "lint": "eslint ."
59
60
  },
60
61
  "engines": {
@@ -0,0 +1,176 @@
1
+ #!/usr/bin/env node
2
+
3
+ /**
4
+ * Activation coverage: proof that every blocking check can still be made red.
5
+ *
6
+ * A defect that turns a check off cannot be found by the check it turns off.
7
+ * That is not a hypothetical — `isTestFile` matched the substring `/test/`,
8
+ * which does not occur in `tests/test_calc.py`, so the entire tamper guard was
9
+ * silent for the standard pytest, Rust and RSpec layouts. Every mechanism that
10
+ * should have caught it was working exactly as designed:
11
+ *
12
+ * - the unit suite sampled the same distribution the implementation was
13
+ * written from, so its fixtures re-confirmed the dialect it already knew;
14
+ * - the doc-sync gate compares counts and versions, and a guard that guards
15
+ * nothing still contributes passing tests;
16
+ * - the nine-way CI matrix varies OS and Node version — dimensions
17
+ * orthogonal to the defect. Nine runs of `test/foo.test.js` never explore
18
+ * `tests/test_calc.py`;
19
+ * - cold review reads the code against its stated intent, and here the code
20
+ * and the intent agreed. The eye supplies the leading slash;
21
+ * - the release gate is a conjunction over those four, and a signal that
22
+ * silently goes absent contributes `true`.
23
+ *
24
+ * The common property: `ok: true` from a check that examined nothing is
25
+ * byte-identical to `ok: true` from a check that examined everything. There is
26
+ * no denominator. This script supplies one.
27
+ *
28
+ * Three steps, all in-process, no dependencies, well under a second:
29
+ *
30
+ * 1. POLICY — the hand-written witness table in test/fixtures/guard-policy.mjs
31
+ * must hold. It is derived from what the tool advertises, never
32
+ * from the regexes that implement it.
33
+ * 2. CANARIES — every known-bad input must produce the finding it names. A
34
+ * canary that comes back clean is not a pass; it is proof that
35
+ * the rule stopped being reachable.
36
+ * 3. MUTANTS — each hand-written mutant of the applicability predicate must
37
+ * kill at least one canary. A surviving mutant means no canary
38
+ * ever required the guard to activate, so the suite would stay
39
+ * green if it silently stopped looking.
40
+ *
41
+ * Usage: node scripts/guard-reach-check.mjs [--json]
42
+ * Exit codes: 0 = every guard reachable, 1 = a guard has gone silent.
43
+ */
44
+
45
+ import { checkTestTampering, checkScope } from "../src/security.mjs";
46
+ import { isTestPath } from "../src/test-paths.mjs";
47
+ import { normalizeScope } from "../src/config.mjs";
48
+ import { parseCollectedTests } from "../src/ops/test-collection.mjs";
49
+ import {
50
+ TEST_PATH_CASES,
51
+ TAMPER_CANARIES,
52
+ PREDICATE_MUTANTS,
53
+ EMPTY_RUN_CANARIES,
54
+ SCOPE_CANARIES,
55
+ } from "../test/fixtures/guard-policy.mjs";
56
+
57
+ /** Build a unified diff for one canary. */
58
+ function canaryDiff(c) {
59
+ const lines = [`--- a/${c.file}`, `+++ b/${c.file}`, "@@ -1,20 +1,20 @@", " // context"];
60
+ for (const l of c.removed) lines.push(`-${l}`);
61
+ for (const l of c.added) lines.push(`+${l}`);
62
+ lines.push(" // context");
63
+ return lines.join("\n");
64
+ }
65
+
66
+ const failures = [];
67
+ const checks = [];
68
+ const add = (name, ok, detail) => {
69
+ checks.push({ name, ok, detail });
70
+ if (!ok) failures.push(`${name}: ${detail}`);
71
+ };
72
+
73
+ // --- 1. Policy contract -----------------------------------------------------
74
+ {
75
+ const wrong = TEST_PATH_CASES.filter((c) => isTestPath(c.path) !== c.expected);
76
+ add(
77
+ "policy: test-path domain",
78
+ wrong.length === 0,
79
+ wrong.length
80
+ ? wrong.map((c) => `${c.path} → ${isTestPath(c.path)}, policy says ${c.expected} (${c.why})`).join("; ")
81
+ : `${TEST_PATH_CASES.length} witnesses hold`
82
+ );
83
+ }
84
+
85
+ {
86
+ const scope = normalizeScope({ deny: [], allow: [], protect: [] });
87
+ const wrong = [];
88
+ for (const c of SCOPE_CANARIES) {
89
+ const res = checkScope([c.path], scope);
90
+ const rule = res.ok ? "none" : res.violations[0].rule;
91
+ if (rule !== c.rule) wrong.push(`${c.path} → ${rule}, policy says ${c.rule} (${c.why})`);
92
+ }
93
+ add("policy: scope tiers", wrong.length === 0, wrong.length ? wrong.join("; ") : `${SCOPE_CANARIES.length} paths tiered as declared`);
94
+ }
95
+
96
+ {
97
+ const missed = EMPTY_RUN_CANARIES.filter((c) => parseCollectedTests(c.output, "").count !== 0);
98
+ add(
99
+ "policy: empty-run detection",
100
+ missed.length === 0,
101
+ missed.length ? `${missed.map((m) => m.id).join(", ")} report zero tests in a spelling the floor cannot read` : `${EMPTY_RUN_CANARIES.length} runners recognised`
102
+ );
103
+ }
104
+
105
+ // --- 2. Canaries ------------------------------------------------------------
106
+ const canaryResults = new Map();
107
+ {
108
+ const silent = [];
109
+ const noDenominator = [];
110
+ for (const c of TAMPER_CANARIES) {
111
+ const res = checkTestTampering(canaryDiff(c));
112
+ const hit = (res.violations || []).some((v) => v.type === c.expect);
113
+ canaryResults.set(c.id, hit);
114
+ if (!hit) silent.push(`${c.id} expected ${c.expect}, got ${JSON.stringify((res.violations || []).map((v) => v.type))}`);
115
+ // A finding with no denominator is the shape this script exists to reject.
116
+ if (hit && !(res.inputsSeen > 0)) noDenominator.push(c.id);
117
+ }
118
+ add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length} canaries red as required`);
119
+ add("canaries: every finding carries a denominator", noDenominator.length === 0, noDenominator.length ? noDenominator.join(", ") : "inputsSeen > 0 on every hit");
120
+ }
121
+
122
+ // --- 3. Predicate mutants ---------------------------------------------------
123
+ {
124
+ const survivors = [];
125
+ for (const mutant of PREDICATE_MUTANTS) {
126
+ let killed = false;
127
+ for (const c of TAMPER_CANARIES) {
128
+ // Only canaries the healthy predicate catches can kill a mutant.
129
+ if (!canaryResults.get(c.id)) continue;
130
+ const res = checkTestTampering(canaryDiff(c), { isTestPath: mutant.fn });
131
+ if (!(res.violations || []).some((v) => v.type === c.expect)) {
132
+ killed = true;
133
+ break;
134
+ }
135
+ }
136
+ if (!killed) survivors.push(`${mutant.id} (${mutant.why})`);
137
+ }
138
+ add(
139
+ "mutants: blinding the predicate breaks a canary",
140
+ survivors.length === 0,
141
+ survivors.length
142
+ ? `survived: ${survivors.join(", ")} — no canary required the guard to activate`
143
+ : `${PREDICATE_MUTANTS.length} mutants killed`
144
+ );
145
+ }
146
+
147
+ // --- Report -----------------------------------------------------------------
148
+ const json = process.argv.includes("--json");
149
+ const activated = [...canaryResults.values()].filter(Boolean).length;
150
+
151
+ if (json) {
152
+ console.log(
153
+ JSON.stringify(
154
+ {
155
+ ok: failures.length === 0,
156
+ checks,
157
+ activationCoverage: { canaries: canaryResults.size, activated },
158
+ },
159
+ null,
160
+ 2
161
+ )
162
+ );
163
+ } else {
164
+ console.log("\n🎯 Guard Reach Check (activation coverage)");
165
+ console.log("-------------------------------------------------------");
166
+ for (const c of checks) console.log(` ${c.ok ? "✅" : "❌"} ${c.name.padEnd(46)} ${c.detail}`);
167
+ console.log("-------------------------------------------------------");
168
+ console.log(` canaries activated: ${activated}/${canaryResults.size}`);
169
+ console.log(
170
+ failures.length === 0
171
+ ? "✅ Every blocking guard can still be made red.\n"
172
+ : `\n❌ ${failures.length} guard(s) may have gone silent. A check that cannot be made red is not a check.\n`
173
+ );
174
+ }
175
+
176
+ process.exit(failures.length === 0 ? 0 : 1);
@@ -41,6 +41,23 @@ try {
41
41
  process.exit(1);
42
42
  }
43
43
 
44
+ // 1a. Activation coverage (blocking).
45
+ //
46
+ // Step 1 proved the suite is green. Green is only evidence if the guards were
47
+ // switched on: a check that silently stopped applying contributes passing
48
+ // tests and a zero exit code exactly like one that ran. This asks the question
49
+ // the suite cannot — can every blocking guard still be made red? — and it runs
50
+ // before the doc-sync gate because a silent guard makes every later signal
51
+ // meaningless.
52
+ console.log("1a. Verifying every blocking guard can still be made red...");
53
+ try {
54
+ execSync("node scripts/guard-reach-check.mjs", { cwd: root, stdio: "inherit" });
55
+ console.log("");
56
+ } catch (_) {
57
+ console.error("\n❌ Release Aborted: a guard has gone silent. See the failing rows above.");
58
+ process.exit(1);
59
+ }
60
+
44
61
  // 1b. Documentation / version consistency gate (blocking).
45
62
  console.log("1b. Verifying documentation is in sync with package.json & test suite...");
46
63
  {
package/src/config.mjs CHANGED
@@ -54,6 +54,9 @@ const CI_DEFINITIONS = [
54
54
  "appveyor.yml",
55
55
  ".teamcity/**",
56
56
  ".githooks/**",
57
+ "buildspec.yml",
58
+ "**/buildspec.yml",
59
+ ".buildspec/**",
57
60
  ];
58
61
 
59
62
  export const BUILTIN_DENY = [
@@ -64,6 +67,24 @@ export const BUILTIN_DENY = [
64
67
  "**/*.key",
65
68
  "**/id_rsa*",
66
69
  ".agent/jules-queue/**",
70
+
71
+ // Shell that runs on `cd`, and credentials in plaintext. Same class as a CI
72
+ // definition: code or secrets that take effect before anyone reviews them.
73
+ "**/.envrc",
74
+ "**/.git-credentials",
75
+ "**/.aws/**",
76
+ "**/.ssh/**",
77
+ "**/.kube/**",
78
+ "**/kubeconfig*",
79
+ "**/.docker/config.json",
80
+ "**/*.p12",
81
+ "**/*.pfx",
82
+ "**/*.p8",
83
+ "**/id_ed25519*",
84
+ "**/credentials.json",
85
+ "**/service-account*.json",
86
+ "**/*.tfstate",
87
+ "**/*.tfstate.*",
67
88
  ...CI_DEFINITIONS,
68
89
  ];
69
90
 
@@ -110,6 +131,53 @@ export const BUILTIN_PROTECT = [
110
131
  "Dockerfile",
111
132
  "**/Dockerfile",
112
133
 
134
+ // Lockfiles decide which code actually *runs*. `package.json` was protected
135
+ // and `package-lock.json` was not, so an agent could change a resolved URL
136
+ // or an integrity hash — swapping the code that gets installed — without
137
+ // touching a single declared dependency, and the gate said nothing. The
138
+ // entropy scanner is deliberately blind to lockfiles too (they are full of
139
+ // hashes), so the change was invisible twice over. `BUILTIN_RESTRICTED` in
140
+ // risk.mjs already knew these mattered; only the risk tier consumed it,
141
+ // never checkScope.
142
+ "package-lock.json",
143
+ "**/package-lock.json",
144
+ "pnpm-lock.yaml",
145
+ "**/pnpm-lock.yaml",
146
+ "yarn.lock",
147
+ "**/yarn.lock",
148
+ "bun.lockb",
149
+ "bun.lock",
150
+ "Cargo.lock",
151
+ "**/Cargo.lock",
152
+ "go.sum",
153
+ "**/go.sum",
154
+ "poetry.lock",
155
+ "uv.lock",
156
+ "Pipfile.lock",
157
+ "Gemfile.lock",
158
+ "composer.lock",
159
+ "gradle.lockfile",
160
+ "npm-shrinkwrap.json",
161
+ "mix.lock",
162
+ "pubspec.lock",
163
+ "Podfile.lock",
164
+ "Package.resolved",
165
+ ".terraform.lock.hcl",
166
+
167
+ // Toolchain pins choose the compiler that runs the whole suite.
168
+ ".nvmrc",
169
+ ".tool-versions",
170
+ ".mise.toml",
171
+ "rust-toolchain",
172
+ "rust-toolchain.toml",
173
+ "**/gradle/wrapper/gradle-wrapper.properties",
174
+ "**/gradle/wrapper/gradle-wrapper.jar",
175
+
176
+ // Who has to approve a change, and what runs on every developer's machine.
177
+ "CODEOWNERS",
178
+ "docs/CODEOWNERS",
179
+ ".pre-commit-config.yaml",
180
+
113
181
  // Test-runner configuration decides which tests run and what counts as a
114
182
  // pass. Rewriting it is the cheapest way to make a suite green without
115
183
  // touching a single assertion — `--passWithNoTests`, an added ignore
package/src/coverage.mjs CHANGED
@@ -300,12 +300,27 @@ export function calculateDiffCoverage(coverageByFile, diffStr = "", options = {}
300
300
  }
301
301
  }
302
302
 
303
- const score = totalLines > 0 ? Math.round((coveredLines / totalLines) * 10000) / 100 : 100;
304
- const ok = score >= minCoverage;
303
+ // 100% of nothing is not 100%.
304
+ //
305
+ // The denominator counts only the added lines V8 actually mapped, and V8
306
+ // maps nothing outside Node. So a Python diff adding three executable lines
307
+ // measured zero of them and was reported as `score: 100` — the best possible
308
+ // number, produced by a measurement that never happened, on 20-odd of the 25
309
+ // stacks this kit claims to support. `mutation.mjs` had the identical bug and
310
+ // was fixed in v0.57.0; this is the same shape one module over.
311
+ //
312
+ // `ok` stays true because nothing failed to be covered, and a gate that
313
+ // blocks every non-Node diff gets switched off. What changes is the claim:
314
+ // `scored: false` and a reason, instead of a number nobody measured.
315
+ const scored = totalLines > 0;
316
+ const score = scored ? Math.round((coveredLines / totalLines) * 10000) / 100 : null;
317
+ const ok = scored ? score >= minCoverage : true;
305
318
 
306
319
  return {
307
320
  ok,
308
321
  score,
322
+ scored,
323
+ ...(scored ? {} : { reason: "No added executable lines were measurable — V8 coverage only observes code Node itself ran, so nothing was scored." }),
309
324
  minCoverage,
310
325
  totalLines,
311
326
  coveredLines,
package/src/engine.mjs CHANGED
@@ -1,5 +1,6 @@
1
1
  import { loadConfig, parseYaml, normalizeScope } from "./config.mjs";
2
2
  import { isTestPath } from "./test-paths.mjs";
3
+ import { checkCollectionFloor } from "./ops/test-collection.mjs";
3
4
  import { checkScope, scanDiff, scanBinaryPayloads, redactSecrets } from "./security.mjs";
4
5
  import { changedFiles, diffBytes, diffText, binaryDiffEntries, symlinkChanges, showFromOrigin, runCmd } from "./git.mjs";
5
6
  import { createProvider, ProviderRateLimitError, ProviderUnavailableError } from "./provider.mjs";
@@ -275,7 +276,11 @@ export async function gate(opts = {}) {
275
276
  }
276
277
 
277
278
  // Phase 3: Diff Secret Scanner & Security Checks
278
- const secretResult = scanDiff(diffStr, { root, allowTestModifications: opts.allowTestModifications === true });
279
+ const secretResult = scanDiff(diffStr, {
280
+ root,
281
+ allowTestModifications: opts.allowTestModifications === true,
282
+ allowTestChanges: opts.allowTestChanges,
283
+ });
279
284
  // A binary file reaches the scanner as one summary line, so its contents were
280
285
  // never looked at — a NUL byte in front of a token was enough to hide it.
281
286
  // Inspect those files directly and fold the verdict in.
@@ -517,7 +522,25 @@ export async function gate(opts = {}) {
517
522
  const ranNoVerification = !executionRecords.some((r) => r && r.kind !== "assert");
518
523
  const missingOracle = verificationRequired && ranNoVerification;
519
524
 
520
- const verifyOk = !failingCmd && !testTampered && !missingOracle;
525
+ // A command that ran is not the same as a command that tested something.
526
+ //
527
+ // `missingOracle` above catches "no stage executed". It cannot catch the
528
+ // case where a stage executed, exited 0, and collected zero tests — which
529
+ // several runners report as success by design: `go test ./...` prints
530
+ // "[no test files]" and exits 0, jest has --passWithNoTests, and
531
+ // `npm test --workspaces` is green when the one package the diff touched
532
+ // has no suite. A repository could invert a function, add an untested one
533
+ // and collect five green phases, verified against nothing at all.
534
+ //
535
+ // The count is read out of the runner's own summary and only a *stated*
536
+ // zero counts. An unrecognised runner yields null and passes: failing on
537
+ // "I could not tell" would break every runner not on the list.
538
+ const collectionFloor = verificationRequired
539
+ ? checkCollectionFloor(testResult, { minTests: trustedVerify.minTests })
540
+ : { ok: true, count: null, runner: null, reason: null };
541
+ const emptySuite = !collectionFloor.ok;
542
+
543
+ const verifyOk = !failingCmd && !testTampered && !missingOracle && !emptySuite;
521
544
 
522
545
  // What actually broke. Without this the verify phase reported `ok: false` and
523
546
  // nothing else — not the stage, not the exit code, not a line of output — so
@@ -564,7 +587,18 @@ export async function gate(opts = {}) {
564
587
  "The gate approves a change because verification passed. Zero stages executed is not a pass.",
565
588
  ],
566
589
  }
567
- : null;
590
+ : emptySuite
591
+ ? {
592
+ stageId: "empty-suite",
593
+ command: testResult?.command || null,
594
+ exitCode: 0,
595
+ stdout: "",
596
+ stderr: collectionFloor.reason,
597
+ diagnostics: [
598
+ "An exit code of 0 from a runner that collected no tests is not evidence about this change.",
599
+ ],
600
+ }
601
+ : null;
568
602
 
569
603
  // Generate & persist Evidence Manifest
570
604
  const evidenceManifest = generateEvidenceManifest(root, {
@@ -579,6 +613,7 @@ export async function gate(opts = {}) {
579
613
  failedStage: failingCmd?.stageId || failingCmd?.phase || null,
580
614
  diagnostics: failingCmd?.diagnostics?.length ? failingCmd.diagnostics : (failingCmd?.stderr ? [failingCmd.stderr] : []),
581
615
  metrics: failingCmd?.metrics || {},
616
+ collection: collectionFloor,
582
617
  ok: verifyOk,
583
618
  });
584
619
  if (testTampered) {
package/src/evidence.mjs CHANGED
@@ -129,7 +129,20 @@ export function computeDirectoryHash(root, options = {}) {
129
129
  .filter((p) => existsSync(join(root, p)) && statSync(join(root, p)).isFile())
130
130
  .sort();
131
131
  } else {
132
- const targetDirs = options.directories || ["test", "tests", "__tests__", "spec", "src"];
132
+ // A sixth spelling of "where do the tests live?", and the one that got
133
+ // missed when the other five were unified behind `isTestPath`: this is a
134
+ // list of *directory names at the repository root*, not a predicate. Go
135
+ // puts its tests beside the code (`internal/calc/calc_test.go`), and every
136
+ // monorepo puts them under `packages/*/test/`. Neither is under a
137
+ // root-level `test/`, so the walk found nothing, `fileCount` was 0, and
138
+ // `strictTestLock` — which requires `fileCount > 0` — switched itself off
139
+ // without saying so. The tree hash then became the SHA-256 of the empty
140
+ // string, and the evidence manifest attested to it.
141
+ //
142
+ // The named directories stay as a fast path; when they yield nothing, walk
143
+ // the repository and let the shared predicate decide. The walk already
144
+ // skips node_modules, vendor, target and the build caches.
145
+ const targetDirs = options.directories || ["test", "tests", "__tests__", "spec", "specs", "src"];
133
146
  for (const dirName of targetDirs) {
134
147
  const dirPath = join(root, dirName);
135
148
  if (existsSync(dirPath)) {
@@ -137,6 +150,11 @@ export function computeDirectoryHash(root, options = {}) {
137
150
  fileList.push(...found);
138
151
  }
139
152
  }
153
+ if (options.testOnly && !fileList.some((f) => isTestPath(f))) {
154
+ for (const f of findFilesRecursively(root, root)) {
155
+ if (isTestPath(f)) fileList.push(f);
156
+ }
157
+ }
140
158
 
141
159
  // Plenty of projects keep `app.test.mjs` or `index.js` beside package.json
142
160
  // rather than under one of the directories above, and those files were
@@ -201,6 +219,7 @@ export function computeEvidenceHash(manifest) {
201
219
  intent: manifest.intent,
202
220
  provenance: manifest.provenance,
203
221
  testIntegrity: manifest.testIntegrity,
222
+ ...(manifest.verification ? { verification: manifest.verification } : {}),
204
223
  ...(manifest.sourceIntegrity ? { sourceIntegrity: manifest.sourceIntegrity } : {}),
205
224
  executionRecords: manifest.executionRecords,
206
225
  securityChecks: manifest.securityChecks,
@@ -342,6 +361,24 @@ export function generateEvidenceManifest(root = process.cwd(), options = {}) {
342
361
  maxDiffKb: options.maxDiffKb || 75,
343
362
  protectedScopeOk: options.protectedScopeOk ?? true,
344
363
  },
364
+ // How many tests the runner said it collected, and whether it said at all.
365
+ //
366
+ // The collection floor fails a *stated* zero and lets an unstated count
367
+ // pass, because failing on "I could not tell" would break every runner not
368
+ // on the list. That is the right call for the verdict and the wrong thing
369
+ // to leave out of the record: a manifest that says nothing here reads as
370
+ // though a suite ran. `counted: false` is the honest shape for a run where
371
+ // the number was never observable — a quiet runner (`cargo test --quiet`,
372
+ // `pytest -q`) suppresses the very line the floor reads.
373
+ ...(options.collection
374
+ ? {
375
+ verification: {
376
+ testsCollected: options.collection.count,
377
+ counted: options.collection.count !== null,
378
+ runner: options.collection.runner,
379
+ },
380
+ }
381
+ : {}),
345
382
  ...(diagnostics.length > 0 ? { diagnostics } : {}),
346
383
  ...(Object.keys(metrics).length > 0 ? { metrics } : {}),
347
384
  };
@@ -0,0 +1,149 @@
1
+ /**
2
+ * How many tests the runner actually collected, read out of its own output.
3
+ *
4
+ * The gate's oracle is one number: the exit code of the verification command.
5
+ * That number cannot distinguish "every test passed" from "there were no
6
+ * tests". Several runners report the second case as success, by design:
7
+ *
8
+ * go test ./... → "? example.com/app [no test files]", exit 0
9
+ * jest --passWithNoTests → "No tests found, exiting with code 0"
10
+ * npm test --workspaces → exit 0 when the changed package has no suite
11
+ * pytest --exitfirst on a path that matches nothing, in some configurations
12
+ *
13
+ * So a repository could invert a function, add an untested one, and collect
14
+ * five green phases — verified against nothing. `verify.required: false` is
15
+ * the switch for a repository that genuinely has no oracle; silently passing
16
+ * is not.
17
+ *
18
+ * The parsing is deliberately one-sided. A count is only returned when the
19
+ * runner stated one in a form recognised here; an unrecognised runner yields
20
+ * `null`, and null is not a failure. Failing on "I could not tell" would break
21
+ * every runner not on this list, which is most of them.
22
+ */
23
+
24
+ /**
25
+ * Patterns that state a test count, per runner family.
26
+ *
27
+ * Each entry captures a single number. The first pattern that matches wins,
28
+ * so the more specific summaries come first.
29
+ */
30
+ const COUNT_PATTERNS = [
31
+ // node:test — spec reporter ("ℹ tests 940") and tap ("# tests 940")
32
+ { name: "node:test", re: /^[^\n]*?(?:ℹ|#)\s*tests\s+(\d+)\s*$/m },
33
+ // pytest — "collected 12 items", "12 passed", "no tests ran in 0.01s"
34
+ { name: "pytest", re: /^\s*collected\s+(\d+)\s+items?/m },
35
+ { name: "pytest", re: /=+\s*(\d+)\s+passed/m },
36
+ // cargo — "running 7 tests"
37
+ { name: "cargo", re: /^\s*running\s+(\d+)\s+tests?\s*$/m },
38
+ // jest / vitest — "Tests: 12 passed, 12 total"
39
+ { name: "jest", re: /^\s*Tests:\s+.*?(\d+)\s+total\s*$/m },
40
+ // mocha — "12 passing"
41
+ { name: "mocha", re: /^\s*(\d+)\s+passing/m },
42
+ // Maven / Surefire — "Tests run: 12, Failures: 0"
43
+ { name: "surefire", re: /\bTests run:\s*(\d+)/i },
44
+ // PHPUnit — "OK (12 tests, 30 assertions)"
45
+ { name: "phpunit", re: /\bOK\s*\((\d+)\s+tests?/i },
46
+ // RSpec / ExUnit — "12 examples, 0 failures" / "12 tests, 0 failures"
47
+ { name: "rspec", re: /^\s*(\d+)\s+examples?,\s*\d+\s+failures?/m },
48
+ { name: "exunit", re: /^\s*(\d+)\s+tests?,\s*\d+\s+failures?/m },
49
+ // dotnet test — "Total tests: 12" / "Passed! - Failed: 0, Passed: 12"
50
+ { name: "dotnet", re: /\bTotal(?:\s+tests)?:\s*(\d+)/i },
51
+ // swift test / XCTest — "Executed 12 tests"
52
+ { name: "xctest", re: /\bExecuted\s+(\d+)\s+tests?/i },
53
+ ];
54
+
55
+ /** Per-test lines, which `go test` only prints under -v. */
56
+ const GO_PER_TEST = /^\s*--- (?:PASS|FAIL|SKIP):/gm;
57
+
58
+ /** Phrases that state, in so many words, that nothing was collected. */
59
+ const EXPLICIT_ZERO = [
60
+ { name: "pytest", re: /\bno tests ran\b/i },
61
+ { name: "pytest", re: /^\s*collected\s+0\s+items?/m },
62
+ { name: "jest", re: /\bNo tests found\b/i },
63
+ { name: "vitest", re: /\bNo test files found\b/i },
64
+ { name: "mocha", re: /^\s*0\s+passing/m },
65
+ { name: "cargo", re: /^\s*running\s+0\s+tests?\s*$/m },
66
+ { name: "phpunit", re: /\bNo tests executed!/i },
67
+ { name: "gradle", re: /^>\s*Task\s+:\S*test\S*\s+NO-SOURCE\s*$/mi },
68
+ { name: "ctest", re: /\bNo tests were found\b/i },
69
+ { name: "flutter", re: /\bNo tests ran\.?/i },
70
+ ];
71
+
72
+ /** Go prints this per package that has no test files at all. */
73
+ const GO_NO_TEST_FILES = /\[no test files\]/;
74
+ /** Any sign that a Go package did run tests. */
75
+ const GO_RAN_SOMETHING = /^(?:ok|FAIL|---\s+(?:PASS|FAIL|SKIP)):?\s/m;
76
+
77
+ /**
78
+ * Read a collected-test count out of a runner's output.
79
+ *
80
+ * @param {string} [stdout]
81
+ * @param {string} [stderr]
82
+ * @returns {{ count: number|null, runner: string|null }}
83
+ * `count` is null when no recognised runner stated one — which is not a
84
+ * finding, only an absence of evidence.
85
+ */
86
+ export function parseCollectedTests(stdout = "", stderr = "") {
87
+ const text = `${stdout || ""}\n${stderr || ""}`;
88
+ if (!text.trim()) return { count: null, runner: null };
89
+
90
+ for (const rule of EXPLICIT_ZERO) {
91
+ if (rule.re.test(text)) return { count: 0, runner: rule.name };
92
+ }
93
+
94
+ // Go states absence per package rather than as a count, so it needs its own
95
+ // pass before the generic patterns.
96
+ if (GO_NO_TEST_FILES.test(text) || GO_RAN_SOMETHING.test(text)) {
97
+ // Only a run where *no* package did anything is a zero: a monorepo where
98
+ // one package has no tests and three do is a normal, healthy repository.
99
+ if (!GO_RAN_SOMETHING.test(text)) return { count: 0, runner: "go" };
100
+ // Something ran. `--- PASS:` lines are per-test but appear only under -v,
101
+ // so their absence means the count was not stated — not that it was zero.
102
+ // Reporting zero here would have failed every ordinary `go test ./...`.
103
+ const perTest = text.match(GO_PER_TEST);
104
+ return { count: perTest && perTest.length > 0 ? perTest.length : null, runner: "go" };
105
+ }
106
+
107
+ for (const rule of COUNT_PATTERNS) {
108
+ const m = rule.re.exec(text);
109
+ if (!m) continue;
110
+ const n = Number(m[1]);
111
+ if (Number.isFinite(n)) return { count: n, runner: rule.name };
112
+ }
113
+
114
+ return { count: null, runner: null };
115
+ }
116
+
117
+ /**
118
+ * Decide whether a passing verification command actually verified anything.
119
+ *
120
+ * @param {object} testResult - the test stage's result ({ ok, stdout, stderr, command })
121
+ * @param {object} [opts]
122
+ * @param {number} [opts.minTests=1] - the floor, from `verify.minTests`.
123
+ * @returns {{ ok: boolean, count: number|null, runner: string|null, reason: string|null }}
124
+ */
125
+ export function checkCollectionFloor(testResult, opts = {}) {
126
+ const minTests = Number.isFinite(opts.minTests) ? opts.minTests : 1;
127
+ if (minTests <= 0) return { ok: true, count: null, runner: null, reason: null };
128
+ // Only a *passing* command can lie about this. A failing one already fails.
129
+ if (!testResult || testResult.ok !== true) {
130
+ return { ok: true, count: null, runner: null, reason: null };
131
+ }
132
+
133
+ const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
134
+ if (count === null || count >= minTests) {
135
+ return { ok: true, count, runner, reason: null };
136
+ }
137
+
138
+ return {
139
+ ok: false,
140
+ count,
141
+ runner,
142
+ reason:
143
+ `The verification command exited 0 without running any tests` +
144
+ (runner ? ` (${runner} reported ${count})` : "") +
145
+ `, so this change was approved against nothing. ` +
146
+ `Point verify.test at a suite that covers this repository, lower the floor with verify.minTests, ` +
147
+ `or — if this repository intentionally uses only the scope and secret phases — set verify.required: false.`,
148
+ };
149
+ }
package/src/security.mjs CHANGED
@@ -1138,8 +1138,11 @@ function locateFindingLine(lines, type, file = null) {
1138
1138
  // assertion — identical once every literal is blanked out — with different
1139
1139
  // values. That does not distinguish an attack from a deliberate change of
1140
1140
  // spec; nothing can, from a diff alone. This reports rather than decides,
1141
- // and `--allow-test-modifications` is the answer when the new expectation is
1142
- // the correct one.
1141
+ // and `--allow-test-change expectation` is the answer when the new
1142
+ // expectation is the correct one. Narrow on purpose: the blunt
1143
+ // `--allow-test-modifications` turns off the other five checks too, and a
1144
+ // check that can only be answered by disabling its neighbours ends up
1145
+ // disabling its neighbours.
1143
1146
 
1144
1147
  // An assertion that states a *specific* expected value. Counting assertions
1145
1148
  // alone let a test be gutted while looking untouched: swapping
@@ -1627,6 +1630,135 @@ const hasLiteralPlaceholder = (shape) =>
1627
1630
  const collapseWhitespace = (s) => s.replace(/\s+/g, " ").trim();
1628
1631
  const shorten = (s) => (s.length > 160 ? `${s.slice(0, 157)}…` : s);
1629
1632
 
1633
+ /**
1634
+ * Split the argument list of the outermost assertion call in `clean`.
1635
+ *
1636
+ * Comments are already stripped by the caller, so only string state has to be
1637
+ * tracked. Returns null whenever the shape is not confidently understood — a
1638
+ * truncated fragment, an unbalanced hunk, a quoting form not handled here —
1639
+ * because every caller uses this to *suppress* a finding, and failing to
1640
+ * understand a statement must never become a reason to stay quiet about it.
1641
+ *
1642
+ * @param {string} clean - comment-stripped statement text
1643
+ * @param {string} lang
1644
+ * @returns {string[] | null} top-level arguments, trimmed
1645
+ */
1646
+ function splitAssertionArgs(clean, lang) {
1647
+ SPECIFIC_ASSERTION.lastIndex = 0;
1648
+ const m = SPECIFIC_ASSERTION.exec(clean);
1649
+ if (!m) return null;
1650
+
1651
+ let i = m.index + m[0].length; // just past the opening paren
1652
+ let depth = 1;
1653
+ let quote = null;
1654
+ let triple = false;
1655
+ const args = [];
1656
+ let start = i;
1657
+
1658
+ while (i < clean.length) {
1659
+ const c = clean[i];
1660
+
1661
+ if (quote !== null) {
1662
+ if (c === "\\") { i += 2; continue; }
1663
+ if (triple && c === quote && clean[i + 1] === quote && clean[i + 2] === quote) {
1664
+ quote = null; triple = false; i += 3; continue;
1665
+ }
1666
+ if (!triple && c === quote) { quote = null; i += 1; continue; }
1667
+ i += 1;
1668
+ continue;
1669
+ }
1670
+
1671
+ if (c === '"' || c === "'" || c === "`") {
1672
+ if (lang === "python" && clean[i + 1] === c && clean[i + 2] === c) {
1673
+ quote = c; triple = true; i += 3; continue;
1674
+ }
1675
+ quote = c; i += 1; continue;
1676
+ }
1677
+
1678
+ if (c === "(" || c === "[" || c === "{") { depth += 1; i += 1; continue; }
1679
+ if (c === ")" || c === "]" || c === "}") {
1680
+ depth -= 1;
1681
+ if (depth === 0) {
1682
+ args.push(clean.slice(start, i).trim());
1683
+ return args;
1684
+ }
1685
+ i += 1;
1686
+ continue;
1687
+ }
1688
+ if (c === "," && depth === 1) {
1689
+ args.push(clean.slice(start, i).trim());
1690
+ start = i + 1;
1691
+ i += 1;
1692
+ continue;
1693
+ }
1694
+ i += 1;
1695
+ }
1696
+ return null; // never closed: an unbalanced fragment, so no suppression
1697
+ }
1698
+
1699
+ /** One plain string literal and nothing else. */
1700
+ const PURE_STRING_LITERAL = new RegExp(
1701
+ [
1702
+ "^'(?:\\\\.|[^'\\\\])*'$",
1703
+ '^"(?:\\\\.|[^"\\\\])*"$',
1704
+ "^`(?:\\\\.|[^`\\\\])*`$",
1705
+ '^"""[\\s\\S]*"""$',
1706
+ "^'''[\\s\\S]*'''$",
1707
+ ].join("|")
1708
+ );
1709
+
1710
+ function isPureStringLiteral(arg) {
1711
+ if (!arg) return false;
1712
+ return PURE_STRING_LITERAL.test(arg.trim());
1713
+ }
1714
+
1715
+ /**
1716
+ * Argument positions that carry a message for a human rather than an expected
1717
+ * value.
1718
+ *
1719
+ * Trailing, for `assert.equal(got, want, "message")` and
1720
+ * `assert_eq!(a, b, "message")`; leading, for Go's
1721
+ * `t.Errorf("got %d want %d", got, want)`. Two arguments is the classic
1722
+ * `(actual, expected)` shape, so a string in last position *there* is the
1723
+ * expected value: `assert.equal(name, "Alice")` must still be judged when
1724
+ * "Alice" becomes "Bob".
1725
+ */
1726
+ function messageArgIndices(args) {
1727
+ const idx = new Set();
1728
+ if (args.length >= 3 && isPureStringLiteral(args[args.length - 1])) idx.add(args.length - 1);
1729
+ if (args.length >= 2 && isPureStringLiteral(args[0])) idx.add(0);
1730
+ return idx;
1731
+ }
1732
+
1733
+ /**
1734
+ * True when two assertions differ only in text written to be read by a person.
1735
+ *
1736
+ * Rewording the message on a failing assertion is among the most common edits
1737
+ * any test file receives, and it says nothing whatsoever about what the suite
1738
+ * checks. But a message is a literal, so blanking literals made the two
1739
+ * statements the same shape and the pairing reported a rewritten expectation
1740
+ * every time somebody improved the wording of a failure. Firing on that is
1741
+ * how an operator learns to pass the override without reading it.
1742
+ */
1743
+ function differsOnlyInMessage(cleanRemoved, cleanAdded, lang) {
1744
+ const a = splitAssertionArgs(cleanRemoved, lang);
1745
+ const b = splitAssertionArgs(cleanAdded, lang);
1746
+ if (!a || !b || a.length !== b.length || a.length === 0) return false;
1747
+
1748
+ const msgIdx = messageArgIndices(a);
1749
+ if (msgIdx.size === 0) return false;
1750
+
1751
+ let sawDifference = false;
1752
+ for (let i = 0; i < a.length; i++) {
1753
+ if (a[i].replace(/\s+/g, "") === b[i].replace(/\s+/g, "")) continue;
1754
+ // A difference outside a message position, or in a position that stopped
1755
+ // being a plain string, is a real change.
1756
+ if (!msgIdx.has(i) || !isPureStringLiteral(b[i])) return false;
1757
+ sawDifference = true;
1758
+ }
1759
+ return sawDifference;
1760
+ }
1761
+
1630
1762
  /**
1631
1763
  * Pair rewritten expectations across the removed and added images of every
1632
1764
  * hunk of one file, and report each pair.
@@ -1664,14 +1796,14 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1664
1796
  if (s.removedLines.length === 0) continue;
1665
1797
  const clean = stripComments(s.text, lang);
1666
1798
  if (!isSpecificAssertion(clean)) continue;
1667
- oldCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1799
+ oldCands.push({ s, clean, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1668
1800
  }
1669
1801
  const newCands = [];
1670
1802
  for (const s of newStmts) {
1671
1803
  if (s.addedLines.length === 0) continue;
1672
1804
  const clean = stripComments(s.text, lang);
1673
1805
  if (!isSpecificAssertion(clean)) continue;
1674
- newCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1806
+ newCands.push({ s, clean, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1675
1807
  }
1676
1808
 
1677
1809
  // The t-th removed candidate of a shape pairs with the t-th added
@@ -1698,11 +1830,32 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1698
1830
  const pairs = [];
1699
1831
  for (const [shape, olds] of oldByShape) {
1700
1832
  const news = newByShape.get(shape) || [];
1701
- const k = Math.min(olds.length, news.length);
1833
+
1834
+ // Cancel the assertions that are byte-identical on both sides before
1835
+ // aligning anything.
1836
+ //
1837
+ // Reordering two assertions removes both and adds both back unchanged.
1838
+ // Positional alignment then matched the first removed against the first
1839
+ // added — a different assertion — and reported two rewritten
1840
+ // expectations for an edit that changed no expected value at all. The
1841
+ // same happened to an assertion that simply moved within its block.
1842
+ // What is present unchanged on both sides did not change; only the
1843
+ // residue can have been rewritten.
1844
+ const survivingNew = news.slice();
1845
+ const survivingOld = [];
1846
+ for (const o of olds) {
1847
+ const twin = survivingNew.findIndex((n) => n.canon === o.canon);
1848
+ if (twin === -1) survivingOld.push(o);
1849
+ else survivingNew.splice(twin, 1);
1850
+ }
1851
+
1852
+ const k = Math.min(survivingOld.length, survivingNew.length);
1702
1853
  for (let t = 0; t < k; t++) {
1703
- if (olds[t].canon !== news[t].canon) {
1704
- pairs.push({ r: olds[t].s, a: news[t].s });
1705
- }
1854
+ const r = survivingOld[t];
1855
+ const a = survivingNew[t];
1856
+ if (r.canon === a.canon) continue;
1857
+ if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
1858
+ pairs.push({ r: r.s, a: a.s });
1706
1859
  }
1707
1860
  }
1708
1861
 
@@ -1717,9 +1870,11 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1717
1870
  const sr = blankLiterals(stripComments(r.text, lang));
1718
1871
  const sa = blankLiterals(stripComments(a.text, lang));
1719
1872
  if (sr === sa && hasLiteralPlaceholder(sr)) {
1720
- const cr = stripComments(r.text, lang).replace(/\s+/g, "");
1721
- const ca = stripComments(a.text, lang).replace(/\s+/g, "");
1722
- if (cr !== ca) pairs.push({ r, a });
1873
+ const clr = stripComments(r.text, lang);
1874
+ const cla = stripComments(a.text, lang);
1875
+ if (clr.replace(/\s+/g, "") !== cla.replace(/\s+/g, "") && !differsOnlyInMessage(clr, cla, lang)) {
1876
+ pairs.push({ r, a });
1877
+ }
1723
1878
  }
1724
1879
  }
1725
1880
  }
@@ -1736,9 +1891,10 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1736
1891
  `"${shorten(collapseWhitespace(p.r.text))}" became "${shorten(collapseWhitespace(p.a.text))}". ` +
1737
1892
  `A deliberately changed spec looks identical to a test bent to match broken ` +
1738
1893
  `output, and a diff alone cannot tell the two apart, so this is flagged for ` +
1739
- `review rather than assumed. If the new expectation is correct, re-run with ` +
1740
- `--allow-test-modifications (which also silences the skip, vacuous, removal ` +
1741
- `and weakening checks for this diff).`,
1894
+ `review rather than assumed. If the new expectation is the correct one, ` +
1895
+ `re-run with --allow-test-change expectation — which allows exactly this ` +
1896
+ `check and leaves the skip, vacuous, commented, removal and weakening ` +
1897
+ `checks doing their job.`,
1742
1898
  });
1743
1899
 
1744
1900
  // Both sides are accounted for here, so they must not also feed the
@@ -1771,17 +1927,85 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1771
1927
  return allPairs;
1772
1928
  }
1773
1929
 
1930
+ /**
1931
+ * The tamper checks, by the name an operator uses to allow one of them.
1932
+ *
1933
+ * There was one override for all six, and it was a switch marked "off". A
1934
+ * deliberate change of spec rewrites what a test expects, which is
1935
+ * indistinguishable from bending a test to match broken output — so the honest
1936
+ * answer to that finding is sometimes an override. But reaching for it also
1937
+ * silenced injected `.skip()`, `expect(true).toBe(true)`, commented-out
1938
+ * assertions and outright deletions, none of which the operator had looked at.
1939
+ * The check with the highest firing rate therefore set the ceiling for every
1940
+ * other check in the bundle: the more useful this one became, the more often
1941
+ * it would be used to turn the others off.
1942
+ */
1943
+ export const TAMPER_KINDS = new Map([
1944
+ ["TEST_SKIP_INJECTION", "skip"],
1945
+ ["VACUOUS_ASSERTION", "vacuous"],
1946
+ ["COMMENTED_ASSERTION", "commented"],
1947
+ ["ASSERTION_REMOVAL", "removal"],
1948
+ ["ASSERTION_WEAKENED", "weakening"],
1949
+ ["ASSERTION_EXPECTATION_CHANGED", "expectation"],
1950
+ ]);
1951
+
1952
+ /** Every kind name, for CLI validation and help text. */
1953
+ export const TAMPER_KIND_NAMES = Object.freeze([...new Set(TAMPER_KINDS.values())].sort());
1954
+
1955
+ /**
1956
+ * Which tamper checks this run is allowed to stay quiet about.
1957
+ *
1958
+ * @param {object} options
1959
+ * @param {boolean} [options.allowTestModifications] - the blunt form: all of them.
1960
+ * @param {string|string[]} [options.allowTestChanges] - kind names, comma-separated or an array.
1961
+ * @returns {{ all: boolean, kinds: Set<string>, unknown: string[] }}
1962
+ */
1963
+ export function resolveAllowedTamperKinds(options = {}) {
1964
+ if (options.allowTestModifications === true) {
1965
+ return { all: true, kinds: new Set(TAMPER_KIND_NAMES), unknown: [] };
1966
+ }
1967
+ const raw = options.allowTestChanges;
1968
+ const list = (Array.isArray(raw) ? raw : [raw])
1969
+ .flatMap((v) => String(v == null ? "" : v).split(","))
1970
+ .map((v) => v.trim().toLowerCase())
1971
+ .filter(Boolean);
1972
+
1973
+ if (list.includes("all")) {
1974
+ return { all: true, kinds: new Set(TAMPER_KIND_NAMES), unknown: [] };
1975
+ }
1976
+ const kinds = new Set();
1977
+ const unknown = [];
1978
+ for (const name of list) {
1979
+ if (TAMPER_KIND_NAMES.includes(name)) kinds.add(name);
1980
+ else unknown.push(name);
1981
+ }
1982
+ return { all: false, kinds, unknown };
1983
+ }
1984
+
1774
1985
  /**
1775
1986
  * Detects test file assertion tampering, weakening, or test skips.
1776
1987
  *
1777
1988
  * @param {string} diffOrText - Unified git diff
1778
1989
  * @param {Object} [options]
1779
1990
  * @param {boolean} [options.allowTestModifications=false]
1780
- * @returns {{ ok: boolean, violations: Array<{ file: string, type: string, line?: number, reason: string }> }}
1991
+ * @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, status: "PASS"|"FAIL"|"NOT_APPLICABLE" }}
1992
+ * `status` distinguishes "checked and clean" from "nothing was checked";
1993
+ * `ok: true` alone cannot, and that ambiguity is the defect class this
1994
+ * field exists to make visible.
1781
1995
  */
1782
1996
  export function checkTestTampering(diffOrText = "", options = {}) {
1783
- if (!diffOrText || typeof diffOrText !== "string") return { ok: true, violations: [] };
1784
- if (options.allowTestModifications === true) return { ok: true, violations: [] };
1997
+ if (!diffOrText || typeof diffOrText !== "string") {
1998
+ return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "empty diff" };
1999
+ }
2000
+ const allowed = resolveAllowedTamperKinds(options);
2001
+ if (allowed.all) {
2002
+ return { ok: true, violations: [], inputsSeen: 0, status: "NOT_APPLICABLE", reason: "all kinds allowed" };
2003
+ }
2004
+
2005
+ // Which predicate decides what this guard even looks at. Injectable so the
2006
+ // meta-check can mutate it: a canary that still passes when the predicate is
2007
+ // replaced by `() => false` was never requiring this guard to activate.
2008
+ const isTestPath_ = typeof options.isTestPath === "function" ? options.isTestPath : isTestPath;
1785
2009
 
1786
2010
  const violations = [];
1787
2011
  const lines = diffOrText.split("\n");
@@ -1790,7 +2014,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1790
2014
  let currentOldLineNo = null;
1791
2015
  let currentNewLineNo = null;
1792
2016
 
1793
- const isTestFile = isTestPath;
2017
+ const isTestFile = isTestPath_;
1794
2018
 
1795
2019
  const SKIP_INJECTIONS = [
1796
2020
  { pattern: /\b(?:it|test|describe|context)\.skip\s*\(/i, desc: "Injected test skip (.skip())" },
@@ -1987,9 +2211,31 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1987
2211
  }
1988
2212
  }
1989
2213
 
2214
+ // A kind the operator has already looked at and accepted is dropped here
2215
+ // rather than never being computed, so the reasoning above stays one code
2216
+ // path regardless of what any given run allows.
2217
+ const reported =
2218
+ allowed.kinds.size === 0
2219
+ ? violations
2220
+ : violations.filter((v) => !allowed.kinds.has(TAMPER_KINDS.get(v.type)));
2221
+
2222
+ // What was examined, not only what was found.
2223
+ //
2224
+ // `ok: true` from a guard that looked at nothing is byte-identical to
2225
+ // `ok: true` from a guard that looked at everything and approved it. That
2226
+ // ambiguity is how a substring bug in the file classifier switched this
2227
+ // entire guard off for the standard pytest, Rust and RSpec layouts while
2228
+ // every signal stayed green. A verdict without a denominator is not a
2229
+ // verdict, so `inputsSeen` reports the number of test files this run
2230
+ // actually reasoned about, and `status` distinguishes "nothing to check"
2231
+ // from "checked and clean".
2232
+ const inputsSeen = fileAssertions.size;
2233
+
1990
2234
  return {
1991
- ok: violations.length === 0,
1992
- violations,
2235
+ ok: reported.length === 0,
2236
+ violations: reported,
2237
+ inputsSeen,
2238
+ status: reported.length > 0 ? "FAIL" : inputsSeen > 0 ? "PASS" : "NOT_APPLICABLE",
1993
2239
  };
1994
2240
  }
1995
2241
 
@@ -39,7 +39,65 @@ export function pytestCmd(env = process.env) {
39
39
  }
40
40
 
41
41
  /**
42
- * Generate a zero-dependency smoke test file using node:test for untested JS/Generic repos.
42
+ * Does this Makefile declare a `test` target?
43
+ *
44
+ * Read rather than assumed: the presence of the file says nothing about
45
+ * whether `make test` will run.
46
+ */
47
+ function makefileHasTestTarget(root) {
48
+ try {
49
+ const text = readFileSync(join(root, "Makefile"), "utf-8");
50
+ return /^\.PHONY:.*\btest\b/m.test(text) || /^test\s*:/m.test(text);
51
+ } catch (_) {
52
+ return false;
53
+ }
54
+ }
55
+
56
+ /**
57
+ * A declared test script that runs no tests and exits 0.
58
+ *
59
+ * This is the single most dangerous input the gate can receive, because every
60
+ * downstream check reads "a command ran and passed". `bootstrapZeroTestRepo`
61
+ * called `"test": "echo 'no tests yet' && exit 0"` an
62
+ * EXISTING_VERIFICATION_ORACLE — it asked whether the field was set, never
63
+ * what was in it.
64
+ *
65
+ * npm's own default (`echo "Error: no test specified" && exit 1`) is not a
66
+ * placeholder by this definition, and correctly so: it exits non-zero, which
67
+ * fails loudly rather than certifying nothing.
68
+ */
69
+ export function isPlaceholderTestScript(cmd) {
70
+ if (typeof cmd !== "string") return false;
71
+ const trimmed = cmd.trim();
72
+ if (!trimmed) return true;
73
+ // Drop the announcements; what matters is what the shell is left doing.
74
+ const remainder = trimmed
75
+ .split(/&&|;/)
76
+ .map((part) => part.trim())
77
+ .filter((part) => part && !/^(?:echo|printf|:)\b/.test(part));
78
+ if (remainder.length === 0) return true;
79
+ return remainder.every((part) => /^(?:exit\s+0|true|:)$/.test(part));
80
+ }
81
+
82
+ /**
83
+ * Generate the fallback verification oracle for a JS/generic repo with no tests.
84
+ *
85
+ * What this used to write could not fail:
86
+ *
87
+ * assert.ok(fs.existsSync(process.cwd()));
88
+ * assert.ok(fs.readdirSync(process.cwd()).length > 0);
89
+ *
90
+ * Both hold for every repository and every change, so the generated "oracle"
91
+ * was green against arbitrary broken code — and worse, it *silenced* the
92
+ * `missingOracle` guard in engine.mjs, which fires only when no command ran at
93
+ * all. A repository that honestly had no oracle was converted into one that
94
+ * claimed to have one. That is the tool writing its own blindness to disk.
95
+ *
96
+ * The other stacks already get a real static gate at this point — `tsc
97
+ * --noEmit`, `cargo check`, `go vet`, `compileall` — each of which fails on a
98
+ * real class of defect. This is the JavaScript equivalent: every source file
99
+ * must parse. It proves the code compiles, not that it works, and the caller
100
+ * says so; but a syntax error fails it, which is one more than before.
43
101
  */
44
102
  export function generateSmokeTestScript(root = process.cwd()) {
45
103
  const agentDir = join(root, ".agent");
@@ -48,15 +106,45 @@ export function generateSmokeTestScript(root = process.cwd()) {
48
106
  } catch (_) {}
49
107
 
50
108
  const smokePath = join(agentDir, "smoke.test.mjs");
51
- const content = `// Auto-generated zero-dependency smoke test gate (.agent/smoke.test.mjs)
109
+ const content = `// Auto-generated zero-dependency parse gate (.agent/smoke.test.mjs)
110
+ //
111
+ // Written by \`agentctl bootstrap\` for a repository that had no test suite.
112
+ // It proves that every source file still parses. It does NOT prove the code
113
+ // is correct — replace it with real tests as soon as there are any.
52
114
  import { test } from "node:test";
53
115
  import assert from "node:assert/strict";
54
- import fs from "node:fs";
116
+ import { readdirSync, statSync } from "node:fs";
117
+ import { join, extname } from "node:path";
118
+ import { spawnSync } from "node:child_process";
119
+
120
+ const SKIP = new Set([".git", "node_modules", "vendor", "dist", "build", "coverage", ".venv", "venv", ".next", ".agent"]);
121
+ const SOURCE = new Set([".js", ".mjs", ".cjs"]);
122
+
123
+ function sources(dir, acc = [], depth = 0) {
124
+ if (depth > 8) return acc;
125
+ for (const entry of readdirSync(dir, { withFileTypes: true })) {
126
+ if (entry.name.startsWith(".") && entry.name !== ".agent") continue;
127
+ if (SKIP.has(entry.name)) continue;
128
+ const full = join(dir, entry.name);
129
+ if (entry.isDirectory()) sources(full, acc, depth + 1);
130
+ else if (SOURCE.has(extname(entry.name))) acc.push(full);
131
+ }
132
+ return acc;
133
+ }
134
+
135
+ test("every source file parses", () => {
136
+ const files = sources(process.cwd());
55
137
 
56
- test("Zero-Test Bootstrapped Smoke Verification", () => {
57
- assert.ok(fs.existsSync(process.cwd()), "Working directory exists and is accessible");
58
- const entries = fs.readdirSync(process.cwd());
59
- assert.ok(entries.length > 0, "Repository contains files");
138
+ // A gate with nothing to check is not a passing gate. Reporting success
139
+ // over an empty file list is exactly the vacuous oracle this replaced.
140
+ assert.ok(files.length > 0, "No JavaScript sources found to verify — this is not an oracle. Set verify.test in .agent/config.yml.");
141
+
142
+ const broken = [];
143
+ for (const file of files) {
144
+ const res = spawnSync(process.execPath, ["--check", file], { encoding: "utf-8" });
145
+ if (res.status !== 0) broken.push(\`\${file}: \${(res.stderr || "").trim().split("\\n")[0]}\`);
146
+ }
147
+ assert.deepEqual(broken, [], \`\${broken.length} file(s) failed to parse\`);
60
148
  });
61
149
  `;
62
150
  writeFileSync(smokePath, content, "utf-8");
@@ -173,7 +261,15 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
173
261
  if (existsSync(join(projectRoot, "Package.swift"))) {
174
262
  return { ...container, stack: "swift", testCmd: "swift test", buildCmd: "swift build", triggerFile: "Package.swift" };
175
263
  }
176
- if (existsSync(join(projectRoot, "app.json")) || existsSync(join(projectRoot, "react-native.config.js"))) {
264
+ // `app.json` is not a manifest, it is an Expo/React Native *configuration*
265
+ // file, and the name is generic enough that unrelated projects use it. Its
266
+ // test command is `npm test`, so without a package.json beside it the
267
+ // detector was claiming a Node stack for a repository that has no Node in
268
+ // it: `Cargo.toml` + `app.json` was measured as `react-native` / `npm test`.
269
+ if (
270
+ (existsSync(join(projectRoot, "app.json")) && existsSync(join(projectRoot, "package.json"))) ||
271
+ existsSync(join(projectRoot, "react-native.config.js"))
272
+ ) {
177
273
  const triggerFile = existsSync(join(projectRoot, "app.json")) ? "app.json" : "react-native.config.js";
178
274
  return { ...container, stack: "react-native", testCmd: "npm test", buildCmd: "npx react-native bundle --platform android --dev false --entry-file index.js --bundle-output android/main.jsbundle", triggerFile };
179
275
  }
@@ -188,7 +284,14 @@ export function detectPolyglotStack(projectRoot = process.cwd()) {
188
284
  if (existsSync(join(projectRoot, "go.mod"))) {
189
285
  return { ...container, stack: "go", testCmd: "go test ./...", buildCmd: "go build ./...", triggerFile: "go.mod" };
190
286
  }
191
- if (existsSync(join(projectRoot, "Makefile"))) {
287
+ // A Makefile is only an oracle if it declares the target we are about to
288
+ // run. `make test` on a Makefile with only a `build:` target exits 2 with
289
+ // "No rule to make target 'test'" — measured on a repository whose
290
+ // package.json declared a perfectly good `vitest run`, because the Makefile
291
+ // was checked first and the presence of the *file* was the whole test. A
292
+ // hard red on day one is how a user learns the gate is broken and turns it
293
+ // off, so the file must earn the claim.
294
+ if (existsSync(join(projectRoot, "Makefile")) && makefileHasTestTarget(projectRoot)) {
192
295
  return { ...container, stack: "make", testCmd: "make test", buildCmd: "make build", triggerFile: "Makefile" };
193
296
  }
194
297
 
@@ -715,7 +818,7 @@ export function bootstrapZeroTestRepo(root = process.cwd(), options = {}) {
715
818
  if (existsSync(join(root, "package.json"))) {
716
819
  try {
717
820
  const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
718
- hasPkgTest = Boolean(pkg?.scripts?.test);
821
+ hasPkgTest = Boolean(pkg?.scripts?.test) && !isPlaceholderTestScript(pkg.scripts.test);
719
822
  } catch (_) {}
720
823
  }
721
824
 
@@ -265,9 +265,13 @@ export function scorePromptFalsifiability(promptText, options = {}) {
265
265
  let isTrivial = false;
266
266
 
267
267
  if (!verifyCmd) {
268
- const detectedOracles = detectStackOracles(rootDir);
269
- if (detectedOracles.length > 0 && detectedOracles[0].testCmd) {
270
- verifyCmd = detectedOracles[0].testCmd;
268
+ // `detectStackOracles` returns an object, not an array: `.length` was
269
+ // undefined, so this branch never ran and every task without an explicit
270
+ // --verify-cmd was scored MISSING_ORACLE even in a repository with a
271
+ // working suite.
272
+ const detected = detectStackOracles(rootDir);
273
+ if (detected?.candidates?.testCmd) {
274
+ verifyCmd = detected.candidates.testCmd;
271
275
  autoDetected = true;
272
276
  }
273
277
  }