jules-orchestrator-kit 0.64.0 → 0.66.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -49,12 +49,15 @@
49
49
  <a id="quickstart"></a>
50
50
  ## Quickstart
51
51
 
52
- Get running in any repository in 3 commands (zero configuration required):
52
+ Get running in any repository in 3 commands. `init` asks seven questions and
53
+ fills in a sensible answer for each; `--yes` accepts all of them, detects the
54
+ stack, and probes the test command it picked before writing it down.
53
55
 
54
56
  ```bash
55
57
  # 1. Scaffold config, AGENTS.md, role prompts and guardrails
56
58
  # (auto-detects Python, Rust, Go, Node, PHP, etc.)
57
- npx jules-orchestrator-kit init
59
+ # Drop --yes to choose provider, plan, profile and workflows yourself.
60
+ npx jules-orchestrator-kit init --yes
58
61
  ```
59
62
 
60
63
  ```bash
@@ -204,7 +207,7 @@ To maximize PR merge rates, dispatch tasks according to deterministic boundaries
204
207
  * **Fail-Closed Security & Secret Redaction:** Evaluates explicit Deny rules before Allow rules against canonicalized, case-folded paths. Redacts high-entropy keys and base64-encoded credentials (such as Kubernetes `Secret` manifests).
205
208
  * **Complexity & Cost Router:** Zero-dependency heuristic classifier (`src/router.mjs`) routing mechanical tasks to lightweight models while reserving primary models for complex refactors, with a `node --check` syntax-verification gate that transparently escalates a FAST-tier result to the primary provider if it left broken JS on disk.
206
209
  * **Terminal UI & Diagnostic Matrix (`agentctl doctor`):** Interactive terminal dashboard, task sidecar manager, and automated transactional self-repair.
207
- * **Verified Test Suite:** Tested with **1085 unit tests across 150 suites passing in < 15.0s**.
210
+ * **Verified Test Suite:** Tested with **1133 unit tests across 163 suites passing in < 15.0s**.
208
211
 
209
212
  <br/>
210
213
 
package/bin/agentctl.mjs CHANGED
@@ -429,6 +429,14 @@ async function main() {
429
429
  if (p.violations) {
430
430
  p.violations.forEach((v) => console.log(` - Violation: ${v.file} (Rule: ${v.rule})`));
431
431
  }
432
+ if (p.unverified) {
433
+ console.log(` - Unverified: ${p.unverified}`);
434
+ }
435
+ if (p.setup) {
436
+ console.log(` - Setup: accepted ${p.setup.length} gate scaffold file(s) this repository did not have yet`);
437
+ p.setup.forEach((f) => console.log(` ${f}`));
438
+ console.log(` Commit them to the base branch and the full protect rules apply from then on.`);
439
+ }
432
440
  if (p.findings) {
433
441
  p.findings.forEach((f) => console.log(` - [${f.severity}] ${f.type}: ${f.description}`));
434
442
  }
@@ -2662,13 +2670,21 @@ async function main() {
2662
2670
  const taskId = values.task || "unknown";
2663
2671
  const agent = values.agent || "jules";
2664
2672
 
2673
+ const { CONFIRM_AFTER } = await import("../src/memory.mjs");
2665
2674
  const res = harvestFailure(root, { exitCode, logPath, diffText, taskId, agent });
2666
2675
  if (values.json) {
2667
2676
  console.log(JSON.stringify(res, null, 2));
2668
2677
  } else {
2669
2678
  if (res.status === "HARVESTED") {
2670
- console.log(`🌾 Harvested failure candidate: ${res.candidate.trigger}`);
2671
- console.log(` Solution: ${res.candidate.solution}`);
2679
+ // No "Solution:" line any more, because there is no solution: the
2680
+ // repair loop exhausting its budget is the definition of not having
2681
+ // found one. It used to print a fixed sentence here and store it as
2682
+ // if it were a remedy.
2683
+ console.log(`🌾 Recorded failure observation: ${res.candidate.trigger}`);
2684
+ console.log(` Seen ${res.occurrences}× — ${res.confirmed ? "stated as a rule" : `not injected into prompts until it recurs ${CONFIRM_AFTER}×`}`);
2685
+ if (!res.confirmed) {
2686
+ console.log(` Record an actual fix for it with: agentctl learning add "<trigger>" "<solution>"`);
2687
+ }
2672
2688
  } else {
2673
2689
  console.log(`⚠️ Harvest rejected: ${res.reason}`);
2674
2690
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "jules-orchestrator-kit",
3
- "version": "0.64.0",
3
+ "version": "0.66.0",
4
4
  "description": "Zero-dependency safety gatekeeper, test oracle generator, and multi-agent coordination protocol for autonomous coding agents — Google Jules, Claude Code, Codex and Gemini CLI.",
5
5
  "repository": {
6
6
  "type": "git",
@@ -49,8 +49,10 @@ import { parseCollectedTests } from "../src/ops/test-collection.mjs";
49
49
  import {
50
50
  TEST_PATH_CASES,
51
51
  TAMPER_CANARIES,
52
+ MULTILINE_CANARIES,
52
53
  PREDICATE_MUTANTS,
53
54
  EMPTY_RUN_CANARIES,
55
+ COUNTED_RUN_CANARIES,
54
56
  SCOPE_CANARIES,
55
57
  INNOCENT_EDITS,
56
58
  UNREADABLE_DIALECTS,
@@ -67,6 +69,9 @@ import {
67
69
  function canaryDiff(c) {
68
70
  const ctx = c.context || "// context";
69
71
  const lines = [`--- a/${c.file}`, `+++ b/${c.file}`, "@@ -1,20 +1,20 @@", ` ${ctx}`];
72
+ // Unchanged lines the change sits inside. An assertion whose keyword is on
73
+ // one of these is the shape the line-level denominator could not see.
74
+ for (const l of c.lead || []) lines.push(` ${l}`);
70
75
  for (const l of c.removed) lines.push(`-${l}`);
71
76
  for (const l of c.added) lines.push(`+${l}`);
72
77
  lines.push(` ${ctx}`);
@@ -105,6 +110,17 @@ const add = (name, ok, detail) => {
105
110
 
106
111
  {
107
112
  const missed = EMPTY_RUN_CANARIES.filter((c) => parseCollectedTests(c.output, "").count !== 0);
113
+ const undercounted = COUNTED_RUN_CANARIES.filter((c) => {
114
+ const n = parseCollectedTests(c.output, "").count;
115
+ return n === null || n < c.atLeast;
116
+ });
117
+ add(
118
+ "policy: a stated count is never read as empty",
119
+ undercounted.length === 0,
120
+ undercounted.length
121
+ ? undercounted.map((c) => `${c.id} (${c.why})`).join("; ")
122
+ : `${COUNTED_RUN_CANARIES.length} healthy runs counted, not rejected`
123
+ );
108
124
  add(
109
125
  "policy: empty-run detection",
110
126
  missed.length === 0,
@@ -118,7 +134,7 @@ const canaryResults = new Map();
118
134
  const silent = [];
119
135
  const noDenominator = [];
120
136
  const noAssertions = [];
121
- for (const c of TAMPER_CANARIES) {
137
+ for (const c of [...TAMPER_CANARIES, ...MULTILINE_CANARIES]) {
122
138
  const res = checkTestTampering(canaryDiff(c));
123
139
  const hit = (res.violations || []).some((v) => v.type === c.expect);
124
140
  canaryResults.set(c.id, hit);
@@ -132,7 +148,7 @@ const canaryResults = new Map();
132
148
  noAssertions.push(`${c.id} (${res.assertionsSeen} assertions parsed)`);
133
149
  }
134
150
  }
135
- add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length} canaries red as required`);
151
+ add("canaries: every tamper rule still fires", silent.length === 0, silent.length ? silent.join("; ") : `${TAMPER_CANARIES.length + MULTILINE_CANARIES.length} canaries red as required`);
136
152
  add("canaries: every finding carries a denominator", noDenominator.length === 0, noDenominator.length ? noDenominator.join(", ") : "inputsSeen > 0 on every hit");
137
153
  add("canaries: assertion rules parsed an assertion", noAssertions.length === 0, noAssertions.length ? noAssertions.join(", ") : "assertionsSeen > 0 on every assertion finding");
138
154
  }
@@ -174,7 +190,7 @@ const canaryResults = new Map();
174
190
  const survivors = [];
175
191
  for (const mutant of PREDICATE_MUTANTS) {
176
192
  let killed = false;
177
- for (const c of TAMPER_CANARIES) {
193
+ for (const c of [...TAMPER_CANARIES, ...MULTILINE_CANARIES]) {
178
194
  // Only canaries the healthy predicate catches can kill a mutant.
179
195
  if (!canaryResults.get(c.id)) continue;
180
196
  const res = checkTestTampering(canaryDiff(c), { isTestPath: mutant.fn });
package/src/engine.mjs CHANGED
@@ -204,6 +204,16 @@ export async function gate(opts = {}) {
204
204
  // Read from the base commit like every other trusted field: an
205
205
  // uncommitted `required: false` must not be able to switch the gate off.
206
206
  required: parsed.verify?.required !== undefined ? parsed.verify.required !== false : config.verify.required !== false,
207
+ // The floor the collection check applies. Omitting it here meant
208
+ // `verify.minTests` was silently dropped and always defaulted to 1
209
+ // — while the failure message told the operator to set exactly
210
+ // that. A remediation hint that does nothing is worse than none.
211
+ minTests:
212
+ parsed.verify?.minTests !== undefined
213
+ ? parsed.verify.minTests
214
+ : parsed.verify?.min_tests !== undefined
215
+ ? parsed.verify.min_tests
216
+ : config.verify.minTests,
207
217
  scope: parsed.verify?.scope || config.verify.scope || "global",
208
218
  timeoutMs: parsed.verify?.timeoutMs || parsed.verify?.timeout_ms || config.verify.timeoutMs,
209
219
  };
@@ -249,7 +259,45 @@ export async function gate(opts = {}) {
249
259
  violation.file = link;
250
260
  }
251
261
 
252
- phases.push({ phase: "scope", ok: scopeResult.ok, violations: scopeResult.violations });
262
+ // Bootstrap: the files that bring a repository under the gate are not agent
263
+ // edits to the gate.
264
+ //
265
+ // `init` writes `.agent/**` and then tells the user to commit it. Doing
266
+ // exactly that produced Exit 3 on the very first run, because the base
267
+ // branch does not have the commit yet and every scaffolded path matches
268
+ // BUILTIN_PROTECT or BUILTIN_DENY. The advice printed alongside it was
269
+ // `--allow-protected` — so a newcomer's first lesson was how to switch the
270
+ // scope guard off. A gate that refuses its own installation is not strict,
271
+ // it is broken.
272
+ //
273
+ // Narrow on purpose, and only where it cannot weaken anything: the base
274
+ // commit must have no gate config at all — in which case `trustedScope` is
275
+ // already built-ins only and nothing in the added files is trusted — and
276
+ // every violating path must be scaffold that the base does not have. A
277
+ // repository already under the gate keeps the full rule, so an agent still
278
+ // cannot touch the policy it is governed by.
279
+ let acceptedScaffold = [];
280
+ if (!scopeResult.ok && !trustedConfigRaw) {
281
+ const violations = scopeResult.violations || [];
282
+ const isScaffold = (f) => typeof f === "string" && f.replace(/\\/g, "/").startsWith(".agent/");
283
+ if (
284
+ violations.length > 0 &&
285
+ violations.every((v) => isScaffold(v.file) && showFromOrigin(root, base, v.file) === null)
286
+ ) {
287
+ acceptedScaffold = violations.map((v) => v.file);
288
+ scopeResult.violations = [];
289
+ scopeResult.ok = true;
290
+ }
291
+ }
292
+
293
+ phases.push({
294
+ phase: "scope",
295
+ ok: scopeResult.ok,
296
+ violations: scopeResult.violations,
297
+ // Reported, never silent: the operator has to see that the gate accepted
298
+ // files it would otherwise have blocked, and why.
299
+ ...(acceptedScaffold.length > 0 ? { setup: acceptedScaffold } : {}),
300
+ });
253
301
  appendTelemetry(root, "gate_phase", { phase: "scope", ok: scopeResult.ok });
254
302
  if (progressBus && progressToken) {
255
303
  progressBus.reportProgress(progressToken, 25, 100, "Phase 1/4: Scope Guard verification complete");
@@ -644,6 +692,13 @@ export async function gate(opts = {}) {
644
692
  postTestHash,
645
693
  tamperDetected: testTampered,
646
694
  },
695
+ // Verified by exit code alone. Not a failure, and not an approval either:
696
+ // an unrecognised runner passes on purpose, but the operator has to be
697
+ // able to tell that apart from a suite the gate actually counted.
698
+ ...(collectionFloor.unverified ? { unverified: collectionFloor.note } : {}),
699
+ ...(collectionFloor.count !== null && collectionFloor.count !== undefined
700
+ ? { testsCollected: collectionFloor.count, runner: collectionFloor.runner }
701
+ : {}),
647
702
  });
648
703
  phases.push({
649
704
  phase: "evidence",
@@ -382,6 +382,15 @@ export const INNOCENT_EDITS = [
382
382
  added: ['const calc = require("../src/calc");'],
383
383
  why: "`require(` is an import, not a claim — the loose net must not read it as one",
384
384
  },
385
+ {
386
+ id: "reword-comment-inside-assertion/node",
387
+ file: "test/calc.test.js",
388
+ context: "// arithmetic",
389
+ lead: [" assert.equal(", " add(1, 2),"],
390
+ removed: [" 3, // the answer", " );"],
391
+ added: [" 3, // the correct answer", " );"],
392
+ why: "line comments were never stripped — `copyCode(i)` left `pending` at the comment and the copy after the loop put it back",
393
+ },
385
394
  {
386
395
  id: "rename-helper/node",
387
396
  file: "test/calc.test.js",
@@ -479,3 +488,66 @@ IMPORT_EXTRACTION_CASES.push(
479
488
  why: "same, in the other place examples live",
480
489
  }
481
490
  );
491
+
492
+ /**
493
+ * Runs that stated a count, which must never be read as empty.
494
+ *
495
+ * The floor was written to be one-sided — only a *stated* zero fails — and
496
+ * then a phrase was allowed to outrank a statement. A healthy 190-test TAP
497
+ * suite whose one skipped fixture printed `# SKIP no tests found` was
498
+ * rejected as empty and attributed to Jest, in a repository that does not
499
+ * use Jest. A false red on a correct repository is how a user learns the
500
+ * gate is broken and turns it off.
501
+ */
502
+ export const COUNTED_RUN_CANARIES = [
503
+ {
504
+ id: "tap with a skip message",
505
+ output: "TAP version 13\n# Subtest: performance\n # SKIP no tests found\nok 1 - performance # SKIP\n1..191\n# tests 191\n# pass 190\n# skip 1",
506
+ atLeast: 1,
507
+ why: "`no tests found` inside a skip comment is not a statement about the run",
508
+ },
509
+ {
510
+ id: "pytest mentioning an empty module",
511
+ output: "collected 12 items\n\ntests/test_a.py ............\n\n12 passed in 0.3s",
512
+ atLeast: 1,
513
+ why: "a stated count is present and must win",
514
+ },
515
+ {
516
+ id: "go, verbose, two tests",
517
+ output: "--- PASS: TestAdd (0.00s)\n--- PASS: TestSub (0.00s)\nok \texample.com/lib\t0.004s",
518
+ atLeast: 1,
519
+ why: "Go's own `ok <package>` line, which a bare `^ok\\s` confused with TAP's `ok 1 - name`",
520
+ },
521
+ ];
522
+
523
+ /**
524
+ * Changes that sit inside an assertion whose keyword never appears in the diff.
525
+ *
526
+ * Counting `+`/`-` lines answered zero here, and nothing among the changed
527
+ * lines looked assertion-shaped either, so the guard reported a clean PASS on
528
+ * a five-element expected list rewritten to one element to match broken
529
+ * output. Measured on a real repository: five green phases, `APPROVED`.
530
+ *
531
+ * The statement machinery that pairs rewrites already assembles context lines
532
+ * together with changed ones. The denominator has to use it.
533
+ */
534
+ export const MULTILINE_CANARIES = [
535
+ {
536
+ id: "expectation/python-multiline",
537
+ file: "tests/test_headers.py",
538
+ context: "# headers",
539
+ lead: [" def test_parse(self):", " self.assertEqual(", " _parse_http_header(x),"],
540
+ removed: [' [("text/xml", {}),', ' ("text/plain", {}),', ' ("more", {})])'],
541
+ added: [' [("text/xml", {"tampered": True})])'],
542
+ expect: "ASSERTION_EXPECTATION_CHANGED",
543
+ },
544
+ {
545
+ id: "expectation/js-multiline",
546
+ file: "test/headers.test.js",
547
+ context: "// headers",
548
+ lead: [" assert.deepStrictEqual(", " parse(input),"],
549
+ removed: [" [{ type: \"a\" }, { type: \"b\" }, { type: \"c\" }]", " );"],
550
+ added: [" [{ type: \"a\" }]", " );"],
551
+ expect: "ASSERTION_EXPECTATION_CHANGED",
552
+ },
553
+ ];
package/src/memory.mjs CHANGED
Binary file
@@ -71,8 +71,15 @@ const EXPLICIT_ZERO = [
71
71
 
72
72
  /** Go prints this per package that has no test files at all. */
73
73
  const GO_NO_TEST_FILES = /\[no test files\]/;
74
- /** Any sign that a Go package did run tests. */
75
- const GO_RAN_SOMETHING = /^(?:ok|FAIL|---\s+(?:PASS|FAIL|SKIP)):?\s/m;
74
+ /**
75
+ * Any sign that a Go package did run tests.
76
+ *
77
+ * The negative lookahead is what separates Go from TAP. `ok 1 - performance`
78
+ * is a TAP result line and `ok example.com/lib 0.004s` is a Go package
79
+ * summary, and a bare `^ok\s` matched both — so a 190-test TAP suite was
80
+ * classified as Go and reported as having stated no count at all.
81
+ */
82
+ const GO_RAN_SOMETHING = /^(?:(?:ok|FAIL)\s+(?!\d+\s)\S+|---\s+(?:PASS|FAIL|SKIP):?\s)/m;
76
83
 
77
84
  /**
78
85
  * Read a collected-test count out of a runner's output.
@@ -87,8 +94,20 @@ export function parseCollectedTests(stdout = "", stderr = "") {
87
94
  const text = `${stdout || ""}\n${stderr || ""}`;
88
95
  if (!text.trim()) return { count: null, runner: null };
89
96
 
90
- for (const rule of EXPLICIT_ZERO) {
91
- if (rule.re.test(text)) return { count: 0, runner: rule.name };
97
+ // A stated count wins over a phrase that merely resembles one.
98
+ //
99
+ // `EXPLICIT_ZERO` used to be consulted first, so any output containing the
100
+ // words "no tests found" was read as a zero — including a healthy TAP run
101
+ // of 190 passing tests whose one skipped fixture printed
102
+ // `# SKIP no tests found`. The gate rejected the suite as empty and named
103
+ // Jest as the runner, in a repository that does not use Jest. A phrase
104
+ // appears anywhere in a stream; a count is stated deliberately, so the
105
+ // count is the better witness and has to be asked first.
106
+ for (const rule of COUNT_PATTERNS) {
107
+ const m = rule.re.exec(text);
108
+ if (!m) continue;
109
+ const n = Number(m[1]);
110
+ if (Number.isFinite(n)) return { count: n, runner: rule.name };
92
111
  }
93
112
 
94
113
  // Go states absence per package rather than as a count, so it needs its own
@@ -104,11 +123,9 @@ export function parseCollectedTests(stdout = "", stderr = "") {
104
123
  return { count: perTest && perTest.length > 0 ? perTest.length : null, runner: "go" };
105
124
  }
106
125
 
107
- for (const rule of COUNT_PATTERNS) {
108
- const m = rule.re.exec(text);
109
- if (!m) continue;
110
- const n = Number(m[1]);
111
- if (Number.isFinite(n)) return { count: n, runner: rule.name };
126
+ // Only now: no runner stated a number, so a declared absence is all there is.
127
+ for (const rule of EXPLICIT_ZERO) {
128
+ if (rule.re.test(text)) return { count: 0, runner: rule.name };
112
129
  }
113
130
 
114
131
  return { count: null, runner: null };
@@ -131,7 +148,24 @@ export function checkCollectionFloor(testResult, opts = {}) {
131
148
  }
132
149
 
133
150
  const { count, runner } = parseCollectedTests(testResult.stdout, testResult.stderr);
134
- if (count === null || count >= minTests) {
151
+
152
+ // Deliberately one-sided: only a *stated* zero fails, because failing on
153
+ // "I could not tell" would break every runner not on the list. But passing
154
+ // and saying nothing makes `echo "all tests passed"` indistinguishable from
155
+ // a real suite, so the absence of evidence is reported as an absence.
156
+ if (count === null) {
157
+ return {
158
+ ok: true,
159
+ count: null,
160
+ runner: null,
161
+ reason: null,
162
+ unverified: true,
163
+ note:
164
+ `The verification command exited 0, but no recognised test runner stated how many tests it ran, ` +
165
+ `so the gate cannot tell a full suite from a command that ran nothing. Verified by exit code alone.`,
166
+ };
167
+ }
168
+ if (count >= minTests) {
135
169
  return { ok: true, count, runner, reason: null };
136
170
  }
137
171
 
package/src/security.mjs CHANGED
@@ -1523,8 +1523,13 @@ function stripComments(text, lang) {
1523
1523
  continue;
1524
1524
  }
1525
1525
 
1526
- if (c === "/" && c2 === "/") { copyCode(i); break; }
1527
- if (lang === "python" && c === "#") { copyCode(i); break; }
1526
+ // `pending = n` is the whole fix. `copyCode(i)` copies the code up to
1527
+ // the comment and leaves `pending` sitting at its start; the
1528
+ // `copyCode(n)` after this loop then copied the comment straight back
1529
+ // in, so no line comment has ever been stripped. Block comments were,
1530
+ // which is why `/* … */` behaved and `// …` did not.
1531
+ if (c === "/" && c2 === "/") { copyCode(i); pending = n; break; }
1532
+ if (lang === "python" && c === "#") { copyCode(i); pending = n; break; }
1528
1533
  if (c === "/" && c2 === "*") {
1529
1534
  copyCode(i);
1530
1535
  state.block = 1;
@@ -1929,6 +1934,9 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1929
1934
  }
1930
1935
 
1931
1936
  const pairs = [];
1937
+ const pairedOld = new Set();
1938
+ const pairedNew = new Set();
1939
+ const cancelled = new Set();
1932
1940
  for (const [shape, olds] of oldByShape) {
1933
1941
  const news = newByShape.get(shape) || [];
1934
1942
 
@@ -1946,20 +1954,69 @@ function detectExpectationRewrites(file, hunks, stats, violations) {
1946
1954
  const survivingOld = [];
1947
1955
  for (const o of olds) {
1948
1956
  const twin = survivingNew.findIndex((n) => n.canon === o.canon);
1949
- if (twin === -1) survivingOld.push(o);
1950
- else survivingNew.splice(twin, 1);
1957
+ if (twin === -1) {
1958
+ survivingOld.push(o);
1959
+ } else {
1960
+ // Present unchanged on both sides: this assertion did not change,
1961
+ // and the argument-level pass below must not be allowed to pair it
1962
+ // with something else and call that a rewrite.
1963
+ cancelled.add(o);
1964
+ cancelled.add(survivingNew[twin]);
1965
+ survivingNew.splice(twin, 1);
1966
+ }
1951
1967
  }
1952
1968
 
1953
1969
  const k = Math.min(survivingOld.length, survivingNew.length);
1954
1970
  for (let t = 0; t < k; t++) {
1955
1971
  const r = survivingOld[t];
1956
1972
  const a = survivingNew[t];
1973
+ // Whatever this pass decides about a pair — reported, or deliberately
1974
+ // let go as a reorder or a reworded message — is the decision. The
1975
+ // argument-level pass below exists only for statements whose shapes
1976
+ // differ so much that they never met in a bucket here; letting it
1977
+ // re-open a case that was already judged turned a comment edit into a
1978
+ // rewritten expectation.
1979
+ pairedOld.add(r);
1980
+ pairedNew.add(a);
1957
1981
  if (r.canon === a.canon) continue;
1958
1982
  if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
1959
1983
  pairs.push({ r: r.s, a: a.s });
1960
1984
  }
1961
1985
  }
1962
1986
 
1987
+ // Same assertion, same subject, different expected value.
1988
+ //
1989
+ // Shape pairing compares the statement with its literals blanked, so it
1990
+ // only ever matched assertions whose structure survived the edit. Shrink
1991
+ // a five-element expected list to one element to match broken output and
1992
+ // the two images land in different shape buckets, never pair, and the
1993
+ // rewrite is not reported at all — measured on a real repository, where
1994
+ // it collected five green phases.
1995
+ //
1996
+ // The arguments are the better witness here: when both sides call the
1997
+ // same assertion with the same number of arguments and the *subject*
1998
+ // argument is untouched, what changed is what the test expects of it.
1999
+ for (const r of oldCands) {
2000
+ if (pairedOld.has(r) || cancelled.has(r)) continue;
2001
+ const ra = splitAssertionArgs(r.clean, lang);
2002
+ if (!ra || ra.length < 2) continue;
2003
+ for (const a of newCands) {
2004
+ if (pairedNew.has(a) || cancelled.has(a)) continue;
2005
+ const aa = splitAssertionArgs(a.clean, lang);
2006
+ if (!aa || aa.length !== ra.length) continue;
2007
+ const same = (i) => ra[i].replace(/\s+/g, "") === aa[i].replace(/\s+/g, "");
2008
+ // The subject has to be the same expression, or these are two
2009
+ // different assertions that merely resemble each other.
2010
+ if (!same(0)) continue;
2011
+ if (ra.every((_, i) => same(i))) continue;
2012
+ if (differsOnlyInMessage(r.clean, a.clean, lang)) continue;
2013
+ pairs.push({ r: r.s, a: a.s });
2014
+ pairedOld.add(r);
2015
+ pairedNew.add(a);
2016
+ break;
2017
+ }
2018
+ }
2019
+
1963
2020
  // Zero-context hunk: each image is a single fragment and the assertion
1964
2021
  // keyword may sit outside the hunk entirely. The fragment pair is taken
1965
2022
  // only when both sides normalize to the same shape *and* that shape
@@ -2388,21 +2445,57 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2388
2445
  // understood. `UNREADABLE` is the state that has no business being silent
2389
2446
  // — assertion-shaped lines were present and none of them parsed, which
2390
2447
  // means this repository speaks a dialect the guard does not.
2448
+ // An assertion is a statement, not a line.
2449
+ //
2450
+ // Counting `+`/`-` lines missed the commonest shape in every language with
2451
+ // multi-line calls: `self.assertEqual(` sits on an unchanged context line
2452
+ // and only its argument lines are edited. Nothing among the changed lines
2453
+ // matched an assertion pattern, nothing looked assertion-shaped either, so
2454
+ // the guard reported `assertionsSeen: 0` and — because no line looked
2455
+ // suspicious — a clean PASS. A five-element expected list rewritten to one
2456
+ // element to match broken output sailed through five green phases.
2457
+ //
2458
+ // The statement machinery that already exists for pairing knows better:
2459
+ // it assembles context lines together with changed ones. Ask it.
2460
+ for (const [file, stats] of fileAssertions.entries()) {
2461
+ const lang = langForTestFile(file);
2462
+ let touched = 0;
2463
+ for (const hunk of stats.hunks) {
2464
+ const oldSlice = [];
2465
+ const newSlice = [];
2466
+ for (const L of hunk.lines) {
2467
+ if (L.kind !== "+") oldSlice.push(L);
2468
+ if (L.kind !== "-") newSlice.push(L);
2469
+ }
2470
+ for (const stmts of [assembleStatements(oldSlice, lang), assembleStatements(newSlice, lang)]) {
2471
+ for (const st of stmts) {
2472
+ const changed = (st.removedLines?.length || 0) + (st.addedLines?.length || 0) > 0;
2473
+ if (changed && ASSERTION_PATTERN.test(stripComments(st.text, lang))) touched++;
2474
+ }
2475
+ }
2476
+ }
2477
+ stats.statementAssertions = touched;
2478
+ }
2479
+
2391
2480
  let examined = 0;
2392
2481
  let assertionsSeen = 0;
2393
2482
  const unreadable = [];
2394
2483
  for (const [file, stats] of fileAssertions.entries()) {
2395
2484
  examined += stats.examined;
2396
- assertionsSeen += stats.recognised;
2485
+ assertionsSeen += Math.max(stats.recognised, stats.statementAssertions || 0);
2397
2486
  if (stats.unreadable.length > 0) {
2398
2487
  unreadable.push({ file, count: stats.unreadable.length, samples: stats.unreadable.slice(0, 3) });
2399
2488
  }
2400
2489
  }
2401
2490
 
2491
+ // Changed lines inside a test file, none of them recognisable as part of an
2492
+ // assertion, is not the same as "checked and clean" — it is the state where
2493
+ // this guard has nothing to say. Saying nothing and saying "approved" have
2494
+ // to look different, which is the whole reason `status` exists.
2402
2495
  const status =
2403
2496
  reported.length > 0
2404
2497
  ? "FAIL"
2405
- : assertionsSeen === 0 && unreadable.length > 0
2498
+ : assertionsSeen === 0 && (unreadable.length > 0 || examined > 0)
2406
2499
  ? "UNREADABLE"
2407
2500
  : examined > 0
2408
2501
  ? "PASS"
@@ -2480,6 +2573,26 @@ export function scanDiff(diffTextStr = "", options = {}) {
2480
2573
  }
2481
2574
  }
2482
2575
 
2576
+ // The gate calls `scanDiff`, not `assertTestIntegrity` — so wiring the
2577
+ // dialect warning into the latter meant it reached nobody. The guard
2578
+ // computed `UNREADABLE`, and the operator was shown an unblemished pass.
2579
+ // A boundary that is not reported is not a boundary.
2580
+ if (tamperingRes.status === "UNREADABLE") {
2581
+ const where = (tamperingRes.unreadable || []).map((u) => u.file);
2582
+ const sample = tamperingRes.unreadable?.[0]?.samples?.[0];
2583
+ findings.push({
2584
+ severity: "MEDIUM",
2585
+ type: "TEST_DIALECT_UNREADABLE",
2586
+ file: where[0] ?? null,
2587
+ line: null,
2588
+ description:
2589
+ `Test Tamper Guard: changed ${tamperingRes.inputsSeen} line(s) in ${tamperingRes.filesSeen} test file(s) ` +
2590
+ `and recognised no assertion among them${sample ? ` (e.g. ${JSON.stringify(sample)})` : ""}. ` +
2591
+ `This change was NOT checked for tampering. That is not a failure — an unlisted assertion library is ` +
2592
+ `normal — but it is not an approval either, so it is reported rather than passed silently.`,
2593
+ });
2594
+ }
2595
+
2483
2596
  return {
2484
2597
  ok: secretsOk && edgeRes.ok && crossPkgRes.ok && tamperingRes.ok,
2485
2598
  findings,
@@ -188,6 +188,56 @@ export function detectEdgeRuntime(projectRoot = process.cwd()) {
188
188
  /**
189
189
  * Detects 24+ polyglot stacks and container environments.
190
190
  */
191
+ /**
192
+ * Test commands worth trying, best first, when the detected one does not run.
193
+ *
194
+ * `init` probes the command it picked. On a repository whose Makefile
195
+ * declares a `test` target that needs a build environment the machine does
196
+ * not have, that probe failed, printed `Oracle verification probe failed`,
197
+ * and the wizard wrote the broken command into the config anyway — in a
198
+ * repository where `pytest` was on PATH and all 360 tests passed in 1.3s.
199
+ * Measuring something and then ignoring the measurement is worse than not
200
+ * measuring: it produces a hard red on day one, which is how a user learns
201
+ * the gate is broken and turns it off.
202
+ *
203
+ * Kept deliberately generic — a per-ecosystem convention, never a per-project
204
+ * or per-provider guess.
205
+ *
206
+ * @param {string} root
207
+ * @param {string} [detected] - the command detection chose; always first.
208
+ * @returns {string[]} ordered, de-duplicated candidates
209
+ */
210
+ export function oracleCandidates(root = process.cwd(), detected = "") {
211
+ const out = [];
212
+ const push = (c) => {
213
+ const v = (c || "").trim();
214
+ if (v && !out.includes(v) && !isPlaceholderTestScript(v)) out.push(v);
215
+ };
216
+ const has = (f) => existsSync(join(root, f));
217
+
218
+ push(detected);
219
+
220
+ if (has("package.json")) {
221
+ try {
222
+ const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
223
+ if (pkg.scripts?.test && !isPlaceholderTestScript(pkg.scripts.test)) push("npm test");
224
+ } catch (_) {}
225
+ }
226
+ if (has("pytest.ini") || has("pyproject.toml") || has("setup.py") || has("tox.ini") || has("setup.cfg")) {
227
+ push(pytestCmd());
228
+ }
229
+ if (has("Cargo.toml")) push("cargo test");
230
+ if (has("go.mod")) push("go test ./...");
231
+ if (has("Gemfile")) push("bundle exec rspec");
232
+ if (has("composer.json")) push("./vendor/bin/phpunit");
233
+ if (has("pom.xml")) push("mvn -q test");
234
+ if (has("build.gradle") || has("build.gradle.kts")) push("./gradlew test");
235
+ if (has("pubspec.yaml")) push("dart test");
236
+ if (has("Package.swift")) push("swift test");
237
+
238
+ return out;
239
+ }
240
+
191
241
  export function detectPolyglotStack(projectRoot = process.cwd()) {
192
242
  const edgeInfo = detectEdgeRuntime(projectRoot);
193
243
  const isDevcontainer = existsSync(join(projectRoot, ".devcontainer", "devcontainer.json"));
@@ -3,7 +3,7 @@ import { join } from "node:path";
3
3
  import { parseYaml, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
4
4
  import { suggestProvider, detectAvailableProviders } from "./provider-readiness.mjs";
5
5
  import { detectDefaultBranch } from "./git.mjs";
6
- import { resolveWorkspaceBoundary } from "./stack-detector.mjs";
6
+ import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
7
7
  import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
8
8
  import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
9
9
  import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
@@ -302,6 +302,51 @@ export function loadPresets(root = process.cwd()) {
302
302
  * @param {object} [options]
303
303
  * @returns {Promise<{ ok: boolean, configPath: string, plan: object }>}
304
304
  */
305
+ /**
306
+ * Probe the chosen test command, and take detection's next choice if it fails.
307
+ *
308
+ * Runs on the non-interactive path too. `--yes` means "do not ask me", not
309
+ * "do not check" — and the user who is not watching is exactly the one who
310
+ * cannot notice that the command written into their config does not run.
311
+ * Before this, the probe lived inside the interactive branch, so
312
+ * `agentctl init --yes` wrote `make test` into a repository where `make test`
313
+ * exits 2 and `npm test` passes, and the first gate run was a hard red.
314
+ *
315
+ * @returns {Promise<string>} the command to save
316
+ */
317
+ async function resolveRunnableOracle(root, testCmd, options = {}) {
318
+ if (!testCmd) return testCmd;
319
+ const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
320
+ const probeRes = await runVerificationProbe(testCmd, root);
321
+ if (probeRes.ok) {
322
+ probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
323
+ return testCmd;
324
+ }
325
+ probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
326
+
327
+ const alternates = oracleCandidates(root, testCmd).filter((c) => c !== testCmd).slice(0, 3);
328
+ for (const cand of alternates) {
329
+ const altSp = spinner(`Trying ${cand}`, options);
330
+ const altRes = await runVerificationProbe(cand, root);
331
+ if (altRes.ok) {
332
+ altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
333
+ return cand;
334
+ }
335
+ altSp.fail(`${cand} also failed (Exit ${altRes.code})`);
336
+ }
337
+
338
+ // Nothing runs. Say so in terms the user can act on, rather than leaving a
339
+ // failed spinner to scroll past and a broken command in the config.
340
+ const out = options.stdout || process.stdout;
341
+ out.write("\n");
342
+ out.write(" \u26a0\ufe0f No test command could be run in this environment.\n");
343
+ out.write(` Keeping "${testCmd}" \u2014 the gate will fail until it runs here.\n`);
344
+ out.write(" Point verify.test in .agent/config.yml at a command that works,\n");
345
+ out.write(" or, if this repository genuinely has no suite, set\n");
346
+ out.write(" verify.required: false deliberately rather than by accident.\n\n");
347
+ return testCmd;
348
+ }
349
+
305
350
  export async function runInitWizard(root = process.cwd(), options = {}) {
306
351
  const interactive = options.interactive !== false && isTTY(options.stdin || process.stdin);
307
352
 
@@ -322,6 +367,7 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
322
367
  let selectedProvider = options.provider || existingConfig.provider;
323
368
  let selectedProfile = options.profile || existingConfig.verify?.profile;
324
369
  let testCmd = options.testCmd;
370
+ let probeInteractive = null;
325
371
  let buildCmd = options.buildCmd;
326
372
  let selectedPresets = options.presets;
327
373
 
@@ -402,16 +448,20 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
402
448
 
403
449
  selectedPresets = await multiSelect(presetOptions, "Select Autonomous Workflows to Enable", options);
404
450
 
405
- const shouldProbe = await confirm("Run verification probe on test command before saving?", true, options);
406
- if (shouldProbe && testCmd) {
407
- const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
408
- const probeRes = await runVerificationProbe(testCmd, root);
409
- if (probeRes.ok) {
410
- probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
411
- } else {
412
- probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
413
- }
414
- }
451
+ probeInteractive = await confirm("Run verification probe on test command before saving?", true, options);
452
+ }
453
+
454
+ // The probe runs whether or not anyone was asked: interactive users can
455
+ // decline it, but silence from `--yes` is not a decline.
456
+ if (probeInteractive !== false && options.probe !== false) {
457
+ // Resolve the command the way planInit will, or there is nothing to
458
+ // probe: on the headless path `testCmd` stays undefined until planInit
459
+ // fills it in from detection, so the probe silently examined nothing —
460
+ // the exact fail-open shape this project keeps finding in itself.
461
+ const effective =
462
+ testCmd || existingConfig.verify?.test || detectStackOracles(root)?.candidates?.testCmd || "";
463
+ const adopted = await resolveRunnableOracle(root, effective, options);
464
+ if (adopted) testCmd = adopted;
415
465
  }
416
466
 
417
467
  // `...options` first, for the same reason as in wizard-task.mjs: spreading it