jules-orchestrator-kit 0.63.0 → 0.65.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/security.mjs CHANGED
@@ -1155,14 +1155,42 @@ function locateFindingLine(lines, type, file = null) {
1155
1155
  // (`expect(\n formatInvoice(bill)\n).toBe(…`) still recognises the chain.
1156
1156
  // The bound is a guess: an argument list longer than 240 characters is
1157
1157
  // rarer than a missed chain.
1158
+ // The dialect list is not decoration. `assertEqual` was recognised only
1159
+ // because `\.?` made the dot optional and the `i` flag let `Equal` match
1160
+ // `equal`; `assertEquals`, one letter longer, fell out of the pattern and
1161
+ // took JUnit, PHPUnit, Minitest, RSpec and XCTest with it. The weak forms
1162
+ // — assertTrue, assertNotNull, XCTAssertTrue — are deliberately absent:
1163
+ // they state no expected value, so their arrival in place of one of these
1164
+ // is a weakening, which is a finding of its own.
1158
1165
  const SPECIFIC_ASSERTION = new RegExp(
1159
1166
  [
1160
1167
  "\\bassert(?:\\.strict)?\\.?(?:strictEqual|deepStrictEqual|deepEqual|notStrictEqual|notDeepStrictEqual|equal|notEqual|match|doesNotMatch|throws|rejects|doesNotThrow)\\s*\\(",
1161
1168
  "\\bexpect\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:toBe|toEqual|toStrictEqual|toMatch|toMatchObject|toContain|toHaveBeenCalledWith|toThrow|toHaveLength|toBeCloseTo)\\s*\\(",
1162
1169
  "\\bassert\\.(?:equals|deepEquals|include|lengthOf)\\s*\\(",
1163
1170
  "assert_eq!|assert_ne!",
1171
+ // The bare comparison form. `assert add(1, 2) == 3` is how pytest is
1172
+ // actually written, and Rust's `assert!(a == b)` and Elixir's
1173
+ // `assert f(x) == 3` follow it; none of them name a comparison
1174
+ // function, so a list of function names could never reach them.
1175
+ // Equality only. `assert!(x != 0)` names no expected value — it is the
1176
+ // weaker claim you arrive at by giving one up, and counting it as
1177
+ // specific would make the downgrade from `assert_eq!(x, 5)` invisible to
1178
+ // the weakening check.
1179
+ "\\bassert\\s+[^\\n]*(?:===|==)(?!=)",
1180
+ "\\bassert!\\s*\\([^\\n]*==(?!=)",
1164
1181
  "\\bt\\.(?:Errorf|Fatalf)\\s*\\(",
1165
1182
  "\\brequire\\.(?:Equal|NotEqual|Len|Contains|Error|NoError)\\s*\\(",
1183
+ // Python unittest, stated rather than inherited from the optional dot.
1184
+ "\\bassert(?:Equal|NotEqual|AlmostEqual|NotAlmostEqual|Regex|NotRegex|Raises|In|NotIn|Is|IsNot|ListEqual|DictEqual|SetEqual|TupleEqual|CountEqual|Greater|Less|GreaterEqual|LessEqual)\\s*\\(",
1185
+ // JUnit / TestNG / PHPUnit
1186
+ "\\bassert(?:Equals|NotEquals|Same|NotSame|ArrayEquals|IterableEquals|LinesMatch|Count|StringContainsString|StringEqualsFile|InstanceOf|Contains|Throws)\\s*\\(",
1187
+ "\\bassertThat\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:isEqualTo|isSameAs|contains|containsExactly|hasSize|isCloseTo|matches)\\s*\\(",
1188
+ // Minitest
1189
+ "\\b(?:assert|refute)_(?:equal|includes|match|nil|same|in_delta|in_epsilon|raises|empty|operator|predicate)\\b",
1190
+ // RSpec
1191
+ "\\bexpect\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:to|not_to|to_not)\\s+(?:eq|eql|equal|be|be_within|match|include|contain_exactly|match_array|have_attributes|raise_error|start_with|end_with)\\b",
1192
+ // XCTest
1193
+ "\\bXCTAssert(?:Equal|NotEqual|EqualWithAccuracy|Identical|NotIdentical|GreaterThan|LessThan|GreaterThanOrEqual|LessThanOrEqual|ThrowsError|NoThrow)\\s*\\(",
1166
1194
  ].join("|"),
1167
1195
  "i"
1168
1196
  );
@@ -1203,6 +1231,15 @@ const TEST_LANG_BY_EXT = new Map([
1203
1231
  [".py", "python"], [".pyi", "python"],
1204
1232
  [".go", "go"],
1205
1233
  [".rs", "rust"],
1234
+ // Approximations, chosen for comment and continuation syntax rather than
1235
+ // for kinship: the C-like family reads correctly under the `js` scanner,
1236
+ // and Ruby under the `python` one because both end a comment at `#` and a
1237
+ // statement at the newline. Naming them beats falling through to `js` by
1238
+ // default, which is how a `#` comment came to be read as code.
1239
+ [".java", "js"], [".kt", "js"], [".kts", "js"], [".scala", "js"], [".groovy", "js"],
1240
+ [".swift", "js"], [".cs", "js"], [".php", "js"], [".c", "js"], [".cc", "js"],
1241
+ [".cpp", "js"], [".h", "js"], [".hpp", "js"], [".m", "js"], [".sol", "js"],
1242
+ [".rb", "python"],
1206
1243
  ]);
1207
1244
 
1208
1245
  function langForTestFile(file) {
@@ -1540,6 +1577,21 @@ function stripComments(text, lang) {
1540
1577
  const CONTINUATION_START = /^[)\],.]/;
1541
1578
  const CONTINUATION_OP_START = /^[+\-*/%<>=&|^:]/;
1542
1579
 
1580
+ // A comment is not a continuation, however much it looks like one.
1581
+ //
1582
+ // `//` begins with a division sign and `--` with a minus, so both matched
1583
+ // CONTINUATION_OP_START and folded the following comment line into the
1584
+ // statement above it. The cost was a false accusation on a virtuous act:
1585
+ // adding an assertion next to a `// ...` line made the new assertion absorb
1586
+ // the comment, stop matching its unchanged twin, and get reported as a
1587
+ // rewritten expectation. Python was unaffected only because `#` is not an
1588
+ // operator — which is why the same fixture passed in pytest and failed in
1589
+ // Jest, and why it survived every suite written against the pytest layout.
1590
+ //
1591
+ // A comment inside an open delimiter still joins: `cur.delta > 0` decides
1592
+ // that before this test is ever reached.
1593
+ const COMMENT_LINE_START = /^(?:\/\/|\/\*|#|--)/;
1594
+
1543
1595
  // A scanner miscount (an unbalanced delimiter inside a regex literal is the
1544
1596
  // usual cause) must not be able to merge a whole file into one statement,
1545
1597
  // which would pair *any* literal change anywhere in the file.
@@ -1578,14 +1630,15 @@ function assembleStatements(sliceLines, lang) {
1578
1630
 
1579
1631
  for (const L of sliceLines) {
1580
1632
  const trimmed = L.text.replace(/^\s+/, "");
1633
+ const startsComment = COMMENT_LINE_START.test(trimmed);
1581
1634
  const joins =
1582
1635
  cur !== null &&
1583
1636
  (cur.delta > 0 ||
1584
1637
  cur.state.str !== null ||
1585
1638
  cur.state.block > 0 ||
1586
1639
  cur.trailingBackslash ||
1587
- CONTINUATION_START.test(trimmed) ||
1588
- CONTINUATION_OP_START.test(trimmed));
1640
+ (!startsComment &&
1641
+ (CONTINUATION_START.test(trimmed) || CONTINUATION_OP_START.test(trimmed))));
1589
1642
 
1590
1643
  if (
1591
1644
  joins &&
@@ -1648,7 +1701,17 @@ function splitAssertionArgs(clean, lang) {
1648
1701
  const m = SPECIFIC_ASSERTION.exec(clean);
1649
1702
  if (!m) return null;
1650
1703
 
1651
- let i = m.index + m[0].length; // just past the opening paren
1704
+ // Not every branch of SPECIFIC_ASSERTION ends at an opening paren:
1705
+ // `assert_eq!`, `assert_equal` and RSpec's `.to eq` all match a bare name.
1706
+ // Starting the walk one character early made every argument boundary wrong,
1707
+ // so a reworded message read as a rewritten value.
1708
+ let i = m.index + m[0].length;
1709
+ if (clean[i - 1] !== "(") {
1710
+ let j = i;
1711
+ while (j < clean.length && /\s/.test(clean[j])) j++;
1712
+ if (clean[j] !== "(") return null;
1713
+ i = j + 1;
1714
+ }
1652
1715
  let depth = 1;
1653
1716
  let quote = null;
1654
1717
  let triple = false;
@@ -1740,7 +1803,45 @@ function messageArgIndices(args) {
1740
1803
  * every time somebody improved the wording of a failure. Firing on that is
1741
1804
  * how an operator learns to pass the override without reading it.
1742
1805
  */
1806
+ /**
1807
+ * Split a statement at a trailing `, "message"` written outside the call.
1808
+ *
1809
+ * RSpec puts the message there — `expect(x).to eq(3), "explain"` — and so do
1810
+ * Ruby and Elixir assertions generally. An argument-position check can never
1811
+ * see it, so rewording one read as a rewritten expectation.
1812
+ */
1813
+ function splitTrailingMessage(clean) {
1814
+ let depth = 0;
1815
+ let quote = null;
1816
+ let lastComma = -1;
1817
+ for (let i = 0; i < clean.length; i++) {
1818
+ const c = clean[i];
1819
+ if (quote !== null) {
1820
+ if (c === "\\") { i += 1; continue; }
1821
+ if (c === quote) quote = null;
1822
+ continue;
1823
+ }
1824
+ if (c === '"' || c === "'" || c === "`") { quote = c; continue; }
1825
+ if (c === "(" || c === "[" || c === "{") depth++;
1826
+ else if (c === ")" || c === "]" || c === "}") depth--;
1827
+ else if (c === "," && depth === 0) lastComma = i;
1828
+ }
1829
+ if (lastComma === -1) return { head: clean, msg: null };
1830
+ const tail = clean.slice(lastComma + 1).trim();
1831
+ if (!isPureStringLiteral(tail)) return { head: clean, msg: null };
1832
+ return { head: clean.slice(0, lastComma), msg: tail };
1833
+ }
1834
+
1743
1835
  function differsOnlyInMessage(cleanRemoved, cleanAdded, lang) {
1836
+ const ta = splitTrailingMessage(cleanRemoved);
1837
+ const tb = splitTrailingMessage(cleanAdded);
1838
+ if (
1839
+ (ta.msg !== null || tb.msg !== null) &&
1840
+ ta.head.replace(/\s+/g, "") === tb.head.replace(/\s+/g, "")
1841
+ ) {
1842
+ return true;
1843
+ }
1844
+
1744
1845
  const a = splitAssertionArgs(cleanRemoved, lang);
1745
1846
  const b = splitAssertionArgs(cleanAdded, lang);
1746
1847
  if (!a || !b || a.length !== b.length || a.length === 0) return false;
@@ -1988,7 +2089,9 @@ export function resolveAllowedTamperKinds(options = {}) {
1988
2089
  * @param {string} diffOrText - Unified git diff
1989
2090
  * @param {Object} [options]
1990
2091
  * @param {boolean} [options.allowTestModifications=false]
1991
- * @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, status: "PASS"|"FAIL"|"NOT_APPLICABLE" }}
2092
+ * @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, filesSeen: number,
2093
+ * assertionsSeen: number, unreadable: Array<{file: string, count: number, samples: string[]}>,
2094
+ * status: "PASS"|"FAIL"|"UNREADABLE"|"NOT_APPLICABLE" }}
1992
2095
  * `status` distinguishes "checked and clean" from "nothing was checked";
1993
2096
  * `ok: true` alone cannot, and that ambiguity is the defect class this
1994
2097
  * field exists to make visible.
@@ -2024,9 +2127,24 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2024
2127
  { pattern: /#\[ignore\]/i, desc: "Injected Rust test ignore attribute (#[ignore])" },
2025
2128
  { pattern: /\/\/\s*@ts-ignore/i, desc: "Injected TypeScript ignore comment (// @ts-ignore)" },
2026
2129
  { pattern: /\/\/\s*@ts-nocheck/i, desc: "Injected TypeScript nocheck directive (// @ts-nocheck)" },
2130
+ // A suite that never runs cannot fail, and `@Disabled` is exactly as
2131
+ // effective at arranging that as `it.skip` is.
2132
+ { pattern: /@Disabled\b/, desc: "Injected JUnit 5 disable annotation (@Disabled)" },
2133
+ { pattern: /@Ignore\b/, desc: "Injected JUnit 4 / TestNG ignore annotation (@Ignore)" },
2134
+ { pattern: /@Test\s*\([^)]*enabled\s*=\s*false/i, desc: "Injected TestNG disabled test (enabled = false)" },
2135
+ { pattern: /@unittest\.skip/i, desc: "Injected unittest skip decorator (@unittest.skip)" },
2136
+ { pattern: /\bmarkTest(?:Skipped|Incomplete)\s*\(/i, desc: "Injected PHPUnit skip (markTestSkipped())" },
2137
+ { pattern: /\bXCTSkip(?:If|Unless|IfNot)?\s*\(/, desc: "Injected XCTest skip (XCTSkip())" },
2138
+ { pattern: /\b(?:xit|xdescribe|xcontext|xspecify)\b\s*["\x27]/i, desc: "Injected RSpec disabled example (xit)" },
2139
+ { pattern: /,\s*skip:\s*(?:true|["\x27])/i, desc: "Injected RSpec skip metadata (skip:)" },
2140
+ { pattern: /^\s*(?:skip|pending)\s*(?:["\x27(]|$)/i, desc: "Injected Minitest/RSpec skip statement" },
2027
2141
  ];
2028
2142
 
2029
- const COMMENTED_ASSERTION = /^\+\s*(?:\/\/|\/\*)\s*(?:expect\(|assert\.|assert\(|t\.expect|t\.assert)/i;
2143
+ // `#` and `--` belong here for the same reason the dialects belong in
2144
+ // ASSERTION_PATTERN: a Ruby or Python assertion commented out is exactly
2145
+ // as gone as a JavaScript one, and was previously not looked for.
2146
+ const COMMENTED_ASSERTION =
2147
+ /^\+\s*(?:\/\/|\/\*|#|--)\s*(?:expect\s*\(|assert(?!ion|ing|ed\b|s\b)[a-zA-Z0-9_$]*\s*[.(]|assert\s|refute_|XCTAssert|t\.expect|t\.assert)/i;
2030
2148
 
2031
2149
  const VACUOUS_ASSERTIONS = [
2032
2150
  { pattern: /\bassert(?:\.ok)?\s*\(\s*true\s*(?:,[^)]*)?\)/i, desc: "Vacuous truth assertion (assert.ok(true))" },
@@ -2037,11 +2155,44 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2037
2155
  { pattern: /\bexpect\s*\(\s*false\s*\)\s*\.toBeFalsy\s*\(/i, desc: "Vacuous falsity expectation (expect(false).toBeFalsy())" },
2038
2156
  { pattern: /\bassert\.(?:isTrue|isOk)\s*\(\s*true\s*(?:,[^)]*)?\)/i, desc: "Vacuous truth assertion (assert.isTrue(true))" },
2039
2157
  { pattern: /\bassert\.(?:isFalse|isNotOk)\s*\(\s*false\s*(?:,[^)]*)?\)/i, desc: "Vacuous falsity assertion (assert.isFalse(false))" },
2158
+ { pattern: /\b(?:XCT)?assertTrue\s*\(\s*true\s*[,)]/i, desc: "Vacuous truth assertion (assertTrue(true))" },
2159
+ { pattern: /\b(?:XCT)?assertFalse\s*\(\s*false\s*[,)]/i, desc: "Vacuous falsity assertion (assertFalse(false))" },
2160
+ { pattern: /\b(?:assertEquals|assertSame|XCTAssertEqual)\s*\(\s*([^,]+?)\s*,\s*\1\s*[,)]/i, desc: "Vacuous identity assertion (assertEquals(X, X))" },
2161
+ { pattern: /\bassert_equal\s*\(?\s*([^,]+?)\s*,\s*\1\s*\)?\s*$/i, desc: "Vacuous identity assertion (assert_equal X, X)" },
2162
+ { pattern: /\bexpect\s*\(\s*true\s*\)\s*\.to\s+be(?:\s+true)?\b/i, desc: "Vacuous truth expectation (expect(true).to be true)" },
2040
2163
  ];
2041
2164
 
2042
- const ASSERTION_PATTERN = /(?:\b(?:assert(?:\.[a-zA-Z0-9_$]+)?|expect|t\.(?:assert|expect|is|equal|true|false|Errorf|Fatalf)|require\.[a-zA-Z0-9_$]+)\b|assert!|assert_eq!|assert_ne!)/i;
2165
+ // Broad on purpose: this is the denominator, not the verdict. A word
2166
+ // boundary immediately after `assert` never falls in `assertEquals`,
2167
+ // `assert_equal` or `XCTAssertEqual`, so five ecosystems contributed no
2168
+ // assertions to count at all and a gutted JUnit suite was arithmetically
2169
+ // indistinguishable from an untouched one. The lookahead keeps prose and
2170
+ // identifiers — `assertion`, `asserts`, `asserted` — out of the count.
2171
+ const ASSERTION_PATTERN =
2172
+ /(?:\b(?:assert(?!ion|ing|ed\b|s\b)[a-zA-Z0-9_$]*(?:\.[a-zA-Z0-9_$]+)?|refute[a-zA-Z0-9_$]*|XCTAssert[a-zA-Z0-9_$]*|XCTFail|expect|t\.(?:assert|expect|is|equal|true|false|Errorf|Fatalf)|require\.[a-zA-Z0-9_$]+)\b|assert!|assert_eq!|assert_ne!)/i;
2173
+ // The loose net. Not a verdict and never a block — its only job is to
2174
+ // notice that a line was plainly an assertion in *some* dialect that
2175
+ // ASSERTION_PATTERN did not recognise. Without it, adding the seventh
2176
+ // ecosystem is indistinguishable from having covered it all along: the
2177
+ // guard returns the same clean PASS either way. This is the denominator
2178
+ // for the denominator.
2179
+ // Deliberately not call-shaped. Haskell's `x `shouldBe` 3` is an
2180
+ // assertion with no parentheses anywhere near it, and a net that only
2181
+ // catches `name(` reports the same confident PASS on it as on a clean
2182
+ // Node suite. `require` and `check` are absent on purpose: in a
2183
+ // CommonJS test file `require("./calc")` is an import, not a claim.
2184
+ const ASSERTION_SHAPED =
2185
+ /\b(?:assert(?!ion|ing|ed\b|s\b)|expect(?!ed\b|ation)|refute)[a-zA-Z0-9_$]*\b|`\s*should[a-zA-Z0-9_$]*\s*`|\b(?:should|must|verify|ensure|confirm)[a-zA-Z0-9_$]*\s*[(!]|\.\s*(?:should|to|to_not|not_to|must)\b|\bBOOST_[A-Z_]+\s*\(|\b[A-Z]+_(?:EQ|NE|TRUE|FALSE|THAT)\s*\(/;
2043
2186
  const isCommentLine = (str) => /^\s*(?:\/\/|\/\*|\*|#|--|;)/.test(str);
2044
2187
 
2188
+ /** Book-keeping only: what this run looked at, before deciding anything. */
2189
+ const countExamined = (stats, text) => {
2190
+ if (!text.trim() || isCommentLine(text)) return;
2191
+ stats.examined++;
2192
+ if (ASSERTION_PATTERN.test(text)) stats.recognised++;
2193
+ else if (ASSERTION_SHAPED.test(text) && stats.unreadable.length < 5) stats.unreadable.push(text.trim().slice(0, 120));
2194
+ };
2195
+
2045
2196
  const fileAssertions = new Map();
2046
2197
  let pendingHunk = false;
2047
2198
 
@@ -2077,7 +2228,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2077
2228
  }
2078
2229
 
2079
2230
  if (!fileAssertions.has(currentFile)) {
2080
- fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [] });
2231
+ fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [] });
2081
2232
  }
2082
2233
  const fileStats = fileAssertions.get(currentFile);
2083
2234
  if (pendingHunk) {
@@ -2089,6 +2240,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2089
2240
  if (line.startsWith("-") && !line.startsWith("---")) {
2090
2241
  const deletedText = line.slice(1);
2091
2242
  if (hunk) hunk.lines.push({ kind: "-", text: deletedText, oldNo: currentOldLineNo, newNo: null });
2243
+ countExamined(fileStats, deletedText);
2092
2244
  if (!isCommentLine(deletedText) && ASSERTION_PATTERN.test(deletedText)) {
2093
2245
  fileStats.removed.push({ line: currentOldLineNo, text: deletedText });
2094
2246
  if (isSpecificAssertion(deletedText)) {
@@ -2099,6 +2251,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2099
2251
  } else if (line.startsWith("+") && !line.startsWith("+++")) {
2100
2252
  const addedText = line.slice(1);
2101
2253
  if (hunk) hunk.lines.push({ kind: "+", text: addedText, oldNo: null, newNo: currentNewLineNo });
2254
+ countExamined(fileStats, addedText);
2102
2255
  let isVacuous = false;
2103
2256
 
2104
2257
  // Check skip injections
@@ -2225,17 +2378,44 @@ export function checkTestTampering(diffOrText = "", options = {}) {
2225
2378
  // `ok: true` from a guard that looked at everything and approved it. That
2226
2379
  // ambiguity is how a substring bug in the file classifier switched this
2227
2380
  // entire guard off for the standard pytest, Rust and RSpec layouts while
2228
- // every signal stayed green. A verdict without a denominator is not a
2229
- // verdict, so `inputsSeen` reports the number of test files this run
2230
- // actually reasoned about, and `status` distinguishes "nothing to check"
2231
- // from "checked and clean".
2232
- const inputsSeen = fileAssertions.size;
2381
+ // every signal stayed green.
2382
+ //
2383
+ // Counting *files* was not enough. A JUnit diff that rewrote an expected
2384
+ // value produced `inputsSeen: 1` and a clean PASS while not one assertion
2385
+ // in it had been recognised — the same ambiguity, one level down, inside
2386
+ // the mechanism built to remove it. So the denominator is now the thing
2387
+ // the rules actually consume: lines examined, and of those, assertions
2388
+ // understood. `UNREADABLE` is the state that has no business being silent
2389
+ // — assertion-shaped lines were present and none of them parsed, which
2390
+ // means this repository speaks a dialect the guard does not.
2391
+ let examined = 0;
2392
+ let assertionsSeen = 0;
2393
+ const unreadable = [];
2394
+ for (const [file, stats] of fileAssertions.entries()) {
2395
+ examined += stats.examined;
2396
+ assertionsSeen += stats.recognised;
2397
+ if (stats.unreadable.length > 0) {
2398
+ unreadable.push({ file, count: stats.unreadable.length, samples: stats.unreadable.slice(0, 3) });
2399
+ }
2400
+ }
2401
+
2402
+ const status =
2403
+ reported.length > 0
2404
+ ? "FAIL"
2405
+ : assertionsSeen === 0 && unreadable.length > 0
2406
+ ? "UNREADABLE"
2407
+ : examined > 0
2408
+ ? "PASS"
2409
+ : "NOT_APPLICABLE";
2233
2410
 
2234
2411
  return {
2235
2412
  ok: reported.length === 0,
2236
2413
  violations: reported,
2237
- inputsSeen,
2238
- status: reported.length > 0 ? "FAIL" : inputsSeen > 0 ? "PASS" : "NOT_APPLICABLE",
2414
+ inputsSeen: examined,
2415
+ filesSeen: fileAssertions.size,
2416
+ assertionsSeen,
2417
+ unreadable,
2418
+ status,
2239
2419
  };
2240
2420
  }
2241
2421
 
@@ -188,6 +188,56 @@ export function detectEdgeRuntime(projectRoot = process.cwd()) {
188
188
  /**
189
189
  * Detects 24+ polyglot stacks and container environments.
190
190
  */
191
+ /**
192
+ * Test commands worth trying, best first, when the detected one does not run.
193
+ *
194
+ * `init` probes the command it picked. On a repository whose Makefile
195
+ * declares a `test` target that needs a build environment the machine does
196
+ * not have, that probe failed, printed `Oracle verification probe failed`,
197
+ * and the wizard wrote the broken command into the config anyway — in a
198
+ * repository where `pytest` was on PATH and all 360 tests passed in 1.3s.
199
+ * Measuring something and then ignoring the measurement is worse than not
200
+ * measuring: it produces a hard red on day one, which is how a user learns
201
+ * the gate is broken and turns it off.
202
+ *
203
+ * Kept deliberately generic — a per-ecosystem convention, never a per-project
204
+ * or per-provider guess.
205
+ *
206
+ * @param {string} root
207
+ * @param {string} [detected] - the command detection chose; always first.
208
+ * @returns {string[]} ordered, de-duplicated candidates
209
+ */
210
+ export function oracleCandidates(root = process.cwd(), detected = "") {
211
+ const out = [];
212
+ const push = (c) => {
213
+ const v = (c || "").trim();
214
+ if (v && !out.includes(v) && !isPlaceholderTestScript(v)) out.push(v);
215
+ };
216
+ const has = (f) => existsSync(join(root, f));
217
+
218
+ push(detected);
219
+
220
+ if (has("package.json")) {
221
+ try {
222
+ const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
223
+ if (pkg.scripts?.test && !isPlaceholderTestScript(pkg.scripts.test)) push("npm test");
224
+ } catch (_) {}
225
+ }
226
+ if (has("pytest.ini") || has("pyproject.toml") || has("setup.py") || has("tox.ini") || has("setup.cfg")) {
227
+ push(pytestCmd());
228
+ }
229
+ if (has("Cargo.toml")) push("cargo test");
230
+ if (has("go.mod")) push("go test ./...");
231
+ if (has("Gemfile")) push("bundle exec rspec");
232
+ if (has("composer.json")) push("./vendor/bin/phpunit");
233
+ if (has("pom.xml")) push("mvn -q test");
234
+ if (has("build.gradle") || has("build.gradle.kts")) push("./gradlew test");
235
+ if (has("pubspec.yaml")) push("dart test");
236
+ if (has("Package.swift")) push("swift test");
237
+
238
+ return out;
239
+ }
240
+
191
241
  export function detectPolyglotStack(projectRoot = process.cwd()) {
192
242
  const edgeInfo = detectEdgeRuntime(projectRoot);
193
243
  const isDevcontainer = existsSync(join(projectRoot, ".devcontainer", "devcontainer.json"));
@@ -3,7 +3,7 @@ import { join } from "node:path";
3
3
  import { parseYaml, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
4
4
  import { suggestProvider, detectAvailableProviders } from "./provider-readiness.mjs";
5
5
  import { detectDefaultBranch } from "./git.mjs";
6
- import { resolveWorkspaceBoundary } from "./stack-detector.mjs";
6
+ import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
7
7
  import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
8
8
  import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
9
9
  import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
@@ -302,6 +302,51 @@ export function loadPresets(root = process.cwd()) {
302
302
  * @param {object} [options]
303
303
  * @returns {Promise<{ ok: boolean, configPath: string, plan: object }>}
304
304
  */
305
+ /**
306
+ * Probe the chosen test command, and take detection's next choice if it fails.
307
+ *
308
+ * Runs on the non-interactive path too. `--yes` means "do not ask me", not
309
+ * "do not check" — and the user who is not watching is exactly the one who
310
+ * cannot notice that the command written into their config does not run.
311
+ * Before this, the probe lived inside the interactive branch, so
312
+ * `agentctl init --yes` wrote `make test` into a repository where `make test`
313
+ * exits 2 and `npm test` passes, and the first gate run was a hard red.
314
+ *
315
+ * @returns {Promise<string>} the command to save
316
+ */
317
+ async function resolveRunnableOracle(root, testCmd, options = {}) {
318
+ if (!testCmd) return testCmd;
319
+ const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
320
+ const probeRes = await runVerificationProbe(testCmd, root);
321
+ if (probeRes.ok) {
322
+ probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
323
+ return testCmd;
324
+ }
325
+ probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
326
+
327
+ const alternates = oracleCandidates(root, testCmd).filter((c) => c !== testCmd).slice(0, 3);
328
+ for (const cand of alternates) {
329
+ const altSp = spinner(`Trying ${cand}`, options);
330
+ const altRes = await runVerificationProbe(cand, root);
331
+ if (altRes.ok) {
332
+ altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
333
+ return cand;
334
+ }
335
+ altSp.fail(`${cand} also failed (Exit ${altRes.code})`);
336
+ }
337
+
338
+ // Nothing runs. Say so in terms the user can act on, rather than leaving a
339
+ // failed spinner to scroll past and a broken command in the config.
340
+ const out = options.stdout || process.stdout;
341
+ out.write("\n");
342
+ out.write(" \u26a0\ufe0f No test command could be run in this environment.\n");
343
+ out.write(` Keeping "${testCmd}" \u2014 the gate will fail until it runs here.\n`);
344
+ out.write(" Point verify.test in .agent/config.yml at a command that works,\n");
345
+ out.write(" or, if this repository genuinely has no suite, set\n");
346
+ out.write(" verify.required: false deliberately rather than by accident.\n\n");
347
+ return testCmd;
348
+ }
349
+
305
350
  export async function runInitWizard(root = process.cwd(), options = {}) {
306
351
  const interactive = options.interactive !== false && isTTY(options.stdin || process.stdin);
307
352
 
@@ -322,6 +367,7 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
322
367
  let selectedProvider = options.provider || existingConfig.provider;
323
368
  let selectedProfile = options.profile || existingConfig.verify?.profile;
324
369
  let testCmd = options.testCmd;
370
+ let probeInteractive = null;
325
371
  let buildCmd = options.buildCmd;
326
372
  let selectedPresets = options.presets;
327
373
 
@@ -402,16 +448,20 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
402
448
 
403
449
  selectedPresets = await multiSelect(presetOptions, "Select Autonomous Workflows to Enable", options);
404
450
 
405
- const shouldProbe = await confirm("Run verification probe on test command before saving?", true, options);
406
- if (shouldProbe && testCmd) {
407
- const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
408
- const probeRes = await runVerificationProbe(testCmd, root);
409
- if (probeRes.ok) {
410
- probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
411
- } else {
412
- probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
413
- }
414
- }
451
+ probeInteractive = await confirm("Run verification probe on test command before saving?", true, options);
452
+ }
453
+
454
+ // The probe runs whether or not anyone was asked: interactive users can
455
+ // decline it, but silence from `--yes` is not a decline.
456
+ if (probeInteractive !== false && options.probe !== false) {
457
+ // Resolve the command the way planInit will, or there is nothing to
458
+ // probe: on the headless path `testCmd` stays undefined until planInit
459
+ // fills it in from detection, so the probe silently examined nothing —
460
+ // the exact fail-open shape this project keeps finding in itself.
461
+ const effective =
462
+ testCmd || existingConfig.verify?.test || detectStackOracles(root)?.candidates?.testCmd || "";
463
+ const adopted = await resolveRunnableOracle(root, effective, options);
464
+ if (adopted) testCmd = adopted;
415
465
  }
416
466
 
417
467
  // `...options` first, for the same reason as in wizard-task.mjs: spreading it