jules-orchestrator-kit 0.63.0 → 0.65.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +6 -3
- package/bin/agentctl.mjs +5 -0
- package/package.json +3 -2
- package/scripts/guard-reach-check.mjs +68 -6
- package/scripts/package-integrity-check.mjs +251 -0
- package/scripts/release.mjs +16 -0
- package/src/assertions.mjs +16 -0
- package/src/engine.mjs +49 -1
- package/src/guard-policy.mjs +512 -0
- package/src/ops/test-collection.mjs +26 -9
- package/src/security.mjs +194 -14
- package/src/stack-detector.mjs +50 -0
- package/src/wizard-init.mjs +61 -11
package/src/security.mjs
CHANGED
|
@@ -1155,14 +1155,42 @@ function locateFindingLine(lines, type, file = null) {
|
|
|
1155
1155
|
// (`expect(\n formatInvoice(bill)\n).toBe(…`) still recognises the chain.
|
|
1156
1156
|
// The bound is a guess: an argument list longer than 240 characters is
|
|
1157
1157
|
// rarer than a missed chain.
|
|
1158
|
+
// The dialect list is not decoration. `assertEqual` was recognised only
|
|
1159
|
+
// because `\.?` made the dot optional and the `i` flag let `Equal` match
|
|
1160
|
+
// `equal`; `assertEquals`, one letter longer, fell out of the pattern and
|
|
1161
|
+
// took JUnit, PHPUnit, Minitest, RSpec and XCTest with it. The weak forms
|
|
1162
|
+
// — assertTrue, assertNotNull, XCTAssertTrue — are deliberately absent:
|
|
1163
|
+
// they state no expected value, so their arrival in place of one of these
|
|
1164
|
+
// is a weakening, which is a finding of its own.
|
|
1158
1165
|
const SPECIFIC_ASSERTION = new RegExp(
|
|
1159
1166
|
[
|
|
1160
1167
|
"\\bassert(?:\\.strict)?\\.?(?:strictEqual|deepStrictEqual|deepEqual|notStrictEqual|notDeepStrictEqual|equal|notEqual|match|doesNotMatch|throws|rejects|doesNotThrow)\\s*\\(",
|
|
1161
1168
|
"\\bexpect\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:toBe|toEqual|toStrictEqual|toMatch|toMatchObject|toContain|toHaveBeenCalledWith|toThrow|toHaveLength|toBeCloseTo)\\s*\\(",
|
|
1162
1169
|
"\\bassert\\.(?:equals|deepEquals|include|lengthOf)\\s*\\(",
|
|
1163
1170
|
"assert_eq!|assert_ne!",
|
|
1171
|
+
// The bare comparison form. `assert add(1, 2) == 3` is how pytest is
|
|
1172
|
+
// actually written, and Rust's `assert!(a == b)` and Elixir's
|
|
1173
|
+
// `assert f(x) == 3` follow it; none of them name a comparison
|
|
1174
|
+
// function, so a list of function names could never reach them.
|
|
1175
|
+
// Equality only. `assert!(x != 0)` names no expected value — it is the
|
|
1176
|
+
// weaker claim you arrive at by giving one up, and counting it as
|
|
1177
|
+
// specific would make the downgrade from `assert_eq!(x, 5)` invisible to
|
|
1178
|
+
// the weakening check.
|
|
1179
|
+
"\\bassert\\s+[^\\n]*(?:===|==)(?!=)",
|
|
1180
|
+
"\\bassert!\\s*\\([^\\n]*==(?!=)",
|
|
1164
1181
|
"\\bt\\.(?:Errorf|Fatalf)\\s*\\(",
|
|
1165
1182
|
"\\brequire\\.(?:Equal|NotEqual|Len|Contains|Error|NoError)\\s*\\(",
|
|
1183
|
+
// Python unittest, stated rather than inherited from the optional dot.
|
|
1184
|
+
"\\bassert(?:Equal|NotEqual|AlmostEqual|NotAlmostEqual|Regex|NotRegex|Raises|In|NotIn|Is|IsNot|ListEqual|DictEqual|SetEqual|TupleEqual|CountEqual|Greater|Less|GreaterEqual|LessEqual)\\s*\\(",
|
|
1185
|
+
// JUnit / TestNG / PHPUnit
|
|
1186
|
+
"\\bassert(?:Equals|NotEquals|Same|NotSame|ArrayEquals|IterableEquals|LinesMatch|Count|StringContainsString|StringEqualsFile|InstanceOf|Contains|Throws)\\s*\\(",
|
|
1187
|
+
"\\bassertThat\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:isEqualTo|isSameAs|contains|containsExactly|hasSize|isCloseTo|matches)\\s*\\(",
|
|
1188
|
+
// Minitest
|
|
1189
|
+
"\\b(?:assert|refute)_(?:equal|includes|match|nil|same|in_delta|in_epsilon|raises|empty|operator|predicate)\\b",
|
|
1190
|
+
// RSpec
|
|
1191
|
+
"\\bexpect\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:to|not_to|to_not)\\s+(?:eq|eql|equal|be|be_within|match|include|contain_exactly|match_array|have_attributes|raise_error|start_with|end_with)\\b",
|
|
1192
|
+
// XCTest
|
|
1193
|
+
"\\bXCTAssert(?:Equal|NotEqual|EqualWithAccuracy|Identical|NotIdentical|GreaterThan|LessThan|GreaterThanOrEqual|LessThanOrEqual|ThrowsError|NoThrow)\\s*\\(",
|
|
1166
1194
|
].join("|"),
|
|
1167
1195
|
"i"
|
|
1168
1196
|
);
|
|
@@ -1203,6 +1231,15 @@ const TEST_LANG_BY_EXT = new Map([
|
|
|
1203
1231
|
[".py", "python"], [".pyi", "python"],
|
|
1204
1232
|
[".go", "go"],
|
|
1205
1233
|
[".rs", "rust"],
|
|
1234
|
+
// Approximations, chosen for comment and continuation syntax rather than
|
|
1235
|
+
// for kinship: the C-like family reads correctly under the `js` scanner,
|
|
1236
|
+
// and Ruby under the `python` one because both end a comment at `#` and a
|
|
1237
|
+
// statement at the newline. Naming them beats falling through to `js` by
|
|
1238
|
+
// default, which is how a `#` comment came to be read as code.
|
|
1239
|
+
[".java", "js"], [".kt", "js"], [".kts", "js"], [".scala", "js"], [".groovy", "js"],
|
|
1240
|
+
[".swift", "js"], [".cs", "js"], [".php", "js"], [".c", "js"], [".cc", "js"],
|
|
1241
|
+
[".cpp", "js"], [".h", "js"], [".hpp", "js"], [".m", "js"], [".sol", "js"],
|
|
1242
|
+
[".rb", "python"],
|
|
1206
1243
|
]);
|
|
1207
1244
|
|
|
1208
1245
|
function langForTestFile(file) {
|
|
@@ -1540,6 +1577,21 @@ function stripComments(text, lang) {
|
|
|
1540
1577
|
const CONTINUATION_START = /^[)\],.]/;
|
|
1541
1578
|
const CONTINUATION_OP_START = /^[+\-*/%<>=&|^:]/;
|
|
1542
1579
|
|
|
1580
|
+
// A comment is not a continuation, however much it looks like one.
|
|
1581
|
+
//
|
|
1582
|
+
// `//` begins with a division sign and `--` with a minus, so both matched
|
|
1583
|
+
// CONTINUATION_OP_START and folded the following comment line into the
|
|
1584
|
+
// statement above it. The cost was a false accusation on a virtuous act:
|
|
1585
|
+
// adding an assertion next to a `// ...` line made the new assertion absorb
|
|
1586
|
+
// the comment, stop matching its unchanged twin, and get reported as a
|
|
1587
|
+
// rewritten expectation. Python was unaffected only because `#` is not an
|
|
1588
|
+
// operator — which is why the same fixture passed in pytest and failed in
|
|
1589
|
+
// Jest, and why it survived every suite written against the pytest layout.
|
|
1590
|
+
//
|
|
1591
|
+
// A comment inside an open delimiter still joins: `cur.delta > 0` decides
|
|
1592
|
+
// that before this test is ever reached.
|
|
1593
|
+
const COMMENT_LINE_START = /^(?:\/\/|\/\*|#|--)/;
|
|
1594
|
+
|
|
1543
1595
|
// A scanner miscount (an unbalanced delimiter inside a regex literal is the
|
|
1544
1596
|
// usual cause) must not be able to merge a whole file into one statement,
|
|
1545
1597
|
// which would pair *any* literal change anywhere in the file.
|
|
@@ -1578,14 +1630,15 @@ function assembleStatements(sliceLines, lang) {
|
|
|
1578
1630
|
|
|
1579
1631
|
for (const L of sliceLines) {
|
|
1580
1632
|
const trimmed = L.text.replace(/^\s+/, "");
|
|
1633
|
+
const startsComment = COMMENT_LINE_START.test(trimmed);
|
|
1581
1634
|
const joins =
|
|
1582
1635
|
cur !== null &&
|
|
1583
1636
|
(cur.delta > 0 ||
|
|
1584
1637
|
cur.state.str !== null ||
|
|
1585
1638
|
cur.state.block > 0 ||
|
|
1586
1639
|
cur.trailingBackslash ||
|
|
1587
|
-
|
|
1588
|
-
|
|
1640
|
+
(!startsComment &&
|
|
1641
|
+
(CONTINUATION_START.test(trimmed) || CONTINUATION_OP_START.test(trimmed))));
|
|
1589
1642
|
|
|
1590
1643
|
if (
|
|
1591
1644
|
joins &&
|
|
@@ -1648,7 +1701,17 @@ function splitAssertionArgs(clean, lang) {
|
|
|
1648
1701
|
const m = SPECIFIC_ASSERTION.exec(clean);
|
|
1649
1702
|
if (!m) return null;
|
|
1650
1703
|
|
|
1651
|
-
|
|
1704
|
+
// Not every branch of SPECIFIC_ASSERTION ends at an opening paren:
|
|
1705
|
+
// `assert_eq!`, `assert_equal` and RSpec's `.to eq` all match a bare name.
|
|
1706
|
+
// Starting the walk one character early made every argument boundary wrong,
|
|
1707
|
+
// so a reworded message read as a rewritten value.
|
|
1708
|
+
let i = m.index + m[0].length;
|
|
1709
|
+
if (clean[i - 1] !== "(") {
|
|
1710
|
+
let j = i;
|
|
1711
|
+
while (j < clean.length && /\s/.test(clean[j])) j++;
|
|
1712
|
+
if (clean[j] !== "(") return null;
|
|
1713
|
+
i = j + 1;
|
|
1714
|
+
}
|
|
1652
1715
|
let depth = 1;
|
|
1653
1716
|
let quote = null;
|
|
1654
1717
|
let triple = false;
|
|
@@ -1740,7 +1803,45 @@ function messageArgIndices(args) {
|
|
|
1740
1803
|
* every time somebody improved the wording of a failure. Firing on that is
|
|
1741
1804
|
* how an operator learns to pass the override without reading it.
|
|
1742
1805
|
*/
|
|
1806
|
+
/**
|
|
1807
|
+
* Split a statement at a trailing `, "message"` written outside the call.
|
|
1808
|
+
*
|
|
1809
|
+
* RSpec puts the message there — `expect(x).to eq(3), "explain"` — and so do
|
|
1810
|
+
* Ruby and Elixir assertions generally. An argument-position check can never
|
|
1811
|
+
* see it, so rewording one read as a rewritten expectation.
|
|
1812
|
+
*/
|
|
1813
|
+
function splitTrailingMessage(clean) {
|
|
1814
|
+
let depth = 0;
|
|
1815
|
+
let quote = null;
|
|
1816
|
+
let lastComma = -1;
|
|
1817
|
+
for (let i = 0; i < clean.length; i++) {
|
|
1818
|
+
const c = clean[i];
|
|
1819
|
+
if (quote !== null) {
|
|
1820
|
+
if (c === "\\") { i += 1; continue; }
|
|
1821
|
+
if (c === quote) quote = null;
|
|
1822
|
+
continue;
|
|
1823
|
+
}
|
|
1824
|
+
if (c === '"' || c === "'" || c === "`") { quote = c; continue; }
|
|
1825
|
+
if (c === "(" || c === "[" || c === "{") depth++;
|
|
1826
|
+
else if (c === ")" || c === "]" || c === "}") depth--;
|
|
1827
|
+
else if (c === "," && depth === 0) lastComma = i;
|
|
1828
|
+
}
|
|
1829
|
+
if (lastComma === -1) return { head: clean, msg: null };
|
|
1830
|
+
const tail = clean.slice(lastComma + 1).trim();
|
|
1831
|
+
if (!isPureStringLiteral(tail)) return { head: clean, msg: null };
|
|
1832
|
+
return { head: clean.slice(0, lastComma), msg: tail };
|
|
1833
|
+
}
|
|
1834
|
+
|
|
1743
1835
|
function differsOnlyInMessage(cleanRemoved, cleanAdded, lang) {
|
|
1836
|
+
const ta = splitTrailingMessage(cleanRemoved);
|
|
1837
|
+
const tb = splitTrailingMessage(cleanAdded);
|
|
1838
|
+
if (
|
|
1839
|
+
(ta.msg !== null || tb.msg !== null) &&
|
|
1840
|
+
ta.head.replace(/\s+/g, "") === tb.head.replace(/\s+/g, "")
|
|
1841
|
+
) {
|
|
1842
|
+
return true;
|
|
1843
|
+
}
|
|
1844
|
+
|
|
1744
1845
|
const a = splitAssertionArgs(cleanRemoved, lang);
|
|
1745
1846
|
const b = splitAssertionArgs(cleanAdded, lang);
|
|
1746
1847
|
if (!a || !b || a.length !== b.length || a.length === 0) return false;
|
|
@@ -1988,7 +2089,9 @@ export function resolveAllowedTamperKinds(options = {}) {
|
|
|
1988
2089
|
* @param {string} diffOrText - Unified git diff
|
|
1989
2090
|
* @param {Object} [options]
|
|
1990
2091
|
* @param {boolean} [options.allowTestModifications=false]
|
|
1991
|
-
* @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number,
|
|
2092
|
+
* @returns {{ ok: boolean, violations: Array<object>, inputsSeen: number, filesSeen: number,
|
|
2093
|
+
* assertionsSeen: number, unreadable: Array<{file: string, count: number, samples: string[]}>,
|
|
2094
|
+
* status: "PASS"|"FAIL"|"UNREADABLE"|"NOT_APPLICABLE" }}
|
|
1992
2095
|
* `status` distinguishes "checked and clean" from "nothing was checked";
|
|
1993
2096
|
* `ok: true` alone cannot, and that ambiguity is the defect class this
|
|
1994
2097
|
* field exists to make visible.
|
|
@@ -2024,9 +2127,24 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2024
2127
|
{ pattern: /#\[ignore\]/i, desc: "Injected Rust test ignore attribute (#[ignore])" },
|
|
2025
2128
|
{ pattern: /\/\/\s*@ts-ignore/i, desc: "Injected TypeScript ignore comment (// @ts-ignore)" },
|
|
2026
2129
|
{ pattern: /\/\/\s*@ts-nocheck/i, desc: "Injected TypeScript nocheck directive (// @ts-nocheck)" },
|
|
2130
|
+
// A suite that never runs cannot fail, and `@Disabled` is exactly as
|
|
2131
|
+
// effective at arranging that as `it.skip` is.
|
|
2132
|
+
{ pattern: /@Disabled\b/, desc: "Injected JUnit 5 disable annotation (@Disabled)" },
|
|
2133
|
+
{ pattern: /@Ignore\b/, desc: "Injected JUnit 4 / TestNG ignore annotation (@Ignore)" },
|
|
2134
|
+
{ pattern: /@Test\s*\([^)]*enabled\s*=\s*false/i, desc: "Injected TestNG disabled test (enabled = false)" },
|
|
2135
|
+
{ pattern: /@unittest\.skip/i, desc: "Injected unittest skip decorator (@unittest.skip)" },
|
|
2136
|
+
{ pattern: /\bmarkTest(?:Skipped|Incomplete)\s*\(/i, desc: "Injected PHPUnit skip (markTestSkipped())" },
|
|
2137
|
+
{ pattern: /\bXCTSkip(?:If|Unless|IfNot)?\s*\(/, desc: "Injected XCTest skip (XCTSkip())" },
|
|
2138
|
+
{ pattern: /\b(?:xit|xdescribe|xcontext|xspecify)\b\s*["\x27]/i, desc: "Injected RSpec disabled example (xit)" },
|
|
2139
|
+
{ pattern: /,\s*skip:\s*(?:true|["\x27])/i, desc: "Injected RSpec skip metadata (skip:)" },
|
|
2140
|
+
{ pattern: /^\s*(?:skip|pending)\s*(?:["\x27(]|$)/i, desc: "Injected Minitest/RSpec skip statement" },
|
|
2027
2141
|
];
|
|
2028
2142
|
|
|
2029
|
-
|
|
2143
|
+
// `#` and `--` belong here for the same reason the dialects belong in
|
|
2144
|
+
// ASSERTION_PATTERN: a Ruby or Python assertion commented out is exactly
|
|
2145
|
+
// as gone as a JavaScript one, and was previously not looked for.
|
|
2146
|
+
const COMMENTED_ASSERTION =
|
|
2147
|
+
/^\+\s*(?:\/\/|\/\*|#|--)\s*(?:expect\s*\(|assert(?!ion|ing|ed\b|s\b)[a-zA-Z0-9_$]*\s*[.(]|assert\s|refute_|XCTAssert|t\.expect|t\.assert)/i;
|
|
2030
2148
|
|
|
2031
2149
|
const VACUOUS_ASSERTIONS = [
|
|
2032
2150
|
{ pattern: /\bassert(?:\.ok)?\s*\(\s*true\s*(?:,[^)]*)?\)/i, desc: "Vacuous truth assertion (assert.ok(true))" },
|
|
@@ -2037,11 +2155,44 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2037
2155
|
{ pattern: /\bexpect\s*\(\s*false\s*\)\s*\.toBeFalsy\s*\(/i, desc: "Vacuous falsity expectation (expect(false).toBeFalsy())" },
|
|
2038
2156
|
{ pattern: /\bassert\.(?:isTrue|isOk)\s*\(\s*true\s*(?:,[^)]*)?\)/i, desc: "Vacuous truth assertion (assert.isTrue(true))" },
|
|
2039
2157
|
{ pattern: /\bassert\.(?:isFalse|isNotOk)\s*\(\s*false\s*(?:,[^)]*)?\)/i, desc: "Vacuous falsity assertion (assert.isFalse(false))" },
|
|
2158
|
+
{ pattern: /\b(?:XCT)?assertTrue\s*\(\s*true\s*[,)]/i, desc: "Vacuous truth assertion (assertTrue(true))" },
|
|
2159
|
+
{ pattern: /\b(?:XCT)?assertFalse\s*\(\s*false\s*[,)]/i, desc: "Vacuous falsity assertion (assertFalse(false))" },
|
|
2160
|
+
{ pattern: /\b(?:assertEquals|assertSame|XCTAssertEqual)\s*\(\s*([^,]+?)\s*,\s*\1\s*[,)]/i, desc: "Vacuous identity assertion (assertEquals(X, X))" },
|
|
2161
|
+
{ pattern: /\bassert_equal\s*\(?\s*([^,]+?)\s*,\s*\1\s*\)?\s*$/i, desc: "Vacuous identity assertion (assert_equal X, X)" },
|
|
2162
|
+
{ pattern: /\bexpect\s*\(\s*true\s*\)\s*\.to\s+be(?:\s+true)?\b/i, desc: "Vacuous truth expectation (expect(true).to be true)" },
|
|
2040
2163
|
];
|
|
2041
2164
|
|
|
2042
|
-
|
|
2165
|
+
// Broad on purpose: this is the denominator, not the verdict. A word
|
|
2166
|
+
// boundary immediately after `assert` never falls in `assertEquals`,
|
|
2167
|
+
// `assert_equal` or `XCTAssertEqual`, so five ecosystems contributed no
|
|
2168
|
+
// assertions to count at all and a gutted JUnit suite was arithmetically
|
|
2169
|
+
// indistinguishable from an untouched one. The lookahead keeps prose and
|
|
2170
|
+
// identifiers — `assertion`, `asserts`, `asserted` — out of the count.
|
|
2171
|
+
const ASSERTION_PATTERN =
|
|
2172
|
+
/(?:\b(?:assert(?!ion|ing|ed\b|s\b)[a-zA-Z0-9_$]*(?:\.[a-zA-Z0-9_$]+)?|refute[a-zA-Z0-9_$]*|XCTAssert[a-zA-Z0-9_$]*|XCTFail|expect|t\.(?:assert|expect|is|equal|true|false|Errorf|Fatalf)|require\.[a-zA-Z0-9_$]+)\b|assert!|assert_eq!|assert_ne!)/i;
|
|
2173
|
+
// The loose net. Not a verdict and never a block — its only job is to
|
|
2174
|
+
// notice that a line was plainly an assertion in *some* dialect that
|
|
2175
|
+
// ASSERTION_PATTERN did not recognise. Without it, adding the seventh
|
|
2176
|
+
// ecosystem is indistinguishable from having covered it all along: the
|
|
2177
|
+
// guard returns the same clean PASS either way. This is the denominator
|
|
2178
|
+
// for the denominator.
|
|
2179
|
+
// Deliberately not call-shaped. Haskell's `x `shouldBe` 3` is an
|
|
2180
|
+
// assertion with no parentheses anywhere near it, and a net that only
|
|
2181
|
+
// catches `name(` reports the same confident PASS on it as on a clean
|
|
2182
|
+
// Node suite. `require` and `check` are absent on purpose: in a
|
|
2183
|
+
// CommonJS test file `require("./calc")` is an import, not a claim.
|
|
2184
|
+
const ASSERTION_SHAPED =
|
|
2185
|
+
/\b(?:assert(?!ion|ing|ed\b|s\b)|expect(?!ed\b|ation)|refute)[a-zA-Z0-9_$]*\b|`\s*should[a-zA-Z0-9_$]*\s*`|\b(?:should|must|verify|ensure|confirm)[a-zA-Z0-9_$]*\s*[(!]|\.\s*(?:should|to|to_not|not_to|must)\b|\bBOOST_[A-Z_]+\s*\(|\b[A-Z]+_(?:EQ|NE|TRUE|FALSE|THAT)\s*\(/;
|
|
2043
2186
|
const isCommentLine = (str) => /^\s*(?:\/\/|\/\*|\*|#|--|;)/.test(str);
|
|
2044
2187
|
|
|
2188
|
+
/** Book-keeping only: what this run looked at, before deciding anything. */
|
|
2189
|
+
const countExamined = (stats, text) => {
|
|
2190
|
+
if (!text.trim() || isCommentLine(text)) return;
|
|
2191
|
+
stats.examined++;
|
|
2192
|
+
if (ASSERTION_PATTERN.test(text)) stats.recognised++;
|
|
2193
|
+
else if (ASSERTION_SHAPED.test(text) && stats.unreadable.length < 5) stats.unreadable.push(text.trim().slice(0, 120));
|
|
2194
|
+
};
|
|
2195
|
+
|
|
2045
2196
|
const fileAssertions = new Map();
|
|
2046
2197
|
let pendingHunk = false;
|
|
2047
2198
|
|
|
@@ -2077,7 +2228,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2077
2228
|
}
|
|
2078
2229
|
|
|
2079
2230
|
if (!fileAssertions.has(currentFile)) {
|
|
2080
|
-
fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [] });
|
|
2231
|
+
fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [], examined: 0, recognised: 0, unreadable: [] });
|
|
2081
2232
|
}
|
|
2082
2233
|
const fileStats = fileAssertions.get(currentFile);
|
|
2083
2234
|
if (pendingHunk) {
|
|
@@ -2089,6 +2240,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2089
2240
|
if (line.startsWith("-") && !line.startsWith("---")) {
|
|
2090
2241
|
const deletedText = line.slice(1);
|
|
2091
2242
|
if (hunk) hunk.lines.push({ kind: "-", text: deletedText, oldNo: currentOldLineNo, newNo: null });
|
|
2243
|
+
countExamined(fileStats, deletedText);
|
|
2092
2244
|
if (!isCommentLine(deletedText) && ASSERTION_PATTERN.test(deletedText)) {
|
|
2093
2245
|
fileStats.removed.push({ line: currentOldLineNo, text: deletedText });
|
|
2094
2246
|
if (isSpecificAssertion(deletedText)) {
|
|
@@ -2099,6 +2251,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2099
2251
|
} else if (line.startsWith("+") && !line.startsWith("+++")) {
|
|
2100
2252
|
const addedText = line.slice(1);
|
|
2101
2253
|
if (hunk) hunk.lines.push({ kind: "+", text: addedText, oldNo: null, newNo: currentNewLineNo });
|
|
2254
|
+
countExamined(fileStats, addedText);
|
|
2102
2255
|
let isVacuous = false;
|
|
2103
2256
|
|
|
2104
2257
|
// Check skip injections
|
|
@@ -2225,17 +2378,44 @@ export function checkTestTampering(diffOrText = "", options = {}) {
|
|
|
2225
2378
|
// `ok: true` from a guard that looked at everything and approved it. That
|
|
2226
2379
|
// ambiguity is how a substring bug in the file classifier switched this
|
|
2227
2380
|
// entire guard off for the standard pytest, Rust and RSpec layouts while
|
|
2228
|
-
// every signal stayed green.
|
|
2229
|
-
//
|
|
2230
|
-
//
|
|
2231
|
-
//
|
|
2232
|
-
|
|
2381
|
+
// every signal stayed green.
|
|
2382
|
+
//
|
|
2383
|
+
// Counting *files* was not enough. A JUnit diff that rewrote an expected
|
|
2384
|
+
// value produced `inputsSeen: 1` and a clean PASS while not one assertion
|
|
2385
|
+
// in it had been recognised — the same ambiguity, one level down, inside
|
|
2386
|
+
// the mechanism built to remove it. So the denominator is now the thing
|
|
2387
|
+
// the rules actually consume: lines examined, and of those, assertions
|
|
2388
|
+
// understood. `UNREADABLE` is the state that has no business being silent
|
|
2389
|
+
// — assertion-shaped lines were present and none of them parsed, which
|
|
2390
|
+
// means this repository speaks a dialect the guard does not.
|
|
2391
|
+
let examined = 0;
|
|
2392
|
+
let assertionsSeen = 0;
|
|
2393
|
+
const unreadable = [];
|
|
2394
|
+
for (const [file, stats] of fileAssertions.entries()) {
|
|
2395
|
+
examined += stats.examined;
|
|
2396
|
+
assertionsSeen += stats.recognised;
|
|
2397
|
+
if (stats.unreadable.length > 0) {
|
|
2398
|
+
unreadable.push({ file, count: stats.unreadable.length, samples: stats.unreadable.slice(0, 3) });
|
|
2399
|
+
}
|
|
2400
|
+
}
|
|
2401
|
+
|
|
2402
|
+
const status =
|
|
2403
|
+
reported.length > 0
|
|
2404
|
+
? "FAIL"
|
|
2405
|
+
: assertionsSeen === 0 && unreadable.length > 0
|
|
2406
|
+
? "UNREADABLE"
|
|
2407
|
+
: examined > 0
|
|
2408
|
+
? "PASS"
|
|
2409
|
+
: "NOT_APPLICABLE";
|
|
2233
2410
|
|
|
2234
2411
|
return {
|
|
2235
2412
|
ok: reported.length === 0,
|
|
2236
2413
|
violations: reported,
|
|
2237
|
-
inputsSeen,
|
|
2238
|
-
|
|
2414
|
+
inputsSeen: examined,
|
|
2415
|
+
filesSeen: fileAssertions.size,
|
|
2416
|
+
assertionsSeen,
|
|
2417
|
+
unreadable,
|
|
2418
|
+
status,
|
|
2239
2419
|
};
|
|
2240
2420
|
}
|
|
2241
2421
|
|
package/src/stack-detector.mjs
CHANGED
|
@@ -188,6 +188,56 @@ export function detectEdgeRuntime(projectRoot = process.cwd()) {
|
|
|
188
188
|
/**
|
|
189
189
|
* Detects 24+ polyglot stacks and container environments.
|
|
190
190
|
*/
|
|
191
|
+
/**
|
|
192
|
+
* Test commands worth trying, best first, when the detected one does not run.
|
|
193
|
+
*
|
|
194
|
+
* `init` probes the command it picked. On a repository whose Makefile
|
|
195
|
+
* declares a `test` target that needs a build environment the machine does
|
|
196
|
+
* not have, that probe failed, printed `Oracle verification probe failed`,
|
|
197
|
+
* and the wizard wrote the broken command into the config anyway — in a
|
|
198
|
+
* repository where `pytest` was on PATH and all 360 tests passed in 1.3s.
|
|
199
|
+
* Measuring something and then ignoring the measurement is worse than not
|
|
200
|
+
* measuring: it produces a hard red on day one, which is how a user learns
|
|
201
|
+
* the gate is broken and turns it off.
|
|
202
|
+
*
|
|
203
|
+
* Kept deliberately generic — a per-ecosystem convention, never a per-project
|
|
204
|
+
* or per-provider guess.
|
|
205
|
+
*
|
|
206
|
+
* @param {string} root
|
|
207
|
+
* @param {string} [detected] - the command detection chose; always first.
|
|
208
|
+
* @returns {string[]} ordered, de-duplicated candidates
|
|
209
|
+
*/
|
|
210
|
+
export function oracleCandidates(root = process.cwd(), detected = "") {
|
|
211
|
+
const out = [];
|
|
212
|
+
const push = (c) => {
|
|
213
|
+
const v = (c || "").trim();
|
|
214
|
+
if (v && !out.includes(v) && !isPlaceholderTestScript(v)) out.push(v);
|
|
215
|
+
};
|
|
216
|
+
const has = (f) => existsSync(join(root, f));
|
|
217
|
+
|
|
218
|
+
push(detected);
|
|
219
|
+
|
|
220
|
+
if (has("package.json")) {
|
|
221
|
+
try {
|
|
222
|
+
const pkg = JSON.parse(readFileSync(join(root, "package.json"), "utf-8"));
|
|
223
|
+
if (pkg.scripts?.test && !isPlaceholderTestScript(pkg.scripts.test)) push("npm test");
|
|
224
|
+
} catch (_) {}
|
|
225
|
+
}
|
|
226
|
+
if (has("pytest.ini") || has("pyproject.toml") || has("setup.py") || has("tox.ini") || has("setup.cfg")) {
|
|
227
|
+
push(pytestCmd());
|
|
228
|
+
}
|
|
229
|
+
if (has("Cargo.toml")) push("cargo test");
|
|
230
|
+
if (has("go.mod")) push("go test ./...");
|
|
231
|
+
if (has("Gemfile")) push("bundle exec rspec");
|
|
232
|
+
if (has("composer.json")) push("./vendor/bin/phpunit");
|
|
233
|
+
if (has("pom.xml")) push("mvn -q test");
|
|
234
|
+
if (has("build.gradle") || has("build.gradle.kts")) push("./gradlew test");
|
|
235
|
+
if (has("pubspec.yaml")) push("dart test");
|
|
236
|
+
if (has("Package.swift")) push("swift test");
|
|
237
|
+
|
|
238
|
+
return out;
|
|
239
|
+
}
|
|
240
|
+
|
|
191
241
|
export function detectPolyglotStack(projectRoot = process.cwd()) {
|
|
192
242
|
const edgeInfo = detectEdgeRuntime(projectRoot);
|
|
193
243
|
const isDevcontainer = existsSync(join(projectRoot, ".devcontainer", "devcontainer.json"));
|
package/src/wizard-init.mjs
CHANGED
|
@@ -3,7 +3,7 @@ import { join } from "node:path";
|
|
|
3
3
|
import { parseYaml, TIER_PRESETS, VENDOR_TIERS, FALLBACK_TIER } from "./config.mjs";
|
|
4
4
|
import { suggestProvider, detectAvailableProviders } from "./provider-readiness.mjs";
|
|
5
5
|
import { detectDefaultBranch } from "./git.mjs";
|
|
6
|
-
import { resolveWorkspaceBoundary } from "./stack-detector.mjs";
|
|
6
|
+
import { resolveWorkspaceBoundary, oracleCandidates } from "./stack-detector.mjs";
|
|
7
7
|
import { PROFILE_NAMES, PROFILE_DESCRIPTIONS } from "./profiles.mjs";
|
|
8
8
|
import { detectStackOracles, runVerificationProbe } from "./wizard-oracle.mjs";
|
|
9
9
|
import { select, multiSelect, input, confirm, spinner, isTTY } from "./tui.mjs";
|
|
@@ -302,6 +302,51 @@ export function loadPresets(root = process.cwd()) {
|
|
|
302
302
|
* @param {object} [options]
|
|
303
303
|
* @returns {Promise<{ ok: boolean, configPath: string, plan: object }>}
|
|
304
304
|
*/
|
|
305
|
+
/**
|
|
306
|
+
* Probe the chosen test command, and take detection's next choice if it fails.
|
|
307
|
+
*
|
|
308
|
+
* Runs on the non-interactive path too. `--yes` means "do not ask me", not
|
|
309
|
+
* "do not check" — and the user who is not watching is exactly the one who
|
|
310
|
+
* cannot notice that the command written into their config does not run.
|
|
311
|
+
* Before this, the probe lived inside the interactive branch, so
|
|
312
|
+
* `agentctl init --yes` wrote `make test` into a repository where `make test`
|
|
313
|
+
* exits 2 and `npm test` passes, and the first gate run was a hard red.
|
|
314
|
+
*
|
|
315
|
+
* @returns {Promise<string>} the command to save
|
|
316
|
+
*/
|
|
317
|
+
async function resolveRunnableOracle(root, testCmd, options = {}) {
|
|
318
|
+
if (!testCmd) return testCmd;
|
|
319
|
+
const probeSp = spinner(`Probing oracle: ${testCmd}`, options);
|
|
320
|
+
const probeRes = await runVerificationProbe(testCmd, root);
|
|
321
|
+
if (probeRes.ok) {
|
|
322
|
+
probeSp.stop(`Oracle verified successfully (${probeRes.durationMs}ms)`);
|
|
323
|
+
return testCmd;
|
|
324
|
+
}
|
|
325
|
+
probeSp.fail(`Oracle verification probe failed (Exit ${probeRes.code})`);
|
|
326
|
+
|
|
327
|
+
const alternates = oracleCandidates(root, testCmd).filter((c) => c !== testCmd).slice(0, 3);
|
|
328
|
+
for (const cand of alternates) {
|
|
329
|
+
const altSp = spinner(`Trying ${cand}`, options);
|
|
330
|
+
const altRes = await runVerificationProbe(cand, root);
|
|
331
|
+
if (altRes.ok) {
|
|
332
|
+
altSp.stop(`${cand} runs here (${altRes.durationMs}ms) — using it instead`);
|
|
333
|
+
return cand;
|
|
334
|
+
}
|
|
335
|
+
altSp.fail(`${cand} also failed (Exit ${altRes.code})`);
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
// Nothing runs. Say so in terms the user can act on, rather than leaving a
|
|
339
|
+
// failed spinner to scroll past and a broken command in the config.
|
|
340
|
+
const out = options.stdout || process.stdout;
|
|
341
|
+
out.write("\n");
|
|
342
|
+
out.write(" \u26a0\ufe0f No test command could be run in this environment.\n");
|
|
343
|
+
out.write(` Keeping "${testCmd}" \u2014 the gate will fail until it runs here.\n`);
|
|
344
|
+
out.write(" Point verify.test in .agent/config.yml at a command that works,\n");
|
|
345
|
+
out.write(" or, if this repository genuinely has no suite, set\n");
|
|
346
|
+
out.write(" verify.required: false deliberately rather than by accident.\n\n");
|
|
347
|
+
return testCmd;
|
|
348
|
+
}
|
|
349
|
+
|
|
305
350
|
export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
306
351
|
const interactive = options.interactive !== false && isTTY(options.stdin || process.stdin);
|
|
307
352
|
|
|
@@ -322,6 +367,7 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
322
367
|
let selectedProvider = options.provider || existingConfig.provider;
|
|
323
368
|
let selectedProfile = options.profile || existingConfig.verify?.profile;
|
|
324
369
|
let testCmd = options.testCmd;
|
|
370
|
+
let probeInteractive = null;
|
|
325
371
|
let buildCmd = options.buildCmd;
|
|
326
372
|
let selectedPresets = options.presets;
|
|
327
373
|
|
|
@@ -402,16 +448,20 @@ export async function runInitWizard(root = process.cwd(), options = {}) {
|
|
|
402
448
|
|
|
403
449
|
selectedPresets = await multiSelect(presetOptions, "Select Autonomous Workflows to Enable", options);
|
|
404
450
|
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
|
|
411
|
-
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
|
|
451
|
+
probeInteractive = await confirm("Run verification probe on test command before saving?", true, options);
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
// The probe runs whether or not anyone was asked: interactive users can
|
|
455
|
+
// decline it, but silence from `--yes` is not a decline.
|
|
456
|
+
if (probeInteractive !== false && options.probe !== false) {
|
|
457
|
+
// Resolve the command the way planInit will, or there is nothing to
|
|
458
|
+
// probe: on the headless path `testCmd` stays undefined until planInit
|
|
459
|
+
// fills it in from detection, so the probe silently examined nothing —
|
|
460
|
+
// the exact fail-open shape this project keeps finding in itself.
|
|
461
|
+
const effective =
|
|
462
|
+
testCmd || existingConfig.verify?.test || detectStackOracles(root)?.candidates?.testCmd || "";
|
|
463
|
+
const adopted = await resolveRunnableOracle(root, effective, options);
|
|
464
|
+
if (adopted) testCmd = adopted;
|
|
415
465
|
}
|
|
416
466
|
|
|
417
467
|
// `...options` first, for the same reason as in wizard-task.mjs: spreading it
|