jules-orchestrator-kit 0.57.0 → 0.59.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/security.mjs CHANGED
@@ -1,4 +1,5 @@
1
1
  import { openSync, readFileSync, writeSync, fsyncSync, closeSync, renameSync, realpathSync, existsSync, lstatSync, unlinkSync } from "node:fs";
2
+ import { isTestPath } from "./test-paths.mjs";
2
3
  import { dirname, join, basename } from "node:path";
3
4
  import { randomBytes } from "node:crypto";
4
5
  import { canonicalizePath, isWindowsAbsolutePath } from "./config.mjs";
@@ -838,9 +839,56 @@ function splitDiffByFile(diffText) {
838
839
 
839
840
  const CANDIDATE_TOKEN_REGEX = /[A-Za-z0-9_-]{24,}/g;
840
841
 
842
+ /** A whole data: URI, base64-encoded or not. Its payload is not a credential. */
843
+ const DATA_URI_TOKEN = /\bdata:[a-z0-9.+-]+\/[a-z0-9.+-]*(?:;[a-z0-9.+=-]+)*,[^\s"'`<>)\]}]*/gi;
844
+
845
+ /** A subresource-integrity digest. High entropy by construction, public by design. */
846
+ const INTEGRITY_TOKEN = /\bsha(?:256|384|512)-[A-Za-z0-9+/=]+/gi;
847
+
848
+ /** A URL, from its scheme to the first character that cannot be part of one. */
849
+ const URL_TOKEN = /\b[a-z][a-z0-9+.-]*:\/\/[^\s"'`<>)\]},;]+/gi;
850
+
851
+ /**
852
+ * Remove from a line the noise that made URLs worth ignoring, and keep the
853
+ * parts of a URL that can carry a credential.
854
+ *
855
+ * This used to be `if (rawLine.includes("://")) continue;` — one substring
856
+ * anywhere on the line switched off entropy analysis for the entire line. So
857
+ * the scanner caught a bare 32-character key and let the identical key through
858
+ * the moment a comment carrying any http link sat beside it. An agent does not
859
+ * need to know why that works to stumble into it; a fetch call and its endpoint
860
+ * on one line is ordinary code.
861
+ *
862
+ * What actually justified the skip is narrower: a CDN path segment or an npm
863
+ * integrity hash looks exactly like a secret and is neither. Those are dropped
864
+ * here. A URL's userinfo and its query values are the opposite — `?api_key=…`
865
+ * and `//user:password@host` are where credentials genuinely hide — so they
866
+ * are carried over and scanned on their own.
867
+ */
868
+ function stripEntropyNoise(rawLine) {
869
+ const carried = [];
870
+ let line = rawLine.replace(DATA_URI_TOKEN, " ").replace(INTEGRITY_TOKEN, " ");
871
+ line = line.replace(URL_TOKEN, (url) => {
872
+ const afterScheme = url.slice(url.indexOf("://") + 3);
873
+ const authority = afterScheme.split(/[/?#]/)[0];
874
+ const at = authority.lastIndexOf("@");
875
+ if (at > 0) carried.push(authority.slice(0, at));
876
+ const q = url.indexOf("?");
877
+ if (q !== -1) {
878
+ for (const pair of url.slice(q + 1).split(/[&;#]/)) {
879
+ const eq = pair.indexOf("=");
880
+ if (eq !== -1) carried.push(pair.slice(eq + 1));
881
+ }
882
+ }
883
+ return " ";
884
+ });
885
+ return carried.length ? `${line} ${carried.join(" ")}` : line;
886
+ }
887
+
841
888
  /**
842
889
  * Checks for high-entropy continuous tokens (>= 24 chars, entropy > 4.5) on added lines.
843
- * Filters out URLs, SRI hashes (sha512-, sha256-, sha384-), and lockfiles to eliminate false positives.
890
+ * Strips URLs, data: URIs, SRI hashes (sha512-, sha256-, sha384-) and skips lockfiles
891
+ * to eliminate false positives.
844
892
  *
845
893
  * @param {string} text - Text to scan
846
894
  * @param {string|null} [file=null] - File path associated with the text
@@ -863,19 +911,11 @@ export function hasHighEntropyToken(text = "", file = null) {
863
911
 
864
912
  const lines = text.split("\n");
865
913
  for (const rawLine of lines) {
866
- if (
867
- rawLine.includes("://") ||
868
- rawLine.includes("data:image/") ||
869
- rawLine.includes("sha512-") ||
870
- rawLine.includes("sha256-") ||
871
- rawLine.includes("sha384-")
872
- ) {
873
- continue;
874
- }
914
+ const line = stripEntropyNoise(rawLine);
875
915
 
876
916
  CANDIDATE_TOKEN_REGEX.lastIndex = 0;
877
917
  let match;
878
- while ((match = CANDIDATE_TOKEN_REGEX.exec(rawLine)) !== null) {
918
+ while ((match = CANDIDATE_TOKEN_REGEX.exec(line)) !== null) {
879
919
  const token = match[0];
880
920
  if (token.startsWith("sha512-") || token.startsWith("sha256-")) continue;
881
921
  if (token.length >= 24) {
@@ -1062,6 +1102,675 @@ function locateFindingLine(lines, type, file = null) {
1062
1102
  return null;
1063
1103
  }
1064
1104
 
1105
+ // ---------------------------------------------------------------------------
1106
+ // Statement-level expectation-rewrite detection (multi-line aware)
1107
+ // ---------------------------------------------------------------------------
1108
+ //
1109
+ // The original pairing ran on physical lines. That caught
1110
+ // `assert.equal(add(1, 2), 3);` becoming `assert.equal(add(1, 2), -1);`, but
1111
+ // the same edit walked straight through the moment it was wrapped across
1112
+ // lines — which every formatter does the day a line runs long, and which an
1113
+ // agent doing an ordinary reformat does on its own:
1114
+ //
1115
+ // -assert.equal(
1116
+ // - add(1, 2),
1117
+ // - 3
1118
+ // -);
1119
+ // +assert.equal(
1120
+ // + add(1, 2),
1121
+ // + -1
1122
+ // +);
1123
+ //
1124
+ // The value lives on a line that carries no assertion keyword, so neither
1125
+ // side ever paired, and the suite went from checking that addition works to
1126
+ // certifying that it is broken. The statement, not the line, is the unit an
1127
+ // agent rewrites, so the pairing now runs on reassembled statements: a run
1128
+ // of physical lines joined while its delimiters are unbalanced, one of its
1129
+ // strings or comments is still open, a Python line-continuation is pending,
1130
+ // or the next line cannot start a statement of its own. The hunk's context
1131
+ // lines belong to both images and are what make the reassembly possible;
1132
+ // when they are absent (a zero-context diff) the only pair that can survive
1133
+ // is the one where each image is a single fragment, and that pair is taken
1134
+ // too, requiring a literal placeholder so a code change cannot masquerade as
1135
+ // a value change.
1136
+ //
1137
+ // The pairing rule is unchanged in spirit: the two sides must be the *same*
1138
+ // assertion — identical once every literal is blanked out — with different
1139
+ // values. That does not distinguish an attack from a deliberate change of
1140
+ // spec; nothing can, from a diff alone. This reports rather than decides,
1141
+ // and `--allow-test-modifications` is the answer when the new expectation is
1142
+ // the correct one.
1143
+
1144
+ // An assertion that states a *specific* expected value. Counting assertions
1145
+ // alone let a test be gutted while looking untouched: swapping
1146
+ // `assert.strictEqual(add(2,3), 5)` for `assert.ok(add(2,3) !== undefined)`
1147
+ // removes one and adds one, so `removed > added` stayed false and the guard
1148
+ // said nothing — while the suite stopped checking the answer.
1149
+ //
1150
+ // The `expect` argument span is a bounded lazy match rather than `[^)]*` so
1151
+ // that a call split across lines with a nested call in its arguments
1152
+ // (`expect(\n formatInvoice(bill)\n).toBe(…`) still recognises the chain.
1153
+ // The bound is a guess: an argument list longer than 240 characters is
1154
+ // rarer than a missed chain.
1155
+ const SPECIFIC_ASSERTION = new RegExp(
1156
+ [
1157
+ "\\bassert(?:\\.strict)?\\.?(?:strictEqual|deepStrictEqual|deepEqual|notStrictEqual|notDeepStrictEqual|equal|notEqual|match|doesNotMatch|throws|rejects|doesNotThrow)\\s*\\(",
1158
+ "\\bexpect\\s*\\([\\s\\S]{0,240}?\\)\\s*\\.(?:toBe|toEqual|toStrictEqual|toMatch|toMatchObject|toContain|toHaveBeenCalledWith|toThrow|toHaveLength|toBeCloseTo)\\s*\\(",
1159
+ "\\bassert\\.(?:equals|deepEquals|include|lengthOf)\\s*\\(",
1160
+ "assert_eq!|assert_ne!",
1161
+ "\\bt\\.(?:Errorf|Fatalf)\\s*\\(",
1162
+ "\\brequire\\.(?:Equal|NotEqual|Len|Contains|Error|NoError)\\s*\\(",
1163
+ ].join("|"),
1164
+ "i"
1165
+ );
1166
+ const isSpecificAssertion = (str) => SPECIFIC_ASSERTION.test(str);
1167
+
1168
+ /**
1169
+ * An assertion with every literal value replaced by a placeholder.
1170
+ *
1171
+ * Two lines that normalize to the same string are the same assertion about
1172
+ * the same expression; whatever differs between them is a value.
1173
+ *
1174
+ * The number form covers hex, octal, binary, underscores and exponents: the
1175
+ * original decimal-only regex never blanked `0xFF`, so an expectation
1176
+ * rewritten from `0xFF` to `0xFE` normalized to two *different* shapes and
1177
+ * the pair was never formed.
1178
+ */
1179
+ const blankLiterals = (str) =>
1180
+ str
1181
+ .replace(/(['"`])(?:\\.|(?!\1)[^\\])*\1/g, "\u0000S")
1182
+ // The sign belongs to the literal: without it `3` and `-1` normalized to
1183
+ // different shapes and the rewritten expectation was never paired.
1184
+ .replace(
1185
+ /(?<![\w$])(?:0[xX][0-9a-fA-F_]+|0[bB][01_]+|0[oO][0-7_]+|-?\d[\d_]*(?:\.[\d_]+)?(?:[eE][+-]?\d+)?)/g,
1186
+ "\u0000N"
1187
+ )
1188
+ .replace(/\b(?:true|false|null|undefined|None|True|False|nil)\b/g, "\u0000B")
1189
+ // Whitespace is dropped, not collapsed: the shape is compared for
1190
+ // equality only, and a reformatted statement must normalize to the same
1191
+ // shape as the original — ` <N> );` and ` <N>);` are the same assertion.
1192
+ .replace(/\s+/g, "");
1193
+
1194
+ // The test languages the gate runs over. The scanner below is written for
1195
+ // these four and nothing else; an unrecognised extension falls back to `js`,
1196
+ // which is the strictest of the four for line joining.
1197
+ const TEST_LANG_BY_EXT = new Map([
1198
+ [".js", "js"], [".mjs", "js"], [".cjs", "js"], [".jsx", "js"],
1199
+ [".ts", "js"], [".mts", "js"], [".cts", "js"], [".tsx", "js"],
1200
+ [".py", "python"], [".pyi", "python"],
1201
+ [".go", "go"],
1202
+ [".rs", "rust"],
1203
+ ]);
1204
+
1205
+ function langForTestFile(file) {
1206
+ const n = String(file || "").toLowerCase();
1207
+ const dot = n.lastIndexOf(".");
1208
+ if (dot === -1) return "js";
1209
+ return TEST_LANG_BY_EXT.get(n.slice(dot)) || "js";
1210
+ }
1211
+
1212
+ function freshScanState() {
1213
+ // str: the open string, or null.
1214
+ // q the quote character
1215
+ // tri Python triple-quoted
1216
+ // raw raw string, no escapes (Go backtick)
1217
+ // rawHashes Rust raw string r#"…"#: terminator is " plus that many #
1218
+ // block: depth of an open /* … */ (nested only in Rust)
1219
+ // accDelta: counted delimiters still open in the current statement
1220
+ // specialStack: counted depth recorded at each non-joining call (see below)
1221
+ return { str: null, block: 0, accDelta: 0, specialStack: [] };
1222
+ }
1223
+
1224
+ // A call whose opening paren must not join lines: the test name is not the
1225
+ // expectation. Without this, `it("old name", () => { expect(f()).toBe(3); })`
1226
+ // would pair against its renamed copy, because a name is a string and so
1227
+ // blanks to the same placeholder as any value change would. The closer of a
1228
+ // non-joining paren is recognised by the depth it was opened at, so the
1229
+ // running balance stays exact.
1230
+ const NON_JOINING_CALL = /\b(?:it|test|describe|context)\s*\($|\bt\.Run\s*\($/;
1231
+
1232
+ /**
1233
+ * Scan one physical line of source.
1234
+ *
1235
+ * Returns { delta, trailingBackslash }. `delta` is the net number of
1236
+ * still-open ( [ { delimiters outside strings and comments; a statement
1237
+ * continues to the next physical line while it is positive, while a string
1238
+ * or block comment is open (tracked on `state`), or on a Python line
1239
+ * continuation.
1240
+ *
1241
+ * This is not a parser, and the approximations are deliberate: JS regex
1242
+ * literals are detected with a one-token look-behind (a `/` that cannot
1243
+ * follow an identifier, number, `)` or `]` starts one), template
1244
+ * interpolation is treated as opaque string content, and Rust lifetimes are
1245
+ * told apart from char literals by shape alone. `stripComments` below
1246
+ * walks the same constructs, so the two must stay in lock-step.
1247
+ */
1248
+ function scanSourceLine(text, lang, state) {
1249
+ let delta = 0;
1250
+ let lastSig = "\n";
1251
+ const n = text.length;
1252
+ let i = 0;
1253
+
1254
+ while (i < n) {
1255
+ const c = text[i];
1256
+ const c2 = i + 1 < n ? text[i + 1] : "";
1257
+
1258
+ if (state.str) {
1259
+ const s = state.str;
1260
+ let closed = false;
1261
+ if (s.rawHashes !== undefined) {
1262
+ if (c === '"') {
1263
+ let j = i + 1;
1264
+ let h = 0;
1265
+ while (j < n && text[j] === "#") { h++; j++; }
1266
+ if (h >= s.rawHashes) { i = j; closed = true; }
1267
+ }
1268
+ } else if (s.raw) {
1269
+ closed = c === s.q;
1270
+ } else if (c === "\\") {
1271
+ i += s.tri ? 1 : 2;
1272
+ continue;
1273
+ } else if (c === s.q) {
1274
+ if (s.tri) {
1275
+ if (text[i + 1] === s.q && text[i + 2] === s.q) { i += 3; closed = true; }
1276
+ else { i += 1; }
1277
+ } else {
1278
+ i += 1;
1279
+ closed = true;
1280
+ }
1281
+ }
1282
+ if (closed) { state.str = null; lastSig = s.q; continue; }
1283
+ i += 1;
1284
+ continue;
1285
+ }
1286
+
1287
+ if (state.block > 0) {
1288
+ if (c === "/" && c2 === "*") {
1289
+ if (lang === "rust") state.block += 1;
1290
+ i += 2;
1291
+ continue;
1292
+ }
1293
+ if (c === "*" && c2 === "/") {
1294
+ state.block -= 1;
1295
+ i += 2;
1296
+ lastSig = "/";
1297
+ continue;
1298
+ }
1299
+ i += 1;
1300
+ continue;
1301
+ }
1302
+
1303
+ // A line comment ends the line.
1304
+ if (c === "/" && c2 === "/") break;
1305
+ if (lang === "python" && c === "#") break;
1306
+
1307
+ if (c === "/" && c2 === "*") {
1308
+ state.block = 1;
1309
+ i += 2;
1310
+ continue;
1311
+ }
1312
+
1313
+ // JS regex literal, best effort. Delimiters inside are not counted.
1314
+ if ((lang === "js" || lang === "ts") && c === "/" && !/[\w$)\]}]/.test(lastSig)) {
1315
+ i += 1;
1316
+ let inClass = false;
1317
+ while (i < n) {
1318
+ const rc = text[i];
1319
+ if (rc === "\\") { i += 2; continue; }
1320
+ if (rc === "[") inClass = true;
1321
+ else if (rc === "]") inClass = false;
1322
+ else if (rc === "/" && !inClass) { i += 1; break; }
1323
+ i += 1;
1324
+ }
1325
+ while (i < n && /[a-z]/i.test(text[i])) i += 1; // flags
1326
+ lastSig = "/";
1327
+ continue;
1328
+ }
1329
+
1330
+ if (c === '"' || c === "'" || (lang === "go" && c === "`")) {
1331
+ if (lang === "python" && c2 === c && text[i + 2] === c) {
1332
+ state.str = { q: c, tri: true };
1333
+ i += 3;
1334
+ } else if (lang === "rust" && c === '"') {
1335
+ if (i > 0 && text[i - 1] === "r") {
1336
+ let j = i - 1;
1337
+ let h = 0;
1338
+ while (j >= 1 && text[j - 1] === "#") { h++; j--; }
1339
+ state.str = { q: '"', rawHashes: h };
1340
+ i += 1;
1341
+ } else {
1342
+ state.str = { q: c };
1343
+ i += 1;
1344
+ }
1345
+ } else if (lang === "rust" && c === "'") {
1346
+ // A char literal is 'X' or '\X' within four characters; anything
1347
+ // else starting with a quote is a lifetime and only the quote is
1348
+ // skipped, or the next line would see a string that never closed.
1349
+ if (c2 === "\\") {
1350
+ const end = text.indexOf("'", i + 2);
1351
+ if (end !== -1 && end - i <= 4) { i = end + 1; lastSig = "'"; continue; }
1352
+ } else if (text[i + 2] === "'" && c2 !== "'") {
1353
+ i += 3;
1354
+ lastSig = "'";
1355
+ continue;
1356
+ }
1357
+ i += 1;
1358
+ continue;
1359
+ } else {
1360
+ state.str = { q: c };
1361
+ i += 1;
1362
+ }
1363
+ lastSig = c;
1364
+ continue;
1365
+ }
1366
+
1367
+ // A backslash on the very last character is a Python line continuation.
1368
+ if (c === "\\" && i + 1 === n) {
1369
+ return { delta, trailingBackslash: true };
1370
+ }
1371
+
1372
+ if (c === "(") {
1373
+ // `before` ends with the paren itself; NON_JOINING_CALL matches on it.
1374
+ const before = text.slice(0, i + 1).replace(/\s+$/, "");
1375
+ if (NON_JOINING_CALL.test(before)) state.specialStack.push(state.accDelta);
1376
+ else { delta += 1; state.accDelta += 1; }
1377
+ lastSig = c;
1378
+ i += 1;
1379
+ continue;
1380
+ }
1381
+ if (c === "[") { delta += 1; state.accDelta += 1; lastSig = c; i += 1; continue; }
1382
+ if (c === "{") {
1383
+ // Go joins braces: the `if got != want { t.Errorf(…) }` block is the
1384
+ // idiomatic Go assertion, and the value lives on its first line. The
1385
+ // other three languages get no brace joining, so a rename of a test
1386
+ // inside a block cannot pair as a value change on its own.
1387
+ if (lang === "go") { delta += 1; state.accDelta += 1; }
1388
+ lastSig = c;
1389
+ i += 1;
1390
+ continue;
1391
+ }
1392
+ if (c === ")") {
1393
+ const top = state.specialStack[state.specialStack.length - 1];
1394
+ if (top === state.accDelta) state.specialStack.pop();
1395
+ else { delta -= 1; state.accDelta -= 1; }
1396
+ lastSig = c;
1397
+ i += 1;
1398
+ continue;
1399
+ }
1400
+ if (c === "]") { delta -= 1; state.accDelta -= 1; lastSig = c; i += 1; continue; }
1401
+ if (c === "}") {
1402
+ if (lang === "go") { delta -= 1; state.accDelta -= 1; }
1403
+ lastSig = c;
1404
+ i += 1;
1405
+ continue;
1406
+ }
1407
+
1408
+ if (!/\s/.test(c)) lastSig = c;
1409
+ i += 1;
1410
+ }
1411
+
1412
+ return { delta, trailingBackslash: false };
1413
+ }
1414
+
1415
+ /**
1416
+ * The same source with every comment blanked out, strings and line
1417
+ * structure untouched. Shapes and keywords are computed on this text so a
1418
+ * commented-out `expect(…)` cannot make a block an assertion, and a number
1419
+ * changed inside a comment cannot pair as a value change.
1420
+ */
1421
+ function stripComments(text, lang) {
1422
+ const state = freshScanState();
1423
+ return text.split("\n").map((line) => {
1424
+ let out = "";
1425
+ let pending = 0;
1426
+ const copyCode = (to) => {
1427
+ out += line.slice(pending, to);
1428
+ pending = to;
1429
+ };
1430
+ let i = 0;
1431
+ const n = line.length;
1432
+
1433
+ while (i < n) {
1434
+ const c = line[i];
1435
+ const c2 = i + 1 < n ? line[i + 1] : "";
1436
+
1437
+ if (state.block > 0) {
1438
+ if (c === "/" && c2 === "*") {
1439
+ if (lang === "rust") state.block += 1;
1440
+ i += 2;
1441
+ continue;
1442
+ }
1443
+ if (c === "*" && c2 === "/") {
1444
+ state.block -= 1;
1445
+ i += 2;
1446
+ if (state.block === 0) pending = i;
1447
+ continue;
1448
+ }
1449
+ i += 1;
1450
+ continue;
1451
+ }
1452
+
1453
+ if (state.str) {
1454
+ const s = state.str;
1455
+ let closed = false;
1456
+ if (s.rawHashes !== undefined) {
1457
+ if (c === '"') {
1458
+ let j = i + 1;
1459
+ let h = 0;
1460
+ while (j < n && line[j] === "#") { h++; j++; }
1461
+ if (h >= s.rawHashes) { i = j; closed = true; }
1462
+ }
1463
+ } else if (s.raw) {
1464
+ closed = c === s.q;
1465
+ } else if (c === "\\") {
1466
+ i += s.tri ? 1 : 2;
1467
+ continue;
1468
+ } else if (c === s.q) {
1469
+ if (s.tri) {
1470
+ if (line[i + 1] === s.q && line[i + 2] === s.q) { i += 3; closed = true; }
1471
+ else { i += 1; }
1472
+ } else {
1473
+ i += 1;
1474
+ closed = true;
1475
+ }
1476
+ }
1477
+ if (closed) {
1478
+ state.str = null;
1479
+ copyCode(i);
1480
+ continue;
1481
+ }
1482
+ i += 1;
1483
+ continue;
1484
+ }
1485
+
1486
+ if (c === "/" && c2 === "/") { copyCode(i); break; }
1487
+ if (lang === "python" && c === "#") { copyCode(i); break; }
1488
+ if (c === "/" && c2 === "*") {
1489
+ copyCode(i);
1490
+ state.block = 1;
1491
+ i += 2;
1492
+ continue;
1493
+ }
1494
+ if (c === '"' || c === "'" || (lang === "go" && c === "`")) {
1495
+ copyCode(i);
1496
+ if (lang === "python" && c2 === c && line[i + 2] === c) {
1497
+ state.str = { q: c, tri: true };
1498
+ i += 3;
1499
+ } else if (lang === "rust" && c === '"') {
1500
+ if (i > 0 && line[i - 1] === "r") {
1501
+ let j = i - 1;
1502
+ let h = 0;
1503
+ while (j >= 1 && line[j - 1] === "#") { h++; j--; }
1504
+ state.str = { q: '"', rawHashes: h };
1505
+ i += 1;
1506
+ } else {
1507
+ state.str = { q: c };
1508
+ i += 1;
1509
+ }
1510
+ } else if (lang === "rust" && c === "'") {
1511
+ if (c2 === "\\") {
1512
+ const end = line.indexOf("'", i + 2);
1513
+ if (end !== -1 && end - i <= 4) { i = end + 1; copyCode(i); continue; }
1514
+ } else if (line[i + 2] === "'" && c2 !== "'") {
1515
+ i += 3;
1516
+ copyCode(i);
1517
+ continue;
1518
+ }
1519
+ i += 1;
1520
+ copyCode(i);
1521
+ continue;
1522
+ } else {
1523
+ state.str = { q: c };
1524
+ i += 1;
1525
+ }
1526
+ continue;
1527
+ }
1528
+ i += 1;
1529
+ }
1530
+ copyCode(n);
1531
+ return out;
1532
+ }).join("\n");
1533
+ }
1534
+
1535
+ // A line that cannot start a statement of its own continues the previous
1536
+ // statement: a closing delimiter, a member-access, or an operator.
1537
+ const CONTINUATION_START = /^[)\],.]/;
1538
+ const CONTINUATION_OP_START = /^[+\-*/%<>=&|^:]/;
1539
+
1540
+ // A scanner miscount (an unbalanced delimiter inside a regex literal is the
1541
+ // usual cause) must not be able to merge a whole file into one statement,
1542
+ // which would pair *any* literal change anywhere in the file.
1543
+ const MAX_STATEMENT_LINES = 100;
1544
+ const MAX_STATEMENT_CHARS = 12000;
1545
+
1546
+ /**
1547
+ * Reassemble physical lines into statements.
1548
+ *
1549
+ * `sliceLines` is one image of a hunk in file order: context lines plus the
1550
+ * removed (or added) lines. Context lines are ordinary file text; a
1551
+ * statement spans them freely, which is exactly what makes a value edit
1552
+ * inside a formatter-wrapped assertion visible to the pairing.
1553
+ *
1554
+ * @param {Array<{ kind: string, text: string, oldNo: number|null, newNo: number|null }>} sliceLines
1555
+ * @param {string} lang
1556
+ * @returns {Array<{ text: string, firstOld: number|null, lastOld: number|null, firstNew: number|null, lastNew: number|null, removedLines: Array, addedLines: Array }>}
1557
+ */
1558
+ function assembleStatements(sliceLines, lang) {
1559
+ const stmts = [];
1560
+ let cur = null;
1561
+
1562
+ const flush = () => {
1563
+ if (!cur) return;
1564
+ stmts.push({
1565
+ text: cur.lines.join("\n"),
1566
+ firstOld: cur.firstOld,
1567
+ lastOld: cur.lastOld,
1568
+ firstNew: cur.firstNew,
1569
+ lastNew: cur.lastNew,
1570
+ removedLines: cur.removedLines,
1571
+ addedLines: cur.addedLines,
1572
+ });
1573
+ cur = null;
1574
+ };
1575
+
1576
+ for (const L of sliceLines) {
1577
+ const trimmed = L.text.replace(/^\s+/, "");
1578
+ const joins =
1579
+ cur !== null &&
1580
+ (cur.delta > 0 ||
1581
+ cur.state.str !== null ||
1582
+ cur.state.block > 0 ||
1583
+ cur.trailingBackslash ||
1584
+ CONTINUATION_START.test(trimmed) ||
1585
+ CONTINUATION_OP_START.test(trimmed));
1586
+
1587
+ if (
1588
+ joins &&
1589
+ cur.lines.length < MAX_STATEMENT_LINES &&
1590
+ cur.chars + L.text.length + 1 <= MAX_STATEMENT_CHARS
1591
+ ) {
1592
+ cur.lines.push(L.text);
1593
+ cur.chars += L.text.length + 1;
1594
+ const sc = scanSourceLine(L.text, lang, cur.state);
1595
+ cur.delta += sc.delta;
1596
+ cur.trailingBackslash = sc.trailingBackslash;
1597
+ cur.lastOld = L.oldNo;
1598
+ cur.lastNew = L.newNo;
1599
+ if (L.kind === "-") cur.removedLines.push(L);
1600
+ else if (L.kind === "+") cur.addedLines.push(L);
1601
+ } else {
1602
+ flush();
1603
+ const st = freshScanState();
1604
+ const sc = scanSourceLine(L.text, lang, st);
1605
+ cur = {
1606
+ lines: [L.text],
1607
+ chars: L.text.length,
1608
+ firstOld: L.oldNo,
1609
+ lastOld: L.oldNo,
1610
+ firstNew: L.newNo,
1611
+ lastNew: L.newNo,
1612
+ delta: sc.delta,
1613
+ state: st,
1614
+ trailingBackslash: sc.trailingBackslash,
1615
+ removedLines: L.kind === "-" ? [L] : [],
1616
+ addedLines: L.kind === "+" ? [L] : [],
1617
+ };
1618
+ }
1619
+ }
1620
+ flush();
1621
+ return stmts;
1622
+ }
1623
+
1624
+ const hasLiteralPlaceholder = (shape) =>
1625
+ shape.includes("\u0000S") || shape.includes("\u0000N") || shape.includes("\u0000B");
1626
+
1627
+ const collapseWhitespace = (s) => s.replace(/\s+/g, " ").trim();
1628
+ const shorten = (s) => (s.length > 160 ? `${s.slice(0, 157)}…` : s);
1629
+
1630
+ /**
1631
+ * Pair rewritten expectations across the removed and added images of every
1632
+ * hunk of one file, and report each pair.
1633
+ *
1634
+ * A statement is only a candidate when it actually contains a removed
1635
+ * (resp. added) line: a context-only statement is unchanged text on both
1636
+ * sides, and letting it pair would flag a genuinely new assertion that
1637
+ * merely has the same shape as one that stayed put.
1638
+ *
1639
+ * @param {string} file
1640
+ * @param {Array<{ lines: Array }>} hunks
1641
+ * @param {object} stats - per-file stats; the paired physical lines are
1642
+ * spliced out of the count pools so the count-based checks below do not
1643
+ * report the same edit a second time.
1644
+ * @param {Array} violations
1645
+ * @returns {Array<{ r: object, a: object }>} the pairs, for the caller
1646
+ */
1647
+ function detectExpectationRewrites(file, hunks, stats, violations) {
1648
+ const lang = langForTestFile(file);
1649
+ const allPairs = [];
1650
+
1651
+ for (const hunk of hunks) {
1652
+ const oldSlice = [];
1653
+ const newSlice = [];
1654
+ for (const L of hunk.lines) {
1655
+ if (L.kind !== "+") oldSlice.push(L);
1656
+ if (L.kind !== "-") newSlice.push(L);
1657
+ }
1658
+ const oldStmts = assembleStatements(oldSlice, lang);
1659
+ const newStmts = assembleStatements(newSlice, lang);
1660
+
1661
+ // Candidate statements in file order, per image.
1662
+ const oldCands = [];
1663
+ for (const s of oldStmts) {
1664
+ if (s.removedLines.length === 0) continue;
1665
+ const clean = stripComments(s.text, lang);
1666
+ if (!isSpecificAssertion(clean)) continue;
1667
+ oldCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1668
+ }
1669
+ const newCands = [];
1670
+ for (const s of newStmts) {
1671
+ if (s.addedLines.length === 0) continue;
1672
+ const clean = stripComments(s.text, lang);
1673
+ if (!isSpecificAssertion(clean)) continue;
1674
+ newCands.push({ s, shape: blankLiterals(clean), canon: clean.replace(/\s+/g, "") });
1675
+ }
1676
+
1677
+ // The t-th removed candidate of a shape pairs with the t-th added
1678
+ // candidate of the same shape. Position alignment is what keeps a
1679
+ // formatter run over a block of same-shape assertions silent: a greedy
1680
+ // "first different text" pairing would match a re-indented
1681
+ // `assert.equal(f(0), 0)` against its neighbour's value and report a
1682
+ // rewrite that did not happen. It also keeps the pairing linear in the
1683
+ // number of statements, which a 4000-line same-shape table would not
1684
+ // survive as a square of comparisons.
1685
+ const oldByShape = new Map();
1686
+ const newByShape = new Map();
1687
+ for (const c of oldCands) {
1688
+ let arr = oldByShape.get(c.shape);
1689
+ if (!arr) arr = oldByShape.set(c.shape, []).get(c.shape);
1690
+ arr.push(c);
1691
+ }
1692
+ for (const c of newCands) {
1693
+ let arr = newByShape.get(c.shape);
1694
+ if (!arr) arr = newByShape.set(c.shape, []).get(c.shape);
1695
+ arr.push(c);
1696
+ }
1697
+
1698
+ const pairs = [];
1699
+ for (const [shape, olds] of oldByShape) {
1700
+ const news = newByShape.get(shape) || [];
1701
+ const k = Math.min(olds.length, news.length);
1702
+ for (let t = 0; t < k; t++) {
1703
+ if (olds[t].canon !== news[t].canon) {
1704
+ pairs.push({ r: olds[t].s, a: news[t].s });
1705
+ }
1706
+ }
1707
+ }
1708
+
1709
+ // Zero-context hunk: each image is a single fragment and the assertion
1710
+ // keyword may sit outside the hunk entirely. The fragment pair is taken
1711
+ // only when both sides normalize to the same shape *and* that shape
1712
+ // holds a literal — a code change cannot fake it.
1713
+ if (pairs.length === 0 && oldStmts.length === 1 && newStmts.length === 1) {
1714
+ const r = oldStmts[0];
1715
+ const a = newStmts[0];
1716
+ if (r.removedLines.length > 0 && a.addedLines.length > 0) {
1717
+ const sr = blankLiterals(stripComments(r.text, lang));
1718
+ const sa = blankLiterals(stripComments(a.text, lang));
1719
+ if (sr === sa && hasLiteralPlaceholder(sr)) {
1720
+ const cr = stripComments(r.text, lang).replace(/\s+/g, "");
1721
+ const ca = stripComments(a.text, lang).replace(/\s+/g, "");
1722
+ if (cr !== ca) pairs.push({ r, a });
1723
+ }
1724
+ }
1725
+ }
1726
+
1727
+ for (const p of pairs) {
1728
+ allPairs.push(p);
1729
+ const addedNo = p.a.addedLines.length > 0 ? p.a.addedLines[0].newNo : (p.r.removedLines[0] ? p.r.removedLines[0].oldNo : null);
1730
+ violations.push({
1731
+ file,
1732
+ line: addedNo,
1733
+ type: "ASSERTION_EXPECTATION_CHANGED",
1734
+ reason:
1735
+ `Test Tamper Guard: Expected value rewritten in ${file}${addedNo ? `:${addedNo}` : ""} — ` +
1736
+ `"${shorten(collapseWhitespace(p.r.text))}" became "${shorten(collapseWhitespace(p.a.text))}". ` +
1737
+ `A deliberately changed spec looks identical to a test bent to match broken ` +
1738
+ `output, and a diff alone cannot tell the two apart, so this is flagged for ` +
1739
+ `review rather than assumed. If the new expectation is correct, re-run with ` +
1740
+ `--allow-test-modifications (which also silences the skip, vacuous, removal ` +
1741
+ `and weakening checks for this diff).`,
1742
+ });
1743
+
1744
+ // Both sides are accounted for here, so they must not also feed the
1745
+ // count-based checks below — the same line reported twice under two
1746
+ // names tells the operator nothing extra. Consumption is symmetric so
1747
+ // a statement that absorbed two old assertion lines but only one new
1748
+ // one still leaves the surplus to the removal check.
1749
+ const removedMatches = p.r.removedLines.filter(
1750
+ (L) => stats.removed.some((e) => e.line === L.oldNo && e.text === L.text)
1751
+ );
1752
+ const addedMatches = p.a.addedLines.filter(
1753
+ (L) => stats.addedTexts.some((e) => e.line === L.newNo && e.text === L.text)
1754
+ );
1755
+ const take = Math.min(removedMatches.length, addedMatches.length);
1756
+ for (let k = 0; k < take; k++) {
1757
+ const L = removedMatches[k];
1758
+ const idx = stats.removed.findIndex((e) => e.line === L.oldNo && e.text === L.text);
1759
+ if (idx === -1) continue;
1760
+ const entry = stats.removed.splice(idx, 1)[0];
1761
+ const rsIdx = stats.removedSpecific.indexOf(entry);
1762
+ if (rsIdx !== -1) {
1763
+ stats.removedSpecific.splice(rsIdx, 1);
1764
+ stats.addedSpecific--;
1765
+ }
1766
+ stats.added--;
1767
+ }
1768
+ }
1769
+ }
1770
+
1771
+ return allPairs;
1772
+ }
1773
+
1065
1774
  /**
1066
1775
  * Detects test file assertion tampering, weakening, or test skips.
1067
1776
  *
@@ -1081,18 +1790,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1081
1790
  let currentOldLineNo = null;
1082
1791
  let currentNewLineNo = null;
1083
1792
 
1084
- const isTestFile = (f) => {
1085
- if (!f) return false;
1086
- const n = f.replace(/\\/g, "/").toLowerCase();
1087
- return (
1088
- n.includes(".test.") ||
1089
- n.includes(".spec.") ||
1090
- n.includes("_test.") ||
1091
- n.includes("/test/") ||
1092
- n.includes("/tests/") ||
1093
- n.includes("/__tests__/")
1094
- );
1095
- };
1793
+ const isTestFile = isTestPath;
1096
1794
 
1097
1795
  const SKIP_INJECTIONS = [
1098
1796
  { pattern: /\b(?:it|test|describe|context)\.skip\s*\(/i, desc: "Injected test skip (.skip())" },
@@ -1120,25 +1818,8 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1120
1818
  const ASSERTION_PATTERN = /(?:\b(?:assert(?:\.[a-zA-Z0-9_$]+)?|expect|t\.(?:assert|expect|is|equal|true|false|Errorf|Fatalf)|require\.[a-zA-Z0-9_$]+)\b|assert!|assert_eq!|assert_ne!)/i;
1121
1819
  const isCommentLine = (str) => /^\s*(?:\/\/|\/\*|\*|#|--|;)/.test(str);
1122
1820
 
1123
- // An assertion that states a *specific* expected value. Counting assertions
1124
- // alone let a test be gutted while looking untouched: swapping
1125
- // `assert.strictEqual(add(2,3), 5)` for `assert.ok(add(2,3) !== undefined)`
1126
- // removes one and adds one, so `removed > added` stayed false and the guard
1127
- // said nothing — while the suite stopped checking the answer.
1128
- const SPECIFIC_ASSERTION = new RegExp(
1129
- [
1130
- "\\bassert(?:\\.strict)?\\.?(?:strictEqual|deepStrictEqual|deepEqual|notStrictEqual|notDeepStrictEqual|equal|notEqual|match|doesNotMatch|throws|rejects|doesNotThrow)\\s*\\(",
1131
- "\\bexpect\\s*\\([^)]*\\)\\s*\\.(?:toBe|toEqual|toStrictEqual|toMatch|toMatchObject|toContain|toHaveBeenCalledWith|toThrow|toHaveLength|toBeCloseTo)\\s*\\(",
1132
- "\\bassert\\.(?:equals|deepEquals|include|lengthOf)\\s*\\(",
1133
- "assert_eq!|assert_ne!",
1134
- "\\bt\\.(?:Errorf|Fatalf)\\s*\\(",
1135
- "\\brequire\\.(?:Equal|NotEqual|Len|Contains|Error|NoError)\\s*\\(",
1136
- ].join("|"),
1137
- "i"
1138
- );
1139
- const isSpecificAssertion = (str) => SPECIFIC_ASSERTION.test(str);
1140
-
1141
1821
  const fileAssertions = new Map();
1822
+ let pendingHunk = false;
1142
1823
 
1143
1824
  for (let i = 0; i < lines.length; i++) {
1144
1825
  const line = lines[i];
@@ -1154,6 +1835,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1154
1835
  currentFile = target && target !== "/dev/null" ? target : lastOldFile;
1155
1836
  currentOldLineNo = null;
1156
1837
  currentNewLineNo = null;
1838
+ pendingHunk = false;
1157
1839
  continue;
1158
1840
  }
1159
1841
 
@@ -1161,20 +1843,28 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1161
1843
  if (hunkMatch) {
1162
1844
  currentOldLineNo = Number(hunkMatch[1]);
1163
1845
  currentNewLineNo = Number(hunkMatch[2]);
1846
+ pendingHunk = true;
1164
1847
  continue;
1165
1848
  }
1166
1849
 
1167
1850
  if (!currentFile || !isTestFile(currentFile)) {
1851
+ pendingHunk = false;
1168
1852
  continue;
1169
1853
  }
1170
1854
 
1171
1855
  if (!fileAssertions.has(currentFile)) {
1172
- fileAssertions.set(currentFile, { removed: [], added: 0, removedSpecific: [], addedSpecific: 0 });
1856
+ fileAssertions.set(currentFile, { removed: [], added: 0, addedTexts: [], removedSpecific: [], addedSpecific: 0, hunks: [] });
1173
1857
  }
1174
1858
  const fileStats = fileAssertions.get(currentFile);
1859
+ if (pendingHunk) {
1860
+ fileStats.hunks.push({ lines: [] });
1861
+ pendingHunk = false;
1862
+ }
1863
+ const hunk = fileStats.hunks.length > 0 ? fileStats.hunks[fileStats.hunks.length - 1] : null;
1175
1864
 
1176
1865
  if (line.startsWith("-") && !line.startsWith("---")) {
1177
1866
  const deletedText = line.slice(1);
1867
+ if (hunk) hunk.lines.push({ kind: "-", text: deletedText, oldNo: currentOldLineNo, newNo: null });
1178
1868
  if (!isCommentLine(deletedText) && ASSERTION_PATTERN.test(deletedText)) {
1179
1869
  fileStats.removed.push({ line: currentOldLineNo, text: deletedText });
1180
1870
  if (isSpecificAssertion(deletedText)) {
@@ -1184,6 +1874,7 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1184
1874
  if (currentOldLineNo !== null) currentOldLineNo++;
1185
1875
  } else if (line.startsWith("+") && !line.startsWith("+++")) {
1186
1876
  const addedText = line.slice(1);
1877
+ if (hunk) hunk.lines.push({ kind: "+", text: addedText, oldNo: null, newNo: currentNewLineNo });
1187
1878
  let isVacuous = false;
1188
1879
 
1189
1880
  // Check skip injections
@@ -1226,17 +1917,38 @@ export function checkTestTampering(diffOrText = "", options = {}) {
1226
1917
  // If valid non-vacuous, non-commented assertion is added, increment added count
1227
1918
  if (!isVacuous && !isCommented && !isCommentLine(addedText) && ASSERTION_PATTERN.test(addedText)) {
1228
1919
  fileStats.added++;
1920
+ fileStats.addedTexts.push({ line: currentNewLineNo, text: addedText });
1229
1921
  if (isSpecificAssertion(addedText)) fileStats.addedSpecific++;
1230
1922
  }
1231
1923
 
1232
1924
  if (currentNewLineNo !== null) currentNewLineNo++;
1233
1925
  } else if (!line.startsWith("\\")) {
1926
+ // Context line. It is file text in both images, so the statement
1927
+ // assembler needs it to reassemble assertions that span a changed
1928
+ // line; `diff --git`/`index` lines that sneak in here are not
1929
+ // diff body and are not collected.
1930
+ if (hunk && (line.startsWith(" ") || line === "")) {
1931
+ hunk.lines.push({ kind: " ", text: line.slice(1), oldNo: currentOldLineNo, newNo: currentNewLineNo });
1932
+ }
1234
1933
  if (currentOldLineNo !== null) currentOldLineNo++;
1235
1934
  if (currentNewLineNo !== null) currentNewLineNo++;
1236
1935
  }
1237
1936
  }
1238
1937
 
1239
1938
  for (const [file, stats] of fileAssertions.entries()) {
1939
+ // An expectation that was rewritten rather than removed.
1940
+ //
1941
+ // Counting assertions cannot see this one: `assert.equal(add(1,2), 3)`
1942
+ // becoming `assert.equal(add(1,2), -1)` takes one specific assertion out
1943
+ // and puts one specific assertion back, so every total stayed level and
1944
+ // the guard said nothing — while the suite went from checking that
1945
+ // addition works to certifying that it is broken. It is the single
1946
+ // cheapest way to make a red suite green, and the one this tool exists
1947
+ // to refuse. The line-based version of this pairing could only see the
1948
+ // single-line spelling of the edit; the reassembled-statement version
1949
+ // above sees every spelling, and explains why on the way.
1950
+ detectExpectationRewrites(file, stats.hunks, stats, violations);
1951
+
1240
1952
  if (stats.removed.length > stats.added) {
1241
1953
  const unreplaced = stats.removed.slice(stats.added);
1242
1954
  for (const item of unreplaced) {