diffninja 0.3.1 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. package/README.md +65 -11
  2. package/dist/executables.d.ts +18 -0
  3. package/dist/executables.js +32 -0
  4. package/dist/git.d.ts +26 -1
  5. package/dist/git.js +56 -4
  6. package/dist/languages/child-env.d.ts +11 -0
  7. package/dist/languages/child-env.js +60 -0
  8. package/dist/languages/grammar-lock.d.ts +569 -0
  9. package/dist/languages/grammar-lock.js +574 -0
  10. package/dist/languages/grammars.d.ts +69 -9
  11. package/dist/languages/grammars.js +186 -119
  12. package/dist/review/call-flow-html.d.ts +3 -1
  13. package/dist/review/call-flow-html.js +13 -11
  14. package/dist/review/change-facts.d.ts +21 -1
  15. package/dist/review/change-facts.js +271 -49
  16. package/dist/review/cli.js +13 -1
  17. package/dist/review/connected-analysis.d.ts +4 -1
  18. package/dist/review/connected-analysis.js +4 -2
  19. package/dist/review/connected-html.d.ts +15 -4
  20. package/dist/review/connected-html.js +342 -35
  21. package/dist/review/connected.js +47 -14
  22. package/dist/review/escape-html.d.ts +5 -1
  23. package/dist/review/escape-html.js +7 -2
  24. package/dist/review/explanation.d.ts +4 -0
  25. package/dist/review/explanation.js +6 -1
  26. package/dist/review/github.d.ts +56 -0
  27. package/dist/review/github.js +234 -33
  28. package/dist/review/grammars-command.d.ts +12 -0
  29. package/dist/review/grammars-command.js +60 -0
  30. package/dist/review/hidden-characters.d.ts +31 -0
  31. package/dist/review/hidden-characters.js +113 -0
  32. package/dist/review/history.js +7 -3
  33. package/dist/review/html.d.ts +3 -2
  34. package/dist/review/html.js +19 -18
  35. package/dist/review/input.js +5 -2
  36. package/dist/review/intent.d.ts +7 -0
  37. package/dist/review/intent.js +25 -3
  38. package/dist/review/markdown.js +11 -0
  39. package/dist/review/mcp-cli.js +4 -1
  40. package/dist/review/mcp.d.ts +7 -1
  41. package/dist/review/mcp.js +145 -59
  42. package/dist/review/pipeline.d.ts +2 -1
  43. package/dist/review/pipeline.js +4 -3
  44. package/dist/review/pr-input.d.ts +7 -0
  45. package/dist/review/pr-input.js +25 -3
  46. package/dist/review/process-html.d.ts +1 -5
  47. package/dist/review/process-html.js +3 -12
  48. package/dist/review/questions.js +11 -3
  49. package/dist/review/reference-check.d.ts +5 -1
  50. package/dist/review/reference-check.js +40 -16
  51. package/dist/review/report-pages.d.ts +24 -9
  52. package/dist/review/report-pages.js +111 -28
  53. package/dist/review/result-budget.d.ts +28 -0
  54. package/dist/review/result-budget.js +136 -0
  55. package/dist/review/service.js +26 -2
  56. package/dist/review/setup.d.ts +1 -1
  57. package/dist/review/setup.js +9 -4
  58. package/dist/review/types.d.ts +19 -5
  59. package/dist/review/types.js +2 -1
  60. package/dist/review/update-check.d.ts +35 -0
  61. package/dist/review/update-check.js +76 -0
  62. package/dist/run.js +11 -5
  63. package/npm-shrinkwrap.json +3483 -0
  64. package/package.json +3 -2
@@ -17,6 +17,8 @@
17
17
  *
18
18
  * - only file types listed here are read; any other file answers no question
19
19
  * at all, never `no`;
20
+ * - a hunk with a changed line longer than {@link MAX_READ_LINE_CHARS} is not
21
+ * read either, rather than answered from part of that line;
20
22
  * - `no` means the changed lines the hunk shows contain no such pattern, never
21
23
  * that the property is absent from the file or the program;
22
24
  * - a line moved without change cancels out, because every question compares
@@ -59,6 +61,16 @@ const CONFIG_FILE = /\.(?:ya?ml|json|jsonc|json5|toml|ini|cfg|conf|properties)$|
59
61
  const JSON_FILE = /\.json[c5]?$/i;
60
62
  /** Evidence text is the source line, bounded so one long line cannot dominate a report. */
61
63
  const EVIDENCE_TEXT_LIMIT = 160;
64
+ /**
65
+ * The longest changed line the analysis reads. A longer one is minified or
66
+ * generated text, or padding that would push a change past any bound on how
67
+ * much of a line is read, so its hunk is left unread and uncertain instead.
68
+ */
69
+ export const MAX_READ_LINE_CHARS = 4000;
70
+ /** What diffninja does not read in a hunk whose facts have no language, as a phrase after "does not read". */
71
+ export function unreadCause(file) {
72
+ return changeFactLanguageOf(file) === null ? "this file type" : `a hunk with a changed line over ${MAX_READ_LINE_CHARS.toLocaleString("en-US")} characters`;
73
+ }
62
74
  export function changeFactLanguageOf(file) {
63
75
  if (C_LIKE_FILE.test(file))
64
76
  return "c-like";
@@ -86,12 +98,17 @@ export function factQuestionsFor(language) {
86
98
  return SQL_FACT_QUESTIONS;
87
99
  return CODE_FACT_QUESTIONS;
88
100
  }
101
+ /** A star followed by whitespace, a slash or the end of the line: how a block comment's continuation line starts. */
102
+ const COMMENT_CONTINUATION = /^\*(?:\s|\/|$)/;
89
103
  /**
90
104
  * One line with strings replaced by `S` and comments removed. `state` carries a
91
105
  * block comment or multi-line string into the next line of the same side.
92
106
  */
93
107
  function scanLine(line, language, state, keepStrings = false) {
94
108
  let out = "";
109
+ // Whether `out` holds anything but whitespace yet, kept as a flag: trimming it
110
+ // for every character would make a long line quadratic to scan.
111
+ let codeSeen = false;
95
112
  // A string's text, or its placeholder: facts ignore string content, while the
96
113
  // formatting-only check must see it, since changing a literal changes behavior.
97
114
  const literal = (from, to) => (keepStrings ? line.slice(from, to) : "S");
@@ -115,33 +132,34 @@ function scanLine(line, language, state, keepStrings = false) {
115
132
  state.open = null;
116
133
  continue;
117
134
  }
118
- const rest = line.slice(index);
119
- if (language === "c-like" && rest.startsWith("//"))
135
+ if (language === "c-like" && line.startsWith("//", index))
120
136
  break;
121
137
  // A hunk can start inside a block comment it never shows opening: a line
122
138
  // that begins `* ` or `*/` is that comment's continuation, not code.
123
- if (language === "c-like" && out.trim() === "" && /^\*(?:\s|\/|$)/.test(rest)) {
124
- if (rest.includes("*/")) {
125
- index = line.indexOf("*/", index) + 2;
139
+ if (language === "c-like" && !codeSeen && COMMENT_CONTINUATION.test(line.slice(index, index + 2))) {
140
+ const close = line.indexOf("*/", index);
141
+ if (close >= 0) {
142
+ index = close + 2;
126
143
  continue;
127
144
  }
128
145
  break;
129
146
  }
130
- if (language === "c-like" && rest.startsWith("/*")) {
147
+ if (language === "c-like" && line.startsWith("/*", index)) {
131
148
  state.open = "comment";
132
149
  index += 2;
133
150
  continue;
134
151
  }
135
- if (language !== "c-like" && rest.startsWith("#"))
152
+ if (language !== "c-like" && line.startsWith("#", index))
136
153
  break;
137
- if (language === "python" && (rest.startsWith('"""') || rest.startsWith("'''"))) {
138
- const delimiter = rest.startsWith('"""') ? '"""' : "'''";
154
+ if (language === "python" && (line.startsWith('"""', index) || line.startsWith("'''", index))) {
155
+ const delimiter = line.startsWith('"""', index) ? '"""' : "'''";
139
156
  const end = closingIndex(line, index + 3, delimiter);
140
157
  if (end < 0) {
141
158
  state.open = delimiter;
142
159
  return out + literal(index, line.length);
143
160
  }
144
161
  out += literal(index, end + 3);
162
+ codeSeen = true;
145
163
  index = end + 3;
146
164
  continue;
147
165
  }
@@ -153,6 +171,7 @@ function scanLine(line, language, state, keepStrings = false) {
153
171
  return out + literal(index, line.length);
154
172
  }
155
173
  out += literal(index, end + 1);
174
+ codeSeen = true;
156
175
  index = end + 1;
157
176
  continue;
158
177
  }
@@ -161,10 +180,13 @@ function scanLine(line, language, state, keepStrings = false) {
161
180
  if (end < 0)
162
181
  return out + literal(index, line.length);
163
182
  out += literal(index, end + 1);
183
+ codeSeen = true;
164
184
  index = end + 1;
165
185
  continue;
166
186
  }
167
187
  out += char;
188
+ if (!codeSeen && char !== undefined && !/\s/.test(char))
189
+ codeSeen = true;
168
190
  index += 1;
169
191
  }
170
192
  return out;
@@ -213,12 +235,30 @@ function compact(code) {
213
235
  .trim()
214
236
  .replace(/ (?=[^\w$])|(?<=[^\w$]) /g, "");
215
237
  }
238
+ /**
239
+ * The line without its `--` comment, as `line.replace(/--.*$/, "")` does it, in
240
+ * linear time. `.` refuses the four line terminators, so the comment must start
241
+ * after the last one; the regex rescanned the rest of the line from every `--`
242
+ * whenever a terminator followed, quadratic on a long line of them.
243
+ */
244
+ export function withoutSqlComment(line) {
245
+ let after = 0;
246
+ for (let index = line.length - 1; index >= 0; index -= 1) {
247
+ const char = line[index];
248
+ if (char === "\n" || char === "\r" || char === "\u2028" || char === "\u2029") {
249
+ after = index + 1;
250
+ break;
251
+ }
252
+ }
253
+ const comment = line.indexOf("--", after);
254
+ return comment === -1 ? line : line.slice(0, comment);
255
+ }
216
256
  function scannerFor(language, file) {
217
257
  if (language === "prose")
218
258
  return { code: (line) => line, literal: (line) => line };
219
259
  if (language === "sql") {
220
260
  // `--` comments go; string text stays, it is what a statement writes.
221
- const scan = (line) => line.replace(/--.*$/, "");
261
+ const scan = withoutSqlComment;
222
262
  return { code: scan, literal: scan };
223
263
  }
224
264
  if (language === "config") {
@@ -231,6 +271,7 @@ function scannerFor(language, file) {
231
271
  literal: (line, state) => scanLine(line, language, state, true),
232
272
  };
233
273
  }
274
+ /** Both sides of the hunk, or null when a changed line is too long to read. */
234
275
  function sidesOf(diff, scanners) {
235
276
  const before = [];
236
277
  const after = [];
@@ -248,6 +289,8 @@ function sidesOf(diff, scanners) {
248
289
  if (!inHunk || line.startsWith("\\"))
249
290
  continue;
250
291
  const marker = line[0];
292
+ if ((marker === "-" || marker === "+") && line.length - 1 > MAX_READ_LINE_CHARS)
293
+ return null;
251
294
  const raw = line.slice(1);
252
295
  const entry = (state, literalState, changed) => ({
253
296
  raw,
@@ -275,7 +318,60 @@ const CONDITION_KEYWORD = /\b(if|elif|while|unless|until|switch|when)\b/;
275
318
  const LOGICAL_OPERATOR = /(&&|\|\||\band\b|\bor\b)/;
276
319
  /** Where a logical operator is a condition: continuing one, or a returned boolean. */
277
320
  const CONDITION_LINE = /^(?:return\b|&&|\|\||!|\()|(?:&&|\|\||\()\s*$/;
278
- const TERNARY = /\s\?\s[^:]*\s:\s/;
321
+ /**
322
+ * Whether a line has a ternary, as the regex `\s\?\s[^:]*\s:\s` matches it, in
323
+ * linear time. That regex rescans the rest of the line from every ` ? ` for the
324
+ * next colon; here that colon is found once and shared by every ` ? ` before it.
325
+ */
326
+ function hasTernary(code) {
327
+ let colon = code.indexOf(":");
328
+ for (let mark = code.indexOf("?"); mark >= 0; mark = code.indexOf("?", mark + 1)) {
329
+ if (!SPACE.test(code[mark - 1] ?? "") || !SPACE.test(code[mark + 1] ?? ""))
330
+ continue;
331
+ if (colon !== -1 && colon < mark + 2)
332
+ colon = code.indexOf(":", mark + 2);
333
+ if (colon === -1)
334
+ return false;
335
+ // The space before the colon cannot be the one after `?`.
336
+ if (colon > mark + 2 && SPACE.test(code[colon - 1] ?? "") && SPACE.test(code[colon + 1] ?? ""))
337
+ return true;
338
+ }
339
+ return false;
340
+ }
341
+ /**
342
+ * `head`, a span up to the first `close` after it, then `tail`: the regex
343
+ * `head[^close]*close tail`, in linear time. With `open`, the span is optional
344
+ * and starts with it: `head(?:open[^close]*close)?tail`. The regex rescans the
345
+ * rest of the line from every head for the next `close`, so a line of heads
346
+ * without one took seconds; here that `close` is found once and shared.
347
+ * `head` must be global and `tail` sticky.
348
+ */
349
+ function spanPattern(head, open, close, tail) {
350
+ const tailAt = (text, at) => {
351
+ tail.lastIndex = at;
352
+ return tail.test(text);
353
+ };
354
+ return {
355
+ test: (text) => {
356
+ let closing = text.indexOf(close);
357
+ for (const match of text.matchAll(head)) {
358
+ let from = (match.index ?? 0) + match[0].length;
359
+ if (open !== null) {
360
+ if (tailAt(text, from))
361
+ return true;
362
+ if (text[from] !== open)
363
+ continue;
364
+ from += 1;
365
+ }
366
+ if (closing !== -1 && closing < from)
367
+ closing = text.indexOf(close, from);
368
+ if (closing !== -1 && tailAt(text, closing + 1))
369
+ return true;
370
+ }
371
+ return false;
372
+ },
373
+ };
374
+ }
279
375
  const LIMIT_WORD = /\b\w*(?:timeout|limit|max|min|size|length|len|offset|index|idx|count|capacity|retries|attempts|ttl|threshold|page|batch|delay|interval|slice|substring|substr|take|skip|range|depth|width|height|bound)\w*/i;
280
376
  const NUMBER = /(?<![\w$.])-?\d[\d_]*(?:\.\d+)?(?:e-?\d+)?\b/gi;
281
377
  /** Non-global twin of {@link NUMBER} for `test`, which a global pattern makes stateful. */
@@ -290,17 +386,21 @@ const ASSERTION = /\b(?:expect|assert\w*|should)\s*[.(]|\bt\.\w+\(|\.to(?:Be|Equ
290
386
  const DEFERRAL = /\b(?:retry|retries|retrying|retried|backoff|requeue|reschedul\w*|dead_?letter|dlq)\w*/i;
291
387
  /** Values a handler returns instead of the error: a default, not a failure. */
292
388
  const DEFAULT_VALUE = String.raw `(?:null|undefined|nil|None|false|False|0|-1|S|\[\s*\]|\{\s*\}|\(\s*\))`;
389
+ /** `catch` and the space after it; {@link spanPattern} reads the binding. */
390
+ const CATCH = /\bcatch\s*/g;
293
391
  const INLINE_DISCARD = [
294
392
  new RegExp(String.raw `\.catch\(\s*(?:\(\s*\w*\s*\)|\w+)\s*=>\s*(?:\{\s*\}|${DEFAULT_VALUE})\s*\)`),
295
393
  /\.catch\(\s*(?:noop|_\.noop|\(\)\s*=>\s*void\s+0)\s*\)/,
296
- /\bcatch\s*(?:\([^)]*\))?\s*\{\s*\}/,
297
- new RegExp(String.raw `\bcatch\s*(?:\([^)]*\))?\s*\{\s*(?:return(?:\s+${DEFAULT_VALUE})?|continue|break)\s*;?\s*\}`),
298
- /\bexcept\b[^:]*:\s*(?:pass|continue)\s*$/,
394
+ spanPattern(CATCH, "(", ")", /\s*\{\s*\}/y),
395
+ spanPattern(CATCH, "(", ")", new RegExp(String.raw `\s*\{\s*(?:return(?:\s+${DEFAULT_VALUE})?|continue|break)\s*;?\s*\}`, "y")),
396
+ spanPattern(/\bexcept\b/g, null, ":", /\s*(?:pass|continue)\s*$/y),
299
397
  /\brescue\s+nil\b/,
300
398
  /^_\s*=\s*err\b/,
301
399
  ];
400
+ const CATCH_BLOCK = spanPattern(CATCH, "(", ")", /\s*\{\s*$/y);
401
+ const GO_ERROR_BLOCK = /\bif\s*\(?\s*err\s*!=\s*nil\s*\)?\s*\{\s*$/;
302
402
  const BLOCK_HANDLER = {
303
- "c-like": /(?:\bcatch\s*(?:\([^)]*\))?|\bif\s*\(?\s*err\s*!=\s*nil\s*\)?)\s*\{\s*$/,
403
+ "c-like": { test: (code) => CATCH_BLOCK.test(code) || GO_ERROR_BLOCK.test(code) },
304
404
  python: /^except\b[^:]*:\s*$/,
305
405
  ruby: /^rescue\b/,
306
406
  };
@@ -331,15 +431,42 @@ function difference(before, after) {
331
431
  }
332
432
  return { removed, added };
333
433
  }
434
+ /**
435
+ * Every comparison on a line is read, operands whole: each operand lies between
436
+ * two operators, so the reads cover each character a bounded number of times.
437
+ */
438
+ const OPERAND_CHAR = /[\w$.[\]]/;
439
+ const SPACE = /\s/;
440
+ /** The operand ending just before `at`: whitespace skipped, then the run of operand characters. */
441
+ function operandBefore(code, at) {
442
+ let end = at;
443
+ while (end > 0 && SPACE.test(code[end - 1] ?? ""))
444
+ end -= 1;
445
+ let start = end;
446
+ while (start > 0 && OPERAND_CHAR.test(code[start - 1] ?? ""))
447
+ start -= 1;
448
+ return code.slice(start, end);
449
+ }
450
+ /** The operand starting after `from`: whitespace skipped, an optional minus, then the run of operand characters. */
451
+ function operandAfter(code, from) {
452
+ let start = from;
453
+ while (start < code.length && SPACE.test(code[start] ?? ""))
454
+ start += 1;
455
+ let end = start;
456
+ if (code[end] === "-")
457
+ end += 1;
458
+ const bodyStart = end;
459
+ while (end < code.length && OPERAND_CHAR.test(code[end] ?? ""))
460
+ end += 1;
461
+ return end === bodyStart ? "" : code.slice(start, end);
462
+ }
334
463
  function comparisonAtoms(code, language) {
335
464
  const atoms = [];
336
465
  const patterns = language === "python" ? [COMPARISON_OPERATOR, PYTHON_COMPARISON_OPERATOR] : [COMPARISON_OPERATOR];
337
466
  for (const pattern of patterns) {
338
467
  for (const match of code.matchAll(pattern)) {
339
468
  const at = match.index ?? 0;
340
- const left = /[\w$.[\]]+$/.exec(code.slice(0, at).trimEnd())?.[0] ?? "";
341
- const right = /^-?[\w$.[\]]+/.exec(code.slice(at + match[0].length).trimStart())?.[0] ?? "";
342
- atoms.push({ left, operator: match[0].trim(), right });
469
+ atoms.push({ left: operandBefore(code, at), operator: match[0].trim(), right: operandAfter(code, at + match[0].length) });
343
470
  }
344
471
  }
345
472
  return atoms;
@@ -364,7 +491,7 @@ function conditionOf(code) {
364
491
  }
365
492
  // A line of a multi-line condition, or a returned boolean; an assignment such
366
493
  // as `const x = a || b` combines values and is not a condition by itself.
367
- if ((LOGICAL_OPERATOR.test(code) && CONDITION_LINE.test(code)) || TERNARY.test(code))
494
+ if ((LOGICAL_OPERATOR.test(code) && CONDITION_LINE.test(code)) || hasTernary(code))
368
495
  return compact(code);
369
496
  return null;
370
497
  }
@@ -391,29 +518,76 @@ function lineHits(lines, pattern) {
391
518
  /** Each bound operator and its strict or non-strict twin. */
392
519
  const STRICTNESS = new Map([["<", "<="], ["<=", "<"], [">", ">="], [">=", ">"]]);
393
520
  const isNumber = (token) => /^-?\d[\d_]*(?:\.\d+)?(?:e-?\d+)?$/i.test(token);
521
+ /**
522
+ * For each group of `items`, the index of its first item and of the first after
523
+ * it with another value. Whatever value a lookup brings, one of the two is the
524
+ * group's first item with a different value: a lookup instead of a scan of
525
+ * every item, which made pairing one side of a hunk with the other quadratic.
526
+ */
527
+ function firstDiffering(items, groupOf, valueOf) {
528
+ const groups = new Map();
529
+ items.forEach((item, index) => {
530
+ const group = groupOf(item);
531
+ if (group === null)
532
+ return;
533
+ const seen = groups.get(group);
534
+ if (seen === undefined)
535
+ groups.set(group, { first: index });
536
+ else if (seen.other === undefined && valueOf(item) !== valueOf(items[seen.first]))
537
+ seen.other = index;
538
+ });
539
+ return (group, value) => {
540
+ const seen = groups.get(group);
541
+ if (seen === undefined)
542
+ return undefined;
543
+ return valueOf(items[seen.first]) !== value ? seen.first : seen.other;
544
+ };
545
+ }
394
546
  /**
395
547
  * A bound whose admitted range changed: the same comparison with a strict and a
396
548
  * non-strict operator swapped, or a numeric side changed; or a line naming a limit
397
549
  * whose only difference is a number.
398
550
  */
399
551
  function limitHit(removed, added, language) {
400
- const removedAtoms = removed.flatMap((hit) => comparisonAtoms(hit.line.code, language).map((atom) => ({ atom, hit })));
401
- const addedAtoms = added.flatMap((hit) => comparisonAtoms(hit.line.code, language).map((atom) => ({ atom, hit })));
402
- for (const { atom: before } of removedAtoms) {
403
- for (const { atom: after, hit } of addedAtoms) {
404
- const sameOperands = before.left === after.left && before.right === after.right;
405
- if (sameOperands && STRICTNESS.get(before.operator) === after.operator)
406
- return hit;
407
- const sameDirection = before.operator === after.operator || STRICTNESS.get(before.operator) === after.operator;
408
- if (!sameDirection)
552
+ // A line has one hit per condition and per comparison; its comparisons are read once, not once per hit.
553
+ const atomsOf = (hits) => {
554
+ const seen = new Set();
555
+ const atoms = [];
556
+ for (const hit of hits) {
557
+ if (seen.has(hit.line))
409
558
  continue;
410
- if (before.left === after.left && isNumber(before.right) && isNumber(after.right) && before.right !== after.right) {
411
- return hit;
412
- }
413
- if (before.right === after.right && isNumber(before.left) && isNumber(after.left) && before.left !== after.left) {
414
- return hit;
415
- }
559
+ seen.add(hit.line);
560
+ for (const atom of comparisonAtoms(hit.line.code, language))
561
+ atoms.push({ atom, hit });
562
+ }
563
+ return atoms;
564
+ };
565
+ // For the first removed comparison that has one, the earliest added comparison
566
+ // that moves its bound, found by lookups in the added side. Comparing every
567
+ // pair was quadratic, and the cap that bounded it let padding hide the change.
568
+ const candidates = atomsOf(added);
569
+ const exact = new Map();
570
+ candidates.forEach(({ atom }, index) => {
571
+ const key = `${atom.left}\n${atom.operator}\n${atom.right}`;
572
+ if (!exact.has(key))
573
+ exact.set(key, index);
574
+ });
575
+ const sameLeft = firstDiffering(candidates, ({ atom }) => (isNumber(atom.right) ? `${atom.left}\n${atom.operator}` : null), ({ atom }) => atom.right);
576
+ const sameRight = firstDiffering(candidates, ({ atom }) => (isNumber(atom.left) ? `${atom.right}\n${atom.operator}` : null), ({ atom }) => atom.left);
577
+ for (const { atom: before } of atomsOf(removed)) {
578
+ const twin = STRICTNESS.get(before.operator);
579
+ // The same operands with the strict and non-strict operator swapped.
580
+ const found = [twin === undefined ? undefined : exact.get(`${before.left}\n${twin}\n${before.right}`)];
581
+ // The same direction, one side the same and the other a different number.
582
+ for (const operator of twin === undefined ? [before.operator] : [before.operator, twin]) {
583
+ if (isNumber(before.right))
584
+ found.push(sameLeft(`${before.left}\n${operator}`, before.right));
585
+ if (isNumber(before.left))
586
+ found.push(sameRight(`${before.right}\n${operator}`, before.left));
416
587
  }
588
+ const earliest = Math.min(...found.filter((index) => index !== undefined));
589
+ if (earliest !== Infinity)
590
+ return candidates[earliest].hit;
417
591
  }
418
592
  return null;
419
593
  }
@@ -421,13 +595,14 @@ function numericLimitHit(before, after) {
421
595
  const limitLines = (lines) => lines
422
596
  .filter((line) => line.changed && line.code !== "" && LIMIT_WORD.test(line.code) && HAS_NUMBER.test(line.code) && !ASSERTION.test(line.code))
423
597
  .map((line) => ({ key: compact(line.code), numberless: compact(line.code.replace(NUMBER, "N")), line }));
424
- const removed = limitLines(before);
598
+ // For the first removed line that has one, the earliest added line of the
599
+ // same form, numbers aside, with other numbers.
425
600
  const added = limitLines(after);
426
- for (const old of removed) {
427
- for (const candidate of added) {
428
- if (old.numberless === candidate.numberless && old.key !== candidate.key)
429
- return { key: candidate.key, line: candidate.line };
430
- }
601
+ const differing = firstDiffering(added, (line) => line.numberless, (line) => line.key);
602
+ for (const old of limitLines(before)) {
603
+ const index = differing(old.numberless, old.key);
604
+ if (index !== undefined)
605
+ return added[index];
431
606
  }
432
607
  return null;
433
608
  }
@@ -441,7 +616,10 @@ function guardHits(lines, changedConditions) {
441
616
  lines.forEach((line, index) => {
442
617
  if (!changedConditions.has(line))
443
618
  return;
444
- const next = lines.slice(index + 1).find((candidate) => candidate.code !== "");
619
+ let after = index + 1;
620
+ while (after < lines.length && lines[after].code === "")
621
+ after += 1;
622
+ const next = lines[after];
445
623
  if (PROPAGATION.test(line.code) || (next !== undefined && PROPAGATION.test(next.code))) {
446
624
  hits.push({ key: `guard:${compact(line.code)}`, line });
447
625
  }
@@ -463,7 +641,8 @@ function discardHits(lines, language) {
463
641
  return;
464
642
  const body = [];
465
643
  let closed = false;
466
- for (const candidate of lines.slice(index + 1)) {
644
+ for (let after = index + 1; after < lines.length; after++) {
645
+ const candidate = lines[after];
467
646
  if (candidate.code === "")
468
647
  continue;
469
648
  if (language === "python" && candidate.indent <= line.indent) {
@@ -521,10 +700,43 @@ const NORMATIVE = /\b(?:must|shall|should|required|requires|never|always|only|ca
521
700
  const REFERENCE = /\]\(\s*<?([^)\s>]+)|(https?:\/\/[^\s)>"'\]]+)/g;
522
701
  /** Settings that turn a failing CI check into a passing or advisory one. */
523
702
  const GATE_WEAKENING = /continue-on-error:\s*true|allow_failure:\s*true|\|\|\s*true\b|\bset\s+\+e\b|--no-verify\b|\bif:\s*false\b|\bskip\b|\bwarn(?:ing)?\b|\bignore\b|fail_?[oO]n_?[eE]rror\W+false|--passWithNoTests|\bexit\s+0\b|--force\b/i;
524
- /** A step that runs a check; fewer of them after the change is a weaker gate. */
525
- const CHECK_STEP = /\b(?:run|script|command)\b.*\b(?:test|tests|lint|check|audit|verify|typecheck|tsc|vitest|jest|pytest|mypy|eslint|oxlint)\b|\buses:.*\b(?:codeql|lint|test|scan)/i;
703
+ /**
704
+ * A step that runs a check; fewer of them after the change is a weaker gate.
705
+ * `^(?=(.*?X))\1` takes the first `run` or `uses:` once, since a lookahead does
706
+ * not backtrack, and a check after any later one is also after the first. The
707
+ * plain `run.*test` rescanned the line from every `run`, quadratic on a long line.
708
+ */
709
+ const CHECK_STEP = /^(?=(.*?\b(?:run|script|command)\b))\1.*\b(?:test|tests|lint|check|audit|verify|typecheck|tsc|vitest|jest|pytest|mypy|eslint|oxlint)\b|^(?=(.*?\buses:))\2.*\b(?:codeql|lint|test|scan)/i;
526
710
  const PERMISSION = /\bpermissions\b|:\s*write(?:-all)?\b|\bwrite-all\b|\bpull_request_target\b|\bsecrets\.|\bid-token\b|\bGITHUB_TOKEN\b|\bprivileged:\s*true|\ballowPrivilegeEscalation\b|\brunAsUser:\s*0\b|^USER\s+root\b|\bsudo\b/i;
527
- const PIN = /\buses:\s*\S+@|\bimage:\s*\S+|^FROM\s|\bversion\b|"[@\w./-]+"\s*:\s*"\s*[\^~<>=*]?\s*(?:v?\d|latest|\*)/i;
711
+ const PIN_SETTING = /\bimage:\s*\S+|^FROM\s|\bversion\b|"[@\w./-]+"\s*:\s*"\s*[\^~<>=*]?\s*(?:v?\d|latest|\*)/i;
712
+ const USES = /\buses:\s*/gi;
713
+ const WHITESPACE = /\s/g;
714
+ /**
715
+ * `uses:` naming a ref, as `uses:\s*\S+@` matches it, in linear time. That regex
716
+ * rescanned the rest of the word from every `uses:` in it; here the next `@` and
717
+ * the next space are found once and shared.
718
+ */
719
+ function usesRef(text) {
720
+ let mark = text.indexOf("@");
721
+ let space = -1;
722
+ for (const match of text.matchAll(USES)) {
723
+ const word = (match.index ?? 0) + match[0].length;
724
+ if (word >= text.length)
725
+ return false;
726
+ if (mark !== -1 && mark <= word)
727
+ mark = text.indexOf("@", word + 1);
728
+ if (mark === -1)
729
+ return false;
730
+ if (space < word) {
731
+ WHITESPACE.lastIndex = word;
732
+ space = WHITESPACE.exec(text)?.index ?? text.length;
733
+ }
734
+ if (mark < space)
735
+ return true;
736
+ }
737
+ return false;
738
+ }
739
+ const PIN = { test: (text) => PIN_SETTING.test(text) || usesRef(text) };
528
740
  /** Changed lines of one side whose text carries each link target they name. */
529
741
  function referenceHits(lines) {
530
742
  const hits = [];
@@ -558,7 +770,9 @@ const CONTRACT = [
558
770
  /^export\s+(?:default\s+)?(?:declare\s+)?(?:async\s+)?(?:function|class|interface|type|enum|const|let|abstract)\b/,
559
771
  /^@(?:Get|Post|Put|Patch|Delete|All|Controller|Resolver|Query|Mutation|Column|PrimaryColumn|PrimaryGeneratedColumn|Entity|ManyToOne|OneToMany|OneToOne|ManyToMany|JoinColumn|Index|Unique|Is[A-Z]\w*|Min|Max|Length|ValidateNested|Type|Transform|Api(?:Property|ResponseProperty)\w*|Field|Prop|Schema)\b/,
560
772
  /^(?:public\s+|static\s+|async\s+|override\s+|readonly\s+)*(?!(?:if|for|while|switch|catch|return|function|await|new|else|do|try)\b)[A-Za-z_$][\w$]*\s*(?:<[^>]*>)?\s*\([^)]*\)\s*(?::\s*[^={;]+)?\s*\{$/,
561
- /^(?:public|protected)\s+(?:static\s+|abstract\s+|final\s+|async\s+|override\s+)*[\w<>[\],.? ]+\s+\w+\s*\(/,
773
+ // Modifiers are part of the type run: listing them in a loop before it made
774
+ // every split between the two a new try, quadratic on a line of modifiers.
775
+ /^(?:public|protected)\s+[\w<>[\],.? ]+\s+\w+\s*\(/,
562
776
  /^def\s+[A-Za-z]\w*\s*\(/,
563
777
  /^class\s+[A-Z]\w*/,
564
778
  /^func\s+(?:\([^)]*\)\s*)?[A-Z]\w*\s*\(/,
@@ -595,8 +809,12 @@ function proseFacts(sides, record) {
595
809
  const limit = numericLimitHit(sides.before, sides.after);
596
810
  record("limitChanged", limit === null ? null : evidenceOf(limit, "added"));
597
811
  }
598
- /** Bookkeeping files whose numbers record state, not bounds. */
599
- const BOOKKEEPING_FILE = /(?:suppressions?|baseline|snapshot|lock)[\w.-]*\.(?:json|ya?ml|toml)$/i;
812
+ /**
813
+ * Bookkeeping files whose numbers record state, not bounds. `^(?=([\s\S]*X))\1`
814
+ * takes the last keyword once: if a plain name runs from any keyword to the
815
+ * extension, one runs from the last. Trying every keyword was quadratic on a long path.
816
+ */
817
+ const BOOKKEEPING_FILE = /^(?=([\s\S]*(?:suppressions?|baseline|snapshot|lock)))\1[\w.-]*\.(?:json|ya?ml|toml)$/i;
600
818
  function configFacts(sides, record, file) {
601
819
  const weakening = difference(textHits(sides.before, GATE_WEAKENING), textHits(sides.after, GATE_WEAKENING));
602
820
  const checksBefore = textHits(sides.before, CHECK_STEP);
@@ -652,6 +870,8 @@ export function changeFactsOf(unit) {
652
870
  if (language === null)
653
871
  return { language, inert: null, answers: {}, evidence: {} };
654
872
  const sides = sidesOf(unit.diff, scannerFor(language, unit.file));
873
+ if (sides === null)
874
+ return { language: null, inert: null, answers: {}, evidence: {} };
655
875
  const answers = {};
656
876
  for (const question of factQuestionsFor(language))
657
877
  answers[question] = "no";
@@ -678,7 +898,9 @@ export function changeFactsOf(unit) {
678
898
  return { language, inert: isInert(sides, language), importsOnly: importsOnly(sides), answers, evidence };
679
899
  }
680
900
  const IMPORT_START = /^(?:import\b|export\s+(?:\*|\{[^}]*\})\s+from\b|from\s+[\w.]+\s+import\b|using\s+[\w.]+\s*;|(?:const|let|var)\s+[\w{}\s,:]+=\s*require\()/;
681
- const IMPORT_LIST_OPEN = /^(?:import\b[^;]*\{[^}]*$|import\s*\($|from\s+[\w.]+\s+import\s*\($|export\s*\{[^}]*$)/;
901
+ // `(?=([^;]*\{))\1` takes the last `{` before any `;` once: if a `}` follows it,
902
+ // one follows every earlier `{`, and trying each of them was quadratic.
903
+ const IMPORT_LIST_OPEN = /^(?:import\b(?=([^;]*\{))\1[^}]*$|import\s*\($|from\s+[\w.]+\s+import\s*\($|export\s*\{[^}]*$)/;
682
904
  const IMPORT_LIST_CLOSE = /[})]/;
683
905
  const IMPORT_LIST_ITEM = /^(?:type\s+)?[\w$]+(?:\s+as\s+[\w$]+)?,?$/;
684
906
  const IMPORT_LIST_END = /^\}\s*from\b/;
@@ -1,6 +1,7 @@
1
1
  #!/usr/bin/env node
2
2
  import { parseArgs } from "node:util";
3
3
  import { runSetup, setupHelp } from "./setup.js";
4
+ import { runGrammars } from "./grammars-command.js";
4
5
  /**
5
6
  * diffninja runs inside an agent CLI through its MCP server; this command only
6
7
  * registers that server. Reviews are requested from the agent, which calls the
@@ -11,7 +12,14 @@ import { runSetup, setupHelp } from "./setup.js";
11
12
  const help = `diffninja. Pull request review inside your coding agent.
12
13
 
13
14
  diffninja setup [--cli claude,codex,omp,pi] [--uninstall] [--dry-run] [--no-install]
14
- Register the diffninja MCP server on every detected agent CLI.
15
+ Register the diffninja MCP server on every detected agent CLI. Runs
16
+ npm to install diffninja globally, and rewrites each JSON config it
17
+ changes (see diffninja setup --help).
18
+ diffninja grammars install [--build] [--dry-run] | status
19
+ Add the tree-sitter grammars call flows use for languages other than
20
+ JavaScript and TypeScript. Downloads code through npm, and only when
21
+ you run it (exact versions, integrity-checked, no install scripts
22
+ unless you pass --build).
15
23
  diffninja --help
16
24
 
17
25
  After setup, ask your agent to review a GitHub pull request link, a diff, or a git
@@ -28,6 +36,10 @@ async function main() {
28
36
  console.log(help);
29
37
  return;
30
38
  }
39
+ if (command === "grammars") {
40
+ runGrammars(args);
41
+ return;
42
+ }
31
43
  if (command !== "setup")
32
44
  throw new Error(NO_TERMINAL_REVIEW);
33
45
  const { values, positionals } = parseArgs({ args, options: {
@@ -11,6 +11,7 @@
11
11
  * MCP client that recorded it.
12
12
  */
13
13
  import { type QuestionKind, type Verdict } from "./questions.js";
14
+ import type { UpdateNotice } from "./update-check.js";
14
15
  import type { AgentSummary, ReviewReport, ReviewStatus, SuggestedComment } from "./types.js";
15
16
  import type { ExplanationChange } from "./explanation.js";
16
17
  /** Most agenda entries the page lists; the full report has the rest. */
@@ -89,7 +90,7 @@ export interface ConnectedAnalysis {
89
90
  };
90
91
  /** Changed files that have call-flow diagrams, for the page's "Call flow" buttons; empty for a patch-only analysis. */
91
92
  readonly callFlowFiles: readonly string[];
92
- /** Line comments the reviewing agent suggested, for the human to add to their review or not. */
93
+ /** Comments the reviewing agent says block the merge, each with its proof, for the human to add to their review or not. */
93
94
  suggestions?: {
94
95
  readonly suggestedBy: string;
95
96
  readonly comments: readonly SuggestedComment[];
@@ -102,6 +103,8 @@ export interface ConnectedAnalysis {
102
103
  * author's stated intent, not a claim that the changes achieve it.
103
104
  */
104
105
  summary?: AgentSummary;
106
+ /** A newer diffninja exists: the page shows a one-line notice with the command. */
107
+ update?: UpdateNotice;
105
108
  /** Present once the reviewing agent's business explanation was accepted for this report. */
106
109
  explanation?: ConnectedExplanation;
107
110
  }
@@ -10,7 +10,7 @@
10
10
  * facts are lexical, and an answer is a closed-set choice attributed to the
11
11
  * MCP client that recorded it.
12
12
  */
13
- import { CHANGE_FACT_QUESTIONS } from "./change-facts.js";
13
+ import { CHANGE_FACT_QUESTIONS, unreadCause } from "./change-facts.js";
14
14
  import { verdictOf } from "./questions.js";
15
15
  /** Most agenda entries the page lists; the full report has the rest. */
16
16
  export const CONNECTED_AGENDA_LIMIT = 5;
@@ -76,7 +76,7 @@ function noteOf(item) {
76
76
  return "Not read by diffninja (binary, rename, mode, or other metadata-only change). Check it yourself.";
77
77
  }
78
78
  if (item.facts?.language === null)
79
- return "diffninja does not read this file type. Read this hunk yourself.";
79
+ return `diffninja does not read ${unreadCause(item.file)}. Read this hunk yourself.`;
80
80
  if (item.facts?.inert === true)
81
81
  return "Formatting or comments only.";
82
82
  if (item.facts?.importsOnly === true)
@@ -178,6 +178,8 @@ export function connectedAnalysisOf(report, snapshotId, reviewId, reportUrl, sco
178
178
  // verbatim, attributed, and never synthesized here.
179
179
  if (report.agentSummary !== undefined)
180
180
  analysis.summary = report.agentSummary;
181
+ if (report.updateNotice !== undefined)
182
+ analysis.update = report.updateNotice;
181
183
  const explanation = report.agentExplanation;
182
184
  if (explanation !== undefined) {
183
185
  analysis.explanation = {
@@ -5,13 +5,24 @@
5
5
  * draft to `/api/preview`, and posts the same payload to `/api/submit`. Every
6
6
  * string that comes from GitHub (paths, diff lines, the diff text itself, error
7
7
  * messages) reaches the DOM through `textContent`; nothing is interpolated into
8
- * markup or into the script. Style and script carry the session CSRF token as
9
- * their CSP nonce, so the server can serve a `default-src 'none'` policy with
10
- * no inline handlers and no inline styles.
8
+ * markup or into the script. Style and script carry a per-response nonce, so the
9
+ * server can serve a `default-src 'none'` policy with no inline handlers and no
10
+ * inline styles; the session's CSRF token is a different secret, and the whole
11
+ * page lives under the session's secret path prefix.
11
12
  *
12
13
  * Drafts live in `sessionStorage`, keyed per pull request and stamped with the
13
14
  * snapshot id. Restoring a draft revalidates every comment anchor against the
14
15
  * snapshot on screen: a comment whose line no longer exists in the new revision
15
16
  * is dropped, never silently carried over.
17
+ *
18
+ * A change is viewed only by the reader's click on its rail checkbox, never by
19
+ * scrolling. GitHub's Viewed mark is per file, so the page marks a file Viewed
20
+ * on GitHub (`POST api/viewed`) exactly when every hunk of it is viewed, and
21
+ * un-marks it when one is not. Progress inside a file stays in `sessionStorage`,
22
+ * keyed by snapshot id.
16
23
  */
17
- export declare function renderConnectedPage(csrf: string): string;
24
+ export declare function renderConnectedPage(options: {
25
+ readonly csrf: string;
26
+ readonly nonce: string;
27
+ readonly base: string;
28
+ }): string;