diffninja 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +259 -0
- package/dist/calltree.d.ts +47 -0
- package/dist/calltree.js +296 -0
- package/dist/cli.d.ts +57 -0
- package/dist/cli.js +340 -0
- package/dist/diff.d.ts +7 -0
- package/dist/diff.js +114 -0
- package/dist/extract.d.ts +26 -0
- package/dist/extract.js +152 -0
- package/dist/git.d.ts +40 -0
- package/dist/git.js +288 -0
- package/dist/index.d.ts +9 -0
- package/dist/index.js +8 -0
- package/dist/infer.d.ts +21 -0
- package/dist/infer.js +189 -0
- package/dist/languages/bash.d.ts +2 -0
- package/dist/languages/bash.js +208 -0
- package/dist/languages/c.d.ts +2 -0
- package/dist/languages/c.js +218 -0
- package/dist/languages/call-syntax.d.ts +125 -0
- package/dist/languages/call-syntax.js +997 -0
- package/dist/languages/cpp.d.ts +2 -0
- package/dist/languages/cpp.js +321 -0
- package/dist/languages/csharp.d.ts +2 -0
- package/dist/languages/csharp.js +324 -0
- package/dist/languages/elixir.d.ts +2 -0
- package/dist/languages/elixir.js +331 -0
- package/dist/languages/go.d.ts +2 -0
- package/dist/languages/go.js +299 -0
- package/dist/languages/grammars.d.ts +50 -0
- package/dist/languages/grammars.js +351 -0
- package/dist/languages/haskell.d.ts +2 -0
- package/dist/languages/haskell.js +250 -0
- package/dist/languages/java.d.ts +2 -0
- package/dist/languages/java.js +351 -0
- package/dist/languages/javascript.d.ts +4 -0
- package/dist/languages/javascript.js +648 -0
- package/dist/languages/kotlin.d.ts +2 -0
- package/dist/languages/kotlin.js +368 -0
- package/dist/languages/lua.d.ts +2 -0
- package/dist/languages/lua.js +212 -0
- package/dist/languages/ocaml.d.ts +2 -0
- package/dist/languages/ocaml.js +291 -0
- package/dist/languages/perl.d.ts +2 -0
- package/dist/languages/perl.js +418 -0
- package/dist/languages/php.d.ts +2 -0
- package/dist/languages/php.js +397 -0
- package/dist/languages/python.d.ts +2 -0
- package/dist/languages/python.js +376 -0
- package/dist/languages/registry.d.ts +7 -0
- package/dist/languages/registry.js +69 -0
- package/dist/languages/ruby.d.ts +2 -0
- package/dist/languages/ruby.js +391 -0
- package/dist/languages/rust.d.ts +2 -0
- package/dist/languages/rust.js +261 -0
- package/dist/languages/scala.d.ts +2 -0
- package/dist/languages/scala.js +307 -0
- package/dist/languages/solidity.d.ts +2 -0
- package/dist/languages/solidity.js +240 -0
- package/dist/languages/swift.d.ts +2 -0
- package/dist/languages/swift.js +268 -0
- package/dist/languages/types.d.ts +36 -0
- package/dist/languages/types.js +74 -0
- package/dist/languages/typescript-contracts.d.ts +57 -0
- package/dist/languages/typescript-contracts.js +528 -0
- package/dist/languages/typescript-dispatch.d.ts +68 -0
- package/dist/languages/typescript-dispatch.js +710 -0
- package/dist/languages/typescript.d.ts +4 -0
- package/dist/languages/typescript.js +722 -0
- package/dist/languages/zig.d.ts +2 -0
- package/dist/languages/zig.js +243 -0
- package/dist/loc.d.ts +17 -0
- package/dist/loc.js +34 -0
- package/dist/reach.d.ts +17 -0
- package/dist/reach.js +65 -0
- package/dist/render.d.ts +18 -0
- package/dist/render.js +83 -0
- package/dist/review/brand.d.ts +8 -0
- package/dist/review/brand.js +25 -0
- package/dist/review/call-context.d.ts +27 -0
- package/dist/review/call-context.js +446 -0
- package/dist/review/call-flow-html.d.ts +32 -0
- package/dist/review/call-flow-html.js +1870 -0
- package/dist/review/call-flow-nav.d.ts +151 -0
- package/dist/review/call-flow-nav.js +317 -0
- package/dist/review/call-flow.d.ts +47 -0
- package/dist/review/call-flow.js +229 -0
- package/dist/review/change-facts.d.ts +69 -0
- package/dist/review/change-facts.js +729 -0
- package/dist/review/cli.d.ts +2 -0
- package/dist/review/cli.js +50 -0
- package/dist/review/connected-analysis.d.ts +100 -0
- package/dist/review/connected-analysis.js +163 -0
- package/dist/review/connected-html.d.ts +17 -0
- package/dist/review/connected-html.js +2853 -0
- package/dist/review/connected.d.ts +23 -0
- package/dist/review/connected.js +141 -0
- package/dist/review/escape-html.d.ts +2 -0
- package/dist/review/escape-html.js +9 -0
- package/dist/review/evidence-html.d.ts +21 -0
- package/dist/review/evidence-html.js +521 -0
- package/dist/review/evidence-syntax.d.ts +132 -0
- package/dist/review/evidence-syntax.js +478 -0
- package/dist/review/evidence-types.d.ts +62 -0
- package/dist/review/evidence-types.js +1 -0
- package/dist/review/evidence.d.ts +31 -0
- package/dist/review/evidence.js +1603 -0
- package/dist/review/file-role.d.ts +9 -0
- package/dist/review/file-role.js +29 -0
- package/dist/review/github.d.ts +204 -0
- package/dist/review/github.js +1245 -0
- package/dist/review/history.d.ts +101 -0
- package/dist/review/history.js +412 -0
- package/dist/review/html.d.ts +34 -0
- package/dist/review/html.js +1104 -0
- package/dist/review/input.d.ts +10 -0
- package/dist/review/input.js +113 -0
- package/dist/review/intent.d.ts +4 -0
- package/dist/review/intent.js +75 -0
- package/dist/review/mcp-cli.d.ts +2 -0
- package/dist/review/mcp-cli.js +25 -0
- package/dist/review/mcp.d.ts +12 -0
- package/dist/review/mcp.js +414 -0
- package/dist/review/module-resolution.d.ts +2 -0
- package/dist/review/module-resolution.js +86 -0
- package/dist/review/palette.d.ts +7 -0
- package/dist/review/palette.js +104 -0
- package/dist/review/pipeline.d.ts +77 -0
- package/dist/review/pipeline.js +227 -0
- package/dist/review/pr-input.d.ts +19 -0
- package/dist/review/pr-input.js +130 -0
- package/dist/review/questions.d.ts +201 -0
- package/dist/review/questions.js +174 -0
- package/dist/review/reference-check.d.ts +7 -0
- package/dist/review/reference-check.js +733 -0
- package/dist/review/report-pages.d.ts +109 -0
- package/dist/review/report-pages.js +328 -0
- package/dist/review/service.d.ts +23 -0
- package/dist/review/service.js +198 -0
- package/dist/review/setup.d.ts +112 -0
- package/dist/review/setup.js +549 -0
- package/dist/review/source.d.ts +26 -0
- package/dist/review/source.js +276 -0
- package/dist/review/toml.d.ts +38 -0
- package/dist/review/toml.js +565 -0
- package/dist/review/types.d.ts +179 -0
- package/dist/review/types.js +1 -0
- package/dist/run.d.ts +49 -0
- package/dist/run.js +311 -0
- package/dist/types.d.ts +366 -0
- package/dist/types.js +83 -0
- package/package.json +88 -0
- package/scripts/ensure-native-grammar.mjs +188 -0
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Project context for a git-range review, read from the local repository only.
|
|
3
|
+
*
|
|
4
|
+
* What a diff cannot show a reviewer is often the history and habits around it:
|
|
5
|
+
* the removed lines came from a fix, a similar change was reverted before, the
|
|
6
|
+
* project writes down rules for contributors, or every sibling file follows a
|
|
7
|
+
* pattern this one does not. All four are read here with plain git commands on
|
|
8
|
+
* the reviewed snapshots — nothing is fetched, checked out, or written — and are
|
|
9
|
+
* reported as pointers for the reviewer and the agent, never as verdicts. Commit
|
|
10
|
+
* subjects are repository text: data to show, not instructions to follow.
|
|
11
|
+
*
|
|
12
|
+
* Every selection is deterministic: the same repository and range yield the same
|
|
13
|
+
* context. A shallow clone cuts history, and lines whose origin lies past the cut
|
|
14
|
+
* are counted as unknown rather than attributed to the boundary commit.
|
|
15
|
+
*/
|
|
16
|
+
import type { ReviewUnit } from "./types.js";
|
|
17
|
+
export interface CommitRef {
|
|
18
|
+
/** Abbreviated to 12 hex digits. */
|
|
19
|
+
readonly commit: string;
|
|
20
|
+
/** Author date, UTC, `YYYY-MM-DD`. */
|
|
21
|
+
readonly date: string;
|
|
22
|
+
/** First line of the commit message, verbatim; untrusted repository text. */
|
|
23
|
+
readonly subject: string;
|
|
24
|
+
}
|
|
25
|
+
/** A commit that last changed some of the lines a hunk removes or rewrites. */
|
|
26
|
+
export interface LineOrigin extends CommitRef {
|
|
27
|
+
/** How many of the hunk's removed lines it last changed. */
|
|
28
|
+
readonly lines: number;
|
|
29
|
+
/** The subject names a fix, workaround, regression, revert, or compatibility concern. */
|
|
30
|
+
readonly notable: boolean;
|
|
31
|
+
}
|
|
32
|
+
export interface HunkHistory {
|
|
33
|
+
/** At most {@link MAX_ORIGINS_PER_HUNK}, most lines first. */
|
|
34
|
+
readonly origins: LineOrigin[];
|
|
35
|
+
/** Removed lines whose origin is past a shallow clone's boundary or could not be read. */
|
|
36
|
+
readonly unknownLines: number;
|
|
37
|
+
}
|
|
38
|
+
/** A revert commit before the base that a reviewer may want to compare this change with. */
|
|
39
|
+
export interface RevertRef extends CommitRef {
|
|
40
|
+
/** `file` when it touched a file this diff changes; otherwise the shared word. */
|
|
41
|
+
readonly reason: {
|
|
42
|
+
readonly kind: "file";
|
|
43
|
+
readonly file: string;
|
|
44
|
+
} | {
|
|
45
|
+
readonly kind: "term";
|
|
46
|
+
readonly term: string;
|
|
47
|
+
};
|
|
48
|
+
}
|
|
49
|
+
/** Identifiers most sibling files share that a new file does not use. */
|
|
50
|
+
export interface PeerConvention {
|
|
51
|
+
readonly file: string;
|
|
52
|
+
/** Glob-like pattern the peers match, e.g. `homeassistant/components/*\/select.py`. */
|
|
53
|
+
readonly pattern: string;
|
|
54
|
+
/** Peers read (at most {@link MAX_PEERS}). */
|
|
55
|
+
readonly peers: number;
|
|
56
|
+
readonly common: readonly {
|
|
57
|
+
readonly name: string;
|
|
58
|
+
readonly peers: number;
|
|
59
|
+
}[];
|
|
60
|
+
}
|
|
61
|
+
export interface ProjectContext {
|
|
62
|
+
/** `shallow` when the clone's history is cut, so origins and reverts may be missing. */
|
|
63
|
+
readonly history: "complete" | "shallow";
|
|
64
|
+
readonly reverts: RevertRef[];
|
|
65
|
+
/** Repository paths of contributor guidelines, nearest to the changed files first. */
|
|
66
|
+
readonly guidelines: string[];
|
|
67
|
+
readonly conventions: PeerConvention[];
|
|
68
|
+
}
|
|
69
|
+
export declare const MAX_ORIGINS_PER_HUNK = 3;
|
|
70
|
+
export declare const MAX_REVERTS = 5;
|
|
71
|
+
export declare const MAX_GUIDELINES = 10;
|
|
72
|
+
export declare const MAX_CONVENTIONS = 5;
|
|
73
|
+
export declare const MAX_PEERS = 300;
|
|
74
|
+
/** Commits scanned for reverts, newest first from the base. */
|
|
75
|
+
export declare const REVERT_SCAN_COMMITS = 5000;
|
|
76
|
+
/** Files blamed per review; the rest carry no history. */
|
|
77
|
+
export declare const MAX_BLAMED_FILES = 60;
|
|
78
|
+
/** Old-side line numbers of the lines one hunk removes. */
|
|
79
|
+
export declare function removedLines(unit: ReviewUnit): number[];
|
|
80
|
+
interface BlameCommit {
|
|
81
|
+
date: string;
|
|
82
|
+
subject: string;
|
|
83
|
+
boundary: boolean;
|
|
84
|
+
}
|
|
85
|
+
/** One `git blame --porcelain` run: origin commit per final line, and each commit's metadata. */
|
|
86
|
+
export interface ParsedBlame {
|
|
87
|
+
readonly byLine: Map<number, string>;
|
|
88
|
+
readonly commits: Map<string, BlameCommit>;
|
|
89
|
+
}
|
|
90
|
+
/** Origin commit per final line number, from `git blame --porcelain`. */
|
|
91
|
+
export declare function parseBlame(output: string): ParsedBlame;
|
|
92
|
+
/** Distinctive word stems in free text or a file name, split on case and punctuation. */
|
|
93
|
+
export declare function topicStems(text: string): Set<string>;
|
|
94
|
+
/** Guideline paths in the head tree, nearest to the changed files first. */
|
|
95
|
+
export declare function guidelinePaths(tree: readonly string[], changed: readonly string[]): string[];
|
|
96
|
+
/**
|
|
97
|
+
* Read the project context for a range and attach each hunk's line history to
|
|
98
|
+
* its unit. Each part fails on its own: a part that cannot be read is empty.
|
|
99
|
+
*/
|
|
100
|
+
export declare function readProjectContext(cwd: string, base: string, head: string, units: ReviewUnit[], goal?: string): ProjectContext;
|
|
101
|
+
export {};
|
|
@@ -0,0 +1,412 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Project context for a git-range review, read from the local repository only.
|
|
3
|
+
*
|
|
4
|
+
* What a diff cannot show a reviewer is often the history and habits around it:
|
|
5
|
+
* the removed lines came from a fix, a similar change was reverted before, the
|
|
6
|
+
* project writes down rules for contributors, or every sibling file follows a
|
|
7
|
+
* pattern this one does not. All four are read here with plain git commands on
|
|
8
|
+
* the reviewed snapshots — nothing is fetched, checked out, or written — and are
|
|
9
|
+
* reported as pointers for the reviewer and the agent, never as verdicts. Commit
|
|
10
|
+
* subjects are repository text: data to show, not instructions to follow.
|
|
11
|
+
*
|
|
12
|
+
* Every selection is deterministic: the same repository and range yield the same
|
|
13
|
+
* context. A shallow clone cuts history, and lines whose origin lies past the cut
|
|
14
|
+
* are counted as unknown rather than attributed to the boundary commit.
|
|
15
|
+
*/
|
|
16
|
+
import { execFileSync } from "node:child_process";
|
|
17
|
+
import { posix } from "node:path";
|
|
18
|
+
import { testLikeFile } from "./file-role.js";
|
|
19
|
+
export const MAX_ORIGINS_PER_HUNK = 3;
|
|
20
|
+
export const MAX_REVERTS = 5;
|
|
21
|
+
export const MAX_GUIDELINES = 10;
|
|
22
|
+
export const MAX_CONVENTIONS = 5;
|
|
23
|
+
export const MAX_PEERS = 300;
|
|
24
|
+
/** Commits scanned for reverts, newest first from the base. */
|
|
25
|
+
export const REVERT_SCAN_COMMITS = 5000;
|
|
26
|
+
/** Files blamed per review; the rest carry no history. */
|
|
27
|
+
export const MAX_BLAMED_FILES = 60;
|
|
28
|
+
const MIN_PEERS = 4;
|
|
29
|
+
/** A peer convention is an identifier at least this share of peers use. */
|
|
30
|
+
const PEER_SHARE = 0.6;
|
|
31
|
+
const MAX_COMMON_NAMES = 8;
|
|
32
|
+
const GIT_TIMEOUT_MS = 30_000;
|
|
33
|
+
const NOTABLE_SUBJECT = /\b(?:fix(?:e[sd])?|bug|workaround|work around|regression|revert(?:s|ed)?|hotfix|security|cve-\d+|vulnerab\w*|msrv|compat\w*|crash\w*|panic\w*|race|deadlock\w*|leak\w*)\b/i;
|
|
34
|
+
/** A revert commit: `Revert "…"`, or `area: revert …` / `[area] Revert …` after a prefix. */
|
|
35
|
+
const REVERT_SUBJECT = /(?:^|:\s*|\]\s*)revert(?:s|ing)?\b/i;
|
|
36
|
+
/** Routine fixes that do not make removed lines worth a question. */
|
|
37
|
+
const ROUTINE_SUBJECT = /\b(?:typos?|spelling|format(?:ting)?|fmt|rustfmt|prettier|lint(?:s|ing)?|clippy|whitespace|docs?|comments?|style|warnings?)\b/i;
|
|
38
|
+
/** A shared word relates two subjects only if at most this share of scanned subjects use it. */
|
|
39
|
+
const RARE_TERM_SHARE = 0.005;
|
|
40
|
+
/** Contributor guideline file names, anywhere in the tree. */
|
|
41
|
+
const GUIDELINE_NAME = /^(?:contributing|contribute|contributors?[-_ ]?guide|agents|claude|copilot-instructions|style(?:[-_ ]?guide)?|code[-_ ]?style|coding[-_ ]?(?:style|standards|guidelines)|guidelines?|reviewing|review[-_ ]?guidelines|development|developing|hacking|conventions)(?:\.(?:md|rst|txt|adoc))?$/i;
|
|
42
|
+
/** Policy documents under a docs directory. */
|
|
43
|
+
const DOCS_POLICY = /(?:^|\/)docs?\/(?:[^/]+\/)*[^/]*(?:contribut|guideline|style|convention|review|policy|versioning|preview|deprecat|compatib)[^/]*\.(?:md|rst|adoc)$/i;
|
|
44
|
+
/** Words too common in commit subjects and file names to relate two changes. */
|
|
45
|
+
const STOP_WORDS = new Set([
|
|
46
|
+
"the", "and", "for", "with", "from", "into", "onto", "that", "this", "when", "then", "than", "also", "only",
|
|
47
|
+
"add", "adds", "added", "use", "uses", "used", "make", "update", "change", "remove", "removed", "revert",
|
|
48
|
+
"reverted", "reverts", "fix", "fixes", "fixed", "test", "tests", "mod", "lib", "main", "index", "init", "src",
|
|
49
|
+
"util", "utils", "more", "less", "support", "new", "allow", "instead", "related", "changes", "commit", "merge",
|
|
50
|
+
"pull", "request", "branch", "file", "files", "code", "docs", "readme", "version", "bump", "move", "rename",
|
|
51
|
+
"some", "type", "types", "value", "values", "error", "errors", "case", "cases", "default", "option", "options",
|
|
52
|
+
"before", "after", "while", "without", "because", "through", "over", "under", "about", "again", "back", "other",
|
|
53
|
+
]);
|
|
54
|
+
function git(cwd, args, input) {
|
|
55
|
+
return execFileSync("git", ["--no-replace-objects", "--no-pager", ...args], {
|
|
56
|
+
cwd, input, encoding: "utf8", maxBuffer: 64 * 1024 * 1024, timeout: GIT_TIMEOUT_MS, stdio: ["pipe", "pipe", "pipe"],
|
|
57
|
+
});
|
|
58
|
+
}
|
|
59
|
+
/** Locale-independent order, so the same repository always sorts the same way. */
|
|
60
|
+
function byCodePoint(a, b) {
|
|
61
|
+
return a < b ? -1 : a > b ? 1 : 0;
|
|
62
|
+
}
|
|
63
|
+
function utcDate(epochSeconds) {
|
|
64
|
+
return new Date(epochSeconds * 1000).toISOString().slice(0, 10);
|
|
65
|
+
}
|
|
66
|
+
/** Old-side line numbers of the lines one hunk removes. */
|
|
67
|
+
export function removedLines(unit) {
|
|
68
|
+
const lines = [];
|
|
69
|
+
let old = unit.oldStart;
|
|
70
|
+
for (const line of unit.diff.split("\n").slice(1)) {
|
|
71
|
+
if (line.startsWith("-"))
|
|
72
|
+
lines.push(old++);
|
|
73
|
+
else if (line.startsWith(" "))
|
|
74
|
+
old++;
|
|
75
|
+
}
|
|
76
|
+
return lines;
|
|
77
|
+
}
|
|
78
|
+
function ranges(lines) {
|
|
79
|
+
const out = [];
|
|
80
|
+
for (const line of [...lines].sort((a, b) => a - b)) {
|
|
81
|
+
const last = out.at(-1);
|
|
82
|
+
if (last !== undefined && line <= last[1] + 1)
|
|
83
|
+
last[1] = Math.max(last[1], line);
|
|
84
|
+
else
|
|
85
|
+
out.push([line, line]);
|
|
86
|
+
}
|
|
87
|
+
return out;
|
|
88
|
+
}
|
|
89
|
+
/** Origin commit per final line number, from `git blame --porcelain`. */
|
|
90
|
+
export function parseBlame(output) {
|
|
91
|
+
const byLine = new Map();
|
|
92
|
+
const commits = new Map();
|
|
93
|
+
let current;
|
|
94
|
+
for (const line of output.split("\n")) {
|
|
95
|
+
const header = /^([0-9a-f]{40}) \d+ (\d+)(?: \d+)?$/.exec(line);
|
|
96
|
+
if (header !== null) {
|
|
97
|
+
current = header[1];
|
|
98
|
+
byLine.set(Number(header[2]), current);
|
|
99
|
+
if (!commits.has(current))
|
|
100
|
+
commits.set(current, { date: "", subject: "", boundary: false });
|
|
101
|
+
continue;
|
|
102
|
+
}
|
|
103
|
+
if (current === undefined || line.startsWith("\t"))
|
|
104
|
+
continue;
|
|
105
|
+
const commit = commits.get(current);
|
|
106
|
+
if (line.startsWith("author-time "))
|
|
107
|
+
commit.date = utcDate(Number(line.slice(12)));
|
|
108
|
+
else if (line.startsWith("summary "))
|
|
109
|
+
commit.subject = line.slice(8);
|
|
110
|
+
else if (line === "boundary")
|
|
111
|
+
commit.boundary = true;
|
|
112
|
+
}
|
|
113
|
+
return { byLine, commits };
|
|
114
|
+
}
|
|
115
|
+
/** Where the lines each hunk removes came from, keyed by unit id. */
|
|
116
|
+
function hunkHistories(cwd, base, units) {
|
|
117
|
+
const byFile = new Map();
|
|
118
|
+
for (const unit of units) {
|
|
119
|
+
if (unit.special !== undefined)
|
|
120
|
+
continue;
|
|
121
|
+
const lines = removedLines(unit);
|
|
122
|
+
if (lines.length === 0)
|
|
123
|
+
continue;
|
|
124
|
+
const list = byFile.get(unit.file) ?? [];
|
|
125
|
+
list.push({ unit, lines });
|
|
126
|
+
byFile.set(unit.file, list);
|
|
127
|
+
}
|
|
128
|
+
const out = new Map();
|
|
129
|
+
for (const [file, hunks] of [...byFile].slice(0, MAX_BLAMED_FILES)) {
|
|
130
|
+
let blame;
|
|
131
|
+
try {
|
|
132
|
+
const args = ranges(hunks.flatMap((hunk) => hunk.lines)).flatMap(([a, b]) => ["-L", `${a},${b}`]);
|
|
133
|
+
blame = parseBlame(git(cwd, ["blame", "--porcelain", ...args, base, "--", file]));
|
|
134
|
+
}
|
|
135
|
+
catch {
|
|
136
|
+
// A file absent at the base (renamed or copied) has no line history here.
|
|
137
|
+
continue;
|
|
138
|
+
}
|
|
139
|
+
for (const { unit, lines } of hunks) {
|
|
140
|
+
const counts = new Map();
|
|
141
|
+
let unknownLines = 0;
|
|
142
|
+
for (const line of lines) {
|
|
143
|
+
const sha = blame.byLine.get(line);
|
|
144
|
+
const commit = sha === undefined ? undefined : blame.commits.get(sha);
|
|
145
|
+
if (sha === undefined || commit === undefined || commit.boundary)
|
|
146
|
+
unknownLines += 1;
|
|
147
|
+
else
|
|
148
|
+
counts.set(sha, (counts.get(sha) ?? 0) + 1);
|
|
149
|
+
}
|
|
150
|
+
const origins = [...counts]
|
|
151
|
+
.map(([sha, count]) => {
|
|
152
|
+
const commit = blame.commits.get(sha);
|
|
153
|
+
const notable = NOTABLE_SUBJECT.test(commit.subject) && !ROUTINE_SUBJECT.test(commit.subject);
|
|
154
|
+
return { commit: sha.slice(0, 12), date: commit.date, subject: commit.subject, lines: count, notable };
|
|
155
|
+
})
|
|
156
|
+
.sort((a, b) => b.lines - a.lines || byCodePoint(b.date, a.date) || byCodePoint(a.commit, b.commit))
|
|
157
|
+
.slice(0, MAX_ORIGINS_PER_HUNK);
|
|
158
|
+
if (origins.length > 0 || unknownLines > 0)
|
|
159
|
+
out.set(unit.id, { origins, unknownLines });
|
|
160
|
+
}
|
|
161
|
+
}
|
|
162
|
+
return out;
|
|
163
|
+
}
|
|
164
|
+
/** A crude, fixed stemmer: enough to relate `sharding`, `sharded`, and `shard`. */
|
|
165
|
+
function stem(word) {
|
|
166
|
+
const lower = word.toLowerCase();
|
|
167
|
+
for (const suffix of ["ing", "ed"]) {
|
|
168
|
+
if (lower.endsWith(suffix) && lower.length - suffix.length >= 4)
|
|
169
|
+
return lower.slice(0, -suffix.length);
|
|
170
|
+
}
|
|
171
|
+
if (lower.endsWith("s") && !lower.endsWith("ss") && lower.length - 1 >= 4)
|
|
172
|
+
return lower.slice(0, -1);
|
|
173
|
+
return lower;
|
|
174
|
+
}
|
|
175
|
+
/** Distinctive word stems in free text or a file name, split on case and punctuation. */
|
|
176
|
+
export function topicStems(text) {
|
|
177
|
+
const words = text.replace(/([a-z0-9])([A-Z])/g, "$1 $2").split(/[^A-Za-z0-9]+/);
|
|
178
|
+
const stems = new Set();
|
|
179
|
+
for (const word of words) {
|
|
180
|
+
if (word.length < 4 || /^\d+$/.test(word) || STOP_WORDS.has(word.toLowerCase()))
|
|
181
|
+
continue;
|
|
182
|
+
const stemmed = stem(word);
|
|
183
|
+
if (!STOP_WORDS.has(stemmed))
|
|
184
|
+
stems.add(stemmed);
|
|
185
|
+
}
|
|
186
|
+
return stems;
|
|
187
|
+
}
|
|
188
|
+
/** The subject a revert names, without the revert wording, for matching. */
|
|
189
|
+
function revertedTopic(subject) {
|
|
190
|
+
const quoted = /"([^"]+)"/.exec(subject);
|
|
191
|
+
return quoted?.[1] ?? subject.replace(REVERT_SUBJECT, " ");
|
|
192
|
+
}
|
|
193
|
+
function reverts(cwd, base, files, goal) {
|
|
194
|
+
const format = "--format=%H%x1f%at%x1f%s";
|
|
195
|
+
const parse = (line) => {
|
|
196
|
+
const [sha, time, subject] = line.split("\x1f");
|
|
197
|
+
return { sha, ref: { commit: sha.slice(0, 12), date: utcDate(Number(time)), subject } };
|
|
198
|
+
};
|
|
199
|
+
const found = [];
|
|
200
|
+
const seen = new Set();
|
|
201
|
+
// Reverts that touched a file this diff changes, newest first.
|
|
202
|
+
if (files.length > 0) {
|
|
203
|
+
const touched = git(cwd, ["log", "-n", "200", "-i", "-E", "--grep=\\brevert", format, base, "--", ...files]);
|
|
204
|
+
for (const line of touched.split("\n").filter(Boolean)) {
|
|
205
|
+
const { sha, ref } = parse(line);
|
|
206
|
+
if (!REVERT_SUBJECT.test(ref.subject) || seen.has(sha))
|
|
207
|
+
continue;
|
|
208
|
+
const file = git(cwd, ["diff-tree", "--no-commit-id", "--name-only", "-r", "--root", sha, "--", ...files]).split("\n").filter(Boolean).sort()[0];
|
|
209
|
+
seen.add(sha);
|
|
210
|
+
found.push({ ...ref, reason: { kind: "file", file: file ?? files[0] } });
|
|
211
|
+
}
|
|
212
|
+
}
|
|
213
|
+
// Reverts anywhere whose subject shares a rare word with the goal or the changed file names.
|
|
214
|
+
// A word most subjects use (the project's name, a common area prefix) relates nothing.
|
|
215
|
+
const topic = topicStems(`${goal} ${files.map((file) => posix.basename(file).replace(/\.[^.]*$/, "")).join(" ")}`);
|
|
216
|
+
if (topic.size > 0) {
|
|
217
|
+
const scanned = git(cwd, ["log", "-n", String(REVERT_SCAN_COMMITS), format, base]).split("\n").filter(Boolean).map(parse);
|
|
218
|
+
const frequency = new Map();
|
|
219
|
+
for (const { ref } of scanned)
|
|
220
|
+
for (const word of topicStems(ref.subject))
|
|
221
|
+
frequency.set(word, (frequency.get(word) ?? 0) + 1);
|
|
222
|
+
const rare = Math.max(3, Math.floor(scanned.length * RARE_TERM_SHARE));
|
|
223
|
+
for (const { sha, ref } of scanned) {
|
|
224
|
+
if (seen.has(sha) || !REVERT_SUBJECT.test(ref.subject))
|
|
225
|
+
continue;
|
|
226
|
+
// The rarest shared word names the relation best.
|
|
227
|
+
const term = [...topicStems(revertedTopic(ref.subject))]
|
|
228
|
+
.filter((word) => topic.has(word) && (frequency.get(word) ?? 0) <= rare)
|
|
229
|
+
.sort((a, b) => (frequency.get(a) ?? 0) - (frequency.get(b) ?? 0) || byCodePoint(a, b))[0];
|
|
230
|
+
if (term === undefined)
|
|
231
|
+
continue;
|
|
232
|
+
seen.add(sha);
|
|
233
|
+
found.push({ ...ref, reason: { kind: "term", term } });
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
return found.slice(0, MAX_REVERTS);
|
|
237
|
+
}
|
|
238
|
+
/** Guideline paths in the head tree, nearest to the changed files first. */
|
|
239
|
+
export function guidelinePaths(tree, changed) {
|
|
240
|
+
const ancestors = new Map();
|
|
241
|
+
for (const file of changed) {
|
|
242
|
+
const parts = file.split("/").slice(0, -1);
|
|
243
|
+
for (let depth = parts.length; depth >= 0; depth -= 1) {
|
|
244
|
+
const dir = parts.slice(0, depth).join("/");
|
|
245
|
+
ancestors.set(dir, Math.max(ancestors.get(dir) ?? -1, depth));
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
const rank = (path) => {
|
|
249
|
+
const dir = posix.dirname(path) === "." ? "" : posix.dirname(path);
|
|
250
|
+
const depth = ancestors.get(dir);
|
|
251
|
+
if (depth !== undefined)
|
|
252
|
+
return 1000 - depth; // guidelines next to the change first, the root last
|
|
253
|
+
if (path.startsWith(".github/"))
|
|
254
|
+
return 2000;
|
|
255
|
+
return 3000;
|
|
256
|
+
};
|
|
257
|
+
// A guideline applies when it sits in a directory above a changed file, or is
|
|
258
|
+
// project-wide (.github, a docs policy page); another package's guide does not.
|
|
259
|
+
const applies = (path) => {
|
|
260
|
+
const dir = posix.dirname(path) === "." ? "" : posix.dirname(path);
|
|
261
|
+
if (GUIDELINE_NAME.test(posix.basename(path)))
|
|
262
|
+
return ancestors.has(dir) || path.startsWith(".github/");
|
|
263
|
+
return DOCS_POLICY.test(path);
|
|
264
|
+
};
|
|
265
|
+
return tree
|
|
266
|
+
.filter((path) => applies(path) && !testLikeFile(path))
|
|
267
|
+
.sort((a, b) => rank(a) - rank(b) || byCodePoint(a, b))
|
|
268
|
+
.slice(0, MAX_GUIDELINES);
|
|
269
|
+
}
|
|
270
|
+
/** Identifier-like words: CamelCase inside, snake_case, or long names; never keywords. */
|
|
271
|
+
const IDENTIFIER = /[A-Za-z_][A-Za-z0-9_]*/g;
|
|
272
|
+
function identifiers(text) {
|
|
273
|
+
const names = new Set();
|
|
274
|
+
for (const [name] of text.matchAll(IDENTIFIER)) {
|
|
275
|
+
if (name.length < 4)
|
|
276
|
+
continue;
|
|
277
|
+
if (/[a-z][A-Z]/.test(name) || (/_/.test(name) && /[A-Za-z]{2}/.test(name)) || name.length >= 12)
|
|
278
|
+
names.add(name);
|
|
279
|
+
}
|
|
280
|
+
return names;
|
|
281
|
+
}
|
|
282
|
+
/** Contents of `<rev>:<path>` blobs, in one `git cat-file --batch` call. */
|
|
283
|
+
function readBlobs(cwd, rev, paths) {
|
|
284
|
+
const out = new Map();
|
|
285
|
+
if (paths.length === 0)
|
|
286
|
+
return out;
|
|
287
|
+
const raw = execFileSync("git", ["--no-replace-objects", "cat-file", "--batch"], {
|
|
288
|
+
cwd, input: paths.map((path) => `${rev}:${path}`).join("\n") + "\n", maxBuffer: 256 * 1024 * 1024, timeout: GIT_TIMEOUT_MS,
|
|
289
|
+
});
|
|
290
|
+
let offset = 0;
|
|
291
|
+
for (const path of paths) {
|
|
292
|
+
const end = raw.indexOf(0x0a, offset);
|
|
293
|
+
const header = raw.subarray(offset, end).toString("utf8");
|
|
294
|
+
offset = end + 1;
|
|
295
|
+
const size = /^[0-9a-f]+ blob (\d+)$/.exec(header);
|
|
296
|
+
if (size === null)
|
|
297
|
+
continue;
|
|
298
|
+
out.set(path, raw.subarray(offset, offset + Number(size[1])).toString("utf8"));
|
|
299
|
+
offset += Number(size[1]) + 1;
|
|
300
|
+
}
|
|
301
|
+
return out;
|
|
302
|
+
}
|
|
303
|
+
/**
|
|
304
|
+
* For each new file, the nearest set of sibling files with the same name
|
|
305
|
+
* (`<prefix>/*\/<rest>`) and the identifiers most of them use that it does not.
|
|
306
|
+
*/
|
|
307
|
+
function conventions(cwd, head, tree, added) {
|
|
308
|
+
const treeSet = new Set(tree);
|
|
309
|
+
const out = [];
|
|
310
|
+
for (const file of added) {
|
|
311
|
+
if (out.length >= MAX_CONVENTIONS)
|
|
312
|
+
break;
|
|
313
|
+
if (testLikeFile(file) || !treeSet.has(file))
|
|
314
|
+
continue;
|
|
315
|
+
const parts = file.split("/");
|
|
316
|
+
for (let wild = parts.length - 2; wild >= 0; wild -= 1) {
|
|
317
|
+
const prefix = parts.slice(0, wild).join("/");
|
|
318
|
+
const rest = parts.slice(wild + 1).join("/");
|
|
319
|
+
const peers = tree
|
|
320
|
+
.filter((path) => path !== file && path.endsWith(`/${rest}`) && (prefix === "" || path.startsWith(`${prefix}/`)))
|
|
321
|
+
.filter((path) => path.split("/").length === parts.length && !added.includes(path))
|
|
322
|
+
.sort()
|
|
323
|
+
.slice(0, MAX_PEERS);
|
|
324
|
+
if (peers.length < MIN_PEERS)
|
|
325
|
+
continue;
|
|
326
|
+
const blobs = readBlobs(cwd, head, [file, ...peers]);
|
|
327
|
+
const own = identifiers(blobs.get(file) ?? "");
|
|
328
|
+
const counts = new Map();
|
|
329
|
+
let read = 0;
|
|
330
|
+
for (const peer of peers) {
|
|
331
|
+
const text = blobs.get(peer);
|
|
332
|
+
if (text === undefined)
|
|
333
|
+
continue;
|
|
334
|
+
read += 1;
|
|
335
|
+
for (const name of identifiers(text))
|
|
336
|
+
counts.set(name, (counts.get(name) ?? 0) + 1);
|
|
337
|
+
}
|
|
338
|
+
const common = [...counts]
|
|
339
|
+
.filter(([name, count]) => !own.has(name) && count >= Math.max(MIN_PEERS, Math.ceil(read * PEER_SHARE)))
|
|
340
|
+
.sort((a, b) => b[1] - a[1] || byCodePoint(a[0], b[0]))
|
|
341
|
+
.slice(0, MAX_COMMON_NAMES)
|
|
342
|
+
.map(([name, count]) => ({ name, peers: count }));
|
|
343
|
+
const pattern = `${prefix === "" ? "" : `${prefix}/`}*/${rest}`;
|
|
344
|
+
if (common.length > 0)
|
|
345
|
+
out.push({ file, pattern, peers: read, common });
|
|
346
|
+
break;
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
return out;
|
|
350
|
+
}
|
|
351
|
+
/** New files: hunks that start at old line 0 and remove nothing. */
|
|
352
|
+
function addedFiles(units) {
|
|
353
|
+
const byFile = new Map();
|
|
354
|
+
for (const unit of units) {
|
|
355
|
+
const isNew = unit.special === undefined && unit.oldStart === 0 && unit.removed === 0;
|
|
356
|
+
byFile.set(unit.file, (byFile.get(unit.file) ?? true) && isNew);
|
|
357
|
+
}
|
|
358
|
+
return [...byFile].filter(([, isNew]) => isNew).map(([file]) => file).sort();
|
|
359
|
+
}
|
|
360
|
+
/**
|
|
361
|
+
* Read the project context for a range and attach each hunk's line history to
|
|
362
|
+
* its unit. Each part fails on its own: a part that cannot be read is empty.
|
|
363
|
+
*/
|
|
364
|
+
export function readProjectContext(cwd, base, head, units, goal = "") {
|
|
365
|
+
const shallow = (() => {
|
|
366
|
+
try {
|
|
367
|
+
return git(cwd, ["rev-parse", "--is-shallow-repository"]).trim() === "true";
|
|
368
|
+
}
|
|
369
|
+
catch {
|
|
370
|
+
return false;
|
|
371
|
+
}
|
|
372
|
+
})();
|
|
373
|
+
try {
|
|
374
|
+
const histories = hunkHistories(cwd, base, units);
|
|
375
|
+
for (const unit of units) {
|
|
376
|
+
const history = histories.get(unit.id);
|
|
377
|
+
if (history !== undefined)
|
|
378
|
+
unit.history = history;
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
catch {
|
|
382
|
+
// Line history is a pointer; its absence is reported by `history` alone.
|
|
383
|
+
}
|
|
384
|
+
const files = [...new Set(units.filter((unit) => unit.special === undefined).map((unit) => unit.file))].sort();
|
|
385
|
+
let revertList = [];
|
|
386
|
+
try {
|
|
387
|
+
revertList = reverts(cwd, base, files.filter((file) => !addedFiles(units).includes(file)), goal);
|
|
388
|
+
}
|
|
389
|
+
catch {
|
|
390
|
+
revertList = [];
|
|
391
|
+
}
|
|
392
|
+
let tree = [];
|
|
393
|
+
try {
|
|
394
|
+
tree = git(cwd, ["ls-tree", "-r", "--name-only", "-z", head]).split("\0").filter(Boolean);
|
|
395
|
+
}
|
|
396
|
+
catch {
|
|
397
|
+
tree = [];
|
|
398
|
+
}
|
|
399
|
+
let conventionList = [];
|
|
400
|
+
try {
|
|
401
|
+
conventionList = conventions(cwd, head, tree, addedFiles(units));
|
|
402
|
+
}
|
|
403
|
+
catch {
|
|
404
|
+
conventionList = [];
|
|
405
|
+
}
|
|
406
|
+
return {
|
|
407
|
+
history: shallow ? "shallow" : "complete",
|
|
408
|
+
reverts: revertList,
|
|
409
|
+
guidelines: guidelinePaths(tree, files),
|
|
410
|
+
conventions: conventionList,
|
|
411
|
+
};
|
|
412
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
import type { ReviewReport } from "./types.js";
|
|
2
|
+
/**
|
|
3
|
+
* Render a review report as one self-contained HTML document.
|
|
4
|
+
*
|
|
5
|
+
* The document opens on the expected outcome: the pull request text, the intent
|
|
6
|
+
* cross-check, the checks that ran with their limits, and a short reading agenda
|
|
7
|
+
* whose top cards link the changed code, its direct caller or its contract. The
|
|
8
|
+
* call flow and the full diff are the other two views of the same document, so a
|
|
9
|
+
* reviewer reaches a hunk from the agenda in one click and every hunk stays
|
|
10
|
+
* reachable, including the ones no evaluation covers.
|
|
11
|
+
*
|
|
12
|
+
* The page is still a plain diff review underneath: file header, hunks with line
|
|
13
|
+
* numbers and green/red lines, ranking deciding order and severity reaching the
|
|
14
|
+
* reviewer as color. Ranking scores, model probabilities and the adapter's own
|
|
15
|
+
* rubric text stay in the JSON sidecar: the only model-derived content on the
|
|
16
|
+
* page is the hunk's own single-sample typed observations, printed from closed
|
|
17
|
+
* sets under a label that says what they are.
|
|
18
|
+
*
|
|
19
|
+
* Everything is server-rendered, so the report is readable with JavaScript
|
|
20
|
+
* disabled; without the script the three views stack as sections and every hunk
|
|
21
|
+
* is one disclosure away. The single inline script is progressive enhancement:
|
|
22
|
+
* view switching, expand and collapse, status filters that keep their fold
|
|
23
|
+
* state, focus mode, and a keyboard cursor. No report string is placed in the
|
|
24
|
+
* script; it reads labels from the DOM. Every string from a diff, path, pull
|
|
25
|
+
* request body or report field is HTML-escaped.
|
|
26
|
+
*/
|
|
27
|
+
export declare function renderReview(report: ReviewReport): string;
|
|
28
|
+
/**
|
|
29
|
+
* The call flows of one review as a page of their own, for the connected pull
|
|
30
|
+
* request page to show beside its diff: every changed file with call flows, or
|
|
31
|
+
* only `file` when given. Tree, Graph and Sequence work as in the report; links
|
|
32
|
+
* into the report's diff are dropped, since the page showing this has its own.
|
|
33
|
+
*/
|
|
34
|
+
export declare function renderCallFlowPage(report: ReviewReport, file?: string): string;
|