argus-reviewer-e2e 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +84 -71
- package/action/action.yml +134 -10
- package/action/approval-review.mjs +13 -3
- package/action/bootstrap.mjs +4 -0
- package/action/emit-review.mjs +16 -0
- package/action/runtime.mjs +32 -0
- package/action/sticky-comment.cjs +1260 -479
- package/dist/cli.d.ts +103 -7
- package/dist/cli.js +1208 -186
- package/dist/config.d.ts +96 -11
- package/dist/config.js +102 -4
- package/dist/detect.d.ts +29 -2
- package/dist/detect.js +98 -7
- package/dist/driver/browser.d.ts +32 -0
- package/dist/driver/browser.js +56 -1
- package/dist/driver/target.d.ts +4 -1
- package/dist/driver/target.js +27 -6
- package/dist/engine/actions.d.ts +5 -0
- package/dist/engine/actions.js +8 -0
- package/dist/engine/explore.d.ts +78 -0
- package/dist/engine/explore.js +373 -0
- package/dist/engine/loop.d.ts +2 -2
- package/dist/engine/loop.js +8 -8
- package/dist/engine/prompts.d.ts +28 -1
- package/dist/engine/prompts.js +88 -0
- package/dist/evidence/ci.d.ts +13 -1
- package/dist/evidence/ci.js +38 -3
- package/dist/evidence/gate.d.ts +8 -0
- package/dist/evidence/gate.js +1 -1
- package/dist/evidence/link.js +1 -1
- package/dist/executor/a0.d.ts +114 -1
- package/dist/executor/a0.js +216 -4
- package/dist/fsutil.d.ts +3 -2
- package/dist/fsutil.js +7 -4
- package/dist/journal/schema.d.ts +1 -1
- package/dist/log.d.ts +2 -1
- package/dist/log.js +10 -2
- package/dist/mention.d.ts +45 -0
- package/dist/mention.js +107 -0
- package/dist/pipeline/app.d.ts +126 -0
- package/dist/pipeline/app.js +250 -0
- package/dist/pipeline/budget.d.ts +1 -0
- package/dist/pipeline/budget.js +1 -1
- package/dist/pipeline/verify.d.ts +20 -3
- package/dist/pipeline/verify.js +189 -35
- package/dist/probe/persist.d.ts +68 -0
- package/dist/probe/persist.js +184 -0
- package/dist/probe/queue.d.ts +12 -0
- package/dist/probe/queue.js +10 -2
- package/dist/report/brand-assets.generated.d.ts +9 -0
- package/dist/report/brand-assets.generated.js +8 -0
- package/dist/report/comment.d.ts +99 -6
- package/dist/report/comment.js +292 -103
- package/dist/report/html.d.ts +50 -0
- package/dist/report/html.js +879 -0
- package/dist/report/manifest.d.ts +29 -0
- package/dist/report/manifest.js +37 -0
- package/dist/report/run.d.ts +54 -1
- package/dist/report/run.js +34 -9
- package/dist/report/viewmodel.d.ts +91 -0
- package/dist/report/viewmodel.js +241 -0
- package/dist/review/adjudicate.d.ts +6 -6
- package/dist/review/adjudicate.js +2 -2
- package/dist/review/inline.d.ts +44 -0
- package/dist/review/inline.js +95 -0
- package/dist/review/packs.d.ts +21 -0
- package/dist/review/packs.js +47 -0
- package/dist/review/scope.d.ts +16 -0
- package/dist/review/scope.js +74 -0
- package/dist/review/secrets.d.ts +10 -10
- package/dist/review/secrets.js +7 -7
- package/dist/review/testfiles.d.ts +18 -0
- package/dist/review/testfiles.js +26 -0
- package/dist/review/triage.d.ts +1 -1
- package/dist/review/triage.js +10 -10
- package/dist/review/validate.d.ts +41 -0
- package/dist/review/validate.js +76 -0
- package/dist/ui/errors.d.ts +54 -0
- package/dist/ui/errors.js +236 -0
- package/dist/ui/style.d.ts +34 -0
- package/dist/ui/style.js +48 -0
- package/dist/ui/summary.d.ts +38 -0
- package/dist/ui/summary.js +101 -0
- package/dist/vision/cost.d.ts +1 -1
- package/dist/vision/decisions.d.ts +9 -3
- package/dist/vision/decisions.js +31 -21
- package/dist/vision/openrouter.d.ts +4 -0
- package/dist/vision/openrouter.js +30 -4
- package/package.json +11 -2
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Inline review comment identity (plan KTD4, U6).
|
|
3
|
+
*
|
|
4
|
+
* The CLI renders each inline comment once and serializes its dedup key; the
|
|
5
|
+
* action poster rebuilds the same key from comments already on the PR. Both
|
|
6
|
+
* sides go through `inlineDedupKey`, and action/sticky-comment.cjs keeps a
|
|
7
|
+
* copy of this file's functions (KTD2), pinned equal by a parity test.
|
|
8
|
+
*
|
|
9
|
+
* Two body formats exist on real PRs:
|
|
10
|
+
* legacy (before U6): `**argus-reviewer <sev>:** <message> \`<category>\``
|
|
11
|
+
* Ocellus (U6): `<!-- argus-reviewer:inline -->`
|
|
12
|
+
* `<glyph> **<severity word>** · <proof meter> <level>`
|
|
13
|
+
* `<message>`
|
|
14
|
+
* A comment posted in either format keys to the same value for the same
|
|
15
|
+
* finding, so upgrading never re-posts every inline comment.
|
|
16
|
+
*/
|
|
17
|
+
/** Hidden first line of every Ocellus inline comment: marks it as Argus's own. */
|
|
18
|
+
export declare const INLINE_SENTINEL = "<!-- argus-reviewer:inline -->";
|
|
19
|
+
/**
|
|
20
|
+
* Strip the prefix the review prompt asks the model for (`L42: <emoji> bug: `)
|
|
21
|
+
* in any order, so the rendered message line starts with the sentence.
|
|
22
|
+
*/
|
|
23
|
+
export declare function normalizeFindingMessage(message: string): string;
|
|
24
|
+
/** Severity and message of an Argus inline comment in either format; undefined for any other body. */
|
|
25
|
+
export declare function parseInlineBody(body: string): {
|
|
26
|
+
severity: string;
|
|
27
|
+
message: string;
|
|
28
|
+
} | undefined;
|
|
29
|
+
/** True when the body is an Argus inline comment (legacy prefix or the sentinel). */
|
|
30
|
+
export declare function isArgusInlineBody(body: string): boolean;
|
|
31
|
+
/** djb2 to 8 hex chars: dedup identity only, not a security boundary. */
|
|
32
|
+
export declare function shortHash(s: string): string;
|
|
33
|
+
/**
|
|
34
|
+
* The fenced suggestion block of a comment body. The CLI's fence is the
|
|
35
|
+
* longest backtick run plus one (min 4), so a run of exactly that length can
|
|
36
|
+
* only be the closing fence.
|
|
37
|
+
*/
|
|
38
|
+
export declare function extractSuggestion(body: string): string;
|
|
39
|
+
/**
|
|
40
|
+
* KTD4 dedup key: `path:line:severity:normalizedMessage:hash8(suggestion)`.
|
|
41
|
+
* A changed suggestion changes the key, so a corrected fix re-posts. A body
|
|
42
|
+
* that parses as neither format keeps the pre-U6 first-line key.
|
|
43
|
+
*/
|
|
44
|
+
export declare function inlineDedupKey(path: string, line: number, body: string): string;
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
import { SEVERITIES, SEVERITY_LABEL } from '../report/viewmodel.js';
|
|
2
|
+
/**
|
|
3
|
+
* Inline review comment identity (plan KTD4, U6).
|
|
4
|
+
*
|
|
5
|
+
* The CLI renders each inline comment once and serializes its dedup key; the
|
|
6
|
+
* action poster rebuilds the same key from comments already on the PR. Both
|
|
7
|
+
* sides go through `inlineDedupKey`, and action/sticky-comment.cjs keeps a
|
|
8
|
+
* copy of this file's functions (KTD2), pinned equal by a parity test.
|
|
9
|
+
*
|
|
10
|
+
* Two body formats exist on real PRs:
|
|
11
|
+
* legacy (before U6): `**argus-reviewer <sev>:** <message> \`<category>\``
|
|
12
|
+
* Ocellus (U6): `<!-- argus-reviewer:inline -->`
|
|
13
|
+
* `<glyph> **<severity word>** · <proof meter> <level>`
|
|
14
|
+
* `<message>`
|
|
15
|
+
* A comment posted in either format keys to the same value for the same
|
|
16
|
+
* finding, so upgrading never re-posts every inline comment.
|
|
17
|
+
*/
|
|
18
|
+
/** Hidden first line of every Ocellus inline comment: marks it as Argus's own. */
|
|
19
|
+
export const INLINE_SENTINEL = '<!-- argus-reviewer:inline -->';
|
|
20
|
+
const LEGACY_PREFIX = '**argus-reviewer';
|
|
21
|
+
const LEGACY_LINE = /^\*\*argus-reviewer ([^:*]+):\*\* ?(.*)$/;
|
|
22
|
+
const SEVERITY_LINE = /^(?:\S+ )?\*\*([^*]+)\*\* · /;
|
|
23
|
+
/** The category code span the legacy renderer appended to the message. */
|
|
24
|
+
const CATEGORY_SUFFIX = /\s*`(?:correctness|security|performance|usability|convention|other)`$/;
|
|
25
|
+
/** One model-message prefix: `L<n>[-<m>]:`, a leading emoji, or a `<severity>:` keyword. */
|
|
26
|
+
const MESSAGE_PREFIX = /^(?:L\d+(?:-\d+)?:|\p{Extended_Pictographic}\u{FE0F}?|(?:bug|risk|nit|q|question):)\s*/iu;
|
|
27
|
+
/**
|
|
28
|
+
* Strip the prefix the review prompt asks the model for (`L42: <emoji> bug: `)
|
|
29
|
+
* in any order, so the rendered message line starts with the sentence.
|
|
30
|
+
*/
|
|
31
|
+
export function normalizeFindingMessage(message) {
|
|
32
|
+
let out = message.trim();
|
|
33
|
+
for (let i = 0; i < 4; i++) {
|
|
34
|
+
const next = out.replace(MESSAGE_PREFIX, '');
|
|
35
|
+
if (next === out)
|
|
36
|
+
break;
|
|
37
|
+
out = next;
|
|
38
|
+
}
|
|
39
|
+
return out;
|
|
40
|
+
}
|
|
41
|
+
/** The message as the dedup key sees it: normalized, no category span, one-spaced. */
|
|
42
|
+
function keyMessage(message) {
|
|
43
|
+
return normalizeFindingMessage(message.replace(CATEGORY_SUFFIX, '')).replace(/\s+/g, ' ').trim();
|
|
44
|
+
}
|
|
45
|
+
const LABEL_TO_SEVERITY = new Map(SEVERITIES.map((s) => [SEVERITY_LABEL[s], s]));
|
|
46
|
+
/** Severity and message of an Argus inline comment in either format; undefined for any other body. */
|
|
47
|
+
export function parseInlineBody(body) {
|
|
48
|
+
const lines = body.split(/\r?\n/);
|
|
49
|
+
if (body.startsWith(LEGACY_PREFIX)) {
|
|
50
|
+
const m = LEGACY_LINE.exec(lines[0] ?? '');
|
|
51
|
+
if (m === null)
|
|
52
|
+
return undefined;
|
|
53
|
+
return { severity: (m[1] ?? '').trim(), message: keyMessage(m[2] ?? '') };
|
|
54
|
+
}
|
|
55
|
+
if (lines[0] === INLINE_SENTINEL) {
|
|
56
|
+
const m = SEVERITY_LINE.exec(lines[1] ?? '');
|
|
57
|
+
if (m === null)
|
|
58
|
+
return undefined;
|
|
59
|
+
const word = (m[1] ?? '').trim();
|
|
60
|
+
return { severity: LABEL_TO_SEVERITY.get(word) ?? word, message: keyMessage(lines[2] ?? '') };
|
|
61
|
+
}
|
|
62
|
+
return undefined;
|
|
63
|
+
}
|
|
64
|
+
/** True when the body is an Argus inline comment (legacy prefix or the sentinel). */
|
|
65
|
+
export function isArgusInlineBody(body) {
|
|
66
|
+
return body.startsWith(LEGACY_PREFIX) || body.startsWith(INLINE_SENTINEL);
|
|
67
|
+
}
|
|
68
|
+
/** djb2 to 8 hex chars: dedup identity only, not a security boundary. */
|
|
69
|
+
export function shortHash(s) {
|
|
70
|
+
let h = 5381;
|
|
71
|
+
for (let i = 0; i < s.length; i++)
|
|
72
|
+
h = ((h << 5) + h + s.charCodeAt(i)) | 0;
|
|
73
|
+
return (h >>> 0).toString(16).padStart(8, '0');
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* The fenced suggestion block of a comment body. The CLI's fence is the
|
|
77
|
+
* longest backtick run plus one (min 4), so a run of exactly that length can
|
|
78
|
+
* only be the closing fence.
|
|
79
|
+
*/
|
|
80
|
+
export function extractSuggestion(body) {
|
|
81
|
+
const m = /\r?\n(`{4,})suggestion\r?\n([\s\S]*?)\r?\n\1/.exec(body);
|
|
82
|
+
return m?.[2] ?? '';
|
|
83
|
+
}
|
|
84
|
+
/**
|
|
85
|
+
* KTD4 dedup key: `path:line:severity:normalizedMessage:hash8(suggestion)`.
|
|
86
|
+
* A changed suggestion changes the key, so a corrected fix re-posts. A body
|
|
87
|
+
* that parses as neither format keeps the pre-U6 first-line key.
|
|
88
|
+
*/
|
|
89
|
+
export function inlineDedupKey(path, line, body) {
|
|
90
|
+
const parsed = parseInlineBody(body);
|
|
91
|
+
const hash = shortHash(extractSuggestion(body));
|
|
92
|
+
if (parsed === undefined)
|
|
93
|
+
return `${path}:${line}:${body.split('\n')[0]}:${hash}`;
|
|
94
|
+
return `${path}:${line}:${parsed.severity}:${parsed.message}:${hash}`;
|
|
95
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Review prompt packs (roadmap E1.U2): named rubric blocks appended to the
|
|
3
|
+
* code-review prompt. Packs shape the rubric only — findings still flow
|
|
4
|
+
* through the same severity gate, dedup, adjudication, and cap path, so a
|
|
5
|
+
* pack can shift recall but cannot bypass posting policy.
|
|
6
|
+
*
|
|
7
|
+
* The `security` pack's deterministic half is the secrets lane
|
|
8
|
+
* (`src/review/secrets.ts`) — regex candidates over the local merge-base
|
|
9
|
+
* diff, adjudicated by the confidence model, masked in every output. It runs regardless of
|
|
10
|
+
* profile selection; the rubric below additionally tunes the model toward
|
|
11
|
+
* security-shaped defects.
|
|
12
|
+
*/
|
|
13
|
+
export declare const REVIEW_PROFILES: readonly ["security", "perf", "debloat"];
|
|
14
|
+
export type ReviewProfile = (typeof REVIEW_PROFILES)[number];
|
|
15
|
+
export declare function isReviewProfile(value: string): value is ReviewProfile;
|
|
16
|
+
/**
|
|
17
|
+
* Render the rubric section for the configured profiles. Returns undefined
|
|
18
|
+
* when no valid profile is configured so the prompt stays byte-identical to
|
|
19
|
+
* the no-packs form.
|
|
20
|
+
*/
|
|
21
|
+
export declare function packRubric(profiles: readonly string[] | undefined): string | undefined;
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Review prompt packs (roadmap E1.U2): named rubric blocks appended to the
|
|
3
|
+
* code-review prompt. Packs shape the rubric only — findings still flow
|
|
4
|
+
* through the same severity gate, dedup, adjudication, and cap path, so a
|
|
5
|
+
* pack can shift recall but cannot bypass posting policy.
|
|
6
|
+
*
|
|
7
|
+
* The `security` pack's deterministic half is the secrets lane
|
|
8
|
+
* (`src/review/secrets.ts`) — regex candidates over the local merge-base
|
|
9
|
+
* diff, adjudicated by the confidence model, masked in every output. It runs regardless of
|
|
10
|
+
* profile selection; the rubric below additionally tunes the model toward
|
|
11
|
+
* security-shaped defects.
|
|
12
|
+
*/
|
|
13
|
+
export const REVIEW_PROFILES = ['security', 'perf', 'debloat'];
|
|
14
|
+
export function isReviewProfile(value) {
|
|
15
|
+
return REVIEW_PROFILES.includes(value);
|
|
16
|
+
}
|
|
17
|
+
const RUBRICS = {
|
|
18
|
+
security: 'Security lens – additionally weigh: injection (SQL, shell, template, prompt), ' +
|
|
19
|
+
'missing or bypassable authorization checks, unsafe deserialization, secret-shaped ' +
|
|
20
|
+
'literals committed in the diff (report file/line and pattern class only – never ' +
|
|
21
|
+
'reproduce the literal), weak or misused crypto, path traversal, SSRF, and ' +
|
|
22
|
+
'unsanitized input reaching HTML or a shell. Only report a path actually ' +
|
|
23
|
+
'exploitable from the changed code.',
|
|
24
|
+
perf: 'Performance lens – additionally weigh: N+1 or per-item queries, work repeated ' +
|
|
25
|
+
'inside loops that could hoist, accidental O(n^2) or worse on unbounded inputs, ' +
|
|
26
|
+
'sync or blocking calls on hot paths, unnecessary copies or allocations of large ' +
|
|
27
|
+
'structures, and missing pagination or bounds on data fetched. Cite the scaling ' +
|
|
28
|
+
'input the cost depends on.',
|
|
29
|
+
debloat: 'Debloat lens – additionally weigh: dead code added but never reachable, logic ' +
|
|
30
|
+
'duplicating an existing helper visible in the diff or index context, dependencies ' +
|
|
31
|
+
'or abstractions introduced for a single trivial use, commented-out code, and ' +
|
|
32
|
+
'params/exports added without a caller. Do not flag removal opportunities that ' +
|
|
33
|
+
'the diff itself already deletes.',
|
|
34
|
+
};
|
|
35
|
+
/**
|
|
36
|
+
* Render the rubric section for the configured profiles. Returns undefined
|
|
37
|
+
* when no valid profile is configured so the prompt stays byte-identical to
|
|
38
|
+
* the no-packs form.
|
|
39
|
+
*/
|
|
40
|
+
export function packRubric(profiles) {
|
|
41
|
+
if (profiles === undefined || profiles.length === 0)
|
|
42
|
+
return undefined;
|
|
43
|
+
const blocks = profiles.filter(isReviewProfile).map((p) => `- ${RUBRICS[p]}`);
|
|
44
|
+
if (blocks.length === 0)
|
|
45
|
+
return undefined;
|
|
46
|
+
return `Active review lenses:\n${blocks.join('\n')}`;
|
|
47
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Review scope: which changed files reach the review model.
|
|
3
|
+
*
|
|
4
|
+
* Generated, fixture, golden and vendored paths carry content that reads
|
|
5
|
+
* like code (sample manifests, rendered comments, lockfiles). Fed to the
|
|
6
|
+
* model it produced findings about files that do not exist, so these
|
|
7
|
+
* paths are excluded by default. `review.exclude` replaces the list.
|
|
8
|
+
*/
|
|
9
|
+
export declare const DEFAULT_REVIEW_EXCLUDE: readonly string[];
|
|
10
|
+
export declare function globMatch(glob: string, path: string): boolean;
|
|
11
|
+
export declare function partitionByExclude<T extends {
|
|
12
|
+
filename: string;
|
|
13
|
+
}>(files: T[], exclude: readonly string[]): {
|
|
14
|
+
kept: T[];
|
|
15
|
+
excluded: T[];
|
|
16
|
+
};
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Review scope: which changed files reach the review model.
|
|
3
|
+
*
|
|
4
|
+
* Generated, fixture, golden and vendored paths carry content that reads
|
|
5
|
+
* like code (sample manifests, rendered comments, lockfiles). Fed to the
|
|
6
|
+
* model it produced findings about files that do not exist, so these
|
|
7
|
+
* paths are excluded by default. `review.exclude` replaces the list.
|
|
8
|
+
*/
|
|
9
|
+
export const DEFAULT_REVIEW_EXCLUDE = [
|
|
10
|
+
'dist/**',
|
|
11
|
+
'fixtures/**',
|
|
12
|
+
'tests/goldens/**',
|
|
13
|
+
'**/package-lock.json',
|
|
14
|
+
'**/npm-shrinkwrap.json',
|
|
15
|
+
'**/yarn.lock',
|
|
16
|
+
'**/pnpm-lock.yaml',
|
|
17
|
+
'**/bun.lock',
|
|
18
|
+
'**/Cargo.lock',
|
|
19
|
+
'**/poetry.lock',
|
|
20
|
+
'**/Gemfile.lock',
|
|
21
|
+
'**/composer.lock',
|
|
22
|
+
'**/go.sum',
|
|
23
|
+
'**/*.generated.*',
|
|
24
|
+
'assets/brand/export/**',
|
|
25
|
+
];
|
|
26
|
+
const globCache = new Map();
|
|
27
|
+
function globToRegExp(glob) {
|
|
28
|
+
const cached = globCache.get(glob);
|
|
29
|
+
if (cached !== undefined)
|
|
30
|
+
return cached;
|
|
31
|
+
let re = '';
|
|
32
|
+
for (let i = 0; i < glob.length; i++) {
|
|
33
|
+
const c = glob[i];
|
|
34
|
+
if (c === '*') {
|
|
35
|
+
if (glob[i + 1] === '*') {
|
|
36
|
+
// `**/` matches zero or more directories; a bare `**` matches anything.
|
|
37
|
+
if (glob[i + 2] === '/') {
|
|
38
|
+
re += '(?:.*/)?';
|
|
39
|
+
i += 2;
|
|
40
|
+
}
|
|
41
|
+
else {
|
|
42
|
+
re += '.*';
|
|
43
|
+
i += 1;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
else {
|
|
47
|
+
re += '[^/]*';
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
else if (c === '?') {
|
|
51
|
+
re += '[^/]';
|
|
52
|
+
}
|
|
53
|
+
else {
|
|
54
|
+
re += c.replace(/[.+^${}()|[\]\\]/g, '\\$&');
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
const out = new RegExp(`^${re}$`);
|
|
58
|
+
globCache.set(glob, out);
|
|
59
|
+
return out;
|
|
60
|
+
}
|
|
61
|
+
export function globMatch(glob, path) {
|
|
62
|
+
return globToRegExp(glob).test(path);
|
|
63
|
+
}
|
|
64
|
+
export function partitionByExclude(files, exclude) {
|
|
65
|
+
const kept = [];
|
|
66
|
+
const excluded = [];
|
|
67
|
+
for (const f of files) {
|
|
68
|
+
if (exclude.some((g) => globMatch(g, f.filename)))
|
|
69
|
+
excluded.push(f);
|
|
70
|
+
else
|
|
71
|
+
kept.push(f);
|
|
72
|
+
}
|
|
73
|
+
return { kept, excluded };
|
|
74
|
+
}
|
package/dist/review/secrets.d.ts
CHANGED
|
@@ -2,12 +2,12 @@ import { type ExecFn } from '../detect.js';
|
|
|
2
2
|
import { DecisionClient } from '../vision/decisions.js';
|
|
3
3
|
/**
|
|
4
4
|
* Deterministic secrets scan over the PR's local merge-base diff,
|
|
5
|
-
* optionally adjudicated by the Decisions API (
|
|
5
|
+
* optionally adjudicated by the Decisions API (the confidence model). The lane is
|
|
6
6
|
* additive-only: findings are unioned into the review AFTER model
|
|
7
7
|
* synthesis so a prompt-injected synthesis can never erase them, and a
|
|
8
|
-
*
|
|
8
|
+
* A confidence-model outage degrades to regex-only findings rather than silence.
|
|
9
9
|
*
|
|
10
|
-
* Masking contract: raw literals transit to
|
|
10
|
+
* Masking contract: raw literals transit to the confidence model inside `state` (KTD9 —
|
|
11
11
|
* adjudication needs the shape, and the full diff already crosses to
|
|
12
12
|
* OpenRouter in the review call) and appear NOWHERE else — not in
|
|
13
13
|
* findings, comments, the report, or logs.
|
|
@@ -19,19 +19,19 @@ export interface SecretCandidate {
|
|
|
19
19
|
patternClass: string;
|
|
20
20
|
/** Diff line text with every occurrence of the literal replaced by `***`. */
|
|
21
21
|
contextExcerpt: string;
|
|
22
|
-
/** Raw literal —
|
|
22
|
+
/** Raw literal — confidence-model `state` only, never emitted. */
|
|
23
23
|
literal: string;
|
|
24
|
-
/** Full raw added-line text —
|
|
24
|
+
/** Full raw added-line text — confidence-model `state` only, never emitted. */
|
|
25
25
|
rawText: string;
|
|
26
26
|
}
|
|
27
27
|
export interface SecretScanRecord {
|
|
28
28
|
file: string;
|
|
29
29
|
line: number;
|
|
30
30
|
patternClass: string;
|
|
31
|
-
/** True when
|
|
31
|
+
/** True when the confidence model answered for this candidate. */
|
|
32
32
|
adjudicated: boolean;
|
|
33
33
|
pLive?: number;
|
|
34
|
-
/**
|
|
34
|
+
/** The confidence model scored below threshold — recorded for audit, not a finding. */
|
|
35
35
|
suppressed?: boolean;
|
|
36
36
|
}
|
|
37
37
|
export interface SecretsScanResult {
|
|
@@ -42,12 +42,12 @@ export interface SecretsScanResult {
|
|
|
42
42
|
severity: string;
|
|
43
43
|
category?: string;
|
|
44
44
|
message: string;
|
|
45
|
-
/**
|
|
45
|
+
/** Confidence-model P(live) carried through adjudication — feeds the review gate. */
|
|
46
46
|
p?: number;
|
|
47
47
|
}[];
|
|
48
48
|
/** Audit records for report.secretsScan — literals never included. */
|
|
49
49
|
records: SecretScanRecord[];
|
|
50
|
-
/** Candidates past MAX_CANDIDATES — reported count-only, never sent to
|
|
50
|
+
/** Candidates past MAX_CANDIDATES — reported count-only, never sent to the confidence model. */
|
|
51
51
|
overflow: number;
|
|
52
52
|
/** Why the lane produced nothing (e.g. base unfetchable). */
|
|
53
53
|
skipped?: string;
|
|
@@ -77,7 +77,7 @@ export declare function materializeMergeBaseDiff(opts: {
|
|
|
77
77
|
skipped: string;
|
|
78
78
|
}>;
|
|
79
79
|
/**
|
|
80
|
-
* Full lane: scan candidates (capped),
|
|
80
|
+
* Full lane: scan candidates (capped), adjudicate with the confidence model when a client and
|
|
81
81
|
* threshold are available, and emit masked findings + audit records.
|
|
82
82
|
* `client === undefined` (decisionModel unset) or any DecisionError →
|
|
83
83
|
* every candidate unadjudicated — regex-only mode, never silence.
|
package/dist/review/secrets.js
CHANGED
|
@@ -87,8 +87,8 @@ export function scanDiffForSecrets(diffText) {
|
|
|
87
87
|
function maskFindingMessage(c, adjudicated) {
|
|
88
88
|
const verdict = adjudicated
|
|
89
89
|
? 'live-looking credential'
|
|
90
|
-
: 'secret-shaped literal (unadjudicated
|
|
91
|
-
return (`L${c.line}: ${adjudicated ? '
|
|
90
|
+
: 'secret-shaped literal (unadjudicated: decision model unavailable)';
|
|
91
|
+
return (`L${c.line}: ${adjudicated ? 'bug' : 'risk'}: ` +
|
|
92
92
|
`${verdict} (${c.patternClass}) added in this PR at \`${c.file}\`. ` +
|
|
93
93
|
`Rotate it and purge it from history.`);
|
|
94
94
|
}
|
|
@@ -128,7 +128,7 @@ export async function materializeMergeBaseDiff(opts) {
|
|
|
128
128
|
return { diff: diff.stdout };
|
|
129
129
|
}
|
|
130
130
|
/**
|
|
131
|
-
* Full lane: scan candidates (capped),
|
|
131
|
+
* Full lane: scan candidates (capped), adjudicate with the confidence model when a client and
|
|
132
132
|
* threshold are available, and emit masked findings + audit records.
|
|
133
133
|
* `client === undefined` (decisionModel unset) or any DecisionError →
|
|
134
134
|
* every candidate unadjudicated — regex-only mode, never silence.
|
|
@@ -169,7 +169,7 @@ export async function scanSecrets(opts) {
|
|
|
169
169
|
}
|
|
170
170
|
catch (e) {
|
|
171
171
|
adjudicationFailed = true;
|
|
172
|
-
debug('secrets', `adjudication failed
|
|
172
|
+
debug('secrets', `adjudication failed, degrading to regex-only: ${describeDecisionError(e)}`);
|
|
173
173
|
}
|
|
174
174
|
}
|
|
175
175
|
const findings = [];
|
|
@@ -194,7 +194,7 @@ export async function scanSecrets(opts) {
|
|
|
194
194
|
severity: adjudicated ? 'bug' : 'risk',
|
|
195
195
|
category: 'security',
|
|
196
196
|
message: maskFindingMessage(c, adjudicated),
|
|
197
|
-
// A
|
|
197
|
+
// A confidence-model-confirmed live secret counts as proven for the review-event
|
|
198
198
|
// gate — pLive IS the true-positive probability for this finding.
|
|
199
199
|
// Unadjudicated findings carry no p (degrade-open, like U8).
|
|
200
200
|
...(adjudicated && pLive !== undefined ? { p: pLive } : {}),
|
|
@@ -215,8 +215,8 @@ export async function scanSecrets(opts) {
|
|
|
215
215
|
line: 0,
|
|
216
216
|
severity: 'risk',
|
|
217
217
|
category: 'security',
|
|
218
|
-
message: `L0:
|
|
219
|
-
`${MAX_CANDIDATES}-candidate adjudication cap and were not evaluated
|
|
218
|
+
message: `L0: risk: ${overflow} secret-shaped literal(s) exceeded the ` +
|
|
219
|
+
`${MAX_CANDIDATES}-candidate adjudication cap and were not evaluated; ` +
|
|
220
220
|
'review the diff for secrets manually.',
|
|
221
221
|
});
|
|
222
222
|
}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Test files describe expected behavior: a finding that restates an
|
|
3
|
+
* assertion is almost never a defect. Findings anchored in a test file
|
|
4
|
+
* are capped at `nit` unless they cite a non-test file from the diff
|
|
5
|
+
* (the real bug is elsewhere) or carry reproduced evidence.
|
|
6
|
+
*/
|
|
7
|
+
export declare function isTestPath(path: string): boolean;
|
|
8
|
+
export declare function capTestFindings<T extends {
|
|
9
|
+
file: string;
|
|
10
|
+
severity: string;
|
|
11
|
+
message: string;
|
|
12
|
+
evidence?: {
|
|
13
|
+
status?: string;
|
|
14
|
+
};
|
|
15
|
+
}>(findings: T[], diffFiles: string[]): {
|
|
16
|
+
findings: T[];
|
|
17
|
+
capped: number;
|
|
18
|
+
};
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Test files describe expected behavior: a finding that restates an
|
|
3
|
+
* assertion is almost never a defect. Findings anchored in a test file
|
|
4
|
+
* are capped at `nit` unless they cite a non-test file from the diff
|
|
5
|
+
* (the real bug is elsewhere) or carry reproduced evidence.
|
|
6
|
+
*/
|
|
7
|
+
const TEST_PATH = /(^|\/)(tests?|__tests__|__mocks__|spec|e2e)\/|\.(test|spec)\.[^/]+$|_test\.(go|py|rb)$|(^|\/)test_[^/]+\.py$/;
|
|
8
|
+
export function isTestPath(path) {
|
|
9
|
+
return TEST_PATH.test(path.replace(/^\.\//, ''));
|
|
10
|
+
}
|
|
11
|
+
const CAPPED = new Set(['bug', 'risk']);
|
|
12
|
+
export function capTestFindings(findings, diffFiles) {
|
|
13
|
+
const sourceFiles = diffFiles.filter((p) => !isTestPath(p));
|
|
14
|
+
let capped = 0;
|
|
15
|
+
const out = findings.map((f) => {
|
|
16
|
+
if (!isTestPath(f.file) || !CAPPED.has(f.severity))
|
|
17
|
+
return f;
|
|
18
|
+
if (f.evidence?.status === 'reproduced')
|
|
19
|
+
return f;
|
|
20
|
+
if (sourceFiles.some((p) => f.message.includes(p)))
|
|
21
|
+
return f;
|
|
22
|
+
capped++;
|
|
23
|
+
return { ...f, severity: 'nit' };
|
|
24
|
+
});
|
|
25
|
+
return { findings: out, capped };
|
|
26
|
+
}
|
package/dist/review/triage.d.ts
CHANGED
|
@@ -17,7 +17,7 @@ export interface TriageRecord {
|
|
|
17
17
|
topRiskArea?: TriageArea;
|
|
18
18
|
/** Choice-answer confidence when the API provides one. */
|
|
19
19
|
topRiskAreaConfidence?: number;
|
|
20
|
-
/** Chars of diff evidence
|
|
20
|
+
/** Chars of diff evidence the confidence model saw — route mode won't downgrade on 0. */
|
|
21
21
|
diffExcerptChars?: number;
|
|
22
22
|
/** decide() failed or answers failed validation — degrade-open marker. */
|
|
23
23
|
unadjudicated?: boolean;
|
package/dist/review/triage.js
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* U7 PR triage lane — one batched `decide` call before chunk review
|
|
3
3
|
* produces a typed-probability triage record (the pace two-round
|
|
4
|
-
* pattern).
|
|
4
|
+
* pattern). The confidence model routes and annotates, never gates: every chunk is still
|
|
5
5
|
* reviewed by a code model and the deterministic verdict stays
|
|
6
6
|
* authoritative. `decide` failure degrades open — the record lands with
|
|
7
7
|
* `unadjudicated` and routing keeps the configured (strong) model.
|
|
8
8
|
*
|
|
9
|
-
* PR title/body in `state` are untrusted text —
|
|
9
|
+
* PR title/body in `state` are untrusted text — the confidence model is the only
|
|
10
10
|
* consumer; they never reach the verdict path.
|
|
11
11
|
*/
|
|
12
12
|
import { debug } from '../debug.js';
|
|
@@ -60,13 +60,13 @@ function buildTriageQuestions() {
|
|
|
60
60
|
},
|
|
61
61
|
[Q.risk]: {
|
|
62
62
|
type: 'score',
|
|
63
|
-
instructions: 'blast radius if this PR merges broken
|
|
63
|
+
instructions: 'blast radius if this PR merges broken – pick the closest rubric level.',
|
|
64
64
|
criteria: [
|
|
65
|
-
'cosmetic only
|
|
66
|
-
'minor
|
|
67
|
-
'moderate
|
|
68
|
-
'high
|
|
69
|
-
'severe
|
|
65
|
+
'cosmetic only – docs, comments, formatting, metadata',
|
|
66
|
+
'minor – internal-only paths, limited blast radius',
|
|
67
|
+
'moderate – user-visible defects plausible',
|
|
68
|
+
'high – security-sensitive or data-handling paths touched',
|
|
69
|
+
'severe – auth, billing, or data-loss surface',
|
|
70
70
|
],
|
|
71
71
|
},
|
|
72
72
|
[Q.area]: {
|
|
@@ -111,7 +111,7 @@ export async function triagePr(opts) {
|
|
|
111
111
|
return rec;
|
|
112
112
|
}
|
|
113
113
|
catch (e) {
|
|
114
|
-
debug('triage', `adjudication failed
|
|
114
|
+
debug('triage', `adjudication failed – degrading to annotate: ${describeDecisionError(e)}`);
|
|
115
115
|
return { mode: opts.mode, model: opts.model ?? JEV_DEFAULT_MODEL, unadjudicated: true };
|
|
116
116
|
}
|
|
117
117
|
}
|
|
@@ -137,7 +137,7 @@ export function routeModel(opts) {
|
|
|
137
137
|
if (rec.risk <= 2 && rec.needsDeepReview < 0.5) {
|
|
138
138
|
return {
|
|
139
139
|
model: opts.lowRiskModel,
|
|
140
|
-
reason: `low risk
|
|
140
|
+
reason: `low risk – risk=${rec.risk}, needsDeepReview=${rec.needsDeepReview.toFixed(2)}`,
|
|
141
141
|
};
|
|
142
142
|
}
|
|
143
143
|
return {
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic finding validation (no model). A finding must point at a
|
|
3
|
+
* file that is in the reviewed diff and at a line inside the diff's
|
|
4
|
+
* changed hunks (plus a small context tolerance). Anything else is a
|
|
5
|
+
* hallucinated or stale anchor; it is dropped here and counted in the
|
|
6
|
+
* report so the drop is never silent.
|
|
7
|
+
*/
|
|
8
|
+
export type DropReason = 'file_not_in_diff' | 'file_excluded' | 'file_deleted' | 'line_beyond_file' | 'line_outside_diff';
|
|
9
|
+
export interface DroppedFinding {
|
|
10
|
+
file: string;
|
|
11
|
+
line?: number;
|
|
12
|
+
reason: DropReason;
|
|
13
|
+
}
|
|
14
|
+
export interface ValidationAudit {
|
|
15
|
+
dropped: number;
|
|
16
|
+
byReason: Partial<Record<DropReason, number>>;
|
|
17
|
+
/** Up to 10 dropped anchors, for the Diagnostics fold. */
|
|
18
|
+
examples: DroppedFinding[];
|
|
19
|
+
}
|
|
20
|
+
/** Lines of slack around a hunk: models are often off by a line or two. */
|
|
21
|
+
export declare const HUNK_TOLERANCE = 2;
|
|
22
|
+
export interface ParsedHunks {
|
|
23
|
+
/** Inclusive new-side [start, end] line ranges, one per hunk with new lines. */
|
|
24
|
+
ranges: [number, number][];
|
|
25
|
+
/** Set when the patch creates the file: its exact line count. */
|
|
26
|
+
newFileLength?: number;
|
|
27
|
+
/** True when no new-side lines survive (file deleted at head). */
|
|
28
|
+
deleted: boolean;
|
|
29
|
+
}
|
|
30
|
+
export declare function parseHunks(patch: string): ParsedHunks;
|
|
31
|
+
export declare function validateFindings<T extends {
|
|
32
|
+
file: string;
|
|
33
|
+
line?: number;
|
|
34
|
+
}>(findings: T[], files: {
|
|
35
|
+
filename: string;
|
|
36
|
+
patch?: string;
|
|
37
|
+
}[], excluded?: ReadonlySet<string>): {
|
|
38
|
+
kept: T[];
|
|
39
|
+
dropped: DroppedFinding[];
|
|
40
|
+
};
|
|
41
|
+
export declare function auditOf(dropped: DroppedFinding[]): ValidationAudit;
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Deterministic finding validation (no model). A finding must point at a
|
|
3
|
+
* file that is in the reviewed diff and at a line inside the diff's
|
|
4
|
+
* changed hunks (plus a small context tolerance). Anything else is a
|
|
5
|
+
* hallucinated or stale anchor; it is dropped here and counted in the
|
|
6
|
+
* report so the drop is never silent.
|
|
7
|
+
*/
|
|
8
|
+
/** Lines of slack around a hunk: models are often off by a line or two. */
|
|
9
|
+
export const HUNK_TOLERANCE = 2;
|
|
10
|
+
const MAX_EXAMPLES = 10;
|
|
11
|
+
const HUNK_HEADER = /^@@ -(\d+)(?:,(\d+))? \+(\d+)(?:,(\d+))? @@/;
|
|
12
|
+
export function parseHunks(patch) {
|
|
13
|
+
const ranges = [];
|
|
14
|
+
let newFileLength;
|
|
15
|
+
let sawHunk = false;
|
|
16
|
+
for (const line of patch.split('\n')) {
|
|
17
|
+
const m = HUNK_HEADER.exec(line);
|
|
18
|
+
if (m === null)
|
|
19
|
+
continue;
|
|
20
|
+
sawHunk = true;
|
|
21
|
+
const oldStart = Number(m[1]);
|
|
22
|
+
const oldCount = m[2] === undefined ? 1 : Number(m[2]);
|
|
23
|
+
const start = Number(m[3]);
|
|
24
|
+
const count = m[4] === undefined ? 1 : Number(m[4]);
|
|
25
|
+
if (count > 0)
|
|
26
|
+
ranges.push([start, start + count - 1]);
|
|
27
|
+
if (oldStart === 0 && oldCount === 0)
|
|
28
|
+
newFileLength = count;
|
|
29
|
+
}
|
|
30
|
+
return {
|
|
31
|
+
ranges,
|
|
32
|
+
...(newFileLength !== undefined ? { newFileLength } : {}),
|
|
33
|
+
deleted: sawHunk && ranges.length === 0,
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
const norm = (p) => p.replace(/^\.\//, '');
|
|
37
|
+
export function validateFindings(findings, files, excluded = new Set()) {
|
|
38
|
+
const byName = new Map();
|
|
39
|
+
for (const f of files)
|
|
40
|
+
byName.set(f.filename, parseHunks(f.patch ?? ''));
|
|
41
|
+
const kept = [];
|
|
42
|
+
const dropped = [];
|
|
43
|
+
for (const f of findings) {
|
|
44
|
+
const file = norm(String(f.file ?? ''));
|
|
45
|
+
const hunks = byName.get(file);
|
|
46
|
+
const hasLine = typeof f.line === 'number' && Number.isFinite(f.line);
|
|
47
|
+
const drop = (reason) => {
|
|
48
|
+
dropped.push({ file: String(f.file), ...(hasLine ? { line: f.line } : {}), reason });
|
|
49
|
+
};
|
|
50
|
+
if (hunks === undefined) {
|
|
51
|
+
drop(excluded.has(file) ? 'file_excluded' : 'file_not_in_diff');
|
|
52
|
+
}
|
|
53
|
+
else if (hunks.deleted) {
|
|
54
|
+
drop('file_deleted');
|
|
55
|
+
}
|
|
56
|
+
else if (!hasLine) {
|
|
57
|
+
kept.push(f);
|
|
58
|
+
}
|
|
59
|
+
else if (hunks.newFileLength !== undefined && f.line > hunks.newFileLength) {
|
|
60
|
+
drop('line_beyond_file');
|
|
61
|
+
}
|
|
62
|
+
else if (!hunks.ranges.some(([s, e]) => f.line >= s - HUNK_TOLERANCE && f.line <= e + HUNK_TOLERANCE)) {
|
|
63
|
+
drop('line_outside_diff');
|
|
64
|
+
}
|
|
65
|
+
else {
|
|
66
|
+
kept.push(f);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
return { kept, dropped };
|
|
70
|
+
}
|
|
71
|
+
export function auditOf(dropped) {
|
|
72
|
+
const byReason = {};
|
|
73
|
+
for (const d of dropped)
|
|
74
|
+
byReason[d.reason] = (byReason[d.reason] ?? 0) + 1;
|
|
75
|
+
return { dropped: dropped.length, byReason, examples: dropped.slice(0, MAX_EXAMPLES) };
|
|
76
|
+
}
|