argus-reviewer-e2e 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -5
- package/action/action.yml +22 -0
- package/action/sticky-comment.mjs +287 -79
- package/dist/api.d.ts +1 -0
- package/dist/api.js +1 -0
- package/dist/cli.d.ts +67 -0
- package/dist/cli.js +463 -86
- package/dist/config.d.ts +105 -2
- package/dist/config.js +136 -9
- package/dist/debug.d.ts +1 -0
- package/dist/debug.js +9 -3
- package/dist/detect.d.ts +11 -1
- package/dist/detect.js +19 -3
- package/dist/evidence/ci.d.ts +47 -2
- package/dist/evidence/ci.js +98 -6
- package/dist/evidence/gate.d.ts +10 -0
- package/dist/evidence/gate.js +29 -0
- package/dist/evidence/link.d.ts +6 -1
- package/dist/evidence/link.js +7 -5
- package/dist/executor/sandbox.d.ts +105 -0
- package/dist/executor/sandbox.js +231 -0
- package/dist/probe/author.d.ts +50 -0
- package/dist/probe/author.js +149 -0
- package/dist/probe/harness.d.ts +30 -0
- package/dist/probe/harness.js +99 -0
- package/dist/probe/queue.d.ts +105 -0
- package/dist/probe/queue.js +446 -0
- package/dist/review/adjudicate.d.ts +63 -0
- package/dist/review/adjudicate.js +111 -0
- package/dist/review/secrets.d.ts +88 -0
- package/dist/review/secrets.js +220 -0
- package/dist/review/triage.d.ts +76 -0
- package/dist/review/triage.js +163 -0
- package/dist/trust.d.ts +50 -0
- package/dist/trust.js +103 -0
- package/dist/vision/cost.d.ts +14 -1
- package/dist/vision/cost.js +13 -0
- package/dist/vision/decisions.d.ts +95 -0
- package/dist/vision/decisions.js +232 -0
- package/package.json +3 -2
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
import { readFile } from 'node:fs/promises';
|
|
2
|
+
import { join } from 'node:path';
|
|
3
|
+
/**
|
|
4
|
+
* Load-failure signatures are checked BEFORE test-failure summaries — a
|
|
5
|
+
* suite that can't import (vitest counts it as "Test Files 1 failed", TAP
|
|
6
|
+
* emits "not ok") must never classify as failed-test, and the exit code
|
|
7
|
+
* binds the result: probe output is attacker-printable, so a forged
|
|
8
|
+
* "N failed" line on a clean exit still classifies clean.
|
|
9
|
+
*/
|
|
10
|
+
const LOAD_ERROR_RE = /Cannot find module|ERR_UNKNOWN_FILE_EXTENSION|ERR_MODULE_NOT_FOUND|Failed to load|SyntaxError/;
|
|
11
|
+
const vitestHarness = {
|
|
12
|
+
kind: 'vitest',
|
|
13
|
+
runCmd: (file) => ['node', 'node_modules/vitest/vitest.mjs', 'run', file],
|
|
14
|
+
classify: ({ exitCode, stdout, stderr }) => {
|
|
15
|
+
const out = `${stdout}\n${stderr}`;
|
|
16
|
+
if (exitCode === 0)
|
|
17
|
+
return 'clean';
|
|
18
|
+
if (LOAD_ERROR_RE.test(out))
|
|
19
|
+
return 'load-error';
|
|
20
|
+
if (/no test files? found/i.test(out))
|
|
21
|
+
return 'not-collected';
|
|
22
|
+
if (/test files?\s+\d+ failed|tests\s+\d+ failed/i.test(out))
|
|
23
|
+
return 'failed-test';
|
|
24
|
+
return 'load-error';
|
|
25
|
+
},
|
|
26
|
+
};
|
|
27
|
+
const jestHarness = {
|
|
28
|
+
kind: 'jest',
|
|
29
|
+
runCmd: (file) => [
|
|
30
|
+
'node',
|
|
31
|
+
'node_modules/jest-cli/bin/jest.js',
|
|
32
|
+
'--runTestsByPath',
|
|
33
|
+
'--cacheDirectory=/tmp/jest',
|
|
34
|
+
file,
|
|
35
|
+
],
|
|
36
|
+
classify: ({ exitCode, stdout, stderr }) => {
|
|
37
|
+
const out = `${stdout}\n${stderr}`;
|
|
38
|
+
if (exitCode === 0)
|
|
39
|
+
return 'clean';
|
|
40
|
+
if (LOAD_ERROR_RE.test(out))
|
|
41
|
+
return 'load-error';
|
|
42
|
+
if (/no tests found/i.test(out))
|
|
43
|
+
return 'not-collected';
|
|
44
|
+
if (/tests:\s+\d+ failed/i.test(out))
|
|
45
|
+
return 'failed-test';
|
|
46
|
+
return 'load-error';
|
|
47
|
+
},
|
|
48
|
+
};
|
|
49
|
+
/**
|
|
50
|
+
* `node --test` in a TS repo only works because `scripts.test` carries the
|
|
51
|
+
* loader flags (`--import tsx`, `--experimental-strip-types`, `--loader`) —
|
|
52
|
+
* capture them verbatim or the authored .ts probe dies with
|
|
53
|
+
* ERR_UNKNOWN_FILE_EXTENSION. Both `--import tsx` and `--import=tsx` forms.
|
|
54
|
+
*/
|
|
55
|
+
const NODE_TEST_FLAG_RE = /(?:--import|--loader)(?:[\s=])\S+|--experimental-(?:strip|transform)-types/g;
|
|
56
|
+
function nodeTestHarness(flags) {
|
|
57
|
+
return {
|
|
58
|
+
kind: 'node-test',
|
|
59
|
+
// --test-reporter=tap pins the format `classify` parses — Node ≥22
|
|
60
|
+
// defaults to the spec reporter (`✖`, not TAP `not ok`) when stdout
|
|
61
|
+
// is a TTY, and real failures would fall through to load-error.
|
|
62
|
+
runCmd: (file) => ['node', ...flags, '--test', '--test-reporter=tap', file],
|
|
63
|
+
classify: ({ exitCode, stdout, stderr }) => {
|
|
64
|
+
const out = `${stdout}\n${stderr}`;
|
|
65
|
+
if (exitCode === 0)
|
|
66
|
+
return 'clean';
|
|
67
|
+
if (LOAD_ERROR_RE.test(out))
|
|
68
|
+
return 'load-error';
|
|
69
|
+
if (/^not ok/im.test(out))
|
|
70
|
+
return 'failed-test';
|
|
71
|
+
return 'load-error';
|
|
72
|
+
},
|
|
73
|
+
};
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Detect the consumer's test harness from package.json. Returns undefined
|
|
77
|
+
* when no supported harness exists — the probe lane degrades to a detail
|
|
78
|
+
* note, not a failure.
|
|
79
|
+
*/
|
|
80
|
+
export async function detectHarness(cwd) {
|
|
81
|
+
let pkg;
|
|
82
|
+
try {
|
|
83
|
+
pkg = JSON.parse(await readFile(join(cwd, 'package.json'), 'utf8'));
|
|
84
|
+
}
|
|
85
|
+
catch {
|
|
86
|
+
return undefined;
|
|
87
|
+
}
|
|
88
|
+
const deps = { ...pkg.dependencies, ...pkg.devDependencies };
|
|
89
|
+
const testScript = pkg.scripts?.test ?? '';
|
|
90
|
+
if (deps['vitest'] !== undefined || /\bvitest\b/.test(testScript))
|
|
91
|
+
return vitestHarness;
|
|
92
|
+
if (deps['jest'] !== undefined || /\bjest\b/.test(testScript))
|
|
93
|
+
return jestHarness;
|
|
94
|
+
if (/\bnode\b.*--test|node:test/.test(testScript)) {
|
|
95
|
+
const flags = testScript.match(NODE_TEST_FLAG_RE)?.flatMap((f) => f.split(/[\s=]+/)) ?? [];
|
|
96
|
+
return nodeTestHarness(flags);
|
|
97
|
+
}
|
|
98
|
+
return undefined;
|
|
99
|
+
}
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { type ProviderRules, type Sandbox } from '../config.js';
|
|
2
|
+
import { type ExecFn } from '../detect.js';
|
|
3
|
+
import type { PrMeta } from '../evidence/ci.js';
|
|
4
|
+
import { type Evidence } from '../evidence/link.js';
|
|
5
|
+
import type { RepoIndex } from '../index/scan.js';
|
|
6
|
+
import type { VisionClient } from '../engine/loop.js';
|
|
7
|
+
import type { CallCost } from '../vision/cost.js';
|
|
8
|
+
import type { Ledger } from '../vision/ledger.js';
|
|
9
|
+
import type { TriageAreaSignal } from '../review/triage.js';
|
|
10
|
+
import { type ProbeOutcome } from './harness.js';
|
|
11
|
+
/**
|
|
12
|
+
* The B.2 probe lane (U4). Called after `linkFindings` inside `code-review`.
|
|
13
|
+
* Selects `not_exercised` blocking-severity findings, authors one probe each,
|
|
14
|
+
* runs it in the sandbox against head AND a merge-base worktree, and upgrades
|
|
15
|
+
* evidence to `reproduced` only on fail-head ∧ clean-base (KTD6). Everything
|
|
16
|
+
* else leaves the finding untouched — additive evidence only, never a new
|
|
17
|
+
* failure surface (KTD8): every error path degrades to a probe record or a
|
|
18
|
+
* debug note, and the queue never changes verdict, ok, or exit code.
|
|
19
|
+
*
|
|
20
|
+
* Trust note: for `pull_request` events the config file ships in the PR's
|
|
21
|
+
* own tree, so for FORK PRs the lane ignores config-supplied
|
|
22
|
+
* image/limits/allowForks and runs on `DEFAULT_SANDBOX` — forks approve via
|
|
23
|
+
* the argus-probe label or a trusted author association, never via config.
|
|
24
|
+
*/
|
|
25
|
+
/** Finding shape shared by the review report, probe targets, and this lane. */
|
|
26
|
+
export interface LinkedFinding {
|
|
27
|
+
file?: string | undefined;
|
|
28
|
+
line?: number | undefined;
|
|
29
|
+
severity?: string | undefined;
|
|
30
|
+
message?: string | undefined;
|
|
31
|
+
evidence: Evidence;
|
|
32
|
+
}
|
|
33
|
+
export type ProbeReportOutcome = 'reproduced' | 'clean' | 'load-error' | 'not-collected' | 'error';
|
|
34
|
+
export interface ProbeRecord {
|
|
35
|
+
/** Probe filename (basename — it is written beside the exemplar test). */
|
|
36
|
+
file: string;
|
|
37
|
+
findingFile: string | undefined;
|
|
38
|
+
findingLine: number | undefined;
|
|
39
|
+
outcome: ProbeReportOutcome;
|
|
40
|
+
/** Raw harness outcomes per checkout, when the run happened ('error' = infra/timeout). */
|
|
41
|
+
headOutcome?: ProbeOutcome | 'error' | undefined;
|
|
42
|
+
baseOutcome?: ProbeOutcome | 'error' | undefined;
|
|
43
|
+
durationMs: number;
|
|
44
|
+
costUsd: number;
|
|
45
|
+
tokens: number;
|
|
46
|
+
detail: string;
|
|
47
|
+
/** Capped, control-char-stripped stdout+stderr for audit. */
|
|
48
|
+
output?: string | undefined;
|
|
49
|
+
}
|
|
50
|
+
export interface ProbeLaneResult {
|
|
51
|
+
records: ProbeRecord[];
|
|
52
|
+
/** Present when the lane was enabled but skipped before running probes. */
|
|
53
|
+
skipReason?: string | undefined;
|
|
54
|
+
}
|
|
55
|
+
export interface ProbeLaneOptions {
|
|
56
|
+
/** PR checkout root (the head tree). */
|
|
57
|
+
cwd: string;
|
|
58
|
+
reportDir: string;
|
|
59
|
+
/** Caller resolves `sandbox.enabled || ARGUS_SANDBOX=1` into this object. */
|
|
60
|
+
sandbox: Sandbox;
|
|
61
|
+
meta: PrMeta | undefined;
|
|
62
|
+
token: string | undefined;
|
|
63
|
+
client: VisionClient;
|
|
64
|
+
model: string;
|
|
65
|
+
provider: ProviderRules | undefined;
|
|
66
|
+
/** Shared codeReviewBudgetUsd ledger — authoring calls record on it. */
|
|
67
|
+
ledger: Ledger;
|
|
68
|
+
/** codeReviewBudgetUsd — authoring stops once spend reaches it. */
|
|
69
|
+
budgetUsd: number | undefined;
|
|
70
|
+
/** Configured blocking severities — the queue only admits those. */
|
|
71
|
+
severityGates: string[];
|
|
72
|
+
index: RepoIndex | undefined;
|
|
73
|
+
/** U9 — triage top_risk_area; advisory reorder of probe candidates. */
|
|
74
|
+
triageArea?: TriageAreaSignal | undefined;
|
|
75
|
+
/** Report sink — authored probe CallCosts are pushed here for the report. */
|
|
76
|
+
calls?: CallCost[] | undefined;
|
|
77
|
+
exec?: ExecFn | undefined;
|
|
78
|
+
log?: ((line: string) => void) | undefined;
|
|
79
|
+
}
|
|
80
|
+
/** Pure selection: not_exercised findings at blocking severities, capped. */
|
|
81
|
+
export declare function selectProbeTargets(findings: LinkedFinding[], severityGates: string[], maxProbes: number, triageArea?: TriageAreaSignal): LinkedFinding[];
|
|
82
|
+
/**
|
|
83
|
+
* Repo-relative path gate for anything model- or index-derived that is read
|
|
84
|
+
* or written on the HOST: no absolute paths, no `..` escapes, no backslashes.
|
|
85
|
+
* The write side is already basename-bound (PROBE_FILENAME_RE); this is the
|
|
86
|
+
* read-side and exemplar-dir boundary — a prompt-injected `../../.env`
|
|
87
|
+
* finding.file or a crafted index entry must never reach fs calls.
|
|
88
|
+
*/
|
|
89
|
+
export declare function isSafeRepoPath(p: string): boolean;
|
|
90
|
+
/**
|
|
91
|
+
* Nearest existing test file to the finding's file — same directory first,
|
|
92
|
+
* then same top-level segment, then any test file. The exemplar sets the
|
|
93
|
+
* probe's write location so the consumer's own include/roots cover it.
|
|
94
|
+
* Index paths are filtered through isSafeRepoPath — a committed/crafted
|
|
95
|
+
* index could otherwise aim the host write outside the checkout.
|
|
96
|
+
*/
|
|
97
|
+
export declare function findExemplarTest(index: RepoIndex | undefined, findingFile: string): string | undefined;
|
|
98
|
+
/**
|
|
99
|
+
* Run the probe lane. Mutates `findings` evidence in place for reproduced
|
|
100
|
+
* results and returns the per-probe audit records for code-review.json.
|
|
101
|
+
* Returns undefined only when the lane is disabled entirely; otherwise a
|
|
102
|
+
* result carrying records and (when it bowed out early) a skipReason the
|
|
103
|
+
* report surface can render.
|
|
104
|
+
*/
|
|
105
|
+
export declare function runProbeLane(findings: LinkedFinding[], o: ProbeLaneOptions): Promise<ProbeLaneResult | undefined>;
|
|
@@ -0,0 +1,446 @@
|
|
|
1
|
+
import { chmod, mkdir, readFile, realpath, rm, stat, writeFile } from 'node:fs/promises';
|
|
2
|
+
import { dirname, isAbsolute, join, posix, relative } from 'node:path';
|
|
3
|
+
import { DEFAULT_SANDBOX } from '../config.js';
|
|
4
|
+
import { defaultExec } from '../detect.js';
|
|
5
|
+
import { mayProbePr } from '../evidence/gate.js';
|
|
6
|
+
import { isTestFile } from '../evidence/link.js';
|
|
7
|
+
import { buildProbeMessages, parseProbe, probeImportsSafe, PROBE_SCHEMA, } from './author.js';
|
|
8
|
+
import { detectHarness } from './harness.js';
|
|
9
|
+
import { checkSandboxPaths, dockerAvailable, resolveSandboxImage, runProbeInSandbox, SANDBOX_OUTPUT_CAP, SCRATCH_DIR_NAME, stripControlChars, sandboxLimits, } from '../executor/sandbox.js';
|
|
10
|
+
/** U9 — below this confidence the triage area signal is ignored. */
|
|
11
|
+
const MIN_AREA_CONFIDENCE = 0.5;
|
|
12
|
+
/**
|
|
13
|
+
* U9 triage-informed ordering: a finding's file path "hits" the flagged
|
|
14
|
+
* risk area when a path segment equals the area token or starts with
|
|
15
|
+
* it at a camelCase/digit boundary — `src/auth/session.ts` and
|
|
16
|
+
* `dataStore.ts` hit auth/data, while `author.ts` and `database.ts`
|
|
17
|
+
* do not. Advisory only.
|
|
18
|
+
*/
|
|
19
|
+
function fileHitsArea(file, area) {
|
|
20
|
+
const token = area.toLowerCase();
|
|
21
|
+
return file.split(/[/._-]+/).some((segment) => {
|
|
22
|
+
const lower = segment.toLowerCase();
|
|
23
|
+
if (lower === token)
|
|
24
|
+
return true;
|
|
25
|
+
if (!lower.startsWith(token))
|
|
26
|
+
return false;
|
|
27
|
+
return /[A-Z0-9]/.test(segment.charAt(token.length));
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
/** Pure selection: not_exercised findings at blocking severities, capped. */
|
|
31
|
+
export function selectProbeTargets(findings, severityGates, maxProbes, triageArea) {
|
|
32
|
+
const eligible = findings.filter((f) => f.evidence.status === 'not_exercised' &&
|
|
33
|
+
f.file !== undefined &&
|
|
34
|
+
isSafeRepoPath(f.file) &&
|
|
35
|
+
severityGates.includes(f.severity ?? ''));
|
|
36
|
+
// U9 — a confident triage top_risk_area reorders candidates so probes
|
|
37
|
+
// prefer the flagged subsystem. Stable sort keeps the original order
|
|
38
|
+
// within each group; probe count/gates/verdict are unchanged.
|
|
39
|
+
if (triageArea !== undefined && triageArea.confidence >= MIN_AREA_CONFIDENCE) {
|
|
40
|
+
// Hit flags precomputed once — the comparator would re-derive them
|
|
41
|
+
// O(n log n) times otherwise.
|
|
42
|
+
const hit = new Map(eligible.map((f) => [f, fileHitsArea(f.file ?? '', triageArea.area)]));
|
|
43
|
+
eligible.sort((a, b) => Number(hit.get(b)) - Number(hit.get(a)));
|
|
44
|
+
}
|
|
45
|
+
return eligible.slice(0, Math.max(0, maxProbes));
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* Repo-relative path gate for anything model- or index-derived that is read
|
|
49
|
+
* or written on the HOST: no absolute paths, no `..` escapes, no backslashes.
|
|
50
|
+
* The write side is already basename-bound (PROBE_FILENAME_RE); this is the
|
|
51
|
+
* read-side and exemplar-dir boundary — a prompt-injected `../../.env`
|
|
52
|
+
* finding.file or a crafted index entry must never reach fs calls.
|
|
53
|
+
*/
|
|
54
|
+
export function isSafeRepoPath(p) {
|
|
55
|
+
if (isAbsolute(p) || p.includes('\\'))
|
|
56
|
+
return false;
|
|
57
|
+
// posix normalize — on win32, normalize() turns `../x` into `..\x` and
|
|
58
|
+
// the startsWith('../') check would miss it.
|
|
59
|
+
const n = posix.normalize(p);
|
|
60
|
+
return n !== '..' && !n.startsWith('../') && !posix.isAbsolute(n);
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Nearest existing test file to the finding's file — same directory first,
|
|
64
|
+
* then same top-level segment, then any test file. The exemplar sets the
|
|
65
|
+
* probe's write location so the consumer's own include/roots cover it.
|
|
66
|
+
* Index paths are filtered through isSafeRepoPath — a committed/crafted
|
|
67
|
+
* index could otherwise aim the host write outside the checkout.
|
|
68
|
+
*/
|
|
69
|
+
export function findExemplarTest(index, findingFile) {
|
|
70
|
+
const tests = index?.entries.map((e) => e.path).filter((p) => isTestFile(p) && isSafeRepoPath(p)) ?? [];
|
|
71
|
+
if (tests.length === 0)
|
|
72
|
+
return undefined;
|
|
73
|
+
const dir = dirname(findingFile);
|
|
74
|
+
const same = tests.find((t) => dirname(t) === dir);
|
|
75
|
+
if (same !== undefined)
|
|
76
|
+
return same;
|
|
77
|
+
const top = findingFile.split('/')[0];
|
|
78
|
+
return tests.find((t) => t.split('/')[0] === top) ?? tests[0];
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* Ensure `baseSha` is fetchable and checked out as a detached worktree at
|
|
82
|
+
* `wtDir`. Stale registrations from killed runs are pruned first — a
|
|
83
|
+
* leftover `probes-base` would otherwise fail `worktree add` and silently
|
|
84
|
+
* degrade every later run to head-only. The auth header rides env config
|
|
85
|
+
* (GIT_CONFIG_*) so the token never appears in the git argv/`ps`.
|
|
86
|
+
* Returns the worktree path, or undefined when the base cannot be
|
|
87
|
+
* materialized — probes then run head-only and can never mark `reproduced`.
|
|
88
|
+
*/
|
|
89
|
+
async function addBaseWorktree(exec, cwd, wtDir, baseSha, token) {
|
|
90
|
+
await exec('git', ['-C', cwd, 'worktree', 'remove', '--force', wtDir], 30_000);
|
|
91
|
+
await exec('git', ['-C', cwd, 'worktree', 'prune'], 15_000);
|
|
92
|
+
const have = await exec('git', ['-C', cwd, 'cat-file', '-e', `${baseSha}^{commit}`], 15_000);
|
|
93
|
+
if (have.code !== 0) {
|
|
94
|
+
// Shallow PR checkouts lack the base — fetch it. Auth rides env config
|
|
95
|
+
// like actions/checkout's extraheader, but env keeps the token out of
|
|
96
|
+
// the process argv where co-tenant jobs could scrape it via /proc.
|
|
97
|
+
const env = token === undefined
|
|
98
|
+
? undefined
|
|
99
|
+
: {
|
|
100
|
+
GIT_CONFIG_COUNT: '1',
|
|
101
|
+
GIT_CONFIG_KEY_0: 'http.https://github.com/.extraheader',
|
|
102
|
+
GIT_CONFIG_VALUE_0: `AUTHORIZATION: basic ${Buffer.from(`x-access-token:${token}`).toString('base64')}`,
|
|
103
|
+
};
|
|
104
|
+
const fetched = await exec('git', ['-C', cwd, 'fetch', '--depth', '1', 'origin', baseSha], 60_000, env);
|
|
105
|
+
if (fetched.code !== 0)
|
|
106
|
+
return undefined;
|
|
107
|
+
}
|
|
108
|
+
const added = await exec('git', ['-C', cwd, 'worktree', 'add', '--detach', wtDir, baseSha], 60_000);
|
|
109
|
+
return added.code === 0 ? wtDir : undefined;
|
|
110
|
+
}
|
|
111
|
+
async function removeBaseWorktree(exec, cwd, wtDir) {
|
|
112
|
+
const res = await exec('git', ['-C', cwd, 'worktree', 'remove', '--force', wtDir], 30_000).catch(() => ({ code: 1, stdout: '', stderr: 'exec failed' }));
|
|
113
|
+
if (res.code !== 0) {
|
|
114
|
+
// rm fallback AND prune — without prune the stale .git/worktrees entry
|
|
115
|
+
// makes the next `worktree add` fail and every later run degrades to
|
|
116
|
+
// head-only.
|
|
117
|
+
await rm(wtDir, { recursive: true, force: true }).catch(() => undefined);
|
|
118
|
+
await exec('git', ['-C', cwd, 'worktree', 'prune'], 15_000).catch(() => undefined);
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
/** Read a repo file for the prompt — windowed around the finding's line, hard-capped. */
|
|
122
|
+
const FILE_WINDOW_LINES = 300;
|
|
123
|
+
const FILE_CAP_BYTES = 64 * 1024;
|
|
124
|
+
const EXEMPLAR_CAP_BYTES = 32 * 1024;
|
|
125
|
+
function windowContent(content, line) {
|
|
126
|
+
if (content.length <= FILE_CAP_BYTES)
|
|
127
|
+
return content;
|
|
128
|
+
if (line === undefined)
|
|
129
|
+
return `${content.slice(0, FILE_CAP_BYTES)}\n…[truncated]`;
|
|
130
|
+
const lines = content.split('\n');
|
|
131
|
+
const lo = Math.max(0, line - 1 - FILE_WINDOW_LINES);
|
|
132
|
+
const hi = Math.min(lines.length, line - 1 + FILE_WINDOW_LINES);
|
|
133
|
+
const windowed = lines.slice(lo, hi).join('\n');
|
|
134
|
+
const capped = windowed.length > FILE_CAP_BYTES ? windowed.slice(0, FILE_CAP_BYTES) : windowed;
|
|
135
|
+
return `${lo > 0 ? `…[${lo} earlier lines omitted]\n` : ''}${capped}${hi < lines.length ? `\n…[${lines.length - hi} later lines omitted]` : ''}`;
|
|
136
|
+
}
|
|
137
|
+
/** One bounded authoring call; returns the validated probe or undefined. */
|
|
138
|
+
async function authorProbe(o, target, harness, exemplarPath, indexPaths) {
|
|
139
|
+
// target.file is model-emitted — require index membership so a
|
|
140
|
+
// prompt-injected path can't make the host read (and exfil) arbitrary
|
|
141
|
+
// files into the authoring prompt.
|
|
142
|
+
const fileTrusted = target.file !== undefined && indexPaths?.has(target.file) === true;
|
|
143
|
+
const [rawFile, rawExemplar] = await Promise.all([
|
|
144
|
+
fileTrusted
|
|
145
|
+
? readFile(join(o.cwd, target.file), 'utf8').catch(() => undefined)
|
|
146
|
+
: Promise.resolve(undefined),
|
|
147
|
+
exemplarPath === undefined
|
|
148
|
+
? Promise.resolve(undefined)
|
|
149
|
+
: readFile(join(o.cwd, exemplarPath), 'utf8').catch(() => undefined),
|
|
150
|
+
]);
|
|
151
|
+
const fileContents = rawFile === undefined ? undefined : windowContent(rawFile, target.line);
|
|
152
|
+
const exemplar = exemplarPath === undefined || rawExemplar === undefined
|
|
153
|
+
? undefined
|
|
154
|
+
: {
|
|
155
|
+
path: exemplarPath,
|
|
156
|
+
content: rawExemplar.length > EXEMPLAR_CAP_BYTES
|
|
157
|
+
? `${rawExemplar.slice(0, EXEMPLAR_CAP_BYTES)}\n…[truncated]`
|
|
158
|
+
: rawExemplar,
|
|
159
|
+
};
|
|
160
|
+
let response;
|
|
161
|
+
try {
|
|
162
|
+
response = await o.client.complete({
|
|
163
|
+
model: o.model,
|
|
164
|
+
messages: buildProbeMessages({
|
|
165
|
+
file: target.file,
|
|
166
|
+
line: target.line,
|
|
167
|
+
severity: target.severity ?? 'bug',
|
|
168
|
+
message: target.message ?? '',
|
|
169
|
+
}, fileContents, exemplar, harness),
|
|
170
|
+
schema: PROBE_SCHEMA,
|
|
171
|
+
kind: 'code',
|
|
172
|
+
...(o.provider !== undefined ? { provider: o.provider } : {}),
|
|
173
|
+
});
|
|
174
|
+
}
|
|
175
|
+
catch (e) {
|
|
176
|
+
return {
|
|
177
|
+
probe: undefined,
|
|
178
|
+
reason: `authoring call failed: ${e.message}`,
|
|
179
|
+
costUsd: 0,
|
|
180
|
+
tokens: 0,
|
|
181
|
+
};
|
|
182
|
+
}
|
|
183
|
+
o.ledger.recordCall(response.cost);
|
|
184
|
+
o.calls?.push(response.cost);
|
|
185
|
+
const parsed = parseProbe(response.content);
|
|
186
|
+
if (!parsed.ok) {
|
|
187
|
+
// Failed parses still cost the call — carry the spend on the record.
|
|
188
|
+
return {
|
|
189
|
+
probe: undefined,
|
|
190
|
+
reason: parsed.reason,
|
|
191
|
+
costUsd: response.cost.costUsd,
|
|
192
|
+
tokens: response.cost.tokens,
|
|
193
|
+
};
|
|
194
|
+
}
|
|
195
|
+
return { probe: parsed.probe, costUsd: response.cost.costUsd, tokens: response.cost.tokens };
|
|
196
|
+
}
|
|
197
|
+
function record(target, probe, outcome, detail, extra = {}) {
|
|
198
|
+
return {
|
|
199
|
+
file: probe?.filename ?? '(none)',
|
|
200
|
+
findingFile: target.file,
|
|
201
|
+
findingLine: target.line,
|
|
202
|
+
outcome,
|
|
203
|
+
durationMs: extra.durationMs ?? 0,
|
|
204
|
+
costUsd: extra.costUsd ?? 0,
|
|
205
|
+
tokens: extra.tokens ?? 0,
|
|
206
|
+
detail,
|
|
207
|
+
headOutcome: extra.headOutcome,
|
|
208
|
+
baseOutcome: extra.baseOutcome,
|
|
209
|
+
output: extra.output,
|
|
210
|
+
};
|
|
211
|
+
}
|
|
212
|
+
/** `timedOut` or the never-ran `-1` sentinel → infra error; else the harness classifies. */
|
|
213
|
+
function outcomeOf(harness, r) {
|
|
214
|
+
return r.timedOut || r.exitCode === -1 ? 'error' : harness.classify(r);
|
|
215
|
+
}
|
|
216
|
+
function probeOutput(stdout, stderr) {
|
|
217
|
+
// Strip control chars + ANSI before capping — attacker-influenced probe
|
|
218
|
+
// output lands in code-review.json and the sticky comment verbatim.
|
|
219
|
+
const combined = stripControlChars(`${stdout}\n${stderr}`)
|
|
220
|
+
// eslint-disable-next-line no-control-regex
|
|
221
|
+
.replace(/\x1b\[[0-9;?]*[a-zA-Z]/g, '')
|
|
222
|
+
.trim();
|
|
223
|
+
return combined === '' ? undefined : combined.slice(0, SANDBOX_OUTPUT_CAP);
|
|
224
|
+
}
|
|
225
|
+
/**
|
|
226
|
+
* Run the probe lane. Mutates `findings` evidence in place for reproduced
|
|
227
|
+
* results and returns the per-probe audit records for code-review.json.
|
|
228
|
+
* Returns undefined only when the lane is disabled entirely; otherwise a
|
|
229
|
+
* result carrying records and (when it bowed out early) a skipReason the
|
|
230
|
+
* report surface can render.
|
|
231
|
+
*/
|
|
232
|
+
export async function runProbeLane(findings, o) {
|
|
233
|
+
const log = o.log ?? (() => undefined);
|
|
234
|
+
if (!o.sandbox.enabled)
|
|
235
|
+
return undefined;
|
|
236
|
+
const skip = (reason) => {
|
|
237
|
+
log(`probes: skipped — ${reason}`);
|
|
238
|
+
return { records: [], skipReason: reason };
|
|
239
|
+
};
|
|
240
|
+
// For fork PRs the config file ships in the PR's own tree — attacker-set
|
|
241
|
+
// image/limits/allowForks would self-approve the gate. Forks always run
|
|
242
|
+
// on DEFAULT_SANDBOX and approve only via trusted author or the label.
|
|
243
|
+
const sandbox = o.meta?.isFork ? { ...DEFAULT_SANDBOX, enabled: true } : o.sandbox;
|
|
244
|
+
const targets = selectProbeTargets(findings, o.severityGates, sandbox.maxProbes, o.triageArea);
|
|
245
|
+
if (targets.length === 0)
|
|
246
|
+
return { records: [] };
|
|
247
|
+
if (!mayProbePr(o.meta, sandbox)) {
|
|
248
|
+
return skip('fork gate (needs argus-probe label on this head or allowForks)');
|
|
249
|
+
}
|
|
250
|
+
// Cheap check first: repos without a supported harness skip before docker
|
|
251
|
+
// ever runs (a possible image pull would be wasted work).
|
|
252
|
+
const harness = await detectHarness(o.cwd);
|
|
253
|
+
if (harness === undefined) {
|
|
254
|
+
return skip('no supported test harness (vitest/jest/node --test)');
|
|
255
|
+
}
|
|
256
|
+
const image = resolveSandboxImage(sandbox.image);
|
|
257
|
+
const exec = o.exec ?? defaultExec;
|
|
258
|
+
if (!(await dockerAvailable(exec, image, o.cwd))) {
|
|
259
|
+
return skip('docker unavailable or cannot see the workspace');
|
|
260
|
+
}
|
|
261
|
+
// Validate reportDir's ancestry BEFORE mkdir — a symlinked reportDir must
|
|
262
|
+
// not create dirs at an arbitrary host path.
|
|
263
|
+
const dirCheck = await checkSandboxPaths(o.cwd, o.reportDir);
|
|
264
|
+
if (!dirCheck.ok)
|
|
265
|
+
return skip(`report dir unsafe — ${dirCheck.reason}`);
|
|
266
|
+
const scratchDir = join(o.reportDir, SCRATCH_DIR_NAME);
|
|
267
|
+
// The single writable mount must actually be writable by nobody (65534).
|
|
268
|
+
// mkdir can reject (EACCES on reportDir, ENOTDIR on a file) — degrade to
|
|
269
|
+
// a skip, not a lane crash.
|
|
270
|
+
try {
|
|
271
|
+
await mkdir(scratchDir, { recursive: true, mode: 0o777 });
|
|
272
|
+
await chmod(scratchDir, 0o777);
|
|
273
|
+
}
|
|
274
|
+
catch (e) {
|
|
275
|
+
return skip(`scratch dir unusable — ${e.message}`);
|
|
276
|
+
}
|
|
277
|
+
const scratchCheck = await checkSandboxPaths(o.cwd, scratchDir);
|
|
278
|
+
if (!scratchCheck.ok)
|
|
279
|
+
return skip(scratchCheck.reason);
|
|
280
|
+
const indexPaths = o.index === undefined ? undefined : new Set(o.index.entries.map((e) => e.path));
|
|
281
|
+
// The double-run needs a merge-base checkout. Started eagerly so the
|
|
282
|
+
// fetch overlaps the first authoring call; failures don't block the lane —
|
|
283
|
+
// head results still record, but without base nothing can upgrade to
|
|
284
|
+
// `reproduced`.
|
|
285
|
+
const wtDir = join(o.reportDir, 'probes-base');
|
|
286
|
+
const basePromise = (o.meta?.baseSha === undefined
|
|
287
|
+
? Promise.resolve(undefined)
|
|
288
|
+
: addBaseWorktree(exec, o.cwd, wtDir, o.meta.baseSha, o.token)).catch(() => undefined);
|
|
289
|
+
const records = [];
|
|
290
|
+
let baseDir;
|
|
291
|
+
try {
|
|
292
|
+
for (let i = 0; i < targets.length; i++) {
|
|
293
|
+
const target = targets[i];
|
|
294
|
+
// Authoring stops at the budget edge but must NOT flag the ledger —
|
|
295
|
+
// `budgetExceeded` flips the review verdict, and KTD8 pins the lane
|
|
296
|
+
// to never change verdict. The spend still records on the ledger.
|
|
297
|
+
if (o.budgetUsd !== undefined && o.ledger.visionCostUsd >= o.budgetUsd) {
|
|
298
|
+
log('probes: authoring budget reached — remaining findings stay not_exercised');
|
|
299
|
+
break;
|
|
300
|
+
}
|
|
301
|
+
const exemplarPath = target.file === undefined ? undefined : findExemplarTest(o.index, target.file);
|
|
302
|
+
const authored = await authorProbe(o, target, harness, exemplarPath, indexPaths);
|
|
303
|
+
if (authored.probe === undefined) {
|
|
304
|
+
records.push(record(target, undefined, 'error', `authoring failed: ${authored.reason}`, {
|
|
305
|
+
costUsd: authored.costUsd,
|
|
306
|
+
tokens: authored.tokens,
|
|
307
|
+
}));
|
|
308
|
+
continue;
|
|
309
|
+
}
|
|
310
|
+
// The host write uses a forced argus-probe- prefix + exclusive create —
|
|
311
|
+
// a model-chosen name can never overwrite (then delete) a real test.
|
|
312
|
+
const probe = authored.probe;
|
|
313
|
+
const safeFilename = `argus-probe-${probe.filename}`;
|
|
314
|
+
const relProbe = exemplarPath === undefined ? safeFilename : join(dirname(exemplarPath), safeFilename);
|
|
315
|
+
if (!isSafeRepoPath(relProbe)) {
|
|
316
|
+
records.push(record(target, probe, 'error', `unsafe probe path: ${relProbe}`));
|
|
317
|
+
continue;
|
|
318
|
+
}
|
|
319
|
+
// `..` imports are legit (tests/ → ../src/x) but must resolve inside
|
|
320
|
+
// the checkout from the probe's write location.
|
|
321
|
+
if (!probeImportsSafe(probe, relProbe)) {
|
|
322
|
+
records.push(record(target, probe, 'error', 'relative import escapes repo'));
|
|
323
|
+
continue;
|
|
324
|
+
}
|
|
325
|
+
// Write on the host — the ro workspace mount exposes it to head, and
|
|
326
|
+
// a copy into the base worktree makes the double-run symmetric.
|
|
327
|
+
baseDir ??= await basePromise;
|
|
328
|
+
let wroteHead = false;
|
|
329
|
+
try {
|
|
330
|
+
await writeFile(join(o.cwd, relProbe), probe.content, { encoding: 'utf8', flag: 'wx' });
|
|
331
|
+
wroteHead = true;
|
|
332
|
+
if (baseDir !== undefined) {
|
|
333
|
+
await mkdir(dirname(join(baseDir, relProbe)), { recursive: true });
|
|
334
|
+
await writeFile(join(baseDir, relProbe), probe.content, 'utf8');
|
|
335
|
+
}
|
|
336
|
+
}
|
|
337
|
+
catch (e) {
|
|
338
|
+
// Only remove what WE created — on EEXIST the path holds a real
|
|
339
|
+
// consumer file and rm would delete it.
|
|
340
|
+
if (wroteHead)
|
|
341
|
+
await rm(join(o.cwd, relProbe), { force: true }).catch(() => undefined);
|
|
342
|
+
records.push(record(target, probe, 'error', `probe write failed: ${e.message}`));
|
|
343
|
+
continue;
|
|
344
|
+
}
|
|
345
|
+
try {
|
|
346
|
+
// Head and base runs are independent — distinct workdirs, scratch
|
|
347
|
+
// dirs, and container names — so they run concurrently. The base
|
|
348
|
+
// worktree has no node_modules (worktrees carry tracked files
|
|
349
|
+
// only), so head's deps are bind-mounted ro — without it every
|
|
350
|
+
// vitest/jest base run load-errors and `reproduced` can never fire.
|
|
351
|
+
const nodeModules = join(o.cwd, 'node_modules');
|
|
352
|
+
const hasNodeModules = (await stat(nodeModules).catch(() => undefined))?.isDirectory() === true;
|
|
353
|
+
const baseScratch = baseDir === undefined ? undefined : join(baseDir, SCRATCH_DIR_NAME);
|
|
354
|
+
if (baseScratch !== undefined) {
|
|
355
|
+
await mkdir(baseScratch, { recursive: true, mode: 0o777 });
|
|
356
|
+
await chmod(baseScratch, 0o777).catch(() => undefined);
|
|
357
|
+
}
|
|
358
|
+
// probes-base sits inside the head run's ro mount — mask it so a
|
|
359
|
+
// head probe can't detect/read the base tree and condition on it.
|
|
360
|
+
// Both sides realpath'd — a symlinked checkout must not silently
|
|
361
|
+
// drop the mask.
|
|
362
|
+
const realWt = baseDir === undefined ? undefined : await realpath(baseDir).catch(() => undefined);
|
|
363
|
+
const wtMask = realWt === undefined ? undefined : relative(scratchCheck.realWork, realWt);
|
|
364
|
+
const [head, base] = await Promise.all([
|
|
365
|
+
runProbeInSandbox({
|
|
366
|
+
workdir: o.cwd,
|
|
367
|
+
scratchDir,
|
|
368
|
+
cmd: harness.runCmd(relProbe),
|
|
369
|
+
image,
|
|
370
|
+
name: `head-${i}`,
|
|
371
|
+
exec,
|
|
372
|
+
masks: wtMask !== undefined && isSafeRepoPath(wtMask) ? [wtMask] : [],
|
|
373
|
+
...sandboxLimits(sandbox),
|
|
374
|
+
}),
|
|
375
|
+
baseDir === undefined || baseScratch === undefined
|
|
376
|
+
? Promise.resolve(undefined)
|
|
377
|
+
: runProbeInSandbox({
|
|
378
|
+
workdir: baseDir,
|
|
379
|
+
scratchDir: baseScratch,
|
|
380
|
+
cmd: harness.runCmd(relProbe),
|
|
381
|
+
image,
|
|
382
|
+
name: `base-${i}`,
|
|
383
|
+
exec,
|
|
384
|
+
roMounts: hasNodeModules
|
|
385
|
+
? [{ host: nodeModules, container: '/work/node_modules' }]
|
|
386
|
+
: [],
|
|
387
|
+
...sandboxLimits(sandbox),
|
|
388
|
+
}),
|
|
389
|
+
]);
|
|
390
|
+
const headOutcome = outcomeOf(harness, head);
|
|
391
|
+
const baseOutcome = base === undefined ? undefined : outcomeOf(harness, base);
|
|
392
|
+
const output = probeOutput(head.stdout, head.stderr) ??
|
|
393
|
+
(base === undefined ? undefined : probeOutput(base.stdout, base.stderr));
|
|
394
|
+
const extra = {
|
|
395
|
+
headOutcome,
|
|
396
|
+
baseOutcome,
|
|
397
|
+
durationMs: head.durationMs,
|
|
398
|
+
costUsd: authored.costUsd,
|
|
399
|
+
tokens: authored.tokens,
|
|
400
|
+
output,
|
|
401
|
+
};
|
|
402
|
+
let outcome;
|
|
403
|
+
let detail;
|
|
404
|
+
if (headOutcome === 'failed-test' && baseOutcome === 'clean') {
|
|
405
|
+
outcome = 'reproduced';
|
|
406
|
+
detail = `reproduced by Argus probe ${probe.filename} (fails on head, clean on base)`;
|
|
407
|
+
target.evidence = { ...target.evidence, status: 'reproduced', detail };
|
|
408
|
+
}
|
|
409
|
+
else if (headOutcome !== 'failed-test') {
|
|
410
|
+
outcome = headOutcome;
|
|
411
|
+
detail = `head outcome: ${headOutcome}`;
|
|
412
|
+
}
|
|
413
|
+
else if (baseOutcome === undefined) {
|
|
414
|
+
outcome = 'error';
|
|
415
|
+
detail = 'fails on head but base checkout unavailable — unverified';
|
|
416
|
+
}
|
|
417
|
+
else if (baseOutcome === 'failed-test') {
|
|
418
|
+
outcome = 'load-error';
|
|
419
|
+
detail = 'fails on both head and base — probe bug or pre-existing defect';
|
|
420
|
+
}
|
|
421
|
+
else {
|
|
422
|
+
// Base run produced no test verdict (infra error / load-error /
|
|
423
|
+
// not-collected) — report it accurately, never claim "fails on both".
|
|
424
|
+
outcome = 'error';
|
|
425
|
+
detail = `fails on head but base run inconclusive (${baseOutcome}) — unverified`;
|
|
426
|
+
}
|
|
427
|
+
records.push(record(target, probe, outcome, detail, extra));
|
|
428
|
+
}
|
|
429
|
+
finally {
|
|
430
|
+
await rm(join(o.cwd, relProbe), { force: true }).catch(() => undefined);
|
|
431
|
+
if (baseDir !== undefined) {
|
|
432
|
+
await rm(join(baseDir, relProbe), { force: true }).catch(() => undefined);
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
}
|
|
437
|
+
finally {
|
|
438
|
+
// The worktree may exist even when the loop never assigned baseDir
|
|
439
|
+
// (e.g. every probe failed authoring) — resolve the promise here so the
|
|
440
|
+
// teardown still runs.
|
|
441
|
+
const bd = baseDir ?? (await basePromise);
|
|
442
|
+
if (bd !== undefined)
|
|
443
|
+
await removeBaseWorktree(exec, o.cwd, wtDir);
|
|
444
|
+
}
|
|
445
|
+
return { records };
|
|
446
|
+
}
|