argus-reviewer-e2e 0.1.3 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -5
- package/action/action.yml +17 -0
- package/action/sticky-comment.mjs +270 -90
- package/dist/api.d.ts +1 -0
- package/dist/api.js +1 -0
- package/dist/cli.d.ts +67 -0
- package/dist/cli.js +383 -49
- package/dist/config.d.ts +63 -2
- package/dist/config.js +108 -5
- package/dist/debug.d.ts +1 -0
- package/dist/debug.js +9 -3
- package/dist/evidence/ci.d.ts +3 -0
- package/dist/evidence/ci.js +3 -1
- package/dist/probe/queue.d.ts +4 -1
- package/dist/probe/queue.js +46 -8
- package/dist/review/adjudicate.d.ts +63 -0
- package/dist/review/adjudicate.js +111 -0
- package/dist/review/secrets.d.ts +88 -0
- package/dist/review/secrets.js +220 -0
- package/dist/review/triage.d.ts +76 -0
- package/dist/review/triage.js +163 -0
- package/dist/trust.d.ts +50 -0
- package/dist/trust.js +103 -0
- package/dist/vision/cost.d.ts +14 -1
- package/dist/vision/cost.js +13 -0
- package/dist/vision/decisions.d.ts +95 -0
- package/dist/vision/decisions.js +232 -0
- package/package.json +3 -2
package/dist/config.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { Trust } from './trust.js';
|
|
1
2
|
export interface ProviderRules {
|
|
2
3
|
only?: string[];
|
|
3
4
|
ignore?: string[];
|
|
@@ -59,6 +60,13 @@ export interface Config {
|
|
|
59
60
|
* PR diffs and post findings. Defaults to the primary `model` if not set.
|
|
60
61
|
*/
|
|
61
62
|
code_model: string | undefined;
|
|
63
|
+
/**
|
|
64
|
+
* OpenRouter Decisions API model for typed adjudication (Jev). Defaults
|
|
65
|
+
* to the pinned `typesafe/jev-1.13-20260917` — alias slugs like
|
|
66
|
+
* `~typesafe/jev-latest` drift silently and thresholds are calibrated
|
|
67
|
+
* to a version. Set to `''` to disable adjudication (regex-only mode).
|
|
68
|
+
*/
|
|
69
|
+
decisionModel: string | undefined;
|
|
62
70
|
/**
|
|
63
71
|
* Hard budget for the `argus-reviewer code-review` lane. When set, the
|
|
64
72
|
* review stops early if the cumulative OpenRouter cost exceeds this cap.
|
|
@@ -152,16 +160,69 @@ export interface Config {
|
|
|
152
160
|
* `resolveConfig` — `enabled: false` by default so the lane is opt-in.
|
|
153
161
|
*/
|
|
154
162
|
sandbox: Sandbox;
|
|
163
|
+
/**
|
|
164
|
+
* Code-review policy knobs. Always populated after `resolveConfig`.
|
|
165
|
+
* `secretsThreshold`: Jev `noul` probability at/above which a
|
|
166
|
+
* secret-shaped diff literal is reported as a finding (below →
|
|
167
|
+
* suppressed but audit-recorded). Default 0.3 — tune after dogfooding.
|
|
168
|
+
* `maxComments`: cap on inline review comments posted per run
|
|
169
|
+
* (default 20) — overflow is summarized count-only in the sticky.
|
|
170
|
+
* `severityGate`: consumer-facing alias over `severity` — 'bug'
|
|
171
|
+
* fails on bugs only, 'risk' fails on bug|risk. Unset → `severity`
|
|
172
|
+
* list is authoritative.
|
|
173
|
+
* `triage`: Jev pre-review lane — 'off' no call, 'annotate' (default)
|
|
174
|
+
* records risk/deep-review/area into the report + sticky, 'route'
|
|
175
|
+
* additionally swaps the code model to `lowRiskModel` on low-risk
|
|
176
|
+
* diffs. Jev routes/annotates, never gates — coverage is constant.
|
|
177
|
+
* `lowRiskModel`: the cheap code-model slug 'route' falls to; unset →
|
|
178
|
+
* route keeps `code_model` (annotate-equivalent).
|
|
179
|
+
* `findingThreshold`: P(false-positive) required to suppress a nit/q
|
|
180
|
+
* finding after Jev adjudication — 1.0 (default) is annotate-only,
|
|
181
|
+
* lowering it suppresses progressively more low-confidence nits.
|
|
182
|
+
* bug/risk are never suppressed.
|
|
183
|
+
*/
|
|
184
|
+
review: {
|
|
185
|
+
secretsThreshold: number;
|
|
186
|
+
maxComments: number;
|
|
187
|
+
severityGate: 'bug' | 'risk' | undefined;
|
|
188
|
+
triage: 'off' | 'annotate' | 'route';
|
|
189
|
+
lowRiskModel: string | undefined;
|
|
190
|
+
findingThreshold: number;
|
|
191
|
+
};
|
|
155
192
|
}
|
|
156
|
-
export type ConfigInput = Partial<Omit<Config, 'provider' | 'sandbox'>> & {
|
|
193
|
+
export type ConfigInput = Partial<Omit<Config, 'provider' | 'sandbox' | 'review'>> & {
|
|
157
194
|
provider?: Partial<ProviderRules>;
|
|
158
195
|
sandbox?: Partial<Sandbox>;
|
|
196
|
+
review?: Partial<Config['review']>;
|
|
159
197
|
};
|
|
160
198
|
export declare const DEFAULT_RECORD_STEP_CAP = 40;
|
|
161
199
|
export declare const DEFAULT_SANDBOX: Sandbox;
|
|
162
200
|
export declare function defineConfig(input: ConfigInput): ConfigInput;
|
|
201
|
+
/**
|
|
202
|
+
* Which severities fail the review status. `review.severityGate` is the
|
|
203
|
+
* consumer-facing alias over `severity` — 'risk' fails on bug|risk,
|
|
204
|
+
* 'bug' on bugs only; unset → the `severity` list is authoritative.
|
|
205
|
+
*/
|
|
206
|
+
export declare function resolveBlockSeverities(config: Config): string[];
|
|
207
|
+
/**
|
|
208
|
+
* Inline-comment cap: `ARGUS_MAX_COMMENTS` (the action's `max-comments`
|
|
209
|
+
* input) wins when it parses as a non-negative integer — it's set by the
|
|
210
|
+
* workflow author, so an untrusted PR config can't reach it (`review`
|
|
211
|
+
* isn't on the untrusted allowlist). Anything else → `review.maxComments`.
|
|
212
|
+
*/
|
|
213
|
+
export declare function resolveMaxComments(env: Record<string, string | undefined>, config: Config): number;
|
|
163
214
|
export declare function resolveConfig(input?: ConfigInput): Config;
|
|
164
|
-
export
|
|
215
|
+
export interface LoadConfigOpts {
|
|
216
|
+
/**
|
|
217
|
+
* Required — there is no default. Every call site must state the
|
|
218
|
+
* checkout's trust so a missed or future caller can't silently execute
|
|
219
|
+
* config code on a hostile tree (see src/trust.ts).
|
|
220
|
+
*/
|
|
221
|
+
trust: Trust;
|
|
222
|
+
/** Human-readable note on security-relevant load decisions (e.g. ctx.err). */
|
|
223
|
+
note?: (line: string) => void;
|
|
224
|
+
}
|
|
225
|
+
export declare function loadConfig(cwd: string, opts: LoadConfigOpts): Promise<Config>;
|
|
165
226
|
/**
|
|
166
227
|
* Provider slugs the harness recognizes for `provider.only/ignore/order`
|
|
167
228
|
* (KTD4). Unknown slugs warn but do not fail — OpenRouter's catalog changes
|
package/dist/config.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { pathToFileURL } from 'node:url';
|
|
2
|
+
import { JEV_DEFAULT_MODEL } from './vision/decisions.js';
|
|
2
3
|
export const DEFAULT_RECORD_STEP_CAP = 40;
|
|
3
4
|
export const DEFAULT_SANDBOX = {
|
|
4
5
|
enabled: false,
|
|
@@ -15,6 +16,7 @@ const defaults = {
|
|
|
15
16
|
escalation_model: 'moonshotai/kimi-k2.5',
|
|
16
17
|
grounding_model: undefined,
|
|
17
18
|
code_model: 'deepseek/deepseek-v4.1-flash',
|
|
19
|
+
decisionModel: JEV_DEFAULT_MODEL,
|
|
18
20
|
codeReviewBudgetUsd: undefined,
|
|
19
21
|
provider: {
|
|
20
22
|
ignore: ['siliconflow', 'novitaai', 'atlascloud', 'streamlake', 'chutes'],
|
|
@@ -38,6 +40,14 @@ const defaults = {
|
|
|
38
40
|
a0: undefined,
|
|
39
41
|
heal: 'local',
|
|
40
42
|
sandbox: { ...DEFAULT_SANDBOX },
|
|
43
|
+
review: {
|
|
44
|
+
secretsThreshold: 0.3,
|
|
45
|
+
maxComments: 20,
|
|
46
|
+
severityGate: undefined,
|
|
47
|
+
triage: 'annotate',
|
|
48
|
+
lowRiskModel: undefined,
|
|
49
|
+
findingThreshold: 1.0,
|
|
50
|
+
},
|
|
41
51
|
};
|
|
42
52
|
export function defineConfig(input) {
|
|
43
53
|
return input;
|
|
@@ -46,6 +56,36 @@ export function defineConfig(input) {
|
|
|
46
56
|
function posInt(v, dflt) {
|
|
47
57
|
return v !== undefined && Number.isFinite(v) && v >= 1 ? Math.floor(v) : dflt;
|
|
48
58
|
}
|
|
59
|
+
/** Probability config values (must be in [0,1]) fall back to their default. */
|
|
60
|
+
function prob01(v, dflt) {
|
|
61
|
+
return v !== undefined && Number.isFinite(v) && v >= 0 && v <= 1 ? v : dflt;
|
|
62
|
+
}
|
|
63
|
+
/**
|
|
64
|
+
* Which severities fail the review status. `review.severityGate` is the
|
|
65
|
+
* consumer-facing alias over `severity` — 'risk' fails on bug|risk,
|
|
66
|
+
* 'bug' on bugs only; unset → the `severity` list is authoritative.
|
|
67
|
+
*/
|
|
68
|
+
export function resolveBlockSeverities(config) {
|
|
69
|
+
if (config.review.severityGate === 'risk')
|
|
70
|
+
return ['bug', 'risk'];
|
|
71
|
+
if (config.review.severityGate === 'bug')
|
|
72
|
+
return ['bug'];
|
|
73
|
+
return config.severity ?? ['bug'];
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Inline-comment cap: `ARGUS_MAX_COMMENTS` (the action's `max-comments`
|
|
77
|
+
* input) wins when it parses as a non-negative integer — it's set by the
|
|
78
|
+
* workflow author, so an untrusted PR config can't reach it (`review`
|
|
79
|
+
* isn't on the untrusted allowlist). Anything else → `review.maxComments`.
|
|
80
|
+
*/
|
|
81
|
+
export function resolveMaxComments(env, config) {
|
|
82
|
+
const raw = env.ARGUS_MAX_COMMENTS?.trim();
|
|
83
|
+
// ^\d+$ — Number() would also accept '0x10', '1e2', ' 4 ', 'Infinity'.
|
|
84
|
+
if (raw !== undefined && /^\d+$/.test(raw)) {
|
|
85
|
+
return Number(raw);
|
|
86
|
+
}
|
|
87
|
+
return config.review.maxComments;
|
|
88
|
+
}
|
|
49
89
|
export function resolveConfig(input = {}) {
|
|
50
90
|
const provider = { ...defaults.provider, ...(input.provider ?? {}) };
|
|
51
91
|
// Wrong-typed sandbox values (e.g. `sandbox: true`, `enabled: 'yes'`,
|
|
@@ -56,23 +96,81 @@ export function resolveConfig(input = {}) {
|
|
|
56
96
|
sandbox.enabled = raw.enabled === true;
|
|
57
97
|
sandbox.allowForks = raw.allowForks === true;
|
|
58
98
|
sandbox.image = typeof raw.image === 'string' && raw.image !== '' ? raw.image : undefined;
|
|
59
|
-
sandbox.memory =
|
|
99
|
+
sandbox.memory =
|
|
100
|
+
typeof raw.memory === 'string' && raw.memory !== '' ? raw.memory : DEFAULT_SANDBOX.memory;
|
|
60
101
|
sandbox.cpus = typeof raw.cpus === 'string' && raw.cpus !== '' ? raw.cpus : DEFAULT_SANDBOX.cpus;
|
|
61
102
|
sandbox.maxProbes = posInt(sandbox.maxProbes, DEFAULT_SANDBOX.maxProbes);
|
|
62
103
|
sandbox.timeoutMs = posInt(sandbox.timeoutMs, DEFAULT_SANDBOX.timeoutMs);
|
|
63
104
|
sandbox.pidsLimit = posInt(sandbox.pidsLimit, DEFAULT_SANDBOX.pidsLimit);
|
|
64
|
-
const
|
|
105
|
+
const rawReview = typeof input.review === 'object' && input.review !== null ? input.review : {};
|
|
106
|
+
const review = { ...defaults.review, ...rawReview };
|
|
107
|
+
// Thresholds must be probabilities — anything else (NaN, >1,
|
|
108
|
+
// negative) would silently suppress or flood the Jev lanes.
|
|
109
|
+
review.secretsThreshold = prob01(review.secretsThreshold, defaults.review.secretsThreshold);
|
|
110
|
+
review.maxComments =
|
|
111
|
+
typeof review.maxComments === 'number' &&
|
|
112
|
+
Number.isInteger(review.maxComments) &&
|
|
113
|
+
review.maxComments >= 0
|
|
114
|
+
? review.maxComments
|
|
115
|
+
: defaults.review.maxComments;
|
|
116
|
+
if (review.severityGate !== 'bug' && review.severityGate !== 'risk') {
|
|
117
|
+
review.severityGate = undefined;
|
|
118
|
+
}
|
|
119
|
+
if (review.triage !== 'off' && review.triage !== 'annotate' && review.triage !== 'route') {
|
|
120
|
+
review.triage = defaults.review.triage;
|
|
121
|
+
}
|
|
122
|
+
if (typeof review.lowRiskModel !== 'string' || review.lowRiskModel === '') {
|
|
123
|
+
review.lowRiskModel = undefined;
|
|
124
|
+
}
|
|
125
|
+
review.findingThreshold = prob01(review.findingThreshold, defaults.review.findingThreshold);
|
|
126
|
+
const resolved = { ...defaults, ...input, provider, sandbox, review };
|
|
65
127
|
resolved.recordStepCap = posInt(resolved.recordStepCap, DEFAULT_RECORD_STEP_CAP);
|
|
66
128
|
if (resolved.heal !== 'a0')
|
|
67
129
|
resolved.heal = 'local';
|
|
130
|
+
// '' is the documented opt-out — an empty slug would send a broken
|
|
131
|
+
// model id to the decisions endpoint on every adjudication call.
|
|
132
|
+
if (resolved.decisionModel === '')
|
|
133
|
+
resolved.decisionModel = undefined;
|
|
68
134
|
return resolved;
|
|
69
135
|
}
|
|
70
|
-
|
|
136
|
+
/**
|
|
137
|
+
* Config keys honored on untrusted checkouts — policy-free fields only.
|
|
138
|
+
* Everything else (exec-bearing fields, model/budget/provider selection,
|
|
139
|
+
* severity/verdict policy, credentials maps, network endpoints, write
|
|
140
|
+
* locations) is ignored: the review policy over hostile code must not be
|
|
141
|
+
* authored by that code.
|
|
142
|
+
*/
|
|
143
|
+
const UNTRUSTED_CONFIG_KEYS = new Set(['logLevel', 'sourceGlobs']);
|
|
144
|
+
function filterUntrustedConfig(input) {
|
|
145
|
+
const out = {};
|
|
146
|
+
for (const [key, value] of Object.entries(input)) {
|
|
147
|
+
if (UNTRUSTED_CONFIG_KEYS.has(key))
|
|
148
|
+
out[key] = value;
|
|
149
|
+
}
|
|
150
|
+
return out;
|
|
151
|
+
}
|
|
152
|
+
export async function loadConfig(cwd, opts) {
|
|
71
153
|
const fs = await import('node:fs/promises');
|
|
72
154
|
const path = await import('node:path');
|
|
155
|
+
const untrusted = opts.trust === 'untrusted';
|
|
73
156
|
const names = ['argus-reviewer.config', 'vision-e2e.config'];
|
|
74
157
|
for (const name of names) {
|
|
75
|
-
|
|
158
|
+
if (untrusted) {
|
|
159
|
+
// Surface skipped .ts candidates — otherwise a hostile config (or a
|
|
160
|
+
// legit consumer debugging "why is my config ignored") is invisible.
|
|
161
|
+
try {
|
|
162
|
+
if ((await fs.stat(path.join(cwd, `${name}.ts`))).isFile()) {
|
|
163
|
+
opts.note?.(`config: ${name}.ts ignored — untrusted checkouts load JSON config only`);
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
catch {
|
|
167
|
+
// no .ts candidate — nothing to note
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
// .ts is tried before .json, so an untrusted checkout must skip the .ts
|
|
171
|
+
// candidate *before* it can shadow a committed .json — importing it
|
|
172
|
+
// executes arbitrary code beside the runner's secrets (#58).
|
|
173
|
+
for (const ext of untrusted ? ['.json'] : ['.ts', '.json']) {
|
|
76
174
|
const file = path.join(cwd, `${name}${ext}`);
|
|
77
175
|
try {
|
|
78
176
|
const stat = await fs.stat(file);
|
|
@@ -80,7 +178,12 @@ export async function loadConfig(cwd) {
|
|
|
80
178
|
continue;
|
|
81
179
|
if (ext === '.json') {
|
|
82
180
|
const raw = await fs.readFile(file, 'utf8');
|
|
83
|
-
|
|
181
|
+
const parsed = JSON.parse(raw);
|
|
182
|
+
if (untrusted) {
|
|
183
|
+
opts.note?.(`config: ${name}.json loaded untrusted — honoring ${[...UNTRUSTED_CONFIG_KEYS].join(', ')} only`);
|
|
184
|
+
return resolveConfig(filterUntrustedConfig(parsed));
|
|
185
|
+
}
|
|
186
|
+
return resolveConfig(parsed);
|
|
84
187
|
}
|
|
85
188
|
// Always transpile .ts to a temp .mjs rather than importing natively:
|
|
86
189
|
// Node's built-in type stripping resolves the module type from the
|
package/dist/debug.d.ts
CHANGED
package/dist/debug.js
CHANGED
|
@@ -1,8 +1,14 @@
|
|
|
1
1
|
import { join } from 'node:path';
|
|
2
2
|
import { liveLog } from './live.js';
|
|
3
3
|
const DEBUG = process.env.ARGUS_DEBUG === '1' || process.env.ARGUS_DEBUG === 'true';
|
|
4
|
-
// debug() has no config access — it
|
|
5
|
-
|
|
4
|
+
// debug() has no config access — it defaults to the conventional cache dir,
|
|
5
|
+
// and callers that resolve a config set the real live dir once it's known
|
|
6
|
+
// (otherwise debug lines and stage lines would split across two dirs).
|
|
7
|
+
const DEFAULT_LIVE_DIR = join(process.cwd(), '.argus-reviewer-cache');
|
|
8
|
+
let liveDir = DEFAULT_LIVE_DIR;
|
|
9
|
+
export function setLiveDir(dir) {
|
|
10
|
+
liveDir = dir ?? DEFAULT_LIVE_DIR;
|
|
11
|
+
}
|
|
6
12
|
function toMsg(arg) {
|
|
7
13
|
if (typeof arg === 'string')
|
|
8
14
|
return arg;
|
|
@@ -18,7 +24,7 @@ export function debug(kind, ...args) {
|
|
|
18
24
|
if (!DEBUG)
|
|
19
25
|
return; // keep debug() free of fs work on the hot path
|
|
20
26
|
for (const arg of args) {
|
|
21
|
-
liveLog(
|
|
27
|
+
liveLog(liveDir, kind, 'debug', toMsg(arg));
|
|
22
28
|
const prefix = `[argus-reviewer:${kind}]`;
|
|
23
29
|
if (typeof arg === 'string') {
|
|
24
30
|
console.error(`${prefix} ${arg}`);
|
package/dist/evidence/ci.d.ts
CHANGED
|
@@ -42,6 +42,9 @@ export interface PrMeta {
|
|
|
42
42
|
* timeline fetch failed (the gate fails closed either way).
|
|
43
43
|
*/
|
|
44
44
|
labelApprovedAt: string | undefined;
|
|
45
|
+
/** PR title/body — triage state only (untrusted text; feeds Jev, never gates). */
|
|
46
|
+
title: string | undefined;
|
|
47
|
+
body: string | undefined;
|
|
45
48
|
}
|
|
46
49
|
/**
|
|
47
50
|
* Shared GitHub GET scaffold — Bearer auth, API headers, 30s abort timeout,
|
package/dist/evidence/ci.js
CHANGED
|
@@ -6,7 +6,7 @@ export const PROBE_LABEL = 'argus-probe';
|
|
|
6
6
|
* FIRST_TIMER, MANNEQUIN, and NONE are not.
|
|
7
7
|
*/
|
|
8
8
|
export function isTrustedAssociation(association) {
|
|
9
|
-
return
|
|
9
|
+
return association === 'MEMBER' || association === 'OWNER' || association === 'COLLABORATOR';
|
|
10
10
|
}
|
|
11
11
|
const GH_API = 'https://api.github.com';
|
|
12
12
|
const MAX_CHECK_RUN_PAGES = 5;
|
|
@@ -85,6 +85,8 @@ export async function fetchPrMeta(repo, pr, token, ctx) {
|
|
|
85
85
|
labels,
|
|
86
86
|
pushedAt: data.head?.repo?.pushed_at,
|
|
87
87
|
labelApprovedAt,
|
|
88
|
+
title: typeof data.title === 'string' ? data.title : undefined,
|
|
89
|
+
body: typeof data.body === 'string' ? data.body : undefined,
|
|
88
90
|
};
|
|
89
91
|
}
|
|
90
92
|
/**
|
package/dist/probe/queue.d.ts
CHANGED
|
@@ -6,6 +6,7 @@ import type { RepoIndex } from '../index/scan.js';
|
|
|
6
6
|
import type { VisionClient } from '../engine/loop.js';
|
|
7
7
|
import type { CallCost } from '../vision/cost.js';
|
|
8
8
|
import type { Ledger } from '../vision/ledger.js';
|
|
9
|
+
import type { TriageAreaSignal } from '../review/triage.js';
|
|
9
10
|
import { type ProbeOutcome } from './harness.js';
|
|
10
11
|
/**
|
|
11
12
|
* The B.2 probe lane (U4). Called after `linkFindings` inside `code-review`.
|
|
@@ -69,13 +70,15 @@ export interface ProbeLaneOptions {
|
|
|
69
70
|
/** Configured blocking severities — the queue only admits those. */
|
|
70
71
|
severityGates: string[];
|
|
71
72
|
index: RepoIndex | undefined;
|
|
73
|
+
/** U9 — triage top_risk_area; advisory reorder of probe candidates. */
|
|
74
|
+
triageArea?: TriageAreaSignal | undefined;
|
|
72
75
|
/** Report sink — authored probe CallCosts are pushed here for the report. */
|
|
73
76
|
calls?: CallCost[] | undefined;
|
|
74
77
|
exec?: ExecFn | undefined;
|
|
75
78
|
log?: ((line: string) => void) | undefined;
|
|
76
79
|
}
|
|
77
80
|
/** Pure selection: not_exercised findings at blocking severities, capped. */
|
|
78
|
-
export declare function selectProbeTargets(findings: LinkedFinding[], severityGates: string[], maxProbes: number): LinkedFinding[];
|
|
81
|
+
export declare function selectProbeTargets(findings: LinkedFinding[], severityGates: string[], maxProbes: number, triageArea?: TriageAreaSignal): LinkedFinding[];
|
|
79
82
|
/**
|
|
80
83
|
* Repo-relative path gate for anything model- or index-derived that is read
|
|
81
84
|
* or written on the HOST: no absolute paths, no `..` escapes, no backslashes.
|
package/dist/probe/queue.js
CHANGED
|
@@ -7,14 +7,42 @@ import { isTestFile } from '../evidence/link.js';
|
|
|
7
7
|
import { buildProbeMessages, parseProbe, probeImportsSafe, PROBE_SCHEMA, } from './author.js';
|
|
8
8
|
import { detectHarness } from './harness.js';
|
|
9
9
|
import { checkSandboxPaths, dockerAvailable, resolveSandboxImage, runProbeInSandbox, SANDBOX_OUTPUT_CAP, SCRATCH_DIR_NAME, stripControlChars, sandboxLimits, } from '../executor/sandbox.js';
|
|
10
|
+
/** U9 — below this confidence the triage area signal is ignored. */
|
|
11
|
+
const MIN_AREA_CONFIDENCE = 0.5;
|
|
12
|
+
/**
|
|
13
|
+
* U9 triage-informed ordering: a finding's file path "hits" the flagged
|
|
14
|
+
* risk area when a path segment equals the area token or starts with
|
|
15
|
+
* it at a camelCase/digit boundary — `src/auth/session.ts` and
|
|
16
|
+
* `dataStore.ts` hit auth/data, while `author.ts` and `database.ts`
|
|
17
|
+
* do not. Advisory only.
|
|
18
|
+
*/
|
|
19
|
+
function fileHitsArea(file, area) {
|
|
20
|
+
const token = area.toLowerCase();
|
|
21
|
+
return file.split(/[/._-]+/).some((segment) => {
|
|
22
|
+
const lower = segment.toLowerCase();
|
|
23
|
+
if (lower === token)
|
|
24
|
+
return true;
|
|
25
|
+
if (!lower.startsWith(token))
|
|
26
|
+
return false;
|
|
27
|
+
return /[A-Z0-9]/.test(segment.charAt(token.length));
|
|
28
|
+
});
|
|
29
|
+
}
|
|
10
30
|
/** Pure selection: not_exercised findings at blocking severities, capped. */
|
|
11
|
-
export function selectProbeTargets(findings, severityGates, maxProbes) {
|
|
12
|
-
|
|
13
|
-
.filter((f) => f.evidence.status === 'not_exercised' &&
|
|
31
|
+
export function selectProbeTargets(findings, severityGates, maxProbes, triageArea) {
|
|
32
|
+
const eligible = findings.filter((f) => f.evidence.status === 'not_exercised' &&
|
|
14
33
|
f.file !== undefined &&
|
|
15
34
|
isSafeRepoPath(f.file) &&
|
|
16
|
-
severityGates.includes(f.severity ?? ''))
|
|
17
|
-
|
|
35
|
+
severityGates.includes(f.severity ?? ''));
|
|
36
|
+
// U9 — a confident triage top_risk_area reorders candidates so probes
|
|
37
|
+
// prefer the flagged subsystem. Stable sort keeps the original order
|
|
38
|
+
// within each group; probe count/gates/verdict are unchanged.
|
|
39
|
+
if (triageArea !== undefined && triageArea.confidence >= MIN_AREA_CONFIDENCE) {
|
|
40
|
+
// Hit flags precomputed once — the comparator would re-derive them
|
|
41
|
+
// O(n log n) times otherwise.
|
|
42
|
+
const hit = new Map(eligible.map((f) => [f, fileHitsArea(f.file ?? '', triageArea.area)]));
|
|
43
|
+
eligible.sort((a, b) => Number(hit.get(b)) - Number(hit.get(a)));
|
|
44
|
+
}
|
|
45
|
+
return eligible.slice(0, Math.max(0, maxProbes));
|
|
18
46
|
}
|
|
19
47
|
/**
|
|
20
48
|
* Repo-relative path gate for anything model- or index-derived that is read
|
|
@@ -145,14 +173,24 @@ async function authorProbe(o, target, harness, exemplarPath, indexPaths) {
|
|
|
145
173
|
});
|
|
146
174
|
}
|
|
147
175
|
catch (e) {
|
|
148
|
-
return {
|
|
176
|
+
return {
|
|
177
|
+
probe: undefined,
|
|
178
|
+
reason: `authoring call failed: ${e.message}`,
|
|
179
|
+
costUsd: 0,
|
|
180
|
+
tokens: 0,
|
|
181
|
+
};
|
|
149
182
|
}
|
|
150
183
|
o.ledger.recordCall(response.cost);
|
|
151
184
|
o.calls?.push(response.cost);
|
|
152
185
|
const parsed = parseProbe(response.content);
|
|
153
186
|
if (!parsed.ok) {
|
|
154
187
|
// Failed parses still cost the call — carry the spend on the record.
|
|
155
|
-
return {
|
|
188
|
+
return {
|
|
189
|
+
probe: undefined,
|
|
190
|
+
reason: parsed.reason,
|
|
191
|
+
costUsd: response.cost.costUsd,
|
|
192
|
+
tokens: response.cost.tokens,
|
|
193
|
+
};
|
|
156
194
|
}
|
|
157
195
|
return { probe: parsed.probe, costUsd: response.cost.costUsd, tokens: response.cost.tokens };
|
|
158
196
|
}
|
|
@@ -203,7 +241,7 @@ export async function runProbeLane(findings, o) {
|
|
|
203
241
|
// image/limits/allowForks would self-approve the gate. Forks always run
|
|
204
242
|
// on DEFAULT_SANDBOX and approve only via trusted author or the label.
|
|
205
243
|
const sandbox = o.meta?.isFork ? { ...DEFAULT_SANDBOX, enabled: true } : o.sandbox;
|
|
206
|
-
const targets = selectProbeTargets(findings, o.severityGates, sandbox.maxProbes);
|
|
244
|
+
const targets = selectProbeTargets(findings, o.severityGates, sandbox.maxProbes, o.triageArea);
|
|
207
245
|
if (targets.length === 0)
|
|
208
246
|
return { records: [] };
|
|
209
247
|
if (!mayProbePr(o.meta, sandbox)) {
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import { DecisionClient } from '../vision/decisions.js';
|
|
2
|
+
/**
|
|
3
|
+
* U8 finding adjudication — one batched Jev decide() scores each
|
|
4
|
+
* synthesized finding's true-positive probability before posting. Same
|
|
5
|
+
* posture as the secrets lane: Jev annotates/routes, never gates.
|
|
6
|
+
* `p` lands on the finding record and the sticky confidence column;
|
|
7
|
+
* suppression is scoped to `nit`/`q` severities whose false-positive
|
|
8
|
+
* confidence exceeds `findingThreshold` — `bug`/`risk` are never
|
|
9
|
+
* suppressed, and a Jev outage leaves every finding unadjudicated and
|
|
10
|
+
* unsuppressed (degrade open).
|
|
11
|
+
*
|
|
12
|
+
* `findingThreshold` is the required P(false positive): a nit/q is
|
|
13
|
+
* suppressed when `1 - p > threshold`, i.e. `p < 1 - threshold`.
|
|
14
|
+
* Default 1.0 → nothing can exceed 100% FP → annotate-only.
|
|
15
|
+
*/
|
|
16
|
+
export interface AdjudicableFinding {
|
|
17
|
+
file: string;
|
|
18
|
+
line?: number;
|
|
19
|
+
severity: string;
|
|
20
|
+
category?: string;
|
|
21
|
+
message: string;
|
|
22
|
+
}
|
|
23
|
+
export interface FindingAdjudicationRecord {
|
|
24
|
+
file: string;
|
|
25
|
+
line?: number;
|
|
26
|
+
severity: string;
|
|
27
|
+
category?: string;
|
|
28
|
+
/** Finding text — capped; suppressed findings keep their message in audit. */
|
|
29
|
+
message: string;
|
|
30
|
+
adjudicated: boolean;
|
|
31
|
+
/** Jev P(true positive) for this finding. */
|
|
32
|
+
p?: number;
|
|
33
|
+
/** nit/q below the FP bar — kept for audit, removed from findings. */
|
|
34
|
+
suppressed?: boolean;
|
|
35
|
+
}
|
|
36
|
+
/** The audit half of the result — what report.findingAdjudication stores. */
|
|
37
|
+
export interface FindingAdjudicationAudit {
|
|
38
|
+
records: FindingAdjudicationRecord[];
|
|
39
|
+
/** Findings past MAX_CANDIDATES — unadjudicated, never suppressed. */
|
|
40
|
+
overflow: number;
|
|
41
|
+
/** Whole-call failure — nothing adjudicated, nothing suppressed. */
|
|
42
|
+
unadjudicated?: boolean;
|
|
43
|
+
}
|
|
44
|
+
export interface FindingAdjudicationResult<T extends AdjudicableFinding> extends FindingAdjudicationAudit {
|
|
45
|
+
/** Surviving findings — adjudicated ones carry `p`. */
|
|
46
|
+
findings: (T & {
|
|
47
|
+
p?: number;
|
|
48
|
+
})[];
|
|
49
|
+
}
|
|
50
|
+
export declare function adjudicateFindings<T extends AdjudicableFinding>(opts: {
|
|
51
|
+
findings: T[];
|
|
52
|
+
/** PR file patches keyed by filename — Jev state context. */
|
|
53
|
+
patchByFile?: Map<string, string>;
|
|
54
|
+
threshold: number;
|
|
55
|
+
/**
|
|
56
|
+
* User-configured blocking severities — a severity listed here drives
|
|
57
|
+
* the verdict, so it must never be suppressed even if it is nit/q
|
|
58
|
+
* (otherwise Jev suppression could flip the commit-status gate).
|
|
59
|
+
*/
|
|
60
|
+
blockSeverities?: string[];
|
|
61
|
+
client: DecisionClient;
|
|
62
|
+
model?: string;
|
|
63
|
+
}): Promise<FindingAdjudicationResult<T>>;
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { debug } from '../debug.js';
|
|
2
|
+
import { describeDecisionError, isNoulAnswer, MAX_CANDIDATES, } from '../vision/decisions.js';
|
|
3
|
+
/** Suppression is scoped to severities that never drive the verdict. */
|
|
4
|
+
const SUPPRESSIBLE = new Set(['nit', 'q']);
|
|
5
|
+
const MAX_PATCH_EXCERPT = 4000;
|
|
6
|
+
/** Aggregate patch-state bound — 50 unique files x 4KB is still ~200KB. */
|
|
7
|
+
const MAX_PATCH_STATE_CHARS = 24_000;
|
|
8
|
+
/** Audit-record message bound — full text already lives in findings. */
|
|
9
|
+
const MAX_RECORD_MESSAGE = 300;
|
|
10
|
+
export async function adjudicateFindings(opts) {
|
|
11
|
+
const capped = opts.findings.slice(0, MAX_CANDIDATES);
|
|
12
|
+
const overflow = opts.findings.length - capped.length;
|
|
13
|
+
const pByIdx = new Array(capped.length);
|
|
14
|
+
let adjudicationFailed = false;
|
|
15
|
+
if (capped.length > 0) {
|
|
16
|
+
try {
|
|
17
|
+
const questions = {};
|
|
18
|
+
capped.forEach((_f, i) => {
|
|
19
|
+
questions[`f_${i}`] = {
|
|
20
|
+
type: 'noul',
|
|
21
|
+
instructions: `state.findings[${i}]: is this code-review finding a real problem the ` +
|
|
22
|
+
'PR author should act on? Its file patch is under state.patches. ' +
|
|
23
|
+
'Answer no for speculative style nits, issues already handled by ' +
|
|
24
|
+
'guards visible in the patch, and findings that merely restate ' +
|
|
25
|
+
'what the code does.',
|
|
26
|
+
};
|
|
27
|
+
});
|
|
28
|
+
// Each finding references its patch by filename — sending patches
|
|
29
|
+
// once keyed by file avoids repeating a 4KB excerpt per finding
|
|
30
|
+
// on the same file; an aggregate bound caps the whole state.
|
|
31
|
+
const patches = {};
|
|
32
|
+
let patchBudget = MAX_PATCH_STATE_CHARS;
|
|
33
|
+
for (const f of capped) {
|
|
34
|
+
if (patchBudget <= 0)
|
|
35
|
+
break;
|
|
36
|
+
if (patches[f.file] !== undefined)
|
|
37
|
+
continue;
|
|
38
|
+
const p = opts.patchByFile?.get(f.file);
|
|
39
|
+
if (p !== undefined) {
|
|
40
|
+
const excerpt = p.slice(0, Math.min(MAX_PATCH_EXCERPT, patchBudget));
|
|
41
|
+
patches[f.file] = excerpt;
|
|
42
|
+
patchBudget -= excerpt.length;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
const state = {
|
|
46
|
+
findings: capped.map((f) => ({
|
|
47
|
+
file: f.file,
|
|
48
|
+
line: f.line,
|
|
49
|
+
severity: f.severity,
|
|
50
|
+
message: f.message,
|
|
51
|
+
})),
|
|
52
|
+
patches,
|
|
53
|
+
};
|
|
54
|
+
const { answers } = await opts.client.decide({
|
|
55
|
+
...(opts.model !== undefined ? { model: opts.model } : {}),
|
|
56
|
+
state,
|
|
57
|
+
questions,
|
|
58
|
+
});
|
|
59
|
+
capped.forEach((_f, i) => {
|
|
60
|
+
const a = answers[`f_${i}`];
|
|
61
|
+
pByIdx[i] = a !== undefined && isNoulAnswer(a) ? a.noul : undefined;
|
|
62
|
+
});
|
|
63
|
+
}
|
|
64
|
+
catch (e) {
|
|
65
|
+
adjudicationFailed = true;
|
|
66
|
+
debug('adjudicate', `decision call failed — no suppression: ${describeDecisionError(e)}`);
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
const blocking = new Set(opts.blockSeverities ?? []);
|
|
70
|
+
// Only Jev may attach p — a model-emitted p on an unadjudicated
|
|
71
|
+
// finding is spoofed confidence, so strip it.
|
|
72
|
+
const stripP = (f) => {
|
|
73
|
+
const out = { ...f };
|
|
74
|
+
delete out.p;
|
|
75
|
+
return out;
|
|
76
|
+
};
|
|
77
|
+
const findings = [];
|
|
78
|
+
const records = [];
|
|
79
|
+
capped.forEach((f, i) => {
|
|
80
|
+
const p = pByIdx[i];
|
|
81
|
+
const adjudicated = p !== undefined && !adjudicationFailed;
|
|
82
|
+
const suppress = adjudicated &&
|
|
83
|
+
SUPPRESSIBLE.has(f.severity) &&
|
|
84
|
+
!blocking.has(f.severity) &&
|
|
85
|
+
p !== undefined &&
|
|
86
|
+
p < 1 - opts.threshold;
|
|
87
|
+
records.push({
|
|
88
|
+
file: f.file,
|
|
89
|
+
...(f.line !== undefined ? { line: f.line } : {}),
|
|
90
|
+
severity: f.severity,
|
|
91
|
+
...(f.category !== undefined ? { category: f.category } : {}),
|
|
92
|
+
message: f.message.slice(0, MAX_RECORD_MESSAGE),
|
|
93
|
+
adjudicated,
|
|
94
|
+
...(p !== undefined ? { p } : {}),
|
|
95
|
+
...(suppress ? { suppressed: true } : {}),
|
|
96
|
+
});
|
|
97
|
+
if (!suppress) {
|
|
98
|
+
findings.push(p !== undefined ? { ...stripP(f), p } : stripP(f));
|
|
99
|
+
}
|
|
100
|
+
});
|
|
101
|
+
// Overflow findings keep their place — never adjudicated, never
|
|
102
|
+
// suppressed; count-only in the audit record like the secrets lane.
|
|
103
|
+
for (const f of opts.findings.slice(MAX_CANDIDATES))
|
|
104
|
+
findings.push(stripP(f));
|
|
105
|
+
return {
|
|
106
|
+
findings,
|
|
107
|
+
records,
|
|
108
|
+
overflow,
|
|
109
|
+
...(adjudicationFailed ? { unadjudicated: true } : {}),
|
|
110
|
+
};
|
|
111
|
+
}
|