argus-reviewer-e2e 0.1.3 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -49
- package/action/action.yml +155 -61
- package/action/approval-review.mjs +393 -0
- package/action/bootstrap.mjs +102 -0
- package/action/cli.mjs +36 -0
- package/action/emit-review.mjs +336 -0
- package/action/runtime.mjs +154 -0
- package/action/sticky-comment.cjs +870 -0
- package/dist/api.d.ts +2 -0
- package/dist/api.js +4 -0
- package/dist/cache/store.d.ts +2 -0
- package/dist/cache/store.js +10 -3
- package/dist/cli.d.ts +137 -0
- package/dist/cli.js +677 -60
- package/dist/config.d.ts +67 -2
- package/dist/config.js +138 -7
- package/dist/debug.d.ts +1 -0
- package/dist/debug.js +9 -3
- package/dist/engine/loop.d.ts +12 -0
- package/dist/engine/loop.js +55 -6
- package/dist/evidence/ci.d.ts +3 -0
- package/dist/evidence/ci.js +3 -1
- package/dist/pipeline/budget.d.ts +11 -0
- package/dist/pipeline/budget.js +48 -0
- package/dist/pipeline/contracts.d.ts +12 -0
- package/dist/pipeline/contracts.js +16 -0
- package/dist/pipeline/verify.d.ts +27 -0
- package/dist/pipeline/verify.js +197 -0
- package/dist/probe/queue.d.ts +4 -1
- package/dist/probe/queue.js +46 -8
- package/dist/report/manifest.d.ts +87 -0
- package/dist/report/manifest.js +143 -0
- package/dist/report/run.d.ts +9 -0
- package/dist/report/run.js +6 -0
- package/dist/review/adjudicate.d.ts +63 -0
- package/dist/review/adjudicate.js +111 -0
- package/dist/review/secrets.d.ts +90 -0
- package/dist/review/secrets.js +224 -0
- package/dist/review/triage.d.ts +76 -0
- package/dist/review/triage.js +163 -0
- package/dist/trust.d.ts +50 -0
- package/dist/trust.js +103 -0
- package/dist/vision/cost.d.ts +14 -1
- package/dist/vision/cost.js +13 -0
- package/dist/vision/decisions.d.ts +95 -0
- package/dist/vision/decisions.js +232 -0
- package/package.json +9 -8
- package/action/sticky-comment.mjs +0 -404
package/dist/config.d.ts
CHANGED
|
@@ -1,3 +1,4 @@
|
|
|
1
|
+
import type { Trust } from './trust.js';
|
|
1
2
|
export interface ProviderRules {
|
|
2
3
|
only?: string[];
|
|
3
4
|
ignore?: string[];
|
|
@@ -59,6 +60,13 @@ export interface Config {
|
|
|
59
60
|
* PR diffs and post findings. Defaults to the primary `model` if not set.
|
|
60
61
|
*/
|
|
61
62
|
code_model: string | undefined;
|
|
63
|
+
/**
|
|
64
|
+
* OpenRouter Decisions API model for typed adjudication (Jev). Defaults
|
|
65
|
+
* to the pinned `typesafe/jev-1.13-20260917` — alias slugs like
|
|
66
|
+
* `~typesafe/jev-latest` drift silently and thresholds are calibrated
|
|
67
|
+
* to a version. Set to `''` to disable adjudication (regex-only mode).
|
|
68
|
+
*/
|
|
69
|
+
decisionModel: string | undefined;
|
|
62
70
|
/**
|
|
63
71
|
* Hard budget for the `argus-reviewer code-review` lane. When set, the
|
|
64
72
|
* review stops early if the cumulative OpenRouter cost exceeds this cap.
|
|
@@ -152,16 +160,73 @@ export interface Config {
|
|
|
152
160
|
* `resolveConfig` — `enabled: false` by default so the lane is opt-in.
|
|
153
161
|
*/
|
|
154
162
|
sandbox: Sandbox;
|
|
163
|
+
/**
|
|
164
|
+
* Code-review policy knobs. Always populated after `resolveConfig`.
|
|
165
|
+
* `secretsThreshold`: Jev `noul` probability at/above which a
|
|
166
|
+
* secret-shaped diff literal is reported as a finding (below →
|
|
167
|
+
* suppressed but audit-recorded). Default 0.3 — tune after dogfooding.
|
|
168
|
+
* `maxComments`: cap on inline review comments posted per run
|
|
169
|
+
* (default 20) — overflow is summarized count-only in the sticky.
|
|
170
|
+
* `severityGate`: consumer-facing alias over `severity` — 'bug'
|
|
171
|
+
* fails on bugs only, 'risk' fails on bug|risk. Unset → `severity`
|
|
172
|
+
* list is authoritative.
|
|
173
|
+
* `triage`: Jev pre-review lane — 'off' no call, 'annotate' (default)
|
|
174
|
+
* records risk/deep-review/area into the report + sticky, 'route'
|
|
175
|
+
* additionally swaps the code model to `lowRiskModel` on low-risk
|
|
176
|
+
* diffs. Jev routes/annotates, never gates — coverage is constant.
|
|
177
|
+
* `lowRiskModel`: the cheap code-model slug 'route' falls to; unset →
|
|
178
|
+
* route keeps `code_model` (annotate-equivalent).
|
|
179
|
+
* `findingThreshold`: P(false-positive) required to suppress a nit/q
|
|
180
|
+
* finding after Jev adjudication — 1.0 (default) is annotate-only,
|
|
181
|
+
* lowering it suppresses progressively more low-confidence nits.
|
|
182
|
+
* bug/risk are never suppressed.
|
|
183
|
+
* `requestChanges`: allow the review event to escalate to
|
|
184
|
+
* REQUEST_CHANGES for proven blockers (probe-reproduced or Jev
|
|
185
|
+
* high-confidence). Default true — set false for advisory-only posting.
|
|
186
|
+
*/
|
|
187
|
+
review: {
|
|
188
|
+
secretsThreshold: number;
|
|
189
|
+
maxComments: number;
|
|
190
|
+
severityGate: 'bug' | 'risk' | undefined;
|
|
191
|
+
triage: 'off' | 'annotate' | 'route';
|
|
192
|
+
lowRiskModel: string | undefined;
|
|
193
|
+
findingThreshold: number;
|
|
194
|
+
requestChanges: boolean;
|
|
195
|
+
};
|
|
155
196
|
}
|
|
156
|
-
export type ConfigInput = Partial<Omit<Config, 'provider' | 'sandbox'>> & {
|
|
197
|
+
export type ConfigInput = Partial<Omit<Config, 'provider' | 'sandbox' | 'review'>> & {
|
|
157
198
|
provider?: Partial<ProviderRules>;
|
|
158
199
|
sandbox?: Partial<Sandbox>;
|
|
200
|
+
review?: Partial<Config['review']>;
|
|
159
201
|
};
|
|
160
202
|
export declare const DEFAULT_RECORD_STEP_CAP = 40;
|
|
161
203
|
export declare const DEFAULT_SANDBOX: Sandbox;
|
|
162
204
|
export declare function defineConfig(input: ConfigInput): ConfigInput;
|
|
205
|
+
/**
|
|
206
|
+
* Which severities fail the review status. `review.severityGate` is the
|
|
207
|
+
* consumer-facing alias over `severity` — 'risk' fails on bug|risk,
|
|
208
|
+
* 'bug' on bugs only; unset → the `severity` list is authoritative.
|
|
209
|
+
*/
|
|
210
|
+
export declare function resolveBlockSeverities(config: Config): string[];
|
|
211
|
+
/**
|
|
212
|
+
* Inline-comment cap: `ARGUS_MAX_COMMENTS` (the action's `max-comments`
|
|
213
|
+
* input) wins when it parses as a non-negative integer — it's set by the
|
|
214
|
+
* workflow author, so an untrusted PR config can't reach it (`review`
|
|
215
|
+
* isn't on the untrusted allowlist). Anything else → `review.maxComments`.
|
|
216
|
+
*/
|
|
217
|
+
export declare function resolveMaxComments(env: Record<string, string | undefined>, config: Config): number;
|
|
163
218
|
export declare function resolveConfig(input?: ConfigInput): Config;
|
|
164
|
-
export
|
|
219
|
+
export interface LoadConfigOpts {
|
|
220
|
+
/**
|
|
221
|
+
* Required — there is no default. Every call site must state the
|
|
222
|
+
* checkout's trust so a missed or future caller can't silently execute
|
|
223
|
+
* config code on a hostile tree (see src/trust.ts).
|
|
224
|
+
*/
|
|
225
|
+
trust: Trust;
|
|
226
|
+
/** Human-readable note on security-relevant load decisions (e.g. ctx.err). */
|
|
227
|
+
note?: (line: string) => void;
|
|
228
|
+
}
|
|
229
|
+
export declare function loadConfig(cwd: string, opts: LoadConfigOpts): Promise<Config>;
|
|
165
230
|
/**
|
|
166
231
|
* Provider slugs the harness recognizes for `provider.only/ignore/order`
|
|
167
232
|
* (KTD4). Unknown slugs warn but do not fail — OpenRouter's catalog changes
|
package/dist/config.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { pathToFileURL } from 'node:url';
|
|
2
|
+
import { JEV_DEFAULT_MODEL } from './vision/decisions.js';
|
|
2
3
|
export const DEFAULT_RECORD_STEP_CAP = 40;
|
|
3
4
|
export const DEFAULT_SANDBOX = {
|
|
4
5
|
enabled: false,
|
|
@@ -15,6 +16,7 @@ const defaults = {
|
|
|
15
16
|
escalation_model: 'moonshotai/kimi-k2.5',
|
|
16
17
|
grounding_model: undefined,
|
|
17
18
|
code_model: 'deepseek/deepseek-v4.1-flash',
|
|
19
|
+
decisionModel: JEV_DEFAULT_MODEL,
|
|
18
20
|
codeReviewBudgetUsd: undefined,
|
|
19
21
|
provider: {
|
|
20
22
|
ignore: ['siliconflow', 'novitaai', 'atlascloud', 'streamlake', 'chutes'],
|
|
@@ -38,6 +40,15 @@ const defaults = {
|
|
|
38
40
|
a0: undefined,
|
|
39
41
|
heal: 'local',
|
|
40
42
|
sandbox: { ...DEFAULT_SANDBOX },
|
|
43
|
+
review: {
|
|
44
|
+
secretsThreshold: 0.3,
|
|
45
|
+
maxComments: 20,
|
|
46
|
+
severityGate: undefined,
|
|
47
|
+
triage: 'annotate',
|
|
48
|
+
lowRiskModel: undefined,
|
|
49
|
+
findingThreshold: 1.0,
|
|
50
|
+
requestChanges: true,
|
|
51
|
+
},
|
|
41
52
|
};
|
|
42
53
|
export function defineConfig(input) {
|
|
43
54
|
return input;
|
|
@@ -46,6 +57,36 @@ export function defineConfig(input) {
|
|
|
46
57
|
function posInt(v, dflt) {
|
|
47
58
|
return v !== undefined && Number.isFinite(v) && v >= 1 ? Math.floor(v) : dflt;
|
|
48
59
|
}
|
|
60
|
+
/** Probability config values (must be in [0,1]) fall back to their default. */
|
|
61
|
+
function prob01(v, dflt) {
|
|
62
|
+
return v !== undefined && Number.isFinite(v) && v >= 0 && v <= 1 ? v : dflt;
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* Which severities fail the review status. `review.severityGate` is the
|
|
66
|
+
* consumer-facing alias over `severity` — 'risk' fails on bug|risk,
|
|
67
|
+
* 'bug' on bugs only; unset → the `severity` list is authoritative.
|
|
68
|
+
*/
|
|
69
|
+
export function resolveBlockSeverities(config) {
|
|
70
|
+
if (config.review.severityGate === 'risk')
|
|
71
|
+
return ['bug', 'risk'];
|
|
72
|
+
if (config.review.severityGate === 'bug')
|
|
73
|
+
return ['bug'];
|
|
74
|
+
return config.severity ?? ['bug'];
|
|
75
|
+
}
|
|
76
|
+
/**
|
|
77
|
+
* Inline-comment cap: `ARGUS_MAX_COMMENTS` (the action's `max-comments`
|
|
78
|
+
* input) wins when it parses as a non-negative integer — it's set by the
|
|
79
|
+
* workflow author, so an untrusted PR config can't reach it (`review`
|
|
80
|
+
* isn't on the untrusted allowlist). Anything else → `review.maxComments`.
|
|
81
|
+
*/
|
|
82
|
+
export function resolveMaxComments(env, config) {
|
|
83
|
+
const raw = env.ARGUS_MAX_COMMENTS?.trim();
|
|
84
|
+
// ^\d+$ — Number() would also accept '0x10', '1e2', ' 4 ', 'Infinity'.
|
|
85
|
+
if (raw !== undefined && /^\d+$/.test(raw)) {
|
|
86
|
+
return Number(raw);
|
|
87
|
+
}
|
|
88
|
+
return config.review.maxComments;
|
|
89
|
+
}
|
|
49
90
|
export function resolveConfig(input = {}) {
|
|
50
91
|
const provider = { ...defaults.provider, ...(input.provider ?? {}) };
|
|
51
92
|
// Wrong-typed sandbox values (e.g. `sandbox: true`, `enabled: 'yes'`,
|
|
@@ -56,23 +97,108 @@ export function resolveConfig(input = {}) {
|
|
|
56
97
|
sandbox.enabled = raw.enabled === true;
|
|
57
98
|
sandbox.allowForks = raw.allowForks === true;
|
|
58
99
|
sandbox.image = typeof raw.image === 'string' && raw.image !== '' ? raw.image : undefined;
|
|
59
|
-
sandbox.memory =
|
|
100
|
+
sandbox.memory =
|
|
101
|
+
typeof raw.memory === 'string' && raw.memory !== '' ? raw.memory : DEFAULT_SANDBOX.memory;
|
|
60
102
|
sandbox.cpus = typeof raw.cpus === 'string' && raw.cpus !== '' ? raw.cpus : DEFAULT_SANDBOX.cpus;
|
|
61
103
|
sandbox.maxProbes = posInt(sandbox.maxProbes, DEFAULT_SANDBOX.maxProbes);
|
|
62
104
|
sandbox.timeoutMs = posInt(sandbox.timeoutMs, DEFAULT_SANDBOX.timeoutMs);
|
|
63
105
|
sandbox.pidsLimit = posInt(sandbox.pidsLimit, DEFAULT_SANDBOX.pidsLimit);
|
|
64
|
-
const
|
|
106
|
+
const rawReview = typeof input.review === 'object' && input.review !== null ? input.review : {};
|
|
107
|
+
const review = { ...defaults.review, ...rawReview };
|
|
108
|
+
// Thresholds must be probabilities — anything else (NaN, >1,
|
|
109
|
+
// negative) would silently suppress or flood the Jev lanes.
|
|
110
|
+
review.secretsThreshold = prob01(review.secretsThreshold, defaults.review.secretsThreshold);
|
|
111
|
+
review.maxComments =
|
|
112
|
+
typeof review.maxComments === 'number' &&
|
|
113
|
+
Number.isInteger(review.maxComments) &&
|
|
114
|
+
review.maxComments >= 0
|
|
115
|
+
? review.maxComments
|
|
116
|
+
: defaults.review.maxComments;
|
|
117
|
+
if (review.severityGate !== 'bug' && review.severityGate !== 'risk') {
|
|
118
|
+
review.severityGate = undefined;
|
|
119
|
+
}
|
|
120
|
+
if (review.triage !== 'off' && review.triage !== 'annotate' && review.triage !== 'route') {
|
|
121
|
+
review.triage = defaults.review.triage;
|
|
122
|
+
}
|
|
123
|
+
if (typeof review.lowRiskModel !== 'string' || review.lowRiskModel === '') {
|
|
124
|
+
review.lowRiskModel = undefined;
|
|
125
|
+
}
|
|
126
|
+
review.findingThreshold = prob01(review.findingThreshold, defaults.review.findingThreshold);
|
|
127
|
+
// Advisory-only escape hatch — only literal `false` opts out; anything
|
|
128
|
+
// else (mis-typed values included) keeps the default-true posture.
|
|
129
|
+
review.requestChanges = review.requestChanges !== false;
|
|
130
|
+
const resolved = { ...defaults, ...input, provider, sandbox, review };
|
|
65
131
|
resolved.recordStepCap = posInt(resolved.recordStepCap, DEFAULT_RECORD_STEP_CAP);
|
|
66
132
|
if (resolved.heal !== 'a0')
|
|
67
133
|
resolved.heal = 'local';
|
|
134
|
+
// '' is the documented opt-out — an empty slug would send a broken
|
|
135
|
+
// model id to the decisions endpoint on every adjudication call.
|
|
136
|
+
if (resolved.decisionModel === '')
|
|
137
|
+
resolved.decisionModel = undefined;
|
|
68
138
|
return resolved;
|
|
69
139
|
}
|
|
70
|
-
|
|
140
|
+
/**
|
|
141
|
+
* Config keys honored on untrusted checkouts — policy-free fields only.
|
|
142
|
+
* Everything else (exec-bearing fields, model/budget/provider selection,
|
|
143
|
+
* severity/verdict policy, credentials maps, network endpoints, write
|
|
144
|
+
* locations) is ignored: the review policy over hostile code must not be
|
|
145
|
+
* authored by that code.
|
|
146
|
+
*/
|
|
147
|
+
const UNTRUSTED_CONFIG_KEYS = new Set(['logLevel', 'sourceGlobs']);
|
|
148
|
+
function filterUntrustedConfig(input) {
|
|
149
|
+
const out = {};
|
|
150
|
+
for (const [key, value] of Object.entries(input)) {
|
|
151
|
+
if (UNTRUSTED_CONFIG_KEYS.has(key))
|
|
152
|
+
out[key] = value;
|
|
153
|
+
}
|
|
154
|
+
return out;
|
|
155
|
+
}
|
|
156
|
+
export async function loadConfig(cwd, opts) {
|
|
71
157
|
const fs = await import('node:fs/promises');
|
|
72
158
|
const path = await import('node:path');
|
|
159
|
+
const untrusted = opts.trust === 'untrusted';
|
|
160
|
+
// cacheDir is the documented CLI default '.argus-reviewer-cache' — normalize
|
|
161
|
+
// it to an absolute path here so engine record/replay persistence (gated on
|
|
162
|
+
// config.cacheDir) writes where every other consumer already falls back to.
|
|
163
|
+
const finish = async (input = {}) => {
|
|
164
|
+
const config = resolveConfig(input);
|
|
165
|
+
const cacheDir = path.resolve(cwd, config.cacheDir ?? '.argus-reviewer-cache');
|
|
166
|
+
if (untrusted) {
|
|
167
|
+
// A hostile PR can commit the default path as a symlink and redirect
|
|
168
|
+
// cache writes outside the checkout — fail closed before writers run.
|
|
169
|
+
let isSymlink = false;
|
|
170
|
+
try {
|
|
171
|
+
isSymlink = (await fs.lstat(cacheDir)).isSymbolicLink();
|
|
172
|
+
}
|
|
173
|
+
catch (e) {
|
|
174
|
+
if (e.code !== 'ENOENT')
|
|
175
|
+
throw e;
|
|
176
|
+
}
|
|
177
|
+
if (isSymlink) {
|
|
178
|
+
throw new Error(`untrusted cache directory must not be a symlink: ${cacheDir}`);
|
|
179
|
+
}
|
|
180
|
+
}
|
|
181
|
+
config.cacheDir = cacheDir;
|
|
182
|
+
return config;
|
|
183
|
+
};
|
|
73
184
|
const names = ['argus-reviewer.config', 'vision-e2e.config'];
|
|
74
185
|
for (const name of names) {
|
|
75
|
-
|
|
186
|
+
if (untrusted) {
|
|
187
|
+
// Surface skipped .ts candidates — otherwise a hostile config (or a
|
|
188
|
+
// legit consumer debugging "why is my config ignored") is invisible.
|
|
189
|
+
try {
|
|
190
|
+
if ((await fs.stat(path.join(cwd, `${name}.ts`))).isFile()) {
|
|
191
|
+
opts.note?.(`config: ${name}.ts ignored — untrusted checkouts load JSON config only`);
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
catch {
|
|
195
|
+
// no .ts candidate — nothing to note
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
// .ts is tried before .json, so an untrusted checkout must skip the .ts
|
|
199
|
+
// candidate *before* it can shadow a committed .json — importing it
|
|
200
|
+
// executes arbitrary code beside the runner's secrets (#58).
|
|
201
|
+
for (const ext of untrusted ? ['.json'] : ['.ts', '.json']) {
|
|
76
202
|
const file = path.join(cwd, `${name}${ext}`);
|
|
77
203
|
try {
|
|
78
204
|
const stat = await fs.stat(file);
|
|
@@ -80,7 +206,12 @@ export async function loadConfig(cwd) {
|
|
|
80
206
|
continue;
|
|
81
207
|
if (ext === '.json') {
|
|
82
208
|
const raw = await fs.readFile(file, 'utf8');
|
|
83
|
-
|
|
209
|
+
const parsed = JSON.parse(raw);
|
|
210
|
+
if (untrusted) {
|
|
211
|
+
opts.note?.(`config: ${name}.json loaded untrusted — honoring ${[...UNTRUSTED_CONFIG_KEYS].join(', ')} only`);
|
|
212
|
+
return finish(filterUntrustedConfig(parsed));
|
|
213
|
+
}
|
|
214
|
+
return finish(parsed);
|
|
84
215
|
}
|
|
85
216
|
// Always transpile .ts to a temp .mjs rather than importing natively:
|
|
86
217
|
// Node's built-in type stripping resolves the module type from the
|
|
@@ -111,7 +242,7 @@ export async function loadConfig(cwd) {
|
|
|
111
242
|
await rm(out, { force: true }).catch(() => undefined);
|
|
112
243
|
}
|
|
113
244
|
const exported = mod.default ?? mod;
|
|
114
|
-
return
|
|
245
|
+
return finish(exported);
|
|
115
246
|
}
|
|
116
247
|
catch (e) {
|
|
117
248
|
const code = e.code;
|
|
@@ -121,7 +252,7 @@ export async function loadConfig(cwd) {
|
|
|
121
252
|
}
|
|
122
253
|
}
|
|
123
254
|
}
|
|
124
|
-
return
|
|
255
|
+
return finish();
|
|
125
256
|
}
|
|
126
257
|
/**
|
|
127
258
|
* Provider slugs the harness recognizes for `provider.only/ignore/order`
|
package/dist/debug.d.ts
CHANGED
package/dist/debug.js
CHANGED
|
@@ -1,8 +1,14 @@
|
|
|
1
1
|
import { join } from 'node:path';
|
|
2
2
|
import { liveLog } from './live.js';
|
|
3
3
|
const DEBUG = process.env.ARGUS_DEBUG === '1' || process.env.ARGUS_DEBUG === 'true';
|
|
4
|
-
// debug() has no config access — it
|
|
5
|
-
|
|
4
|
+
// debug() has no config access — it defaults to the conventional cache dir,
|
|
5
|
+
// and callers that resolve a config set the real live dir once it's known
|
|
6
|
+
// (otherwise debug lines and stage lines would split across two dirs).
|
|
7
|
+
const DEFAULT_LIVE_DIR = join(process.cwd(), '.argus-reviewer-cache');
|
|
8
|
+
let liveDir = DEFAULT_LIVE_DIR;
|
|
9
|
+
export function setLiveDir(dir) {
|
|
10
|
+
liveDir = dir ?? DEFAULT_LIVE_DIR;
|
|
11
|
+
}
|
|
6
12
|
function toMsg(arg) {
|
|
7
13
|
if (typeof arg === 'string')
|
|
8
14
|
return arg;
|
|
@@ -18,7 +24,7 @@ export function debug(kind, ...args) {
|
|
|
18
24
|
if (!DEBUG)
|
|
19
25
|
return; // keep debug() free of fs work on the hot path
|
|
20
26
|
for (const arg of args) {
|
|
21
|
-
liveLog(
|
|
27
|
+
liveLog(liveDir, kind, 'debug', toMsg(arg));
|
|
22
28
|
const prefix = `[argus-reviewer:${kind}]`;
|
|
23
29
|
if (typeof arg === 'string') {
|
|
24
30
|
console.error(`${prefix} ${arg}`);
|
package/dist/engine/loop.d.ts
CHANGED
|
@@ -62,6 +62,15 @@ export interface RunResult {
|
|
|
62
62
|
reason?: string;
|
|
63
63
|
steps: StepResult[];
|
|
64
64
|
visionCalls: number;
|
|
65
|
+
cache: CacheStats;
|
|
66
|
+
}
|
|
67
|
+
export interface CacheStats {
|
|
68
|
+
hits: number;
|
|
69
|
+
misses: number;
|
|
70
|
+
heals: number;
|
|
71
|
+
staleEntries: number;
|
|
72
|
+
assertionHits: number;
|
|
73
|
+
assertionMisses: number;
|
|
65
74
|
}
|
|
66
75
|
export interface AssertResult extends AssertionResult {
|
|
67
76
|
cached: boolean;
|
|
@@ -81,6 +90,7 @@ export declare class Engine {
|
|
|
81
90
|
private _fingerprints;
|
|
82
91
|
private _assertCache;
|
|
83
92
|
private _errors;
|
|
93
|
+
private _cacheStats;
|
|
84
94
|
/** Structured, non-fatal anomalies — journaled as evidence, never thrown. */
|
|
85
95
|
get errorRecords(): ErrorRecord[];
|
|
86
96
|
private _note;
|
|
@@ -88,6 +98,7 @@ export declare class Engine {
|
|
|
88
98
|
/** Assertion verdicts collected/known this run — persist into the flow cache. */
|
|
89
99
|
get assertEntries(): CachedAssert[];
|
|
90
100
|
get visionCalls(): number;
|
|
101
|
+
get cacheStats(): CacheStats;
|
|
91
102
|
record(instruction: string, tdApi?: TestDriverApi, options?: RecordOptions): Promise<RunResult>;
|
|
92
103
|
replay(flow: FlowCache, options?: ReplayOptions): Promise<RunResult>;
|
|
93
104
|
/**
|
|
@@ -114,5 +125,6 @@ export declare class Engine {
|
|
|
114
125
|
private _buildFingerprint;
|
|
115
126
|
private _regionScreenshot;
|
|
116
127
|
private _result;
|
|
128
|
+
private _resetCacheStats;
|
|
117
129
|
}
|
|
118
130
|
export declare function instructionMatchesNode(instruction: string, nodeSnippet: string): boolean;
|
package/dist/engine/loop.js
CHANGED
|
@@ -9,6 +9,14 @@ export class Engine {
|
|
|
9
9
|
_fingerprints = [];
|
|
10
10
|
_assertCache = new Map();
|
|
11
11
|
_errors = [];
|
|
12
|
+
_cacheStats = {
|
|
13
|
+
hits: 0,
|
|
14
|
+
misses: 0,
|
|
15
|
+
heals: 0,
|
|
16
|
+
staleEntries: 0,
|
|
17
|
+
assertionHits: 0,
|
|
18
|
+
assertionMisses: 0,
|
|
19
|
+
};
|
|
12
20
|
/** Structured, non-fatal anomalies — journaled as evidence, never thrown. */
|
|
13
21
|
get errorRecords() {
|
|
14
22
|
return this._errors;
|
|
@@ -31,10 +39,14 @@ export class Engine {
|
|
|
31
39
|
get visionCalls() {
|
|
32
40
|
return this._visionCalls;
|
|
33
41
|
}
|
|
42
|
+
get cacheStats() {
|
|
43
|
+
return { ...this._cacheStats };
|
|
44
|
+
}
|
|
34
45
|
async record(instruction, tdApi = this._opts.actions, options = {}) {
|
|
35
46
|
this._visionCalls = 0;
|
|
36
47
|
this._steps = [];
|
|
37
48
|
this._fingerprints = [];
|
|
49
|
+
this._resetCacheStats();
|
|
38
50
|
const cap = options.stepCap ?? this._opts.config.recordStepCap ?? DEFAULT_RECORD_STEP_CAP;
|
|
39
51
|
let observation = await this._opts.driver.observe({ grid: true });
|
|
40
52
|
for (let i = 0; i < cap; i++) {
|
|
@@ -82,16 +94,18 @@ export class Engine {
|
|
|
82
94
|
async replay(flow, options = {}) {
|
|
83
95
|
this._visionCalls = 0;
|
|
84
96
|
this._steps = [];
|
|
97
|
+
this._resetCacheStats();
|
|
85
98
|
for (let i = 0; i < flow.steps.length; i++) {
|
|
86
99
|
const step = flow.steps[i];
|
|
87
100
|
if (!step)
|
|
88
101
|
continue;
|
|
89
|
-
|
|
102
|
+
const observation = await this._opts.driver.observe({ grid: true });
|
|
90
103
|
// Diff-invalidated entries skip hash verification entirely and go
|
|
91
104
|
// straight to the heal path — the diff already told us they're stale.
|
|
92
105
|
let resolve;
|
|
93
106
|
if (step.stale !== undefined) {
|
|
94
107
|
this._note('heal', 'cache entry invalidated by diff', step.stale);
|
|
108
|
+
this._cacheStats.staleEntries++;
|
|
95
109
|
resolve = { matched: false, currentHash: '', regionMatched: false, a11yMatched: false };
|
|
96
110
|
}
|
|
97
111
|
else {
|
|
@@ -99,10 +113,12 @@ export class Engine {
|
|
|
99
113
|
resolve = new Fingerprint(step).resolve(regionBuffer, observation.a11yYaml);
|
|
100
114
|
}
|
|
101
115
|
if (resolve.matched) {
|
|
116
|
+
this._cacheStats.hits++;
|
|
102
117
|
await this._executeAction(this._opts.actions, step.action);
|
|
103
118
|
this._steps.push({ instruction: step.instruction, action: step.action.action, ok: true });
|
|
104
119
|
continue;
|
|
105
120
|
}
|
|
121
|
+
this._cacheStats.misses++;
|
|
106
122
|
if (this._opts.ledger.replayOnly || !this._opts.ledger.canSpend(0.001)) {
|
|
107
123
|
this._steps.push({
|
|
108
124
|
instruction: step.instruction,
|
|
@@ -144,9 +160,10 @@ export class Engine {
|
|
|
144
160
|
return this._result(false);
|
|
145
161
|
}
|
|
146
162
|
const resolved = await this._resolveAction(action);
|
|
147
|
-
|
|
163
|
+
await this._executeAction(this._opts.actions, action);
|
|
148
164
|
const newFingerprint = await this._buildFingerprint(step.instruction, action, resolved, response.model);
|
|
149
165
|
flow.steps[i] = newFingerprint;
|
|
166
|
+
this._cacheStats.heals++;
|
|
150
167
|
this._note('heal', 'fingerprint mismatch healed by model', step.instruction);
|
|
151
168
|
this._steps.push({
|
|
152
169
|
instruction: step.instruction,
|
|
@@ -155,7 +172,6 @@ export class Engine {
|
|
|
155
172
|
healed: true,
|
|
156
173
|
model: response.model,
|
|
157
174
|
});
|
|
158
|
-
observation = nextObservation;
|
|
159
175
|
}
|
|
160
176
|
if (options.flowName && this._opts.config.cacheDir) {
|
|
161
177
|
await saveFlow(this._opts.config.cacheDir, options.flowName, flow.steps);
|
|
@@ -177,6 +193,7 @@ export class Engine {
|
|
|
177
193
|
const regionBuffer = await this._regionScreenshot(cached.bbox);
|
|
178
194
|
const resolve = new Fingerprint(cached).resolve(regionBuffer, observation.a11yYaml);
|
|
179
195
|
if (resolve.matched) {
|
|
196
|
+
this._cacheStats.hits++;
|
|
180
197
|
return {
|
|
181
198
|
ok: true,
|
|
182
199
|
reason: undefined,
|
|
@@ -186,6 +203,7 @@ export class Engine {
|
|
|
186
203
|
model: undefined,
|
|
187
204
|
};
|
|
188
205
|
}
|
|
206
|
+
this._cacheStats.misses++;
|
|
189
207
|
if (this._opts.ledger.replayOnly || !this._opts.ledger.canSpend(0.001)) {
|
|
190
208
|
return {
|
|
191
209
|
ok: false,
|
|
@@ -197,6 +215,13 @@ export class Engine {
|
|
|
197
215
|
};
|
|
198
216
|
}
|
|
199
217
|
}
|
|
218
|
+
else if (cached?.stale !== undefined) {
|
|
219
|
+
this._cacheStats.misses++;
|
|
220
|
+
this._cacheStats.staleEntries++;
|
|
221
|
+
}
|
|
222
|
+
else {
|
|
223
|
+
this._cacheStats.misses++;
|
|
224
|
+
}
|
|
200
225
|
// A diff-invalidated (stale) entry is a fresh ground, not a heal — heal
|
|
201
226
|
// implies the fingerprint *checked out as wrong*, stale means we never
|
|
202
227
|
// verified it. Keeping the kind split honest keeps the heal-rate signal
|
|
@@ -205,8 +230,11 @@ export class Engine {
|
|
|
205
230
|
const isStale = cached !== undefined && cached.stale !== undefined;
|
|
206
231
|
const useHeal = cached !== undefined && !isStale;
|
|
207
232
|
const primary = await this._locateWithModel(instruction, observation, useHeal);
|
|
208
|
-
if (primary.ok)
|
|
233
|
+
if (primary.ok) {
|
|
234
|
+
if (useHeal)
|
|
235
|
+
this._cacheStats.heals++;
|
|
209
236
|
return primary;
|
|
237
|
+
}
|
|
210
238
|
// Semantic escalation fallback (issue #14): the model answered but could
|
|
211
239
|
// not ground — provider-level OpenRouter fallback only covers unavailable
|
|
212
240
|
// models, not bad answers. Retry once with escalation_model as primary on
|
|
@@ -221,7 +249,10 @@ export class Engine {
|
|
|
221
249
|
}
|
|
222
250
|
this._note('locate', 'escalating to fallback model', `failed=${failedModel} esc=${esc} reason=${(primary.reason ?? '').slice(0, 80)}`);
|
|
223
251
|
const fresh = await this._opts.driver.observe({ grid: true });
|
|
224
|
-
|
|
252
|
+
const escalated = await this._locateWithModel(instruction, fresh, useHeal, esc);
|
|
253
|
+
if (escalated.ok && useHeal)
|
|
254
|
+
this._cacheStats.heals++;
|
|
255
|
+
return escalated;
|
|
225
256
|
}
|
|
226
257
|
/**
|
|
227
258
|
* One locate attempt against a specific model: initial call plus the
|
|
@@ -388,8 +419,10 @@ export class Engine {
|
|
|
388
419
|
const key = `${question}${a11yHash}`;
|
|
389
420
|
const cached = this._assertCache.get(key);
|
|
390
421
|
if (cached) {
|
|
422
|
+
this._cacheStats.assertionHits++;
|
|
391
423
|
return { verdict: cached.verdict, reasoning: cached.reasoning, cached: true };
|
|
392
424
|
}
|
|
425
|
+
this._cacheStats.assertionMisses++;
|
|
393
426
|
if (this._opts.ledger.replayOnly || !this._opts.ledger.canSpend(0.001)) {
|
|
394
427
|
return { verdict: 'fail', reasoning: 'budget exceeded or replay-only', cached: false };
|
|
395
428
|
}
|
|
@@ -621,7 +654,23 @@ export class Engine {
|
|
|
621
654
|
return Buffer.from(raw);
|
|
622
655
|
}
|
|
623
656
|
_result(ok, reason) {
|
|
624
|
-
return {
|
|
657
|
+
return {
|
|
658
|
+
ok,
|
|
659
|
+
steps: this._steps,
|
|
660
|
+
visionCalls: this._visionCalls,
|
|
661
|
+
cache: this.cacheStats,
|
|
662
|
+
...(reason ? { reason } : {}),
|
|
663
|
+
};
|
|
664
|
+
}
|
|
665
|
+
_resetCacheStats() {
|
|
666
|
+
this._cacheStats = {
|
|
667
|
+
hits: 0,
|
|
668
|
+
misses: 0,
|
|
669
|
+
heals: 0,
|
|
670
|
+
staleEntries: 0,
|
|
671
|
+
assertionHits: 0,
|
|
672
|
+
assertionMisses: 0,
|
|
673
|
+
};
|
|
625
674
|
}
|
|
626
675
|
}
|
|
627
676
|
/**
|
package/dist/evidence/ci.d.ts
CHANGED
|
@@ -42,6 +42,9 @@ export interface PrMeta {
|
|
|
42
42
|
* timeline fetch failed (the gate fails closed either way).
|
|
43
43
|
*/
|
|
44
44
|
labelApprovedAt: string | undefined;
|
|
45
|
+
/** PR title/body — triage state only (untrusted text; feeds Jev, never gates). */
|
|
46
|
+
title: string | undefined;
|
|
47
|
+
body: string | undefined;
|
|
45
48
|
}
|
|
46
49
|
/**
|
|
47
50
|
* Shared GitHub GET scaffold — Bearer auth, API headers, 30s abort timeout,
|
package/dist/evidence/ci.js
CHANGED
|
@@ -6,7 +6,7 @@ export const PROBE_LABEL = 'argus-probe';
|
|
|
6
6
|
* FIRST_TIMER, MANNEQUIN, and NONE are not.
|
|
7
7
|
*/
|
|
8
8
|
export function isTrustedAssociation(association) {
|
|
9
|
-
return
|
|
9
|
+
return association === 'MEMBER' || association === 'OWNER' || association === 'COLLABORATOR';
|
|
10
10
|
}
|
|
11
11
|
const GH_API = 'https://api.github.com';
|
|
12
12
|
const MAX_CHECK_RUN_PAGES = 5;
|
|
@@ -85,6 +85,8 @@ export async function fetchPrMeta(repo, pr, token, ctx) {
|
|
|
85
85
|
labels,
|
|
86
86
|
pushedAt: data.head?.repo?.pushed_at,
|
|
87
87
|
labelApprovedAt,
|
|
88
|
+
title: typeof data.title === 'string' ? data.title : undefined,
|
|
89
|
+
body: typeof data.body === 'string' ? data.body : undefined,
|
|
88
90
|
};
|
|
89
91
|
}
|
|
90
92
|
/**
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { CallCost } from '../vision/cost.js';
|
|
2
|
+
import type { BudgetSummary, LaneId } from '../report/manifest.js';
|
|
3
|
+
export interface BudgetOptions {
|
|
4
|
+
limitUsd?: number;
|
|
5
|
+
maxDurationMs?: number;
|
|
6
|
+
maxTasks?: number;
|
|
7
|
+
}
|
|
8
|
+
export declare function createBudget(lane: LaneId, options?: BudgetOptions): BudgetSummary;
|
|
9
|
+
export declare function addProviderCalls(budget: BudgetSummary, calls: CallCost[] | undefined): BudgetSummary;
|
|
10
|
+
export declare function addA0Task(budget: BudgetSummary, elapsedMs: number, metered: boolean): BudgetSummary;
|
|
11
|
+
export declare function budgetCanSpend(budget: BudgetSummary, nextCostUsd: number): boolean;
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
export function createBudget(lane, options = {}) {
|
|
2
|
+
void lane;
|
|
3
|
+
return {
|
|
4
|
+
limitUsd: options.limitUsd,
|
|
5
|
+
spentUsd: 0,
|
|
6
|
+
exceeded: false,
|
|
7
|
+
maxDurationMs: options.maxDurationMs,
|
|
8
|
+
elapsedMs: 0,
|
|
9
|
+
maxTasks: options.maxTasks,
|
|
10
|
+
tasks: 0,
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
/**
|
|
14
|
+
* USD spend is summed as IEEE-754 doubles, so a lane that lands exactly on its
|
|
15
|
+
* cap (0.1 + 0.2 === 0.30000000000000004, not 0.3) would otherwise read as over
|
|
16
|
+
* budget and abort a run that never exceeded the cap. Compare with a tolerance
|
|
17
|
+
* far below the 1e-6 precision reports render, and far above double noise here.
|
|
18
|
+
*/
|
|
19
|
+
const USD_EPSILON = 1e-9;
|
|
20
|
+
function overLimit(spentUsd, limitUsd) {
|
|
21
|
+
return spentUsd > limitUsd + USD_EPSILON;
|
|
22
|
+
}
|
|
23
|
+
export function addProviderCalls(budget, calls) {
|
|
24
|
+
if (calls === undefined || calls.length === 0)
|
|
25
|
+
return budget;
|
|
26
|
+
const spentUsd = calls.reduce((sum, call) => sum + call.costUsd, 0);
|
|
27
|
+
const exceeded = budget.exceeded ||
|
|
28
|
+
(budget.limitUsd !== undefined && overLimit(budget.spentUsd + spentUsd, budget.limitUsd));
|
|
29
|
+
return {
|
|
30
|
+
...budget,
|
|
31
|
+
spentUsd: budget.spentUsd + spentUsd,
|
|
32
|
+
exceeded,
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
export function addA0Task(budget, elapsedMs, metered) {
|
|
36
|
+
const tasks = budget.tasks + 1;
|
|
37
|
+
const elapsed = budget.elapsedMs + elapsedMs;
|
|
38
|
+
const exceeded = budget.exceeded ||
|
|
39
|
+
(budget.maxTasks !== undefined && tasks > budget.maxTasks) ||
|
|
40
|
+
(budget.maxDurationMs !== undefined && elapsed > budget.maxDurationMs) ||
|
|
41
|
+
(budget.limitUsd !== undefined && metered && overLimit(budget.spentUsd, budget.limitUsd));
|
|
42
|
+
return { ...budget, tasks, elapsedMs: elapsed, exceeded };
|
|
43
|
+
}
|
|
44
|
+
export function budgetCanSpend(budget, nextCostUsd) {
|
|
45
|
+
return (!budget.exceeded &&
|
|
46
|
+
(budget.limitUsd === undefined ||
|
|
47
|
+
budget.spentUsd + nextCostUsd <= budget.limitUsd + USD_EPSILON));
|
|
48
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import { LANE_IDS, type LaneId } from '../report/manifest.js';
|
|
2
|
+
export { LANE_IDS };
|
|
3
|
+
export type { LaneId };
|
|
4
|
+
export interface LaneSelection {
|
|
5
|
+
review: boolean;
|
|
6
|
+
flow: boolean;
|
|
7
|
+
app: boolean;
|
|
8
|
+
a0: boolean;
|
|
9
|
+
}
|
|
10
|
+
export declare function defaultLaneSelection(): LaneSelection;
|
|
11
|
+
export declare function selectionFromFlags(flags: Partial<Record<LaneId, boolean>>): LaneSelection;
|
|
12
|
+
export declare function selectedLanes(selection: LaneSelection): LaneId[];
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
import { LANE_IDS } from '../report/manifest.js';
|
|
2
|
+
export { LANE_IDS };
|
|
3
|
+
export function defaultLaneSelection() {
|
|
4
|
+
return { review: true, flow: false, app: false, a0: false };
|
|
5
|
+
}
|
|
6
|
+
export function selectionFromFlags(flags) {
|
|
7
|
+
const selection = defaultLaneSelection();
|
|
8
|
+
for (const lane of LANE_IDS) {
|
|
9
|
+
if (flags[lane] !== undefined)
|
|
10
|
+
selection[lane] = flags[lane];
|
|
11
|
+
}
|
|
12
|
+
return selection;
|
|
13
|
+
}
|
|
14
|
+
export function selectedLanes(selection) {
|
|
15
|
+
return LANE_IDS.filter((lane) => selection[lane]);
|
|
16
|
+
}
|