vigiles 25.0.0 → 26.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/adapters/claude-code/run-scripts.d.ts +5 -2
- package/dist/adapters/claude-code/run-scripts.js +6 -3
- package/dist/adapters/codex/eval.d.ts +18 -0
- package/dist/adapters/codex/eval.js +27 -0
- package/dist/cli.d.ts +2 -1
- package/dist/cli.js +157 -74
- package/dist/core/adopt.d.ts +50 -1
- package/dist/core/adopt.js +100 -6
- package/dist/core/bash-effects.d.ts +11 -0
- package/dist/core/bash-effects.js +52 -12
- package/dist/core/bash-equivalents.d.ts +18 -0
- package/dist/core/bash-equivalents.js +239 -0
- package/dist/core/coverage.d.ts +3 -3
- package/dist/core/coverage.js +10 -6
- package/dist/core/hook-program.js +48 -2
- package/dist/core/orphans.d.ts +8 -1
- package/dist/core/orphans.js +8 -2
- package/dist/core/types.d.ts +14 -6
- package/dist/eval-cache.d.ts +7 -0
- package/dist/eval-lock.d.ts +20 -0
- package/dist/eval.d.ts +117 -4
- package/dist/eval.js +164 -32
- package/dist/exclude.d.ts +25 -0
- package/dist/exclude.js +132 -0
- package/dist/guardrail-check.d.ts +43 -24
- package/dist/guardrail-check.js +57 -0
- package/dist/scan-behavioral.d.ts +10 -0
- package/dist/scan-behavioral.js +1 -0
- package/dist/test.d.ts +1 -1
- package/dist/test.js +3 -2
- package/package.json +1 -1
package/dist/eval.js
CHANGED
|
@@ -1,8 +1,11 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
-
exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = void 0;
|
|
3
|
+
exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = exports.spawnAgent = exports.EFFORT_ENV_VAR = void 0;
|
|
4
4
|
exports.resolveSpawnEnv = resolveSpawnEnv;
|
|
5
|
-
exports.
|
|
5
|
+
exports.pinEffortEnv = pinEffortEnv;
|
|
6
|
+
exports.effortRejection = effortRejection;
|
|
7
|
+
exports.withEffortGuard = withEffortGuard;
|
|
8
|
+
exports.buildAgentArgs = buildAgentArgs;
|
|
6
9
|
exports.unregisteredSkillFiles = unregisteredSkillFiles;
|
|
7
10
|
exports.runEval = runEval;
|
|
8
11
|
exports.measureWith = measureWith;
|
|
@@ -99,35 +102,136 @@ function writeFiles(cwd, files) {
|
|
|
99
102
|
* leak the host environment into an untrusted, model-driven run.
|
|
100
103
|
*/
|
|
101
104
|
function resolveSpawnEnv(a, base = process.env) {
|
|
102
|
-
|
|
105
|
+
const resolved = a.replaceEnv ? (a.env ?? {}) : { ...base, ...a.env };
|
|
106
|
+
return pinEffortEnv(resolved, a.effort);
|
|
103
107
|
}
|
|
108
|
+
/**
|
|
109
|
+
* The env var name the harness reads for the reasoning budget. It sits ABOVE the
|
|
110
|
+
* `--effort` flag in the CLI's own precedence chain, so passing the flag alone
|
|
111
|
+
* does NOT pin the level.
|
|
112
|
+
*/
|
|
113
|
+
exports.EFFORT_ENV_VAR = "CLAUDE_CODE_EFFORT_LEVEL";
|
|
114
|
+
/**
|
|
115
|
+
* Pin the effort the run actually gets, so the recorded effort is the effort
|
|
116
|
+
* that ran.
|
|
117
|
+
*
|
|
118
|
+
* WHY THIS EXISTS AND WHY IT IS NOT OPTIONAL. Effort has THREE inputs — the
|
|
119
|
+
* `--effort` flag, the `effortLevel` settings key, and `CLAUDE_CODE_EFFORT_LEVEL`
|
|
120
|
+
* — and the env var wins over the flag. `EPHEMERAL_ALLOW_PREFIXES` passes
|
|
121
|
+
* `CLAUDE_*` through by design (the CLI reads several such knobs and dropping one
|
|
122
|
+
* is the failure mode), so an ambient `CLAUDE_CODE_EFFORT_LEVEL=max` in the
|
|
123
|
+
* author's shell survives even the SCRUBBED ephemeral env. Without this pin,
|
|
124
|
+
* hashing effort into the lock would make the lock CONFIDENTLY WRONG: it would
|
|
125
|
+
* record `low` over a run that executed at `max` — the exact defect the feature
|
|
126
|
+
* exists to prevent, reintroduced by the fix for it.
|
|
127
|
+
*
|
|
128
|
+
* Both directions matter, so both are handled:
|
|
129
|
+
* - effort DECLARED → set the var, overriding whatever the shell had.
|
|
130
|
+
* - effort OMITTED → DELETE an inherited var, so "omit" means the harness
|
|
131
|
+
* default rather than "whatever this machine happened to
|
|
132
|
+
* export". An omitted effort must not be a hidden input.
|
|
133
|
+
*/
|
|
134
|
+
function pinEffortEnv(env, effort) {
|
|
135
|
+
// Rebuilt WITHOUT the key rather than deleting or assigning `undefined`:
|
|
136
|
+
// omission has to be provable here, and whether a spawn drops an
|
|
137
|
+
// `undefined`-valued env entry is a Node-version detail we should not lean on.
|
|
138
|
+
const { [exports.EFFORT_ENV_VAR]: _inherited, ...rest } = env;
|
|
139
|
+
return effort === undefined
|
|
140
|
+
? rest
|
|
141
|
+
: { ...rest, [exports.EFFORT_ENV_VAR]: String(effort) };
|
|
142
|
+
}
|
|
143
|
+
/**
|
|
144
|
+
* The harness's own rejection of an `--effort` value, or null. Pure.
|
|
145
|
+
*
|
|
146
|
+
* The CLI does NOT fail on a bad level — it prints this to stderr and silently
|
|
147
|
+
* runs at its default. That silent substitution is precisely the bug class this
|
|
148
|
+
* feature addresses (a number produced by a configuration nobody asked for), so
|
|
149
|
+
* a rejected value must never become a sample. Matched on the binary's own
|
|
150
|
+
* wording, the same shape as {@link isRateLimited}.
|
|
151
|
+
*/
|
|
152
|
+
function effortRejection(out) {
|
|
153
|
+
const text = `${out.stderr ?? ""}\n${out.stdout}`;
|
|
154
|
+
const m = /Unknown --effort value[^\n]*/.exec(text);
|
|
155
|
+
return m ? m[0].trim() : null;
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Wrap a runner so a run the harness rejected on `--effort` FAILS LOUDLY.
|
|
159
|
+
*
|
|
160
|
+
* Applied ONCE, around the real runner, rather than as a guard repeated at each
|
|
161
|
+
* of the five `runner(...)` call sites — a guard per call site is the shape that
|
|
162
|
+
* left four of five compilers unprotected in #173.
|
|
163
|
+
*
|
|
164
|
+
* It THROWS rather than counting the trial as `runError`. A `runError` trial is
|
|
165
|
+
* dropped from the denominator, which is right for a transient (a rate limit) and
|
|
166
|
+
* wrong here: an unusable effort value is deterministic and repeatable, so every
|
|
167
|
+
* trial fails it and the run would report a rate computed over ZERO samples. A
|
|
168
|
+
* configuration mistake should stop the run and name itself.
|
|
169
|
+
*/
|
|
170
|
+
function withEffortGuard(runner) {
|
|
171
|
+
// 🔴 DELIBERATELY NOT `async`. An `async` wrapper turns the wrapped runner's
|
|
172
|
+
// SYNCHRONOUS throws into rejected promises, and the real runner refuses
|
|
173
|
+
// synchronously on purpose — `refuseDuringEvalLoad` / `refuseUnderForeignRunner`
|
|
174
|
+
// stop a paid eval from billing when a foreign test runner collects it. Making
|
|
175
|
+
// this `async` silently downgraded those refusals from "throws at the call" to
|
|
176
|
+
// "returns a promise that rejects", which `assert.throws` cannot see and an
|
|
177
|
+
// un-awaited caller would not notice. Caught by the full suite, not by the
|
|
178
|
+
// targeted one; pinned below by `withEffortGuard preserves a SYNCHRONOUS throw`.
|
|
179
|
+
return (a) => {
|
|
180
|
+
const pending = runner(a);
|
|
181
|
+
return pending.then((out) => {
|
|
182
|
+
const rejection = effortRejection(out);
|
|
183
|
+
if (rejection !== null) {
|
|
184
|
+
throw new Error(`the harness rejected effort ${JSON.stringify(a.effort)}: ${rejection}`);
|
|
185
|
+
}
|
|
186
|
+
return out;
|
|
187
|
+
});
|
|
188
|
+
};
|
|
189
|
+
}
|
|
190
|
+
/**
|
|
191
|
+
* Build the real runner's argv. Pure and exported so the FLAGS are provable —
|
|
192
|
+
* `spawnAgentRaw` is `v8 ignore`d (it spawns a subprocess), so an argv assembled
|
|
193
|
+
* inline there could not be asserted at all. Mirrors `buildCodexArgs`.
|
|
194
|
+
*/
|
|
195
|
+
function buildAgentArgs(a) {
|
|
196
|
+
return [
|
|
197
|
+
"-p",
|
|
198
|
+
a.task,
|
|
199
|
+
// stream-json (+ --verbose, required with -p) so the per-turn tool_use
|
|
200
|
+
// events survive into `ctx.toolCalls` — the unified Trace, same as the
|
|
201
|
+
// harness tier. The terminal `result` event still carries num_turns/output.
|
|
202
|
+
"--output-format",
|
|
203
|
+
"stream-json",
|
|
204
|
+
"--verbose",
|
|
205
|
+
"--model",
|
|
206
|
+
a.model,
|
|
207
|
+
...(a.effort !== undefined ? ["--effort", String(a.effort)] : []),
|
|
208
|
+
"--permission-mode",
|
|
209
|
+
"acceptEdits",
|
|
210
|
+
...(a.pluginDir !== undefined
|
|
211
|
+
? ["--plugin-dir", (0, node_path_1.resolve)(a.pluginDir)]
|
|
212
|
+
: []),
|
|
213
|
+
...(a.hasSettings ? ["--settings", "settings.json"] : []),
|
|
214
|
+
"--allowedTools",
|
|
215
|
+
...a.tools,
|
|
216
|
+
];
|
|
217
|
+
}
|
|
218
|
+
/**
|
|
219
|
+
* The real `claude`-spawning runner (composition root). Exported so other
|
|
220
|
+
* real-model entries (e.g. the `audit` trigger tier) bind the same runner.
|
|
221
|
+
*
|
|
222
|
+
* The effort guard is composed in HERE, at the single definition, rather than at
|
|
223
|
+
* each of the places that bind this runner — so every consumer, including ones
|
|
224
|
+
* not yet written, is covered by construction. Guarding each call site instead is
|
|
225
|
+
* the shape that left four of five compilers unprotected in #173.
|
|
226
|
+
*/
|
|
227
|
+
exports.spawnAgent = withEffortGuard(spawnAgentRaw);
|
|
104
228
|
/* v8 ignore start -- real claude subprocess; exercised by bench/, not the unit gate */
|
|
105
|
-
/** The
|
|
106
|
-
|
|
107
|
-
function spawnAgent(a) {
|
|
229
|
+
/** The unguarded spawn itself; wrapped by {@link spawnAgent}, never bound raw. */
|
|
230
|
+
function spawnAgentRaw(a) {
|
|
108
231
|
(0, eval_load_phase_js_1.refuseDuringEvalLoad)("spawning `claude`");
|
|
109
232
|
(0, foreign_runner_js_1.refuseUnderForeignRunner)("spawning `claude`");
|
|
110
233
|
return new Promise((resolvePromise) => {
|
|
111
|
-
const args =
|
|
112
|
-
"-p",
|
|
113
|
-
a.task,
|
|
114
|
-
// stream-json (+ --verbose, required with -p) so the per-turn tool_use
|
|
115
|
-
// events survive into `ctx.toolCalls` — the unified Trace, same as the
|
|
116
|
-
// harness tier. The terminal `result` event still carries num_turns/output.
|
|
117
|
-
"--output-format",
|
|
118
|
-
"stream-json",
|
|
119
|
-
"--verbose",
|
|
120
|
-
"--model",
|
|
121
|
-
a.model,
|
|
122
|
-
"--permission-mode",
|
|
123
|
-
"acceptEdits",
|
|
124
|
-
...(a.pluginDir !== undefined
|
|
125
|
-
? ["--plugin-dir", (0, node_path_1.resolve)(a.pluginDir)]
|
|
126
|
-
: []),
|
|
127
|
-
...(a.hasSettings ? ["--settings", "settings.json"] : []),
|
|
128
|
-
"--allowedTools",
|
|
129
|
-
...a.tools,
|
|
130
|
-
];
|
|
234
|
+
const args = buildAgentArgs(a);
|
|
131
235
|
const child = (0, node_child_process_1.spawn)(runtime_js_1.claudeCodeRuntime.agentBinary, args, {
|
|
132
236
|
cwd: a.cwd,
|
|
133
237
|
// The security-critical env resolution (overlay vs. scrubbed replacement)
|
|
@@ -194,7 +298,7 @@ function warnUnregisteredSkillArms(arms) {
|
|
|
194
298
|
}
|
|
195
299
|
async function runEval(spec) {
|
|
196
300
|
warnUnregisteredSkillArms(spec.arms);
|
|
197
|
-
const report = await runEvalWith(spec, spawnAgent);
|
|
301
|
+
const report = await runEvalWith(spec, exports.spawnAgent);
|
|
198
302
|
// Surface what the run spent — tokens + API-equivalent $, and a LOUD warning if
|
|
199
303
|
// it was billed to a metered API key instead of the subscription. See eval-cost.ts.
|
|
200
304
|
(0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromEvalReport)(report));
|
|
@@ -237,6 +341,7 @@ async function measureWith(spec, runner) {
|
|
|
237
341
|
task: spec.task,
|
|
238
342
|
trials: spec.trials ?? 5,
|
|
239
343
|
model: spec.model ?? "sonnet",
|
|
344
|
+
effort: spec.effort,
|
|
240
345
|
allowedTools: spec.allowedTools,
|
|
241
346
|
timeoutMs: spec.timeoutMs,
|
|
242
347
|
spacingSec: spec.spacingSec,
|
|
@@ -266,7 +371,7 @@ async function measureWith(spec, runner) {
|
|
|
266
371
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
|
|
267
372
|
/** Score a check vocabulary across trials against the real `claude` CLI. */
|
|
268
373
|
async function measure(spec) {
|
|
269
|
-
const report = await measureWith(spec, spawnAgent);
|
|
374
|
+
const report = await measureWith(spec, exports.spawnAgent);
|
|
270
375
|
(0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromArm)(report.usage));
|
|
271
376
|
return report;
|
|
272
377
|
}
|
|
@@ -283,6 +388,7 @@ async function measureArmsWith(spec, runner) {
|
|
|
283
388
|
task: spec.task,
|
|
284
389
|
trials: spec.trials ?? 5,
|
|
285
390
|
model: spec.model ?? "sonnet",
|
|
391
|
+
effort: spec.effort,
|
|
286
392
|
allowedTools: spec.allowedTools,
|
|
287
393
|
timeoutMs: spec.timeoutMs,
|
|
288
394
|
spacingSec: spec.spacingSec,
|
|
@@ -337,7 +443,7 @@ function stubArmPluginDirs(arms) {
|
|
|
337
443
|
/** Score checks across arms against the real `claude` CLI. */
|
|
338
444
|
async function measureArms(spec) {
|
|
339
445
|
warnUnregisteredSkillArms(spec.arms);
|
|
340
|
-
const report = await measureArmsWith(spec, spawnAgent);
|
|
446
|
+
const report = await measureArmsWith(spec, exports.spawnAgent);
|
|
341
447
|
// Sum every arm's spend — an A/B run pays for both arms.
|
|
342
448
|
(0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.sumCosts)(Object.values(report.arms).map((a) => (0, eval_cost_js_1.costFromArm)(a.usage))));
|
|
343
449
|
return report;
|
|
@@ -552,6 +658,7 @@ async function runSkillSelectionTrial(args) {
|
|
|
552
658
|
task: args.prompt,
|
|
553
659
|
cwd,
|
|
554
660
|
model: args.model,
|
|
661
|
+
effort: args.effort,
|
|
555
662
|
tools: args.tools ?? ["Read", "Edit", "Write", "Bash", "Skill"],
|
|
556
663
|
hasSettings: false,
|
|
557
664
|
pluginDir: args.pluginDir,
|
|
@@ -637,6 +744,7 @@ async function runWithCache(runArgs, keyParts, runner, cfg) {
|
|
|
637
744
|
const key = (0, eval_cache_js_1.cacheKey)({
|
|
638
745
|
task: runArgs.task,
|
|
639
746
|
model: runArgs.model,
|
|
747
|
+
effort: runArgs.effort,
|
|
640
748
|
tools: runArgs.tools,
|
|
641
749
|
files: keyParts.files,
|
|
642
750
|
settings: keyParts.settings,
|
|
@@ -967,6 +1075,7 @@ async function executeTrial(spec, arm, trialIndex, runner, cfg) {
|
|
|
967
1075
|
cwd,
|
|
968
1076
|
// A model comparison is a harness A/B: an arm may override the model.
|
|
969
1077
|
model: arm.model ?? cfg.model,
|
|
1078
|
+
effort: arm.effort ?? cfg.effort,
|
|
970
1079
|
tools: cfg.tools,
|
|
971
1080
|
hasSettings,
|
|
972
1081
|
pluginDir: arm.pluginDir,
|
|
@@ -1100,8 +1209,16 @@ async function withEvalLock(args, produce) {
|
|
|
1100
1209
|
}
|
|
1101
1210
|
if (!isDatedModel(args.model))
|
|
1102
1211
|
warnFloatingModel(args.model);
|
|
1212
|
+
// OVERLAP, DELIBERATE — do not delete this as dead. Effort reaches the hash
|
|
1213
|
+
// twice: here (the CHOKEPOINT every seam passes through, so a future seam that
|
|
1214
|
+
// forgets to fold effort into its own `inputs` is still covered) and inside
|
|
1215
|
+
// each seam's `inputs` (which alone can see a PER-ARM override this line
|
|
1216
|
+
// cannot). Measured 2026-09-01: removing either one alone leaves the suite
|
|
1217
|
+
// green; removing BOTH fails `changing effort makes a committed lock STALE`.
|
|
1218
|
+
// That is two populations covered, not one line duplicated.
|
|
1103
1219
|
const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
|
|
1104
1220
|
model: args.model,
|
|
1221
|
+
effort: args.effort,
|
|
1105
1222
|
evalApiVersion: lock.evalApiVersion,
|
|
1106
1223
|
inputs: args.inputs,
|
|
1107
1224
|
});
|
|
@@ -1116,6 +1233,7 @@ async function withEvalLock(args, produce) {
|
|
|
1116
1233
|
name: args.name,
|
|
1117
1234
|
inputsHash,
|
|
1118
1235
|
model: args.model,
|
|
1236
|
+
effort: args.effort,
|
|
1119
1237
|
harnessVersionKey: harnessVersion(),
|
|
1120
1238
|
evalApiVersion: lock.evalApiVersion,
|
|
1121
1239
|
builtAt: new Date().toISOString(),
|
|
@@ -1166,6 +1284,10 @@ function evalArmsInputs(spec, cfg) {
|
|
|
1166
1284
|
const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
|
|
1167
1285
|
arms[name] = {
|
|
1168
1286
|
model: arm.model ?? cfg.model,
|
|
1287
|
+
// The per-ARM half of the overlap documented at `inputsHash` — the
|
|
1288
|
+
// chokepoint sees only the eval-level effort, so an arm that overrides it
|
|
1289
|
+
// would otherwise hash identically to its sibling.
|
|
1290
|
+
effort: arm.effort ?? cfg.effort,
|
|
1169
1291
|
tools: [...cfg.tools].sort(),
|
|
1170
1292
|
files: stripPluginRoot(resolved.files, absRoot),
|
|
1171
1293
|
settings: stripPluginRoot(resolved.settings, absRoot),
|
|
@@ -1203,6 +1325,7 @@ async function runEvalWith(spec, runner) {
|
|
|
1203
1325
|
const backoffMs = spec.retryBackoffMs ?? 1000;
|
|
1204
1326
|
const cfg = {
|
|
1205
1327
|
model: spec.model ?? "haiku",
|
|
1328
|
+
effort: spec.effort,
|
|
1206
1329
|
tools: spec.allowedTools ?? ["Read", "Edit", "Write", "Bash"],
|
|
1207
1330
|
timeoutMs: spec.timeoutMs ?? 240000,
|
|
1208
1331
|
cache: spec.cache ?? "off",
|
|
@@ -1232,7 +1355,7 @@ async function runEvalWith(spec, runner) {
|
|
|
1232
1355
|
// resolveHarness/hashDir work). `check` replays the committed report below
|
|
1233
1356
|
// without ever entering the run pool — so no model is driven in CI.
|
|
1234
1357
|
const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
|
|
1235
|
-
return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
|
|
1358
|
+
return withEvalLock({ name: spec.name, inputs, model: cfg.model, effort: cfg.effort, lock }, async () => {
|
|
1236
1359
|
const results = await runPool(units, concurrency, worker);
|
|
1237
1360
|
const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
|
|
1238
1361
|
return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
|
|
@@ -1285,7 +1408,7 @@ function formatEvalReport(report) {
|
|
|
1285
1408
|
* The asymmetry reflects default-vs-injected, not a hexagonal violation.
|
|
1286
1409
|
*/
|
|
1287
1410
|
exports.claudeEvalDriver = {
|
|
1288
|
-
runner: spawnAgent,
|
|
1411
|
+
runner: exports.spawnAgent,
|
|
1289
1412
|
parse: parseClaudeRun,
|
|
1290
1413
|
harness: "claude-code",
|
|
1291
1414
|
};
|
|
@@ -1591,6 +1714,7 @@ async function runTriggerTrial(prompt, cfg, runner) {
|
|
|
1591
1714
|
task: prompt,
|
|
1592
1715
|
cwd,
|
|
1593
1716
|
model: cfg.model,
|
|
1717
|
+
effort: cfg.effort,
|
|
1594
1718
|
tools: cfg.tools,
|
|
1595
1719
|
hasSettings: false,
|
|
1596
1720
|
pluginDir: cfg.pluginDir,
|
|
@@ -1690,6 +1814,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1690
1814
|
// Sonnet, not haiku: trigger-rate is a selection measurement and haiku
|
|
1691
1815
|
// under-selects, producing false-negative recall (see TriggerRateSpec.model).
|
|
1692
1816
|
model,
|
|
1817
|
+
effort: spec.effort,
|
|
1693
1818
|
tools: spec.allowedTools ?? ["Read", "Edit", "Write", "Bash", "Skill"],
|
|
1694
1819
|
timeoutMs: spec.timeoutMs ?? 240000,
|
|
1695
1820
|
spacing: (spec.spacingSec ?? 4) * 1000,
|
|
@@ -1715,6 +1840,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1715
1840
|
? [...spec.irrelevantPrompts]
|
|
1716
1841
|
: undefined,
|
|
1717
1842
|
model: cfg.model,
|
|
1843
|
+
effort: cfg.effort,
|
|
1718
1844
|
tools: [...cfg.tools].sort(),
|
|
1719
1845
|
fixture: spec.fixture,
|
|
1720
1846
|
competitors,
|
|
@@ -1723,7 +1849,13 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1723
1849
|
// STALE if the eval is switched to another harness.
|
|
1724
1850
|
harness,
|
|
1725
1851
|
};
|
|
1726
|
-
return await withEvalLock({
|
|
1852
|
+
return await withEvalLock({
|
|
1853
|
+
name: spec.name,
|
|
1854
|
+
inputs: triggerInputs,
|
|
1855
|
+
model: cfg.model,
|
|
1856
|
+
effort: cfg.effort,
|
|
1857
|
+
lock,
|
|
1858
|
+
}, async () => {
|
|
1727
1859
|
const relevant = await runTriggerSet(spec.prompts, cfg, runner);
|
|
1728
1860
|
const base = {
|
|
1729
1861
|
rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import type { IgnoreLike } from "glob";
|
|
2
|
+
/** Always excluded, whatever the config says. Root-relative, like `exclude`. */
|
|
3
|
+
export declare const EXCLUDE_FLOOR: readonly string[];
|
|
4
|
+
/** The parsed `.vigilesrc.json#exclude`, in the forms a walk consumes. */
|
|
5
|
+
export interface ExcludeSet {
|
|
6
|
+
/** Absolute repo root every pattern is relative to. */
|
|
7
|
+
readonly root: string;
|
|
8
|
+
/** The user's patterns, as written (for messages). */
|
|
9
|
+
readonly patterns: readonly string[];
|
|
10
|
+
/**
|
|
11
|
+
* The string-list face for a glob rooted AT `root`: the floor, then each user
|
|
12
|
+
* pattern normalized so a bare directory name excludes its subtree (`bench` →
|
|
13
|
+
* `bench`, `bench/**`), which is what "tsconfig-style" promises.
|
|
14
|
+
*/
|
|
15
|
+
readonly ignore: readonly string[];
|
|
16
|
+
/** The function face for `globSync`, correct whatever the glob's `cwd` is. */
|
|
17
|
+
readonly globIgnore: IgnoreLike;
|
|
18
|
+
/** Is this root-relative path excluded (floor or user pattern)? */
|
|
19
|
+
matches(rel: string): boolean;
|
|
20
|
+
/** The pattern that excludes this root-relative path, or null when none does. */
|
|
21
|
+
explain(rel: string): string | null;
|
|
22
|
+
}
|
|
23
|
+
/** Parse `.vigilesrc.json#exclude` once, against `root`. */
|
|
24
|
+
export declare function excludeSet(root: string, patterns: readonly string[] | undefined): ExcludeSet;
|
|
25
|
+
//# sourceMappingURL=exclude.d.ts.map
|
package/dist/exclude.js
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.EXCLUDE_FLOOR = void 0;
|
|
4
|
+
exports.excludeSet = excludeSet;
|
|
5
|
+
/**
|
|
6
|
+
* The ONE exclusion policy for every walk that polices the user's repository.
|
|
7
|
+
*
|
|
8
|
+
* `.vigilesrc.json#exclude` is documented as "tsconfig-style" — a list of paths or
|
|
9
|
+
* globs the repo's own lint should not police (vendored corpora, benchmark
|
|
10
|
+
* fixtures, frozen reproductions). Issue #192: it was honoured by three walks,
|
|
11
|
+
* ignored by six, and the three that honoured it did not agree with each other.
|
|
12
|
+
*
|
|
13
|
+
* 🔴 TWO DIALECTS WERE ALREADY IN `main`, AND THEY DISAGREED ON `"bench"`.
|
|
14
|
+
* `glob`'s string `ignore` treats a bare directory name as a file pattern:
|
|
15
|
+
* measured 2026-09-03 on glob 13, `ignore: ["bench"]` and `["bench/"]` exclude
|
|
16
|
+
* NOTHING, only `["bench/**"]` works. The minimatch helper inside
|
|
17
|
+
* `discoverNestedBundles` accepted the bare name. tsc and ESLint both treat the
|
|
18
|
+
* bare name as the directory (measured the same day). So a user who wrote the
|
|
19
|
+
* key the way its JSDoc promised got the nested-bundle pass filtered and every
|
|
20
|
+
* glob-backed pass unfiltered — with no way to tell from the output.
|
|
21
|
+
*
|
|
22
|
+
* This module is the fix in the shape the fence-parser fix took
|
|
23
|
+
* (`core/markdown.ts`): not one WALK — the walks legitimately have different
|
|
24
|
+
* scopes — but ONE PREDICATE, parsed once from config, that every walk consumes.
|
|
25
|
+
* It offers the predicate in the two forms a walk needs and nothing else:
|
|
26
|
+
*
|
|
27
|
+
* - `globIgnore` — an `IgnoreLike` for `globSync`, keyed on the path's position
|
|
28
|
+
* RELATIVE TO THE REPO ROOT, so a glob rooted below the root (`vigiles lint
|
|
29
|
+
* some/dir`) still applies a root-relative `exclude` correctly. The string
|
|
30
|
+
* list could not: `bench/**` relative to `some/dir` matches nothing.
|
|
31
|
+
* - `ignore` — the normalized string list, for pure detectors in `core/` that
|
|
32
|
+
* take an ignore list by injection and glob from the repo root themselves.
|
|
33
|
+
* - `matches(rel)` / `explain(rel)` — for `readdirSync`-style walks and for the
|
|
34
|
+
* one line printed when an explicitly named path is processed anyway.
|
|
35
|
+
*
|
|
36
|
+
* 🔴 EVERY IN-SCOPE WALK TAKES AN `ExcludeSet` AS A REQUIRED PARAMETER. The
|
|
37
|
+
* shape that let `findSpecs` go a year without the key was
|
|
38
|
+
* `exclude: readonly string[] = []` — optional-with-default lets a call site
|
|
39
|
+
* forget, and a forgotten argument is indistinguishable from an empty config.
|
|
40
|
+
* Do not reintroduce an optional `ExcludeSet` anywhere in scope.
|
|
41
|
+
*
|
|
42
|
+
* EXPLICITLY NAMED PATHS WIN, LOUDLY. Four tools were measured (2026-09-03):
|
|
43
|
+
* ripgrep and tsc process an explicitly named ignored file silently; ESLint
|
|
44
|
+
* skips it with a warning; prettier skips it and prints "All matched files use
|
|
45
|
+
* Prettier code style!" — the silent no-op this repo has burned itself on. vigiles
|
|
46
|
+
* takes the rg/tsc semantics (an argument is an instruction) with ESLint's
|
|
47
|
+
* loudness: the path is processed and ONE line says which pattern it matched.
|
|
48
|
+
* `exclude` filters DISCOVERY, never an argument.
|
|
49
|
+
*
|
|
50
|
+
* WHAT DELIBERATELY DOES NOT GO THROUGH THIS MODULE — so the next reader does not
|
|
51
|
+
* "fix" it (the classification is issue #192's comment):
|
|
52
|
+
*
|
|
53
|
+
* - `core/compile.ts#validateGlobRef` — verifies a spec's own `glob()` reference
|
|
54
|
+
* resolves to ≥1 file. That is reference verification of the user's claim
|
|
55
|
+
* about the repo, not lint scope; excluding a dir must not make a true ref
|
|
56
|
+
* false.
|
|
57
|
+
* - `cli.ts#specReferencedElsewhere` — eject's "is this spec compiled anywhere
|
|
58
|
+
* else" safety check. Wider is safer: a target under an excluded dir is still
|
|
59
|
+
* a target that would be orphaned.
|
|
60
|
+
* - `core/validate.ts#expandGlobs` — expands a pattern the user TYPED. An
|
|
61
|
+
* argument wins (see above); it is not discovery.
|
|
62
|
+
* - Surface walks INSIDE a bundle (`plugin-loader.ts#readTree`,
|
|
63
|
+
* `skill-reachability.ts`): `exclude` applies at BUNDLE granularity via
|
|
64
|
+
* `discoverNestedBundles`; a skill inside your own `skills/` is yours.
|
|
65
|
+
* - Internal machinery that never enumerates the user's repo as lint surface:
|
|
66
|
+
* eval temp installs (`eval.ts`, `adapters/codex/eval.ts`), the eval cache and
|
|
67
|
+
* locks (`eval-cache.ts`, `eval-lock.ts`, `run-script.ts#snapshotTree`),
|
|
68
|
+
* `.vigiles/hooks/` discovery (`hook-install.ts`, a fixed dir), sidecars
|
|
69
|
+
* (`core/sidecar.ts`), linter catalogs / rulesDirs / toolchain paths
|
|
70
|
+
* (`core/linters.ts`, `core/generate-schema.ts`, `core/generate-types.ts`),
|
|
71
|
+
* `init`'s shallow adoptable-surface sweep (`cli.ts#discoverAdoptableSurfaces`)
|
|
72
|
+
* and lint-config collection (`cli.ts#safeReaddir`), and
|
|
73
|
+
* `core/generate-harness.ts` (an explicit, non-recursive dir argument).
|
|
74
|
+
*
|
|
75
|
+
* The floor (`node_modules`, `dist`, `.git`, `.vigiles`) lives here too, so a
|
|
76
|
+
* walk cannot carry its own private copy of it — the original `findSpecs` list
|
|
77
|
+
* lacked `.vigiles/**` while the three `core/` detectors had it.
|
|
78
|
+
*/
|
|
79
|
+
const minimatch_1 = require("minimatch");
|
|
80
|
+
const node_path_1 = require("node:path");
|
|
81
|
+
/** Always excluded, whatever the config says. Root-relative, like `exclude`. */
|
|
82
|
+
exports.EXCLUDE_FLOOR = [
|
|
83
|
+
"node_modules/**",
|
|
84
|
+
"dist/**",
|
|
85
|
+
".git/**",
|
|
86
|
+
".vigiles/**",
|
|
87
|
+
];
|
|
88
|
+
/** `a\b\c` → `a/b/c`, drop a leading `./`, drop a trailing `/`. */
|
|
89
|
+
function normalizeRel(rel) {
|
|
90
|
+
let r = rel
|
|
91
|
+
.split(node_path_1.sep)
|
|
92
|
+
.join("/")
|
|
93
|
+
.replace(/^(?:\.\/)+/, "");
|
|
94
|
+
while (r.endsWith("/"))
|
|
95
|
+
r = r.slice(0, -1);
|
|
96
|
+
return r;
|
|
97
|
+
}
|
|
98
|
+
/** Parse `.vigilesrc.json#exclude` once, against `root`. */
|
|
99
|
+
function excludeSet(root, patterns) {
|
|
100
|
+
const user = (patterns ?? []).map(normalizeRel).filter((p) => p !== "");
|
|
101
|
+
const all = [...exports.EXCLUDE_FLOOR, ...user];
|
|
102
|
+
// One compiled matcher pair per pattern: the pattern itself and its subtree.
|
|
103
|
+
const compiled = all.map((p) => ({
|
|
104
|
+
pattern: p,
|
|
105
|
+
self: new minimatch_1.Minimatch(p, { dot: true }),
|
|
106
|
+
subtree: new minimatch_1.Minimatch(`${p}/**`, { dot: true }),
|
|
107
|
+
}));
|
|
108
|
+
const explain = (rel) => {
|
|
109
|
+
const r = normalizeRel(rel);
|
|
110
|
+
if (r === "" || r === "." || r.startsWith("../"))
|
|
111
|
+
return null;
|
|
112
|
+
for (const c of compiled) {
|
|
113
|
+
if (c.self.match(r) || c.subtree.match(r))
|
|
114
|
+
return c.pattern;
|
|
115
|
+
}
|
|
116
|
+
return null;
|
|
117
|
+
};
|
|
118
|
+
const matches = (rel) => explain(rel) !== null;
|
|
119
|
+
const relOf = (p) => normalizeRel((0, node_path_1.relative)(root, p.fullpath()));
|
|
120
|
+
return {
|
|
121
|
+
root,
|
|
122
|
+
patterns: user,
|
|
123
|
+
ignore: [...exports.EXCLUDE_FLOOR, ...user.flatMap((p) => [p, `${p}/**`])],
|
|
124
|
+
globIgnore: {
|
|
125
|
+
ignored: (p) => matches(relOf(p)),
|
|
126
|
+
childrenIgnored: (p) => matches(relOf(p)),
|
|
127
|
+
},
|
|
128
|
+
matches,
|
|
129
|
+
explain,
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
//# sourceMappingURL=exclude.js.map
|
|
@@ -1,27 +1,3 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Guardrail verification — "prove your safety hook ACTUALLY blocks."
|
|
3
|
-
*
|
|
4
|
-
* The #1 verified Claude Code hook pain is FALSE CONFIDENCE: a developer ships a
|
|
5
|
-
* PreToolUse safety hook, believes they're protected, and finds out otherwise only
|
|
6
|
-
* when the agent force-pushes to main. The failure is silent — exit 1 instead of
|
|
7
|
-
* exit 2, the wrong JSON field, PostToolUse-can't-block, a wrong `jq` path, a missing
|
|
8
|
-
* `chmod +x` — all produce a hook that LOOKS like a guard and enforces nothing, with
|
|
9
|
-
* no error. (Crosley: "three different teams believed they had blocked force pushes";
|
|
10
|
-
* RFC #45427, closed not-planned. Full corpus: research/hook-pain-points.md.)
|
|
11
|
-
*
|
|
12
|
-
* This is the deterministic answer: feed a curated **disaster event** (`git push
|
|
13
|
-
* --force`, `rm -rf /`, `git commit --no-verify`, `cat ~/.ssh/*`, `curl … | sh`) to
|
|
14
|
-
* the hook via {@link runHook} and check the normalized decision is BLOCK. No model,
|
|
15
|
-
* no API key, runs in CI, works on a hand-written hook with NO vigiles spec — it
|
|
16
|
-
* verifies the hook's decision LOGIC, so it sidesteps CC's runtime delivery bugs
|
|
17
|
-
* (the model routing around a tool entirely, #45427 / #32376) which it deliberately
|
|
18
|
-
* does NOT claim to fix. (#34692, the old subagent-delivery gap, is fixed as of CC
|
|
19
|
-
* 2.1.241 — see src/subagent-delivery.test.ts.)
|
|
20
|
-
*
|
|
21
|
-
* Pure-ish (wraps the existing runHook tier). The catalog is harness-neutral data;
|
|
22
|
-
* the scaffold-test generator emits a test that calls these, and the same engine
|
|
23
|
-
* backs an informational coverage report.
|
|
24
|
-
*/
|
|
25
1
|
import { type RunHookOptions } from "./run-hook.js";
|
|
26
2
|
/** A category of dangerous action a guard might be meant to block. */
|
|
27
3
|
export type DisasterCategory = "destructive-git" | "destructive-fs" | "bypass-verification" | "secret-exfiltration" | "remote-code";
|
|
@@ -61,6 +37,49 @@ export interface VerifyGuardrailOptions extends RunHookOptions {
|
|
|
61
37
|
/** The PreToolUse event name to wrap each disaster in (default "PreToolUse"). */
|
|
62
38
|
readonly event?: string;
|
|
63
39
|
}
|
|
40
|
+
/**
|
|
41
|
+
* The same dangerous commands, spelled the other ways a shell reads identically.
|
|
42
|
+
*
|
|
43
|
+
* Takes a battery of hook test cases (shell commands wrapped as `PreToolUse`
|
|
44
|
+
* events — `DISASTER_CATALOG` is the shipped one) and returns MORE test cases:
|
|
45
|
+
* every command re-spelled with a quoted flag (`git push "--force"`), the short
|
|
46
|
+
* form of a flag (`-f`), an absolute or escaped head (`/usr/bin/git`, `\git`), or a
|
|
47
|
+
* pass-through wrapper (`sudo …`, `env …`). The shell runs each rewrite exactly as
|
|
48
|
+
* it runs the original. Feed them to `assertBlocksDisasters` alongside the
|
|
49
|
+
* originals:
|
|
50
|
+
*
|
|
51
|
+
* assertBlocksDisasters(hook, {
|
|
52
|
+
* events: [...DISASTER_CATALOG, ...experimental_alternateSpellings(DISASTER_CATALOG)],
|
|
53
|
+
* });
|
|
54
|
+
*
|
|
55
|
+
* WHAT BREAKS WITHOUT IT. A guard whose rule is "the command contains `--force`"
|
|
56
|
+
* blocks all seven catalog commands, so the battery is green — and lets
|
|
57
|
+
* `git push "--force"` through, because the quotes make it a different string.
|
|
58
|
+
* Measured 2026-09-02 on the shipped dogfood guard BEFORE this existed: 7/7
|
|
59
|
+
* originals blocked, **8 of 30** hand-written re-spellings blocked.
|
|
60
|
+
*
|
|
61
|
+
* WHY THE OUTPUT IS TRUSTWORTHY. Nothing new is judged. "Dangerous" is inherited
|
|
62
|
+
* from the original a human put in the battery; "the same command" is decided by
|
|
63
|
+
* the shell parser vigiles already uses for `runs()`/`touches()` (see
|
|
64
|
+
* {@link sameOperation}). A rewrite that fails that check throws rather than being
|
|
65
|
+
* emitted, so the battery can never quietly shrink.
|
|
66
|
+
*
|
|
67
|
+
* It returns ONLY the rewrites (never the originals), so pass the originals
|
|
68
|
+
* alongside as above. Each rewrite keeps its original's `tool` and `category`
|
|
69
|
+
* and takes the original's id with an index suffix (`force-push~4`), so a report
|
|
70
|
+
* names which spelling got through.
|
|
71
|
+
*
|
|
72
|
+
* In promptfoo's vocabulary each rewrite rule here is a "strategy"; the difference
|
|
73
|
+
* is that promptfoo's encodings (base64, leetspeak) may or may not be decoded by the
|
|
74
|
+
* target, whereas every rewrite here is one the shell provably executes identically
|
|
75
|
+
* — a miss is a guard bug, never an ambiguous input.
|
|
76
|
+
*
|
|
77
|
+
* @experimental Days old with a single consumer (this repo's own dogfood) and no
|
|
78
|
+
* external use. The set of rewrite rules and the id-suffix shape are the parts most
|
|
79
|
+
* likely to move; the prefix says so at every call site, which an import line or a
|
|
80
|
+
* doc note cannot.
|
|
81
|
+
*/
|
|
82
|
+
export declare function experimental_alternateSpellings(events: readonly DisasterEvent[]): readonly DisasterEvent[];
|
|
64
83
|
/**
|
|
65
84
|
* Run a hook command against the disaster battery and report which events it blocks.
|
|
66
85
|
* `hookCommand` is the exact shell the hook registers (e.g. `bash hooks/guard.sh` or
|
package/dist/guardrail-check.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.DISASTER_CATALOG = void 0;
|
|
4
|
+
exports.experimental_alternateSpellings = experimental_alternateSpellings;
|
|
4
5
|
exports.verifyGuardrail = verifyGuardrail;
|
|
5
6
|
exports.unblockedDisasters = unblockedDisasters;
|
|
6
7
|
exports.assertBlocksDisasters = assertBlocksDisasters;
|
|
@@ -29,6 +30,7 @@ exports.formatGuardrailReport = formatGuardrailReport;
|
|
|
29
30
|
* the scaffold-test generator emits a test that calls these, and the same engine
|
|
30
31
|
* backs an informational coverage report.
|
|
31
32
|
*/
|
|
33
|
+
const bash_equivalents_js_1 = require("./core/bash-equivalents.js");
|
|
32
34
|
const run_hook_js_1 = require("./run-hook.js");
|
|
33
35
|
/**
|
|
34
36
|
* The curated battery. Deliberately small and high-signal: each is a textbook
|
|
@@ -99,6 +101,61 @@ function selectEvents(opts) {
|
|
|
99
101
|
}
|
|
100
102
|
return exports.DISASTER_CATALOG;
|
|
101
103
|
}
|
|
104
|
+
/**
|
|
105
|
+
* The same dangerous commands, spelled the other ways a shell reads identically.
|
|
106
|
+
*
|
|
107
|
+
* Takes a battery of hook test cases (shell commands wrapped as `PreToolUse`
|
|
108
|
+
* events — `DISASTER_CATALOG` is the shipped one) and returns MORE test cases:
|
|
109
|
+
* every command re-spelled with a quoted flag (`git push "--force"`), the short
|
|
110
|
+
* form of a flag (`-f`), an absolute or escaped head (`/usr/bin/git`, `\git`), or a
|
|
111
|
+
* pass-through wrapper (`sudo …`, `env …`). The shell runs each rewrite exactly as
|
|
112
|
+
* it runs the original. Feed them to `assertBlocksDisasters` alongside the
|
|
113
|
+
* originals:
|
|
114
|
+
*
|
|
115
|
+
* assertBlocksDisasters(hook, {
|
|
116
|
+
* events: [...DISASTER_CATALOG, ...experimental_alternateSpellings(DISASTER_CATALOG)],
|
|
117
|
+
* });
|
|
118
|
+
*
|
|
119
|
+
* WHAT BREAKS WITHOUT IT. A guard whose rule is "the command contains `--force`"
|
|
120
|
+
* blocks all seven catalog commands, so the battery is green — and lets
|
|
121
|
+
* `git push "--force"` through, because the quotes make it a different string.
|
|
122
|
+
* Measured 2026-09-02 on the shipped dogfood guard BEFORE this existed: 7/7
|
|
123
|
+
* originals blocked, **8 of 30** hand-written re-spellings blocked.
|
|
124
|
+
*
|
|
125
|
+
* WHY THE OUTPUT IS TRUSTWORTHY. Nothing new is judged. "Dangerous" is inherited
|
|
126
|
+
* from the original a human put in the battery; "the same command" is decided by
|
|
127
|
+
* the shell parser vigiles already uses for `runs()`/`touches()` (see
|
|
128
|
+
* {@link sameOperation}). A rewrite that fails that check throws rather than being
|
|
129
|
+
* emitted, so the battery can never quietly shrink.
|
|
130
|
+
*
|
|
131
|
+
* It returns ONLY the rewrites (never the originals), so pass the originals
|
|
132
|
+
* alongside as above. Each rewrite keeps its original's `tool` and `category`
|
|
133
|
+
* and takes the original's id with an index suffix (`force-push~4`), so a report
|
|
134
|
+
* names which spelling got through.
|
|
135
|
+
*
|
|
136
|
+
* In promptfoo's vocabulary each rewrite rule here is a "strategy"; the difference
|
|
137
|
+
* is that promptfoo's encodings (base64, leetspeak) may or may not be decoded by the
|
|
138
|
+
* target, whereas every rewrite here is one the shell provably executes identically
|
|
139
|
+
* — a miss is a guard bug, never an ambiguous input.
|
|
140
|
+
*
|
|
141
|
+
* @experimental Days old with a single consumer (this repo's own dogfood) and no
|
|
142
|
+
* external use. The set of rewrite rules and the id-suffix shape are the parts most
|
|
143
|
+
* likely to move; the prefix says so at every call site, which an import line or a
|
|
144
|
+
* doc note cannot.
|
|
145
|
+
*/
|
|
146
|
+
function experimental_alternateSpellings(events) {
|
|
147
|
+
return events.flatMap((event) => {
|
|
148
|
+
const command = event.input["command"];
|
|
149
|
+
if (typeof command !== "string")
|
|
150
|
+
return [];
|
|
151
|
+
return (0, bash_equivalents_js_1.equivalentCommands)(command).map((variant, i) => ({
|
|
152
|
+
...event,
|
|
153
|
+
id: `${event.id}~${String(i + 1)}`,
|
|
154
|
+
label: `${event.label} — spelled: ${variant}`,
|
|
155
|
+
input: { ...event.input, command: variant },
|
|
156
|
+
}));
|
|
157
|
+
});
|
|
158
|
+
}
|
|
102
159
|
/**
|
|
103
160
|
* Run a hook command against the disaster battery and report which events it blocks.
|
|
104
161
|
* `hookCommand` is the exact shell the hook registers (e.g. `bash hooks/guard.sh` or
|
|
@@ -89,6 +89,16 @@ export interface SelectionOptions {
|
|
|
89
89
|
readonly trials?: number;
|
|
90
90
|
/** Selector model — defaults to Sonnet (a weaker model under-selects). */
|
|
91
91
|
readonly model?: string;
|
|
92
|
+
/**
|
|
93
|
+
* Reasoning budget (`claude --effort`) for the selector, or undefined for the
|
|
94
|
+
* harness default. Present for {@link measureSelectionMatrix}, the ASSERTABLE
|
|
95
|
+
* test primitive, where the configuration a number came from has to be pinnable.
|
|
96
|
+
*
|
|
97
|
+
* Deliberately NOT exposed as an `audit` CLI flag: `audit` is a local report,
|
|
98
|
+
* not a reproducibility surface, and a flag nobody can act on is surface without
|
|
99
|
+
* a use. The audit probe therefore leaves this unset and runs at the default.
|
|
100
|
+
*/
|
|
101
|
+
readonly effort?: string | number;
|
|
92
102
|
/** Parallel runs across the prompts × trials grid (default 1). */
|
|
93
103
|
readonly concurrency?: number;
|
|
94
104
|
/** Which harness drives it (default `"claude-code"`; others report n/a). */
|