vigiles 25.0.0 → 26.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/eval.js CHANGED
@@ -1,8 +1,11 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
- exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = void 0;
3
+ exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = exports.spawnAgent = exports.EFFORT_ENV_VAR = void 0;
4
4
  exports.resolveSpawnEnv = resolveSpawnEnv;
5
- exports.spawnAgent = spawnAgent;
5
+ exports.pinEffortEnv = pinEffortEnv;
6
+ exports.effortRejection = effortRejection;
7
+ exports.withEffortGuard = withEffortGuard;
8
+ exports.buildAgentArgs = buildAgentArgs;
6
9
  exports.unregisteredSkillFiles = unregisteredSkillFiles;
7
10
  exports.runEval = runEval;
8
11
  exports.measureWith = measureWith;
@@ -99,35 +102,136 @@ function writeFiles(cwd, files) {
99
102
  * leak the host environment into an untrusted, model-driven run.
100
103
  */
101
104
  function resolveSpawnEnv(a, base = process.env) {
102
- return a.replaceEnv ? (a.env ?? {}) : { ...base, ...a.env };
105
+ const resolved = a.replaceEnv ? (a.env ?? {}) : { ...base, ...a.env };
106
+ return pinEffortEnv(resolved, a.effort);
103
107
  }
108
+ /**
109
+ * The env var name the harness reads for the reasoning budget. It sits ABOVE the
110
+ * `--effort` flag in the CLI's own precedence chain, so passing the flag alone
111
+ * does NOT pin the level.
112
+ */
113
+ exports.EFFORT_ENV_VAR = "CLAUDE_CODE_EFFORT_LEVEL";
114
+ /**
115
+ * Pin the effort the run actually gets, so the recorded effort is the effort
116
+ * that ran.
117
+ *
118
+ * WHY THIS EXISTS AND WHY IT IS NOT OPTIONAL. Effort has THREE inputs — the
119
+ * `--effort` flag, the `effortLevel` settings key, and `CLAUDE_CODE_EFFORT_LEVEL`
120
+ * — and the env var wins over the flag. `EPHEMERAL_ALLOW_PREFIXES` passes
121
+ * `CLAUDE_*` through by design (the CLI reads several such knobs and dropping one
122
+ * is the failure mode), so an ambient `CLAUDE_CODE_EFFORT_LEVEL=max` in the
123
+ * author's shell survives even the SCRUBBED ephemeral env. Without this pin,
124
+ * hashing effort into the lock would make the lock CONFIDENTLY WRONG: it would
125
+ * record `low` over a run that executed at `max` — the exact defect the feature
126
+ * exists to prevent, reintroduced by the fix for it.
127
+ *
128
+ * Both directions matter, so both are handled:
129
+ * - effort DECLARED → set the var, overriding whatever the shell had.
130
+ * - effort OMITTED → DELETE an inherited var, so "omit" means the harness
131
+ * default rather than "whatever this machine happened to
132
+ * export". An omitted effort must not be a hidden input.
133
+ */
134
+ function pinEffortEnv(env, effort) {
135
+ // Rebuilt WITHOUT the key rather than deleting or assigning `undefined`:
136
+ // omission has to be provable here, and whether a spawn drops an
137
+ // `undefined`-valued env entry is a Node-version detail we should not lean on.
138
+ const { [exports.EFFORT_ENV_VAR]: _inherited, ...rest } = env;
139
+ return effort === undefined
140
+ ? rest
141
+ : { ...rest, [exports.EFFORT_ENV_VAR]: String(effort) };
142
+ }
143
+ /**
144
+ * The harness's own rejection of an `--effort` value, or null. Pure.
145
+ *
146
+ * The CLI does NOT fail on a bad level — it prints this to stderr and silently
147
+ * runs at its default. That silent substitution is precisely the bug class this
148
+ * feature addresses (a number produced by a configuration nobody asked for), so
149
+ * a rejected value must never become a sample. Matched on the binary's own
150
+ * wording, the same shape as {@link isRateLimited}.
151
+ */
152
+ function effortRejection(out) {
153
+ const text = `${out.stderr ?? ""}\n${out.stdout}`;
154
+ const m = /Unknown --effort value[^\n]*/.exec(text);
155
+ return m ? m[0].trim() : null;
156
+ }
157
+ /**
158
+ * Wrap a runner so a run the harness rejected on `--effort` FAILS LOUDLY.
159
+ *
160
+ * Applied ONCE, around the real runner, rather than as a guard repeated at each
161
+ * of the five `runner(...)` call sites — a guard per call site is the shape that
162
+ * left four of five compilers unprotected in #173.
163
+ *
164
+ * It THROWS rather than counting the trial as `runError`. A `runError` trial is
165
+ * dropped from the denominator, which is right for a transient (a rate limit) and
166
+ * wrong here: an unusable effort value is deterministic and repeatable, so every
167
+ * trial fails it and the run would report a rate computed over ZERO samples. A
168
+ * configuration mistake should stop the run and name itself.
169
+ */
170
+ function withEffortGuard(runner) {
171
+ // 🔴 DELIBERATELY NOT `async`. An `async` wrapper turns the wrapped runner's
172
+ // SYNCHRONOUS throws into rejected promises, and the real runner refuses
173
+ // synchronously on purpose — `refuseDuringEvalLoad` / `refuseUnderForeignRunner`
174
+ // stop a paid eval from billing when a foreign test runner collects it. Making
175
+ // this `async` silently downgraded those refusals from "throws at the call" to
176
+ // "returns a promise that rejects", which `assert.throws` cannot see and an
177
+ // un-awaited caller would not notice. Caught by the full suite, not by the
178
+ // targeted one; pinned below by `withEffortGuard preserves a SYNCHRONOUS throw`.
179
+ return (a) => {
180
+ const pending = runner(a);
181
+ return pending.then((out) => {
182
+ const rejection = effortRejection(out);
183
+ if (rejection !== null) {
184
+ throw new Error(`the harness rejected effort ${JSON.stringify(a.effort)}: ${rejection}`);
185
+ }
186
+ return out;
187
+ });
188
+ };
189
+ }
190
+ /**
191
+ * Build the real runner's argv. Pure and exported so the FLAGS are provable —
192
+ * `spawnAgentRaw` is `v8 ignore`d (it spawns a subprocess), so an argv assembled
193
+ * inline there could not be asserted at all. Mirrors `buildCodexArgs`.
194
+ */
195
+ function buildAgentArgs(a) {
196
+ return [
197
+ "-p",
198
+ a.task,
199
+ // stream-json (+ --verbose, required with -p) so the per-turn tool_use
200
+ // events survive into `ctx.toolCalls` — the unified Trace, same as the
201
+ // harness tier. The terminal `result` event still carries num_turns/output.
202
+ "--output-format",
203
+ "stream-json",
204
+ "--verbose",
205
+ "--model",
206
+ a.model,
207
+ ...(a.effort !== undefined ? ["--effort", String(a.effort)] : []),
208
+ "--permission-mode",
209
+ "acceptEdits",
210
+ ...(a.pluginDir !== undefined
211
+ ? ["--plugin-dir", (0, node_path_1.resolve)(a.pluginDir)]
212
+ : []),
213
+ ...(a.hasSettings ? ["--settings", "settings.json"] : []),
214
+ "--allowedTools",
215
+ ...a.tools,
216
+ ];
217
+ }
218
+ /**
219
+ * The real `claude`-spawning runner (composition root). Exported so other
220
+ * real-model entries (e.g. the `audit` trigger tier) bind the same runner.
221
+ *
222
+ * The effort guard is composed in HERE, at the single definition, rather than at
223
+ * each of the places that bind this runner — so every consumer, including ones
224
+ * not yet written, is covered by construction. Guarding each call site instead is
225
+ * the shape that left four of five compilers unprotected in #173.
226
+ */
227
+ exports.spawnAgent = withEffortGuard(spawnAgentRaw);
104
228
  /* v8 ignore start -- real claude subprocess; exercised by bench/, not the unit gate */
105
- /** The real `claude`-spawning runner (composition root). Exported so other
106
- * real-model entries (e.g. the `audit` trigger tier) bind the same runner. */
107
- function spawnAgent(a) {
229
+ /** The unguarded spawn itself; wrapped by {@link spawnAgent}, never bound raw. */
230
+ function spawnAgentRaw(a) {
108
231
  (0, eval_load_phase_js_1.refuseDuringEvalLoad)("spawning `claude`");
109
232
  (0, foreign_runner_js_1.refuseUnderForeignRunner)("spawning `claude`");
110
233
  return new Promise((resolvePromise) => {
111
- const args = [
112
- "-p",
113
- a.task,
114
- // stream-json (+ --verbose, required with -p) so the per-turn tool_use
115
- // events survive into `ctx.toolCalls` — the unified Trace, same as the
116
- // harness tier. The terminal `result` event still carries num_turns/output.
117
- "--output-format",
118
- "stream-json",
119
- "--verbose",
120
- "--model",
121
- a.model,
122
- "--permission-mode",
123
- "acceptEdits",
124
- ...(a.pluginDir !== undefined
125
- ? ["--plugin-dir", (0, node_path_1.resolve)(a.pluginDir)]
126
- : []),
127
- ...(a.hasSettings ? ["--settings", "settings.json"] : []),
128
- "--allowedTools",
129
- ...a.tools,
130
- ];
234
+ const args = buildAgentArgs(a);
131
235
  const child = (0, node_child_process_1.spawn)(runtime_js_1.claudeCodeRuntime.agentBinary, args, {
132
236
  cwd: a.cwd,
133
237
  // The security-critical env resolution (overlay vs. scrubbed replacement)
@@ -194,7 +298,7 @@ function warnUnregisteredSkillArms(arms) {
194
298
  }
195
299
  async function runEval(spec) {
196
300
  warnUnregisteredSkillArms(spec.arms);
197
- const report = await runEvalWith(spec, spawnAgent);
301
+ const report = await runEvalWith(spec, exports.spawnAgent);
198
302
  // Surface what the run spent — tokens + API-equivalent $, and a LOUD warning if
199
303
  // it was billed to a metered API key instead of the subscription. See eval-cost.ts.
200
304
  (0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromEvalReport)(report));
@@ -237,6 +341,7 @@ async function measureWith(spec, runner) {
237
341
  task: spec.task,
238
342
  trials: spec.trials ?? 5,
239
343
  model: spec.model ?? "sonnet",
344
+ effort: spec.effort,
240
345
  allowedTools: spec.allowedTools,
241
346
  timeoutMs: spec.timeoutMs,
242
347
  spacingSec: spec.spacingSec,
@@ -266,7 +371,7 @@ async function measureWith(spec, runner) {
266
371
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
267
372
  /** Score a check vocabulary across trials against the real `claude` CLI. */
268
373
  async function measure(spec) {
269
- const report = await measureWith(spec, spawnAgent);
374
+ const report = await measureWith(spec, exports.spawnAgent);
270
375
  (0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromArm)(report.usage));
271
376
  return report;
272
377
  }
@@ -283,6 +388,7 @@ async function measureArmsWith(spec, runner) {
283
388
  task: spec.task,
284
389
  trials: spec.trials ?? 5,
285
390
  model: spec.model ?? "sonnet",
391
+ effort: spec.effort,
286
392
  allowedTools: spec.allowedTools,
287
393
  timeoutMs: spec.timeoutMs,
288
394
  spacingSec: spec.spacingSec,
@@ -337,7 +443,7 @@ function stubArmPluginDirs(arms) {
337
443
  /** Score checks across arms against the real `claude` CLI. */
338
444
  async function measureArms(spec) {
339
445
  warnUnregisteredSkillArms(spec.arms);
340
- const report = await measureArmsWith(spec, spawnAgent);
446
+ const report = await measureArmsWith(spec, exports.spawnAgent);
341
447
  // Sum every arm's spend — an A/B run pays for both arms.
342
448
  (0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.sumCosts)(Object.values(report.arms).map((a) => (0, eval_cost_js_1.costFromArm)(a.usage))));
343
449
  return report;
@@ -552,6 +658,7 @@ async function runSkillSelectionTrial(args) {
552
658
  task: args.prompt,
553
659
  cwd,
554
660
  model: args.model,
661
+ effort: args.effort,
555
662
  tools: args.tools ?? ["Read", "Edit", "Write", "Bash", "Skill"],
556
663
  hasSettings: false,
557
664
  pluginDir: args.pluginDir,
@@ -637,6 +744,7 @@ async function runWithCache(runArgs, keyParts, runner, cfg) {
637
744
  const key = (0, eval_cache_js_1.cacheKey)({
638
745
  task: runArgs.task,
639
746
  model: runArgs.model,
747
+ effort: runArgs.effort,
640
748
  tools: runArgs.tools,
641
749
  files: keyParts.files,
642
750
  settings: keyParts.settings,
@@ -967,6 +1075,7 @@ async function executeTrial(spec, arm, trialIndex, runner, cfg) {
967
1075
  cwd,
968
1076
  // A model comparison is a harness A/B: an arm may override the model.
969
1077
  model: arm.model ?? cfg.model,
1078
+ effort: arm.effort ?? cfg.effort,
970
1079
  tools: cfg.tools,
971
1080
  hasSettings,
972
1081
  pluginDir: arm.pluginDir,
@@ -1100,8 +1209,16 @@ async function withEvalLock(args, produce) {
1100
1209
  }
1101
1210
  if (!isDatedModel(args.model))
1102
1211
  warnFloatingModel(args.model);
1212
+ // OVERLAP, DELIBERATE — do not delete this as dead. Effort reaches the hash
1213
+ // twice: here (the CHOKEPOINT every seam passes through, so a future seam that
1214
+ // forgets to fold effort into its own `inputs` is still covered) and inside
1215
+ // each seam's `inputs` (which alone can see a PER-ARM override this line
1216
+ // cannot). Measured 2026-09-01: removing either one alone leaves the suite
1217
+ // green; removing BOTH fails `changing effort makes a committed lock STALE`.
1218
+ // That is two populations covered, not one line duplicated.
1103
1219
  const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
1104
1220
  model: args.model,
1221
+ effort: args.effort,
1105
1222
  evalApiVersion: lock.evalApiVersion,
1106
1223
  inputs: args.inputs,
1107
1224
  });
@@ -1116,6 +1233,7 @@ async function withEvalLock(args, produce) {
1116
1233
  name: args.name,
1117
1234
  inputsHash,
1118
1235
  model: args.model,
1236
+ effort: args.effort,
1119
1237
  harnessVersionKey: harnessVersion(),
1120
1238
  evalApiVersion: lock.evalApiVersion,
1121
1239
  builtAt: new Date().toISOString(),
@@ -1166,6 +1284,10 @@ function evalArmsInputs(spec, cfg) {
1166
1284
  const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
1167
1285
  arms[name] = {
1168
1286
  model: arm.model ?? cfg.model,
1287
+ // The per-ARM half of the overlap documented at `inputsHash` — the
1288
+ // chokepoint sees only the eval-level effort, so an arm that overrides it
1289
+ // would otherwise hash identically to its sibling.
1290
+ effort: arm.effort ?? cfg.effort,
1169
1291
  tools: [...cfg.tools].sort(),
1170
1292
  files: stripPluginRoot(resolved.files, absRoot),
1171
1293
  settings: stripPluginRoot(resolved.settings, absRoot),
@@ -1203,6 +1325,7 @@ async function runEvalWith(spec, runner) {
1203
1325
  const backoffMs = spec.retryBackoffMs ?? 1000;
1204
1326
  const cfg = {
1205
1327
  model: spec.model ?? "haiku",
1328
+ effort: spec.effort,
1206
1329
  tools: spec.allowedTools ?? ["Read", "Edit", "Write", "Bash"],
1207
1330
  timeoutMs: spec.timeoutMs ?? 240000,
1208
1331
  cache: spec.cache ?? "off",
@@ -1232,7 +1355,7 @@ async function runEvalWith(spec, runner) {
1232
1355
  // resolveHarness/hashDir work). `check` replays the committed report below
1233
1356
  // without ever entering the run pool — so no model is driven in CI.
1234
1357
  const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
1235
- return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
1358
+ return withEvalLock({ name: spec.name, inputs, model: cfg.model, effort: cfg.effort, lock }, async () => {
1236
1359
  const results = await runPool(units, concurrency, worker);
1237
1360
  const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
1238
1361
  return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
@@ -1285,7 +1408,7 @@ function formatEvalReport(report) {
1285
1408
  * The asymmetry reflects default-vs-injected, not a hexagonal violation.
1286
1409
  */
1287
1410
  exports.claudeEvalDriver = {
1288
- runner: spawnAgent,
1411
+ runner: exports.spawnAgent,
1289
1412
  parse: parseClaudeRun,
1290
1413
  harness: "claude-code",
1291
1414
  };
@@ -1591,6 +1714,7 @@ async function runTriggerTrial(prompt, cfg, runner) {
1591
1714
  task: prompt,
1592
1715
  cwd,
1593
1716
  model: cfg.model,
1717
+ effort: cfg.effort,
1594
1718
  tools: cfg.tools,
1595
1719
  hasSettings: false,
1596
1720
  pluginDir: cfg.pluginDir,
@@ -1690,6 +1814,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1690
1814
  // Sonnet, not haiku: trigger-rate is a selection measurement and haiku
1691
1815
  // under-selects, producing false-negative recall (see TriggerRateSpec.model).
1692
1816
  model,
1817
+ effort: spec.effort,
1693
1818
  tools: spec.allowedTools ?? ["Read", "Edit", "Write", "Bash", "Skill"],
1694
1819
  timeoutMs: spec.timeoutMs ?? 240000,
1695
1820
  spacing: (spec.spacingSec ?? 4) * 1000,
@@ -1715,6 +1840,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1715
1840
  ? [...spec.irrelevantPrompts]
1716
1841
  : undefined,
1717
1842
  model: cfg.model,
1843
+ effort: cfg.effort,
1718
1844
  tools: [...cfg.tools].sort(),
1719
1845
  fixture: spec.fixture,
1720
1846
  competitors,
@@ -1723,7 +1849,13 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1723
1849
  // STALE if the eval is switched to another harness.
1724
1850
  harness,
1725
1851
  };
1726
- return await withEvalLock({ name: spec.name, inputs: triggerInputs, model: cfg.model, lock }, async () => {
1852
+ return await withEvalLock({
1853
+ name: spec.name,
1854
+ inputs: triggerInputs,
1855
+ model: cfg.model,
1856
+ effort: cfg.effort,
1857
+ lock,
1858
+ }, async () => {
1727
1859
  const relevant = await runTriggerSet(spec.prompts, cfg, runner);
1728
1860
  const base = {
1729
1861
  rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
@@ -0,0 +1,25 @@
1
+ import type { IgnoreLike } from "glob";
2
+ /** Always excluded, whatever the config says. Root-relative, like `exclude`. */
3
+ export declare const EXCLUDE_FLOOR: readonly string[];
4
+ /** The parsed `.vigilesrc.json#exclude`, in the forms a walk consumes. */
5
+ export interface ExcludeSet {
6
+ /** Absolute repo root every pattern is relative to. */
7
+ readonly root: string;
8
+ /** The user's patterns, as written (for messages). */
9
+ readonly patterns: readonly string[];
10
+ /**
11
+ * The string-list face for a glob rooted AT `root`: the floor, then each user
12
+ * pattern normalized so a bare directory name excludes its subtree (`bench` →
13
+ * `bench`, `bench/**`), which is what "tsconfig-style" promises.
14
+ */
15
+ readonly ignore: readonly string[];
16
+ /** The function face for `globSync`, correct whatever the glob's `cwd` is. */
17
+ readonly globIgnore: IgnoreLike;
18
+ /** Is this root-relative path excluded (floor or user pattern)? */
19
+ matches(rel: string): boolean;
20
+ /** The pattern that excludes this root-relative path, or null when none does. */
21
+ explain(rel: string): string | null;
22
+ }
23
+ /** Parse `.vigilesrc.json#exclude` once, against `root`. */
24
+ export declare function excludeSet(root: string, patterns: readonly string[] | undefined): ExcludeSet;
25
+ //# sourceMappingURL=exclude.d.ts.map
@@ -0,0 +1,132 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.EXCLUDE_FLOOR = void 0;
4
+ exports.excludeSet = excludeSet;
5
+ /**
6
+ * The ONE exclusion policy for every walk that polices the user's repository.
7
+ *
8
+ * `.vigilesrc.json#exclude` is documented as "tsconfig-style" — a list of paths or
9
+ * globs the repo's own lint should not police (vendored corpora, benchmark
10
+ * fixtures, frozen reproductions). Issue #192: it was honoured by three walks,
11
+ * ignored by six, and the three that honoured it did not agree with each other.
12
+ *
13
+ * 🔴 TWO DIALECTS WERE ALREADY IN `main`, AND THEY DISAGREED ON `"bench"`.
14
+ * `glob`'s string `ignore` treats a bare directory name as a file pattern:
15
+ * measured 2026-09-03 on glob 13, `ignore: ["bench"]` and `["bench/"]` exclude
16
+ * NOTHING, only `["bench/**"]` works. The minimatch helper inside
17
+ * `discoverNestedBundles` accepted the bare name. tsc and ESLint both treat the
18
+ * bare name as the directory (measured the same day). So a user who wrote the
19
+ * key the way its JSDoc promised got the nested-bundle pass filtered and every
20
+ * glob-backed pass unfiltered — with no way to tell from the output.
21
+ *
22
+ * This module is the fix in the shape the fence-parser fix took
23
+ * (`core/markdown.ts`): not one WALK — the walks legitimately have different
24
+ * scopes — but ONE PREDICATE, parsed once from config, that every walk consumes.
25
+ * It offers the predicate in the two forms a walk needs and nothing else:
26
+ *
27
+ * - `globIgnore` — an `IgnoreLike` for `globSync`, keyed on the path's position
28
+ * RELATIVE TO THE REPO ROOT, so a glob rooted below the root (`vigiles lint
29
+ * some/dir`) still applies a root-relative `exclude` correctly. The string
30
+ * list could not: `bench/**` relative to `some/dir` matches nothing.
31
+ * - `ignore` — the normalized string list, for pure detectors in `core/` that
32
+ * take an ignore list by injection and glob from the repo root themselves.
33
+ * - `matches(rel)` / `explain(rel)` — for `readdirSync`-style walks and for the
34
+ * one line printed when an explicitly named path is processed anyway.
35
+ *
36
+ * 🔴 EVERY IN-SCOPE WALK TAKES AN `ExcludeSet` AS A REQUIRED PARAMETER. The
37
+ * shape that let `findSpecs` go a year without the key was
38
+ * `exclude: readonly string[] = []` — optional-with-default lets a call site
39
+ * forget, and a forgotten argument is indistinguishable from an empty config.
40
+ * Do not reintroduce an optional `ExcludeSet` anywhere in scope.
41
+ *
42
+ * EXPLICITLY NAMED PATHS WIN, LOUDLY. Four tools were measured (2026-09-03):
43
+ * ripgrep and tsc process an explicitly named ignored file silently; ESLint
44
+ * skips it with a warning; prettier skips it and prints "All matched files use
45
+ * Prettier code style!" — the silent no-op this repo has burned itself on. vigiles
46
+ * takes the rg/tsc semantics (an argument is an instruction) with ESLint's
47
+ * loudness: the path is processed and ONE line says which pattern it matched.
48
+ * `exclude` filters DISCOVERY, never an argument.
49
+ *
50
+ * WHAT DELIBERATELY DOES NOT GO THROUGH THIS MODULE — so the next reader does not
51
+ * "fix" it (the classification is issue #192's comment):
52
+ *
53
+ * - `core/compile.ts#validateGlobRef` — verifies a spec's own `glob()` reference
54
+ * resolves to ≥1 file. That is reference verification of the user's claim
55
+ * about the repo, not lint scope; excluding a dir must not make a true ref
56
+ * false.
57
+ * - `cli.ts#specReferencedElsewhere` — eject's "is this spec compiled anywhere
58
+ * else" safety check. Wider is safer: a target under an excluded dir is still
59
+ * a target that would be orphaned.
60
+ * - `core/validate.ts#expandGlobs` — expands a pattern the user TYPED. An
61
+ * argument wins (see above); it is not discovery.
62
+ * - Surface walks INSIDE a bundle (`plugin-loader.ts#readTree`,
63
+ * `skill-reachability.ts`): `exclude` applies at BUNDLE granularity via
64
+ * `discoverNestedBundles`; a skill inside your own `skills/` is yours.
65
+ * - Internal machinery that never enumerates the user's repo as lint surface:
66
+ * eval temp installs (`eval.ts`, `adapters/codex/eval.ts`), the eval cache and
67
+ * locks (`eval-cache.ts`, `eval-lock.ts`, `run-script.ts#snapshotTree`),
68
+ * `.vigiles/hooks/` discovery (`hook-install.ts`, a fixed dir), sidecars
69
+ * (`core/sidecar.ts`), linter catalogs / rulesDirs / toolchain paths
70
+ * (`core/linters.ts`, `core/generate-schema.ts`, `core/generate-types.ts`),
71
+ * `init`'s shallow adoptable-surface sweep (`cli.ts#discoverAdoptableSurfaces`)
72
+ * and lint-config collection (`cli.ts#safeReaddir`), and
73
+ * `core/generate-harness.ts` (an explicit, non-recursive dir argument).
74
+ *
75
+ * The floor (`node_modules`, `dist`, `.git`, `.vigiles`) lives here too, so a
76
+ * walk cannot carry its own private copy of it — the original `findSpecs` list
77
+ * lacked `.vigiles/**` while the three `core/` detectors had it.
78
+ */
79
+ const minimatch_1 = require("minimatch");
80
+ const node_path_1 = require("node:path");
81
+ /** Always excluded, whatever the config says. Root-relative, like `exclude`. */
82
+ exports.EXCLUDE_FLOOR = [
83
+ "node_modules/**",
84
+ "dist/**",
85
+ ".git/**",
86
+ ".vigiles/**",
87
+ ];
88
+ /** `a\b\c` → `a/b/c`, drop a leading `./`, drop a trailing `/`. */
89
+ function normalizeRel(rel) {
90
+ let r = rel
91
+ .split(node_path_1.sep)
92
+ .join("/")
93
+ .replace(/^(?:\.\/)+/, "");
94
+ while (r.endsWith("/"))
95
+ r = r.slice(0, -1);
96
+ return r;
97
+ }
98
+ /** Parse `.vigilesrc.json#exclude` once, against `root`. */
99
+ function excludeSet(root, patterns) {
100
+ const user = (patterns ?? []).map(normalizeRel).filter((p) => p !== "");
101
+ const all = [...exports.EXCLUDE_FLOOR, ...user];
102
+ // One compiled matcher pair per pattern: the pattern itself and its subtree.
103
+ const compiled = all.map((p) => ({
104
+ pattern: p,
105
+ self: new minimatch_1.Minimatch(p, { dot: true }),
106
+ subtree: new minimatch_1.Minimatch(`${p}/**`, { dot: true }),
107
+ }));
108
+ const explain = (rel) => {
109
+ const r = normalizeRel(rel);
110
+ if (r === "" || r === "." || r.startsWith("../"))
111
+ return null;
112
+ for (const c of compiled) {
113
+ if (c.self.match(r) || c.subtree.match(r))
114
+ return c.pattern;
115
+ }
116
+ return null;
117
+ };
118
+ const matches = (rel) => explain(rel) !== null;
119
+ const relOf = (p) => normalizeRel((0, node_path_1.relative)(root, p.fullpath()));
120
+ return {
121
+ root,
122
+ patterns: user,
123
+ ignore: [...exports.EXCLUDE_FLOOR, ...user.flatMap((p) => [p, `${p}/**`])],
124
+ globIgnore: {
125
+ ignored: (p) => matches(relOf(p)),
126
+ childrenIgnored: (p) => matches(relOf(p)),
127
+ },
128
+ matches,
129
+ explain,
130
+ };
131
+ }
132
+ //# sourceMappingURL=exclude.js.map
@@ -1,27 +1,3 @@
1
- /**
2
- * Guardrail verification — "prove your safety hook ACTUALLY blocks."
3
- *
4
- * The #1 verified Claude Code hook pain is FALSE CONFIDENCE: a developer ships a
5
- * PreToolUse safety hook, believes they're protected, and finds out otherwise only
6
- * when the agent force-pushes to main. The failure is silent — exit 1 instead of
7
- * exit 2, the wrong JSON field, PostToolUse-can't-block, a wrong `jq` path, a missing
8
- * `chmod +x` — all produce a hook that LOOKS like a guard and enforces nothing, with
9
- * no error. (Crosley: "three different teams believed they had blocked force pushes";
10
- * RFC #45427, closed not-planned. Full corpus: research/hook-pain-points.md.)
11
- *
12
- * This is the deterministic answer: feed a curated **disaster event** (`git push
13
- * --force`, `rm -rf /`, `git commit --no-verify`, `cat ~/.ssh/*`, `curl … | sh`) to
14
- * the hook via {@link runHook} and check the normalized decision is BLOCK. No model,
15
- * no API key, runs in CI, works on a hand-written hook with NO vigiles spec — it
16
- * verifies the hook's decision LOGIC, so it sidesteps CC's runtime delivery bugs
17
- * (the model routing around a tool entirely, #45427 / #32376) which it deliberately
18
- * does NOT claim to fix. (#34692, the old subagent-delivery gap, is fixed as of CC
19
- * 2.1.241 — see src/subagent-delivery.test.ts.)
20
- *
21
- * Pure-ish (wraps the existing runHook tier). The catalog is harness-neutral data;
22
- * the scaffold-test generator emits a test that calls these, and the same engine
23
- * backs an informational coverage report.
24
- */
25
1
  import { type RunHookOptions } from "./run-hook.js";
26
2
  /** A category of dangerous action a guard might be meant to block. */
27
3
  export type DisasterCategory = "destructive-git" | "destructive-fs" | "bypass-verification" | "secret-exfiltration" | "remote-code";
@@ -61,6 +37,49 @@ export interface VerifyGuardrailOptions extends RunHookOptions {
61
37
  /** The PreToolUse event name to wrap each disaster in (default "PreToolUse"). */
62
38
  readonly event?: string;
63
39
  }
40
+ /**
41
+ * The same dangerous commands, spelled the other ways a shell reads identically.
42
+ *
43
+ * Takes a battery of hook test cases (shell commands wrapped as `PreToolUse`
44
+ * events — `DISASTER_CATALOG` is the shipped one) and returns MORE test cases:
45
+ * every command re-spelled with a quoted flag (`git push "--force"`), the short
46
+ * form of a flag (`-f`), an absolute or escaped head (`/usr/bin/git`, `\git`), or a
47
+ * pass-through wrapper (`sudo …`, `env …`). The shell runs each rewrite exactly as
48
+ * it runs the original. Feed them to `assertBlocksDisasters` alongside the
49
+ * originals:
50
+ *
51
+ * assertBlocksDisasters(hook, {
52
+ * events: [...DISASTER_CATALOG, ...experimental_alternateSpellings(DISASTER_CATALOG)],
53
+ * });
54
+ *
55
+ * WHAT BREAKS WITHOUT IT. A guard whose rule is "the command contains `--force`"
56
+ * blocks all seven catalog commands, so the battery is green — and lets
57
+ * `git push "--force"` through, because the quotes make it a different string.
58
+ * Measured 2026-09-02 on the shipped dogfood guard BEFORE this existed: 7/7
59
+ * originals blocked, **8 of 30** hand-written re-spellings blocked.
60
+ *
61
+ * WHY THE OUTPUT IS TRUSTWORTHY. Nothing new is judged. "Dangerous" is inherited
62
+ * from the original a human put in the battery; "the same command" is decided by
63
+ * the shell parser vigiles already uses for `runs()`/`touches()` (see
64
+ * {@link sameOperation}). A rewrite that fails that check throws rather than being
65
+ * emitted, so the battery can never quietly shrink.
66
+ *
67
+ * It returns ONLY the rewrites (never the originals), so pass the originals
68
+ * alongside as above. Each rewrite keeps its original's `tool` and `category`
69
+ * and takes the original's id with an index suffix (`force-push~4`), so a report
70
+ * names which spelling got through.
71
+ *
72
+ * In promptfoo's vocabulary each rewrite rule here is a "strategy"; the difference
73
+ * is that promptfoo's encodings (base64, leetspeak) may or may not be decoded by the
74
+ * target, whereas every rewrite here is one the shell provably executes identically
75
+ * — a miss is a guard bug, never an ambiguous input.
76
+ *
77
+ * @experimental Days old with a single consumer (this repo's own dogfood) and no
78
+ * external use. The set of rewrite rules and the id-suffix shape are the parts most
79
+ * likely to move; the prefix says so at every call site, which an import line or a
80
+ * doc note cannot.
81
+ */
82
+ export declare function experimental_alternateSpellings(events: readonly DisasterEvent[]): readonly DisasterEvent[];
64
83
  /**
65
84
  * Run a hook command against the disaster battery and report which events it blocks.
66
85
  * `hookCommand` is the exact shell the hook registers (e.g. `bash hooks/guard.sh` or
@@ -1,6 +1,7 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.DISASTER_CATALOG = void 0;
4
+ exports.experimental_alternateSpellings = experimental_alternateSpellings;
4
5
  exports.verifyGuardrail = verifyGuardrail;
5
6
  exports.unblockedDisasters = unblockedDisasters;
6
7
  exports.assertBlocksDisasters = assertBlocksDisasters;
@@ -29,6 +30,7 @@ exports.formatGuardrailReport = formatGuardrailReport;
29
30
  * the scaffold-test generator emits a test that calls these, and the same engine
30
31
  * backs an informational coverage report.
31
32
  */
33
+ const bash_equivalents_js_1 = require("./core/bash-equivalents.js");
32
34
  const run_hook_js_1 = require("./run-hook.js");
33
35
  /**
34
36
  * The curated battery. Deliberately small and high-signal: each is a textbook
@@ -99,6 +101,61 @@ function selectEvents(opts) {
99
101
  }
100
102
  return exports.DISASTER_CATALOG;
101
103
  }
104
+ /**
105
+ * The same dangerous commands, spelled the other ways a shell reads identically.
106
+ *
107
+ * Takes a battery of hook test cases (shell commands wrapped as `PreToolUse`
108
+ * events — `DISASTER_CATALOG` is the shipped one) and returns MORE test cases:
109
+ * every command re-spelled with a quoted flag (`git push "--force"`), the short
110
+ * form of a flag (`-f`), an absolute or escaped head (`/usr/bin/git`, `\git`), or a
111
+ * pass-through wrapper (`sudo …`, `env …`). The shell runs each rewrite exactly as
112
+ * it runs the original. Feed them to `assertBlocksDisasters` alongside the
113
+ * originals:
114
+ *
115
+ * assertBlocksDisasters(hook, {
116
+ * events: [...DISASTER_CATALOG, ...experimental_alternateSpellings(DISASTER_CATALOG)],
117
+ * });
118
+ *
119
+ * WHAT BREAKS WITHOUT IT. A guard whose rule is "the command contains `--force`"
120
+ * blocks all seven catalog commands, so the battery is green — and lets
121
+ * `git push "--force"` through, because the quotes make it a different string.
122
+ * Measured 2026-09-02 on the shipped dogfood guard BEFORE this existed: 7/7
123
+ * originals blocked, **8 of 30** hand-written re-spellings blocked.
124
+ *
125
+ * WHY THE OUTPUT IS TRUSTWORTHY. Nothing new is judged. "Dangerous" is inherited
126
+ * from the original a human put in the battery; "the same command" is decided by
127
+ * the shell parser vigiles already uses for `runs()`/`touches()` (see
128
+ * {@link sameOperation}). A rewrite that fails that check throws rather than being
129
+ * emitted, so the battery can never quietly shrink.
130
+ *
131
+ * It returns ONLY the rewrites (never the originals), so pass the originals
132
+ * alongside as above. Each rewrite keeps its original's `tool` and `category`
133
+ * and takes the original's id with an index suffix (`force-push~4`), so a report
134
+ * names which spelling got through.
135
+ *
136
+ * In promptfoo's vocabulary each rewrite rule here is a "strategy"; the difference
137
+ * is that promptfoo's encodings (base64, leetspeak) may or may not be decoded by the
138
+ * target, whereas every rewrite here is one the shell provably executes identically
139
+ * — a miss is a guard bug, never an ambiguous input.
140
+ *
141
+ * @experimental Days old with a single consumer (this repo's own dogfood) and no
142
+ * external use. The set of rewrite rules and the id-suffix shape are the parts most
143
+ * likely to move; the prefix says so at every call site, which an import line or a
144
+ * doc note cannot.
145
+ */
146
+ function experimental_alternateSpellings(events) {
147
+ return events.flatMap((event) => {
148
+ const command = event.input["command"];
149
+ if (typeof command !== "string")
150
+ return [];
151
+ return (0, bash_equivalents_js_1.equivalentCommands)(command).map((variant, i) => ({
152
+ ...event,
153
+ id: `${event.id}~${String(i + 1)}`,
154
+ label: `${event.label} — spelled: ${variant}`,
155
+ input: { ...event.input, command: variant },
156
+ }));
157
+ });
158
+ }
102
159
  /**
103
160
  * Run a hook command against the disaster battery and report which events it blocks.
104
161
  * `hookCommand` is the exact shell the hook registers (e.g. `bash hooks/guard.sh` or
@@ -89,6 +89,16 @@ export interface SelectionOptions {
89
89
  readonly trials?: number;
90
90
  /** Selector model — defaults to Sonnet (a weaker model under-selects). */
91
91
  readonly model?: string;
92
+ /**
93
+ * Reasoning budget (`claude --effort`) for the selector, or undefined for the
94
+ * harness default. Present for {@link measureSelectionMatrix}, the ASSERTABLE
95
+ * test primitive, where the configuration a number came from has to be pinnable.
96
+ *
97
+ * Deliberately NOT exposed as an `audit` CLI flag: `audit` is a local report,
98
+ * not a reproducibility surface, and a flag nobody can act on is surface without
99
+ * a use. The audit probe therefore leaves this unset and runs at the default.
100
+ */
101
+ readonly effort?: string | number;
92
102
  /** Parallel runs across the prompts × trials grid (default 1). */
93
103
  readonly concurrency?: number;
94
104
  /** Which harness drives it (default `"claude-code"`; others report n/a). */