vigiles 11.0.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/eval.d.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import { parseToolCalls, parseHooks, parseSubagents, type ToolCall, type Trace } from "./harness-test.js";
2
2
  import { type CacheMode } from "./eval-cache.js";
3
+ import { type EvalLockOptions } from "./eval-lock.js";
3
4
  import type { Check, CheckJSON } from "./check.js";
4
5
  import { type Comparison } from "./stats.js";
5
6
  import { type ToolIntercept } from "./tool-intercept.js";
@@ -159,6 +160,16 @@ export interface EvalSpec<M extends Metrics> {
159
160
  * to today). See {@link ToolStub} and `research/eval-coverage-and-isolation.md`.
160
161
  */
161
162
  readonly stubs?: readonly ToolStub[];
163
+ /**
164
+ * **The eval LOCK** — a committed staleness gate for CI (see `src/eval-lock.ts`).
165
+ * With a `name` set, `vigiles eval --update` (local, on your subscription)
166
+ * records the report to `.vigiles/eval-locks/<name>.lock.json`; `--check` (CI)
167
+ * verifies the committed result against the current inputs WITHOUT a model call,
168
+ * failing "stale" when they diverge. Mode normally comes from the CLI; set
169
+ * `lock` to drive it programmatically or point a test at a throwaway dir. The
170
+ * lock engages only when `name` is set (it keys the lock file).
171
+ */
172
+ readonly lock?: EvalLockOptions;
162
173
  }
163
174
  /** Per-metric summary statistics across an arm's runs. */
164
175
  export interface MetricStat {
@@ -479,16 +490,6 @@ export declare function modelTier(id: string): number | null;
479
490
  * {@link modelTier}); an unrankable model/floor is never "below" (fail-open).
480
491
  */
481
492
  export declare function belowModelFloor(model: string, floor: string): boolean;
482
- /**
483
- * Reduce a raw `--version` string to the **major.minor** cache-key token. We key
484
- * the cache on major.minor, NOT the patch: a patch release rarely changes agent
485
- * behaviour, so keying patches would churn the cache on every release for no
486
- * signal; a minor/major bump is where the system prompt / tool defs actually move.
487
- * (If a specific patch is known to matter, clear the cache or bump
488
- * `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
489
- * found. Pure + tested.
490
- */
491
- export declare function harnessVersionKey(raw: string): string;
492
493
  /**
493
494
  * Build an **ephemeral run environment** for a model-driven run: a NEW env object
494
495
  * with a *fresh* `HOME` (and `TMPDIR`) pointed at the throwaway `opts.home`, only
@@ -542,15 +543,6 @@ export declare function seedEphemeralHome(throwawayHome: string, realHome: strin
542
543
  export declare function isRateLimited(out: RunOut): boolean;
543
544
  /** Map `worker` over `items` with at most `concurrency` in flight, order preserved. */
544
545
  export declare function runPool<T, R>(items: readonly T[], concurrency: number, worker: (item: T) => Promise<R>): Promise<R[]>;
545
- /**
546
- * The eval orchestration — every arm × trial via `runner`, run through the cache
547
- * and a rate-limit retry, with at most `concurrency` in flight and an optional
548
- * `maxCostUsd` budget cap; metric + usage computed per run and aggregated per
549
- * arm. Exported with an injectable `runner` so the loop, `measure` context,
550
- * caching, pooling, and aggregation are unit-testable without spawning a model
551
- * (pass a fake returning canned stream-json). `runEval` is this with the real
552
- * agent runner.
553
- */
554
546
  export declare function runEvalWith<M extends Metrics>(spec: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
555
547
  /** Format an eval report as a compact table for the console (mean ± se, pass^k). */
556
548
  export declare function formatEvalReport(report: EvalReport): string;
@@ -563,6 +555,19 @@ export declare function formatEvalReport(report: EvalReport): string;
563
555
  * (reuse the bare predicates, e.g. `(t) => skillResolved(t, "x:y")`).
564
556
  */
565
557
  export interface TriggerRateSpec {
558
+ /**
559
+ * A stable name for this trigger eval — required to engage the {@link lock}
560
+ * (it keys the committed `.vigiles/eval-locks/<name>.lock.json`). Two evals over
561
+ * the same skill with different prompts get distinct names. Optional otherwise.
562
+ */
563
+ readonly name?: string;
564
+ /**
565
+ * **The eval LOCK** — the CI staleness gate (see `src/eval-lock.ts`). With
566
+ * `name` set, `vigiles eval --update` records this trigger-rate report locally
567
+ * and `--check` verifies it against the current inputs (skill contents, prompts,
568
+ * model) with NO model call. Mode normally comes from the CLI flags.
569
+ */
570
+ readonly lock?: EvalLockOptions;
566
571
  /**
567
572
  * Plugin dir installed natively (`--plugin-dir`) so its skills/commands
568
573
  * activate. Provide this OR {@link skillsDir}, not both.
@@ -722,6 +727,14 @@ export interface EvalDriver {
722
727
  readonly runner: AgentRunner;
723
728
  readonly parse: ModelOutputParser;
724
729
  readonly runError?: (out: RunOut) => string | null;
730
+ /**
731
+ * The harness this driver runs (e.g. `"claude-code"`, `"codex"`). Folded into a
732
+ * trigger-rate eval's LOCK hash so a report recorded on one harness is marked
733
+ * STALE if the eval is later switched to another (a different harness can fire a
734
+ * skill differently). Optional for back-compat — absent defaults to
735
+ * `"claude-code"`, so an existing single-harness lock is unaffected.
736
+ */
737
+ readonly harness?: string;
725
738
  }
726
739
  /** The default (Claude Code) eval driver: real `claude` + stream-json parsing. */
727
740
  export declare const claudeEvalDriver: EvalDriver;
@@ -801,7 +814,7 @@ export declare function packageInstallSet(opts: {
801
814
  dir: string;
802
815
  added: number;
803
816
  };
804
- export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null): Promise<TriggerRateReport>;
817
+ export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null, harness?: string): Promise<TriggerRateReport>;
805
818
  /**
806
819
  * Measure a skill/behaviour's real trigger rate across prompts × trials. Defaults
807
820
  * to the real `claude` CLI (`claudeEvalDriver`); pass `{ evalDriver }` to drive a
package/dist/eval.js CHANGED
@@ -22,7 +22,6 @@ exports.aggregateUsage = aggregateUsage;
22
22
  exports.isDatedModel = isDatedModel;
23
23
  exports.modelTier = modelTier;
24
24
  exports.belowModelFloor = belowModelFloor;
25
- exports.harnessVersionKey = harnessVersionKey;
26
25
  exports.ephemeralRunEnv = ephemeralRunEnv;
27
26
  exports.seedEphemeralHome = seedEphemeralHome;
28
27
  exports.isRateLimited = isRateLimited;
@@ -71,6 +70,7 @@ const runtime_js_1 = require("./adapters/claude-code/runtime.js");
71
70
  const proofs_js_1 = require("./core/proofs.js");
72
71
  const harness_test_js_1 = require("./harness-test.js");
73
72
  const eval_cache_js_1 = require("./eval-cache.js");
73
+ const eval_lock_js_1 = require("./eval-lock.js");
74
74
  const stats_js_1 = require("./stats.js");
75
75
  const tool_intercept_js_1 = require("./tool-intercept.js");
76
76
  const tool_stub_js_1 = require("./tool-stub.js");
@@ -640,32 +640,22 @@ function belowModelFloor(model, floor) {
640
640
  const f = modelTier(floor);
641
641
  return m !== null && f !== null && m < f;
642
642
  }
643
- /**
644
- * Reduce a raw `--version` string to the **major.minor** cache-key token. We key
645
- * the cache on major.minor, NOT the patch: a patch release rarely changes agent
646
- * behaviour, so keying patches would churn the cache on every release for no
647
- * signal; a minor/major bump is where the system prompt / tool defs actually move.
648
- * (If a specific patch is known to matter, clear the cache or bump
649
- * `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
650
- * found. Pure + tested.
651
- */
652
- function harnessVersionKey(raw) {
653
- const m = /(\d+)\.(\d+)\.\d+/.exec(raw);
654
- return m ? `${m[1]}.${m[2]}` : raw.trim();
655
- }
656
643
  /* v8 ignore start -- spawns the real harness binary; memoized, cache-path only */
657
644
  let cachedHarnessVersion;
658
645
  /**
659
- * The harness binary version (`claude --version`) reduced to major.minor, for the
660
- * cache key — so a CLI **minor/major** upgrade (new system prompt / tool defs)
661
- * invalidates a stale replay, while patches don't churn it. Memoized (one spawn
662
- * per process), resolved only on the cache path, "unknown" if the binary isn't
663
- * found (then it doesn't partition the key).
646
+ * The harness binary version (`claude --version`), reduced to its
647
+ * behaviorally-significant token by the runtime port's
648
+ * {@link HarnessRuntime.versionKey} — so a Claude Code **minor/major** upgrade
649
+ * (new system prompt / tool defs) invalidates a stale replay while patches don't
650
+ * churn it. The reduction is per-harness (CC → `major.minor`, Codex → `""`) and
651
+ * lives on the adapter, not here. Memoized (one spawn per process), resolved only
652
+ * on the cache path, "unknown" if the binary isn't found (then it doesn't
653
+ * partition the key).
664
654
  */
665
655
  function harnessVersion() {
666
656
  if (cachedHarnessVersion === undefined) {
667
657
  try {
668
- cachedHarnessVersion = harnessVersionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
658
+ cachedHarnessVersion = runtime_js_1.claudeCodeRuntime.versionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
669
659
  encoding: "utf-8",
670
660
  stdio: ["ignore", "pipe", "ignore"],
671
661
  }));
@@ -983,15 +973,141 @@ function aggregateArms(armNames, results) {
983
973
  }
984
974
  return { arms, totalCostUsd };
985
975
  }
976
+ function resolveLock(over) {
977
+ return {
978
+ mode: over?.mode ?? (0, eval_lock_js_1.lockModeFromEnv)(),
979
+ dir: over?.dir ?? (0, node_path_1.resolve)(process.cwd(), eval_lock_js_1.DEFAULT_LOCK_DIR),
980
+ evalApiVersion: over?.evalApiVersion ?? (0, eval_lock_js_1.evalApiVersionFromEnv)(),
981
+ };
982
+ }
983
+ /** Emit a vigiles message (GitHub annotation under Actions, else stderr/stdout). */
984
+ function emitLockMessage(msg, warn) {
985
+ if (process.env.GITHUB_ACTIONS)
986
+ console.log(`::${warn ? "warning" : "notice"}::${msg}`);
987
+ else if (warn)
988
+ console.warn(msg);
989
+ else
990
+ console.log(msg);
991
+ }
992
+ /**
993
+ * Run a named eval through the lock. `off` → just `produce()`. `check` → replay a
994
+ * matching committed lock (NO model call) or throw "stale". `update` → `produce()`,
995
+ * write the lock, print the human-facing delta. The model is driven ONLY on the
996
+ * run path — never on a clean `check`, which is what keeps the CI gate binary-free.
997
+ * A lock-on run with no `name` is a LOUD skip (the lock needs a name to key the file).
998
+ */
999
+ async function withEvalLock(args, produce) {
1000
+ const { lock } = args;
1001
+ if (lock.mode === "off")
1002
+ return produce();
1003
+ if (!args.name) {
1004
+ // An unnamed eval can't be keyed to a lock. In `check` (the CI gate) running
1005
+ // it would call the model — violating the no-model-in-CI contract and failing
1006
+ // for missing auth — so FAIL LOUDLY instead of silently hitting the model.
1007
+ // `update` (local, model available) keeps producing: the eval just isn't gated.
1008
+ if (lock.mode === "check") {
1009
+ throw new Error(`vigiles eval --check: an unnamed eval cannot run in CI — it has no ` +
1010
+ `committed lock to replay, and running it would call the model. Add a ` +
1011
+ `\`name\` to the spec to gate it, or exclude it from the --check run.`);
1012
+ }
1013
+ emitLockMessage(`vigiles eval --${lock.mode}: skipped the lock for an unnamed eval — set ` +
1014
+ `\`name\` on the spec to enable the staleness gate for it.`, true);
1015
+ return produce();
1016
+ }
1017
+ if (!isDatedModel(args.model))
1018
+ warnFloatingModel(args.model);
1019
+ const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
1020
+ model: args.model,
1021
+ evalApiVersion: lock.evalApiVersion,
1022
+ inputs: args.inputs,
1023
+ });
1024
+ const existing = (0, eval_lock_js_1.readLock)(lock.dir, args.name);
1025
+ const decision = (0, eval_lock_js_1.decideLock)(lock.mode, args.name, inputsHash, existing);
1026
+ if (decision.kind === "stale")
1027
+ throw new Error(decision.reason);
1028
+ if (decision.kind === "replay")
1029
+ return decision.report;
1030
+ const report = await produce();
1031
+ const builtLock = (0, eval_lock_js_1.buildLock)({
1032
+ name: args.name,
1033
+ inputsHash,
1034
+ model: args.model,
1035
+ harnessVersionKey: harnessVersion(),
1036
+ evalApiVersion: lock.evalApiVersion,
1037
+ builtAt: new Date().toISOString(),
1038
+ report,
1039
+ });
1040
+ const deltas = existing ? (0, eval_lock_js_1.diffReportNumbers)(existing.report, report) : [];
1041
+ (0, eval_lock_js_1.writeLock)(lock.dir, builtLock);
1042
+ emitLockMessage((0, eval_lock_js_1.formatLockUpdate)(args.name, deltas, existing === null), false);
1043
+ return report;
1044
+ }
1045
+ /**
1046
+ * Strip the machine-specific plugin-root prefix from a resolved value before it
1047
+ * enters the lock hash. `resolveHarness` expands `${PLUGIN_ROOT}` in a plugin's
1048
+ * hook commands to the checkout's ABSOLUTE path, so a lock recorded at
1049
+ * `/home/dev/...` would be falsely STALE when `--check` recomputes it at
1050
+ * `/home/runner/...` in CI (or any other machine). Normalizing the prefix back to
1051
+ * a token makes the hash location-independent. No-op when there's no plugin root.
1052
+ */
1053
+ function stripPluginRoot(value, absRoot) {
1054
+ if (absRoot === "")
1055
+ return value;
1056
+ const json = JSON.stringify(value);
1057
+ // A hookless plugin resolves to `settings: undefined`, which `JSON.stringify`
1058
+ // returns as `undefined` (not a string) — pass it through rather than `.split`
1059
+ // a non-string (which would throw before the eval can run).
1060
+ if (json === undefined)
1061
+ return value;
1062
+ return JSON.parse(json.split(absRoot).join("${PLUGIN_ROOT}"));
1063
+ }
986
1064
  /**
987
- * The eval orchestration — every arm × trial via `runner`, run through the cache
988
- * and a rate-limit retry, with at most `concurrency` in flight and an optional
989
- * `maxCostUsd` budget cap; metric + usage computed per run and aggregated per
990
- * arm. Exported with an injectable `runner` so the loop, `measure` context,
991
- * caching, pooling, and aggregation are unit-testable without spawning a model
992
- * (pass a fake returning canned stream-json). `runEval` is this with the real
993
- * agent runner.
1065
+ * Resolve each arm's model-affecting inputs into a canonical object for the lock
1066
+ * hash — WITHOUT running the model (it reads files + hashes plugin dirs only). The
1067
+ * trial count is excluded (a sample-size knob, not a behavior input); `measure` is
1068
+ * excluded by design (the script re-asserts against the replayed report).
994
1069
  */
1070
+ function evalArmsInputs(spec, cfg) {
1071
+ const arms = {};
1072
+ for (const [name, arm] of Object.entries(spec.arms)) {
1073
+ const resolved = (0, plugin_loader_js_1.resolveHarness)({
1074
+ plugin: arm.plugin,
1075
+ settings: arm.settings,
1076
+ files: { ...spec.fixture, ...arm.files },
1077
+ });
1078
+ // The plugin root `resolveHarness` expanded into the resolved files/settings
1079
+ // is this checkout's absolute path — normalize it out so the hash is the same
1080
+ // on the dev's machine and in CI (else every plugin-with-root-hooks eval is
1081
+ // falsely stale across machines).
1082
+ const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
1083
+ arms[name] = {
1084
+ model: arm.model ?? cfg.model,
1085
+ tools: [...cfg.tools].sort(),
1086
+ files: stripPluginRoot(resolved.files, absRoot),
1087
+ settings: stripPluginRoot(resolved.settings, absRoot),
1088
+ pluginDirHash: arm.pluginDir ? (0, eval_cache_js_1.hashDir)(arm.pluginDir) : undefined,
1089
+ interceptTools: arm.interceptTools
1090
+ ? (0, tool_intercept_js_1.serializeIntercepts)(arm.interceptTools)
1091
+ : undefined,
1092
+ };
1093
+ }
1094
+ // Tool stubs (`spec.stubs`) are written onto PATH before each trial, so a
1095
+ // change to a canned CLI output IS a model-facing input change — fold a
1096
+ // canonical (name-sorted) view into the hash so `--check` catches it. Sorted
1097
+ // for a stable key regardless of declaration order; an empty list is the
1098
+ // byte-identical-to-before default. Each ToolStub is plain serializable data.
1099
+ const stubs = [...(spec.stubs ?? [])].sort((a, b) => a.name.localeCompare(b.name));
1100
+ // `ephemeralEnv` swaps the trial's environment (scrubbed env + throwaway HOME
1101
+ // vs the inherited process env), which can move tool/hook/agent behavior — a
1102
+ // model-facing input, so it belongs in the hash. Normalize to a bool so a
1103
+ // record under one mode can't be replayed under the other with the same hash.
1104
+ return {
1105
+ task: spec.task,
1106
+ arms,
1107
+ stubs,
1108
+ ephemeralEnv: spec.ephemeralEnv === true,
1109
+ };
1110
+ }
995
1111
  async function runEvalWith(spec, runner) {
996
1112
  const trials = spec.trials ?? 5;
997
1113
  const spacing = (spec.spacingSec ?? 4) * 1000;
@@ -1024,9 +1140,16 @@ async function runEvalWith(spec, runner) {
1024
1140
  await sleep(spacing);
1025
1141
  return { armName: unit.armName, skipped: false, row, usage };
1026
1142
  };
1027
- const results = await runPool(units, concurrency, worker);
1028
- const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
1029
- return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
1143
+ const lock = resolveLock(spec.lock);
1144
+ // Resolve the lock inputs only when the lock is active (off skips the
1145
+ // resolveHarness/hashDir work). `check` replays the committed report below
1146
+ // without ever entering the run pool — so no model is driven in CI.
1147
+ const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
1148
+ return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
1149
+ const results = await runPool(units, concurrency, worker);
1150
+ const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
1151
+ return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
1152
+ });
1030
1153
  }
1031
1154
  /** Render one metric: `name=mean±se pass^k=…` (se/pass^k shown when measured). */
1032
1155
  function formatMetric(name, mean, stat) {
@@ -1065,6 +1188,7 @@ function formatEvalReport(report) {
1065
1188
  exports.claudeEvalDriver = {
1066
1189
  runner: spawnAgent,
1067
1190
  parse: parseClaudeRun,
1191
+ harness: "claude-code",
1068
1192
  };
1069
1193
  /**
1070
1194
  * Package loose `<skillsDir>/<name>/SKILL.md` skills into a throwaway plugin dir
@@ -1439,7 +1563,7 @@ function assertTriggerDiversity(spec) {
1439
1563
  });
1440
1564
  }
1441
1565
  }
1442
- async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError) {
1566
+ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
1443
1567
  // Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
1444
1568
  assertTriggerDiversity(spec);
1445
1569
  // Model floor (default Sonnet): trigger-rate under-measures selection on a
@@ -1469,26 +1593,50 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1469
1593
  parse,
1470
1594
  runError,
1471
1595
  };
1596
+ const lock = resolveLock(spec.lock);
1472
1597
  try {
1473
- const relevant = await runTriggerSet(spec.prompts, cfg, runner);
1474
- const base = {
1475
- rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
1476
- n: relevant.n,
1477
- perPrompt: relevant.perPrompt,
1478
- competitors,
1479
- errored: positiveOrUndefined(relevant.errored),
1480
- };
1481
- if ((spec.irrelevantPrompts?.length ?? 0) === 0)
1482
- return base;
1483
- const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
1484
- const fires = relevant.fired + irrelevant.fired;
1485
- return {
1486
- ...base,
1487
- errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
1488
- falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
1489
- precision: fires > 0 ? relevant.fired / fires : undefined,
1490
- perIrrelevant: irrelevant.perPrompt,
1491
- };
1598
+ // The lock hashes the skill UNDER TEST (its dir contents) + the prompts +
1599
+ // model + tools — the inputs that steer whether it fires. `--check` replays
1600
+ // the committed report with no model; `--update` records it. Resolved only
1601
+ // when the lock is active. `trials` is excluded (a sample-size knob).
1602
+ const triggerInputs = lock.mode === "off"
1603
+ ? undefined
1604
+ : {
1605
+ pluginDirHash: (0, eval_cache_js_1.hashDir)(pluginDir),
1606
+ prompts: [...spec.prompts],
1607
+ irrelevantPrompts: spec.irrelevantPrompts
1608
+ ? [...spec.irrelevantPrompts]
1609
+ : undefined,
1610
+ model: cfg.model,
1611
+ tools: [...cfg.tools].sort(),
1612
+ fixture: spec.fixture,
1613
+ competitors,
1614
+ // The harness the driver runs is a model-facing input — a Claude vs
1615
+ // Codex run can fire a skill differently — so a recorded report is
1616
+ // STALE if the eval is switched to another harness.
1617
+ harness,
1618
+ };
1619
+ return await withEvalLock({ name: spec.name, inputs: triggerInputs, model: cfg.model, lock }, async () => {
1620
+ const relevant = await runTriggerSet(spec.prompts, cfg, runner);
1621
+ const base = {
1622
+ rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
1623
+ n: relevant.n,
1624
+ perPrompt: relevant.perPrompt,
1625
+ competitors,
1626
+ errored: positiveOrUndefined(relevant.errored),
1627
+ };
1628
+ if ((spec.irrelevantPrompts?.length ?? 0) === 0)
1629
+ return base;
1630
+ const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
1631
+ const fires = relevant.fired + irrelevant.fired;
1632
+ return {
1633
+ ...base,
1634
+ errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
1635
+ falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
1636
+ precision: fires > 0 ? relevant.fired / fires : undefined,
1637
+ perIrrelevant: irrelevant.perPrompt,
1638
+ };
1639
+ });
1492
1640
  }
1493
1641
  finally {
1494
1642
  // Remove the throwaway plugin dir we built from a loose `skillsDir`.
@@ -1506,7 +1654,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1506
1654
  */
1507
1655
  async function measureTriggerRate(spec, opts = {}) {
1508
1656
  const d = opts.evalDriver ?? exports.claudeEvalDriver;
1509
- return measureTriggerRateWith(spec, d.runner, d.parse, d.runError);
1657
+ return measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
1510
1658
  }
1511
1659
  /* v8 ignore stop */
1512
1660
  /** Format a trigger-rate report: overall %, then each prompt's rate. */
@@ -158,6 +158,43 @@ export interface InstallPlan {
158
158
  export declare function planPluginInstall(harnesses: readonly string[], opts: {
159
159
  hasClaude: boolean;
160
160
  }): InstallPlan[];
161
+ /** One vigiles-managed Codex hook: a `[[hooks.<event>]]` entry. */
162
+ export interface CodexPluginHook {
163
+ readonly event: string;
164
+ /** Codex matcher (anchored regex, the dialect convention). */
165
+ readonly matcher: string;
166
+ /** The shell command Codex runs (a direct `npx vigiles …`, no plugin root). */
167
+ readonly command: string;
168
+ /** A unique command substring → idempotent re-merge (replaces in place). */
169
+ readonly key: string;
170
+ }
171
+ /**
172
+ * vigiles's proactive nudges, wired into a Codex repo's `.codex/config.toml`.
173
+ *
174
+ * Codex has no global plugin store (unlike Claude Code's marketplace), so its
175
+ * config is repo-committed — the idiomatic place for these. They run as DIRECT
176
+ * `npx vigiles hook-runtime …` commands (NOT vendored bash scripts): the runtime
177
+ * entrypoints read the event JSON on stdin and emit the `hookSpecificOutput.
178
+ * additionalContext` shape Codex honors on `PostToolUse` (confirmed against the
179
+ * official hooks docs + encoded in `HookProtocol.injectableEvents`). Safety: only
180
+ * an INTENTIONAL `exit 2` blocks an edit (the refs nudge, when `unmarked-refs` is
181
+ * `error`); an npx-resolution failure exits non-2, so a missing dep never blocks.
182
+ *
183
+ * Deliberately NOT here (a loud, documented deferral — no-silent-skips): the
184
+ * SessionStart lint summary (CC delivers it as plain stdout, whose SessionStart
185
+ * prepend is unconfirmed on Codex — vs the JSON `additionalContext` these use) and
186
+ * the compile-on-edit / pre-edit-block guards (filename-gated bash with no
187
+ * harness-neutral `hook-runtime` entrypoint yet). Those stay manual on Codex.
188
+ */
189
+ export declare function codexPluginHooks(): CodexPluginHook[];
190
+ /**
191
+ * Idempotently merge {@link codexPluginHooks} into a parsed `.codex/config.toml`
192
+ * object. Pure — the IO (read/parse/serialize/write) stays in cli.ts's
193
+ * `wireCodexHooks`. Each vigiles hook is keyed by a unique command substring, so
194
+ * a re-run REPLACES its own entry in place and leaves the user's own Codex hooks
195
+ * (and every other config key) untouched. Returns a new object.
196
+ */
197
+ export declare function applyCodexPluginHooks(existing: Record<string, unknown>): Record<string, unknown>;
161
198
  /**
162
199
  * Resolve the final plan: defaults, then flags, then interactive answers (each
163
200
  * layer overrides the previous only where it has an opinion). `--target` pins a
@@ -17,6 +17,8 @@ exports.mergeProjectConfig = mergeProjectConfig;
17
17
  exports.shouldPrompt = shouldPrompt;
18
18
  exports.collectSetupAnswers = collectSetupAnswers;
19
19
  exports.planPluginInstall = planPluginInstall;
20
+ exports.codexPluginHooks = codexPluginHooks;
21
+ exports.applyCodexPluginHooks = applyCodexPluginHooks;
20
22
  exports.resolvePlan = resolvePlan;
21
23
  function flagValue(args, prefix) {
22
24
  return args.find((a) => a.startsWith(prefix))?.slice(prefix.length);
@@ -260,8 +262,10 @@ function planPluginInstall(harnesses, opts) {
260
262
  if (harness === "codex") {
261
263
  // The cross-agent `skills` CLI with `-g -y` installs to the global store
262
264
  // ~/.agents/skills/ (NOT the repo, and NOT ~/.codex/ — verified against
263
- // the real CLI). Skills only; Codex hooks (.codex/config.toml [hooks])
264
- // are not wired automatically.
265
+ // the real CLI). Skills install globally; the proactive NUDGE hooks
266
+ // (eval-lock + refs) are wired into the repo's .codex/config.toml by
267
+ // `init` (see codexPluginHooks / wireCodexHooks) — Codex config is
268
+ // repo-committed, so that's the idiomatic place.
265
269
  return {
266
270
  harness,
267
271
  commands: ["npx --yes skills add zernie/vigiles -a codex -g -y"],
@@ -269,9 +273,10 @@ function planPluginInstall(harnesses, opts) {
269
273
  manualSteps: ["npx skills add zernie/vigiles -a codex -g -y"],
270
274
  notes: [
271
275
  "Codex reads AGENTS.md directly; the skills install globally to ~/.agents/skills/ (not the repo).",
272
- "Codex hooks (.codex/config.toml [hooks]) are not auto-wired yet — add them manually for compile-on-edit.",
276
+ "The eval-lock + refs NUDGE hooks are wired into .codex/config.toml (repo-committed, the Codex norm).",
277
+ "Still manual on Codex: the SessionStart lint summary + compile-on-edit/pre-edit guards (no harness-neutral entrypoint yet).",
273
278
  ],
274
- vendors: false,
279
+ vendors: true,
275
280
  };
276
281
  }
277
282
  return {
@@ -284,6 +289,63 @@ function planPluginInstall(harnesses, opts) {
284
289
  };
285
290
  });
286
291
  }
292
+ /**
293
+ * vigiles's proactive nudges, wired into a Codex repo's `.codex/config.toml`.
294
+ *
295
+ * Codex has no global plugin store (unlike Claude Code's marketplace), so its
296
+ * config is repo-committed — the idiomatic place for these. They run as DIRECT
297
+ * `npx vigiles hook-runtime …` commands (NOT vendored bash scripts): the runtime
298
+ * entrypoints read the event JSON on stdin and emit the `hookSpecificOutput.
299
+ * additionalContext` shape Codex honors on `PostToolUse` (confirmed against the
300
+ * official hooks docs + encoded in `HookProtocol.injectableEvents`). Safety: only
301
+ * an INTENTIONAL `exit 2` blocks an edit (the refs nudge, when `unmarked-refs` is
302
+ * `error`); an npx-resolution failure exits non-2, so a missing dep never blocks.
303
+ *
304
+ * Deliberately NOT here (a loud, documented deferral — no-silent-skips): the
305
+ * SessionStart lint summary (CC delivers it as plain stdout, whose SessionStart
306
+ * prepend is unconfirmed on Codex — vs the JSON `additionalContext` these use) and
307
+ * the compile-on-edit / pre-edit-block guards (filename-gated bash with no
308
+ * harness-neutral `hook-runtime` entrypoint yet). Those stay manual on Codex.
309
+ */
310
+ function codexPluginHooks() {
311
+ // Codex's file-edit tool is `apply_patch` (its dialect vocabulary —
312
+ // src/adapters/codex/dialect.ts), NOT Claude's `Edit`/`Write`. A PostToolUse
313
+ // matcher keyed on CC tool names would never fire on Codex, so the nudges must
314
+ // match the Codex tool name. (Both nudge entrypoints also self-gate on the
315
+ // edited file, so a non-edit event no-ops regardless.)
316
+ return [
317
+ {
318
+ event: "PostToolUse",
319
+ matcher: "^apply_patch$",
320
+ command: "npx --no-install vigiles hook-runtime eval-lock-nudge",
321
+ key: "hook-runtime eval-lock-nudge",
322
+ },
323
+ {
324
+ event: "PostToolUse",
325
+ matcher: "^apply_patch$",
326
+ command: "npx --no-install vigiles hook-runtime refs",
327
+ key: "hook-runtime refs",
328
+ },
329
+ ];
330
+ }
331
+ /**
332
+ * Idempotently merge {@link codexPluginHooks} into a parsed `.codex/config.toml`
333
+ * object. Pure — the IO (read/parse/serialize/write) stays in cli.ts's
334
+ * `wireCodexHooks`. Each vigiles hook is keyed by a unique command substring, so
335
+ * a re-run REPLACES its own entry in place and leaves the user's own Codex hooks
336
+ * (and every other config key) untouched. Returns a new object.
337
+ */
338
+ function applyCodexPluginHooks(existing) {
339
+ const config = existing;
340
+ const hooks = {
341
+ ...(config.hooks ?? {}),
342
+ };
343
+ for (const h of codexPluginHooks()) {
344
+ const kept = (hooks[h.event] ?? []).filter((e) => !e.command.includes(h.key));
345
+ hooks[h.event] = [...kept, { matcher: h.matcher, command: h.command }];
346
+ }
347
+ return { ...existing, hooks };
348
+ }
287
349
  /**
288
350
  * Resolve the final plan: defaults, then flags, then interactive answers (each
289
351
  * layer overrides the previous only where it has an opinion). `--target` pins a
@@ -0,0 +1,21 @@
1
+ #!/usr/bin/env bash
2
+ # PostToolUse hook — after the agent edits an eval input (a SKILL.md trigger
3
+ # surface or an *.eval.* script) and committed eval locks exist, inject a
4
+ # NON-BLOCKING reminder to re-run `vigiles eval --update`. Self-gating: silent
5
+ # until you've committed a lock, so it never fires in a repo that doesn't use
6
+ # evals. Never blocks the edit. Runs as its OWN PostToolUse entry so its stdout
7
+ # stays clean JSON. See docs/harness-testing.md (the eval lock).
8
+
9
+ set -uo pipefail
10
+
11
+ INPUT=$(cat)
12
+
13
+ # No npx / not a Node project → nothing to do, never disrupt the edit.
14
+ command -v npx >/dev/null 2>&1 || exit 0
15
+ [ -f package.json ] || exit 0
16
+
17
+ printf '%s' "$INPUT" | npx vigiles hook-runtime eval-lock-nudge
18
+
19
+ # Always exit 0 — this hook only nudges, it never blocks (the gate is CI's
20
+ # `vigiles eval --check`). Swallow any tool/setup error so it can't disrupt edits.
21
+ exit 0
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "11.0.0",
3
+ "version": "12.0.0",
4
4
  "description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
5
5
  "keywords": [
6
6
  "claude-code",
@@ -190,6 +190,33 @@ job that asserts the capability is present, run **`vigiles test --no-skip`** so
190
190
  skipped tier fails — a green-with-skips is untested surface. Keep unit +
191
191
  deterministic tests in CI (free); run evals locally or on a schedule with auth.
192
192
 
193
+ ## Step 5 — Lock the eval so CI stays honest (you do this automatically)
194
+
195
+ Real-model evals run on the user's subscription — locally, never in CI. So **as
196
+ part of writing an eval, you keep its result fresh for them.** Do these two things
197
+ without being asked:
198
+
199
+ 1. **Give every eval a `name`.** That's what the lock keys on.
200
+ 2. **After running it, commit the lock:**
201
+
202
+ ```bash
203
+ vigiles eval --update # records the result → .vigiles/eval-locks/<name>.lock.json
204
+ ```
205
+
206
+ Then commit that file. CI runs `vigiles eval --check` (no model) to verify it
207
+ still matches the inputs — so a later edit that forgets to re-eval fails loud
208
+ instead of shipping stale numbers.
209
+
210
+ **When you later change a skill's description or prompts, re-run `vigiles eval
211
+ --update` and commit the updated lock** — the change altered what the eval
212
+ measures. (vigiles also nudges you: when a lock exists, a `SKILL.md` edit triggers
213
+ a non-blocking reminder.)
214
+
215
+ Why it's cheap: `--check` only hashes inputs (skill text, prompts, model). A
216
+ **threshold** change in the test re-uses the saved numbers (no model); only an
217
+ **input** change needs a fresh `--update`. Full mechanics:
218
+ [`docs/harness-testing.md`](../../docs/harness-testing.md#keep-eval-results-fresh-in-ci-the-lock).
219
+
193
220
  ## When the user didn't say what to test
194
221
 
195
222
  Don't ask them to specify — **pick something real and demonstrate.** Scan the