vigiles 11.0.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +9 -0
- package/README.md +9 -4
- package/action.yml +13 -2
- package/dist/adapter-conformance.js +6 -0
- package/dist/adapter-registry.d.ts +20 -0
- package/dist/adapter-registry.js +27 -0
- package/dist/adapters/claude-code/hook-protocol.js +4 -0
- package/dist/adapters/claude-code/runtime.js +12 -0
- package/dist/adapters/codex/eval.js +3 -0
- package/dist/adapters/codex/hook-protocol.d.ts +9 -1
- package/dist/adapters/codex/hook-protocol.js +10 -0
- package/dist/adapters/codex/runtime.js +10 -0
- package/dist/adapters/opencode/runtime.js +4 -0
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +1 -0
- package/dist/cli.js +211 -29
- package/dist/core/hook-protocol.d.ts +15 -0
- package/dist/core/runtime.d.ts +20 -0
- package/dist/core/types.d.ts +12 -0
- package/dist/eval-cache.d.ts +6 -0
- package/dist/eval-cache.js +2 -0
- package/dist/eval-lock.d.ts +192 -0
- package/dist/eval-lock.js +286 -0
- package/dist/eval.d.ts +33 -20
- package/dist/eval.js +199 -51
- package/dist/setup-plan.d.ts +37 -0
- package/dist/setup-plan.js +66 -4
- package/hooks/eval-lock-nudge.sh +21 -0
- package/package.json +1 -1
- package/skills/test-harness/SKILL.md +27 -0
package/dist/eval.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import { parseToolCalls, parseHooks, parseSubagents, type ToolCall, type Trace } from "./harness-test.js";
|
|
2
2
|
import { type CacheMode } from "./eval-cache.js";
|
|
3
|
+
import { type EvalLockOptions } from "./eval-lock.js";
|
|
3
4
|
import type { Check, CheckJSON } from "./check.js";
|
|
4
5
|
import { type Comparison } from "./stats.js";
|
|
5
6
|
import { type ToolIntercept } from "./tool-intercept.js";
|
|
@@ -159,6 +160,16 @@ export interface EvalSpec<M extends Metrics> {
|
|
|
159
160
|
* to today). See {@link ToolStub} and `research/eval-coverage-and-isolation.md`.
|
|
160
161
|
*/
|
|
161
162
|
readonly stubs?: readonly ToolStub[];
|
|
163
|
+
/**
|
|
164
|
+
* **The eval LOCK** — a committed staleness gate for CI (see `src/eval-lock.ts`).
|
|
165
|
+
* With a `name` set, `vigiles eval --update` (local, on your subscription)
|
|
166
|
+
* records the report to `.vigiles/eval-locks/<name>.lock.json`; `--check` (CI)
|
|
167
|
+
* verifies the committed result against the current inputs WITHOUT a model call,
|
|
168
|
+
* failing "stale" when they diverge. Mode normally comes from the CLI; set
|
|
169
|
+
* `lock` to drive it programmatically or point a test at a throwaway dir. The
|
|
170
|
+
* lock engages only when `name` is set (it keys the lock file).
|
|
171
|
+
*/
|
|
172
|
+
readonly lock?: EvalLockOptions;
|
|
162
173
|
}
|
|
163
174
|
/** Per-metric summary statistics across an arm's runs. */
|
|
164
175
|
export interface MetricStat {
|
|
@@ -479,16 +490,6 @@ export declare function modelTier(id: string): number | null;
|
|
|
479
490
|
* {@link modelTier}); an unrankable model/floor is never "below" (fail-open).
|
|
480
491
|
*/
|
|
481
492
|
export declare function belowModelFloor(model: string, floor: string): boolean;
|
|
482
|
-
/**
|
|
483
|
-
* Reduce a raw `--version` string to the **major.minor** cache-key token. We key
|
|
484
|
-
* the cache on major.minor, NOT the patch: a patch release rarely changes agent
|
|
485
|
-
* behaviour, so keying patches would churn the cache on every release for no
|
|
486
|
-
* signal; a minor/major bump is where the system prompt / tool defs actually move.
|
|
487
|
-
* (If a specific patch is known to matter, clear the cache or bump
|
|
488
|
-
* `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
|
|
489
|
-
* found. Pure + tested.
|
|
490
|
-
*/
|
|
491
|
-
export declare function harnessVersionKey(raw: string): string;
|
|
492
493
|
/**
|
|
493
494
|
* Build an **ephemeral run environment** for a model-driven run: a NEW env object
|
|
494
495
|
* with a *fresh* `HOME` (and `TMPDIR`) pointed at the throwaway `opts.home`, only
|
|
@@ -542,15 +543,6 @@ export declare function seedEphemeralHome(throwawayHome: string, realHome: strin
|
|
|
542
543
|
export declare function isRateLimited(out: RunOut): boolean;
|
|
543
544
|
/** Map `worker` over `items` with at most `concurrency` in flight, order preserved. */
|
|
544
545
|
export declare function runPool<T, R>(items: readonly T[], concurrency: number, worker: (item: T) => Promise<R>): Promise<R[]>;
|
|
545
|
-
/**
|
|
546
|
-
* The eval orchestration — every arm × trial via `runner`, run through the cache
|
|
547
|
-
* and a rate-limit retry, with at most `concurrency` in flight and an optional
|
|
548
|
-
* `maxCostUsd` budget cap; metric + usage computed per run and aggregated per
|
|
549
|
-
* arm. Exported with an injectable `runner` so the loop, `measure` context,
|
|
550
|
-
* caching, pooling, and aggregation are unit-testable without spawning a model
|
|
551
|
-
* (pass a fake returning canned stream-json). `runEval` is this with the real
|
|
552
|
-
* agent runner.
|
|
553
|
-
*/
|
|
554
546
|
export declare function runEvalWith<M extends Metrics>(spec: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
|
|
555
547
|
/** Format an eval report as a compact table for the console (mean ± se, pass^k). */
|
|
556
548
|
export declare function formatEvalReport(report: EvalReport): string;
|
|
@@ -563,6 +555,19 @@ export declare function formatEvalReport(report: EvalReport): string;
|
|
|
563
555
|
* (reuse the bare predicates, e.g. `(t) => skillResolved(t, "x:y")`).
|
|
564
556
|
*/
|
|
565
557
|
export interface TriggerRateSpec {
|
|
558
|
+
/**
|
|
559
|
+
* A stable name for this trigger eval — required to engage the {@link lock}
|
|
560
|
+
* (it keys the committed `.vigiles/eval-locks/<name>.lock.json`). Two evals over
|
|
561
|
+
* the same skill with different prompts get distinct names. Optional otherwise.
|
|
562
|
+
*/
|
|
563
|
+
readonly name?: string;
|
|
564
|
+
/**
|
|
565
|
+
* **The eval LOCK** — the CI staleness gate (see `src/eval-lock.ts`). With
|
|
566
|
+
* `name` set, `vigiles eval --update` records this trigger-rate report locally
|
|
567
|
+
* and `--check` verifies it against the current inputs (skill contents, prompts,
|
|
568
|
+
* model) with NO model call. Mode normally comes from the CLI flags.
|
|
569
|
+
*/
|
|
570
|
+
readonly lock?: EvalLockOptions;
|
|
566
571
|
/**
|
|
567
572
|
* Plugin dir installed natively (`--plugin-dir`) so its skills/commands
|
|
568
573
|
* activate. Provide this OR {@link skillsDir}, not both.
|
|
@@ -722,6 +727,14 @@ export interface EvalDriver {
|
|
|
722
727
|
readonly runner: AgentRunner;
|
|
723
728
|
readonly parse: ModelOutputParser;
|
|
724
729
|
readonly runError?: (out: RunOut) => string | null;
|
|
730
|
+
/**
|
|
731
|
+
* The harness this driver runs (e.g. `"claude-code"`, `"codex"`). Folded into a
|
|
732
|
+
* trigger-rate eval's LOCK hash so a report recorded on one harness is marked
|
|
733
|
+
* STALE if the eval is later switched to another (a different harness can fire a
|
|
734
|
+
* skill differently). Optional for back-compat — absent defaults to
|
|
735
|
+
* `"claude-code"`, so an existing single-harness lock is unaffected.
|
|
736
|
+
*/
|
|
737
|
+
readonly harness?: string;
|
|
725
738
|
}
|
|
726
739
|
/** The default (Claude Code) eval driver: real `claude` + stream-json parsing. */
|
|
727
740
|
export declare const claudeEvalDriver: EvalDriver;
|
|
@@ -801,7 +814,7 @@ export declare function packageInstallSet(opts: {
|
|
|
801
814
|
dir: string;
|
|
802
815
|
added: number;
|
|
803
816
|
};
|
|
804
|
-
export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null): Promise<TriggerRateReport>;
|
|
817
|
+
export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null, harness?: string): Promise<TriggerRateReport>;
|
|
805
818
|
/**
|
|
806
819
|
* Measure a skill/behaviour's real trigger rate across prompts × trials. Defaults
|
|
807
820
|
* to the real `claude` CLI (`claudeEvalDriver`); pass `{ evalDriver }` to drive a
|
package/dist/eval.js
CHANGED
|
@@ -22,7 +22,6 @@ exports.aggregateUsage = aggregateUsage;
|
|
|
22
22
|
exports.isDatedModel = isDatedModel;
|
|
23
23
|
exports.modelTier = modelTier;
|
|
24
24
|
exports.belowModelFloor = belowModelFloor;
|
|
25
|
-
exports.harnessVersionKey = harnessVersionKey;
|
|
26
25
|
exports.ephemeralRunEnv = ephemeralRunEnv;
|
|
27
26
|
exports.seedEphemeralHome = seedEphemeralHome;
|
|
28
27
|
exports.isRateLimited = isRateLimited;
|
|
@@ -71,6 +70,7 @@ const runtime_js_1 = require("./adapters/claude-code/runtime.js");
|
|
|
71
70
|
const proofs_js_1 = require("./core/proofs.js");
|
|
72
71
|
const harness_test_js_1 = require("./harness-test.js");
|
|
73
72
|
const eval_cache_js_1 = require("./eval-cache.js");
|
|
73
|
+
const eval_lock_js_1 = require("./eval-lock.js");
|
|
74
74
|
const stats_js_1 = require("./stats.js");
|
|
75
75
|
const tool_intercept_js_1 = require("./tool-intercept.js");
|
|
76
76
|
const tool_stub_js_1 = require("./tool-stub.js");
|
|
@@ -640,32 +640,22 @@ function belowModelFloor(model, floor) {
|
|
|
640
640
|
const f = modelTier(floor);
|
|
641
641
|
return m !== null && f !== null && m < f;
|
|
642
642
|
}
|
|
643
|
-
/**
|
|
644
|
-
* Reduce a raw `--version` string to the **major.minor** cache-key token. We key
|
|
645
|
-
* the cache on major.minor, NOT the patch: a patch release rarely changes agent
|
|
646
|
-
* behaviour, so keying patches would churn the cache on every release for no
|
|
647
|
-
* signal; a minor/major bump is where the system prompt / tool defs actually move.
|
|
648
|
-
* (If a specific patch is known to matter, clear the cache or bump
|
|
649
|
-
* `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
|
|
650
|
-
* found. Pure + tested.
|
|
651
|
-
*/
|
|
652
|
-
function harnessVersionKey(raw) {
|
|
653
|
-
const m = /(\d+)\.(\d+)\.\d+/.exec(raw);
|
|
654
|
-
return m ? `${m[1]}.${m[2]}` : raw.trim();
|
|
655
|
-
}
|
|
656
643
|
/* v8 ignore start -- spawns the real harness binary; memoized, cache-path only */
|
|
657
644
|
let cachedHarnessVersion;
|
|
658
645
|
/**
|
|
659
|
-
* The harness binary version (`claude --version`) reduced to
|
|
660
|
-
*
|
|
661
|
-
*
|
|
662
|
-
*
|
|
663
|
-
*
|
|
646
|
+
* The harness binary version (`claude --version`), reduced to its
|
|
647
|
+
* behaviorally-significant token by the runtime port's
|
|
648
|
+
* {@link HarnessRuntime.versionKey} — so a Claude Code **minor/major** upgrade
|
|
649
|
+
* (new system prompt / tool defs) invalidates a stale replay while patches don't
|
|
650
|
+
* churn it. The reduction is per-harness (CC → `major.minor`, Codex → `""`) and
|
|
651
|
+
* lives on the adapter, not here. Memoized (one spawn per process), resolved only
|
|
652
|
+
* on the cache path, "unknown" if the binary isn't found (then it doesn't
|
|
653
|
+
* partition the key).
|
|
664
654
|
*/
|
|
665
655
|
function harnessVersion() {
|
|
666
656
|
if (cachedHarnessVersion === undefined) {
|
|
667
657
|
try {
|
|
668
|
-
cachedHarnessVersion =
|
|
658
|
+
cachedHarnessVersion = runtime_js_1.claudeCodeRuntime.versionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
|
|
669
659
|
encoding: "utf-8",
|
|
670
660
|
stdio: ["ignore", "pipe", "ignore"],
|
|
671
661
|
}));
|
|
@@ -983,15 +973,141 @@ function aggregateArms(armNames, results) {
|
|
|
983
973
|
}
|
|
984
974
|
return { arms, totalCostUsd };
|
|
985
975
|
}
|
|
976
|
+
function resolveLock(over) {
|
|
977
|
+
return {
|
|
978
|
+
mode: over?.mode ?? (0, eval_lock_js_1.lockModeFromEnv)(),
|
|
979
|
+
dir: over?.dir ?? (0, node_path_1.resolve)(process.cwd(), eval_lock_js_1.DEFAULT_LOCK_DIR),
|
|
980
|
+
evalApiVersion: over?.evalApiVersion ?? (0, eval_lock_js_1.evalApiVersionFromEnv)(),
|
|
981
|
+
};
|
|
982
|
+
}
|
|
983
|
+
/** Emit a vigiles message (GitHub annotation under Actions, else stderr/stdout). */
|
|
984
|
+
function emitLockMessage(msg, warn) {
|
|
985
|
+
if (process.env.GITHUB_ACTIONS)
|
|
986
|
+
console.log(`::${warn ? "warning" : "notice"}::${msg}`);
|
|
987
|
+
else if (warn)
|
|
988
|
+
console.warn(msg);
|
|
989
|
+
else
|
|
990
|
+
console.log(msg);
|
|
991
|
+
}
|
|
992
|
+
/**
|
|
993
|
+
* Run a named eval through the lock. `off` → just `produce()`. `check` → replay a
|
|
994
|
+
* matching committed lock (NO model call) or throw "stale". `update` → `produce()`,
|
|
995
|
+
* write the lock, print the human-facing delta. The model is driven ONLY on the
|
|
996
|
+
* run path — never on a clean `check`, which is what keeps the CI gate binary-free.
|
|
997
|
+
* A lock-on run with no `name` is a LOUD skip (the lock needs a name to key the file).
|
|
998
|
+
*/
|
|
999
|
+
async function withEvalLock(args, produce) {
|
|
1000
|
+
const { lock } = args;
|
|
1001
|
+
if (lock.mode === "off")
|
|
1002
|
+
return produce();
|
|
1003
|
+
if (!args.name) {
|
|
1004
|
+
// An unnamed eval can't be keyed to a lock. In `check` (the CI gate) running
|
|
1005
|
+
// it would call the model — violating the no-model-in-CI contract and failing
|
|
1006
|
+
// for missing auth — so FAIL LOUDLY instead of silently hitting the model.
|
|
1007
|
+
// `update` (local, model available) keeps producing: the eval just isn't gated.
|
|
1008
|
+
if (lock.mode === "check") {
|
|
1009
|
+
throw new Error(`vigiles eval --check: an unnamed eval cannot run in CI — it has no ` +
|
|
1010
|
+
`committed lock to replay, and running it would call the model. Add a ` +
|
|
1011
|
+
`\`name\` to the spec to gate it, or exclude it from the --check run.`);
|
|
1012
|
+
}
|
|
1013
|
+
emitLockMessage(`vigiles eval --${lock.mode}: skipped the lock for an unnamed eval — set ` +
|
|
1014
|
+
`\`name\` on the spec to enable the staleness gate for it.`, true);
|
|
1015
|
+
return produce();
|
|
1016
|
+
}
|
|
1017
|
+
if (!isDatedModel(args.model))
|
|
1018
|
+
warnFloatingModel(args.model);
|
|
1019
|
+
const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
|
|
1020
|
+
model: args.model,
|
|
1021
|
+
evalApiVersion: lock.evalApiVersion,
|
|
1022
|
+
inputs: args.inputs,
|
|
1023
|
+
});
|
|
1024
|
+
const existing = (0, eval_lock_js_1.readLock)(lock.dir, args.name);
|
|
1025
|
+
const decision = (0, eval_lock_js_1.decideLock)(lock.mode, args.name, inputsHash, existing);
|
|
1026
|
+
if (decision.kind === "stale")
|
|
1027
|
+
throw new Error(decision.reason);
|
|
1028
|
+
if (decision.kind === "replay")
|
|
1029
|
+
return decision.report;
|
|
1030
|
+
const report = await produce();
|
|
1031
|
+
const builtLock = (0, eval_lock_js_1.buildLock)({
|
|
1032
|
+
name: args.name,
|
|
1033
|
+
inputsHash,
|
|
1034
|
+
model: args.model,
|
|
1035
|
+
harnessVersionKey: harnessVersion(),
|
|
1036
|
+
evalApiVersion: lock.evalApiVersion,
|
|
1037
|
+
builtAt: new Date().toISOString(),
|
|
1038
|
+
report,
|
|
1039
|
+
});
|
|
1040
|
+
const deltas = existing ? (0, eval_lock_js_1.diffReportNumbers)(existing.report, report) : [];
|
|
1041
|
+
(0, eval_lock_js_1.writeLock)(lock.dir, builtLock);
|
|
1042
|
+
emitLockMessage((0, eval_lock_js_1.formatLockUpdate)(args.name, deltas, existing === null), false);
|
|
1043
|
+
return report;
|
|
1044
|
+
}
|
|
1045
|
+
/**
|
|
1046
|
+
* Strip the machine-specific plugin-root prefix from a resolved value before it
|
|
1047
|
+
* enters the lock hash. `resolveHarness` expands `${PLUGIN_ROOT}` in a plugin's
|
|
1048
|
+
* hook commands to the checkout's ABSOLUTE path, so a lock recorded at
|
|
1049
|
+
* `/home/dev/...` would be falsely STALE when `--check` recomputes it at
|
|
1050
|
+
* `/home/runner/...` in CI (or any other machine). Normalizing the prefix back to
|
|
1051
|
+
* a token makes the hash location-independent. No-op when there's no plugin root.
|
|
1052
|
+
*/
|
|
1053
|
+
function stripPluginRoot(value, absRoot) {
|
|
1054
|
+
if (absRoot === "")
|
|
1055
|
+
return value;
|
|
1056
|
+
const json = JSON.stringify(value);
|
|
1057
|
+
// A hookless plugin resolves to `settings: undefined`, which `JSON.stringify`
|
|
1058
|
+
// returns as `undefined` (not a string) — pass it through rather than `.split`
|
|
1059
|
+
// a non-string (which would throw before the eval can run).
|
|
1060
|
+
if (json === undefined)
|
|
1061
|
+
return value;
|
|
1062
|
+
return JSON.parse(json.split(absRoot).join("${PLUGIN_ROOT}"));
|
|
1063
|
+
}
|
|
986
1064
|
/**
|
|
987
|
-
*
|
|
988
|
-
*
|
|
989
|
-
*
|
|
990
|
-
*
|
|
991
|
-
* caching, pooling, and aggregation are unit-testable without spawning a model
|
|
992
|
-
* (pass a fake returning canned stream-json). `runEval` is this with the real
|
|
993
|
-
* agent runner.
|
|
1065
|
+
* Resolve each arm's model-affecting inputs into a canonical object for the lock
|
|
1066
|
+
* hash — WITHOUT running the model (it reads files + hashes plugin dirs only). The
|
|
1067
|
+
* trial count is excluded (a sample-size knob, not a behavior input); `measure` is
|
|
1068
|
+
* excluded by design (the script re-asserts against the replayed report).
|
|
994
1069
|
*/
|
|
1070
|
+
function evalArmsInputs(spec, cfg) {
|
|
1071
|
+
const arms = {};
|
|
1072
|
+
for (const [name, arm] of Object.entries(spec.arms)) {
|
|
1073
|
+
const resolved = (0, plugin_loader_js_1.resolveHarness)({
|
|
1074
|
+
plugin: arm.plugin,
|
|
1075
|
+
settings: arm.settings,
|
|
1076
|
+
files: { ...spec.fixture, ...arm.files },
|
|
1077
|
+
});
|
|
1078
|
+
// The plugin root `resolveHarness` expanded into the resolved files/settings
|
|
1079
|
+
// is this checkout's absolute path — normalize it out so the hash is the same
|
|
1080
|
+
// on the dev's machine and in CI (else every plugin-with-root-hooks eval is
|
|
1081
|
+
// falsely stale across machines).
|
|
1082
|
+
const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
|
|
1083
|
+
arms[name] = {
|
|
1084
|
+
model: arm.model ?? cfg.model,
|
|
1085
|
+
tools: [...cfg.tools].sort(),
|
|
1086
|
+
files: stripPluginRoot(resolved.files, absRoot),
|
|
1087
|
+
settings: stripPluginRoot(resolved.settings, absRoot),
|
|
1088
|
+
pluginDirHash: arm.pluginDir ? (0, eval_cache_js_1.hashDir)(arm.pluginDir) : undefined,
|
|
1089
|
+
interceptTools: arm.interceptTools
|
|
1090
|
+
? (0, tool_intercept_js_1.serializeIntercepts)(arm.interceptTools)
|
|
1091
|
+
: undefined,
|
|
1092
|
+
};
|
|
1093
|
+
}
|
|
1094
|
+
// Tool stubs (`spec.stubs`) are written onto PATH before each trial, so a
|
|
1095
|
+
// change to a canned CLI output IS a model-facing input change — fold a
|
|
1096
|
+
// canonical (name-sorted) view into the hash so `--check` catches it. Sorted
|
|
1097
|
+
// for a stable key regardless of declaration order; an empty list is the
|
|
1098
|
+
// byte-identical-to-before default. Each ToolStub is plain serializable data.
|
|
1099
|
+
const stubs = [...(spec.stubs ?? [])].sort((a, b) => a.name.localeCompare(b.name));
|
|
1100
|
+
// `ephemeralEnv` swaps the trial's environment (scrubbed env + throwaway HOME
|
|
1101
|
+
// vs the inherited process env), which can move tool/hook/agent behavior — a
|
|
1102
|
+
// model-facing input, so it belongs in the hash. Normalize to a bool so a
|
|
1103
|
+
// record under one mode can't be replayed under the other with the same hash.
|
|
1104
|
+
return {
|
|
1105
|
+
task: spec.task,
|
|
1106
|
+
arms,
|
|
1107
|
+
stubs,
|
|
1108
|
+
ephemeralEnv: spec.ephemeralEnv === true,
|
|
1109
|
+
};
|
|
1110
|
+
}
|
|
995
1111
|
async function runEvalWith(spec, runner) {
|
|
996
1112
|
const trials = spec.trials ?? 5;
|
|
997
1113
|
const spacing = (spec.spacingSec ?? 4) * 1000;
|
|
@@ -1024,9 +1140,16 @@ async function runEvalWith(spec, runner) {
|
|
|
1024
1140
|
await sleep(spacing);
|
|
1025
1141
|
return { armName: unit.armName, skipped: false, row, usage };
|
|
1026
1142
|
};
|
|
1027
|
-
const
|
|
1028
|
-
|
|
1029
|
-
|
|
1143
|
+
const lock = resolveLock(spec.lock);
|
|
1144
|
+
// Resolve the lock inputs only when the lock is active (off skips the
|
|
1145
|
+
// resolveHarness/hashDir work). `check` replays the committed report below
|
|
1146
|
+
// without ever entering the run pool — so no model is driven in CI.
|
|
1147
|
+
const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
|
|
1148
|
+
return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
|
|
1149
|
+
const results = await runPool(units, concurrency, worker);
|
|
1150
|
+
const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
|
|
1151
|
+
return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
|
|
1152
|
+
});
|
|
1030
1153
|
}
|
|
1031
1154
|
/** Render one metric: `name=mean±se pass^k=…` (se/pass^k shown when measured). */
|
|
1032
1155
|
function formatMetric(name, mean, stat) {
|
|
@@ -1065,6 +1188,7 @@ function formatEvalReport(report) {
|
|
|
1065
1188
|
exports.claudeEvalDriver = {
|
|
1066
1189
|
runner: spawnAgent,
|
|
1067
1190
|
parse: parseClaudeRun,
|
|
1191
|
+
harness: "claude-code",
|
|
1068
1192
|
};
|
|
1069
1193
|
/**
|
|
1070
1194
|
* Package loose `<skillsDir>/<name>/SKILL.md` skills into a throwaway plugin dir
|
|
@@ -1439,7 +1563,7 @@ function assertTriggerDiversity(spec) {
|
|
|
1439
1563
|
});
|
|
1440
1564
|
}
|
|
1441
1565
|
}
|
|
1442
|
-
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError) {
|
|
1566
|
+
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
|
|
1443
1567
|
// Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
|
|
1444
1568
|
assertTriggerDiversity(spec);
|
|
1445
1569
|
// Model floor (default Sonnet): trigger-rate under-measures selection on a
|
|
@@ -1469,26 +1593,50 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1469
1593
|
parse,
|
|
1470
1594
|
runError,
|
|
1471
1595
|
};
|
|
1596
|
+
const lock = resolveLock(spec.lock);
|
|
1472
1597
|
try {
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1598
|
+
// The lock hashes the skill UNDER TEST (its dir contents) + the prompts +
|
|
1599
|
+
// model + tools — the inputs that steer whether it fires. `--check` replays
|
|
1600
|
+
// the committed report with no model; `--update` records it. Resolved only
|
|
1601
|
+
// when the lock is active. `trials` is excluded (a sample-size knob).
|
|
1602
|
+
const triggerInputs = lock.mode === "off"
|
|
1603
|
+
? undefined
|
|
1604
|
+
: {
|
|
1605
|
+
pluginDirHash: (0, eval_cache_js_1.hashDir)(pluginDir),
|
|
1606
|
+
prompts: [...spec.prompts],
|
|
1607
|
+
irrelevantPrompts: spec.irrelevantPrompts
|
|
1608
|
+
? [...spec.irrelevantPrompts]
|
|
1609
|
+
: undefined,
|
|
1610
|
+
model: cfg.model,
|
|
1611
|
+
tools: [...cfg.tools].sort(),
|
|
1612
|
+
fixture: spec.fixture,
|
|
1613
|
+
competitors,
|
|
1614
|
+
// The harness the driver runs is a model-facing input — a Claude vs
|
|
1615
|
+
// Codex run can fire a skill differently — so a recorded report is
|
|
1616
|
+
// STALE if the eval is switched to another harness.
|
|
1617
|
+
harness,
|
|
1618
|
+
};
|
|
1619
|
+
return await withEvalLock({ name: spec.name, inputs: triggerInputs, model: cfg.model, lock }, async () => {
|
|
1620
|
+
const relevant = await runTriggerSet(spec.prompts, cfg, runner);
|
|
1621
|
+
const base = {
|
|
1622
|
+
rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
|
|
1623
|
+
n: relevant.n,
|
|
1624
|
+
perPrompt: relevant.perPrompt,
|
|
1625
|
+
competitors,
|
|
1626
|
+
errored: positiveOrUndefined(relevant.errored),
|
|
1627
|
+
};
|
|
1628
|
+
if ((spec.irrelevantPrompts?.length ?? 0) === 0)
|
|
1629
|
+
return base;
|
|
1630
|
+
const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
|
|
1631
|
+
const fires = relevant.fired + irrelevant.fired;
|
|
1632
|
+
return {
|
|
1633
|
+
...base,
|
|
1634
|
+
errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
|
|
1635
|
+
falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
|
|
1636
|
+
precision: fires > 0 ? relevant.fired / fires : undefined,
|
|
1637
|
+
perIrrelevant: irrelevant.perPrompt,
|
|
1638
|
+
};
|
|
1639
|
+
});
|
|
1492
1640
|
}
|
|
1493
1641
|
finally {
|
|
1494
1642
|
// Remove the throwaway plugin dir we built from a loose `skillsDir`.
|
|
@@ -1506,7 +1654,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1506
1654
|
*/
|
|
1507
1655
|
async function measureTriggerRate(spec, opts = {}) {
|
|
1508
1656
|
const d = opts.evalDriver ?? exports.claudeEvalDriver;
|
|
1509
|
-
return measureTriggerRateWith(spec, d.runner, d.parse, d.runError);
|
|
1657
|
+
return measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
|
|
1510
1658
|
}
|
|
1511
1659
|
/* v8 ignore stop */
|
|
1512
1660
|
/** Format a trigger-rate report: overall %, then each prompt's rate. */
|
package/dist/setup-plan.d.ts
CHANGED
|
@@ -158,6 +158,43 @@ export interface InstallPlan {
|
|
|
158
158
|
export declare function planPluginInstall(harnesses: readonly string[], opts: {
|
|
159
159
|
hasClaude: boolean;
|
|
160
160
|
}): InstallPlan[];
|
|
161
|
+
/** One vigiles-managed Codex hook: a `[[hooks.<event>]]` entry. */
|
|
162
|
+
export interface CodexPluginHook {
|
|
163
|
+
readonly event: string;
|
|
164
|
+
/** Codex matcher (anchored regex, the dialect convention). */
|
|
165
|
+
readonly matcher: string;
|
|
166
|
+
/** The shell command Codex runs (a direct `npx vigiles …`, no plugin root). */
|
|
167
|
+
readonly command: string;
|
|
168
|
+
/** A unique command substring → idempotent re-merge (replaces in place). */
|
|
169
|
+
readonly key: string;
|
|
170
|
+
}
|
|
171
|
+
/**
|
|
172
|
+
* vigiles's proactive nudges, wired into a Codex repo's `.codex/config.toml`.
|
|
173
|
+
*
|
|
174
|
+
* Codex has no global plugin store (unlike Claude Code's marketplace), so its
|
|
175
|
+
* config is repo-committed — the idiomatic place for these. They run as DIRECT
|
|
176
|
+
* `npx vigiles hook-runtime …` commands (NOT vendored bash scripts): the runtime
|
|
177
|
+
* entrypoints read the event JSON on stdin and emit the `hookSpecificOutput.
|
|
178
|
+
* additionalContext` shape Codex honors on `PostToolUse` (confirmed against the
|
|
179
|
+
* official hooks docs + encoded in `HookProtocol.injectableEvents`). Safety: only
|
|
180
|
+
* an INTENTIONAL `exit 2` blocks an edit (the refs nudge, when `unmarked-refs` is
|
|
181
|
+
* `error`); an npx-resolution failure exits non-2, so a missing dep never blocks.
|
|
182
|
+
*
|
|
183
|
+
* Deliberately NOT here (a loud, documented deferral — no-silent-skips): the
|
|
184
|
+
* SessionStart lint summary (CC delivers it as plain stdout, whose SessionStart
|
|
185
|
+
* prepend is unconfirmed on Codex — vs the JSON `additionalContext` these use) and
|
|
186
|
+
* the compile-on-edit / pre-edit-block guards (filename-gated bash with no
|
|
187
|
+
* harness-neutral `hook-runtime` entrypoint yet). Those stay manual on Codex.
|
|
188
|
+
*/
|
|
189
|
+
export declare function codexPluginHooks(): CodexPluginHook[];
|
|
190
|
+
/**
|
|
191
|
+
* Idempotently merge {@link codexPluginHooks} into a parsed `.codex/config.toml`
|
|
192
|
+
* object. Pure — the IO (read/parse/serialize/write) stays in cli.ts's
|
|
193
|
+
* `wireCodexHooks`. Each vigiles hook is keyed by a unique command substring, so
|
|
194
|
+
* a re-run REPLACES its own entry in place and leaves the user's own Codex hooks
|
|
195
|
+
* (and every other config key) untouched. Returns a new object.
|
|
196
|
+
*/
|
|
197
|
+
export declare function applyCodexPluginHooks(existing: Record<string, unknown>): Record<string, unknown>;
|
|
161
198
|
/**
|
|
162
199
|
* Resolve the final plan: defaults, then flags, then interactive answers (each
|
|
163
200
|
* layer overrides the previous only where it has an opinion). `--target` pins a
|
package/dist/setup-plan.js
CHANGED
|
@@ -17,6 +17,8 @@ exports.mergeProjectConfig = mergeProjectConfig;
|
|
|
17
17
|
exports.shouldPrompt = shouldPrompt;
|
|
18
18
|
exports.collectSetupAnswers = collectSetupAnswers;
|
|
19
19
|
exports.planPluginInstall = planPluginInstall;
|
|
20
|
+
exports.codexPluginHooks = codexPluginHooks;
|
|
21
|
+
exports.applyCodexPluginHooks = applyCodexPluginHooks;
|
|
20
22
|
exports.resolvePlan = resolvePlan;
|
|
21
23
|
function flagValue(args, prefix) {
|
|
22
24
|
return args.find((a) => a.startsWith(prefix))?.slice(prefix.length);
|
|
@@ -260,8 +262,10 @@ function planPluginInstall(harnesses, opts) {
|
|
|
260
262
|
if (harness === "codex") {
|
|
261
263
|
// The cross-agent `skills` CLI with `-g -y` installs to the global store
|
|
262
264
|
// ~/.agents/skills/ (NOT the repo, and NOT ~/.codex/ — verified against
|
|
263
|
-
// the real CLI). Skills
|
|
264
|
-
// are
|
|
265
|
+
// the real CLI). Skills install globally; the proactive NUDGE hooks
|
|
266
|
+
// (eval-lock + refs) are wired into the repo's .codex/config.toml by
|
|
267
|
+
// `init` (see codexPluginHooks / wireCodexHooks) — Codex config is
|
|
268
|
+
// repo-committed, so that's the idiomatic place.
|
|
265
269
|
return {
|
|
266
270
|
harness,
|
|
267
271
|
commands: ["npx --yes skills add zernie/vigiles -a codex -g -y"],
|
|
@@ -269,9 +273,10 @@ function planPluginInstall(harnesses, opts) {
|
|
|
269
273
|
manualSteps: ["npx skills add zernie/vigiles -a codex -g -y"],
|
|
270
274
|
notes: [
|
|
271
275
|
"Codex reads AGENTS.md directly; the skills install globally to ~/.agents/skills/ (not the repo).",
|
|
272
|
-
"
|
|
276
|
+
"The eval-lock + refs NUDGE hooks are wired into .codex/config.toml (repo-committed, the Codex norm).",
|
|
277
|
+
"Still manual on Codex: the SessionStart lint summary + compile-on-edit/pre-edit guards (no harness-neutral entrypoint yet).",
|
|
273
278
|
],
|
|
274
|
-
vendors:
|
|
279
|
+
vendors: true,
|
|
275
280
|
};
|
|
276
281
|
}
|
|
277
282
|
return {
|
|
@@ -284,6 +289,63 @@ function planPluginInstall(harnesses, opts) {
|
|
|
284
289
|
};
|
|
285
290
|
});
|
|
286
291
|
}
|
|
292
|
+
/**
|
|
293
|
+
* vigiles's proactive nudges, wired into a Codex repo's `.codex/config.toml`.
|
|
294
|
+
*
|
|
295
|
+
* Codex has no global plugin store (unlike Claude Code's marketplace), so its
|
|
296
|
+
* config is repo-committed — the idiomatic place for these. They run as DIRECT
|
|
297
|
+
* `npx vigiles hook-runtime …` commands (NOT vendored bash scripts): the runtime
|
|
298
|
+
* entrypoints read the event JSON on stdin and emit the `hookSpecificOutput.
|
|
299
|
+
* additionalContext` shape Codex honors on `PostToolUse` (confirmed against the
|
|
300
|
+
* official hooks docs + encoded in `HookProtocol.injectableEvents`). Safety: only
|
|
301
|
+
* an INTENTIONAL `exit 2` blocks an edit (the refs nudge, when `unmarked-refs` is
|
|
302
|
+
* `error`); an npx-resolution failure exits non-2, so a missing dep never blocks.
|
|
303
|
+
*
|
|
304
|
+
* Deliberately NOT here (a loud, documented deferral — no-silent-skips): the
|
|
305
|
+
* SessionStart lint summary (CC delivers it as plain stdout, whose SessionStart
|
|
306
|
+
* prepend is unconfirmed on Codex — vs the JSON `additionalContext` these use) and
|
|
307
|
+
* the compile-on-edit / pre-edit-block guards (filename-gated bash with no
|
|
308
|
+
* harness-neutral `hook-runtime` entrypoint yet). Those stay manual on Codex.
|
|
309
|
+
*/
|
|
310
|
+
function codexPluginHooks() {
|
|
311
|
+
// Codex's file-edit tool is `apply_patch` (its dialect vocabulary —
|
|
312
|
+
// src/adapters/codex/dialect.ts), NOT Claude's `Edit`/`Write`. A PostToolUse
|
|
313
|
+
// matcher keyed on CC tool names would never fire on Codex, so the nudges must
|
|
314
|
+
// match the Codex tool name. (Both nudge entrypoints also self-gate on the
|
|
315
|
+
// edited file, so a non-edit event no-ops regardless.)
|
|
316
|
+
return [
|
|
317
|
+
{
|
|
318
|
+
event: "PostToolUse",
|
|
319
|
+
matcher: "^apply_patch$",
|
|
320
|
+
command: "npx --no-install vigiles hook-runtime eval-lock-nudge",
|
|
321
|
+
key: "hook-runtime eval-lock-nudge",
|
|
322
|
+
},
|
|
323
|
+
{
|
|
324
|
+
event: "PostToolUse",
|
|
325
|
+
matcher: "^apply_patch$",
|
|
326
|
+
command: "npx --no-install vigiles hook-runtime refs",
|
|
327
|
+
key: "hook-runtime refs",
|
|
328
|
+
},
|
|
329
|
+
];
|
|
330
|
+
}
|
|
331
|
+
/**
|
|
332
|
+
* Idempotently merge {@link codexPluginHooks} into a parsed `.codex/config.toml`
|
|
333
|
+
* object. Pure — the IO (read/parse/serialize/write) stays in cli.ts's
|
|
334
|
+
* `wireCodexHooks`. Each vigiles hook is keyed by a unique command substring, so
|
|
335
|
+
* a re-run REPLACES its own entry in place and leaves the user's own Codex hooks
|
|
336
|
+
* (and every other config key) untouched. Returns a new object.
|
|
337
|
+
*/
|
|
338
|
+
function applyCodexPluginHooks(existing) {
|
|
339
|
+
const config = existing;
|
|
340
|
+
const hooks = {
|
|
341
|
+
...(config.hooks ?? {}),
|
|
342
|
+
};
|
|
343
|
+
for (const h of codexPluginHooks()) {
|
|
344
|
+
const kept = (hooks[h.event] ?? []).filter((e) => !e.command.includes(h.key));
|
|
345
|
+
hooks[h.event] = [...kept, { matcher: h.matcher, command: h.command }];
|
|
346
|
+
}
|
|
347
|
+
return { ...existing, hooks };
|
|
348
|
+
}
|
|
287
349
|
/**
|
|
288
350
|
* Resolve the final plan: defaults, then flags, then interactive answers (each
|
|
289
351
|
* layer overrides the previous only where it has an opinion). `--target` pins a
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# PostToolUse hook — after the agent edits an eval input (a SKILL.md trigger
|
|
3
|
+
# surface or an *.eval.* script) and committed eval locks exist, inject a
|
|
4
|
+
# NON-BLOCKING reminder to re-run `vigiles eval --update`. Self-gating: silent
|
|
5
|
+
# until you've committed a lock, so it never fires in a repo that doesn't use
|
|
6
|
+
# evals. Never blocks the edit. Runs as its OWN PostToolUse entry so its stdout
|
|
7
|
+
# stays clean JSON. See docs/harness-testing.md (the eval lock).
|
|
8
|
+
|
|
9
|
+
set -uo pipefail
|
|
10
|
+
|
|
11
|
+
INPUT=$(cat)
|
|
12
|
+
|
|
13
|
+
# No npx / not a Node project → nothing to do, never disrupt the edit.
|
|
14
|
+
command -v npx >/dev/null 2>&1 || exit 0
|
|
15
|
+
[ -f package.json ] || exit 0
|
|
16
|
+
|
|
17
|
+
printf '%s' "$INPUT" | npx vigiles hook-runtime eval-lock-nudge
|
|
18
|
+
|
|
19
|
+
# Always exit 0 — this hook only nudges, it never blocks (the gate is CI's
|
|
20
|
+
# `vigiles eval --check`). Swallow any tool/setup error so it can't disrupt edits.
|
|
21
|
+
exit 0
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "12.0.0",
|
|
4
4
|
"description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|
|
@@ -190,6 +190,33 @@ job that asserts the capability is present, run **`vigiles test --no-skip`** so
|
|
|
190
190
|
skipped tier fails — a green-with-skips is untested surface. Keep unit +
|
|
191
191
|
deterministic tests in CI (free); run evals locally or on a schedule with auth.
|
|
192
192
|
|
|
193
|
+
## Step 5 — Lock the eval so CI stays honest (you do this automatically)
|
|
194
|
+
|
|
195
|
+
Real-model evals run on the user's subscription — locally, never in CI. So **as
|
|
196
|
+
part of writing an eval, you keep its result fresh for them.** Do these two things
|
|
197
|
+
without being asked:
|
|
198
|
+
|
|
199
|
+
1. **Give every eval a `name`.** That's what the lock keys on.
|
|
200
|
+
2. **After running it, commit the lock:**
|
|
201
|
+
|
|
202
|
+
```bash
|
|
203
|
+
vigiles eval --update # records the result → .vigiles/eval-locks/<name>.lock.json
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
Then commit that file. CI runs `vigiles eval --check` (no model) to verify it
|
|
207
|
+
still matches the inputs — so a later edit that forgets to re-eval fails loud
|
|
208
|
+
instead of shipping stale numbers.
|
|
209
|
+
|
|
210
|
+
**When you later change a skill's description or prompts, re-run `vigiles eval
|
|
211
|
+
--update` and commit the updated lock** — the change altered what the eval
|
|
212
|
+
measures. (vigiles also nudges you: when a lock exists, a `SKILL.md` edit triggers
|
|
213
|
+
a non-blocking reminder.)
|
|
214
|
+
|
|
215
|
+
Why it's cheap: `--check` only hashes inputs (skill text, prompts, model). A
|
|
216
|
+
**threshold** change in the test re-uses the saved numbers (no model); only an
|
|
217
|
+
**input** change needs a fresh `--update`. Full mechanics:
|
|
218
|
+
[`docs/harness-testing.md`](../../docs/harness-testing.md#keep-eval-results-fresh-in-ci-the-lock).
|
|
219
|
+
|
|
193
220
|
## When the user didn't say what to test
|
|
194
221
|
|
|
195
222
|
Don't ask them to specify — **pick something real and demonstrate.** Scan the
|