vigiles 11.0.0 → 12.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.claude-plugin/plugin.json +9 -0
  2. package/README.md +10 -5
  3. package/action.yml +13 -2
  4. package/dist/adapter-conformance.js +6 -0
  5. package/dist/adapter-registry.d.ts +20 -0
  6. package/dist/adapter-registry.js +27 -0
  7. package/dist/adapters/claude-code/hook-protocol.js +4 -0
  8. package/dist/adapters/claude-code/runtime.js +12 -0
  9. package/dist/adapters/codex/eval.js +3 -0
  10. package/dist/adapters/codex/hook-protocol.d.ts +9 -1
  11. package/dist/adapters/codex/hook-protocol.js +10 -0
  12. package/dist/adapters/codex/runtime.js +10 -0
  13. package/dist/adapters/opencode/runtime.js +4 -0
  14. package/dist/claude-code.d.ts +2 -0
  15. package/dist/claude-code.js +9 -1
  16. package/dist/cli-commands.d.ts +1 -1
  17. package/dist/cli-commands.js +1 -0
  18. package/dist/cli.js +245 -29
  19. package/dist/core/hook-protocol.d.ts +15 -0
  20. package/dist/core/rule-meta.js +8 -0
  21. package/dist/core/runtime.d.ts +20 -0
  22. package/dist/core/skill-description-budget.d.ts +42 -0
  23. package/dist/core/skill-description-budget.js +47 -0
  24. package/dist/core/types.d.ts +21 -0
  25. package/dist/core/validate.js +3 -0
  26. package/dist/doc-command-coverage.d.ts +20 -0
  27. package/dist/doc-command-coverage.js +60 -0
  28. package/dist/eval-cache.d.ts +6 -0
  29. package/dist/eval-cache.js +2 -0
  30. package/dist/eval-cost.d.ts +75 -0
  31. package/dist/eval-cost.js +134 -0
  32. package/dist/eval-lock.d.ts +192 -0
  33. package/dist/eval-lock.js +286 -0
  34. package/dist/eval.d.ts +37 -20
  35. package/dist/eval.js +227 -56
  36. package/dist/research-index.d.ts +31 -0
  37. package/dist/research-index.js +48 -0
  38. package/dist/scan-behavioral.d.ts +42 -0
  39. package/dist/scan-behavioral.js +67 -0
  40. package/dist/scan.d.ts +3 -23
  41. package/dist/scan.js +18 -69
  42. package/dist/setup-plan.d.ts +38 -1
  43. package/dist/setup-plan.js +67 -4
  44. package/hooks/eval-lock-nudge.sh +21 -0
  45. package/package.json +1 -1
  46. package/skills/adopt-spec/SKILL.md +10 -1
  47. package/skills/edit-spec/SKILL.md +1 -0
  48. package/skills/strengthen/SKILL.md +4 -0
  49. package/skills/test-harness/SKILL.md +44 -0
@@ -0,0 +1,286 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.DEFAULT_LOCK_DIR = exports.LOCK_VERSION = void 0;
4
+ exports.evalInputsHash = evalInputsHash;
5
+ exports.lockSlug = lockSlug;
6
+ exports.lockPath = lockPath;
7
+ exports.readLock = readLock;
8
+ exports.anyLocksCommitted = anyLocksCommitted;
9
+ exports.isEvalInputFile = isEvalInputFile;
10
+ exports.evalLockNudge = evalLockNudge;
11
+ exports.writeLock = writeLock;
12
+ exports.buildLock = buildLock;
13
+ exports.decideLock = decideLock;
14
+ exports.diffReportNumbers = diffReportNumbers;
15
+ exports.formatLockUpdate = formatLockUpdate;
16
+ exports.lockModeFromEnv = lockModeFromEnv;
17
+ exports.evalApiVersionFromEnv = evalApiVersionFromEnv;
18
+ /**
19
+ * vigiles — the eval LOCK (the CI staleness gate for evals run on a subscription).
20
+ *
21
+ * Real-model evals authenticate as your own `claude` CLI on your Claude
22
+ * subscription, so they only run **locally** — never in CI (no metered key, and
23
+ * the subscription can't be driven from a headless runner). That leaves a hole:
24
+ * how does CI know the committed eval numbers still match the current inputs?
25
+ * Someone edits a skill, forgets to re-eval, and ships stale results.
26
+ *
27
+ * The lock closes it with an **integrity hash**, NOT a cache. The two are
28
+ * different mechanisms and must not be confused:
29
+ *
30
+ * - the eval CACHE ({@link ./eval-cache}) is a LOCAL speed optimization
31
+ * (gitignored, keyed store, skips the model call when re-scoring `measure`);
32
+ * - the eval LOCK is a COMMITTED staleness stamp (`.vigiles/eval-locks/<slug>.lock.json`),
33
+ * reviewed in the git diff, checked in CI without ever touching the model.
34
+ *
35
+ * It is the snapshot/lockfile pattern (`Cargo.lock` + `npm ci`; `jest --ci` /
36
+ * `cargo-insta`): you produce numbers locally with `--update`, commit the lock,
37
+ * and CI runs `--check` — recompute the input hash, compare, and fail "stale,
38
+ * re-run `--update`" on a mismatch. The committed diff of `recall: 0.90 → 0.65`
39
+ * IS the quality gate a human reviews. The same integrity-hash-of-inputs idea
40
+ * vigiles already ships in `core/integrity.ts` (compiled markdown) and
41
+ * `core/sidecar.ts` (spec inputs), applied a third time to eval results.
42
+ *
43
+ * Honest scope (no fiction): the lock promises "your committed results match your
44
+ * current inputs," NOT "your results reflect current model behavior." Model /
45
+ * harness drift is only caught when YOU re-run `--update` locally — there is no
46
+ * automated live run, by design. The clean split that makes replay sound: the
47
+ * lock stores only the model's OBSERVED BEHAVIOR (the report); the script's own
48
+ * assertions (`assertTriggerRate` / `assertSignificant`) re-run live against the
49
+ * replayed report, so a threshold-only edit is a valid replay (no model call)
50
+ * while an input change is stale. See `research/cache-invalidation.md`.
51
+ *
52
+ * Pure + model-free (the only side effects are the two small fs helpers); the
53
+ * inputs hash reuses `canonical` from `eval-cache.ts` so the lock and the cache
54
+ * canonicalize identically.
55
+ */
56
+ const node_fs_1 = require("node:fs");
57
+ const node_path_1 = require("node:path");
58
+ const hash_js_1 = require("./core/hash.js");
59
+ const eval_cache_js_1 = require("./eval-cache.js");
60
+ /**
61
+ * On-disk lock-format version, salted into nothing (the lock is keyed by name,
62
+ * not by hash) but VALIDATED on read so an incompatible shape fails loud rather
63
+ * than deserializing into a stale structure. Bump on a breaking shape change.
64
+ */
65
+ exports.LOCK_VERSION = 1;
66
+ /** Default directory for committed eval locks (tracked, NOT gitignored). */
67
+ exports.DEFAULT_LOCK_DIR = ".vigiles/eval-locks";
68
+ /**
69
+ * Why the harness binary version is **NOT** hashed (only recorded as provenance):
70
+ * `--check` runs in CI where `claude` is PINNED to a fixed version, while a dev's
71
+ * local `claude` is whatever they have — folding the version into the hash would
72
+ * false-trip `--check` on every PR where those differ. It is also the lock's
73
+ * honest scope: the gate verifies your committed results match your current
74
+ * *author-controlled inputs*, not current model/harness behavior (there is no
75
+ * automated live run). Harness/model drift is caught when YOU re-run `--update`
76
+ * locally and review the moved numbers in the git diff. Keeping the version out
77
+ * of the hash is what lets `--check` stay binary-free + deterministic in CI.
78
+ * (The eval CACHE still keys on it — that's local replay soundness, a different
79
+ * axis.) See research/cache-invalidation.md.
80
+ */
81
+ /** Deterministic content hash of a lock's model-affecting inputs. */
82
+ function evalInputsHash(input) {
83
+ return (0, hash_js_1.sha256short)(JSON.stringify((0, eval_cache_js_1.canonical)(input)));
84
+ }
85
+ /** Filesystem-safe slug for a report name (the lock filename). */
86
+ function lockSlug(name) {
87
+ const slug = name
88
+ .toLowerCase()
89
+ .replace(/[^a-z0-9]+/g, "-")
90
+ .replace(/^-+|-+$/g, "");
91
+ return slug || "eval";
92
+ }
93
+ /** Path to a named eval's lock file under `dir`. */
94
+ function lockPath(dir, name) {
95
+ return (0, node_path_1.join)(dir, `${lockSlug(name)}.lock.json`);
96
+ }
97
+ /**
98
+ * Read a named eval's lock. A MISS (no file) returns `null`. A CORRUPT or
99
+ * wrong-version file **throws** — a broken lock is a real failure the CI gate
100
+ * must surface, not silently treat as "no lock" (which would let a stale eval
101
+ * pass). The message says how to recover.
102
+ */
103
+ function readLock(dir, name) {
104
+ const path = lockPath(dir, name);
105
+ if (!(0, node_fs_1.existsSync)(path))
106
+ return null;
107
+ const raw = (0, node_fs_1.readFileSync)(path, "utf-8");
108
+ let data;
109
+ try {
110
+ data = JSON.parse(raw);
111
+ }
112
+ catch {
113
+ throw new Error(`eval lock: corrupt lock ${path} (invalid JSON) — delete it and re-run \`vigiles eval --update\``);
114
+ }
115
+ if (typeof data !== "object" || data === null)
116
+ throw new Error(`eval lock: ${path} is not a JSON object`);
117
+ const obj = data;
118
+ if (obj.version !== exports.LOCK_VERSION)
119
+ throw new Error(`eval lock: ${path} has unsupported version ${String(obj.version)} ` +
120
+ `(expected ${String(exports.LOCK_VERSION)}) — re-run \`vigiles eval --update\``);
121
+ // Slug collision guard: two distinct names can normalize to the same file
122
+ // (`foo/bar` and `foo bar` → `foo-bar.lock.json`). A lock whose stored `name`
123
+ // differs from the one requested belongs to the OTHER eval — treat it as a
124
+ // MISS (not this eval's lock) so `--check` degrades to "stale → re-run" and
125
+ // NEVER replays the wrong eval's report. The stored name is the source of truth.
126
+ if (obj.name !== name)
127
+ return null;
128
+ return obj;
129
+ }
130
+ /**
131
+ * Whether ANY lock has been committed under `dir`. The CI staleness gate
132
+ * (`eval --check`) uses this to stay a NO-OP until the feature is in use: a repo
133
+ * that has never run `eval --update` has no locks, so there is nothing to verify
134
+ * and CI passes green. Once the first lock is committed, every named eval is held
135
+ * to having a fresh one (a new unlocked eval then reads as stale). The graduated,
136
+ * opt-in-by-committing behavior that keeps a fresh `init` from going red.
137
+ */
138
+ function anyLocksCommitted(dir) {
139
+ if (!(0, node_fs_1.existsSync)(dir))
140
+ return false;
141
+ return (0, node_fs_1.readdirSync)(dir).some((f) => f.endsWith(".lock.json"));
142
+ }
143
+ /**
144
+ * Does an edited path plausibly change an eval's INPUTS — so a committed lock may
145
+ * now be stale? Two surfaces feed the hash: a skill's trigger surface (`SKILL.md`)
146
+ * and the eval script that holds the prompts/spec (`*.eval.{mjs,cjs,js,mts,cts,ts}`).
147
+ * Pure (string-only) so the nudge hook stays cheap and never runs an eval script.
148
+ */
149
+ function isEvalInputFile(path) {
150
+ const p = path.replace(/\\/g, "/");
151
+ if (/(^|\/)SKILL\.md$/.test(p))
152
+ return true;
153
+ return /\.eval\.(mjs|cjs|js|mts|cts|ts)$/.test(p);
154
+ }
155
+ /**
156
+ * The NON-BLOCKING nudge to emit after an eval-input edit when committed locks
157
+ * exist, or `null` for no nudge. Self-gating: it stays silent until you've opted
158
+ * into the lock (committed one), so it can't annoy a repo that doesn't use evals.
159
+ * It deliberately does NOT recompute staleness (that needs the eval script + is
160
+ * the job of `eval --check`) — a reminder, not a gate. The honest harness-neutral
161
+ * reminder; how it reaches the agent (both CC and Codex inject `additionalContext`
162
+ * on `PostToolUse`) is the caller's concern. See docs/harness-testing-*.md.
163
+ */
164
+ function evalLockNudge(filePath, lockDir) {
165
+ if (!isEvalInputFile(filePath))
166
+ return null;
167
+ if (!anyLocksCommitted(lockDir))
168
+ return null;
169
+ return (`vigiles: you edited ${filePath}, which can change an eval's inputs — a ` +
170
+ `committed eval lock may now be stale. When you're done, run ` +
171
+ `\`vigiles eval --update\` (local, on your subscription) and commit the ` +
172
+ `updated lock; CI's \`vigiles eval --check\` will otherwise flag it stale. ` +
173
+ `This is a reminder, not a block.`);
174
+ }
175
+ /** Write a named eval's lock (pretty JSON for a reviewable git diff). */
176
+ function writeLock(dir, lock) {
177
+ (0, node_fs_1.mkdirSync)(dir, { recursive: true });
178
+ (0, node_fs_1.writeFileSync)(lockPath(dir, lock.name), JSON.stringify(lock, null, 2) + "\n");
179
+ }
180
+ /**
181
+ * Build a fresh lock envelope from a just-recorded report (the `--update` write).
182
+ * `builtAt` is passed in (never read from the clock here) so the module stays
183
+ * pure + deterministically testable; the CLI stamps the real timestamp.
184
+ */
185
+ function buildLock(args) {
186
+ return { version: exports.LOCK_VERSION, ...args };
187
+ }
188
+ /**
189
+ * Decide what `check` mode should do given the current input hash and the
190
+ * committed lock. `off`/`update` always `run` (update records afterwards). `check`
191
+ * replays a matching lock (no model) and is `stale` on a missing lock or a hash
192
+ * mismatch — the deterministic CI gate.
193
+ */
194
+ function decideLock(mode, name, currentHash, existing) {
195
+ if (mode !== "check")
196
+ return { kind: "run" };
197
+ if (!existing)
198
+ return {
199
+ kind: "stale",
200
+ reason: `eval lock missing for "${name}" — no committed results to verify against. ` +
201
+ `Run \`vigiles eval --update\` locally (on your subscription) and commit the lock.`,
202
+ };
203
+ if (existing.inputsHash !== currentHash)
204
+ return {
205
+ kind: "stale",
206
+ reason: `eval lock STALE for "${name}" — the inputs changed since the committed results ` +
207
+ `were recorded (skill/prompts/model/harness/apiVersion). Re-run ` +
208
+ `\`vigiles eval --update\` locally and commit the updated lock.`,
209
+ };
210
+ return { kind: "replay", report: existing.report };
211
+ }
212
+ /**
213
+ * Collect the numeric leaves that changed between two recorded reports — the
214
+ * human-facing delta printed at `--update` time (e.g. `rate: 0.900 → 0.650`).
215
+ * Generic over any report shape (walks numbers by dotted path), so it works for
216
+ * `EvalReport`, `TriggerRateReport`, and `CheckReport` without per-type code. The
217
+ * committed git diff is the primary review surface; this is the at-a-glance echo.
218
+ */
219
+ function diffReportNumbers(before, after) {
220
+ const out = [];
221
+ walkNumberLeaves(before, after, "", out);
222
+ return out;
223
+ }
224
+ function walkNumberLeaves(a, b, path, out) {
225
+ if (typeof a === "number" && typeof b === "number") {
226
+ if (a !== b)
227
+ out.push({ path, before: a, after: b });
228
+ }
229
+ else if (Array.isArray(a) && Array.isArray(b)) {
230
+ walkArrayLeaves(a, b, path, out);
231
+ }
232
+ else if (isRecord(a) && isRecord(b)) {
233
+ walkRecordLeaves(a, b, path, out);
234
+ }
235
+ }
236
+ function walkArrayLeaves(a, b, path, out) {
237
+ const len = Math.min(a.length, b.length);
238
+ for (let i = 0; i < len; i++)
239
+ walkNumberLeaves(a[i], b[i], `${path}[${String(i)}]`, out);
240
+ }
241
+ function walkRecordLeaves(a, b, path, out) {
242
+ for (const k of Object.keys(a))
243
+ if (k in b)
244
+ walkNumberLeaves(a[k], b[k], path ? `${path}.${k}` : k, out);
245
+ }
246
+ function isRecord(v) {
247
+ return v !== null && typeof v === "object";
248
+ }
249
+ /** Render the `--update` result for a human: NEW lock, or the per-number deltas. */
250
+ function formatLockUpdate(name, deltas, isNew) {
251
+ if (isNew)
252
+ return `eval lock: recorded NEW lock for "${name}"`;
253
+ if (deltas.length === 0)
254
+ return `eval lock: "${name}" updated — no numeric change vs the prior lock`;
255
+ const lines = [
256
+ `eval lock: "${name}" updated — ${String(deltas.length)} value(s) moved:`,
257
+ ];
258
+ for (const d of deltas) {
259
+ const dir = d.after > d.before ? "▲" : "▼";
260
+ lines.push(` ${dir} ${d.path}: ${d.before.toFixed(3)} → ${d.after.toFixed(3)}`);
261
+ }
262
+ lines.push(" review the committed lock diff — this is the eval quality gate.");
263
+ return lines.join("\n");
264
+ }
265
+ /**
266
+ * Read the lock mode from the environment (`VIGILES_EVAL_LOCK`), set by the CLI's
267
+ * `eval --check` / `--update` flags. A run knob (like `VIGILES_TRIALS`): the CLI
268
+ * is the only place that should set it. Anything unrecognized → `off`.
269
+ */
270
+ function lockModeFromEnv(env = process.env) {
271
+ const v = env.VIGILES_EVAL_LOCK;
272
+ return v === "check" || v === "update" ? v : "off";
273
+ }
274
+ /**
275
+ * The behavior epoch (`evalApiVersion`) for this run, read from the env the CLI
276
+ * populates from `.vigilesrc.json` `eval.apiVersion`. Default 1. A malformed
277
+ * value falls back to 1 (never throws) — the lock stays usable.
278
+ */
279
+ function evalApiVersionFromEnv(env = process.env) {
280
+ const raw = env.VIGILES_EVAL_API_VERSION;
281
+ if (raw === undefined)
282
+ return 1;
283
+ const n = Number.parseInt(raw, 10);
284
+ return Number.isFinite(n) && n >= 0 ? n : 1;
285
+ }
286
+ //# sourceMappingURL=eval-lock.js.map
package/dist/eval.d.ts CHANGED
@@ -1,5 +1,6 @@
1
1
  import { parseToolCalls, parseHooks, parseSubagents, type ToolCall, type Trace } from "./harness-test.js";
2
2
  import { type CacheMode } from "./eval-cache.js";
3
+ import { type EvalLockOptions } from "./eval-lock.js";
3
4
  import type { Check, CheckJSON } from "./check.js";
4
5
  import { type Comparison } from "./stats.js";
5
6
  import { type ToolIntercept } from "./tool-intercept.js";
@@ -159,6 +160,16 @@ export interface EvalSpec<M extends Metrics> {
159
160
  * to today). See {@link ToolStub} and `research/eval-coverage-and-isolation.md`.
160
161
  */
161
162
  readonly stubs?: readonly ToolStub[];
163
+ /**
164
+ * **The eval LOCK** — a committed staleness gate for CI (see `src/eval-lock.ts`).
165
+ * With a `name` set, `vigiles eval --update` (local, on your subscription)
166
+ * records the report to `.vigiles/eval-locks/<name>.lock.json`; `--check` (CI)
167
+ * verifies the committed result against the current inputs WITHOUT a model call,
168
+ * failing "stale" when they diverge. Mode normally comes from the CLI; set
169
+ * `lock` to drive it programmatically or point a test at a throwaway dir. The
170
+ * lock engages only when `name` is set (it keys the lock file).
171
+ */
172
+ readonly lock?: EvalLockOptions;
162
173
  }
163
174
  /** Per-metric summary statistics across an arm's runs. */
164
175
  export interface MetricStat {
@@ -316,6 +327,8 @@ export interface CheckRate {
316
327
  export interface CheckReport {
317
328
  readonly n: number;
318
329
  readonly perCheck: readonly CheckRate[];
330
+ /** Cost / latency / token totals for the run (the same source as `runEval`). */
331
+ readonly usage: ArmUsage;
319
332
  }
320
333
  /**
321
334
  * Score a check vocabulary across trials — the scored counterpart to
@@ -479,16 +492,6 @@ export declare function modelTier(id: string): number | null;
479
492
  * {@link modelTier}); an unrankable model/floor is never "below" (fail-open).
480
493
  */
481
494
  export declare function belowModelFloor(model: string, floor: string): boolean;
482
- /**
483
- * Reduce a raw `--version` string to the **major.minor** cache-key token. We key
484
- * the cache on major.minor, NOT the patch: a patch release rarely changes agent
485
- * behaviour, so keying patches would churn the cache on every release for no
486
- * signal; a minor/major bump is where the system prompt / tool defs actually move.
487
- * (If a specific patch is known to matter, clear the cache or bump
488
- * `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
489
- * found. Pure + tested.
490
- */
491
- export declare function harnessVersionKey(raw: string): string;
492
495
  /**
493
496
  * Build an **ephemeral run environment** for a model-driven run: a NEW env object
494
497
  * with a *fresh* `HOME` (and `TMPDIR`) pointed at the throwaway `opts.home`, only
@@ -542,15 +545,6 @@ export declare function seedEphemeralHome(throwawayHome: string, realHome: strin
542
545
  export declare function isRateLimited(out: RunOut): boolean;
543
546
  /** Map `worker` over `items` with at most `concurrency` in flight, order preserved. */
544
547
  export declare function runPool<T, R>(items: readonly T[], concurrency: number, worker: (item: T) => Promise<R>): Promise<R[]>;
545
- /**
546
- * The eval orchestration — every arm × trial via `runner`, run through the cache
547
- * and a rate-limit retry, with at most `concurrency` in flight and an optional
548
- * `maxCostUsd` budget cap; metric + usage computed per run and aggregated per
549
- * arm. Exported with an injectable `runner` so the loop, `measure` context,
550
- * caching, pooling, and aggregation are unit-testable without spawning a model
551
- * (pass a fake returning canned stream-json). `runEval` is this with the real
552
- * agent runner.
553
- */
554
548
  export declare function runEvalWith<M extends Metrics>(spec: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
555
549
  /** Format an eval report as a compact table for the console (mean ± se, pass^k). */
556
550
  export declare function formatEvalReport(report: EvalReport): string;
@@ -563,6 +557,19 @@ export declare function formatEvalReport(report: EvalReport): string;
563
557
  * (reuse the bare predicates, e.g. `(t) => skillResolved(t, "x:y")`).
564
558
  */
565
559
  export interface TriggerRateSpec {
560
+ /**
561
+ * A stable name for this trigger eval — required to engage the {@link lock}
562
+ * (it keys the committed `.vigiles/eval-locks/<name>.lock.json`). Two evals over
563
+ * the same skill with different prompts get distinct names. Optional otherwise.
564
+ */
565
+ readonly name?: string;
566
+ /**
567
+ * **The eval LOCK** — the CI staleness gate (see `src/eval-lock.ts`). With
568
+ * `name` set, `vigiles eval --update` records this trigger-rate report locally
569
+ * and `--check` verifies it against the current inputs (skill contents, prompts,
570
+ * model) with NO model call. Mode normally comes from the CLI flags.
571
+ */
572
+ readonly lock?: EvalLockOptions;
566
573
  /**
567
574
  * Plugin dir installed natively (`--plugin-dir`) so its skills/commands
568
575
  * activate. Provide this OR {@link skillsDir}, not both.
@@ -709,6 +716,8 @@ export interface TriggerRateReport {
709
716
  * `n` means the measurement is thin (e.g. a Codex usage limit was hit); re-run.
710
717
  */
711
718
  readonly errored?: number;
719
+ /** Cost / tokens SPENT across all runs (relevant + irrelevant) — feeds the cost summary. */
720
+ readonly usage: ArmUsage;
712
721
  }
713
722
  /**
714
723
  * An eval-tier transport: how to RUN a real harness turn and PARSE its output.
@@ -722,6 +731,14 @@ export interface EvalDriver {
722
731
  readonly runner: AgentRunner;
723
732
  readonly parse: ModelOutputParser;
724
733
  readonly runError?: (out: RunOut) => string | null;
734
+ /**
735
+ * The harness this driver runs (e.g. `"claude-code"`, `"codex"`). Folded into a
736
+ * trigger-rate eval's LOCK hash so a report recorded on one harness is marked
737
+ * STALE if the eval is later switched to another (a different harness can fire a
738
+ * skill differently). Optional for back-compat — absent defaults to
739
+ * `"claude-code"`, so an existing single-harness lock is unaffected.
740
+ */
741
+ readonly harness?: string;
725
742
  }
726
743
  /** The default (Claude Code) eval driver: real `claude` + stream-json parsing. */
727
744
  export declare const claudeEvalDriver: EvalDriver;
@@ -801,7 +818,7 @@ export declare function packageInstallSet(opts: {
801
818
  dir: string;
802
819
  added: number;
803
820
  };
804
- export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null): Promise<TriggerRateReport>;
821
+ export declare function measureTriggerRateWith(spec: TriggerRateSpec, runner: AgentRunner, parse?: ModelOutputParser, runError?: (out: RunOut) => string | null, harness?: string): Promise<TriggerRateReport>;
805
822
  /**
806
823
  * Measure a skill/behaviour's real trigger rate across prompts × trials. Defaults
807
824
  * to the real `claude` CLI (`claudeEvalDriver`); pass `{ evalDriver }` to drive a