vigiles 11.0.0 → 12.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.claude-plugin/plugin.json +9 -0
  2. package/README.md +10 -5
  3. package/action.yml +13 -2
  4. package/dist/adapter-conformance.js +6 -0
  5. package/dist/adapter-registry.d.ts +20 -0
  6. package/dist/adapter-registry.js +27 -0
  7. package/dist/adapters/claude-code/hook-protocol.js +4 -0
  8. package/dist/adapters/claude-code/runtime.js +12 -0
  9. package/dist/adapters/codex/eval.js +3 -0
  10. package/dist/adapters/codex/hook-protocol.d.ts +9 -1
  11. package/dist/adapters/codex/hook-protocol.js +10 -0
  12. package/dist/adapters/codex/runtime.js +10 -0
  13. package/dist/adapters/opencode/runtime.js +4 -0
  14. package/dist/claude-code.d.ts +2 -0
  15. package/dist/claude-code.js +9 -1
  16. package/dist/cli-commands.d.ts +1 -1
  17. package/dist/cli-commands.js +1 -0
  18. package/dist/cli.js +245 -29
  19. package/dist/core/hook-protocol.d.ts +15 -0
  20. package/dist/core/rule-meta.js +8 -0
  21. package/dist/core/runtime.d.ts +20 -0
  22. package/dist/core/skill-description-budget.d.ts +42 -0
  23. package/dist/core/skill-description-budget.js +47 -0
  24. package/dist/core/types.d.ts +21 -0
  25. package/dist/core/validate.js +3 -0
  26. package/dist/doc-command-coverage.d.ts +20 -0
  27. package/dist/doc-command-coverage.js +60 -0
  28. package/dist/eval-cache.d.ts +6 -0
  29. package/dist/eval-cache.js +2 -0
  30. package/dist/eval-cost.d.ts +75 -0
  31. package/dist/eval-cost.js +134 -0
  32. package/dist/eval-lock.d.ts +192 -0
  33. package/dist/eval-lock.js +286 -0
  34. package/dist/eval.d.ts +37 -20
  35. package/dist/eval.js +227 -56
  36. package/dist/research-index.d.ts +31 -0
  37. package/dist/research-index.js +48 -0
  38. package/dist/scan-behavioral.d.ts +42 -0
  39. package/dist/scan-behavioral.js +67 -0
  40. package/dist/scan.d.ts +3 -23
  41. package/dist/scan.js +18 -69
  42. package/dist/setup-plan.d.ts +38 -1
  43. package/dist/setup-plan.js +67 -4
  44. package/hooks/eval-lock-nudge.sh +21 -0
  45. package/package.json +1 -1
  46. package/skills/adopt-spec/SKILL.md +10 -1
  47. package/skills/edit-spec/SKILL.md +1 -0
  48. package/skills/strengthen/SKILL.md +4 -0
  49. package/skills/test-harness/SKILL.md +44 -0
@@ -0,0 +1,60 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.COVERAGE_EXEMPT = void 0;
4
+ exports.verbMentioned = verbMentioned;
5
+ exports.findUndocumentedVerbs = findUndocumentedVerbs;
6
+ /**
7
+ * Doc-command coverage — the INVERSE of self-command-refs, and the deterministic
8
+ * FLOOR under the `document-the-why` rule. self-command-refs checks that every
9
+ * `vigiles <cmd>` reference in the docs resolves to a REAL command (docs → code).
10
+ * This checks the other direction (code → docs): every public CLI VERB must be
11
+ * MENTIONED somewhere under `docs/`, so a verb shipped without a doc home is a
12
+ * failing test, not a thing a reader discovers is missing.
13
+ *
14
+ * HIGH-PRECISION, and biased toward NOT crying wolf: the risk here is a FALSE
15
+ * "undocumented" alarm on a verb that IS documented, so "mentioned" is matched
16
+ * GENEROUSLY — a verb counts as documented if it appears in a COMMAND context:
17
+ * `vigiles <verb>` (covers `npx vigiles <verb>`) OR a backtick immediately
18
+ * followed by the verb (`` `<verb>` ``, `` `<verb> ./x` ``). A bare English word
19
+ * ("test", "audit", "eval", "compile" all double as prose) is NOT enough — it
20
+ * must sit in a command context — so the check still fires on a genuinely
21
+ * undocumented verb while never flagging a documented one.
22
+ *
23
+ * `hook-runtime` is excluded by default: it is the HIDDEN runtime-entrypoint
24
+ * umbrella (cohesive-cli-surface keeps it OUT of the human verb surface), not a
25
+ * verb a user is expected to read about beside `audit`/`lint`. Source of truth
26
+ * for the verb set: {@link VERBS}.
27
+ */
28
+ const cli_commands_js_1 = require("./cli-commands.js");
29
+ /** The hidden runtime umbrella — not a human-facing verb to document. */
30
+ exports.COVERAGE_EXEMPT = ["hook-runtime"];
31
+ function escapeRegExp(s) {
32
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
33
+ }
34
+ /**
35
+ * Whether `verb` appears in a COMMAND context anywhere in `content`:
36
+ * `vigiles <verb>` or a backtick-prefixed `` `<verb> ``. Generous on purpose
37
+ * (see the file header) — over-counting a verb as documented is the SAFE
38
+ * direction; under-counting would cry wolf.
39
+ */
40
+ function verbMentioned(verb, content) {
41
+ const esc = escapeRegExp(verb);
42
+ return new RegExp(String.raw `(\bvigiles\s+|\x60)${esc}\b`).test(content);
43
+ }
44
+ /**
45
+ * Find public verbs not MENTIONED in any of the given doc files. Pure — the
46
+ * caller supplies file contents (so it runs over the repo's `docs/` in a test,
47
+ * or any file set). `verbs` defaults to the canonical {@link VERBS}; `exempt`
48
+ * drops the hidden umbrella.
49
+ */
50
+ function findUndocumentedVerbs(docs, verbs = cli_commands_js_1.VERBS, exempt = exports.COVERAGE_EXEMPT) {
51
+ const mentioned = new Set();
52
+ for (const { content } of docs) {
53
+ for (const verb of verbs) {
54
+ if (!mentioned.has(verb) && verbMentioned(verb, content))
55
+ mentioned.add(verb);
56
+ }
57
+ }
58
+ return verbs.filter((v) => !exempt.includes(v) && !mentioned.has(v));
59
+ }
60
+ //# sourceMappingURL=doc-command-coverage.js.map
@@ -43,6 +43,12 @@ export interface CacheRecord {
43
43
  /** Text files present in the cwd after the run (relative path → contents). */
44
44
  readonly files: Record<string, string>;
45
45
  }
46
+ /**
47
+ * Canonicalize a value so the key is stable regardless of object key order —
48
+ * recursively sorts object keys. Arrays keep order (it's significant for tools).
49
+ * Exported so the eval LOCK ({@link ./eval-lock}) hashes its inputs the same way.
50
+ */
51
+ export declare function canonical(value: unknown): unknown;
46
52
  /**
47
53
  * Cache record-format version, SALTED into every key (Jest `CACHE_VERSION` /
48
54
  * webpack `cache.version` pattern). Bump when the `CacheRecord` shape — or how a
@@ -1,6 +1,7 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.CACHE_FORMAT_VERSION = void 0;
4
+ exports.canonical = canonical;
4
5
  exports.cacheKey = cacheKey;
5
6
  exports.readCache = readCache;
6
7
  exports.writeCache = writeCache;
@@ -32,6 +33,7 @@ const SKIP_DIRS = new Set(["node_modules", ".git"]);
32
33
  /**
33
34
  * Canonicalize a value so the key is stable regardless of object key order —
34
35
  * recursively sorts object keys. Arrays keep order (it's significant for tools).
36
+ * Exported so the eval LOCK ({@link ./eval-lock}) hashes its inputs the same way.
35
37
  */
36
38
  function canonical(value) {
37
39
  if (Array.isArray(value))
@@ -0,0 +1,75 @@
1
+ /**
2
+ * Eval cost transparency — make what a real-model run SPENT impossible to miss.
3
+ * vigiles's whole affordability pitch is "runs on your Claude subscription, not a
4
+ * metered API," so every real-model run should say — out loud — how many tokens it
5
+ * spent, the API-equivalent dollar cost, and (loudly) if it was billed to a
6
+ * METERED API key instead of your subscription.
7
+ *
8
+ * HONEST SCOPE: we surface tokens + the API-equivalent `$` (`total_cost_usd` from
9
+ * the `claude` CLI) + a running session tally. We deliberately do NOT show a
10
+ * "% of your subscription" — Anthropic does not expose a subscription's quota or
11
+ * limit programmatically (and the real limits are rolling rate windows, not a
12
+ * dollar bucket), so any percentage would be fiction. See docs/eval-architecture.md.
13
+ *
14
+ * Pure + injectable (env + an output sink), so the whole thing is unit-tested
15
+ * without a model or a real key.
16
+ */
17
+ import type { EvalUsage, ArmUsage, EvalReport } from "./eval.js";
18
+ /** A normalized cost/token snapshot — the common shape a report renders from. */
19
+ export interface CostSummary {
20
+ /** API-equivalent cost (`total_cost_usd`) — the number that matters. */
21
+ readonly costUsd: number;
22
+ readonly inputTokens: number;
23
+ readonly outputTokens: number;
24
+ readonly cacheCreationTokens: number;
25
+ readonly cacheReadTokens: number;
26
+ }
27
+ /** Total tokens across all four billing buckets. */
28
+ export declare function totalTokens(c: CostSummary): number;
29
+ /** A per-run {@link EvalUsage} → the common snapshot. */
30
+ export declare function costFromRun(u: EvalUsage): CostSummary;
31
+ /** An aggregated per-arm {@link ArmUsage} → the common snapshot. */
32
+ export declare function costFromArm(u: ArmUsage): CostSummary;
33
+ /** Sum any number of snapshots (e.g. every arm of an A/B). */
34
+ export declare function sumCosts(costs: readonly CostSummary[]): CostSummary;
35
+ /** The whole-{@link EvalReport} cost — every arm summed. */
36
+ export declare function costFromEvalReport(report: EvalReport): CostSummary;
37
+ /**
38
+ * How the run was billed. `metered` is true when a real Anthropic API key is in
39
+ * the environment — the `claude` CLI bills those PER TOKEN, whereas a
40
+ * subscription run (auth via `~/.claude`, no key var) costs $0 beyond the sub.
41
+ * The mock/deterministic tier never reaches here (it has no real cost), so a
42
+ * present key means a real metered run.
43
+ */
44
+ export interface Billing {
45
+ readonly metered: boolean;
46
+ /** Which env var carried the key (for the actionable "unset X" message). */
47
+ readonly keyVar: string | null;
48
+ }
49
+ export declare function detectBilling(env?: NodeJS.ProcessEnv): Billing;
50
+ /** Add a run to the running session total and return the new total. */
51
+ export declare function recordSessionCost(c: CostSummary): CostSummary;
52
+ /** The session total so far. */
53
+ export declare function sessionCost(): CostSummary;
54
+ /** Reset the session tally (test seam). */
55
+ export declare function resetSessionCost(): void;
56
+ /**
57
+ * The human-readable cost block for a run. Shows tokens + API-equivalent `$`, the
58
+ * billed-to line (a LOUD warning + an actionable fix when metered, a green ✅ when
59
+ * on the subscription), and the session tally when it exceeds this run.
60
+ */
61
+ export declare function formatCostSummary(c: CostSummary, opts: {
62
+ billing: Billing;
63
+ session?: CostSummary | null;
64
+ }): string;
65
+ /**
66
+ * Record `c` into the session tally and emit its cost block. The default sink is
67
+ * stderr (so a run's cost never pollutes `--json` stdout). Injectable env + sink
68
+ * keep it fully testable. Returns the emitted text (also handy for a skill to
69
+ * relay to the user).
70
+ */
71
+ export declare function emitCostSummary(c: CostSummary, opts?: {
72
+ env?: NodeJS.ProcessEnv;
73
+ out?: (s: string) => void;
74
+ }): string;
75
+ //# sourceMappingURL=eval-cost.d.ts.map
@@ -0,0 +1,134 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.totalTokens = totalTokens;
4
+ exports.costFromRun = costFromRun;
5
+ exports.costFromArm = costFromArm;
6
+ exports.sumCosts = sumCosts;
7
+ exports.costFromEvalReport = costFromEvalReport;
8
+ exports.detectBilling = detectBilling;
9
+ exports.recordSessionCost = recordSessionCost;
10
+ exports.sessionCost = sessionCost;
11
+ exports.resetSessionCost = resetSessionCost;
12
+ exports.formatCostSummary = formatCostSummary;
13
+ exports.emitCostSummary = emitCostSummary;
14
+ const ZERO = {
15
+ costUsd: 0,
16
+ inputTokens: 0,
17
+ outputTokens: 0,
18
+ cacheCreationTokens: 0,
19
+ cacheReadTokens: 0,
20
+ };
21
+ /** Total tokens across all four billing buckets. */
22
+ function totalTokens(c) {
23
+ return (c.inputTokens + c.outputTokens + c.cacheCreationTokens + c.cacheReadTokens);
24
+ }
25
+ /** A per-run {@link EvalUsage} → the common snapshot. */
26
+ function costFromRun(u) {
27
+ return {
28
+ costUsd: u.costUsd,
29
+ inputTokens: u.inputTokens,
30
+ outputTokens: u.outputTokens,
31
+ cacheCreationTokens: u.cacheCreationTokens,
32
+ cacheReadTokens: u.cacheReadTokens,
33
+ };
34
+ }
35
+ /** An aggregated per-arm {@link ArmUsage} → the common snapshot. */
36
+ function costFromArm(u) {
37
+ return {
38
+ costUsd: u.totalCostUsd,
39
+ inputTokens: u.totalInputTokens,
40
+ outputTokens: u.totalOutputTokens,
41
+ cacheCreationTokens: u.totalCacheCreationTokens,
42
+ cacheReadTokens: u.totalCacheReadTokens,
43
+ };
44
+ }
45
+ /** Sum any number of snapshots (e.g. every arm of an A/B). */
46
+ function sumCosts(costs) {
47
+ return costs.reduce((a, c) => ({
48
+ costUsd: a.costUsd + c.costUsd,
49
+ inputTokens: a.inputTokens + c.inputTokens,
50
+ outputTokens: a.outputTokens + c.outputTokens,
51
+ cacheCreationTokens: a.cacheCreationTokens + c.cacheCreationTokens,
52
+ cacheReadTokens: a.cacheReadTokens + c.cacheReadTokens,
53
+ }), ZERO);
54
+ }
55
+ /** The whole-{@link EvalReport} cost — every arm summed. */
56
+ function costFromEvalReport(report) {
57
+ return sumCosts(Object.values(report.arms).map((a) => costFromArm(a.usage)));
58
+ }
59
+ const KEY_VARS = ["ANTHROPIC_API_KEY", "ANTHROPIC_AUTH_TOKEN"];
60
+ function detectBilling(env = process.env) {
61
+ for (const v of KEY_VARS) {
62
+ const val = env[v];
63
+ if (val !== undefined && val.trim() !== "")
64
+ return { metered: true, keyVar: v };
65
+ }
66
+ return { metered: false, keyVar: null };
67
+ }
68
+ // --- Session tally (within one process) ------------------------------------
69
+ let SESSION = ZERO;
70
+ /** Add a run to the running session total and return the new total. */
71
+ function recordSessionCost(c) {
72
+ SESSION = sumCosts([SESSION, c]);
73
+ return SESSION;
74
+ }
75
+ /** The session total so far. */
76
+ function sessionCost() {
77
+ return SESSION;
78
+ }
79
+ /** Reset the session tally (test seam). */
80
+ function resetSessionCost() {
81
+ SESSION = ZERO;
82
+ }
83
+ // --- Formatting ------------------------------------------------------------
84
+ function fmtUsd(n) {
85
+ return `$${n < 0.01 && n > 0 ? n.toFixed(4) : n.toFixed(2)}`;
86
+ }
87
+ function fmtInt(n) {
88
+ return Math.round(n).toLocaleString("en-US");
89
+ }
90
+ function fmtK(n) {
91
+ return n >= 1000 ? `${(n / 1000).toFixed(1)}k` : String(Math.round(n));
92
+ }
93
+ /**
94
+ * The human-readable cost block for a run. Shows tokens + API-equivalent `$`, the
95
+ * billed-to line (a LOUD warning + an actionable fix when metered, a green ✅ when
96
+ * on the subscription), and the session tally when it exceeds this run.
97
+ */
98
+ function formatCostSummary(c, opts) {
99
+ const lines = [];
100
+ lines.push(` Spent: ${fmtInt(totalTokens(c))} tokens ` +
101
+ `(${fmtK(c.inputTokens)} in · ${fmtK(c.outputTokens)} out · ${fmtK(c.cacheReadTokens)} cache) ` +
102
+ `· ~${fmtUsd(c.costUsd)} API-equivalent`);
103
+ if (opts.billing.metered) {
104
+ const v = opts.billing.keyVar ?? "ANTHROPIC_API_KEY";
105
+ lines.push(` ⚠ Billed to: METERED API (${v} is set) — you paid ~${fmtUsd(c.costUsd)} this run.`, ` Run it free on your Claude subscription: unset ${v}, then \`claude login\`.`);
106
+ }
107
+ else {
108
+ lines.push(` Billed to: your Claude subscription — $0 metered ✅`);
109
+ }
110
+ if (opts.session && opts.session.costUsd > c.costUsd) {
111
+ lines.push(` Session so far: ${fmtInt(totalTokens(opts.session))} tokens · ~${fmtUsd(opts.session.costUsd)} API-equivalent`);
112
+ }
113
+ return lines.join("\n");
114
+ }
115
+ /**
116
+ * Record `c` into the session tally and emit its cost block. The default sink is
117
+ * stderr (so a run's cost never pollutes `--json` stdout). Injectable env + sink
118
+ * keep it fully testable. Returns the emitted text (also handy for a skill to
119
+ * relay to the user).
120
+ */
121
+ function emitCostSummary(c, opts = {}) {
122
+ // A no-cost run (a replay, a zero-trial eval) has nothing to report.
123
+ if (totalTokens(c) === 0 && c.costUsd === 0)
124
+ return "";
125
+ const billing = detectBilling(opts.env ?? process.env);
126
+ const session = recordSessionCost(c);
127
+ const text = formatCostSummary(c, { billing, session });
128
+ (opts.out ??
129
+ ((s) => {
130
+ console.error(s);
131
+ }))(text);
132
+ return text;
133
+ }
134
+ //# sourceMappingURL=eval-cost.js.map
@@ -0,0 +1,192 @@
1
+ import { type SHA256Hash } from "./core/hash.js";
2
+ /** Lock mode: never touch the lock / verify-only (CI) / record-and-write (local). */
3
+ export type LockMode = "off" | "check" | "update";
4
+ /**
5
+ * Per-spec lock overrides (additive on `EvalSpec`/`TriggerRateSpec`). Normally the
6
+ * mode comes from the CLI (`eval --check`/`--update` → `VIGILES_EVAL_LOCK`) and
7
+ * the dir/epoch from defaults/config; set these to drive the lock programmatically
8
+ * (or to point a test at a throwaway dir). Each field falls back to its env/default.
9
+ */
10
+ export interface EvalLockOptions {
11
+ /** Override the lock mode (else `VIGILES_EVAL_LOCK`, else `off`). */
12
+ readonly mode?: LockMode;
13
+ /** Override the lock directory (else `<cwd>/.vigiles/eval-locks`). */
14
+ readonly dir?: string;
15
+ /** Override the behavior epoch (else `VIGILES_EVAL_API_VERSION`, else 1). */
16
+ readonly evalApiVersion?: number;
17
+ }
18
+ /**
19
+ * On-disk lock-format version, salted into nothing (the lock is keyed by name,
20
+ * not by hash) but VALIDATED on read so an incompatible shape fails loud rather
21
+ * than deserializing into a stale structure. Bump on a breaking shape change.
22
+ */
23
+ export declare const LOCK_VERSION = 1;
24
+ /** Default directory for committed eval locks (tracked, NOT gitignored). */
25
+ export declare const DEFAULT_LOCK_DIR = ".vigiles/eval-locks";
26
+ /**
27
+ * The model-affecting inputs hashed into a lock's `inputsHash`. Everything here
28
+ * is something that, if it changes, means the recorded model behavior is stale
29
+ * and you MUST re-drive the model (→ subscription → local). Deliberately EXCLUDED:
30
+ * the scoring `measure`/assertions (re-run live against the replayed report), the
31
+ * trial count (a sample-size knob, not a behavior input), and per-run env noise.
32
+ */
33
+ export interface EvalLockInputs {
34
+ /** Model id used (folded in; a floating alias can't detect weight drift — warned). */
35
+ readonly model: string;
36
+ /**
37
+ * A hand-bumped behavior epoch the project owns (`.vigilesrc.json`
38
+ * `eval.apiVersion`), bumped when a harness-side change YOU made (a CLAUDE.md
39
+ * edit, a global hook) would shift eval outputs but isn't otherwise in the
40
+ * inputs. The escape hatch for "force a re-eval."
41
+ */
42
+ readonly evalApiVersion: number;
43
+ /**
44
+ * The seam-specific canonical input object — the tasks/prompts/files/settings/
45
+ * sorted-tools/pluginDirHash/serialized-checks that steer the model. Assembled
46
+ * by each entry point (it knows its own shape) and hashed opaquely here.
47
+ */
48
+ readonly inputs: unknown;
49
+ }
50
+ /**
51
+ * Why the harness binary version is **NOT** hashed (only recorded as provenance):
52
+ * `--check` runs in CI where `claude` is PINNED to a fixed version, while a dev's
53
+ * local `claude` is whatever they have — folding the version into the hash would
54
+ * false-trip `--check` on every PR where those differ. It is also the lock's
55
+ * honest scope: the gate verifies your committed results match your current
56
+ * *author-controlled inputs*, not current model/harness behavior (there is no
57
+ * automated live run). Harness/model drift is caught when YOU re-run `--update`
58
+ * locally and review the moved numbers in the git diff. Keeping the version out
59
+ * of the hash is what lets `--check` stay binary-free + deterministic in CI.
60
+ * (The eval CACHE still keys on it — that's local replay soundness, a different
61
+ * axis.) See research/cache-invalidation.md.
62
+ */
63
+ /** Deterministic content hash of a lock's model-affecting inputs. */
64
+ export declare function evalInputsHash(input: EvalLockInputs): SHA256Hash;
65
+ /** A committed eval lock: the integrity stamp + the replayable recorded report. */
66
+ export interface EvalLock {
67
+ readonly version: number;
68
+ /** The eval's report name (human-facing; also the lock filename slug source). */
69
+ readonly name: string;
70
+ /** Hash of the model-affecting inputs ({@link evalInputsHash}). */
71
+ readonly inputsHash: string;
72
+ /** The model id the report was produced against (for the drift warning). */
73
+ readonly model: string;
74
+ /** The harness version token at record time (provenance; already in the hash). */
75
+ readonly harnessVersionKey: string;
76
+ /** The behavior epoch at record time (provenance; already in the hash). */
77
+ readonly evalApiVersion: number;
78
+ /** ISO-8601 timestamp the lock was recorded (provenance; NOT in the hash). */
79
+ readonly builtAt: string;
80
+ /**
81
+ * The entry point's recorded report — the model's observed behavior, REPLAYED
82
+ * verbatim on `--check` so the script's own assertions judge it. Stored as the
83
+ * exact return type of the entry point (`EvalReport` / `TriggerRateReport` /
84
+ * `CheckReport`) so replay is transparent to the caller.
85
+ */
86
+ readonly report: unknown;
87
+ }
88
+ /** Filesystem-safe slug for a report name (the lock filename). */
89
+ export declare function lockSlug(name: string): string;
90
+ /** Path to a named eval's lock file under `dir`. */
91
+ export declare function lockPath(dir: string, name: string): string;
92
+ /**
93
+ * Read a named eval's lock. A MISS (no file) returns `null`. A CORRUPT or
94
+ * wrong-version file **throws** — a broken lock is a real failure the CI gate
95
+ * must surface, not silently treat as "no lock" (which would let a stale eval
96
+ * pass). The message says how to recover.
97
+ */
98
+ export declare function readLock(dir: string, name: string): EvalLock | null;
99
+ /**
100
+ * Whether ANY lock has been committed under `dir`. The CI staleness gate
101
+ * (`eval --check`) uses this to stay a NO-OP until the feature is in use: a repo
102
+ * that has never run `eval --update` has no locks, so there is nothing to verify
103
+ * and CI passes green. Once the first lock is committed, every named eval is held
104
+ * to having a fresh one (a new unlocked eval then reads as stale). The graduated,
105
+ * opt-in-by-committing behavior that keeps a fresh `init` from going red.
106
+ */
107
+ export declare function anyLocksCommitted(dir: string): boolean;
108
+ /**
109
+ * Does an edited path plausibly change an eval's INPUTS — so a committed lock may
110
+ * now be stale? Two surfaces feed the hash: a skill's trigger surface (`SKILL.md`)
111
+ * and the eval script that holds the prompts/spec (`*.eval.{mjs,cjs,js,mts,cts,ts}`).
112
+ * Pure (string-only) so the nudge hook stays cheap and never runs an eval script.
113
+ */
114
+ export declare function isEvalInputFile(path: string): boolean;
115
+ /**
116
+ * The NON-BLOCKING nudge to emit after an eval-input edit when committed locks
117
+ * exist, or `null` for no nudge. Self-gating: it stays silent until you've opted
118
+ * into the lock (committed one), so it can't annoy a repo that doesn't use evals.
119
+ * It deliberately does NOT recompute staleness (that needs the eval script + is
120
+ * the job of `eval --check`) — a reminder, not a gate. The honest harness-neutral
121
+ * reminder; how it reaches the agent (both CC and Codex inject `additionalContext`
122
+ * on `PostToolUse`) is the caller's concern. See docs/harness-testing-*.md.
123
+ */
124
+ export declare function evalLockNudge(filePath: string, lockDir: string): string | null;
125
+ /** Write a named eval's lock (pretty JSON for a reviewable git diff). */
126
+ export declare function writeLock(dir: string, lock: EvalLock): void;
127
+ /**
128
+ * Build a fresh lock envelope from a just-recorded report (the `--update` write).
129
+ * `builtAt` is passed in (never read from the clock here) so the module stays
130
+ * pure + deterministically testable; the CLI stamps the real timestamp.
131
+ */
132
+ export declare function buildLock(args: {
133
+ readonly name: string;
134
+ readonly inputsHash: string;
135
+ readonly model: string;
136
+ readonly harnessVersionKey: string;
137
+ readonly evalApiVersion: number;
138
+ readonly builtAt: string;
139
+ readonly report: unknown;
140
+ }): EvalLock;
141
+ /** What the lock layer decides an entry point should do for this run. */
142
+ export type LockDecision =
143
+ /** Drive the model normally (mode `off`, or `update`, or `check` with no lock-skip). */
144
+ {
145
+ readonly kind: "run";
146
+ }
147
+ /** `check` + a matching fresh lock → return the recorded report, NO model call. */
148
+ | {
149
+ readonly kind: "replay";
150
+ readonly report: unknown;
151
+ }
152
+ /** `check` + a missing/stale lock → fail; the caller throws `reason`. */
153
+ | {
154
+ readonly kind: "stale";
155
+ readonly reason: string;
156
+ };
157
+ /**
158
+ * Decide what `check` mode should do given the current input hash and the
159
+ * committed lock. `off`/`update` always `run` (update records afterwards). `check`
160
+ * replays a matching lock (no model) and is `stale` on a missing lock or a hash
161
+ * mismatch — the deterministic CI gate.
162
+ */
163
+ export declare function decideLock(mode: LockMode, name: string, currentHash: string, existing: EvalLock | null): LockDecision;
164
+ /** A single numeric leaf that moved between the prior lock and a fresh `--update`. */
165
+ export interface NumberDelta {
166
+ readonly path: string;
167
+ readonly before: number;
168
+ readonly after: number;
169
+ }
170
+ /**
171
+ * Collect the numeric leaves that changed between two recorded reports — the
172
+ * human-facing delta printed at `--update` time (e.g. `rate: 0.900 → 0.650`).
173
+ * Generic over any report shape (walks numbers by dotted path), so it works for
174
+ * `EvalReport`, `TriggerRateReport`, and `CheckReport` without per-type code. The
175
+ * committed git diff is the primary review surface; this is the at-a-glance echo.
176
+ */
177
+ export declare function diffReportNumbers(before: unknown, after: unknown): NumberDelta[];
178
+ /** Render the `--update` result for a human: NEW lock, or the per-number deltas. */
179
+ export declare function formatLockUpdate(name: string, deltas: readonly NumberDelta[], isNew: boolean): string;
180
+ /**
181
+ * Read the lock mode from the environment (`VIGILES_EVAL_LOCK`), set by the CLI's
182
+ * `eval --check` / `--update` flags. A run knob (like `VIGILES_TRIALS`): the CLI
183
+ * is the only place that should set it. Anything unrecognized → `off`.
184
+ */
185
+ export declare function lockModeFromEnv(env?: NodeJS.ProcessEnv): LockMode;
186
+ /**
187
+ * The behavior epoch (`evalApiVersion`) for this run, read from the env the CLI
188
+ * populates from `.vigilesrc.json` `eval.apiVersion`. Default 1. A malformed
189
+ * value falls back to 1 (never throws) — the lock stays usable.
190
+ */
191
+ export declare function evalApiVersionFromEnv(env?: NodeJS.ProcessEnv): number;
192
+ //# sourceMappingURL=eval-lock.d.ts.map