vigiles 11.0.0 → 12.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/.claude-plugin/plugin.json +9 -0
  2. package/README.md +10 -5
  3. package/action.yml +13 -2
  4. package/dist/adapter-conformance.js +6 -0
  5. package/dist/adapter-registry.d.ts +20 -0
  6. package/dist/adapter-registry.js +27 -0
  7. package/dist/adapters/claude-code/hook-protocol.js +4 -0
  8. package/dist/adapters/claude-code/runtime.js +12 -0
  9. package/dist/adapters/codex/eval.js +3 -0
  10. package/dist/adapters/codex/hook-protocol.d.ts +9 -1
  11. package/dist/adapters/codex/hook-protocol.js +10 -0
  12. package/dist/adapters/codex/runtime.js +10 -0
  13. package/dist/adapters/opencode/runtime.js +4 -0
  14. package/dist/claude-code.d.ts +2 -0
  15. package/dist/claude-code.js +9 -1
  16. package/dist/cli-commands.d.ts +1 -1
  17. package/dist/cli-commands.js +1 -0
  18. package/dist/cli.js +245 -29
  19. package/dist/core/hook-protocol.d.ts +15 -0
  20. package/dist/core/rule-meta.js +8 -0
  21. package/dist/core/runtime.d.ts +20 -0
  22. package/dist/core/skill-description-budget.d.ts +42 -0
  23. package/dist/core/skill-description-budget.js +47 -0
  24. package/dist/core/types.d.ts +21 -0
  25. package/dist/core/validate.js +3 -0
  26. package/dist/doc-command-coverage.d.ts +20 -0
  27. package/dist/doc-command-coverage.js +60 -0
  28. package/dist/eval-cache.d.ts +6 -0
  29. package/dist/eval-cache.js +2 -0
  30. package/dist/eval-cost.d.ts +75 -0
  31. package/dist/eval-cost.js +134 -0
  32. package/dist/eval-lock.d.ts +192 -0
  33. package/dist/eval-lock.js +286 -0
  34. package/dist/eval.d.ts +37 -20
  35. package/dist/eval.js +227 -56
  36. package/dist/research-index.d.ts +31 -0
  37. package/dist/research-index.js +48 -0
  38. package/dist/scan-behavioral.d.ts +42 -0
  39. package/dist/scan-behavioral.js +67 -0
  40. package/dist/scan.d.ts +3 -23
  41. package/dist/scan.js +18 -69
  42. package/dist/setup-plan.d.ts +38 -1
  43. package/dist/setup-plan.js +67 -4
  44. package/hooks/eval-lock-nudge.sh +21 -0
  45. package/package.json +1 -1
  46. package/skills/adopt-spec/SKILL.md +10 -1
  47. package/skills/edit-spec/SKILL.md +1 -0
  48. package/skills/strengthen/SKILL.md +4 -0
  49. package/skills/test-harness/SKILL.md +44 -0
package/dist/cli.js CHANGED
@@ -17,6 +17,7 @@ const generate_types_js_1 = require("./core/generate-types.js");
17
17
  const generate_harness_js_1 = require("./core/generate-harness.js");
18
18
  const capability_diff_js_1 = require("./core/capability-diff.js");
19
19
  const validate_js_1 = require("./core/validate.js");
20
+ const eval_lock_js_1 = require("./eval-lock.js");
20
21
  const cli_flags_js_1 = require("./cli-flags.js");
21
22
  const setup_plan_js_1 = require("./setup-plan.js");
22
23
  const types_js_1 = require("./core/types.js");
@@ -675,6 +676,7 @@ function lintExitCode(report) {
675
676
  report.hookScriptErrors > 0 ||
676
677
  report.disallowedToolErrors > 0 ||
677
678
  report.descriptionOverlapErrors > 0 ||
679
+ report.descriptionBudgetErrors > 0 ||
678
680
  report.frontmatterValidErrors > 0 ||
679
681
  report.mcpHookErrors > 0 ||
680
682
  report.preferCompiledHookErrors > 0 ||
@@ -1026,6 +1028,10 @@ async function runLint(restArgs, flags, config) {
1026
1028
  // 7k. Description-overlap — two model-invocable skills with near-identical
1027
1029
  // descriptions collide in the selector (deterministic NCD precision proxy).
1028
1030
  const descriptionOverlap = checkDescriptionOverlap(config, silent, adapter);
1031
+ // 7k². Skill-description-budget — a model-invocable skill whose description is
1032
+ // so long the trigger signal is buried (heuristic proxy; degrades recall +
1033
+ // precision). Generous 500-char budget; warn-tier, never gates.
1034
+ const descriptionBudget = checkDescriptionBudget(config, silent, adapter);
1029
1035
  // 7l. Frontmatter-valid — a `---` block that isn't valid YAML (warn; js-yaml is
1030
1036
  // stricter than some loaders, so verify before enforcing).
1031
1037
  const frontmatterValid = checkFrontmatterValid(config, silent, adapter);
@@ -1119,6 +1125,8 @@ async function runLint(restArgs, flags, config) {
1119
1125
  disallowedToolErrors: disallowedTools.errors,
1120
1126
  descriptionOverlapIssues: descriptionOverlap.issues,
1121
1127
  descriptionOverlapErrors: descriptionOverlap.errors,
1128
+ descriptionBudgetIssues: descriptionBudget.issues,
1129
+ descriptionBudgetErrors: descriptionBudget.errors,
1122
1130
  frontmatterValidIssues: frontmatterValid.issues,
1123
1131
  frontmatterValidErrors: frontmatterValid.errors,
1124
1132
  mcpHookIssues: mcpHookTargets.issues,
@@ -1601,7 +1609,19 @@ function specReferencedElsewhere(specFile, ejectedFile) {
1601
1609
  /** Full GitHub Actions workflow that wires the production `zernie/vigiles@v1`
1602
1610
  * Action (lint pillar) and, when the test pillar is set up, a deterministic
1603
1611
  * harness job. */
1604
- function vigilesWorkflow(plan) {
1612
+ /** The npm package(s) that provide each harness's CLI binary — the deterministic
1613
+ * harness tier spawns the real agent CLI against a mock model (no API key). A repo
1614
+ * targeting both harnesses installs both. */
1615
+ function harnessTestBinaries(harnesses) {
1616
+ const pkgs = [];
1617
+ if (harnesses.includes("claude"))
1618
+ pkgs.push("@anthropic-ai/claude-code");
1619
+ if (harnesses.includes("codex"))
1620
+ pkgs.push("@openai/codex");
1621
+ // Fall back to Claude Code if the set is somehow empty (back-compatible default).
1622
+ return (pkgs.length > 0 ? pkgs : ["@anthropic-ai/claude-code"]).join(" ");
1623
+ }
1624
+ function vigilesWorkflow(plan, harnesses) {
1605
1625
  const harness = plan.test
1606
1626
  ? `
1607
1627
  harness:
@@ -1615,8 +1635,23 @@ function vigilesWorkflow(plan) {
1615
1635
  with:
1616
1636
  node-version: "20"
1617
1637
  - run: npm install
1618
- - run: npm i -g @anthropic-ai/claude-code # mock tier needs the binary, no API key
1638
+ - run: npm i -g ${harnessTestBinaries(harnesses)} # mock tier needs the binary, no API key
1619
1639
  - run: npx vigiles test
1640
+
1641
+ eval-check:
1642
+ # Eval staleness gate — real-model evals run LOCALLY on your subscription
1643
+ # (\`npx vigiles eval --update\`, which commits a lock); this job VERIFIES those
1644
+ # committed results against the current inputs with NO model call. It stays a
1645
+ # green no-op until you commit your first lock. See docs/harness-testing.md.
1646
+ runs-on: ubuntu-latest
1647
+ steps:
1648
+ - uses: actions/checkout@v4
1649
+ - uses: actions/setup-node@v4
1650
+ with:
1651
+ node-version: "20"
1652
+ - uses: zernie/vigiles@v1
1653
+ with:
1654
+ command: eval-check
1620
1655
  `
1621
1656
  : "";
1622
1657
  return `name: vigiles
@@ -1687,7 +1722,7 @@ function rewriteRemovedSubcommands(content) {
1687
1722
  * commit hint). An existing workflow is never clobbered unless `--force`, but a
1688
1723
  * STALE one (old bare-`npx vigiles` API, or a removed subcommand) is reported
1689
1724
  * loudly instead of silently skipped — and rewritten in place with `--force`. */
1690
- function wireGha(plan) {
1725
+ function wireGha(plan, harnesses) {
1691
1726
  const dir = (0, node_path_1.resolve)(process.cwd(), ".github", "workflows");
1692
1727
  const path = (0, node_path_1.resolve)(dir, "vigiles.yml");
1693
1728
  const rel = ".github/workflows/vigiles.yml";
@@ -1708,7 +1743,7 @@ function wireGha(plan) {
1708
1743
  }
1709
1744
  else if (workflowUsesStaleApi(content)) {
1710
1745
  if (plan.force) {
1711
- (0, node_fs_1.writeFileSync)(path, vigilesWorkflow(plan));
1746
+ (0, node_fs_1.writeFileSync)(path, vigilesWorkflow(plan, harnesses));
1712
1747
  console.log(`✓ Regenerated ${rel} (was a stale bare \`npx vigiles\`)`);
1713
1748
  return [rel];
1714
1749
  }
@@ -1724,7 +1759,7 @@ function wireGha(plan) {
1724
1759
  }
1725
1760
  if (!(0, node_fs_1.existsSync)(dir))
1726
1761
  (0, node_fs_1.mkdirSync)(dir, { recursive: true });
1727
- (0, node_fs_1.writeFileSync)(path, vigilesWorkflow(plan));
1762
+ (0, node_fs_1.writeFileSync)(path, vigilesWorkflow(plan, harnesses));
1728
1763
  console.log("✓ Created .github/workflows/vigiles.yml (uses zernie/vigiles@v1)");
1729
1764
  return [".github/workflows/vigiles.yml"];
1730
1765
  }
@@ -2130,6 +2165,36 @@ function installPlugins(harnesses) {
2130
2165
  console.log("");
2131
2166
  reportInstall(plan, runInstall(plan, exec));
2132
2167
  }
2168
+ // Claude Code gets its hooks from the global marketplace plugin; Codex has no
2169
+ // global store, so wire vigiles's proactive nudge hooks into the repo's
2170
+ // .codex/config.toml directly (the idiomatic, repo-committed place).
2171
+ if (harnesses.includes("codex"))
2172
+ wireCodexHooks();
2173
+ }
2174
+ /**
2175
+ * Wire vigiles's proactive nudge hooks into `.codex/config.toml` (idempotently).
2176
+ * Codex honors `additionalContext` on `PostToolUse`, and these run as direct
2177
+ * `npx vigiles hook-runtime …` commands (no plugin root / vendored script), so a
2178
+ * Codex user gets the same eval-lock + refs nudges a Claude Code user gets from
2179
+ * the marketplace plugin. The pure merge is `applyCodexPluginHooks` (unit-tested
2180
+ * in setup-plan.test.ts) — this only does the read/parse/write IO.
2181
+ */
2182
+ function wireCodexHooks() {
2183
+ const path = (0, node_path_1.resolve)(process.cwd(), ".codex", "config.toml");
2184
+ let config = {};
2185
+ if ((0, node_fs_1.existsSync)(path)) {
2186
+ try {
2187
+ config = (0, toml_1.parse)((0, node_fs_1.readFileSync)(path, "utf-8"));
2188
+ }
2189
+ catch {
2190
+ console.log("⚠ .codex/config.toml is not valid TOML — skipping Codex hook wiring (fix it, then re-run `vigiles init`).");
2191
+ return;
2192
+ }
2193
+ }
2194
+ const merged = (0, setup_plan_js_1.applyCodexPluginHooks)(config);
2195
+ (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(path), { recursive: true });
2196
+ (0, node_fs_1.writeFileSync)(path, (0, hook_install_js_1.serializeConfig)(merged, "toml"));
2197
+ console.log("✓ Wired the eval-lock + refs nudge hooks into .codex/config.toml (commit it)");
2133
2198
  }
2134
2199
  /** Add/upgrade `vigiles` in the project's `devDependencies` (and move it out of
2135
2200
  * `dependencies` if it's there). Returns the files it wrote (for the commit
@@ -2293,7 +2358,7 @@ async function setup(args) {
2293
2358
  // CI — the production Action (+ a harness job when Pillar 2 is on).
2294
2359
  if (plan.gha) {
2295
2360
  console.log("");
2296
- written.push(...wireGha(plan));
2361
+ written.push(...wireGha(plan, harnesses));
2297
2362
  }
2298
2363
  // Plugin/skill install — per-harness (Claude marketplace / Codex direct).
2299
2364
  if (plan.plugin) {
@@ -2717,6 +2782,33 @@ function checkDescriptionOverlap(config, silent, adapter) {
2717
2782
  }
2718
2783
  return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2719
2784
  }
2785
+ /**
2786
+ * Apply the `skill-description-budget` rule: a model-invocable skill whose
2787
+ * description is so long the trigger signal is buried — the selector weighs the
2788
+ * opening most, so a bloated description hurts recall + precision. A
2789
+ * deterministic heuristic proxy (generous 500-char budget). Reuses `scanPlugin`'s
2790
+ * `descriptionBudgetIssues`. Warning by default; "error" gates CI.
2791
+ */
2792
+ function checkDescriptionBudget(config, silent, adapter) {
2793
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["skill-description-budget"]);
2794
+ if (!sev)
2795
+ return { issues: 0, errors: 0 };
2796
+ let found;
2797
+ try {
2798
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).descriptionBudgetIssues;
2799
+ }
2800
+ catch {
2801
+ return { issues: 0, errors: 0 };
2802
+ }
2803
+ if (found.length > 0 && !silent) {
2804
+ console.log("\nSkill-description-budget check:\n");
2805
+ for (const issue of found) {
2806
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${issue.message}`);
2807
+ ghAnnotate(sev === "error" ? "error" : "warning", issue.message);
2808
+ }
2809
+ }
2810
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2811
+ }
2720
2812
  /**
2721
2813
  * Apply the `lethal-trifecta` rule: a unit (subagent / model-invocable skill)
2722
2814
  * whose declared tools hold all three legs (read-private + ingest-untrusted +
@@ -3406,10 +3498,55 @@ async function handleGenerateHarness(args, restArgs) {
3406
3498
  * tier needs it, just like the node:test suite). `--trials=N` is forwarded to
3407
3499
  * eval scripts via the `VIGILES_TRIALS` env var.
3408
3500
  */
3501
+ /**
3502
+ * Resolve the eval LOCK env from the `eval` flags. `--update` records each named
3503
+ * eval's report to a committed `.vigiles/eval-locks/<name>.lock.json` (run locally
3504
+ * on your subscription); `--check` (CI) verifies the committed result against the
3505
+ * current inputs WITHOUT a model call. `--check` is a green NO-OP until the first
3506
+ * lock is committed (smooth adoption). Returns the env to thread, or `"skip"` to
3507
+ * exit green now. `--check`+`--update` together is a usage error (exit 2). The
3508
+ * behavior epoch comes from `.vigilesrc.json` `eval.apiVersion` (committed).
3509
+ */
3510
+ function resolveEvalLockEnv(args) {
3511
+ const wantCheck = args.includes("--check");
3512
+ const wantUpdate = args.includes("--update");
3513
+ if (wantCheck && wantUpdate) {
3514
+ console.error("vigiles eval: --check and --update are mutually exclusive (one verifies, one records).");
3515
+ process.exit(2);
3516
+ }
3517
+ if (wantCheck &&
3518
+ !(0, eval_lock_js_1.anyLocksCommitted)((0, node_path_1.resolve)(process.cwd(), eval_lock_js_1.DEFAULT_LOCK_DIR))) {
3519
+ console.log("ℹ vigiles eval --check: no committed eval locks found — nothing to verify.\n" +
3520
+ " Run `vigiles eval --update` locally (on your subscription) and commit the\n" +
3521
+ " lock to enable the CI staleness gate.");
3522
+ return "skip";
3523
+ }
3524
+ const env = {};
3525
+ if (wantCheck)
3526
+ env.VIGILES_EVAL_LOCK = "check";
3527
+ if (wantUpdate)
3528
+ env.VIGILES_EVAL_LOCK = "update";
3529
+ if (wantCheck || wantUpdate) {
3530
+ const apiVersion = (0, validate_js_1.loadConfig)().eval?.apiVersion;
3531
+ if (apiVersion !== undefined)
3532
+ env.VIGILES_EVAL_API_VERSION = String(apiVersion);
3533
+ }
3534
+ return env;
3535
+ }
3409
3536
  function handleRunScripts(kind, args, restArgs) {
3410
3537
  const cwd = process.cwd();
3411
3538
  // Harness/eval scripts may be authored in JS or TS (see run-scripts.ts).
3412
3539
  const defaultGlob = (0, run_scripts_js_1.scriptGlob)(kind === "test" ? "harness" : "eval");
3540
+ // The eval LOCK flags (`--check`/`--update`) are resolved BEFORE file discovery
3541
+ // so mutual-exclusion + the cold-start no-op are honored regardless of file
3542
+ // count. Returns the env to thread to scripts, or `"skip"` to exit green now.
3543
+ let lockEnv = {};
3544
+ if (kind === "eval") {
3545
+ const r = resolveEvalLockEnv(args);
3546
+ if (r === "skip")
3547
+ return;
3548
+ lockEnv = r;
3549
+ }
3413
3550
  const files = (0, run_scripts_js_1.discoverScripts)(restArgs, defaultGlob, cwd);
3414
3551
  // `--min=N`: a CI gate asserts at least N scripts actually RAN — so a bad path,
3415
3552
  // a renamed file, or a glob that matched nothing fails LOUD instead of passing
@@ -3438,7 +3575,7 @@ function handleRunScripts(kind, args, restArgs) {
3438
3575
  // it's part of the measurement definition, so it belongs in the spec
3439
3576
  // (`model` / `minModel`), version-controlled, not a hidden override.
3440
3577
  const trialsFlag = args.find((a) => a.startsWith("--trials="));
3441
- const env = {};
3578
+ const env = { ...lockEnv };
3442
3579
  if (trialsFlag)
3443
3580
  env.VIGILES_TRIALS = trialsFlag.split("=")[1];
3444
3581
  console.log(`Running ${String(files.length)} ${kind} file(s):\n`);
@@ -3617,6 +3754,8 @@ function printUsage(command) {
3617
3754
  console.log(" --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt");
3618
3755
  console.log(" vigiles test [files...] Run *.harness.mjs deterministic harness tests");
3619
3756
  console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
3757
+ console.log(" --update records each named eval's result to a committed lock (run locally on your subscription)");
3758
+ console.log(" --check verifies committed eval results against current inputs WITHOUT a model — the CI staleness gate");
3620
3759
  console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
3621
3760
  console.log("");
3622
3761
  console.log("Examples:");
@@ -3932,6 +4071,9 @@ async function handleHookRuntime(kind, restArgs) {
3932
4071
  case "refs":
3933
4072
  refsHookCommand();
3934
4073
  return;
4074
+ case "eval-lock-nudge":
4075
+ evalLockNudgeHookCommand();
4076
+ return;
3935
4077
  case "effect-enter":
3936
4078
  (0, effect_region_js_1.setEffectActive)(process.cwd());
3937
4079
  console.log("Effect boundary entered.");
@@ -3978,6 +4120,45 @@ const INSTRUCTION_FILE = /^(SKILL|CLAUDE|AGENTS)\.md$/;
3978
4120
  function isInstructionFile(file) {
3979
4121
  return INSTRUCTION_FILE.test((0, node_path_1.basename)(file));
3980
4122
  }
4123
+ /**
4124
+ * PostToolUse-hook entrypoint: when the agent edits an eval input (a `SKILL.md`
4125
+ * trigger surface or an `*.eval.*` script), and committed eval locks exist, inject
4126
+ * a NON-BLOCKING reminder to re-run `vigiles eval --update`. Self-gating (silent
4127
+ * until a lock is committed), never blocks, never runs an eval — a reminder, not a
4128
+ * gate (the gate is `eval --check` in CI). The harness-neutral nudge lives in
4129
+ * `evalLockNudge`; both CC and Codex deliver it as `additionalContext` on
4130
+ * `PostToolUse` (confirmed — see docs/harness-testing-codex.md).
4131
+ */
4132
+ function evalLockNudgeHookCommand() {
4133
+ let raw = "";
4134
+ try {
4135
+ raw = (0, node_fs_1.readFileSync)(0, "utf-8");
4136
+ }
4137
+ catch {
4138
+ /* no stdin → nothing to do */
4139
+ }
4140
+ let file = "";
4141
+ try {
4142
+ const j = JSON.parse(raw);
4143
+ file = j.tool_input?.file_path ?? "";
4144
+ }
4145
+ catch {
4146
+ /* malformed → nothing to do */
4147
+ }
4148
+ if (!file)
4149
+ return;
4150
+ const cwd = process.cwd();
4151
+ const target = (0, node_path_1.relative)(cwd, (0, node_path_1.resolve)(cwd, file)) || file;
4152
+ const msg = (0, eval_lock_js_1.evalLockNudge)(target, (0, node_path_1.resolve)(cwd, eval_lock_js_1.DEFAULT_LOCK_DIR));
4153
+ if (!msg)
4154
+ return;
4155
+ process.stdout.write(JSON.stringify({
4156
+ hookSpecificOutput: {
4157
+ hookEventName: "PostToolUse",
4158
+ additionalContext: msg,
4159
+ },
4160
+ }) + "\n");
4161
+ }
3981
4162
  /**
3982
4163
  * PostToolUse-hook entrypoint: when the agent edits an instruction file, force
3983
4164
  * every code reference to carry a file-qualified mark (`path.ext#symbol`) and
@@ -4153,31 +4334,63 @@ async function installHookFile(file, adapter, registeredProviders = []) {
4153
4334
  : (0, hook_install_js_1.mergeHooksJson)(existing, compiled.hooks, file);
4154
4335
  (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(settingsAbs), { recursive: true });
4155
4336
  (0, node_fs_1.writeFileSync)(settingsAbs, (0, hook_install_js_1.serializeConfig)(merged, format));
4156
- // Honest gap (no silent skips): the gate/deny path is exit-2 and cross-harness,
4157
- // but an inject/react hook's OUTPUT shape is confirmed only for Claude Code. On
4158
- // another harness it would emit CC-shaped output that may not be read — exactly
4159
- // the silent failure this feature exists to prevent. Flag it, loudly.
4337
+ // No silent skips: warn loudly only where a hook's OUTPUT genuinely may not
4338
+ // apply on this harness. INJECT's `additionalContext` shape is now CONFIRMED
4339
+ // shared with Codex (per the official hooks docs), so an inject hook only
4340
+ // warns when its event isn't in the harness's `injectableEvents`. REACT's
4341
+ // output is still Claude-Code-confirmed only. The gate (deny→exit 2) path is
4342
+ // cross-harness and never warns.
4160
4343
  const role = (0, hook_program_js_1.dispatchKind)(program);
4161
- const warning = adapter.name !== "claude-code" && (role === "inject" || role === "react")
4162
- ? `${role} output is only confirmed for Claude Code. On ${adapter.name}, ` +
4163
- `the gate (deny→exit 2) path works, but this hook's ${role} output is ` +
4164
- `CC-shaped and unverified — it may silently not apply. Use a gate hook on ` +
4165
- `${adapter.name} for now, or confirm against the real binary first ` +
4166
- `(research/compiled-hooks-codex.md §Deferred).`
4167
- : undefined;
4344
+ const event = typeof program.on === "string" ? program.on : "";
4345
+ const injectable = adapter.hookProtocol?.injectableEvents ?? [];
4346
+ const matcher = (0, hook_program_js_1.hookRouting)(program).matcher;
4347
+ let warning;
4348
+ if (adapter.name !== "claude-code") {
4349
+ if (role === "inject" && !injectable.includes(event)) {
4350
+ warning =
4351
+ `this inject hook targets "${event}", which ${adapter.name} does not ` +
4352
+ `honor for additionalContext — the injected text won't reach the agent. ` +
4353
+ `Use an event ${adapter.name} supports: ${injectable.join(", ")}.`;
4354
+ }
4355
+ else if (role === "react") {
4356
+ warning =
4357
+ `react output is confirmed only for Claude Code; on ${adapter.name} this ` +
4358
+ `hook's react output is unverified (the gate deny→exit 2 path IS ` +
4359
+ `cross-harness). Confirm against the real binary first.`;
4360
+ }
4361
+ else if (matcher !== undefined) {
4362
+ // A tool-matched gate carries TOOL NAMES in its matcher. vigiles does not
4363
+ // yet translate tool vocabularies across dialects, so a matcher authored
4364
+ // with Claude Code names (`Edit`/`Write`/`Bash`) won't fire on a harness
4365
+ // that names the same tools differently (Codex: `apply_patch`/`shell`).
4366
+ // Warn LOUDLY rather than report a silently-non-firing success.
4367
+ warning =
4368
+ `this hook matches tool(s) "${matcher}" — if those are Claude Code tool ` +
4369
+ `names, they may not match ${adapter.name}'s vocabulary (e.g. ` +
4370
+ `apply_patch/shell), so the hook may not fire. Verify the matcher uses ` +
4371
+ `${adapter.name}'s tool names (cross-dialect matcher translation is not ` +
4372
+ `yet automatic).`;
4373
+ }
4374
+ }
4168
4375
  return { role, settingsPath: adapter.layout.settingsPath, warning };
4169
4376
  }
4170
4377
  /**
4171
4378
  * Compile + install every hook (explicit paths, else discovered under
4172
- * `.vigiles/hooks/`) into the active harness's config. Returns false if any
4173
- * hook failed to compile.
4379
+ * `.vigiles/hooks/`) into EVERY enabled harness's config. A typed hook is
4380
+ * harness-neutral, so when a repo targets both harnesses the SAME hook is merged
4381
+ * into `.claude/settings.json` AND `.codex/config.toml` (each in its native
4382
+ * format, with per-harness warnings) — never just the first. The harness set is
4383
+ * resolved from the `--harness=` flag, else `config.harness`, else auto-detect.
4384
+ * Returns false if any hook failed to compile for any harness.
4174
4385
  */
4175
- async function installHooks(hookFiles, harnessFlag) {
4386
+ async function installHooks(hookFiles, harnessFlag, configHarness) {
4176
4387
  if (hookFiles.length === 0)
4177
4388
  return true;
4178
- const adapter = harnessFlag
4179
- ? (0, adapter_registry_js_1.resolveAdapter)(process.cwd(), harnessFlag)
4180
- : (0, adapter_registry_js_1.detectAdapterResult)(process.cwd()).adapter;
4389
+ const adapters = (0, adapter_registry_js_1.resolveHarnessAdapters)({
4390
+ root: process.cwd(),
4391
+ flag: harnessFlag,
4392
+ configHarness,
4393
+ });
4181
4394
  // Validate registered providers first → the names a hook's provider() ref may
4182
4395
  // resolve to (an unsafe provider fails the whole compile, like a bad hook).
4183
4396
  let registeredProviders;
@@ -4194,10 +4407,13 @@ async function installHooks(hookFiles, harnessFlag) {
4194
4407
  let ok = true;
4195
4408
  for (const file of hookFiles) {
4196
4409
  try {
4197
- const r = await installHookFile(file, adapter, registeredProviders);
4198
- console.log(`✓ ${file} → ${r.settingsPath} (role: ${r.role}, harness: ${adapter.name})`);
4199
- if (r.warning)
4200
- console.warn(`⚠ ${r.warning}`);
4410
+ // Fan out: the same compiled hook lands in each enabled harness's config.
4411
+ for (const adapter of adapters) {
4412
+ const r = await installHookFile(file, adapter, registeredProviders);
4413
+ console.log(`✓ ${file} → ${r.settingsPath} (role: ${r.role}, harness: ${adapter.name})`);
4414
+ if (r.warning)
4415
+ console.warn(`⚠ ${r.warning}`);
4416
+ }
4201
4417
  }
4202
4418
  catch (e) {
4203
4419
  if (e instanceof hook_program_js_1.HookCompileError) {
@@ -4781,7 +4997,7 @@ async function main() {
4781
4997
  let valid = true;
4782
4998
  if (specs.length > 0)
4783
4999
  valid = (await compile(specs, config, { harnessFlag })) && valid;
4784
- valid = (await installHooks(hooks, harnessFlag)) && valid;
5000
+ valid = (await installHooks(hooks, harnessFlag, config.harness)) && valid;
4785
5001
  // Keep an existing whole-harness registry in sync (cheap, opt-in) so the
4786
5002
  // user never hand-runs `generate-harness`. Skipped when no harness.gen.ts.
4787
5003
  if (specs.length > 0)
@@ -31,5 +31,20 @@ export interface HookProtocol {
31
31
  * Used by `compileHookProgram` when rendering the settings block.
32
32
  */
33
33
  readonly matcherStyle?: "exact" | "regex";
34
+ /**
35
+ * The events whose hook can inject **developer context** into the agent by
36
+ * printing `{ hookSpecificOutput: { hookEventName, additionalContext } }` on
37
+ * stdout. The inject *shape* is shared across Claude Code and Codex (so the
38
+ * runtime emits it once, not per-harness); the genuinely per-harness fact is
39
+ * **which events honor it** — and encoding it here is what makes "this harness
40
+ * can deliver an inject hook" a TESTED contract instead of an assumption. Both
41
+ * Claude Code and Codex support the main lifecycle events (SessionStart,
42
+ * UserPromptSubmit, PostToolUse); a few (Stop, SubagentStop, PreCompact) carry
43
+ * no context on either. An empty list means the harness cannot inject context
44
+ * from a hook at all. Verified for Codex against the official hooks docs
45
+ * (developers.openai.com/codex/hooks). The conformance kit asserts a
46
+ * shell-hook harness declares a non-empty set.
47
+ */
48
+ readonly injectableEvents: readonly string[];
34
49
  }
35
50
  //# sourceMappingURL=hook-protocol.d.ts.map
@@ -202,6 +202,14 @@ exports.RULE_META = {
202
202
  summary: "Two model-invocable skills aren't near-identical (wrong one fires).",
203
203
  detector: "findDescriptionOverlaps",
204
204
  },
205
+ "skill-description-budget": {
206
+ id: "skill-description-budget",
207
+ bucket: "heuristic-behavioral",
208
+ surface: ["skill"],
209
+ defaultSeverity: "warn",
210
+ summary: "A model-invocable skill's description isn't so long the trigger is buried.",
211
+ detector: "findDescriptionBudgetIssues",
212
+ },
205
213
  "frontmatter-valid": {
206
214
  id: "frontmatter-valid",
207
215
  bucket: "heuristic-behavioral",
@@ -34,5 +34,25 @@ export interface HarnessRuntime {
34
34
  readonly args: readonly string[];
35
35
  readonly env: Record<string, string>;
36
36
  };
37
+ /**
38
+ * Reduce a raw `--version` string to the **behaviorally-significant** token the
39
+ * cache + lock key on — so a harness upgrade that actually moves agent behavior
40
+ * (new system prompt / tool defs) invalidates a stale replay, while churn that
41
+ * doesn't shouldn't partition the key. **What counts as significant is
42
+ * per-harness**, which is exactly why this lives on the port rather than as a
43
+ * universal `major.minor` rule:
44
+ *
45
+ * - **Claude Code** is semver-ish — `major.minor` bumps roughly quarterly
46
+ * (0.2 → 1.0 → 2.0 → 2.1 over 16 months) while patches ship ~daily — so it
47
+ * returns `major.minor`: a real behavior boundary, rare enough not to churn.
48
+ * - **Codex** is perpetual `0.x` where the *minor* IS the patch cadence (~2
49
+ * bumps/week), so `major.minor` would churn weekly — it returns `""`, opting
50
+ * out of version partitioning and relying on the dated model id +
51
+ * `evalApiVersion` instead.
52
+ *
53
+ * `""` means "don't partition on the harness version." Pure (no spawn — the
54
+ * caller resolves the raw string via `agentBinary --version`); unit-testable.
55
+ */
56
+ versionKey(raw: string): string;
37
57
  }
38
58
  //# sourceMappingURL=runtime.d.ts.map
@@ -0,0 +1,42 @@
1
+ /**
2
+ * Skill-description budget — a DETERMINISTIC proxy for a behavioral risk, and the
3
+ * deterministic sibling of {@link findDescriptionOverlaps}. A model-invocable
4
+ * skill is selected on its `description`, and the selector weighs the OPENING of
5
+ * it most; a long, buried description dilutes the trigger signal and degrades
6
+ * both recall ("did it fire when it should?") and precision ("did it stay quiet
7
+ * when it shouldn't?"). This catches a trigger-class problem with NO model.
8
+ *
9
+ * HEURISTIC-BEHAVIORAL bucket: the threshold is a PROXY (no character count
10
+ * PROVES a description triggers badly), so the ceiling is WARN — it never gates.
11
+ * Calibrated FP-safe: the default budget (500 chars) sits well above a normal
12
+ * one-to-three-sentence description, so only a genuinely bloated description
13
+ * fires. Reports the per-skill overflow, never a unilateral defect.
14
+ */
15
+ /** A skill identified by name + its trigger-surface description. */
16
+ export interface BudgetedSurface {
17
+ readonly name: string;
18
+ readonly description: string;
19
+ }
20
+ export interface DescriptionBudgetIssue {
21
+ readonly name: string;
22
+ /** Length of the description in characters. */
23
+ readonly length: number;
24
+ /** The budget it exceeded. */
25
+ readonly budget: number;
26
+ readonly message: string;
27
+ }
28
+ /**
29
+ * The default description-length budget, in characters. A concise what+when
30
+ * description is comfortably under this; only a bloated one (multiple long
31
+ * sentences, embedded examples, disambiguation prose) exceeds it. Generous on
32
+ * purpose — warn-tier, don't cry wolf. Exported so a caller / test sees it.
33
+ */
34
+ export declare const DEFAULT_DESCRIPTION_BUDGET = 500;
35
+ /**
36
+ * Find model-invocable skills whose `description` exceeds `budget` characters.
37
+ * Returns one {@link DescriptionBudgetIssue} per over-budget skill, longest
38
+ * first. Pure; pass only the surfaces that compete for auto-selection
39
+ * (model-invocable, described) so a user-invoked skill isn't a false alarm.
40
+ */
41
+ export declare function findDescriptionBudgetIssues(surfaces: readonly BudgetedSurface[], budget?: number): DescriptionBudgetIssue[];
42
+ //# sourceMappingURL=skill-description-budget.d.ts.map
@@ -0,0 +1,47 @@
1
+ "use strict";
2
+ /**
3
+ * Skill-description budget — a DETERMINISTIC proxy for a behavioral risk, and the
4
+ * deterministic sibling of {@link findDescriptionOverlaps}. A model-invocable
5
+ * skill is selected on its `description`, and the selector weighs the OPENING of
6
+ * it most; a long, buried description dilutes the trigger signal and degrades
7
+ * both recall ("did it fire when it should?") and precision ("did it stay quiet
8
+ * when it shouldn't?"). This catches a trigger-class problem with NO model.
9
+ *
10
+ * HEURISTIC-BEHAVIORAL bucket: the threshold is a PROXY (no character count
11
+ * PROVES a description triggers badly), so the ceiling is WARN — it never gates.
12
+ * Calibrated FP-safe: the default budget (500 chars) sits well above a normal
13
+ * one-to-three-sentence description, so only a genuinely bloated description
14
+ * fires. Reports the per-skill overflow, never a unilateral defect.
15
+ */
16
+ Object.defineProperty(exports, "__esModule", { value: true });
17
+ exports.DEFAULT_DESCRIPTION_BUDGET = void 0;
18
+ exports.findDescriptionBudgetIssues = findDescriptionBudgetIssues;
19
+ /**
20
+ * The default description-length budget, in characters. A concise what+when
21
+ * description is comfortably under this; only a bloated one (multiple long
22
+ * sentences, embedded examples, disambiguation prose) exceeds it. Generous on
23
+ * purpose — warn-tier, don't cry wolf. Exported so a caller / test sees it.
24
+ */
25
+ exports.DEFAULT_DESCRIPTION_BUDGET = 500;
26
+ /**
27
+ * Find model-invocable skills whose `description` exceeds `budget` characters.
28
+ * Returns one {@link DescriptionBudgetIssue} per over-budget skill, longest
29
+ * first. Pure; pass only the surfaces that compete for auto-selection
30
+ * (model-invocable, described) so a user-invoked skill isn't a false alarm.
31
+ */
32
+ function findDescriptionBudgetIssues(surfaces, budget = exports.DEFAULT_DESCRIPTION_BUDGET) {
33
+ const issues = [];
34
+ for (const s of surfaces) {
35
+ const length = Array.from(s.description).length;
36
+ if (length <= budget)
37
+ continue;
38
+ issues.push({
39
+ name: s.name,
40
+ length,
41
+ budget,
42
+ message: `skill "${s.name}" has a ${String(length)}-char description (budget ${String(budget)}) — the selector weighs the opening most, so a long description buries the trigger signal and hurts recall + precision. Tighten it to a concise what + when.`,
43
+ });
44
+ }
45
+ return issues.sort((a, b) => b.length - a.length);
46
+ }
47
+ //# sourceMappingURL=skill-description-budget.js.map
@@ -202,6 +202,15 @@ export interface RulesConfig {
202
202
  * as `scan` (descriptionOverlaps).
203
203
  */
204
204
  "description-overlap"?: RuleSeverity;
205
+ /**
206
+ * Flag a model-invocable skill whose `description` is so long the trigger
207
+ * signal is buried — the selector weighs the opening most, so a bloated
208
+ * description hurts recall + precision. A DETERMINISTIC heuristic proxy for a
209
+ * `--trigger`-class behavioral bug; calibrated FP-safe (generous default
210
+ * budget, 500 chars). Default "warn" — a proxy, never gates. Same detector as
211
+ * `scan` (descriptionBudgetIssues).
212
+ */
213
+ "skill-description-budget"?: RuleSeverity;
205
214
  /**
206
215
  * Flag a skill/agent whose `---` frontmatter block EXISTS but isn't valid YAML
207
216
  * — fields may not parse as intended. CAVEAT: a real YAML parser (js-yaml) is
@@ -346,6 +355,18 @@ export interface VigilesConfig {
346
355
  audit?: {
347
356
  measure?: boolean;
348
357
  };
358
+ /**
359
+ * `vigiles eval` preferences. `apiVersion` is the hand-bumped **behavior epoch**
360
+ * folded into the eval LOCK's input hash (`src/eval-lock.ts`): bump it when a
361
+ * harness-side change YOU made (a CLAUDE.md edit, a global hook) would shift
362
+ * eval outputs but isn't otherwise visible to the lock — so `vigiles eval
363
+ * --check` reports the committed eval results STALE and forces a local re-run.
364
+ * Default 1. Distinct from the (auto-resolved) `claude` CLI version, which is
365
+ * recorded as provenance but deliberately NOT hashed.
366
+ */
367
+ eval?: {
368
+ apiVersion?: number;
369
+ };
349
370
  }
350
371
  /** Valid marker types for rule detection. */
351
372
  export type MarkerType = "headings" | "checkboxes";
@@ -67,6 +67,9 @@ exports.DEFAULT_RULES = {
67
67
  "disallowed-tools-contract": "warn",
68
68
  // Deterministic NCD precision proxy (near-identical skill descriptions) — warn.
69
69
  "description-overlap": "warn",
70
+ // A model-invocable skill's description so long the trigger signal is buried —
71
+ // WARN only (heuristic proxy, generous 500-char budget); never gates.
72
+ "skill-description-budget": "warn",
70
73
  // Malformed-YAML frontmatter — WARN only (js-yaml is stricter than some loaders).
71
74
  "frontmatter-valid": "warn",
72
75
  // A mcp_tool hook incomplete / targeting an undeclared server — on by default at warn.
@@ -0,0 +1,20 @@
1
+ /** The hidden runtime umbrella — not a human-facing verb to document. */
2
+ export declare const COVERAGE_EXEMPT: readonly ["hook-runtime"];
3
+ /**
4
+ * Whether `verb` appears in a COMMAND context anywhere in `content`:
5
+ * `vigiles <verb>` or a backtick-prefixed `` `<verb> ``. Generous on purpose
6
+ * (see the file header) — over-counting a verb as documented is the SAFE
7
+ * direction; under-counting would cry wolf.
8
+ */
9
+ export declare function verbMentioned(verb: string, content: string): boolean;
10
+ /**
11
+ * Find public verbs not MENTIONED in any of the given doc files. Pure — the
12
+ * caller supplies file contents (so it runs over the repo's `docs/` in a test,
13
+ * or any file set). `verbs` defaults to the canonical {@link VERBS}; `exempt`
14
+ * drops the hidden umbrella.
15
+ */
16
+ export declare function findUndocumentedVerbs(docs: readonly {
17
+ readonly path: string;
18
+ readonly content: string;
19
+ }[], verbs?: readonly string[], exempt?: readonly string[]): string[];
20
+ //# sourceMappingURL=doc-command-coverage.d.ts.map