vigiles 15.0.2 → 15.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. package/dist/adapters/claude-code/layout.js +3 -0
  2. package/dist/adapters/claude-code/run-scripts.d.ts +21 -9
  3. package/dist/adapters/claude-code/run-scripts.js +33 -22
  4. package/dist/adapters/codex/driver.js +3 -1
  5. package/dist/adapters/codex/eval.js +5 -0
  6. package/dist/audit-score.js +32 -6
  7. package/dist/check-count.d.ts +62 -3
  8. package/dist/check-count.js +123 -5
  9. package/dist/check.d.ts +55 -0
  10. package/dist/check.js +91 -0
  11. package/dist/cli.js +456 -54
  12. package/dist/core/bash-effects.d.ts +41 -0
  13. package/dist/core/bash-effects.js +278 -0
  14. package/dist/core/foreign-runner.d.ts +185 -0
  15. package/dist/core/foreign-runner.js +228 -0
  16. package/dist/core/harness-driver.d.ts +18 -1
  17. package/dist/core/hook-program.d.ts +319 -20
  18. package/dist/core/hook-program.js +743 -49
  19. package/dist/core/layout.d.ts +13 -0
  20. package/dist/core/lethal-trifecta.d.ts +93 -0
  21. package/dist/core/lethal-trifecta.js +409 -65
  22. package/dist/core/markdown.d.ts +32 -0
  23. package/dist/core/markdown.js +36 -0
  24. package/dist/core/merge-conflict.d.ts +50 -0
  25. package/dist/core/merge-conflict.js +84 -0
  26. package/dist/core/test-file-ext.d.ts +80 -0
  27. package/dist/core/test-file-ext.js +90 -0
  28. package/dist/core/types.d.ts +11 -0
  29. package/dist/coverage-artifact.d.ts +383 -0
  30. package/dist/coverage-artifact.js +586 -0
  31. package/dist/coverage-evidence.d.ts +87 -79
  32. package/dist/coverage-evidence.js +212 -211
  33. package/dist/coverage-probe.d.ts +125 -0
  34. package/dist/coverage-probe.js +700 -0
  35. package/dist/doc-commands.d.ts +119 -0
  36. package/dist/doc-commands.js +158 -0
  37. package/dist/eval-lock.d.ts +9 -0
  38. package/dist/eval-lock.js +14 -2
  39. package/dist/eval.js +28 -0
  40. package/dist/fs-walk.d.ts +80 -0
  41. package/dist/fs-walk.js +148 -0
  42. package/dist/harness-assert.js +42 -3
  43. package/dist/harness-test.js +31 -2
  44. package/dist/judge.js +4 -0
  45. package/dist/leaderboard.d.ts +16 -1
  46. package/dist/leaderboard.js +19 -2
  47. package/dist/load-hook.js +6 -1
  48. package/dist/mock-model.d.ts +58 -4
  49. package/dist/mock-model.js +133 -5
  50. package/dist/observe.d.ts +1 -1
  51. package/dist/observe.js +44 -1
  52. package/dist/plugin-loader.js +52 -13
  53. package/dist/run-hook.d.ts +58 -0
  54. package/dist/run-hook.js +70 -0
  55. package/dist/run-script.js +75 -0
  56. package/dist/scaffold-test.d.ts +3 -0
  57. package/dist/scaffold-test.js +10 -5
  58. package/dist/scan-behavioral.js +4 -0
  59. package/dist/scan-core.d.ts +22 -0
  60. package/dist/scan-core.js +46 -14
  61. package/dist/scan-files.js +19 -2
  62. package/dist/scan.d.ts +22 -5
  63. package/dist/scan.js +67 -13
  64. package/dist/skill-contract.d.ts +46 -0
  65. package/dist/skill-contract.js +201 -0
  66. package/dist/skill-refs.d.ts +67 -0
  67. package/dist/skill-refs.js +117 -0
  68. package/dist/test-coverage-files.d.ts +7 -1
  69. package/dist/test-coverage-files.js +66 -55
  70. package/dist/test-coverage.d.ts +155 -19
  71. package/dist/test-coverage.js +430 -91
  72. package/dist/testing.d.ts +8 -2
  73. package/dist/testing.js +19 -1
  74. package/dist/trigger-containment.d.ts +85 -0
  75. package/dist/trigger-containment.js +125 -0
  76. package/dist/ts-runner-caps.d.ts +16 -0
  77. package/dist/ts-runner-caps.js +43 -0
  78. package/dist/unit.d.ts +2 -2
  79. package/dist/unit.js +2 -1
  80. package/hooks/eval-lock-nudge.sh +14 -6
  81. package/package.json +2 -2
  82. package/skills/test-harness/SKILL.md +68 -2
package/dist/cli.js CHANGED
@@ -55,6 +55,7 @@ const compile_generator_js_1 = require("./core/compile-generator.js");
55
55
  const action_gate_js_1 = require("./action-gate.js");
56
56
  const guards_js_1 = require("./core/guards.js");
57
57
  const hook_program_js_1 = require("./core/hook-program.js");
58
+ const merge_conflict_js_1 = require("./core/merge-conflict.js");
58
59
  const hook_install_js_1 = require("./hook-install.js");
59
60
  const hook_providers_js_1 = require("./core/hook-providers.js");
60
61
  const toml_1 = require("@iarna/toml");
@@ -68,6 +69,7 @@ const skill_runtime_js_1 = require("./adapters/claude-code/skill-runtime.js");
68
69
  const linters_js_1 = require("./core/linters.js");
69
70
  const harness_test_js_1 = require("./harness-test.js");
70
71
  const run_scripts_js_1 = require("./adapters/claude-code/run-scripts.js");
72
+ const coverage_artifact_js_1 = require("./coverage-artifact.js");
71
73
  const integrity_js_1 = require("./core/integrity.js");
72
74
  const adopt_js_1 = require("./core/adopt.js");
73
75
  const coverage_js_1 = require("./core/coverage.js");
@@ -2907,35 +2909,90 @@ function checkIntegrityForFiles(files, severity, silent) {
2907
2909
  return severity === "error" ? errorCount : 0;
2908
2910
  }
2909
2911
  /**
2910
- * Apply the per-kind `untested-skill` / `untested-subagent` / `untested-hook` rules:
2911
- * find skills/agents/hooks with no test or eval (see src/test-coverage.ts). Each
2912
- * kind is gated by its OWN rule severity — a kind set to `false` is not scanned;
2913
- * "warn" prints but never fails CI; "error" fails (exit 2). Returns the raw
2914
- * untested count plus the severity-gated error count.
2912
+ * The layout of the harness this repo actually targets — `--harness=`/config/
2913
+ * auto-detect, the SAME precedence `lint`, `compile` and `audit` use.
2914
+ *
2915
+ * 🔴 ONE RESOLVER, BECAUSE THE OTHER CALLERS DEFAULTED. `lint` threads
2916
+ * `adapter.layout` into `findUntestedSurfaces`; two entry points that call the very
2917
+ * same detector passed only a path, so `test-coverage.ts` fell back to
2918
+ * `claudeCodeLayout`. Reproduced 2026-08-12 on a `.codex/config.toml` fixture whose
2919
+ * only surface is `.codex/skills/foo/SKILL.md`: with the layout, one surface is
2920
+ * discovered; without it, ZERO — and a zero-surface scan is not an error, it is a
2921
+ * silence. The edit-time nudge said nothing, and `vigiles test`'s coverage
2922
+ * recorder threw away every probe an execution had legitimately produced, because
2923
+ * `recordsFrom` cannot resolve a probe against a surface list that is empty.
2924
+ *
2925
+ * Non-throwing on purpose: an unknown `--harness=` is a hard error where the user
2926
+ * typed it, but these callers are a hook and a post-run recorder, and neither may
2927
+ * turn a bad config key into a failed edit or a red test run.
2915
2928
  */
2916
- function checkUntestedSurfaces(config, silent, adapter, scanRoot) {
2929
+ function harnessLayoutFor(root, config, flag) {
2930
+ try {
2931
+ return (0, adapter_registry_js_1.resolveHarnessSelection)({
2932
+ root,
2933
+ flag,
2934
+ configHarness: (0, adapter_registry_js_1.normalizeHarnessList)(config?.harness),
2935
+ }).adapter.layout;
2936
+ }
2937
+ catch {
2938
+ return adapter_registry_js_1.defaultAdapter.layout;
2939
+ }
2940
+ }
2941
+ /**
2942
+ * The `untested-*` rules AS THE DETECTOR TAKES THEM: which kinds this repo
2943
+ * enabled, and the discovery options merged from whichever of the three rules
2944
+ * carries them (`testGlobs` / `exclude` / `testExtension` are shared).
2945
+ *
2946
+ * 🔴 ONE READER, BECAUSE THE SECOND ONE DRIFTED. `vigiles lint` read the config
2947
+ * here; the PostToolUse nudge (`evalLockNudgeHookCommand`) called the same
2948
+ * detector with nothing but `basePath`. Reproduced 2026-08-12 on a two-file
2949
+ * fixture: with `"untested-skill": false` lint printed nothing and the nudge
2950
+ * still told the agent the skill was untested; with a configured
2951
+ * `testGlobs: ["**\/*.check.mjs"]` lint printed "all 1 surface(s) have a test"
2952
+ * while the nudge said "no test or eval covers it". A nudge contradicting the
2953
+ * linter of the same repo, in the same second, teaches people to ignore both.
2954
+ */
2955
+ function untestedRules(config) {
2917
2956
  const rules = config?.rules;
2918
2957
  const skillSev = (0, types_js_1.ruleSeverity)(rules?.["untested-skill"]);
2919
2958
  const agentSev = (0, types_js_1.ruleSeverity)(rules?.["untested-subagent"]);
2920
2959
  const hookSev = (0, types_js_1.ruleSeverity)(rules?.["untested-hook"]);
2921
- if (!skillSev && !agentSev && !hookSev)
2922
- return { untested: 0, errors: 0 };
2923
- const sevFor = (kind) => kind === "skill" ? skillSev : kind === "agent" ? agentSev : hookSev;
2924
- // Test-discovery options (testGlobs/exclude) are shared; merge them from
2925
- // whichever of the three rules carries them.
2926
2960
  const opts = {
2927
2961
  ...(0, types_js_1.ruleOptions)(rules?.["untested-skill"]),
2928
2962
  ...(0, types_js_1.ruleOptions)(rules?.["untested-subagent"]),
2929
2963
  ...(0, types_js_1.ruleOptions)(rules?.["untested-hook"]),
2930
2964
  };
2965
+ return {
2966
+ severity: (kind) => kind === "skill" ? skillSev : kind === "agent" ? agentSev : hookSev,
2967
+ anyEnabled: skillSev !== false || agentSev !== false || hookSev !== false,
2968
+ options: {
2969
+ skills: skillSev !== false,
2970
+ agents: agentSev !== false,
2971
+ hooks: hookSev !== false,
2972
+ testGlobs: opts.testGlobs,
2973
+ exclude: opts.exclude,
2974
+ // Without this the `testExtension` documented on TestCoverageOptions was a
2975
+ // config key nothing read: a TypeScript-shaped repo got `.ts` suggestions
2976
+ // however the author configured it. Prose isn't policy, in our own CLI.
2977
+ testExtension: opts.testExtension,
2978
+ },
2979
+ };
2980
+ }
2981
+ /**
2982
+ * Apply the per-kind `untested-skill` / `untested-subagent` / `untested-hook` rules:
2983
+ * find skills/agents/hooks with no test or eval (see src/test-coverage.ts). Each
2984
+ * kind is gated by its OWN rule severity — a kind set to `false` is not scanned;
2985
+ * "warn" prints but never fails CI; "error" fails (exit 2). Returns the raw
2986
+ * untested count plus the severity-gated error count.
2987
+ */
2988
+ function checkUntestedSurfaces(config, silent, adapter, scanRoot) {
2989
+ const { severity: sevFor, anyEnabled, options } = untestedRules(config);
2990
+ if (!anyEnabled)
2991
+ return { untested: 0, errors: 0 };
2931
2992
  const report = (0, test_coverage_js_1.findUntestedSurfaces)({
2993
+ ...options,
2932
2994
  basePath: scanRoot,
2933
2995
  layout: adapter.layout,
2934
- skills: skillSev !== false,
2935
- agents: agentSev !== false,
2936
- hooks: hookSev !== false,
2937
- testGlobs: opts.testGlobs,
2938
- exclude: opts.exclude,
2939
2996
  });
2940
2997
  if (!silent) {
2941
2998
  console.log("\nUntested surfaces:\n");
@@ -3267,12 +3324,19 @@ function checkDescriptionBudget(config, silent, adapter, scanRoot) {
3267
3324
  }
3268
3325
  /**
3269
3326
  * Apply the `lethal-trifecta` rule: a unit (subagent / model-invocable skill)
3270
- * whose declared tools hold all three legs (read-private + ingest-untrusted +
3271
- * exfiltrate) is a prompt-injection exfil path (Meta's Rule of Two). Reuses
3272
- * `scanPlugin`'s `trifectaFindings` (a capability SET-intersection, one detector,
3273
- * no drift). Warning by default; "error" gates CI. Surfaces across BOTH subagents
3274
- * and skills, so it is NOT gated on the `subagents` capability — a skill-only
3275
- * harness still has the surface.
3327
+ * holding all three legs (read-private + ingest-untrusted + exfiltrate) is a
3328
+ * prompt-injection exfil path (Meta's Rule of Two). Reuses `scanPlugin`'s
3329
+ * `trifectaFindings` (one detector, no drift). Warning by default; "error" gates
3330
+ * CI. Surfaces across BOTH subagents and skills, so it is NOT gated on the
3331
+ * `subagents` capability — a skill-only harness still has the surface.
3332
+ *
3333
+ * 🔴 The unfenced-skill group is printed ONCE and gets NO per-file GitHub
3334
+ * annotation. A skill's `allowed-tools:` pre-approves rather than restricts
3335
+ * (measured 2026-08-11), so every skill without a `disallowed-tools:` fence is in
3336
+ * this state — annotating each of them would stamp the same sentence on every
3337
+ * SKILL.md in the repo on every CI run, which is how a rule gets switched off. The
3338
+ * COUNT is unchanged (units, not lines), so exit codes and the CI gate are
3339
+ * unaffected by the collapse.
3276
3340
  */
3277
3341
  function checkLethalTrifecta(config, silent, adapter, scanRoot) {
3278
3342
  const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["lethal-trifecta"]);
@@ -3286,11 +3350,23 @@ function checkLethalTrifecta(config, silent, adapter, scanRoot) {
3286
3350
  return { issues: 0, errors: 0 };
3287
3351
  }
3288
3352
  if (found.length > 0 && !silent) {
3353
+ const mark = sev === "error" ? "✗" : "⚠";
3354
+ const level = sev === "error" ? "error" : "warning";
3289
3355
  console.log("\nLethal-trifecta check:\n");
3356
+ const unfenced = found.filter((t) => t.kind === "skill" && t.finding.fence === "none");
3290
3357
  for (const t of found) {
3358
+ if (unfenced.includes(t))
3359
+ continue;
3291
3360
  const msg = `${t.kind} ${t.name}: ${t.finding.message}`;
3292
- console.log(` ${sev === "error" ? "✗" : "⚠"} ${t.path}: ${msg}`);
3293
- ghAnnotate(sev === "error" ? "error" : "warning", msg, t.path);
3361
+ console.log(` ${mark} ${t.path}: ${msg}`);
3362
+ ghAnnotate(level, msg, t.path);
3363
+ }
3364
+ if (unfenced.length > 0) {
3365
+ console.log(` ${mark} ${String(unfenced.length)} skill(s) declare no \`disallowed-tools:\` fence, so each ` +
3366
+ `inherits every tool the session grants and holds all three legs. ` +
3367
+ `\`allowed-tools:\` pre-approves, it does not restrict. ` +
3368
+ `Add one \`disallowed-tools:\` line per skill to drop a leg: ` +
3369
+ unfenced.map((t) => t.name).join(", "));
3294
3370
  }
3295
3371
  }
3296
3372
  return { issues: found.length, errors: sev === "error" ? found.length : 0 };
@@ -4028,6 +4104,160 @@ function resolveEvalLockEnv(args) {
4028
4104
  }
4029
4105
  return env;
4030
4106
  }
4107
+ /**
4108
+ * Turn a completed run into `.vigiles/coverage.json` — the composition root for
4109
+ * the execution tier of coverage (`coverage-artifact.ts`).
4110
+ *
4111
+ * It resolves here, and nowhere earlier, because resolution needs DISCOVERY: a
4112
+ * script reports the reference it saw (`hooks/pre-edit.sh`, `plugin:my-skill`)
4113
+ * and only the repo can say whether that is a surface. Doing it inside the test
4114
+ * process would mean scanning a user's repo from inside their test.
4115
+ *
4116
+ * Best-effort throughout: a repo that cannot be scanned or a `.vigiles/` that
4117
+ * cannot be written must never turn a green run red.
4118
+ *
4119
+ * 🔴 A RUN RETRACTS AS WELL AS RECORDS. The scripts that executed are handed to
4120
+ * `mergeRuns` so their previous records go before the new ones land — otherwise
4121
+ * a script edited to stop exercising a surface leaves that surface permanently
4122
+ * "measured by a run" (see the retraction note on `mergeRuns`). This is why the
4123
+ * cheap early return on "nothing new to record" is gone: a run that now reports
4124
+ * NOTHING is exactly the case worth writing down.
4125
+ */
4126
+ /**
4127
+ * Resolve this run's probes against the repo's discovered surfaces.
4128
+ *
4129
+ * `harnessFlag` is the `--harness=` the user typed on THIS run — see the layout
4130
+ * note below for why the flag has to travel this far.
4131
+ */
4132
+ function resolveRecords(cwd, runs, tier, harnessFlag) {
4133
+ // 🔴 DISCOVERY MUST USE THE ACTIVE LAYOUT, or resolution silently resolves
4134
+ // nothing. This called the detector with a path alone, so a Codex repo (surfaces
4135
+ // only under `.codex/skills/`) discovered ZERO surfaces, `recordsFrom` matched
4136
+ // every successful probe against an empty list and dropped it, and `vigiles
4137
+ // test` / `vigiles eval` never granted execution coverage there — while
4138
+ // `vigiles lint`, which passes `adapter.layout` to the same function, saw the
4139
+ // surfaces perfectly well. Empty discovery does not fail; it just quietly means
4140
+ // "nothing ran".
4141
+ //
4142
+ // 🔴 AND THE ACTIVE LAYOUT INCLUDES THE FLAG. The first fix passed only
4143
+ // `(cwd, loadConfig())`, so `--harness=` — which `resolveHarnessSelection` ranks
4144
+ // ABOVE config and auto-detect, and which `cli-flag-check.ts` accepts on every
4145
+ // verb — was dropped exactly here. `vigiles test --harness=codex` in a repo that
4146
+ // auto-detects (or is configured) as Claude Code then discovered the Claude
4147
+ // surfaces, matched the run's Codex probes against them, resolved none, and
4148
+ // recorded no execution coverage: the same silent empty set as before, reached
4149
+ // through the one input that exists to override the other two.
4150
+ /**
4151
+ * The plugin namespaces that mean THIS repo, for {@link resolveProbe}.
4152
+ *
4153
+ * A skill activation is reported as `plugin:skill` and a subagent dispatch as
4154
+ * `plugin:agent`, so resolving one needs to know which plugin we ARE — otherwise
4155
+ * `other-plugin:foo` credits a local `foo` (that was the defect). Two sources,
4156
+ * and both are needed:
4157
+ *
4158
+ * 1. `.claude-plugin/plugin.json#name` — the repo's own declared name, used
4159
+ * when the run installed the repo AS a plugin (`pluginDir`).
4160
+ * 2. `vigiles-loose-skills` — the synthetic name OUR OWN packaging gives a
4161
+ * loose `.claude/skills` dir (`packageSkillsDir`, and `underTestSource`'s
4162
+ * fallback when a plugin manifest has no name). A repo that is not a plugin
4163
+ * still reports namespaced ids under it, so omitting it would drop every
4164
+ * trigger-rate record for the documented one-liner.
4165
+ *
4166
+ * An unreadable or nameless manifest contributes nothing rather than a guess.
4167
+ */
4168
+ function selfNamespaces(cwd) {
4169
+ const names = new Set(["vigiles-loose-skills"]);
4170
+ try {
4171
+ const raw = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(cwd, ".claude-plugin", "plugin.json"), "utf-8");
4172
+ const name = JSON.parse(raw).name;
4173
+ if (typeof name === "string" && name.trim())
4174
+ names.add(name.trim());
4175
+ }
4176
+ catch {
4177
+ /* no manifest, or unreadable → the synthetic name alone */
4178
+ }
4179
+ return [...names];
4180
+ }
4181
+ const scan = (0, test_coverage_js_1.findUntestedSurfaces)({
4182
+ basePath: cwd,
4183
+ layout: harnessLayoutFor(cwd, (0, validate_js_1.loadConfig)(), harnessFlag),
4184
+ });
4185
+ return (0, coverage_artifact_js_1.recordsFrom)({
4186
+ runs,
4187
+ surfaces: [...scan.covered, ...scan.untested],
4188
+ tier,
4189
+ at: new Date().toISOString(),
4190
+ selfNamespaces: selfNamespaces(cwd),
4191
+ // The root an ABSOLUTE command ref must lie beneath to be OURS. Without it a
4192
+ // harness that executed `/tmp/fixture/hooks/pre.sh` credited this repo's own
4193
+ // `hooks/pre.sh`, because the tail matched. See the ladder note on
4194
+ // `resolveProbe`; absent a root, absolute refs abstain rather than guess.
4195
+ root: cwd,
4196
+ readSurface: (p) => {
4197
+ try {
4198
+ return (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(cwd, p), "utf-8");
4199
+ }
4200
+ catch {
4201
+ return null;
4202
+ }
4203
+ },
4204
+ });
4205
+ }
4206
+ /**
4207
+ * Merge one run's records into `.vigiles/coverage.json`, retractions included.
4208
+ *
4209
+ * The verb's `kind` maps to the coverage TIER here, so the one call site stays a
4210
+ * single line. `harnessFlag` is the caller's `--harness=` — a SHARED flag (`cli-flag-check.ts`)
4211
+ * that every verb accepts and that `resolveHarnessSelection` ranks above config
4212
+ * and auto-detection. It has to travel all the way to discovery: see the layout
4213
+ * note on {@link resolveRecords} for what dropping it silently did.
4214
+ */
4215
+ function recordRunCoverage(cwd, results, kind, harnessFlag) {
4216
+ const tier = kind === "test" ? "harness" : "eval";
4217
+ try {
4218
+ const previous = (0, coverage_artifact_js_1.readCoverageArtifact)(cwd);
4219
+ // `root` so an ABSOLUTE target (`vigiles test /abs/x.harness.mjs`) retracts
4220
+ // what the same file recorded when the default glob found it relatively —
4221
+ // `discoverScripts` passes an existing file's argument through verbatim, so
4222
+ // the spelling that reaches `by` is whatever the person typed.
4223
+ const executed = { scripts: (0, coverage_artifact_js_1.executedScripts)(results), tier, root: cwd };
4224
+ const runs = (0, coverage_artifact_js_1.runsFromResults)(results);
4225
+ // Discovery is only needed to RESOLVE new probes; a pure retraction needs
4226
+ // nothing but the artifact, so a repo scan is skipped when there are none.
4227
+ const records = runs.length === 0 ? [] : resolveRecords(cwd, runs, tier, harnessFlag);
4228
+ const merged = (0, coverage_artifact_js_1.mergeRuns)(previous?.runs ?? [], records, executed);
4229
+ // Nothing to add and nothing withdrawn — leave the file exactly as it is, so
4230
+ // a repo with no artifact still gets none (the "absent artifact = today's
4231
+ // behaviour" property) and a green no-op run doesn't churn the timestamp.
4232
+ if (records.length === 0 && merged.length === (previous?.runs.length ?? 0))
4233
+ return;
4234
+ const commit = gitHead(cwd);
4235
+ (0, coverage_artifact_js_1.writeCoverageArtifact)(cwd, {
4236
+ v: coverage_artifact_js_1.COVERAGE_ARTIFACT_VERSION,
4237
+ generated: new Date().toISOString(),
4238
+ ...(commit ? { commit } : {}),
4239
+ runs: merged,
4240
+ });
4241
+ }
4242
+ catch {
4243
+ /* recording is never allowed to fail a run */
4244
+ }
4245
+ }
4246
+ /** Short HEAD of the checkout, or "" when this is not a git repo. */
4247
+ function gitHead(cwd) {
4248
+ try {
4249
+ const { spawnSync } = require("node:child_process");
4250
+ const r = spawnSync("git", ["rev-parse", "--short", "HEAD"], {
4251
+ cwd,
4252
+ encoding: "utf-8",
4253
+ stdio: ["ignore", "pipe", "ignore"],
4254
+ });
4255
+ return r.status === 0 ? (r.stdout ?? "").trim() : "";
4256
+ }
4257
+ catch {
4258
+ return "";
4259
+ }
4260
+ }
4031
4261
  async function handleRunScripts(kind, args, restArgs) {
4032
4262
  const cwd = process.cwd();
4033
4263
  // Harness/eval scripts may be authored in JS or TS (see run-scripts.ts).
@@ -4102,6 +4332,11 @@ async function handleRunScripts(kind, args, restArgs) {
4102
4332
  env.VIGILES_TRIALS = trialsFlag.split("=")[1];
4103
4333
  console.log(`Running ${String(files.length)} ${kind} file(s):\n`);
4104
4334
  const results = (0, run_scripts_js_1.runScripts)(files, cwd, env);
4335
+ // Write down WHAT the run exercised, so `lint`/`audit` can answer "tested?"
4336
+ // from execution instead of from a matching file name. Not a new verb and not
4337
+ // a flag: the run already happened, and this is the runner recording what it
4338
+ // saw — the same shape as the flight-recorder ledger it already appends to.
4339
+ recordRunCoverage(cwd, results, kind, harnessFlagFrom(args));
4105
4340
  console.log("\n" + (0, run_scripts_js_1.formatScriptSummary)(results));
4106
4341
  if ((0, run_scripts_js_1.anyFailed)(results))
4107
4342
  process.exit(1);
@@ -4590,13 +4825,20 @@ function isInstructionFile(file) {
4590
4825
  return INSTRUCTION_FILE.test((0, node_path_1.basename)(file));
4591
4826
  }
4592
4827
  /**
4593
- * PostToolUse-hook entrypoint: when the agent edits an eval input (a `SKILL.md`
4594
- * trigger surface or an `*.eval.*` script), and committed eval locks exist, inject
4595
- * a NON-BLOCKING reminder to re-run `vigiles eval --update`. Self-gating (silent
4596
- * until a lock is committed), never blocks, never runs an eval — a reminder, not a
4597
- * gate (the gate is `eval --check` in CI). The harness-neutral nudge lives in
4598
- * `evalLockNudge`; both CC and Codex deliver it as `additionalContext` on
4599
- * `PostToolUse` (confirmed — see docs/harness-testing-codex.md).
4828
+ * PostToolUse-hook entrypoint for the two edit-time test reminders. When the agent
4829
+ * edits a skill/agent surface or an `*.eval.*` script it injects, NON-BLOCKING:
4830
+ *
4831
+ * 1. `skillTestNudge` — the surface has no test/eval at all, or has a harness but
4832
+ * was never EVALUATED (so nothing has measured that its description fires).
4833
+ * Reuses the `untested-skill` detector and hands off to the `test-harness`
4834
+ * skill for the tier→API table — the agent knows it needs a test, not which
4835
+ * runner. Fires at edit time, closing the gap that `untested-skill` only ran
4836
+ * when someone typed `vigiles lint`.
4837
+ * 2. `evalLockNudge` — a committed lock may now be stale; re-run `eval --update`.
4838
+ *
4839
+ * Never blocks, never runs an eval — reminders, not gates (the gate is
4840
+ * `eval --check` in CI). Both CC and Codex deliver these as `additionalContext`
4841
+ * on `PostToolUse` (confirmed — see docs/harness-testing-codex.md).
4600
4842
  */
4601
4843
  function evalLockNudgeHookCommand() {
4602
4844
  let raw = "";
@@ -4618,7 +4860,36 @@ function evalLockNudgeHookCommand() {
4618
4860
  return;
4619
4861
  const cwd = process.cwd();
4620
4862
  const target = (0, node_path_1.relative)(cwd, (0, node_path_1.resolve)(cwd, file)) || file;
4621
- const msg = (0, eval_lock_js_1.evalLockNudge)(target, (0, node_path_1.resolve)(cwd, eval_lock_js_1.DEFAULT_LOCK_DIR));
4863
+ // 🔴 THE SAME CONFIG `vigiles lint` READS. This used to pass `basePath` alone,
4864
+ // so a repo that had switched `untested-skill` off, or pointed `testGlobs` at
4865
+ // its own test names, got a nudge asserting the opposite of what its own
4866
+ // linter said (see `untestedRules`). A rule the author DISABLED must not come
4867
+ // back through a hook; a test the author CONFIGURED must count here too.
4868
+ //
4869
+ // The disable travels inside `options` — a kind whose rule is off arrives as
4870
+ // `skills: false` / `agents: false` and is not scanned, so the detector has
4871
+ // nothing to report. No extra "is anything enabled" guard here on purpose: a
4872
+ // second gate saying the same thing is a branch no test can distinguish from
4873
+ // its absence (measured — the mutation passed), i.e. the dead-fragment class
4874
+ // this same change removed from the runner table.
4875
+ const config = (0, validate_js_1.loadConfig)();
4876
+ const { options } = untestedRules(config);
4877
+ // 🔴 THE SAME LAYOUT `vigiles lint` RESOLVES, for the same reason as the config
4878
+ // above. This used to pass `basePath` alone, so the detector fell back to the
4879
+ // Claude Code layout: in a Codex repo whose surfaces live at
4880
+ // `.codex/skills/foo/SKILL.md`, it discovered NOTHING, found the edited skill in
4881
+ // nothing, and stayed silent — while `vigiles lint`, one `adapter.layout` away,
4882
+ // reported that very skill as untested. A hook that contradicts the linter by
4883
+ // omission is worse than no hook: nobody is looking for the message that never
4884
+ // came.
4885
+ const layout = harnessLayoutFor(cwd, config);
4886
+ // Two nudges, one entry, most-urgent first. The lock reminder is SELF-GATING:
4887
+ // silent until a lock is committed — so a repo that has never written a test
4888
+ // heard nothing at all, while a repo that already tests got reminded. That is
4889
+ // backwards, and `untested-skill` already stated the missing half correctly;
4890
+ // it just lived in `vigiles lint`, which someone has to run by hand.
4891
+ const msg = (0, test_coverage_js_1.skillTestNudge)(target, { ...options, basePath: cwd, layout }) ??
4892
+ (0, eval_lock_js_1.evalLockNudge)(target, (0, node_path_1.resolve)(cwd, eval_lock_js_1.DEFAULT_LOCK_DIR));
4622
4893
  if (!msg)
4623
4894
  return;
4624
4895
  process.stdout.write(JSON.stringify({
@@ -5006,6 +5277,63 @@ function emitGate(decision, on, mode, file) {
5006
5277
  return; // emit nothing, exit 0
5007
5278
  }
5008
5279
  }
5280
+ /**
5281
+ * Every `package.json` between the hook file and the filesystem root, plus the
5282
+ * project's `.vigilesrc.json` — the files Node and the runtime must PARSE for a
5283
+ * compiled hook to load at all. Repo-relative-ish paths, for a message.
5284
+ *
5285
+ * Walking UP is not decoration: `vigiles/hook` is a bare specifier, so Node reads
5286
+ * the nearest `package.json` (and every one above it) while resolving it. The
5287
+ * observed wedge came from a `package.json` the author was not thinking about at
5288
+ * the time — it had merge-conflict markers in it, nothing to do with hooks.
5289
+ */
5290
+ function hookLoadPathFiles(hookFile) {
5291
+ const files = [];
5292
+ let dir = (0, node_path_1.dirname)((0, node_path_1.resolve)(process.cwd(), hookFile));
5293
+ for (;;) {
5294
+ const pkg = (0, node_path_1.resolve)(dir, "package.json");
5295
+ files.push(pkg);
5296
+ // 🔴 THE WALK STOPS AT THE FIRST ONE THAT EXISTS, and this list is now also
5297
+ // the set of writes the repair door accepts, so its length is a blast
5298
+ // radius. Unbounded, it reached `/home/package.json` and `/package.json` —
5299
+ // files Node never opens once a nearer one is found. MEASURED against this
5300
+ // runtime, hook at `gp/p/repo/.claude/hooks/`, conflict markers planted at
5301
+ // one ancestor:
5302
+ //
5303
+ // repo pkg PRESENT , parent conflicted → loads fine
5304
+ // repo pkg absent , parent conflicted → WEDGES (cause: ../package.json)
5305
+ // repo pkg absent , parent absent, gp broken → WEDGES (cause: ../../package.json)
5306
+ // repo pkg PRESENT , parent absent, gp broken → loads fine
5307
+ // repo pkg absent , parent HEALTHY, gp broken→ loads fine
5308
+ //
5309
+ // A missing one is still pushed before the check: `package.json` may be the
5310
+ // file the author has to CREATE, and it is the commonest repair of all.
5311
+ if ((0, node_fs_1.existsSync)(pkg))
5312
+ break;
5313
+ const up = (0, node_path_1.dirname)(dir);
5314
+ if (up === dir)
5315
+ break;
5316
+ dir = up;
5317
+ }
5318
+ files.push((0, node_path_1.resolve)(process.cwd(), ".vigilesrc.json"));
5319
+ return files;
5320
+ }
5321
+ /**
5322
+ * The conflicted files on this hook's load path, if any — the difference between
5323
+ * "your hook is broken" and "your repo is mid-merge and the hook is collateral".
5324
+ */
5325
+ function conflictedLoadPathFiles(hookFile) {
5326
+ return hookLoadPathFiles(hookFile)
5327
+ .filter((p) => {
5328
+ try {
5329
+ return ((0, node_fs_1.existsSync)(p) && (0, merge_conflict_js_1.hasMergeConflictMarkers)((0, node_fs_1.readFileSync)(p, "utf-8")));
5330
+ }
5331
+ catch {
5332
+ return false; // unreadable is a different problem; don't guess about it
5333
+ }
5334
+ })
5335
+ .map((p) => (0, node_path_1.relative)(process.cwd(), p) || p);
5336
+ }
5009
5337
  /**
5010
5338
  * Print the loud stderr banner that accompanies a REPAIR-only pass-through, and
5011
5339
  * return true — the caller allows exactly this one tool call. Shared by the two
@@ -5013,10 +5341,11 @@ function emitGate(decision, on, mode, file) {
5013
5341
  */
5014
5342
  function announceRepairEscape(file, why) {
5015
5343
  console.error(`vigiles: hook ${file} ${why}.\n` +
5016
- `vigiles: ALLOWING this one call because it is the repair action ` +
5017
- `(\`vigiles compile ${file}\`, or an edit to the hook itself) — without ` +
5018
- `this the gate blocks the only command that can fix it.\n` +
5019
- `vigiles: every OTHER tool call stays BLOCKED until the hook is recompiled.`);
5344
+ `vigiles: ALLOWING this one call because it is a repair or recovery action ` +
5345
+ `(a write to ${file}, to ${hookStampPath(file)}, or to one of ` +
5346
+ `${merge_conflict_js_1.HARNESS_CONFIG_FILES.join(", ")}) ` +
5347
+ `— without this the gate blocks the only actions that can fix it.\n` +
5348
+ `vigiles: every OTHER tool call stays BLOCKED until the hook loads again.`);
5020
5349
  return true;
5021
5350
  }
5022
5351
  /**
@@ -5039,12 +5368,16 @@ function verifyStampOrRefuse(file, event) {
5039
5368
  const { stamp } = JSON.parse((0, node_fs_1.readFileSync)(stampPath, "utf-8"));
5040
5369
  const source = (0, node_fs_1.readFileSync)((0, node_path_1.resolve)(process.cwd(), file), "utf-8");
5041
5370
  if (stamp && !(0, hook_program_js_1.verifyHookStamp)(source, stamp)) {
5042
- if ((0, hook_program_js_1.isStampRepairEvent)(event, file)) {
5371
+ if ((0, hook_program_js_1.isStampRepairEvent)(event, file, process.cwd())) {
5043
5372
  announceRepairEscape(file, "does not match its compiled stamp");
5044
5373
  return;
5045
5374
  }
5046
- console.error(`vigiles: hook ${file} does not match its compiled stamp (tampered). ` +
5047
- `If YOU edited it, run \`vigiles compile ${file}\` to regenerate the stamp.`);
5375
+ console.error(`vigiles: hook ${file} does not match its compiled stamp (tampered).\n` +
5376
+ `vigiles: if YOU edited it, the way out is a FILE WRITE, not a command — ` +
5377
+ `this refusal blocks the recompile too. Either edit ${file} back to what ` +
5378
+ `was compiled, or clear its stamp by writing \`{}\` into ` +
5379
+ `${stampPath}. The hook then runs UNSTAMPED but still ENFORCES, so ` +
5380
+ `\`vigiles compile ${file}\` goes through the normal gate.`);
5048
5381
  process.exit(2);
5049
5382
  }
5050
5383
  }
@@ -5052,15 +5385,28 @@ function verifyStampOrRefuse(file, event) {
5052
5385
  /* unreadable sidecar → don't block a live session on it */
5053
5386
  }
5054
5387
  }
5388
+ /**
5389
+ * Say so, ON STDERR, when this event's path cannot be matched against a
5390
+ * repo-relative prefix — an absolute `file_path` and no project root anywhere.
5391
+ * The failure it announces is otherwise invisible: the hook runs, exits 0, and
5392
+ * decides on nothing. Deliberately silent for a relative `file_path` (decidable
5393
+ * without a root) so it cannot be mistaken for a react hook's `notice`.
5394
+ */
5395
+ function warnIfPathUndecidable(event, root) {
5396
+ const warning = (0, hook_program_js_1.undecidablePathWarning)(event.tool_input?.file_path, root);
5397
+ if (warning !== undefined)
5398
+ console.error(warning);
5399
+ }
5055
5400
  /**
5056
5401
  * `vigiles hook-runtime run-program <file>` — the runtime the compiled hooks block
5057
5402
  * points at. Reads the live event on stdin, loads the typed program, verifies
5058
5403
  * its stamp, and dispatches by role: a gate exits 2 + reason on `deny`; an
5059
5404
  * inject prints `additionalContext`; a react runs its effect-classified
5060
5405
  * command. A hook that won't load — or whose stamp is stale — fails CLOSED
5061
- * (exit 2), never silent-allow, with ONE loudly-announced exception: the repair
5062
- * action itself ({@link isStampRepairEvent}), or the repo wedges with no way to
5063
- * recompile.
5406
+ * (exit 2), never silent-allow, with two loudly-announced exceptions: the repair
5407
+ * action itself ({@link isStampRepairEvent}), and — on a LOAD failure only — the
5408
+ * load-path repair WRITE ({@link isLoadPathRepairEvent}), or the repo wedges
5409
+ * with no way to fix whatever broke the load path.
5064
5410
  */
5065
5411
  async function runHookProgramCommand(file) {
5066
5412
  if (!file) {
@@ -5082,21 +5428,70 @@ async function runHookProgramCommand(file) {
5082
5428
  catch {
5083
5429
  /* malformed → empty event */
5084
5430
  }
5431
+ // The root repo-relative path prefixes resolve against. `$CLAUDE_PROJECT_DIR`
5432
+ // first (the same root the harness resolved THIS hook's own path against),
5433
+ // then the payload's `cwd`; never `process.cwd()`, which under a git worktree
5434
+ // can be a different checkout. See `projectRootOf`.
5435
+ const projectRoot = (0, hook_program_js_1.projectRootOf)(event, process.env);
5085
5436
  let program;
5086
5437
  try {
5087
5438
  program = await loadHookProgram(file);
5088
5439
  }
5089
5440
  catch {
5090
- // Same bootstrap escape as the stale stamp below: an edit that leaves the
5091
- // hook unloadable (a typo mid-edit) otherwise blocks the recompile that
5092
- // would fix it. A hook that can't load enforces nothing either way, so
5093
- // refusing the repair only wedges the repo. Everything else still fails
5094
- // CLOSED (exit 2), never silent-allow.
5095
- if ((0, hook_program_js_1.isStampRepairEvent)(event, file)) {
5096
- announceRepairEscape(file, "cannot be loaded");
5441
+ // A LOAD failure is a fact about the harness, not a verdict about the
5442
+ // command that happened to arrive — so it must still fail CLOSED (a gate
5443
+ // that cannot run must not wave traffic through), but the two things it owes
5444
+ // the author are different from a `deny`'s: name the real cause, and leave a
5445
+ // way back.
5446
+ //
5447
+ // Escapes, both announced loudly on stderr:
5448
+ // - the stale-stamp one (an edit to the hook itself / `vigiles compile`),
5449
+ // for the hook broken mid-edit;
5450
+ // - the RECOVERY set, for the case where the hook is fine and something
5451
+ // else on its load path is not (observed 2026-08-10: `package.json` left
5452
+ // holding merge-conflict markers → Node can't resolve `vigiles/hook` →
5453
+ // no compiled hook loads → the Bash gate refuses `git merge --abort`,
5454
+ // the one command that undoes the cause. Irreversible from inside the
5455
+ // session; it was fixed by hand-editing the JSON, because file tools do
5456
+ // not go through PreToolUse(Bash)).
5457
+ // Everything else stays BLOCKED, and the escapes are whitelists of commands
5458
+ // that are WRITES — see `isLoadPathRepairEvent` for why no command is one.
5459
+ const conflicted = conflictedLoadPathFiles(file);
5460
+ const cause = conflicted.length > 0
5461
+ ? `cannot be loaded — ${conflicted.join(", ")} contains merge-conflict ` +
5462
+ `markers, so Node cannot resolve \`vigiles/hook\` from it (the hook itself ` +
5463
+ `may be fine)`
5464
+ : "cannot be loaded";
5465
+ if ((0, hook_program_js_1.isLoadPathRepairEvent)(event, file, {
5466
+ // The root the REST of this runtime already uses: `hookStampPath` and
5467
+ // `verifyStampOrRefuse` read the hook and its sidecar via `process.cwd()`,
5468
+ // so a repair accepted against any other root would name a file the
5469
+ // runtime never reads. The hook's own path cannot supply it (a hook sits
5470
+ // at any depth, and a `.git` probe would be a disk read core does not do).
5471
+ root: process.cwd(),
5472
+ loadPathFiles: hookLoadPathFiles(file),
5473
+ })) {
5474
+ announceRepairEscape(file, cause);
5097
5475
  return;
5098
5476
  }
5099
- console.error(`vigiles: cannot load hook program ${file}`);
5477
+ console.error(`vigiles: hook ${file} ${cause}.\n` +
5478
+ `vigiles: this is the state of the HARNESS, not a decision about your ` +
5479
+ `command — the gate never ran. Blocking anyway (a gate that cannot run ` +
5480
+ `must not pass traffic).\n` +
5481
+ `vigiles: the way out is a FILE WRITE, not a command — under a tool that ` +
5482
+ `WRITES (Write/Edit/MultiEdit); a Read of the same path repairs nothing ` +
5483
+ `and is refused. Fix whichever of ` +
5484
+ `${file}, ${merge_conflict_js_1.HARNESS_CONFIG_FILES.join(", ")} is broken — those writes are ` +
5485
+ `allowed even while this refuses, and a Bash gate never gated file tools ` +
5486
+ `at all. The hook then loads and the gate decides normally again.\n` +
5487
+ `vigiles: those paths resolve under ${process.cwd()} — plus any ancestor ` +
5488
+ `\`package.json\` Node actually reads, so whatever is named above as the ` +
5489
+ `cause is writable. A path in a DIFFERENT checkout is refused: it cannot ` +
5490
+ `repair this failure.\n` +
5491
+ `vigiles: no command is allowed, deliberately. \`git merge --abort\` and ` +
5492
+ `\`git checkout\` RUN \`.git/hooks/*\` (measured: reference-transaction, ` +
5493
+ `post-checkout), and \`vigiles compile\` loads the hook through the same ` +
5494
+ `resolver that just failed.`);
5100
5495
  process.exit(2);
5101
5496
  return;
5102
5497
  }
@@ -5108,7 +5503,8 @@ async function runHookProgramCommand(file) {
5108
5503
  return;
5109
5504
  }
5110
5505
  case "react": {
5111
- const reaction = (0, hook_program_js_1.runReact)(program, event);
5506
+ warnIfPathUndecidable(event, projectRoot);
5507
+ const reaction = (0, hook_program_js_1.runReact)(program, event, projectRoot);
5112
5508
  if (reaction.kind === "run") {
5113
5509
  const { spawnSync } = require("node:child_process");
5114
5510
  const res = spawnSync(reaction.command, {
@@ -5123,12 +5519,18 @@ async function runHookProgramCommand(file) {
5123
5519
  }
5124
5520
  case "file-gate": {
5125
5521
  const ctx = await gatherHookContext(program);
5126
- emitGate((0, hook_program_js_1.decideFileGate)(program, event, ctx), program.on, (0, hook_program_js_1.hookMode)(program), file);
5522
+ warnIfPathUndecidable(event, projectRoot);
5523
+ emitGate((0, hook_program_js_1.decideFileGate)(program, event, ctx, projectRoot), program.on, (0, hook_program_js_1.hookMode)(program), file);
5127
5524
  return;
5128
5525
  }
5129
5526
  case "bash-gate": {
5130
5527
  const ctx = await gatherHookContext(program);
5131
- emitGate((0, hook_program_js_1.decideProgram)(program, event, ctx), program.on, (0, hook_program_js_1.hookMode)(program), file);
5528
+ // The same `projectRoot` the file gates get: without it every
5529
+ // repo-relative prefix in a DENYLIST matcher (`touches`/`writesTo`) is
5530
+ // matched by over-blocking alone, and with it an absolute token is placed
5531
+ // exactly. Measured bypass this closes: `sed -i s/a/b/ <abs>/paper.tex`
5532
+ // exited 0 against a guard that blocked the relative spelling.
5533
+ emitGate((0, hook_program_js_1.decideProgram)(program, event, ctx, projectRoot), program.on, (0, hook_program_js_1.hookMode)(program), file);
5132
5534
  return;
5133
5535
  }
5134
5536
  case "prompt-gate": {
@@ -5864,7 +6266,7 @@ async function main() {
5864
6266
  // The flight recorder: a compact summary of what the harness actually
5865
6267
  // DID in real sessions (hook/agent decisions), read off the local
5866
6268
  // agent-readable ledger. Empty (skipped) until something is recorded.
5867
- const ledgerSummary = (0, observe_js_1.formatLedgerSummary)(ledgerRecords);
6269
+ const ledgerSummary = (0, observe_js_1.formatLedgerSummary)(ledgerRecords, (0, eval_lock_js_1.countLocks)((0, node_path_1.resolve)(root, eval_lock_js_1.DEFAULT_LOCK_DIR)));
5868
6270
  if (ledgerSummary)
5869
6271
  console.log("\n" + ledgerSummary);
5870
6272
  }