vigiles 15.0.0 → 15.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -297,12 +297,16 @@ export interface RulesConfig {
297
297
  */
298
298
  "hook-block-ineffective"?: RuleSeverity;
299
299
  /**
300
- * Flag a hook `matcher` string that silently never fires — a close typo of a
301
- * built-in tool (`bash`→`Bash`), or a malformed/undeclared MCP form
302
- * (`mcp_memory_*` instead of `mcp__memory__.*`, or a server the plugin doesn't
303
- * declare). High-precision (close-typo only; MCP gated on a declared set,
304
- * built-ins allowlisted; wildcards/regex skipped). Default "warn"; raise to
305
- * "error" to gate CI. Same detector as `scan` (hookMatcherFindings). See
300
+ * Flag a hook `matcher` string that doesn't fire the way it reads — a close
301
+ * typo of a built-in tool (`bash`→`Bash`), a matcher that doesn't COMPILE, an
302
+ * MCP pattern that can match no tool name at all (`mcp_memory_*` instead of
303
+ * `mcp__memory__.*`), an MCP pattern too narrow for real server naming
304
+ * (`mcp__[^_]+__[^_]+` can't cross the `_` in `mcp__Google_Calendar__…`), or a
305
+ * server the plugin doesn't declare. A matcher is a PATTERN, so patterns are
306
+ * validated by compiling and probing, never by literal shape. High-precision
307
+ * (close-typo only; MCP gated on a declared set, built-ins allowlisted;
308
+ * match-all and alternation skipped). Default "warn"; raise to "error" to gate
309
+ * CI. Same detector as `scan` (hookMatcherFindings). See
306
310
  * docs/rules/hook-matcher.md.
307
311
  */
308
312
  "hook-matcher"?: RuleSeverity;
@@ -94,8 +94,9 @@ exports.DEFAULT_RULES = {
94
94
  // A hook that looks like it blocks but silently doesn't (#19009) — WARN by
95
95
  // default (FP-safe literal patterns); raise to error to gate CI.
96
96
  "hook-block-ineffective": "warn",
97
- // A hook matcher that never fires (tool typo / wrong MCP form) — WARN by
98
- // default (high-precision); raise to error to gate CI.
97
+ // A hook matcher that doesn't fire as written (tool typo, an MCP pattern
98
+ // that matches no tool name, or one too narrow for real server names) — WARN
99
+ // by default (high-precision); raise to error to gate CI.
99
100
  "hook-matcher": "warn",
100
101
  };
101
102
  const DEFAULT_CONFIG = {
package/dist/eval.js CHANGED
@@ -75,6 +75,7 @@ const harness_test_js_1 = require("./harness-test.js");
75
75
  const eval_cache_js_1 = require("./eval-cache.js");
76
76
  const eval_lock_js_1 = require("./eval-lock.js");
77
77
  const stats_js_1 = require("./stats.js");
78
+ const check_count_js_1 = require("./check-count.js");
78
79
  const tool_intercept_js_1 = require("./tool-intercept.js");
79
80
  const tool_stub_js_1 = require("./tool-stub.js");
80
81
  function writeFiles(cwd, files) {
@@ -1165,6 +1166,9 @@ function evalArmsInputs(spec, cfg) {
1165
1166
  };
1166
1167
  }
1167
1168
  async function runEvalWith(spec, runner) {
1169
+ // Tell the CLI runner this script exercised the harness, so a file that runs
1170
+ // NOTHING can be told apart from one that ran and passed. See check-count.ts.
1171
+ (0, check_count_js_1.recordCheck)();
1168
1172
  const trials = spec.trials ?? 5;
1169
1173
  const spacing = (spec.spacingSec ?? 4) * 1000;
1170
1174
  const concurrency = spec.concurrency ?? 1;
@@ -1637,6 +1641,8 @@ function assertTriggerDiversity(spec) {
1637
1641
  }
1638
1642
  }
1639
1643
  async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
1644
+ // Tell the CLI runner this script exercised the harness (see check-count.ts).
1645
+ (0, check_count_js_1.recordCheck)();
1640
1646
  // Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
1641
1647
  assertTriggerDiversity(spec);
1642
1648
  // Model floor (default Sonnet): trigger-rate under-measures selection on a
@@ -64,6 +64,7 @@ exports.assertTriggerRate = assertTriggerRate;
64
64
  const harness_test_js_1 = require("./harness-test.js");
65
65
  const hook_program_js_1 = require("./core/hook-program.js");
66
66
  const check_js_1 = require("./check.js");
67
+ const check_count_js_1 = require("./check-count.js");
67
68
  const agent_result_js_1 = require("./adapters/claude-code/agent-result.js");
68
69
  const stats_js_1 = require("./stats.js");
69
70
  const eval_baseline_js_1 = require("./eval-baseline.js");
@@ -142,6 +143,15 @@ function assertHookAllowed(r) {
142
143
  fail(`expected the hook to allow, but it blocked (exit ${String(r.exitCode)}, decision ${String(r.decision)})`);
143
144
  }
144
145
  }
146
+ /**
147
+ * `runHookProgram`, counted. An in-process hook decision is an observation the
148
+ * CLI runner can see — without it, a `*.harness.*` file that only tests compiled
149
+ * hooks would look like it did nothing at all. See check-count.ts.
150
+ */
151
+ function runCountedHookProgram(hook, event) {
152
+ (0, check_count_js_1.recordCheck)();
153
+ return (0, hook_program_js_1.runHookProgram)(hook, event);
154
+ }
145
155
  /** Render a {@link HookProgramOutcome} for an assertion message. */
146
156
  function describeOutcome(o) {
147
157
  if (o.kind === "decision")
@@ -157,14 +167,14 @@ function describeOutcome(o) {
157
167
  * check, use {@link assertHookBlocked} over `runHook`.)
158
168
  */
159
169
  function assertHookDenies(hook, event) {
160
- const o = (0, hook_program_js_1.runHookProgram)(hook, event);
170
+ const o = runCountedHookProgram(hook, event);
161
171
  if (o.kind !== "decision" || o.decision.kind !== "deny") {
162
172
  fail(`expected the hook to deny, got ${describeOutcome(o)}`);
163
173
  }
164
174
  }
165
175
  /** Assert a COMPILED hook allows an event (in-process). The twin of {@link assertHookDenies}. */
166
176
  function assertHookAllows(hook, event) {
167
- const o = (0, hook_program_js_1.runHookProgram)(hook, event);
177
+ const o = runCountedHookProgram(hook, event);
168
178
  if (o.kind !== "decision" || o.decision.kind !== "allow") {
169
179
  fail(`expected the hook to allow, got ${describeOutcome(o)}`);
170
180
  }
@@ -181,7 +191,7 @@ function assertHookAllows(hook, event) {
181
191
  * stdout-vs-stderr never enters into it.
182
192
  */
183
193
  function assertHookNotices(hook, event, matcher) {
184
- const o = (0, hook_program_js_1.runHookProgram)(hook, event);
194
+ const o = runCountedHookProgram(hook, event);
185
195
  if (o.kind !== "reaction" || o.reaction.kind !== "notice") {
186
196
  fail(`expected the hook to notice, got ${describeOutcome(o)}`);
187
197
  }
@@ -204,7 +214,7 @@ function assertHookNotices(hook, event, matcher) {
204
214
  * A `run(…)` reaction is not silent for this purpose: the hook still reacted.
205
215
  */
206
216
  function assertHookSilent(hook, event) {
207
- const o = (0, hook_program_js_1.runHookProgram)(hook, event);
217
+ const o = runCountedHookProgram(hook, event);
208
218
  if (o.kind !== "reaction") {
209
219
  fail(`expected a react hook, got ${describeOutcome(o)}`);
210
220
  }
@@ -46,6 +46,7 @@ const node_fs_1 = require("node:fs");
46
46
  const node_os_1 = require("node:os");
47
47
  const node_path_1 = require("node:path");
48
48
  const adapter_conformance_js_1 = require("./adapter-conformance.js");
49
+ const check_count_js_1 = require("./check-count.js");
49
50
  const runtime_js_1 = require("./adapters/claude-code/runtime.js");
50
51
  const mock_model_js_1 = require("./mock-model.js");
51
52
  const plugin_loader_js_1 = require("./adapters/claude-code/plugin-loader.js");
@@ -410,6 +411,9 @@ function makeResult(cwd, out, parsed, turns, modelRequests) {
410
411
  * only — requesting confinement for another harness throws.
411
412
  */
412
413
  async function runHarnessTest(spec, opts = {}) {
414
+ // Tell the CLI runner this script exercised the harness, so a file that runs
415
+ // NOTHING can be told apart from one that ran and passed. See check-count.ts.
416
+ (0, check_count_js_1.recordCheck)();
413
417
  const adapter = opts.adapter;
414
418
  // Default (no adapter): the unchanged Claude Code driver — keeps the
415
419
  // sandbox/confined path and behaviour byte-for-byte identical.
@@ -35,6 +35,7 @@ const node_os_1 = require("node:os");
35
35
  const node_path_1 = require("node:path");
36
36
  const egress_js_1 = require("./egress.js");
37
37
  const sandbox_js_1 = require("./sandbox.js");
38
+ const check_count_js_1 = require("./check-count.js");
38
39
  /**
39
40
  * The run orchestration with injectable spawn seams: pick direct vs. confined
40
41
  * via the safe-by-default policy (`decideSandbox`), then assemble the result.
@@ -42,6 +43,11 @@ const sandbox_js_1 = require("./sandbox.js");
42
43
  * with fake spawners — no real bwrap.
43
44
  */
44
45
  function runScriptWith(command, stdin, opts, deps) {
46
+ // Tell the CLI runner this script exercised the harness, so a `*.harness.*`
47
+ // file that runs NOTHING can be told apart from one that ran and passed. Here,
48
+ // at the primitive, so `runHook` and a bare `runScript` both count. See
49
+ // check-count.ts.
50
+ (0, check_count_js_1.recordCheck)();
45
51
  // Allowlisted egress is its own confined path (bwrap netns + slirp4netns +
46
52
  // nft); it can't run unconfined, so it refuses outright when the tooling is
47
53
  // absent rather than falling back to a direct run that ignores the allowlist.
package/dist/scan-core.js CHANGED
@@ -69,6 +69,44 @@ function frontmatter(md) {
69
69
  color: (0, frontmatter_read_js_1.frontmatterScalar)(fm, "color"),
70
70
  };
71
71
  }
72
+ /**
73
+ * Whether this unit's declared TOOL CONTRACT is unreadable — the frontmatter
74
+ * block exists but is not valid YAML.
75
+ *
76
+ * 🔴 WHY SCORING MUST NOT USE THE SALVAGE. The shared reader is deliberately
77
+ * lenient: on a block js-yaml rejects it falls back to a regex salvage, so the
78
+ * live PreToolUse rail still has *something* to enforce and the other fields
79
+ * keep working. That is right for a rail and wrong for a SCORE. Measured
80
+ * 2026-08-08: `readFrontmatter(bad)` returns `{data: null, malformed: true}` —
81
+ * the tool KNOWS the block is broken — while `frontmatterList(…, "allowed-tools")`
82
+ * on the same block returns `["Read","Bash"]`, the narrow contract the author
83
+ * MEANT. Strict js-yaml on it throws `bad indentation of a mapping entry`. So a
84
+ * unit whose contract a strict loader rejects was graded as though it had
85
+ * declared exactly that narrow contract: the Safety ring read BETTER than the
86
+ * truth, on the optimistic branch, in the tool whose own thesis is that the
87
+ * presence of a declaration is not the enforcement of it.
88
+ *
89
+ * The trifecta detector is therefore told the list is a SALVAGE, and reads it as
90
+ * one: it can only make the verdict worse (a salvaged all-three still convicts),
91
+ * and anything short of that falls back to what a strict loader really yields —
92
+ * no contract, i.e. inherits-all. The finding says which happened, so the author
93
+ * can tell a dropped grade from a real capability. Strictly one-directional, the
94
+ * same shape as the inherits-all monotonicity fix (#119). `frontmatter-valid`
95
+ * reports the broken block itself; this is the half that stops the SCORE
96
+ * disagreeing with it.
97
+ *
98
+ * DELIBERATELY NOT WIDER. The typo / never-available / MCP-server /
99
+ * disallowed-tools cross-references keep using the salvage: they are diagnostics,
100
+ * and suppressing them on a malformed file DELETES findings, which moves the
101
+ * grade the optimistic way — the direction this whole fix exists to close. A real
102
+ * vendored plugin in `test/dogfood` proves the point: `madappgang-frontend`'s
103
+ * `tester.md` has both a malformed description and an explicit all-three-legs
104
+ * tool list, and its `AskUserQuestion` never-available finding is true whether or
105
+ * not the block parses.
106
+ */
107
+ function contractIsUnreadable(md) {
108
+ return (0, frontmatter_read_js_1.readFrontmatter)(md).malformed;
109
+ }
72
110
  function escapeRe(s) {
73
111
  return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
74
112
  }
@@ -257,11 +295,17 @@ function scanSkills(files, cls, ctx) {
257
295
  // invocable skill can be hijacked by attacker content, so a user-invoked one is
258
296
  // excluded. A skill with no `allowed-tools` line inherits all → advisory.
259
297
  const skillTools = (0, agent_tools_js_1.parseAgentToolList)(md, "allowed-tools");
298
+ // Whether that list came out of a block a strict loader REJECTS — in which
299
+ // case it is a salvage, and the trifecta detector must read it as one. See
300
+ // `contractIsUnreadable`.
301
+ const contractUnreadable = contractIsUnreadable(md);
260
302
  // No `allowed-tools:` line (null) → inherits all → wildcard sentinel; an
261
303
  // EXPLICIT empty `[]` means zero tools → no trifecta (don't collapse them).
262
304
  const trifecta = userInvoked
263
305
  ? null
264
- : (0, lethal_trifecta_js_1.lethalTrifectaIssues)(skillTools ?? ["*"], dialect);
306
+ : (0, lethal_trifecta_js_1.lethalTrifectaIssues)(skillTools ?? ["*"], dialect, {
307
+ contractUnreadable,
308
+ });
265
309
  out.push({
266
310
  name: fm.name ?? skillName(path),
267
311
  // Report the real on-disk path, not the synthetic materialize key (E1).
@@ -323,6 +367,14 @@ ctx) {
323
367
  if (!cls.isAgent(path))
324
368
  continue;
325
369
  const tools = (0, agent_tools_js_1.parseAgentTools)(md);
370
+ // Whether that list came out of a block a strict loader REJECTS — see
371
+ // `contractIsUnreadable`. Scoped to the trifecta (the Safety ring) on
372
+ // purpose: the typo / MCP / disallowed-tools cross-references below are
373
+ // DIAGNOSTICS, and dropping them on a malformed file would delete real
374
+ // findings — moving the grade in the optimistic direction this fix exists to
375
+ // stop. A salvage is too weak to earn a unit a clean bill of health; it is
376
+ // plenty strong enough to convict.
377
+ const contractUnreadable = contractIsUnreadable(md);
326
378
  // An inherits-all agent (no `tools:` line) grants access to every tool
327
379
  // including every side-effecting one — pass the wildcard sentinel so
328
380
  // effectSurface correctly classifies it as `"unrestricted"`.
@@ -359,7 +411,9 @@ ctx) {
359
411
  // inherits-all agent (no `tools:` line → tools === null) is the advisory
360
412
  // case — pass the wildcard sentinel so it's distinguished from an EXPLICIT
361
413
  // empty `tools: []` (zero tools → no trifecta). One detector, no drift.
362
- trifecta: (0, lethal_trifecta_js_1.lethalTrifectaIssues)(tools ?? ["*"], dialect),
414
+ trifecta: (0, lethal_trifecta_js_1.lethalTrifectaIssues)(tools ?? ["*"], dialect, {
415
+ contractUnreadable,
416
+ }),
363
417
  });
364
418
  }
365
419
  return out.sort((a, b) => a.name.localeCompare(b.name));
package/dist/scan.d.ts CHANGED
@@ -257,9 +257,10 @@ export interface ScanReport {
257
257
  */
258
258
  readonly hookBlockFindings: readonly HookBlockFinding[];
259
259
  /**
260
- * Hook `matcher` strings that silently never fire — a tool-name typo or a
261
- * malformed/undeclared MCP form. Shared by `scan` and the `hook-matcher` lint
262
- * rule (one detector, no drift).
260
+ * Hook `matcher` strings that don't fire as written — a tool-name typo, a
261
+ * matcher that doesn't compile, an MCP pattern that matches no tool name or is
262
+ * too narrow for real server naming, or an undeclared MCP server. Shared by
263
+ * `scan` and the `hook-matcher` lint rule (one detector, no drift).
263
264
  */
264
265
  readonly hookMatcherFindings: readonly HookMatcherFinding[];
265
266
  /** Skills/agents whose `---` block isn't valid YAML — informational (may still load via salvage). */
package/dist/scan.js CHANGED
@@ -431,7 +431,7 @@ function formatScanReport(r) {
431
431
  out.push(...section("Misplaced plugin directories", r.pluginLayoutIssues.map((p) => ` ✗ ${p.message}`)));
432
432
  out.push(...section("Lethal trifecta across delegation (blast radius)", r.delegationTrifecta.map((d) => ` ⚠ ${d.finding.name} (${d.path}): ${d.finding.message}`)));
433
433
  out.push(...section("Ineffective hook guards (false confidence)", r.hookBlockFindings.map((h) => ` ✗ [${h.event}] ${h.scriptPath ?? "(inline)"}: ${h.message}`)));
434
- out.push(...section("Hook matchers that never fire", r.hookMatcherFindings.map((m) => ` ✗ ${m.message}`)));
434
+ out.push(...section("Hook matchers that don't fire as written", r.hookMatcherFindings.map((m) => ` ✗ ${m.message}`)));
435
435
  const facts = [];
436
436
  if (r.commands > 0)
437
437
  facts.push(`Commands: ${String(r.commands)}`);
@@ -212,7 +212,7 @@ function reportDeductions(r) {
212
212
  {
213
213
  n: r.hookMatcherFindings.length,
214
214
  weight: exports.W_MISSING_HOOK,
215
- label: "hook matcher(s) that never fire (typo / wrong MCP form)",
215
+ label: "hook matcher(s) that don't fire as written (dead, or too narrow for real MCP names)",
216
216
  },
217
217
  // NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
218
218
  // ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
package/dist/testing.d.ts CHANGED
@@ -10,6 +10,7 @@
10
10
  * boundary forbids importing `src/adapters/*` from here. See
11
11
  * `research/adapter-api-design.md`.
12
12
  */
13
+ export { recordCheck } from "./check-count.js";
13
14
  export { runScript } from "./run-script.js";
14
15
  export type { RunScriptOptions, ScriptRunResult } from "./run-script.js";
15
16
  export { runHook, propertyHook } from "./run-hook.js";
package/dist/testing.js CHANGED
@@ -30,7 +30,15 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
30
30
  for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
31
31
  };
32
32
  Object.defineProperty(exports, "__esModule", { value: true });
33
- exports.runHarness = exports.runHarnessTest = exports.judge = exports.hookFired = exports.loadHook = exports.stubSkillBody = exports.parseClaudeRun = exports.claudeEvalDriver = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.measureTriggerRate = exports.measureArms = exports.measure = exports.runEval = exports.propertyHook = exports.runHook = exports.runScript = void 0;
33
+ exports.runHarness = exports.runHarnessTest = exports.judge = exports.hookFired = exports.loadHook = exports.stubSkillBody = exports.parseClaudeRun = exports.claudeEvalDriver = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.measureTriggerRate = exports.measureArms = exports.measure = exports.runEval = exports.propertyHook = exports.runHook = exports.runScript = exports.recordCheck = void 0;
34
+ // --- reporting: how much did this script actually do? ---
35
+ // `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
36
+ // prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
37
+ // exporting an object of tests nobody calls). The tiers below count themselves;
38
+ // call `recordCheck()` yourself when you assert some OTHER way — `node:assert`,
39
+ // vitest's `expect` — so those are visible to the runner too. See check-count.ts.
40
+ var check_count_js_1 = require("./check-count.js");
41
+ Object.defineProperty(exports, "recordCheck", { enumerable: true, get: function () { return check_count_js_1.recordCheck; } });
34
42
  // --- unit tier: runScript (the primitive) + runHook (it, plus a decision) ---
35
43
  // `runScript` runs any program and reports what it DID (exit, both streams,
36
44
  // writes, egress). `runHook` is that plus the hook protocol: event to stdin,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "15.0.0",
3
+ "version": "15.0.2",
4
4
  "description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
5
5
  "keywords": [
6
6
  "claude-code",