vigiles 26.0.1 → 26.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +5 -4
  2. package/dist/adapters/claude-code/hook-condition.d.ts +46 -0
  3. package/dist/adapters/claude-code/hook-condition.js +142 -0
  4. package/dist/adapters/claude-code/hook-protocol.js +5 -0
  5. package/dist/audit-report.template.html +2 -2
  6. package/dist/cli.js +67 -115
  7. package/dist/core/bash-effects.d.ts +22 -0
  8. package/dist/core/bash-effects.js +10 -0
  9. package/dist/core/command-files.d.ts +107 -0
  10. package/dist/core/command-files.js +407 -0
  11. package/dist/core/hook-condition.d.ts +96 -0
  12. package/dist/core/hook-condition.js +63 -0
  13. package/dist/core/hook-matcher.d.ts +50 -0
  14. package/dist/core/hook-matcher.js +77 -2
  15. package/dist/core/hook-normalize.d.ts +51 -0
  16. package/dist/core/hook-normalize.js +61 -1
  17. package/dist/core/hook-program.d.ts +62 -1
  18. package/dist/core/hook-program.js +15 -1
  19. package/dist/core/hook-protocol.d.ts +16 -0
  20. package/dist/core/linters.js +97 -58
  21. package/dist/core/shell-vars.d.ts +74 -0
  22. package/dist/core/shell-vars.js +270 -0
  23. package/dist/core/skill-resources.d.ts +22 -1
  24. package/dist/core/skill-resources.js +2 -1
  25. package/dist/doc-test-script-coverage.d.ts +52 -0
  26. package/dist/doc-test-script-coverage.js +66 -0
  27. package/dist/guardrail-check.d.ts +29 -0
  28. package/dist/guardrail-check.js +69 -10
  29. package/dist/harness-assert.d.ts +8 -5
  30. package/dist/harness-assert.js +8 -5
  31. package/dist/harness-resolve-hooks.mjs +14 -37
  32. package/dist/hook-state-store.d.ts +143 -0
  33. package/dist/hook-state-store.js +241 -0
  34. package/dist/hook.d.ts +3 -1
  35. package/dist/hook.js +3 -1
  36. package/dist/run-hook.d.ts +33 -1
  37. package/dist/run-hook.js +46 -2
  38. package/dist/run-script.d.ts +94 -0
  39. package/dist/run-script.js +47 -26
  40. package/dist/scan-core.js +21 -3
  41. package/dist/score-core.d.ts +21 -1
  42. package/dist/score-core.js +30 -6
  43. package/dist/self-resolve.d.mts +20 -0
  44. package/dist/self-resolve.mjs +75 -0
  45. package/dist/spec-hooks.d.mts +10 -0
  46. package/dist/spec-hooks.mjs +17 -0
  47. package/dist/test.d.ts +5 -0
  48. package/dist/test.js +23 -2
  49. package/dist/verify-plugin-guards.d.ts +194 -0
  50. package/dist/verify-plugin-guards.js +822 -0
  51. package/package.json +1 -1
@@ -137,6 +137,62 @@ export interface RunScriptDeps {
137
137
  /** Run the command confined with an allowlisted-egress netns. */
138
138
  readonly egress: ScriptSpawner;
139
139
  }
140
+ /**
141
+ * True when the exit code is the shell's own report that it never reached the
142
+ * program — so nothing of the harness ran and no surface may be credited.
143
+ *
144
+ * 🔴 THE NUMBERS ARE MEASURED, NOT CITED. 126/127 are POSIX *conventions*; what
145
+ * matters is what `spawnSync(cmd, { shell: true })` actually returns here.
146
+ * Measured 2026-08-12, `/bin/sh` → dash (Debian), each case a real file:
147
+ *
148
+ * ```
149
+ * 126 | direct non-executable | ./noexec.sh | out="" | Permission denied
150
+ * 127 | direct missing | ./missing.sh | out="" | not found
151
+ * 127 | bare unknown command | nosuchcmd-xyz | out="" | not found
152
+ * 127 | bad shebang (exists+x) | ./badshebang.sh | out="" | env: 'nosuchinterp'
153
+ * 126 | is a directory | ./ | out="" | Permission denied
154
+ * 3 | real hook, exit 3 | ./exit3.sh | out="ran" |
155
+ * 0 | real hook, exit 0 | ./ok.sh | out="ran" |
156
+ * ```
157
+ *
158
+ * The exposure this closes: a harness may legitimately assert that a hook is
159
+ * NOT executable (`assert.equal(runHook("./hooks/a.sh").exitCode, 126)`). That
160
+ * test passes, records a check — and the unconditional probe used to credit
161
+ * `hooks/a.sh` with an execution-tier record although only `/bin/sh` ran. The
162
+ * file exists and IS a discovered surface, so `resolveProbe` resolves it happily.
163
+ *
164
+ * ⚠️ TWO SHAPES ARE DELIBERATELY MISSED, both toward SILENCE (a missed probe
165
+ * costs one coverage line; a false grant costs the claim). Measured, same run:
166
+ *
167
+ * ```
168
+ * 2 | sh + missing | sh ./missing.sh | out="" | cannot open
169
+ * 0 | sh + non-executable | sh ./noexec.sh | out="ran" |
170
+ * 0 | bash + non-executable | bash ./noexec.sh | out="ran" |
171
+ * 127 | ran, THEN failed to launch| ./ok.sh && ./missing.sh | out="ran" |
172
+ * ```
173
+ *
174
+ * - `sh <missing>` exits **2** under dash, which is Claude Code's BLOCK code —
175
+ * indistinguishable from a gate legitimately denying, so it cannot be encoded.
176
+ * It is also mostly moot: a path that does not exist was never discovered as a
177
+ * surface, and `resolveProbe` matches only discovered surfaces.
178
+ * - `sh <file>` / `bash <file>` on a non-executable file exit **0 and print
179
+ * "ran"** — the interpreter reads the file as an argument, so the exec bit is
180
+ * irrelevant and the hook genuinely EXECUTED. Those must keep attributing, and
181
+ * do.
182
+ * - A compound whose last leaf fails to launch reports 126/127 for the whole
183
+ * line even though an earlier leaf ran. We abstain: silence, not a false grant.
184
+ *
185
+ * A hook that deliberately exits 126/127 itself is missed the same way, and the
186
+ * same direction.
187
+ *
188
+ * @internal Exported so the guard sweep asks the SAME question rather than
189
+ * re-deriving it (one-detector-no-drift). It reads these codes for the opposite
190
+ * purpose — not "may I credit this file with coverage?" but "may I score this
191
+ * guard at all?" — and the answer is the same fact: the shell never reached the
192
+ * program, so nothing the exit code says is the program's opinion. Not part of
193
+ * the public API.
194
+ */
195
+ export declare function shellNeverLaunched(status: number | null | undefined): boolean;
140
196
  /**
141
197
  * The run orchestration with injectable spawn seams: pick direct vs. confined
142
198
  * via the safe-by-default policy (`decideSandbox`), then assemble the result.
@@ -144,6 +200,44 @@ export interface RunScriptDeps {
144
200
  * with fake spawners — no real bwrap.
145
201
  */
146
202
  export declare function runScriptWith(command: string, stdin: string, opts: RunScriptOptions, deps: RunScriptDeps): ScriptRunResult;
203
+ /**
204
+ * Which of the three ways to start a script this run gets — or the refusal.
205
+ *
206
+ * 🔴 THE ONE PLACE THAT DECIDES, because a SECOND place that decided the same
207
+ * thing got it wrong. `experimental_verifyPluginGuards` must know whether a run
208
+ * will be confined BEFORE it runs anything: a confined run starts in a fresh
209
+ * empty directory, so a relative script that exists here will not exist there,
210
+ * and an interpreter that cannot open its script exits 2 — this harness's DENY
211
+ * code — which is then scored as a block. It answered that question with its own
212
+ * copy of the expression below, and the copy was a term short: `recordEgress`
213
+ * selects the netns recorder and therefore confinement, and the copy did not
214
+ * say so. So a `recordEgress` sweep pre-flighted against the host's cwd and env,
215
+ * then ran somewhere neither existed.
216
+ *
217
+ * Returning the ROUTE rather than a boolean is what makes that unrepeatable: the
218
+ * runner below dispatches on it and owns no policy of its own, and the pre-flight
219
+ * asks the same function instead of re-deriving the answer. A future option that
220
+ * selects confinement is added HERE, once, and every reader inherits it.
221
+ *
222
+ * A refusal counts as non-direct, and deliberately: the sweep is describing the
223
+ * environment a run WOULD have, and the environment it would have refused to run
224
+ * unconfined in is the confined one.
225
+ */
226
+ export type ScriptRunRoute = {
227
+ readonly kind: "egress";
228
+ } | {
229
+ readonly kind: "sandboxed";
230
+ } | {
231
+ readonly kind: "direct";
232
+ } | {
233
+ readonly kind: "refuse";
234
+ readonly reason: string;
235
+ };
236
+ /** What a run with these options will do, given what the machine can offer. */
237
+ export declare function routeScriptRun(opts: Pick<RunScriptOptions, "egress" | "sandbox" | "trusted" | "recordEgress">, available: {
238
+ readonly sandbox: boolean;
239
+ readonly egress: boolean;
240
+ }): ScriptRunRoute;
147
241
  export declare const REAL_DEPS: RunScriptDeps;
148
242
  export declare function egressRoutes(): boolean;
149
243
  /**
@@ -1,7 +1,9 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.REAL_DEPS = void 0;
4
+ exports.shellNeverLaunched = shellNeverLaunched;
4
5
  exports.runScriptWith = runScriptWith;
6
+ exports.routeScriptRun = routeScriptRun;
5
7
  exports.egressRoutes = egressRoutes;
6
8
  exports.runScript = runScript;
7
9
  /**
@@ -84,6 +86,13 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
84
86
  *
85
87
  * A hook that deliberately exits 126/127 itself is missed the same way, and the
86
88
  * same direction.
89
+ *
90
+ * @internal Exported so the guard sweep asks the SAME question rather than
91
+ * re-deriving it (one-detector-no-drift). It reads these codes for the opposite
92
+ * purpose — not "may I credit this file with coverage?" but "may I score this
93
+ * guard at all?" — and the answer is the same fact: the shell never reached the
94
+ * program, so nothing the exit code says is the program's opinion. Not part of
95
+ * the public API.
87
96
  */
88
97
  function shellNeverLaunched(status) {
89
98
  return status === 126 || status === 127;
@@ -100,22 +109,31 @@ function runScriptWith(command, stdin, opts, deps) {
100
109
  // at the primitive, so `runHook` and a bare `runScript` both count. See
101
110
  // check-count.ts.
102
111
  (0, check_count_js_1.recordCheck)();
103
- // Allowlisted egress is its own confined path (bwrap netns + slirp4netns +
104
- // nft); it can't run unconfined, so it refuses outright when the tooling is
105
- // absent rather than falling back to a direct run that ignores the allowlist.
106
- const res = opts.egress
107
- ? runEgress(command, stdin, opts, deps)
108
- : runConfinedOrDirect(command, stdin, opts, deps);
112
+ // The route is DECIDED elsewhere (`routeScriptRun`) and only dispatched here,
113
+ // so this function holds no confinement policy a second reader could fall
114
+ // behind — see the type's header for the sweep that fell behind it.
115
+ const route = routeScriptRun(opts, {
116
+ sandbox: deps.available,
117
+ egress: deps.egressAvailable,
118
+ });
119
+ if (route.kind === "refuse")
120
+ throw new Error(route.reason);
121
+ const spawn = route.kind === "egress"
122
+ ? deps.egress
123
+ : route.kind === "sandboxed"
124
+ ? deps.sandboxed
125
+ : deps.direct;
126
+ const res = spawn(command, stdin, opts);
109
127
  // …and WHICH surface it exercised, read off the command line that WAS
110
128
  // executed (plus `opts.env`, because the documented idiom passes the hook path
111
129
  // through one). Attribution by execution, not by file name — see
112
130
  // coverage-probe.ts. Derived here at the primitive so `runHook` and a bare
113
131
  // `runScript` both attribute without either knowing about coverage.
114
132
  //
115
- // 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM. Both
116
- // branches above can THROW before any spawner is reached: `runEgress` refuses
117
- // when the allowlist sandbox is missing, and `runConfinedOrDirect` refuses when
118
- // confinement was required and bwrap is absent. A harness that asserts exactly
133
+ // 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM. The route
134
+ // above can be `refuse` and THROW before any spawner is reached: the allowlist
135
+ // sandbox is missing, or confinement was required and bwrap is absent. A
136
+ // harness that asserts exactly
119
137
  // that refusal — `assert.throws(() => runHook(untrusted…))`, a legitimate and
120
138
  // documented test — caught the error, exited 0 with checks recorded, and the
121
139
  // runner wrote an execution-tier coverage record for a hook that never ran.
@@ -140,18 +158,21 @@ function runScriptWith(command, stdin, opts, deps) {
140
158
  filesWritten: res.filesWritten,
141
159
  };
142
160
  }
143
- /** The allowlisted-egress branch: confine-with-netns, or refuse if unavailable. */
144
- function runEgress(command, stdin, opts, deps) {
145
- if (!deps.egressAvailable) {
146
- throw new Error("refusing to run egress: { allow } without the allowlist sandbox: it " +
147
- "needs Linux + bubblewrap (bwrap) + slirp4netns + nft — install them to " +
148
- "run with a packet-layer egress allowlist, or use recordEgress to record " +
149
- "and block instead");
150
- }
151
- return deps.egress(command, stdin, opts);
152
- }
153
- /** The default branch: pick direct vs. confined via the safe-by-default policy. */
154
- function runConfinedOrDirect(command, stdin, opts, deps) {
161
+ /** What a run with these options will do, given what the machine can offer. */
162
+ function routeScriptRun(opts, available) {
163
+ // Allowlisted egress is its own confined path (bwrap netns + slirp4netns +
164
+ // nft); it can't run unconfined, so it refuses outright when the tooling is
165
+ // absent rather than falling back to a direct run that ignores the allowlist.
166
+ if (opts.egress)
167
+ return available.egress
168
+ ? { kind: "egress" }
169
+ : {
170
+ kind: "refuse",
171
+ reason: "refusing to run egress: { allow } without the allowlist sandbox: it " +
172
+ "needs Linux + bubblewrap (bwrap) + slirp4netns + nft — install them to " +
173
+ "run with a packet-layer egress allowlist, or use recordEgress to record " +
174
+ "and block instead",
175
+ };
155
176
  // Confinement follows provenance: a trusted hook (the default) runs directly;
156
177
  // marking a hook untrusted defaults it to "auto" (confine-or-refuse), so
157
178
  // foreign code is never run unconfined by accident. An explicit `sandbox`
@@ -164,13 +185,13 @@ function runConfinedOrDirect(command, stdin, opts, deps) {
164
185
  const decision = (0, sandbox_js_1.decideSandbox)({
165
186
  trusted: false,
166
187
  mode,
167
- available: deps.available,
188
+ available: available.sandbox,
168
189
  });
169
190
  if (decision.action === "throw")
170
- throw new Error(decision.reason);
191
+ return { kind: "refuse", reason: decision.reason };
171
192
  return decision.action === "sandbox"
172
- ? deps.sandboxed(command, stdin, opts)
173
- : deps.direct(command, stdin, opts);
193
+ ? { kind: "sandboxed" }
194
+ : { kind: "direct" };
174
195
  }
175
196
  /** Run the hook command directly through a shell (the default, unconfined). */
176
197
  function directSpawn(command, stdin, opts) {
package/dist/scan-core.js CHANGED
@@ -225,9 +225,23 @@ function firstBodyParagraph(md) {
225
225
  }
226
226
  return para.join(" ").trim() || undefined;
227
227
  }
228
- /** The body of a SKILL.md with the leading `---` frontmatter block stripped. */
228
+ /**
229
+ * The body of a SKILL.md with the leading `---` frontmatter block stripped, PLUS
230
+ * how many lines that block consumed.
231
+ *
232
+ * 🔴 The count is returned rather than discarded because a downstream detector
233
+ * emits LINE NUMBERS. `markdownRefs` counts from one over whatever string it is
234
+ * given, so a coordinate computed over the body is short by exactly the stripped
235
+ * frontmatter — on a 5-line block, the ref on file line 11 printed as 6 (#206).
236
+ * A struct instead of a bare string so the offset is in the caller's hand at the
237
+ * call site, not something to remember.
238
+ */
229
239
  function skillBody(md) {
230
- return md.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n?/, "");
240
+ const body = md.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n?/, "");
241
+ const consumed = md.slice(0, md.length - body.length);
242
+ // Newlines in the consumed prefix == lines before the body starts. Counts a
243
+ // CRLF file identically (the `\r` rides along inside the line).
244
+ return { body, strippedLines: consumed.split("\n").length - 1 };
231
245
  }
232
246
  /**
233
247
  * The on-disk path for a materialized file key. `loadPlugin` prefixes each
@@ -337,10 +351,14 @@ function scanSkills(files, cls, ctx) {
337
351
  // a ref under one of those declared dirs against the repo root — OPT-IN, so a
338
352
  // repo that doesn't set it is byte-identical to before (no masking of a real
339
353
  // missing bundled resource). See feedback P1-4.
340
- const resourceIssues = (0, skill_resources_js_1.skillResourceIssues)(skillBody(md), skillDir, {
354
+ const { body: bodyMd, strippedLines } = skillBody(md);
355
+ const resourceIssues = (0, skill_resources_js_1.skillResourceIssues)(bodyMd, skillDir, {
341
356
  repoRoot: sharedDirsRoot,
342
357
  sharedDirs,
343
358
  existsSync: ctx.existsSync,
359
+ // Findings carry a line number the author is meant to OPEN, so give the
360
+ // detector back the frontmatter it never saw (#206).
361
+ lineOffset: strippedLines,
344
362
  });
345
363
  // The lethal trifecta is a property of what a unit CAN do. For a SKILL that is
346
364
  // NOT its `allowed-tools:` — measured 2026-08-11, and documented by Claude Code
@@ -163,7 +163,27 @@ export declare function trifectaExposure(r: ScanReport): TrifectaExposure;
163
163
  * surfaced separately, never scored).
164
164
  */
165
165
  export declare function reportDeductions(r: ScanReport): Deduction[];
166
- /** True when a report has no loadable plugin surface at all (the empty machine). */
166
+ /**
167
+ * True when a report has no loadable plugin surface at all (the empty machine).
168
+ *
169
+ * 🔴 THE ONE PLACE the surface set is enumerated, and every scorer must ask it
170
+ * rather than write the list again. `scoreReport` used to keep its own copy and
171
+ * that copy had dropped `inlineHooks` — so a plugin whose entire contribution is
172
+ * hooks declared inline in `plugin.json` scored 0 / F with "no loadable plugin
173
+ * surface" while the same report printed `inlineHooks: 2` (#199). Measured on a
174
+ * two-inline-hook fixture: this predicate said `false` and `scoreReport` said
175
+ * `0`, against `auditScore`'s `100` — a 100-point disagreement between two
176
+ * numbers documented to be equal.
177
+ *
178
+ * A hooks-only plugin is a legitimate shape, the same way a command-only or
179
+ * MCP-only one is: a repo can ship gates and nothing else. Zero has to mean "no
180
+ * surface of any kind", because it reads as "this plugin is broken" — a
181
+ * different statement from "this plugin has no skills or subagents".
182
+ *
183
+ * Hooks count in BOTH forms: `hooks[]` is the script-backed ones, `inlineHooks`
184
+ * the shell one-liners that carry no script file to path-check. The second is
185
+ * still a gate that runs.
186
+ */
167
187
  export declare function isEmptyMachine(r: ScanReport): boolean;
168
188
  /**
169
189
  * THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
@@ -307,7 +307,27 @@ function reportDeductions(r) {
307
307
  // advisory note below). The score ranks what's BROKEN.
308
308
  ];
309
309
  }
310
- /** True when a report has no loadable plugin surface at all (the empty machine). */
310
+ /**
311
+ * True when a report has no loadable plugin surface at all (the empty machine).
312
+ *
313
+ * 🔴 THE ONE PLACE the surface set is enumerated, and every scorer must ask it
314
+ * rather than write the list again. `scoreReport` used to keep its own copy and
315
+ * that copy had dropped `inlineHooks` — so a plugin whose entire contribution is
316
+ * hooks declared inline in `plugin.json` scored 0 / F with "no loadable plugin
317
+ * surface" while the same report printed `inlineHooks: 2` (#199). Measured on a
318
+ * two-inline-hook fixture: this predicate said `false` and `scoreReport` said
319
+ * `0`, against `auditScore`'s `100` — a 100-point disagreement between two
320
+ * numbers documented to be equal.
321
+ *
322
+ * A hooks-only plugin is a legitimate shape, the same way a command-only or
323
+ * MCP-only one is: a repo can ship gates and nothing else. Zero has to mean "no
324
+ * surface of any kind", because it reads as "this plugin is broken" — a
325
+ * different statement from "this plugin has no skills or subagents".
326
+ *
327
+ * Hooks count in BOTH forms: `hooks[]` is the script-backed ones, `inlineHooks`
328
+ * the shell one-liners that carry no script file to path-check. The second is
329
+ * still a gate that runs.
330
+ */
311
331
  function isEmptyMachine(r) {
312
332
  const surfaces = r.skills.length +
313
333
  r.agents.length +
@@ -341,11 +361,15 @@ function pluralizeLabel(n, label) {
341
361
  /** Deterministic structural-health score for one scanned plugin. */
342
362
  function scoreReport(r) {
343
363
  // An empty/unloadable machine isn't healthy — it's a non-plugin or a broken
344
- // load. A command-only or MCP-only plugin (commands/*.md or .mcp.json with no
345
- // skills/agents/hooks) IS a legitimate plugin, though — Anthropic ships
346
- // command-only plugins in its own marketplace — so it must NOT score 0.
347
- const surfaces = r.skills.length + r.agents.length + r.hooks.length + r.commands;
348
- if (surfaces === 0 && !r.mcp) {
364
+ // load. A command-only, MCP-only or HOOKS-only plugin IS a legitimate plugin,
365
+ // though — Anthropic ships command-only plugins in its own marketplace, and a
366
+ // repo can reasonably ship gates and nothing else — so none of them scores 0.
367
+ //
368
+ // 🔴 Asked of `isEmptyMachine`, never re-enumerated here. The hand-written copy
369
+ // that used to sit on this line had dropped `inlineHooks`, so a hooks-only
370
+ // plugin scored 0 while `auditScore` — which does ask the predicate — scored the
371
+ // same report 100 (#199). Two lists mean two answers.
372
+ if (isEmptyMachine(r)) {
349
373
  return { score: 0, issues: ["no loadable plugin surface"] };
350
374
  }
351
375
  const deductions = reportDeductions(r);
@@ -0,0 +1,20 @@
1
+ /** The shape a Node resolve hook must return. */
2
+ export type SelfResolved = {
3
+ url: string;
4
+ format?: string | null;
5
+ shortCircuit?: boolean;
6
+ };
7
+ /**
8
+ * Resolve `vigiles` / `vigiles/<subpath>` against the CLI's OWN installation,
9
+ * or `null` when this is not our specifier / we cannot serve it.
10
+ *
11
+ * `null` rather than a throw of its own: the caller is inside a `catch` and the
12
+ * ORIGINAL resolution error is the accurate one to re-raise. An unknown subpath
13
+ * (`vigiles/nope`) would otherwise surface as `ERR_PACKAGE_PATH_NOT_EXPORTED`
14
+ * from our rescue, blaming the rescue for the author's typo.
15
+ *
16
+ * The root is read from the environment at CALL time, not at module load, so a
17
+ * test can drive both branches in one process.
18
+ */
19
+ export declare function resolveSelfSpecifier(specifier: string): SelfResolved | null;
20
+ //# sourceMappingURL=self-resolve.d.mts.map
@@ -0,0 +1,75 @@
1
+ /**
2
+ * ONE rescue for the bare `vigiles` specifier, shared by both module-resolution
3
+ * hooks this package registers.
4
+ *
5
+ * 🔴 WHY IT EXISTS. A spec does `import { instructionFile } from "vigiles/spec"`
6
+ * and a harness does `import { runHook } from "vigiles"`, so the package has to
7
+ * sit in a `node_modules` Node can reach from that file. In a repo that already
8
+ * has a `package.json`, the obvious way to put it there is `npm install` in the
9
+ * root — which installs the whole dependency tree. Measured by an adopter
10
+ * (#184): **840 packages in 2 minutes** where vigiles alone is 42 and about
11
+ * 90 MB; the other 798 were a model-eval framework and an agent SDK that the
12
+ * gate — `lint`, `compile` and `test`, all deterministic reads — never touches.
13
+ * One run sat 11 minutes in that step before being cancelled. In a repo with NO
14
+ * `package.json` at all (Python, Rust, Go) there is no install to run: the
15
+ * typed-spec path was simply unavailable, and `compile` exited 1 with
16
+ * `ERR_MODULE_NOT_FOUND`.
17
+ *
18
+ * ⚠️ `NODE_PATH` does NOT solve this, and that was measured rather than assumed:
19
+ * Node ignores it for ESM resolution, and both a spec and a harness are ESM. So
20
+ * the only ways to resolve a bare specifier from elsewhere are a real
21
+ * `node_modules` entry (a symlink) or a resolver hook. This is the hook half.
22
+ *
23
+ * 🔴 WHY IT IS SHARED. It was written once for harness scripts and the spec host
24
+ * did not get it, so `vigiles test` worked on a repo without `package.json` and
25
+ * `vigiles compile` did not — the same question answered two different ways by
26
+ * two files. Copying the branch into the second hook would have made that
27
+ * divergence permanent instead of closing it (`one-detector-no-drift`), so the
28
+ * branch lives here and both hooks call it.
29
+ *
30
+ * Scope is deliberately narrow: ONLY the `vigiles` specifier and its subpaths,
31
+ * and only when normal resolution has already FAILED. A repo that has vigiles
32
+ * installed locally keeps resolving to the local copy — the caller tries
33
+ * `nextResolve` first and only reaches this on the way out — so nothing changes
34
+ * for a repo that already worked. This only fills the hole where resolution
35
+ * would otherwise throw.
36
+ */
37
+ import { createRequire } from "node:module";
38
+ import { pathToFileURL } from "node:url";
39
+ /**
40
+ * Resolve `vigiles` / `vigiles/<subpath>` against the CLI's OWN installation,
41
+ * or `null` when this is not our specifier / we cannot serve it.
42
+ *
43
+ * `null` rather than a throw of its own: the caller is inside a `catch` and the
44
+ * ORIGINAL resolution error is the accurate one to re-raise. An unknown subpath
45
+ * (`vigiles/nope`) would otherwise surface as `ERR_PACKAGE_PATH_NOT_EXPORTED`
46
+ * from our rescue, blaming the rescue for the author's typo.
47
+ *
48
+ * The root is read from the environment at CALL time, not at module load, so a
49
+ * test can drive both branches in one process.
50
+ */
51
+ export function resolveSelfSpecifier(specifier) {
52
+ // Handed in by the parent process (the CLI), which knows where it is
53
+ // installed. Absent = nobody promised us a root; stay out of the way.
54
+ const self = process.env.VIGILES_SELF_ROOT;
55
+ if (!self)
56
+ return null;
57
+ if (specifier !== "vigiles" && !specifier.startsWith("vigiles/"))
58
+ return null;
59
+ try {
60
+ const require = createRequire(pathToFileURL(`${self}/package.json`));
61
+ // Resolve through the package's own `exports` map rather than guessing a
62
+ // file path, so a subpath like `vigiles/eval` obeys the same contract it
63
+ // would from a normal install. Node's self-reference rule is what makes
64
+ // this work when `self` IS the vigiles package rather than a copy inside
65
+ // some `node_modules`.
66
+ return {
67
+ url: pathToFileURL(require.resolve(specifier, { paths: [self] })).href,
68
+ shortCircuit: true,
69
+ };
70
+ }
71
+ catch {
72
+ return null;
73
+ }
74
+ }
75
+ //# sourceMappingURL=self-resolve.mjs.map
@@ -19,6 +19,8 @@ type Loaded = {
19
19
  };
20
20
  type NextLoad = (url: string, context: LoadContext) => Loaded | Promise<Loaded>;
21
21
  /**
22
+ * Two rescues, both attempted ONLY after normal resolution has failed.
23
+ *
22
24
  * `./x.js` → `./x.ts` when the sibling exists.
23
25
  *
24
26
  * This is the TypeScript ESM convention (`tsc` under `nodenext` requires the
@@ -27,6 +29,14 @@ type NextLoad = (url: string, context: LoadContext) => Loaded | Promise<Loaded>;
27
29
  * dogfood specs import `src/core/spec.js`, a file that does not exist on disk.
28
30
  * Attempted only AFTER normal resolution fails, so it can never shadow a real
29
31
  * `.js` file.
32
+ *
33
+ * Then `vigiles` / `vigiles/<subpath>` against the CLI's own install. Without
34
+ * it, a repo with no `node_modules/vigiles` — every Python, Rust or Go repo,
35
+ * where there is not even an install to run — could not compile a spec at all:
36
+ * `init` scaffolded `import { instructionFile } from "vigiles/spec"` and
37
+ * `compile` exited 1 with `ERR_MODULE_NOT_FOUND`. `vigiles test` had carried
38
+ * this rescue since #184; the spec host did not, which is the whole reason the
39
+ * branch now lives in one shared module rather than in each hook.
30
40
  */
31
41
  export declare function resolve(specifier: string, context: ResolveContext, nextResolve: NextResolve): Promise<Resolved>;
32
42
  /** Transpile `.ts`/`.mts` with the TypeScript this package already ships. */
@@ -18,12 +18,18 @@
18
18
  * `./x.js` → `./x.ts` specifier rewrite, and bare specifiers. NOT tsconfig
19
19
  * `paths`, JSX, or decorator configuration — specs are configuration modules,
20
20
  * not applications.
21
+ *
22
+ * The one bare specifier it does more than pass through is `vigiles` itself —
23
+ * see `./self-resolve.mjs`, shared with the harness hook.
21
24
  */
22
25
  import { existsSync, readFileSync } from "node:fs";
23
26
  import { fileURLToPath } from "node:url";
24
27
  import ts from "typescript";
28
+ import { resolveSelfSpecifier } from "./self-resolve.mjs";
25
29
  const TS_SOURCE = /\.m?ts$/;
26
30
  /**
31
+ * Two rescues, both attempted ONLY after normal resolution has failed.
32
+ *
27
33
  * `./x.js` → `./x.ts` when the sibling exists.
28
34
  *
29
35
  * This is the TypeScript ESM convention (`tsc` under `nodenext` requires the
@@ -32,6 +38,14 @@ const TS_SOURCE = /\.m?ts$/;
32
38
  * dogfood specs import `src/core/spec.js`, a file that does not exist on disk.
33
39
  * Attempted only AFTER normal resolution fails, so it can never shadow a real
34
40
  * `.js` file.
41
+ *
42
+ * Then `vigiles` / `vigiles/<subpath>` against the CLI's own install. Without
43
+ * it, a repo with no `node_modules/vigiles` — every Python, Rust or Go repo,
44
+ * where there is not even an install to run — could not compile a spec at all:
45
+ * `init` scaffolded `import { instructionFile } from "vigiles/spec"` and
46
+ * `compile` exited 1 with `ERR_MODULE_NOT_FOUND`. `vigiles test` had carried
47
+ * this rescue since #184; the spec host did not, which is the whole reason the
48
+ * branch now lives in one shared module rather than in each hook.
35
49
  */
36
50
  export async function resolve(specifier, context, nextResolve) {
37
51
  try {
@@ -44,6 +58,9 @@ export async function resolve(specifier, context, nextResolve) {
44
58
  return { url: candidate.href, format: "module", shortCircuit: true };
45
59
  }
46
60
  }
61
+ const rescued = resolveSelfSpecifier(specifier);
62
+ if (rescued)
63
+ return rescued;
47
64
  throw err;
48
65
  }
49
66
  }
package/dist/test.d.ts CHANGED
@@ -58,10 +58,15 @@ export type { HookRunResult, BlockMechanism, RunHookOptions, HookInput, HookOutp
58
58
  export * from "./harness-assert.js";
59
59
  export { experimental_emitTool, experimental_parseEmitted, experimental_assertEmittedOk, type EmitFieldSchema, type EmitObjectSchema, type EmitPropertySchema, type EmitTrackSchema, type EmitToolDefinition, type ExperimentalEmitTool, } from "./experimental-emit.js";
60
60
  export { loadHook } from "./load-hook.js";
61
+ export { experimental_hookState } from "./hook-state-store.js";
62
+ export type { HookStateHandle, SeedStateOptions } from "./hook-state-store.js";
63
+ export type { StateFact, Duration } from "./core/hook-state.js";
61
64
  export { evalChecks, assertChecks, tool, toolWith, notTool, onlyTools, skill, output, hookFired, received, turns, wrote, didNotWrite, subagent, blocked, allowed, mcp, cost, latency, tokens, inputTokens, outputTokens, cacheTokens, } from "./check.js";
62
65
  export type { ArgMatcher, Check, CheckJSON, CheckResult, JudgeFn, } from "./check.js";
63
66
  export { DISASTER_CATALOG, verifyGuardrail, unblockedDisasters, assertBlocksDisasters, formatGuardrailReport, experimental_alternateSpellings, } from "./guardrail-check.js";
64
67
  export type { DisasterEvent, DisasterCategory, GuardrailResult, VerifyGuardrailOptions, } from "./guardrail-check.js";
68
+ export { experimental_verifyPluginGuards, experimental_formatPluginGuardReport, } from "./verify-plugin-guards.js";
69
+ export type { PluginGuardReport, SweptHook, SweptHookOutcome, NonCommandHookAction, VerifyPluginGuardsOptions, } from "./verify-plugin-guards.js";
65
70
  export * from "./tool-stub.js";
66
71
  export { runHarnessTest, runHarness, parseToolCalls, parseSubagents, parseResultEvent, parseOutput, parseHooks, decideSandbox, specTrusted, sandboxAvailable, } from "./harness-test.js";
67
72
  export type { HarnessTestSpec, Trace, SubagentTrace, HarnessTestResult, RunHarnessTestOptions, ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, SandboxMode, } from "./harness-test.js";
package/dist/test.js CHANGED
@@ -66,8 +66,8 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
66
66
  for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
67
67
  };
68
68
  Object.defineProperty(exports, "__esModule", { value: true });
69
- exports.specTrusted = exports.decideSandbox = exports.parseHooks = exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.experimental_alternateSpellings = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.loadHook = exports.experimental_assertEmittedOk = exports.experimental_parseEmitted = exports.experimental_emitTool = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
70
- exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = void 0;
69
+ exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.experimental_formatPluginGuardReport = exports.experimental_verifyPluginGuards = exports.experimental_alternateSpellings = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.experimental_hookState = exports.loadHook = exports.experimental_assertEmittedOk = exports.experimental_parseEmitted = exports.experimental_emitTool = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
70
+ exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = void 0;
71
71
  // --- reporting: how much did this script actually do? ---
72
72
  // `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
73
73
  // prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
@@ -113,6 +113,15 @@ Object.defineProperty(exports, "experimental_assertEmittedOk", { enumerable: tru
113
113
  // the CLI runtime uses, so a hook that loads in a test loads identically in prod.
114
114
  var load_hook_js_1 = require("./load-hook.js");
115
115
  Object.defineProperty(exports, "loadHook", { enumerable: true, get: function () { return load_hook_js_1.loadHook; } });
116
+ // Seed + read a compiled hook's NAMED STATE from a test. A throttled hook only
117
+ // does anything interesting against an OLD fact, and "old" is not something a
118
+ // test can produce by waiting — so without this a consumer testing a throttle
119
+ // had to reconstruct the store's private path by hand (the dogfood repo did,
120
+ // and it broke when the facts were renamed). It is on THIS barrel and not on
121
+ // `vigiles/hook` on purpose: `vigiles/hook` is the only import a compiled hook
122
+ // may have, and its guarantee is that it hands out no writer.
123
+ var hook_state_store_js_1 = require("./hook-state-store.js");
124
+ Object.defineProperty(exports, "experimental_hookState", { enumerable: true, get: function () { return hook_state_store_js_1.experimental_hookState; } });
116
125
  // --- the declarative check vocabulary ---
117
126
  // Enumerated rather than `export *` ON PURPOSE: `judged` also lives in check.ts
118
127
  // and its default judge is a real model call, so it is the one member of this
@@ -153,6 +162,18 @@ Object.defineProperty(exports, "unblockedDisasters", { enumerable: true, get: fu
153
162
  Object.defineProperty(exports, "assertBlocksDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.assertBlocksDisasters; } });
154
163
  Object.defineProperty(exports, "formatGuardrailReport", { enumerable: true, get: function () { return guardrail_check_js_1.formatGuardrailReport; } });
155
164
  Object.defineProperty(exports, "experimental_alternateSpellings", { enumerable: true, get: function () { return guardrail_check_js_1.experimental_alternateSpellings; } });
165
+ // The same battery, pointed at a DIRECTORY instead of one command string. It
166
+ // reads each hook's event, matcher, command and condition off the same
167
+ // registration, so the pairing mistake `verifyGuardrail`'s own comment has to ask
168
+ // callers to avoid ("pass the hook's declared `if` here") cannot be made. It
169
+ // belongs on THIS barrel and beside the battery for the same reason the battery
170
+ // does: nothing here calls a model.
171
+ // The report + its renderer ship together: the sweep's one motivating use is
172
+ // "point the battery at YOUR hooks", and without the formatter that is a
173
+ // hand-written fold over a discriminated union at every call site.
174
+ var verify_plugin_guards_js_1 = require("./verify-plugin-guards.js");
175
+ Object.defineProperty(exports, "experimental_verifyPluginGuards", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_verifyPluginGuards; } });
176
+ Object.defineProperty(exports, "experimental_formatPluginGuardReport", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_formatPluginGuardReport; } });
156
177
  // Tool stubs on PATH (rung R2): shadow a CLI tool with a recorded canned result.
157
178
  __exportStar(require("./tool-stub.js"), exports);
158
179
  // The assembled machine — AGNOSTIC SURFACE ONLY. The Claude-Code transport