vigiles 26.1.1 → 26.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/core/bash-effects.d.ts +22 -0
- package/dist/core/bash-effects.js +10 -0
- package/dist/core/command-files.d.ts +107 -0
- package/dist/core/command-files.js +407 -0
- package/dist/core/hook-matcher.d.ts +50 -0
- package/dist/core/hook-matcher.js +77 -2
- package/dist/core/hook-normalize.d.ts +33 -0
- package/dist/core/hook-normalize.js +45 -0
- package/dist/core/shell-vars.d.ts +74 -0
- package/dist/core/shell-vars.js +270 -0
- package/dist/guardrail-check.d.ts +15 -0
- package/dist/guardrail-check.js +43 -9
- package/dist/run-script.d.ts +94 -0
- package/dist/run-script.js +47 -26
- package/dist/test.d.ts +2 -0
- package/dist/test.js +14 -2
- package/dist/verify-plugin-guards.d.ts +194 -0
- package/dist/verify-plugin-guards.js +822 -0
- package/package.json +1 -1
package/dist/run-script.d.ts
CHANGED
|
@@ -137,6 +137,62 @@ export interface RunScriptDeps {
|
|
|
137
137
|
/** Run the command confined with an allowlisted-egress netns. */
|
|
138
138
|
readonly egress: ScriptSpawner;
|
|
139
139
|
}
|
|
140
|
+
/**
|
|
141
|
+
* True when the exit code is the shell's own report that it never reached the
|
|
142
|
+
* program — so nothing of the harness ran and no surface may be credited.
|
|
143
|
+
*
|
|
144
|
+
* 🔴 THE NUMBERS ARE MEASURED, NOT CITED. 126/127 are POSIX *conventions*; what
|
|
145
|
+
* matters is what `spawnSync(cmd, { shell: true })` actually returns here.
|
|
146
|
+
* Measured 2026-08-12, `/bin/sh` → dash (Debian), each case a real file:
|
|
147
|
+
*
|
|
148
|
+
* ```
|
|
149
|
+
* 126 | direct non-executable | ./noexec.sh | out="" | Permission denied
|
|
150
|
+
* 127 | direct missing | ./missing.sh | out="" | not found
|
|
151
|
+
* 127 | bare unknown command | nosuchcmd-xyz | out="" | not found
|
|
152
|
+
* 127 | bad shebang (exists+x) | ./badshebang.sh | out="" | env: 'nosuchinterp'
|
|
153
|
+
* 126 | is a directory | ./ | out="" | Permission denied
|
|
154
|
+
* 3 | real hook, exit 3 | ./exit3.sh | out="ran" |
|
|
155
|
+
* 0 | real hook, exit 0 | ./ok.sh | out="ran" |
|
|
156
|
+
* ```
|
|
157
|
+
*
|
|
158
|
+
* The exposure this closes: a harness may legitimately assert that a hook is
|
|
159
|
+
* NOT executable (`assert.equal(runHook("./hooks/a.sh").exitCode, 126)`). That
|
|
160
|
+
* test passes, records a check — and the unconditional probe used to credit
|
|
161
|
+
* `hooks/a.sh` with an execution-tier record although only `/bin/sh` ran. The
|
|
162
|
+
* file exists and IS a discovered surface, so `resolveProbe` resolves it happily.
|
|
163
|
+
*
|
|
164
|
+
* ⚠️ TWO SHAPES ARE DELIBERATELY MISSED, both toward SILENCE (a missed probe
|
|
165
|
+
* costs one coverage line; a false grant costs the claim). Measured, same run:
|
|
166
|
+
*
|
|
167
|
+
* ```
|
|
168
|
+
* 2 | sh + missing | sh ./missing.sh | out="" | cannot open
|
|
169
|
+
* 0 | sh + non-executable | sh ./noexec.sh | out="ran" |
|
|
170
|
+
* 0 | bash + non-executable | bash ./noexec.sh | out="ran" |
|
|
171
|
+
* 127 | ran, THEN failed to launch| ./ok.sh && ./missing.sh | out="ran" |
|
|
172
|
+
* ```
|
|
173
|
+
*
|
|
174
|
+
* - `sh <missing>` exits **2** under dash, which is Claude Code's BLOCK code —
|
|
175
|
+
* indistinguishable from a gate legitimately denying, so it cannot be encoded.
|
|
176
|
+
* It is also mostly moot: a path that does not exist was never discovered as a
|
|
177
|
+
* surface, and `resolveProbe` matches only discovered surfaces.
|
|
178
|
+
* - `sh <file>` / `bash <file>` on a non-executable file exit **0 and print
|
|
179
|
+
* "ran"** — the interpreter reads the file as an argument, so the exec bit is
|
|
180
|
+
* irrelevant and the hook genuinely EXECUTED. Those must keep attributing, and
|
|
181
|
+
* do.
|
|
182
|
+
* - A compound whose last leaf fails to launch reports 126/127 for the whole
|
|
183
|
+
* line even though an earlier leaf ran. We abstain: silence, not a false grant.
|
|
184
|
+
*
|
|
185
|
+
* A hook that deliberately exits 126/127 itself is missed the same way, and the
|
|
186
|
+
* same direction.
|
|
187
|
+
*
|
|
188
|
+
* @internal Exported so the guard sweep asks the SAME question rather than
|
|
189
|
+
* re-deriving it (one-detector-no-drift). It reads these codes for the opposite
|
|
190
|
+
* purpose — not "may I credit this file with coverage?" but "may I score this
|
|
191
|
+
* guard at all?" — and the answer is the same fact: the shell never reached the
|
|
192
|
+
* program, so nothing the exit code says is the program's opinion. Not part of
|
|
193
|
+
* the public API.
|
|
194
|
+
*/
|
|
195
|
+
export declare function shellNeverLaunched(status: number | null | undefined): boolean;
|
|
140
196
|
/**
|
|
141
197
|
* The run orchestration with injectable spawn seams: pick direct vs. confined
|
|
142
198
|
* via the safe-by-default policy (`decideSandbox`), then assemble the result.
|
|
@@ -144,6 +200,44 @@ export interface RunScriptDeps {
|
|
|
144
200
|
* with fake spawners — no real bwrap.
|
|
145
201
|
*/
|
|
146
202
|
export declare function runScriptWith(command: string, stdin: string, opts: RunScriptOptions, deps: RunScriptDeps): ScriptRunResult;
|
|
203
|
+
/**
|
|
204
|
+
* Which of the three ways to start a script this run gets — or the refusal.
|
|
205
|
+
*
|
|
206
|
+
* 🔴 THE ONE PLACE THAT DECIDES, because a SECOND place that decided the same
|
|
207
|
+
* thing got it wrong. `experimental_verifyPluginGuards` must know whether a run
|
|
208
|
+
* will be confined BEFORE it runs anything: a confined run starts in a fresh
|
|
209
|
+
* empty directory, so a relative script that exists here will not exist there,
|
|
210
|
+
* and an interpreter that cannot open its script exits 2 — this harness's DENY
|
|
211
|
+
* code — which is then scored as a block. It answered that question with its own
|
|
212
|
+
* copy of the expression below, and the copy was a term short: `recordEgress`
|
|
213
|
+
* selects the netns recorder and therefore confinement, and the copy did not
|
|
214
|
+
* say so. So a `recordEgress` sweep pre-flighted against the host's cwd and env,
|
|
215
|
+
* then ran somewhere neither existed.
|
|
216
|
+
*
|
|
217
|
+
* Returning the ROUTE rather than a boolean is what makes that unrepeatable: the
|
|
218
|
+
* runner below dispatches on it and owns no policy of its own, and the pre-flight
|
|
219
|
+
* asks the same function instead of re-deriving the answer. A future option that
|
|
220
|
+
* selects confinement is added HERE, once, and every reader inherits it.
|
|
221
|
+
*
|
|
222
|
+
* A refusal counts as non-direct, and deliberately: the sweep is describing the
|
|
223
|
+
* environment a run WOULD have, and the environment it would have refused to run
|
|
224
|
+
* unconfined in is the confined one.
|
|
225
|
+
*/
|
|
226
|
+
export type ScriptRunRoute = {
|
|
227
|
+
readonly kind: "egress";
|
|
228
|
+
} | {
|
|
229
|
+
readonly kind: "sandboxed";
|
|
230
|
+
} | {
|
|
231
|
+
readonly kind: "direct";
|
|
232
|
+
} | {
|
|
233
|
+
readonly kind: "refuse";
|
|
234
|
+
readonly reason: string;
|
|
235
|
+
};
|
|
236
|
+
/** What a run with these options will do, given what the machine can offer. */
|
|
237
|
+
export declare function routeScriptRun(opts: Pick<RunScriptOptions, "egress" | "sandbox" | "trusted" | "recordEgress">, available: {
|
|
238
|
+
readonly sandbox: boolean;
|
|
239
|
+
readonly egress: boolean;
|
|
240
|
+
}): ScriptRunRoute;
|
|
147
241
|
export declare const REAL_DEPS: RunScriptDeps;
|
|
148
242
|
export declare function egressRoutes(): boolean;
|
|
149
243
|
/**
|
package/dist/run-script.js
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.REAL_DEPS = void 0;
|
|
4
|
+
exports.shellNeverLaunched = shellNeverLaunched;
|
|
4
5
|
exports.runScriptWith = runScriptWith;
|
|
6
|
+
exports.routeScriptRun = routeScriptRun;
|
|
5
7
|
exports.egressRoutes = egressRoutes;
|
|
6
8
|
exports.runScript = runScript;
|
|
7
9
|
/**
|
|
@@ -84,6 +86,13 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
|
|
|
84
86
|
*
|
|
85
87
|
* A hook that deliberately exits 126/127 itself is missed the same way, and the
|
|
86
88
|
* same direction.
|
|
89
|
+
*
|
|
90
|
+
* @internal Exported so the guard sweep asks the SAME question rather than
|
|
91
|
+
* re-deriving it (one-detector-no-drift). It reads these codes for the opposite
|
|
92
|
+
* purpose — not "may I credit this file with coverage?" but "may I score this
|
|
93
|
+
* guard at all?" — and the answer is the same fact: the shell never reached the
|
|
94
|
+
* program, so nothing the exit code says is the program's opinion. Not part of
|
|
95
|
+
* the public API.
|
|
87
96
|
*/
|
|
88
97
|
function shellNeverLaunched(status) {
|
|
89
98
|
return status === 126 || status === 127;
|
|
@@ -100,22 +109,31 @@ function runScriptWith(command, stdin, opts, deps) {
|
|
|
100
109
|
// at the primitive, so `runHook` and a bare `runScript` both count. See
|
|
101
110
|
// check-count.ts.
|
|
102
111
|
(0, check_count_js_1.recordCheck)();
|
|
103
|
-
//
|
|
104
|
-
//
|
|
105
|
-
//
|
|
106
|
-
const
|
|
107
|
-
|
|
108
|
-
:
|
|
112
|
+
// The route is DECIDED elsewhere (`routeScriptRun`) and only dispatched here,
|
|
113
|
+
// so this function holds no confinement policy a second reader could fall
|
|
114
|
+
// behind — see the type's header for the sweep that fell behind it.
|
|
115
|
+
const route = routeScriptRun(opts, {
|
|
116
|
+
sandbox: deps.available,
|
|
117
|
+
egress: deps.egressAvailable,
|
|
118
|
+
});
|
|
119
|
+
if (route.kind === "refuse")
|
|
120
|
+
throw new Error(route.reason);
|
|
121
|
+
const spawn = route.kind === "egress"
|
|
122
|
+
? deps.egress
|
|
123
|
+
: route.kind === "sandboxed"
|
|
124
|
+
? deps.sandboxed
|
|
125
|
+
: deps.direct;
|
|
126
|
+
const res = spawn(command, stdin, opts);
|
|
109
127
|
// …and WHICH surface it exercised, read off the command line that WAS
|
|
110
128
|
// executed (plus `opts.env`, because the documented idiom passes the hook path
|
|
111
129
|
// through one). Attribution by execution, not by file name — see
|
|
112
130
|
// coverage-probe.ts. Derived here at the primitive so `runHook` and a bare
|
|
113
131
|
// `runScript` both attribute without either knowing about coverage.
|
|
114
132
|
//
|
|
115
|
-
// 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM.
|
|
116
|
-
//
|
|
117
|
-
//
|
|
118
|
-
//
|
|
133
|
+
// 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM. The route
|
|
134
|
+
// above can be `refuse` and THROW before any spawner is reached: the allowlist
|
|
135
|
+
// sandbox is missing, or confinement was required and bwrap is absent. A
|
|
136
|
+
// harness that asserts exactly
|
|
119
137
|
// that refusal — `assert.throws(() => runHook(untrusted…))`, a legitimate and
|
|
120
138
|
// documented test — caught the error, exited 0 with checks recorded, and the
|
|
121
139
|
// runner wrote an execution-tier coverage record for a hook that never ran.
|
|
@@ -140,18 +158,21 @@ function runScriptWith(command, stdin, opts, deps) {
|
|
|
140
158
|
filesWritten: res.filesWritten,
|
|
141
159
|
};
|
|
142
160
|
}
|
|
143
|
-
/**
|
|
144
|
-
function
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
161
|
+
/** What a run with these options will do, given what the machine can offer. */
|
|
162
|
+
function routeScriptRun(opts, available) {
|
|
163
|
+
// Allowlisted egress is its own confined path (bwrap netns + slirp4netns +
|
|
164
|
+
// nft); it can't run unconfined, so it refuses outright when the tooling is
|
|
165
|
+
// absent rather than falling back to a direct run that ignores the allowlist.
|
|
166
|
+
if (opts.egress)
|
|
167
|
+
return available.egress
|
|
168
|
+
? { kind: "egress" }
|
|
169
|
+
: {
|
|
170
|
+
kind: "refuse",
|
|
171
|
+
reason: "refusing to run egress: { allow } without the allowlist sandbox: it " +
|
|
172
|
+
"needs Linux + bubblewrap (bwrap) + slirp4netns + nft — install them to " +
|
|
173
|
+
"run with a packet-layer egress allowlist, or use recordEgress to record " +
|
|
174
|
+
"and block instead",
|
|
175
|
+
};
|
|
155
176
|
// Confinement follows provenance: a trusted hook (the default) runs directly;
|
|
156
177
|
// marking a hook untrusted defaults it to "auto" (confine-or-refuse), so
|
|
157
178
|
// foreign code is never run unconfined by accident. An explicit `sandbox`
|
|
@@ -164,13 +185,13 @@ function runConfinedOrDirect(command, stdin, opts, deps) {
|
|
|
164
185
|
const decision = (0, sandbox_js_1.decideSandbox)({
|
|
165
186
|
trusted: false,
|
|
166
187
|
mode,
|
|
167
|
-
available:
|
|
188
|
+
available: available.sandbox,
|
|
168
189
|
});
|
|
169
190
|
if (decision.action === "throw")
|
|
170
|
-
|
|
191
|
+
return { kind: "refuse", reason: decision.reason };
|
|
171
192
|
return decision.action === "sandbox"
|
|
172
|
-
?
|
|
173
|
-
:
|
|
193
|
+
? { kind: "sandboxed" }
|
|
194
|
+
: { kind: "direct" };
|
|
174
195
|
}
|
|
175
196
|
/** Run the hook command directly through a shell (the default, unconfined). */
|
|
176
197
|
function directSpawn(command, stdin, opts) {
|
package/dist/test.d.ts
CHANGED
|
@@ -65,6 +65,8 @@ export { evalChecks, assertChecks, tool, toolWith, notTool, onlyTools, skill, ou
|
|
|
65
65
|
export type { ArgMatcher, Check, CheckJSON, CheckResult, JudgeFn, } from "./check.js";
|
|
66
66
|
export { DISASTER_CATALOG, verifyGuardrail, unblockedDisasters, assertBlocksDisasters, formatGuardrailReport, experimental_alternateSpellings, } from "./guardrail-check.js";
|
|
67
67
|
export type { DisasterEvent, DisasterCategory, GuardrailResult, VerifyGuardrailOptions, } from "./guardrail-check.js";
|
|
68
|
+
export { experimental_verifyPluginGuards, experimental_formatPluginGuardReport, } from "./verify-plugin-guards.js";
|
|
69
|
+
export type { PluginGuardReport, SweptHook, SweptHookOutcome, NonCommandHookAction, VerifyPluginGuardsOptions, } from "./verify-plugin-guards.js";
|
|
68
70
|
export * from "./tool-stub.js";
|
|
69
71
|
export { runHarnessTest, runHarness, parseToolCalls, parseSubagents, parseResultEvent, parseOutput, parseHooks, decideSandbox, specTrusted, sandboxAvailable, } from "./harness-test.js";
|
|
70
72
|
export type { HarnessTestSpec, Trace, SubagentTrace, HarnessTestResult, RunHarnessTestOptions, ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, SandboxMode, } from "./harness-test.js";
|
package/dist/test.js
CHANGED
|
@@ -66,8 +66,8 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
|
66
66
|
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
67
67
|
};
|
|
68
68
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
69
|
-
exports.
|
|
70
|
-
exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = void 0;
|
|
69
|
+
exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.experimental_formatPluginGuardReport = exports.experimental_verifyPluginGuards = exports.experimental_alternateSpellings = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.experimental_hookState = exports.loadHook = exports.experimental_assertEmittedOk = exports.experimental_parseEmitted = exports.experimental_emitTool = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
|
|
70
|
+
exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = void 0;
|
|
71
71
|
// --- reporting: how much did this script actually do? ---
|
|
72
72
|
// `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
|
|
73
73
|
// prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
|
|
@@ -162,6 +162,18 @@ Object.defineProperty(exports, "unblockedDisasters", { enumerable: true, get: fu
|
|
|
162
162
|
Object.defineProperty(exports, "assertBlocksDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.assertBlocksDisasters; } });
|
|
163
163
|
Object.defineProperty(exports, "formatGuardrailReport", { enumerable: true, get: function () { return guardrail_check_js_1.formatGuardrailReport; } });
|
|
164
164
|
Object.defineProperty(exports, "experimental_alternateSpellings", { enumerable: true, get: function () { return guardrail_check_js_1.experimental_alternateSpellings; } });
|
|
165
|
+
// The same battery, pointed at a DIRECTORY instead of one command string. It
|
|
166
|
+
// reads each hook's event, matcher, command and condition off the same
|
|
167
|
+
// registration, so the pairing mistake `verifyGuardrail`'s own comment has to ask
|
|
168
|
+
// callers to avoid ("pass the hook's declared `if` here") cannot be made. It
|
|
169
|
+
// belongs on THIS barrel and beside the battery for the same reason the battery
|
|
170
|
+
// does: nothing here calls a model.
|
|
171
|
+
// The report + its renderer ship together: the sweep's one motivating use is
|
|
172
|
+
// "point the battery at YOUR hooks", and without the formatter that is a
|
|
173
|
+
// hand-written fold over a discriminated union at every call site.
|
|
174
|
+
var verify_plugin_guards_js_1 = require("./verify-plugin-guards.js");
|
|
175
|
+
Object.defineProperty(exports, "experimental_verifyPluginGuards", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_verifyPluginGuards; } });
|
|
176
|
+
Object.defineProperty(exports, "experimental_formatPluginGuardReport", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_formatPluginGuardReport; } });
|
|
165
177
|
// Tool stubs on PATH (rung R2): shadow a CLI tool with a recorded canned result.
|
|
166
178
|
__exportStar(require("./tool-stub.js"), exports);
|
|
167
179
|
// The assembled machine — AGNOSTIC SURFACE ONLY. The Claude-Code transport
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
import { type NonCommandHookAction } from "./core/hook-normalize.js";
|
|
2
|
+
export type { NonCommandHookAction } from "./core/hook-normalize.js";
|
|
3
|
+
import type { HarnessAdapter } from "./core/adapter.js";
|
|
4
|
+
import { type DisasterCategory, type DisasterEvent, type GuardrailResult } from "./guardrail-check.js";
|
|
5
|
+
import type { RunHookOptions } from "./run-hook.js";
|
|
6
|
+
/** The hook a sweep looked at, as its config declares it. */
|
|
7
|
+
export interface SweptHook {
|
|
8
|
+
/** The event it registers under, e.g. `"PreToolUse"`. */
|
|
9
|
+
readonly event: string;
|
|
10
|
+
/** Its tool matcher, or `null` when it declares none (matches everything). */
|
|
11
|
+
readonly matcher: string | null;
|
|
12
|
+
/** Its condition as written (Claude Code's `if`), or `null` when unconditional. */
|
|
13
|
+
readonly condition: string | null;
|
|
14
|
+
/** The command, with the harness's plugin-root token already expanded. */
|
|
15
|
+
readonly command: string;
|
|
16
|
+
/**
|
|
17
|
+
* Its position in the flattened registration list, so two hooks sharing a
|
|
18
|
+
* command are still distinguishable in a report. Stable for one sweep of one
|
|
19
|
+
* directory; not an identity across versions.
|
|
20
|
+
*/
|
|
21
|
+
readonly index: number;
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* What happened to one hook. A DISCRIMINATED UNION, not a result list plus three
|
|
25
|
+
* nullable fields: `results` exists on exactly the outcome that has them, so
|
|
26
|
+
* "read the score of a hook that was never run" is a type error rather than a
|
|
27
|
+
* `0/7` someone quotes.
|
|
28
|
+
*/
|
|
29
|
+
export type SweptHookOutcome = {
|
|
30
|
+
/** The battery reached this hook; `results` holds one entry per event. */
|
|
31
|
+
readonly status: "measured";
|
|
32
|
+
readonly hook: SweptHook;
|
|
33
|
+
/** One result per battery event, in catalog order. */
|
|
34
|
+
readonly results: readonly GuardrailResult[];
|
|
35
|
+
/** Ids of the events it denied. */
|
|
36
|
+
readonly blocked: readonly string[];
|
|
37
|
+
/** Ids it ran on and let through. */
|
|
38
|
+
readonly allowed: readonly string[];
|
|
39
|
+
/** Ids the harness would never have handed it (condition did not match). */
|
|
40
|
+
readonly notRun: readonly string[];
|
|
41
|
+
} | {
|
|
42
|
+
/**
|
|
43
|
+
* The battery does not apply to this hook — a different event, or a matcher
|
|
44
|
+
* that selects none of the battery's tools. NOT a score of zero.
|
|
45
|
+
*/
|
|
46
|
+
readonly status: "not-applicable";
|
|
47
|
+
readonly hook: SweptHook;
|
|
48
|
+
/** One line naming which of the two it is, and against what. */
|
|
49
|
+
readonly reason: string;
|
|
50
|
+
} | {
|
|
51
|
+
/**
|
|
52
|
+
* The command names a variable nothing has set, so the program we would run
|
|
53
|
+
* is not the program the harness runs. Refusing is the honest answer; the
|
|
54
|
+
* fix is in the caller's hands (pass `env`).
|
|
55
|
+
*/
|
|
56
|
+
readonly status: "unresolved";
|
|
57
|
+
readonly hook: SweptHook;
|
|
58
|
+
/** One line naming the unset variables. */
|
|
59
|
+
readonly reason: string;
|
|
60
|
+
};
|
|
61
|
+
/** The whole sweep. */
|
|
62
|
+
export interface PluginGuardReport {
|
|
63
|
+
/** The directory swept, resolved to an absolute path. */
|
|
64
|
+
readonly dir: string;
|
|
65
|
+
/** The adapter that read it, e.g. `"claude-code"`. */
|
|
66
|
+
readonly harness: string;
|
|
67
|
+
/** The event each disaster was delivered as (default `"PreToolUse"`). */
|
|
68
|
+
readonly event: string;
|
|
69
|
+
/** The battery that was used, so a report says what it measured against. */
|
|
70
|
+
readonly events: readonly DisasterEvent[];
|
|
71
|
+
/** One outcome per declared COMMAND hook, in config order. */
|
|
72
|
+
readonly hooks: readonly SweptHookOutcome[];
|
|
73
|
+
/**
|
|
74
|
+
* Declared actions this tier cannot drive because they are not commands —
|
|
75
|
+
* `prompt`, `http`, `mcp_tool`, `agent`.
|
|
76
|
+
*
|
|
77
|
+
* 🔴 THEY USED TO BE DROPPED, AND DROPPING THEM MANUFACTURED THE FALSE EMPTY
|
|
78
|
+
* `notes` exists to prevent. A repository whose hooks are all `prompt` actions
|
|
79
|
+
* has declared guards; it was reported as declaring none, in the words of the
|
|
80
|
+
* one sentence this report writes to be sure nobody reads an empty result as a
|
|
81
|
+
* clean bill of health. Not measured is a limit of the tier and says so; not
|
|
82
|
+
* declared is an accusation about the repository, and it was not true.
|
|
83
|
+
*
|
|
84
|
+
* They carry no score and never will here — a shell battery cannot drive a
|
|
85
|
+
* prompt — so they are a separate list rather than a fourth outcome status
|
|
86
|
+
* with an invented command.
|
|
87
|
+
*/
|
|
88
|
+
readonly unmeasurable: readonly NonCommandHookAction[];
|
|
89
|
+
/**
|
|
90
|
+
* Why the sweep measured less than a reader might assume — no COMMAND hooks
|
|
91
|
+
* declared, a harness with no shell hooks, every hook on another event, or an
|
|
92
|
+
* action this tier cannot drive. Empty only when at least one hook was
|
|
93
|
+
* measured AND nothing was left undrivable: an undrivable action is a gap in
|
|
94
|
+
* COVERAGE, so it is said even when other hooks scored.
|
|
95
|
+
*
|
|
96
|
+
* 🔴 THIS IS THE EMPTY CASE'S VOICE. `hooks: []` on its own reads as a clean
|
|
97
|
+
* bill of health, which is the exact false confidence the battery exists to
|
|
98
|
+
* remove — so a sweep that measured nothing always says so in words.
|
|
99
|
+
*/
|
|
100
|
+
readonly notes: readonly string[];
|
|
101
|
+
}
|
|
102
|
+
/** Options for {@link experimental_verifyPluginGuards}. */
|
|
103
|
+
export interface VerifyPluginGuardsOptions extends Omit<RunHookOptions, "condition" | "protocol"> {
|
|
104
|
+
/**
|
|
105
|
+
* The harness to read the repo as. Defaults to Claude Code, so an existing
|
|
106
|
+
* Claude Code repo needs nothing. The condition grammar and the block protocol
|
|
107
|
+
* both come from this adapter's `hookProtocol`.
|
|
108
|
+
*/
|
|
109
|
+
readonly adapter?: HarnessAdapter;
|
|
110
|
+
/** Restrict the battery to these categories (default: the whole catalog). */
|
|
111
|
+
readonly categories?: readonly DisasterCategory[];
|
|
112
|
+
/** Override the battery entirely. */
|
|
113
|
+
readonly events?: readonly DisasterEvent[];
|
|
114
|
+
/** The event each disaster is delivered as (default `"PreToolUse"`). */
|
|
115
|
+
readonly event?: string;
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Run the disaster battery against every hook a plugin or repo declares, using
|
|
119
|
+
* each hook's OWN event, matcher and condition, and report per hook.
|
|
120
|
+
*
|
|
121
|
+
* ```ts
|
|
122
|
+
* import { experimental_verifyPluginGuards } from "vigiles";
|
|
123
|
+
*
|
|
124
|
+
* const report = experimental_verifyPluginGuards(".");
|
|
125
|
+
* for (const h of report.hooks) {
|
|
126
|
+
* if (h.status === "measured")
|
|
127
|
+
* console.log(`${h.blocked.length}/${h.results.length} ${h.hook.command}`);
|
|
128
|
+
* else console.log(`⊘ ${h.status} ${h.hook.command} — ${h.reason}`);
|
|
129
|
+
* }
|
|
130
|
+
* for (const note of report.notes) console.log(note);
|
|
131
|
+
* ```
|
|
132
|
+
*
|
|
133
|
+
* Nothing here needs a model or a key. The hooks it finds are the ones the
|
|
134
|
+
* harness would load, so a hook that is present on disk but not registered is
|
|
135
|
+
* absent from the report by construction — which is the correct answer, and the
|
|
136
|
+
* one you would not get by globbing `hooks/*.sh`.
|
|
137
|
+
*
|
|
138
|
+
* ⚠️ It RUNS each reachable hook. A hook is a program you did not necessarily
|
|
139
|
+
* write, so point this at a repo whose hooks you are willing to execute, or pass
|
|
140
|
+
* `trusted: false` / `sandbox: "auto"` (inherited from {@link RunHookOptions}) to
|
|
141
|
+
* confine them. `verifyGuardrail` has always had the same property; sweeping a
|
|
142
|
+
* whole plugin makes it worth saying out loud.
|
|
143
|
+
*
|
|
144
|
+
* @experimental Days old, with no consumer outside this repository. The REPORT
|
|
145
|
+
* SHAPE is the part most likely to move — specifically whether `not-applicable`
|
|
146
|
+
* stays one status or splits by cause, and whether the per-hook counts stay id
|
|
147
|
+
* arrays. The prefix comes off when that shape survives sweeping several real
|
|
148
|
+
* third-party repos unchanged; see docs/experimental.md.
|
|
149
|
+
*
|
|
150
|
+
* @param dir - the plugin or repo root to read hooks from.
|
|
151
|
+
*/
|
|
152
|
+
export declare function experimental_verifyPluginGuards(dir: string, opts?: VerifyPluginGuardsOptions): PluginGuardReport;
|
|
153
|
+
/**
|
|
154
|
+
* Render a {@link PluginGuardReport} as terminal text.
|
|
155
|
+
*
|
|
156
|
+
* ```ts
|
|
157
|
+
* import {
|
|
158
|
+
* experimental_verifyPluginGuards,
|
|
159
|
+
* experimental_formatPluginGuardReport,
|
|
160
|
+
* } from "vigiles";
|
|
161
|
+
*
|
|
162
|
+
* console.log(
|
|
163
|
+
* experimental_formatPluginGuardReport(experimental_verifyPluginGuards(".")),
|
|
164
|
+
* );
|
|
165
|
+
* ```
|
|
166
|
+
*
|
|
167
|
+
* NEUTRAL, the same way {@link formatGuardrailReport} is: it reports what each
|
|
168
|
+
* hook blocks without deciding whether that was the hook's job. A repo's config
|
|
169
|
+
* never says which of its hooks is meant to be a bash-safety guard, so a verdict
|
|
170
|
+
* here would be invented rather than read.
|
|
171
|
+
*
|
|
172
|
+
* 🔴 A HOOK THE BATTERY NEVER REACHED IS NEVER GIVEN A NUMBER. A `measured` hook
|
|
173
|
+
* prints `blocks n/7`; a `not-applicable` or `unresolved` one prints its REASON
|
|
174
|
+
* under a `⊘` heading and no count at all, because a rendered `0/7` is the same
|
|
175
|
+
* false confidence the discriminated union exists to prevent, reintroduced one
|
|
176
|
+
* layer up where the type system can no longer see it. For the same reason the
|
|
177
|
+
* report's `notes` are printed FIRST and in full: a sweep that measured nothing
|
|
178
|
+
* has to say so in words, since an output with no rows reads as a clean bill of
|
|
179
|
+
* health.
|
|
180
|
+
*
|
|
181
|
+
* MANY HOOKS STAY READABLE by grouping the unmeasured half BY REASON — a repo
|
|
182
|
+
* with thirty hooks usually has two or three distinct reasons — and naming at
|
|
183
|
+
* most {@link HOOKS_PER_REASON} hooks per reason before counting the rest. The
|
|
184
|
+
* measured half is never collapsed: those are the hooks you came for.
|
|
185
|
+
*
|
|
186
|
+
* @experimental It renders {@link PluginGuardReport}, whose SHAPE is the part
|
|
187
|
+
* most likely to move (see {@link experimental_verifyPluginGuards}), so a stable
|
|
188
|
+
* name here would promise a stability its only input does not have. The prefix
|
|
189
|
+
* comes off with the same change that takes it off the report.
|
|
190
|
+
*
|
|
191
|
+
* @param report - a sweep from {@link experimental_verifyPluginGuards}.
|
|
192
|
+
*/
|
|
193
|
+
export declare function experimental_formatPluginGuardReport(report: PluginGuardReport): string;
|
|
194
|
+
//# sourceMappingURL=verify-plugin-guards.d.ts.map
|