vigiles 26.0.1 → 26.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -4
- package/dist/adapters/claude-code/hook-condition.d.ts +46 -0
- package/dist/adapters/claude-code/hook-condition.js +142 -0
- package/dist/adapters/claude-code/hook-protocol.js +5 -0
- package/dist/audit-report.template.html +2 -2
- package/dist/cli.js +67 -115
- package/dist/core/bash-effects.d.ts +22 -0
- package/dist/core/bash-effects.js +10 -0
- package/dist/core/command-files.d.ts +107 -0
- package/dist/core/command-files.js +407 -0
- package/dist/core/hook-condition.d.ts +96 -0
- package/dist/core/hook-condition.js +63 -0
- package/dist/core/hook-matcher.d.ts +50 -0
- package/dist/core/hook-matcher.js +77 -2
- package/dist/core/hook-normalize.d.ts +51 -0
- package/dist/core/hook-normalize.js +61 -1
- package/dist/core/hook-program.d.ts +62 -1
- package/dist/core/hook-program.js +15 -1
- package/dist/core/hook-protocol.d.ts +16 -0
- package/dist/core/linters.js +97 -58
- package/dist/core/shell-vars.d.ts +74 -0
- package/dist/core/shell-vars.js +270 -0
- package/dist/core/skill-resources.d.ts +22 -1
- package/dist/core/skill-resources.js +2 -1
- package/dist/doc-test-script-coverage.d.ts +52 -0
- package/dist/doc-test-script-coverage.js +66 -0
- package/dist/guardrail-check.d.ts +29 -0
- package/dist/guardrail-check.js +69 -10
- package/dist/harness-assert.d.ts +8 -5
- package/dist/harness-assert.js +8 -5
- package/dist/harness-resolve-hooks.mjs +14 -37
- package/dist/hook-state-store.d.ts +143 -0
- package/dist/hook-state-store.js +241 -0
- package/dist/hook.d.ts +3 -1
- package/dist/hook.js +3 -1
- package/dist/run-hook.d.ts +33 -1
- package/dist/run-hook.js +46 -2
- package/dist/run-script.d.ts +94 -0
- package/dist/run-script.js +47 -26
- package/dist/scan-core.js +21 -3
- package/dist/score-core.d.ts +21 -1
- package/dist/score-core.js +30 -6
- package/dist/self-resolve.d.mts +20 -0
- package/dist/self-resolve.mjs +75 -0
- package/dist/spec-hooks.d.mts +10 -0
- package/dist/spec-hooks.mjs +17 -0
- package/dist/test.d.ts +5 -0
- package/dist/test.js +23 -2
- package/dist/verify-plugin-guards.d.ts +194 -0
- package/dist/verify-plugin-guards.js +822 -0
- package/package.json +1 -1
package/dist/run-script.d.ts
CHANGED
|
@@ -137,6 +137,62 @@ export interface RunScriptDeps {
|
|
|
137
137
|
/** Run the command confined with an allowlisted-egress netns. */
|
|
138
138
|
readonly egress: ScriptSpawner;
|
|
139
139
|
}
|
|
140
|
+
/**
|
|
141
|
+
* True when the exit code is the shell's own report that it never reached the
|
|
142
|
+
* program — so nothing of the harness ran and no surface may be credited.
|
|
143
|
+
*
|
|
144
|
+
* 🔴 THE NUMBERS ARE MEASURED, NOT CITED. 126/127 are POSIX *conventions*; what
|
|
145
|
+
* matters is what `spawnSync(cmd, { shell: true })` actually returns here.
|
|
146
|
+
* Measured 2026-08-12, `/bin/sh` → dash (Debian), each case a real file:
|
|
147
|
+
*
|
|
148
|
+
* ```
|
|
149
|
+
* 126 | direct non-executable | ./noexec.sh | out="" | Permission denied
|
|
150
|
+
* 127 | direct missing | ./missing.sh | out="" | not found
|
|
151
|
+
* 127 | bare unknown command | nosuchcmd-xyz | out="" | not found
|
|
152
|
+
* 127 | bad shebang (exists+x) | ./badshebang.sh | out="" | env: 'nosuchinterp'
|
|
153
|
+
* 126 | is a directory | ./ | out="" | Permission denied
|
|
154
|
+
* 3 | real hook, exit 3 | ./exit3.sh | out="ran" |
|
|
155
|
+
* 0 | real hook, exit 0 | ./ok.sh | out="ran" |
|
|
156
|
+
* ```
|
|
157
|
+
*
|
|
158
|
+
* The exposure this closes: a harness may legitimately assert that a hook is
|
|
159
|
+
* NOT executable (`assert.equal(runHook("./hooks/a.sh").exitCode, 126)`). That
|
|
160
|
+
* test passes, records a check — and the unconditional probe used to credit
|
|
161
|
+
* `hooks/a.sh` with an execution-tier record although only `/bin/sh` ran. The
|
|
162
|
+
* file exists and IS a discovered surface, so `resolveProbe` resolves it happily.
|
|
163
|
+
*
|
|
164
|
+
* ⚠️ TWO SHAPES ARE DELIBERATELY MISSED, both toward SILENCE (a missed probe
|
|
165
|
+
* costs one coverage line; a false grant costs the claim). Measured, same run:
|
|
166
|
+
*
|
|
167
|
+
* ```
|
|
168
|
+
* 2 | sh + missing | sh ./missing.sh | out="" | cannot open
|
|
169
|
+
* 0 | sh + non-executable | sh ./noexec.sh | out="ran" |
|
|
170
|
+
* 0 | bash + non-executable | bash ./noexec.sh | out="ran" |
|
|
171
|
+
* 127 | ran, THEN failed to launch| ./ok.sh && ./missing.sh | out="ran" |
|
|
172
|
+
* ```
|
|
173
|
+
*
|
|
174
|
+
* - `sh <missing>` exits **2** under dash, which is Claude Code's BLOCK code —
|
|
175
|
+
* indistinguishable from a gate legitimately denying, so it cannot be encoded.
|
|
176
|
+
* It is also mostly moot: a path that does not exist was never discovered as a
|
|
177
|
+
* surface, and `resolveProbe` matches only discovered surfaces.
|
|
178
|
+
* - `sh <file>` / `bash <file>` on a non-executable file exit **0 and print
|
|
179
|
+
* "ran"** — the interpreter reads the file as an argument, so the exec bit is
|
|
180
|
+
* irrelevant and the hook genuinely EXECUTED. Those must keep attributing, and
|
|
181
|
+
* do.
|
|
182
|
+
* - A compound whose last leaf fails to launch reports 126/127 for the whole
|
|
183
|
+
* line even though an earlier leaf ran. We abstain: silence, not a false grant.
|
|
184
|
+
*
|
|
185
|
+
* A hook that deliberately exits 126/127 itself is missed the same way, and the
|
|
186
|
+
* same direction.
|
|
187
|
+
*
|
|
188
|
+
* @internal Exported so the guard sweep asks the SAME question rather than
|
|
189
|
+
* re-deriving it (one-detector-no-drift). It reads these codes for the opposite
|
|
190
|
+
* purpose — not "may I credit this file with coverage?" but "may I score this
|
|
191
|
+
* guard at all?" — and the answer is the same fact: the shell never reached the
|
|
192
|
+
* program, so nothing the exit code says is the program's opinion. Not part of
|
|
193
|
+
* the public API.
|
|
194
|
+
*/
|
|
195
|
+
export declare function shellNeverLaunched(status: number | null | undefined): boolean;
|
|
140
196
|
/**
|
|
141
197
|
* The run orchestration with injectable spawn seams: pick direct vs. confined
|
|
142
198
|
* via the safe-by-default policy (`decideSandbox`), then assemble the result.
|
|
@@ -144,6 +200,44 @@ export interface RunScriptDeps {
|
|
|
144
200
|
* with fake spawners — no real bwrap.
|
|
145
201
|
*/
|
|
146
202
|
export declare function runScriptWith(command: string, stdin: string, opts: RunScriptOptions, deps: RunScriptDeps): ScriptRunResult;
|
|
203
|
+
/**
|
|
204
|
+
* Which of the three ways to start a script this run gets — or the refusal.
|
|
205
|
+
*
|
|
206
|
+
* 🔴 THE ONE PLACE THAT DECIDES, because a SECOND place that decided the same
|
|
207
|
+
* thing got it wrong. `experimental_verifyPluginGuards` must know whether a run
|
|
208
|
+
* will be confined BEFORE it runs anything: a confined run starts in a fresh
|
|
209
|
+
* empty directory, so a relative script that exists here will not exist there,
|
|
210
|
+
* and an interpreter that cannot open its script exits 2 — this harness's DENY
|
|
211
|
+
* code — which is then scored as a block. It answered that question with its own
|
|
212
|
+
* copy of the expression below, and the copy was a term short: `recordEgress`
|
|
213
|
+
* selects the netns recorder and therefore confinement, and the copy did not
|
|
214
|
+
* say so. So a `recordEgress` sweep pre-flighted against the host's cwd and env,
|
|
215
|
+
* then ran somewhere neither existed.
|
|
216
|
+
*
|
|
217
|
+
* Returning the ROUTE rather than a boolean is what makes that unrepeatable: the
|
|
218
|
+
* runner below dispatches on it and owns no policy of its own, and the pre-flight
|
|
219
|
+
* asks the same function instead of re-deriving the answer. A future option that
|
|
220
|
+
* selects confinement is added HERE, once, and every reader inherits it.
|
|
221
|
+
*
|
|
222
|
+
* A refusal counts as non-direct, and deliberately: the sweep is describing the
|
|
223
|
+
* environment a run WOULD have, and the environment it would have refused to run
|
|
224
|
+
* unconfined in is the confined one.
|
|
225
|
+
*/
|
|
226
|
+
export type ScriptRunRoute = {
|
|
227
|
+
readonly kind: "egress";
|
|
228
|
+
} | {
|
|
229
|
+
readonly kind: "sandboxed";
|
|
230
|
+
} | {
|
|
231
|
+
readonly kind: "direct";
|
|
232
|
+
} | {
|
|
233
|
+
readonly kind: "refuse";
|
|
234
|
+
readonly reason: string;
|
|
235
|
+
};
|
|
236
|
+
/** What a run with these options will do, given what the machine can offer. */
|
|
237
|
+
export declare function routeScriptRun(opts: Pick<RunScriptOptions, "egress" | "sandbox" | "trusted" | "recordEgress">, available: {
|
|
238
|
+
readonly sandbox: boolean;
|
|
239
|
+
readonly egress: boolean;
|
|
240
|
+
}): ScriptRunRoute;
|
|
147
241
|
export declare const REAL_DEPS: RunScriptDeps;
|
|
148
242
|
export declare function egressRoutes(): boolean;
|
|
149
243
|
/**
|
package/dist/run-script.js
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.REAL_DEPS = void 0;
|
|
4
|
+
exports.shellNeverLaunched = shellNeverLaunched;
|
|
4
5
|
exports.runScriptWith = runScriptWith;
|
|
6
|
+
exports.routeScriptRun = routeScriptRun;
|
|
5
7
|
exports.egressRoutes = egressRoutes;
|
|
6
8
|
exports.runScript = runScript;
|
|
7
9
|
/**
|
|
@@ -84,6 +86,13 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
|
|
|
84
86
|
*
|
|
85
87
|
* A hook that deliberately exits 126/127 itself is missed the same way, and the
|
|
86
88
|
* same direction.
|
|
89
|
+
*
|
|
90
|
+
* @internal Exported so the guard sweep asks the SAME question rather than
|
|
91
|
+
* re-deriving it (one-detector-no-drift). It reads these codes for the opposite
|
|
92
|
+
* purpose — not "may I credit this file with coverage?" but "may I score this
|
|
93
|
+
* guard at all?" — and the answer is the same fact: the shell never reached the
|
|
94
|
+
* program, so nothing the exit code says is the program's opinion. Not part of
|
|
95
|
+
* the public API.
|
|
87
96
|
*/
|
|
88
97
|
function shellNeverLaunched(status) {
|
|
89
98
|
return status === 126 || status === 127;
|
|
@@ -100,22 +109,31 @@ function runScriptWith(command, stdin, opts, deps) {
|
|
|
100
109
|
// at the primitive, so `runHook` and a bare `runScript` both count. See
|
|
101
110
|
// check-count.ts.
|
|
102
111
|
(0, check_count_js_1.recordCheck)();
|
|
103
|
-
//
|
|
104
|
-
//
|
|
105
|
-
//
|
|
106
|
-
const
|
|
107
|
-
|
|
108
|
-
:
|
|
112
|
+
// The route is DECIDED elsewhere (`routeScriptRun`) and only dispatched here,
|
|
113
|
+
// so this function holds no confinement policy a second reader could fall
|
|
114
|
+
// behind — see the type's header for the sweep that fell behind it.
|
|
115
|
+
const route = routeScriptRun(opts, {
|
|
116
|
+
sandbox: deps.available,
|
|
117
|
+
egress: deps.egressAvailable,
|
|
118
|
+
});
|
|
119
|
+
if (route.kind === "refuse")
|
|
120
|
+
throw new Error(route.reason);
|
|
121
|
+
const spawn = route.kind === "egress"
|
|
122
|
+
? deps.egress
|
|
123
|
+
: route.kind === "sandboxed"
|
|
124
|
+
? deps.sandboxed
|
|
125
|
+
: deps.direct;
|
|
126
|
+
const res = spawn(command, stdin, opts);
|
|
109
127
|
// …and WHICH surface it exercised, read off the command line that WAS
|
|
110
128
|
// executed (plus `opts.env`, because the documented idiom passes the hook path
|
|
111
129
|
// through one). Attribution by execution, not by file name — see
|
|
112
130
|
// coverage-probe.ts. Derived here at the primitive so `runHook` and a bare
|
|
113
131
|
// `runScript` both attribute without either knowing about coverage.
|
|
114
132
|
//
|
|
115
|
-
// 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM.
|
|
116
|
-
//
|
|
117
|
-
//
|
|
118
|
-
//
|
|
133
|
+
// 🔴 AFTER THE SPAWN, NOT BEFORE, AND THE ORDER IS THE WHOLE CLAIM. The route
|
|
134
|
+
// above can be `refuse` and THROW before any spawner is reached: the allowlist
|
|
135
|
+
// sandbox is missing, or confinement was required and bwrap is absent. A
|
|
136
|
+
// harness that asserts exactly
|
|
119
137
|
// that refusal — `assert.throws(() => runHook(untrusted…))`, a legitimate and
|
|
120
138
|
// documented test — caught the error, exited 0 with checks recorded, and the
|
|
121
139
|
// runner wrote an execution-tier coverage record for a hook that never ran.
|
|
@@ -140,18 +158,21 @@ function runScriptWith(command, stdin, opts, deps) {
|
|
|
140
158
|
filesWritten: res.filesWritten,
|
|
141
159
|
};
|
|
142
160
|
}
|
|
143
|
-
/**
|
|
144
|
-
function
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
161
|
+
/** What a run with these options will do, given what the machine can offer. */
|
|
162
|
+
function routeScriptRun(opts, available) {
|
|
163
|
+
// Allowlisted egress is its own confined path (bwrap netns + slirp4netns +
|
|
164
|
+
// nft); it can't run unconfined, so it refuses outright when the tooling is
|
|
165
|
+
// absent rather than falling back to a direct run that ignores the allowlist.
|
|
166
|
+
if (opts.egress)
|
|
167
|
+
return available.egress
|
|
168
|
+
? { kind: "egress" }
|
|
169
|
+
: {
|
|
170
|
+
kind: "refuse",
|
|
171
|
+
reason: "refusing to run egress: { allow } without the allowlist sandbox: it " +
|
|
172
|
+
"needs Linux + bubblewrap (bwrap) + slirp4netns + nft — install them to " +
|
|
173
|
+
"run with a packet-layer egress allowlist, or use recordEgress to record " +
|
|
174
|
+
"and block instead",
|
|
175
|
+
};
|
|
155
176
|
// Confinement follows provenance: a trusted hook (the default) runs directly;
|
|
156
177
|
// marking a hook untrusted defaults it to "auto" (confine-or-refuse), so
|
|
157
178
|
// foreign code is never run unconfined by accident. An explicit `sandbox`
|
|
@@ -164,13 +185,13 @@ function runConfinedOrDirect(command, stdin, opts, deps) {
|
|
|
164
185
|
const decision = (0, sandbox_js_1.decideSandbox)({
|
|
165
186
|
trusted: false,
|
|
166
187
|
mode,
|
|
167
|
-
available:
|
|
188
|
+
available: available.sandbox,
|
|
168
189
|
});
|
|
169
190
|
if (decision.action === "throw")
|
|
170
|
-
|
|
191
|
+
return { kind: "refuse", reason: decision.reason };
|
|
171
192
|
return decision.action === "sandbox"
|
|
172
|
-
?
|
|
173
|
-
:
|
|
193
|
+
? { kind: "sandboxed" }
|
|
194
|
+
: { kind: "direct" };
|
|
174
195
|
}
|
|
175
196
|
/** Run the hook command directly through a shell (the default, unconfined). */
|
|
176
197
|
function directSpawn(command, stdin, opts) {
|
package/dist/scan-core.js
CHANGED
|
@@ -225,9 +225,23 @@ function firstBodyParagraph(md) {
|
|
|
225
225
|
}
|
|
226
226
|
return para.join(" ").trim() || undefined;
|
|
227
227
|
}
|
|
228
|
-
/**
|
|
228
|
+
/**
|
|
229
|
+
* The body of a SKILL.md with the leading `---` frontmatter block stripped, PLUS
|
|
230
|
+
* how many lines that block consumed.
|
|
231
|
+
*
|
|
232
|
+
* 🔴 The count is returned rather than discarded because a downstream detector
|
|
233
|
+
* emits LINE NUMBERS. `markdownRefs` counts from one over whatever string it is
|
|
234
|
+
* given, so a coordinate computed over the body is short by exactly the stripped
|
|
235
|
+
* frontmatter — on a 5-line block, the ref on file line 11 printed as 6 (#206).
|
|
236
|
+
* A struct instead of a bare string so the offset is in the caller's hand at the
|
|
237
|
+
* call site, not something to remember.
|
|
238
|
+
*/
|
|
229
239
|
function skillBody(md) {
|
|
230
|
-
|
|
240
|
+
const body = md.replace(/^---\r?\n[\s\S]*?\r?\n---\r?\n?/, "");
|
|
241
|
+
const consumed = md.slice(0, md.length - body.length);
|
|
242
|
+
// Newlines in the consumed prefix == lines before the body starts. Counts a
|
|
243
|
+
// CRLF file identically (the `\r` rides along inside the line).
|
|
244
|
+
return { body, strippedLines: consumed.split("\n").length - 1 };
|
|
231
245
|
}
|
|
232
246
|
/**
|
|
233
247
|
* The on-disk path for a materialized file key. `loadPlugin` prefixes each
|
|
@@ -337,10 +351,14 @@ function scanSkills(files, cls, ctx) {
|
|
|
337
351
|
// a ref under one of those declared dirs against the repo root — OPT-IN, so a
|
|
338
352
|
// repo that doesn't set it is byte-identical to before (no masking of a real
|
|
339
353
|
// missing bundled resource). See feedback P1-4.
|
|
340
|
-
const
|
|
354
|
+
const { body: bodyMd, strippedLines } = skillBody(md);
|
|
355
|
+
const resourceIssues = (0, skill_resources_js_1.skillResourceIssues)(bodyMd, skillDir, {
|
|
341
356
|
repoRoot: sharedDirsRoot,
|
|
342
357
|
sharedDirs,
|
|
343
358
|
existsSync: ctx.existsSync,
|
|
359
|
+
// Findings carry a line number the author is meant to OPEN, so give the
|
|
360
|
+
// detector back the frontmatter it never saw (#206).
|
|
361
|
+
lineOffset: strippedLines,
|
|
344
362
|
});
|
|
345
363
|
// The lethal trifecta is a property of what a unit CAN do. For a SKILL that is
|
|
346
364
|
// NOT its `allowed-tools:` — measured 2026-08-11, and documented by Claude Code
|
package/dist/score-core.d.ts
CHANGED
|
@@ -163,7 +163,27 @@ export declare function trifectaExposure(r: ScanReport): TrifectaExposure;
|
|
|
163
163
|
* surfaced separately, never scored).
|
|
164
164
|
*/
|
|
165
165
|
export declare function reportDeductions(r: ScanReport): Deduction[];
|
|
166
|
-
/**
|
|
166
|
+
/**
|
|
167
|
+
* True when a report has no loadable plugin surface at all (the empty machine).
|
|
168
|
+
*
|
|
169
|
+
* 🔴 THE ONE PLACE the surface set is enumerated, and every scorer must ask it
|
|
170
|
+
* rather than write the list again. `scoreReport` used to keep its own copy and
|
|
171
|
+
* that copy had dropped `inlineHooks` — so a plugin whose entire contribution is
|
|
172
|
+
* hooks declared inline in `plugin.json` scored 0 / F with "no loadable plugin
|
|
173
|
+
* surface" while the same report printed `inlineHooks: 2` (#199). Measured on a
|
|
174
|
+
* two-inline-hook fixture: this predicate said `false` and `scoreReport` said
|
|
175
|
+
* `0`, against `auditScore`'s `100` — a 100-point disagreement between two
|
|
176
|
+
* numbers documented to be equal.
|
|
177
|
+
*
|
|
178
|
+
* A hooks-only plugin is a legitimate shape, the same way a command-only or
|
|
179
|
+
* MCP-only one is: a repo can ship gates and nothing else. Zero has to mean "no
|
|
180
|
+
* surface of any kind", because it reads as "this plugin is broken" — a
|
|
181
|
+
* different statement from "this plugin has no skills or subagents".
|
|
182
|
+
*
|
|
183
|
+
* Hooks count in BOTH forms: `hooks[]` is the script-backed ones, `inlineHooks`
|
|
184
|
+
* the shell one-liners that carry no script file to path-check. The second is
|
|
185
|
+
* still a gate that runs.
|
|
186
|
+
*/
|
|
167
187
|
export declare function isEmptyMachine(r: ScanReport): boolean;
|
|
168
188
|
/**
|
|
169
189
|
* THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
|
package/dist/score-core.js
CHANGED
|
@@ -307,7 +307,27 @@ function reportDeductions(r) {
|
|
|
307
307
|
// advisory note below). The score ranks what's BROKEN.
|
|
308
308
|
];
|
|
309
309
|
}
|
|
310
|
-
/**
|
|
310
|
+
/**
|
|
311
|
+
* True when a report has no loadable plugin surface at all (the empty machine).
|
|
312
|
+
*
|
|
313
|
+
* 🔴 THE ONE PLACE the surface set is enumerated, and every scorer must ask it
|
|
314
|
+
* rather than write the list again. `scoreReport` used to keep its own copy and
|
|
315
|
+
* that copy had dropped `inlineHooks` — so a plugin whose entire contribution is
|
|
316
|
+
* hooks declared inline in `plugin.json` scored 0 / F with "no loadable plugin
|
|
317
|
+
* surface" while the same report printed `inlineHooks: 2` (#199). Measured on a
|
|
318
|
+
* two-inline-hook fixture: this predicate said `false` and `scoreReport` said
|
|
319
|
+
* `0`, against `auditScore`'s `100` — a 100-point disagreement between two
|
|
320
|
+
* numbers documented to be equal.
|
|
321
|
+
*
|
|
322
|
+
* A hooks-only plugin is a legitimate shape, the same way a command-only or
|
|
323
|
+
* MCP-only one is: a repo can ship gates and nothing else. Zero has to mean "no
|
|
324
|
+
* surface of any kind", because it reads as "this plugin is broken" — a
|
|
325
|
+
* different statement from "this plugin has no skills or subagents".
|
|
326
|
+
*
|
|
327
|
+
* Hooks count in BOTH forms: `hooks[]` is the script-backed ones, `inlineHooks`
|
|
328
|
+
* the shell one-liners that carry no script file to path-check. The second is
|
|
329
|
+
* still a gate that runs.
|
|
330
|
+
*/
|
|
311
331
|
function isEmptyMachine(r) {
|
|
312
332
|
const surfaces = r.skills.length +
|
|
313
333
|
r.agents.length +
|
|
@@ -341,11 +361,15 @@ function pluralizeLabel(n, label) {
|
|
|
341
361
|
/** Deterministic structural-health score for one scanned plugin. */
|
|
342
362
|
function scoreReport(r) {
|
|
343
363
|
// An empty/unloadable machine isn't healthy — it's a non-plugin or a broken
|
|
344
|
-
// load. A command-only or
|
|
345
|
-
//
|
|
346
|
-
//
|
|
347
|
-
|
|
348
|
-
|
|
364
|
+
// load. A command-only, MCP-only or HOOKS-only plugin IS a legitimate plugin,
|
|
365
|
+
// though — Anthropic ships command-only plugins in its own marketplace, and a
|
|
366
|
+
// repo can reasonably ship gates and nothing else — so none of them scores 0.
|
|
367
|
+
//
|
|
368
|
+
// 🔴 Asked of `isEmptyMachine`, never re-enumerated here. The hand-written copy
|
|
369
|
+
// that used to sit on this line had dropped `inlineHooks`, so a hooks-only
|
|
370
|
+
// plugin scored 0 while `auditScore` — which does ask the predicate — scored the
|
|
371
|
+
// same report 100 (#199). Two lists mean two answers.
|
|
372
|
+
if (isEmptyMachine(r)) {
|
|
349
373
|
return { score: 0, issues: ["no loadable plugin surface"] };
|
|
350
374
|
}
|
|
351
375
|
const deductions = reportDeductions(r);
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/** The shape a Node resolve hook must return. */
|
|
2
|
+
export type SelfResolved = {
|
|
3
|
+
url: string;
|
|
4
|
+
format?: string | null;
|
|
5
|
+
shortCircuit?: boolean;
|
|
6
|
+
};
|
|
7
|
+
/**
|
|
8
|
+
* Resolve `vigiles` / `vigiles/<subpath>` against the CLI's OWN installation,
|
|
9
|
+
* or `null` when this is not our specifier / we cannot serve it.
|
|
10
|
+
*
|
|
11
|
+
* `null` rather than a throw of its own: the caller is inside a `catch` and the
|
|
12
|
+
* ORIGINAL resolution error is the accurate one to re-raise. An unknown subpath
|
|
13
|
+
* (`vigiles/nope`) would otherwise surface as `ERR_PACKAGE_PATH_NOT_EXPORTED`
|
|
14
|
+
* from our rescue, blaming the rescue for the author's typo.
|
|
15
|
+
*
|
|
16
|
+
* The root is read from the environment at CALL time, not at module load, so a
|
|
17
|
+
* test can drive both branches in one process.
|
|
18
|
+
*/
|
|
19
|
+
export declare function resolveSelfSpecifier(specifier: string): SelfResolved | null;
|
|
20
|
+
//# sourceMappingURL=self-resolve.d.mts.map
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ONE rescue for the bare `vigiles` specifier, shared by both module-resolution
|
|
3
|
+
* hooks this package registers.
|
|
4
|
+
*
|
|
5
|
+
* 🔴 WHY IT EXISTS. A spec does `import { instructionFile } from "vigiles/spec"`
|
|
6
|
+
* and a harness does `import { runHook } from "vigiles"`, so the package has to
|
|
7
|
+
* sit in a `node_modules` Node can reach from that file. In a repo that already
|
|
8
|
+
* has a `package.json`, the obvious way to put it there is `npm install` in the
|
|
9
|
+
* root — which installs the whole dependency tree. Measured by an adopter
|
|
10
|
+
* (#184): **840 packages in 2 minutes** where vigiles alone is 42 and about
|
|
11
|
+
* 90 MB; the other 798 were a model-eval framework and an agent SDK that the
|
|
12
|
+
* gate — `lint`, `compile` and `test`, all deterministic reads — never touches.
|
|
13
|
+
* One run sat 11 minutes in that step before being cancelled. In a repo with NO
|
|
14
|
+
* `package.json` at all (Python, Rust, Go) there is no install to run: the
|
|
15
|
+
* typed-spec path was simply unavailable, and `compile` exited 1 with
|
|
16
|
+
* `ERR_MODULE_NOT_FOUND`.
|
|
17
|
+
*
|
|
18
|
+
* ⚠️ `NODE_PATH` does NOT solve this, and that was measured rather than assumed:
|
|
19
|
+
* Node ignores it for ESM resolution, and both a spec and a harness are ESM. So
|
|
20
|
+
* the only ways to resolve a bare specifier from elsewhere are a real
|
|
21
|
+
* `node_modules` entry (a symlink) or a resolver hook. This is the hook half.
|
|
22
|
+
*
|
|
23
|
+
* 🔴 WHY IT IS SHARED. It was written once for harness scripts and the spec host
|
|
24
|
+
* did not get it, so `vigiles test` worked on a repo without `package.json` and
|
|
25
|
+
* `vigiles compile` did not — the same question answered two different ways by
|
|
26
|
+
* two files. Copying the branch into the second hook would have made that
|
|
27
|
+
* divergence permanent instead of closing it (`one-detector-no-drift`), so the
|
|
28
|
+
* branch lives here and both hooks call it.
|
|
29
|
+
*
|
|
30
|
+
* Scope is deliberately narrow: ONLY the `vigiles` specifier and its subpaths,
|
|
31
|
+
* and only when normal resolution has already FAILED. A repo that has vigiles
|
|
32
|
+
* installed locally keeps resolving to the local copy — the caller tries
|
|
33
|
+
* `nextResolve` first and only reaches this on the way out — so nothing changes
|
|
34
|
+
* for a repo that already worked. This only fills the hole where resolution
|
|
35
|
+
* would otherwise throw.
|
|
36
|
+
*/
|
|
37
|
+
import { createRequire } from "node:module";
|
|
38
|
+
import { pathToFileURL } from "node:url";
|
|
39
|
+
/**
|
|
40
|
+
* Resolve `vigiles` / `vigiles/<subpath>` against the CLI's OWN installation,
|
|
41
|
+
* or `null` when this is not our specifier / we cannot serve it.
|
|
42
|
+
*
|
|
43
|
+
* `null` rather than a throw of its own: the caller is inside a `catch` and the
|
|
44
|
+
* ORIGINAL resolution error is the accurate one to re-raise. An unknown subpath
|
|
45
|
+
* (`vigiles/nope`) would otherwise surface as `ERR_PACKAGE_PATH_NOT_EXPORTED`
|
|
46
|
+
* from our rescue, blaming the rescue for the author's typo.
|
|
47
|
+
*
|
|
48
|
+
* The root is read from the environment at CALL time, not at module load, so a
|
|
49
|
+
* test can drive both branches in one process.
|
|
50
|
+
*/
|
|
51
|
+
export function resolveSelfSpecifier(specifier) {
|
|
52
|
+
// Handed in by the parent process (the CLI), which knows where it is
|
|
53
|
+
// installed. Absent = nobody promised us a root; stay out of the way.
|
|
54
|
+
const self = process.env.VIGILES_SELF_ROOT;
|
|
55
|
+
if (!self)
|
|
56
|
+
return null;
|
|
57
|
+
if (specifier !== "vigiles" && !specifier.startsWith("vigiles/"))
|
|
58
|
+
return null;
|
|
59
|
+
try {
|
|
60
|
+
const require = createRequire(pathToFileURL(`${self}/package.json`));
|
|
61
|
+
// Resolve through the package's own `exports` map rather than guessing a
|
|
62
|
+
// file path, so a subpath like `vigiles/eval` obeys the same contract it
|
|
63
|
+
// would from a normal install. Node's self-reference rule is what makes
|
|
64
|
+
// this work when `self` IS the vigiles package rather than a copy inside
|
|
65
|
+
// some `node_modules`.
|
|
66
|
+
return {
|
|
67
|
+
url: pathToFileURL(require.resolve(specifier, { paths: [self] })).href,
|
|
68
|
+
shortCircuit: true,
|
|
69
|
+
};
|
|
70
|
+
}
|
|
71
|
+
catch {
|
|
72
|
+
return null;
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
//# sourceMappingURL=self-resolve.mjs.map
|
package/dist/spec-hooks.d.mts
CHANGED
|
@@ -19,6 +19,8 @@ type Loaded = {
|
|
|
19
19
|
};
|
|
20
20
|
type NextLoad = (url: string, context: LoadContext) => Loaded | Promise<Loaded>;
|
|
21
21
|
/**
|
|
22
|
+
* Two rescues, both attempted ONLY after normal resolution has failed.
|
|
23
|
+
*
|
|
22
24
|
* `./x.js` → `./x.ts` when the sibling exists.
|
|
23
25
|
*
|
|
24
26
|
* This is the TypeScript ESM convention (`tsc` under `nodenext` requires the
|
|
@@ -27,6 +29,14 @@ type NextLoad = (url: string, context: LoadContext) => Loaded | Promise<Loaded>;
|
|
|
27
29
|
* dogfood specs import `src/core/spec.js`, a file that does not exist on disk.
|
|
28
30
|
* Attempted only AFTER normal resolution fails, so it can never shadow a real
|
|
29
31
|
* `.js` file.
|
|
32
|
+
*
|
|
33
|
+
* Then `vigiles` / `vigiles/<subpath>` against the CLI's own install. Without
|
|
34
|
+
* it, a repo with no `node_modules/vigiles` — every Python, Rust or Go repo,
|
|
35
|
+
* where there is not even an install to run — could not compile a spec at all:
|
|
36
|
+
* `init` scaffolded `import { instructionFile } from "vigiles/spec"` and
|
|
37
|
+
* `compile` exited 1 with `ERR_MODULE_NOT_FOUND`. `vigiles test` had carried
|
|
38
|
+
* this rescue since #184; the spec host did not, which is the whole reason the
|
|
39
|
+
* branch now lives in one shared module rather than in each hook.
|
|
30
40
|
*/
|
|
31
41
|
export declare function resolve(specifier: string, context: ResolveContext, nextResolve: NextResolve): Promise<Resolved>;
|
|
32
42
|
/** Transpile `.ts`/`.mts` with the TypeScript this package already ships. */
|
package/dist/spec-hooks.mjs
CHANGED
|
@@ -18,12 +18,18 @@
|
|
|
18
18
|
* `./x.js` → `./x.ts` specifier rewrite, and bare specifiers. NOT tsconfig
|
|
19
19
|
* `paths`, JSX, or decorator configuration — specs are configuration modules,
|
|
20
20
|
* not applications.
|
|
21
|
+
*
|
|
22
|
+
* The one bare specifier it does more than pass through is `vigiles` itself —
|
|
23
|
+
* see `./self-resolve.mjs`, shared with the harness hook.
|
|
21
24
|
*/
|
|
22
25
|
import { existsSync, readFileSync } from "node:fs";
|
|
23
26
|
import { fileURLToPath } from "node:url";
|
|
24
27
|
import ts from "typescript";
|
|
28
|
+
import { resolveSelfSpecifier } from "./self-resolve.mjs";
|
|
25
29
|
const TS_SOURCE = /\.m?ts$/;
|
|
26
30
|
/**
|
|
31
|
+
* Two rescues, both attempted ONLY after normal resolution has failed.
|
|
32
|
+
*
|
|
27
33
|
* `./x.js` → `./x.ts` when the sibling exists.
|
|
28
34
|
*
|
|
29
35
|
* This is the TypeScript ESM convention (`tsc` under `nodenext` requires the
|
|
@@ -32,6 +38,14 @@ const TS_SOURCE = /\.m?ts$/;
|
|
|
32
38
|
* dogfood specs import `src/core/spec.js`, a file that does not exist on disk.
|
|
33
39
|
* Attempted only AFTER normal resolution fails, so it can never shadow a real
|
|
34
40
|
* `.js` file.
|
|
41
|
+
*
|
|
42
|
+
* Then `vigiles` / `vigiles/<subpath>` against the CLI's own install. Without
|
|
43
|
+
* it, a repo with no `node_modules/vigiles` — every Python, Rust or Go repo,
|
|
44
|
+
* where there is not even an install to run — could not compile a spec at all:
|
|
45
|
+
* `init` scaffolded `import { instructionFile } from "vigiles/spec"` and
|
|
46
|
+
* `compile` exited 1 with `ERR_MODULE_NOT_FOUND`. `vigiles test` had carried
|
|
47
|
+
* this rescue since #184; the spec host did not, which is the whole reason the
|
|
48
|
+
* branch now lives in one shared module rather than in each hook.
|
|
35
49
|
*/
|
|
36
50
|
export async function resolve(specifier, context, nextResolve) {
|
|
37
51
|
try {
|
|
@@ -44,6 +58,9 @@ export async function resolve(specifier, context, nextResolve) {
|
|
|
44
58
|
return { url: candidate.href, format: "module", shortCircuit: true };
|
|
45
59
|
}
|
|
46
60
|
}
|
|
61
|
+
const rescued = resolveSelfSpecifier(specifier);
|
|
62
|
+
if (rescued)
|
|
63
|
+
return rescued;
|
|
47
64
|
throw err;
|
|
48
65
|
}
|
|
49
66
|
}
|
package/dist/test.d.ts
CHANGED
|
@@ -58,10 +58,15 @@ export type { HookRunResult, BlockMechanism, RunHookOptions, HookInput, HookOutp
|
|
|
58
58
|
export * from "./harness-assert.js";
|
|
59
59
|
export { experimental_emitTool, experimental_parseEmitted, experimental_assertEmittedOk, type EmitFieldSchema, type EmitObjectSchema, type EmitPropertySchema, type EmitTrackSchema, type EmitToolDefinition, type ExperimentalEmitTool, } from "./experimental-emit.js";
|
|
60
60
|
export { loadHook } from "./load-hook.js";
|
|
61
|
+
export { experimental_hookState } from "./hook-state-store.js";
|
|
62
|
+
export type { HookStateHandle, SeedStateOptions } from "./hook-state-store.js";
|
|
63
|
+
export type { StateFact, Duration } from "./core/hook-state.js";
|
|
61
64
|
export { evalChecks, assertChecks, tool, toolWith, notTool, onlyTools, skill, output, hookFired, received, turns, wrote, didNotWrite, subagent, blocked, allowed, mcp, cost, latency, tokens, inputTokens, outputTokens, cacheTokens, } from "./check.js";
|
|
62
65
|
export type { ArgMatcher, Check, CheckJSON, CheckResult, JudgeFn, } from "./check.js";
|
|
63
66
|
export { DISASTER_CATALOG, verifyGuardrail, unblockedDisasters, assertBlocksDisasters, formatGuardrailReport, experimental_alternateSpellings, } from "./guardrail-check.js";
|
|
64
67
|
export type { DisasterEvent, DisasterCategory, GuardrailResult, VerifyGuardrailOptions, } from "./guardrail-check.js";
|
|
68
|
+
export { experimental_verifyPluginGuards, experimental_formatPluginGuardReport, } from "./verify-plugin-guards.js";
|
|
69
|
+
export type { PluginGuardReport, SweptHook, SweptHookOutcome, NonCommandHookAction, VerifyPluginGuardsOptions, } from "./verify-plugin-guards.js";
|
|
65
70
|
export * from "./tool-stub.js";
|
|
66
71
|
export { runHarnessTest, runHarness, parseToolCalls, parseSubagents, parseResultEvent, parseOutput, parseHooks, decideSandbox, specTrusted, sandboxAvailable, } from "./harness-test.js";
|
|
67
72
|
export type { HarnessTestSpec, Trace, SubagentTrace, HarnessTestResult, RunHarnessTestOptions, ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, SandboxMode, } from "./harness-test.js";
|
package/dist/test.js
CHANGED
|
@@ -66,8 +66,8 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
|
66
66
|
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
67
67
|
};
|
|
68
68
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
69
|
-
exports.
|
|
70
|
-
exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = void 0;
|
|
69
|
+
exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.experimental_formatPluginGuardReport = exports.experimental_verifyPluginGuards = exports.experimental_alternateSpellings = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.experimental_hookState = exports.loadHook = exports.experimental_assertEmittedOk = exports.experimental_parseEmitted = exports.experimental_emitTool = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
|
|
70
|
+
exports.experimental_makeDockerRuntime = exports.experimental_dockerRuntime = exports.experimental_withServices = exports.experimental_startServices = exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = void 0;
|
|
71
71
|
// --- reporting: how much did this script actually do? ---
|
|
72
72
|
// `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
|
|
73
73
|
// prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
|
|
@@ -113,6 +113,15 @@ Object.defineProperty(exports, "experimental_assertEmittedOk", { enumerable: tru
|
|
|
113
113
|
// the CLI runtime uses, so a hook that loads in a test loads identically in prod.
|
|
114
114
|
var load_hook_js_1 = require("./load-hook.js");
|
|
115
115
|
Object.defineProperty(exports, "loadHook", { enumerable: true, get: function () { return load_hook_js_1.loadHook; } });
|
|
116
|
+
// Seed + read a compiled hook's NAMED STATE from a test. A throttled hook only
|
|
117
|
+
// does anything interesting against an OLD fact, and "old" is not something a
|
|
118
|
+
// test can produce by waiting — so without this a consumer testing a throttle
|
|
119
|
+
// had to reconstruct the store's private path by hand (the dogfood repo did,
|
|
120
|
+
// and it broke when the facts were renamed). It is on THIS barrel and not on
|
|
121
|
+
// `vigiles/hook` on purpose: `vigiles/hook` is the only import a compiled hook
|
|
122
|
+
// may have, and its guarantee is that it hands out no writer.
|
|
123
|
+
var hook_state_store_js_1 = require("./hook-state-store.js");
|
|
124
|
+
Object.defineProperty(exports, "experimental_hookState", { enumerable: true, get: function () { return hook_state_store_js_1.experimental_hookState; } });
|
|
116
125
|
// --- the declarative check vocabulary ---
|
|
117
126
|
// Enumerated rather than `export *` ON PURPOSE: `judged` also lives in check.ts
|
|
118
127
|
// and its default judge is a real model call, so it is the one member of this
|
|
@@ -153,6 +162,18 @@ Object.defineProperty(exports, "unblockedDisasters", { enumerable: true, get: fu
|
|
|
153
162
|
Object.defineProperty(exports, "assertBlocksDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.assertBlocksDisasters; } });
|
|
154
163
|
Object.defineProperty(exports, "formatGuardrailReport", { enumerable: true, get: function () { return guardrail_check_js_1.formatGuardrailReport; } });
|
|
155
164
|
Object.defineProperty(exports, "experimental_alternateSpellings", { enumerable: true, get: function () { return guardrail_check_js_1.experimental_alternateSpellings; } });
|
|
165
|
+
// The same battery, pointed at a DIRECTORY instead of one command string. It
|
|
166
|
+
// reads each hook's event, matcher, command and condition off the same
|
|
167
|
+
// registration, so the pairing mistake `verifyGuardrail`'s own comment has to ask
|
|
168
|
+
// callers to avoid ("pass the hook's declared `if` here") cannot be made. It
|
|
169
|
+
// belongs on THIS barrel and beside the battery for the same reason the battery
|
|
170
|
+
// does: nothing here calls a model.
|
|
171
|
+
// The report + its renderer ship together: the sweep's one motivating use is
|
|
172
|
+
// "point the battery at YOUR hooks", and without the formatter that is a
|
|
173
|
+
// hand-written fold over a discriminated union at every call site.
|
|
174
|
+
var verify_plugin_guards_js_1 = require("./verify-plugin-guards.js");
|
|
175
|
+
Object.defineProperty(exports, "experimental_verifyPluginGuards", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_verifyPluginGuards; } });
|
|
176
|
+
Object.defineProperty(exports, "experimental_formatPluginGuardReport", { enumerable: true, get: function () { return verify_plugin_guards_js_1.experimental_formatPluginGuardReport; } });
|
|
156
177
|
// Tool stubs on PATH (rung R2): shadow a CLI tool with a recorded canned result.
|
|
157
178
|
__exportStar(require("./tool-stub.js"), exports);
|
|
158
179
|
// The assembled machine — AGNOSTIC SURFACE ONLY. The Claude-Code transport
|