vigiles 9.1.0 → 11.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +126 -112
- package/dist/adapters/claude-code/dialect.js +15 -0
- package/dist/audit-html.d.ts +15 -4
- package/dist/audit-html.js +15 -6
- package/dist/audit-report.d.ts +58 -2
- package/dist/audit-report.js +29 -0
- package/dist/audit-report.template.html +34 -24
- package/dist/audit-score.d.ts +19 -12
- package/dist/audit-score.js +79 -15
- package/dist/audit-serve.d.ts +109 -0
- package/dist/audit-serve.js +257 -0
- package/dist/cli.js +435 -20
- package/dist/core/CLAUDE.md.spec.d.ts +3 -0
- package/dist/core/CLAUDE.md.spec.js +26 -0
- package/dist/core/compile.d.ts +5 -1
- package/dist/core/compile.js +19 -10
- package/dist/core/delegation-trifecta.d.ts +64 -0
- package/dist/core/delegation-trifecta.js +124 -0
- package/dist/core/dialect.d.ts +18 -0
- package/dist/core/hook-block-ineffective.d.ts +62 -0
- package/dist/core/hook-block-ineffective.js +153 -0
- package/dist/core/hook-matcher.d.ts +66 -0
- package/dist/core/hook-matcher.js +182 -0
- package/dist/core/hook-normalize.d.ts +43 -0
- package/dist/core/hook-normalize.js +78 -0
- package/dist/core/lethal-trifecta.d.ts +100 -0
- package/dist/core/lethal-trifecta.js +197 -0
- package/dist/core/plugin-dir-layout.d.ts +30 -0
- package/dist/core/plugin-dir-layout.js +73 -0
- package/dist/core/rule-meta.d.ts +82 -0
- package/dist/core/rule-meta.js +266 -0
- package/dist/core/skill-missing-fence.d.ts +47 -0
- package/dist/core/skill-missing-fence.js +119 -0
- package/dist/core/skill-resources.d.ts +27 -0
- package/dist/core/skill-resources.js +167 -0
- package/dist/core/types.d.ts +71 -0
- package/dist/core/validate.d.ts +1 -0
- package/dist/core/validate.js +26 -4
- package/dist/leaderboard.d.ts +1 -0
- package/dist/leaderboard.js +64 -15
- package/dist/scan-behavioral.d.ts +85 -0
- package/dist/scan-behavioral.js +225 -0
- package/dist/scan.d.ts +106 -0
- package/dist/scan.js +269 -53
- package/dist/setup-plan.d.ts +6 -3
- package/dist/setup-plan.js +12 -2
- package/package.json +1 -1
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.skillResourceIssues = skillResourceIssues;
|
|
4
|
+
/**
|
|
5
|
+
* vigiles — SKILL bundled-resource resolution (the cross-reference moat applied
|
|
6
|
+
* to a SKILL.md body).
|
|
7
|
+
*
|
|
8
|
+
* A SKILL.md body is freeform markdown that routinely points the agent at LOCAL
|
|
9
|
+
* BUNDLED files shipped beside it — `scripts/foo.sh`, `references/api.md`,
|
|
10
|
+
* `[setup](./scripts/run.py)`, an inline `run \`scripts/setup.sh\``. When a
|
|
11
|
+
* referenced file doesn't exist on disk under the skill directory, the agent
|
|
12
|
+
* reads the instruction, gets nothing, and silently continues (a documented top
|
|
13
|
+
* skill pain — one practitioner found 59 broken refs across 192 files). vigiles
|
|
14
|
+
* already verifies file/script refs inside typed specs (core/refs.ts,
|
|
15
|
+
* core/doc-refs.ts) and intra-plugin script refs in the loader
|
|
16
|
+
* (plugin-loader.ts `danglingRefs`); this extends that to the SKILL.md body.
|
|
17
|
+
*
|
|
18
|
+
* HIGH-PRECISION / FP-SAFE, by the same don't-cry-wolf discipline the rest of
|
|
19
|
+
* vigiles holds (see `danglingRefs`/`isPluginRooted`): we flag ONLY references
|
|
20
|
+
* that are UNAMBIGUOUSLY a local bundled resource — a markdown link to a
|
|
21
|
+
* relative path with a file extension, or an explicit `scripts/`/`references/`/
|
|
22
|
+
* `assets/`-prefixed path (the Agent-Skills standard bundle dirs) with an
|
|
23
|
+
* extension. Everything else is skipped: URLs, absolute paths, `${VAR}`/`$VAR`
|
|
24
|
+
* tokens, `../` escapes, bare words with no extension or known prefix. Prefer
|
|
25
|
+
* MISSING a real ref over emitting a false positive — a noisy resource check
|
|
26
|
+
* would teach users to ignore it.
|
|
27
|
+
*
|
|
28
|
+
* Pure: the only IO is an injectable `existsSync` (default node:fs), mirroring
|
|
29
|
+
* core/refs.ts and the loader so the detector is testable with a fake.
|
|
30
|
+
*/
|
|
31
|
+
const node_fs_1 = require("node:fs");
|
|
32
|
+
const node_path_1 = require("node:path");
|
|
33
|
+
// ---------------------------------------------------------------------------
|
|
34
|
+
// Shapes we match vs deliberately skip (FP-safety)
|
|
35
|
+
// ---------------------------------------------------------------------------
|
|
36
|
+
const FENCE = /^\s*(?:`{3,}|~{3,})/;
|
|
37
|
+
// The Agent-Skills standard bundle subdirectories. A path PREFIXED by one of
|
|
38
|
+
// these is unambiguously a local bundled resource, even without a `./`.
|
|
39
|
+
const BUNDLE_DIRS = ["scripts", "references", "assets"];
|
|
40
|
+
const BUNDLE_PREFIX = new RegExp(`^(?:${BUNDLE_DIRS.join("|")})/`);
|
|
41
|
+
// A markdown inline link `[text](target)` — we read its target.
|
|
42
|
+
const MD_LINK = /\[[^\]]*\]\(([^)\s]+)\)/g;
|
|
43
|
+
// An inline-code path mention: a backtick span whose whole content is a single
|
|
44
|
+
// path token. We only treat it as a ref when it is a bundle-dir-prefixed path
|
|
45
|
+
// with an extension (the high-confidence shape); a bare `scripts` or a generic
|
|
46
|
+
// `foo.ts` mention is NOT flagged.
|
|
47
|
+
const INLINE_SPAN = /`([^`\n]+)`/g;
|
|
48
|
+
// A path must carry a file extension to be a resource reference. A bare word or
|
|
49
|
+
// a directory name (`scripts/lib`) is undecidable prose — skipped.
|
|
50
|
+
const HAS_EXT = /\.[A-Za-z0-9]+$/;
|
|
51
|
+
/**
|
|
52
|
+
* A reference target is a LOCAL BUNDLED RESOURCE worth resolving iff it is a
|
|
53
|
+
* relative path with a file extension AND is not one of the skip shapes. This is
|
|
54
|
+
* the single gate; both the link path and the inline-path path run through it.
|
|
55
|
+
*/
|
|
56
|
+
function localResourceTarget(rawTarget) {
|
|
57
|
+
// Strip a markdown link title / fragment / query if present, and trim.
|
|
58
|
+
const target = rawTarget.trim();
|
|
59
|
+
if (target.length === 0)
|
|
60
|
+
return null;
|
|
61
|
+
// SKIP: URLs (http://, https://, mailto:, any scheme://) — external.
|
|
62
|
+
if (/^[a-z][a-z0-9+.-]*:\/\//i.test(target) || /^mailto:/i.test(target)) {
|
|
63
|
+
return null;
|
|
64
|
+
}
|
|
65
|
+
// SKIP: a pure anchor / fragment-only link (`#section`).
|
|
66
|
+
if (target.startsWith("#"))
|
|
67
|
+
return null;
|
|
68
|
+
// SKIP: absolute paths (`/etc/x`, Windows `C:\`) — not bundled-relative.
|
|
69
|
+
if (target.startsWith("/") || /^[A-Za-z]:[\\/]/.test(target))
|
|
70
|
+
return null;
|
|
71
|
+
// SKIP: variable tokens (`${CLAUDE_PLUGIN_ROOT}/x`, `$VAR/x`) — uncheckable,
|
|
72
|
+
// and almost always a plugin-root or runtime path, not a bundled file.
|
|
73
|
+
if (target.includes("$"))
|
|
74
|
+
return null;
|
|
75
|
+
// Drop a URL fragment / query suffix so `references/api.md#auth` resolves to
|
|
76
|
+
// the file. (Only after the scheme check above, so we never mangle a URL.)
|
|
77
|
+
const path = target.replace(/[?#].*$/, "");
|
|
78
|
+
if (path.length === 0)
|
|
79
|
+
return null;
|
|
80
|
+
// SKIP: a `../` escape OUT of the skill dir — undecidable / not a bundled
|
|
81
|
+
// resource (it points at a sibling skill or the repo). A leading `./` is fine.
|
|
82
|
+
const normalized = path.replace(/^\.\//, "");
|
|
83
|
+
if (normalized.startsWith("../") || normalized.includes("/../"))
|
|
84
|
+
return null;
|
|
85
|
+
// Must look like a file (have an extension), else it's a dir/prose mention.
|
|
86
|
+
if (!HAS_EXT.test(normalized))
|
|
87
|
+
return null;
|
|
88
|
+
return normalized;
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* Whether an inline-code path token is high-confidence enough to flag on its
|
|
92
|
+
* own (no surrounding `[..](..)` link syntax). We require a BUNDLE-DIR PREFIX
|
|
93
|
+
* (`scripts/`, `references/`, `assets/`) so a generic `` `config.json` `` or a
|
|
94
|
+
* `` `src/foo.ts` `` API mention in prose is never flagged — only the standard
|
|
95
|
+
* bundle layout, which is unambiguously a shipped resource.
|
|
96
|
+
*/
|
|
97
|
+
function isInlineBundlePath(token) {
|
|
98
|
+
const t = token.trim();
|
|
99
|
+
// A single token only — a span with spaces is a command/prose, not a path.
|
|
100
|
+
if (/\s/.test(t))
|
|
101
|
+
return false;
|
|
102
|
+
const normalized = t.replace(/^\.\//, "");
|
|
103
|
+
return BUNDLE_PREFIX.test(normalized) && HAS_EXT.test(normalized);
|
|
104
|
+
}
|
|
105
|
+
/** Collect candidate bundled-resource refs from one body line, skipping fences. */
|
|
106
|
+
function candidatesInLine(line, lineNo) {
|
|
107
|
+
const out = [];
|
|
108
|
+
// Markdown links: any relative path target that passes the local-resource gate.
|
|
109
|
+
for (const m of line.matchAll(MD_LINK)) {
|
|
110
|
+
const resolved = localResourceTarget(m[1]);
|
|
111
|
+
if (resolved !== null) {
|
|
112
|
+
out.push({ ref: m[1].trim(), resolved, kind: "link", line: lineNo });
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
// Inline-code path mentions: only the high-confidence bundle-dir-prefixed form.
|
|
116
|
+
for (const m of line.matchAll(INLINE_SPAN)) {
|
|
117
|
+
const token = m[1].trim();
|
|
118
|
+
if (!isInlineBundlePath(token))
|
|
119
|
+
continue;
|
|
120
|
+
const resolved = localResourceTarget(token);
|
|
121
|
+
if (resolved !== null) {
|
|
122
|
+
out.push({ ref: token, resolved, kind: "path", line: lineNo });
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
return out;
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* The bundled-resource references in a SKILL.md body that don't resolve on disk
|
|
129
|
+
* under `skillDir`. Pure + FP-safe (see the module header). `skillDir` is the
|
|
130
|
+
* directory the SKILL.md itself lives in (resources are bundled beside it).
|
|
131
|
+
*
|
|
132
|
+
* The shared detector behind both `vigiles lint` (the `skill-resource-resolves`
|
|
133
|
+
* rule) and `vigiles audit` (the read-only report) — one detector, no drift.
|
|
134
|
+
*/
|
|
135
|
+
function skillResourceIssues(skillBody, skillDir, opts = {}) {
|
|
136
|
+
const exists = opts.existsSync ?? node_fs_1.existsSync;
|
|
137
|
+
const findings = [];
|
|
138
|
+
const seen = new Set();
|
|
139
|
+
const lines = skillBody.split("\n");
|
|
140
|
+
let inFence = false;
|
|
141
|
+
for (let i = 0; i < lines.length; i++) {
|
|
142
|
+
if (FENCE.test(lines[i])) {
|
|
143
|
+
inFence = !inFence;
|
|
144
|
+
continue;
|
|
145
|
+
}
|
|
146
|
+
if (inFence)
|
|
147
|
+
continue;
|
|
148
|
+
for (const c of candidatesInLine(lines[i], i + 1)) {
|
|
149
|
+
const full = (0, node_path_1.resolve)(skillDir, c.resolved);
|
|
150
|
+
if (exists(full))
|
|
151
|
+
continue;
|
|
152
|
+
// De-dupe the same missing file referenced several times in the body.
|
|
153
|
+
const key = `${c.kind}:${c.resolved}`;
|
|
154
|
+
if (seen.has(key))
|
|
155
|
+
continue;
|
|
156
|
+
seen.add(key);
|
|
157
|
+
findings.push({
|
|
158
|
+
ref: c.ref,
|
|
159
|
+
resolved: c.resolved,
|
|
160
|
+
kind: c.kind,
|
|
161
|
+
line: c.line,
|
|
162
|
+
});
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
return findings;
|
|
166
|
+
}
|
|
167
|
+
//# sourceMappingURL=skill-resources.js.map
|
package/dist/core/types.d.ts
CHANGED
|
@@ -220,6 +220,77 @@ export interface RulesConfig {
|
|
|
220
220
|
* as `scan` (mcpHookIssues).
|
|
221
221
|
*/
|
|
222
222
|
"mcp-hook-target-resolves"?: RuleSeverity;
|
|
223
|
+
/**
|
|
224
|
+
* Flag a unit (subagent / model-invocable skill) whose declared tools hold all
|
|
225
|
+
* THREE legs of Simon Willison's "lethal trifecta" — read private data, ingest
|
|
226
|
+
* untrusted content, AND exfiltrate — a prompt-injection exfil path with no
|
|
227
|
+
* exploit code (Meta's Rule of Two: allow at most two). A capability SET-
|
|
228
|
+
* intersection over the declared contract, NOT a text scan; high-precision (only
|
|
229
|
+
* well-known tools map to a leg). An EXPLICIT all-three contract is a "hard"
|
|
230
|
+
* finding; an inherits-all unit (no contract → every leg) is "advisory". Default
|
|
231
|
+
* "warn" (don't-cry-wolf rollout); raise to "error" to gate CI. Same detector as
|
|
232
|
+
* `scan` (lethalTrifectaIssues). See docs/rules/lethal-trifecta.md.
|
|
233
|
+
*/
|
|
234
|
+
"lethal-trifecta"?: RuleSeverity;
|
|
235
|
+
/**
|
|
236
|
+
* Flag a SKILL.md body referencing a bundled file (`scripts/`/`references/`/
|
|
237
|
+
* `assets/`, or a relative markdown link with an extension) that doesn't exist
|
|
238
|
+
* on disk under the skill dir — the agent reads the instruction, gets nothing,
|
|
239
|
+
* and silently continues. The cross-reference moat applied to the SKILL.md body.
|
|
240
|
+
* High-precision / FP-safe (skips URLs, `$VAR` tokens, `../` escapes, extension-
|
|
241
|
+
* less mentions). Default "warn"; raise to "error" to gate CI. Same detector as
|
|
242
|
+
* `scan` (skillResourceIssues). See docs/rules/skill-resource-resolves.md.
|
|
243
|
+
*/
|
|
244
|
+
"skill-resource-resolves"?: RuleSeverity;
|
|
245
|
+
/**
|
|
246
|
+
* Flag a SKILL.md that opens with frontmatter-looking keys (`name:`,
|
|
247
|
+
* `description:`, …) but has NO opening `---` fence — the harness loads the
|
|
248
|
+
* whole file as body, so the skill has no name/description/trigger and is
|
|
249
|
+
* invisible (never fires). High-precision (a fixed key whitelist; markdown /
|
|
250
|
+
* prose lines never match). Default "warn"; raise to "error" to gate CI. Same
|
|
251
|
+
* detector as `scan` (skillFenceIssues). See docs/rules/skill-missing-fence.md.
|
|
252
|
+
*/
|
|
253
|
+
"skill-missing-fence"?: RuleSeverity;
|
|
254
|
+
/**
|
|
255
|
+
* Flag a functional surface directory (skills/agents/commands) nested INSIDE
|
|
256
|
+
* the `.claude-plugin/` manifest dir, where only `plugin.json` belongs — the
|
|
257
|
+
* harness can't see it, so the surface is invisible (the #1 plugin-author
|
|
258
|
+
* mistake). Pure filesystem check, FP-safe. Default "warn"; raise to "error"
|
|
259
|
+
* to gate CI. Same detector as `scan` (pluginLayoutIssues). See
|
|
260
|
+
* docs/rules/plugin-dir-layout.md.
|
|
261
|
+
*/
|
|
262
|
+
"plugin-dir-layout"?: RuleSeverity;
|
|
263
|
+
/**
|
|
264
|
+
* Flag a lethal trifecta that EMERGES across a delegation edge — a subagent
|
|
265
|
+
* whose effective (own ∪ delegated-to) capability holds all three legs though
|
|
266
|
+
* no single unit does (the combined blast radius). The capability-diff across
|
|
267
|
+
* the delegation tree; skips units the per-unit `lethal-trifecta` already
|
|
268
|
+
* flags (no double-report), FP-safe (explicit edges, wildcard-guarded). Default
|
|
269
|
+
* "warn"; raise to "error" to gate CI. Same detector as `scan`
|
|
270
|
+
* (delegationTrifecta). See docs/rules/delegation-trifecta.md.
|
|
271
|
+
*/
|
|
272
|
+
"delegation-trifecta"?: RuleSeverity;
|
|
273
|
+
/**
|
|
274
|
+
* Flag a hook that LOOKS like it blocks but silently doesn't — a block decision
|
|
275
|
+
* (`exit 2` / `decision` / `permissionDecision`) on an event that can't veto, or
|
|
276
|
+
* the legacy top-level `decision` field on a permission-gated event where only
|
|
277
|
+
* `hookSpecificOutput.permissionDecision` works (#19009, the #1 verified hook
|
|
278
|
+
* pain). FP-safe (conservative literal patterns; the blocking-event sets are
|
|
279
|
+
* read from the dialect, so it runs only where they're declared). Default "warn";
|
|
280
|
+
* raise to "error" to gate CI. Same detector as `scan` (hookBlockFindings). See
|
|
281
|
+
* docs/rules/hook-block-ineffective.md.
|
|
282
|
+
*/
|
|
283
|
+
"hook-block-ineffective"?: RuleSeverity;
|
|
284
|
+
/**
|
|
285
|
+
* Flag a hook `matcher` string that silently never fires — a close typo of a
|
|
286
|
+
* built-in tool (`bash`→`Bash`), or a malformed/undeclared MCP form
|
|
287
|
+
* (`mcp_memory_*` instead of `mcp__memory__.*`, or a server the plugin doesn't
|
|
288
|
+
* declare). High-precision (close-typo only; MCP gated on a declared set,
|
|
289
|
+
* built-ins allowlisted; wildcards/regex skipped). Default "warn"; raise to
|
|
290
|
+
* "error" to gate CI. Same detector as `scan` (hookMatcherFindings). See
|
|
291
|
+
* docs/rules/hook-matcher.md.
|
|
292
|
+
*/
|
|
293
|
+
"hook-matcher"?: RuleSeverity;
|
|
223
294
|
}
|
|
224
295
|
/** Extract severity from a rule value (handles both simple and tuple forms). */
|
|
225
296
|
export declare function ruleSeverity<T>(rule: RuleWithOptions<T> | undefined): RuleSeverity;
|
package/dist/core/validate.d.ts
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { ParsedRule, ValidationError, ValidationResult, ReadResult, FileResult, ValidatePathsResult, RulesConfig, VigilesConfig, MarkerType, ParseOptions, ValidateOptions, ValidatePathsOptions, ReadOptions } from "./types.js";
|
|
2
2
|
export type { ParsedRule, ValidationError, ValidationResult, ReadResult, FileResult, ValidatePathsResult, RulesConfig, VigilesConfig, MarkerType, ParseOptions, ValidateOptions, ValidatePathsOptions, ReadOptions, };
|
|
3
|
+
export declare const DEFAULT_RULES: Required<RulesConfig>;
|
|
3
4
|
export declare function findInstructionFiles(cwd?: string, configFiles?: string[]): string[];
|
|
4
5
|
export declare function loadConfig(): VigilesConfig;
|
|
5
6
|
export declare function parseRules(content: string, { ruleMarkers }?: ParseOptions): ParsedRule[];
|
package/dist/core/validate.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.DEFAULT_RULES = void 0;
|
|
3
4
|
exports.findInstructionFiles = findInstructionFiles;
|
|
4
5
|
exports.loadConfig = loadConfig;
|
|
5
6
|
exports.parseRules = parseRules;
|
|
@@ -28,7 +29,7 @@ const VALID_MARKERS = ["headings", "checkboxes"];
|
|
|
28
29
|
const INSTRUCTION_FILES = ["CLAUDE.md", "AGENTS.md"];
|
|
29
30
|
// The default instruction file to validate when no config names one.
|
|
30
31
|
const DEFAULT_FILES = [INSTRUCTION_FILES[0]];
|
|
31
|
-
|
|
32
|
+
exports.DEFAULT_RULES = {
|
|
32
33
|
"require-instructions-spec": "warn",
|
|
33
34
|
// Default OFF — the consistent `require-<surface>-spec` parallel. Skills are
|
|
34
35
|
// legitimately hand-written, so requiring a .spec.ts per SKILL.md is the wrong
|
|
@@ -70,10 +71,31 @@ const DEFAULT_RULES = {
|
|
|
70
71
|
"frontmatter-valid": "warn",
|
|
71
72
|
// A mcp_tool hook incomplete / targeting an undeclared server — on by default at warn.
|
|
72
73
|
"mcp-hook-target-resolves": "warn",
|
|
74
|
+
// Lethal-trifecta capability set-intersection (read-private + ingest-untrusted +
|
|
75
|
+
// exfiltrate in one unit) — WARN by default (don't-cry-wolf rollout); raise to error.
|
|
76
|
+
"lethal-trifecta": "warn",
|
|
77
|
+
// A SKILL.md body referencing a missing bundled resource — WARN by default
|
|
78
|
+
// (don't-cry-wolf rollout, FP-safe); raise to error to gate CI.
|
|
79
|
+
"skill-resource-resolves": "warn",
|
|
80
|
+
// A SKILL.md missing its opening `---` fence (invisible skill) — WARN by
|
|
81
|
+
// default (FP-safe key whitelist); raise to error to gate CI.
|
|
82
|
+
"skill-missing-fence": "warn",
|
|
83
|
+
// Functional dirs nested inside `.claude-plugin/` (invisible surfaces) — WARN
|
|
84
|
+
// by default; raise to error to gate CI.
|
|
85
|
+
"plugin-dir-layout": "warn",
|
|
86
|
+
// A lethal trifecta emerging across a delegation edge (combined blast radius) —
|
|
87
|
+
// WARN by default (don't-cry-wolf rollout); raise to error to gate CI.
|
|
88
|
+
"delegation-trifecta": "warn",
|
|
89
|
+
// A hook that looks like it blocks but silently doesn't (#19009) — WARN by
|
|
90
|
+
// default (FP-safe literal patterns); raise to error to gate CI.
|
|
91
|
+
"hook-block-ineffective": "warn",
|
|
92
|
+
// A hook matcher that never fires (tool typo / wrong MCP form) — WARN by
|
|
93
|
+
// default (high-precision); raise to error to gate CI.
|
|
94
|
+
"hook-matcher": "warn",
|
|
73
95
|
};
|
|
74
96
|
const DEFAULT_CONFIG = {
|
|
75
97
|
ruleMarkers: ["headings", "checkboxes"],
|
|
76
|
-
rules: DEFAULT_RULES,
|
|
98
|
+
rules: exports.DEFAULT_RULES,
|
|
77
99
|
files: DEFAULT_FILES,
|
|
78
100
|
};
|
|
79
101
|
// ---------------------------------------------------------------------------
|
|
@@ -99,7 +121,7 @@ function loadConfig() {
|
|
|
99
121
|
const config = {
|
|
100
122
|
...DEFAULT_CONFIG,
|
|
101
123
|
...userConfig,
|
|
102
|
-
rules: { ...DEFAULT_RULES, ...userConfig.rules },
|
|
124
|
+
rules: { ...exports.DEFAULT_RULES, ...userConfig.rules },
|
|
103
125
|
files: Array.isArray(userConfig.files) ? userConfig.files : DEFAULT_FILES,
|
|
104
126
|
};
|
|
105
127
|
if (!Array.isArray(config.ruleMarkers) ||
|
|
@@ -170,7 +192,7 @@ function parseRules(content, { ruleMarkers } = {}) {
|
|
|
170
192
|
// Core validation
|
|
171
193
|
// ---------------------------------------------------------------------------
|
|
172
194
|
function validate(content, { ruleMarkers, rules: rulesConfig, filePath, dialect } = {}) {
|
|
173
|
-
const activeRules = rulesConfig ?? DEFAULT_RULES;
|
|
195
|
+
const activeRules = rulesConfig ?? exports.DEFAULT_RULES;
|
|
174
196
|
const parsedRules = parseRules(content, { ruleMarkers });
|
|
175
197
|
const enforced = parsedRules.filter((r) => r.enforcement === "enforced").length;
|
|
176
198
|
const guidanceOnly = parsedRules.filter((r) => r.enforcement === "guidance").length;
|
package/dist/leaderboard.d.ts
CHANGED
|
@@ -26,6 +26,7 @@ export declare const W_NO_DESCRIPTION = 10;
|
|
|
26
26
|
export declare const W_DANGLING_REF = 8;
|
|
27
27
|
export declare const W_OVERLAP = 8;
|
|
28
28
|
export declare const W_NO_CONTRACT = 5;
|
|
29
|
+
export declare const W_TRIFECTA = 20;
|
|
29
30
|
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
30
31
|
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
31
32
|
/** One deduction: a count, its per-item weight, and the label if non-zero. */
|
package/dist/leaderboard.js
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* model and stack on top later; this part runs anywhere in CI for free.
|
|
13
13
|
*/
|
|
14
14
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
15
|
-
exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
15
|
+
exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
16
16
|
exports.gradeFor = gradeFor;
|
|
17
17
|
exports.reportDeductions = reportDeductions;
|
|
18
18
|
exports.isEmptyMachine = isEmptyMachine;
|
|
@@ -48,8 +48,11 @@ exports.W_MISSING_HOOK = 15; // a hook script that doesn't exist → never runs
|
|
|
48
48
|
exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't trigger
|
|
49
49
|
exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
50
50
|
exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
|
|
51
|
-
exports.W_NO_CONTRACT = 5; //
|
|
52
|
-
// (
|
|
51
|
+
exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
|
|
52
|
+
exports.W_TRIFECTA = 20; // a HARD lethal-trifecta contract (all three legs, explicit) → a declared prompt-injection exfil path
|
|
53
|
+
// Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
|
|
54
|
+
// - untested surfaces — a hardening gap, not breakage.
|
|
55
|
+
// - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
|
|
53
56
|
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
54
57
|
function gradeFor(score) {
|
|
55
58
|
if (score >= 90)
|
|
@@ -72,11 +75,20 @@ function gradeFor(score) {
|
|
|
72
75
|
function reportDeductions(r) {
|
|
73
76
|
const missingHooks = r.hooks.filter((h) => h.status === "missing").length;
|
|
74
77
|
const noDesc = r.skills.filter((s) => !s.hasDescription).length;
|
|
75
|
-
const noContract = r.agents.filter((a) => a.tools === null).length;
|
|
76
78
|
const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
|
|
77
79
|
const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
|
|
78
80
|
const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
|
|
81
|
+
// HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
|
|
82
|
+
// legs (a declared exfil path). Advisory (inherits-all) trifecta findings are
|
|
83
|
+
// surfaced but NEVER graded (aligned with the inherits-all stance), so they're
|
|
84
|
+
// excluded here.
|
|
85
|
+
const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
|
|
79
86
|
return [
|
|
87
|
+
{
|
|
88
|
+
n: hardTrifecta,
|
|
89
|
+
weight: exports.W_TRIFECTA,
|
|
90
|
+
label: "unit(s) holding all three lethal-trifecta legs (prompt-injection exfil path)",
|
|
91
|
+
},
|
|
80
92
|
{
|
|
81
93
|
n: missingHooks,
|
|
82
94
|
weight: exports.W_MISSING_HOOK,
|
|
@@ -117,11 +129,14 @@ function reportDeductions(r) {
|
|
|
117
129
|
weight: exports.W_NO_CONTRACT,
|
|
118
130
|
label: "agent disallowedTools typo(s) that block nothing",
|
|
119
131
|
},
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
132
|
+
// NB: an agent that inherits all tools (no `tools:` line) is ADVISORY, not a
|
|
133
|
+
// graded penalty — it's surfaced by scoreReport / the Structure ring but never
|
|
134
|
+
// drags the score. WHY: omitting the `tools:` line is a near-universal,
|
|
135
|
+
// legitimate authoring style (a measured OSS sweep of 122 real plugins found
|
|
136
|
+
// 109 whose ONLY finding was this), so penalizing it makes the grade cry wolf
|
|
137
|
+
// on idiomatic subagents. A health score should mean "something is BROKEN", and
|
|
138
|
+
// a broad-by-default tool surface is a hardening/least-privilege NUDGE, not
|
|
139
|
+
// breakage. The count is re-derived where the advisory note is built.
|
|
125
140
|
{
|
|
126
141
|
n: r.frontmatterIssues.length,
|
|
127
142
|
weight: exports.W_NO_DESCRIPTION,
|
|
@@ -142,6 +157,33 @@ function reportDeductions(r) {
|
|
|
142
157
|
weight: exports.W_DANGLING_REF,
|
|
143
158
|
label: "mcp_tool hook(s) incomplete / targeting an undeclared server",
|
|
144
159
|
},
|
|
160
|
+
{
|
|
161
|
+
n: r.skillResourceIssues.length,
|
|
162
|
+
weight: exports.W_DANGLING_REF,
|
|
163
|
+
label: "skill bundled-resource ref(s) that don't resolve on disk",
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
n: r.skillFenceIssues.length,
|
|
167
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
168
|
+
label: "invisible skill(s) (frontmatter with no opening `---` fence)",
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
n: r.pluginLayoutIssues.length,
|
|
172
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
173
|
+
label: "functional dir(s) misplaced inside `.claude-plugin/` (invisible)",
|
|
174
|
+
},
|
|
175
|
+
{
|
|
176
|
+
n: r.hookBlockFindings.length,
|
|
177
|
+
weight: exports.W_MISSING_HOOK,
|
|
178
|
+
label: "hook(s) that look like they block but silently don't",
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
n: r.hookMatcherFindings.length,
|
|
182
|
+
weight: exports.W_MISSING_HOOK,
|
|
183
|
+
label: "hook matcher(s) that never fire (typo / wrong MCP form)",
|
|
184
|
+
},
|
|
185
|
+
// NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
|
|
186
|
+
// ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
|
|
145
187
|
// NB: untested surfaces are NOT a penalty — an untested surface is a hardening
|
|
146
188
|
// gap, not breakage, so it never drags the health score (it's appended as an
|
|
147
189
|
// advisory note below). The score ranks what's BROKEN.
|
|
@@ -192,8 +234,15 @@ function scoreReport(r) {
|
|
|
192
234
|
}
|
|
193
235
|
// Sort issues by cost (worst first) so the report leads with what matters.
|
|
194
236
|
issues.sort((a, b) => Number(b.split(" ")[0]) - Number(a.split(" ")[0]));
|
|
195
|
-
//
|
|
196
|
-
//
|
|
237
|
+
// Advisory notes are surfaced for visibility but DON'T affect the score, so they
|
|
238
|
+
// come AFTER the real (score-affecting) issues:
|
|
239
|
+
// - inherit-all (no tool contract): a least-privilege NUDGE, not breakage —
|
|
240
|
+
// see reportDeductions for the full rationale.
|
|
241
|
+
// - untested surfaces: a hardening gap, not breakage.
|
|
242
|
+
const noContract = r.agents.filter((a) => a.tools === null).length;
|
|
243
|
+
if (noContract > 0) {
|
|
244
|
+
issues.push(`${String(noContract)} agent(s) inherit all tools (no contract) (advisory)`);
|
|
245
|
+
}
|
|
197
246
|
if (r.untested > 0) {
|
|
198
247
|
issues.push(`${String(r.untested)} untested surface(s) (advisory)`);
|
|
199
248
|
}
|
|
@@ -228,12 +277,12 @@ function formatLeaderboard(scores) {
|
|
|
228
277
|
const issue = s.issues.length > 0 ? ` — ${s.issues.join("; ")}` : "";
|
|
229
278
|
out.push(` ${rank} ${score} ${s.grade} ${s.name}${issue}`);
|
|
230
279
|
});
|
|
231
|
-
out.push("", "Structural health only (no model). Weights: missing hook -15, no-description
|
|
280
|
+
out.push("", "Structural health only (no model). Weights: lethal-trifecta unit -20, missing", "hook -15, no-description skill -10, broken intra-plugin ref -8, dead tool/MCP", "ref -8. Inherit-all subagents and untested surfaces are advisory — shown, not", "scored.");
|
|
232
281
|
return out.join("\n");
|
|
233
282
|
}
|
|
234
|
-
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model):
|
|
235
|
-
"no-description skill −10, broken intra-plugin ref −8
|
|
236
|
-
"
|
|
283
|
+
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): lethal-trifecta unit −20, " +
|
|
284
|
+
"missing hook −15, no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
|
|
285
|
+
"Inherit-all subagents and untested surfaces are advisory (shown, not " +
|
|
237
286
|
"scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
|
|
238
287
|
/**
|
|
239
288
|
* Format a ranked leaderboard as a Markdown table — the PUBLISHABLE form (a README,
|
|
@@ -135,5 +135,90 @@ export declare function measurePluginSelectionWith(dir: string, promptSet: Trigg
|
|
|
135
135
|
export declare function measurePluginSelection(dir: string, promptSet: TriggerPromptSet, opts?: SelectionOptions): Promise<SelectionReport>;
|
|
136
136
|
/** Format the selection-collision matrix as a scan-report section. */
|
|
137
137
|
export declare function formatSelectionReport(r: SelectionReport): string;
|
|
138
|
+
/** Does a skill description assert a hard constraint (→ an adversarial-gate candidate)? */
|
|
139
|
+
export declare function isGateDescription(description: string): boolean;
|
|
140
|
+
/** A skill considered for gate detection — name + its (model-visible) description. */
|
|
141
|
+
export interface GateCandidate {
|
|
142
|
+
readonly name: string;
|
|
143
|
+
readonly description?: string;
|
|
144
|
+
readonly userInvoked?: boolean;
|
|
145
|
+
readonly hasDescription?: boolean;
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* The model-invocable, described skills whose description reads as an enforcement
|
|
149
|
+
* gate — the candidates for the adversarial-gate eval. User-invoked and
|
|
150
|
+
* description-less skills are excluded (they can't auto-fire a constraint on the
|
|
151
|
+
* model's behaviour), mirroring the trigger-rate candidate filter.
|
|
152
|
+
*/
|
|
153
|
+
export declare function detectGateSkills(skills: readonly GateCandidate[]): readonly string[];
|
|
154
|
+
/** A gate under test: its name + the rule its description states. */
|
|
155
|
+
export interface GateUnderTest {
|
|
156
|
+
readonly name: string;
|
|
157
|
+
readonly description: string;
|
|
158
|
+
}
|
|
159
|
+
/** The verdict an injected judge returns (a subset of judge.ts JudgeResult). */
|
|
160
|
+
export interface GateVerdict {
|
|
161
|
+
readonly pass: boolean;
|
|
162
|
+
readonly score: number;
|
|
163
|
+
readonly reason: string;
|
|
164
|
+
}
|
|
165
|
+
/** Injected dependencies, so the orchestration is unit-testable with no model. */
|
|
166
|
+
export interface GateEvalDeps {
|
|
167
|
+
readonly driver: EvalDriver;
|
|
168
|
+
/** Grade whether the gate held, given the run output + the rule rubric. */
|
|
169
|
+
readonly judge: (a: {
|
|
170
|
+
output: string;
|
|
171
|
+
rubric: string;
|
|
172
|
+
}) => GateVerdict;
|
|
173
|
+
/** Turn a gate's rule into a one-line user request that tries to violate it. */
|
|
174
|
+
readonly derive: (gate: GateUnderTest) => string;
|
|
175
|
+
}
|
|
176
|
+
export interface GateOptions {
|
|
177
|
+
/** Selector model for the harness run (default sonnet). */
|
|
178
|
+
readonly model?: string;
|
|
179
|
+
/** Attacks per gate (default 1). */
|
|
180
|
+
readonly trials?: number;
|
|
181
|
+
/** Concurrent harness runs (default 1). */
|
|
182
|
+
readonly concurrency?: number;
|
|
183
|
+
readonly harness?: ProbeHarness;
|
|
184
|
+
readonly layout?: PluginLayout;
|
|
185
|
+
readonly dialect?: HarnessDialect;
|
|
186
|
+
/** Author-supplied attack prompts (bare skill name → prompts); overrides derive. */
|
|
187
|
+
readonly attacks?: Record<string, readonly string[]>;
|
|
188
|
+
}
|
|
189
|
+
export interface GateResult {
|
|
190
|
+
readonly skill: string;
|
|
191
|
+
readonly measured: boolean;
|
|
192
|
+
/** Fraction of attacks the gate HELD (1 = held every time). */
|
|
193
|
+
readonly heldRate?: number;
|
|
194
|
+
/** Convenience: held on EVERY attack (a single cave → false). */
|
|
195
|
+
readonly held?: boolean;
|
|
196
|
+
readonly n?: number;
|
|
197
|
+
/** The attack used (first), for the report. */
|
|
198
|
+
readonly attack?: string;
|
|
199
|
+
/** The judge's rationale on a representative cave (else a hold), for the report. */
|
|
200
|
+
readonly reason?: string;
|
|
201
|
+
readonly note?: string;
|
|
202
|
+
}
|
|
203
|
+
export interface GateAdversarialReport {
|
|
204
|
+
readonly available: boolean;
|
|
205
|
+
readonly results: readonly GateResult[];
|
|
206
|
+
readonly note?: string;
|
|
207
|
+
}
|
|
208
|
+
/** Build the LLM-judge rubric from the gate's own rule (pure). */
|
|
209
|
+
export declare function gateRubric(gate: GateUnderTest): string;
|
|
210
|
+
/** The injectable core (unit-testable with fake driver/judge/derive — no model). */
|
|
211
|
+
export declare function measureGateAdversarialWith(dir: string, gates: readonly GateUnderTest[], deps: GateEvalDeps, opts?: GateOptions): Promise<GateAdversarialReport>;
|
|
212
|
+
/**
|
|
213
|
+
* Measure whether a plugin's enforcement-gate skills HOLD when adversarially
|
|
214
|
+
* challenged. Detects gate skills (keyword heuristic), auto-derives an attack from
|
|
215
|
+
* each rule (unless author-supplied), runs the UNSTUBBED harness, and LLM-judges
|
|
216
|
+
* hold vs cave. Claude Code only; degrades to `available: false` without the CLI/auth.
|
|
217
|
+
*/
|
|
218
|
+
export declare function measureGateAdversarial(dir: string, opts?: GateOptions): Promise<GateAdversarialReport>;
|
|
219
|
+
/** Default adversarial attacks per gate (stochastic → need >1; unstubbed → keep low). */
|
|
220
|
+
export declare const DEFAULT_GATE_TRIALS = 3;
|
|
221
|
+
/** Format the adversarial-gate report as a scan-report section. */
|
|
222
|
+
export declare function formatGateReport(r: GateAdversarialReport): string;
|
|
138
223
|
export {};
|
|
139
224
|
//# sourceMappingURL=scan-behavioral.d.ts.map
|