vigiles 14.7.0 → 14.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/agent-runtime.d.ts +2 -19
- package/dist/adapters/claude-code/agent-runtime.js +5 -30
- package/dist/adapters/claude-code/agent-tools.d.ts +20 -0
- package/dist/adapters/claude-code/agent-tools.js +40 -0
- package/dist/audit-report.d.ts +1 -1
- package/dist/audit-report.template.html +37 -32
- package/dist/audit-score.d.ts +1 -1
- package/dist/audit-score.js +20 -20
- package/dist/audit-verdict.d.ts +1 -1
- package/dist/audit-verdict.js +3 -3
- package/dist/core/assert-never.d.ts +9 -0
- package/dist/core/assert-never.js +14 -0
- package/dist/core/description-overlap.js +2 -2
- package/dist/core/effects.js +3 -3
- package/dist/core/hash.d.ts +1 -2
- package/dist/core/hash.js +6 -4
- package/dist/core/hook-block-ineffective.d.ts +55 -6
- package/dist/core/hook-block-ineffective.js +9 -14
- package/dist/core/mcp-contract-message.d.ts +22 -0
- package/dist/core/mcp-contract-message.js +29 -0
- package/dist/core/mcp.d.ts +4 -12
- package/dist/core/mcp.js +3 -14
- package/dist/core/ncd.d.ts +12 -0
- package/dist/core/ncd.js +50 -0
- package/dist/core/plugin-dir-layout.d.ts +5 -5
- package/dist/core/plugin-dir-layout.js +10 -22
- package/dist/core/proofs.d.ts +2 -11
- package/dist/core/proofs.js +4 -39
- package/dist/core/skill-resources.d.ts +3 -3
- package/dist/core/skill-resources.js +9 -8
- package/dist/leaderboard.d.ts +2 -51
- package/dist/leaderboard.js +20 -225
- package/dist/optimize.d.ts +1 -1
- package/dist/optimize.js +3 -3
- package/dist/posix-path.d.ts +40 -0
- package/dist/posix-path.js +293 -0
- package/dist/scan-core.d.ts +154 -0
- package/dist/scan-core.js +690 -0
- package/dist/scan-files.d.ts +28 -0
- package/dist/scan-files.js +489 -0
- package/dist/scan.d.ts +11 -34
- package/dist/scan.js +55 -668
- package/dist/score-core.d.ts +83 -0
- package/dist/score-core.js +236 -0
- package/dist/test-coverage-files.d.ts +11 -0
- package/dist/test-coverage-files.js +208 -0
- package/package.json +1 -1
|
@@ -12,10 +12,11 @@ exports.pluginDirLayoutIssues = pluginDirLayoutIssues;
|
|
|
12
12
|
* completely invisible to the harness. Only `plugin.json` belongs inside the
|
|
13
13
|
* manifest directory; everything else must live at the root.
|
|
14
14
|
*
|
|
15
|
-
* Pure + FP-safe: the only IO is
|
|
16
|
-
* (
|
|
17
|
-
*
|
|
18
|
-
* filesystem in tests
|
|
15
|
+
* Pure + FP-safe + node-free: the only IO is a REQUIRED, injected `existsSync`
|
|
16
|
+
* and `isDirectory` (the disk caller passes `node:fs`; the browser engine passes
|
|
17
|
+
* a map-backed pair), so the detector is fully testable with fakes, never touches
|
|
18
|
+
* the filesystem in tests, and statically imports no `node:` builtin — it bundles
|
|
19
|
+
* clean in a browser (path ops come from the node-free `posix-path`).
|
|
19
20
|
*
|
|
20
21
|
* Harness-agnostic: the surface directory names are INJECTED from the layout
|
|
21
22
|
* (PluginLayout.skillDir / agentDir / commandDir / hookDir, or equivalent), never
|
|
@@ -23,23 +24,10 @@ exports.pluginDirLayoutIssues = pluginDirLayoutIssues;
|
|
|
23
24
|
* `plugin-dir-layout` rule) and `vigiles audit` (the read-only report) — one
|
|
24
25
|
* detector, no drift.
|
|
25
26
|
*/
|
|
26
|
-
const
|
|
27
|
-
const node_path_1 = require("node:path");
|
|
27
|
+
const posix_path_js_1 = require("../posix-path.js");
|
|
28
28
|
// ---------------------------------------------------------------------------
|
|
29
29
|
// Detector
|
|
30
30
|
// ---------------------------------------------------------------------------
|
|
31
|
-
/**
|
|
32
|
-
* Default `isDirectory` implementation — wraps statSync so that a missing or
|
|
33
|
-
* unreadable path returns `false` instead of throwing.
|
|
34
|
-
*/
|
|
35
|
-
function defaultIsDirectory(p) {
|
|
36
|
-
try {
|
|
37
|
-
return (0, node_fs_1.statSync)(p).isDirectory();
|
|
38
|
-
}
|
|
39
|
-
catch {
|
|
40
|
-
return false;
|
|
41
|
-
}
|
|
42
|
-
}
|
|
43
31
|
/**
|
|
44
32
|
* Surface directories found nested INSIDE the manifest directory, where they are
|
|
45
33
|
* invisible to the harness.
|
|
@@ -53,12 +41,12 @@ function defaultIsDirectory(p) {
|
|
|
53
41
|
* dirs.
|
|
54
42
|
*/
|
|
55
43
|
function pluginDirLayoutIssues(manifestDir, surfaceDirNames, opts) {
|
|
56
|
-
const exists = opts
|
|
57
|
-
const isDir = opts
|
|
58
|
-
const manifestBase = (0,
|
|
44
|
+
const exists = opts.existsSync;
|
|
45
|
+
const isDir = opts.isDirectory;
|
|
46
|
+
const manifestBase = (0, posix_path_js_1.basename)(manifestDir);
|
|
59
47
|
const findings = [];
|
|
60
48
|
for (const name of surfaceDirNames) {
|
|
61
|
-
const candidate = (0,
|
|
49
|
+
const candidate = (0, posix_path_js_1.join)(manifestDir, name);
|
|
62
50
|
if (exists(candidate) && isDir(candidate)) {
|
|
63
51
|
findings.push({
|
|
64
52
|
dir: name,
|
package/dist/core/proofs.d.ts
CHANGED
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
* 5. MerkleHistory — tamper-evident spec evolution audit trail
|
|
11
11
|
* 6. propertyTest() — random mutation + invariant checking
|
|
12
12
|
*/
|
|
13
|
+
import { ncd } from "./ncd.js";
|
|
13
14
|
import type { Rule, ClaudeSpec } from "./spec.js";
|
|
14
15
|
export interface MonotonicityViolation {
|
|
15
16
|
ruleId: string;
|
|
@@ -49,17 +50,7 @@ export declare function latticeJoin(a: Rule["_kind"], b: Rule["_kind"]): Rule["_
|
|
|
49
50
|
export declare function latticeMeet(a: Rule["_kind"], b: Rule["_kind"]): Rule["_kind"];
|
|
50
51
|
/** Get the numeric strength of a rule kind. */
|
|
51
52
|
export declare function ruleStrength(kind: Rule["_kind"]): number;
|
|
52
|
-
|
|
53
|
-
* Normalized Compression Distance — information-theoretic similarity.
|
|
54
|
-
*
|
|
55
|
-
* NCD(x, y) = (C(xy) - min(C(x), C(y))) / max(C(x), C(y))
|
|
56
|
-
*
|
|
57
|
-
* Range: [0, 1+ε] where 0 = identical information content.
|
|
58
|
-
* Deterministic. No model dependency. Approximates the universal distance metric.
|
|
59
|
-
*
|
|
60
|
-
* Reference: Li, Chen, Li, Ma, Vitányi (2004) "The Similarity Metric"
|
|
61
|
-
*/
|
|
62
|
-
export declare function ncd(a: string, b: string): number;
|
|
53
|
+
export { ncd };
|
|
63
54
|
export interface NCDPair {
|
|
64
55
|
idA: string;
|
|
65
56
|
idB: string;
|
package/dist/core/proofs.js
CHANGED
|
@@ -12,19 +12,19 @@
|
|
|
12
12
|
* 6. propertyTest() — random mutation + invariant checking
|
|
13
13
|
*/
|
|
14
14
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
15
|
-
exports.MerkleHistory = exports.BloomFilter = void 0;
|
|
15
|
+
exports.MerkleHistory = exports.BloomFilter = exports.ncd = void 0;
|
|
16
16
|
exports.checkMonotonicity = checkMonotonicity;
|
|
17
17
|
exports.latticeJoin = latticeJoin;
|
|
18
18
|
exports.latticeMeet = latticeMeet;
|
|
19
19
|
exports.ruleStrength = ruleStrength;
|
|
20
|
-
exports.ncd = ncd;
|
|
21
20
|
exports.findSimilarRules = findSimilarRules;
|
|
22
21
|
exports.ruleToBloomFilter = ruleToBloomFilter;
|
|
23
22
|
exports.fixedPoint = fixedPoint;
|
|
24
23
|
exports.propertyTest = propertyTest;
|
|
25
24
|
exports.fitness = fitness;
|
|
26
|
-
const node_zlib_1 = require("node:zlib");
|
|
27
25
|
const hash_js_1 = require("./hash.js");
|
|
26
|
+
const ncd_js_1 = require("./ncd.js");
|
|
27
|
+
Object.defineProperty(exports, "ncd", { enumerable: true, get: function () { return ncd_js_1.ncd; } });
|
|
28
28
|
// ---------------------------------------------------------------------------
|
|
29
29
|
// 1. Monotonicity Lattice — partial order on rule strength
|
|
30
30
|
// ---------------------------------------------------------------------------
|
|
@@ -127,41 +127,6 @@ function latticeMeet(a, b) {
|
|
|
127
127
|
function ruleStrength(kind) {
|
|
128
128
|
return STRENGTH[kind];
|
|
129
129
|
}
|
|
130
|
-
// ---------------------------------------------------------------------------
|
|
131
|
-
// 2. Normalized Compression Distance (NCD)
|
|
132
|
-
// ---------------------------------------------------------------------------
|
|
133
|
-
/**
|
|
134
|
-
* Compute the compressed size of a string using gzip.
|
|
135
|
-
* This approximates Kolmogorov complexity — the length of the shortest
|
|
136
|
-
* program that produces the string.
|
|
137
|
-
*/
|
|
138
|
-
function compressedSize(s) {
|
|
139
|
-
return (0, node_zlib_1.gzipSync)(Buffer.from(s, "utf-8"), { level: 9 }).length;
|
|
140
|
-
}
|
|
141
|
-
/**
|
|
142
|
-
* Normalized Compression Distance — information-theoretic similarity.
|
|
143
|
-
*
|
|
144
|
-
* NCD(x, y) = (C(xy) - min(C(x), C(y))) / max(C(x), C(y))
|
|
145
|
-
*
|
|
146
|
-
* Range: [0, 1+ε] where 0 = identical information content.
|
|
147
|
-
* Deterministic. No model dependency. Approximates the universal distance metric.
|
|
148
|
-
*
|
|
149
|
-
* Reference: Li, Chen, Li, Ma, Vitányi (2004) "The Similarity Metric"
|
|
150
|
-
*/
|
|
151
|
-
function ncd(a, b) {
|
|
152
|
-
if (a === b)
|
|
153
|
-
return 0;
|
|
154
|
-
if (a.length === 0 && b.length === 0)
|
|
155
|
-
return 0;
|
|
156
|
-
const ca = compressedSize(a);
|
|
157
|
-
const cb = compressedSize(b);
|
|
158
|
-
const cab = compressedSize(a + b);
|
|
159
|
-
const minC = Math.min(ca, cb);
|
|
160
|
-
const maxC = Math.max(ca, cb);
|
|
161
|
-
if (maxC === 0)
|
|
162
|
-
return 0;
|
|
163
|
-
return (cab - minC) / maxC;
|
|
164
|
-
}
|
|
165
130
|
/**
|
|
166
131
|
* Find all rule pairs with NCD below a similarity threshold.
|
|
167
132
|
* Returns pairs sorted by distance (most similar first).
|
|
@@ -175,7 +140,7 @@ function findSimilarRules(rules, threshold = 0.5) {
|
|
|
175
140
|
const [idB, ruleB] = entries[j];
|
|
176
141
|
const textA = ruleToText(ruleA);
|
|
177
142
|
const textB = ruleToText(ruleB);
|
|
178
|
-
const d = ncd(textA, textB);
|
|
143
|
+
const d = (0, ncd_js_1.ncd)(textA, textB);
|
|
179
144
|
if (d < threshold) {
|
|
180
145
|
pairs.push({ idA, idB, distance: d });
|
|
181
146
|
}
|
|
@@ -12,8 +12,8 @@ export interface SkillResourceFinding {
|
|
|
12
12
|
readonly line: number;
|
|
13
13
|
}
|
|
14
14
|
export interface SkillResourceOptions {
|
|
15
|
-
/**
|
|
16
|
-
readonly existsSync
|
|
15
|
+
/** REQUIRED, injected existence check (disk: node:fs existsSync). */
|
|
16
|
+
readonly existsSync: (p: string) => boolean;
|
|
17
17
|
/**
|
|
18
18
|
* Repo root, used only together with `sharedDirs` (below). Off by default.
|
|
19
19
|
*/
|
|
@@ -39,5 +39,5 @@ export interface SkillResourceOptions {
|
|
|
39
39
|
* The shared detector behind both `vigiles lint` (the `skill-resource-resolves`
|
|
40
40
|
* rule) and `vigiles audit` (the read-only report) — one detector, no drift.
|
|
41
41
|
*/
|
|
42
|
-
export declare function skillResourceIssues(skillBody: string, skillDir: string, opts
|
|
42
|
+
export declare function skillResourceIssues(skillBody: string, skillDir: string, opts: SkillResourceOptions): SkillResourceFinding[];
|
|
43
43
|
//# sourceMappingURL=skill-resources.d.ts.map
|
|
@@ -34,11 +34,12 @@ exports.skillResourceIssues = skillResourceIssues;
|
|
|
34
34
|
* (example / e.g. / such as / would be / template / →). Markdown links are
|
|
35
35
|
* unchanged — a link is already an act-on-it reference. See `inlinePathIsUsed`.
|
|
36
36
|
*
|
|
37
|
-
* Pure: the only IO is
|
|
38
|
-
*
|
|
37
|
+
* Pure + node-free: the only IO is a REQUIRED, injected `existsSync` (the disk
|
|
38
|
+
* caller passes `node:fs`; the browser engine a map-backed check), so the
|
|
39
|
+
* detector is testable with a fake and statically imports no `node:` builtin —
|
|
40
|
+
* it bundles clean in a browser (path ops come from the node-free `posix-path`).
|
|
39
41
|
*/
|
|
40
|
-
const
|
|
41
|
-
const node_path_1 = require("node:path");
|
|
42
|
+
const posix_path_js_1 = require("../posix-path.js");
|
|
42
43
|
// ---------------------------------------------------------------------------
|
|
43
44
|
// Shapes we match vs deliberately skip (FP-safety)
|
|
44
45
|
// ---------------------------------------------------------------------------
|
|
@@ -202,19 +203,19 @@ function candidatesInLine(line, lineNo) {
|
|
|
202
203
|
* The shared detector behind both `vigiles lint` (the `skill-resource-resolves`
|
|
203
204
|
* rule) and `vigiles audit` (the read-only report) — one detector, no drift.
|
|
204
205
|
*/
|
|
205
|
-
function skillResourceIssues(skillBody, skillDir, opts
|
|
206
|
-
const exists = opts.existsSync
|
|
206
|
+
function skillResourceIssues(skillBody, skillDir, opts) {
|
|
207
|
+
const exists = opts.existsSync;
|
|
207
208
|
const sharedDirs = new Set(opts.sharedDirs ?? []);
|
|
208
209
|
// A ref resolves if it exists under the skill's own dir. If (and only if) its
|
|
209
210
|
// first segment is a DECLARED shared dir, it may also resolve against the repo
|
|
210
211
|
// root — the opt-in shared-tree case. No shared dirs → skill-dir-only (unchanged).
|
|
211
212
|
const resolvesAnywhere = (rel) => {
|
|
212
|
-
if (exists((0,
|
|
213
|
+
if (exists((0, posix_path_js_1.resolve)(skillDir, rel)))
|
|
213
214
|
return true;
|
|
214
215
|
const firstSeg = rel.split("/")[0];
|
|
215
216
|
return (opts.repoRoot !== undefined &&
|
|
216
217
|
sharedDirs.has(firstSeg) &&
|
|
217
|
-
exists((0,
|
|
218
|
+
exists((0, posix_path_js_1.resolve)(opts.repoRoot, rel)));
|
|
218
219
|
};
|
|
219
220
|
const findings = [];
|
|
220
221
|
const seen = new Set();
|
package/dist/leaderboard.d.ts
CHANGED
|
@@ -10,57 +10,8 @@
|
|
|
10
10
|
* The behavioural columns (real trigger-rate, observed egress, safety) need a
|
|
11
11
|
* model and stack on top later; this part runs anywhere in CI for free.
|
|
12
12
|
*/
|
|
13
|
-
import { type
|
|
14
|
-
export
|
|
15
|
-
readonly dir: string;
|
|
16
|
-
readonly name: string;
|
|
17
|
-
/** 0–100 structural-health score (100 = no structural issues found). */
|
|
18
|
-
readonly score: number;
|
|
19
|
-
readonly grade: "A" | "B" | "C" | "D" | "F";
|
|
20
|
-
/** Human-readable deductions, worst first. */
|
|
21
|
-
readonly issues: readonly string[];
|
|
22
|
-
readonly report: ScanReport;
|
|
23
|
-
}
|
|
24
|
-
export declare const W_MISSING_HOOK = 15;
|
|
25
|
-
export declare const W_NO_DESCRIPTION = 10;
|
|
26
|
-
export declare const W_DANGLING_REF = 8;
|
|
27
|
-
export declare const W_OVERLAP = 8;
|
|
28
|
-
export declare const W_NO_CONTRACT = 5;
|
|
29
|
-
export declare const W_TRIFECTA = 10;
|
|
30
|
-
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
31
|
-
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
32
|
-
/** One deduction: a count, its per-item weight, and the label if non-zero. */
|
|
33
|
-
export interface Deduction {
|
|
34
|
-
readonly n: number;
|
|
35
|
-
readonly weight: number;
|
|
36
|
-
readonly label: string;
|
|
37
|
-
}
|
|
38
|
-
/**
|
|
39
|
-
* The COMPLETE graded-penalty list a report incurs — the single source of truth
|
|
40
|
-
* BOTH the leaderboard's single health number and the audit's category rings
|
|
41
|
-
* read, so the overall can never drift between the two surfaces. Each entry is a
|
|
42
|
-
* graded penalty; untested surfaces are deliberately ABSENT (they're advisory,
|
|
43
|
-
* surfaced separately, never scored).
|
|
44
|
-
*/
|
|
45
|
-
export declare function reportDeductions(r: ScanReport): Deduction[];
|
|
46
|
-
/** True when a report has no loadable plugin surface at all (the empty machine). */
|
|
47
|
-
export declare function isEmptyMachine(r: ScanReport): boolean;
|
|
48
|
-
/**
|
|
49
|
-
* THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
|
|
50
|
-
* [0,100]. Both the leaderboard's single health number AND the audit's headline
|
|
51
|
-
* overall read this, so the two can never disagree (the summed model is the
|
|
52
|
-
* honest one — averaging rings would dilute a real problem). Returns the score
|
|
53
|
-
* plus the per-item deductions so callers render their own issue/finding lists.
|
|
54
|
-
*/
|
|
55
|
-
export declare function computeIntegrityScore(deductions: readonly Deduction[]): {
|
|
56
|
-
score: number;
|
|
57
|
-
penalty: number;
|
|
58
|
-
};
|
|
59
|
-
/** Deterministic structural-health score for one scanned plugin. */
|
|
60
|
-
export declare function scoreReport(r: ScanReport): {
|
|
61
|
-
score: number;
|
|
62
|
-
issues: string[];
|
|
63
|
-
};
|
|
13
|
+
import { type PluginScore } from "./score-core.js";
|
|
14
|
+
export { type PluginScore, type Deduction, W_MISSING_HOOK, W_NO_DESCRIPTION, W_DANGLING_REF, W_OVERLAP, W_NO_CONTRACT, W_TRIFECTA, gradeFor, reportDeductions, isEmptyMachine, computeIntegrityScore, scoreReport, } from "./score-core.js";
|
|
64
15
|
/** Scan + score each directory, ranked best-first (ties broken by name). */
|
|
65
16
|
export declare function rankPlugins(dirs: readonly string[]): PluginScore[];
|
|
66
17
|
/** Format a ranked leaderboard as human-readable text. */
|
package/dist/leaderboard.js
CHANGED
|
@@ -12,18 +12,30 @@
|
|
|
12
12
|
* model and stack on top later; this part runs anywhere in CI for free.
|
|
13
13
|
*/
|
|
14
14
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
15
|
-
exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
16
|
-
exports.gradeFor = gradeFor;
|
|
17
|
-
exports.reportDeductions = reportDeductions;
|
|
18
|
-
exports.isEmptyMachine = isEmptyMachine;
|
|
19
|
-
exports.computeIntegrityScore = computeIntegrityScore;
|
|
20
|
-
exports.scoreReport = scoreReport;
|
|
15
|
+
exports.scoreReport = exports.computeIntegrityScore = exports.isEmptyMachine = exports.reportDeductions = exports.gradeFor = exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
21
16
|
exports.rankPlugins = rankPlugins;
|
|
22
17
|
exports.formatLeaderboard = formatLeaderboard;
|
|
23
18
|
exports.formatLeaderboardMarkdown = formatLeaderboardMarkdown;
|
|
24
19
|
const node_fs_1 = require("node:fs");
|
|
25
20
|
const node_path_1 = require("node:path");
|
|
26
21
|
const scan_js_1 = require("./scan.js");
|
|
22
|
+
// The pure, node-free scoring core lives in ./score-core.js (extracted so the
|
|
23
|
+
// in-browser audit's report builder can import scoring WITHOUT this module's
|
|
24
|
+
// node-only scanPlugin/pluginLabel → scan.ts → @ast-grep/napi chain). Re-exported
|
|
25
|
+
// here so every existing `from "./leaderboard.js"` consumer keeps working.
|
|
26
|
+
const score_core_js_1 = require("./score-core.js");
|
|
27
|
+
var score_core_js_2 = require("./score-core.js");
|
|
28
|
+
Object.defineProperty(exports, "W_MISSING_HOOK", { enumerable: true, get: function () { return score_core_js_2.W_MISSING_HOOK; } });
|
|
29
|
+
Object.defineProperty(exports, "W_NO_DESCRIPTION", { enumerable: true, get: function () { return score_core_js_2.W_NO_DESCRIPTION; } });
|
|
30
|
+
Object.defineProperty(exports, "W_DANGLING_REF", { enumerable: true, get: function () { return score_core_js_2.W_DANGLING_REF; } });
|
|
31
|
+
Object.defineProperty(exports, "W_OVERLAP", { enumerable: true, get: function () { return score_core_js_2.W_OVERLAP; } });
|
|
32
|
+
Object.defineProperty(exports, "W_NO_CONTRACT", { enumerable: true, get: function () { return score_core_js_2.W_NO_CONTRACT; } });
|
|
33
|
+
Object.defineProperty(exports, "W_TRIFECTA", { enumerable: true, get: function () { return score_core_js_2.W_TRIFECTA; } });
|
|
34
|
+
Object.defineProperty(exports, "gradeFor", { enumerable: true, get: function () { return score_core_js_2.gradeFor; } });
|
|
35
|
+
Object.defineProperty(exports, "reportDeductions", { enumerable: true, get: function () { return score_core_js_2.reportDeductions; } });
|
|
36
|
+
Object.defineProperty(exports, "isEmptyMachine", { enumerable: true, get: function () { return score_core_js_2.isEmptyMachine; } });
|
|
37
|
+
Object.defineProperty(exports, "computeIntegrityScore", { enumerable: true, get: function () { return score_core_js_2.computeIntegrityScore; } });
|
|
38
|
+
Object.defineProperty(exports, "scoreReport", { enumerable: true, get: function () { return score_core_js_2.scoreReport; } });
|
|
27
39
|
/** The plugin's declared name (`.claude-plugin/plugin.json`), for a real label in
|
|
28
40
|
* the ranking instead of a SHA-pinned dir basename. Falls back to the basename. */
|
|
29
41
|
function pluginLabel(dir) {
|
|
@@ -41,233 +53,16 @@ function pluginLabel(dir) {
|
|
|
41
53
|
}
|
|
42
54
|
return (0, node_path_1.basename)(dir) || dir;
|
|
43
55
|
}
|
|
44
|
-
// Penalty weights — broken-at-runtime costs most, footguns less, nudges least.
|
|
45
|
-
// Exported so the category view (audit-score.ts) reuses the SAME weights and the
|
|
46
|
-
// two surfaces can never drift on a per-item cost.
|
|
47
|
-
exports.W_MISSING_HOOK = 15; // a hook script that doesn't exist → never runs
|
|
48
|
-
exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't trigger
|
|
49
|
-
exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
50
|
-
exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
|
|
51
|
-
exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
|
|
52
|
-
exports.W_TRIFECTA = 10; // a HARD lethal-trifecta contract (all three legs, explicit) → a prompt-injection exfil path. HALF the old 20: a DING, not a fail — a trifecta is a real risk worth surfacing in the grade, but official plugins ship the pattern by design, so it dents the score (e.g. feature-dev's 3 hard units → −30 → C) without a catastrophic F.
|
|
53
|
-
// Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
|
|
54
|
-
// - untested surfaces — a hardening gap, not breakage.
|
|
55
|
-
// - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
|
|
56
|
-
// - an inherits-all (severity "advisory") trifecta finding — shown by the Safety
|
|
57
|
-
// ring but never scored; only the HARD, explicit all-three-legs finding grades.
|
|
58
|
-
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
59
|
-
function gradeFor(score) {
|
|
60
|
-
if (score >= 90)
|
|
61
|
-
return "A";
|
|
62
|
-
if (score >= 80)
|
|
63
|
-
return "B";
|
|
64
|
-
if (score >= 70)
|
|
65
|
-
return "C";
|
|
66
|
-
if (score >= 60)
|
|
67
|
-
return "D";
|
|
68
|
-
return "F";
|
|
69
|
-
}
|
|
70
|
-
/**
|
|
71
|
-
* The COMPLETE graded-penalty list a report incurs — the single source of truth
|
|
72
|
-
* BOTH the leaderboard's single health number and the audit's category rings
|
|
73
|
-
* read, so the overall can never drift between the two surfaces. Each entry is a
|
|
74
|
-
* graded penalty; untested surfaces are deliberately ABSENT (they're advisory,
|
|
75
|
-
* surfaced separately, never scored).
|
|
76
|
-
*/
|
|
77
|
-
function reportDeductions(r) {
|
|
78
|
-
const missingHooks = r.hooks.filter((h) => h.status === "missing").length;
|
|
79
|
-
const noDesc = r.skills.filter((s) => !s.hasDescription).length;
|
|
80
|
-
const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
|
|
81
|
-
const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
|
|
82
|
-
const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
|
|
83
|
-
// HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
|
|
84
|
-
// legs (a prompt-injection exfil path). Graded at W_TRIFECTA=10 (HALF the old
|
|
85
|
-
// 20): a DING that surfaces a real risk in the grade without a catastrophic F
|
|
86
|
-
// for an accepted design pattern official plugins ship. Advisory (inherits-all)
|
|
87
|
-
// trifecta findings are surfaced but NEVER graded (aligned with the inherits-all
|
|
88
|
-
// stance), so they're excluded here.
|
|
89
|
-
const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
|
|
90
|
-
return [
|
|
91
|
-
{
|
|
92
|
-
n: hardTrifecta,
|
|
93
|
-
weight: exports.W_TRIFECTA,
|
|
94
|
-
label: "unit(s) holding all three lethal-trifecta legs (prompt-injection exfil path)",
|
|
95
|
-
},
|
|
96
|
-
{
|
|
97
|
-
n: missingHooks,
|
|
98
|
-
weight: exports.W_MISSING_HOOK,
|
|
99
|
-
label: "hook script(s) MISSING",
|
|
100
|
-
},
|
|
101
|
-
{
|
|
102
|
-
n: r.hookEventIssues.length,
|
|
103
|
-
weight: exports.W_MISSING_HOOK,
|
|
104
|
-
label: "hook(s) on an unknown event (never fire)",
|
|
105
|
-
},
|
|
106
|
-
{
|
|
107
|
-
n: noDesc,
|
|
108
|
-
weight: exports.W_NO_DESCRIPTION,
|
|
109
|
-
label: "skill(s) with no usable description",
|
|
110
|
-
},
|
|
111
|
-
{
|
|
112
|
-
n: r.descriptionOverlaps.length,
|
|
113
|
-
weight: exports.W_OVERLAP,
|
|
114
|
-
label: "near-identical skill description(s) (wrong one fires)",
|
|
115
|
-
},
|
|
116
|
-
{
|
|
117
|
-
n: r.danglingRefs.length,
|
|
118
|
-
weight: exports.W_DANGLING_REF,
|
|
119
|
-
label: "broken intra-plugin reference(s)",
|
|
120
|
-
},
|
|
121
|
-
{
|
|
122
|
-
n: deadTools,
|
|
123
|
-
weight: exports.W_DANGLING_REF,
|
|
124
|
-
label: "unavailable agent tool(s) (typo / never-available)",
|
|
125
|
-
},
|
|
126
|
-
{
|
|
127
|
-
n: deadMcpTools,
|
|
128
|
-
weight: exports.W_DANGLING_REF,
|
|
129
|
-
label: "agent MCP tool(s) whose server isn't declared (can't resolve)",
|
|
130
|
-
},
|
|
131
|
-
{
|
|
132
|
-
n: deadDisallowed,
|
|
133
|
-
weight: exports.W_NO_CONTRACT,
|
|
134
|
-
label: "agent disallowedTools typo(s) that block nothing",
|
|
135
|
-
},
|
|
136
|
-
// NB: an agent that inherits all tools (no `tools:` line) is ADVISORY, not a
|
|
137
|
-
// graded penalty — it's surfaced by scoreReport / the Structure ring but never
|
|
138
|
-
// drags the score. WHY: omitting the `tools:` line is a near-universal,
|
|
139
|
-
// legitimate authoring style (a measured OSS sweep of 122 real plugins found
|
|
140
|
-
// 109 whose ONLY finding was this), so penalizing it makes the grade cry wolf
|
|
141
|
-
// on idiomatic subagents. A health score should mean "something is BROKEN", and
|
|
142
|
-
// a broad-by-default tool surface is a hardening/least-privilege NUDGE, not
|
|
143
|
-
// breakage. The count is re-derived where the advisory note is built.
|
|
144
|
-
{
|
|
145
|
-
n: r.frontmatterIssues.length,
|
|
146
|
-
weight: exports.W_NO_DESCRIPTION,
|
|
147
|
-
label: "surface(s) missing required frontmatter (name/description)",
|
|
148
|
-
},
|
|
149
|
-
{
|
|
150
|
-
n: r.frontmatterValueIssues.length,
|
|
151
|
-
weight: exports.W_NO_CONTRACT,
|
|
152
|
-
label: "agent(s) with an invalid model/color (typo → silent fallback)",
|
|
153
|
-
},
|
|
154
|
-
{
|
|
155
|
-
n: r.mcpIssues.length,
|
|
156
|
-
weight: exports.W_DANGLING_REF,
|
|
157
|
-
label: "MCP server(s) that can't start (no command/url)",
|
|
158
|
-
},
|
|
159
|
-
{
|
|
160
|
-
n: r.mcpHookIssues.length,
|
|
161
|
-
weight: exports.W_DANGLING_REF,
|
|
162
|
-
label: "mcp_tool hook(s) incomplete / targeting an undeclared server",
|
|
163
|
-
},
|
|
164
|
-
{
|
|
165
|
-
n: r.skillResourceIssues.length,
|
|
166
|
-
weight: exports.W_DANGLING_REF,
|
|
167
|
-
label: "skill bundled-resource ref(s) that don't resolve on disk",
|
|
168
|
-
},
|
|
169
|
-
{
|
|
170
|
-
n: r.skillFenceIssues.length,
|
|
171
|
-
weight: exports.W_NO_DESCRIPTION,
|
|
172
|
-
label: "invisible skill(s) (frontmatter with no opening `---` fence)",
|
|
173
|
-
},
|
|
174
|
-
{
|
|
175
|
-
n: r.pluginLayoutIssues.length,
|
|
176
|
-
weight: exports.W_NO_DESCRIPTION,
|
|
177
|
-
label: "functional dir(s) misplaced inside `.claude-plugin/` (invisible)",
|
|
178
|
-
},
|
|
179
|
-
{
|
|
180
|
-
n: r.hookBlockFindings.length,
|
|
181
|
-
weight: exports.W_MISSING_HOOK,
|
|
182
|
-
label: "hook(s) that look like they block but silently don't",
|
|
183
|
-
},
|
|
184
|
-
{
|
|
185
|
-
n: r.hookMatcherFindings.length,
|
|
186
|
-
weight: exports.W_MISSING_HOOK,
|
|
187
|
-
label: "hook matcher(s) that never fire (typo / wrong MCP form)",
|
|
188
|
-
},
|
|
189
|
-
// NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
|
|
190
|
-
// ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
|
|
191
|
-
// NB: untested surfaces are NOT a penalty — an untested surface is a hardening
|
|
192
|
-
// gap, not breakage, so it never drags the health score (it's appended as an
|
|
193
|
-
// advisory note below). The score ranks what's BROKEN.
|
|
194
|
-
];
|
|
195
|
-
}
|
|
196
|
-
/** True when a report has no loadable plugin surface at all (the empty machine). */
|
|
197
|
-
function isEmptyMachine(r) {
|
|
198
|
-
const surfaces = r.skills.length +
|
|
199
|
-
r.agents.length +
|
|
200
|
-
r.hooks.length +
|
|
201
|
-
r.inlineHooks +
|
|
202
|
-
r.commands;
|
|
203
|
-
return surfaces === 0 && !r.mcp;
|
|
204
|
-
}
|
|
205
|
-
/**
|
|
206
|
-
* THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
|
|
207
|
-
* [0,100]. Both the leaderboard's single health number AND the audit's headline
|
|
208
|
-
* overall read this, so the two can never disagree (the summed model is the
|
|
209
|
-
* honest one — averaging rings would dilute a real problem). Returns the score
|
|
210
|
-
* plus the per-item deductions so callers render their own issue/finding lists.
|
|
211
|
-
*/
|
|
212
|
-
function computeIntegrityScore(deductions) {
|
|
213
|
-
let penalty = 0;
|
|
214
|
-
for (const d of deductions) {
|
|
215
|
-
if (d.n <= 0)
|
|
216
|
-
continue;
|
|
217
|
-
penalty += d.n * d.weight;
|
|
218
|
-
}
|
|
219
|
-
return { score: Math.max(0, 100 - penalty), penalty };
|
|
220
|
-
}
|
|
221
|
-
/** Resolve the terse "thing(s)" plural placeholder against a count:
|
|
222
|
-
* n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools").
|
|
223
|
-
* (Kept local — audit-score.ts has its own copy to avoid a circular import.) */
|
|
224
|
-
function pluralizeLabel(n, label) {
|
|
225
|
-
return label.replace(/\(s\)/g, n === 1 ? "" : "s");
|
|
226
|
-
}
|
|
227
|
-
/** Deterministic structural-health score for one scanned plugin. */
|
|
228
|
-
function scoreReport(r) {
|
|
229
|
-
// An empty/unloadable machine isn't healthy — it's a non-plugin or a broken
|
|
230
|
-
// load. A command-only or MCP-only plugin (commands/*.md or .mcp.json with no
|
|
231
|
-
// skills/agents/hooks) IS a legitimate plugin, though — Anthropic ships
|
|
232
|
-
// command-only plugins in its own marketplace — so it must NOT score 0.
|
|
233
|
-
const surfaces = r.skills.length + r.agents.length + r.hooks.length + r.commands;
|
|
234
|
-
if (surfaces === 0 && !r.mcp) {
|
|
235
|
-
return { score: 0, issues: ["no loadable plugin surface"] };
|
|
236
|
-
}
|
|
237
|
-
const deductions = reportDeductions(r);
|
|
238
|
-
const { score } = computeIntegrityScore(deductions);
|
|
239
|
-
const issues = [];
|
|
240
|
-
for (const d of deductions) {
|
|
241
|
-
if (d.n === 0)
|
|
242
|
-
continue;
|
|
243
|
-
issues.push(`${String(d.n)} ${pluralizeLabel(d.n, d.label)}`);
|
|
244
|
-
}
|
|
245
|
-
// Sort issues by cost (worst first) so the report leads with what matters.
|
|
246
|
-
issues.sort((a, b) => Number(b.split(" ")[0]) - Number(a.split(" ")[0]));
|
|
247
|
-
// Advisory notes are surfaced for visibility but DON'T affect the score, so they
|
|
248
|
-
// come AFTER the real (score-affecting) issues:
|
|
249
|
-
// - inherit-all (no tool contract): a least-privilege NUDGE, not breakage —
|
|
250
|
-
// see reportDeductions for the full rationale.
|
|
251
|
-
// - untested surfaces: a hardening gap, not breakage.
|
|
252
|
-
const noContract = r.agents.filter((a) => a.tools === null).length;
|
|
253
|
-
if (noContract > 0) {
|
|
254
|
-
issues.push(`${String(noContract)} ${pluralizeLabel(noContract, "agent(s) inherit all tools (no contract) (advisory)")}`);
|
|
255
|
-
}
|
|
256
|
-
if (r.untested > 0) {
|
|
257
|
-
issues.push(`${String(r.untested)} ${pluralizeLabel(r.untested, "untested surface(s) (advisory)")}`);
|
|
258
|
-
}
|
|
259
|
-
return { score, issues };
|
|
260
|
-
}
|
|
261
56
|
/** Scan + score each directory, ranked best-first (ties broken by name). */
|
|
262
57
|
function rankPlugins(dirs) {
|
|
263
58
|
const scored = dirs.map((dir) => {
|
|
264
59
|
const report = (0, scan_js_1.scanPlugin)(dir);
|
|
265
|
-
const { score, issues } = scoreReport(report);
|
|
60
|
+
const { score, issues } = (0, score_core_js_1.scoreReport)(report);
|
|
266
61
|
return {
|
|
267
62
|
dir,
|
|
268
63
|
name: pluginLabel(dir),
|
|
269
64
|
score,
|
|
270
|
-
grade: gradeFor(score),
|
|
65
|
+
grade: (0, score_core_js_1.gradeFor)(score),
|
|
271
66
|
issues,
|
|
272
67
|
report,
|
|
273
68
|
};
|
package/dist/optimize.d.ts
CHANGED
|
@@ -26,7 +26,7 @@
|
|
|
26
26
|
* the ranked fix list + the hand-off to the measured layer. Same findings, the
|
|
27
27
|
* optimization framing. See research/measurement-authority.md (A2) + roadmap §P1.
|
|
28
28
|
*/
|
|
29
|
-
import { type PluginScore } from "./
|
|
29
|
+
import { type PluginScore } from "./score-core.js";
|
|
30
30
|
import { type ExplanationConfidence } from "./score-explainer.js";
|
|
31
31
|
import type { ScanReport } from "./scan.js";
|
|
32
32
|
/**
|
package/dist/optimize.js
CHANGED
|
@@ -31,7 +31,7 @@ exports.formatOptimize = formatOptimize;
|
|
|
31
31
|
* the ranked fix list + the hand-off to the measured layer. Same findings, the
|
|
32
32
|
* optimization framing. See research/measurement-authority.md (A2) + roadmap §P1.
|
|
33
33
|
*/
|
|
34
|
-
const
|
|
34
|
+
const score_core_js_1 = require("./score-core.js");
|
|
35
35
|
const score_explainer_js_1 = require("./score-explainer.js");
|
|
36
36
|
function actionFor(e) {
|
|
37
37
|
return e.symptom === "wrong-skill-fires" ? "differentiate" : "fix";
|
|
@@ -46,7 +46,7 @@ function isEmptyMachine(r) {
|
|
|
46
46
|
* before `possible` proxies, via explainScore's own ordering). Pure over the report.
|
|
47
47
|
*/
|
|
48
48
|
function optimize(report) {
|
|
49
|
-
const { score } = (0,
|
|
49
|
+
const { score } = (0, score_core_js_1.scoreReport)(report);
|
|
50
50
|
const recommendations = (0, score_explainer_js_1.explainScore)(report).map((e) => ({
|
|
51
51
|
surface: e.surface,
|
|
52
52
|
action: actionFor(e),
|
|
@@ -58,7 +58,7 @@ function optimize(report) {
|
|
|
58
58
|
return {
|
|
59
59
|
dir: report.dir,
|
|
60
60
|
score,
|
|
61
|
-
grade: (0,
|
|
61
|
+
grade: (0, score_core_js_1.gradeFor)(score),
|
|
62
62
|
recommendations,
|
|
63
63
|
empty: isEmptyMachine(report),
|
|
64
64
|
};
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A tiny, node-free POSIX `path` — the browser-safe substitute for `node:path`
|
|
3
|
+
* inside the in-browser audit engine (`scan-core.ts` / `scan-files.ts` /
|
|
4
|
+
* `test-coverage-files.ts` and the core detectors they reach). A Vite bundle of
|
|
5
|
+
* that engine must not pull `node:path`, so these are pure string ops.
|
|
6
|
+
*
|
|
7
|
+
* The audit engine's file-map keys are always POSIX (`/`-separated), and the
|
|
8
|
+
* disk-side `scanPlugin` feeds it absolute POSIX roots (`resolve(dir)` on Linux),
|
|
9
|
+
* so a faithful port of Node's `path.posix` algorithm is byte-identical to
|
|
10
|
+
* `node:path` for every input the engine passes — which is exactly what the
|
|
11
|
+
* parity firewall (`scan-files.test.ts`) proves. `resolve` deliberately falls
|
|
12
|
+
* back to `/` (never `process.cwd()`) so it stays pure and process-free; every
|
|
13
|
+
* call site passes an absolute first segment, so the fallback is never reached.
|
|
14
|
+
*
|
|
15
|
+
* NOTE — the functions below are VERBATIM ports of Node's `lib/path.js` POSIX
|
|
16
|
+
* implementations (charCode scan, `normalizeString`, `basename`, `relative`).
|
|
17
|
+
* Their branch depth / cyclomatic complexity is inherent to that battle-tested
|
|
18
|
+
* algorithm; rewriting it to satisfy the complexity linters would risk a subtle
|
|
19
|
+
* behavioural divergence from `node:path` (which the disk-vs-browser parity gate
|
|
20
|
+
* relies on), so the metric rules are disabled for this file only.
|
|
21
|
+
*/
|
|
22
|
+
/** POSIX `path.isAbsolute`. */
|
|
23
|
+
export declare function isAbsolute(path: string): boolean;
|
|
24
|
+
/** POSIX `path.normalize`. */
|
|
25
|
+
export declare function normalize(path: string): string;
|
|
26
|
+
/** POSIX `path.join`. */
|
|
27
|
+
export declare function join(...parts: string[]): string;
|
|
28
|
+
/**
|
|
29
|
+
* POSIX `path.resolve`. Right-to-left until an absolute segment is found, then
|
|
30
|
+
* normalize. The `i === -1` fallback is `/` (not `process.cwd()`) so this stays
|
|
31
|
+
* pure; every engine call site passes an absolute first segment.
|
|
32
|
+
*/
|
|
33
|
+
export declare function resolve(...parts: string[]): string;
|
|
34
|
+
/** POSIX `path.dirname`. */
|
|
35
|
+
export declare function dirname(path: string): string;
|
|
36
|
+
/** POSIX `path.basename` (with an optional `suffix` to strip, like `node:path`). */
|
|
37
|
+
export declare function basename(path: string, suffix?: string): string;
|
|
38
|
+
/** POSIX `path.relative`. */
|
|
39
|
+
export declare function relative(from: string, to: string): string;
|
|
40
|
+
//# sourceMappingURL=posix-path.d.ts.map
|