vigiles 14.7.0 → 14.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/agent-runtime.d.ts +2 -19
- package/dist/adapters/claude-code/agent-runtime.js +5 -30
- package/dist/adapters/claude-code/agent-tools.d.ts +20 -0
- package/dist/adapters/claude-code/agent-tools.js +40 -0
- package/dist/audit-report.d.ts +1 -1
- package/dist/audit-report.template.html +15 -15
- package/dist/audit-score.d.ts +1 -1
- package/dist/audit-score.js +19 -19
- package/dist/audit-verdict.d.ts +1 -1
- package/dist/audit-verdict.js +3 -3
- package/dist/core/assert-never.d.ts +9 -0
- package/dist/core/assert-never.js +14 -0
- package/dist/core/description-overlap.js +2 -2
- package/dist/core/effects.js +3 -3
- package/dist/core/hash.d.ts +1 -2
- package/dist/core/hash.js +6 -4
- package/dist/core/hook-block-ineffective.d.ts +55 -6
- package/dist/core/hook-block-ineffective.js +9 -14
- package/dist/core/mcp-contract-message.d.ts +22 -0
- package/dist/core/mcp-contract-message.js +29 -0
- package/dist/core/mcp.d.ts +4 -12
- package/dist/core/mcp.js +3 -14
- package/dist/core/ncd.d.ts +12 -0
- package/dist/core/ncd.js +50 -0
- package/dist/core/plugin-dir-layout.d.ts +5 -5
- package/dist/core/plugin-dir-layout.js +10 -22
- package/dist/core/proofs.d.ts +2 -11
- package/dist/core/proofs.js +4 -39
- package/dist/core/skill-resources.d.ts +3 -3
- package/dist/core/skill-resources.js +9 -8
- package/dist/leaderboard.d.ts +2 -51
- package/dist/leaderboard.js +20 -225
- package/dist/optimize.d.ts +1 -1
- package/dist/optimize.js +3 -3
- package/dist/posix-path.d.ts +40 -0
- package/dist/posix-path.js +293 -0
- package/dist/scan-core.d.ts +154 -0
- package/dist/scan-core.js +690 -0
- package/dist/scan-files.d.ts +28 -0
- package/dist/scan-files.js +489 -0
- package/dist/scan.d.ts +11 -34
- package/dist/scan.js +55 -668
- package/dist/score-core.d.ts +73 -0
- package/dist/score-core.js +226 -0
- package/dist/test-coverage-files.d.ts +11 -0
- package/dist/test-coverage-files.js +208 -0
- package/package.json +1 -1
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural-health SCORING core — extracted as a NODE-FREE leaf.
|
|
3
|
+
*
|
|
4
|
+
* The scoring functions (`scoreReport`/`reportDeductions`/`computeIntegrityScore`/
|
|
5
|
+
* `gradeFor` + the penalty weights) are pure: they take a {@link ScanReport} and
|
|
6
|
+
* return numbers/labels, touching no filesystem. They lived in `leaderboard.ts`,
|
|
7
|
+
* but that module ALSO holds the node-only `rankPlugins`/`pluginLabel` (which call
|
|
8
|
+
* `scanPlugin` + `node:fs`), so importing ANY symbol from it eagerly pulled the
|
|
9
|
+
* whole `scan.ts` → `@ast-grep/napi` (a native binary) + `node:fs` chain into the
|
|
10
|
+
* graph — tainting `buildAuditReport` and blocking the in-browser audit engine
|
|
11
|
+
* (`scanFiles` is node-free, but the report BUILDER re-tainted it via
|
|
12
|
+
* audit-score/optimize/audit-verdict → leaderboard → scan).
|
|
13
|
+
*
|
|
14
|
+
* Splitting the pure core out here — exactly like `core/ncd.ts` was extracted from
|
|
15
|
+
* `proofs.ts` to keep `node:crypto` out of the browser graph — lets the report
|
|
16
|
+
* builder import scoring WITHOUT the node-only leaderboard code. `leaderboard.ts`
|
|
17
|
+
* re-exports these for its own consumers (rankPlugins/formatters), so nothing that
|
|
18
|
+
* imported them from `./leaderboard.js` breaks. The `ScanReport` import is
|
|
19
|
+
* TYPE-ONLY (elided at build), so no runtime `scan.ts` dependency enters this leaf.
|
|
20
|
+
* See src/scan-files.ts (BROWSER_ROOT) and research/report-view-and-browser-demo.md.
|
|
21
|
+
*/
|
|
22
|
+
import type { ScanReport } from "./scan.js";
|
|
23
|
+
export interface PluginScore {
|
|
24
|
+
readonly dir: string;
|
|
25
|
+
readonly name: string;
|
|
26
|
+
/** 0–100 structural-health score (100 = no structural issues found). */
|
|
27
|
+
readonly score: number;
|
|
28
|
+
readonly grade: "A" | "B" | "C" | "D" | "F";
|
|
29
|
+
/** Human-readable deductions, worst first. */
|
|
30
|
+
readonly issues: readonly string[];
|
|
31
|
+
readonly report: ScanReport;
|
|
32
|
+
}
|
|
33
|
+
export declare const W_MISSING_HOOK = 15;
|
|
34
|
+
export declare const W_NO_DESCRIPTION = 10;
|
|
35
|
+
export declare const W_DANGLING_REF = 8;
|
|
36
|
+
export declare const W_OVERLAP = 8;
|
|
37
|
+
export declare const W_NO_CONTRACT = 5;
|
|
38
|
+
export declare const W_TRIFECTA = 10;
|
|
39
|
+
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
40
|
+
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
41
|
+
/** One deduction: a count, its per-item weight, and the label if non-zero. */
|
|
42
|
+
export interface Deduction {
|
|
43
|
+
readonly n: number;
|
|
44
|
+
readonly weight: number;
|
|
45
|
+
readonly label: string;
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* The COMPLETE graded-penalty list a report incurs — the single source of truth
|
|
49
|
+
* BOTH the leaderboard's single health number and the audit's category rings
|
|
50
|
+
* read, so the overall can never drift between the two surfaces. Each entry is a
|
|
51
|
+
* graded penalty; untested surfaces are deliberately ABSENT (they're advisory,
|
|
52
|
+
* surfaced separately, never scored).
|
|
53
|
+
*/
|
|
54
|
+
export declare function reportDeductions(r: ScanReport): Deduction[];
|
|
55
|
+
/** True when a report has no loadable plugin surface at all (the empty machine). */
|
|
56
|
+
export declare function isEmptyMachine(r: ScanReport): boolean;
|
|
57
|
+
/**
|
|
58
|
+
* THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
|
|
59
|
+
* [0,100]. Both the leaderboard's single health number AND the audit's headline
|
|
60
|
+
* overall read this, so the two can never disagree (the summed model is the
|
|
61
|
+
* honest one — averaging rings would dilute a real problem). Returns the score
|
|
62
|
+
* plus the per-item deductions so callers render their own issue/finding lists.
|
|
63
|
+
*/
|
|
64
|
+
export declare function computeIntegrityScore(deductions: readonly Deduction[]): {
|
|
65
|
+
score: number;
|
|
66
|
+
penalty: number;
|
|
67
|
+
};
|
|
68
|
+
/** Deterministic structural-health score for one scanned plugin. */
|
|
69
|
+
export declare function scoreReport(r: ScanReport): {
|
|
70
|
+
score: number;
|
|
71
|
+
issues: string[];
|
|
72
|
+
};
|
|
73
|
+
//# sourceMappingURL=score-core.d.ts.map
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
4
|
+
exports.gradeFor = gradeFor;
|
|
5
|
+
exports.reportDeductions = reportDeductions;
|
|
6
|
+
exports.isEmptyMachine = isEmptyMachine;
|
|
7
|
+
exports.computeIntegrityScore = computeIntegrityScore;
|
|
8
|
+
exports.scoreReport = scoreReport;
|
|
9
|
+
// Penalty weights — broken-at-runtime costs most, footguns less, nudges least.
|
|
10
|
+
// Exported so the category view (audit-score.ts) reuses the SAME weights and the
|
|
11
|
+
// two surfaces can never drift on a per-item cost.
|
|
12
|
+
exports.W_MISSING_HOOK = 15; // a hook script that doesn't exist → never runs
|
|
13
|
+
exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't trigger
|
|
14
|
+
exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
15
|
+
exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
|
|
16
|
+
exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
|
|
17
|
+
exports.W_TRIFECTA = 10; // a HARD lethal-trifecta contract (all three legs, explicit) → a prompt-injection exfil path. HALF the old 20: a DING, not a fail — a trifecta is a real risk worth surfacing in the grade, but official plugins ship the pattern by design, so it dents the score (e.g. feature-dev's 3 hard units → −30 → C) without a catastrophic F.
|
|
18
|
+
// Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
|
|
19
|
+
// - untested surfaces — a hardening gap, not breakage.
|
|
20
|
+
// - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
|
|
21
|
+
// - an inherits-all (severity "advisory") trifecta finding — shown by the Safety
|
|
22
|
+
// ring but never scored; only the HARD, explicit all-three-legs finding grades.
|
|
23
|
+
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
24
|
+
function gradeFor(score) {
|
|
25
|
+
if (score >= 90)
|
|
26
|
+
return "A";
|
|
27
|
+
if (score >= 80)
|
|
28
|
+
return "B";
|
|
29
|
+
if (score >= 70)
|
|
30
|
+
return "C";
|
|
31
|
+
if (score >= 60)
|
|
32
|
+
return "D";
|
|
33
|
+
return "F";
|
|
34
|
+
}
|
|
35
|
+
/**
|
|
36
|
+
* The COMPLETE graded-penalty list a report incurs — the single source of truth
|
|
37
|
+
* BOTH the leaderboard's single health number and the audit's category rings
|
|
38
|
+
* read, so the overall can never drift between the two surfaces. Each entry is a
|
|
39
|
+
* graded penalty; untested surfaces are deliberately ABSENT (they're advisory,
|
|
40
|
+
* surfaced separately, never scored).
|
|
41
|
+
*/
|
|
42
|
+
function reportDeductions(r) {
|
|
43
|
+
const missingHooks = r.hooks.filter((h) => h.status === "missing").length;
|
|
44
|
+
const noDesc = r.skills.filter((s) => !s.hasDescription).length;
|
|
45
|
+
const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
|
|
46
|
+
const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
|
|
47
|
+
const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
|
|
48
|
+
// HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
|
|
49
|
+
// legs (a prompt-injection exfil path). Graded at W_TRIFECTA=10 (HALF the old
|
|
50
|
+
// 20): a DING that surfaces a real risk in the grade without a catastrophic F
|
|
51
|
+
// for an accepted design pattern official plugins ship. Advisory (inherits-all)
|
|
52
|
+
// trifecta findings are surfaced but NEVER graded (aligned with the inherits-all
|
|
53
|
+
// stance), so they're excluded here.
|
|
54
|
+
const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
|
|
55
|
+
return [
|
|
56
|
+
{
|
|
57
|
+
n: hardTrifecta,
|
|
58
|
+
weight: exports.W_TRIFECTA,
|
|
59
|
+
label: "unit(s) holding all three lethal-trifecta legs (prompt-injection exfil path)",
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
n: missingHooks,
|
|
63
|
+
weight: exports.W_MISSING_HOOK,
|
|
64
|
+
label: "hook script(s) MISSING",
|
|
65
|
+
},
|
|
66
|
+
{
|
|
67
|
+
n: r.hookEventIssues.length,
|
|
68
|
+
weight: exports.W_MISSING_HOOK,
|
|
69
|
+
label: "hook(s) on an unknown event (never fire)",
|
|
70
|
+
},
|
|
71
|
+
{
|
|
72
|
+
n: noDesc,
|
|
73
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
74
|
+
label: "skill(s) with no usable description",
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
n: r.descriptionOverlaps.length,
|
|
78
|
+
weight: exports.W_OVERLAP,
|
|
79
|
+
label: "near-identical skill description(s) (wrong one fires)",
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
n: r.danglingRefs.length,
|
|
83
|
+
weight: exports.W_DANGLING_REF,
|
|
84
|
+
label: "broken intra-plugin reference(s)",
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
n: deadTools,
|
|
88
|
+
weight: exports.W_DANGLING_REF,
|
|
89
|
+
label: "unavailable agent tool(s) (typo / never-available)",
|
|
90
|
+
},
|
|
91
|
+
{
|
|
92
|
+
n: deadMcpTools,
|
|
93
|
+
weight: exports.W_DANGLING_REF,
|
|
94
|
+
label: "agent MCP tool(s) whose server isn't declared (can't resolve)",
|
|
95
|
+
},
|
|
96
|
+
{
|
|
97
|
+
n: deadDisallowed,
|
|
98
|
+
weight: exports.W_NO_CONTRACT,
|
|
99
|
+
label: "agent disallowedTools typo(s) that block nothing",
|
|
100
|
+
},
|
|
101
|
+
// NB: an agent that inherits all tools (no `tools:` line) is ADVISORY, not a
|
|
102
|
+
// graded penalty — it's surfaced by scoreReport / the Structure ring but never
|
|
103
|
+
// drags the score. WHY: omitting the `tools:` line is a near-universal,
|
|
104
|
+
// legitimate authoring style (a measured OSS sweep of 122 real plugins found
|
|
105
|
+
// 109 whose ONLY finding was this), so penalizing it makes the grade cry wolf
|
|
106
|
+
// on idiomatic subagents. A health score should mean "something is BROKEN", and
|
|
107
|
+
// a broad-by-default tool surface is a hardening/least-privilege NUDGE, not
|
|
108
|
+
// breakage. The count is re-derived where the advisory note is built.
|
|
109
|
+
{
|
|
110
|
+
n: r.frontmatterIssues.length,
|
|
111
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
112
|
+
label: "surface(s) missing required frontmatter (name/description)",
|
|
113
|
+
},
|
|
114
|
+
{
|
|
115
|
+
n: r.frontmatterValueIssues.length,
|
|
116
|
+
weight: exports.W_NO_CONTRACT,
|
|
117
|
+
label: "agent(s) with an invalid model/color (typo → silent fallback)",
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
n: r.mcpIssues.length,
|
|
121
|
+
weight: exports.W_DANGLING_REF,
|
|
122
|
+
label: "MCP server(s) that can't start (no command/url)",
|
|
123
|
+
},
|
|
124
|
+
{
|
|
125
|
+
n: r.mcpHookIssues.length,
|
|
126
|
+
weight: exports.W_DANGLING_REF,
|
|
127
|
+
label: "mcp_tool hook(s) incomplete / targeting an undeclared server",
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
n: r.skillResourceIssues.length,
|
|
131
|
+
weight: exports.W_DANGLING_REF,
|
|
132
|
+
label: "skill bundled-resource ref(s) that don't resolve on disk",
|
|
133
|
+
},
|
|
134
|
+
{
|
|
135
|
+
n: r.skillFenceIssues.length,
|
|
136
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
137
|
+
label: "invisible skill(s) (frontmatter with no opening `---` fence)",
|
|
138
|
+
},
|
|
139
|
+
{
|
|
140
|
+
n: r.pluginLayoutIssues.length,
|
|
141
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
142
|
+
label: "functional dir(s) misplaced inside `.claude-plugin/` (invisible)",
|
|
143
|
+
},
|
|
144
|
+
{
|
|
145
|
+
n: r.hookBlockFindings.length,
|
|
146
|
+
weight: exports.W_MISSING_HOOK,
|
|
147
|
+
label: "hook(s) that look like they block but silently don't",
|
|
148
|
+
},
|
|
149
|
+
{
|
|
150
|
+
n: r.hookMatcherFindings.length,
|
|
151
|
+
weight: exports.W_MISSING_HOOK,
|
|
152
|
+
label: "hook matcher(s) that never fire (typo / wrong MCP form)",
|
|
153
|
+
},
|
|
154
|
+
// NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
|
|
155
|
+
// ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
|
|
156
|
+
// NB: untested surfaces are NOT a penalty — an untested surface is a hardening
|
|
157
|
+
// gap, not breakage, so it never drags the health score (it's appended as an
|
|
158
|
+
// advisory note below). The score ranks what's BROKEN.
|
|
159
|
+
];
|
|
160
|
+
}
|
|
161
|
+
/** True when a report has no loadable plugin surface at all (the empty machine). */
|
|
162
|
+
function isEmptyMachine(r) {
|
|
163
|
+
const surfaces = r.skills.length +
|
|
164
|
+
r.agents.length +
|
|
165
|
+
r.hooks.length +
|
|
166
|
+
r.inlineHooks +
|
|
167
|
+
r.commands;
|
|
168
|
+
return surfaces === 0 && !r.mcp;
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* THE shared integrity score — `100 − Σ(all graded penalties)`, clamped to
|
|
172
|
+
* [0,100]. Both the leaderboard's single health number AND the audit's headline
|
|
173
|
+
* overall read this, so the two can never disagree (the summed model is the
|
|
174
|
+
* honest one — averaging rings would dilute a real problem). Returns the score
|
|
175
|
+
* plus the per-item deductions so callers render their own issue/finding lists.
|
|
176
|
+
*/
|
|
177
|
+
function computeIntegrityScore(deductions) {
|
|
178
|
+
let penalty = 0;
|
|
179
|
+
for (const d of deductions) {
|
|
180
|
+
if (d.n <= 0)
|
|
181
|
+
continue;
|
|
182
|
+
penalty += d.n * d.weight;
|
|
183
|
+
}
|
|
184
|
+
return { score: Math.max(0, 100 - penalty), penalty };
|
|
185
|
+
}
|
|
186
|
+
/** Resolve the terse "thing(s)" plural placeholder against a count:
|
|
187
|
+
* n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools").
|
|
188
|
+
* (Kept local — audit-score.ts has its own copy to avoid a circular import.) */
|
|
189
|
+
function pluralizeLabel(n, label) {
|
|
190
|
+
return label.replace(/\(s\)/g, n === 1 ? "" : "s");
|
|
191
|
+
}
|
|
192
|
+
/** Deterministic structural-health score for one scanned plugin. */
|
|
193
|
+
function scoreReport(r) {
|
|
194
|
+
// An empty/unloadable machine isn't healthy — it's a non-plugin or a broken
|
|
195
|
+
// load. A command-only or MCP-only plugin (commands/*.md or .mcp.json with no
|
|
196
|
+
// skills/agents/hooks) IS a legitimate plugin, though — Anthropic ships
|
|
197
|
+
// command-only plugins in its own marketplace — so it must NOT score 0.
|
|
198
|
+
const surfaces = r.skills.length + r.agents.length + r.hooks.length + r.commands;
|
|
199
|
+
if (surfaces === 0 && !r.mcp) {
|
|
200
|
+
return { score: 0, issues: ["no loadable plugin surface"] };
|
|
201
|
+
}
|
|
202
|
+
const deductions = reportDeductions(r);
|
|
203
|
+
const { score } = computeIntegrityScore(deductions);
|
|
204
|
+
const issues = [];
|
|
205
|
+
for (const d of deductions) {
|
|
206
|
+
if (d.n === 0)
|
|
207
|
+
continue;
|
|
208
|
+
issues.push(`${String(d.n)} ${pluralizeLabel(d.n, d.label)}`);
|
|
209
|
+
}
|
|
210
|
+
// Sort issues by cost (worst first) so the report leads with what matters.
|
|
211
|
+
issues.sort((a, b) => Number(b.split(" ")[0]) - Number(a.split(" ")[0]));
|
|
212
|
+
// Advisory notes are surfaced for visibility but DON'T affect the score, so they
|
|
213
|
+
// come AFTER the real (score-affecting) issues:
|
|
214
|
+
// - inherit-all (no tool contract): a least-privilege NUDGE, not breakage —
|
|
215
|
+
// see reportDeductions for the full rationale.
|
|
216
|
+
// - untested surfaces: a hardening gap, not breakage.
|
|
217
|
+
const noContract = r.agents.filter((a) => a.tools === null).length;
|
|
218
|
+
if (noContract > 0) {
|
|
219
|
+
issues.push(`${String(noContract)} ${pluralizeLabel(noContract, "agent(s) inherit all tools (no contract) (advisory)")}`);
|
|
220
|
+
}
|
|
221
|
+
if (r.untested > 0) {
|
|
222
|
+
issues.push(`${String(r.untested)} ${pluralizeLabel(r.untested, "untested surface(s) (advisory)")}`);
|
|
223
|
+
}
|
|
224
|
+
return { score, issues };
|
|
225
|
+
}
|
|
226
|
+
//# sourceMappingURL=score-core.js.map
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import type { PluginLayout } from "./core/layout.js";
|
|
2
|
+
import type { Surface } from "./test-coverage.js";
|
|
3
|
+
/**
|
|
4
|
+
* The untested harness surfaces (skills / agents / hooks) in a file map — the
|
|
5
|
+
* browser-safe twin of `findUntestedSurfaces`, returning only the `untested` list
|
|
6
|
+
* (`scanFiles` needs the count). Same coverage rules as the disk detector.
|
|
7
|
+
*/
|
|
8
|
+
export declare function findUntestedSurfacesInFiles(files: Record<string, string>, layout: PluginLayout): {
|
|
9
|
+
untested: Surface[];
|
|
10
|
+
};
|
|
11
|
+
//# sourceMappingURL=test-coverage-files.d.ts.map
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.findUntestedSurfacesInFiles = findUntestedSurfacesInFiles;
|
|
4
|
+
/**
|
|
5
|
+
* Browser-safe twin of `findUntestedSurfaces` (src/test-coverage.ts) — the same
|
|
6
|
+
* untested-surface detection over an in-memory repo-relative file map instead of
|
|
7
|
+
* a disk walk. `findUntestedSurfaces` globs the filesystem; this reimplements each
|
|
8
|
+
* of its `globSync` calls as an `Object.keys(files).filter(...)` over the map, so
|
|
9
|
+
* `scanFiles` reaches the SAME `untested` count with no disk I/O.
|
|
10
|
+
*
|
|
11
|
+
* Mirrors the disk detector exactly: the skill/agent surface globs (both the plain
|
|
12
|
+
* and the materialize-root form), the hook-script discovery from the
|
|
13
|
+
* manifest/settings/`.local` settings, the test-file globs (harness/eval/test
|
|
14
|
+
* suffixes) with the same ignore set, the `vigiles:ignore-test` exemption, and the
|
|
15
|
+
* colocation + content-reference coverage rules. See the parity test in
|
|
16
|
+
* src/scan-files.test.ts.
|
|
17
|
+
*/
|
|
18
|
+
const posix_path_js_1 = require("./posix-path.js");
|
|
19
|
+
// Mirrors src/test-coverage.ts constants.
|
|
20
|
+
const DEFAULT_TEST_SUFFIXES = [
|
|
21
|
+
".harness.mjs",
|
|
22
|
+
".eval.mjs",
|
|
23
|
+
".test.ts",
|
|
24
|
+
".test.mts",
|
|
25
|
+
".test.cts",
|
|
26
|
+
".test.js",
|
|
27
|
+
".test.mjs",
|
|
28
|
+
".test.cjs",
|
|
29
|
+
];
|
|
30
|
+
const IGNORE_MARKER = "vigiles:ignore-test";
|
|
31
|
+
const SCRIPT_RE = /[\w./${}@-]+\.(?:sh|mjs|cjs|js|ts|py|rb)/g;
|
|
32
|
+
/** Mirror of the DEFAULT_IGNORE globs (root-anchored), as a key predicate. */
|
|
33
|
+
function isIgnored(key) {
|
|
34
|
+
return /^(?:node_modules|dist|\.vigiles|\.git)\//.test(key);
|
|
35
|
+
}
|
|
36
|
+
/** Escape a literal path segment for use inside a RegExp. */
|
|
37
|
+
function escapeRe(s) {
|
|
38
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
39
|
+
}
|
|
40
|
+
/** The two prefixes a surface dir can occupy (plain + materialize-root form). */
|
|
41
|
+
function surfacePrefixes(dir, materializeRoot) {
|
|
42
|
+
if (!dir)
|
|
43
|
+
return [];
|
|
44
|
+
const matForm = materializeRoot ? `${materializeRoot}/${dir}` : dir;
|
|
45
|
+
return [...new Set([dir, matForm])];
|
|
46
|
+
}
|
|
47
|
+
/** Keys matching `<prefix>/<leafRe>` for any prefix, deduped + sorted (globSync order). */
|
|
48
|
+
function matchSurface(files, prefixes, leafRe) {
|
|
49
|
+
const found = new Set();
|
|
50
|
+
for (const prefix of prefixes) {
|
|
51
|
+
const re = new RegExp(`^${escapeRe(prefix)}/${leafRe}$`);
|
|
52
|
+
for (const key of Object.keys(files)) {
|
|
53
|
+
if (isIgnored(key))
|
|
54
|
+
continue;
|
|
55
|
+
if (re.test(key))
|
|
56
|
+
found.add(key);
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
return [...found].sort();
|
|
60
|
+
}
|
|
61
|
+
function discoverSkills(files, layout) {
|
|
62
|
+
const out = [];
|
|
63
|
+
if (layout.skillDir) {
|
|
64
|
+
const prefixes = surfacePrefixes(layout.skillDir, layout.materializeRoot);
|
|
65
|
+
for (const path of matchSurface(files, prefixes, "[^/]+/SKILL\\.md")) {
|
|
66
|
+
const name = (0, posix_path_js_1.basename)((0, posix_path_js_1.dirname)(path));
|
|
67
|
+
const content = files[path];
|
|
68
|
+
out.push({
|
|
69
|
+
kind: "skill",
|
|
70
|
+
path,
|
|
71
|
+
name,
|
|
72
|
+
tokens: [`${layout.skillDir}/${name}`, `:${name}`],
|
|
73
|
+
ignored: content.includes(IGNORE_MARKER),
|
|
74
|
+
});
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
// Single-skill-directory target: a bare `SKILL.md` at the repo root.
|
|
78
|
+
if (Object.prototype.hasOwnProperty.call(files, "SKILL.md")) {
|
|
79
|
+
const name = "SKILL"; // browser has no real base dir name (see BROWSER_ROOT).
|
|
80
|
+
const content = files["SKILL.md"];
|
|
81
|
+
out.push({
|
|
82
|
+
kind: "skill",
|
|
83
|
+
path: "SKILL.md",
|
|
84
|
+
name,
|
|
85
|
+
tokens: [`${layout.skillDir}/${name}`, `:${name}`],
|
|
86
|
+
ignored: content.includes(IGNORE_MARKER),
|
|
87
|
+
});
|
|
88
|
+
}
|
|
89
|
+
return out;
|
|
90
|
+
}
|
|
91
|
+
function discoverAgents(files, layout) {
|
|
92
|
+
const out = [];
|
|
93
|
+
if (!layout.agentDir)
|
|
94
|
+
return out;
|
|
95
|
+
const prefixes = surfacePrefixes(layout.agentDir, layout.materializeRoot);
|
|
96
|
+
for (const path of matchSurface(files, prefixes, "[^/]+\\.md")) {
|
|
97
|
+
if (path.endsWith(".spec.ts"))
|
|
98
|
+
continue;
|
|
99
|
+
const content = files[path];
|
|
100
|
+
const name = (0, posix_path_js_1.basename)(path, ".md");
|
|
101
|
+
const dir = (0, posix_path_js_1.dirname)(path);
|
|
102
|
+
out.push({
|
|
103
|
+
kind: "agent",
|
|
104
|
+
path,
|
|
105
|
+
name,
|
|
106
|
+
tokens: [`${dir}/${name}`],
|
|
107
|
+
ignored: content.includes(IGNORE_MARKER),
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
return out;
|
|
111
|
+
}
|
|
112
|
+
/** Hook-script paths referenced from a manifest's `hooks` block (file hooks only). */
|
|
113
|
+
function hookScripts(files, manifest, pluginRootToken) {
|
|
114
|
+
const text = files[manifest];
|
|
115
|
+
if (text === undefined)
|
|
116
|
+
return [];
|
|
117
|
+
let hooks;
|
|
118
|
+
try {
|
|
119
|
+
hooks = JSON.parse(text).hooks;
|
|
120
|
+
}
|
|
121
|
+
catch {
|
|
122
|
+
return [];
|
|
123
|
+
}
|
|
124
|
+
if (hooks === undefined)
|
|
125
|
+
return [];
|
|
126
|
+
const raw = JSON.stringify(hooks);
|
|
127
|
+
const unbraced = pluginRootToken.replace(/^\$\{(.+)\}$/, "$$$1");
|
|
128
|
+
const scripts = new Set();
|
|
129
|
+
for (const m of raw.matchAll(SCRIPT_RE)) {
|
|
130
|
+
const rel = m[0]
|
|
131
|
+
.replaceAll(pluginRootToken, "")
|
|
132
|
+
.replaceAll(unbraced, "")
|
|
133
|
+
.replace(/^\/+/, "")
|
|
134
|
+
.replace(/^\.\//, "");
|
|
135
|
+
if (Object.prototype.hasOwnProperty.call(files, rel))
|
|
136
|
+
scripts.add(rel);
|
|
137
|
+
}
|
|
138
|
+
return [...scripts];
|
|
139
|
+
}
|
|
140
|
+
function discoverHooks(files, layout) {
|
|
141
|
+
const scripts = new Set();
|
|
142
|
+
const localSettings = layout.settingsPath.replace(/(\.[^./]+)$/, ".local$1");
|
|
143
|
+
const manifests = [
|
|
144
|
+
...new Set([layout.manifestPath, layout.settingsPath, localSettings]),
|
|
145
|
+
];
|
|
146
|
+
for (const m of manifests) {
|
|
147
|
+
for (const s of hookScripts(files, m, layout.pluginRootToken)) {
|
|
148
|
+
scripts.add(s);
|
|
149
|
+
}
|
|
150
|
+
}
|
|
151
|
+
return [...scripts].sort().map((path) => ({
|
|
152
|
+
kind: "hook",
|
|
153
|
+
path,
|
|
154
|
+
name: (0, posix_path_js_1.basename)(path).replace(/\.[^.]+$/, ""),
|
|
155
|
+
tokens: [path],
|
|
156
|
+
ignored: false,
|
|
157
|
+
}));
|
|
158
|
+
}
|
|
159
|
+
function discoverTests(files) {
|
|
160
|
+
const out = [];
|
|
161
|
+
for (const [path, content] of Object.entries(files)) {
|
|
162
|
+
if (isIgnored(path))
|
|
163
|
+
continue;
|
|
164
|
+
if (DEFAULT_TEST_SUFFIXES.some((s) => path.endsWith(s))) {
|
|
165
|
+
out.push({ path, content });
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
return out;
|
|
169
|
+
}
|
|
170
|
+
/** Mirror of test-coverage.ts `isColocated`. */
|
|
171
|
+
function isColocated(surface, testPath) {
|
|
172
|
+
if (surface.kind === "skill") {
|
|
173
|
+
const dir = (0, posix_path_js_1.dirname)(surface.path);
|
|
174
|
+
return dir === "."
|
|
175
|
+
? (0, posix_path_js_1.dirname)(testPath) === "."
|
|
176
|
+
: testPath.startsWith(`${dir}/`);
|
|
177
|
+
}
|
|
178
|
+
return ((0, posix_path_js_1.dirname)(testPath) === (0, posix_path_js_1.dirname)(surface.path) &&
|
|
179
|
+
(0, posix_path_js_1.basename)(testPath).startsWith(`${surface.name}.`));
|
|
180
|
+
}
|
|
181
|
+
function isCovered(surface, tests) {
|
|
182
|
+
for (const t of tests) {
|
|
183
|
+
if (t.path === surface.path)
|
|
184
|
+
continue;
|
|
185
|
+
if (isColocated(surface, t.path))
|
|
186
|
+
return true;
|
|
187
|
+
if (surface.tokens.some((tok) => t.content.includes(tok)))
|
|
188
|
+
return true;
|
|
189
|
+
}
|
|
190
|
+
return false;
|
|
191
|
+
}
|
|
192
|
+
/**
|
|
193
|
+
* The untested harness surfaces (skills / agents / hooks) in a file map — the
|
|
194
|
+
* browser-safe twin of `findUntestedSurfaces`, returning only the `untested` list
|
|
195
|
+
* (`scanFiles` needs the count). Same coverage rules as the disk detector.
|
|
196
|
+
*/
|
|
197
|
+
function findUntestedSurfacesInFiles(files, layout) {
|
|
198
|
+
const surfaces = [
|
|
199
|
+
...discoverSkills(files, layout),
|
|
200
|
+
...discoverAgents(files, layout),
|
|
201
|
+
...discoverHooks(files, layout),
|
|
202
|
+
];
|
|
203
|
+
const considered = surfaces.filter((s) => !s.ignored);
|
|
204
|
+
const tests = discoverTests(files);
|
|
205
|
+
const untested = considered.filter((s) => !isCovered(s, tests));
|
|
206
|
+
return { untested };
|
|
207
|
+
}
|
|
208
|
+
//# sourceMappingURL=test-coverage-files.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "14.
|
|
3
|
+
"version": "14.8.0",
|
|
4
4
|
"description": "Lint & test the harness your AI agent runs on — verify the references in your CLAUDE.md / AGENTS.md and test that your hooks and skills actually work.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|