vigiles 6.0.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. package/README.md +189 -88
  2. package/dist/action-gate.js +1 -1
  3. package/dist/adapters/claude-code/agent-runtime.d.ts +46 -11
  4. package/dist/adapters/claude-code/agent-runtime.js +95 -24
  5. package/dist/adapters/claude-code/effect-region.js +1 -1
  6. package/dist/adapters/claude-code/skill-runtime.d.ts +1 -1
  7. package/dist/adapters/claude-code/skill-runtime.js +1 -1
  8. package/dist/adapters/codex/hook-protocol.js +3 -0
  9. package/dist/adapters/codex/mock-model.js +1 -1
  10. package/dist/cli-commands.d.ts +19 -0
  11. package/dist/cli-commands.js +47 -0
  12. package/dist/cli.d.ts +1 -1
  13. package/dist/cli.js +1054 -201
  14. package/dist/core/adopt.d.ts +65 -0
  15. package/dist/core/adopt.js +199 -0
  16. package/dist/core/bash-effects.d.ts +12 -0
  17. package/dist/core/bash-effects.js +31 -0
  18. package/dist/core/capability-diff.d.ts +46 -0
  19. package/dist/core/capability-diff.js +97 -0
  20. package/dist/core/compose.d.ts +1 -1
  21. package/dist/core/compose.js +1 -1
  22. package/dist/core/evolve.d.ts +4 -0
  23. package/dist/core/evolve.js +4 -0
  24. package/dist/core/frontmatter.d.ts +8 -7
  25. package/dist/core/frontmatter.js +8 -7
  26. package/dist/core/generate-harness.d.ts +1 -1
  27. package/dist/core/generate-harness.js +3 -3
  28. package/dist/core/generate-schema.js +1 -1
  29. package/dist/core/guards.d.ts +126 -0
  30. package/dist/core/guards.js +309 -0
  31. package/dist/core/harness-driver.d.ts +1 -1
  32. package/dist/core/hook-program.d.ts +459 -0
  33. package/dist/core/hook-program.js +468 -0
  34. package/dist/core/hook-protocol.d.ts +7 -0
  35. package/dist/core/hook-providers.d.ts +138 -0
  36. package/dist/core/hook-providers.js +155 -0
  37. package/dist/core/hook-spec.d.ts +74 -0
  38. package/dist/core/hook-spec.js +130 -0
  39. package/dist/core/inline.d.ts +6 -6
  40. package/dist/core/inline.js +7 -7
  41. package/dist/core/integrity.d.ts +31 -0
  42. package/dist/core/integrity.js +45 -0
  43. package/dist/core/mcp-tool.d.ts +12 -0
  44. package/dist/core/mcp-tool.js +20 -0
  45. package/dist/core/mcp.d.ts +13 -0
  46. package/dist/core/mcp.js +67 -0
  47. package/dist/core/orphans.js +1 -1
  48. package/dist/core/spec.d.ts +40 -2
  49. package/dist/core/spec.js +16 -1
  50. package/dist/core/types.d.ts +37 -5
  51. package/dist/core/validate.js +26 -26
  52. package/dist/dialect-drift.d.ts +65 -0
  53. package/dist/dialect-drift.js +216 -0
  54. package/dist/eval.d.ts +40 -5
  55. package/dist/eval.js +59 -5
  56. package/dist/guardrail-check.d.ts +85 -0
  57. package/dist/guardrail-check.js +152 -0
  58. package/dist/harness-assert.d.ts +10 -0
  59. package/dist/harness-assert.js +30 -0
  60. package/dist/hook-install.d.ts +43 -0
  61. package/dist/hook-install.js +91 -0
  62. package/dist/hook.d.ts +52 -0
  63. package/dist/hook.js +98 -0
  64. package/dist/leaderboard.d.ts +6 -0
  65. package/dist/leaderboard.js +43 -1
  66. package/dist/linting.d.ts +9 -5
  67. package/dist/linting.js +17 -5
  68. package/dist/optimize.js +1 -1
  69. package/dist/scaffold-test.js +21 -7
  70. package/dist/scan-behavioral.d.ts +60 -0
  71. package/dist/scan-behavioral.js +239 -1
  72. package/dist/scan-trigger-suggest.d.ts +54 -0
  73. package/dist/scan-trigger-suggest.js +70 -0
  74. package/dist/scan.d.ts +31 -1
  75. package/dist/scan.js +65 -3
  76. package/dist/score-explainer.js +1 -1
  77. package/dist/self-command-refs.d.ts +21 -0
  78. package/dist/self-command-refs.js +125 -0
  79. package/dist/setup-plan.d.ts +59 -1
  80. package/dist/setup-plan.js +103 -5
  81. package/dist/testing.d.ts +5 -3
  82. package/dist/testing.js +37 -23
  83. package/dist/tool-intercept.d.ts +4 -4
  84. package/dist/tool-intercept.js +5 -5
  85. package/dist/unit.d.ts +2 -0
  86. package/dist/unit.js +8 -1
  87. package/hooks/post-edit.sh +1 -1
  88. package/hooks/refs-nudge.sh +1 -1
  89. package/package.json +5 -3
  90. package/skills/adopt-spec/SKILL.md +7 -7
  91. package/skills/linter-docs/eslint.md +1 -1
  92. package/skills/strengthen/SKILL.md +1 -1
@@ -17,12 +17,17 @@ Object.defineProperty(exports, "__esModule", { value: true });
17
17
  exports.probePluginTriggersWith = probePluginTriggersWith;
18
18
  exports.probePluginTriggers = probePluginTriggers;
19
19
  exports.formatBehavioralReport = formatBehavioralReport;
20
+ exports.buildSelectionReport = buildSelectionReport;
21
+ exports.measurePluginSelectionWith = measurePluginSelectionWith;
22
+ exports.measurePluginSelection = measurePluginSelection;
23
+ exports.formatSelectionReport = formatSelectionReport;
20
24
  const node_fs_1 = require("node:fs");
21
25
  const node_path_1 = require("node:path");
22
26
  const scan_js_1 = require("./scan.js");
23
27
  const eval_js_1 = require("./eval.js");
24
28
  const harness_assert_js_1 = require("./harness-assert.js");
25
29
  const harness_test_js_1 = require("./harness-test.js");
30
+ const plugin_loader_js_1 = require("./adapters/claude-code/plugin-loader.js");
26
31
  const eval_js_2 = require("./adapters/codex/eval.js");
27
32
  const driver_js_1 = require("./adapters/codex/driver.js");
28
33
  function buildProbe(dir, harness) {
@@ -56,6 +61,39 @@ function pluginName(dir) {
56
61
  return null;
57
62
  }
58
63
  }
64
+ /**
65
+ * Does the plugin declare a SessionStart hook? The STUBBED measurement path rebuilds
66
+ * the plugin to skills-only (`packageSkillsDir`), DROPPING `hooks/` — so a SessionStart
67
+ * hook that primes skill selection (e.g. superpowers' `using-superpowers` gateway
68
+ * injection) is silently lost, and a recall collapse to 0 under stubbing is then a
69
+ * measurement ARTIFACT, not a real miss. Detect it to LABEL honestly (Layer 1) rather
70
+ * than report a misleading 0%. See `research/plugin-selection-collision.md`.
71
+ */
72
+ function hasSessionStartHook(dir) {
73
+ try {
74
+ const hooks = (0, plugin_loader_js_1.loadPlugin)(dir).settings.hooks;
75
+ return hooks !== undefined && Object.keys(hooks).includes("SessionStart");
76
+ }
77
+ catch {
78
+ return false;
79
+ }
80
+ }
81
+ const HOOK_PRIMED_NOTE = "hook-primed — the stubbed run dropped the plugin's SessionStart hook (which can " +
82
+ "prime skill selection), so 0% recall is likely a measurement artifact; re-run " +
83
+ "against the full plugin install to measure faithfully";
84
+ /**
85
+ * Layer-1 honesty: a STUBBED run on a SessionStart-hooked plugin where EVERY measured
86
+ * skill sits at recall 0 is the dropped-hook artifact — not a real result. The
87
+ * all-zero gate keeps a genuine single-skill miss reported as real (if siblings fired,
88
+ * the hook ran or wasn't needed). Applied to both the trigger column and the matrix.
89
+ */
90
+ function isStubbedHookArtifact(dir, stub, recalls) {
91
+ if (!stub || recalls.length === 0)
92
+ return false;
93
+ if (!recalls.every((r) => r === 0))
94
+ return false;
95
+ return hasSessionStartHook(dir);
96
+ }
59
97
  /** Probe one skill via the harness probe's eval driver → result, never throwing. */
60
98
  async function probeSkill(ctx, name, ps) {
61
99
  try {
@@ -108,7 +146,19 @@ async function probePluginTriggersWith(dir, promptSet, probe, opts = {}) {
108
146
  }
109
147
  results.push(await probeSkill(ctx, s.name, ps));
110
148
  }
111
- return { available: true, results };
149
+ return {
150
+ available: true,
151
+ results: relabelTriggerArtifact(dir, probe, results),
152
+ };
153
+ }
154
+ /** Relabel an all-zero-recall stubbed run on a hooked plugin as unmeasured (Layer 1). */
155
+ function relabelTriggerArtifact(dir, probe, results) {
156
+ const recalls = results.filter((r) => r.measured).map((r) => r.recall ?? 0);
157
+ if (!isStubbedHookArtifact(dir, probe.stub, recalls))
158
+ return [...results];
159
+ return results.map((r) => r.measured && (r.recall ?? 0) === 0
160
+ ? { skill: r.skill, measured: false, note: HOOK_PRIMED_NOTE }
161
+ : r);
112
162
  }
113
163
  /**
114
164
  * Probe a plugin's skills against the real harness (default Claude Code; Codex via
@@ -147,4 +197,192 @@ function formatBehavioralReport(b) {
147
197
  }
148
198
  return lines.join("\n");
149
199
  }
200
+ /** Map a namespaced skill id (`ns:name`) to its bare name. */
201
+ function bareSkillName(id) {
202
+ const i = id.lastIndexOf(":");
203
+ return i >= 0 ? id.slice(i + 1) : id;
204
+ }
205
+ function emptySelection(skills) {
206
+ return {
207
+ available: true,
208
+ skills,
209
+ matrix: skills.map(() => []),
210
+ perSkill: [],
211
+ collisionRate: 0,
212
+ n: 0,
213
+ };
214
+ }
215
+ function buildSkillStat(a) {
216
+ const { skill, i, skills, row, n, collisions } = a;
217
+ const collidesWith = skills
218
+ .map((s, j) => ({ skill: s, rate: n > 0 ? row[j] / n : 0 }))
219
+ .filter((c) => c.skill !== skill && c.rate > 0)
220
+ .sort((x, y) => y.rate - x.rate);
221
+ return {
222
+ skill,
223
+ recall: n > 0 ? row[i] / n : 0,
224
+ collisionRate: n > 0 ? collisions / n : 0,
225
+ n,
226
+ collidesWith,
227
+ };
228
+ }
229
+ /**
230
+ * Pure aggregation: fold per-run fired-skill sets into the N×N selection matrix +
231
+ * per-skill recall/collision + the plugin-level collision rate. Separated from the
232
+ * model-driving so it's unit-testable with synthetic runs (no model).
233
+ */
234
+ function buildSelectionReport(skills, runs) {
235
+ const index = new Map(skills.map((s, i) => [s, i]));
236
+ const n = skills.length;
237
+ const matrix = skills.map(() => new Array(n).fill(0));
238
+ const nBy = new Array(n).fill(0);
239
+ const collisionBy = new Array(n).fill(0);
240
+ let collisionTotal = 0;
241
+ for (const run of runs) {
242
+ const i = index.get(run.intended);
243
+ if (i === undefined)
244
+ continue;
245
+ nBy[i] += 1;
246
+ let collided = false;
247
+ for (const fb of run.firedBare) {
248
+ const j = index.get(fb);
249
+ if (j === undefined)
250
+ continue;
251
+ matrix[i][j] += 1;
252
+ if (j !== i)
253
+ collided = true;
254
+ }
255
+ if (collided) {
256
+ collisionBy[i] += 1;
257
+ collisionTotal += 1;
258
+ }
259
+ }
260
+ const total = nBy.reduce((a, b) => a + b, 0);
261
+ return {
262
+ available: true,
263
+ skills,
264
+ matrix,
265
+ perSkill: skills.map((s, i) => buildSkillStat({
266
+ skill: s,
267
+ i,
268
+ skills,
269
+ row: matrix[i],
270
+ n: nBy[i],
271
+ collisions: collisionBy[i],
272
+ })),
273
+ collisionRate: total > 0 ? collisionTotal / total : 0,
274
+ n: total,
275
+ };
276
+ }
277
+ /** Build the prompts × trials work list across every skill that has prompts. */
278
+ function selectionJobs(candidates, promptSet, trials) {
279
+ return candidates.flatMap((c) => {
280
+ const ps = promptSet[c.name];
281
+ if (!ps || ps.prompts.length === 0)
282
+ return [];
283
+ return ps.prompts.flatMap((prompt) => Array.from({ length: trials }, () => ({ intended: c.name, prompt })));
284
+ });
285
+ }
286
+ /** The injectable core (for tests): drive the matrix via a fake/real probe. */
287
+ async function measurePluginSelectionWith(dir, promptSet, probe, opts = {}) {
288
+ const candidates = (0, scan_js_1.scanPlugin)(dir).skills.filter((s) => !s.userInvoked && s.hasDescription);
289
+ const skills = candidates.map((c) => c.name);
290
+ if (skills.length < 2) {
291
+ return {
292
+ ...emptySelection(skills),
293
+ note: "needs ≥2 model-invocable skills to measure cross-skill collision",
294
+ };
295
+ }
296
+ const own = new Set(skills);
297
+ const jobs = selectionJobs(candidates, promptSet, Math.max(1, opts.trials ?? 1));
298
+ if (jobs.length === 0) {
299
+ return {
300
+ ...emptySelection(skills),
301
+ note: "no prompts supplied for any skill",
302
+ };
303
+ }
304
+ // Selection is decided at the frontmatter (the selector picks BEFORE the body
305
+ // loads), so stub each body to a no-op: the run stops AT selection instead of
306
+ // executing the whole workflow — the same affordability trick trigger-rate uses.
307
+ const pluginDir = probe.stub ? (0, eval_js_1.stubbedPluginDir)(dir) : dir;
308
+ try {
309
+ const d = probe.evalDriver;
310
+ const outcomes = await (0, eval_js_1.runPool)(jobs, Math.max(1, opts.concurrency ?? 1), (job) => (0, eval_js_1.runSkillSelectionTrial)({
311
+ prompt: job.prompt,
312
+ pluginDir,
313
+ runner: d.runner,
314
+ parse: d.parse,
315
+ runError: d.runError,
316
+ model: opts.model ?? "sonnet",
317
+ }));
318
+ const runs = [];
319
+ jobs.forEach((job, k) => {
320
+ if (outcomes[k].errored)
321
+ return;
322
+ runs.push({
323
+ intended: job.intended,
324
+ firedBare: outcomes[k].fired
325
+ .map(bareSkillName)
326
+ .filter((b) => own.has(b)),
327
+ });
328
+ });
329
+ const report = buildSelectionReport(skills, runs);
330
+ // Layer-1 honesty: an all-zero-recall stubbed run on a hooked plugin is the
331
+ // dropped-hook artifact — flag it instead of presenting a 0%-collision result
332
+ // computed from a plugin that never fired.
333
+ const recalls = report.perSkill.filter((s) => s.n > 0).map((s) => s.recall);
334
+ return isStubbedHookArtifact(dir, probe.stub, recalls)
335
+ ? { ...report, note: HOOK_PRIMED_NOTE }
336
+ : report;
337
+ }
338
+ finally {
339
+ if (probe.stub)
340
+ (0, node_fs_1.rmSync)(pluginDir, { recursive: true, force: true });
341
+ }
342
+ }
343
+ /**
344
+ * Measure a plugin's cross-skill selection-collision matrix against the real
345
+ * harness (Claude Code only — Codex has no skill-selection event). Needs the
346
+ * `claude` CLI + model auth; degrades to `available: false` otherwise.
347
+ */
348
+ async function measurePluginSelection(dir, promptSet, opts = {}) {
349
+ const harness = opts.harness ?? "claude-code";
350
+ if (harness !== "claude-code") {
351
+ return {
352
+ ...emptySelection([]),
353
+ available: false,
354
+ note: `selection-collision is Claude Code only (no skill-selection event on ${harness})`,
355
+ };
356
+ }
357
+ const probe = buildProbe(dir, harness);
358
+ if (!probe.available()) {
359
+ return {
360
+ ...emptySelection([]),
361
+ available: false,
362
+ note: "needs the claude CLI + model auth",
363
+ };
364
+ }
365
+ return measurePluginSelectionWith(dir, promptSet, probe, opts);
366
+ }
367
+ /** Format the selection-collision matrix as a scan-report section. */
368
+ function formatSelectionReport(r) {
369
+ if (!r.available)
370
+ return `Selection-collision: unavailable — ${r.note ?? "n/a"}`;
371
+ if (r.n === 0)
372
+ return `Selection-collision: ${r.note ?? "not measured"}`;
373
+ const lines = [
374
+ `Selection-collision: ${pct(r.collisionRate)} of ${String(r.n)} runs hit a sibling skill`,
375
+ ];
376
+ if (r.note)
377
+ lines.push(` ⓘ ${r.note}`);
378
+ for (const s of r.perSkill) {
379
+ if (s.n === 0)
380
+ continue;
381
+ const mark = s.collisionRate === 0 ? "✓" : "⚠";
382
+ const top = s.collidesWith[0];
383
+ const tail = top ? ` — top collider: ${top.skill} ${pct(top.rate)}` : "";
384
+ lines.push(` ${mark} ${s.skill} — recall ${pct(s.recall)}, collision ${pct(s.collisionRate)}${tail}`);
385
+ }
386
+ return lines.join("\n");
387
+ }
150
388
  //# sourceMappingURL=scan-behavioral.js.map
@@ -0,0 +1,54 @@
1
+ /**
2
+ * scan → trigger-tier nudge. When a plugin ships model-invocable skills AND a
3
+ * real model is reachable, surface the (real-model) `scan --trigger` tier that
4
+ * measures whether those skills actually FIRE (recall + precision). Pure
5
+ * decision + helpers; the IO (prompt / scaffold write / running the measure)
6
+ * lives in the CLI. Honors `great-agent-flow`: an agent (non-TTY / `--json` /
7
+ * `--no-interactive`) is HINTED, never prompted — a `scan` must never hang.
8
+ */
9
+ /** Only the env vars that signal a reachable model (parse, don't validate). */
10
+ export interface ModelEnv {
11
+ readonly ANTHROPIC_API_KEY?: string;
12
+ readonly CLAUDECODE?: string;
13
+ readonly CLAUDE_CODE_ENTRYPOINT?: string;
14
+ }
15
+ /**
16
+ * Is a real model reachable for the eval / trigger tier? Either a metered API
17
+ * key (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session
18
+ * (`CLAUDECODE=1` / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter
19
+ * drives the `claude` CLI on the user's subscription, no key needed. Keeping
20
+ * this a tiny, env-only predicate (not a live probe) means it never spends a
21
+ * token just to decide whether to suggest spending one.
22
+ */
23
+ export declare function hasModelAccess(env: ModelEnv): boolean;
24
+ export type TriggerSuggestion = "prompt" | "hint" | "none";
25
+ export interface SuggestOpts {
26
+ /** A real model is reachable (`hasModelAccess`). */
27
+ readonly modelAccess: boolean;
28
+ /** stdout is an interactive terminal (a human is watching). */
29
+ readonly isTTY: boolean;
30
+ /** Count of model-invocable, described skills worth measuring. */
31
+ readonly triggerableSkills: number;
32
+ /** `--json` — machine output; never decorate or prompt. */
33
+ readonly json: boolean;
34
+ /** `--no-interactive` — explicit agent/CI mode; hint, never prompt. */
35
+ readonly noInteractive: boolean;
36
+ }
37
+ /**
38
+ * Decide how `scan` surfaces the trigger tier after its report:
39
+ * - `"none"` — no model-invocable skills, no model access, or `--json`.
40
+ * - `"hint"` — model access + skills but non-interactive (agent / CI / non-TTY
41
+ * / `--no-interactive`): a one-line, non-blocking hint.
42
+ * - `"prompt"` — model access + skills + a human at a TTY: offer to set it up.
43
+ */
44
+ export declare function decideTriggerSuggestion(o: SuggestOpts): TriggerSuggestion;
45
+ /** The one-line, non-blocking hint (the agent/CI surface). */
46
+ export declare function formatTriggerHint(dir: string, triggerableSkills: number): string;
47
+ /**
48
+ * A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
49
+ * → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
50
+ * placeholders the user replaces with real requests. Deterministic; written
51
+ * only on an explicit human "yes" so a plain `scan` never spends a token.
52
+ */
53
+ export declare function scaffoldTriggerPrompts(skillNames: readonly string[]): string;
54
+ //# sourceMappingURL=scan-trigger-suggest.d.ts.map
@@ -0,0 +1,70 @@
1
+ "use strict";
2
+ /**
3
+ * scan → trigger-tier nudge. When a plugin ships model-invocable skills AND a
4
+ * real model is reachable, surface the (real-model) `scan --trigger` tier that
5
+ * measures whether those skills actually FIRE (recall + precision). Pure
6
+ * decision + helpers; the IO (prompt / scaffold write / running the measure)
7
+ * lives in the CLI. Honors `great-agent-flow`: an agent (non-TTY / `--json` /
8
+ * `--no-interactive`) is HINTED, never prompted — a `scan` must never hang.
9
+ */
10
+ Object.defineProperty(exports, "__esModule", { value: true });
11
+ exports.hasModelAccess = hasModelAccess;
12
+ exports.decideTriggerSuggestion = decideTriggerSuggestion;
13
+ exports.formatTriggerHint = formatTriggerHint;
14
+ exports.scaffoldTriggerPrompts = scaffoldTriggerPrompts;
15
+ /**
16
+ * Is a real model reachable for the eval / trigger tier? Either a metered API
17
+ * key (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session
18
+ * (`CLAUDECODE=1` / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter
19
+ * drives the `claude` CLI on the user's subscription, no key needed. Keeping
20
+ * this a tiny, env-only predicate (not a live probe) means it never spends a
21
+ * token just to decide whether to suggest spending one.
22
+ */
23
+ function hasModelAccess(env) {
24
+ return Boolean(env.ANTHROPIC_API_KEY ||
25
+ env.CLAUDECODE === "1" ||
26
+ env.CLAUDE_CODE_ENTRYPOINT);
27
+ }
28
+ /**
29
+ * Decide how `scan` surfaces the trigger tier after its report:
30
+ * - `"none"` — no model-invocable skills, no model access, or `--json`.
31
+ * - `"hint"` — model access + skills but non-interactive (agent / CI / non-TTY
32
+ * / `--no-interactive`): a one-line, non-blocking hint.
33
+ * - `"prompt"` — model access + skills + a human at a TTY: offer to set it up.
34
+ */
35
+ function decideTriggerSuggestion(o) {
36
+ if (o.json || o.triggerableSkills < 1 || !o.modelAccess)
37
+ return "none";
38
+ if (o.noInteractive || !o.isTTY)
39
+ return "hint";
40
+ return "prompt";
41
+ }
42
+ /** The one-line, non-blocking hint (the agent/CI surface). */
43
+ function formatTriggerHint(dir, triggerableSkills) {
44
+ const n = triggerableSkills;
45
+ return (`ℹ ${String(n)} model-invocable skill${n === 1 ? "" : "s"} + model access detected — ` +
46
+ `measure whether they actually FIRE (recall + precision) with:\n` +
47
+ ` vigiles scan ${dir} --trigger --prompts=trigger-prompts.json`);
48
+ }
49
+ /**
50
+ * A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
51
+ * → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
52
+ * placeholders the user replaces with real requests. Deterministic; written
53
+ * only on an explicit human "yes" so a plain `scan` never spends a token.
54
+ */
55
+ function scaffoldTriggerPrompts(skillNames) {
56
+ const obj = {};
57
+ for (const name of skillNames) {
58
+ obj[name] = {
59
+ prompts: [
60
+ `TODO: a request that SHOULD trigger "${name}"`,
61
+ `TODO: a differently-phrased request that should also trigger it`,
62
+ ],
63
+ irrelevant: [
64
+ `TODO: an unrelated request that should NOT trigger "${name}"`,
65
+ ],
66
+ };
67
+ }
68
+ return JSON.stringify(obj, null, 2) + "\n";
69
+ }
70
+ //# sourceMappingURL=scan-trigger-suggest.js.map
package/dist/scan.d.ts CHANGED
@@ -18,6 +18,7 @@ import { type HookEventIssue } from "./core/hook-events.js";
18
18
  import { type McpIssue } from "./core/mcp-config.js";
19
19
  import { type DescriptionOverlap } from "./core/description-overlap.js";
20
20
  import { type McpToolIssue } from "./core/mcp-tool.js";
21
+ import { type McpContractToolError } from "./core/mcp.js";
21
22
  import { type McpHookIssue } from "./core/mcp-hook.js";
22
23
  import { type PurityLevel, type EffectSurface } from "./core/effects.js";
23
24
  /** A named writing system. The label `unexpectedScript` reports + the config's expectation parse into this. */
@@ -93,7 +94,7 @@ export interface ScanHook {
93
94
  * Every cc/codex repo has one even when it ships no plugin surface, so `scan`
94
95
  * reports it — otherwise a plain instruction-only repo looks empty. `hasSpec` is
95
96
  * the deterministic fact that a `<file>.spec.ts` sits beside it (spec-managed vs
96
- * hand-written); it is informational, NOT the `require-spec` gate (that's lint).
97
+ * hand-written); it is informational, NOT the `require-instructions-spec` gate (that's lint).
97
98
  */
98
99
  export interface ScanInstructions {
99
100
  readonly file: string;
@@ -108,6 +109,13 @@ export interface ScanReport {
108
109
  readonly hooks: readonly ScanHook[];
109
110
  /** Hook entries with no script file (inline shell one-liners) — can't be path-checked. */
110
111
  readonly inlineHooks: number;
112
+ /**
113
+ * Hand-written hook commands that are NOT compiled `vigiles/hook` artifacts (a
114
+ * compiled hook's command invokes `vigiles hook-runtime run-program`). The basis
115
+ * for the `prefer-compiled-hooks` recommendation — a single nudge regardless of
116
+ * count. Zero when there are no hooks or every hook is vigiles-managed.
117
+ */
118
+ readonly manualHookCount: number;
111
119
  readonly commands: number;
112
120
  readonly mcp: boolean;
113
121
  /**
@@ -174,8 +182,30 @@ export interface SurfaceClassifier {
174
182
  * no drift). The ≥20% guard avoids a near-empty string tripping on one letter.
175
183
  */
176
184
  export declare function unexpectedScript(text: string, expected?: Script): Script | null;
185
+ /**
186
+ * A compiled `vigiles/hook` artifact runs through the `hook-runtime run-program`
187
+ * runtime entrypoint; any other hook command is hand-written (a shell script or
188
+ * an inline one-liner) the author maintains directly. The basis for the
189
+ * `prefer-compiled-hooks` nudge.
190
+ */
191
+ export declare function isManagedHookCommand(command: string): boolean;
192
+ /** The `prefer-compiled-hooks` recommendation message (shared by `lint` + `scan`). */
193
+ export declare function preferCompiledHooksMessage(count: number): string;
177
194
  /** Scan a plugin/repo directory and report its surfaces + structural issues. */
178
195
  export declare function scanPlugin(dir: string, layout?: PluginLayout, dialect?: HarnessDialect): ScanReport;
196
+ /**
197
+ * LIVE MCP tool resolution for a scanned plugin — the opt-in (`scan --verify-mcp`)
198
+ * dynamic check no static linter can do: it STARTS each declared MCP server and
199
+ * checks every `mcp__server__tool` the plugin's agents reference actually exists on
200
+ * it (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
201
+ * already-computed `report` (its agents' tool lists) + the declared server configs;
202
+ * returns `[]` when the plugin declares no MCP servers (nothing to start). Async +
203
+ * side-effecting (spawns servers) — which is exactly why it's opt-in, not a default
204
+ * lint rule. See `verifyMcpContractTools` (core/mcp.ts).
205
+ */
206
+ export declare function verifyLiveMcpTools(report: ScanReport, layout: PluginLayout, dialect: HarnessDialect, timeoutMs?: number): Promise<McpContractToolError[]>;
207
+ /** Render the live MCP tool-check result (human-readable). */
208
+ export declare function formatMcpContractReport(errors: readonly McpContractToolError[]): string;
179
209
  /**
180
210
  * A plugin MARKETPLACE (`.claude-plugin/marketplace.json`) decomposed into its
181
211
  * members. A marketplace either VENDORS its plugins in-tree (string `source`
package/dist/scan.js CHANGED
@@ -14,7 +14,11 @@
14
14
  */
15
15
  Object.defineProperty(exports, "__esModule", { value: true });
16
16
  exports.unexpectedScript = unexpectedScript;
17
+ exports.isManagedHookCommand = isManagedHookCommand;
18
+ exports.preferCompiledHooksMessage = preferCompiledHooksMessage;
17
19
  exports.scanPlugin = scanPlugin;
20
+ exports.verifyLiveMcpTools = verifyLiveMcpTools;
21
+ exports.formatMcpContractReport = formatMcpContractReport;
18
22
  exports.inspectMarketplace = inspectMarketplace;
19
23
  exports.expandMarketplace = expandMarketplace;
20
24
  exports.formatScanReport = formatScanReport;
@@ -31,6 +35,7 @@ const linters_js_1 = require("./core/linters.js");
31
35
  const frontmatter_read_js_1 = require("./core/frontmatter-read.js");
32
36
  const description_overlap_js_1 = require("./core/description-overlap.js");
33
37
  const mcp_tool_js_1 = require("./core/mcp-tool.js");
38
+ const mcp_js_1 = require("./core/mcp.js");
34
39
  const mcp_hook_js_1 = require("./core/mcp-hook.js");
35
40
  const agent_runtime_js_1 = require("./adapters/claude-code/agent-runtime.js");
36
41
  const test_coverage_js_1 = require("./test-coverage.js");
@@ -262,10 +267,32 @@ function resolveScript(token, root, pluginRootToken) {
262
267
  // target is INTENTIONAL, not a broken reference. Don't flag scripts in such a
263
268
  // command as MISSING (a false positive caught on gmickel/flow-next's ralph-guard).
264
269
  const EXISTENCE_GUARD = /(?:\[\[?\s*!?\s*-[efsx]\s)|(?:\btest\s+!?\s*-[efsx]\s)/;
270
+ /**
271
+ * A compiled `vigiles/hook` artifact runs through the `hook-runtime run-program`
272
+ * runtime entrypoint; any other hook command is hand-written (a shell script or
273
+ * an inline one-liner) the author maintains directly. The basis for the
274
+ * `prefer-compiled-hooks` nudge.
275
+ */
276
+ function isManagedHookCommand(command) {
277
+ return /\bhook-runtime\b/.test(command);
278
+ }
279
+ /** The `prefer-compiled-hooks` recommendation message (shared by `lint` + `scan`). */
280
+ function preferCompiledHooksMessage(count) {
281
+ return (`${String(count)} hand-written hook command(s) — if any gate the agent ` +
282
+ `(a block/deny decision), compiled hooks (\`vigiles/hook\`) make whole hook ` +
283
+ `bug classes unrepresentable at authoring time, and \`guardrail-check\` proves ` +
284
+ `an existing one blocks. See docs/compiled-hooks.md.`);
285
+ }
265
286
  /** Pull script-file hook commands out of the resolved settings; count inline ones. */
266
287
  function scanHooks(settings, root, pluginRootToken) {
267
288
  const text = JSON.stringify(settings.hooks ?? {});
268
289
  const commands = [...text.matchAll(/"command":\s*"((?:[^"\\]|\\.)*)"/g)].map((m) => m[1]);
290
+ // A hand-written hook is any non-empty command that isn't a vigiles-managed
291
+ // (compiled) hook-runtime invocation — the basis for the prefer-compiled-hooks nudge.
292
+ const manual = commands.filter((c) => {
293
+ const u = c.replace(/\\(.)/g, "$1").trim();
294
+ return u !== "" && !isManagedHookCommand(u);
295
+ }).length;
269
296
  const byScript = new Map();
270
297
  let inline = 0;
271
298
  for (const cmd of commands) {
@@ -287,7 +314,7 @@ function scanHooks(settings, root, pluginRootToken) {
287
314
  }
288
315
  }
289
316
  const hooks = [...byScript.values()].sort((a, b) => a.script.localeCompare(b.script));
290
- return { hooks, inline };
317
+ return { hooks, inline, manual };
291
318
  }
292
319
  // ---------------------------------------------------------------------------
293
320
  // Public API
@@ -484,7 +511,7 @@ function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
484
511
  const lay = layout ?? layout_js_1.claudeCodeLayout;
485
512
  const cls = makeClassifier(lay);
486
513
  const loaded = (0, plugin_loader_js_1.loadPlugin)(dir, lay);
487
- const { hooks, inline } = scanHooks(loaded.settings, (0, node_path_1.resolve)(dir), lay.pluginRootToken);
514
+ const { hooks, inline, manual } = scanHooks(loaded.settings, (0, node_path_1.resolve)(dir), lay.pluginRootToken);
488
515
  // Hook-event keys are a CLOSED platform set — an unrecognized one is a dead
489
516
  // registration (the hook never fires), so flag every unknown (not just typos).
490
517
  // ONLY for the canonical object-keyed-by-event shape: a plugin shipping a
@@ -518,6 +545,7 @@ function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
518
545
  agents,
519
546
  hooks,
520
547
  inlineHooks: inline,
548
+ manualHookCount: manual,
521
549
  commands: Object.keys(loaded.files).filter(cls.isCommand).length,
522
550
  mcp: loaded.warnings.some((w) => w.includes("MCP server")),
523
551
  danglingRefs: (0, plugin_loader_js_2.danglingRefs)((0, node_path_1.resolve)(dir), lay),
@@ -535,6 +563,35 @@ function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
535
563
  puritySummary,
536
564
  };
537
565
  }
566
+ /**
567
+ * LIVE MCP tool resolution for a scanned plugin — the opt-in (`scan --verify-mcp`)
568
+ * dynamic check no static linter can do: it STARTS each declared MCP server and
569
+ * checks every `mcp__server__tool` the plugin's agents reference actually exists on
570
+ * it (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
571
+ * already-computed `report` (its agents' tool lists) + the declared server configs;
572
+ * returns `[]` when the plugin declares no MCP servers (nothing to start). Async +
573
+ * side-effecting (spawns servers) — which is exactly why it's opt-in, not a default
574
+ * lint rule. See `verifyMcpContractTools` (core/mcp.ts).
575
+ */
576
+ async function verifyLiveMcpTools(report, layout, dialect, timeoutMs = 10000) {
577
+ // collectMcpServers yields the raw JSON server entries; a malformed one (no
578
+ // command) just fails to start → server-unreachable (handled), so the cast is safe.
579
+ const servers = collectMcpServers((0, node_path_1.resolve)(report.dir), layout);
580
+ if (Object.keys(servers).length === 0)
581
+ return [];
582
+ const tools = report.agents.flatMap((a) => a.tools ?? []);
583
+ return (0, mcp_js_1.verifyMcpContractTools)(tools, servers, dialect, timeoutMs);
584
+ }
585
+ /** Render the live MCP tool-check result (human-readable). */
586
+ function formatMcpContractReport(errors) {
587
+ if (errors.length === 0) {
588
+ return "Live MCP tool check: every referenced mcp__server__tool resolves ✓";
589
+ }
590
+ const lines = [`Live MCP tool check — ${String(errors.length)} issue(s):`];
591
+ for (const e of errors)
592
+ lines.push(" ✗ " + (0, mcp_js_1.mcpContractToolMessage)(e));
593
+ return lines.join("\n");
594
+ }
538
595
  /**
539
596
  * Read a `marketplace.json` beside the layout's plugin manifest and classify its
540
597
  * members into on-disk vs external. Returns `null` when `dir` is not a
@@ -710,13 +767,18 @@ function formatScanReport(r) {
710
767
  // verdict — it points at the behavioral column, it doesn't fail the scan.
711
768
  const mismatched = r.skills.filter((s) => s.descriptionScript);
712
769
  if (mismatched.length > 0) {
713
- out.push(`⚠ ${String(mismatched.length)} skill(s) have descriptions in an unexpected script (cross-language trigger risk) — measure with \`scan --trigger\``, "");
770
+ out.push(`⚠ ${String(mismatched.length)} skill(s) have descriptions in an unexpected script (cross-language trigger risk) — measure with \`vigiles measure\``, "");
714
771
  }
715
772
  // Skill-metadata is a RECOMMENDATION, not a structural defect (the skill loads
716
773
  // via fallbacks) — reported as a soft note, never counted in the verdict.
717
774
  if (r.skillMetaIssues.length > 0) {
718
775
  out.push(`ℹ ${String(r.skillMetaIssues.length)} skill(s) lack an explicit frontmatter name/description (recommended for a reliable trigger surface) — they still load via fallback`, "");
719
776
  }
777
+ // One discovery nudge toward compiled hooks (never per-hook); the hand-written
778
+ // shell lane stays first-class, so this is a recommendation, not a defect.
779
+ if (r.manualHookCount > 0) {
780
+ out.push(`ℹ ${preferCompiledHooksMessage(r.manualHookCount)}`, "");
781
+ }
720
782
  // Malformed-YAML frontmatter is INFORMATIONAL, not a structural defect: js-yaml
721
783
  // is stricter than some loaders (a colon/quote/<example> in a one-line
722
784
  // description trips it though the file may still load), and the other fields are
@@ -155,7 +155,7 @@ function explainSurface(report, surface) {
155
155
  /** Render explanations for a CLI/report — grouped under the symptom, fix called out. */
156
156
  function formatExplanations(exps) {
157
157
  if (exps.length === 0) {
158
- return "No deterministic cause found — the cause is likely behavioral (measure with `scan --trigger` / an eval).";
158
+ return "No deterministic cause found — the cause is likely behavioral (measure with `vigiles measure` / an eval).";
159
159
  }
160
160
  const lines = [];
161
161
  for (const e of exps) {
@@ -0,0 +1,21 @@
1
+ export interface CommandRefIssue {
2
+ readonly file: string;
3
+ readonly line: number;
4
+ /** The offending invocation, e.g. `vigiles compile-hook`. */
5
+ readonly ref: string;
6
+ readonly reason: string;
7
+ }
8
+ export interface KnownCommands {
9
+ readonly verbs: readonly string[];
10
+ readonly kinds: readonly string[];
11
+ }
12
+ /**
13
+ * Find stale/unknown vigiles command references across the given files. Pure —
14
+ * the caller supplies file contents (so it works over the repo in a test, or any
15
+ * file set).
16
+ */
17
+ export declare function findStaleCommandRefs(files: readonly {
18
+ readonly path: string;
19
+ readonly content: string;
20
+ }[], known?: KnownCommands): CommandRefIssue[];
21
+ //# sourceMappingURL=self-command-refs.d.ts.map