vigiles 8.0.0 → 9.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/optimize.js CHANGED
@@ -1,13 +1,14 @@
1
1
  "use strict";
2
2
  Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.optimize = optimize;
4
+ exports.formatRecommendations = formatRecommendations;
4
5
  exports.formatOptimize = formatOptimize;
5
6
  /**
6
- * The per-repo harness optimizer's DETERMINISTIC spine — shipped as the
7
- * `vigiles scan --fix-plan` lens (NOT its own `optimize` verb: until the measured
8
- * A/B half lands, an "optimizer" that only re-prints scan's findings doesn't earn
9
- * a separate command, so it's folded into scan as one more view on the same
10
- * report; see research/roadmap.md §P2 "reconsider an `optimize` verb").
7
+ * The per-repo harness optimizer's DETERMINISTIC spine — folded INLINE into the
8
+ * default `vigiles audit` report (each finding carries its fix; NOT its own
9
+ * `optimize` verb, and no longer a `--fix-plan` flag: until the measured A/B half
10
+ * lands, an "optimizer" that only re-prints audit's findings doesn't earn a
11
+ * separate surface; see research/roadmap.md §P2 "reconsider an `optimize` verb").
11
12
  *
12
13
  * A2 in the measurement-authority pivot is the ADOPTION product: measure a user's
13
14
  * own skills/model/rules on their tasks and recommend add/drop/swap with a MEASURED
@@ -67,6 +68,27 @@ const ACTION_LABEL = {
67
68
  differentiate: "DIFFERENTIATE",
68
69
  };
69
70
  const measureHint = (dir) => `\`vigiles measure ${dir} --prompts=<file>\` — real-model, runs on your subscription`;
71
+ /**
72
+ * Just the deterministic fix list (no score header) — folded into the default
73
+ * `vigiles audit` report so every finding carries its fix inline (replaces the
74
+ * former `--fix-plan`/`--explain` flags). Empty string when there's nothing to
75
+ * fix (or no loadable surface), so the caller can skip the section entirely.
76
+ */
77
+ function formatRecommendations(rep) {
78
+ if (rep.empty || rep.recommendations.length === 0)
79
+ return "";
80
+ const lines = [
81
+ `${String(rep.recommendations.length)} deterministic fix(es) — free, no model:`,
82
+ "",
83
+ ];
84
+ for (const r of rep.recommendations) {
85
+ const mark = r.confidence === "likely" ? "✗" : "⚠";
86
+ lines.push(`${mark} [${ACTION_LABEL[r.action]}] ${r.surface}`);
87
+ lines.push(` why: ${r.rationale} [${r.detector}]`);
88
+ lines.push(` → ${r.fix}`);
89
+ }
90
+ return lines.join("\n");
91
+ }
70
92
  /** Render an optimization plan for the CLI. */
71
93
  function formatOptimize(rep) {
72
94
  const head = `Harness health: ${String(rep.score)}/100 (${rep.grade}) — ${rep.dir}`;
@@ -1,8 +1,8 @@
1
1
  /**
2
- * `vigiles scan --trigger` — the BEHAVIORAL column of the scan report.
2
+ * `vigiles audit` model trigger tier — the BEHAVIORAL column of the audit report.
3
3
  *
4
4
  * Structural `scan`/`scanPlugin` is deterministic, no-model, CI-free — and stays
5
- * that way. This is the opt-in, model-gated column that stacks on top: for each
5
+ * that way. This is the model-gated column that stacks on top: for each
6
6
  * model-invocable skill in a plugin, it measures how reliably the description
7
7
  * actually FIRES (recall, + precision when irrelevant prompts are supplied),
8
8
  * reusing `measureTriggerRate`. It degrades honestly when the `claude` CLI / auth
@@ -12,6 +12,8 @@
12
12
  * path in prose is undecidable, and the deterministic-input discipline is what
13
13
  * makes the column trustworthy. See `research/plugin-behavioral-findings.md`.
14
14
  */
15
+ import type { PluginLayout } from "./core/layout.js";
16
+ import type { HarnessDialect } from "./core/dialect.js";
15
17
  import { type EvalDriver } from "./eval.js";
16
18
  import { type Trace } from "./harness-test.js";
17
19
  /** Which harness drives the behavioral column (default Claude Code). */
@@ -46,6 +48,10 @@ export interface ProbeOptions {
46
48
  readonly minDistance?: number;
47
49
  /** Which harness to drive (default `"claude-code"`). */
48
50
  readonly harness?: ProbeHarness;
51
+ /** Layout + dialect for candidate discovery — so a Codex repo's skills (under
52
+ * the Codex layout) are found, not silently missed by the default CC layout. */
53
+ readonly layout?: PluginLayout;
54
+ readonly dialect?: HarnessDialect;
49
55
  }
50
56
  /**
51
57
  * Per-harness probe wiring: the eval driver (runner+parse), how to build the
@@ -1,9 +1,9 @@
1
1
  "use strict";
2
2
  /**
3
- * `vigiles scan --trigger` — the BEHAVIORAL column of the scan report.
3
+ * `vigiles audit` model trigger tier — the BEHAVIORAL column of the audit report.
4
4
  *
5
5
  * Structural `scan`/`scanPlugin` is deterministic, no-model, CI-free — and stays
6
- * that way. This is the opt-in, model-gated column that stacks on top: for each
6
+ * that way. This is the model-gated column that stacks on top: for each
7
7
  * model-invocable skill in a plugin, it measures how reliably the description
8
8
  * actually FIRES (recall, + precision when irrelevant prompts are supplied),
9
9
  * reusing `measureTriggerRate`. It degrades honestly when the `claude` CLI / auth
@@ -131,8 +131,10 @@ async function probeSkill(ctx, name, ps) {
131
131
  async function probePluginTriggersWith(dir, promptSet, probe, opts = {}) {
132
132
  const ctx = { dir, opts, probe };
133
133
  // Only model-invocable, describable skills can auto-trigger; user-invoked and
134
- // description-less ones can't, so they're not behavioral candidates.
135
- const candidates = (0, scan_js_1.scanPlugin)(dir).skills.filter((s) => !s.userInvoked && s.hasDescription);
134
+ // description-less ones can't, so they're not behavioral candidates. Discover
135
+ // them with the resolved layout/dialect (default CC) so a Codex repo's skills
136
+ // aren't missed by the wrong layout.
137
+ const candidates = (0, scan_js_1.scanPlugin)(dir, opts.layout, opts.dialect).skills.filter((s) => !s.userInvoked && s.hasDescription);
136
138
  const results = [];
137
139
  for (const s of candidates) {
138
140
  const ps = promptSet[s.name];
@@ -1,10 +1,15 @@
1
1
  /**
2
- * scan → trigger-tier nudge. When a plugin ships model-invocable skills AND a
3
- * real model is reachable, surface the (real-model) `scan --trigger` tier that
4
- * measures whether those skills actually FIRE (recall + precision). Pure
5
- * decision + helpers; the IO (prompt / scaffold write / running the measure)
6
- * lives in the CLI. Honors `great-agent-flow`: an agent (non-TTY / `--json` /
7
- * `--no-interactive`) is HINTED, never prompted — a `scan` must never hang.
2
+ * audit → the ONE read-vs-run decision. `audit` is a Lighthouse-style LOCAL
3
+ * report: a deterministic READ by default — safe + identical on every OS, nothing
4
+ * executes — NOT a CI step (CI uses `vigiles lint`, the deterministic gate). The
5
+ * executing checks (safety battery, live MCP resolution, skill-firing trigger-rate)
6
+ * run ONLY when there's a human to consent: at a TTY `audit` ASKS once (and
7
+ * remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
8
+ * `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
9
+ * never silently executes. There is deliberately NO execution flag: automation
10
+ * tests the harness through the `vigiles/testing` API + skills (the layered tiers),
11
+ * not through the report verb. The IO (prompt / run / remember) lives in the CLI;
12
+ * this is the pure decision + helpers.
8
13
  */
9
14
  /** Only the env vars that signal a reachable model (parse, don't validate). */
10
15
  export interface ModelEnv {
@@ -13,42 +18,74 @@ export interface ModelEnv {
13
18
  readonly CLAUDE_CODE_ENTRYPOINT?: string;
14
19
  }
15
20
  /**
16
- * Is a real model reachable for the eval / trigger tier? Either a metered API
17
- * key (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session
18
- * (`CLAUDECODE=1` / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter
19
- * drives the `claude` CLI on the user's subscription, no key needed. Keeping
20
- * this a tiny, env-only predicate (not a live probe) means it never spends a
21
- * token just to decide whether to suggest spending one.
21
+ * Is a real model reachable for the trigger tier? Either a metered API key
22
+ * (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session (`CLAUDECODE=1`
23
+ * / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter drives the `claude`
24
+ * CLI on the user's subscription, no key needed and $0 metered. A tiny env-only
25
+ * predicate (not a live probe), so it never spends a token just to decide.
22
26
  */
23
27
  export declare function hasModelAccess(env: ModelEnv): boolean;
24
- export type TriggerSuggestion = "prompt" | "hint" | "none";
25
- export interface SuggestOpts {
26
- /** A real model is reachable (`hasModelAccess`). */
27
- readonly modelAccess: boolean;
28
- /** stdout is an interactive terminal (a human is watching). */
28
+ /**
29
+ * Is the reachable model METERED (a paid API key) rather than a subscription?
30
+ * Only affects the consent DISCLOSURE wording (a metered key bills per token; a
31
+ * subscription is $0 metered) — the run/skip decision itself is consent-driven,
32
+ * not metered-driven.
33
+ */
34
+ export declare function isMeteredAccess(env: ModelEnv): boolean;
35
+ /** Why the executing checks were skipped (drives the "not run" nudge). */
36
+ export type ExecuteSkipReason = "nothing" | "headless" | "remembered-no";
37
+ /**
38
+ * What `audit` should do about the EXECUTING checks (battery + live MCP +
39
+ * trigger-rate), as ONE bundle:
40
+ * - `run` — run them now (a remembered yes).
41
+ * - `ask` — interactive human + something to run + no sticky choice: ask once,
42
+ * then remember.
43
+ * - `skip` — stay a deterministic read; the `reason` drives a one-line nudge.
44
+ */
45
+ export type ExecuteDecision = {
46
+ readonly kind: "run";
47
+ } | {
48
+ readonly kind: "ask";
49
+ } | {
50
+ readonly kind: "skip";
51
+ readonly reason: ExecuteSkipReason;
52
+ };
53
+ export interface ExecuteEnv {
54
+ /** Is there ANY executable surface — runnable hooks, an own-repo MCP server, or
55
+ * a model-invocable skill? Nothing to run → never ask, never nudge. */
56
+ readonly hasExecutable: boolean;
57
+ /** Both stdin AND stdout are a terminal (a human who can answer + wait). */
29
58
  readonly isTTY: boolean;
30
- /** Count of model-invocable, described skills worth measuring. */
31
- readonly triggerableSkills: number;
32
- /** `--json` — machine output; never decorate or prompt. */
59
+ /** `--json` — machine output; stays a read even at a TTY (never prompt). */
33
60
  readonly json: boolean;
34
- /** `--no-interactive` — explicit agent/CI mode; hint, never prompt. */
61
+ /** `--no-interactive` / `--yes` — explicit agent/CI mode (never prompt). */
35
62
  readonly noInteractive: boolean;
63
+ /** Sticky remembered choice from `.vigilesrc.json` (`audit.measure`), or undefined. */
64
+ readonly remembered?: boolean;
36
65
  }
37
66
  /**
38
- * Decide how `scan` surfaces the trigger tier after its report:
39
- * - `"none"` — no model-invocable skills, no model access, or `--json`.
40
- * - `"hint"` — model access + skills but non-interactive (agent / CI / non-TTY
41
- * / `--no-interactive`): a one-line, non-blocking hint.
42
- * - `"prompt"` — model access + skills + a human at a TTY: offer to set it up.
67
+ * Decide what `audit` does with the executing checks. Total + pure; the first
68
+ * matching rule wins. There is NO execution flag — `audit` is a local report, so
69
+ * the executing checks need a human to consent:
70
+ * 1. nothing executable → skip "nothing" (a clean read; no nudge)
71
+ * 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
72
+ * skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
73
+ * 3. sticky no → skip "remembered-no"
74
+ * 4. sticky yes → run
75
+ * 5. interactive human, no sticky choice → ask (then remember)
76
+ */
77
+ export declare function decideExecute(o: ExecuteEnv): ExecuteDecision;
78
+ /**
79
+ * The one-line "executing checks not run" nudge for a skipped read (the
80
+ * no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
81
+ * gap). There is no flag to point at — `audit` runs them only interactively, and
82
+ * automation uses the `vigiles/testing` API.
43
83
  */
44
- export declare function decideTriggerSuggestion(o: SuggestOpts): TriggerSuggestion;
45
- /** The one-line, non-blocking hint (the agent/CI surface). */
46
- export declare function formatTriggerHint(dir: string, triggerableSkills: number): string;
84
+ export declare function formatExecuteSkip(reason: ExecuteSkipReason): string | null;
47
85
  /**
48
86
  * A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
49
87
  * → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
50
- * placeholders the user replaces with real requests. Deterministic; written
51
- * only on an explicit human "yes" so a plain `scan` never spends a token.
88
+ * placeholders the user replaces with real requests.
52
89
  */
53
90
  export declare function scaffoldTriggerPrompts(skillNames: readonly string[]): string;
54
91
  //# sourceMappingURL=scan-trigger-suggest.d.ts.map
@@ -1,24 +1,29 @@
1
1
  "use strict";
2
2
  /**
3
- * scan → trigger-tier nudge. When a plugin ships model-invocable skills AND a
4
- * real model is reachable, surface the (real-model) `scan --trigger` tier that
5
- * measures whether those skills actually FIRE (recall + precision). Pure
6
- * decision + helpers; the IO (prompt / scaffold write / running the measure)
7
- * lives in the CLI. Honors `great-agent-flow`: an agent (non-TTY / `--json` /
8
- * `--no-interactive`) is HINTED, never prompted — a `scan` must never hang.
3
+ * audit → the ONE read-vs-run decision. `audit` is a Lighthouse-style LOCAL
4
+ * report: a deterministic READ by default — safe + identical on every OS, nothing
5
+ * executes — NOT a CI step (CI uses `vigiles lint`, the deterministic gate). The
6
+ * executing checks (safety battery, live MCP resolution, skill-firing trigger-rate)
7
+ * run ONLY when there's a human to consent: at a TTY `audit` ASKS once (and
8
+ * remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
9
+ * `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
10
+ * never silently executes. There is deliberately NO execution flag: automation
11
+ * tests the harness through the `vigiles/testing` API + skills (the layered tiers),
12
+ * not through the report verb. The IO (prompt / run / remember) lives in the CLI;
13
+ * this is the pure decision + helpers.
9
14
  */
10
15
  Object.defineProperty(exports, "__esModule", { value: true });
11
16
  exports.hasModelAccess = hasModelAccess;
12
- exports.decideTriggerSuggestion = decideTriggerSuggestion;
13
- exports.formatTriggerHint = formatTriggerHint;
17
+ exports.isMeteredAccess = isMeteredAccess;
18
+ exports.decideExecute = decideExecute;
19
+ exports.formatExecuteSkip = formatExecuteSkip;
14
20
  exports.scaffoldTriggerPrompts = scaffoldTriggerPrompts;
15
21
  /**
16
- * Is a real model reachable for the eval / trigger tier? Either a metered API
17
- * key (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session
18
- * (`CLAUDECODE=1` / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter
19
- * drives the `claude` CLI on the user's subscription, no key needed. Keeping
20
- * this a tiny, env-only predicate (not a live probe) means it never spends a
21
- * token just to decide whether to suggest spending one.
22
+ * Is a real model reachable for the trigger tier? Either a metered API key
23
+ * (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session (`CLAUDECODE=1`
24
+ * / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter drives the `claude`
25
+ * CLI on the user's subscription, no key needed and $0 metered. A tiny env-only
26
+ * predicate (not a live probe), so it never spends a token just to decide.
22
27
  */
23
28
  function hasModelAccess(env) {
24
29
  return Boolean(env.ANTHROPIC_API_KEY ||
@@ -26,31 +31,59 @@ function hasModelAccess(env) {
26
31
  env.CLAUDE_CODE_ENTRYPOINT);
27
32
  }
28
33
  /**
29
- * Decide how `scan` surfaces the trigger tier after its report:
30
- * - `"none"` — no model-invocable skills, no model access, or `--json`.
31
- * - `"hint"` — model access + skills but non-interactive (agent / CI / non-TTY
32
- * / `--no-interactive`): a one-line, non-blocking hint.
33
- * - `"prompt"` — model access + skills + a human at a TTY: offer to set it up.
34
+ * Is the reachable model METERED (a paid API key) rather than a subscription?
35
+ * Only affects the consent DISCLOSURE wording (a metered key bills per token; a
36
+ * subscription is $0 metered) — the run/skip decision itself is consent-driven,
37
+ * not metered-driven.
34
38
  */
35
- function decideTriggerSuggestion(o) {
36
- if (o.json || o.triggerableSkills < 1 || !o.modelAccess)
37
- return "none";
38
- if (o.noInteractive || !o.isTTY)
39
- return "hint";
40
- return "prompt";
39
+ function isMeteredAccess(env) {
40
+ return Boolean(env.ANTHROPIC_API_KEY);
41
41
  }
42
- /** The one-line, non-blocking hint (the agent/CI surface). */
43
- function formatTriggerHint(dir, triggerableSkills) {
44
- const n = triggerableSkills;
45
- return (`ℹ ${String(n)} model-invocable skill${n === 1 ? "" : "s"} + model access detected — ` +
46
- `measure whether they actually FIRE (recall + precision) with:\n` +
47
- ` vigiles scan ${dir} --trigger --prompts=trigger-prompts.json`);
42
+ /**
43
+ * Decide what `audit` does with the executing checks. Total + pure; the first
44
+ * matching rule wins. There is NO execution flag — `audit` is a local report, so
45
+ * the executing checks need a human to consent:
46
+ * 1. nothing executable → skip "nothing" (a clean read; no nudge)
47
+ * 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
48
+ * skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
49
+ * 3. sticky no → skip "remembered-no"
50
+ * 4. sticky yes → run
51
+ * 5. interactive human, no sticky choice → ask (then remember)
52
+ */
53
+ function decideExecute(o) {
54
+ if (!o.hasExecutable)
55
+ return { kind: "skip", reason: "nothing" };
56
+ if (o.json || o.noInteractive || !o.isTTY)
57
+ return { kind: "skip", reason: "headless" };
58
+ if (o.remembered === false)
59
+ return { kind: "skip", reason: "remembered-no" };
60
+ if (o.remembered === true)
61
+ return { kind: "run" };
62
+ return { kind: "ask" };
63
+ }
64
+ /**
65
+ * The one-line "executing checks not run" nudge for a skipped read (the
66
+ * no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
67
+ * gap). There is no flag to point at — `audit` runs them only interactively, and
68
+ * automation uses the `vigiles/testing` API.
69
+ */
70
+ function formatExecuteSkip(reason) {
71
+ switch (reason) {
72
+ case "nothing":
73
+ return null;
74
+ case "headless":
75
+ return ("\nℹ Executing checks (safety battery · live MCP · skill firing) skipped — " +
76
+ "`audit` runs them only interactively (a terminal). For automation, test the " +
77
+ "harness with the `vigiles/testing` API.");
78
+ case "remembered-no":
79
+ return ("\nℹ Executing checks not run (you disabled them — edit .vigilesrc.json " +
80
+ "`audit.measure` to re-enable).");
81
+ }
48
82
  }
49
83
  /**
50
84
  * A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
51
85
  * → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
52
- * placeholders the user replaces with real requests. Deterministic; written
53
- * only on an explicit human "yes" so a plain `scan` never spends a token.
86
+ * placeholders the user replaces with real requests.
54
87
  */
55
88
  function scaffoldTriggerPrompts(skillNames) {
56
89
  const obj = {};
package/dist/scan.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  /**
2
- * `vigiles scan <dir>` — point vigiles at any plugin/repo and see what it ships
2
+ * `vigiles audit <dir>` — point vigiles at any plugin/repo and see what it ships
3
3
  * and what's broken, with **no model and no API key**.
4
4
  *
5
5
  * This is the deterministic substrate under the plugin/skill leaderboard
@@ -27,14 +27,20 @@ export interface ScanSkill {
27
27
  readonly name: string;
28
28
  readonly path: string;
29
29
  readonly hasDescription: boolean;
30
+ /**
31
+ * The skill's effective description (frontmatter `description`, else the first
32
+ * body paragraph — the same text the selector keys on), trimmed; `undefined`
33
+ * when neither exists. Feeds the model trigger tier's auto-generated probes.
34
+ */
35
+ readonly description?: string;
30
36
  readonly userInvoked: boolean;
31
37
  /**
32
38
  * The description's dominant script when it DIFFERS from the expected one
33
39
  * (default `"Latin"`), else null. The model's skill-selection context is
34
40
  * English-centric, so a description in another script carries a cross-language
35
41
  * trigger risk — it may under-fire on English prompts. A RISK flag, not a
36
- * defect (a language-matched audience is fine); measure the real gap with
37
- * `scan --trigger`.
42
+ * defect (a language-matched audience is fine); measure the real gap with the
43
+ * `audit` trigger tier / `measureTriggerRate`.
38
44
  */
39
45
  readonly descriptionScript: Script | null;
40
46
  }
@@ -86,8 +92,25 @@ export interface FrontmatterValueIssue {
86
92
  /** ok = file present; missing = referenced but absent; unresolved = path still has an unexpanded var, can't check. */
87
93
  export type HookStatus = "ok" | "missing" | "unresolved";
88
94
  export interface ScanHook {
95
+ /**
96
+ * The full hook command as it would be run (plugin-root token expanded, shell
97
+ * quotes stripped). Present on script-based hooks; empty string on hooks whose
98
+ * command is entirely inline (no script file) — but inline hooks never appear
99
+ * in `hooks[]`, they are counted by `inlineHooks`, so in practice `command` is
100
+ * always non-empty when a `ScanHook` is in the list.
101
+ */
102
+ readonly command: string;
89
103
  readonly script: string;
90
104
  readonly status: HookStatus;
105
+ /**
106
+ * The hook EVENT this script is registered under (`PreToolUse`, `PostToolUse`,
107
+ * `SessionStart`, …), when it can be determined from the canonical
108
+ * object-keyed-by-event settings shape; `undefined` for a non-object/array
109
+ * config. The safety battery uses it to test only the blocking-capable
110
+ * `PreToolUse` guards — so a `SessionStart`/`PostToolUse` hook isn't unfairly
111
+ * scored against "does it block rm -rf".
112
+ */
113
+ readonly event?: string;
91
114
  }
92
115
  /**
93
116
  * The repo's top-level instruction file (`CLAUDE.md` / `AGENTS.md`), if present.
@@ -194,14 +217,16 @@ export declare function preferCompiledHooksMessage(count: number): string;
194
217
  /** Scan a plugin/repo directory and report its surfaces + structural issues. */
195
218
  export declare function scanPlugin(dir: string, layout?: PluginLayout, dialect?: HarnessDialect): ScanReport;
196
219
  /**
197
- * LIVE MCP tool resolution for a scanned plugin — the opt-in (`scan --verify-mcp`)
198
- * dynamic check no static linter can do: it STARTS each declared MCP server and
199
- * checks every `mcp__server__tool` the plugin's agents reference actually exists on
200
- * it (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
220
+ * LIVE MCP tool resolution for a scanned plugin — the dynamic check no static
221
+ * linter can do: it STARTS each declared MCP server and checks every
222
+ * `mcp__server__tool` the plugin's agents reference actually exists on it
223
+ * (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
201
224
  * already-computed `report` (its agents' tool lists) + the declared server configs;
202
225
  * returns `[]` when the plugin declares no MCP servers (nothing to start). Async +
203
- * side-effecting (spawns servers) — which is exactly why it's opt-in, not a default
204
- * lint rule. See `verifyMcpContractTools` (core/mcp.ts).
226
+ * side-effecting (spawns servers) — so `audit` runs it by default only for the
227
+ * user's OWN repo (own-repo, like running your own tools); a FOREIGN plugin's
228
+ * servers are never spawned, and `--fast` opts out. See `verifyMcpContractTools`
229
+ * (core/mcp.ts).
205
230
  */
206
231
  export declare function verifyLiveMcpTools(report: ScanReport, layout: PluginLayout, dialect: HarnessDialect, timeoutMs?: number): Promise<McpContractToolError[]>;
207
232
  /** Render the live MCP tool-check result (human-readable). */
@@ -227,7 +252,7 @@ export interface MarketplaceInfo {
227
252
  * Read a `marketplace.json` beside the layout's plugin manifest and classify its
228
253
  * members into on-disk vs external. Returns `null` when `dir` is not a
229
254
  * marketplace. The source of truth behind {@link expandMarketplace} and the
230
- * curated-marketplace report in `vigiles scan`.
255
+ * curated-marketplace report in `vigiles audit`.
231
256
  */
232
257
  export declare function inspectMarketplace(dir: string, layout?: PluginLayout): MarketplaceInfo | null;
233
258
  /**
@@ -235,7 +260,7 @@ export declare function inspectMarketplace(dir: string, layout?: PluginLayout):
235
260
  * plugin manifest, e.g. `.claude-plugin/marketplace.json`), expand it into the
236
261
  * absolute dirs of its member plugins. Returns `null` when there's no
237
262
  * marketplace, `[]` when it's a marketplace whose members are all external (not
238
- * on disk). Used by `vigiles scan` to rank a whole marketplace — wshobson/agents
263
+ * on disk). Used by `vigiles audit` to rank a whole marketplace — wshobson/agents
239
264
  * alone ships 80+ plugins under one `marketplace.json`. See {@link inspectMarketplace}.
240
265
  */
241
266
  export declare function expandMarketplace(dir: string, layout?: PluginLayout): string[] | null;
package/dist/scan.js CHANGED
@@ -1,6 +1,6 @@
1
1
  "use strict";
2
2
  /**
3
- * `vigiles scan <dir>` — point vigiles at any plugin/repo and see what it ships
3
+ * `vigiles audit <dir>` — point vigiles at any plugin/repo and see what it ships
4
4
  * and what's broken, with **no model and no API key**.
5
5
  *
6
6
  * This is the deterministic substrate under the plugin/skill leaderboard
@@ -69,9 +69,21 @@ function makeClassifier(layout) {
69
69
  const skillRe = skill ? new RegExp(`${skill}[^/]+/SKILL\\.md$`) : null;
70
70
  const agentRe = agent ? new RegExp(`${agent}[^/]+\\.md$`) : null;
71
71
  const commandRe = command ? new RegExp(`${command}.+\\.md$`) : null;
72
+ // A subagent lives at the plugin's TOP-LEVEL `agents/` dir, never recursively
73
+ // under a skill (`skills/<x>/agents/`). Those nested files are skill-internal
74
+ // worker docs (e.g. Anthropic's own skill-creator), NOT dispatchable Claude
75
+ // Code subagents — flagging them is a false positive. The agent dir nested
76
+ // under the skill dir is excluded; a genuine top-level `agents/foo.md` still
77
+ // matches. See scan.test.ts for the regression.
78
+ const nestedAgentRe = layout.skillDir && layout.agentDir
79
+ ? new RegExp(`(?:^|/)${escapeRe(layout.skillDir)}/.+/${escapeRe(layout.agentDir)}/`)
80
+ : null;
81
+ const isAgent = (f) => (agentRe?.test(f) ?? false) &&
82
+ !f.endsWith(".spec.ts") &&
83
+ !(nestedAgentRe?.test(f) ?? false);
72
84
  return {
73
85
  isSkill: (f) => skillRe?.test(f) ?? false,
74
- isAgent: (f) => (agentRe?.test(f) ?? false) && !f.endsWith(".spec.ts"),
86
+ isAgent,
75
87
  isCommand: (f) => commandRe?.test(f) ?? false,
76
88
  };
77
89
  }
@@ -168,6 +180,7 @@ function scanSkills(files, cls) {
168
180
  name: fm.name ?? skillName(path),
169
181
  path,
170
182
  hasDescription: Boolean(effectiveDesc && effectiveDesc.length >= 20),
183
+ description: effectiveDesc?.trim(),
171
184
  userInvoked: /^\s*disable-model-invocation:\s*true\s*$/m.test(md),
172
185
  descriptionScript: effectiveDesc ? unexpectedScript(effectiveDesc) : null,
173
186
  });
@@ -244,7 +257,7 @@ function scanAgents(files, dialect, declaredServers, cls) {
244
257
  * quotes. A token that still carries any `$VAR` after that is genuinely
245
258
  * uncheckable.
246
259
  */
247
- function resolveScript(token, root, pluginRootToken) {
260
+ function resolveScript(token, root, pluginRootToken, fullCommand) {
248
261
  // "${CLAUDE_PLUGIN_ROOT}" → unbraced "$CLAUDE_PLUGIN_ROOT".
249
262
  const unbraced = pluginRootToken.replace(/^\$\{(.+)\}$/, "$$$1");
250
263
  const cleaned = token
@@ -252,14 +265,23 @@ function resolveScript(token, root, pluginRootToken) {
252
265
  .replaceAll(pluginRootToken, root)
253
266
  .replaceAll(unbraced, root);
254
267
  if (cleaned.includes("$"))
255
- return { script: token, status: "unresolved" };
268
+ return { command: fullCommand, script: token, status: "unresolved" };
256
269
  // A relative hook path (`./hooks/x.sh`, `scripts/x.py`) is the plugin's own —
257
270
  // resolve it against the PLUGIN ROOT, not the scanner's cwd. Without this, a
258
271
  // plugin that references `./hooks/x.sh` (the file IS present) was reported
259
272
  // MISSING because existsSync() checked cwd-relative (a false positive caught on
260
273
  // ananddtyagi/cc-marketplace). The displayed `script` stays as the author wrote it.
261
274
  const abs = (0, node_path_1.isAbsolute)(cleaned) ? cleaned : (0, node_path_1.resolve)(root, cleaned);
262
- return { script: cleaned, status: (0, node_fs_1.existsSync)(abs) ? "ok" : "missing" };
275
+ // Resolve the full command the same way we resolve the script token (expand
276
+ // plugin-root, strip outer quotes) so the CLI can pass it to verifyGuardrail.
277
+ const resolvedCommand = fullCommand
278
+ .replaceAll(pluginRootToken, root)
279
+ .replaceAll(unbraced, root);
280
+ return {
281
+ command: resolvedCommand,
282
+ script: cleaned,
283
+ status: (0, node_fs_1.existsSync)(abs) ? "ok" : "missing",
284
+ };
263
285
  }
264
286
  // A shell existence guard around a command — `[ ! -f x ] || x`, `[ -f x ] && x`,
265
287
  // `test -f x && …`. Authors use it to make a hook OPTIONAL (run the script only
@@ -284,9 +306,42 @@ function preferCompiledHooksMessage(count) {
284
306
  `an existing one blocks. See docs/compiled-hooks.md.`);
285
307
  }
286
308
  /** Pull script-file hook commands out of the resolved settings; count inline ones. */
309
+ /**
310
+ * Best-effort map of each script token → the hook EVENT it's registered under,
311
+ * by walking the canonical object-keyed-by-event settings shape
312
+ * (`{ PreToolUse: [{ hooks: [{ command }] }], … }`). Lets the safety battery
313
+ * scope itself to `PreToolUse` (the only event that can block a tool call), so a
314
+ * `SessionStart`/`PostToolUse`/`Stop` hook isn't tested against the disaster
315
+ * catalog. Returns an empty map for a non-object/array config (event → unknown).
316
+ */
317
+ function eventsByScript(hooks) {
318
+ const map = new Map();
319
+ if (!hooks || typeof hooks !== "object" || Array.isArray(hooks))
320
+ return map;
321
+ for (const [event, arr] of Object.entries(hooks)) {
322
+ if (!Array.isArray(arr))
323
+ continue;
324
+ for (const entry of arr) {
325
+ const hookList = entry.hooks;
326
+ if (!Array.isArray(hookList))
327
+ continue;
328
+ for (const h of hookList) {
329
+ const cmd = h.command;
330
+ if (typeof cmd !== "string")
331
+ continue;
332
+ for (const tok of cmd.match(SCRIPT_RE) ?? []) {
333
+ if (!map.has(tok))
334
+ map.set(tok, event);
335
+ }
336
+ }
337
+ }
338
+ }
339
+ return map;
340
+ }
287
341
  function scanHooks(settings, root, pluginRootToken) {
288
342
  const text = JSON.stringify(settings.hooks ?? {});
289
343
  const commands = [...text.matchAll(/"command":\s*"((?:[^"\\]|\\.)*)"/g)].map((m) => m[1]);
344
+ const evMap = eventsByScript(settings.hooks);
290
345
  // A hand-written hook is any non-empty command that isn't a vigiles-managed
291
346
  // (compiled) hook-runtime invocation — the basis for the prefer-compiled-hooks nudge.
292
347
  const manual = commands.filter((c) => {
@@ -309,8 +364,9 @@ function scanHooks(settings, root, pluginRootToken) {
309
364
  continue;
310
365
  }
311
366
  for (const tok of found) {
312
- const hook = resolveScript(tok, root, pluginRootToken);
313
- byScript.set(hook.script, hook);
367
+ const hook = resolveScript(tok, root, pluginRootToken, unescaped);
368
+ const event = evMap.get(tok);
369
+ byScript.set(hook.script, event ? { ...hook, event } : hook);
314
370
  }
315
371
  }
316
372
  const hooks = [...byScript.values()].sort((a, b) => a.script.localeCompare(b.script));
@@ -564,14 +620,16 @@ function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
564
620
  };
565
621
  }
566
622
  /**
567
- * LIVE MCP tool resolution for a scanned plugin — the opt-in (`scan --verify-mcp`)
568
- * dynamic check no static linter can do: it STARTS each declared MCP server and
569
- * checks every `mcp__server__tool` the plugin's agents reference actually exists on
570
- * it (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
623
+ * LIVE MCP tool resolution for a scanned plugin — the dynamic check no static
624
+ * linter can do: it STARTS each declared MCP server and checks every
625
+ * `mcp__server__tool` the plugin's agents reference actually exists on it
626
+ * (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
571
627
  * already-computed `report` (its agents' tool lists) + the declared server configs;
572
628
  * returns `[]` when the plugin declares no MCP servers (nothing to start). Async +
573
- * side-effecting (spawns servers) — which is exactly why it's opt-in, not a default
574
- * lint rule. See `verifyMcpContractTools` (core/mcp.ts).
629
+ * side-effecting (spawns servers) — so `audit` runs it by default only for the
630
+ * user's OWN repo (own-repo, like running your own tools); a FOREIGN plugin's
631
+ * servers are never spawned, and `--fast` opts out. See `verifyMcpContractTools`
632
+ * (core/mcp.ts).
575
633
  */
576
634
  async function verifyLiveMcpTools(report, layout, dialect, timeoutMs = 10000) {
577
635
  // collectMcpServers yields the raw JSON server entries; a malformed one (no
@@ -596,7 +654,7 @@ function formatMcpContractReport(errors) {
596
654
  * Read a `marketplace.json` beside the layout's plugin manifest and classify its
597
655
  * members into on-disk vs external. Returns `null` when `dir` is not a
598
656
  * marketplace. The source of truth behind {@link expandMarketplace} and the
599
- * curated-marketplace report in `vigiles scan`.
657
+ * curated-marketplace report in `vigiles audit`.
600
658
  */
601
659
  function inspectMarketplace(dir, layout = layout_js_1.claudeCodeLayout) {
602
660
  const mpPath = (0, node_path_1.join)(dir, (0, node_path_1.dirname)(layout.manifestPath), "marketplace.json");
@@ -648,7 +706,7 @@ function inspectMarketplace(dir, layout = layout_js_1.claudeCodeLayout) {
648
706
  * plugin manifest, e.g. `.claude-plugin/marketplace.json`), expand it into the
649
707
  * absolute dirs of its member plugins. Returns `null` when there's no
650
708
  * marketplace, `[]` when it's a marketplace whose members are all external (not
651
- * on disk). Used by `vigiles scan` to rank a whole marketplace — wshobson/agents
709
+ * on disk). Used by `vigiles audit` to rank a whole marketplace — wshobson/agents
652
710
  * alone ships 80+ plugins under one `marketplace.json`. See {@link inspectMarketplace}.
653
711
  */
654
712
  function expandMarketplace(dir, layout = layout_js_1.claudeCodeLayout) {
@@ -35,7 +35,7 @@ export type BehavioralSymptom = "wrong-skill-fires" | "skill-never-fires" | "age
35
35
  * never-available tool can't be called); the cause is near-certain.
36
36
  * - `"possible"` — a high-precision PROXY for a behavioral risk (a description
37
37
  * overlap / a foreign-script description); deterministic to detect, but whether
38
- * it actually moved behaviour is confirmed by `scan --trigger`.
38
+ * it actually moved behaviour is confirmed by the `audit` trigger tier.
39
39
  */
40
40
  export type ExplanationConfidence = "likely" | "possible";
41
41
  export interface ScoreExplanation {