vigiles 8.0.0 → 9.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +63 -35
- package/dist/adoptability.d.ts +55 -0
- package/dist/adoptability.js +196 -0
- package/dist/audit-html.d.ts +20 -0
- package/dist/audit-html.js +61 -0
- package/dist/audit-prompts.d.ts +46 -0
- package/dist/audit-prompts.js +90 -0
- package/dist/audit-report.d.ts +70 -0
- package/dist/audit-report.js +51 -0
- package/dist/audit-report.template.html +110 -0
- package/dist/audit-score.d.ts +44 -0
- package/dist/audit-score.js +221 -0
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +1 -1
- package/dist/cli.js +340 -111
- package/dist/core/inline.js +11 -1
- package/dist/core/types.d.ts +12 -0
- package/dist/dialect-drift.js +1 -1
- package/dist/eval.d.ts +1 -1
- package/dist/eval.js +1 -1
- package/dist/optimize.d.ts +12 -5
- package/dist/optimize.js +27 -5
- package/dist/scan-behavioral.d.ts +8 -2
- package/dist/scan-behavioral.js +6 -4
- package/dist/scan-trigger-suggest.d.ts +68 -31
- package/dist/scan-trigger-suggest.js +66 -33
- package/dist/scan.d.ts +36 -11
- package/dist/scan.js +60 -14
- package/dist/score-explainer.d.ts +1 -1
- package/package.json +4 -2
- package/skills/test-harness/SKILL.md +1 -1
package/dist/eval.d.ts
CHANGED
|
@@ -251,7 +251,7 @@ export type AgentRunner = (args: AgentRunArgs) => Promise<RunOut>;
|
|
|
251
251
|
*/
|
|
252
252
|
export declare function resolveSpawnEnv(a: Pick<AgentRunArgs, "env" | "replaceEnv">, base?: NodeJS.ProcessEnv): NodeJS.ProcessEnv;
|
|
253
253
|
/** The real `claude`-spawning runner (composition root). Exported so other
|
|
254
|
-
* real-model entries (e.g. `
|
|
254
|
+
* real-model entries (e.g. the `audit` trigger tier) bind the same runner. */
|
|
255
255
|
export declare function spawnAgent(a: AgentRunArgs): Promise<RunOut>;
|
|
256
256
|
/**
|
|
257
257
|
* Run the eval: every arm × every trial against the real `claude` CLI, with the
|
package/dist/eval.js
CHANGED
|
@@ -96,7 +96,7 @@ function resolveSpawnEnv(a, base = process.env) {
|
|
|
96
96
|
}
|
|
97
97
|
/* v8 ignore start -- real claude subprocess; exercised by bench/, not the unit gate */
|
|
98
98
|
/** The real `claude`-spawning runner (composition root). Exported so other
|
|
99
|
-
* real-model entries (e.g. `
|
|
99
|
+
* real-model entries (e.g. the `audit` trigger tier) bind the same runner. */
|
|
100
100
|
function spawnAgent(a) {
|
|
101
101
|
return new Promise((resolvePromise) => {
|
|
102
102
|
const args = [
|
package/dist/optimize.d.ts
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The per-repo harness optimizer's DETERMINISTIC spine —
|
|
3
|
-
* `vigiles
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
2
|
+
* The per-repo harness optimizer's DETERMINISTIC spine — folded INLINE into the
|
|
3
|
+
* default `vigiles audit` report (each finding carries its fix; NOT its own
|
|
4
|
+
* `optimize` verb, and no longer a `--fix-plan` flag: until the measured A/B half
|
|
5
|
+
* lands, an "optimizer" that only re-prints audit's findings doesn't earn a
|
|
6
|
+
* separate surface; see research/roadmap.md §P2 "reconsider an `optimize` verb").
|
|
7
7
|
*
|
|
8
8
|
* A2 in the measurement-authority pivot is the ADOPTION product: measure a user's
|
|
9
9
|
* own skills/model/rules on their tasks and recommend add/drop/swap with a MEASURED
|
|
@@ -69,6 +69,13 @@ export interface OptimizeReport {
|
|
|
69
69
|
* before `possible` proxies, via explainScore's own ordering). Pure over the report.
|
|
70
70
|
*/
|
|
71
71
|
export declare function optimize(report: ScanReport): OptimizeReport;
|
|
72
|
+
/**
|
|
73
|
+
* Just the deterministic fix list (no score header) — folded into the default
|
|
74
|
+
* `vigiles audit` report so every finding carries its fix inline (replaces the
|
|
75
|
+
* former `--fix-plan`/`--explain` flags). Empty string when there's nothing to
|
|
76
|
+
* fix (or no loadable surface), so the caller can skip the section entirely.
|
|
77
|
+
*/
|
|
78
|
+
export declare function formatRecommendations(rep: OptimizeReport): string;
|
|
72
79
|
/** Render an optimization plan for the CLI. */
|
|
73
80
|
export declare function formatOptimize(rep: OptimizeReport): string;
|
|
74
81
|
//# sourceMappingURL=optimize.d.ts.map
|
package/dist/optimize.js
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.optimize = optimize;
|
|
4
|
+
exports.formatRecommendations = formatRecommendations;
|
|
4
5
|
exports.formatOptimize = formatOptimize;
|
|
5
6
|
/**
|
|
6
|
-
* The per-repo harness optimizer's DETERMINISTIC spine —
|
|
7
|
-
* `vigiles
|
|
8
|
-
*
|
|
9
|
-
*
|
|
10
|
-
*
|
|
7
|
+
* The per-repo harness optimizer's DETERMINISTIC spine — folded INLINE into the
|
|
8
|
+
* default `vigiles audit` report (each finding carries its fix; NOT its own
|
|
9
|
+
* `optimize` verb, and no longer a `--fix-plan` flag: until the measured A/B half
|
|
10
|
+
* lands, an "optimizer" that only re-prints audit's findings doesn't earn a
|
|
11
|
+
* separate surface; see research/roadmap.md §P2 "reconsider an `optimize` verb").
|
|
11
12
|
*
|
|
12
13
|
* A2 in the measurement-authority pivot is the ADOPTION product: measure a user's
|
|
13
14
|
* own skills/model/rules on their tasks and recommend add/drop/swap with a MEASURED
|
|
@@ -67,6 +68,27 @@ const ACTION_LABEL = {
|
|
|
67
68
|
differentiate: "DIFFERENTIATE",
|
|
68
69
|
};
|
|
69
70
|
const measureHint = (dir) => `\`vigiles measure ${dir} --prompts=<file>\` — real-model, runs on your subscription`;
|
|
71
|
+
/**
|
|
72
|
+
* Just the deterministic fix list (no score header) — folded into the default
|
|
73
|
+
* `vigiles audit` report so every finding carries its fix inline (replaces the
|
|
74
|
+
* former `--fix-plan`/`--explain` flags). Empty string when there's nothing to
|
|
75
|
+
* fix (or no loadable surface), so the caller can skip the section entirely.
|
|
76
|
+
*/
|
|
77
|
+
function formatRecommendations(rep) {
|
|
78
|
+
if (rep.empty || rep.recommendations.length === 0)
|
|
79
|
+
return "";
|
|
80
|
+
const lines = [
|
|
81
|
+
`${String(rep.recommendations.length)} deterministic fix(es) — free, no model:`,
|
|
82
|
+
"",
|
|
83
|
+
];
|
|
84
|
+
for (const r of rep.recommendations) {
|
|
85
|
+
const mark = r.confidence === "likely" ? "✗" : "⚠";
|
|
86
|
+
lines.push(`${mark} [${ACTION_LABEL[r.action]}] ${r.surface}`);
|
|
87
|
+
lines.push(` why: ${r.rationale} [${r.detector}]`);
|
|
88
|
+
lines.push(` → ${r.fix}`);
|
|
89
|
+
}
|
|
90
|
+
return lines.join("\n");
|
|
91
|
+
}
|
|
70
92
|
/** Render an optimization plan for the CLI. */
|
|
71
93
|
function formatOptimize(rep) {
|
|
72
94
|
const head = `Harness health: ${String(rep.score)}/100 (${rep.grade}) — ${rep.dir}`;
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `vigiles
|
|
2
|
+
* `vigiles audit` model trigger tier — the BEHAVIORAL column of the audit report.
|
|
3
3
|
*
|
|
4
4
|
* Structural `scan`/`scanPlugin` is deterministic, no-model, CI-free — and stays
|
|
5
|
-
* that way. This is the
|
|
5
|
+
* that way. This is the model-gated column that stacks on top: for each
|
|
6
6
|
* model-invocable skill in a plugin, it measures how reliably the description
|
|
7
7
|
* actually FIRES (recall, + precision when irrelevant prompts are supplied),
|
|
8
8
|
* reusing `measureTriggerRate`. It degrades honestly when the `claude` CLI / auth
|
|
@@ -12,6 +12,8 @@
|
|
|
12
12
|
* path in prose is undecidable, and the deterministic-input discipline is what
|
|
13
13
|
* makes the column trustworthy. See `research/plugin-behavioral-findings.md`.
|
|
14
14
|
*/
|
|
15
|
+
import type { PluginLayout } from "./core/layout.js";
|
|
16
|
+
import type { HarnessDialect } from "./core/dialect.js";
|
|
15
17
|
import { type EvalDriver } from "./eval.js";
|
|
16
18
|
import { type Trace } from "./harness-test.js";
|
|
17
19
|
/** Which harness drives the behavioral column (default Claude Code). */
|
|
@@ -46,6 +48,10 @@ export interface ProbeOptions {
|
|
|
46
48
|
readonly minDistance?: number;
|
|
47
49
|
/** Which harness to drive (default `"claude-code"`). */
|
|
48
50
|
readonly harness?: ProbeHarness;
|
|
51
|
+
/** Layout + dialect for candidate discovery — so a Codex repo's skills (under
|
|
52
|
+
* the Codex layout) are found, not silently missed by the default CC layout. */
|
|
53
|
+
readonly layout?: PluginLayout;
|
|
54
|
+
readonly dialect?: HarnessDialect;
|
|
49
55
|
}
|
|
50
56
|
/**
|
|
51
57
|
* Per-harness probe wiring: the eval driver (runner+parse), how to build the
|
package/dist/scan-behavioral.js
CHANGED
|
@@ -1,9 +1,9 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
* `vigiles
|
|
3
|
+
* `vigiles audit` model trigger tier — the BEHAVIORAL column of the audit report.
|
|
4
4
|
*
|
|
5
5
|
* Structural `scan`/`scanPlugin` is deterministic, no-model, CI-free — and stays
|
|
6
|
-
* that way. This is the
|
|
6
|
+
* that way. This is the model-gated column that stacks on top: for each
|
|
7
7
|
* model-invocable skill in a plugin, it measures how reliably the description
|
|
8
8
|
* actually FIRES (recall, + precision when irrelevant prompts are supplied),
|
|
9
9
|
* reusing `measureTriggerRate`. It degrades honestly when the `claude` CLI / auth
|
|
@@ -131,8 +131,10 @@ async function probeSkill(ctx, name, ps) {
|
|
|
131
131
|
async function probePluginTriggersWith(dir, promptSet, probe, opts = {}) {
|
|
132
132
|
const ctx = { dir, opts, probe };
|
|
133
133
|
// Only model-invocable, describable skills can auto-trigger; user-invoked and
|
|
134
|
-
// description-less ones can't, so they're not behavioral candidates.
|
|
135
|
-
|
|
134
|
+
// description-less ones can't, so they're not behavioral candidates. Discover
|
|
135
|
+
// them with the resolved layout/dialect (default CC) so a Codex repo's skills
|
|
136
|
+
// aren't missed by the wrong layout.
|
|
137
|
+
const candidates = (0, scan_js_1.scanPlugin)(dir, opts.layout, opts.dialect).skills.filter((s) => !s.userInvoked && s.hasDescription);
|
|
136
138
|
const results = [];
|
|
137
139
|
for (const s of candidates) {
|
|
138
140
|
const ps = promptSet[s.name];
|
|
@@ -1,10 +1,15 @@
|
|
|
1
1
|
/**
|
|
2
|
-
*
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
2
|
+
* audit → the ONE read-vs-run decision. `audit` is a Lighthouse-style LOCAL
|
|
3
|
+
* report: a deterministic READ by default — safe + identical on every OS, nothing
|
|
4
|
+
* executes — NOT a CI step (CI uses `vigiles lint`, the deterministic gate). The
|
|
5
|
+
* executing checks (safety battery, live MCP resolution, skill-firing trigger-rate)
|
|
6
|
+
* run ONLY when there's a human to consent: at a TTY `audit` ASKS once (and
|
|
7
|
+
* remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
|
|
8
|
+
* `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
|
|
9
|
+
* never silently executes. There is deliberately NO execution flag: automation
|
|
10
|
+
* tests the harness through the `vigiles/testing` API + skills (the layered tiers),
|
|
11
|
+
* not through the report verb. The IO (prompt / run / remember) lives in the CLI;
|
|
12
|
+
* this is the pure decision + helpers.
|
|
8
13
|
*/
|
|
9
14
|
/** Only the env vars that signal a reachable model (parse, don't validate). */
|
|
10
15
|
export interface ModelEnv {
|
|
@@ -13,42 +18,74 @@ export interface ModelEnv {
|
|
|
13
18
|
readonly CLAUDE_CODE_ENTRYPOINT?: string;
|
|
14
19
|
}
|
|
15
20
|
/**
|
|
16
|
-
* Is a real model reachable for the
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
* token just to decide whether to suggest spending one.
|
|
21
|
+
* Is a real model reachable for the trigger tier? Either a metered API key
|
|
22
|
+
* (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session (`CLAUDECODE=1`
|
|
23
|
+
* / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter drives the `claude`
|
|
24
|
+
* CLI on the user's subscription, no key needed and $0 metered. A tiny env-only
|
|
25
|
+
* predicate (not a live probe), so it never spends a token just to decide.
|
|
22
26
|
*/
|
|
23
27
|
export declare function hasModelAccess(env: ModelEnv): boolean;
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
28
|
+
/**
|
|
29
|
+
* Is the reachable model METERED (a paid API key) rather than a subscription?
|
|
30
|
+
* Only affects the consent DISCLOSURE wording (a metered key bills per token; a
|
|
31
|
+
* subscription is $0 metered) — the run/skip decision itself is consent-driven,
|
|
32
|
+
* not metered-driven.
|
|
33
|
+
*/
|
|
34
|
+
export declare function isMeteredAccess(env: ModelEnv): boolean;
|
|
35
|
+
/** Why the executing checks were skipped (drives the "not run" nudge). */
|
|
36
|
+
export type ExecuteSkipReason = "nothing" | "headless" | "remembered-no";
|
|
37
|
+
/**
|
|
38
|
+
* What `audit` should do about the EXECUTING checks (battery + live MCP +
|
|
39
|
+
* trigger-rate), as ONE bundle:
|
|
40
|
+
* - `run` — run them now (a remembered yes).
|
|
41
|
+
* - `ask` — interactive human + something to run + no sticky choice: ask once,
|
|
42
|
+
* then remember.
|
|
43
|
+
* - `skip` — stay a deterministic read; the `reason` drives a one-line nudge.
|
|
44
|
+
*/
|
|
45
|
+
export type ExecuteDecision = {
|
|
46
|
+
readonly kind: "run";
|
|
47
|
+
} | {
|
|
48
|
+
readonly kind: "ask";
|
|
49
|
+
} | {
|
|
50
|
+
readonly kind: "skip";
|
|
51
|
+
readonly reason: ExecuteSkipReason;
|
|
52
|
+
};
|
|
53
|
+
export interface ExecuteEnv {
|
|
54
|
+
/** Is there ANY executable surface — runnable hooks, an own-repo MCP server, or
|
|
55
|
+
* a model-invocable skill? Nothing to run → never ask, never nudge. */
|
|
56
|
+
readonly hasExecutable: boolean;
|
|
57
|
+
/** Both stdin AND stdout are a terminal (a human who can answer + wait). */
|
|
29
58
|
readonly isTTY: boolean;
|
|
30
|
-
/**
|
|
31
|
-
readonly triggerableSkills: number;
|
|
32
|
-
/** `--json` — machine output; never decorate or prompt. */
|
|
59
|
+
/** `--json` — machine output; stays a read even at a TTY (never prompt). */
|
|
33
60
|
readonly json: boolean;
|
|
34
|
-
/** `--no-interactive` — explicit agent/CI mode
|
|
61
|
+
/** `--no-interactive` / `--yes` — explicit agent/CI mode (never prompt). */
|
|
35
62
|
readonly noInteractive: boolean;
|
|
63
|
+
/** Sticky remembered choice from `.vigilesrc.json` (`audit.measure`), or undefined. */
|
|
64
|
+
readonly remembered?: boolean;
|
|
36
65
|
}
|
|
37
66
|
/**
|
|
38
|
-
* Decide
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
67
|
+
* Decide what `audit` does with the executing checks. Total + pure; the first
|
|
68
|
+
* matching rule wins. There is NO execution flag — `audit` is a local report, so
|
|
69
|
+
* the executing checks need a human to consent:
|
|
70
|
+
* 1. nothing executable → skip "nothing" (a clean read; no nudge)
|
|
71
|
+
* 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
|
|
72
|
+
* skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
|
|
73
|
+
* 3. sticky no → skip "remembered-no"
|
|
74
|
+
* 4. sticky yes → run
|
|
75
|
+
* 5. interactive human, no sticky choice → ask (then remember)
|
|
76
|
+
*/
|
|
77
|
+
export declare function decideExecute(o: ExecuteEnv): ExecuteDecision;
|
|
78
|
+
/**
|
|
79
|
+
* The one-line "executing checks not run" nudge for a skipped read (the
|
|
80
|
+
* no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
|
|
81
|
+
* gap). There is no flag to point at — `audit` runs them only interactively, and
|
|
82
|
+
* automation uses the `vigiles/testing` API.
|
|
43
83
|
*/
|
|
44
|
-
export declare function
|
|
45
|
-
/** The one-line, non-blocking hint (the agent/CI surface). */
|
|
46
|
-
export declare function formatTriggerHint(dir: string, triggerableSkills: number): string;
|
|
84
|
+
export declare function formatExecuteSkip(reason: ExecuteSkipReason): string | null;
|
|
47
85
|
/**
|
|
48
86
|
* A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
|
|
49
87
|
* → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
|
|
50
|
-
* placeholders the user replaces with real requests.
|
|
51
|
-
* only on an explicit human "yes" so a plain `scan` never spends a token.
|
|
88
|
+
* placeholders the user replaces with real requests.
|
|
52
89
|
*/
|
|
53
90
|
export declare function scaffoldTriggerPrompts(skillNames: readonly string[]): string;
|
|
54
91
|
//# sourceMappingURL=scan-trigger-suggest.d.ts.map
|
|
@@ -1,24 +1,29 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
3
|
+
* audit → the ONE read-vs-run decision. `audit` is a Lighthouse-style LOCAL
|
|
4
|
+
* report: a deterministic READ by default — safe + identical on every OS, nothing
|
|
5
|
+
* executes — NOT a CI step (CI uses `vigiles lint`, the deterministic gate). The
|
|
6
|
+
* executing checks (safety battery, live MCP resolution, skill-firing trigger-rate)
|
|
7
|
+
* run ONLY when there's a human to consent: at a TTY `audit` ASKS once (and
|
|
8
|
+
* remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
|
|
9
|
+
* `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
|
|
10
|
+
* never silently executes. There is deliberately NO execution flag: automation
|
|
11
|
+
* tests the harness through the `vigiles/testing` API + skills (the layered tiers),
|
|
12
|
+
* not through the report verb. The IO (prompt / run / remember) lives in the CLI;
|
|
13
|
+
* this is the pure decision + helpers.
|
|
9
14
|
*/
|
|
10
15
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
11
16
|
exports.hasModelAccess = hasModelAccess;
|
|
12
|
-
exports.
|
|
13
|
-
exports.
|
|
17
|
+
exports.isMeteredAccess = isMeteredAccess;
|
|
18
|
+
exports.decideExecute = decideExecute;
|
|
19
|
+
exports.formatExecuteSkip = formatExecuteSkip;
|
|
14
20
|
exports.scaffoldTriggerPrompts = scaffoldTriggerPrompts;
|
|
15
21
|
/**
|
|
16
|
-
* Is a real model reachable for the
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
* token just to decide whether to suggest spending one.
|
|
22
|
+
* Is a real model reachable for the trigger tier? Either a metered API key
|
|
23
|
+
* (`ANTHROPIC_API_KEY`), OR an authenticated Claude Code session (`CLAUDECODE=1`
|
|
24
|
+
* / `CLAUDE_CODE_ENTRYPOINT`, web/desktop/CLI) — the latter drives the `claude`
|
|
25
|
+
* CLI on the user's subscription, no key needed and $0 metered. A tiny env-only
|
|
26
|
+
* predicate (not a live probe), so it never spends a token just to decide.
|
|
22
27
|
*/
|
|
23
28
|
function hasModelAccess(env) {
|
|
24
29
|
return Boolean(env.ANTHROPIC_API_KEY ||
|
|
@@ -26,31 +31,59 @@ function hasModelAccess(env) {
|
|
|
26
31
|
env.CLAUDE_CODE_ENTRYPOINT);
|
|
27
32
|
}
|
|
28
33
|
/**
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
-
* - `"prompt"` — model access + skills + a human at a TTY: offer to set it up.
|
|
34
|
+
* Is the reachable model METERED (a paid API key) rather than a subscription?
|
|
35
|
+
* Only affects the consent DISCLOSURE wording (a metered key bills per token; a
|
|
36
|
+
* subscription is $0 metered) — the run/skip decision itself is consent-driven,
|
|
37
|
+
* not metered-driven.
|
|
34
38
|
*/
|
|
35
|
-
function
|
|
36
|
-
|
|
37
|
-
return "none";
|
|
38
|
-
if (o.noInteractive || !o.isTTY)
|
|
39
|
-
return "hint";
|
|
40
|
-
return "prompt";
|
|
39
|
+
function isMeteredAccess(env) {
|
|
40
|
+
return Boolean(env.ANTHROPIC_API_KEY);
|
|
41
41
|
}
|
|
42
|
-
/**
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
42
|
+
/**
|
|
43
|
+
* Decide what `audit` does with the executing checks. Total + pure; the first
|
|
44
|
+
* matching rule wins. There is NO execution flag — `audit` is a local report, so
|
|
45
|
+
* the executing checks need a human to consent:
|
|
46
|
+
* 1. nothing executable → skip "nothing" (a clean read; no nudge)
|
|
47
|
+
* 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
|
|
48
|
+
* skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
|
|
49
|
+
* 3. sticky no → skip "remembered-no"
|
|
50
|
+
* 4. sticky yes → run
|
|
51
|
+
* 5. interactive human, no sticky choice → ask (then remember)
|
|
52
|
+
*/
|
|
53
|
+
function decideExecute(o) {
|
|
54
|
+
if (!o.hasExecutable)
|
|
55
|
+
return { kind: "skip", reason: "nothing" };
|
|
56
|
+
if (o.json || o.noInteractive || !o.isTTY)
|
|
57
|
+
return { kind: "skip", reason: "headless" };
|
|
58
|
+
if (o.remembered === false)
|
|
59
|
+
return { kind: "skip", reason: "remembered-no" };
|
|
60
|
+
if (o.remembered === true)
|
|
61
|
+
return { kind: "run" };
|
|
62
|
+
return { kind: "ask" };
|
|
63
|
+
}
|
|
64
|
+
/**
|
|
65
|
+
* The one-line "executing checks not run" nudge for a skipped read (the
|
|
66
|
+
* no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
|
|
67
|
+
* gap). There is no flag to point at — `audit` runs them only interactively, and
|
|
68
|
+
* automation uses the `vigiles/testing` API.
|
|
69
|
+
*/
|
|
70
|
+
function formatExecuteSkip(reason) {
|
|
71
|
+
switch (reason) {
|
|
72
|
+
case "nothing":
|
|
73
|
+
return null;
|
|
74
|
+
case "headless":
|
|
75
|
+
return ("\nℹ Executing checks (safety battery · live MCP · skill firing) skipped — " +
|
|
76
|
+
"`audit` runs them only interactively (a terminal). For automation, test the " +
|
|
77
|
+
"harness with the `vigiles/testing` API.");
|
|
78
|
+
case "remembered-no":
|
|
79
|
+
return ("\nℹ Executing checks not run (you disabled them — edit .vigilesrc.json " +
|
|
80
|
+
"`audit.measure` to re-enable).");
|
|
81
|
+
}
|
|
48
82
|
}
|
|
49
83
|
/**
|
|
50
84
|
* A starter `--prompts` file (the real `TriggerPromptSet` shape: bare skill name
|
|
51
85
|
* → `{ prompts, irrelevant }`). One entry per triggerable skill, with TODO
|
|
52
|
-
* placeholders the user replaces with real requests.
|
|
53
|
-
* only on an explicit human "yes" so a plain `scan` never spends a token.
|
|
86
|
+
* placeholders the user replaces with real requests.
|
|
54
87
|
*/
|
|
55
88
|
function scaffoldTriggerPrompts(skillNames) {
|
|
56
89
|
const obj = {};
|
package/dist/scan.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* `vigiles
|
|
2
|
+
* `vigiles audit <dir>` — point vigiles at any plugin/repo and see what it ships
|
|
3
3
|
* and what's broken, with **no model and no API key**.
|
|
4
4
|
*
|
|
5
5
|
* This is the deterministic substrate under the plugin/skill leaderboard
|
|
@@ -27,14 +27,20 @@ export interface ScanSkill {
|
|
|
27
27
|
readonly name: string;
|
|
28
28
|
readonly path: string;
|
|
29
29
|
readonly hasDescription: boolean;
|
|
30
|
+
/**
|
|
31
|
+
* The skill's effective description (frontmatter `description`, else the first
|
|
32
|
+
* body paragraph — the same text the selector keys on), trimmed; `undefined`
|
|
33
|
+
* when neither exists. Feeds the model trigger tier's auto-generated probes.
|
|
34
|
+
*/
|
|
35
|
+
readonly description?: string;
|
|
30
36
|
readonly userInvoked: boolean;
|
|
31
37
|
/**
|
|
32
38
|
* The description's dominant script when it DIFFERS from the expected one
|
|
33
39
|
* (default `"Latin"`), else null. The model's skill-selection context is
|
|
34
40
|
* English-centric, so a description in another script carries a cross-language
|
|
35
41
|
* trigger risk — it may under-fire on English prompts. A RISK flag, not a
|
|
36
|
-
* defect (a language-matched audience is fine); measure the real gap with
|
|
37
|
-
* `
|
|
42
|
+
* defect (a language-matched audience is fine); measure the real gap with the
|
|
43
|
+
* `audit` trigger tier / `measureTriggerRate`.
|
|
38
44
|
*/
|
|
39
45
|
readonly descriptionScript: Script | null;
|
|
40
46
|
}
|
|
@@ -86,8 +92,25 @@ export interface FrontmatterValueIssue {
|
|
|
86
92
|
/** ok = file present; missing = referenced but absent; unresolved = path still has an unexpanded var, can't check. */
|
|
87
93
|
export type HookStatus = "ok" | "missing" | "unresolved";
|
|
88
94
|
export interface ScanHook {
|
|
95
|
+
/**
|
|
96
|
+
* The full hook command as it would be run (plugin-root token expanded, shell
|
|
97
|
+
* quotes stripped). Present on script-based hooks; empty string on hooks whose
|
|
98
|
+
* command is entirely inline (no script file) — but inline hooks never appear
|
|
99
|
+
* in `hooks[]`, they are counted by `inlineHooks`, so in practice `command` is
|
|
100
|
+
* always non-empty when a `ScanHook` is in the list.
|
|
101
|
+
*/
|
|
102
|
+
readonly command: string;
|
|
89
103
|
readonly script: string;
|
|
90
104
|
readonly status: HookStatus;
|
|
105
|
+
/**
|
|
106
|
+
* The hook EVENT this script is registered under (`PreToolUse`, `PostToolUse`,
|
|
107
|
+
* `SessionStart`, …), when it can be determined from the canonical
|
|
108
|
+
* object-keyed-by-event settings shape; `undefined` for a non-object/array
|
|
109
|
+
* config. The safety battery uses it to test only the blocking-capable
|
|
110
|
+
* `PreToolUse` guards — so a `SessionStart`/`PostToolUse` hook isn't unfairly
|
|
111
|
+
* scored against "does it block rm -rf".
|
|
112
|
+
*/
|
|
113
|
+
readonly event?: string;
|
|
91
114
|
}
|
|
92
115
|
/**
|
|
93
116
|
* The repo's top-level instruction file (`CLAUDE.md` / `AGENTS.md`), if present.
|
|
@@ -194,14 +217,16 @@ export declare function preferCompiledHooksMessage(count: number): string;
|
|
|
194
217
|
/** Scan a plugin/repo directory and report its surfaces + structural issues. */
|
|
195
218
|
export declare function scanPlugin(dir: string, layout?: PluginLayout, dialect?: HarnessDialect): ScanReport;
|
|
196
219
|
/**
|
|
197
|
-
* LIVE MCP tool resolution for a scanned plugin — the
|
|
198
|
-
*
|
|
199
|
-
*
|
|
200
|
-
*
|
|
220
|
+
* LIVE MCP tool resolution for a scanned plugin — the dynamic check no static
|
|
221
|
+
* linter can do: it STARTS each declared MCP server and checks every
|
|
222
|
+
* `mcp__server__tool` the plugin's agents reference actually exists on it
|
|
223
|
+
* (catching rename/removal rot, e.g. `create_issue`→`issue_write`). Reuses the
|
|
201
224
|
* already-computed `report` (its agents' tool lists) + the declared server configs;
|
|
202
225
|
* returns `[]` when the plugin declares no MCP servers (nothing to start). Async +
|
|
203
|
-
* side-effecting (spawns servers) —
|
|
204
|
-
*
|
|
226
|
+
* side-effecting (spawns servers) — so `audit` runs it by default only for the
|
|
227
|
+
* user's OWN repo (own-repo, like running your own tools); a FOREIGN plugin's
|
|
228
|
+
* servers are never spawned, and `--fast` opts out. See `verifyMcpContractTools`
|
|
229
|
+
* (core/mcp.ts).
|
|
205
230
|
*/
|
|
206
231
|
export declare function verifyLiveMcpTools(report: ScanReport, layout: PluginLayout, dialect: HarnessDialect, timeoutMs?: number): Promise<McpContractToolError[]>;
|
|
207
232
|
/** Render the live MCP tool-check result (human-readable). */
|
|
@@ -227,7 +252,7 @@ export interface MarketplaceInfo {
|
|
|
227
252
|
* Read a `marketplace.json` beside the layout's plugin manifest and classify its
|
|
228
253
|
* members into on-disk vs external. Returns `null` when `dir` is not a
|
|
229
254
|
* marketplace. The source of truth behind {@link expandMarketplace} and the
|
|
230
|
-
* curated-marketplace report in `vigiles
|
|
255
|
+
* curated-marketplace report in `vigiles audit`.
|
|
231
256
|
*/
|
|
232
257
|
export declare function inspectMarketplace(dir: string, layout?: PluginLayout): MarketplaceInfo | null;
|
|
233
258
|
/**
|
|
@@ -235,7 +260,7 @@ export declare function inspectMarketplace(dir: string, layout?: PluginLayout):
|
|
|
235
260
|
* plugin manifest, e.g. `.claude-plugin/marketplace.json`), expand it into the
|
|
236
261
|
* absolute dirs of its member plugins. Returns `null` when there's no
|
|
237
262
|
* marketplace, `[]` when it's a marketplace whose members are all external (not
|
|
238
|
-
* on disk). Used by `vigiles
|
|
263
|
+
* on disk). Used by `vigiles audit` to rank a whole marketplace — wshobson/agents
|
|
239
264
|
* alone ships 80+ plugins under one `marketplace.json`. See {@link inspectMarketplace}.
|
|
240
265
|
*/
|
|
241
266
|
export declare function expandMarketplace(dir: string, layout?: PluginLayout): string[] | null;
|