vigiles 9.1.0 → 11.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/README.md +126 -112
  2. package/dist/adapters/claude-code/dialect.js +15 -0
  3. package/dist/audit-html.d.ts +15 -4
  4. package/dist/audit-html.js +15 -6
  5. package/dist/audit-report.d.ts +58 -2
  6. package/dist/audit-report.js +29 -0
  7. package/dist/audit-report.template.html +34 -24
  8. package/dist/audit-score.d.ts +19 -12
  9. package/dist/audit-score.js +79 -15
  10. package/dist/audit-serve.d.ts +109 -0
  11. package/dist/audit-serve.js +257 -0
  12. package/dist/cli.js +435 -20
  13. package/dist/core/CLAUDE.md.spec.d.ts +3 -0
  14. package/dist/core/CLAUDE.md.spec.js +26 -0
  15. package/dist/core/compile.d.ts +5 -1
  16. package/dist/core/compile.js +19 -10
  17. package/dist/core/delegation-trifecta.d.ts +64 -0
  18. package/dist/core/delegation-trifecta.js +124 -0
  19. package/dist/core/dialect.d.ts +18 -0
  20. package/dist/core/hook-block-ineffective.d.ts +62 -0
  21. package/dist/core/hook-block-ineffective.js +153 -0
  22. package/dist/core/hook-matcher.d.ts +66 -0
  23. package/dist/core/hook-matcher.js +182 -0
  24. package/dist/core/hook-normalize.d.ts +43 -0
  25. package/dist/core/hook-normalize.js +78 -0
  26. package/dist/core/lethal-trifecta.d.ts +100 -0
  27. package/dist/core/lethal-trifecta.js +197 -0
  28. package/dist/core/plugin-dir-layout.d.ts +30 -0
  29. package/dist/core/plugin-dir-layout.js +73 -0
  30. package/dist/core/rule-meta.d.ts +82 -0
  31. package/dist/core/rule-meta.js +266 -0
  32. package/dist/core/skill-missing-fence.d.ts +47 -0
  33. package/dist/core/skill-missing-fence.js +119 -0
  34. package/dist/core/skill-resources.d.ts +27 -0
  35. package/dist/core/skill-resources.js +167 -0
  36. package/dist/core/types.d.ts +71 -0
  37. package/dist/core/validate.d.ts +1 -0
  38. package/dist/core/validate.js +26 -4
  39. package/dist/leaderboard.d.ts +1 -0
  40. package/dist/leaderboard.js +64 -15
  41. package/dist/scan-behavioral.d.ts +85 -0
  42. package/dist/scan-behavioral.js +225 -0
  43. package/dist/scan.d.ts +106 -0
  44. package/dist/scan.js +269 -53
  45. package/dist/setup-plan.d.ts +6 -3
  46. package/dist/setup-plan.js +12 -2
  47. package/package.json +1 -1
package/dist/cli.js CHANGED
@@ -34,6 +34,7 @@ const optimize_js_1 = require("./optimize.js");
34
34
  const audit_score_js_1 = require("./audit-score.js");
35
35
  const audit_prompts_js_1 = require("./audit-prompts.js");
36
36
  const audit_html_js_1 = require("./audit-html.js");
37
+ const audit_serve_js_1 = require("./audit-serve.js");
37
38
  const audit_report_js_1 = require("./audit-report.js");
38
39
  const adoptability_js_1 = require("./adoptability.js");
39
40
  const compile_js_1 = require("./core/compile.js");
@@ -140,6 +141,14 @@ function printErrors(specFile, errors) {
140
141
  console.log(`::error file=${specFile}::${err.message}`);
141
142
  }
142
143
  }
144
+ /** Non-blocking advisories — printed, but never fail the compile. */
145
+ function printWarnings(specFile, warnings) {
146
+ for (const w of warnings) {
147
+ const pathInfo = w.path ? ` (${w.path})` : "";
148
+ console.log(` ⚠ [${w.type}] ${w.message}${pathInfo}`);
149
+ console.log(`::warning file=${specFile}::${w.message}`);
150
+ }
151
+ }
143
152
  // ---------------------------------------------------------------------------
144
153
  // Commands
145
154
  // ---------------------------------------------------------------------------
@@ -238,7 +247,7 @@ function writeInstructionMirrors(primaryOutput, harnesses) {
238
247
  /** Compile a declarative SkillSpec → SKILL.md. */
239
248
  function compileSkillToFile(spec, specPath, dialect) {
240
249
  const outputPath = specPath.replace(/\.spec\.ts$/, "");
241
- const { markdown, errors } = (0, compile_js_1.compileSkill)(spec, {
250
+ const { markdown, errors, warnings } = (0, compile_js_1.compileSkill)(spec, {
242
251
  basePath: process.cwd(),
243
252
  specFile: specPath,
244
253
  // The SKILL.md frontmatter profile comes from the resolved harness — a Codex
@@ -248,16 +257,18 @@ function compileSkillToFile(spec, specPath, dialect) {
248
257
  (0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
249
258
  if (errors.length === 0) {
250
259
  console.log(`\n✓ ${specPath} → ${outputPath}`);
260
+ printWarnings(specPath, warnings);
251
261
  return true;
252
262
  }
253
263
  console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
254
264
  printErrors(specPath, errors);
265
+ printWarnings(specPath, warnings);
255
266
  return false;
256
267
  }
257
268
  /** Compile a subagent spec → agents/<name>.md (with its result-contract section). */
258
269
  function compileAgentToFile(spec, specPath, dialect) {
259
270
  const outputPath = specPath.replace(/\.spec\.ts$/, "");
260
- const { markdown, errors } = (0, compile_js_1.compileAgent)(spec, {
271
+ const { markdown, errors, warnings } = (0, compile_js_1.compileAgent)(spec, {
261
272
  basePath: process.cwd(),
262
273
  specFile: specPath,
263
274
  dialect,
@@ -265,10 +276,12 @@ function compileAgentToFile(spec, specPath, dialect) {
265
276
  (0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
266
277
  if (errors.length === 0) {
267
278
  console.log(`\n✓ ${specPath} → ${outputPath}`);
279
+ printWarnings(specPath, warnings);
268
280
  return true;
269
281
  }
270
282
  console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
271
283
  printErrors(specPath, errors);
284
+ printWarnings(specPath, warnings);
272
285
  return false;
273
286
  }
274
287
  /**
@@ -665,6 +678,13 @@ function lintExitCode(report) {
665
678
  report.frontmatterValidErrors > 0 ||
666
679
  report.mcpHookErrors > 0 ||
667
680
  report.preferCompiledHookErrors > 0 ||
681
+ report.lethalTrifectaErrors > 0 ||
682
+ report.skillResourceErrors > 0 ||
683
+ report.skillFenceErrors > 0 ||
684
+ report.pluginLayoutErrors > 0 ||
685
+ report.delegationTrifectaErrors > 0 ||
686
+ report.hookBlockErrors > 0 ||
687
+ report.hookMatcherErrors > 0 ||
668
688
  report.symbolRefErrors > 0 ||
669
689
  report.mcpRefErrors > 0)
670
690
  return 2;
@@ -819,8 +839,26 @@ function verifyFrontmatterRules(filePath, silent, exclude, linterOptions) {
819
839
  };
820
840
  }
821
841
  /**
822
- * Verify inline `<!-- vigiles:enforce -->` comments and `vigiles:` YAML
823
- * frontmatter in instruction files that aren't managed by a spec.
842
+ * Frontmatter mode (Level 1 — a `vigiles:` YAML block) is DISABLED in lint:
843
+ * KEPT IN CODE (`src/core/frontmatter.ts`, `verifyFrontmatterRules`,
844
+ * `vigiles generate schema`), but INERT — lint no longer reads or verifies a
845
+ * `vigiles:` block, so it never fires and never fails a build.
846
+ *
847
+ * WHY disabled-not-removed: the three-rung adoption ladder (inline / frontmatter
848
+ * / typed spec) collapsed to TWO on-ramps — inline comments (the zero-TS floor)
849
+ * and the typed `.spec.ts` (the source of truth). Frontmatter mode was the
850
+ * weakest middle rung and an undocumented-but-live surface that muddied the
851
+ * spec-first story (it literally confused a review). With ~no users to break,
852
+ * gating it off makes lint coherent (verify compiled output + inline marks +
853
+ * specs, nothing else) while preserving the code so the decision is reversible:
854
+ * flip this to `true` to re-enable. See `research/pre-release-focus.md` and the
855
+ * parked note in `docs/markdown-mode.md`.
856
+ */
857
+ const FRONTMATTER_MODE_ENABLED = false;
858
+ /**
859
+ * Verify inline `<!-- vigiles:enforce -->` comments (and, when
860
+ * {@link FRONTMATTER_MODE_ENABLED}, `vigiles:` YAML frontmatter) in instruction
861
+ * files that aren't managed by a spec.
824
862
  *
825
863
  * Spec mode is the source of truth when it exists, so a literal
826
864
  * `<!-- vigiles:enforce ... -->` snippet that survived into compiled
@@ -861,9 +899,13 @@ function verifyMarkdownModeRules(files, silent, config) {
861
899
  const inline = verifyInlineRules(filePath, silent, linterOptions);
862
900
  totals.inlineErrors += inline.errorCount;
863
901
  totals.inlineRules += inline.ruleCount;
864
- const fm = verifyFrontmatterRules(filePath, silent, new Set(inline.ruleNames), linterOptions);
865
- totals.frontmatterErrors += fm.errorCount;
866
- totals.frontmatterRules += fm.ruleCount;
902
+ // Frontmatter mode is DISABLED (kept in code, inert in lint) — a `vigiles:`
903
+ // block is ignored, never verified. See FRONTMATTER_MODE_ENABLED.
904
+ if (FRONTMATTER_MODE_ENABLED) {
905
+ const fm = verifyFrontmatterRules(filePath, silent, new Set(inline.ruleNames), linterOptions);
906
+ totals.frontmatterErrors += fm.errorCount;
907
+ totals.frontmatterRules += fm.ruleCount;
908
+ }
867
909
  }
868
910
  if (!silent &&
869
911
  files.length > 0 &&
@@ -993,6 +1035,29 @@ async function runLint(restArgs, flags, config) {
993
1035
  // 7n. Prefer-compiled-hooks — ONE discovery nudge (not per-hook) toward
994
1036
  // compiled `vigiles/hook` artifacts when hand-written hooks ship. Recommendation.
995
1037
  const preferCompiledHooks = checkPreferCompiledHooks(config, silent, adapter);
1038
+ // 7o. Lethal-trifecta — a unit (subagent / model-invocable skill) whose tools
1039
+ // hold all three legs (read-private + ingest-untrusted + exfiltrate) is a
1040
+ // prompt-injection exfil path (Rule of Two). Capability SET-intersection.
1041
+ const lethalTrifecta = checkLethalTrifecta(config, silent, adapter);
1042
+ // 7p. Skill-resource — a SKILL.md body referencing a bundled file that doesn't
1043
+ // exist on disk under the skill dir (the agent gets nothing). FP-safe.
1044
+ const skillResources = checkSkillResourceResolves(config, silent, adapter);
1045
+ // 7q. Skill-missing-fence — a SKILL.md opening with `name:`/`description:` but no
1046
+ // `---` fence loads as plain body (invisible — no name/description/trigger).
1047
+ const skillFence = checkSkillMissingFence(config, silent, adapter);
1048
+ // 7r. Plugin-dir-layout — functional surface dirs (skills/agents/commands) nested
1049
+ // inside the `.claude-plugin/` manifest dir where the harness can't see them.
1050
+ const pluginLayout = checkPluginDirLayout(config, silent, adapter);
1051
+ // 7s. Delegation-trifecta — a lethal trifecta that emerges across a delegation
1052
+ // edge (a subagent's own ∪ delegated-to capability) though no single unit trips it.
1053
+ const delegationTrifecta = checkDelegationTrifecta(config, silent, adapter);
1054
+ // 7t. Hook-block-ineffective — a hook that looks like it blocks but silently
1055
+ // doesn't (block decision on a non-blocking event, or the legacy `decision`
1056
+ // field on a permission-gated event). The #1 verified hook pain (#19009).
1057
+ const hookBlock = checkHookBlockIneffective(config, silent, adapter);
1058
+ // 7u. Hook-matcher — a hook `matcher` that never fires (tool-name typo, or a
1059
+ // malformed/undeclared MCP form).
1060
+ const hookMatcher = checkHookMatcher(config, silent, adapter);
996
1061
  // 8. Validate vigiles builder calls inside markdown code blocks. Default
997
1062
  // is to validate every ref; illustrative blocks opt out via
998
1063
  // `<!-- vigiles:ignore -->` (single block) or
@@ -1060,6 +1125,20 @@ async function runLint(restArgs, flags, config) {
1060
1125
  mcpHookErrors: mcpHookTargets.errors,
1061
1126
  preferCompiledHookIssues: preferCompiledHooks.issues,
1062
1127
  preferCompiledHookErrors: preferCompiledHooks.errors,
1128
+ lethalTrifectaIssues: lethalTrifecta.issues,
1129
+ lethalTrifectaErrors: lethalTrifecta.errors,
1130
+ skillResourceIssues: skillResources.issues,
1131
+ skillResourceErrors: skillResources.errors,
1132
+ skillFenceIssues: skillFence.issues,
1133
+ skillFenceErrors: skillFence.errors,
1134
+ pluginLayoutIssues: pluginLayout.issues,
1135
+ pluginLayoutErrors: pluginLayout.errors,
1136
+ delegationTrifectaIssues: delegationTrifecta.issues,
1137
+ delegationTrifectaErrors: delegationTrifecta.errors,
1138
+ hookBlockIssues: hookBlock.issues,
1139
+ hookBlockErrors: hookBlock.errors,
1140
+ hookMatcherIssues: hookMatcher.issues,
1141
+ hookMatcherErrors: hookMatcher.errors,
1063
1142
  docRefErrors: docRefReport.errors.length,
1064
1143
  symbolRefErrors,
1065
1144
  mcpRefErrors,
@@ -2638,6 +2717,211 @@ function checkDescriptionOverlap(config, silent, adapter) {
2638
2717
  }
2639
2718
  return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2640
2719
  }
2720
+ /**
2721
+ * Apply the `lethal-trifecta` rule: a unit (subagent / model-invocable skill)
2722
+ * whose declared tools hold all three legs (read-private + ingest-untrusted +
2723
+ * exfiltrate) is a prompt-injection exfil path (Meta's Rule of Two). Reuses
2724
+ * `scanPlugin`'s `trifectaFindings` (a capability SET-intersection, one detector,
2725
+ * no drift). Warning by default; "error" gates CI. Surfaces across BOTH subagents
2726
+ * and skills, so it is NOT gated on the `subagents` capability — a skill-only
2727
+ * harness still has the surface.
2728
+ */
2729
+ function checkLethalTrifecta(config, silent, adapter) {
2730
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["lethal-trifecta"]);
2731
+ if (!sev)
2732
+ return { issues: 0, errors: 0 };
2733
+ let found;
2734
+ try {
2735
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).trifectaFindings;
2736
+ }
2737
+ catch {
2738
+ return { issues: 0, errors: 0 };
2739
+ }
2740
+ if (found.length > 0 && !silent) {
2741
+ console.log("\nLethal-trifecta check:\n");
2742
+ for (const t of found) {
2743
+ const msg = `${t.kind} ${t.name}: ${t.finding.message}`;
2744
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${t.path}: ${msg}`);
2745
+ ghAnnotate(sev === "error" ? "error" : "warning", msg, t.path);
2746
+ }
2747
+ }
2748
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2749
+ }
2750
+ /**
2751
+ * Apply the `skill-resource-resolves` rule: a SKILL.md body referencing a bundled
2752
+ * file (`scripts/`/`references/`/`assets/` or a relative markdown link with an
2753
+ * extension) that doesn't exist on disk — the agent reads the instruction and gets
2754
+ * nothing. Reuses `scanPlugin`'s `skillResourceIssues` (high-precision / FP-safe,
2755
+ * one detector, no drift). Warning by default; "error" gates CI.
2756
+ */
2757
+ function checkSkillResourceResolves(config, silent, adapter) {
2758
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["skill-resource-resolves"]);
2759
+ if (!sev)
2760
+ return { issues: 0, errors: 0 };
2761
+ let found;
2762
+ try {
2763
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).skillResourceIssues;
2764
+ }
2765
+ catch {
2766
+ return { issues: 0, errors: 0 };
2767
+ }
2768
+ if (found.length > 0 && !silent) {
2769
+ console.log("\nSkill-resource check:\n");
2770
+ for (const s of found) {
2771
+ const msg = `${s.name}: bundled resource "${s.finding.ref}" (line ${String(s.finding.line)}) is referenced but missing — the agent reads the instruction and gets nothing.`;
2772
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${s.path}: ${msg}`);
2773
+ ghAnnotate(sev === "error" ? "error" : "warning", msg, s.path);
2774
+ }
2775
+ }
2776
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2777
+ }
2778
+ /**
2779
+ * Apply the `skill-missing-fence` rule: a SKILL.md that opens with
2780
+ * frontmatter-looking keys (`name:`/`description:`) but no `---` fence loads as
2781
+ * pure body — no name, no description, no trigger (the skill is invisible).
2782
+ * Reuses `scanPlugin`'s `skillFenceIssues` (one detector, no drift). Warning by
2783
+ * default; "error" gates CI.
2784
+ */
2785
+ function checkSkillMissingFence(config, silent, adapter) {
2786
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["skill-missing-fence"]);
2787
+ if (!sev)
2788
+ return { issues: 0, errors: 0 };
2789
+ let found;
2790
+ try {
2791
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).skillFenceIssues;
2792
+ }
2793
+ catch {
2794
+ return { issues: 0, errors: 0 };
2795
+ }
2796
+ if (found.length > 0 && !silent) {
2797
+ console.log("\nSkill-missing-fence check:\n");
2798
+ for (const s of found) {
2799
+ const msg = `${s.name}: ${s.finding.message}`;
2800
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${s.path}: ${msg}`);
2801
+ ghAnnotate(sev === "error" ? "error" : "warning", msg, s.path);
2802
+ }
2803
+ }
2804
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2805
+ }
2806
+ /**
2807
+ * Apply the `plugin-dir-layout` rule: functional surface dirs (skills/agents/
2808
+ * commands) nested inside the `.claude-plugin/` manifest dir where the harness
2809
+ * can't see them (the #1 plugin-author mistake). Reuses `scanPlugin`'s
2810
+ * `pluginLayoutIssues` (one detector, no drift). Warning by default; "error"
2811
+ * gates CI.
2812
+ */
2813
+ function checkPluginDirLayout(config, silent, adapter) {
2814
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["plugin-dir-layout"]);
2815
+ if (!sev)
2816
+ return { issues: 0, errors: 0 };
2817
+ let found;
2818
+ try {
2819
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).pluginLayoutIssues;
2820
+ }
2821
+ catch {
2822
+ return { issues: 0, errors: 0 };
2823
+ }
2824
+ if (found.length > 0 && !silent) {
2825
+ console.log("\nPlugin-dir-layout check:\n");
2826
+ for (const p of found) {
2827
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${p.message}`);
2828
+ ghAnnotate(sev === "error" ? "error" : "warning", p.message);
2829
+ }
2830
+ }
2831
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2832
+ }
2833
+ /**
2834
+ * Apply the `delegation-trifecta` rule: a lethal trifecta that EMERGES across a
2835
+ * delegation edge — a subagent whose effective (own ∪ delegated-to) capability
2836
+ * holds all three legs though no single unit does. Reuses `scanPlugin`'s
2837
+ * `delegationTrifecta` (one detector, no drift). Warning by default; "error"
2838
+ * gates CI. Surfaces across the subagent graph, so it is NOT gated on a
2839
+ * capability the way a surface-specific rule is.
2840
+ */
2841
+ function checkDelegationTrifecta(config, silent, adapter) {
2842
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["delegation-trifecta"]);
2843
+ if (!sev)
2844
+ return { issues: 0, errors: 0 };
2845
+ let found;
2846
+ try {
2847
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).delegationTrifecta;
2848
+ }
2849
+ catch {
2850
+ return { issues: 0, errors: 0 };
2851
+ }
2852
+ if (found.length > 0 && !silent) {
2853
+ console.log("\nDelegation-trifecta check:\n");
2854
+ for (const d of found) {
2855
+ const msg = `${d.finding.name}: ${d.finding.message}`;
2856
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${d.path}: ${msg}`);
2857
+ ghAnnotate(sev === "error" ? "error" : "warning", msg, d.path);
2858
+ }
2859
+ }
2860
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2861
+ }
2862
+ /**
2863
+ * Apply the `hook-block-ineffective` rule: a hook that LOOKS like it blocks but
2864
+ * silently doesn't — a block decision (`exit 2` / `decision` / `permissionDecision`)
2865
+ * on a non-blocking event, or the legacy top-level `decision` field on a
2866
+ * permission-gated event (#19009, the #1 verified hook pain). Reuses `scanPlugin`'s
2867
+ * `hookBlockFindings` (one detector, no drift). Warning by default; "error" gates CI.
2868
+ */
2869
+ function checkHookBlockIneffective(config, silent, adapter) {
2870
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["hook-block-ineffective"]);
2871
+ if (!sev)
2872
+ return { issues: 0, errors: 0 };
2873
+ if (!adapter.capabilities.shellHooks) {
2874
+ reportNotApplicable("Hook-block check", "shell hooks", adapter, silent);
2875
+ return { issues: 0, errors: 0 };
2876
+ }
2877
+ let found;
2878
+ try {
2879
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).hookBlockFindings;
2880
+ }
2881
+ catch {
2882
+ return { issues: 0, errors: 0 };
2883
+ }
2884
+ if (found.length > 0 && !silent) {
2885
+ console.log("\nHook-block check:\n");
2886
+ for (const h of found) {
2887
+ const where = h.scriptPath ?? "(inline)";
2888
+ const msg = `[${h.event}] ${where}: ${h.message}`;
2889
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${msg}`);
2890
+ ghAnnotate(sev === "error" ? "error" : "warning", msg, h.scriptPath ?? undefined);
2891
+ }
2892
+ }
2893
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2894
+ }
2895
+ /**
2896
+ * Apply the `hook-matcher` rule: a hook `matcher` string that silently never
2897
+ * fires — a tool-name typo (`bash`→`Bash`) or a malformed/undeclared MCP form.
2898
+ * Reuses `scanPlugin`'s `hookMatcherFindings` (one detector, no drift). Warning
2899
+ * by default; "error" gates CI.
2900
+ */
2901
+ function checkHookMatcher(config, silent, adapter) {
2902
+ const sev = (0, types_js_1.ruleSeverity)(config?.rules?.["hook-matcher"]);
2903
+ if (!sev)
2904
+ return { issues: 0, errors: 0 };
2905
+ if (!adapter.capabilities.shellHooks) {
2906
+ reportNotApplicable("Hook-matcher check", "shell hooks", adapter, silent);
2907
+ return { issues: 0, errors: 0 };
2908
+ }
2909
+ let found;
2910
+ try {
2911
+ found = (0, scan_js_1.scanPlugin)(process.cwd(), adapter.layout, adapter.dialect).hookMatcherFindings;
2912
+ }
2913
+ catch {
2914
+ return { issues: 0, errors: 0 };
2915
+ }
2916
+ if (found.length > 0 && !silent) {
2917
+ console.log("\nHook-matcher check:\n");
2918
+ for (const m of found) {
2919
+ console.log(` ${sev === "error" ? "✗" : "⚠"} ${m.message}`);
2920
+ ghAnnotate(sev === "error" ? "error" : "warning", m.message);
2921
+ }
2922
+ }
2923
+ return { issues: found.length, errors: sev === "error" ? found.length : 0 };
2924
+ }
2641
2925
  /**
2642
2926
  * Apply the `mcp-hook-target-resolves` rule: a `type: "mcp_tool"` hook action
2643
2927
  * that's incomplete (no `server`/`tool`) or targets a server the plugin doesn't
@@ -3330,6 +3614,7 @@ function printUsage(command) {
3330
3614
  console.log(" vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)");
3331
3615
  console.log(" writes vigiles-report.html + vigiles-report.json (--no-html/--no-json) · --json for machine output. NOT a CI step — use `vigiles lint` in CI.");
3332
3616
  console.log(" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles/testing API");
3617
+ console.log(" --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt");
3333
3618
  console.log(" vigiles test [files...] Run *.harness.mjs deterministic harness tests");
3334
3619
  console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
3335
3620
  console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
@@ -4186,6 +4471,63 @@ function writeAuditHtml(report) {
4186
4471
  console.log(`\n⚠ skipped vigiles-report.html: ${e instanceof Error ? e.message : String(e)}`);
4187
4472
  }
4188
4473
  }
4474
+ /**
4475
+ * Start the live (`--serve`) adoption server: render the report with a per-run
4476
+ * token, serve it on loopback, and run `init` in-process when a button POSTs. The
4477
+ * security model lives in src/audit-serve.ts (token + Origin + allowlist). Blocks
4478
+ * until the user stops it (Ctrl-C or the page's Done). Own-repo only — the caller
4479
+ * gates this via decideServeGate, so adopt always writes into the current repo.
4480
+ */
4481
+ async function runAuditServe(report, adoptable, cliErr) {
4482
+ const token = (0, audit_serve_js_1.newToken)();
4483
+ const surfaces = new Set((adoptable?.surfaces ?? []).map((s) => s.path));
4484
+ let html;
4485
+ try {
4486
+ html = (0, audit_html_js_1.renderAuditHtml)(report, { token });
4487
+ }
4488
+ catch (e) {
4489
+ console.log(`\n⚠ can't serve the live report: ${cliErr(e)}`);
4490
+ return;
4491
+ }
4492
+ const adoptOne = (target) => {
4493
+ try {
4494
+ scaffoldSpec(["--target=" + target]); // in-process; writes into cwd (own repo)
4495
+ return Promise.resolve({
4496
+ ok: true,
4497
+ message: `created spec for ${target}`,
4498
+ });
4499
+ }
4500
+ catch (e) {
4501
+ return Promise.resolve({ ok: false, message: cliErr(e) });
4502
+ }
4503
+ };
4504
+ console.log("\n Live report — create specs with one click. Ctrl-C to stop.\n");
4505
+ await (0, audit_serve_js_1.serveAudit)({
4506
+ token,
4507
+ surfaces,
4508
+ html,
4509
+ runAdopt: adoptOne,
4510
+ runAdoptAll: () => {
4511
+ try {
4512
+ for (const p of surfaces)
4513
+ scaffoldSpec(["--target=" + p]);
4514
+ return Promise.resolve({
4515
+ ok: true,
4516
+ message: `created ${String(surfaces.size)} spec(s)`,
4517
+ });
4518
+ }
4519
+ catch (e) {
4520
+ return Promise.resolve({ ok: false, message: cliErr(e) });
4521
+ }
4522
+ },
4523
+ onListening: (url) => {
4524
+ console.log(` ${url}`);
4525
+ if (process.stdout.isTTY)
4526
+ openBestEffort(url);
4527
+ },
4528
+ });
4529
+ console.log("\n✓ live report closed");
4530
+ }
4189
4531
  /**
4190
4532
  * Run the model trigger tier with no `--prompts`: auto-generate diverse probe
4191
4533
  * prompts from each skill's description and measure trigger-rate (recall +
@@ -4210,19 +4552,54 @@ async function runAutoTrigger(dir, report, adapter, args) {
4210
4552
  if (!json) {
4211
4553
  console.log("\nℹ auto-generated probe prompts from skill descriptions (pass --prompts=<file> for a curated set).");
4212
4554
  }
4555
+ const model = flagValue(args, "--model");
4213
4556
  const trigger = await (0, scan_behavioral_js_1.probePluginTriggers)(dir, promptSet, {
4214
4557
  minPrompts: audit_prompts_js_1.AUTO_RECALL_COUNT,
4215
4558
  minDistance: audit_prompts_js_1.AUTO_MIN_DISTANCE,
4216
- model: flagValue(args, "--model"),
4559
+ model,
4217
4560
  harness,
4218
4561
  // Discover candidates with the resolved adapter's layout/dialect — a Codex
4219
4562
  // repo's skills live under the Codex layout, not the default CC one.
4220
4563
  layout: adapter.layout,
4221
4564
  dialect: adapter.dialect,
4222
4565
  });
4223
- console.log(json
4224
- ? JSON.stringify({ trigger }, null, 2)
4225
- : "\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
4566
+ // Second behavioral eval (same consent): the selection-collision matrix — does
4567
+ // one skill HIJACK a sibling's prompt? This is the MEASURED confirmation of the
4568
+ // deterministic description-overlap proxy (the Triggering ring flags look-alikes;
4569
+ // this proves the wrong one actually fires). Only meaningful with ≥2 model-
4570
+ // invocable skills (a lone skill can't collide); reuses the same auto prompts.
4571
+ const collisions = skills.length >= 2
4572
+ ? await (0, scan_behavioral_js_1.measurePluginSelection)(dir, promptSet, { model, harness })
4573
+ : null;
4574
+ // Third behavioral eval (same consent): adversarial-gate — do enforcement-gate
4575
+ // skills HOLD when the agent is told to violate them? Auto-derives its own
4576
+ // attacks; a no-op (no model calls) when the plugin declares no gate skills.
4577
+ const gates = await (0, scan_behavioral_js_1.measureGateAdversarial)(dir, {
4578
+ model,
4579
+ harness,
4580
+ layout: adapter.layout,
4581
+ dialect: adapter.dialect,
4582
+ });
4583
+ // Show the gate section when gate skills were DETECTED — even if the eval
4584
+ // couldn't RUN (a Codex audit, or no `claude` CLI) it returns available:false
4585
+ // with empty results, and `formatGateReport` renders the "unavailable" note.
4586
+ // The consent prompt already advertised these gate skills, so a skipped check
4587
+ // must be reported LOUDLY, never silently omitted as if there were none.
4588
+ const hasGates = gates.results.length > 0 || (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length > 0;
4589
+ if (json) {
4590
+ console.log(JSON.stringify({
4591
+ trigger,
4592
+ ...(collisions ? { collisions } : {}),
4593
+ ...(hasGates ? { gates } : {}),
4594
+ }, null, 2));
4595
+ }
4596
+ else {
4597
+ console.log("\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
4598
+ if (collisions)
4599
+ console.log("\n" + (0, scan_behavioral_js_1.formatSelectionReport)(collisions));
4600
+ if (hasGates)
4601
+ console.log("\n" + (0, scan_behavioral_js_1.formatGateReport)(gates));
4602
+ }
4226
4603
  }
4227
4604
  /**
4228
4605
  * The ONE read-vs-run decision for a single-plugin `audit`. A plain `audit` is a
@@ -4272,7 +4649,17 @@ function buildExecuteDisclosure(s, harness) {
4272
4649
  if (s.hasMcp)
4273
4650
  lines.push(" · start your MCP servers — connects to their backends");
4274
4651
  if (s.triggerableSkills > 0) {
4275
- lines.push(` · measure whether skills fire (${triggerCostWording(harness)})`);
4652
+ // ≥2 model-invocable skills also get the selection-collision matrix (does one
4653
+ // skill hijack a sibling's prompt) — disclose it so the consent stays honest.
4654
+ const what = s.triggerableSkills >= 2
4655
+ ? "measure whether skills fire and collide"
4656
+ : "measure whether skills fire";
4657
+ lines.push(` · ${what} (${triggerCostWording(harness)})`);
4658
+ }
4659
+ if (s.gateSkills > 0) {
4660
+ // The adversarial-gate eval runs the FULL (unstubbed) skill — the most
4661
+ // expensive check — so disclose it separately when gate skills are present.
4662
+ lines.push(` · test whether ${String(s.gateSkills)} enforcement-gate skill${s.gateSkills === 1 ? "" : "s"} hold under pressure — runs the full skill (${triggerCostWording(harness)})`);
4276
4663
  }
4277
4664
  if (s.adoptableRefs) {
4278
4665
  lines.push(` · draft + verify your instruction file's references (${triggerCostWording(harness)})`);
@@ -4452,7 +4839,10 @@ async function main() {
4452
4839
  // through to a misleading "empty machine / no structural issues" report
4453
4840
  // (obra/superpowers-marketplace, anthropics/claude-plugins-community).
4454
4841
  if (json) {
4455
- console.log(JSON.stringify(market, null, 2));
4842
+ console.log(JSON.stringify((0, audit_report_js_1.buildMarketplaceReport)(market, {
4843
+ vigilesVersion: getVersion(),
4844
+ dir: (0, node_path_1.resolve)(dirs[0]),
4845
+ }), null, 2));
4456
4846
  }
4457
4847
  else {
4458
4848
  console.log(`Marketplace "${market.name}": ${String(market.total)} plugin(s), all external ` +
@@ -4468,7 +4858,12 @@ async function main() {
4468
4858
  const text = args.includes("--md")
4469
4859
  ? (0, leaderboard_js_1.formatLeaderboardMarkdown)(scores)
4470
4860
  : (0, leaderboard_js_1.formatLeaderboard)(scores);
4471
- console.log(json ? JSON.stringify(scores, null, 2) : text);
4861
+ console.log(json
4862
+ ? JSON.stringify((0, audit_report_js_1.buildLeaderboardReport)(scores, {
4863
+ vigilesVersion: getVersion(),
4864
+ dir: (0, node_path_1.resolve)(dirs[0]),
4865
+ }), null, 2)
4866
+ : text);
4472
4867
  }
4473
4868
  else {
4474
4869
  const root = (0, node_path_1.resolve)(targets[0]);
@@ -4546,6 +4941,7 @@ async function main() {
4546
4941
  const surfaces = {
4547
4942
  hasMcp: report.mcp && !isForeign,
4548
4943
  triggerableSkills: report.skills.filter((s) => s.hasDescription && !s.userInvoked).length,
4944
+ gateSkills: (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length,
4549
4945
  adoptableRefs: adapter.name === "claude-code" &&
4550
4946
  (0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, adapter.layout.instructionFile)),
4551
4947
  };
@@ -4632,17 +5028,36 @@ async function main() {
4632
5028
  const finalReport = adoptabilityResult
4633
5029
  ? { ...auditReport, adoptability: adoptabilityResult }
4634
5030
  : auditReport;
4635
- // The shareable HTML report — written by default (--no-html to skip), and
4636
- // opened best-effort only for a human at a TTY (never spawn a browser for
4637
- // an agent / CI run).
4638
- if (!json && !args.includes("--no-html")) {
4639
- writeAuditHtml(finalReport);
4640
- }
4641
5031
  // The versioned JSON artifact — the upload/CI boundary (a hosted dashboard
4642
5032
  // ingests this). Written by default in the human path; --no-json to skip.
4643
5033
  if (!json && !args.includes("--no-json")) {
4644
5034
  writeAuditJson(finalReport);
4645
5035
  }
5036
+ // The HTML report has two deliveries. STATIC (default): write the
5037
+ // shareable file whose buttons copy the `init` command. LIVE (`--serve`,
5038
+ // or a TTY "yes"): start a loopback server whose buttons run `init` for
5039
+ // you. The gate keeps the default a terminating, headless-safe read and
5040
+ // restricts the write-server to your own repo (decideServeGate).
5041
+ const serveGate = (0, audit_serve_js_1.decideServeGate)({
5042
+ serveFlag: args.includes("--serve"),
5043
+ noServeFlag: args.includes("--no-serve"),
5044
+ json,
5045
+ isTTY: process.stdout.isTTY && process.stdin.isTTY,
5046
+ ownRepo: !isForeign,
5047
+ adoptableCount: finalReport.adoptable?.surfaces.length ?? 0,
5048
+ });
5049
+ let serveLive = serveGate === "serve";
5050
+ if (serveGate === "ask") {
5051
+ const ans = (await askOnce("\nOpen the live report to create specs with one click? [y/N] ")).toLowerCase();
5052
+ serveLive = ans === "y" || ans === "yes";
5053
+ }
5054
+ const errMsg = (e) => e instanceof Error ? e.message : String(e);
5055
+ if (serveLive) {
5056
+ await runAuditServe(finalReport, finalReport.adoptable, errMsg);
5057
+ }
5058
+ else if (!json && !args.includes("--no-html")) {
5059
+ writeAuditHtml(finalReport);
5060
+ }
4646
5061
  }
4647
5062
  break;
4648
5063
  }
@@ -0,0 +1,3 @@
1
+ declare const _default: import("./spec.js").ClaudeSpec;
2
+ export default _default;
3
+ //# sourceMappingURL=CLAUDE.md.spec.d.ts.map
@@ -0,0 +1,26 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ /**
4
+ * Directory-scoped guidance for working in `src/core/` (the harness-agnostic
5
+ * detectors + domain).
6
+ *
7
+ * The full project rule set is the ROOT `CLAUDE.md` (compiled from
8
+ * `CLAUDE.md.spec.ts`). This nested spec adds only the discipline that belongs
9
+ * next to the detectors themselves — Claude Code loads it as directory memory
10
+ * whenever you work in `src/core/`. Source of truth; `src/core/CLAUDE.md` is a
11
+ * compiled build artifact (`vigiles compile`).
12
+ */
13
+ const spec_js_1 = require("./spec.js");
14
+ exports.default = (0, spec_js_1.claude)({
15
+ sections: {
16
+ scope: `Working in \`src/core/\`? This is the harness-AGNOSTIC domain (spec, compile, linters, the lint/audit detectors). The root \`CLAUDE.md\` holds the full positioning + rule set — read it first. Two invariants live closest to this code: the core must not import an adapter (\`core ⊄ adapter\`, eslint-enforced) and must not hard-code a Claude Code literal (read it from the injected layout/dialect). This file adds the rule for ADDING or CHANGING a detector.`,
17
+ },
18
+ keyFiles: {
19
+ "src/core/rule-meta.ts": "The RuleMeta registry — every rule's decidability bucket + severity + detector, the single source the detector-meta rule enforces.",
20
+ "src/core/types.ts": "RulesConfig — the rule-name keys the registry is keyed on.",
21
+ },
22
+ rules: {
23
+ "detector-meta": (0, spec_js_1.guidance)("A deterministic DETECTOR here is one half of a RULE — and a rule is not done until it is DECLARED. Three things move together (sibling of one-detector-no-drift + rules-docs-in-sync): (1) the pure detector function (shared by `lint` AND `audit`, never reimplemented per surface; read the layout/dialect, never a CC literal); (2) its entry in `src/core/rule-meta.ts` — the `Record<RuleName, RuleMeta>` won't typecheck without it — declaring its DECIDABILITY BUCKET (structural-closed = a type could prevent it / external-decidable = needs the world, error-capable / heuristic-behavioral = warn-or-measure-only), surface, defaultSeverity, the detector name, and any upstreamPrevention; (3) its `docs/rules/<name>.md` (the coverage test binds the registry to the docs by an EXACT set match, so a missing meta or doc fails CI). The bucket is the CEILING, not a preference — a heuristic proxy may NEVER default to `error` (it cries wolf); a structural/external fact MAY, once proven FP-safe. Before writing a new detector, CLASSIFY the defect into a bucket — that decides whether it can ever gate. The full model + the prose behind the buckets is the root `lint-rule-calibration` rule and `research/enforcement-model.md`."),
24
+ },
25
+ });
26
+ //# sourceMappingURL=CLAUDE.md.spec.js.map