vigiles 9.1.0 → 10.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -34,6 +34,7 @@ const optimize_js_1 = require("./optimize.js");
34
34
  const audit_score_js_1 = require("./audit-score.js");
35
35
  const audit_prompts_js_1 = require("./audit-prompts.js");
36
36
  const audit_html_js_1 = require("./audit-html.js");
37
+ const audit_serve_js_1 = require("./audit-serve.js");
37
38
  const audit_report_js_1 = require("./audit-report.js");
38
39
  const adoptability_js_1 = require("./adoptability.js");
39
40
  const compile_js_1 = require("./core/compile.js");
@@ -140,6 +141,14 @@ function printErrors(specFile, errors) {
140
141
  console.log(`::error file=${specFile}::${err.message}`);
141
142
  }
142
143
  }
144
+ /** Non-blocking advisories — printed, but never fail the compile. */
145
+ function printWarnings(specFile, warnings) {
146
+ for (const w of warnings) {
147
+ const pathInfo = w.path ? ` (${w.path})` : "";
148
+ console.log(` ⚠ [${w.type}] ${w.message}${pathInfo}`);
149
+ console.log(`::warning file=${specFile}::${w.message}`);
150
+ }
151
+ }
143
152
  // ---------------------------------------------------------------------------
144
153
  // Commands
145
154
  // ---------------------------------------------------------------------------
@@ -238,7 +247,7 @@ function writeInstructionMirrors(primaryOutput, harnesses) {
238
247
  /** Compile a declarative SkillSpec → SKILL.md. */
239
248
  function compileSkillToFile(spec, specPath, dialect) {
240
249
  const outputPath = specPath.replace(/\.spec\.ts$/, "");
241
- const { markdown, errors } = (0, compile_js_1.compileSkill)(spec, {
250
+ const { markdown, errors, warnings } = (0, compile_js_1.compileSkill)(spec, {
242
251
  basePath: process.cwd(),
243
252
  specFile: specPath,
244
253
  // The SKILL.md frontmatter profile comes from the resolved harness — a Codex
@@ -248,16 +257,18 @@ function compileSkillToFile(spec, specPath, dialect) {
248
257
  (0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
249
258
  if (errors.length === 0) {
250
259
  console.log(`\n✓ ${specPath} → ${outputPath}`);
260
+ printWarnings(specPath, warnings);
251
261
  return true;
252
262
  }
253
263
  console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
254
264
  printErrors(specPath, errors);
265
+ printWarnings(specPath, warnings);
255
266
  return false;
256
267
  }
257
268
  /** Compile a subagent spec → agents/<name>.md (with its result-contract section). */
258
269
  function compileAgentToFile(spec, specPath, dialect) {
259
270
  const outputPath = specPath.replace(/\.spec\.ts$/, "");
260
- const { markdown, errors } = (0, compile_js_1.compileAgent)(spec, {
271
+ const { markdown, errors, warnings } = (0, compile_js_1.compileAgent)(spec, {
261
272
  basePath: process.cwd(),
262
273
  specFile: specPath,
263
274
  dialect,
@@ -265,10 +276,12 @@ function compileAgentToFile(spec, specPath, dialect) {
265
276
  (0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
266
277
  if (errors.length === 0) {
267
278
  console.log(`\n✓ ${specPath} → ${outputPath}`);
279
+ printWarnings(specPath, warnings);
268
280
  return true;
269
281
  }
270
282
  console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
271
283
  printErrors(specPath, errors);
284
+ printWarnings(specPath, warnings);
272
285
  return false;
273
286
  }
274
287
  /**
@@ -819,8 +832,26 @@ function verifyFrontmatterRules(filePath, silent, exclude, linterOptions) {
819
832
  };
820
833
  }
821
834
  /**
822
- * Verify inline `<!-- vigiles:enforce -->` comments and `vigiles:` YAML
823
- * frontmatter in instruction files that aren't managed by a spec.
835
+ * Frontmatter mode (Level 1 — a `vigiles:` YAML block) is DISABLED in lint:
836
+ * KEPT IN CODE (`src/core/frontmatter.ts`, `verifyFrontmatterRules`,
837
+ * `vigiles generate schema`), but INERT — lint no longer reads or verifies a
838
+ * `vigiles:` block, so it never fires and never fails a build.
839
+ *
840
+ * WHY disabled-not-removed: the three-rung adoption ladder (inline / frontmatter
841
+ * / typed spec) collapsed to TWO on-ramps — inline comments (the zero-TS floor)
842
+ * and the typed `.spec.ts` (the source of truth). Frontmatter mode was the
843
+ * weakest middle rung and an undocumented-but-live surface that muddied the
844
+ * spec-first story (it literally confused a review). With ~no users to break,
845
+ * gating it off makes lint coherent (verify compiled output + inline marks +
846
+ * specs, nothing else) while preserving the code so the decision is reversible:
847
+ * flip this to `true` to re-enable. See `research/pre-release-focus.md` and the
848
+ * parked note in `docs/markdown-mode.md`.
849
+ */
850
+ const FRONTMATTER_MODE_ENABLED = false;
851
+ /**
852
+ * Verify inline `<!-- vigiles:enforce -->` comments (and, when
853
+ * {@link FRONTMATTER_MODE_ENABLED}, `vigiles:` YAML frontmatter) in instruction
854
+ * files that aren't managed by a spec.
824
855
  *
825
856
  * Spec mode is the source of truth when it exists, so a literal
826
857
  * `<!-- vigiles:enforce ... -->` snippet that survived into compiled
@@ -861,9 +892,13 @@ function verifyMarkdownModeRules(files, silent, config) {
861
892
  const inline = verifyInlineRules(filePath, silent, linterOptions);
862
893
  totals.inlineErrors += inline.errorCount;
863
894
  totals.inlineRules += inline.ruleCount;
864
- const fm = verifyFrontmatterRules(filePath, silent, new Set(inline.ruleNames), linterOptions);
865
- totals.frontmatterErrors += fm.errorCount;
866
- totals.frontmatterRules += fm.ruleCount;
895
+ // Frontmatter mode is DISABLED (kept in code, inert in lint) — a `vigiles:`
896
+ // block is ignored, never verified. See FRONTMATTER_MODE_ENABLED.
897
+ if (FRONTMATTER_MODE_ENABLED) {
898
+ const fm = verifyFrontmatterRules(filePath, silent, new Set(inline.ruleNames), linterOptions);
899
+ totals.frontmatterErrors += fm.errorCount;
900
+ totals.frontmatterRules += fm.ruleCount;
901
+ }
867
902
  }
868
903
  if (!silent &&
869
904
  files.length > 0 &&
@@ -3330,6 +3365,7 @@ function printUsage(command) {
3330
3365
  console.log(" vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)");
3331
3366
  console.log(" writes vigiles-report.html + vigiles-report.json (--no-html/--no-json) · --json for machine output. NOT a CI step — use `vigiles lint` in CI.");
3332
3367
  console.log(" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles/testing API");
3368
+ console.log(" --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt");
3333
3369
  console.log(" vigiles test [files...] Run *.harness.mjs deterministic harness tests");
3334
3370
  console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
3335
3371
  console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
@@ -4186,6 +4222,63 @@ function writeAuditHtml(report) {
4186
4222
  console.log(`\n⚠ skipped vigiles-report.html: ${e instanceof Error ? e.message : String(e)}`);
4187
4223
  }
4188
4224
  }
4225
+ /**
4226
+ * Start the live (`--serve`) adoption server: render the report with a per-run
4227
+ * token, serve it on loopback, and run `init` in-process when a button POSTs. The
4228
+ * security model lives in src/audit-serve.ts (token + Origin + allowlist). Blocks
4229
+ * until the user stops it (Ctrl-C or the page's Done). Own-repo only — the caller
4230
+ * gates this via decideServeGate, so adopt always writes into the current repo.
4231
+ */
4232
+ async function runAuditServe(report, adoptable, cliErr) {
4233
+ const token = (0, audit_serve_js_1.newToken)();
4234
+ const surfaces = new Set((adoptable?.surfaces ?? []).map((s) => s.path));
4235
+ let html;
4236
+ try {
4237
+ html = (0, audit_html_js_1.renderAuditHtml)(report, { token });
4238
+ }
4239
+ catch (e) {
4240
+ console.log(`\n⚠ can't serve the live report: ${cliErr(e)}`);
4241
+ return;
4242
+ }
4243
+ const adoptOne = (target) => {
4244
+ try {
4245
+ scaffoldSpec(["--target=" + target]); // in-process; writes into cwd (own repo)
4246
+ return Promise.resolve({
4247
+ ok: true,
4248
+ message: `created spec for ${target}`,
4249
+ });
4250
+ }
4251
+ catch (e) {
4252
+ return Promise.resolve({ ok: false, message: cliErr(e) });
4253
+ }
4254
+ };
4255
+ console.log("\n Live report — create specs with one click. Ctrl-C to stop.\n");
4256
+ await (0, audit_serve_js_1.serveAudit)({
4257
+ token,
4258
+ surfaces,
4259
+ html,
4260
+ runAdopt: adoptOne,
4261
+ runAdoptAll: () => {
4262
+ try {
4263
+ for (const p of surfaces)
4264
+ scaffoldSpec(["--target=" + p]);
4265
+ return Promise.resolve({
4266
+ ok: true,
4267
+ message: `created ${String(surfaces.size)} spec(s)`,
4268
+ });
4269
+ }
4270
+ catch (e) {
4271
+ return Promise.resolve({ ok: false, message: cliErr(e) });
4272
+ }
4273
+ },
4274
+ onListening: (url) => {
4275
+ console.log(` ${url}`);
4276
+ if (process.stdout.isTTY)
4277
+ openBestEffort(url);
4278
+ },
4279
+ });
4280
+ console.log("\n✓ live report closed");
4281
+ }
4189
4282
  /**
4190
4283
  * Run the model trigger tier with no `--prompts`: auto-generate diverse probe
4191
4284
  * prompts from each skill's description and measure trigger-rate (recall +
@@ -4210,19 +4303,54 @@ async function runAutoTrigger(dir, report, adapter, args) {
4210
4303
  if (!json) {
4211
4304
  console.log("\nℹ auto-generated probe prompts from skill descriptions (pass --prompts=<file> for a curated set).");
4212
4305
  }
4306
+ const model = flagValue(args, "--model");
4213
4307
  const trigger = await (0, scan_behavioral_js_1.probePluginTriggers)(dir, promptSet, {
4214
4308
  minPrompts: audit_prompts_js_1.AUTO_RECALL_COUNT,
4215
4309
  minDistance: audit_prompts_js_1.AUTO_MIN_DISTANCE,
4216
- model: flagValue(args, "--model"),
4310
+ model,
4217
4311
  harness,
4218
4312
  // Discover candidates with the resolved adapter's layout/dialect — a Codex
4219
4313
  // repo's skills live under the Codex layout, not the default CC one.
4220
4314
  layout: adapter.layout,
4221
4315
  dialect: adapter.dialect,
4222
4316
  });
4223
- console.log(json
4224
- ? JSON.stringify({ trigger }, null, 2)
4225
- : "\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
4317
+ // Second behavioral eval (same consent): the selection-collision matrix — does
4318
+ // one skill HIJACK a sibling's prompt? This is the MEASURED confirmation of the
4319
+ // deterministic description-overlap proxy (the Triggering ring flags look-alikes;
4320
+ // this proves the wrong one actually fires). Only meaningful with ≥2 model-
4321
+ // invocable skills (a lone skill can't collide); reuses the same auto prompts.
4322
+ const collisions = skills.length >= 2
4323
+ ? await (0, scan_behavioral_js_1.measurePluginSelection)(dir, promptSet, { model, harness })
4324
+ : null;
4325
+ // Third behavioral eval (same consent): adversarial-gate — do enforcement-gate
4326
+ // skills HOLD when the agent is told to violate them? Auto-derives its own
4327
+ // attacks; a no-op (no model calls) when the plugin declares no gate skills.
4328
+ const gates = await (0, scan_behavioral_js_1.measureGateAdversarial)(dir, {
4329
+ model,
4330
+ harness,
4331
+ layout: adapter.layout,
4332
+ dialect: adapter.dialect,
4333
+ });
4334
+ // Show the gate section when gate skills were DETECTED — even if the eval
4335
+ // couldn't RUN (a Codex audit, or no `claude` CLI) it returns available:false
4336
+ // with empty results, and `formatGateReport` renders the "unavailable" note.
4337
+ // The consent prompt already advertised these gate skills, so a skipped check
4338
+ // must be reported LOUDLY, never silently omitted as if there were none.
4339
+ const hasGates = gates.results.length > 0 || (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length > 0;
4340
+ if (json) {
4341
+ console.log(JSON.stringify({
4342
+ trigger,
4343
+ ...(collisions ? { collisions } : {}),
4344
+ ...(hasGates ? { gates } : {}),
4345
+ }, null, 2));
4346
+ }
4347
+ else {
4348
+ console.log("\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
4349
+ if (collisions)
4350
+ console.log("\n" + (0, scan_behavioral_js_1.formatSelectionReport)(collisions));
4351
+ if (hasGates)
4352
+ console.log("\n" + (0, scan_behavioral_js_1.formatGateReport)(gates));
4353
+ }
4226
4354
  }
4227
4355
  /**
4228
4356
  * The ONE read-vs-run decision for a single-plugin `audit`. A plain `audit` is a
@@ -4272,7 +4400,17 @@ function buildExecuteDisclosure(s, harness) {
4272
4400
  if (s.hasMcp)
4273
4401
  lines.push(" · start your MCP servers — connects to their backends");
4274
4402
  if (s.triggerableSkills > 0) {
4275
- lines.push(` · measure whether skills fire (${triggerCostWording(harness)})`);
4403
+ // ≥2 model-invocable skills also get the selection-collision matrix (does one
4404
+ // skill hijack a sibling's prompt) — disclose it so the consent stays honest.
4405
+ const what = s.triggerableSkills >= 2
4406
+ ? "measure whether skills fire and collide"
4407
+ : "measure whether skills fire";
4408
+ lines.push(` · ${what} (${triggerCostWording(harness)})`);
4409
+ }
4410
+ if (s.gateSkills > 0) {
4411
+ // The adversarial-gate eval runs the FULL (unstubbed) skill — the most
4412
+ // expensive check — so disclose it separately when gate skills are present.
4413
+ lines.push(` · test whether ${String(s.gateSkills)} enforcement-gate skill${s.gateSkills === 1 ? "" : "s"} hold under pressure — runs the full skill (${triggerCostWording(harness)})`);
4276
4414
  }
4277
4415
  if (s.adoptableRefs) {
4278
4416
  lines.push(` · draft + verify your instruction file's references (${triggerCostWording(harness)})`);
@@ -4452,7 +4590,10 @@ async function main() {
4452
4590
  // through to a misleading "empty machine / no structural issues" report
4453
4591
  // (obra/superpowers-marketplace, anthropics/claude-plugins-community).
4454
4592
  if (json) {
4455
- console.log(JSON.stringify(market, null, 2));
4593
+ console.log(JSON.stringify((0, audit_report_js_1.buildMarketplaceReport)(market, {
4594
+ vigilesVersion: getVersion(),
4595
+ dir: (0, node_path_1.resolve)(dirs[0]),
4596
+ }), null, 2));
4456
4597
  }
4457
4598
  else {
4458
4599
  console.log(`Marketplace "${market.name}": ${String(market.total)} plugin(s), all external ` +
@@ -4468,7 +4609,12 @@ async function main() {
4468
4609
  const text = args.includes("--md")
4469
4610
  ? (0, leaderboard_js_1.formatLeaderboardMarkdown)(scores)
4470
4611
  : (0, leaderboard_js_1.formatLeaderboard)(scores);
4471
- console.log(json ? JSON.stringify(scores, null, 2) : text);
4612
+ console.log(json
4613
+ ? JSON.stringify((0, audit_report_js_1.buildLeaderboardReport)(scores, {
4614
+ vigilesVersion: getVersion(),
4615
+ dir: (0, node_path_1.resolve)(dirs[0]),
4616
+ }), null, 2)
4617
+ : text);
4472
4618
  }
4473
4619
  else {
4474
4620
  const root = (0, node_path_1.resolve)(targets[0]);
@@ -4546,6 +4692,7 @@ async function main() {
4546
4692
  const surfaces = {
4547
4693
  hasMcp: report.mcp && !isForeign,
4548
4694
  triggerableSkills: report.skills.filter((s) => s.hasDescription && !s.userInvoked).length,
4695
+ gateSkills: (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length,
4549
4696
  adoptableRefs: adapter.name === "claude-code" &&
4550
4697
  (0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, adapter.layout.instructionFile)),
4551
4698
  };
@@ -4632,17 +4779,36 @@ async function main() {
4632
4779
  const finalReport = adoptabilityResult
4633
4780
  ? { ...auditReport, adoptability: adoptabilityResult }
4634
4781
  : auditReport;
4635
- // The shareable HTML report — written by default (--no-html to skip), and
4636
- // opened best-effort only for a human at a TTY (never spawn a browser for
4637
- // an agent / CI run).
4638
- if (!json && !args.includes("--no-html")) {
4639
- writeAuditHtml(finalReport);
4640
- }
4641
4782
  // The versioned JSON artifact — the upload/CI boundary (a hosted dashboard
4642
4783
  // ingests this). Written by default in the human path; --no-json to skip.
4643
4784
  if (!json && !args.includes("--no-json")) {
4644
4785
  writeAuditJson(finalReport);
4645
4786
  }
4787
+ // The HTML report has two deliveries. STATIC (default): write the
4788
+ // shareable file whose buttons copy the `init` command. LIVE (`--serve`,
4789
+ // or a TTY "yes"): start a loopback server whose buttons run `init` for
4790
+ // you. The gate keeps the default a terminating, headless-safe read and
4791
+ // restricts the write-server to your own repo (decideServeGate).
4792
+ const serveGate = (0, audit_serve_js_1.decideServeGate)({
4793
+ serveFlag: args.includes("--serve"),
4794
+ noServeFlag: args.includes("--no-serve"),
4795
+ json,
4796
+ isTTY: process.stdout.isTTY && process.stdin.isTTY,
4797
+ ownRepo: !isForeign,
4798
+ adoptableCount: finalReport.adoptable?.surfaces.length ?? 0,
4799
+ });
4800
+ let serveLive = serveGate === "serve";
4801
+ if (serveGate === "ask") {
4802
+ const ans = (await askOnce("\nOpen the live report to create specs with one click? [y/N] ")).toLowerCase();
4803
+ serveLive = ans === "y" || ans === "yes";
4804
+ }
4805
+ const errMsg = (e) => e instanceof Error ? e.message : String(e);
4806
+ if (serveLive) {
4807
+ await runAuditServe(finalReport, finalReport.adoptable, errMsg);
4808
+ }
4809
+ else if (!json && !args.includes("--no-html")) {
4810
+ writeAuditHtml(finalReport);
4811
+ }
4646
4812
  }
4647
4813
  break;
4648
4814
  }
@@ -25,7 +25,7 @@ export declare function verifyHash(content: string): {
25
25
  */
26
26
  /** @internal */ export declare function estimateTokens(text: string): number;
27
27
  export interface CompileError {
28
- type: "stale-file" | "stale-command" | "stale-ref" | "invalid-rule" | "budget-exceeded" | "section-too-long" | "section-has-header" | "reserved-section-key" | "spec-name-mismatch" | "unknown-tool" | "invalid-railway" | "purity-violation" | "output-without-fork" | "effect-in-skill";
28
+ type: "stale-file" | "stale-command" | "stale-ref" | "invalid-rule" | "budget-exceeded" | "section-too-long" | "section-has-header" | "reserved-section-key" | "spec-name-mismatch" | "unknown-tool" | "invalid-railway" | "purity-violation" | "output-without-fork" | "effect-in-skill" | "inline-code-too-long";
29
29
  message: string;
30
30
  path?: string;
31
31
  }
@@ -76,6 +76,8 @@ export declare function compileClaude(spec: ClaudeSpec, options?: CompileClaudeO
76
76
  export interface CompileSkillResult {
77
77
  markdown: string;
78
78
  errors: CompileError[];
79
+ /** Non-blocking advisories (e.g. an over-long inline code block). */
80
+ warnings: CompileError[];
79
81
  }
80
82
  /**
81
83
  * Compile a SkillSpec into SKILL.md markdown with YAML frontmatter.
@@ -90,6 +92,8 @@ export declare function compileSkill(spec: SkillSpec, options?: {
90
92
  export interface CompileAgentResult {
91
93
  markdown: string;
92
94
  errors: CompileError[];
95
+ /** Non-blocking advisories (e.g. an over-long inline code block). */
96
+ warnings: CompileError[];
93
97
  }
94
98
  /**
95
99
  * Compile an AgentSpec into a subagent markdown file with YAML frontmatter.
@@ -664,11 +664,17 @@ function renderSkillSections(spec) {
664
664
  return sections.join("\n\n");
665
665
  }
666
666
  const DEFAULT_MAX_INLINE_CODE_LINES = 20;
667
- /** Flag inline fenced code blocks longer than `max` lines (0 = disabled). */
667
+ /**
668
+ * Flag inline fenced code blocks longer than `max` lines (0 = disabled). These
669
+ * are WARNINGS, not errors: a big inline code block is an authoring smell worth
670
+ * surfacing ("extract it to a file"), but it never breaks the harness — and a
671
+ * faithful adoption of an existing skill/subagent (`init`) must still compile.
672
+ * Callers route the result into a result's `warnings` channel, never `errors`.
673
+ */
668
674
  function checkInlineCode(markdown, max) {
669
675
  if (max <= 0)
670
676
  return [];
671
- const errs = [];
677
+ const warns = [];
672
678
  const lines = markdown.split("\n");
673
679
  let start = -1;
674
680
  let lang = "";
@@ -683,16 +689,16 @@ function checkInlineCode(markdown, max) {
683
689
  else {
684
690
  const len = i - start - 1;
685
691
  if (len > max) {
686
- errs.push({
687
- type: "section-too-long",
688
- message: `Inline ${lang || "code"} block is ${String(len)} lines (max ${String(max)}); extract it to a file and reference it with file().`,
692
+ warns.push({
693
+ type: "inline-code-too-long",
694
+ message: `Inline ${lang || "code"} block is ${String(len)} lines (max ${String(max)}); consider extracting it to a file and referencing it with file().`,
689
695
  });
690
696
  }
691
697
  start = -1;
692
698
  lang = "";
693
699
  }
694
700
  }
695
- return errs;
701
+ return warns;
696
702
  }
697
703
  /**
698
704
  * Compile a SkillSpec into SKILL.md markdown with YAML frontmatter.
@@ -756,14 +762,16 @@ function compileSkill(spec, options = {}) {
756
762
  }
757
763
  }
758
764
  const sections = renderSkillSections(spec);
759
- errors.push(...checkInlineCode(sections, spec.maxInlineCodeLines ?? DEFAULT_MAX_INLINE_CODE_LINES));
765
+ // Over-long inline code blocks are WARNINGS, not errors — they don't block
766
+ // compilation (so adoption always compiles), just nudge toward file().
767
+ const warnings = checkInlineCode(sections, spec.maxInlineCodeLines ?? DEFAULT_MAX_INLINE_CODE_LINES);
760
768
  const marker = purityMarker(spec.purity);
761
769
  const content = renderSkillFrontmatter(spec, profile) +
762
770
  "\n\n" +
763
771
  (marker ? marker + "\n\n" : "") +
764
772
  sections.trim() +
765
773
  "\n";
766
- return { markdown: addHash(content, specFile), errors };
774
+ return { markdown: addHash(content, specFile), errors, warnings };
767
775
  }
768
776
  // ---------------------------------------------------------------------------
769
777
  // Compile a subagent spec → agents/<name>.md
@@ -944,14 +952,15 @@ function compileAgent(spec, options) {
944
952
  if (spec.output)
945
953
  sections.push(renderOutputContract(spec.output));
946
954
  const body = sections.join("\n\n");
947
- errors.push(...checkInlineCode(body, DEFAULT_MAX_INLINE_CODE_LINES));
955
+ // Over-long inline code blocks are WARNINGS, not errors (see checkInlineCode).
956
+ const warnings = checkInlineCode(body, DEFAULT_MAX_INLINE_CODE_LINES);
948
957
  const marker = purityMarker(spec.purity);
949
958
  const content = renderAgentFrontmatter(spec) +
950
959
  "\n\n" +
951
960
  (marker ? marker + "\n\n" : "") +
952
961
  body.trim() +
953
962
  "\n";
954
- return { markdown: addHash(content, specFile), errors };
963
+ return { markdown: addHash(content, specFile), errors, warnings };
955
964
  }
956
965
  /** Verify a railway: non-empty, bounded recovery, every delegate target real. */
957
966
  function validateRailway(rw, knownAgents) {
@@ -48,8 +48,10 @@ exports.W_MISSING_HOOK = 15; // a hook script that doesn't exist → never runs
48
48
  exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't trigger
49
49
  exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
50
50
  exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
51
- exports.W_NO_CONTRACT = 5; // an agent with no `tools:` line → inherits everything
52
- // (untested surfaces are advisory, not a penalty — see scoreReport)
51
+ exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
52
+ // Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
53
+ // - untested surfaces — a hardening gap, not breakage.
54
+ // - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
53
55
  /** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
54
56
  function gradeFor(score) {
55
57
  if (score >= 90)
@@ -72,7 +74,6 @@ function gradeFor(score) {
72
74
  function reportDeductions(r) {
73
75
  const missingHooks = r.hooks.filter((h) => h.status === "missing").length;
74
76
  const noDesc = r.skills.filter((s) => !s.hasDescription).length;
75
- const noContract = r.agents.filter((a) => a.tools === null).length;
76
77
  const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
77
78
  const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
78
79
  const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
@@ -117,11 +118,14 @@ function reportDeductions(r) {
117
118
  weight: exports.W_NO_CONTRACT,
118
119
  label: "agent disallowedTools typo(s) that block nothing",
119
120
  },
120
- {
121
- n: noContract,
122
- weight: exports.W_NO_CONTRACT,
123
- label: "agent(s) inherit all tools (no contract)",
124
- },
121
+ // NB: an agent that inherits all tools (no `tools:` line) is ADVISORY, not a
122
+ // graded penalty — it's surfaced by scoreReport / the Structure ring but never
123
+ // drags the score. WHY: omitting the `tools:` line is a near-universal,
124
+ // legitimate authoring style (a measured OSS sweep of 122 real plugins found
125
+ // 109 whose ONLY finding was this), so penalizing it makes the grade cry wolf
126
+ // on idiomatic subagents. A health score should mean "something is BROKEN", and
127
+ // a broad-by-default tool surface is a hardening/least-privilege NUDGE, not
128
+ // breakage. The count is re-derived where the advisory note is built.
125
129
  {
126
130
  n: r.frontmatterIssues.length,
127
131
  weight: exports.W_NO_DESCRIPTION,
@@ -192,8 +196,15 @@ function scoreReport(r) {
192
196
  }
193
197
  // Sort issues by cost (worst first) so the report leads with what matters.
194
198
  issues.sort((a, b) => Number(b.split(" ")[0]) - Number(a.split(" ")[0]));
195
- // Untested surfaces are advisory — surfaced for visibility, but they don't
196
- // affect the score, so they come AFTER the real (score-affecting) issues.
199
+ // Advisory notes are surfaced for visibility but DON'T affect the score, so they
200
+ // come AFTER the real (score-affecting) issues:
201
+ // - inherit-all (no tool contract): a least-privilege NUDGE, not breakage —
202
+ // see reportDeductions for the full rationale.
203
+ // - untested surfaces: a hardening gap, not breakage.
204
+ const noContract = r.agents.filter((a) => a.tools === null).length;
205
+ if (noContract > 0) {
206
+ issues.push(`${String(noContract)} agent(s) inherit all tools (no contract) (advisory)`);
207
+ }
197
208
  if (r.untested > 0) {
198
209
  issues.push(`${String(r.untested)} untested surface(s) (advisory)`);
199
210
  }
@@ -228,12 +239,12 @@ function formatLeaderboard(scores) {
228
239
  const issue = s.issues.length > 0 ? ` — ${s.issues.join("; ")}` : "";
229
240
  out.push(` ${rank} ${score} ${s.grade} ${s.name}${issue}`);
230
241
  });
231
- out.push("", "Structural health only (no model). Weights: missing hook -15, no-description", "skill -10, broken intra-plugin ref -8, agent-without-tool-contract -5.", "Untested surfaces are advisory — shown, but they don't affect the score.");
242
+ out.push("", "Structural health only (no model). Weights: missing hook -15, no-description", "skill -10, broken intra-plugin ref -8, dead tool/MCP ref -8.", "Inherit-all subagents and untested surfaces are advisory — shown, not scored.");
232
243
  return out.join("\n");
233
244
  }
234
245
  const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): missing hook −15, " +
235
- "no-description skill −10, broken intra-plugin ref −8, " +
236
- "agent-without-tool-contract −5. Untested surfaces are advisory (shown, not " +
246
+ "no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
247
+ "Inherit-all subagents and untested surfaces are advisory (shown, not " +
237
248
  "scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
238
249
  /**
239
250
  * Format a ranked leaderboard as a Markdown table — the PUBLISHABLE form (a README,
@@ -135,5 +135,90 @@ export declare function measurePluginSelectionWith(dir: string, promptSet: Trigg
135
135
  export declare function measurePluginSelection(dir: string, promptSet: TriggerPromptSet, opts?: SelectionOptions): Promise<SelectionReport>;
136
136
  /** Format the selection-collision matrix as a scan-report section. */
137
137
  export declare function formatSelectionReport(r: SelectionReport): string;
138
+ /** Does a skill description assert a hard constraint (→ an adversarial-gate candidate)? */
139
+ export declare function isGateDescription(description: string): boolean;
140
+ /** A skill considered for gate detection — name + its (model-visible) description. */
141
+ export interface GateCandidate {
142
+ readonly name: string;
143
+ readonly description?: string;
144
+ readonly userInvoked?: boolean;
145
+ readonly hasDescription?: boolean;
146
+ }
147
+ /**
148
+ * The model-invocable, described skills whose description reads as an enforcement
149
+ * gate — the candidates for the adversarial-gate eval. User-invoked and
150
+ * description-less skills are excluded (they can't auto-fire a constraint on the
151
+ * model's behaviour), mirroring the trigger-rate candidate filter.
152
+ */
153
+ export declare function detectGateSkills(skills: readonly GateCandidate[]): readonly string[];
154
+ /** A gate under test: its name + the rule its description states. */
155
+ export interface GateUnderTest {
156
+ readonly name: string;
157
+ readonly description: string;
158
+ }
159
+ /** The verdict an injected judge returns (a subset of judge.ts JudgeResult). */
160
+ export interface GateVerdict {
161
+ readonly pass: boolean;
162
+ readonly score: number;
163
+ readonly reason: string;
164
+ }
165
+ /** Injected dependencies, so the orchestration is unit-testable with no model. */
166
+ export interface GateEvalDeps {
167
+ readonly driver: EvalDriver;
168
+ /** Grade whether the gate held, given the run output + the rule rubric. */
169
+ readonly judge: (a: {
170
+ output: string;
171
+ rubric: string;
172
+ }) => GateVerdict;
173
+ /** Turn a gate's rule into a one-line user request that tries to violate it. */
174
+ readonly derive: (gate: GateUnderTest) => string;
175
+ }
176
+ export interface GateOptions {
177
+ /** Selector model for the harness run (default sonnet). */
178
+ readonly model?: string;
179
+ /** Attacks per gate (default 1). */
180
+ readonly trials?: number;
181
+ /** Concurrent harness runs (default 1). */
182
+ readonly concurrency?: number;
183
+ readonly harness?: ProbeHarness;
184
+ readonly layout?: PluginLayout;
185
+ readonly dialect?: HarnessDialect;
186
+ /** Author-supplied attack prompts (bare skill name → prompts); overrides derive. */
187
+ readonly attacks?: Record<string, readonly string[]>;
188
+ }
189
+ export interface GateResult {
190
+ readonly skill: string;
191
+ readonly measured: boolean;
192
+ /** Fraction of attacks the gate HELD (1 = held every time). */
193
+ readonly heldRate?: number;
194
+ /** Convenience: held on EVERY attack (a single cave → false). */
195
+ readonly held?: boolean;
196
+ readonly n?: number;
197
+ /** The attack used (first), for the report. */
198
+ readonly attack?: string;
199
+ /** The judge's rationale on a representative cave (else a hold), for the report. */
200
+ readonly reason?: string;
201
+ readonly note?: string;
202
+ }
203
+ export interface GateAdversarialReport {
204
+ readonly available: boolean;
205
+ readonly results: readonly GateResult[];
206
+ readonly note?: string;
207
+ }
208
+ /** Build the LLM-judge rubric from the gate's own rule (pure). */
209
+ export declare function gateRubric(gate: GateUnderTest): string;
210
+ /** The injectable core (unit-testable with fake driver/judge/derive — no model). */
211
+ export declare function measureGateAdversarialWith(dir: string, gates: readonly GateUnderTest[], deps: GateEvalDeps, opts?: GateOptions): Promise<GateAdversarialReport>;
212
+ /**
213
+ * Measure whether a plugin's enforcement-gate skills HOLD when adversarially
214
+ * challenged. Detects gate skills (keyword heuristic), auto-derives an attack from
215
+ * each rule (unless author-supplied), runs the UNSTUBBED harness, and LLM-judges
216
+ * hold vs cave. Claude Code only; degrades to `available: false` without the CLI/auth.
217
+ */
218
+ export declare function measureGateAdversarial(dir: string, opts?: GateOptions): Promise<GateAdversarialReport>;
219
+ /** Default adversarial attacks per gate (stochastic → need >1; unstubbed → keep low). */
220
+ export declare const DEFAULT_GATE_TRIALS = 3;
221
+ /** Format the adversarial-gate report as a scan-report section. */
222
+ export declare function formatGateReport(r: GateAdversarialReport): string;
138
223
  export {};
139
224
  //# sourceMappingURL=scan-behavioral.d.ts.map