vigiles 10.0.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. package/.claude-plugin/plugin.json +9 -0
  2. package/README.md +121 -86
  3. package/action.yml +13 -2
  4. package/dist/adapter-conformance.js +6 -0
  5. package/dist/adapter-registry.d.ts +20 -0
  6. package/dist/adapter-registry.js +27 -0
  7. package/dist/adapters/claude-code/dialect.js +15 -0
  8. package/dist/adapters/claude-code/hook-protocol.js +4 -0
  9. package/dist/adapters/claude-code/runtime.js +12 -0
  10. package/dist/adapters/codex/eval.js +3 -0
  11. package/dist/adapters/codex/hook-protocol.d.ts +9 -1
  12. package/dist/adapters/codex/hook-protocol.js +10 -0
  13. package/dist/adapters/codex/runtime.js +10 -0
  14. package/dist/adapters/opencode/runtime.js +4 -0
  15. package/dist/audit-report.d.ts +1 -1
  16. package/dist/audit-report.template.html +1 -1
  17. package/dist/audit-score.d.ts +19 -12
  18. package/dist/audit-score.js +65 -11
  19. package/dist/cli-commands.d.ts +1 -1
  20. package/dist/cli-commands.js +1 -0
  21. package/dist/cli.js +460 -29
  22. package/dist/core/CLAUDE.md.spec.d.ts +3 -0
  23. package/dist/core/CLAUDE.md.spec.js +26 -0
  24. package/dist/core/delegation-trifecta.d.ts +64 -0
  25. package/dist/core/delegation-trifecta.js +124 -0
  26. package/dist/core/dialect.d.ts +18 -0
  27. package/dist/core/hook-block-ineffective.d.ts +62 -0
  28. package/dist/core/hook-block-ineffective.js +153 -0
  29. package/dist/core/hook-matcher.d.ts +66 -0
  30. package/dist/core/hook-matcher.js +182 -0
  31. package/dist/core/hook-normalize.d.ts +43 -0
  32. package/dist/core/hook-normalize.js +78 -0
  33. package/dist/core/hook-protocol.d.ts +15 -0
  34. package/dist/core/lethal-trifecta.d.ts +100 -0
  35. package/dist/core/lethal-trifecta.js +197 -0
  36. package/dist/core/plugin-dir-layout.d.ts +30 -0
  37. package/dist/core/plugin-dir-layout.js +73 -0
  38. package/dist/core/rule-meta.d.ts +82 -0
  39. package/dist/core/rule-meta.js +266 -0
  40. package/dist/core/runtime.d.ts +20 -0
  41. package/dist/core/skill-missing-fence.d.ts +47 -0
  42. package/dist/core/skill-missing-fence.js +119 -0
  43. package/dist/core/skill-resources.d.ts +27 -0
  44. package/dist/core/skill-resources.js +167 -0
  45. package/dist/core/types.d.ts +83 -0
  46. package/dist/core/validate.d.ts +1 -0
  47. package/dist/core/validate.js +26 -4
  48. package/dist/eval-cache.d.ts +6 -0
  49. package/dist/eval-cache.js +2 -0
  50. package/dist/eval-lock.d.ts +192 -0
  51. package/dist/eval-lock.js +286 -0
  52. package/dist/eval.d.ts +33 -20
  53. package/dist/eval.js +199 -51
  54. package/dist/leaderboard.d.ts +1 -0
  55. package/dist/leaderboard.js +42 -4
  56. package/dist/scan.d.ts +106 -0
  57. package/dist/scan.js +251 -45
  58. package/dist/setup-plan.d.ts +43 -3
  59. package/dist/setup-plan.js +78 -6
  60. package/hooks/eval-lock-nudge.sh +21 -0
  61. package/package.json +1 -1
  62. package/skills/test-harness/SKILL.md +27 -0
package/dist/eval.js CHANGED
@@ -22,7 +22,6 @@ exports.aggregateUsage = aggregateUsage;
22
22
  exports.isDatedModel = isDatedModel;
23
23
  exports.modelTier = modelTier;
24
24
  exports.belowModelFloor = belowModelFloor;
25
- exports.harnessVersionKey = harnessVersionKey;
26
25
  exports.ephemeralRunEnv = ephemeralRunEnv;
27
26
  exports.seedEphemeralHome = seedEphemeralHome;
28
27
  exports.isRateLimited = isRateLimited;
@@ -71,6 +70,7 @@ const runtime_js_1 = require("./adapters/claude-code/runtime.js");
71
70
  const proofs_js_1 = require("./core/proofs.js");
72
71
  const harness_test_js_1 = require("./harness-test.js");
73
72
  const eval_cache_js_1 = require("./eval-cache.js");
73
+ const eval_lock_js_1 = require("./eval-lock.js");
74
74
  const stats_js_1 = require("./stats.js");
75
75
  const tool_intercept_js_1 = require("./tool-intercept.js");
76
76
  const tool_stub_js_1 = require("./tool-stub.js");
@@ -640,32 +640,22 @@ function belowModelFloor(model, floor) {
640
640
  const f = modelTier(floor);
641
641
  return m !== null && f !== null && m < f;
642
642
  }
643
- /**
644
- * Reduce a raw `--version` string to the **major.minor** cache-key token. We key
645
- * the cache on major.minor, NOT the patch: a patch release rarely changes agent
646
- * behaviour, so keying patches would churn the cache on every release for no
647
- * signal; a minor/major bump is where the system prompt / tool defs actually move.
648
- * (If a specific patch is known to matter, clear the cache or bump
649
- * `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
650
- * found. Pure + tested.
651
- */
652
- function harnessVersionKey(raw) {
653
- const m = /(\d+)\.(\d+)\.\d+/.exec(raw);
654
- return m ? `${m[1]}.${m[2]}` : raw.trim();
655
- }
656
643
  /* v8 ignore start -- spawns the real harness binary; memoized, cache-path only */
657
644
  let cachedHarnessVersion;
658
645
  /**
659
- * The harness binary version (`claude --version`) reduced to major.minor, for the
660
- * cache key — so a CLI **minor/major** upgrade (new system prompt / tool defs)
661
- * invalidates a stale replay, while patches don't churn it. Memoized (one spawn
662
- * per process), resolved only on the cache path, "unknown" if the binary isn't
663
- * found (then it doesn't partition the key).
646
+ * The harness binary version (`claude --version`), reduced to its
647
+ * behaviorally-significant token by the runtime port's
648
+ * {@link HarnessRuntime.versionKey} — so a Claude Code **minor/major** upgrade
649
+ * (new system prompt / tool defs) invalidates a stale replay while patches don't
650
+ * churn it. The reduction is per-harness (CC → `major.minor`, Codex → `""`) and
651
+ * lives on the adapter, not here. Memoized (one spawn per process), resolved only
652
+ * on the cache path, "unknown" if the binary isn't found (then it doesn't
653
+ * partition the key).
664
654
  */
665
655
  function harnessVersion() {
666
656
  if (cachedHarnessVersion === undefined) {
667
657
  try {
668
- cachedHarnessVersion = harnessVersionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
658
+ cachedHarnessVersion = runtime_js_1.claudeCodeRuntime.versionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
669
659
  encoding: "utf-8",
670
660
  stdio: ["ignore", "pipe", "ignore"],
671
661
  }));
@@ -983,15 +973,141 @@ function aggregateArms(armNames, results) {
983
973
  }
984
974
  return { arms, totalCostUsd };
985
975
  }
976
+ function resolveLock(over) {
977
+ return {
978
+ mode: over?.mode ?? (0, eval_lock_js_1.lockModeFromEnv)(),
979
+ dir: over?.dir ?? (0, node_path_1.resolve)(process.cwd(), eval_lock_js_1.DEFAULT_LOCK_DIR),
980
+ evalApiVersion: over?.evalApiVersion ?? (0, eval_lock_js_1.evalApiVersionFromEnv)(),
981
+ };
982
+ }
983
+ /** Emit a vigiles message (GitHub annotation under Actions, else stderr/stdout). */
984
+ function emitLockMessage(msg, warn) {
985
+ if (process.env.GITHUB_ACTIONS)
986
+ console.log(`::${warn ? "warning" : "notice"}::${msg}`);
987
+ else if (warn)
988
+ console.warn(msg);
989
+ else
990
+ console.log(msg);
991
+ }
992
+ /**
993
+ * Run a named eval through the lock. `off` → just `produce()`. `check` → replay a
994
+ * matching committed lock (NO model call) or throw "stale". `update` → `produce()`,
995
+ * write the lock, print the human-facing delta. The model is driven ONLY on the
996
+ * run path — never on a clean `check`, which is what keeps the CI gate binary-free.
997
+ * A lock-on run with no `name` is a LOUD skip (the lock needs a name to key the file).
998
+ */
999
+ async function withEvalLock(args, produce) {
1000
+ const { lock } = args;
1001
+ if (lock.mode === "off")
1002
+ return produce();
1003
+ if (!args.name) {
1004
+ // An unnamed eval can't be keyed to a lock. In `check` (the CI gate) running
1005
+ // it would call the model — violating the no-model-in-CI contract and failing
1006
+ // for missing auth — so FAIL LOUDLY instead of silently hitting the model.
1007
+ // `update` (local, model available) keeps producing: the eval just isn't gated.
1008
+ if (lock.mode === "check") {
1009
+ throw new Error(`vigiles eval --check: an unnamed eval cannot run in CI — it has no ` +
1010
+ `committed lock to replay, and running it would call the model. Add a ` +
1011
+ `\`name\` to the spec to gate it, or exclude it from the --check run.`);
1012
+ }
1013
+ emitLockMessage(`vigiles eval --${lock.mode}: skipped the lock for an unnamed eval — set ` +
1014
+ `\`name\` on the spec to enable the staleness gate for it.`, true);
1015
+ return produce();
1016
+ }
1017
+ if (!isDatedModel(args.model))
1018
+ warnFloatingModel(args.model);
1019
+ const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
1020
+ model: args.model,
1021
+ evalApiVersion: lock.evalApiVersion,
1022
+ inputs: args.inputs,
1023
+ });
1024
+ const existing = (0, eval_lock_js_1.readLock)(lock.dir, args.name);
1025
+ const decision = (0, eval_lock_js_1.decideLock)(lock.mode, args.name, inputsHash, existing);
1026
+ if (decision.kind === "stale")
1027
+ throw new Error(decision.reason);
1028
+ if (decision.kind === "replay")
1029
+ return decision.report;
1030
+ const report = await produce();
1031
+ const builtLock = (0, eval_lock_js_1.buildLock)({
1032
+ name: args.name,
1033
+ inputsHash,
1034
+ model: args.model,
1035
+ harnessVersionKey: harnessVersion(),
1036
+ evalApiVersion: lock.evalApiVersion,
1037
+ builtAt: new Date().toISOString(),
1038
+ report,
1039
+ });
1040
+ const deltas = existing ? (0, eval_lock_js_1.diffReportNumbers)(existing.report, report) : [];
1041
+ (0, eval_lock_js_1.writeLock)(lock.dir, builtLock);
1042
+ emitLockMessage((0, eval_lock_js_1.formatLockUpdate)(args.name, deltas, existing === null), false);
1043
+ return report;
1044
+ }
1045
+ /**
1046
+ * Strip the machine-specific plugin-root prefix from a resolved value before it
1047
+ * enters the lock hash. `resolveHarness` expands `${PLUGIN_ROOT}` in a plugin's
1048
+ * hook commands to the checkout's ABSOLUTE path, so a lock recorded at
1049
+ * `/home/dev/...` would be falsely STALE when `--check` recomputes it at
1050
+ * `/home/runner/...` in CI (or any other machine). Normalizing the prefix back to
1051
+ * a token makes the hash location-independent. No-op when there's no plugin root.
1052
+ */
1053
+ function stripPluginRoot(value, absRoot) {
1054
+ if (absRoot === "")
1055
+ return value;
1056
+ const json = JSON.stringify(value);
1057
+ // A hookless plugin resolves to `settings: undefined`, which `JSON.stringify`
1058
+ // returns as `undefined` (not a string) — pass it through rather than `.split`
1059
+ // a non-string (which would throw before the eval can run).
1060
+ if (json === undefined)
1061
+ return value;
1062
+ return JSON.parse(json.split(absRoot).join("${PLUGIN_ROOT}"));
1063
+ }
986
1064
  /**
987
- * The eval orchestration — every arm × trial via `runner`, run through the cache
988
- * and a rate-limit retry, with at most `concurrency` in flight and an optional
989
- * `maxCostUsd` budget cap; metric + usage computed per run and aggregated per
990
- * arm. Exported with an injectable `runner` so the loop, `measure` context,
991
- * caching, pooling, and aggregation are unit-testable without spawning a model
992
- * (pass a fake returning canned stream-json). `runEval` is this with the real
993
- * agent runner.
1065
+ * Resolve each arm's model-affecting inputs into a canonical object for the lock
1066
+ * hash — WITHOUT running the model (it reads files + hashes plugin dirs only). The
1067
+ * trial count is excluded (a sample-size knob, not a behavior input); `measure` is
1068
+ * excluded by design (the script re-asserts against the replayed report).
994
1069
  */
1070
+ function evalArmsInputs(spec, cfg) {
1071
+ const arms = {};
1072
+ for (const [name, arm] of Object.entries(spec.arms)) {
1073
+ const resolved = (0, plugin_loader_js_1.resolveHarness)({
1074
+ plugin: arm.plugin,
1075
+ settings: arm.settings,
1076
+ files: { ...spec.fixture, ...arm.files },
1077
+ });
1078
+ // The plugin root `resolveHarness` expanded into the resolved files/settings
1079
+ // is this checkout's absolute path — normalize it out so the hash is the same
1080
+ // on the dev's machine and in CI (else every plugin-with-root-hooks eval is
1081
+ // falsely stale across machines).
1082
+ const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
1083
+ arms[name] = {
1084
+ model: arm.model ?? cfg.model,
1085
+ tools: [...cfg.tools].sort(),
1086
+ files: stripPluginRoot(resolved.files, absRoot),
1087
+ settings: stripPluginRoot(resolved.settings, absRoot),
1088
+ pluginDirHash: arm.pluginDir ? (0, eval_cache_js_1.hashDir)(arm.pluginDir) : undefined,
1089
+ interceptTools: arm.interceptTools
1090
+ ? (0, tool_intercept_js_1.serializeIntercepts)(arm.interceptTools)
1091
+ : undefined,
1092
+ };
1093
+ }
1094
+ // Tool stubs (`spec.stubs`) are written onto PATH before each trial, so a
1095
+ // change to a canned CLI output IS a model-facing input change — fold a
1096
+ // canonical (name-sorted) view into the hash so `--check` catches it. Sorted
1097
+ // for a stable key regardless of declaration order; an empty list is the
1098
+ // byte-identical-to-before default. Each ToolStub is plain serializable data.
1099
+ const stubs = [...(spec.stubs ?? [])].sort((a, b) => a.name.localeCompare(b.name));
1100
+ // `ephemeralEnv` swaps the trial's environment (scrubbed env + throwaway HOME
1101
+ // vs the inherited process env), which can move tool/hook/agent behavior — a
1102
+ // model-facing input, so it belongs in the hash. Normalize to a bool so a
1103
+ // record under one mode can't be replayed under the other with the same hash.
1104
+ return {
1105
+ task: spec.task,
1106
+ arms,
1107
+ stubs,
1108
+ ephemeralEnv: spec.ephemeralEnv === true,
1109
+ };
1110
+ }
995
1111
  async function runEvalWith(spec, runner) {
996
1112
  const trials = spec.trials ?? 5;
997
1113
  const spacing = (spec.spacingSec ?? 4) * 1000;
@@ -1024,9 +1140,16 @@ async function runEvalWith(spec, runner) {
1024
1140
  await sleep(spacing);
1025
1141
  return { armName: unit.armName, skipped: false, row, usage };
1026
1142
  };
1027
- const results = await runPool(units, concurrency, worker);
1028
- const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
1029
- return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
1143
+ const lock = resolveLock(spec.lock);
1144
+ // Resolve the lock inputs only when the lock is active (off skips the
1145
+ // resolveHarness/hashDir work). `check` replays the committed report below
1146
+ // without ever entering the run pool — so no model is driven in CI.
1147
+ const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
1148
+ return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
1149
+ const results = await runPool(units, concurrency, worker);
1150
+ const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
1151
+ return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
1152
+ });
1030
1153
  }
1031
1154
  /** Render one metric: `name=mean±se pass^k=…` (se/pass^k shown when measured). */
1032
1155
  function formatMetric(name, mean, stat) {
@@ -1065,6 +1188,7 @@ function formatEvalReport(report) {
1065
1188
  exports.claudeEvalDriver = {
1066
1189
  runner: spawnAgent,
1067
1190
  parse: parseClaudeRun,
1191
+ harness: "claude-code",
1068
1192
  };
1069
1193
  /**
1070
1194
  * Package loose `<skillsDir>/<name>/SKILL.md` skills into a throwaway plugin dir
@@ -1439,7 +1563,7 @@ function assertTriggerDiversity(spec) {
1439
1563
  });
1440
1564
  }
1441
1565
  }
1442
- async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError) {
1566
+ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
1443
1567
  // Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
1444
1568
  assertTriggerDiversity(spec);
1445
1569
  // Model floor (default Sonnet): trigger-rate under-measures selection on a
@@ -1469,26 +1593,50 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1469
1593
  parse,
1470
1594
  runError,
1471
1595
  };
1596
+ const lock = resolveLock(spec.lock);
1472
1597
  try {
1473
- const relevant = await runTriggerSet(spec.prompts, cfg, runner);
1474
- const base = {
1475
- rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
1476
- n: relevant.n,
1477
- perPrompt: relevant.perPrompt,
1478
- competitors,
1479
- errored: positiveOrUndefined(relevant.errored),
1480
- };
1481
- if ((spec.irrelevantPrompts?.length ?? 0) === 0)
1482
- return base;
1483
- const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
1484
- const fires = relevant.fired + irrelevant.fired;
1485
- return {
1486
- ...base,
1487
- errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
1488
- falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
1489
- precision: fires > 0 ? relevant.fired / fires : undefined,
1490
- perIrrelevant: irrelevant.perPrompt,
1491
- };
1598
+ // The lock hashes the skill UNDER TEST (its dir contents) + the prompts +
1599
+ // model + tools — the inputs that steer whether it fires. `--check` replays
1600
+ // the committed report with no model; `--update` records it. Resolved only
1601
+ // when the lock is active. `trials` is excluded (a sample-size knob).
1602
+ const triggerInputs = lock.mode === "off"
1603
+ ? undefined
1604
+ : {
1605
+ pluginDirHash: (0, eval_cache_js_1.hashDir)(pluginDir),
1606
+ prompts: [...spec.prompts],
1607
+ irrelevantPrompts: spec.irrelevantPrompts
1608
+ ? [...spec.irrelevantPrompts]
1609
+ : undefined,
1610
+ model: cfg.model,
1611
+ tools: [...cfg.tools].sort(),
1612
+ fixture: spec.fixture,
1613
+ competitors,
1614
+ // The harness the driver runs is a model-facing input — a Claude vs
1615
+ // Codex run can fire a skill differently — so a recorded report is
1616
+ // STALE if the eval is switched to another harness.
1617
+ harness,
1618
+ };
1619
+ return await withEvalLock({ name: spec.name, inputs: triggerInputs, model: cfg.model, lock }, async () => {
1620
+ const relevant = await runTriggerSet(spec.prompts, cfg, runner);
1621
+ const base = {
1622
+ rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
1623
+ n: relevant.n,
1624
+ perPrompt: relevant.perPrompt,
1625
+ competitors,
1626
+ errored: positiveOrUndefined(relevant.errored),
1627
+ };
1628
+ if ((spec.irrelevantPrompts?.length ?? 0) === 0)
1629
+ return base;
1630
+ const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
1631
+ const fires = relevant.fired + irrelevant.fired;
1632
+ return {
1633
+ ...base,
1634
+ errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
1635
+ falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
1636
+ precision: fires > 0 ? relevant.fired / fires : undefined,
1637
+ perIrrelevant: irrelevant.perPrompt,
1638
+ };
1639
+ });
1492
1640
  }
1493
1641
  finally {
1494
1642
  // Remove the throwaway plugin dir we built from a loose `skillsDir`.
@@ -1506,7 +1654,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
1506
1654
  */
1507
1655
  async function measureTriggerRate(spec, opts = {}) {
1508
1656
  const d = opts.evalDriver ?? exports.claudeEvalDriver;
1509
- return measureTriggerRateWith(spec, d.runner, d.parse, d.runError);
1657
+ return measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
1510
1658
  }
1511
1659
  /* v8 ignore stop */
1512
1660
  /** Format a trigger-rate report: overall %, then each prompt's rate. */
@@ -26,6 +26,7 @@ export declare const W_NO_DESCRIPTION = 10;
26
26
  export declare const W_DANGLING_REF = 8;
27
27
  export declare const W_OVERLAP = 8;
28
28
  export declare const W_NO_CONTRACT = 5;
29
+ export declare const W_TRIFECTA = 20;
29
30
  /** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
30
31
  export declare function gradeFor(score: number): PluginScore["grade"];
31
32
  /** One deduction: a count, its per-item weight, and the label if non-zero. */
@@ -12,7 +12,7 @@
12
12
  * model and stack on top later; this part runs anywhere in CI for free.
13
13
  */
14
14
  Object.defineProperty(exports, "__esModule", { value: true });
15
- exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
15
+ exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
16
16
  exports.gradeFor = gradeFor;
17
17
  exports.reportDeductions = reportDeductions;
18
18
  exports.isEmptyMachine = isEmptyMachine;
@@ -49,6 +49,7 @@ exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't t
49
49
  exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
50
50
  exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
51
51
  exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
52
+ exports.W_TRIFECTA = 20; // a HARD lethal-trifecta contract (all three legs, explicit) → a declared prompt-injection exfil path
52
53
  // Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
53
54
  // - untested surfaces — a hardening gap, not breakage.
54
55
  // - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
@@ -77,7 +78,17 @@ function reportDeductions(r) {
77
78
  const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
78
79
  const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
79
80
  const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
81
+ // HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
82
+ // legs (a declared exfil path). Advisory (inherits-all) trifecta findings are
83
+ // surfaced but NEVER graded (aligned with the inherits-all stance), so they're
84
+ // excluded here.
85
+ const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
80
86
  return [
87
+ {
88
+ n: hardTrifecta,
89
+ weight: exports.W_TRIFECTA,
90
+ label: "unit(s) holding all three lethal-trifecta legs (prompt-injection exfil path)",
91
+ },
81
92
  {
82
93
  n: missingHooks,
83
94
  weight: exports.W_MISSING_HOOK,
@@ -146,6 +157,33 @@ function reportDeductions(r) {
146
157
  weight: exports.W_DANGLING_REF,
147
158
  label: "mcp_tool hook(s) incomplete / targeting an undeclared server",
148
159
  },
160
+ {
161
+ n: r.skillResourceIssues.length,
162
+ weight: exports.W_DANGLING_REF,
163
+ label: "skill bundled-resource ref(s) that don't resolve on disk",
164
+ },
165
+ {
166
+ n: r.skillFenceIssues.length,
167
+ weight: exports.W_NO_DESCRIPTION,
168
+ label: "invisible skill(s) (frontmatter with no opening `---` fence)",
169
+ },
170
+ {
171
+ n: r.pluginLayoutIssues.length,
172
+ weight: exports.W_NO_DESCRIPTION,
173
+ label: "functional dir(s) misplaced inside `.claude-plugin/` (invisible)",
174
+ },
175
+ {
176
+ n: r.hookBlockFindings.length,
177
+ weight: exports.W_MISSING_HOOK,
178
+ label: "hook(s) that look like they block but silently don't",
179
+ },
180
+ {
181
+ n: r.hookMatcherFindings.length,
182
+ weight: exports.W_MISSING_HOOK,
183
+ label: "hook matcher(s) that never fire (typo / wrong MCP form)",
184
+ },
185
+ // NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
186
+ // ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
149
187
  // NB: untested surfaces are NOT a penalty — an untested surface is a hardening
150
188
  // gap, not breakage, so it never drags the health score (it's appended as an
151
189
  // advisory note below). The score ranks what's BROKEN.
@@ -239,11 +277,11 @@ function formatLeaderboard(scores) {
239
277
  const issue = s.issues.length > 0 ? ` — ${s.issues.join("; ")}` : "";
240
278
  out.push(` ${rank} ${score} ${s.grade} ${s.name}${issue}`);
241
279
  });
242
- out.push("", "Structural health only (no model). Weights: missing hook -15, no-description", "skill -10, broken intra-plugin ref -8, dead tool/MCP ref -8.", "Inherit-all subagents and untested surfaces are advisory — shown, not scored.");
280
+ out.push("", "Structural health only (no model). Weights: lethal-trifecta unit -20, missing", "hook -15, no-description skill -10, broken intra-plugin ref -8, dead tool/MCP", "ref -8. Inherit-all subagents and untested surfaces are advisory — shown, not", "scored.");
243
281
  return out.join("\n");
244
282
  }
245
- const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): missing hook −15, " +
246
- "no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
283
+ const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): lethal-trifecta unit −20, " +
284
+ "missing hook −15, no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
247
285
  "Inherit-all subagents and untested surfaces are advisory (shown, not " +
248
286
  "scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
249
287
  /**
package/dist/scan.d.ts CHANGED
@@ -20,6 +20,13 @@ import { type DescriptionOverlap } from "./core/description-overlap.js";
20
20
  import { type McpToolIssue } from "./core/mcp-tool.js";
21
21
  import { type McpContractToolError } from "./core/mcp.js";
22
22
  import { type McpHookIssue } from "./core/mcp-hook.js";
23
+ import { type TrifectaFinding } from "./core/lethal-trifecta.js";
24
+ import { type SkillResourceFinding } from "./core/skill-resources.js";
25
+ import { type SkillFenceFinding } from "./core/skill-missing-fence.js";
26
+ import { type PluginLayoutFinding } from "./core/plugin-dir-layout.js";
27
+ import { type DelegationTrifectaFinding } from "./core/delegation-trifecta.js";
28
+ import { type HookBlockFinding } from "./core/hook-block-ineffective.js";
29
+ import { type HookMatcherFinding } from "./core/hook-matcher.js";
23
30
  import { type PurityLevel, type EffectSurface } from "./core/effects.js";
24
31
  /** A named writing system. The label `unexpectedScript` reports + the config's expectation parse into this. */
25
32
  export type Script = "Latin" | "Cyrillic" | "Han" | "Japanese" | "Korean" | "Arabic" | "Hebrew" | "Greek" | "Devanagari" | "Thai";
@@ -43,6 +50,27 @@ export interface ScanSkill {
43
50
  * `audit` trigger tier / `measureTriggerRate`.
44
51
  */
45
52
  readonly descriptionScript: Script | null;
53
+ /**
54
+ * SKILL.md body references to a bundled file (`scripts/`/`references/`/`assets/`
55
+ * or a relative markdown link with an extension) that don't resolve on disk
56
+ * under the skill dir — the agent reads the instruction and gets nothing.
57
+ * Computed by `skillResourceIssues()` (one detector, no drift).
58
+ */
59
+ readonly resourceIssues: readonly SkillResourceFinding[];
60
+ /**
61
+ * Lethal-trifecta finding when a MODEL-INVOCABLE skill's declared `allowed-tools`
62
+ * hold all three legs (read-private + ingest-untrusted + exfiltrate), else null.
63
+ * A user-invoked skill is excluded (it can't be selected by attacker content).
64
+ * Computed by `lethalTrifectaIssues()` (one detector, no drift).
65
+ */
66
+ readonly trifecta: TrifectaFinding | null;
67
+ /**
68
+ * Set when the SKILL.md opens with frontmatter-looking keys (`name:`, …) but
69
+ * has NO opening `---` fence, so the whole file loads as body — no name, no
70
+ * description, no trigger (the skill is invisible). Computed by
71
+ * `skillMissingFence()` (one detector, no drift).
72
+ */
73
+ readonly fenceIssue: SkillFenceFinding | null;
46
74
  }
47
75
  export interface ScanAgent {
48
76
  readonly name: string;
@@ -68,6 +96,37 @@ export interface ScanAgent {
68
96
  * unknown-effect (MCP or unrecognized) tool names in the declared contract.
69
97
  */
70
98
  readonly effectBuckets: Pick<EffectSurface, "readOnly" | "sideEffecting" | "unknown">;
99
+ /**
100
+ * Lethal-trifecta finding when the subagent's declared tools hold all three legs
101
+ * (read-private + ingest-untrusted + exfiltrate), else null. An inherits-all
102
+ * agent (no `tools:` line) is the "advisory" case. Computed by
103
+ * `lethalTrifectaIssues()` (one detector, no drift).
104
+ */
105
+ readonly trifecta: TrifectaFinding | null;
106
+ }
107
+ /** A lethal-trifecta finding tagged with the surface (subagent/skill) that holds it. */
108
+ export interface ScanTrifectaFinding {
109
+ readonly path: string;
110
+ readonly kind: "subagent" | "skill";
111
+ readonly name: string;
112
+ readonly finding: TrifectaFinding;
113
+ }
114
+ /** A SKILL.md body resource reference that doesn't resolve, tagged with the skill path. */
115
+ export interface ScanSkillResourceFinding {
116
+ readonly path: string;
117
+ readonly name: string;
118
+ readonly finding: SkillResourceFinding;
119
+ }
120
+ /** A missing-frontmatter-fence finding tagged with the skill path. */
121
+ export interface ScanSkillFenceFinding {
122
+ readonly path: string;
123
+ readonly name: string;
124
+ readonly finding: SkillFenceFinding;
125
+ }
126
+ /** A delegation-trifecta finding tagged with the subagent path that holds it. */
127
+ export interface ScanDelegationFinding {
128
+ readonly path: string;
129
+ readonly finding: DelegationTrifectaFinding;
71
130
  }
72
131
  /** A skill/agent whose frontmatter is missing a required field (name / description). */
73
132
  export interface FrontmatterIssue {
@@ -162,6 +221,53 @@ export interface ScanReport {
162
221
  readonly mcpHookIssues: readonly McpHookIssue[];
163
222
  /** Pairs of model-invocable skills whose descriptions are near-identical (precision collision). */
164
223
  readonly descriptionOverlaps: readonly DescriptionOverlap[];
224
+ /**
225
+ * Lethal-trifecta findings across subagents + model-invocable skills — a unit
226
+ * holding all three legs (read-private + ingest-untrusted + exfiltrate). Each
227
+ * carries the surface path + kind for reporting/annotations. Shared by `scan`
228
+ * (the report) and the `lethal-trifecta` lint rule (one detector, no drift).
229
+ */
230
+ readonly trifectaFindings: readonly ScanTrifectaFinding[];
231
+ /**
232
+ * SKILL.md body references to a bundled resource that doesn't resolve on disk,
233
+ * across all skills — each carries the skill path for reporting/annotations.
234
+ * Shared by `scan` and the `skill-resource-resolves` lint rule (one detector, no
235
+ * drift).
236
+ */
237
+ readonly skillResourceIssues: readonly ScanSkillResourceFinding[];
238
+ /**
239
+ * Skills whose frontmatter-looking opening has NO `---` fence, so they load as
240
+ * pure body (invisible — no name/description/trigger). Shared by `scan` and the
241
+ * `skill-missing-fence` lint rule (one detector, no drift).
242
+ */
243
+ readonly skillFenceIssues: readonly ScanSkillFenceFinding[];
244
+ /**
245
+ * Functional surface dirs (skills/agents/commands) nested INSIDE the manifest
246
+ * dir (`.claude-plugin/`) where the harness can't see them. Shared by `scan`
247
+ * and the `plugin-dir-layout` lint rule (one detector, no drift).
248
+ */
249
+ readonly pluginLayoutIssues: readonly PluginLayoutFinding[];
250
+ /**
251
+ * Lethal trifectas that EMERGE across a delegation edge — a subagent whose
252
+ * effective (own ∪ delegated-to) capability holds all three legs though no
253
+ * single unit does. Shared by `scan` and the `delegation-trifecta` lint rule
254
+ * (one detector, no drift).
255
+ */
256
+ readonly delegationTrifecta: readonly ScanDelegationFinding[];
257
+ /**
258
+ * Hooks that LOOK like they block but silently don't — a block decision on a
259
+ * non-blocking event, or the legacy `decision` field on a permission-gated
260
+ * event (#19009, the #1 verified hook pain). Shared by `scan` and the
261
+ * `hook-block-ineffective` lint rule (one detector, no drift). Empty when the
262
+ * dialect doesn't declare its blocking-event semantics.
263
+ */
264
+ readonly hookBlockFindings: readonly HookBlockFinding[];
265
+ /**
266
+ * Hook `matcher` strings that silently never fire — a tool-name typo or a
267
+ * malformed/undeclared MCP form. Shared by `scan` and the `hook-matcher` lint
268
+ * rule (one detector, no drift).
269
+ */
270
+ readonly hookMatcherFindings: readonly HookMatcherFinding[];
165
271
  /** Skills/agents whose `---` block isn't valid YAML — informational (may still load via salvage). */
166
272
  readonly malformedFrontmatter: readonly FrontmatterParseIssue[];
167
273
  readonly warnings: readonly string[];