vigiles 10.0.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +9 -0
- package/README.md +121 -86
- package/action.yml +13 -2
- package/dist/adapter-conformance.js +6 -0
- package/dist/adapter-registry.d.ts +20 -0
- package/dist/adapter-registry.js +27 -0
- package/dist/adapters/claude-code/dialect.js +15 -0
- package/dist/adapters/claude-code/hook-protocol.js +4 -0
- package/dist/adapters/claude-code/runtime.js +12 -0
- package/dist/adapters/codex/eval.js +3 -0
- package/dist/adapters/codex/hook-protocol.d.ts +9 -1
- package/dist/adapters/codex/hook-protocol.js +10 -0
- package/dist/adapters/codex/runtime.js +10 -0
- package/dist/adapters/opencode/runtime.js +4 -0
- package/dist/audit-report.d.ts +1 -1
- package/dist/audit-report.template.html +1 -1
- package/dist/audit-score.d.ts +19 -12
- package/dist/audit-score.js +65 -11
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +1 -0
- package/dist/cli.js +460 -29
- package/dist/core/CLAUDE.md.spec.d.ts +3 -0
- package/dist/core/CLAUDE.md.spec.js +26 -0
- package/dist/core/delegation-trifecta.d.ts +64 -0
- package/dist/core/delegation-trifecta.js +124 -0
- package/dist/core/dialect.d.ts +18 -0
- package/dist/core/hook-block-ineffective.d.ts +62 -0
- package/dist/core/hook-block-ineffective.js +153 -0
- package/dist/core/hook-matcher.d.ts +66 -0
- package/dist/core/hook-matcher.js +182 -0
- package/dist/core/hook-normalize.d.ts +43 -0
- package/dist/core/hook-normalize.js +78 -0
- package/dist/core/hook-protocol.d.ts +15 -0
- package/dist/core/lethal-trifecta.d.ts +100 -0
- package/dist/core/lethal-trifecta.js +197 -0
- package/dist/core/plugin-dir-layout.d.ts +30 -0
- package/dist/core/plugin-dir-layout.js +73 -0
- package/dist/core/rule-meta.d.ts +82 -0
- package/dist/core/rule-meta.js +266 -0
- package/dist/core/runtime.d.ts +20 -0
- package/dist/core/skill-missing-fence.d.ts +47 -0
- package/dist/core/skill-missing-fence.js +119 -0
- package/dist/core/skill-resources.d.ts +27 -0
- package/dist/core/skill-resources.js +167 -0
- package/dist/core/types.d.ts +83 -0
- package/dist/core/validate.d.ts +1 -0
- package/dist/core/validate.js +26 -4
- package/dist/eval-cache.d.ts +6 -0
- package/dist/eval-cache.js +2 -0
- package/dist/eval-lock.d.ts +192 -0
- package/dist/eval-lock.js +286 -0
- package/dist/eval.d.ts +33 -20
- package/dist/eval.js +199 -51
- package/dist/leaderboard.d.ts +1 -0
- package/dist/leaderboard.js +42 -4
- package/dist/scan.d.ts +106 -0
- package/dist/scan.js +251 -45
- package/dist/setup-plan.d.ts +43 -3
- package/dist/setup-plan.js +78 -6
- package/hooks/eval-lock-nudge.sh +21 -0
- package/package.json +1 -1
- package/skills/test-harness/SKILL.md +27 -0
package/dist/eval.js
CHANGED
|
@@ -22,7 +22,6 @@ exports.aggregateUsage = aggregateUsage;
|
|
|
22
22
|
exports.isDatedModel = isDatedModel;
|
|
23
23
|
exports.modelTier = modelTier;
|
|
24
24
|
exports.belowModelFloor = belowModelFloor;
|
|
25
|
-
exports.harnessVersionKey = harnessVersionKey;
|
|
26
25
|
exports.ephemeralRunEnv = ephemeralRunEnv;
|
|
27
26
|
exports.seedEphemeralHome = seedEphemeralHome;
|
|
28
27
|
exports.isRateLimited = isRateLimited;
|
|
@@ -71,6 +70,7 @@ const runtime_js_1 = require("./adapters/claude-code/runtime.js");
|
|
|
71
70
|
const proofs_js_1 = require("./core/proofs.js");
|
|
72
71
|
const harness_test_js_1 = require("./harness-test.js");
|
|
73
72
|
const eval_cache_js_1 = require("./eval-cache.js");
|
|
73
|
+
const eval_lock_js_1 = require("./eval-lock.js");
|
|
74
74
|
const stats_js_1 = require("./stats.js");
|
|
75
75
|
const tool_intercept_js_1 = require("./tool-intercept.js");
|
|
76
76
|
const tool_stub_js_1 = require("./tool-stub.js");
|
|
@@ -640,32 +640,22 @@ function belowModelFloor(model, floor) {
|
|
|
640
640
|
const f = modelTier(floor);
|
|
641
641
|
return m !== null && f !== null && m < f;
|
|
642
642
|
}
|
|
643
|
-
/**
|
|
644
|
-
* Reduce a raw `--version` string to the **major.minor** cache-key token. We key
|
|
645
|
-
* the cache on major.minor, NOT the patch: a patch release rarely changes agent
|
|
646
|
-
* behaviour, so keying patches would churn the cache on every release for no
|
|
647
|
-
* signal; a minor/major bump is where the system prompt / tool defs actually move.
|
|
648
|
-
* (If a specific patch is known to matter, clear the cache or bump
|
|
649
|
-
* `CACHE_FORMAT_VERSION`.) Falls back to the trimmed raw string when no semver is
|
|
650
|
-
* found. Pure + tested.
|
|
651
|
-
*/
|
|
652
|
-
function harnessVersionKey(raw) {
|
|
653
|
-
const m = /(\d+)\.(\d+)\.\d+/.exec(raw);
|
|
654
|
-
return m ? `${m[1]}.${m[2]}` : raw.trim();
|
|
655
|
-
}
|
|
656
643
|
/* v8 ignore start -- spawns the real harness binary; memoized, cache-path only */
|
|
657
644
|
let cachedHarnessVersion;
|
|
658
645
|
/**
|
|
659
|
-
* The harness binary version (`claude --version`) reduced to
|
|
660
|
-
*
|
|
661
|
-
*
|
|
662
|
-
*
|
|
663
|
-
*
|
|
646
|
+
* The harness binary version (`claude --version`), reduced to its
|
|
647
|
+
* behaviorally-significant token by the runtime port's
|
|
648
|
+
* {@link HarnessRuntime.versionKey} — so a Claude Code **minor/major** upgrade
|
|
649
|
+
* (new system prompt / tool defs) invalidates a stale replay while patches don't
|
|
650
|
+
* churn it. The reduction is per-harness (CC → `major.minor`, Codex → `""`) and
|
|
651
|
+
* lives on the adapter, not here. Memoized (one spawn per process), resolved only
|
|
652
|
+
* on the cache path, "unknown" if the binary isn't found (then it doesn't
|
|
653
|
+
* partition the key).
|
|
664
654
|
*/
|
|
665
655
|
function harnessVersion() {
|
|
666
656
|
if (cachedHarnessVersion === undefined) {
|
|
667
657
|
try {
|
|
668
|
-
cachedHarnessVersion =
|
|
658
|
+
cachedHarnessVersion = runtime_js_1.claudeCodeRuntime.versionKey((0, node_child_process_1.execSync)(`${runtime_js_1.claudeCodeRuntime.agentBinary} --version`, {
|
|
669
659
|
encoding: "utf-8",
|
|
670
660
|
stdio: ["ignore", "pipe", "ignore"],
|
|
671
661
|
}));
|
|
@@ -983,15 +973,141 @@ function aggregateArms(armNames, results) {
|
|
|
983
973
|
}
|
|
984
974
|
return { arms, totalCostUsd };
|
|
985
975
|
}
|
|
976
|
+
function resolveLock(over) {
|
|
977
|
+
return {
|
|
978
|
+
mode: over?.mode ?? (0, eval_lock_js_1.lockModeFromEnv)(),
|
|
979
|
+
dir: over?.dir ?? (0, node_path_1.resolve)(process.cwd(), eval_lock_js_1.DEFAULT_LOCK_DIR),
|
|
980
|
+
evalApiVersion: over?.evalApiVersion ?? (0, eval_lock_js_1.evalApiVersionFromEnv)(),
|
|
981
|
+
};
|
|
982
|
+
}
|
|
983
|
+
/** Emit a vigiles message (GitHub annotation under Actions, else stderr/stdout). */
|
|
984
|
+
function emitLockMessage(msg, warn) {
|
|
985
|
+
if (process.env.GITHUB_ACTIONS)
|
|
986
|
+
console.log(`::${warn ? "warning" : "notice"}::${msg}`);
|
|
987
|
+
else if (warn)
|
|
988
|
+
console.warn(msg);
|
|
989
|
+
else
|
|
990
|
+
console.log(msg);
|
|
991
|
+
}
|
|
992
|
+
/**
|
|
993
|
+
* Run a named eval through the lock. `off` → just `produce()`. `check` → replay a
|
|
994
|
+
* matching committed lock (NO model call) or throw "stale". `update` → `produce()`,
|
|
995
|
+
* write the lock, print the human-facing delta. The model is driven ONLY on the
|
|
996
|
+
* run path — never on a clean `check`, which is what keeps the CI gate binary-free.
|
|
997
|
+
* A lock-on run with no `name` is a LOUD skip (the lock needs a name to key the file).
|
|
998
|
+
*/
|
|
999
|
+
async function withEvalLock(args, produce) {
|
|
1000
|
+
const { lock } = args;
|
|
1001
|
+
if (lock.mode === "off")
|
|
1002
|
+
return produce();
|
|
1003
|
+
if (!args.name) {
|
|
1004
|
+
// An unnamed eval can't be keyed to a lock. In `check` (the CI gate) running
|
|
1005
|
+
// it would call the model — violating the no-model-in-CI contract and failing
|
|
1006
|
+
// for missing auth — so FAIL LOUDLY instead of silently hitting the model.
|
|
1007
|
+
// `update` (local, model available) keeps producing: the eval just isn't gated.
|
|
1008
|
+
if (lock.mode === "check") {
|
|
1009
|
+
throw new Error(`vigiles eval --check: an unnamed eval cannot run in CI — it has no ` +
|
|
1010
|
+
`committed lock to replay, and running it would call the model. Add a ` +
|
|
1011
|
+
`\`name\` to the spec to gate it, or exclude it from the --check run.`);
|
|
1012
|
+
}
|
|
1013
|
+
emitLockMessage(`vigiles eval --${lock.mode}: skipped the lock for an unnamed eval — set ` +
|
|
1014
|
+
`\`name\` on the spec to enable the staleness gate for it.`, true);
|
|
1015
|
+
return produce();
|
|
1016
|
+
}
|
|
1017
|
+
if (!isDatedModel(args.model))
|
|
1018
|
+
warnFloatingModel(args.model);
|
|
1019
|
+
const inputsHash = (0, eval_lock_js_1.evalInputsHash)({
|
|
1020
|
+
model: args.model,
|
|
1021
|
+
evalApiVersion: lock.evalApiVersion,
|
|
1022
|
+
inputs: args.inputs,
|
|
1023
|
+
});
|
|
1024
|
+
const existing = (0, eval_lock_js_1.readLock)(lock.dir, args.name);
|
|
1025
|
+
const decision = (0, eval_lock_js_1.decideLock)(lock.mode, args.name, inputsHash, existing);
|
|
1026
|
+
if (decision.kind === "stale")
|
|
1027
|
+
throw new Error(decision.reason);
|
|
1028
|
+
if (decision.kind === "replay")
|
|
1029
|
+
return decision.report;
|
|
1030
|
+
const report = await produce();
|
|
1031
|
+
const builtLock = (0, eval_lock_js_1.buildLock)({
|
|
1032
|
+
name: args.name,
|
|
1033
|
+
inputsHash,
|
|
1034
|
+
model: args.model,
|
|
1035
|
+
harnessVersionKey: harnessVersion(),
|
|
1036
|
+
evalApiVersion: lock.evalApiVersion,
|
|
1037
|
+
builtAt: new Date().toISOString(),
|
|
1038
|
+
report,
|
|
1039
|
+
});
|
|
1040
|
+
const deltas = existing ? (0, eval_lock_js_1.diffReportNumbers)(existing.report, report) : [];
|
|
1041
|
+
(0, eval_lock_js_1.writeLock)(lock.dir, builtLock);
|
|
1042
|
+
emitLockMessage((0, eval_lock_js_1.formatLockUpdate)(args.name, deltas, existing === null), false);
|
|
1043
|
+
return report;
|
|
1044
|
+
}
|
|
1045
|
+
/**
|
|
1046
|
+
* Strip the machine-specific plugin-root prefix from a resolved value before it
|
|
1047
|
+
* enters the lock hash. `resolveHarness` expands `${PLUGIN_ROOT}` in a plugin's
|
|
1048
|
+
* hook commands to the checkout's ABSOLUTE path, so a lock recorded at
|
|
1049
|
+
* `/home/dev/...` would be falsely STALE when `--check` recomputes it at
|
|
1050
|
+
* `/home/runner/...` in CI (or any other machine). Normalizing the prefix back to
|
|
1051
|
+
* a token makes the hash location-independent. No-op when there's no plugin root.
|
|
1052
|
+
*/
|
|
1053
|
+
function stripPluginRoot(value, absRoot) {
|
|
1054
|
+
if (absRoot === "")
|
|
1055
|
+
return value;
|
|
1056
|
+
const json = JSON.stringify(value);
|
|
1057
|
+
// A hookless plugin resolves to `settings: undefined`, which `JSON.stringify`
|
|
1058
|
+
// returns as `undefined` (not a string) — pass it through rather than `.split`
|
|
1059
|
+
// a non-string (which would throw before the eval can run).
|
|
1060
|
+
if (json === undefined)
|
|
1061
|
+
return value;
|
|
1062
|
+
return JSON.parse(json.split(absRoot).join("${PLUGIN_ROOT}"));
|
|
1063
|
+
}
|
|
986
1064
|
/**
|
|
987
|
-
*
|
|
988
|
-
*
|
|
989
|
-
*
|
|
990
|
-
*
|
|
991
|
-
* caching, pooling, and aggregation are unit-testable without spawning a model
|
|
992
|
-
* (pass a fake returning canned stream-json). `runEval` is this with the real
|
|
993
|
-
* agent runner.
|
|
1065
|
+
* Resolve each arm's model-affecting inputs into a canonical object for the lock
|
|
1066
|
+
* hash — WITHOUT running the model (it reads files + hashes plugin dirs only). The
|
|
1067
|
+
* trial count is excluded (a sample-size knob, not a behavior input); `measure` is
|
|
1068
|
+
* excluded by design (the script re-asserts against the replayed report).
|
|
994
1069
|
*/
|
|
1070
|
+
function evalArmsInputs(spec, cfg) {
|
|
1071
|
+
const arms = {};
|
|
1072
|
+
for (const [name, arm] of Object.entries(spec.arms)) {
|
|
1073
|
+
const resolved = (0, plugin_loader_js_1.resolveHarness)({
|
|
1074
|
+
plugin: arm.plugin,
|
|
1075
|
+
settings: arm.settings,
|
|
1076
|
+
files: { ...spec.fixture, ...arm.files },
|
|
1077
|
+
});
|
|
1078
|
+
// The plugin root `resolveHarness` expanded into the resolved files/settings
|
|
1079
|
+
// is this checkout's absolute path — normalize it out so the hash is the same
|
|
1080
|
+
// on the dev's machine and in CI (else every plugin-with-root-hooks eval is
|
|
1081
|
+
// falsely stale across machines).
|
|
1082
|
+
const absRoot = arm.plugin ? (0, node_path_1.resolve)(process.cwd(), arm.plugin) : "";
|
|
1083
|
+
arms[name] = {
|
|
1084
|
+
model: arm.model ?? cfg.model,
|
|
1085
|
+
tools: [...cfg.tools].sort(),
|
|
1086
|
+
files: stripPluginRoot(resolved.files, absRoot),
|
|
1087
|
+
settings: stripPluginRoot(resolved.settings, absRoot),
|
|
1088
|
+
pluginDirHash: arm.pluginDir ? (0, eval_cache_js_1.hashDir)(arm.pluginDir) : undefined,
|
|
1089
|
+
interceptTools: arm.interceptTools
|
|
1090
|
+
? (0, tool_intercept_js_1.serializeIntercepts)(arm.interceptTools)
|
|
1091
|
+
: undefined,
|
|
1092
|
+
};
|
|
1093
|
+
}
|
|
1094
|
+
// Tool stubs (`spec.stubs`) are written onto PATH before each trial, so a
|
|
1095
|
+
// change to a canned CLI output IS a model-facing input change — fold a
|
|
1096
|
+
// canonical (name-sorted) view into the hash so `--check` catches it. Sorted
|
|
1097
|
+
// for a stable key regardless of declaration order; an empty list is the
|
|
1098
|
+
// byte-identical-to-before default. Each ToolStub is plain serializable data.
|
|
1099
|
+
const stubs = [...(spec.stubs ?? [])].sort((a, b) => a.name.localeCompare(b.name));
|
|
1100
|
+
// `ephemeralEnv` swaps the trial's environment (scrubbed env + throwaway HOME
|
|
1101
|
+
// vs the inherited process env), which can move tool/hook/agent behavior — a
|
|
1102
|
+
// model-facing input, so it belongs in the hash. Normalize to a bool so a
|
|
1103
|
+
// record under one mode can't be replayed under the other with the same hash.
|
|
1104
|
+
return {
|
|
1105
|
+
task: spec.task,
|
|
1106
|
+
arms,
|
|
1107
|
+
stubs,
|
|
1108
|
+
ephemeralEnv: spec.ephemeralEnv === true,
|
|
1109
|
+
};
|
|
1110
|
+
}
|
|
995
1111
|
async function runEvalWith(spec, runner) {
|
|
996
1112
|
const trials = spec.trials ?? 5;
|
|
997
1113
|
const spacing = (spec.spacingSec ?? 4) * 1000;
|
|
@@ -1024,9 +1140,16 @@ async function runEvalWith(spec, runner) {
|
|
|
1024
1140
|
await sleep(spacing);
|
|
1025
1141
|
return { armName: unit.armName, skipped: false, row, usage };
|
|
1026
1142
|
};
|
|
1027
|
-
const
|
|
1028
|
-
|
|
1029
|
-
|
|
1143
|
+
const lock = resolveLock(spec.lock);
|
|
1144
|
+
// Resolve the lock inputs only when the lock is active (off skips the
|
|
1145
|
+
// resolveHarness/hashDir work). `check` replays the committed report below
|
|
1146
|
+
// without ever entering the run pool — so no model is driven in CI.
|
|
1147
|
+
const inputs = lock.mode === "off" ? undefined : evalArmsInputs(spec, cfg);
|
|
1148
|
+
return withEvalLock({ name: spec.name, inputs, model: cfg.model, lock }, async () => {
|
|
1149
|
+
const results = await runPool(units, concurrency, worker);
|
|
1150
|
+
const { arms, totalCostUsd } = aggregateArms(Object.keys(spec.arms), results);
|
|
1151
|
+
return { name: spec.name ?? "eval", trials, arms, totalCostUsd, aborted };
|
|
1152
|
+
});
|
|
1030
1153
|
}
|
|
1031
1154
|
/** Render one metric: `name=mean±se pass^k=…` (se/pass^k shown when measured). */
|
|
1032
1155
|
function formatMetric(name, mean, stat) {
|
|
@@ -1065,6 +1188,7 @@ function formatEvalReport(report) {
|
|
|
1065
1188
|
exports.claudeEvalDriver = {
|
|
1066
1189
|
runner: spawnAgent,
|
|
1067
1190
|
parse: parseClaudeRun,
|
|
1191
|
+
harness: "claude-code",
|
|
1068
1192
|
};
|
|
1069
1193
|
/**
|
|
1070
1194
|
* Package loose `<skillsDir>/<name>/SKILL.md` skills into a throwaway plugin dir
|
|
@@ -1439,7 +1563,7 @@ function assertTriggerDiversity(spec) {
|
|
|
1439
1563
|
});
|
|
1440
1564
|
}
|
|
1441
1565
|
}
|
|
1442
|
-
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError) {
|
|
1566
|
+
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
|
|
1443
1567
|
// Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
|
|
1444
1568
|
assertTriggerDiversity(spec);
|
|
1445
1569
|
// Model floor (default Sonnet): trigger-rate under-measures selection on a
|
|
@@ -1469,26 +1593,50 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1469
1593
|
parse,
|
|
1470
1594
|
runError,
|
|
1471
1595
|
};
|
|
1596
|
+
const lock = resolveLock(spec.lock);
|
|
1472
1597
|
try {
|
|
1473
|
-
|
|
1474
|
-
|
|
1475
|
-
|
|
1476
|
-
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
|
|
1481
|
-
|
|
1482
|
-
|
|
1483
|
-
|
|
1484
|
-
|
|
1485
|
-
|
|
1486
|
-
|
|
1487
|
-
|
|
1488
|
-
|
|
1489
|
-
|
|
1490
|
-
|
|
1491
|
-
|
|
1598
|
+
// The lock hashes the skill UNDER TEST (its dir contents) + the prompts +
|
|
1599
|
+
// model + tools — the inputs that steer whether it fires. `--check` replays
|
|
1600
|
+
// the committed report with no model; `--update` records it. Resolved only
|
|
1601
|
+
// when the lock is active. `trials` is excluded (a sample-size knob).
|
|
1602
|
+
const triggerInputs = lock.mode === "off"
|
|
1603
|
+
? undefined
|
|
1604
|
+
: {
|
|
1605
|
+
pluginDirHash: (0, eval_cache_js_1.hashDir)(pluginDir),
|
|
1606
|
+
prompts: [...spec.prompts],
|
|
1607
|
+
irrelevantPrompts: spec.irrelevantPrompts
|
|
1608
|
+
? [...spec.irrelevantPrompts]
|
|
1609
|
+
: undefined,
|
|
1610
|
+
model: cfg.model,
|
|
1611
|
+
tools: [...cfg.tools].sort(),
|
|
1612
|
+
fixture: spec.fixture,
|
|
1613
|
+
competitors,
|
|
1614
|
+
// The harness the driver runs is a model-facing input — a Claude vs
|
|
1615
|
+
// Codex run can fire a skill differently — so a recorded report is
|
|
1616
|
+
// STALE if the eval is switched to another harness.
|
|
1617
|
+
harness,
|
|
1618
|
+
};
|
|
1619
|
+
return await withEvalLock({ name: spec.name, inputs: triggerInputs, model: cfg.model, lock }, async () => {
|
|
1620
|
+
const relevant = await runTriggerSet(spec.prompts, cfg, runner);
|
|
1621
|
+
const base = {
|
|
1622
|
+
rate: relevant.n > 0 ? relevant.fired / relevant.n : 0,
|
|
1623
|
+
n: relevant.n,
|
|
1624
|
+
perPrompt: relevant.perPrompt,
|
|
1625
|
+
competitors,
|
|
1626
|
+
errored: positiveOrUndefined(relevant.errored),
|
|
1627
|
+
};
|
|
1628
|
+
if ((spec.irrelevantPrompts?.length ?? 0) === 0)
|
|
1629
|
+
return base;
|
|
1630
|
+
const irrelevant = await runTriggerSet(spec.irrelevantPrompts ?? [], cfg, runner);
|
|
1631
|
+
const fires = relevant.fired + irrelevant.fired;
|
|
1632
|
+
return {
|
|
1633
|
+
...base,
|
|
1634
|
+
errored: positiveOrUndefined(relevant.errored + irrelevant.errored),
|
|
1635
|
+
falsePositiveRate: irrelevant.n > 0 ? irrelevant.fired / irrelevant.n : 0,
|
|
1636
|
+
precision: fires > 0 ? relevant.fired / fires : undefined,
|
|
1637
|
+
perIrrelevant: irrelevant.perPrompt,
|
|
1638
|
+
};
|
|
1639
|
+
});
|
|
1492
1640
|
}
|
|
1493
1641
|
finally {
|
|
1494
1642
|
// Remove the throwaway plugin dir we built from a loose `skillsDir`.
|
|
@@ -1506,7 +1654,7 @@ async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runE
|
|
|
1506
1654
|
*/
|
|
1507
1655
|
async function measureTriggerRate(spec, opts = {}) {
|
|
1508
1656
|
const d = opts.evalDriver ?? exports.claudeEvalDriver;
|
|
1509
|
-
return measureTriggerRateWith(spec, d.runner, d.parse, d.runError);
|
|
1657
|
+
return measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
|
|
1510
1658
|
}
|
|
1511
1659
|
/* v8 ignore stop */
|
|
1512
1660
|
/** Format a trigger-rate report: overall %, then each prompt's rate. */
|
package/dist/leaderboard.d.ts
CHANGED
|
@@ -26,6 +26,7 @@ export declare const W_NO_DESCRIPTION = 10;
|
|
|
26
26
|
export declare const W_DANGLING_REF = 8;
|
|
27
27
|
export declare const W_OVERLAP = 8;
|
|
28
28
|
export declare const W_NO_CONTRACT = 5;
|
|
29
|
+
export declare const W_TRIFECTA = 20;
|
|
29
30
|
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
30
31
|
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
31
32
|
/** One deduction: a count, its per-item weight, and the label if non-zero. */
|
package/dist/leaderboard.js
CHANGED
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
* model and stack on top later; this part runs anywhere in CI for free.
|
|
13
13
|
*/
|
|
14
14
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
15
|
-
exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
15
|
+
exports.W_TRIFECTA = exports.W_NO_CONTRACT = exports.W_OVERLAP = exports.W_DANGLING_REF = exports.W_NO_DESCRIPTION = exports.W_MISSING_HOOK = void 0;
|
|
16
16
|
exports.gradeFor = gradeFor;
|
|
17
17
|
exports.reportDeductions = reportDeductions;
|
|
18
18
|
exports.isEmptyMachine = isEmptyMachine;
|
|
@@ -49,6 +49,7 @@ exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't t
|
|
|
49
49
|
exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
50
50
|
exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
|
|
51
51
|
exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
|
|
52
|
+
exports.W_TRIFECTA = 20; // a HARD lethal-trifecta contract (all three legs, explicit) → a declared prompt-injection exfil path
|
|
52
53
|
// Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
|
|
53
54
|
// - untested surfaces — a hardening gap, not breakage.
|
|
54
55
|
// - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
|
|
@@ -77,7 +78,17 @@ function reportDeductions(r) {
|
|
|
77
78
|
const deadTools = r.agents.reduce((n, a) => n + a.toolIssues.length, 0);
|
|
78
79
|
const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
|
|
79
80
|
const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
|
|
81
|
+
// HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
|
|
82
|
+
// legs (a declared exfil path). Advisory (inherits-all) trifecta findings are
|
|
83
|
+
// surfaced but NEVER graded (aligned with the inherits-all stance), so they're
|
|
84
|
+
// excluded here.
|
|
85
|
+
const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
|
|
80
86
|
return [
|
|
87
|
+
{
|
|
88
|
+
n: hardTrifecta,
|
|
89
|
+
weight: exports.W_TRIFECTA,
|
|
90
|
+
label: "unit(s) holding all three lethal-trifecta legs (prompt-injection exfil path)",
|
|
91
|
+
},
|
|
81
92
|
{
|
|
82
93
|
n: missingHooks,
|
|
83
94
|
weight: exports.W_MISSING_HOOK,
|
|
@@ -146,6 +157,33 @@ function reportDeductions(r) {
|
|
|
146
157
|
weight: exports.W_DANGLING_REF,
|
|
147
158
|
label: "mcp_tool hook(s) incomplete / targeting an undeclared server",
|
|
148
159
|
},
|
|
160
|
+
{
|
|
161
|
+
n: r.skillResourceIssues.length,
|
|
162
|
+
weight: exports.W_DANGLING_REF,
|
|
163
|
+
label: "skill bundled-resource ref(s) that don't resolve on disk",
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
n: r.skillFenceIssues.length,
|
|
167
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
168
|
+
label: "invisible skill(s) (frontmatter with no opening `---` fence)",
|
|
169
|
+
},
|
|
170
|
+
{
|
|
171
|
+
n: r.pluginLayoutIssues.length,
|
|
172
|
+
weight: exports.W_NO_DESCRIPTION,
|
|
173
|
+
label: "functional dir(s) misplaced inside `.claude-plugin/` (invisible)",
|
|
174
|
+
},
|
|
175
|
+
{
|
|
176
|
+
n: r.hookBlockFindings.length,
|
|
177
|
+
weight: exports.W_MISSING_HOOK,
|
|
178
|
+
label: "hook(s) that look like they block but silently don't",
|
|
179
|
+
},
|
|
180
|
+
{
|
|
181
|
+
n: r.hookMatcherFindings.length,
|
|
182
|
+
weight: exports.W_MISSING_HOOK,
|
|
183
|
+
label: "hook matcher(s) that never fire (typo / wrong MCP form)",
|
|
184
|
+
},
|
|
185
|
+
// NB: delegationTrifecta (like the advisory per-unit/inherits-all trifecta) is a
|
|
186
|
+
// ⚠ RISK, surfaced but NOT graded — only the HARD per-unit trifecta above scores.
|
|
149
187
|
// NB: untested surfaces are NOT a penalty — an untested surface is a hardening
|
|
150
188
|
// gap, not breakage, so it never drags the health score (it's appended as an
|
|
151
189
|
// advisory note below). The score ranks what's BROKEN.
|
|
@@ -239,11 +277,11 @@ function formatLeaderboard(scores) {
|
|
|
239
277
|
const issue = s.issues.length > 0 ? ` — ${s.issues.join("; ")}` : "";
|
|
240
278
|
out.push(` ${rank} ${score} ${s.grade} ${s.name}${issue}`);
|
|
241
279
|
});
|
|
242
|
-
out.push("", "Structural health only (no model). Weights: missing hook -15, no-description
|
|
280
|
+
out.push("", "Structural health only (no model). Weights: lethal-trifecta unit -20, missing", "hook -15, no-description skill -10, broken intra-plugin ref -8, dead tool/MCP", "ref -8. Inherit-all subagents and untested surfaces are advisory — shown, not", "scored.");
|
|
243
281
|
return out.join("\n");
|
|
244
282
|
}
|
|
245
|
-
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model):
|
|
246
|
-
"no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
|
|
283
|
+
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): lethal-trifecta unit −20, " +
|
|
284
|
+
"missing hook −15, no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
|
|
247
285
|
"Inherit-all subagents and untested surfaces are advisory (shown, not " +
|
|
248
286
|
"scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
|
|
249
287
|
/**
|
package/dist/scan.d.ts
CHANGED
|
@@ -20,6 +20,13 @@ import { type DescriptionOverlap } from "./core/description-overlap.js";
|
|
|
20
20
|
import { type McpToolIssue } from "./core/mcp-tool.js";
|
|
21
21
|
import { type McpContractToolError } from "./core/mcp.js";
|
|
22
22
|
import { type McpHookIssue } from "./core/mcp-hook.js";
|
|
23
|
+
import { type TrifectaFinding } from "./core/lethal-trifecta.js";
|
|
24
|
+
import { type SkillResourceFinding } from "./core/skill-resources.js";
|
|
25
|
+
import { type SkillFenceFinding } from "./core/skill-missing-fence.js";
|
|
26
|
+
import { type PluginLayoutFinding } from "./core/plugin-dir-layout.js";
|
|
27
|
+
import { type DelegationTrifectaFinding } from "./core/delegation-trifecta.js";
|
|
28
|
+
import { type HookBlockFinding } from "./core/hook-block-ineffective.js";
|
|
29
|
+
import { type HookMatcherFinding } from "./core/hook-matcher.js";
|
|
23
30
|
import { type PurityLevel, type EffectSurface } from "./core/effects.js";
|
|
24
31
|
/** A named writing system. The label `unexpectedScript` reports + the config's expectation parse into this. */
|
|
25
32
|
export type Script = "Latin" | "Cyrillic" | "Han" | "Japanese" | "Korean" | "Arabic" | "Hebrew" | "Greek" | "Devanagari" | "Thai";
|
|
@@ -43,6 +50,27 @@ export interface ScanSkill {
|
|
|
43
50
|
* `audit` trigger tier / `measureTriggerRate`.
|
|
44
51
|
*/
|
|
45
52
|
readonly descriptionScript: Script | null;
|
|
53
|
+
/**
|
|
54
|
+
* SKILL.md body references to a bundled file (`scripts/`/`references/`/`assets/`
|
|
55
|
+
* or a relative markdown link with an extension) that don't resolve on disk
|
|
56
|
+
* under the skill dir — the agent reads the instruction and gets nothing.
|
|
57
|
+
* Computed by `skillResourceIssues()` (one detector, no drift).
|
|
58
|
+
*/
|
|
59
|
+
readonly resourceIssues: readonly SkillResourceFinding[];
|
|
60
|
+
/**
|
|
61
|
+
* Lethal-trifecta finding when a MODEL-INVOCABLE skill's declared `allowed-tools`
|
|
62
|
+
* hold all three legs (read-private + ingest-untrusted + exfiltrate), else null.
|
|
63
|
+
* A user-invoked skill is excluded (it can't be selected by attacker content).
|
|
64
|
+
* Computed by `lethalTrifectaIssues()` (one detector, no drift).
|
|
65
|
+
*/
|
|
66
|
+
readonly trifecta: TrifectaFinding | null;
|
|
67
|
+
/**
|
|
68
|
+
* Set when the SKILL.md opens with frontmatter-looking keys (`name:`, …) but
|
|
69
|
+
* has NO opening `---` fence, so the whole file loads as body — no name, no
|
|
70
|
+
* description, no trigger (the skill is invisible). Computed by
|
|
71
|
+
* `skillMissingFence()` (one detector, no drift).
|
|
72
|
+
*/
|
|
73
|
+
readonly fenceIssue: SkillFenceFinding | null;
|
|
46
74
|
}
|
|
47
75
|
export interface ScanAgent {
|
|
48
76
|
readonly name: string;
|
|
@@ -68,6 +96,37 @@ export interface ScanAgent {
|
|
|
68
96
|
* unknown-effect (MCP or unrecognized) tool names in the declared contract.
|
|
69
97
|
*/
|
|
70
98
|
readonly effectBuckets: Pick<EffectSurface, "readOnly" | "sideEffecting" | "unknown">;
|
|
99
|
+
/**
|
|
100
|
+
* Lethal-trifecta finding when the subagent's declared tools hold all three legs
|
|
101
|
+
* (read-private + ingest-untrusted + exfiltrate), else null. An inherits-all
|
|
102
|
+
* agent (no `tools:` line) is the "advisory" case. Computed by
|
|
103
|
+
* `lethalTrifectaIssues()` (one detector, no drift).
|
|
104
|
+
*/
|
|
105
|
+
readonly trifecta: TrifectaFinding | null;
|
|
106
|
+
}
|
|
107
|
+
/** A lethal-trifecta finding tagged with the surface (subagent/skill) that holds it. */
|
|
108
|
+
export interface ScanTrifectaFinding {
|
|
109
|
+
readonly path: string;
|
|
110
|
+
readonly kind: "subagent" | "skill";
|
|
111
|
+
readonly name: string;
|
|
112
|
+
readonly finding: TrifectaFinding;
|
|
113
|
+
}
|
|
114
|
+
/** A SKILL.md body resource reference that doesn't resolve, tagged with the skill path. */
|
|
115
|
+
export interface ScanSkillResourceFinding {
|
|
116
|
+
readonly path: string;
|
|
117
|
+
readonly name: string;
|
|
118
|
+
readonly finding: SkillResourceFinding;
|
|
119
|
+
}
|
|
120
|
+
/** A missing-frontmatter-fence finding tagged with the skill path. */
|
|
121
|
+
export interface ScanSkillFenceFinding {
|
|
122
|
+
readonly path: string;
|
|
123
|
+
readonly name: string;
|
|
124
|
+
readonly finding: SkillFenceFinding;
|
|
125
|
+
}
|
|
126
|
+
/** A delegation-trifecta finding tagged with the subagent path that holds it. */
|
|
127
|
+
export interface ScanDelegationFinding {
|
|
128
|
+
readonly path: string;
|
|
129
|
+
readonly finding: DelegationTrifectaFinding;
|
|
71
130
|
}
|
|
72
131
|
/** A skill/agent whose frontmatter is missing a required field (name / description). */
|
|
73
132
|
export interface FrontmatterIssue {
|
|
@@ -162,6 +221,53 @@ export interface ScanReport {
|
|
|
162
221
|
readonly mcpHookIssues: readonly McpHookIssue[];
|
|
163
222
|
/** Pairs of model-invocable skills whose descriptions are near-identical (precision collision). */
|
|
164
223
|
readonly descriptionOverlaps: readonly DescriptionOverlap[];
|
|
224
|
+
/**
|
|
225
|
+
* Lethal-trifecta findings across subagents + model-invocable skills — a unit
|
|
226
|
+
* holding all three legs (read-private + ingest-untrusted + exfiltrate). Each
|
|
227
|
+
* carries the surface path + kind for reporting/annotations. Shared by `scan`
|
|
228
|
+
* (the report) and the `lethal-trifecta` lint rule (one detector, no drift).
|
|
229
|
+
*/
|
|
230
|
+
readonly trifectaFindings: readonly ScanTrifectaFinding[];
|
|
231
|
+
/**
|
|
232
|
+
* SKILL.md body references to a bundled resource that doesn't resolve on disk,
|
|
233
|
+
* across all skills — each carries the skill path for reporting/annotations.
|
|
234
|
+
* Shared by `scan` and the `skill-resource-resolves` lint rule (one detector, no
|
|
235
|
+
* drift).
|
|
236
|
+
*/
|
|
237
|
+
readonly skillResourceIssues: readonly ScanSkillResourceFinding[];
|
|
238
|
+
/**
|
|
239
|
+
* Skills whose frontmatter-looking opening has NO `---` fence, so they load as
|
|
240
|
+
* pure body (invisible — no name/description/trigger). Shared by `scan` and the
|
|
241
|
+
* `skill-missing-fence` lint rule (one detector, no drift).
|
|
242
|
+
*/
|
|
243
|
+
readonly skillFenceIssues: readonly ScanSkillFenceFinding[];
|
|
244
|
+
/**
|
|
245
|
+
* Functional surface dirs (skills/agents/commands) nested INSIDE the manifest
|
|
246
|
+
* dir (`.claude-plugin/`) where the harness can't see them. Shared by `scan`
|
|
247
|
+
* and the `plugin-dir-layout` lint rule (one detector, no drift).
|
|
248
|
+
*/
|
|
249
|
+
readonly pluginLayoutIssues: readonly PluginLayoutFinding[];
|
|
250
|
+
/**
|
|
251
|
+
* Lethal trifectas that EMERGE across a delegation edge — a subagent whose
|
|
252
|
+
* effective (own ∪ delegated-to) capability holds all three legs though no
|
|
253
|
+
* single unit does. Shared by `scan` and the `delegation-trifecta` lint rule
|
|
254
|
+
* (one detector, no drift).
|
|
255
|
+
*/
|
|
256
|
+
readonly delegationTrifecta: readonly ScanDelegationFinding[];
|
|
257
|
+
/**
|
|
258
|
+
* Hooks that LOOK like they block but silently don't — a block decision on a
|
|
259
|
+
* non-blocking event, or the legacy `decision` field on a permission-gated
|
|
260
|
+
* event (#19009, the #1 verified hook pain). Shared by `scan` and the
|
|
261
|
+
* `hook-block-ineffective` lint rule (one detector, no drift). Empty when the
|
|
262
|
+
* dialect doesn't declare its blocking-event semantics.
|
|
263
|
+
*/
|
|
264
|
+
readonly hookBlockFindings: readonly HookBlockFinding[];
|
|
265
|
+
/**
|
|
266
|
+
* Hook `matcher` strings that silently never fire — a tool-name typo or a
|
|
267
|
+
* malformed/undeclared MCP form. Shared by `scan` and the `hook-matcher` lint
|
|
268
|
+
* rule (one detector, no drift).
|
|
269
|
+
*/
|
|
270
|
+
readonly hookMatcherFindings: readonly HookMatcherFinding[];
|
|
165
271
|
/** Skills/agents whose `---` block isn't valid YAML — informational (may still load via salvage). */
|
|
166
272
|
readonly malformedFrontmatter: readonly FrontmatterParseIssue[];
|
|
167
273
|
readonly warnings: readonly string[];
|