vigiles 31.0.0 → 32.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/eval.js CHANGED
@@ -87,6 +87,7 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
87
87
  const tool_intercept_js_1 = require("./tool-intercept.js");
88
88
  const tool_stub_js_1 = require("./tool-stub.js");
89
89
  const tmp_root_js_1 = require("./core/tmp-root.js");
90
+ const edit_distance_js_1 = require("./core/edit-distance.js");
90
91
  function writeFiles(cwd, files) {
91
92
  for (const [p, content] of Object.entries(files)) {
92
93
  const full = (0, node_path_1.resolve)(cwd, p);
@@ -315,8 +316,12 @@ async function runEval(spec) {
315
316
  * `runner` so the orchestration is unit-testable without a model.
316
317
  */
317
318
  async function measureWith(spec, runner) {
318
- if (spec.stubSkillBodies && !spec.pluginDir)
319
- throw new Error("measure: `stubSkillBodies` requires `pluginDir`.");
319
+ assertKnownKeys(spec, MEASURE_SPEC_KEYS, {
320
+ caller: "measure",
321
+ type: "MeasureSpec",
322
+ });
323
+ if (spec.stubSkillBodies && !spec.pluginDir && !spec.skillsDir)
324
+ throw new Error("measure: `stubSkillBodies` requires `pluginDir` or `skillsDir`.");
320
325
  // stubSkillBodies replaces each skill BODY with a no-op (the run stops at
321
326
  // selection), so there is no output to grade — a `judged` check would score an
322
327
  // empty body and mislead. The docs warn against this pairing; enforce it.
@@ -325,51 +330,49 @@ async function measureWith(spec, runner) {
325
330
  throw new Error("measure: `stubSkillBodies` is for firing/`skill()` checks only — it stubs " +
326
331
  "the skill body, so there's no output for a `judged` check to grade. Drop " +
327
332
  "`stubSkillBodies`, or remove the `judged` check.");
328
- const stubbed = spec.stubSkillBodies
329
- ? stubbedPluginDir(spec.pluginDir)
330
- : undefined;
331
- const pluginDir = stubbed ?? spec.pluginDir;
332
- try {
333
- const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
334
- const report = await runEvalWith({
335
- fixture: spec.fixture,
336
- arms: {
337
- run: {
338
- settings: spec.settings,
339
- plugin: spec.plugin,
340
- pluginDir,
341
- interceptTools: spec.interceptTools,
342
- },
343
- },
344
- task: spec.task,
345
- trials: spec.trials ?? 5,
346
- model: spec.model ?? "sonnet",
347
- effort: spec.effort,
348
- allowedTools: spec.allowedTools,
349
- timeoutMs: spec.timeoutMs,
350
- spacingSec: spec.spacingSec,
351
- measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
352
- }, runner);
353
- const arm = report.arms.run;
354
- return {
355
- n: arm?.runs ?? 0,
356
- perCheck: keyed.map(([k, c]) => {
357
- const s = arm?.stats[k];
358
- return {
359
- check: c.toJSON(),
360
- rate: s?.mean ?? 0,
361
- se: s?.se ?? 0,
362
- passK: s?.passK ?? 0,
363
- n: s?.n ?? 0,
364
- };
365
- }),
366
- usage: arm?.usage ?? aggregateUsage([]),
367
- };
368
- }
369
- finally {
370
- if (stubbed)
371
- (0, node_fs_1.rmSync)(stubbed, { recursive: true, force: true });
372
- }
333
+ const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
334
+ // The install source (plugin / pluginDir / skillsDir, stubbed or not) is
335
+ // resolved by runEvalWith, once per arm — measure is one arm, so it just
336
+ // forwards the fields and reads the arm's report back.
337
+ const arm = {
338
+ settings: spec.settings,
339
+ plugin: spec.plugin,
340
+ pluginDir: spec.pluginDir,
341
+ skillsDir: spec.skillsDir,
342
+ interceptTools: spec.interceptTools,
343
+ };
344
+ const report = await runEvalWith({
345
+ fixture: spec.fixture,
346
+ arms: { run: arm },
347
+ stubSkillBodies: spec.stubSkillBodies,
348
+ task: spec.task,
349
+ trials: spec.trials ?? 5,
350
+ model: spec.model ?? "sonnet",
351
+ effort: spec.effort,
352
+ allowedTools: spec.allowedTools,
353
+ timeoutMs: spec.timeoutMs,
354
+ spacingSec: spec.spacingSec,
355
+ measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
356
+ }, runner);
357
+ return checkReportOf(report.arms.run, keyed, installNamespace(arm));
358
+ }
359
+ /** Read one arm's {@link ArmReport} back into a {@link CheckReport}. */
360
+ function checkReportOf(arm, keyed, namespace) {
361
+ return {
362
+ n: arm?.runs ?? 0,
363
+ perCheck: keyed.map(([k, c]) => {
364
+ const s = arm?.stats[k];
365
+ return {
366
+ check: c.toJSON(),
367
+ rate: s?.mean ?? 0,
368
+ se: s?.se ?? 0,
369
+ passK: s?.passK ?? 0,
370
+ n: s?.n ?? 0,
371
+ };
372
+ }),
373
+ usage: arm?.usage ?? aggregateUsage([]),
374
+ namespace,
375
+ };
373
376
  }
374
377
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
375
378
  /** Score a check vocabulary across trials against the real `claude` CLI. */
@@ -380,67 +383,36 @@ async function measure(spec) {
380
383
  }
381
384
  /** Score checks across arms (injectable runner). Reuses `runEvalWith`. */
382
385
  async function measureArmsWith(spec, runner) {
386
+ assertKnownKeys(spec, ARMS_MEASURE_SPEC_KEYS, {
387
+ caller: "measureArms",
388
+ type: "ArmsMeasureSpec",
389
+ });
390
+ for (const [name, arm] of Object.entries(spec.arms))
391
+ assertKnownKeys(arm, EVAL_ARM_KEYS, {
392
+ caller: "measureArms",
393
+ type: "EvalArm",
394
+ arm: name,
395
+ });
383
396
  const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
384
- const { arms: runArms, temps } = spec.stubSkillBodies
385
- ? stubArmPluginDirs(spec.arms)
386
- : { arms: spec.arms, temps: [] };
387
- try {
388
- const report = await runEvalWith({
389
- fixture: spec.fixture,
390
- arms: runArms,
391
- task: spec.task,
392
- trials: spec.trials ?? 5,
393
- model: spec.model ?? "sonnet",
394
- effort: spec.effort,
395
- allowedTools: spec.allowedTools,
396
- timeoutMs: spec.timeoutMs,
397
- spacingSec: spec.spacingSec,
398
- measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
399
- }, runner);
400
- const arms = {};
401
- for (const [armName, arm] of Object.entries(report.arms)) {
402
- arms[armName] = {
403
- n: arm.runs,
404
- perCheck: keyed.map(([k, c]) => {
405
- const s = arm.stats[k];
406
- return {
407
- check: c.toJSON(),
408
- rate: s?.mean ?? 0,
409
- se: s?.se ?? 0,
410
- passK: s?.passK ?? 0,
411
- n: s?.n ?? 0,
412
- };
413
- }),
414
- usage: arm.usage,
415
- };
416
- }
417
- return { arms };
418
- }
419
- finally {
420
- for (const t of temps)
421
- (0, node_fs_1.rmSync)(t, { recursive: true, force: true });
422
- }
423
- }
424
- /**
425
- * Repackage every arm that sets a `pluginDir` with its skill bodies stubbed
426
- * (frontmatter kept), for an A/B firing comparison. Returns the rewritten arms
427
- * plus the throwaway dirs the caller must remove. Arms without a `pluginDir` pass
428
- * through unchanged. See {@link stubbedPluginDir}.
429
- */
430
- function stubArmPluginDirs(arms) {
431
- const out = {};
432
- const temps = [];
433
- for (const [name, arm] of Object.entries(arms)) {
434
- if (arm.pluginDir) {
435
- const stubbed = stubbedPluginDir(arm.pluginDir);
436
- temps.push(stubbed);
437
- out[name] = { ...arm, pluginDir: stubbed };
438
- }
439
- else {
440
- out[name] = arm;
441
- }
397
+ // Per-arm install sources (and the stub) are resolved by runEvalWith.
398
+ const report = await runEvalWith({
399
+ fixture: spec.fixture,
400
+ arms: spec.arms,
401
+ stubSkillBodies: spec.stubSkillBodies,
402
+ task: spec.task,
403
+ trials: spec.trials ?? 5,
404
+ model: spec.model ?? "sonnet",
405
+ effort: spec.effort,
406
+ allowedTools: spec.allowedTools,
407
+ timeoutMs: spec.timeoutMs,
408
+ spacingSec: spec.spacingSec,
409
+ measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
410
+ }, runner);
411
+ const arms = {};
412
+ for (const [armName, arm] of Object.entries(report.arms)) {
413
+ arms[armName] = checkReportOf(arm, keyed, installNamespace(spec.arms[armName] ?? {}));
442
414
  }
443
- return { arms: out, temps };
415
+ return { arms };
444
416
  }
445
417
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
446
418
  /** Score checks across arms against the real `claude` CLI. */
@@ -482,6 +454,28 @@ function formatCheckReport(report) {
482
454
  lines.push(` ${(c.rate * 100).toFixed(0)}% ± ${(c.se * 100).toFixed(0)}% ${checkLabel(c.check)}` +
483
455
  ` (pass^k ${String(c.passK)})`);
484
456
  }
457
+ // The `skill()` twin of the trigger-rate TOTAL-zero note (see
458
+ // formatTriggerRateReport): every skill() check at 0% over real runs is far
459
+ // more often a wiring mistake — a bare id where the namespaced one is matched,
460
+ // `pluginDir` handed a loose dir, an empty cwd — than a finding, and it reads
461
+ // exactly like a finding. Only when EVERY skill check is at zero: a partial rate
462
+ // is a real measurement, and a note that hedges on good data gets ignored.
463
+ const skillChecks = report.perCheck.filter((c) => c.check.kind === "skill");
464
+ if (report.n > 0 &&
465
+ skillChecks.length > 0 &&
466
+ skillChecks.every((c) => c.rate === 0))
467
+ lines.push("⚠ nothing resolved for ANY skill() check. That is usually SETUP, not the skill — check, in order:\n" +
468
+ (report.namespace !== undefined
469
+ ? " 1. the id in `skill()` — your skills installed under " +
470
+ `\`${report.namespace}\`, so \`skill()\` matches ` +
471
+ `\`${report.namespace}:<skill>\`; a bare name silently never matches;\n`
472
+ : " 1. the id in `skill()` — it matches the NAMESPACED id " +
473
+ "(`<plugin>:<skill>`); a bare name silently never matches;\n") +
474
+ " 2. the install field — a loose `.claude/skills` dir needs `skillsDir`, " +
475
+ "not `pluginDir` (which wants a full plugin manifest);\n" +
476
+ " 3. the `fixture` — a run starts in an EMPTY cwd, so a prompt about a " +
477
+ "file that does not exist is one the model is right to decline.\n" +
478
+ " Rule out all three before recording this as a fact about the skill.");
485
479
  return lines.join("\n");
486
480
  }
487
481
  /**
@@ -1317,10 +1311,175 @@ function evalArmsInputs(spec, cfg) {
1317
1311
  ephemeralEnv: spec.ephemeralEnv === true,
1318
1312
  };
1319
1313
  }
1320
- async function runEvalWith(spec, runner) {
1314
+ // ---------------------------------------------------------------------------
1315
+ // The spec boundary — refuse a field the spec does not declare.
1316
+ //
1317
+ // A spec usually arrives from a plain-JS `*.eval.mjs` file, where nothing
1318
+ // type-checks the object literal, so a key that is misspelled (`skillDir`) or
1319
+ // belongs to a sibling runner (`prompts` on measure) used to be dropped without a
1320
+ // word — and the run then reported a confident number about a setup the author
1321
+ // never asked for (issue #307). The precedent is `driverMisplaced` in
1322
+ // eval-entry.ts: a field that would silently do nothing is REFUSED, not ignored.
1323
+ //
1324
+ // Each key table is `satisfies Record<keyof Spec, true>`, so the compiler fails
1325
+ // when a spec gains or loses a field the table does not: the list cannot drift.
1326
+ // ---------------------------------------------------------------------------
1327
+ const EVAL_ARM_KEYS = {
1328
+ files: true,
1329
+ settings: true,
1330
+ plugin: true,
1331
+ pluginDir: true,
1332
+ skillsDir: true,
1333
+ interceptTools: true,
1334
+ model: true,
1335
+ effort: true,
1336
+ };
1337
+ const EVAL_SPEC_KEYS = {
1338
+ name: true,
1339
+ fixture: true,
1340
+ arms: true,
1341
+ stubSkillBodies: true,
1342
+ task: true,
1343
+ measure: true,
1344
+ trials: true,
1345
+ model: true,
1346
+ effort: true,
1347
+ allowedTools: true,
1348
+ timeoutMs: true,
1349
+ spacingSec: true,
1350
+ cache: true,
1351
+ cacheDir: true,
1352
+ concurrency: true,
1353
+ maxCostUsd: true,
1354
+ rateLimitRetries: true,
1355
+ retryBackoffMs: true,
1356
+ ephemeralEnv: true,
1357
+ stubs: true,
1358
+ lock: true,
1359
+ };
1360
+ const MEASURE_SPEC_KEYS = {
1361
+ fixture: true,
1362
+ settings: true,
1363
+ plugin: true,
1364
+ pluginDir: true,
1365
+ skillsDir: true,
1366
+ stubSkillBodies: true,
1367
+ interceptTools: true,
1368
+ task: true,
1369
+ checks: true,
1370
+ trials: true,
1371
+ model: true,
1372
+ effort: true,
1373
+ allowedTools: true,
1374
+ timeoutMs: true,
1375
+ spacingSec: true,
1376
+ };
1377
+ const ARMS_MEASURE_SPEC_KEYS = {
1378
+ fixture: true,
1379
+ arms: true,
1380
+ task: true,
1381
+ checks: true,
1382
+ stubSkillBodies: true,
1383
+ trials: true,
1384
+ model: true,
1385
+ allowedTools: true,
1386
+ timeoutMs: true,
1387
+ effort: true,
1388
+ spacingSec: true,
1389
+ };
1390
+ const TRIGGER_RATE_SPEC_KEYS = {
1391
+ name: true,
1392
+ lock: true,
1393
+ pluginDir: true,
1394
+ skillsDir: true,
1395
+ prompts: true,
1396
+ irrelevantPrompts: true,
1397
+ fired: true,
1398
+ installSet: true,
1399
+ stubSkillBodies: true,
1400
+ minPrompts: true,
1401
+ minDistance: true,
1402
+ trials: true,
1403
+ model: true,
1404
+ effort: true,
1405
+ minModel: true,
1406
+ allowedTools: true,
1407
+ timeoutMs: true,
1408
+ spacingSec: true,
1409
+ fixture: true,
1410
+ concurrency: true,
1411
+ };
1412
+ /**
1413
+ * The closest declared field to an unknown one, or undefined when nothing is
1414
+ * close — a wrong suggestion is worse than none (it invites "fixing" a field the
1415
+ * author never meant). Same tight threshold as the CLI's unknown-flag hint.
1416
+ */
1417
+ function nearestField(unknown, known) {
1418
+ let best;
1419
+ for (const name of known) {
1420
+ const d = (0, edit_distance_js_1.editDistance)(unknown.toLowerCase(), name.toLowerCase());
1421
+ if (!best || d < best.d)
1422
+ best = { name, d };
1423
+ }
1424
+ return best && best.d <= Math.max(2, Math.floor(unknown.length / 4))
1425
+ ? best.name
1426
+ : undefined;
1427
+ }
1428
+ /**
1429
+ * Throw when `spec` carries a field outside `known` — every unknown one named,
1430
+ * with a did-you-mean where a declared field is one typo away. `at` says which
1431
+ * runner and spec type the message is about (`arm` labels a nested arm). Runs
1432
+ * before anything is packaged or spent.
1433
+ */
1434
+ function assertKnownKeys(spec, known, at) {
1435
+ const declared = Object.keys(known);
1436
+ const where = at.arm === undefined ? "" : ` on arm "${at.arm}"`;
1437
+ const problems = Object.keys(spec)
1438
+ .filter((key) => !Object.hasOwn(known, key))
1439
+ .map((key) => {
1440
+ const near = nearestField(key, declared);
1441
+ const hint = near === undefined ? "" : ` — did you mean \`${near}\`?`;
1442
+ return `${at.caller}: unknown ${at.type} field "${key}"${where}${hint}`;
1443
+ });
1444
+ if (problems.length === 0)
1445
+ return;
1446
+ throw new Error(`${problems.join("\n")}\n ${at.type} fields: ${declared.join(", ")}.\n` +
1447
+ " An unknown field is refused rather than ignored: a dropped field would " +
1448
+ "make the run measure a setup you did not ask for.");
1449
+ }
1450
+ async function runEvalWith(input, runner) {
1451
+ // Refuse a stray field BEFORE spending a token: an eval file is plain JS, so a
1452
+ // typo'd or misplaced key would otherwise vanish and the run would report a
1453
+ // confident number about the wrong setup (issue #307).
1454
+ assertKnownKeys(input, EVAL_SPEC_KEYS, {
1455
+ caller: "runEval",
1456
+ type: "EvalSpec",
1457
+ });
1458
+ for (const [name, arm] of Object.entries(input.arms))
1459
+ assertKnownKeys(arm, EVAL_ARM_KEYS, {
1460
+ caller: "runEval",
1461
+ type: "EvalArm",
1462
+ arm: name,
1463
+ });
1321
1464
  // Tell the CLI runner this script exercised the harness, so a file that runs
1322
1465
  // NOTHING can be told apart from one that ran and passed. See check-count.ts.
1323
1466
  (0, check_count_js_1.recordCheck)();
1467
+ // Every arm's install source (`pluginDir` as-is / stubbed, or a loose
1468
+ // `skillsDir` packaged into a throwaway plugin) is resolved HERE, once — the
1469
+ // one place that decides what `--plugin-dir` receives, so measure / measureArms
1470
+ // / runEval cannot disagree about it. The throwaways are removed afterward.
1471
+ const { arms: resolvedArms, packaged } = resolveArmInstalls(input.arms, input.stubSkillBodies ?? false);
1472
+ const spec = { ...input, arms: resolvedArms };
1473
+ try {
1474
+ return await runResolvedEval(spec, runner);
1475
+ }
1476
+ finally {
1477
+ for (const dir of packaged)
1478
+ (0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
1479
+ }
1480
+ }
1481
+ /** {@link runEvalWith} after its arms' install sources are concrete plugin dirs. */
1482
+ async function runResolvedEval(spec, runner) {
1324
1483
  const trials = spec.trials ?? 5;
1325
1484
  const spacing = (spec.spacingSec ?? 4) * 1000;
1326
1485
  const concurrency = spec.concurrency ?? 1;
@@ -1493,7 +1652,7 @@ function packageSkillsDir(skillsDir, opts = {}) {
1493
1652
  throw new Error(`skillsDir not found: ${skillsDir} (resolved ${abs})`);
1494
1653
  const root = (0, tmp_root_js_1.makeTmpDir)("skills");
1495
1654
  (0, node_fs_1.mkdirSync)((0, node_path_1.join)(root, ".claude-plugin"), { recursive: true });
1496
- (0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? "vigiles-loose-skills", version: "0.0.0" }, null, 2));
1655
+ (0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? LOOSE_SKILLS_NAMESPACE, version: "0.0.0" }, null, 2));
1497
1656
  const skillsOut = (0, node_path_1.join)(root, "skills");
1498
1657
  (0, node_fs_1.mkdirSync)(skillsOut, { recursive: true });
1499
1658
  let copied = 0;
@@ -1704,16 +1863,96 @@ function packageInstallSet(opts) {
1704
1863
  throw e;
1705
1864
  }
1706
1865
  }
1707
- /** The under-test skills source + plugin name (the namespace `fired` matches). */
1708
- function underTestSource(spec) {
1709
- if (spec.skillsDir)
1710
- return { src: spec.skillsDir, name: "vigiles-loose-skills" };
1711
- if (spec.pluginDir)
1712
- return {
1713
- src: skillsDirOf(spec.pluginDir),
1714
- name: pluginName(spec.pluginDir) ?? "vigiles-loose-skills",
1715
- };
1716
- throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
1866
+ // ---------------------------------------------------------------------------
1867
+ // The install source — the ONE place that decides what `--plugin-dir` receives.
1868
+ //
1869
+ // Every runner that installs skills (runEval / measure / measureArms via an
1870
+ // EvalArm, measureTriggerRate via its spec) accepts the same two fields —
1871
+ // `pluginDir` (a complete plugin, used as-is or re-packaged with stubbed bodies)
1872
+ // and `skillsDir` (a loose `<name>/SKILL.md` dir, packaged into a throwaway
1873
+ // plugin) — and they all resolve through `resolveSkillInstall`. Before this
1874
+ // existed each runner re-derived the rule for itself, and the one written last
1875
+ // was the only one that knew about loose dirs (issue #307).
1876
+ // ---------------------------------------------------------------------------
1877
+ /** The plugin name a packaged loose skills dir installs under. */
1878
+ const LOOSE_SKILLS_NAMESPACE = "vigiles-loose-skills";
1879
+ /**
1880
+ * The plugin namespace a source's skills install under — the `<plugin>` half of
1881
+ * the `<plugin>:<skill>` id `skill()` / `skillResolved` match. A loose dir gets
1882
+ * the packager's name; a plugin its declared name, falling back to the same
1883
+ * synthetic one when its manifest has none. Pure (reads the manifest only).
1884
+ */
1885
+ function installNamespace(src) {
1886
+ if (src.skillsDir)
1887
+ return LOOSE_SKILLS_NAMESPACE;
1888
+ if (src.pluginDir)
1889
+ return pluginName(src.pluginDir) ?? LOOSE_SKILLS_NAMESPACE;
1890
+ return undefined;
1891
+ }
1892
+ /**
1893
+ * Resolve a {@link SkillSource} to a concrete plugin dir. `stub` strips skill
1894
+ * bodies (frontmatter kept) into a throwaway; `installSet` merges competitor
1895
+ * skills in (the whole-harness tier, `measureTriggerRate` only). Exactly one of
1896
+ * `pluginDir` / `skillsDir` may be set; neither resolves to nothing installed —
1897
+ * the caller decides whether that is legal (an arm: yes; a trigger run: no).
1898
+ */
1899
+ function resolveSkillInstall(src, opts) {
1900
+ if (src.pluginDir && src.skillsDir)
1901
+ throw new Error(`${opts.caller}: set \`pluginDir\` OR \`skillsDir\`, not both.`);
1902
+ const namespace = installNamespace(src);
1903
+ if (namespace === undefined)
1904
+ return {};
1905
+ const installSet = opts.installSet ?? [];
1906
+ if (installSet.length > 0) {
1907
+ // Whole-harness tier: merge the under-test skills with the install set so
1908
+ // selection is competitive (the realistic, differentiated measurement).
1909
+ const { dir } = packageInstallSet({
1910
+ underTestSrc: src.skillsDir ?? skillsDirOf(src.pluginDir),
1911
+ name: namespace,
1912
+ installSet,
1913
+ stub: opts.stub,
1914
+ });
1915
+ return { pluginDir: dir, packaged: dir, namespace };
1916
+ }
1917
+ if (src.skillsDir) {
1918
+ const dir = packageSkillsDir(src.skillsDir, { stub: opts.stub });
1919
+ return { pluginDir: dir, packaged: dir, namespace };
1920
+ }
1921
+ // A real plugin: as-is, or re-packaged from its skills/ with bodies stripped —
1922
+ // keeping the original plugin NAME so `<name>:<skill>` still matches.
1923
+ const pluginDir = src.pluginDir;
1924
+ if (!opts.stub)
1925
+ return { pluginDir, namespace };
1926
+ const dir = stubbedPluginDir(pluginDir);
1927
+ return { pluginDir: dir, packaged: dir, namespace };
1928
+ }
1929
+ /**
1930
+ * Resolve every arm's install source (see {@link resolveSkillInstall}) so each
1931
+ * arm carries only a concrete `pluginDir` — `skillsDir` is consumed here. Returns
1932
+ * the rewritten arms plus the throwaway dirs the caller removes afterward. A
1933
+ * failure part-way removes what was already built (no leaked temp dirs).
1934
+ */
1935
+ function resolveArmInstalls(arms, stub) {
1936
+ const out = {};
1937
+ const packaged = [];
1938
+ try {
1939
+ for (const [name, arm] of Object.entries(arms)) {
1940
+ const { skillsDir: _consumed, ...rest } = arm;
1941
+ const r = resolveSkillInstall(arm, {
1942
+ caller: `eval arm "${name}"`,
1943
+ stub,
1944
+ });
1945
+ if (r.packaged)
1946
+ packaged.push(r.packaged);
1947
+ out[name] = { ...rest, pluginDir: r.pluginDir };
1948
+ }
1949
+ }
1950
+ catch (e) {
1951
+ for (const dir of packaged)
1952
+ (0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
1953
+ throw e;
1954
+ }
1955
+ return { arms: out, packaged };
1717
1956
  }
1718
1957
  /** Number of `<name>/SKILL.md` skills installed in a plugin — the selection pool. */
1719
1958
  function countSkills(pluginDir) {
@@ -1727,46 +1966,22 @@ function countSkills(pluginDir) {
1727
1966
  return n;
1728
1967
  }
1729
1968
  function resolveTriggerPluginDir(spec) {
1730
- if (spec.pluginDir && spec.skillsDir)
1731
- throw new Error("measureTriggerRate: set `pluginDir` OR `skillsDir`, not both.");
1732
- const stub = spec.stubSkillBodies ?? true; // trigger = frontmatter; body never needed
1733
- const installSet = spec.installSet ?? [];
1734
- let pluginDir;
1735
- let packaged;
1736
- if (installSet.length > 0) {
1737
- // Whole-harness tier: merge the under-test skills with the install set so
1738
- // selection is competitive (the realistic, differentiated measurement).
1739
- const { src, name } = underTestSource(spec);
1740
- ({ dir: pluginDir } = packageInstallSet({
1741
- underTestSrc: src,
1742
- name,
1743
- installSet,
1744
- stub,
1745
- }));
1746
- packaged = pluginDir;
1747
- }
1748
- else if (spec.skillsDir) {
1749
- packaged = packageSkillsDir(spec.skillsDir, { stub });
1750
- pluginDir = packaged;
1751
- }
1752
- else if (spec.pluginDir) {
1753
- // Stub a real plugin: build a minimal plugin from its skills/ with bodies
1754
- // stripped — keep the original plugin NAME so `<name>:<skill>` still matches.
1755
- packaged = stub ? stubbedPluginDir(spec.pluginDir) : undefined;
1756
- pluginDir = packaged ?? spec.pluginDir;
1757
- }
1758
- else {
1969
+ const r = resolveSkillInstall(spec, {
1970
+ caller: "measureTriggerRate",
1971
+ stub: spec.stubSkillBodies ?? true, // trigger = frontmatter; body never needed
1972
+ installSet: spec.installSet,
1973
+ });
1974
+ if (r.pluginDir === undefined || r.namespace === undefined)
1759
1975
  throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
1760
- }
1761
1976
  // `competitors` is the REAL selection pressure: every OTHER skill installed in
1762
1977
  // the resolved plugin (siblings already in the source + any installSet), not
1763
1978
  // just the installSet delta — so a multi-skill plugin is never mislabeled
1764
1979
  // "isolated". `max(0, …)` guards a 0-skill pool.
1765
1980
  return {
1766
- pluginDir,
1767
- packaged,
1768
- competitors: Math.max(0, countSkills(pluginDir) - 1),
1769
- namespace: underTestSource(spec).name,
1981
+ pluginDir: r.pluginDir,
1982
+ packaged: r.packaged,
1983
+ competitors: Math.max(0, countSkills(r.pluginDir) - 1),
1984
+ namespace: r.namespace,
1770
1985
  };
1771
1986
  }
1772
1987
  /** Run one prompt set × trials through `runner`, aggregating fired counts. */
@@ -1859,6 +2074,10 @@ function assertTriggerDiversity(spec) {
1859
2074
  }
1860
2075
  }
1861
2076
  async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
2077
+ assertKnownKeys(spec, TRIGGER_RATE_SPEC_KEYS, {
2078
+ caller: "measureTriggerRate",
2079
+ type: "TriggerRateSpec",
2080
+ });
1862
2081
  // Tell the CLI runner this script exercised the harness (see check-count.ts).
1863
2082
  (0, check_count_js_1.recordCheck)();
1864
2083
  // Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
package/dist/exclude.d.ts CHANGED
@@ -8,12 +8,10 @@ export interface ExcludeSet {
8
8
  /** The user's patterns, as written (for messages). */
9
9
  readonly patterns: readonly string[];
10
10
  /**
11
- * The string-list face for a glob rooted AT `root`: the floor, then each user
12
- * pattern normalized so a bare directory name excludes its subtree (`bench` →
13
- * `bench`, `bench/**`), which is what "tsconfig-style" promises.
11
+ * The function face for `globSync`, correct whatever the glob's `cwd` is. A
12
+ * bare directory name excludes its subtree (`bench` → `bench`, `bench/**`),
13
+ * which is what "tsconfig-style" promises.
14
14
  */
15
- readonly ignore: readonly string[];
16
- /** The function face for `globSync`, correct whatever the glob's `cwd` is. */
17
15
  readonly globIgnore: IgnoreLike;
18
16
  /** Is this root-relative path excluded (floor or user pattern)? */
19
17
  matches(rel: string): boolean;
package/dist/exclude.js CHANGED
@@ -4,8 +4,9 @@ exports.excludesNothing = exports.EXCLUDE_FLOOR = void 0;
4
4
  exports.excludeSet = excludeSet;
5
5
  exports.excludedBy = excludedBy;
6
6
  /**
7
- * The ONE exclusion policy for every walk that polices the user's repository (#192) — the parsed `.vigilesrc.json#exclude` as an `ExcludeSet`, built ONCE where `loadConfig()` runs and taken as a REQUIRED parameter by every in-scope discovery (`findSpecs`, `findInstructionFiles`, `discoverNestedBundles`, `collectDocumentedRules`, `gatherInstructionFiles` in cli.ts; the string face handed to `findDocRefs`, `findOrphanDocs` (`repoExclude`), `findUntestedSurfaces`/`skillTestNudge`, `discoverScripts`, `computeScriptCoverage`).
8
- * Two faces: `globIgnore` (an `IgnoreLike` keyed on the path's position relative to the REPO root, so a glob rooted below it — `vigiles lint some/dir` — still applies a root-relative exclude) and `ignore` (the normalized string list for pure core detectors that glob from the root).
7
+ * The ONE exclusion policy for every walk that polices the user's repository (#192) — the parsed `.vigilesrc.json#exclude` as an `ExcludeSet`, built ONCE where `loadConfig()` runs and taken as a REQUIRED parameter by every in-scope discovery (`findSpecs`, `findInstructionFiles`, `discoverNestedBundles`, `collectDocumentedRules`, `gatherInstructionFiles` in cli.ts; the `globIgnore` face handed to `findDocRefs`, `findOrphanDocs` (`repoExclude`), `discoverScripts`, `computeScriptCoverage`; the whole set to `findUntestedSurfaces`/`skillTestNudge`/`scanPlugin`).
8
+ * Faces: `globIgnore` (an `IgnoreLike` keyed on the path's position relative to the REPO root, so a glob rooted anywhere — `vigiles lint some/dir`, a nested bundle — still applies a root-relative exclude), `matches`/`explain` (a root-relative path), and `excludedBy` below (an absolute path).
9
+ * 🔴 There is deliberately NO string-list face any more (#281). It was `ignore`, correct only for a glob rooted AT `root` — a precondition that lived in a comment, and two callers that globbed from a nested bundle broke it: the repo exclude never reached the bundle, and a root-relative pattern aliased into it. `core/glob-ignore.ts` unions a detector's own string floor with `globIgnore` instead.
9
10
  * A bare directory name excludes its subtree, as tsconfig/ESLint do — measured 2026-09-03: glob's own string `ignore` treated `bench` and `bench/` as matching NOTHING while the minimatch helper in `discoverNestedBundles` accepted them, so the two walks that honoured `exclude` disagreed.
10
11
  * The floor (node_modules/dist/.git/.vigiles) lives here, not per walk.
11
12
  * `exclude` filters DISCOVERY only: an explicitly named path is processed and ONE line names the pattern it matched (rg/tsc semantics with ESLint's loudness; never prettier's silent 'all clean').
@@ -57,7 +58,6 @@ function excludeSet(root, patterns) {
57
58
  return {
58
59
  root,
59
60
  patterns: user,
60
- ignore: [...exports.EXCLUDE_FLOOR, ...user.flatMap((p) => [p, `${p}/**`])],
61
61
  globIgnore: {
62
62
  ignored: (p) => matches(relOf(p)),
63
63
  childrenIgnored: (p) => matches(relOf(p)),
package/dist/scan.d.ts CHANGED
@@ -423,9 +423,9 @@ export declare function scanPlugin(dir: string, layout: PluginLayout, dialect: H
423
423
  * would be twenty-odd mechanical edits for one behavioural change.
424
424
  *
425
425
  * ⚠️ Omitting it is NOT "the repo excludes nothing" — it is "this caller has
426
- * no ExcludeSet to give", and the walk then reads everything. Today only
427
- * `audit` supplies one; the `lint` rule checkers below still do not (they
428
- * share a `(config, silent, adapter, root)` signature through `overBundles`).
426
+ * no ExcludeSet to give", and the walk then reads everything. `audit` and
427
+ * every `lint` rule checker supply one (the checkers through the per-bundle
428
+ * context `overBundles` builds, so a new checker cannot forget it).
429
429
  */
430
430
  excludes?: ExcludeSet;
431
431
  /**