vigiles 31.0.1 → 32.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli-main.js CHANGED
@@ -4871,7 +4871,7 @@ function resolveRecords(cwd, runs, tier, harnessFlag) {
4871
4871
  * 1. `.claude-plugin/plugin.json#name` — the repo's own declared name, used
4872
4872
  * when the run installed the repo AS a plugin (`pluginDir`).
4873
4873
  * 2. `vigiles-loose-skills` — the synthetic name OUR OWN packaging gives a
4874
- * loose `.claude/skills` dir (`packageSkillsDir`, and `underTestSource`'s
4874
+ * loose `.claude/skills` dir (`packageSkillsDir`, and `installNamespace`'s
4875
4875
  * fallback when a plugin manifest has no name). A repo that is not a plugin
4876
4876
  * still reports namespaced ids under it, so omitting it would drop every
4877
4877
  * trigger-rate record for the documented one-liner.
package/dist/eval.d.ts CHANGED
@@ -25,9 +25,21 @@ export interface EvalArm {
25
25
  * arm, so its skills/commands/agents activate the real way — the real model
26
26
  * can trigger a skill by its description (vs. `plugin`, which materializes a
27
27
  * file subset that does not register skills). Point at a COMPLETE plugin. Lets
28
- * an arm be "skill installed" vs "off" to measure real activation.
28
+ * an arm be "skill installed" vs "off" to measure real activation. Provide this
29
+ * OR {@link skillsDir}, not both.
29
30
  */
30
31
  readonly pluginDir?: string;
32
+ /**
33
+ * A directory of LOOSE skills (`<skillsDir>/<name>/SKILL.md`, e.g. a repo's
34
+ * `.claude/skills`) to install for this arm. vigiles packages them into a
35
+ * throwaway `--plugin-dir` (manifest + `skills/<name>/`, each skill dir copied
36
+ * whole) for the run and removes it afterward — the one-liner for repo-local
37
+ * skills that aren't a published plugin. The skills install under the
38
+ * namespace `vigiles-loose-skills`, so a `skill()` check / `skillResolved`
39
+ * matches `vigiles-loose-skills:<name>` (the report's `namespace` says so).
40
+ * Provide this OR {@link pluginDir}, not both.
41
+ */
42
+ readonly skillsDir?: string;
31
43
  /**
32
44
  * Tools to intercept for this arm (the tool-call spy). Each
33
45
  * {@link ToolIntercept} is denied its real execution by an auto-wired PreToolUse
@@ -83,6 +95,15 @@ export interface EvalSpec<M extends Metrics> {
83
95
  readonly fixture?: Record<string, string>;
84
96
  /** The arms to compare, by name. */
85
97
  readonly arms: Record<string, EvalArm>;
98
+ /**
99
+ * Stub each arm's skill BODIES (frontmatter kept) before the run — for firing
100
+ * comparisons (does the skill get SELECTED?), where a selected skill should
101
+ * stop at selection instead of running its (often expensive) procedure. Every
102
+ * arm with a `pluginDir` / `skillsDir` is repackaged with bodies stripped; arms
103
+ * without one are untouched. Don't combine with quality metrics: the body is
104
+ * gone, so there is nothing to grade. See {@link stubSkillBody}.
105
+ */
106
+ readonly stubSkillBodies?: boolean;
86
107
  /** The task prompt given to the agent. */
87
108
  readonly task: string;
88
109
  /** Compute this run's metrics from its outcome. */
@@ -328,15 +349,24 @@ export interface MeasureSpec {
328
349
  readonly settings?: unknown;
329
350
  /** A real plugin/repo to load (materialized) — see `EvalArm.plugin`. */
330
351
  readonly plugin?: string;
331
- /** A complete plugin dir to install natively (`--plugin-dir`) so skills activate. */
352
+ /**
353
+ * A complete plugin dir to install natively (`--plugin-dir`) so skills activate
354
+ * — see {@link EvalArm.pluginDir}. Provide this OR `skillsDir`, not both.
355
+ */
332
356
  readonly pluginDir?: string;
333
357
  /**
334
- * Stub each skill BODY in `pluginDir` (frontmatter/trigger surface kept) before
335
- * the run — for checks about whether a skill FIRES (`skill()`), not what it
336
- * produces. A selected skill stops at selection instead of running its (often
337
- * expensive) procedure, so a description/firing run costs a fraction of the
338
- * tokens. Do NOT combine with `judged`/quality checks: the body is gone, so
339
- * there's nothing to grade. Requires `pluginDir`. See {@link stubSkillBody}.
358
+ * A loose `<skillsDir>/<name>/SKILL.md` directory (e.g. a repo's `.claude/skills`)
359
+ * to install, auto-packaged into a throwaway plugin — see {@link EvalArm.skillsDir}.
360
+ * Installs under `vigiles-loose-skills`, so `skill("vigiles-loose-skills:<name>")`.
361
+ */
362
+ readonly skillsDir?: string;
363
+ /**
364
+ * Stub each skill BODY (frontmatter/trigger surface kept) before the run — for
365
+ * checks about whether a skill FIRES (`skill()`), not what it produces. A
366
+ * selected skill stops at selection instead of running its (often expensive)
367
+ * procedure, so a description/firing run costs a fraction of the tokens. Do NOT
368
+ * combine with `judged`/quality checks: the body is gone, so there's nothing to
369
+ * grade. Requires `pluginDir` or `skillsDir`. See {@link stubSkillBody}.
340
370
  */
341
371
  readonly stubSkillBodies?: boolean;
342
372
  /** Tools to intercept (the tool-call spy) — see {@link EvalArm.interceptTools}. */
@@ -385,6 +415,15 @@ export interface CheckReport {
385
415
  readonly perCheck: readonly CheckRate[];
386
416
  /** Cost / latency / token totals for the run (the same source as `runEval`). */
387
417
  readonly usage: ArmUsage;
418
+ /**
419
+ * The plugin namespace the skills actually installed under — the `<plugin>`
420
+ * half of the `<plugin>:<skill>` id a `skill()` check matches. Undefined when
421
+ * the run installed nothing (no `pluginDir` / `skillsDir`). Reported because
422
+ * with `skillsDir` the name is chosen by the packager, not the caller, so the
423
+ * most common cause of a 0% `skill()` rate was a value the caller never saw.
424
+ * Mirrors {@link TriggerRateReport.namespace}.
425
+ */
426
+ readonly namespace?: string;
388
427
  }
389
428
  /**
390
429
  * Score a check vocabulary across trials — the scored counterpart to
@@ -407,8 +446,8 @@ export interface ArmsMeasureSpec {
407
446
  * Stub each arm's skill BODIES (frontmatter kept) before the run — the A/B
408
447
  * counterpart to {@link MeasureSpec.stubSkillBodies}. For firing comparisons
409
448
  * (does description variant A fire more than B?), every arm that sets
410
- * `pluginDir` is repackaged with bodies stripped so each run stops at
411
- * selection — a fraction of the tokens. Arms without a `pluginDir` are left
449
+ * `pluginDir` / `skillsDir` is repackaged with bodies stripped so each run
450
+ * stops at selection — a fraction of the tokens. Arms without one are left
412
451
  * untouched. Don't combine with `judged`/quality checks. See {@link stubSkillBody}.
413
452
  */
414
453
  readonly stubSkillBodies?: boolean;
@@ -593,7 +632,7 @@ export declare function seedEphemeralHome(throwawayHome: string, realHome: strin
593
632
  export declare function isRateLimited(out: RunOut): boolean;
594
633
  /** Map `worker` over `items` with at most `concurrency` in flight, order preserved. */
595
634
  export declare function runPool<T, R>(items: readonly T[], concurrency: number, worker: (item: T) => Promise<R>): Promise<R[]>;
596
- export declare function runEvalWith<M extends Metrics>(spec: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
635
+ export declare function runEvalWith<M extends Metrics>(input: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
597
636
  /** Format an eval report as a compact table for the console (mean ± se, pass^k). */
598
637
  export declare function formatEvalReport(report: EvalReport): string;
599
638
  /**
package/dist/eval.js CHANGED
@@ -87,6 +87,7 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
87
87
  const tool_intercept_js_1 = require("./tool-intercept.js");
88
88
  const tool_stub_js_1 = require("./tool-stub.js");
89
89
  const tmp_root_js_1 = require("./core/tmp-root.js");
90
+ const edit_distance_js_1 = require("./core/edit-distance.js");
90
91
  function writeFiles(cwd, files) {
91
92
  for (const [p, content] of Object.entries(files)) {
92
93
  const full = (0, node_path_1.resolve)(cwd, p);
@@ -315,8 +316,12 @@ async function runEval(spec) {
315
316
  * `runner` so the orchestration is unit-testable without a model.
316
317
  */
317
318
  async function measureWith(spec, runner) {
318
- if (spec.stubSkillBodies && !spec.pluginDir)
319
- throw new Error("measure: `stubSkillBodies` requires `pluginDir`.");
319
+ assertKnownKeys(spec, MEASURE_SPEC_KEYS, {
320
+ caller: "measure",
321
+ type: "MeasureSpec",
322
+ });
323
+ if (spec.stubSkillBodies && !spec.pluginDir && !spec.skillsDir)
324
+ throw new Error("measure: `stubSkillBodies` requires `pluginDir` or `skillsDir`.");
320
325
  // stubSkillBodies replaces each skill BODY with a no-op (the run stops at
321
326
  // selection), so there is no output to grade — a `judged` check would score an
322
327
  // empty body and mislead. The docs warn against this pairing; enforce it.
@@ -325,51 +330,49 @@ async function measureWith(spec, runner) {
325
330
  throw new Error("measure: `stubSkillBodies` is for firing/`skill()` checks only — it stubs " +
326
331
  "the skill body, so there's no output for a `judged` check to grade. Drop " +
327
332
  "`stubSkillBodies`, or remove the `judged` check.");
328
- const stubbed = spec.stubSkillBodies
329
- ? stubbedPluginDir(spec.pluginDir)
330
- : undefined;
331
- const pluginDir = stubbed ?? spec.pluginDir;
332
- try {
333
- const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
334
- const report = await runEvalWith({
335
- fixture: spec.fixture,
336
- arms: {
337
- run: {
338
- settings: spec.settings,
339
- plugin: spec.plugin,
340
- pluginDir,
341
- interceptTools: spec.interceptTools,
342
- },
343
- },
344
- task: spec.task,
345
- trials: spec.trials ?? 5,
346
- model: spec.model ?? "sonnet",
347
- effort: spec.effort,
348
- allowedTools: spec.allowedTools,
349
- timeoutMs: spec.timeoutMs,
350
- spacingSec: spec.spacingSec,
351
- measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
352
- }, runner);
353
- const arm = report.arms.run;
354
- return {
355
- n: arm?.runs ?? 0,
356
- perCheck: keyed.map(([k, c]) => {
357
- const s = arm?.stats[k];
358
- return {
359
- check: c.toJSON(),
360
- rate: s?.mean ?? 0,
361
- se: s?.se ?? 0,
362
- passK: s?.passK ?? 0,
363
- n: s?.n ?? 0,
364
- };
365
- }),
366
- usage: arm?.usage ?? aggregateUsage([]),
367
- };
368
- }
369
- finally {
370
- if (stubbed)
371
- (0, node_fs_1.rmSync)(stubbed, { recursive: true, force: true });
372
- }
333
+ const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
334
+ // The install source (plugin / pluginDir / skillsDir, stubbed or not) is
335
+ // resolved by runEvalWith, once per arm — measure is one arm, so it just
336
+ // forwards the fields and reads the arm's report back.
337
+ const arm = {
338
+ settings: spec.settings,
339
+ plugin: spec.plugin,
340
+ pluginDir: spec.pluginDir,
341
+ skillsDir: spec.skillsDir,
342
+ interceptTools: spec.interceptTools,
343
+ };
344
+ const report = await runEvalWith({
345
+ fixture: spec.fixture,
346
+ arms: { run: arm },
347
+ stubSkillBodies: spec.stubSkillBodies,
348
+ task: spec.task,
349
+ trials: spec.trials ?? 5,
350
+ model: spec.model ?? "sonnet",
351
+ effort: spec.effort,
352
+ allowedTools: spec.allowedTools,
353
+ timeoutMs: spec.timeoutMs,
354
+ spacingSec: spec.spacingSec,
355
+ measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
356
+ }, runner);
357
+ return checkReportOf(report.arms.run, keyed, installNamespace(arm));
358
+ }
359
+ /** Read one arm's {@link ArmReport} back into a {@link CheckReport}. */
360
+ function checkReportOf(arm, keyed, namespace) {
361
+ return {
362
+ n: arm?.runs ?? 0,
363
+ perCheck: keyed.map(([k, c]) => {
364
+ const s = arm?.stats[k];
365
+ return {
366
+ check: c.toJSON(),
367
+ rate: s?.mean ?? 0,
368
+ se: s?.se ?? 0,
369
+ passK: s?.passK ?? 0,
370
+ n: s?.n ?? 0,
371
+ };
372
+ }),
373
+ usage: arm?.usage ?? aggregateUsage([]),
374
+ namespace,
375
+ };
373
376
  }
374
377
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
375
378
  /** Score a check vocabulary across trials against the real `claude` CLI. */
@@ -380,67 +383,36 @@ async function measure(spec) {
380
383
  }
381
384
  /** Score checks across arms (injectable runner). Reuses `runEvalWith`. */
382
385
  async function measureArmsWith(spec, runner) {
386
+ assertKnownKeys(spec, ARMS_MEASURE_SPEC_KEYS, {
387
+ caller: "measureArms",
388
+ type: "ArmsMeasureSpec",
389
+ });
390
+ for (const [name, arm] of Object.entries(spec.arms))
391
+ assertKnownKeys(arm, EVAL_ARM_KEYS, {
392
+ caller: "measureArms",
393
+ type: "EvalArm",
394
+ arm: name,
395
+ });
383
396
  const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
384
- const { arms: runArms, temps } = spec.stubSkillBodies
385
- ? stubArmPluginDirs(spec.arms)
386
- : { arms: spec.arms, temps: [] };
387
- try {
388
- const report = await runEvalWith({
389
- fixture: spec.fixture,
390
- arms: runArms,
391
- task: spec.task,
392
- trials: spec.trials ?? 5,
393
- model: spec.model ?? "sonnet",
394
- effort: spec.effort,
395
- allowedTools: spec.allowedTools,
396
- timeoutMs: spec.timeoutMs,
397
- spacingSec: spec.spacingSec,
398
- measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
399
- }, runner);
400
- const arms = {};
401
- for (const [armName, arm] of Object.entries(report.arms)) {
402
- arms[armName] = {
403
- n: arm.runs,
404
- perCheck: keyed.map(([k, c]) => {
405
- const s = arm.stats[k];
406
- return {
407
- check: c.toJSON(),
408
- rate: s?.mean ?? 0,
409
- se: s?.se ?? 0,
410
- passK: s?.passK ?? 0,
411
- n: s?.n ?? 0,
412
- };
413
- }),
414
- usage: arm.usage,
415
- };
416
- }
417
- return { arms };
418
- }
419
- finally {
420
- for (const t of temps)
421
- (0, node_fs_1.rmSync)(t, { recursive: true, force: true });
422
- }
423
- }
424
- /**
425
- * Repackage every arm that sets a `pluginDir` with its skill bodies stubbed
426
- * (frontmatter kept), for an A/B firing comparison. Returns the rewritten arms
427
- * plus the throwaway dirs the caller must remove. Arms without a `pluginDir` pass
428
- * through unchanged. See {@link stubbedPluginDir}.
429
- */
430
- function stubArmPluginDirs(arms) {
431
- const out = {};
432
- const temps = [];
433
- for (const [name, arm] of Object.entries(arms)) {
434
- if (arm.pluginDir) {
435
- const stubbed = stubbedPluginDir(arm.pluginDir);
436
- temps.push(stubbed);
437
- out[name] = { ...arm, pluginDir: stubbed };
438
- }
439
- else {
440
- out[name] = arm;
441
- }
397
+ // Per-arm install sources (and the stub) are resolved by runEvalWith.
398
+ const report = await runEvalWith({
399
+ fixture: spec.fixture,
400
+ arms: spec.arms,
401
+ stubSkillBodies: spec.stubSkillBodies,
402
+ task: spec.task,
403
+ trials: spec.trials ?? 5,
404
+ model: spec.model ?? "sonnet",
405
+ effort: spec.effort,
406
+ allowedTools: spec.allowedTools,
407
+ timeoutMs: spec.timeoutMs,
408
+ spacingSec: spec.spacingSec,
409
+ measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
410
+ }, runner);
411
+ const arms = {};
412
+ for (const [armName, arm] of Object.entries(report.arms)) {
413
+ arms[armName] = checkReportOf(arm, keyed, installNamespace(spec.arms[armName] ?? {}));
442
414
  }
443
- return { arms: out, temps };
415
+ return { arms };
444
416
  }
445
417
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
446
418
  /** Score checks across arms against the real `claude` CLI. */
@@ -482,6 +454,28 @@ function formatCheckReport(report) {
482
454
  lines.push(` ${(c.rate * 100).toFixed(0)}% ± ${(c.se * 100).toFixed(0)}% ${checkLabel(c.check)}` +
483
455
  ` (pass^k ${String(c.passK)})`);
484
456
  }
457
+ // The `skill()` twin of the trigger-rate TOTAL-zero note (see
458
+ // formatTriggerRateReport): every skill() check at 0% over real runs is far
459
+ // more often a wiring mistake — a bare id where the namespaced one is matched,
460
+ // `pluginDir` handed a loose dir, an empty cwd — than a finding, and it reads
461
+ // exactly like a finding. Only when EVERY skill check is at zero: a partial rate
462
+ // is a real measurement, and a note that hedges on good data gets ignored.
463
+ const skillChecks = report.perCheck.filter((c) => c.check.kind === "skill");
464
+ if (report.n > 0 &&
465
+ skillChecks.length > 0 &&
466
+ skillChecks.every((c) => c.rate === 0))
467
+ lines.push("⚠ nothing resolved for ANY skill() check. That is usually SETUP, not the skill — check, in order:\n" +
468
+ (report.namespace !== undefined
469
+ ? " 1. the id in `skill()` — your skills installed under " +
470
+ `\`${report.namespace}\`, so \`skill()\` matches ` +
471
+ `\`${report.namespace}:<skill>\`; a bare name silently never matches;\n`
472
+ : " 1. the id in `skill()` — it matches the NAMESPACED id " +
473
+ "(`<plugin>:<skill>`); a bare name silently never matches;\n") +
474
+ " 2. the install field — a loose `.claude/skills` dir needs `skillsDir`, " +
475
+ "not `pluginDir` (which wants a full plugin manifest);\n" +
476
+ " 3. the `fixture` — a run starts in an EMPTY cwd, so a prompt about a " +
477
+ "file that does not exist is one the model is right to decline.\n" +
478
+ " Rule out all three before recording this as a fact about the skill.");
485
479
  return lines.join("\n");
486
480
  }
487
481
  /**
@@ -1317,10 +1311,175 @@ function evalArmsInputs(spec, cfg) {
1317
1311
  ephemeralEnv: spec.ephemeralEnv === true,
1318
1312
  };
1319
1313
  }
1320
- async function runEvalWith(spec, runner) {
1314
+ // ---------------------------------------------------------------------------
1315
+ // The spec boundary — refuse a field the spec does not declare.
1316
+ //
1317
+ // A spec usually arrives from a plain-JS `*.eval.mjs` file, where nothing
1318
+ // type-checks the object literal, so a key that is misspelled (`skillDir`) or
1319
+ // belongs to a sibling runner (`prompts` on measure) used to be dropped without a
1320
+ // word — and the run then reported a confident number about a setup the author
1321
+ // never asked for (issue #307). The precedent is `driverMisplaced` in
1322
+ // eval-entry.ts: a field that would silently do nothing is REFUSED, not ignored.
1323
+ //
1324
+ // Each key table is `satisfies Record<keyof Spec, true>`, so the compiler fails
1325
+ // when a spec gains or loses a field the table does not: the list cannot drift.
1326
+ // ---------------------------------------------------------------------------
1327
+ const EVAL_ARM_KEYS = {
1328
+ files: true,
1329
+ settings: true,
1330
+ plugin: true,
1331
+ pluginDir: true,
1332
+ skillsDir: true,
1333
+ interceptTools: true,
1334
+ model: true,
1335
+ effort: true,
1336
+ };
1337
+ const EVAL_SPEC_KEYS = {
1338
+ name: true,
1339
+ fixture: true,
1340
+ arms: true,
1341
+ stubSkillBodies: true,
1342
+ task: true,
1343
+ measure: true,
1344
+ trials: true,
1345
+ model: true,
1346
+ effort: true,
1347
+ allowedTools: true,
1348
+ timeoutMs: true,
1349
+ spacingSec: true,
1350
+ cache: true,
1351
+ cacheDir: true,
1352
+ concurrency: true,
1353
+ maxCostUsd: true,
1354
+ rateLimitRetries: true,
1355
+ retryBackoffMs: true,
1356
+ ephemeralEnv: true,
1357
+ stubs: true,
1358
+ lock: true,
1359
+ };
1360
+ const MEASURE_SPEC_KEYS = {
1361
+ fixture: true,
1362
+ settings: true,
1363
+ plugin: true,
1364
+ pluginDir: true,
1365
+ skillsDir: true,
1366
+ stubSkillBodies: true,
1367
+ interceptTools: true,
1368
+ task: true,
1369
+ checks: true,
1370
+ trials: true,
1371
+ model: true,
1372
+ effort: true,
1373
+ allowedTools: true,
1374
+ timeoutMs: true,
1375
+ spacingSec: true,
1376
+ };
1377
+ const ARMS_MEASURE_SPEC_KEYS = {
1378
+ fixture: true,
1379
+ arms: true,
1380
+ task: true,
1381
+ checks: true,
1382
+ stubSkillBodies: true,
1383
+ trials: true,
1384
+ model: true,
1385
+ allowedTools: true,
1386
+ timeoutMs: true,
1387
+ effort: true,
1388
+ spacingSec: true,
1389
+ };
1390
+ const TRIGGER_RATE_SPEC_KEYS = {
1391
+ name: true,
1392
+ lock: true,
1393
+ pluginDir: true,
1394
+ skillsDir: true,
1395
+ prompts: true,
1396
+ irrelevantPrompts: true,
1397
+ fired: true,
1398
+ installSet: true,
1399
+ stubSkillBodies: true,
1400
+ minPrompts: true,
1401
+ minDistance: true,
1402
+ trials: true,
1403
+ model: true,
1404
+ effort: true,
1405
+ minModel: true,
1406
+ allowedTools: true,
1407
+ timeoutMs: true,
1408
+ spacingSec: true,
1409
+ fixture: true,
1410
+ concurrency: true,
1411
+ };
1412
+ /**
1413
+ * The closest declared field to an unknown one, or undefined when nothing is
1414
+ * close — a wrong suggestion is worse than none (it invites "fixing" a field the
1415
+ * author never meant). Same tight threshold as the CLI's unknown-flag hint.
1416
+ */
1417
+ function nearestField(unknown, known) {
1418
+ let best;
1419
+ for (const name of known) {
1420
+ const d = (0, edit_distance_js_1.editDistance)(unknown.toLowerCase(), name.toLowerCase());
1421
+ if (!best || d < best.d)
1422
+ best = { name, d };
1423
+ }
1424
+ return best && best.d <= Math.max(2, Math.floor(unknown.length / 4))
1425
+ ? best.name
1426
+ : undefined;
1427
+ }
1428
+ /**
1429
+ * Throw when `spec` carries a field outside `known` — every unknown one named,
1430
+ * with a did-you-mean where a declared field is one typo away. `at` says which
1431
+ * runner and spec type the message is about (`arm` labels a nested arm). Runs
1432
+ * before anything is packaged or spent.
1433
+ */
1434
+ function assertKnownKeys(spec, known, at) {
1435
+ const declared = Object.keys(known);
1436
+ const where = at.arm === undefined ? "" : ` on arm "${at.arm}"`;
1437
+ const problems = Object.keys(spec)
1438
+ .filter((key) => !Object.hasOwn(known, key))
1439
+ .map((key) => {
1440
+ const near = nearestField(key, declared);
1441
+ const hint = near === undefined ? "" : ` — did you mean \`${near}\`?`;
1442
+ return `${at.caller}: unknown ${at.type} field "${key}"${where}${hint}`;
1443
+ });
1444
+ if (problems.length === 0)
1445
+ return;
1446
+ throw new Error(`${problems.join("\n")}\n ${at.type} fields: ${declared.join(", ")}.\n` +
1447
+ " An unknown field is refused rather than ignored: a dropped field would " +
1448
+ "make the run measure a setup you did not ask for.");
1449
+ }
1450
+ async function runEvalWith(input, runner) {
1451
+ // Refuse a stray field BEFORE spending a token: an eval file is plain JS, so a
1452
+ // typo'd or misplaced key would otherwise vanish and the run would report a
1453
+ // confident number about the wrong setup (issue #307).
1454
+ assertKnownKeys(input, EVAL_SPEC_KEYS, {
1455
+ caller: "runEval",
1456
+ type: "EvalSpec",
1457
+ });
1458
+ for (const [name, arm] of Object.entries(input.arms))
1459
+ assertKnownKeys(arm, EVAL_ARM_KEYS, {
1460
+ caller: "runEval",
1461
+ type: "EvalArm",
1462
+ arm: name,
1463
+ });
1321
1464
  // Tell the CLI runner this script exercised the harness, so a file that runs
1322
1465
  // NOTHING can be told apart from one that ran and passed. See check-count.ts.
1323
1466
  (0, check_count_js_1.recordCheck)();
1467
+ // Every arm's install source (`pluginDir` as-is / stubbed, or a loose
1468
+ // `skillsDir` packaged into a throwaway plugin) is resolved HERE, once — the
1469
+ // one place that decides what `--plugin-dir` receives, so measure / measureArms
1470
+ // / runEval cannot disagree about it. The throwaways are removed afterward.
1471
+ const { arms: resolvedArms, packaged } = resolveArmInstalls(input.arms, input.stubSkillBodies ?? false);
1472
+ const spec = { ...input, arms: resolvedArms };
1473
+ try {
1474
+ return await runResolvedEval(spec, runner);
1475
+ }
1476
+ finally {
1477
+ for (const dir of packaged)
1478
+ (0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
1479
+ }
1480
+ }
1481
+ /** {@link runEvalWith} after its arms' install sources are concrete plugin dirs. */
1482
+ async function runResolvedEval(spec, runner) {
1324
1483
  const trials = spec.trials ?? 5;
1325
1484
  const spacing = (spec.spacingSec ?? 4) * 1000;
1326
1485
  const concurrency = spec.concurrency ?? 1;
@@ -1493,7 +1652,7 @@ function packageSkillsDir(skillsDir, opts = {}) {
1493
1652
  throw new Error(`skillsDir not found: ${skillsDir} (resolved ${abs})`);
1494
1653
  const root = (0, tmp_root_js_1.makeTmpDir)("skills");
1495
1654
  (0, node_fs_1.mkdirSync)((0, node_path_1.join)(root, ".claude-plugin"), { recursive: true });
1496
- (0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? "vigiles-loose-skills", version: "0.0.0" }, null, 2));
1655
+ (0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? LOOSE_SKILLS_NAMESPACE, version: "0.0.0" }, null, 2));
1497
1656
  const skillsOut = (0, node_path_1.join)(root, "skills");
1498
1657
  (0, node_fs_1.mkdirSync)(skillsOut, { recursive: true });
1499
1658
  let copied = 0;
@@ -1704,16 +1863,96 @@ function packageInstallSet(opts) {
1704
1863
  throw e;
1705
1864
  }
1706
1865
  }
1707
- /** The under-test skills source + plugin name (the namespace `fired` matches). */
1708
- function underTestSource(spec) {
1709
- if (spec.skillsDir)
1710
- return { src: spec.skillsDir, name: "vigiles-loose-skills" };
1711
- if (spec.pluginDir)
1712
- return {
1713
- src: skillsDirOf(spec.pluginDir),
1714
- name: pluginName(spec.pluginDir) ?? "vigiles-loose-skills",
1715
- };
1716
- throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
1866
+ // ---------------------------------------------------------------------------
1867
+ // The install source — the ONE place that decides what `--plugin-dir` receives.
1868
+ //
1869
+ // Every runner that installs skills (runEval / measure / measureArms via an
1870
+ // EvalArm, measureTriggerRate via its spec) accepts the same two fields —
1871
+ // `pluginDir` (a complete plugin, used as-is or re-packaged with stubbed bodies)
1872
+ // and `skillsDir` (a loose `<name>/SKILL.md` dir, packaged into a throwaway
1873
+ // plugin) — and they all resolve through `resolveSkillInstall`. Before this
1874
+ // existed each runner re-derived the rule for itself, and the one written last
1875
+ // was the only one that knew about loose dirs (issue #307).
1876
+ // ---------------------------------------------------------------------------
1877
+ /** The plugin name a packaged loose skills dir installs under. */
1878
+ const LOOSE_SKILLS_NAMESPACE = "vigiles-loose-skills";
1879
+ /**
1880
+ * The plugin namespace a source's skills install under — the `<plugin>` half of
1881
+ * the `<plugin>:<skill>` id `skill()` / `skillResolved` match. A loose dir gets
1882
+ * the packager's name; a plugin its declared name, falling back to the same
1883
+ * synthetic one when its manifest has none. Pure (reads the manifest only).
1884
+ */
1885
+ function installNamespace(src) {
1886
+ if (src.skillsDir)
1887
+ return LOOSE_SKILLS_NAMESPACE;
1888
+ if (src.pluginDir)
1889
+ return pluginName(src.pluginDir) ?? LOOSE_SKILLS_NAMESPACE;
1890
+ return undefined;
1891
+ }
1892
+ /**
1893
+ * Resolve a {@link SkillSource} to a concrete plugin dir. `stub` strips skill
1894
+ * bodies (frontmatter kept) into a throwaway; `installSet` merges competitor
1895
+ * skills in (the whole-harness tier, `measureTriggerRate` only). Exactly one of
1896
+ * `pluginDir` / `skillsDir` may be set; neither resolves to nothing installed —
1897
+ * the caller decides whether that is legal (an arm: yes; a trigger run: no).
1898
+ */
1899
+ function resolveSkillInstall(src, opts) {
1900
+ if (src.pluginDir && src.skillsDir)
1901
+ throw new Error(`${opts.caller}: set \`pluginDir\` OR \`skillsDir\`, not both.`);
1902
+ const namespace = installNamespace(src);
1903
+ if (namespace === undefined)
1904
+ return {};
1905
+ const installSet = opts.installSet ?? [];
1906
+ if (installSet.length > 0) {
1907
+ // Whole-harness tier: merge the under-test skills with the install set so
1908
+ // selection is competitive (the realistic, differentiated measurement).
1909
+ const { dir } = packageInstallSet({
1910
+ underTestSrc: src.skillsDir ?? skillsDirOf(src.pluginDir),
1911
+ name: namespace,
1912
+ installSet,
1913
+ stub: opts.stub,
1914
+ });
1915
+ return { pluginDir: dir, packaged: dir, namespace };
1916
+ }
1917
+ if (src.skillsDir) {
1918
+ const dir = packageSkillsDir(src.skillsDir, { stub: opts.stub });
1919
+ return { pluginDir: dir, packaged: dir, namespace };
1920
+ }
1921
+ // A real plugin: as-is, or re-packaged from its skills/ with bodies stripped —
1922
+ // keeping the original plugin NAME so `<name>:<skill>` still matches.
1923
+ const pluginDir = src.pluginDir;
1924
+ if (!opts.stub)
1925
+ return { pluginDir, namespace };
1926
+ const dir = stubbedPluginDir(pluginDir);
1927
+ return { pluginDir: dir, packaged: dir, namespace };
1928
+ }
1929
+ /**
1930
+ * Resolve every arm's install source (see {@link resolveSkillInstall}) so each
1931
+ * arm carries only a concrete `pluginDir` — `skillsDir` is consumed here. Returns
1932
+ * the rewritten arms plus the throwaway dirs the caller removes afterward. A
1933
+ * failure part-way removes what was already built (no leaked temp dirs).
1934
+ */
1935
+ function resolveArmInstalls(arms, stub) {
1936
+ const out = {};
1937
+ const packaged = [];
1938
+ try {
1939
+ for (const [name, arm] of Object.entries(arms)) {
1940
+ const { skillsDir: _consumed, ...rest } = arm;
1941
+ const r = resolveSkillInstall(arm, {
1942
+ caller: `eval arm "${name}"`,
1943
+ stub,
1944
+ });
1945
+ if (r.packaged)
1946
+ packaged.push(r.packaged);
1947
+ out[name] = { ...rest, pluginDir: r.pluginDir };
1948
+ }
1949
+ }
1950
+ catch (e) {
1951
+ for (const dir of packaged)
1952
+ (0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
1953
+ throw e;
1954
+ }
1955
+ return { arms: out, packaged };
1717
1956
  }
1718
1957
  /** Number of `<name>/SKILL.md` skills installed in a plugin — the selection pool. */
1719
1958
  function countSkills(pluginDir) {
@@ -1727,46 +1966,22 @@ function countSkills(pluginDir) {
1727
1966
  return n;
1728
1967
  }
1729
1968
  function resolveTriggerPluginDir(spec) {
1730
- if (spec.pluginDir && spec.skillsDir)
1731
- throw new Error("measureTriggerRate: set `pluginDir` OR `skillsDir`, not both.");
1732
- const stub = spec.stubSkillBodies ?? true; // trigger = frontmatter; body never needed
1733
- const installSet = spec.installSet ?? [];
1734
- let pluginDir;
1735
- let packaged;
1736
- if (installSet.length > 0) {
1737
- // Whole-harness tier: merge the under-test skills with the install set so
1738
- // selection is competitive (the realistic, differentiated measurement).
1739
- const { src, name } = underTestSource(spec);
1740
- ({ dir: pluginDir } = packageInstallSet({
1741
- underTestSrc: src,
1742
- name,
1743
- installSet,
1744
- stub,
1745
- }));
1746
- packaged = pluginDir;
1747
- }
1748
- else if (spec.skillsDir) {
1749
- packaged = packageSkillsDir(spec.skillsDir, { stub });
1750
- pluginDir = packaged;
1751
- }
1752
- else if (spec.pluginDir) {
1753
- // Stub a real plugin: build a minimal plugin from its skills/ with bodies
1754
- // stripped — keep the original plugin NAME so `<name>:<skill>` still matches.
1755
- packaged = stub ? stubbedPluginDir(spec.pluginDir) : undefined;
1756
- pluginDir = packaged ?? spec.pluginDir;
1757
- }
1758
- else {
1969
+ const r = resolveSkillInstall(spec, {
1970
+ caller: "measureTriggerRate",
1971
+ stub: spec.stubSkillBodies ?? true, // trigger = frontmatter; body never needed
1972
+ installSet: spec.installSet,
1973
+ });
1974
+ if (r.pluginDir === undefined || r.namespace === undefined)
1759
1975
  throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
1760
- }
1761
1976
  // `competitors` is the REAL selection pressure: every OTHER skill installed in
1762
1977
  // the resolved plugin (siblings already in the source + any installSet), not
1763
1978
  // just the installSet delta — so a multi-skill plugin is never mislabeled
1764
1979
  // "isolated". `max(0, …)` guards a 0-skill pool.
1765
1980
  return {
1766
- pluginDir,
1767
- packaged,
1768
- competitors: Math.max(0, countSkills(pluginDir) - 1),
1769
- namespace: underTestSource(spec).name,
1981
+ pluginDir: r.pluginDir,
1982
+ packaged: r.packaged,
1983
+ competitors: Math.max(0, countSkills(r.pluginDir) - 1),
1984
+ namespace: r.namespace,
1770
1985
  };
1771
1986
  }
1772
1987
  /** Run one prompt set × trials through `runner`, aggregating fired counts. */
@@ -1859,6 +2074,10 @@ function assertTriggerDiversity(spec) {
1859
2074
  }
1860
2075
  }
1861
2076
  async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
2077
+ assertKnownKeys(spec, TRIGGER_RATE_SPEC_KEYS, {
2078
+ caller: "measureTriggerRate",
2079
+ type: "TriggerRateSpec",
2080
+ });
1862
2081
  // Tell the CLI runner this script exercised the harness (see check-count.ts).
1863
2082
  (0, check_count_js_1.recordCheck)();
1864
2083
  // Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "31.0.1",
3
+ "version": "32.0.0",
4
4
  "description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
5
5
  "keywords": [
6
6
  "claude-code",