vigiles 31.0.0 → 32.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/run-scripts.d.ts +5 -3
- package/dist/adapters/claude-code/run-scripts.js +5 -3
- package/dist/cli-main.d.ts +8 -0
- package/dist/cli-main.js +297 -192
- package/dist/core/coverage.d.ts +4 -3
- package/dist/core/coverage.js +4 -3
- package/dist/core/doc-refs.d.ts +3 -1
- package/dist/core/doc-refs.js +2 -1
- package/dist/core/frame.d.ts +90 -0
- package/dist/core/frame.js +58 -0
- package/dist/core/glob-ignore.d.ts +28 -0
- package/dist/core/glob-ignore.js +42 -0
- package/dist/core/orphans.d.ts +5 -3
- package/dist/core/orphans.js +5 -4
- package/dist/eval.d.ts +50 -11
- package/dist/eval.js +371 -152
- package/dist/exclude.d.ts +3 -5
- package/dist/exclude.js +3 -3
- package/dist/scan.d.ts +3 -3
- package/dist/scan.js +8 -7
- package/dist/test-coverage.d.ts +20 -3
- package/dist/test-coverage.js +40 -14
- package/package.json +3 -2
package/dist/eval.js
CHANGED
|
@@ -87,6 +87,7 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
|
|
|
87
87
|
const tool_intercept_js_1 = require("./tool-intercept.js");
|
|
88
88
|
const tool_stub_js_1 = require("./tool-stub.js");
|
|
89
89
|
const tmp_root_js_1 = require("./core/tmp-root.js");
|
|
90
|
+
const edit_distance_js_1 = require("./core/edit-distance.js");
|
|
90
91
|
function writeFiles(cwd, files) {
|
|
91
92
|
for (const [p, content] of Object.entries(files)) {
|
|
92
93
|
const full = (0, node_path_1.resolve)(cwd, p);
|
|
@@ -315,8 +316,12 @@ async function runEval(spec) {
|
|
|
315
316
|
* `runner` so the orchestration is unit-testable without a model.
|
|
316
317
|
*/
|
|
317
318
|
async function measureWith(spec, runner) {
|
|
318
|
-
|
|
319
|
-
|
|
319
|
+
assertKnownKeys(spec, MEASURE_SPEC_KEYS, {
|
|
320
|
+
caller: "measure",
|
|
321
|
+
type: "MeasureSpec",
|
|
322
|
+
});
|
|
323
|
+
if (spec.stubSkillBodies && !spec.pluginDir && !spec.skillsDir)
|
|
324
|
+
throw new Error("measure: `stubSkillBodies` requires `pluginDir` or `skillsDir`.");
|
|
320
325
|
// stubSkillBodies replaces each skill BODY with a no-op (the run stops at
|
|
321
326
|
// selection), so there is no output to grade — a `judged` check would score an
|
|
322
327
|
// empty body and mislead. The docs warn against this pairing; enforce it.
|
|
@@ -325,51 +330,49 @@ async function measureWith(spec, runner) {
|
|
|
325
330
|
throw new Error("measure: `stubSkillBodies` is for firing/`skill()` checks only — it stubs " +
|
|
326
331
|
"the skill body, so there's no output for a `judged` check to grade. Drop " +
|
|
327
332
|
"`stubSkillBodies`, or remove the `judged` check.");
|
|
328
|
-
const
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
}
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
(0, node_fs_1.rmSync)(stubbed, { recursive: true, force: true });
|
|
372
|
-
}
|
|
333
|
+
const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
|
|
334
|
+
// The install source (plugin / pluginDir / skillsDir, stubbed or not) is
|
|
335
|
+
// resolved by runEvalWith, once per arm — measure is one arm, so it just
|
|
336
|
+
// forwards the fields and reads the arm's report back.
|
|
337
|
+
const arm = {
|
|
338
|
+
settings: spec.settings,
|
|
339
|
+
plugin: spec.plugin,
|
|
340
|
+
pluginDir: spec.pluginDir,
|
|
341
|
+
skillsDir: spec.skillsDir,
|
|
342
|
+
interceptTools: spec.interceptTools,
|
|
343
|
+
};
|
|
344
|
+
const report = await runEvalWith({
|
|
345
|
+
fixture: spec.fixture,
|
|
346
|
+
arms: { run: arm },
|
|
347
|
+
stubSkillBodies: spec.stubSkillBodies,
|
|
348
|
+
task: spec.task,
|
|
349
|
+
trials: spec.trials ?? 5,
|
|
350
|
+
model: spec.model ?? "sonnet",
|
|
351
|
+
effort: spec.effort,
|
|
352
|
+
allowedTools: spec.allowedTools,
|
|
353
|
+
timeoutMs: spec.timeoutMs,
|
|
354
|
+
spacingSec: spec.spacingSec,
|
|
355
|
+
measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
|
|
356
|
+
}, runner);
|
|
357
|
+
return checkReportOf(report.arms.run, keyed, installNamespace(arm));
|
|
358
|
+
}
|
|
359
|
+
/** Read one arm's {@link ArmReport} back into a {@link CheckReport}. */
|
|
360
|
+
function checkReportOf(arm, keyed, namespace) {
|
|
361
|
+
return {
|
|
362
|
+
n: arm?.runs ?? 0,
|
|
363
|
+
perCheck: keyed.map(([k, c]) => {
|
|
364
|
+
const s = arm?.stats[k];
|
|
365
|
+
return {
|
|
366
|
+
check: c.toJSON(),
|
|
367
|
+
rate: s?.mean ?? 0,
|
|
368
|
+
se: s?.se ?? 0,
|
|
369
|
+
passK: s?.passK ?? 0,
|
|
370
|
+
n: s?.n ?? 0,
|
|
371
|
+
};
|
|
372
|
+
}),
|
|
373
|
+
usage: arm?.usage ?? aggregateUsage([]),
|
|
374
|
+
namespace,
|
|
375
|
+
};
|
|
373
376
|
}
|
|
374
377
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
|
|
375
378
|
/** Score a check vocabulary across trials against the real `claude` CLI. */
|
|
@@ -380,67 +383,36 @@ async function measure(spec) {
|
|
|
380
383
|
}
|
|
381
384
|
/** Score checks across arms (injectable runner). Reuses `runEvalWith`. */
|
|
382
385
|
async function measureArmsWith(spec, runner) {
|
|
386
|
+
assertKnownKeys(spec, ARMS_MEASURE_SPEC_KEYS, {
|
|
387
|
+
caller: "measureArms",
|
|
388
|
+
type: "ArmsMeasureSpec",
|
|
389
|
+
});
|
|
390
|
+
for (const [name, arm] of Object.entries(spec.arms))
|
|
391
|
+
assertKnownKeys(arm, EVAL_ARM_KEYS, {
|
|
392
|
+
caller: "measureArms",
|
|
393
|
+
type: "EvalArm",
|
|
394
|
+
arm: name,
|
|
395
|
+
});
|
|
383
396
|
const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
:
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
for (const [armName, arm] of Object.entries(report.arms)) {
|
|
402
|
-
arms[armName] = {
|
|
403
|
-
n: arm.runs,
|
|
404
|
-
perCheck: keyed.map(([k, c]) => {
|
|
405
|
-
const s = arm.stats[k];
|
|
406
|
-
return {
|
|
407
|
-
check: c.toJSON(),
|
|
408
|
-
rate: s?.mean ?? 0,
|
|
409
|
-
se: s?.se ?? 0,
|
|
410
|
-
passK: s?.passK ?? 0,
|
|
411
|
-
n: s?.n ?? 0,
|
|
412
|
-
};
|
|
413
|
-
}),
|
|
414
|
-
usage: arm.usage,
|
|
415
|
-
};
|
|
416
|
-
}
|
|
417
|
-
return { arms };
|
|
418
|
-
}
|
|
419
|
-
finally {
|
|
420
|
-
for (const t of temps)
|
|
421
|
-
(0, node_fs_1.rmSync)(t, { recursive: true, force: true });
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
/**
|
|
425
|
-
* Repackage every arm that sets a `pluginDir` with its skill bodies stubbed
|
|
426
|
-
* (frontmatter kept), for an A/B firing comparison. Returns the rewritten arms
|
|
427
|
-
* plus the throwaway dirs the caller must remove. Arms without a `pluginDir` pass
|
|
428
|
-
* through unchanged. See {@link stubbedPluginDir}.
|
|
429
|
-
*/
|
|
430
|
-
function stubArmPluginDirs(arms) {
|
|
431
|
-
const out = {};
|
|
432
|
-
const temps = [];
|
|
433
|
-
for (const [name, arm] of Object.entries(arms)) {
|
|
434
|
-
if (arm.pluginDir) {
|
|
435
|
-
const stubbed = stubbedPluginDir(arm.pluginDir);
|
|
436
|
-
temps.push(stubbed);
|
|
437
|
-
out[name] = { ...arm, pluginDir: stubbed };
|
|
438
|
-
}
|
|
439
|
-
else {
|
|
440
|
-
out[name] = arm;
|
|
441
|
-
}
|
|
397
|
+
// Per-arm install sources (and the stub) are resolved by runEvalWith.
|
|
398
|
+
const report = await runEvalWith({
|
|
399
|
+
fixture: spec.fixture,
|
|
400
|
+
arms: spec.arms,
|
|
401
|
+
stubSkillBodies: spec.stubSkillBodies,
|
|
402
|
+
task: spec.task,
|
|
403
|
+
trials: spec.trials ?? 5,
|
|
404
|
+
model: spec.model ?? "sonnet",
|
|
405
|
+
effort: spec.effort,
|
|
406
|
+
allowedTools: spec.allowedTools,
|
|
407
|
+
timeoutMs: spec.timeoutMs,
|
|
408
|
+
spacingSec: spec.spacingSec,
|
|
409
|
+
measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
|
|
410
|
+
}, runner);
|
|
411
|
+
const arms = {};
|
|
412
|
+
for (const [armName, arm] of Object.entries(report.arms)) {
|
|
413
|
+
arms[armName] = checkReportOf(arm, keyed, installNamespace(spec.arms[armName] ?? {}));
|
|
442
414
|
}
|
|
443
|
-
return { arms
|
|
415
|
+
return { arms };
|
|
444
416
|
}
|
|
445
417
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
|
|
446
418
|
/** Score checks across arms against the real `claude` CLI. */
|
|
@@ -482,6 +454,28 @@ function formatCheckReport(report) {
|
|
|
482
454
|
lines.push(` ${(c.rate * 100).toFixed(0)}% ± ${(c.se * 100).toFixed(0)}% ${checkLabel(c.check)}` +
|
|
483
455
|
` (pass^k ${String(c.passK)})`);
|
|
484
456
|
}
|
|
457
|
+
// The `skill()` twin of the trigger-rate TOTAL-zero note (see
|
|
458
|
+
// formatTriggerRateReport): every skill() check at 0% over real runs is far
|
|
459
|
+
// more often a wiring mistake — a bare id where the namespaced one is matched,
|
|
460
|
+
// `pluginDir` handed a loose dir, an empty cwd — than a finding, and it reads
|
|
461
|
+
// exactly like a finding. Only when EVERY skill check is at zero: a partial rate
|
|
462
|
+
// is a real measurement, and a note that hedges on good data gets ignored.
|
|
463
|
+
const skillChecks = report.perCheck.filter((c) => c.check.kind === "skill");
|
|
464
|
+
if (report.n > 0 &&
|
|
465
|
+
skillChecks.length > 0 &&
|
|
466
|
+
skillChecks.every((c) => c.rate === 0))
|
|
467
|
+
lines.push("⚠ nothing resolved for ANY skill() check. That is usually SETUP, not the skill — check, in order:\n" +
|
|
468
|
+
(report.namespace !== undefined
|
|
469
|
+
? " 1. the id in `skill()` — your skills installed under " +
|
|
470
|
+
`\`${report.namespace}\`, so \`skill()\` matches ` +
|
|
471
|
+
`\`${report.namespace}:<skill>\`; a bare name silently never matches;\n`
|
|
472
|
+
: " 1. the id in `skill()` — it matches the NAMESPACED id " +
|
|
473
|
+
"(`<plugin>:<skill>`); a bare name silently never matches;\n") +
|
|
474
|
+
" 2. the install field — a loose `.claude/skills` dir needs `skillsDir`, " +
|
|
475
|
+
"not `pluginDir` (which wants a full plugin manifest);\n" +
|
|
476
|
+
" 3. the `fixture` — a run starts in an EMPTY cwd, so a prompt about a " +
|
|
477
|
+
"file that does not exist is one the model is right to decline.\n" +
|
|
478
|
+
" Rule out all three before recording this as a fact about the skill.");
|
|
485
479
|
return lines.join("\n");
|
|
486
480
|
}
|
|
487
481
|
/**
|
|
@@ -1317,10 +1311,175 @@ function evalArmsInputs(spec, cfg) {
|
|
|
1317
1311
|
ephemeralEnv: spec.ephemeralEnv === true,
|
|
1318
1312
|
};
|
|
1319
1313
|
}
|
|
1320
|
-
|
|
1314
|
+
// ---------------------------------------------------------------------------
|
|
1315
|
+
// The spec boundary — refuse a field the spec does not declare.
|
|
1316
|
+
//
|
|
1317
|
+
// A spec usually arrives from a plain-JS `*.eval.mjs` file, where nothing
|
|
1318
|
+
// type-checks the object literal, so a key that is misspelled (`skillDir`) or
|
|
1319
|
+
// belongs to a sibling runner (`prompts` on measure) used to be dropped without a
|
|
1320
|
+
// word — and the run then reported a confident number about a setup the author
|
|
1321
|
+
// never asked for (issue #307). The precedent is `driverMisplaced` in
|
|
1322
|
+
// eval-entry.ts: a field that would silently do nothing is REFUSED, not ignored.
|
|
1323
|
+
//
|
|
1324
|
+
// Each key table is `satisfies Record<keyof Spec, true>`, so the compiler fails
|
|
1325
|
+
// when a spec gains or loses a field the table does not: the list cannot drift.
|
|
1326
|
+
// ---------------------------------------------------------------------------
|
|
1327
|
+
const EVAL_ARM_KEYS = {
|
|
1328
|
+
files: true,
|
|
1329
|
+
settings: true,
|
|
1330
|
+
plugin: true,
|
|
1331
|
+
pluginDir: true,
|
|
1332
|
+
skillsDir: true,
|
|
1333
|
+
interceptTools: true,
|
|
1334
|
+
model: true,
|
|
1335
|
+
effort: true,
|
|
1336
|
+
};
|
|
1337
|
+
const EVAL_SPEC_KEYS = {
|
|
1338
|
+
name: true,
|
|
1339
|
+
fixture: true,
|
|
1340
|
+
arms: true,
|
|
1341
|
+
stubSkillBodies: true,
|
|
1342
|
+
task: true,
|
|
1343
|
+
measure: true,
|
|
1344
|
+
trials: true,
|
|
1345
|
+
model: true,
|
|
1346
|
+
effort: true,
|
|
1347
|
+
allowedTools: true,
|
|
1348
|
+
timeoutMs: true,
|
|
1349
|
+
spacingSec: true,
|
|
1350
|
+
cache: true,
|
|
1351
|
+
cacheDir: true,
|
|
1352
|
+
concurrency: true,
|
|
1353
|
+
maxCostUsd: true,
|
|
1354
|
+
rateLimitRetries: true,
|
|
1355
|
+
retryBackoffMs: true,
|
|
1356
|
+
ephemeralEnv: true,
|
|
1357
|
+
stubs: true,
|
|
1358
|
+
lock: true,
|
|
1359
|
+
};
|
|
1360
|
+
const MEASURE_SPEC_KEYS = {
|
|
1361
|
+
fixture: true,
|
|
1362
|
+
settings: true,
|
|
1363
|
+
plugin: true,
|
|
1364
|
+
pluginDir: true,
|
|
1365
|
+
skillsDir: true,
|
|
1366
|
+
stubSkillBodies: true,
|
|
1367
|
+
interceptTools: true,
|
|
1368
|
+
task: true,
|
|
1369
|
+
checks: true,
|
|
1370
|
+
trials: true,
|
|
1371
|
+
model: true,
|
|
1372
|
+
effort: true,
|
|
1373
|
+
allowedTools: true,
|
|
1374
|
+
timeoutMs: true,
|
|
1375
|
+
spacingSec: true,
|
|
1376
|
+
};
|
|
1377
|
+
const ARMS_MEASURE_SPEC_KEYS = {
|
|
1378
|
+
fixture: true,
|
|
1379
|
+
arms: true,
|
|
1380
|
+
task: true,
|
|
1381
|
+
checks: true,
|
|
1382
|
+
stubSkillBodies: true,
|
|
1383
|
+
trials: true,
|
|
1384
|
+
model: true,
|
|
1385
|
+
allowedTools: true,
|
|
1386
|
+
timeoutMs: true,
|
|
1387
|
+
effort: true,
|
|
1388
|
+
spacingSec: true,
|
|
1389
|
+
};
|
|
1390
|
+
const TRIGGER_RATE_SPEC_KEYS = {
|
|
1391
|
+
name: true,
|
|
1392
|
+
lock: true,
|
|
1393
|
+
pluginDir: true,
|
|
1394
|
+
skillsDir: true,
|
|
1395
|
+
prompts: true,
|
|
1396
|
+
irrelevantPrompts: true,
|
|
1397
|
+
fired: true,
|
|
1398
|
+
installSet: true,
|
|
1399
|
+
stubSkillBodies: true,
|
|
1400
|
+
minPrompts: true,
|
|
1401
|
+
minDistance: true,
|
|
1402
|
+
trials: true,
|
|
1403
|
+
model: true,
|
|
1404
|
+
effort: true,
|
|
1405
|
+
minModel: true,
|
|
1406
|
+
allowedTools: true,
|
|
1407
|
+
timeoutMs: true,
|
|
1408
|
+
spacingSec: true,
|
|
1409
|
+
fixture: true,
|
|
1410
|
+
concurrency: true,
|
|
1411
|
+
};
|
|
1412
|
+
/**
|
|
1413
|
+
* The closest declared field to an unknown one, or undefined when nothing is
|
|
1414
|
+
* close — a wrong suggestion is worse than none (it invites "fixing" a field the
|
|
1415
|
+
* author never meant). Same tight threshold as the CLI's unknown-flag hint.
|
|
1416
|
+
*/
|
|
1417
|
+
function nearestField(unknown, known) {
|
|
1418
|
+
let best;
|
|
1419
|
+
for (const name of known) {
|
|
1420
|
+
const d = (0, edit_distance_js_1.editDistance)(unknown.toLowerCase(), name.toLowerCase());
|
|
1421
|
+
if (!best || d < best.d)
|
|
1422
|
+
best = { name, d };
|
|
1423
|
+
}
|
|
1424
|
+
return best && best.d <= Math.max(2, Math.floor(unknown.length / 4))
|
|
1425
|
+
? best.name
|
|
1426
|
+
: undefined;
|
|
1427
|
+
}
|
|
1428
|
+
/**
|
|
1429
|
+
* Throw when `spec` carries a field outside `known` — every unknown one named,
|
|
1430
|
+
* with a did-you-mean where a declared field is one typo away. `at` says which
|
|
1431
|
+
* runner and spec type the message is about (`arm` labels a nested arm). Runs
|
|
1432
|
+
* before anything is packaged or spent.
|
|
1433
|
+
*/
|
|
1434
|
+
function assertKnownKeys(spec, known, at) {
|
|
1435
|
+
const declared = Object.keys(known);
|
|
1436
|
+
const where = at.arm === undefined ? "" : ` on arm "${at.arm}"`;
|
|
1437
|
+
const problems = Object.keys(spec)
|
|
1438
|
+
.filter((key) => !Object.hasOwn(known, key))
|
|
1439
|
+
.map((key) => {
|
|
1440
|
+
const near = nearestField(key, declared);
|
|
1441
|
+
const hint = near === undefined ? "" : ` — did you mean \`${near}\`?`;
|
|
1442
|
+
return `${at.caller}: unknown ${at.type} field "${key}"${where}${hint}`;
|
|
1443
|
+
});
|
|
1444
|
+
if (problems.length === 0)
|
|
1445
|
+
return;
|
|
1446
|
+
throw new Error(`${problems.join("\n")}\n ${at.type} fields: ${declared.join(", ")}.\n` +
|
|
1447
|
+
" An unknown field is refused rather than ignored: a dropped field would " +
|
|
1448
|
+
"make the run measure a setup you did not ask for.");
|
|
1449
|
+
}
|
|
1450
|
+
async function runEvalWith(input, runner) {
|
|
1451
|
+
// Refuse a stray field BEFORE spending a token: an eval file is plain JS, so a
|
|
1452
|
+
// typo'd or misplaced key would otherwise vanish and the run would report a
|
|
1453
|
+
// confident number about the wrong setup (issue #307).
|
|
1454
|
+
assertKnownKeys(input, EVAL_SPEC_KEYS, {
|
|
1455
|
+
caller: "runEval",
|
|
1456
|
+
type: "EvalSpec",
|
|
1457
|
+
});
|
|
1458
|
+
for (const [name, arm] of Object.entries(input.arms))
|
|
1459
|
+
assertKnownKeys(arm, EVAL_ARM_KEYS, {
|
|
1460
|
+
caller: "runEval",
|
|
1461
|
+
type: "EvalArm",
|
|
1462
|
+
arm: name,
|
|
1463
|
+
});
|
|
1321
1464
|
// Tell the CLI runner this script exercised the harness, so a file that runs
|
|
1322
1465
|
// NOTHING can be told apart from one that ran and passed. See check-count.ts.
|
|
1323
1466
|
(0, check_count_js_1.recordCheck)();
|
|
1467
|
+
// Every arm's install source (`pluginDir` as-is / stubbed, or a loose
|
|
1468
|
+
// `skillsDir` packaged into a throwaway plugin) is resolved HERE, once — the
|
|
1469
|
+
// one place that decides what `--plugin-dir` receives, so measure / measureArms
|
|
1470
|
+
// / runEval cannot disagree about it. The throwaways are removed afterward.
|
|
1471
|
+
const { arms: resolvedArms, packaged } = resolveArmInstalls(input.arms, input.stubSkillBodies ?? false);
|
|
1472
|
+
const spec = { ...input, arms: resolvedArms };
|
|
1473
|
+
try {
|
|
1474
|
+
return await runResolvedEval(spec, runner);
|
|
1475
|
+
}
|
|
1476
|
+
finally {
|
|
1477
|
+
for (const dir of packaged)
|
|
1478
|
+
(0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
|
|
1479
|
+
}
|
|
1480
|
+
}
|
|
1481
|
+
/** {@link runEvalWith} after its arms' install sources are concrete plugin dirs. */
|
|
1482
|
+
async function runResolvedEval(spec, runner) {
|
|
1324
1483
|
const trials = spec.trials ?? 5;
|
|
1325
1484
|
const spacing = (spec.spacingSec ?? 4) * 1000;
|
|
1326
1485
|
const concurrency = spec.concurrency ?? 1;
|
|
@@ -1493,7 +1652,7 @@ function packageSkillsDir(skillsDir, opts = {}) {
|
|
|
1493
1652
|
throw new Error(`skillsDir not found: ${skillsDir} (resolved ${abs})`);
|
|
1494
1653
|
const root = (0, tmp_root_js_1.makeTmpDir)("skills");
|
|
1495
1654
|
(0, node_fs_1.mkdirSync)((0, node_path_1.join)(root, ".claude-plugin"), { recursive: true });
|
|
1496
|
-
(0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ??
|
|
1655
|
+
(0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? LOOSE_SKILLS_NAMESPACE, version: "0.0.0" }, null, 2));
|
|
1497
1656
|
const skillsOut = (0, node_path_1.join)(root, "skills");
|
|
1498
1657
|
(0, node_fs_1.mkdirSync)(skillsOut, { recursive: true });
|
|
1499
1658
|
let copied = 0;
|
|
@@ -1704,16 +1863,96 @@ function packageInstallSet(opts) {
|
|
|
1704
1863
|
throw e;
|
|
1705
1864
|
}
|
|
1706
1865
|
}
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1866
|
+
// ---------------------------------------------------------------------------
|
|
1867
|
+
// The install source — the ONE place that decides what `--plugin-dir` receives.
|
|
1868
|
+
//
|
|
1869
|
+
// Every runner that installs skills (runEval / measure / measureArms via an
|
|
1870
|
+
// EvalArm, measureTriggerRate via its spec) accepts the same two fields —
|
|
1871
|
+
// `pluginDir` (a complete plugin, used as-is or re-packaged with stubbed bodies)
|
|
1872
|
+
// and `skillsDir` (a loose `<name>/SKILL.md` dir, packaged into a throwaway
|
|
1873
|
+
// plugin) — and they all resolve through `resolveSkillInstall`. Before this
|
|
1874
|
+
// existed each runner re-derived the rule for itself, and the one written last
|
|
1875
|
+
// was the only one that knew about loose dirs (issue #307).
|
|
1876
|
+
// ---------------------------------------------------------------------------
|
|
1877
|
+
/** The plugin name a packaged loose skills dir installs under. */
|
|
1878
|
+
const LOOSE_SKILLS_NAMESPACE = "vigiles-loose-skills";
|
|
1879
|
+
/**
|
|
1880
|
+
* The plugin namespace a source's skills install under — the `<plugin>` half of
|
|
1881
|
+
* the `<plugin>:<skill>` id `skill()` / `skillResolved` match. A loose dir gets
|
|
1882
|
+
* the packager's name; a plugin its declared name, falling back to the same
|
|
1883
|
+
* synthetic one when its manifest has none. Pure (reads the manifest only).
|
|
1884
|
+
*/
|
|
1885
|
+
function installNamespace(src) {
|
|
1886
|
+
if (src.skillsDir)
|
|
1887
|
+
return LOOSE_SKILLS_NAMESPACE;
|
|
1888
|
+
if (src.pluginDir)
|
|
1889
|
+
return pluginName(src.pluginDir) ?? LOOSE_SKILLS_NAMESPACE;
|
|
1890
|
+
return undefined;
|
|
1891
|
+
}
|
|
1892
|
+
/**
|
|
1893
|
+
* Resolve a {@link SkillSource} to a concrete plugin dir. `stub` strips skill
|
|
1894
|
+
* bodies (frontmatter kept) into a throwaway; `installSet` merges competitor
|
|
1895
|
+
* skills in (the whole-harness tier, `measureTriggerRate` only). Exactly one of
|
|
1896
|
+
* `pluginDir` / `skillsDir` may be set; neither resolves to nothing installed —
|
|
1897
|
+
* the caller decides whether that is legal (an arm: yes; a trigger run: no).
|
|
1898
|
+
*/
|
|
1899
|
+
function resolveSkillInstall(src, opts) {
|
|
1900
|
+
if (src.pluginDir && src.skillsDir)
|
|
1901
|
+
throw new Error(`${opts.caller}: set \`pluginDir\` OR \`skillsDir\`, not both.`);
|
|
1902
|
+
const namespace = installNamespace(src);
|
|
1903
|
+
if (namespace === undefined)
|
|
1904
|
+
return {};
|
|
1905
|
+
const installSet = opts.installSet ?? [];
|
|
1906
|
+
if (installSet.length > 0) {
|
|
1907
|
+
// Whole-harness tier: merge the under-test skills with the install set so
|
|
1908
|
+
// selection is competitive (the realistic, differentiated measurement).
|
|
1909
|
+
const { dir } = packageInstallSet({
|
|
1910
|
+
underTestSrc: src.skillsDir ?? skillsDirOf(src.pluginDir),
|
|
1911
|
+
name: namespace,
|
|
1912
|
+
installSet,
|
|
1913
|
+
stub: opts.stub,
|
|
1914
|
+
});
|
|
1915
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1916
|
+
}
|
|
1917
|
+
if (src.skillsDir) {
|
|
1918
|
+
const dir = packageSkillsDir(src.skillsDir, { stub: opts.stub });
|
|
1919
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1920
|
+
}
|
|
1921
|
+
// A real plugin: as-is, or re-packaged from its skills/ with bodies stripped —
|
|
1922
|
+
// keeping the original plugin NAME so `<name>:<skill>` still matches.
|
|
1923
|
+
const pluginDir = src.pluginDir;
|
|
1924
|
+
if (!opts.stub)
|
|
1925
|
+
return { pluginDir, namespace };
|
|
1926
|
+
const dir = stubbedPluginDir(pluginDir);
|
|
1927
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1928
|
+
}
|
|
1929
|
+
/**
|
|
1930
|
+
* Resolve every arm's install source (see {@link resolveSkillInstall}) so each
|
|
1931
|
+
* arm carries only a concrete `pluginDir` — `skillsDir` is consumed here. Returns
|
|
1932
|
+
* the rewritten arms plus the throwaway dirs the caller removes afterward. A
|
|
1933
|
+
* failure part-way removes what was already built (no leaked temp dirs).
|
|
1934
|
+
*/
|
|
1935
|
+
function resolveArmInstalls(arms, stub) {
|
|
1936
|
+
const out = {};
|
|
1937
|
+
const packaged = [];
|
|
1938
|
+
try {
|
|
1939
|
+
for (const [name, arm] of Object.entries(arms)) {
|
|
1940
|
+
const { skillsDir: _consumed, ...rest } = arm;
|
|
1941
|
+
const r = resolveSkillInstall(arm, {
|
|
1942
|
+
caller: `eval arm "${name}"`,
|
|
1943
|
+
stub,
|
|
1944
|
+
});
|
|
1945
|
+
if (r.packaged)
|
|
1946
|
+
packaged.push(r.packaged);
|
|
1947
|
+
out[name] = { ...rest, pluginDir: r.pluginDir };
|
|
1948
|
+
}
|
|
1949
|
+
}
|
|
1950
|
+
catch (e) {
|
|
1951
|
+
for (const dir of packaged)
|
|
1952
|
+
(0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
|
|
1953
|
+
throw e;
|
|
1954
|
+
}
|
|
1955
|
+
return { arms: out, packaged };
|
|
1717
1956
|
}
|
|
1718
1957
|
/** Number of `<name>/SKILL.md` skills installed in a plugin — the selection pool. */
|
|
1719
1958
|
function countSkills(pluginDir) {
|
|
@@ -1727,46 +1966,22 @@ function countSkills(pluginDir) {
|
|
|
1727
1966
|
return n;
|
|
1728
1967
|
}
|
|
1729
1968
|
function resolveTriggerPluginDir(spec) {
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
if (installSet.length > 0) {
|
|
1737
|
-
// Whole-harness tier: merge the under-test skills with the install set so
|
|
1738
|
-
// selection is competitive (the realistic, differentiated measurement).
|
|
1739
|
-
const { src, name } = underTestSource(spec);
|
|
1740
|
-
({ dir: pluginDir } = packageInstallSet({
|
|
1741
|
-
underTestSrc: src,
|
|
1742
|
-
name,
|
|
1743
|
-
installSet,
|
|
1744
|
-
stub,
|
|
1745
|
-
}));
|
|
1746
|
-
packaged = pluginDir;
|
|
1747
|
-
}
|
|
1748
|
-
else if (spec.skillsDir) {
|
|
1749
|
-
packaged = packageSkillsDir(spec.skillsDir, { stub });
|
|
1750
|
-
pluginDir = packaged;
|
|
1751
|
-
}
|
|
1752
|
-
else if (spec.pluginDir) {
|
|
1753
|
-
// Stub a real plugin: build a minimal plugin from its skills/ with bodies
|
|
1754
|
-
// stripped — keep the original plugin NAME so `<name>:<skill>` still matches.
|
|
1755
|
-
packaged = stub ? stubbedPluginDir(spec.pluginDir) : undefined;
|
|
1756
|
-
pluginDir = packaged ?? spec.pluginDir;
|
|
1757
|
-
}
|
|
1758
|
-
else {
|
|
1969
|
+
const r = resolveSkillInstall(spec, {
|
|
1970
|
+
caller: "measureTriggerRate",
|
|
1971
|
+
stub: spec.stubSkillBodies ?? true, // trigger = frontmatter; body never needed
|
|
1972
|
+
installSet: spec.installSet,
|
|
1973
|
+
});
|
|
1974
|
+
if (r.pluginDir === undefined || r.namespace === undefined)
|
|
1759
1975
|
throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
|
|
1760
|
-
}
|
|
1761
1976
|
// `competitors` is the REAL selection pressure: every OTHER skill installed in
|
|
1762
1977
|
// the resolved plugin (siblings already in the source + any installSet), not
|
|
1763
1978
|
// just the installSet delta — so a multi-skill plugin is never mislabeled
|
|
1764
1979
|
// "isolated". `max(0, …)` guards a 0-skill pool.
|
|
1765
1980
|
return {
|
|
1766
|
-
pluginDir,
|
|
1767
|
-
packaged,
|
|
1768
|
-
competitors: Math.max(0, countSkills(pluginDir) - 1),
|
|
1769
|
-
namespace:
|
|
1981
|
+
pluginDir: r.pluginDir,
|
|
1982
|
+
packaged: r.packaged,
|
|
1983
|
+
competitors: Math.max(0, countSkills(r.pluginDir) - 1),
|
|
1984
|
+
namespace: r.namespace,
|
|
1770
1985
|
};
|
|
1771
1986
|
}
|
|
1772
1987
|
/** Run one prompt set × trials through `runner`, aggregating fired counts. */
|
|
@@ -1859,6 +2074,10 @@ function assertTriggerDiversity(spec) {
|
|
|
1859
2074
|
}
|
|
1860
2075
|
}
|
|
1861
2076
|
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
|
|
2077
|
+
assertKnownKeys(spec, TRIGGER_RATE_SPEC_KEYS, {
|
|
2078
|
+
caller: "measureTriggerRate",
|
|
2079
|
+
type: "TriggerRateSpec",
|
|
2080
|
+
});
|
|
1862
2081
|
// Tell the CLI runner this script exercised the harness (see check-count.ts).
|
|
1863
2082
|
(0, check_count_js_1.recordCheck)();
|
|
1864
2083
|
// Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
|
package/dist/exclude.d.ts
CHANGED
|
@@ -8,12 +8,10 @@ export interface ExcludeSet {
|
|
|
8
8
|
/** The user's patterns, as written (for messages). */
|
|
9
9
|
readonly patterns: readonly string[];
|
|
10
10
|
/**
|
|
11
|
-
* The
|
|
12
|
-
*
|
|
13
|
-
*
|
|
11
|
+
* The function face for `globSync`, correct whatever the glob's `cwd` is. A
|
|
12
|
+
* bare directory name excludes its subtree (`bench` → `bench`, `bench/**`),
|
|
13
|
+
* which is what "tsconfig-style" promises.
|
|
14
14
|
*/
|
|
15
|
-
readonly ignore: readonly string[];
|
|
16
|
-
/** The function face for `globSync`, correct whatever the glob's `cwd` is. */
|
|
17
15
|
readonly globIgnore: IgnoreLike;
|
|
18
16
|
/** Is this root-relative path excluded (floor or user pattern)? */
|
|
19
17
|
matches(rel: string): boolean;
|
package/dist/exclude.js
CHANGED
|
@@ -4,8 +4,9 @@ exports.excludesNothing = exports.EXCLUDE_FLOOR = void 0;
|
|
|
4
4
|
exports.excludeSet = excludeSet;
|
|
5
5
|
exports.excludedBy = excludedBy;
|
|
6
6
|
/**
|
|
7
|
-
* The ONE exclusion policy for every walk that polices the user's repository (#192) — the parsed `.vigilesrc.json#exclude` as an `ExcludeSet`, built ONCE where `loadConfig()` runs and taken as a REQUIRED parameter by every in-scope discovery (`findSpecs`, `findInstructionFiles`, `discoverNestedBundles`, `collectDocumentedRules`, `gatherInstructionFiles` in cli.ts; the
|
|
8
|
-
*
|
|
7
|
+
* The ONE exclusion policy for every walk that polices the user's repository (#192) — the parsed `.vigilesrc.json#exclude` as an `ExcludeSet`, built ONCE where `loadConfig()` runs and taken as a REQUIRED parameter by every in-scope discovery (`findSpecs`, `findInstructionFiles`, `discoverNestedBundles`, `collectDocumentedRules`, `gatherInstructionFiles` in cli.ts; the `globIgnore` face handed to `findDocRefs`, `findOrphanDocs` (`repoExclude`), `discoverScripts`, `computeScriptCoverage`; the whole set to `findUntestedSurfaces`/`skillTestNudge`/`scanPlugin`).
|
|
8
|
+
* Faces: `globIgnore` (an `IgnoreLike` keyed on the path's position relative to the REPO root, so a glob rooted anywhere — `vigiles lint some/dir`, a nested bundle — still applies a root-relative exclude), `matches`/`explain` (a root-relative path), and `excludedBy` below (an absolute path).
|
|
9
|
+
* 🔴 There is deliberately NO string-list face any more (#281). It was `ignore`, correct only for a glob rooted AT `root` — a precondition that lived in a comment, and two callers that globbed from a nested bundle broke it: the repo exclude never reached the bundle, and a root-relative pattern aliased into it. `core/glob-ignore.ts` unions a detector's own string floor with `globIgnore` instead.
|
|
9
10
|
* A bare directory name excludes its subtree, as tsconfig/ESLint do — measured 2026-09-03: glob's own string `ignore` treated `bench` and `bench/` as matching NOTHING while the minimatch helper in `discoverNestedBundles` accepted them, so the two walks that honoured `exclude` disagreed.
|
|
10
11
|
* The floor (node_modules/dist/.git/.vigiles) lives here, not per walk.
|
|
11
12
|
* `exclude` filters DISCOVERY only: an explicitly named path is processed and ONE line names the pattern it matched (rg/tsc semantics with ESLint's loudness; never prettier's silent 'all clean').
|
|
@@ -57,7 +58,6 @@ function excludeSet(root, patterns) {
|
|
|
57
58
|
return {
|
|
58
59
|
root,
|
|
59
60
|
patterns: user,
|
|
60
|
-
ignore: [...exports.EXCLUDE_FLOOR, ...user.flatMap((p) => [p, `${p}/**`])],
|
|
61
61
|
globIgnore: {
|
|
62
62
|
ignored: (p) => matches(relOf(p)),
|
|
63
63
|
childrenIgnored: (p) => matches(relOf(p)),
|
package/dist/scan.d.ts
CHANGED
|
@@ -423,9 +423,9 @@ export declare function scanPlugin(dir: string, layout: PluginLayout, dialect: H
|
|
|
423
423
|
* would be twenty-odd mechanical edits for one behavioural change.
|
|
424
424
|
*
|
|
425
425
|
* ⚠️ Omitting it is NOT "the repo excludes nothing" — it is "this caller has
|
|
426
|
-
* no ExcludeSet to give", and the walk then reads everything.
|
|
427
|
-
* `
|
|
428
|
-
*
|
|
426
|
+
* no ExcludeSet to give", and the walk then reads everything. `audit` and
|
|
427
|
+
* every `lint` rule checker supply one (the checkers through the per-bundle
|
|
428
|
+
* context `overBundles` builds, so a new checker cannot forget it).
|
|
429
429
|
*/
|
|
430
430
|
excludes?: ExcludeSet;
|
|
431
431
|
/**
|