vigiles 31.0.1 → 32.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli-main.js +1 -1
- package/dist/eval.d.ts +50 -11
- package/dist/eval.js +371 -152
- package/package.json +1 -1
package/dist/cli-main.js
CHANGED
|
@@ -4871,7 +4871,7 @@ function resolveRecords(cwd, runs, tier, harnessFlag) {
|
|
|
4871
4871
|
* 1. `.claude-plugin/plugin.json#name` — the repo's own declared name, used
|
|
4872
4872
|
* when the run installed the repo AS a plugin (`pluginDir`).
|
|
4873
4873
|
* 2. `vigiles-loose-skills` — the synthetic name OUR OWN packaging gives a
|
|
4874
|
-
* loose `.claude/skills` dir (`packageSkillsDir`, and `
|
|
4874
|
+
* loose `.claude/skills` dir (`packageSkillsDir`, and `installNamespace`'s
|
|
4875
4875
|
* fallback when a plugin manifest has no name). A repo that is not a plugin
|
|
4876
4876
|
* still reports namespaced ids under it, so omitting it would drop every
|
|
4877
4877
|
* trigger-rate record for the documented one-liner.
|
package/dist/eval.d.ts
CHANGED
|
@@ -25,9 +25,21 @@ export interface EvalArm {
|
|
|
25
25
|
* arm, so its skills/commands/agents activate the real way — the real model
|
|
26
26
|
* can trigger a skill by its description (vs. `plugin`, which materializes a
|
|
27
27
|
* file subset that does not register skills). Point at a COMPLETE plugin. Lets
|
|
28
|
-
* an arm be "skill installed" vs "off" to measure real activation.
|
|
28
|
+
* an arm be "skill installed" vs "off" to measure real activation. Provide this
|
|
29
|
+
* OR {@link skillsDir}, not both.
|
|
29
30
|
*/
|
|
30
31
|
readonly pluginDir?: string;
|
|
32
|
+
/**
|
|
33
|
+
* A directory of LOOSE skills (`<skillsDir>/<name>/SKILL.md`, e.g. a repo's
|
|
34
|
+
* `.claude/skills`) to install for this arm. vigiles packages them into a
|
|
35
|
+
* throwaway `--plugin-dir` (manifest + `skills/<name>/`, each skill dir copied
|
|
36
|
+
* whole) for the run and removes it afterward — the one-liner for repo-local
|
|
37
|
+
* skills that aren't a published plugin. The skills install under the
|
|
38
|
+
* namespace `vigiles-loose-skills`, so a `skill()` check / `skillResolved`
|
|
39
|
+
* matches `vigiles-loose-skills:<name>` (the report's `namespace` says so).
|
|
40
|
+
* Provide this OR {@link pluginDir}, not both.
|
|
41
|
+
*/
|
|
42
|
+
readonly skillsDir?: string;
|
|
31
43
|
/**
|
|
32
44
|
* Tools to intercept for this arm (the tool-call spy). Each
|
|
33
45
|
* {@link ToolIntercept} is denied its real execution by an auto-wired PreToolUse
|
|
@@ -83,6 +95,15 @@ export interface EvalSpec<M extends Metrics> {
|
|
|
83
95
|
readonly fixture?: Record<string, string>;
|
|
84
96
|
/** The arms to compare, by name. */
|
|
85
97
|
readonly arms: Record<string, EvalArm>;
|
|
98
|
+
/**
|
|
99
|
+
* Stub each arm's skill BODIES (frontmatter kept) before the run — for firing
|
|
100
|
+
* comparisons (does the skill get SELECTED?), where a selected skill should
|
|
101
|
+
* stop at selection instead of running its (often expensive) procedure. Every
|
|
102
|
+
* arm with a `pluginDir` / `skillsDir` is repackaged with bodies stripped; arms
|
|
103
|
+
* without one are untouched. Don't combine with quality metrics: the body is
|
|
104
|
+
* gone, so there is nothing to grade. See {@link stubSkillBody}.
|
|
105
|
+
*/
|
|
106
|
+
readonly stubSkillBodies?: boolean;
|
|
86
107
|
/** The task prompt given to the agent. */
|
|
87
108
|
readonly task: string;
|
|
88
109
|
/** Compute this run's metrics from its outcome. */
|
|
@@ -328,15 +349,24 @@ export interface MeasureSpec {
|
|
|
328
349
|
readonly settings?: unknown;
|
|
329
350
|
/** A real plugin/repo to load (materialized) — see `EvalArm.plugin`. */
|
|
330
351
|
readonly plugin?: string;
|
|
331
|
-
/**
|
|
352
|
+
/**
|
|
353
|
+
* A complete plugin dir to install natively (`--plugin-dir`) so skills activate
|
|
354
|
+
* — see {@link EvalArm.pluginDir}. Provide this OR `skillsDir`, not both.
|
|
355
|
+
*/
|
|
332
356
|
readonly pluginDir?: string;
|
|
333
357
|
/**
|
|
334
|
-
*
|
|
335
|
-
*
|
|
336
|
-
*
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
358
|
+
* A loose `<skillsDir>/<name>/SKILL.md` directory (e.g. a repo's `.claude/skills`)
|
|
359
|
+
* to install, auto-packaged into a throwaway plugin — see {@link EvalArm.skillsDir}.
|
|
360
|
+
* Installs under `vigiles-loose-skills`, so `skill("vigiles-loose-skills:<name>")`.
|
|
361
|
+
*/
|
|
362
|
+
readonly skillsDir?: string;
|
|
363
|
+
/**
|
|
364
|
+
* Stub each skill BODY (frontmatter/trigger surface kept) before the run — for
|
|
365
|
+
* checks about whether a skill FIRES (`skill()`), not what it produces. A
|
|
366
|
+
* selected skill stops at selection instead of running its (often expensive)
|
|
367
|
+
* procedure, so a description/firing run costs a fraction of the tokens. Do NOT
|
|
368
|
+
* combine with `judged`/quality checks: the body is gone, so there's nothing to
|
|
369
|
+
* grade. Requires `pluginDir` or `skillsDir`. See {@link stubSkillBody}.
|
|
340
370
|
*/
|
|
341
371
|
readonly stubSkillBodies?: boolean;
|
|
342
372
|
/** Tools to intercept (the tool-call spy) — see {@link EvalArm.interceptTools}. */
|
|
@@ -385,6 +415,15 @@ export interface CheckReport {
|
|
|
385
415
|
readonly perCheck: readonly CheckRate[];
|
|
386
416
|
/** Cost / latency / token totals for the run (the same source as `runEval`). */
|
|
387
417
|
readonly usage: ArmUsage;
|
|
418
|
+
/**
|
|
419
|
+
* The plugin namespace the skills actually installed under — the `<plugin>`
|
|
420
|
+
* half of the `<plugin>:<skill>` id a `skill()` check matches. Undefined when
|
|
421
|
+
* the run installed nothing (no `pluginDir` / `skillsDir`). Reported because
|
|
422
|
+
* with `skillsDir` the name is chosen by the packager, not the caller, so the
|
|
423
|
+
* most common cause of a 0% `skill()` rate was a value the caller never saw.
|
|
424
|
+
* Mirrors {@link TriggerRateReport.namespace}.
|
|
425
|
+
*/
|
|
426
|
+
readonly namespace?: string;
|
|
388
427
|
}
|
|
389
428
|
/**
|
|
390
429
|
* Score a check vocabulary across trials — the scored counterpart to
|
|
@@ -407,8 +446,8 @@ export interface ArmsMeasureSpec {
|
|
|
407
446
|
* Stub each arm's skill BODIES (frontmatter kept) before the run — the A/B
|
|
408
447
|
* counterpart to {@link MeasureSpec.stubSkillBodies}. For firing comparisons
|
|
409
448
|
* (does description variant A fire more than B?), every arm that sets
|
|
410
|
-
* `pluginDir` is repackaged with bodies stripped so each run
|
|
411
|
-
* selection — a fraction of the tokens. Arms without
|
|
449
|
+
* `pluginDir` / `skillsDir` is repackaged with bodies stripped so each run
|
|
450
|
+
* stops at selection — a fraction of the tokens. Arms without one are left
|
|
412
451
|
* untouched. Don't combine with `judged`/quality checks. See {@link stubSkillBody}.
|
|
413
452
|
*/
|
|
414
453
|
readonly stubSkillBodies?: boolean;
|
|
@@ -593,7 +632,7 @@ export declare function seedEphemeralHome(throwawayHome: string, realHome: strin
|
|
|
593
632
|
export declare function isRateLimited(out: RunOut): boolean;
|
|
594
633
|
/** Map `worker` over `items` with at most `concurrency` in flight, order preserved. */
|
|
595
634
|
export declare function runPool<T, R>(items: readonly T[], concurrency: number, worker: (item: T) => Promise<R>): Promise<R[]>;
|
|
596
|
-
export declare function runEvalWith<M extends Metrics>(
|
|
635
|
+
export declare function runEvalWith<M extends Metrics>(input: EvalSpec<M>, runner: AgentRunner): Promise<EvalReport>;
|
|
597
636
|
/** Format an eval report as a compact table for the console (mean ± se, pass^k). */
|
|
598
637
|
export declare function formatEvalReport(report: EvalReport): string;
|
|
599
638
|
/**
|
package/dist/eval.js
CHANGED
|
@@ -87,6 +87,7 @@ const coverage_probe_js_1 = require("./coverage-probe.js");
|
|
|
87
87
|
const tool_intercept_js_1 = require("./tool-intercept.js");
|
|
88
88
|
const tool_stub_js_1 = require("./tool-stub.js");
|
|
89
89
|
const tmp_root_js_1 = require("./core/tmp-root.js");
|
|
90
|
+
const edit_distance_js_1 = require("./core/edit-distance.js");
|
|
90
91
|
function writeFiles(cwd, files) {
|
|
91
92
|
for (const [p, content] of Object.entries(files)) {
|
|
92
93
|
const full = (0, node_path_1.resolve)(cwd, p);
|
|
@@ -315,8 +316,12 @@ async function runEval(spec) {
|
|
|
315
316
|
* `runner` so the orchestration is unit-testable without a model.
|
|
316
317
|
*/
|
|
317
318
|
async function measureWith(spec, runner) {
|
|
318
|
-
|
|
319
|
-
|
|
319
|
+
assertKnownKeys(spec, MEASURE_SPEC_KEYS, {
|
|
320
|
+
caller: "measure",
|
|
321
|
+
type: "MeasureSpec",
|
|
322
|
+
});
|
|
323
|
+
if (spec.stubSkillBodies && !spec.pluginDir && !spec.skillsDir)
|
|
324
|
+
throw new Error("measure: `stubSkillBodies` requires `pluginDir` or `skillsDir`.");
|
|
320
325
|
// stubSkillBodies replaces each skill BODY with a no-op (the run stops at
|
|
321
326
|
// selection), so there is no output to grade — a `judged` check would score an
|
|
322
327
|
// empty body and mislead. The docs warn against this pairing; enforce it.
|
|
@@ -325,51 +330,49 @@ async function measureWith(spec, runner) {
|
|
|
325
330
|
throw new Error("measure: `stubSkillBodies` is for firing/`skill()` checks only — it stubs " +
|
|
326
331
|
"the skill body, so there's no output for a `judged` check to grade. Drop " +
|
|
327
332
|
"`stubSkillBodies`, or remove the `judged` check.");
|
|
328
|
-
const
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
}
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
(0, node_fs_1.rmSync)(stubbed, { recursive: true, force: true });
|
|
372
|
-
}
|
|
333
|
+
const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
|
|
334
|
+
// The install source (plugin / pluginDir / skillsDir, stubbed or not) is
|
|
335
|
+
// resolved by runEvalWith, once per arm — measure is one arm, so it just
|
|
336
|
+
// forwards the fields and reads the arm's report back.
|
|
337
|
+
const arm = {
|
|
338
|
+
settings: spec.settings,
|
|
339
|
+
plugin: spec.plugin,
|
|
340
|
+
pluginDir: spec.pluginDir,
|
|
341
|
+
skillsDir: spec.skillsDir,
|
|
342
|
+
interceptTools: spec.interceptTools,
|
|
343
|
+
};
|
|
344
|
+
const report = await runEvalWith({
|
|
345
|
+
fixture: spec.fixture,
|
|
346
|
+
arms: { run: arm },
|
|
347
|
+
stubSkillBodies: spec.stubSkillBodies,
|
|
348
|
+
task: spec.task,
|
|
349
|
+
trials: spec.trials ?? 5,
|
|
350
|
+
model: spec.model ?? "sonnet",
|
|
351
|
+
effort: spec.effort,
|
|
352
|
+
allowedTools: spec.allowedTools,
|
|
353
|
+
timeoutMs: spec.timeoutMs,
|
|
354
|
+
spacingSec: spec.spacingSec,
|
|
355
|
+
measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
|
|
356
|
+
}, runner);
|
|
357
|
+
return checkReportOf(report.arms.run, keyed, installNamespace(arm));
|
|
358
|
+
}
|
|
359
|
+
/** Read one arm's {@link ArmReport} back into a {@link CheckReport}. */
|
|
360
|
+
function checkReportOf(arm, keyed, namespace) {
|
|
361
|
+
return {
|
|
362
|
+
n: arm?.runs ?? 0,
|
|
363
|
+
perCheck: keyed.map(([k, c]) => {
|
|
364
|
+
const s = arm?.stats[k];
|
|
365
|
+
return {
|
|
366
|
+
check: c.toJSON(),
|
|
367
|
+
rate: s?.mean ?? 0,
|
|
368
|
+
se: s?.se ?? 0,
|
|
369
|
+
passK: s?.passK ?? 0,
|
|
370
|
+
n: s?.n ?? 0,
|
|
371
|
+
};
|
|
372
|
+
}),
|
|
373
|
+
usage: arm?.usage ?? aggregateUsage([]),
|
|
374
|
+
namespace,
|
|
375
|
+
};
|
|
373
376
|
}
|
|
374
377
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureWith */
|
|
375
378
|
/** Score a check vocabulary across trials against the real `claude` CLI. */
|
|
@@ -380,67 +383,36 @@ async function measure(spec) {
|
|
|
380
383
|
}
|
|
381
384
|
/** Score checks across arms (injectable runner). Reuses `runEvalWith`. */
|
|
382
385
|
async function measureArmsWith(spec, runner) {
|
|
386
|
+
assertKnownKeys(spec, ARMS_MEASURE_SPEC_KEYS, {
|
|
387
|
+
caller: "measureArms",
|
|
388
|
+
type: "ArmsMeasureSpec",
|
|
389
|
+
});
|
|
390
|
+
for (const [name, arm] of Object.entries(spec.arms))
|
|
391
|
+
assertKnownKeys(arm, EVAL_ARM_KEYS, {
|
|
392
|
+
caller: "measureArms",
|
|
393
|
+
type: "EvalArm",
|
|
394
|
+
arm: name,
|
|
395
|
+
});
|
|
383
396
|
const keyed = spec.checks.map((c, i) => [`c${String(i)}`, c]);
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
:
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
for (const [armName, arm] of Object.entries(report.arms)) {
|
|
402
|
-
arms[armName] = {
|
|
403
|
-
n: arm.runs,
|
|
404
|
-
perCheck: keyed.map(([k, c]) => {
|
|
405
|
-
const s = arm.stats[k];
|
|
406
|
-
return {
|
|
407
|
-
check: c.toJSON(),
|
|
408
|
-
rate: s?.mean ?? 0,
|
|
409
|
-
se: s?.se ?? 0,
|
|
410
|
-
passK: s?.passK ?? 0,
|
|
411
|
-
n: s?.n ?? 0,
|
|
412
|
-
};
|
|
413
|
-
}),
|
|
414
|
-
usage: arm.usage,
|
|
415
|
-
};
|
|
416
|
-
}
|
|
417
|
-
return { arms };
|
|
418
|
-
}
|
|
419
|
-
finally {
|
|
420
|
-
for (const t of temps)
|
|
421
|
-
(0, node_fs_1.rmSync)(t, { recursive: true, force: true });
|
|
422
|
-
}
|
|
423
|
-
}
|
|
424
|
-
/**
|
|
425
|
-
* Repackage every arm that sets a `pluginDir` with its skill bodies stubbed
|
|
426
|
-
* (frontmatter kept), for an A/B firing comparison. Returns the rewritten arms
|
|
427
|
-
* plus the throwaway dirs the caller must remove. Arms without a `pluginDir` pass
|
|
428
|
-
* through unchanged. See {@link stubbedPluginDir}.
|
|
429
|
-
*/
|
|
430
|
-
function stubArmPluginDirs(arms) {
|
|
431
|
-
const out = {};
|
|
432
|
-
const temps = [];
|
|
433
|
-
for (const [name, arm] of Object.entries(arms)) {
|
|
434
|
-
if (arm.pluginDir) {
|
|
435
|
-
const stubbed = stubbedPluginDir(arm.pluginDir);
|
|
436
|
-
temps.push(stubbed);
|
|
437
|
-
out[name] = { ...arm, pluginDir: stubbed };
|
|
438
|
-
}
|
|
439
|
-
else {
|
|
440
|
-
out[name] = arm;
|
|
441
|
-
}
|
|
397
|
+
// Per-arm install sources (and the stub) are resolved by runEvalWith.
|
|
398
|
+
const report = await runEvalWith({
|
|
399
|
+
fixture: spec.fixture,
|
|
400
|
+
arms: spec.arms,
|
|
401
|
+
stubSkillBodies: spec.stubSkillBodies,
|
|
402
|
+
task: spec.task,
|
|
403
|
+
trials: spec.trials ?? 5,
|
|
404
|
+
model: spec.model ?? "sonnet",
|
|
405
|
+
effort: spec.effort,
|
|
406
|
+
allowedTools: spec.allowedTools,
|
|
407
|
+
timeoutMs: spec.timeoutMs,
|
|
408
|
+
spacingSec: spec.spacingSec,
|
|
409
|
+
measure: (ctx) => Object.fromEntries(keyed.map(([k, c]) => [k, c.eval(ctx).pass])),
|
|
410
|
+
}, runner);
|
|
411
|
+
const arms = {};
|
|
412
|
+
for (const [armName, arm] of Object.entries(report.arms)) {
|
|
413
|
+
arms[armName] = checkReportOf(arm, keyed, installNamespace(spec.arms[armName] ?? {}));
|
|
442
414
|
}
|
|
443
|
-
return { arms
|
|
415
|
+
return { arms };
|
|
444
416
|
}
|
|
445
417
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
|
|
446
418
|
/** Score checks across arms against the real `claude` CLI. */
|
|
@@ -482,6 +454,28 @@ function formatCheckReport(report) {
|
|
|
482
454
|
lines.push(` ${(c.rate * 100).toFixed(0)}% ± ${(c.se * 100).toFixed(0)}% ${checkLabel(c.check)}` +
|
|
483
455
|
` (pass^k ${String(c.passK)})`);
|
|
484
456
|
}
|
|
457
|
+
// The `skill()` twin of the trigger-rate TOTAL-zero note (see
|
|
458
|
+
// formatTriggerRateReport): every skill() check at 0% over real runs is far
|
|
459
|
+
// more often a wiring mistake — a bare id where the namespaced one is matched,
|
|
460
|
+
// `pluginDir` handed a loose dir, an empty cwd — than a finding, and it reads
|
|
461
|
+
// exactly like a finding. Only when EVERY skill check is at zero: a partial rate
|
|
462
|
+
// is a real measurement, and a note that hedges on good data gets ignored.
|
|
463
|
+
const skillChecks = report.perCheck.filter((c) => c.check.kind === "skill");
|
|
464
|
+
if (report.n > 0 &&
|
|
465
|
+
skillChecks.length > 0 &&
|
|
466
|
+
skillChecks.every((c) => c.rate === 0))
|
|
467
|
+
lines.push("⚠ nothing resolved for ANY skill() check. That is usually SETUP, not the skill — check, in order:\n" +
|
|
468
|
+
(report.namespace !== undefined
|
|
469
|
+
? " 1. the id in `skill()` — your skills installed under " +
|
|
470
|
+
`\`${report.namespace}\`, so \`skill()\` matches ` +
|
|
471
|
+
`\`${report.namespace}:<skill>\`; a bare name silently never matches;\n`
|
|
472
|
+
: " 1. the id in `skill()` — it matches the NAMESPACED id " +
|
|
473
|
+
"(`<plugin>:<skill>`); a bare name silently never matches;\n") +
|
|
474
|
+
" 2. the install field — a loose `.claude/skills` dir needs `skillsDir`, " +
|
|
475
|
+
"not `pluginDir` (which wants a full plugin manifest);\n" +
|
|
476
|
+
" 3. the `fixture` — a run starts in an EMPTY cwd, so a prompt about a " +
|
|
477
|
+
"file that does not exist is one the model is right to decline.\n" +
|
|
478
|
+
" Rule out all three before recording this as a fact about the skill.");
|
|
485
479
|
return lines.join("\n");
|
|
486
480
|
}
|
|
487
481
|
/**
|
|
@@ -1317,10 +1311,175 @@ function evalArmsInputs(spec, cfg) {
|
|
|
1317
1311
|
ephemeralEnv: spec.ephemeralEnv === true,
|
|
1318
1312
|
};
|
|
1319
1313
|
}
|
|
1320
|
-
|
|
1314
|
+
// ---------------------------------------------------------------------------
|
|
1315
|
+
// The spec boundary — refuse a field the spec does not declare.
|
|
1316
|
+
//
|
|
1317
|
+
// A spec usually arrives from a plain-JS `*.eval.mjs` file, where nothing
|
|
1318
|
+
// type-checks the object literal, so a key that is misspelled (`skillDir`) or
|
|
1319
|
+
// belongs to a sibling runner (`prompts` on measure) used to be dropped without a
|
|
1320
|
+
// word — and the run then reported a confident number about a setup the author
|
|
1321
|
+
// never asked for (issue #307). The precedent is `driverMisplaced` in
|
|
1322
|
+
// eval-entry.ts: a field that would silently do nothing is REFUSED, not ignored.
|
|
1323
|
+
//
|
|
1324
|
+
// Each key table is `satisfies Record<keyof Spec, true>`, so the compiler fails
|
|
1325
|
+
// when a spec gains or loses a field the table does not: the list cannot drift.
|
|
1326
|
+
// ---------------------------------------------------------------------------
|
|
1327
|
+
const EVAL_ARM_KEYS = {
|
|
1328
|
+
files: true,
|
|
1329
|
+
settings: true,
|
|
1330
|
+
plugin: true,
|
|
1331
|
+
pluginDir: true,
|
|
1332
|
+
skillsDir: true,
|
|
1333
|
+
interceptTools: true,
|
|
1334
|
+
model: true,
|
|
1335
|
+
effort: true,
|
|
1336
|
+
};
|
|
1337
|
+
const EVAL_SPEC_KEYS = {
|
|
1338
|
+
name: true,
|
|
1339
|
+
fixture: true,
|
|
1340
|
+
arms: true,
|
|
1341
|
+
stubSkillBodies: true,
|
|
1342
|
+
task: true,
|
|
1343
|
+
measure: true,
|
|
1344
|
+
trials: true,
|
|
1345
|
+
model: true,
|
|
1346
|
+
effort: true,
|
|
1347
|
+
allowedTools: true,
|
|
1348
|
+
timeoutMs: true,
|
|
1349
|
+
spacingSec: true,
|
|
1350
|
+
cache: true,
|
|
1351
|
+
cacheDir: true,
|
|
1352
|
+
concurrency: true,
|
|
1353
|
+
maxCostUsd: true,
|
|
1354
|
+
rateLimitRetries: true,
|
|
1355
|
+
retryBackoffMs: true,
|
|
1356
|
+
ephemeralEnv: true,
|
|
1357
|
+
stubs: true,
|
|
1358
|
+
lock: true,
|
|
1359
|
+
};
|
|
1360
|
+
const MEASURE_SPEC_KEYS = {
|
|
1361
|
+
fixture: true,
|
|
1362
|
+
settings: true,
|
|
1363
|
+
plugin: true,
|
|
1364
|
+
pluginDir: true,
|
|
1365
|
+
skillsDir: true,
|
|
1366
|
+
stubSkillBodies: true,
|
|
1367
|
+
interceptTools: true,
|
|
1368
|
+
task: true,
|
|
1369
|
+
checks: true,
|
|
1370
|
+
trials: true,
|
|
1371
|
+
model: true,
|
|
1372
|
+
effort: true,
|
|
1373
|
+
allowedTools: true,
|
|
1374
|
+
timeoutMs: true,
|
|
1375
|
+
spacingSec: true,
|
|
1376
|
+
};
|
|
1377
|
+
const ARMS_MEASURE_SPEC_KEYS = {
|
|
1378
|
+
fixture: true,
|
|
1379
|
+
arms: true,
|
|
1380
|
+
task: true,
|
|
1381
|
+
checks: true,
|
|
1382
|
+
stubSkillBodies: true,
|
|
1383
|
+
trials: true,
|
|
1384
|
+
model: true,
|
|
1385
|
+
allowedTools: true,
|
|
1386
|
+
timeoutMs: true,
|
|
1387
|
+
effort: true,
|
|
1388
|
+
spacingSec: true,
|
|
1389
|
+
};
|
|
1390
|
+
const TRIGGER_RATE_SPEC_KEYS = {
|
|
1391
|
+
name: true,
|
|
1392
|
+
lock: true,
|
|
1393
|
+
pluginDir: true,
|
|
1394
|
+
skillsDir: true,
|
|
1395
|
+
prompts: true,
|
|
1396
|
+
irrelevantPrompts: true,
|
|
1397
|
+
fired: true,
|
|
1398
|
+
installSet: true,
|
|
1399
|
+
stubSkillBodies: true,
|
|
1400
|
+
minPrompts: true,
|
|
1401
|
+
minDistance: true,
|
|
1402
|
+
trials: true,
|
|
1403
|
+
model: true,
|
|
1404
|
+
effort: true,
|
|
1405
|
+
minModel: true,
|
|
1406
|
+
allowedTools: true,
|
|
1407
|
+
timeoutMs: true,
|
|
1408
|
+
spacingSec: true,
|
|
1409
|
+
fixture: true,
|
|
1410
|
+
concurrency: true,
|
|
1411
|
+
};
|
|
1412
|
+
/**
|
|
1413
|
+
* The closest declared field to an unknown one, or undefined when nothing is
|
|
1414
|
+
* close — a wrong suggestion is worse than none (it invites "fixing" a field the
|
|
1415
|
+
* author never meant). Same tight threshold as the CLI's unknown-flag hint.
|
|
1416
|
+
*/
|
|
1417
|
+
function nearestField(unknown, known) {
|
|
1418
|
+
let best;
|
|
1419
|
+
for (const name of known) {
|
|
1420
|
+
const d = (0, edit_distance_js_1.editDistance)(unknown.toLowerCase(), name.toLowerCase());
|
|
1421
|
+
if (!best || d < best.d)
|
|
1422
|
+
best = { name, d };
|
|
1423
|
+
}
|
|
1424
|
+
return best && best.d <= Math.max(2, Math.floor(unknown.length / 4))
|
|
1425
|
+
? best.name
|
|
1426
|
+
: undefined;
|
|
1427
|
+
}
|
|
1428
|
+
/**
|
|
1429
|
+
* Throw when `spec` carries a field outside `known` — every unknown one named,
|
|
1430
|
+
* with a did-you-mean where a declared field is one typo away. `at` says which
|
|
1431
|
+
* runner and spec type the message is about (`arm` labels a nested arm). Runs
|
|
1432
|
+
* before anything is packaged or spent.
|
|
1433
|
+
*/
|
|
1434
|
+
function assertKnownKeys(spec, known, at) {
|
|
1435
|
+
const declared = Object.keys(known);
|
|
1436
|
+
const where = at.arm === undefined ? "" : ` on arm "${at.arm}"`;
|
|
1437
|
+
const problems = Object.keys(spec)
|
|
1438
|
+
.filter((key) => !Object.hasOwn(known, key))
|
|
1439
|
+
.map((key) => {
|
|
1440
|
+
const near = nearestField(key, declared);
|
|
1441
|
+
const hint = near === undefined ? "" : ` — did you mean \`${near}\`?`;
|
|
1442
|
+
return `${at.caller}: unknown ${at.type} field "${key}"${where}${hint}`;
|
|
1443
|
+
});
|
|
1444
|
+
if (problems.length === 0)
|
|
1445
|
+
return;
|
|
1446
|
+
throw new Error(`${problems.join("\n")}\n ${at.type} fields: ${declared.join(", ")}.\n` +
|
|
1447
|
+
" An unknown field is refused rather than ignored: a dropped field would " +
|
|
1448
|
+
"make the run measure a setup you did not ask for.");
|
|
1449
|
+
}
|
|
1450
|
+
async function runEvalWith(input, runner) {
|
|
1451
|
+
// Refuse a stray field BEFORE spending a token: an eval file is plain JS, so a
|
|
1452
|
+
// typo'd or misplaced key would otherwise vanish and the run would report a
|
|
1453
|
+
// confident number about the wrong setup (issue #307).
|
|
1454
|
+
assertKnownKeys(input, EVAL_SPEC_KEYS, {
|
|
1455
|
+
caller: "runEval",
|
|
1456
|
+
type: "EvalSpec",
|
|
1457
|
+
});
|
|
1458
|
+
for (const [name, arm] of Object.entries(input.arms))
|
|
1459
|
+
assertKnownKeys(arm, EVAL_ARM_KEYS, {
|
|
1460
|
+
caller: "runEval",
|
|
1461
|
+
type: "EvalArm",
|
|
1462
|
+
arm: name,
|
|
1463
|
+
});
|
|
1321
1464
|
// Tell the CLI runner this script exercised the harness, so a file that runs
|
|
1322
1465
|
// NOTHING can be told apart from one that ran and passed. See check-count.ts.
|
|
1323
1466
|
(0, check_count_js_1.recordCheck)();
|
|
1467
|
+
// Every arm's install source (`pluginDir` as-is / stubbed, or a loose
|
|
1468
|
+
// `skillsDir` packaged into a throwaway plugin) is resolved HERE, once — the
|
|
1469
|
+
// one place that decides what `--plugin-dir` receives, so measure / measureArms
|
|
1470
|
+
// / runEval cannot disagree about it. The throwaways are removed afterward.
|
|
1471
|
+
const { arms: resolvedArms, packaged } = resolveArmInstalls(input.arms, input.stubSkillBodies ?? false);
|
|
1472
|
+
const spec = { ...input, arms: resolvedArms };
|
|
1473
|
+
try {
|
|
1474
|
+
return await runResolvedEval(spec, runner);
|
|
1475
|
+
}
|
|
1476
|
+
finally {
|
|
1477
|
+
for (const dir of packaged)
|
|
1478
|
+
(0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
|
|
1479
|
+
}
|
|
1480
|
+
}
|
|
1481
|
+
/** {@link runEvalWith} after its arms' install sources are concrete plugin dirs. */
|
|
1482
|
+
async function runResolvedEval(spec, runner) {
|
|
1324
1483
|
const trials = spec.trials ?? 5;
|
|
1325
1484
|
const spacing = (spec.spacingSec ?? 4) * 1000;
|
|
1326
1485
|
const concurrency = spec.concurrency ?? 1;
|
|
@@ -1493,7 +1652,7 @@ function packageSkillsDir(skillsDir, opts = {}) {
|
|
|
1493
1652
|
throw new Error(`skillsDir not found: ${skillsDir} (resolved ${abs})`);
|
|
1494
1653
|
const root = (0, tmp_root_js_1.makeTmpDir)("skills");
|
|
1495
1654
|
(0, node_fs_1.mkdirSync)((0, node_path_1.join)(root, ".claude-plugin"), { recursive: true });
|
|
1496
|
-
(0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ??
|
|
1655
|
+
(0, node_fs_1.writeFileSync)((0, node_path_1.join)(root, ".claude-plugin", "plugin.json"), JSON.stringify({ name: opts.name ?? LOOSE_SKILLS_NAMESPACE, version: "0.0.0" }, null, 2));
|
|
1497
1656
|
const skillsOut = (0, node_path_1.join)(root, "skills");
|
|
1498
1657
|
(0, node_fs_1.mkdirSync)(skillsOut, { recursive: true });
|
|
1499
1658
|
let copied = 0;
|
|
@@ -1704,16 +1863,96 @@ function packageInstallSet(opts) {
|
|
|
1704
1863
|
throw e;
|
|
1705
1864
|
}
|
|
1706
1865
|
}
|
|
1707
|
-
|
|
1708
|
-
|
|
1709
|
-
|
|
1710
|
-
|
|
1711
|
-
|
|
1712
|
-
|
|
1713
|
-
|
|
1714
|
-
|
|
1715
|
-
|
|
1716
|
-
|
|
1866
|
+
// ---------------------------------------------------------------------------
|
|
1867
|
+
// The install source — the ONE place that decides what `--plugin-dir` receives.
|
|
1868
|
+
//
|
|
1869
|
+
// Every runner that installs skills (runEval / measure / measureArms via an
|
|
1870
|
+
// EvalArm, measureTriggerRate via its spec) accepts the same two fields —
|
|
1871
|
+
// `pluginDir` (a complete plugin, used as-is or re-packaged with stubbed bodies)
|
|
1872
|
+
// and `skillsDir` (a loose `<name>/SKILL.md` dir, packaged into a throwaway
|
|
1873
|
+
// plugin) — and they all resolve through `resolveSkillInstall`. Before this
|
|
1874
|
+
// existed each runner re-derived the rule for itself, and the one written last
|
|
1875
|
+
// was the only one that knew about loose dirs (issue #307).
|
|
1876
|
+
// ---------------------------------------------------------------------------
|
|
1877
|
+
/** The plugin name a packaged loose skills dir installs under. */
|
|
1878
|
+
const LOOSE_SKILLS_NAMESPACE = "vigiles-loose-skills";
|
|
1879
|
+
/**
|
|
1880
|
+
* The plugin namespace a source's skills install under — the `<plugin>` half of
|
|
1881
|
+
* the `<plugin>:<skill>` id `skill()` / `skillResolved` match. A loose dir gets
|
|
1882
|
+
* the packager's name; a plugin its declared name, falling back to the same
|
|
1883
|
+
* synthetic one when its manifest has none. Pure (reads the manifest only).
|
|
1884
|
+
*/
|
|
1885
|
+
function installNamespace(src) {
|
|
1886
|
+
if (src.skillsDir)
|
|
1887
|
+
return LOOSE_SKILLS_NAMESPACE;
|
|
1888
|
+
if (src.pluginDir)
|
|
1889
|
+
return pluginName(src.pluginDir) ?? LOOSE_SKILLS_NAMESPACE;
|
|
1890
|
+
return undefined;
|
|
1891
|
+
}
|
|
1892
|
+
/**
|
|
1893
|
+
* Resolve a {@link SkillSource} to a concrete plugin dir. `stub` strips skill
|
|
1894
|
+
* bodies (frontmatter kept) into a throwaway; `installSet` merges competitor
|
|
1895
|
+
* skills in (the whole-harness tier, `measureTriggerRate` only). Exactly one of
|
|
1896
|
+
* `pluginDir` / `skillsDir` may be set; neither resolves to nothing installed —
|
|
1897
|
+
* the caller decides whether that is legal (an arm: yes; a trigger run: no).
|
|
1898
|
+
*/
|
|
1899
|
+
function resolveSkillInstall(src, opts) {
|
|
1900
|
+
if (src.pluginDir && src.skillsDir)
|
|
1901
|
+
throw new Error(`${opts.caller}: set \`pluginDir\` OR \`skillsDir\`, not both.`);
|
|
1902
|
+
const namespace = installNamespace(src);
|
|
1903
|
+
if (namespace === undefined)
|
|
1904
|
+
return {};
|
|
1905
|
+
const installSet = opts.installSet ?? [];
|
|
1906
|
+
if (installSet.length > 0) {
|
|
1907
|
+
// Whole-harness tier: merge the under-test skills with the install set so
|
|
1908
|
+
// selection is competitive (the realistic, differentiated measurement).
|
|
1909
|
+
const { dir } = packageInstallSet({
|
|
1910
|
+
underTestSrc: src.skillsDir ?? skillsDirOf(src.pluginDir),
|
|
1911
|
+
name: namespace,
|
|
1912
|
+
installSet,
|
|
1913
|
+
stub: opts.stub,
|
|
1914
|
+
});
|
|
1915
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1916
|
+
}
|
|
1917
|
+
if (src.skillsDir) {
|
|
1918
|
+
const dir = packageSkillsDir(src.skillsDir, { stub: opts.stub });
|
|
1919
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1920
|
+
}
|
|
1921
|
+
// A real plugin: as-is, or re-packaged from its skills/ with bodies stripped —
|
|
1922
|
+
// keeping the original plugin NAME so `<name>:<skill>` still matches.
|
|
1923
|
+
const pluginDir = src.pluginDir;
|
|
1924
|
+
if (!opts.stub)
|
|
1925
|
+
return { pluginDir, namespace };
|
|
1926
|
+
const dir = stubbedPluginDir(pluginDir);
|
|
1927
|
+
return { pluginDir: dir, packaged: dir, namespace };
|
|
1928
|
+
}
|
|
1929
|
+
/**
|
|
1930
|
+
* Resolve every arm's install source (see {@link resolveSkillInstall}) so each
|
|
1931
|
+
* arm carries only a concrete `pluginDir` — `skillsDir` is consumed here. Returns
|
|
1932
|
+
* the rewritten arms plus the throwaway dirs the caller removes afterward. A
|
|
1933
|
+
* failure part-way removes what was already built (no leaked temp dirs).
|
|
1934
|
+
*/
|
|
1935
|
+
function resolveArmInstalls(arms, stub) {
|
|
1936
|
+
const out = {};
|
|
1937
|
+
const packaged = [];
|
|
1938
|
+
try {
|
|
1939
|
+
for (const [name, arm] of Object.entries(arms)) {
|
|
1940
|
+
const { skillsDir: _consumed, ...rest } = arm;
|
|
1941
|
+
const r = resolveSkillInstall(arm, {
|
|
1942
|
+
caller: `eval arm "${name}"`,
|
|
1943
|
+
stub,
|
|
1944
|
+
});
|
|
1945
|
+
if (r.packaged)
|
|
1946
|
+
packaged.push(r.packaged);
|
|
1947
|
+
out[name] = { ...rest, pluginDir: r.pluginDir };
|
|
1948
|
+
}
|
|
1949
|
+
}
|
|
1950
|
+
catch (e) {
|
|
1951
|
+
for (const dir of packaged)
|
|
1952
|
+
(0, node_fs_1.rmSync)(dir, { recursive: true, force: true });
|
|
1953
|
+
throw e;
|
|
1954
|
+
}
|
|
1955
|
+
return { arms: out, packaged };
|
|
1717
1956
|
}
|
|
1718
1957
|
/** Number of `<name>/SKILL.md` skills installed in a plugin — the selection pool. */
|
|
1719
1958
|
function countSkills(pluginDir) {
|
|
@@ -1727,46 +1966,22 @@ function countSkills(pluginDir) {
|
|
|
1727
1966
|
return n;
|
|
1728
1967
|
}
|
|
1729
1968
|
function resolveTriggerPluginDir(spec) {
|
|
1730
|
-
|
|
1731
|
-
|
|
1732
|
-
|
|
1733
|
-
|
|
1734
|
-
|
|
1735
|
-
|
|
1736
|
-
if (installSet.length > 0) {
|
|
1737
|
-
// Whole-harness tier: merge the under-test skills with the install set so
|
|
1738
|
-
// selection is competitive (the realistic, differentiated measurement).
|
|
1739
|
-
const { src, name } = underTestSource(spec);
|
|
1740
|
-
({ dir: pluginDir } = packageInstallSet({
|
|
1741
|
-
underTestSrc: src,
|
|
1742
|
-
name,
|
|
1743
|
-
installSet,
|
|
1744
|
-
stub,
|
|
1745
|
-
}));
|
|
1746
|
-
packaged = pluginDir;
|
|
1747
|
-
}
|
|
1748
|
-
else if (spec.skillsDir) {
|
|
1749
|
-
packaged = packageSkillsDir(spec.skillsDir, { stub });
|
|
1750
|
-
pluginDir = packaged;
|
|
1751
|
-
}
|
|
1752
|
-
else if (spec.pluginDir) {
|
|
1753
|
-
// Stub a real plugin: build a minimal plugin from its skills/ with bodies
|
|
1754
|
-
// stripped — keep the original plugin NAME so `<name>:<skill>` still matches.
|
|
1755
|
-
packaged = stub ? stubbedPluginDir(spec.pluginDir) : undefined;
|
|
1756
|
-
pluginDir = packaged ?? spec.pluginDir;
|
|
1757
|
-
}
|
|
1758
|
-
else {
|
|
1969
|
+
const r = resolveSkillInstall(spec, {
|
|
1970
|
+
caller: "measureTriggerRate",
|
|
1971
|
+
stub: spec.stubSkillBodies ?? true, // trigger = frontmatter; body never needed
|
|
1972
|
+
installSet: spec.installSet,
|
|
1973
|
+
});
|
|
1974
|
+
if (r.pluginDir === undefined || r.namespace === undefined)
|
|
1759
1975
|
throw new Error("measureTriggerRate: provide `pluginDir` or `skillsDir`.");
|
|
1760
|
-
}
|
|
1761
1976
|
// `competitors` is the REAL selection pressure: every OTHER skill installed in
|
|
1762
1977
|
// the resolved plugin (siblings already in the source + any installSet), not
|
|
1763
1978
|
// just the installSet delta — so a multi-skill plugin is never mislabeled
|
|
1764
1979
|
// "isolated". `max(0, …)` guards a 0-skill pool.
|
|
1765
1980
|
return {
|
|
1766
|
-
pluginDir,
|
|
1767
|
-
packaged,
|
|
1768
|
-
competitors: Math.max(0, countSkills(pluginDir) - 1),
|
|
1769
|
-
namespace:
|
|
1981
|
+
pluginDir: r.pluginDir,
|
|
1982
|
+
packaged: r.packaged,
|
|
1983
|
+
competitors: Math.max(0, countSkills(r.pluginDir) - 1),
|
|
1984
|
+
namespace: r.namespace,
|
|
1770
1985
|
};
|
|
1771
1986
|
}
|
|
1772
1987
|
/** Run one prompt set × trials through `runner`, aggregating fired counts. */
|
|
@@ -1859,6 +2074,10 @@ function assertTriggerDiversity(spec) {
|
|
|
1859
2074
|
}
|
|
1860
2075
|
}
|
|
1861
2076
|
async function measureTriggerRateWith(spec, runner, parse = parseClaudeRun, runError, harness = "claude-code") {
|
|
2077
|
+
assertKnownKeys(spec, TRIGGER_RATE_SPEC_KEYS, {
|
|
2078
|
+
caller: "measureTriggerRate",
|
|
2079
|
+
type: "TriggerRateSpec",
|
|
2080
|
+
});
|
|
1862
2081
|
// Tell the CLI runner this script exercised the harness (see check-count.ts).
|
|
1863
2082
|
(0, check_count_js_1.recordCheck)();
|
|
1864
2083
|
// Deterministic gate FIRST — before spending a token (or packaging a skillsDir).
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "32.0.0",
|
|
4
4
|
"description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|