vigiles 12.1.0 → 12.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +128 -154
- package/action.yml +73 -0
- package/dist/adapters/claude-code/run-scripts.d.ts +41 -0
- package/dist/adapters/claude-code/run-scripts.js +26 -0
- package/dist/audit-report.d.ts +11 -0
- package/dist/audit-report.js +1 -0
- package/dist/audit-report.template.html +36 -26
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +0 -1
- package/dist/cli.js +111 -139
- package/dist/eval.d.ts +15 -0
- package/dist/eval.js +60 -0
- package/dist/observe.d.ts +109 -0
- package/dist/observe.js +164 -0
- package/dist/scaffold-test.js +3 -2
- package/package.json +2 -2
- package/skills/debug-my-harness/SKILL.md +56 -0
- package/dist/core/hook-spec.d.ts +0 -74
- package/dist/core/hook-spec.js +0 -130
package/dist/cli-commands.d.ts
CHANGED
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
* recognizes exactly these, so this list can't silently drift from the code.
|
|
12
12
|
*/
|
|
13
13
|
/** Human-facing verbs (printed in help; typed by a human/agent/CI). */
|
|
14
|
-
export declare const VERBS: readonly ["init", "compile", "eject", "lint", "test", "eval", "audit", "
|
|
14
|
+
export declare const VERBS: readonly ["init", "compile", "eject", "lint", "test", "eval", "audit", "generate", "hook-runtime"];
|
|
15
15
|
/** Runtime entrypoint kinds under `vigiles hook-runtime <kind>` (emitted, not typed). */
|
|
16
16
|
export declare const HOOK_RUNTIME_KINDS: readonly ["run-program", "agent", "agent-start", "agent-done", "skill", "skill-tool", "skill-start", "skill-done", "run-skill", "intercept-tool", "guard", "action", "refs", "eval-lock-nudge", "effect-enter", "effect-exit"];
|
|
17
17
|
export type Verb = (typeof VERBS)[number];
|
package/dist/cli-commands.js
CHANGED
package/dist/cli.js
CHANGED
|
@@ -22,8 +22,6 @@ const cli_flags_js_1 = require("./cli-flags.js");
|
|
|
22
22
|
const setup_plan_js_1 = require("./setup-plan.js");
|
|
23
23
|
const types_js_1 = require("./core/types.js");
|
|
24
24
|
const test_coverage_js_1 = require("./test-coverage.js");
|
|
25
|
-
const scaffold_test_js_1 = require("./scaffold-test.js");
|
|
26
|
-
const effects_js_1 = require("./core/effects.js");
|
|
27
25
|
const scan_js_1 = require("./scan.js");
|
|
28
26
|
const scan_trigger_suggest_js_1 = require("./scan-trigger-suggest.js");
|
|
29
27
|
const dialect_drift_js_1 = require("./dialect-drift.js");
|
|
@@ -52,6 +50,7 @@ const hook_install_js_1 = require("./hook-install.js");
|
|
|
52
50
|
const hook_providers_js_1 = require("./core/hook-providers.js");
|
|
53
51
|
const toml_1 = require("@iarna/toml");
|
|
54
52
|
const agent_runtime_js_1 = require("./adapters/claude-code/agent-runtime.js");
|
|
53
|
+
const observe_js_1 = require("./observe.js");
|
|
55
54
|
const effect_region_js_1 = require("./adapters/claude-code/effect-region.js");
|
|
56
55
|
const tool_intercept_js_1 = require("./tool-intercept.js");
|
|
57
56
|
const refs_js_1 = require("./core/refs.js");
|
|
@@ -3488,11 +3487,29 @@ async function handleGenerateHarness(args, restArgs) {
|
|
|
3488
3487
|
console.log(`\n✓ Generated ${(0, generate_harness_js_1.labelFor)(process.cwd(), fullOut)}`);
|
|
3489
3488
|
console.log(" `tsc --noEmit` over this file now checks every delegate target resolves.");
|
|
3490
3489
|
}
|
|
3490
|
+
/** Minimal TTY yes/no prompt (readline). Returns true only on an explicit y/yes. */
|
|
3491
|
+
async function promptYesNo(question) {
|
|
3492
|
+
const readline = await import("node:readline");
|
|
3493
|
+
const rl = readline.createInterface({
|
|
3494
|
+
input: process.stdin,
|
|
3495
|
+
output: process.stdout,
|
|
3496
|
+
});
|
|
3497
|
+
try {
|
|
3498
|
+
const answer = await new Promise((res) => {
|
|
3499
|
+
rl.question(question, res);
|
|
3500
|
+
});
|
|
3501
|
+
return /^y(es)?$/i.test(answer.trim());
|
|
3502
|
+
}
|
|
3503
|
+
finally {
|
|
3504
|
+
rl.close();
|
|
3505
|
+
}
|
|
3506
|
+
}
|
|
3491
3507
|
/**
|
|
3492
3508
|
* `vigiles test` / `vigiles eval` — discover and run the two-tier harness
|
|
3493
3509
|
* scripts (deterministic `*.harness.mjs` / real-model `*.eval.mjs`) as child
|
|
3494
3510
|
* `node` processes, aggregating exit codes so they work as a CI command. See
|
|
3495
|
-
* src/run-scripts.ts.
|
|
3511
|
+
* src/run-scripts.ts. A bare `vigiles eval` (no target) asks before fanning out
|
|
3512
|
+
* over the whole tree — it spends model quota (see `decideRunScripts`).
|
|
3496
3513
|
*
|
|
3497
3514
|
* `vigiles test` skips clean when the `claude` CLI is absent (the deterministic
|
|
3498
3515
|
* tier needs it, just like the node:test suite). `--trials=N` is forwarded to
|
|
@@ -3533,7 +3550,7 @@ function resolveEvalLockEnv(args) {
|
|
|
3533
3550
|
}
|
|
3534
3551
|
return env;
|
|
3535
3552
|
}
|
|
3536
|
-
function handleRunScripts(kind, args, restArgs) {
|
|
3553
|
+
async function handleRunScripts(kind, args, restArgs) {
|
|
3537
3554
|
const cwd = process.cwd();
|
|
3538
3555
|
// Harness/eval scripts may be authored in JS or TS (see run-scripts.ts).
|
|
3539
3556
|
const defaultGlob = (0, run_scripts_js_1.scriptGlob)(kind === "test" ? "harness" : "eval");
|
|
@@ -3564,6 +3581,33 @@ function handleRunScripts(kind, args, restArgs) {
|
|
|
3564
3581
|
console.log(`No ${defaultGlob} files found.`);
|
|
3565
3582
|
return;
|
|
3566
3583
|
}
|
|
3584
|
+
// Consent gate for a bare `vigiles eval`: it runs the REAL model on your
|
|
3585
|
+
// subscription, and a no-target run discovered the whole tree — so never fan out
|
|
3586
|
+
// over an unbounded glob without explicit intent. Mirrors audit's read-vs-run
|
|
3587
|
+
// consent. `test` is free → always runs (decideRunScripts returns "run").
|
|
3588
|
+
const runDecision = (0, run_scripts_js_1.decideRunScripts)({
|
|
3589
|
+
kind,
|
|
3590
|
+
explicitTargets: restArgs.length > 0,
|
|
3591
|
+
matchedCount: files.length,
|
|
3592
|
+
isTTY: (process.stdin.isTTY ?? false) && (process.stdout.isTTY ?? false),
|
|
3593
|
+
all: args.includes("--all"),
|
|
3594
|
+
yes: args.includes("--yes") || args.includes("--no-interactive"),
|
|
3595
|
+
});
|
|
3596
|
+
if (runDecision.kind === "refuse") {
|
|
3597
|
+
console.error(`✗ vigiles eval: ${String(runDecision.count)} eval file(s) matched the whole tree, and each ` +
|
|
3598
|
+
"runs the real model on your subscription. Refusing to fire them all non-interactively.\n" +
|
|
3599
|
+
" → name the eval(s): vigiles eval path/to/x.eval.mjs\n" +
|
|
3600
|
+
" → or opt in to all: vigiles eval --all");
|
|
3601
|
+
process.exit(2);
|
|
3602
|
+
}
|
|
3603
|
+
if (runDecision.kind === "confirm") {
|
|
3604
|
+
const ok = await promptYesNo(`About to run ${String(runDecision.count)} eval file(s) against the real model on your ` +
|
|
3605
|
+
"subscription (uses your Claude quota). Continue? [y/N] ");
|
|
3606
|
+
if (!ok) {
|
|
3607
|
+
console.log("Aborted. Name specific eval(s), or pass --all to run them all.");
|
|
3608
|
+
return;
|
|
3609
|
+
}
|
|
3610
|
+
}
|
|
3567
3611
|
// No blanket skip: unit-tier (runHook) tests need no `claude`, so always run.
|
|
3568
3612
|
// A script whose tier DOES need `claude` self-reports `⊘ SKIPPED` (exit 77) —
|
|
3569
3613
|
// loud, never a silent green. Just flag up front that some may skip.
|
|
@@ -3611,135 +3655,6 @@ function capabilitiesOfReport(report, dialect) {
|
|
|
3611
3655
|
}));
|
|
3612
3656
|
return (0, generate_harness_js_1.computeHarnessCapabilities)(agents, dialect);
|
|
3613
3657
|
}
|
|
3614
|
-
/**
|
|
3615
|
-
* The plugin's declared name for the namespaced skill id, read from the layout's
|
|
3616
|
-
* manifest (adapter-aware path, not a hardcoded `.claude-plugin/`), falling back to
|
|
3617
|
-
* the dir basename. JSON manifests only for now (a TOML/Codex manifest → basename).
|
|
3618
|
-
*/
|
|
3619
|
-
function pluginNameFor(dir, manifestPath) {
|
|
3620
|
-
try {
|
|
3621
|
-
const manifest = JSON.parse((0, node_fs_1.readFileSync)((0, node_path_1.resolve)(dir, manifestPath), "utf-8"));
|
|
3622
|
-
if (typeof manifest.name === "string" && manifest.name)
|
|
3623
|
-
return manifest.name;
|
|
3624
|
-
}
|
|
3625
|
-
catch {
|
|
3626
|
-
/* missing / non-JSON manifest → fall back */
|
|
3627
|
-
}
|
|
3628
|
-
return (0, node_path_1.basename)(dir);
|
|
3629
|
-
}
|
|
3630
|
-
/** Enrich an untested Surface with the metadata the right template needs. */
|
|
3631
|
-
/** Extract `"name": type` fields from one rendered `vigiles:ok`/`err` shape block. */
|
|
3632
|
-
function parseContractFields(block) {
|
|
3633
|
-
const fields = [];
|
|
3634
|
-
const re = /"([^"]+)"\s*:\s*(string\[\]|string|number|boolean)/g;
|
|
3635
|
-
let m;
|
|
3636
|
-
while ((m = re.exec(block)) !== null) {
|
|
3637
|
-
fields.push({ name: m[1], type: m[2] });
|
|
3638
|
-
}
|
|
3639
|
-
return fields;
|
|
3640
|
-
}
|
|
3641
|
-
/**
|
|
3642
|
-
* Parse a subagent's compiled `## Output contract` (the `vigiles:ok` / `vigiles:err`
|
|
3643
|
-
* blocks the compiler emits) back into a typed `ResultContract`, so the generator
|
|
3644
|
-
* can write an `assertAgentOk` test against the real fields. Returns null when the
|
|
3645
|
-
* agent has no result() contract.
|
|
3646
|
-
*/
|
|
3647
|
-
function parseResultContract(md) {
|
|
3648
|
-
const ok = /```vigiles:ok\n([\s\S]*?)```/.exec(md);
|
|
3649
|
-
const err = /```vigiles:err\n([\s\S]*?)```/.exec(md);
|
|
3650
|
-
if (!ok && !err)
|
|
3651
|
-
return null;
|
|
3652
|
-
const okFields = ok ? parseContractFields(ok[1]) : [];
|
|
3653
|
-
const errFields = err ? parseContractFields(err[1]) : [];
|
|
3654
|
-
if (okFields.length === 0 && errFields.length === 0)
|
|
3655
|
-
return null;
|
|
3656
|
-
return { ok: okFields, err: errFields };
|
|
3657
|
-
}
|
|
3658
|
-
function scaffoldInputFor(s, report, pluginName, dir, dialect) {
|
|
3659
|
-
const base = { kind: s.kind, name: s.name, path: s.path };
|
|
3660
|
-
switch (s.kind) {
|
|
3661
|
-
case "skill": {
|
|
3662
|
-
const sk = report.skills.find((x) => x.name === s.name);
|
|
3663
|
-
return { ...base, pluginName, userInvoked: sk?.userInvoked };
|
|
3664
|
-
}
|
|
3665
|
-
case "agent": {
|
|
3666
|
-
const ag = report.agents.find((x) => x.name === s.name);
|
|
3667
|
-
const tools = ag?.tools ?? null;
|
|
3668
|
-
const sideEffectingTools = tools
|
|
3669
|
-
? (0, effects_js_1.effectSurface)(tools, dialect).sideEffecting
|
|
3670
|
-
: undefined;
|
|
3671
|
-
let resultContract = null;
|
|
3672
|
-
try {
|
|
3673
|
-
resultContract = parseResultContract((0, node_fs_1.readFileSync)((0, node_path_1.resolve)(dir, s.path), "utf-8"));
|
|
3674
|
-
}
|
|
3675
|
-
catch {
|
|
3676
|
-
// agent .md unreadable → no contract to generate against
|
|
3677
|
-
}
|
|
3678
|
-
return { ...base, tools, sideEffectingTools, resultContract };
|
|
3679
|
-
}
|
|
3680
|
-
case "hook":
|
|
3681
|
-
return { ...base, hookCommand: `bash ${s.path}` };
|
|
3682
|
-
}
|
|
3683
|
-
}
|
|
3684
|
-
/**
|
|
3685
|
-
* `vigiles scaffold-test [dir]` — generate a runnable STARTER test for each
|
|
3686
|
-
* untested skill/agent/hook (B1, test-gen from free-form). Reuses the
|
|
3687
|
-
* untested-surface detector for the list + `scan` for the metadata, then emits the
|
|
3688
|
-
* cheapest meaningful tier per kind (hook → `runHook`, skill → `measureTriggerRate`,
|
|
3689
|
-
* subagent → `runHarnessTest`) at the surface's suggested test path. Dry-run by
|
|
3690
|
-
* default (prints the scaffolds); `--write` creates the files (never clobbering an
|
|
3691
|
-
* existing one); `--json` for the agent-consumable `{ path, content }[]`.
|
|
3692
|
-
*/
|
|
3693
|
-
function handleScaffoldTest(restArgs, args) {
|
|
3694
|
-
const dir = (0, node_path_1.resolve)(restArgs[0] ?? ".");
|
|
3695
|
-
const write = args.includes("--write");
|
|
3696
|
-
const json = args.includes("--json");
|
|
3697
|
-
const harnessFlag = harnessFlagFrom(args);
|
|
3698
|
-
const adapter = harnessFlag
|
|
3699
|
-
? (0, adapter_registry_js_1.resolveAdapter)(dir, harnessFlag)
|
|
3700
|
-
: (0, adapter_registry_js_1.detectAdapterResult)(dir).adapter;
|
|
3701
|
-
const { untested } = (0, test_coverage_js_1.findUntestedSurfaces)({
|
|
3702
|
-
basePath: dir,
|
|
3703
|
-
layout: adapter.layout,
|
|
3704
|
-
});
|
|
3705
|
-
const report = (0, scan_js_1.scanPlugin)(dir, adapter.layout, adapter.dialect);
|
|
3706
|
-
const pluginName = pluginNameFor(dir, adapter.layout.manifestPath);
|
|
3707
|
-
const scaffolds = untested.map((s) => (0, scaffold_test_js_1.scaffoldTest)(scaffoldInputFor(s, report, pluginName, dir, adapter.dialect)));
|
|
3708
|
-
if (json) {
|
|
3709
|
-
console.log(JSON.stringify(scaffolds, null, 2));
|
|
3710
|
-
return;
|
|
3711
|
-
}
|
|
3712
|
-
if (!write) {
|
|
3713
|
-
console.log((0, scaffold_test_js_1.formatScaffolds)(scaffolds));
|
|
3714
|
-
for (const s of scaffolds) {
|
|
3715
|
-
console.log(`\n# ${s.path}\n`);
|
|
3716
|
-
console.log(s.content);
|
|
3717
|
-
}
|
|
3718
|
-
if (scaffolds.length > 0) {
|
|
3719
|
-
console.log("Re-run with --write to create these files.");
|
|
3720
|
-
}
|
|
3721
|
-
return;
|
|
3722
|
-
}
|
|
3723
|
-
const written = [];
|
|
3724
|
-
const skipped = [];
|
|
3725
|
-
for (const s of scaffolds) {
|
|
3726
|
-
const target = (0, node_path_1.resolve)(dir, s.path);
|
|
3727
|
-
if ((0, node_fs_1.existsSync)(target)) {
|
|
3728
|
-
skipped.push(s.path);
|
|
3729
|
-
continue;
|
|
3730
|
-
}
|
|
3731
|
-
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(target), { recursive: true });
|
|
3732
|
-
(0, node_fs_1.writeFileSync)(target, s.content);
|
|
3733
|
-
written.push(s.path);
|
|
3734
|
-
}
|
|
3735
|
-
for (const p of written)
|
|
3736
|
-
console.log(`✓ wrote ${p}`);
|
|
3737
|
-
for (const p of skipped)
|
|
3738
|
-
console.log(`⊘ skipped ${p} (already exists)`);
|
|
3739
|
-
if (written.length === 0 && skipped.length === 0) {
|
|
3740
|
-
console.log("Nothing to scaffold — every surface already has a test.");
|
|
3741
|
-
}
|
|
3742
|
-
}
|
|
3743
3658
|
function printUsage(command) {
|
|
3744
3659
|
console.log("vigiles — compile typed specs to instruction files");
|
|
3745
3660
|
console.log("");
|
|
@@ -3756,7 +3671,6 @@ function printUsage(command) {
|
|
|
3756
3671
|
console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
|
|
3757
3672
|
console.log(" --update records each named eval's result to a committed lock (run locally on your subscription)");
|
|
3758
3673
|
console.log(" --check verifies committed eval results against current inputs WITHOUT a model — the CI staleness gate");
|
|
3759
|
-
console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
|
|
3760
3674
|
console.log("");
|
|
3761
3675
|
console.log("Examples:");
|
|
3762
3676
|
console.log(" vigiles init Auto-detect project, create specs, wire CI");
|
|
@@ -3862,6 +3776,11 @@ function skillStartCommand(target) {
|
|
|
3862
3776
|
process.exit(2);
|
|
3863
3777
|
}
|
|
3864
3778
|
(0, skill_runtime_js_1.setActiveSkill)(process.cwd(), target);
|
|
3779
|
+
// Record the fire in the flight recorder: the skill NAME is the parent dir of
|
|
3780
|
+
// its SKILL.md (skills/<name>/SKILL.md), falling back to the raw target.
|
|
3781
|
+
const parts = target.replace(/\\/g, "/").split("/").filter(Boolean);
|
|
3782
|
+
const name = parts.length >= 2 ? parts[parts.length - 2] : (parts[0] ?? target);
|
|
3783
|
+
(0, observe_js_1.appendObservation)({ kind: "skill", name, fired: true });
|
|
3865
3784
|
console.log(`Active skill: ${target}`);
|
|
3866
3785
|
}
|
|
3867
3786
|
/**
|
|
@@ -3962,6 +3881,13 @@ function agentHookCommand() {
|
|
|
3962
3881
|
return;
|
|
3963
3882
|
const decision = (0, agent_runtime_js_1.evaluatePreToolUse)(cwd, tool, command);
|
|
3964
3883
|
if (!decision.allow) {
|
|
3884
|
+
(0, observe_js_1.appendObservation)({
|
|
3885
|
+
kind: "agent",
|
|
3886
|
+
name: (0, agent_runtime_js_1.readActiveAgent)(cwd) ?? "unknown",
|
|
3887
|
+
tool,
|
|
3888
|
+
allowed: false,
|
|
3889
|
+
reason: decision.message,
|
|
3890
|
+
});
|
|
3965
3891
|
console.error(decision.message);
|
|
3966
3892
|
process.exit(2);
|
|
3967
3893
|
}
|
|
@@ -4510,10 +4436,26 @@ function emitGate(decision, on, mode, file) {
|
|
|
4510
4436
|
const action = (0, hook_program_js_1.gateAction)(decision, mode);
|
|
4511
4437
|
switch (action.kind) {
|
|
4512
4438
|
case "block":
|
|
4439
|
+
(0, observe_js_1.appendObservation)({
|
|
4440
|
+
kind: "hook",
|
|
4441
|
+
event: on,
|
|
4442
|
+
decision: "deny",
|
|
4443
|
+
mode: "enforce",
|
|
4444
|
+
rule: file,
|
|
4445
|
+
reason: action.reason,
|
|
4446
|
+
});
|
|
4513
4447
|
console.error(action.reason);
|
|
4514
4448
|
process.exit(2);
|
|
4515
4449
|
return;
|
|
4516
4450
|
case "ask":
|
|
4451
|
+
(0, observe_js_1.appendObservation)({
|
|
4452
|
+
kind: "hook",
|
|
4453
|
+
event: on,
|
|
4454
|
+
decision: "ask",
|
|
4455
|
+
mode: "enforce",
|
|
4456
|
+
rule: file,
|
|
4457
|
+
reason: action.reason,
|
|
4458
|
+
});
|
|
4517
4459
|
process.stdout.write(JSON.stringify({
|
|
4518
4460
|
hookSpecificOutput: {
|
|
4519
4461
|
hookEventName: on,
|
|
@@ -4523,6 +4465,14 @@ function emitGate(decision, on, mode, file) {
|
|
|
4523
4465
|
}) + "\n");
|
|
4524
4466
|
return;
|
|
4525
4467
|
case "observe":
|
|
4468
|
+
(0, observe_js_1.appendObservation)({
|
|
4469
|
+
kind: "hook",
|
|
4470
|
+
event: on,
|
|
4471
|
+
decision: action.would,
|
|
4472
|
+
mode: "observe",
|
|
4473
|
+
rule: file,
|
|
4474
|
+
reason: action.reason,
|
|
4475
|
+
});
|
|
4526
4476
|
recordObservation(file, on, action.would, action.reason);
|
|
4527
4477
|
console.error(`⚠ [vigiles observe] ${on}: would ${action.would} — ${action.reason}`);
|
|
4528
4478
|
return; // exit 0 — observe never blocks
|
|
@@ -5029,10 +4979,10 @@ async function main() {
|
|
|
5029
4979
|
break;
|
|
5030
4980
|
}
|
|
5031
4981
|
case "test":
|
|
5032
|
-
handleRunScripts("test", args, restArgs);
|
|
4982
|
+
await handleRunScripts("test", args, restArgs);
|
|
5033
4983
|
break;
|
|
5034
4984
|
case "eval":
|
|
5035
|
-
handleRunScripts("eval", args, restArgs);
|
|
4985
|
+
await handleRunScripts("eval", args, restArgs);
|
|
5036
4986
|
break;
|
|
5037
4987
|
case "audit": {
|
|
5038
4988
|
// The Lighthouse run: a plain `audit` is a deterministic READ — rings, each
|
|
@@ -5112,10 +5062,14 @@ async function main() {
|
|
|
5112
5062
|
// Surfaced in the AuditReport (the report's "Create spec" command-emit
|
|
5113
5063
|
// buttons read it) and the terminal nudge below.
|
|
5114
5064
|
const adoptableSurfaces = discoverAdoptableForAudit(root, adapter.layout.instructionFile);
|
|
5065
|
+
// Read the local flight recorder ONCE — feeds both the JSON report
|
|
5066
|
+
// (structured summary, the product boundary) and the terminal render.
|
|
5067
|
+
const ledgerRecords = (0, observe_js_1.readObservations)(root);
|
|
5115
5068
|
const auditReport = (0, audit_report_js_1.buildAuditReport)(report, {
|
|
5116
5069
|
harness: adapter.name,
|
|
5117
5070
|
vigilesVersion: getVersion(),
|
|
5118
5071
|
adoptableSurfaces,
|
|
5072
|
+
observations: (0, observe_js_1.summarizeObservations)(ledgerRecords),
|
|
5119
5073
|
});
|
|
5120
5074
|
const sc = auditReport.score;
|
|
5121
5075
|
const plan = (0, optimize_js_1.optimize)(report);
|
|
@@ -5146,6 +5100,12 @@ async function main() {
|
|
|
5146
5100
|
.length);
|
|
5147
5101
|
if (fireNudge)
|
|
5148
5102
|
console.log("\n" + fireNudge);
|
|
5103
|
+
// The flight recorder: a compact summary of what the harness actually
|
|
5104
|
+
// DID in real sessions (hook/agent decisions), read off the local
|
|
5105
|
+
// agent-readable ledger. Empty (skipped) until something is recorded.
|
|
5106
|
+
const ledgerSummary = (0, observe_js_1.formatLedgerSummary)(ledgerRecords);
|
|
5107
|
+
if (ledgerSummary)
|
|
5108
|
+
console.log("\n" + ledgerSummary);
|
|
5149
5109
|
}
|
|
5150
5110
|
// ONE read-vs-run decision for the EXECUTING checks (live MCP + skill
|
|
5151
5111
|
// firing). A plain `audit` is a deterministic READ; these run only on
|
|
@@ -5179,6 +5139,21 @@ async function main() {
|
|
|
5179
5139
|
// default; `--fail-on-widen` exits non-zero (the opt-in CI gate).
|
|
5180
5140
|
const beforeReport = (0, scan_js_1.scanPlugin)((0, node_path_1.resolve)(capBase), adapter.layout, adapter.dialect);
|
|
5181
5141
|
const diff = (0, capability_diff_js_1.diffCapabilities)(capabilitiesOfReport(beforeReport, adapter.dialect), capabilitiesOfReport(report, adapter.dialect));
|
|
5142
|
+
// Feed the flight recorder: the blast-radius change (moat #2) as a record.
|
|
5143
|
+
// Write to the AUDITED root's ledger (not the caller's cwd) — the same
|
|
5144
|
+
// `root` the audit reads back via `readObservations(root)`, so a
|
|
5145
|
+
// `vigiles audit ./after --capability-diff=./before` from a parent dir
|
|
5146
|
+
// records into ./after/.vigiles/, not the parent workspace.
|
|
5147
|
+
(0, observe_js_1.appendObservation)({
|
|
5148
|
+
kind: "capability-diff",
|
|
5149
|
+
added: [
|
|
5150
|
+
...diff.addedSideEffecting,
|
|
5151
|
+
...diff.addedUnknown,
|
|
5152
|
+
...diff.addedReadOnly,
|
|
5153
|
+
],
|
|
5154
|
+
removed: [...diff.removed],
|
|
5155
|
+
widened: diff.widened,
|
|
5156
|
+
}, root);
|
|
5182
5157
|
console.log(json
|
|
5183
5158
|
? JSON.stringify({ capabilityDiff: diff }, null, 2)
|
|
5184
5159
|
: "\n" + (0, capability_diff_js_1.formatCapabilityDiff)(diff));
|
|
@@ -5277,9 +5252,6 @@ async function main() {
|
|
|
5277
5252
|
}
|
|
5278
5253
|
break;
|
|
5279
5254
|
}
|
|
5280
|
-
case "scaffold-test":
|
|
5281
|
-
handleScaffoldTest(restArgs, args);
|
|
5282
|
-
break;
|
|
5283
5255
|
// --- Plumbing ---
|
|
5284
5256
|
case "generate":
|
|
5285
5257
|
await handleGenerate(restArgs, args);
|
package/dist/eval.d.ts
CHANGED
|
@@ -270,6 +270,21 @@ export declare function spawnAgent(a: AgentRunArgs): Promise<RunOut>;
|
|
|
270
270
|
* working model auth (e.g. `ANTHROPIC_API_KEY`). Thin wrapper over
|
|
271
271
|
* `runEvalWith` with the real agent runner.
|
|
272
272
|
*/
|
|
273
|
+
/**
|
|
274
|
+
* A `SKILL.md` written straight into a run's cwd (via an arm's `files`) is NOT
|
|
275
|
+
* registered as a skill by the harness. Claude Code — and Codex — load skills only
|
|
276
|
+
* from a plugin / `.claude/skills` layout, so a bare cwd `SKILL.md` sits unread:
|
|
277
|
+
* the arm silently measures NOTHING (baseline and "skill" become the same run). This
|
|
278
|
+
* is the exact footgun the ecosystem benchmark hit — a skill file delivered where it
|
|
279
|
+
* can never activate, with no error. Detect it so {@link runEval} / {@link measureArms}
|
|
280
|
+
* can WARN and point at `pluginDir` (a real `--plugin-dir` install) or `skillsDir`.
|
|
281
|
+
*
|
|
282
|
+
* HIGH-PRECISION (don't cry wolf): only a file whose basename is `SKILL.md` AND that
|
|
283
|
+
* carries real skill frontmatter (a `---` block naming `name`/`description`) is flagged
|
|
284
|
+
* — so an empty scratch `SKILL.md` a task is asked to AUTHOR is never flagged. Pure +
|
|
285
|
+
* exported for testing.
|
|
286
|
+
*/
|
|
287
|
+
export declare function unregisteredSkillFiles(files: Record<string, string> | undefined): string[];
|
|
273
288
|
export declare function runEval<M extends Metrics>(spec: EvalSpec<M>): Promise<EvalReport>;
|
|
274
289
|
/** A task run N times, scored against a `Trace` check vocabulary. */
|
|
275
290
|
export interface MeasureSpec {
|
package/dist/eval.js
CHANGED
|
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
3
3
|
exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = void 0;
|
|
4
4
|
exports.resolveSpawnEnv = resolveSpawnEnv;
|
|
5
5
|
exports.spawnAgent = spawnAgent;
|
|
6
|
+
exports.unregisteredSkillFiles = unregisteredSkillFiles;
|
|
6
7
|
exports.runEval = runEval;
|
|
7
8
|
exports.measureWith = measureWith;
|
|
8
9
|
exports.measure = measure;
|
|
@@ -65,6 +66,7 @@ const node_child_process_1 = require("node:child_process");
|
|
|
65
66
|
const node_fs_1 = require("node:fs");
|
|
66
67
|
const node_os_1 = require("node:os");
|
|
67
68
|
const node_path_1 = require("node:path");
|
|
69
|
+
const observe_js_1 = require("./observe.js");
|
|
68
70
|
const plugin_loader_js_1 = require("./adapters/claude-code/plugin-loader.js");
|
|
69
71
|
const runtime_js_1 = require("./adapters/claude-code/runtime.js");
|
|
70
72
|
const eval_cost_js_1 = require("./eval-cost.js");
|
|
@@ -144,7 +146,48 @@ function spawnAgent(a) {
|
|
|
144
146
|
* working model auth (e.g. `ANTHROPIC_API_KEY`). Thin wrapper over
|
|
145
147
|
* `runEvalWith` with the real agent runner.
|
|
146
148
|
*/
|
|
149
|
+
/**
|
|
150
|
+
* A `SKILL.md` written straight into a run's cwd (via an arm's `files`) is NOT
|
|
151
|
+
* registered as a skill by the harness. Claude Code — and Codex — load skills only
|
|
152
|
+
* from a plugin / `.claude/skills` layout, so a bare cwd `SKILL.md` sits unread:
|
|
153
|
+
* the arm silently measures NOTHING (baseline and "skill" become the same run). This
|
|
154
|
+
* is the exact footgun the ecosystem benchmark hit — a skill file delivered where it
|
|
155
|
+
* can never activate, with no error. Detect it so {@link runEval} / {@link measureArms}
|
|
156
|
+
* can WARN and point at `pluginDir` (a real `--plugin-dir` install) or `skillsDir`.
|
|
157
|
+
*
|
|
158
|
+
* HIGH-PRECISION (don't cry wolf): only a file whose basename is `SKILL.md` AND that
|
|
159
|
+
* carries real skill frontmatter (a `---` block naming `name`/`description`) is flagged
|
|
160
|
+
* — so an empty scratch `SKILL.md` a task is asked to AUTHOR is never flagged. Pure +
|
|
161
|
+
* exported for testing.
|
|
162
|
+
*/
|
|
163
|
+
function unregisteredSkillFiles(files) {
|
|
164
|
+
if (files === undefined)
|
|
165
|
+
return [];
|
|
166
|
+
return Object.entries(files)
|
|
167
|
+
.filter(([p, c]) => skillBasename(p) && hasSkillFrontmatter(c))
|
|
168
|
+
.map(([p]) => p);
|
|
169
|
+
}
|
|
170
|
+
function skillBasename(path) {
|
|
171
|
+
return (path.split(/[\\/]/).pop() ?? path) === "SKILL.md";
|
|
172
|
+
}
|
|
173
|
+
function hasSkillFrontmatter(content) {
|
|
174
|
+
const m = /^\uFEFF?\s*---\s*\r?\n([\s\S]*?)\r?\n---/.exec(content);
|
|
175
|
+
return m !== null && /(^|\n)\s*(name|description)\s*:/.test(m[1]);
|
|
176
|
+
}
|
|
177
|
+
/** Warn (loud, non-fatal) for every arm that drops an unregistered skill file. */
|
|
178
|
+
function warnUnregisteredSkillArms(arms) {
|
|
179
|
+
for (const [name, arm] of Object.entries(arms)) {
|
|
180
|
+
for (const path of unregisteredSkillFiles(arm.files)) {
|
|
181
|
+
console.warn(`⚠ eval arm "${name}": files["${path}"] is a SKILL.md with skill ` +
|
|
182
|
+
`frontmatter, but a SKILL.md written to the run cwd is NOT registered as a ` +
|
|
183
|
+
`skill by the harness — it never activates, so this arm measures nothing. ` +
|
|
184
|
+
`Install it via \`pluginDir\` (a real --plugin-dir plugin) or \`skillsDir\`, ` +
|
|
185
|
+
`not \`files\`. See docs/harness-testing.md.`);
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
}
|
|
147
189
|
async function runEval(spec) {
|
|
190
|
+
warnUnregisteredSkillArms(spec.arms);
|
|
148
191
|
const report = await runEvalWith(spec, spawnAgent);
|
|
149
192
|
// Surface what the run spent — tokens + API-equivalent $, and a LOUD warning if
|
|
150
193
|
// it was billed to a metered API key instead of the subscription. See eval-cost.ts.
|
|
@@ -287,6 +330,7 @@ function stubArmPluginDirs(arms) {
|
|
|
287
330
|
/* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
|
|
288
331
|
/** Score checks across arms against the real `claude` CLI. */
|
|
289
332
|
async function measureArms(spec) {
|
|
333
|
+
warnUnregisteredSkillArms(spec.arms);
|
|
290
334
|
const report = await measureArmsWith(spec, spawnAgent);
|
|
291
335
|
// Sum every arm's spend — an A/B run pays for both arms.
|
|
292
336
|
(0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.sumCosts)(Object.values(report.arms).map((a) => (0, eval_cost_js_1.costFromArm)(a.usage))));
|
|
@@ -1677,6 +1721,22 @@ async function measureTriggerRate(spec, opts = {}) {
|
|
|
1677
1721
|
const report = await measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
|
|
1678
1722
|
// Surface what the run spent (tokens + API-equivalent $ + metered warning).
|
|
1679
1723
|
(0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromArm)(report.usage));
|
|
1724
|
+
// Feed the flight recorder: recall (+ precision when measured) for this skill.
|
|
1725
|
+
const evalName = spec.name ?? "trigger-rate";
|
|
1726
|
+
(0, observe_js_1.appendObservation)({
|
|
1727
|
+
kind: "eval",
|
|
1728
|
+
name: evalName,
|
|
1729
|
+
metric: "recall",
|
|
1730
|
+
value: report.rate,
|
|
1731
|
+
});
|
|
1732
|
+
if (report.precision !== undefined) {
|
|
1733
|
+
(0, observe_js_1.appendObservation)({
|
|
1734
|
+
kind: "eval",
|
|
1735
|
+
name: evalName,
|
|
1736
|
+
metric: "precision",
|
|
1737
|
+
value: report.precision,
|
|
1738
|
+
});
|
|
1739
|
+
}
|
|
1680
1740
|
return report;
|
|
1681
1741
|
}
|
|
1682
1742
|
/* v8 ignore stop */
|
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
/** Bumped when the record shape changes in a non-additive way. */
|
|
2
|
+
export declare const OBSERVE_VERSION = 1;
|
|
3
|
+
/** The ledger filename under the `.vigiles/` directory. */
|
|
4
|
+
export declare const LEDGER_FILE = "runs.jsonl";
|
|
5
|
+
/** Fields every record carries; `v`/`ts` are stamped by the writer, not the caller. */
|
|
6
|
+
export interface ObservationBase {
|
|
7
|
+
/** schema version (`OBSERVE_VERSION`) */
|
|
8
|
+
v: number;
|
|
9
|
+
/** ISO-8601 timestamp */
|
|
10
|
+
ts: string;
|
|
11
|
+
}
|
|
12
|
+
/** A gate/hook decision observed in a real session. */
|
|
13
|
+
export interface HookObservation extends ObservationBase {
|
|
14
|
+
kind: "hook";
|
|
15
|
+
event: string;
|
|
16
|
+
decision: "allow" | "deny" | "ask";
|
|
17
|
+
/** the compiled-hook rule/name, when known */
|
|
18
|
+
rule?: string;
|
|
19
|
+
/** enforce actually blocked; observe recorded a would-be block */
|
|
20
|
+
mode?: "enforce" | "observe";
|
|
21
|
+
/** the bash command inspected, when the gate keyed on one */
|
|
22
|
+
cmd?: string;
|
|
23
|
+
reason?: string;
|
|
24
|
+
}
|
|
25
|
+
/** A subagent tool-contract decision (the PreToolUse rail). */
|
|
26
|
+
export interface AgentObservation extends ObservationBase {
|
|
27
|
+
kind: "agent";
|
|
28
|
+
/** the dispatched subagent */
|
|
29
|
+
name: string;
|
|
30
|
+
tool: string;
|
|
31
|
+
allowed: boolean;
|
|
32
|
+
reason?: string;
|
|
33
|
+
}
|
|
34
|
+
/** Whether a skill fired for a turn (behavioral surface — best-effort per harness). */
|
|
35
|
+
export interface SkillObservation extends ObservationBase {
|
|
36
|
+
kind: "skill";
|
|
37
|
+
name: string;
|
|
38
|
+
fired: boolean;
|
|
39
|
+
}
|
|
40
|
+
/** A measured eval outcome (recall, cost, a check rate, …). */
|
|
41
|
+
export interface EvalObservation extends ObservationBase {
|
|
42
|
+
kind: "eval";
|
|
43
|
+
name: string;
|
|
44
|
+
metric: string;
|
|
45
|
+
value: number;
|
|
46
|
+
}
|
|
47
|
+
/** A capability/blast-radius change observed at PR time. */
|
|
48
|
+
export interface CapabilityDiffObservation extends ObservationBase {
|
|
49
|
+
kind: "capability-diff";
|
|
50
|
+
pr?: number;
|
|
51
|
+
added: string[];
|
|
52
|
+
removed?: string[];
|
|
53
|
+
/** true when the change loosened the agent's effect surface */
|
|
54
|
+
widened: boolean;
|
|
55
|
+
}
|
|
56
|
+
/** The discriminated union every reader narrows on `kind`. */
|
|
57
|
+
export type ObservationRecord = HookObservation | AgentObservation | SkillObservation | EvalObservation | CapabilityDiffObservation;
|
|
58
|
+
/** What a caller supplies — the writer stamps `v` + `ts`. */
|
|
59
|
+
export type ObservationInput = Omit<HookObservation, "v" | "ts"> | Omit<AgentObservation, "v" | "ts"> | Omit<SkillObservation, "v" | "ts"> | Omit<EvalObservation, "v" | "ts"> | Omit<CapabilityDiffObservation, "v" | "ts">;
|
|
60
|
+
/** Serialize one record to a single JSONL line (trailing newline included). */
|
|
61
|
+
export declare function formatObservation(record: ObservationRecord): string;
|
|
62
|
+
/**
|
|
63
|
+
* Append one observation to `<cwd>/.vigiles/runs.jsonl`. Best-effort: any failure is
|
|
64
|
+
* swallowed so recording can never break a live session (the `ts`/`clock` here is a
|
|
65
|
+
* runtime side effect, intentionally — this module records reality, it is not a spec).
|
|
66
|
+
*/
|
|
67
|
+
export declare function appendObservation(input: ObservationInput, cwd?: string): void;
|
|
68
|
+
/**
|
|
69
|
+
* Read the ledger back. Tolerant by design: a malformed or partially-written line is
|
|
70
|
+
* skipped rather than throwing, so a torn append never nukes the whole read.
|
|
71
|
+
*/
|
|
72
|
+
export declare function readObservations(cwd?: string): ObservationRecord[];
|
|
73
|
+
/** Filter the ledger to one record kind, narrowing the type for the caller. */
|
|
74
|
+
export declare function observationsOfKind<K extends ObservationRecord["kind"]>(records: readonly ObservationRecord[], kind: K): Extract<ObservationRecord, {
|
|
75
|
+
kind: K;
|
|
76
|
+
}>[];
|
|
77
|
+
/** A denial rendered as a structured `{label, reason}` — the shared shape the
|
|
78
|
+
* terminal line and the JSON summary both derive from (one-detector-no-drift). */
|
|
79
|
+
export interface LedgerDenial {
|
|
80
|
+
readonly label: string;
|
|
81
|
+
readonly reason: string;
|
|
82
|
+
}
|
|
83
|
+
/** Per-kind record count. */
|
|
84
|
+
export interface LedgerCount {
|
|
85
|
+
readonly kind: ObservationRecord["kind"];
|
|
86
|
+
readonly count: number;
|
|
87
|
+
}
|
|
88
|
+
/** The structured ledger summary carried in the versioned AuditReport JSON. */
|
|
89
|
+
export interface LedgerSummary {
|
|
90
|
+
readonly total: number;
|
|
91
|
+
readonly counts: readonly LedgerCount[];
|
|
92
|
+
readonly denials: number;
|
|
93
|
+
/** The most recent denials (blocked gates / out-of-contract tool calls). */
|
|
94
|
+
readonly recentDenials: readonly LedgerDenial[];
|
|
95
|
+
}
|
|
96
|
+
/**
|
|
97
|
+
* The structured ledger summary for the AuditReport JSON — total, per-kind counts,
|
|
98
|
+
* and the recent denials. `undefined` when nothing is recorded, so the report field
|
|
99
|
+
* stays absent (additive/optional). Shares `isDenial`/`denialParts` with the
|
|
100
|
+
* terminal `formatLedgerSummary` so the two can't drift.
|
|
101
|
+
*/
|
|
102
|
+
export declare function summarizeObservations(records: readonly ObservationRecord[]): LedgerSummary | undefined;
|
|
103
|
+
/**
|
|
104
|
+
* A compact human summary of the ledger for `vigiles audit` — total, counts by kind,
|
|
105
|
+
* and the recent high-signal denials. Empty string when there is nothing recorded, so
|
|
106
|
+
* the caller can skip the section entirely.
|
|
107
|
+
*/
|
|
108
|
+
export declare function formatLedgerSummary(records: readonly ObservationRecord[]): string;
|
|
109
|
+
//# sourceMappingURL=observe.d.ts.map
|