vigiles 12.1.0 → 12.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,7 +11,7 @@
11
11
  * recognizes exactly these, so this list can't silently drift from the code.
12
12
  */
13
13
  /** Human-facing verbs (printed in help; typed by a human/agent/CI). */
14
- export declare const VERBS: readonly ["init", "compile", "eject", "lint", "test", "eval", "audit", "scaffold-test", "generate", "hook-runtime"];
14
+ export declare const VERBS: readonly ["init", "compile", "eject", "lint", "test", "eval", "audit", "generate", "hook-runtime"];
15
15
  /** Runtime entrypoint kinds under `vigiles hook-runtime <kind>` (emitted, not typed). */
16
16
  export declare const HOOK_RUNTIME_KINDS: readonly ["run-program", "agent", "agent-start", "agent-done", "skill", "skill-tool", "skill-start", "skill-done", "run-skill", "intercept-tool", "guard", "action", "refs", "eval-lock-nudge", "effect-enter", "effect-exit"];
17
17
  export type Verb = (typeof VERBS)[number];
@@ -22,7 +22,6 @@ exports.VERBS = [
22
22
  "test",
23
23
  "eval",
24
24
  "audit",
25
- "scaffold-test",
26
25
  "generate",
27
26
  "hook-runtime",
28
27
  ];
package/dist/cli.js CHANGED
@@ -22,8 +22,6 @@ const cli_flags_js_1 = require("./cli-flags.js");
22
22
  const setup_plan_js_1 = require("./setup-plan.js");
23
23
  const types_js_1 = require("./core/types.js");
24
24
  const test_coverage_js_1 = require("./test-coverage.js");
25
- const scaffold_test_js_1 = require("./scaffold-test.js");
26
- const effects_js_1 = require("./core/effects.js");
27
25
  const scan_js_1 = require("./scan.js");
28
26
  const scan_trigger_suggest_js_1 = require("./scan-trigger-suggest.js");
29
27
  const dialect_drift_js_1 = require("./dialect-drift.js");
@@ -52,6 +50,7 @@ const hook_install_js_1 = require("./hook-install.js");
52
50
  const hook_providers_js_1 = require("./core/hook-providers.js");
53
51
  const toml_1 = require("@iarna/toml");
54
52
  const agent_runtime_js_1 = require("./adapters/claude-code/agent-runtime.js");
53
+ const observe_js_1 = require("./observe.js");
55
54
  const effect_region_js_1 = require("./adapters/claude-code/effect-region.js");
56
55
  const tool_intercept_js_1 = require("./tool-intercept.js");
57
56
  const refs_js_1 = require("./core/refs.js");
@@ -3488,11 +3487,29 @@ async function handleGenerateHarness(args, restArgs) {
3488
3487
  console.log(`\n✓ Generated ${(0, generate_harness_js_1.labelFor)(process.cwd(), fullOut)}`);
3489
3488
  console.log(" `tsc --noEmit` over this file now checks every delegate target resolves.");
3490
3489
  }
3490
+ /** Minimal TTY yes/no prompt (readline). Returns true only on an explicit y/yes. */
3491
+ async function promptYesNo(question) {
3492
+ const readline = await import("node:readline");
3493
+ const rl = readline.createInterface({
3494
+ input: process.stdin,
3495
+ output: process.stdout,
3496
+ });
3497
+ try {
3498
+ const answer = await new Promise((res) => {
3499
+ rl.question(question, res);
3500
+ });
3501
+ return /^y(es)?$/i.test(answer.trim());
3502
+ }
3503
+ finally {
3504
+ rl.close();
3505
+ }
3506
+ }
3491
3507
  /**
3492
3508
  * `vigiles test` / `vigiles eval` — discover and run the two-tier harness
3493
3509
  * scripts (deterministic `*.harness.mjs` / real-model `*.eval.mjs`) as child
3494
3510
  * `node` processes, aggregating exit codes so they work as a CI command. See
3495
- * src/run-scripts.ts.
3511
+ * src/run-scripts.ts. A bare `vigiles eval` (no target) asks before fanning out
3512
+ * over the whole tree — it spends model quota (see `decideRunScripts`).
3496
3513
  *
3497
3514
  * `vigiles test` skips clean when the `claude` CLI is absent (the deterministic
3498
3515
  * tier needs it, just like the node:test suite). `--trials=N` is forwarded to
@@ -3533,7 +3550,7 @@ function resolveEvalLockEnv(args) {
3533
3550
  }
3534
3551
  return env;
3535
3552
  }
3536
- function handleRunScripts(kind, args, restArgs) {
3553
+ async function handleRunScripts(kind, args, restArgs) {
3537
3554
  const cwd = process.cwd();
3538
3555
  // Harness/eval scripts may be authored in JS or TS (see run-scripts.ts).
3539
3556
  const defaultGlob = (0, run_scripts_js_1.scriptGlob)(kind === "test" ? "harness" : "eval");
@@ -3564,6 +3581,33 @@ function handleRunScripts(kind, args, restArgs) {
3564
3581
  console.log(`No ${defaultGlob} files found.`);
3565
3582
  return;
3566
3583
  }
3584
+ // Consent gate for a bare `vigiles eval`: it runs the REAL model on your
3585
+ // subscription, and a no-target run discovered the whole tree — so never fan out
3586
+ // over an unbounded glob without explicit intent. Mirrors audit's read-vs-run
3587
+ // consent. `test` is free → always runs (decideRunScripts returns "run").
3588
+ const runDecision = (0, run_scripts_js_1.decideRunScripts)({
3589
+ kind,
3590
+ explicitTargets: restArgs.length > 0,
3591
+ matchedCount: files.length,
3592
+ isTTY: (process.stdin.isTTY ?? false) && (process.stdout.isTTY ?? false),
3593
+ all: args.includes("--all"),
3594
+ yes: args.includes("--yes") || args.includes("--no-interactive"),
3595
+ });
3596
+ if (runDecision.kind === "refuse") {
3597
+ console.error(`✗ vigiles eval: ${String(runDecision.count)} eval file(s) matched the whole tree, and each ` +
3598
+ "runs the real model on your subscription. Refusing to fire them all non-interactively.\n" +
3599
+ " → name the eval(s): vigiles eval path/to/x.eval.mjs\n" +
3600
+ " → or opt in to all: vigiles eval --all");
3601
+ process.exit(2);
3602
+ }
3603
+ if (runDecision.kind === "confirm") {
3604
+ const ok = await promptYesNo(`About to run ${String(runDecision.count)} eval file(s) against the real model on your ` +
3605
+ "subscription (uses your Claude quota). Continue? [y/N] ");
3606
+ if (!ok) {
3607
+ console.log("Aborted. Name specific eval(s), or pass --all to run them all.");
3608
+ return;
3609
+ }
3610
+ }
3567
3611
  // No blanket skip: unit-tier (runHook) tests need no `claude`, so always run.
3568
3612
  // A script whose tier DOES need `claude` self-reports `⊘ SKIPPED` (exit 77) —
3569
3613
  // loud, never a silent green. Just flag up front that some may skip.
@@ -3611,135 +3655,6 @@ function capabilitiesOfReport(report, dialect) {
3611
3655
  }));
3612
3656
  return (0, generate_harness_js_1.computeHarnessCapabilities)(agents, dialect);
3613
3657
  }
3614
- /**
3615
- * The plugin's declared name for the namespaced skill id, read from the layout's
3616
- * manifest (adapter-aware path, not a hardcoded `.claude-plugin/`), falling back to
3617
- * the dir basename. JSON manifests only for now (a TOML/Codex manifest → basename).
3618
- */
3619
- function pluginNameFor(dir, manifestPath) {
3620
- try {
3621
- const manifest = JSON.parse((0, node_fs_1.readFileSync)((0, node_path_1.resolve)(dir, manifestPath), "utf-8"));
3622
- if (typeof manifest.name === "string" && manifest.name)
3623
- return manifest.name;
3624
- }
3625
- catch {
3626
- /* missing / non-JSON manifest → fall back */
3627
- }
3628
- return (0, node_path_1.basename)(dir);
3629
- }
3630
- /** Enrich an untested Surface with the metadata the right template needs. */
3631
- /** Extract `"name": type` fields from one rendered `vigiles:ok`/`err` shape block. */
3632
- function parseContractFields(block) {
3633
- const fields = [];
3634
- const re = /"([^"]+)"\s*:\s*(string\[\]|string|number|boolean)/g;
3635
- let m;
3636
- while ((m = re.exec(block)) !== null) {
3637
- fields.push({ name: m[1], type: m[2] });
3638
- }
3639
- return fields;
3640
- }
3641
- /**
3642
- * Parse a subagent's compiled `## Output contract` (the `vigiles:ok` / `vigiles:err`
3643
- * blocks the compiler emits) back into a typed `ResultContract`, so the generator
3644
- * can write an `assertAgentOk` test against the real fields. Returns null when the
3645
- * agent has no result() contract.
3646
- */
3647
- function parseResultContract(md) {
3648
- const ok = /```vigiles:ok\n([\s\S]*?)```/.exec(md);
3649
- const err = /```vigiles:err\n([\s\S]*?)```/.exec(md);
3650
- if (!ok && !err)
3651
- return null;
3652
- const okFields = ok ? parseContractFields(ok[1]) : [];
3653
- const errFields = err ? parseContractFields(err[1]) : [];
3654
- if (okFields.length === 0 && errFields.length === 0)
3655
- return null;
3656
- return { ok: okFields, err: errFields };
3657
- }
3658
- function scaffoldInputFor(s, report, pluginName, dir, dialect) {
3659
- const base = { kind: s.kind, name: s.name, path: s.path };
3660
- switch (s.kind) {
3661
- case "skill": {
3662
- const sk = report.skills.find((x) => x.name === s.name);
3663
- return { ...base, pluginName, userInvoked: sk?.userInvoked };
3664
- }
3665
- case "agent": {
3666
- const ag = report.agents.find((x) => x.name === s.name);
3667
- const tools = ag?.tools ?? null;
3668
- const sideEffectingTools = tools
3669
- ? (0, effects_js_1.effectSurface)(tools, dialect).sideEffecting
3670
- : undefined;
3671
- let resultContract = null;
3672
- try {
3673
- resultContract = parseResultContract((0, node_fs_1.readFileSync)((0, node_path_1.resolve)(dir, s.path), "utf-8"));
3674
- }
3675
- catch {
3676
- // agent .md unreadable → no contract to generate against
3677
- }
3678
- return { ...base, tools, sideEffectingTools, resultContract };
3679
- }
3680
- case "hook":
3681
- return { ...base, hookCommand: `bash ${s.path}` };
3682
- }
3683
- }
3684
- /**
3685
- * `vigiles scaffold-test [dir]` — generate a runnable STARTER test for each
3686
- * untested skill/agent/hook (B1, test-gen from free-form). Reuses the
3687
- * untested-surface detector for the list + `scan` for the metadata, then emits the
3688
- * cheapest meaningful tier per kind (hook → `runHook`, skill → `measureTriggerRate`,
3689
- * subagent → `runHarnessTest`) at the surface's suggested test path. Dry-run by
3690
- * default (prints the scaffolds); `--write` creates the files (never clobbering an
3691
- * existing one); `--json` for the agent-consumable `{ path, content }[]`.
3692
- */
3693
- function handleScaffoldTest(restArgs, args) {
3694
- const dir = (0, node_path_1.resolve)(restArgs[0] ?? ".");
3695
- const write = args.includes("--write");
3696
- const json = args.includes("--json");
3697
- const harnessFlag = harnessFlagFrom(args);
3698
- const adapter = harnessFlag
3699
- ? (0, adapter_registry_js_1.resolveAdapter)(dir, harnessFlag)
3700
- : (0, adapter_registry_js_1.detectAdapterResult)(dir).adapter;
3701
- const { untested } = (0, test_coverage_js_1.findUntestedSurfaces)({
3702
- basePath: dir,
3703
- layout: adapter.layout,
3704
- });
3705
- const report = (0, scan_js_1.scanPlugin)(dir, adapter.layout, adapter.dialect);
3706
- const pluginName = pluginNameFor(dir, adapter.layout.manifestPath);
3707
- const scaffolds = untested.map((s) => (0, scaffold_test_js_1.scaffoldTest)(scaffoldInputFor(s, report, pluginName, dir, adapter.dialect)));
3708
- if (json) {
3709
- console.log(JSON.stringify(scaffolds, null, 2));
3710
- return;
3711
- }
3712
- if (!write) {
3713
- console.log((0, scaffold_test_js_1.formatScaffolds)(scaffolds));
3714
- for (const s of scaffolds) {
3715
- console.log(`\n# ${s.path}\n`);
3716
- console.log(s.content);
3717
- }
3718
- if (scaffolds.length > 0) {
3719
- console.log("Re-run with --write to create these files.");
3720
- }
3721
- return;
3722
- }
3723
- const written = [];
3724
- const skipped = [];
3725
- for (const s of scaffolds) {
3726
- const target = (0, node_path_1.resolve)(dir, s.path);
3727
- if ((0, node_fs_1.existsSync)(target)) {
3728
- skipped.push(s.path);
3729
- continue;
3730
- }
3731
- (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(target), { recursive: true });
3732
- (0, node_fs_1.writeFileSync)(target, s.content);
3733
- written.push(s.path);
3734
- }
3735
- for (const p of written)
3736
- console.log(`✓ wrote ${p}`);
3737
- for (const p of skipped)
3738
- console.log(`⊘ skipped ${p} (already exists)`);
3739
- if (written.length === 0 && skipped.length === 0) {
3740
- console.log("Nothing to scaffold — every surface already has a test.");
3741
- }
3742
- }
3743
3658
  function printUsage(command) {
3744
3659
  console.log("vigiles — compile typed specs to instruction files");
3745
3660
  console.log("");
@@ -3756,7 +3671,6 @@ function printUsage(command) {
3756
3671
  console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
3757
3672
  console.log(" --update records each named eval's result to a committed lock (run locally on your subscription)");
3758
3673
  console.log(" --check verifies committed eval results against current inputs WITHOUT a model — the CI staleness gate");
3759
- console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
3760
3674
  console.log("");
3761
3675
  console.log("Examples:");
3762
3676
  console.log(" vigiles init Auto-detect project, create specs, wire CI");
@@ -3862,6 +3776,11 @@ function skillStartCommand(target) {
3862
3776
  process.exit(2);
3863
3777
  }
3864
3778
  (0, skill_runtime_js_1.setActiveSkill)(process.cwd(), target);
3779
+ // Record the fire in the flight recorder: the skill NAME is the parent dir of
3780
+ // its SKILL.md (skills/<name>/SKILL.md), falling back to the raw target.
3781
+ const parts = target.replace(/\\/g, "/").split("/").filter(Boolean);
3782
+ const name = parts.length >= 2 ? parts[parts.length - 2] : (parts[0] ?? target);
3783
+ (0, observe_js_1.appendObservation)({ kind: "skill", name, fired: true });
3865
3784
  console.log(`Active skill: ${target}`);
3866
3785
  }
3867
3786
  /**
@@ -3962,6 +3881,13 @@ function agentHookCommand() {
3962
3881
  return;
3963
3882
  const decision = (0, agent_runtime_js_1.evaluatePreToolUse)(cwd, tool, command);
3964
3883
  if (!decision.allow) {
3884
+ (0, observe_js_1.appendObservation)({
3885
+ kind: "agent",
3886
+ name: (0, agent_runtime_js_1.readActiveAgent)(cwd) ?? "unknown",
3887
+ tool,
3888
+ allowed: false,
3889
+ reason: decision.message,
3890
+ });
3965
3891
  console.error(decision.message);
3966
3892
  process.exit(2);
3967
3893
  }
@@ -4510,10 +4436,26 @@ function emitGate(decision, on, mode, file) {
4510
4436
  const action = (0, hook_program_js_1.gateAction)(decision, mode);
4511
4437
  switch (action.kind) {
4512
4438
  case "block":
4439
+ (0, observe_js_1.appendObservation)({
4440
+ kind: "hook",
4441
+ event: on,
4442
+ decision: "deny",
4443
+ mode: "enforce",
4444
+ rule: file,
4445
+ reason: action.reason,
4446
+ });
4513
4447
  console.error(action.reason);
4514
4448
  process.exit(2);
4515
4449
  return;
4516
4450
  case "ask":
4451
+ (0, observe_js_1.appendObservation)({
4452
+ kind: "hook",
4453
+ event: on,
4454
+ decision: "ask",
4455
+ mode: "enforce",
4456
+ rule: file,
4457
+ reason: action.reason,
4458
+ });
4517
4459
  process.stdout.write(JSON.stringify({
4518
4460
  hookSpecificOutput: {
4519
4461
  hookEventName: on,
@@ -4523,6 +4465,14 @@ function emitGate(decision, on, mode, file) {
4523
4465
  }) + "\n");
4524
4466
  return;
4525
4467
  case "observe":
4468
+ (0, observe_js_1.appendObservation)({
4469
+ kind: "hook",
4470
+ event: on,
4471
+ decision: action.would,
4472
+ mode: "observe",
4473
+ rule: file,
4474
+ reason: action.reason,
4475
+ });
4526
4476
  recordObservation(file, on, action.would, action.reason);
4527
4477
  console.error(`⚠ [vigiles observe] ${on}: would ${action.would} — ${action.reason}`);
4528
4478
  return; // exit 0 — observe never blocks
@@ -5029,10 +4979,10 @@ async function main() {
5029
4979
  break;
5030
4980
  }
5031
4981
  case "test":
5032
- handleRunScripts("test", args, restArgs);
4982
+ await handleRunScripts("test", args, restArgs);
5033
4983
  break;
5034
4984
  case "eval":
5035
- handleRunScripts("eval", args, restArgs);
4985
+ await handleRunScripts("eval", args, restArgs);
5036
4986
  break;
5037
4987
  case "audit": {
5038
4988
  // The Lighthouse run: a plain `audit` is a deterministic READ — rings, each
@@ -5112,10 +5062,14 @@ async function main() {
5112
5062
  // Surfaced in the AuditReport (the report's "Create spec" command-emit
5113
5063
  // buttons read it) and the terminal nudge below.
5114
5064
  const adoptableSurfaces = discoverAdoptableForAudit(root, adapter.layout.instructionFile);
5065
+ // Read the local flight recorder ONCE — feeds both the JSON report
5066
+ // (structured summary, the product boundary) and the terminal render.
5067
+ const ledgerRecords = (0, observe_js_1.readObservations)(root);
5115
5068
  const auditReport = (0, audit_report_js_1.buildAuditReport)(report, {
5116
5069
  harness: adapter.name,
5117
5070
  vigilesVersion: getVersion(),
5118
5071
  adoptableSurfaces,
5072
+ observations: (0, observe_js_1.summarizeObservations)(ledgerRecords),
5119
5073
  });
5120
5074
  const sc = auditReport.score;
5121
5075
  const plan = (0, optimize_js_1.optimize)(report);
@@ -5146,6 +5100,12 @@ async function main() {
5146
5100
  .length);
5147
5101
  if (fireNudge)
5148
5102
  console.log("\n" + fireNudge);
5103
+ // The flight recorder: a compact summary of what the harness actually
5104
+ // DID in real sessions (hook/agent decisions), read off the local
5105
+ // agent-readable ledger. Empty (skipped) until something is recorded.
5106
+ const ledgerSummary = (0, observe_js_1.formatLedgerSummary)(ledgerRecords);
5107
+ if (ledgerSummary)
5108
+ console.log("\n" + ledgerSummary);
5149
5109
  }
5150
5110
  // ONE read-vs-run decision for the EXECUTING checks (live MCP + skill
5151
5111
  // firing). A plain `audit` is a deterministic READ; these run only on
@@ -5179,6 +5139,21 @@ async function main() {
5179
5139
  // default; `--fail-on-widen` exits non-zero (the opt-in CI gate).
5180
5140
  const beforeReport = (0, scan_js_1.scanPlugin)((0, node_path_1.resolve)(capBase), adapter.layout, adapter.dialect);
5181
5141
  const diff = (0, capability_diff_js_1.diffCapabilities)(capabilitiesOfReport(beforeReport, adapter.dialect), capabilitiesOfReport(report, adapter.dialect));
5142
+ // Feed the flight recorder: the blast-radius change (moat #2) as a record.
5143
+ // Write to the AUDITED root's ledger (not the caller's cwd) — the same
5144
+ // `root` the audit reads back via `readObservations(root)`, so a
5145
+ // `vigiles audit ./after --capability-diff=./before` from a parent dir
5146
+ // records into ./after/.vigiles/, not the parent workspace.
5147
+ (0, observe_js_1.appendObservation)({
5148
+ kind: "capability-diff",
5149
+ added: [
5150
+ ...diff.addedSideEffecting,
5151
+ ...diff.addedUnknown,
5152
+ ...diff.addedReadOnly,
5153
+ ],
5154
+ removed: [...diff.removed],
5155
+ widened: diff.widened,
5156
+ }, root);
5182
5157
  console.log(json
5183
5158
  ? JSON.stringify({ capabilityDiff: diff }, null, 2)
5184
5159
  : "\n" + (0, capability_diff_js_1.formatCapabilityDiff)(diff));
@@ -5277,9 +5252,6 @@ async function main() {
5277
5252
  }
5278
5253
  break;
5279
5254
  }
5280
- case "scaffold-test":
5281
- handleScaffoldTest(restArgs, args);
5282
- break;
5283
5255
  // --- Plumbing ---
5284
5256
  case "generate":
5285
5257
  await handleGenerate(restArgs, args);
package/dist/eval.d.ts CHANGED
@@ -270,6 +270,21 @@ export declare function spawnAgent(a: AgentRunArgs): Promise<RunOut>;
270
270
  * working model auth (e.g. `ANTHROPIC_API_KEY`). Thin wrapper over
271
271
  * `runEvalWith` with the real agent runner.
272
272
  */
273
+ /**
274
+ * A `SKILL.md` written straight into a run's cwd (via an arm's `files`) is NOT
275
+ * registered as a skill by the harness. Claude Code — and Codex — load skills only
276
+ * from a plugin / `.claude/skills` layout, so a bare cwd `SKILL.md` sits unread:
277
+ * the arm silently measures NOTHING (baseline and "skill" become the same run). This
278
+ * is the exact footgun the ecosystem benchmark hit — a skill file delivered where it
279
+ * can never activate, with no error. Detect it so {@link runEval} / {@link measureArms}
280
+ * can WARN and point at `pluginDir` (a real `--plugin-dir` install) or `skillsDir`.
281
+ *
282
+ * HIGH-PRECISION (don't cry wolf): only a file whose basename is `SKILL.md` AND that
283
+ * carries real skill frontmatter (a `---` block naming `name`/`description`) is flagged
284
+ * — so an empty scratch `SKILL.md` a task is asked to AUTHOR is never flagged. Pure +
285
+ * exported for testing.
286
+ */
287
+ export declare function unregisteredSkillFiles(files: Record<string, string> | undefined): string[];
273
288
  export declare function runEval<M extends Metrics>(spec: EvalSpec<M>): Promise<EvalReport>;
274
289
  /** A task run N times, scored against a `Trace` check vocabulary. */
275
290
  export interface MeasureSpec {
package/dist/eval.js CHANGED
@@ -3,6 +3,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
3
3
  exports.claudeEvalDriver = exports.EPHEMERAL_HOME_KEEP = void 0;
4
4
  exports.resolveSpawnEnv = resolveSpawnEnv;
5
5
  exports.spawnAgent = spawnAgent;
6
+ exports.unregisteredSkillFiles = unregisteredSkillFiles;
6
7
  exports.runEval = runEval;
7
8
  exports.measureWith = measureWith;
8
9
  exports.measure = measure;
@@ -65,6 +66,7 @@ const node_child_process_1 = require("node:child_process");
65
66
  const node_fs_1 = require("node:fs");
66
67
  const node_os_1 = require("node:os");
67
68
  const node_path_1 = require("node:path");
69
+ const observe_js_1 = require("./observe.js");
68
70
  const plugin_loader_js_1 = require("./adapters/claude-code/plugin-loader.js");
69
71
  const runtime_js_1 = require("./adapters/claude-code/runtime.js");
70
72
  const eval_cost_js_1 = require("./eval-cost.js");
@@ -144,7 +146,48 @@ function spawnAgent(a) {
144
146
  * working model auth (e.g. `ANTHROPIC_API_KEY`). Thin wrapper over
145
147
  * `runEvalWith` with the real agent runner.
146
148
  */
149
+ /**
150
+ * A `SKILL.md` written straight into a run's cwd (via an arm's `files`) is NOT
151
+ * registered as a skill by the harness. Claude Code — and Codex — load skills only
152
+ * from a plugin / `.claude/skills` layout, so a bare cwd `SKILL.md` sits unread:
153
+ * the arm silently measures NOTHING (baseline and "skill" become the same run). This
154
+ * is the exact footgun the ecosystem benchmark hit — a skill file delivered where it
155
+ * can never activate, with no error. Detect it so {@link runEval} / {@link measureArms}
156
+ * can WARN and point at `pluginDir` (a real `--plugin-dir` install) or `skillsDir`.
157
+ *
158
+ * HIGH-PRECISION (don't cry wolf): only a file whose basename is `SKILL.md` AND that
159
+ * carries real skill frontmatter (a `---` block naming `name`/`description`) is flagged
160
+ * — so an empty scratch `SKILL.md` a task is asked to AUTHOR is never flagged. Pure +
161
+ * exported for testing.
162
+ */
163
+ function unregisteredSkillFiles(files) {
164
+ if (files === undefined)
165
+ return [];
166
+ return Object.entries(files)
167
+ .filter(([p, c]) => skillBasename(p) && hasSkillFrontmatter(c))
168
+ .map(([p]) => p);
169
+ }
170
+ function skillBasename(path) {
171
+ return (path.split(/[\\/]/).pop() ?? path) === "SKILL.md";
172
+ }
173
+ function hasSkillFrontmatter(content) {
174
+ const m = /^\uFEFF?\s*---\s*\r?\n([\s\S]*?)\r?\n---/.exec(content);
175
+ return m !== null && /(^|\n)\s*(name|description)\s*:/.test(m[1]);
176
+ }
177
+ /** Warn (loud, non-fatal) for every arm that drops an unregistered skill file. */
178
+ function warnUnregisteredSkillArms(arms) {
179
+ for (const [name, arm] of Object.entries(arms)) {
180
+ for (const path of unregisteredSkillFiles(arm.files)) {
181
+ console.warn(`⚠ eval arm "${name}": files["${path}"] is a SKILL.md with skill ` +
182
+ `frontmatter, but a SKILL.md written to the run cwd is NOT registered as a ` +
183
+ `skill by the harness — it never activates, so this arm measures nothing. ` +
184
+ `Install it via \`pluginDir\` (a real --plugin-dir plugin) or \`skillsDir\`, ` +
185
+ `not \`files\`. See docs/harness-testing.md.`);
186
+ }
187
+ }
188
+ }
147
189
  async function runEval(spec) {
190
+ warnUnregisteredSkillArms(spec.arms);
148
191
  const report = await runEvalWith(spec, spawnAgent);
149
192
  // Surface what the run spent — tokens + API-equivalent $, and a LOUD warning if
150
193
  // it was billed to a metered API key instead of the subscription. See eval-cost.ts.
@@ -287,6 +330,7 @@ function stubArmPluginDirs(arms) {
287
330
  /* v8 ignore start -- real claude subprocess; thin wrapper over measureArmsWith */
288
331
  /** Score checks across arms against the real `claude` CLI. */
289
332
  async function measureArms(spec) {
333
+ warnUnregisteredSkillArms(spec.arms);
290
334
  const report = await measureArmsWith(spec, spawnAgent);
291
335
  // Sum every arm's spend — an A/B run pays for both arms.
292
336
  (0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.sumCosts)(Object.values(report.arms).map((a) => (0, eval_cost_js_1.costFromArm)(a.usage))));
@@ -1677,6 +1721,22 @@ async function measureTriggerRate(spec, opts = {}) {
1677
1721
  const report = await measureTriggerRateWith(spec, d.runner, d.parse, d.runError, d.harness ?? "claude-code");
1678
1722
  // Surface what the run spent (tokens + API-equivalent $ + metered warning).
1679
1723
  (0, eval_cost_js_1.emitCostSummary)((0, eval_cost_js_1.costFromArm)(report.usage));
1724
+ // Feed the flight recorder: recall (+ precision when measured) for this skill.
1725
+ const evalName = spec.name ?? "trigger-rate";
1726
+ (0, observe_js_1.appendObservation)({
1727
+ kind: "eval",
1728
+ name: evalName,
1729
+ metric: "recall",
1730
+ value: report.rate,
1731
+ });
1732
+ if (report.precision !== undefined) {
1733
+ (0, observe_js_1.appendObservation)({
1734
+ kind: "eval",
1735
+ name: evalName,
1736
+ metric: "precision",
1737
+ value: report.precision,
1738
+ });
1739
+ }
1680
1740
  return report;
1681
1741
  }
1682
1742
  /* v8 ignore stop */
@@ -0,0 +1,109 @@
1
+ /** Bumped when the record shape changes in a non-additive way. */
2
+ export declare const OBSERVE_VERSION = 1;
3
+ /** The ledger filename under the `.vigiles/` directory. */
4
+ export declare const LEDGER_FILE = "runs.jsonl";
5
+ /** Fields every record carries; `v`/`ts` are stamped by the writer, not the caller. */
6
+ export interface ObservationBase {
7
+ /** schema version (`OBSERVE_VERSION`) */
8
+ v: number;
9
+ /** ISO-8601 timestamp */
10
+ ts: string;
11
+ }
12
+ /** A gate/hook decision observed in a real session. */
13
+ export interface HookObservation extends ObservationBase {
14
+ kind: "hook";
15
+ event: string;
16
+ decision: "allow" | "deny" | "ask";
17
+ /** the compiled-hook rule/name, when known */
18
+ rule?: string;
19
+ /** enforce actually blocked; observe recorded a would-be block */
20
+ mode?: "enforce" | "observe";
21
+ /** the bash command inspected, when the gate keyed on one */
22
+ cmd?: string;
23
+ reason?: string;
24
+ }
25
+ /** A subagent tool-contract decision (the PreToolUse rail). */
26
+ export interface AgentObservation extends ObservationBase {
27
+ kind: "agent";
28
+ /** the dispatched subagent */
29
+ name: string;
30
+ tool: string;
31
+ allowed: boolean;
32
+ reason?: string;
33
+ }
34
+ /** Whether a skill fired for a turn (behavioral surface — best-effort per harness). */
35
+ export interface SkillObservation extends ObservationBase {
36
+ kind: "skill";
37
+ name: string;
38
+ fired: boolean;
39
+ }
40
+ /** A measured eval outcome (recall, cost, a check rate, …). */
41
+ export interface EvalObservation extends ObservationBase {
42
+ kind: "eval";
43
+ name: string;
44
+ metric: string;
45
+ value: number;
46
+ }
47
+ /** A capability/blast-radius change observed at PR time. */
48
+ export interface CapabilityDiffObservation extends ObservationBase {
49
+ kind: "capability-diff";
50
+ pr?: number;
51
+ added: string[];
52
+ removed?: string[];
53
+ /** true when the change loosened the agent's effect surface */
54
+ widened: boolean;
55
+ }
56
+ /** The discriminated union every reader narrows on `kind`. */
57
+ export type ObservationRecord = HookObservation | AgentObservation | SkillObservation | EvalObservation | CapabilityDiffObservation;
58
+ /** What a caller supplies — the writer stamps `v` + `ts`. */
59
+ export type ObservationInput = Omit<HookObservation, "v" | "ts"> | Omit<AgentObservation, "v" | "ts"> | Omit<SkillObservation, "v" | "ts"> | Omit<EvalObservation, "v" | "ts"> | Omit<CapabilityDiffObservation, "v" | "ts">;
60
+ /** Serialize one record to a single JSONL line (trailing newline included). */
61
+ export declare function formatObservation(record: ObservationRecord): string;
62
+ /**
63
+ * Append one observation to `<cwd>/.vigiles/runs.jsonl`. Best-effort: any failure is
64
+ * swallowed so recording can never break a live session (the `ts`/`clock` here is a
65
+ * runtime side effect, intentionally — this module records reality, it is not a spec).
66
+ */
67
+ export declare function appendObservation(input: ObservationInput, cwd?: string): void;
68
+ /**
69
+ * Read the ledger back. Tolerant by design: a malformed or partially-written line is
70
+ * skipped rather than throwing, so a torn append never nukes the whole read.
71
+ */
72
+ export declare function readObservations(cwd?: string): ObservationRecord[];
73
+ /** Filter the ledger to one record kind, narrowing the type for the caller. */
74
+ export declare function observationsOfKind<K extends ObservationRecord["kind"]>(records: readonly ObservationRecord[], kind: K): Extract<ObservationRecord, {
75
+ kind: K;
76
+ }>[];
77
+ /** A denial rendered as a structured `{label, reason}` — the shared shape the
78
+ * terminal line and the JSON summary both derive from (one-detector-no-drift). */
79
+ export interface LedgerDenial {
80
+ readonly label: string;
81
+ readonly reason: string;
82
+ }
83
+ /** Per-kind record count. */
84
+ export interface LedgerCount {
85
+ readonly kind: ObservationRecord["kind"];
86
+ readonly count: number;
87
+ }
88
+ /** The structured ledger summary carried in the versioned AuditReport JSON. */
89
+ export interface LedgerSummary {
90
+ readonly total: number;
91
+ readonly counts: readonly LedgerCount[];
92
+ readonly denials: number;
93
+ /** The most recent denials (blocked gates / out-of-contract tool calls). */
94
+ readonly recentDenials: readonly LedgerDenial[];
95
+ }
96
+ /**
97
+ * The structured ledger summary for the AuditReport JSON — total, per-kind counts,
98
+ * and the recent denials. `undefined` when nothing is recorded, so the report field
99
+ * stays absent (additive/optional). Shares `isDenial`/`denialParts` with the
100
+ * terminal `formatLedgerSummary` so the two can't drift.
101
+ */
102
+ export declare function summarizeObservations(records: readonly ObservationRecord[]): LedgerSummary | undefined;
103
+ /**
104
+ * A compact human summary of the ledger for `vigiles audit` — total, counts by kind,
105
+ * and the recent high-signal denials. Empty string when there is nothing recorded, so
106
+ * the caller can skip the section entirely.
107
+ */
108
+ export declare function formatLedgerSummary(records: readonly ObservationRecord[]): string;
109
+ //# sourceMappingURL=observe.d.ts.map