vigiles 15.4.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -57,7 +57,7 @@ import {
57
57
  verifyGuardrail,
58
58
  formatGuardrailReport,
59
59
  // assertBlocksDisasters, // uncomment to gate CI on the battery (see below)
60
- } from "vigiles/unit";
60
+ } from "vigiles";
61
61
 
62
62
  const cmd = ${JSON.stringify(cmd)};
63
63
 
@@ -89,19 +89,15 @@ function skillScaffold(input) {
89
89
  ? "\n// NOTE: this skill is user-invoked (disableModelInvocation). Trigger-rate\n// measures MODEL-invocable skills; either make it model-invocable or test its\n// slash-command invocation with runHarnessTest instead.\n"
90
90
  : "";
91
91
  return `${header(`Starter trigger-rate eval for the \`${id}\` skill (recall + precision).`, `npx vigiles eval ${suggestedPath(input)} # real model, on your subscription`)}
92
- import {
93
- measureTriggerRate,
94
- formatTriggerRateReport,
95
- assertTriggerRate,
96
- skillResolved,
97
- } from "vigiles/testing";
92
+ import { paid_measureTriggerRate } from "vigiles/eval"; // paid_ = a real model runs
93
+ import { formatTriggerRateReport, assertTriggerRate, skillResolved } from "vigiles";
98
94
  import { fileURLToPath } from "node:url";
99
95
  ${note}
100
96
  // TODO: point at the plugin root (the dir holding .claude-plugin/ or skills/).
101
97
  const pluginDir = fileURLToPath(new URL("../../", import.meta.url));
102
98
  const skill = ${JSON.stringify(id)};
103
99
 
104
- const report = await measureTriggerRate({
100
+ const report = await paid_measureTriggerRate({
105
101
  pluginDir,
106
102
  stubSkillBodies: true, // firing is a frontmatter property — stub bodies, pay less
107
103
  prompts: [
@@ -152,7 +148,7 @@ function outcomeSection(input, contract) {
152
148
  : "";
153
149
  return `import assert from "node:assert/strict";
154
150
  import { result } from "vigiles/spec";
155
- import { assertAgentOk } from "vigiles/testing";
151
+ import { assertAgentOk } from "vigiles";
156
152
 
157
153
  // Reconstructed from ${input.name}'s ## Output contract (its compiled .md) — the
158
154
  // typed result() the spec wrote. assertAgentOk parses + validates the outcome
@@ -177,7 +173,7 @@ function fallbackSection(input) {
177
173
  const toolHint = input.tools && input.tools.length > 0
178
174
  ? `assertToolUsed(r, ${JSON.stringify(input.tools[0])}); // its declared contract: ${input.tools.join(", ")}`
179
175
  : `assertToolUsed(r, "Task"); // TODO: assert what the subagent should do`;
180
- return `import { runHarnessTest, assertToolUsed } from "vigiles/testing";
176
+ return `import { runHarnessTest, assertToolUsed } from "vigiles";
181
177
 
182
178
  // ${input.name} has no result() contract, so its outcome can't be asserted
183
179
  // deterministically — add one (result() on its agent() spec) for a no-judge
@@ -239,7 +235,7 @@ ${checks.join("\n")}
239
235
  function agentScaffold(input) {
240
236
  const head = header(`Starter harness test for the \`${input.name}\` subagent.`, `npx vigiles test ${suggestedPath(input)}`);
241
237
  const safetyImport = input.sideEffectingTools && input.sideEffectingTools.length > 0
242
- ? `import { notTool, didNotWrite, assertChecks } from "vigiles/testing";\n`
238
+ ? `import { notTool, didNotWrite, assertChecks } from "vigiles";\n`
243
239
  : "";
244
240
  const body = input.resultContract
245
241
  ? outcomeSection(input, input.resultContract)
@@ -7,7 +7,7 @@
7
7
  * remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
8
8
  * `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
9
9
  * never silently executes. There is deliberately NO execution flag: automation
10
- * tests the harness through the `vigiles/testing` API + skills (the layered tiers),
10
+ * tests the harness through the `vigiles` testing API + skills (the layered tiers),
11
11
  * not through the report verb. The IO (prompt / run / remember) lives in the CLI;
12
12
  * this is the pure decision + helpers.
13
13
  */
@@ -69,7 +69,7 @@ export interface ExecuteEnv {
69
69
  * the executing checks need a human to consent:
70
70
  * 1. nothing executable → skip "nothing" (a clean read; no nudge)
71
71
  * 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
72
- * skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
72
+ * skip "headless" (no one to ask; automation uses the `vigiles` testing API)
73
73
  * 3. sticky no → skip "remembered-no"
74
74
  * 4. sticky yes → run
75
75
  * 5. interactive human, no sticky choice → ask (then remember)
@@ -79,7 +79,7 @@ export declare function decideExecute(o: ExecuteEnv): ExecuteDecision;
79
79
  * The one-line "executing checks not run" nudge for a skipped read (the
80
80
  * no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
81
81
  * gap). There is no flag to point at — `audit` runs them only interactively, and
82
- * automation uses the `vigiles/testing` API.
82
+ * automation uses the `vigiles` testing API.
83
83
  */
84
84
  export declare function formatExecuteSkip(reason: ExecuteSkipReason): string | null;
85
85
  /**
@@ -8,7 +8,7 @@
8
8
  * remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
9
9
  * `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
10
10
  * never silently executes. There is deliberately NO execution flag: automation
11
- * tests the harness through the `vigiles/testing` API + skills (the layered tiers),
11
+ * tests the harness through the `vigiles` testing API + skills (the layered tiers),
12
12
  * not through the report verb. The IO (prompt / run / remember) lives in the CLI;
13
13
  * this is the pure decision + helpers.
14
14
  */
@@ -45,7 +45,7 @@ function isMeteredAccess(env) {
45
45
  * the executing checks need a human to consent:
46
46
  * 1. nothing executable → skip "nothing" (a clean read; no nudge)
47
47
  * 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
48
- * skip "headless" (no one to ask; automation uses the `vigiles/testing` API)
48
+ * skip "headless" (no one to ask; automation uses the `vigiles` testing API)
49
49
  * 3. sticky no → skip "remembered-no"
50
50
  * 4. sticky yes → run
51
51
  * 5. interactive human, no sticky choice → ask (then remember)
@@ -65,7 +65,7 @@ function decideExecute(o) {
65
65
  * The one-line "executing checks not run" nudge for a skipped read (the
66
66
  * no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
67
67
  * gap). There is no flag to point at — `audit` runs them only interactively, and
68
- * automation uses the `vigiles/testing` API.
68
+ * automation uses the `vigiles` testing API.
69
69
  */
70
70
  function formatExecuteSkip(reason) {
71
71
  switch (reason) {
@@ -74,7 +74,7 @@ function formatExecuteSkip(reason) {
74
74
  case "headless":
75
75
  return ("\nℹ Executing checks (safety battery · live MCP · skill firing) skipped — " +
76
76
  "`audit` runs them only interactively (a terminal). For automation, test the " +
77
- "harness with the `vigiles/testing` API.");
77
+ "harness with the `vigiles` testing API.");
78
78
  case "remembered-no":
79
79
  return ("\nℹ Executing checks not run (you disabled them — edit .vigilesrc.json " +
80
80
  "`audit.measure` to re-enable).");
@@ -595,7 +595,7 @@ function evalTierQuestion(kind) {
595
595
  //
596
596
  // ⚠️ It does NOT offer `{ evalDriver }` here, and that is checked rather
597
597
  // than assumed: `runEval`/`measure`/`measureArms` all hard-wire
598
- // `spawnAgent`, and `runEvalWith` is not exported from `vigiles/testing` —
598
+ // `spawnAgent`, and `runEvalWith` is not exported from `vigiles/eval` —
599
599
  // `measureTriggerRate` is the only public call with that seam. Suggesting
600
600
  // an argument the function does not take is this finding's own defect.
601
601
  return (`A deterministic test covers it, but that test drove a SCRIPTED model. ` +
package/dist/test.d.ts ADDED
@@ -0,0 +1,75 @@
1
+ /**
2
+ * `vigiles` (the package root) — the **free** testing surface: everything you can
3
+ * run without a model call, and therefore without a bill.
4
+ *
5
+ * This is the front door. There is no `vigiles/test` subpath: the bare package
6
+ * name IS the testing surface, because a second name for the same door is the
7
+ * duplication this reorganisation removed (the old `.` + `./spec` pair pointed at
8
+ * one module under two names). A subpath is warranted when something genuinely
9
+ * DIFFERENT lives behind it — another runtime, another driver, another protocol.
10
+ * Here the one real difference is whether a call can spend money, so there are
11
+ * exactly two doors: this one, and [`vigiles/eval`](./eval-surface.ts).
12
+ *
13
+ * ## What replaced what
14
+ *
15
+ * The old split was by TEST TIER — `vigiles/unit`, `vigiles/integration`,
16
+ * `vigiles/e2e`, `vigiles/testing`. Tiers are a property of test-file layout and
17
+ * runner config, not of a module graph: this repo already expresses them as
18
+ * vitest projects (`test:unit`, `test:integration`, `test:e2e`), so the exports
19
+ * map was a second, competing implementation of the same idea. `src/e2e.ts` had
20
+ * degenerated to `export * from "./integration.js"` — a barrel adding zero
21
+ * symbols, which its own docstring called "the definition of a non-tier".
22
+ *
23
+ * ## Two honest costs of the new split
24
+ *
25
+ * **1. The type coupling crosses the boundary and cannot be removed.** About two
26
+ * dozen helpers here — `assertImproves`, `assertNoRegression`, `assertReliable`,
27
+ * `assertSignificant`, `assertTriggerRate`, `reliable`, `improvement`,
28
+ * `significantlyBeats`, `compareArms`, `diffReports`, `diffToJUnit`,
29
+ * `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`, `toBaselineFile`,
30
+ * `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`, `outputTokens`,
31
+ * `cacheTokens`, plus `formatEvalReport` / `formatCheckReport` /
32
+ * `formatTriggerRateReport` / `checkReportToJUnit` / `assertRates` — operate on
33
+ * eval RESULTS while spending nothing themselves. They belong on the free path.
34
+ * Their TYPES (`EvalReport`, `CheckReport`, `TriggerRateReport`, `Metrics`, …)
35
+ * are defined on the paid side. So those types are re-exported from BOTH barrels:
36
+ * a user of the free surface must never import `vigiles/eval` for a type alone.
37
+ * The duplication is deliberate and is the price of splitting on cost.
38
+ *
39
+ * **2. Free is not the same as fast.** The old "real CLI, no API key" tier
40
+ * (`vigiles/integration`) collapses into this barrel, so the import path no longer
41
+ * warns you that `runHarnessTest` spawns a real `claude` binary under bubblewrap
42
+ * and can take ~40 seconds. Nothing here bills you; plenty here is slow. Duration
43
+ * now lives where duration always lived in practice — in the runner config
44
+ * (`--project integration`), not in the import.
45
+ *
46
+ * **What was lost, stated plainly:** the symmetry with the CLI verbs is now
47
+ * one-sided. `vigiles/eval` rhymes with `vigiles eval`, but the free half answers
48
+ * to the plain package name rather than to `vigiles test`. That is accepted, not
49
+ * overlooked. `@playwright/test` — an end-to-end tool by definition — ships no
50
+ * subpath naming a test type at all, and keeps the whole fast/slow/expensive
51
+ * distinction in its config rather than in its imports.
52
+ */
53
+ export { recordCheck } from "./check-count.js";
54
+ export { runScript } from "./run-script.js";
55
+ export type { RunScriptOptions, ScriptRunResult } from "./run-script.js";
56
+ export { runHook, parseHookOutput, decideHook, propertyHook, fileToolEvents, egressRoutes, } from "./run-hook.js";
57
+ export type { HookRunResult, RunHookOptions, HookInput, HookOutput, HookPropertyResult, FileToolEventOptions, } from "./run-hook.js";
58
+ export * from "./harness-assert.js";
59
+ export { loadHook } from "./load-hook.js";
60
+ export { evalChecks, assertChecks, tool, toolWith, notTool, onlyTools, skill, output, hookFired, received, turns, wrote, didNotWrite, subagent, blocked, allowed, mcp, cost, latency, tokens, inputTokens, outputTokens, cacheTokens, } from "./check.js";
61
+ export type { ArgMatcher, Check, CheckJSON, CheckResult, JudgeFn, } from "./check.js";
62
+ export { DISASTER_CATALOG, verifyGuardrail, unblockedDisasters, assertBlocksDisasters, formatGuardrailReport, } from "./guardrail-check.js";
63
+ export type { DisasterEvent, DisasterCategory, GuardrailResult, VerifyGuardrailOptions, } from "./guardrail-check.js";
64
+ export * from "./tool-stub.js";
65
+ export { runHarnessTest, runHarness, parseToolCalls, parseSubagents, parseResultEvent, parseOutput, parseHooks, decideSandbox, specTrusted, sandboxAvailable, } from "./harness-test.js";
66
+ export type { HarnessTestSpec, Trace, SubagentTrace, HarnessTestResult, RunHarnessTestOptions, ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, SandboxMode, } from "./harness-test.js";
67
+ export { commandsIn, mustInclude, mustNotInclude } from "./doc-commands.js";
68
+ export type { DocCommand } from "./doc-commands.js";
69
+ export { skillContract } from "./skill-contract.js";
70
+ export type { SkillContract, SkillContractOptions } from "./skill-contract.js";
71
+ export { compareContainment, formatContainment, } from "./trigger-containment.js";
72
+ export type { ContainmentInput, ContainmentVerdict, } from "./trigger-containment.js";
73
+ export { assertRates, assertPromptDiversity, checkPromptDiversity, checkReportToJUnit, formatCheckReport, formatEvalReport, formatTriggerRateReport, parseClaudeRun, stubSkillBody, } from "./eval.js";
74
+ export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
75
+ //# sourceMappingURL=test.d.ts.map
package/dist/test.js ADDED
@@ -0,0 +1,192 @@
1
+ "use strict";
2
+ /**
3
+ * `vigiles` (the package root) — the **free** testing surface: everything you can
4
+ * run without a model call, and therefore without a bill.
5
+ *
6
+ * This is the front door. There is no `vigiles/test` subpath: the bare package
7
+ * name IS the testing surface, because a second name for the same door is the
8
+ * duplication this reorganisation removed (the old `.` + `./spec` pair pointed at
9
+ * one module under two names). A subpath is warranted when something genuinely
10
+ * DIFFERENT lives behind it — another runtime, another driver, another protocol.
11
+ * Here the one real difference is whether a call can spend money, so there are
12
+ * exactly two doors: this one, and [`vigiles/eval`](./eval-surface.ts).
13
+ *
14
+ * ## What replaced what
15
+ *
16
+ * The old split was by TEST TIER — `vigiles/unit`, `vigiles/integration`,
17
+ * `vigiles/e2e`, `vigiles/testing`. Tiers are a property of test-file layout and
18
+ * runner config, not of a module graph: this repo already expresses them as
19
+ * vitest projects (`test:unit`, `test:integration`, `test:e2e`), so the exports
20
+ * map was a second, competing implementation of the same idea. `src/e2e.ts` had
21
+ * degenerated to `export * from "./integration.js"` — a barrel adding zero
22
+ * symbols, which its own docstring called "the definition of a non-tier".
23
+ *
24
+ * ## Two honest costs of the new split
25
+ *
26
+ * **1. The type coupling crosses the boundary and cannot be removed.** About two
27
+ * dozen helpers here — `assertImproves`, `assertNoRegression`, `assertReliable`,
28
+ * `assertSignificant`, `assertTriggerRate`, `reliable`, `improvement`,
29
+ * `significantlyBeats`, `compareArms`, `diffReports`, `diffToJUnit`,
30
+ * `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`, `toBaselineFile`,
31
+ * `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`, `outputTokens`,
32
+ * `cacheTokens`, plus `formatEvalReport` / `formatCheckReport` /
33
+ * `formatTriggerRateReport` / `checkReportToJUnit` / `assertRates` — operate on
34
+ * eval RESULTS while spending nothing themselves. They belong on the free path.
35
+ * Their TYPES (`EvalReport`, `CheckReport`, `TriggerRateReport`, `Metrics`, …)
36
+ * are defined on the paid side. So those types are re-exported from BOTH barrels:
37
+ * a user of the free surface must never import `vigiles/eval` for a type alone.
38
+ * The duplication is deliberate and is the price of splitting on cost.
39
+ *
40
+ * **2. Free is not the same as fast.** The old "real CLI, no API key" tier
41
+ * (`vigiles/integration`) collapses into this barrel, so the import path no longer
42
+ * warns you that `runHarnessTest` spawns a real `claude` binary under bubblewrap
43
+ * and can take ~40 seconds. Nothing here bills you; plenty here is slow. Duration
44
+ * now lives where duration always lived in practice — in the runner config
45
+ * (`--project integration`), not in the import.
46
+ *
47
+ * **What was lost, stated plainly:** the symmetry with the CLI verbs is now
48
+ * one-sided. `vigiles/eval` rhymes with `vigiles eval`, but the free half answers
49
+ * to the plain package name rather than to `vigiles test`. That is accepted, not
50
+ * overlooked. `@playwright/test` — an end-to-end tool by definition — ships no
51
+ * subpath naming a test type at all, and keeps the whole fast/slow/expensive
52
+ * distinction in its config rather than in its imports.
53
+ */
54
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
55
+ if (k2 === undefined) k2 = k;
56
+ var desc = Object.getOwnPropertyDescriptor(m, k);
57
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
58
+ desc = { enumerable: true, get: function() { return m[k]; } };
59
+ }
60
+ Object.defineProperty(o, k2, desc);
61
+ }) : (function(o, m, k, k2) {
62
+ if (k2 === undefined) k2 = k;
63
+ o[k2] = m[k];
64
+ }));
65
+ var __exportStar = (this && this.__exportStar) || function(m, exports) {
66
+ for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
67
+ };
68
+ Object.defineProperty(exports, "__esModule", { value: true });
69
+ exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.loadHook = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
70
+ exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
71
+ // --- reporting: how much did this script actually do? ---
72
+ // `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
73
+ // prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
74
+ // exporting an object of tests nobody calls). Call `recordCheck()` yourself when
75
+ // you assert some OTHER way — `node:assert`, vitest's `expect` — so those are
76
+ // visible to the runner too. See check-count.ts.
77
+ var check_count_js_1 = require("./check-count.js");
78
+ Object.defineProperty(exports, "recordCheck", { enumerable: true, get: function () { return check_count_js_1.recordCheck; } });
79
+ // --- the process primitives: runScript (any program) + runHook (plus a decision) ---
80
+ // `runScript` runs any program and reports what it DID (exit, both streams,
81
+ // writes, egress). `runHook` is that plus the hook protocol: event to stdin,
82
+ // exit code to allow/deny. Testing a plain helper script? Reach for runScript —
83
+ // its result carries no `decision`, because a script does not have one.
84
+ var run_script_js_1 = require("./run-script.js");
85
+ Object.defineProperty(exports, "runScript", { enumerable: true, get: function () { return run_script_js_1.runScript; } });
86
+ var run_hook_js_1 = require("./run-hook.js");
87
+ Object.defineProperty(exports, "runHook", { enumerable: true, get: function () { return run_hook_js_1.runHook; } });
88
+ Object.defineProperty(exports, "parseHookOutput", { enumerable: true, get: function () { return run_hook_js_1.parseHookOutput; } });
89
+ Object.defineProperty(exports, "decideHook", { enumerable: true, get: function () { return run_hook_js_1.decideHook; } });
90
+ Object.defineProperty(exports, "propertyHook", { enumerable: true, get: function () { return run_hook_js_1.propertyHook; } });
91
+ Object.defineProperty(exports, "fileToolEvents", { enumerable: true, get: function () { return run_hook_js_1.fileToolEvents; } });
92
+ Object.defineProperty(exports, "egressRoutes", { enumerable: true, get: function () { return run_hook_js_1.egressRoutes; } });
93
+ // --- assertions and the eval-RESULT analysis helpers (free; see cost #1 above) ---
94
+ __exportStar(require("./harness-assert.js"), exports);
95
+ // The compiled-hook LOADER. The in-process assertions above take the hook
96
+ // OBJECT, but a `.harness.mjs` test only has its PATH — without this the
97
+ // intended in-process test path is unreachable from the file format
98
+ // `vigiles test` actually runs, and authors fall back to spawning the runtime as
99
+ // a subprocess (the very plumbing compiled hooks exist to remove). Same loader
100
+ // the CLI runtime uses, so a hook that loads in a test loads identically in prod.
101
+ var load_hook_js_1 = require("./load-hook.js");
102
+ Object.defineProperty(exports, "loadHook", { enumerable: true, get: function () { return load_hook_js_1.loadHook; } });
103
+ // --- the declarative check vocabulary ---
104
+ // Enumerated rather than `export *` ON PURPOSE: `judged` also lives in check.ts
105
+ // and its default judge is a real model call, so it is the one member of this
106
+ // module that belongs on the paid barrel (as `paid_judged`). Listing the members
107
+ // makes "a free import that can bill you" unrepresentable here rather than merely
108
+ // documented — the previous barrel carried a paragraph apologising for exactly
109
+ // that. `hookFired` is re-exported explicitly so this `Check<Trace>` wins over the
110
+ // legacy boolean predicate of the same name from harness-assert.
111
+ var check_js_1 = require("./check.js");
112
+ Object.defineProperty(exports, "evalChecks", { enumerable: true, get: function () { return check_js_1.evalChecks; } });
113
+ Object.defineProperty(exports, "assertChecks", { enumerable: true, get: function () { return check_js_1.assertChecks; } });
114
+ Object.defineProperty(exports, "tool", { enumerable: true, get: function () { return check_js_1.tool; } });
115
+ Object.defineProperty(exports, "toolWith", { enumerable: true, get: function () { return check_js_1.toolWith; } });
116
+ Object.defineProperty(exports, "notTool", { enumerable: true, get: function () { return check_js_1.notTool; } });
117
+ Object.defineProperty(exports, "onlyTools", { enumerable: true, get: function () { return check_js_1.onlyTools; } });
118
+ Object.defineProperty(exports, "skill", { enumerable: true, get: function () { return check_js_1.skill; } });
119
+ Object.defineProperty(exports, "output", { enumerable: true, get: function () { return check_js_1.output; } });
120
+ Object.defineProperty(exports, "hookFired", { enumerable: true, get: function () { return check_js_1.hookFired; } });
121
+ Object.defineProperty(exports, "received", { enumerable: true, get: function () { return check_js_1.received; } });
122
+ Object.defineProperty(exports, "turns", { enumerable: true, get: function () { return check_js_1.turns; } });
123
+ Object.defineProperty(exports, "wrote", { enumerable: true, get: function () { return check_js_1.wrote; } });
124
+ Object.defineProperty(exports, "didNotWrite", { enumerable: true, get: function () { return check_js_1.didNotWrite; } });
125
+ Object.defineProperty(exports, "subagent", { enumerable: true, get: function () { return check_js_1.subagent; } });
126
+ Object.defineProperty(exports, "blocked", { enumerable: true, get: function () { return check_js_1.blocked; } });
127
+ Object.defineProperty(exports, "allowed", { enumerable: true, get: function () { return check_js_1.allowed; } });
128
+ Object.defineProperty(exports, "mcp", { enumerable: true, get: function () { return check_js_1.mcp; } });
129
+ Object.defineProperty(exports, "cost", { enumerable: true, get: function () { return check_js_1.cost; } });
130
+ Object.defineProperty(exports, "latency", { enumerable: true, get: function () { return check_js_1.latency; } });
131
+ Object.defineProperty(exports, "tokens", { enumerable: true, get: function () { return check_js_1.tokens; } });
132
+ Object.defineProperty(exports, "inputTokens", { enumerable: true, get: function () { return check_js_1.inputTokens; } });
133
+ Object.defineProperty(exports, "outputTokens", { enumerable: true, get: function () { return check_js_1.outputTokens; } });
134
+ Object.defineProperty(exports, "cacheTokens", { enumerable: true, get: function () { return check_js_1.cacheTokens; } });
135
+ // Guardrail verification — "prove your safety hook actually blocks" (over runHook).
136
+ var guardrail_check_js_1 = require("./guardrail-check.js");
137
+ Object.defineProperty(exports, "DISASTER_CATALOG", { enumerable: true, get: function () { return guardrail_check_js_1.DISASTER_CATALOG; } });
138
+ Object.defineProperty(exports, "verifyGuardrail", { enumerable: true, get: function () { return guardrail_check_js_1.verifyGuardrail; } });
139
+ Object.defineProperty(exports, "unblockedDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.unblockedDisasters; } });
140
+ Object.defineProperty(exports, "assertBlocksDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.assertBlocksDisasters; } });
141
+ Object.defineProperty(exports, "formatGuardrailReport", { enumerable: true, get: function () { return guardrail_check_js_1.formatGuardrailReport; } });
142
+ // Tool stubs on PATH (rung R2): shadow a CLI tool with a recorded canned result.
143
+ __exportStar(require("./tool-stub.js"), exports);
144
+ // The assembled machine — AGNOSTIC SURFACE ONLY. The Claude-Code transport
145
+ // (`scriptModel`, `claudeCodeDriver`, `buildClaudeArgs`, `claudeAvailable`,
146
+ // `loadPlugin`, `resolveHarness`) is deliberately NOT re-exported here, so this
147
+ // surface stays harness-agnostic — import those from `vigiles/claude-code` (or
148
+ // your harness's package). See `research/code-adapter-architecture.md`.
149
+ // Slow but free: see cost #2 in the module doc.
150
+ var harness_test_js_1 = require("./harness-test.js");
151
+ Object.defineProperty(exports, "runHarnessTest", { enumerable: true, get: function () { return harness_test_js_1.runHarnessTest; } });
152
+ Object.defineProperty(exports, "runHarness", { enumerable: true, get: function () { return harness_test_js_1.runHarness; } });
153
+ Object.defineProperty(exports, "parseToolCalls", { enumerable: true, get: function () { return harness_test_js_1.parseToolCalls; } });
154
+ Object.defineProperty(exports, "parseSubagents", { enumerable: true, get: function () { return harness_test_js_1.parseSubagents; } });
155
+ Object.defineProperty(exports, "parseResultEvent", { enumerable: true, get: function () { return harness_test_js_1.parseResultEvent; } });
156
+ Object.defineProperty(exports, "parseOutput", { enumerable: true, get: function () { return harness_test_js_1.parseOutput; } });
157
+ Object.defineProperty(exports, "parseHooks", { enumerable: true, get: function () { return harness_test_js_1.parseHooks; } });
158
+ Object.defineProperty(exports, "decideSandbox", { enumerable: true, get: function () { return harness_test_js_1.decideSandbox; } });
159
+ Object.defineProperty(exports, "specTrusted", { enumerable: true, get: function () { return harness_test_js_1.specTrusted; } });
160
+ Object.defineProperty(exports, "sandboxAvailable", { enumerable: true, get: function () { return harness_test_js_1.sandboxAvailable; } });
161
+ // A document's own rule about its own COMMANDS, made checkable. vigiles cannot
162
+ // infer "always pass -g" from prose; what was missing was a cheap way to declare
163
+ // it — measured at ~95 lines of markdown parsing around ~5 lines of rule, which
164
+ // is why nobody wrote the check and the rule stayed prose.
165
+ var doc_commands_js_1 = require("./doc-commands.js");
166
+ Object.defineProperty(exports, "commandsIn", { enumerable: true, get: function () { return doc_commands_js_1.commandsIn; } });
167
+ Object.defineProperty(exports, "mustInclude", { enumerable: true, get: function () { return doc_commands_js_1.mustInclude; } });
168
+ Object.defineProperty(exports, "mustNotInclude", { enumerable: true, get: function () { return doc_commands_js_1.mustNotInclude; } });
169
+ // A skill's own `allowed-tools:` frontmatter, read back as checks — the wiring
170
+ // from a declaration to the existing check vocabulary, not new machinery.
171
+ var skill_contract_js_1 = require("./skill-contract.js");
172
+ Object.defineProperty(exports, "skillContract", { enumerable: true, get: function () { return skill_contract_js_1.skillContract; } });
173
+ // Is a WEAKER model a valid lower bound for trigger-rate? Pure comparison of two
174
+ // runs; the open question the model floor makes people ask.
175
+ var trigger_containment_js_1 = require("./trigger-containment.js");
176
+ Object.defineProperty(exports, "compareContainment", { enumerable: true, get: function () { return trigger_containment_js_1.compareContainment; } });
177
+ Object.defineProperty(exports, "formatContainment", { enumerable: true, get: function () { return trigger_containment_js_1.formatContainment; } });
178
+ // --- free analysis OVER eval results (the coupling from cost #1) ---
179
+ // These read a report that `vigiles/eval` produced. They spend nothing, so they
180
+ // live here and carry no `paid_` prefix; their argument types are defined over
181
+ // there, so those types are re-exported below from this barrel too.
182
+ var eval_js_1 = require("./eval.js");
183
+ Object.defineProperty(exports, "assertRates", { enumerable: true, get: function () { return eval_js_1.assertRates; } });
184
+ Object.defineProperty(exports, "assertPromptDiversity", { enumerable: true, get: function () { return eval_js_1.assertPromptDiversity; } });
185
+ Object.defineProperty(exports, "checkPromptDiversity", { enumerable: true, get: function () { return eval_js_1.checkPromptDiversity; } });
186
+ Object.defineProperty(exports, "checkReportToJUnit", { enumerable: true, get: function () { return eval_js_1.checkReportToJUnit; } });
187
+ Object.defineProperty(exports, "formatCheckReport", { enumerable: true, get: function () { return eval_js_1.formatCheckReport; } });
188
+ Object.defineProperty(exports, "formatEvalReport", { enumerable: true, get: function () { return eval_js_1.formatEvalReport; } });
189
+ Object.defineProperty(exports, "formatTriggerRateReport", { enumerable: true, get: function () { return eval_js_1.formatTriggerRateReport; } });
190
+ Object.defineProperty(exports, "parseClaudeRun", { enumerable: true, get: function () { return eval_js_1.parseClaudeRun; } });
191
+ Object.defineProperty(exports, "stubSkillBody", { enumerable: true, get: function () { return eval_js_1.stubSkillBody; } });
192
+ //# sourceMappingURL=test.js.map
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "15.4.0",
3
+ "version": "16.0.0",
4
4
  "description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
5
5
  "keywords": [
6
6
  "claude-code",
@@ -35,17 +35,14 @@
35
35
  "bin": {
36
36
  "vigiles": "dist/cli.js"
37
37
  },
38
- "main": "./dist/core/spec.js",
39
- "types": "./dist/core/spec.d.ts",
38
+ "main": "./dist/test.js",
39
+ "types": "./dist/test.d.ts",
40
40
  "exports": {
41
- ".": "./dist/core/spec.js",
41
+ ".": "./dist/test.js",
42
42
  "./spec": "./dist/core/spec.js",
43
+ "./eval": "./dist/eval-surface.js",
43
44
  "./linting": "./dist/linting.js",
44
- "./testing": "./dist/testing.js",
45
- "./unit": "./dist/unit.js",
46
45
  "./hook": "./dist/hook.js",
47
- "./integration": "./dist/integration.js",
48
- "./e2e": "./dist/e2e.js",
49
46
  "./claude-code": "./dist/claude-code.js",
50
47
  "./codex": "./dist/codex.js",
51
48
  "./adapter": "./dist/adapter.js",
@@ -92,7 +89,9 @@
92
89
  "test:types": "npm run build && tsc --noEmit -p test/types/tsconfig.json",
93
90
  "api:report": "npm run build && node scripts/api-extractor.mjs --local",
94
91
  "api:check": "npm run build && node scripts/api-extractor.mjs",
92
+ "check": "node scripts/check.mjs",
95
93
  "docs:check": "npm run build && node scripts/check-doc-imports.mjs . docs README.md",
94
+ "exports:check": "npm run build && node scripts/check-export-prefixes.mjs .",
96
95
  "experimental:check": "npm run api:check && node scripts/check-experimental-naming.mjs",
97
96
  "docs:api": "typedoc"
98
97
  },
@@ -100,6 +99,7 @@
100
99
  "@eslint/js": "^10.0.1",
101
100
  "@jackchuka/mdschema": "^0.12.8",
102
101
  "@microsoft/api-extractor": "^7.58.9",
102
+ "@semantic-release/commit-analyzer": "^13.0.1",
103
103
  "@types/js-yaml": "^4.0.9",
104
104
  "@types/markdown-it": "^14.1.2",
105
105
  "@types/minimatch": "^5.1.2",
@@ -107,6 +107,7 @@
107
107
  "@typescript-eslint/eslint-plugin": "^8.58.0",
108
108
  "@typescript-eslint/parser": "^8.58.0",
109
109
  "@vitest/coverage-v8": "^4.1.8",
110
+ "conventional-changelog-conventionalcommits": "^10.3.0",
110
111
  "eslint": "^10.1.0",
111
112
  "eslint-import-resolver-typescript": "^4.4.5",
112
113
  "eslint-plugin-boundaries": "^6.0.2",
@@ -58,7 +58,7 @@ anything; every one of them ships today.
58
58
 
59
59
  | The question you're actually asking | Use |
60
60
  | ----------------------------------------------------------------------- | -------------------------------------------------------------------------------------- |
61
- | Which tools did it call, and with what arguments? | `trace.toolCalls` · `tool` / `toolWith` checks · `parseToolCalls` (`vigiles/e2e`) |
61
+ | Which tools did it call, and with what arguments? | `trace.toolCalls` · `tool` / `toolWith` checks · `parseToolCalls` (`vigiles`) |
62
62
  | Did it call a tool it must not? | `notTool(name)` |
63
63
  | Did it call **only** tools from a known set? | `onlyTools([...])` — the white-list, symmetric to `assertWroteOnly` |
64
64
  | Did it stay inside the `allowed-tools` its own frontmatter declares? | `skillContract(dir).surface` — builds that check FROM the declaration |
@@ -80,7 +80,7 @@ the `allowed-tools:` the skill already claims and hands back ready checks, so
80
80
  the claim is verified instead of restated:
81
81
 
82
82
  ```ts
83
- import { skillContract, assertChecks } from "vigiles/testing";
83
+ import { skillContract, assertChecks } from "vigiles";
84
84
 
85
85
  const c = skillContract(".claude/skills/my-skill");
86
86
  assertChecks(trace, [c.activation, ...c.surface]);
@@ -157,7 +157,7 @@ Pick one concrete thing to pin down — a specific `PreToolUse` hook, a specific
157
157
  **Unit (`runHook`)** — hand a hook a synthesized event, assert the decision:
158
158
 
159
159
  ```ts
160
- import { runHook, assertHookBlocked } from "vigiles/testing";
160
+ import { runHook, assertHookBlocked } from "vigiles";
161
161
 
162
162
  const r = runHook(hookCommand, {
163
163
  hook_event_name: "PreToolUse",
@@ -188,9 +188,9 @@ import {
188
188
  runHarnessTest,
189
189
  assertHookFired,
190
190
  assertRequestContains,
191
- } from "vigiles/testing";
191
+ } from "vigiles";
192
192
  // `scriptModel` is the Claude-Code TRANSPORT, deliberately not re-exported from
193
- // the harness-agnostic `vigiles/testing` — import it from the harness package:
193
+ // the harness-agnostic root surface — import it from the harness package:
194
194
  import { scriptModel } from "vigiles/claude-code";
195
195
 
196
196
  const r = await runHarnessTest({
@@ -202,34 +202,36 @@ assertHookFired(r, "SessionStart");
202
202
  assertRequestContains(r, "expected injected text"); // did it actually land?
203
203
  ```
204
204
 
205
- **Eval — absolute (`measure` + `judged`)** — testing _one_ skill, the usual case:
205
+ **Eval — absolute (`paid_measure` + `paid_judged`)** — testing _one_ skill, the usual case:
206
206
  score its output directly against a rubric. No on/off baseline — this is the
207
207
  "is it any good?" oracle (what promptfoo/DeepEval lead with), and the right
208
208
  default when there's nothing to compare against:
209
209
 
210
210
  ```ts
211
- import { measure, judged, skill, assertRates } from "vigiles/testing";
211
+ import { paid_measure, paid_judged } from "vigiles/eval"; // `paid_` = these call a model
212
+ import { skill, assertRates } from "vigiles";
212
213
 
213
- const report = await measure({
214
+ const report = await paid_measure({
214
215
  pluginDir: "./",
215
216
  task: "…a task the skill should handle…",
216
217
  checks: [
217
218
  skill("my-plugin:my-skill"), // it fired
218
- judged("the answer correctly does X and avoids Y"), // …and the output is good
219
+ paid_judged("the answer correctly does X and avoids Y"), // …and the output is good
219
220
  ],
220
221
  trials: 6,
221
222
  });
222
223
  assertRates(report, { min: 0.8 }); // each check passes ≥ 80% of trials
223
224
  ```
224
225
 
225
- **Eval — relative (`runEval` + `assertSignificant`)** — when the question is
226
+ **Eval — relative (`paid_runEval` + `assertSignificant`)** — when the question is
226
227
  _lift over no-skill_ (regression, or proving a change isn't noise): A/B the
227
228
  change on vs off and gate on significance, not eyeballing:
228
229
 
229
230
  ```ts
230
- import { runEval, assertSignificant } from "vigiles/testing";
231
+ import { paid_runEval } from "vigiles/eval"; // `paid_` = a real model runs
232
+ import { assertSignificant } from "vigiles";
231
233
 
232
- const report = await runEval({
234
+ const report = await paid_runEval({
233
235
  arms: { off: {}, on: { pluginDir: "./" } },
234
236
  task: "…a task the harness change should affect…",
235
237
  measure: (ctx) => ({ ok: /* a bare predicate over the trace */ true }),
@@ -261,7 +263,7 @@ unrepresentable:
261
263
  Use **`runScript`** — it runs any command and reports what it did:
262
264
 
263
265
  ```ts
264
- import { runScript } from "vigiles/testing";
266
+ import { runScript } from "vigiles";
265
267
 
266
268
  const r = runScript("bash scripts/check-links.sh", { cwd: repoDir });
267
269
  assert.equal(r.exitCode, 0);
@@ -293,7 +295,7 @@ npx vigiles eval --trials=6 # *.eval.{mjs,ts} — real model (local / night
293
295
  Unit-tier `runHook` tests need no `claude` and **always run** — write and run them
294
296
  even with no `claude` installed. A tier that genuinely can't run reports a loud
295
297
  `⊘ SKIPPED` (tallied separately, never a fake `✓`); a standalone script emits one
296
- via `skip(reason)` from `vigiles/testing`. A skip passes by default, but in a CI
298
+ via `skip(reason)` from `vigiles`. A skip passes by default, but in a CI
297
299
  job that asserts the capability is present, run **`vigiles test --no-skip`** so a
298
300
  skipped tier fails — a green-with-skips is untested surface. Keep unit +
299
301
  deterministic tests in CI (free); run evals locally or on a schedule with auth.
package/dist/e2e.d.ts DELETED
@@ -1,18 +0,0 @@
1
- /**
2
- * `vigiles/e2e` — DEPRECATED back-compat alias for [`vigiles/integration`](./integration.ts).
3
- *
4
- * There is no separate "e2e" tier: real **egress** is a *capability* of the
5
- * harness/integration scope (`egressRoutes()` + `runHook`'s `egress: { allow }`),
6
- * not a different kind of test — the old `e2e` barrel added exactly one symbol
7
- * over `integration`, which is the definition of a non-tier. It now lives on
8
- * `vigiles/integration`; this entry re-exports it unchanged so existing imports
9
- * keep working. See `research/testing-api-design.md` Part 4 (two scopes, not
10
- * four tiers). Prefer `vigiles/integration`.
11
- *
12
- * NOT here: **evals** (`runEval` / `measure` / `measureTriggerRate` / `judge`) —
13
- * those are non-deterministic measurement, a different axis. (They live on
14
- * `vigiles/testing`; `vigiles/eval` is named here in the original comment and does
15
- * not exist as an entry point.)
16
- */
17
- export * from "./integration.js";
18
- //# sourceMappingURL=e2e.d.ts.map
package/dist/e2e.js DELETED
@@ -1,34 +0,0 @@
1
- "use strict";
2
- var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
3
- if (k2 === undefined) k2 = k;
4
- var desc = Object.getOwnPropertyDescriptor(m, k);
5
- if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
6
- desc = { enumerable: true, get: function() { return m[k]; } };
7
- }
8
- Object.defineProperty(o, k2, desc);
9
- }) : (function(o, m, k, k2) {
10
- if (k2 === undefined) k2 = k;
11
- o[k2] = m[k];
12
- }));
13
- var __exportStar = (this && this.__exportStar) || function(m, exports) {
14
- for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
15
- };
16
- Object.defineProperty(exports, "__esModule", { value: true });
17
- /**
18
- * `vigiles/e2e` — DEPRECATED back-compat alias for [`vigiles/integration`](./integration.ts).
19
- *
20
- * There is no separate "e2e" tier: real **egress** is a *capability* of the
21
- * harness/integration scope (`egressRoutes()` + `runHook`'s `egress: { allow }`),
22
- * not a different kind of test — the old `e2e` barrel added exactly one symbol
23
- * over `integration`, which is the definition of a non-tier. It now lives on
24
- * `vigiles/integration`; this entry re-exports it unchanged so existing imports
25
- * keep working. See `research/testing-api-design.md` Part 4 (two scopes, not
26
- * four tiers). Prefer `vigiles/integration`.
27
- *
28
- * NOT here: **evals** (`runEval` / `measure` / `measureTriggerRate` / `judge`) —
29
- * those are non-deterministic measurement, a different axis. (They live on
30
- * `vigiles/testing`; `vigiles/eval` is named here in the original comment and does
31
- * not exist as an entry point.)
32
- */
33
- __exportStar(require("./integration.js"), exports);
34
- //# sourceMappingURL=e2e.js.map