vigiles 15.4.0 → 16.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -23,7 +23,7 @@
23
23
  * design). NB the EXECUTING
24
24
  * "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
25
25
  * running arbitrary hooks safely needs cross-platform confinement that isn't
26
- * shipped yet, so the battery lives in the `vigiles/testing` API via
26
+ * shipped yet, so the battery lives in the `vigiles` testing API via
27
27
  * `guardrail-check`/`assertBlocksDisasters`, where you opt in explicitly.
28
28
  *
29
29
  * TESTED vs EVALUATED are two rings, not one, because a harness and an eval differ
@@ -28,7 +28,7 @@ exports.formatAuditScore = formatAuditScore;
28
28
  * design). NB the EXECUTING
29
29
  * "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
30
30
  * running arbitrary hooks safely needs cross-platform confinement that isn't
31
- * shipped yet, so the battery lives in the `vigiles/testing` API via
31
+ * shipped yet, so the battery lives in the `vigiles` testing API via
32
32
  * `guardrail-check`/`assertBlocksDisasters`, where you opt in explicitly.
33
33
  *
34
34
  * TESTED vs EVALUATED are two rings, not one, because a harness and an eval differ
@@ -49,7 +49,7 @@ const score_core_js_1 = require("./score-core.js");
49
49
  const W_UNTESTED = 3;
50
50
  /** The command that answers the firing question — named, not alluded to. */
51
51
  const MEASURE_FIRING_COMMAND = "run `npx vigiles audit` interactively to measure, or add a `*.eval.mjs` " +
52
- "(`measureTriggerRate`, vigiles/testing)";
52
+ "(`paid_measureTriggerRate`, vigiles/eval)";
53
53
  /** Resolve the terse "thing(s)" plural placeholder against a count:
54
54
  * n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools"). */
55
55
  function pluralizeLabel(n, label) {
@@ -46,7 +46,7 @@ exports.armCheckReport = armCheckReport;
46
46
  * tool avoids.
47
47
  *
48
48
  * WHAT A MISSING COUNT MEANS: nothing at all. A script that never imports
49
- * `vigiles/testing` cannot report, so the runner sees no file and treats it
49
+ * `vigiles` cannot report, so the runner sees no file and treats it
50
50
  * exactly as before — a plain pass. Silence is the legacy branch, never a
51
51
  * verdict; only a count of literally zero is a finding. (The alternative —
52
52
  * force-loading this module into every child with `node --import` so silence
@@ -2,7 +2,7 @@
2
2
  * `vigiles/claude-code` — the Claude Code-specific harness pieces a *different*
3
3
  * harness would swap out: the plugin/repo loader (reads real Claude Code plugin
4
4
  * layouts) and the scriptable Anthropic Messages mock. Split from
5
- * `vigiles/testing` on purpose — the test API above is the stable surface; this
5
+ * the `vigiles` root on purpose — the test API above is the stable surface; this
6
6
  * is the adapter, so a future `vigiles/<other-harness>` can sit beside it.
7
7
  */
8
8
  export * from "./adapters/claude-code/plugin-loader.js";
@@ -19,7 +19,7 @@ exports.formatSelectionReport = exports.assertNoCollision = exports.measureSelec
19
19
  * `vigiles/claude-code` — the Claude Code-specific harness pieces a *different*
20
20
  * harness would swap out: the plugin/repo loader (reads real Claude Code plugin
21
21
  * layouts) and the scriptable Anthropic Messages mock. Split from
22
- * `vigiles/testing` on purpose — the test API above is the stable surface; this
22
+ * the `vigiles` root on purpose — the test API above is the stable surface; this
23
23
  * is the adapter, so a future `vigiles/<other-harness>` can sit beside it.
24
24
  */
25
25
  __exportStar(require("./adapters/claude-code/plugin-loader.js"), exports);
@@ -28,7 +28,7 @@ __exportStar(require("./mock-model.js"), exports);
28
28
  // helpers + the `claude` capability probe. Agnostic users never need these
29
29
  // (`runHarnessTest` defaults to the CC driver), but they're exposed here — beside
30
30
  // the Codex driver in `vigiles/codex` — for CC-specific tests/tooling. They are
31
- // deliberately NOT on the agnostic `vigiles/testing` surface.
31
+ // deliberately NOT on the agnostic `vigiles` surface.
32
32
  var harness_test_js_1 = require("./harness-test.js");
33
33
  Object.defineProperty(exports, "claudeCodeDriver", { enumerable: true, get: function () { return harness_test_js_1.claudeCodeDriver; } });
34
34
  Object.defineProperty(exports, "buildClaudeArgs", { enumerable: true, get: function () { return harness_test_js_1.buildClaudeArgs; } });
@@ -50,7 +50,7 @@ Object.defineProperty(exports, "agent", { enumerable: true, get: function () { r
50
50
  Object.defineProperty(exports, "experimental_skill", { enumerable: true, get: function () { return typed_spec_js_1.experimental_skill; } });
51
51
  // Selection-collision — a Claude-Code-ONLY behavioral measurement (Codex has no
52
52
  // skill-selection event to read), so it lives on this surface, not the agnostic
53
- // `vigiles/testing`. `measureSelectionMatrix` builds the N×N "which skill fired?"
53
+ // `vigiles`. `measureSelectionMatrix` builds the N×N "which skill fired?"
54
54
  // matrix (diagonal = recall, off-diagonal = collision); `assertNoCollision` gates it.
55
55
  var scan_behavioral_js_1 = require("./scan-behavioral.js");
56
56
  Object.defineProperty(exports, "measureSelectionMatrix", { enumerable: true, get: function () { return scan_behavioral_js_1.measureSelectionMatrix; } });
package/dist/cli.js CHANGED
@@ -1702,7 +1702,7 @@ function formatTriggerNudge(triggerableSkills) {
1702
1702
  return "";
1703
1703
  const n = triggerableSkills;
1704
1704
  return (`ℹ Do your ${String(n)} skill${n === 1 ? "" : "s"} actually fire? The deterministic read can't tell — ` +
1705
- `run \`audit\` interactively to measure, or test with \`measureTriggerRate\` (vigiles/testing).`);
1705
+ `run \`audit\` interactively to measure, or test with \`paid_measureTriggerRate\` (vigiles/eval).`);
1706
1706
  }
1707
1707
  /** A terminal summary of the rule map: the CONFIDENT lane counts + the POSSIBLE
1708
1708
  * (review) and SKIPPED tiers, with the honest caveat that detection is a
@@ -1973,7 +1973,7 @@ function vigilesWorkflow(plan, harnesses, hasPackageJson) {
1973
1973
  `
1974
1974
  : "";
1975
1975
  // The test pillar's harness job runs `*.harness.mjs` files that
1976
- // `import "vigiles/testing"`, so it needs vigiles RESOLVABLE — a package.json +
1976
+ // `import "vigiles"`, so it needs vigiles RESOLVABLE — a package.json +
1977
1977
  // install. A repo with no package.json can't run the JS test tier in CI at all
1978
1978
  // (the import won't resolve even though the CLI itself came from npx), so the
1979
1979
  // harness job is emitted ONLY when a package.json exists (Codex review / #111 /
@@ -2136,7 +2136,7 @@ const STARTER_HARNESS = `/**
2136
2136
  *
2137
2137
  * Guide: https://github.com/zernie/vigiles/blob/main/docs/harness-testing.md
2138
2138
  */
2139
- import { runHook } from "vigiles/testing";
2139
+ import { runHook } from "vigiles";
2140
2140
  import assert from "node:assert/strict";
2141
2141
 
2142
2142
  // EXAMPLE — replace with one of YOUR hooks. This PreToolUse Bash guard blocks a
@@ -4384,7 +4384,7 @@ const COMMAND_HELP = {
4384
4384
  usage: " vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)",
4385
4385
  detail: [
4386
4386
  " writes vigiles-report.html + .json (auto-gitignored; --out=<dir> · --no-html/--no-json · --no-open · --json for machine output). NOT a CI step — use `vigiles lint` in CI.",
4387
- " the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles/testing API",
4387
+ " the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles testing API",
4388
4388
  " --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt",
4389
4389
  ],
4390
4390
  },
@@ -4970,7 +4970,7 @@ function refsHookCommand() {
4970
4970
  // ---------------------------------------------------------------------------
4971
4971
  /**
4972
4972
  * Load a compiled-hook program's default export — the SHARED loader, also the
4973
- * public `vigiles/testing` `loadHook` a `.harness.mjs` test uses, so a hook that
4973
+ * public `vigiles` `loadHook` a `.harness.mjs` test uses, so a hook that
4974
4974
  * loads in a test loads identically here (one loader, no drift).
4975
4975
  */
4976
4976
  const loadHookProgram = load_hook_js_1.loadHook;
@@ -6135,7 +6135,7 @@ async function main() {
6135
6135
  // uses `vigiles lint`). The executing checks (safety battery + live MCP +
6136
6136
  // skill firing) run only on consent: at a TTY `audit` asks once (remembered
6137
6137
  // in `.vigilesrc.json`), headless it stays a read + a one-line nudge. There
6138
- // is NO execution flag — automation tests the harness via the vigiles/testing
6138
+ // is NO execution flag — automation tests the harness via the vigiles testing
6139
6139
  // API + skills, not the report verb. See the `audit-side-effect-free` rule.
6140
6140
  const dirs = restArgs.length > 0 ? restArgs : ["."];
6141
6141
  const json = args.includes("--json");
@@ -6337,10 +6337,10 @@ async function main() {
6337
6337
  // linter also executes it). A plain `audit` is a deterministic READ;
6338
6338
  // these run only on consent — ASK once at a TTY (remembered); headless
6339
6339
  // stays a read + a nudge (no execution flag — automation uses the
6340
- // vigiles/testing API). Resolved AFTER the report (so the read leads) but
6340
+ // vigiles testing API). Resolved AFTER the report (so the read leads) but
6341
6341
  // BEFORE the rule map is routed, so a first-time "yes" enriches THIS run.
6342
6342
  // (The safety battery is NOT here — it needs cross-platform confinement
6343
- // that isn't shipped, so it lives in the vigiles/testing API.)
6343
+ // that isn't shipped, so it lives in the vigiles testing API.)
6344
6344
  // `surfaces` + `execDecision` were resolved above (the ring needed them);
6345
6345
  // this only turns an `ask` into a prompt and remembers the answer.
6346
6346
  const { execute, note: execNote } = await resolveExecution(surfaces, execDecision, json, adapter.name);
package/dist/codex.d.ts CHANGED
@@ -1,6 +1,6 @@
1
1
  /**
2
2
  * `vigiles/codex` — the OpenAI Codex harness adapter. Sits beside
3
- * `vigiles/claude-code`: same harness-agnostic `vigiles/testing` core, a
3
+ * `vigiles/claude-code`: same harness-agnostic `vigiles` testing core, a
4
4
  * different transport (the `codex` binary + the OpenAI **Responses** SSE mock).
5
5
  *
6
6
  * Pillar 2 (harness testing) is proven here — `startCodexMock` serves the
package/dist/codex.js CHANGED
@@ -16,7 +16,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
16
16
  Object.defineProperty(exports, "__esModule", { value: true });
17
17
  /**
18
18
  * `vigiles/codex` — the OpenAI Codex harness adapter. Sits beside
19
- * `vigiles/claude-code`: same harness-agnostic `vigiles/testing` core, a
19
+ * `vigiles/claude-code`: same harness-agnostic `vigiles` testing core, a
20
20
  * different transport (the `codex` binary + the OpenAI **Responses** SSE mock).
21
21
  *
22
22
  * Pillar 2 (harness testing) is proven here — `startCodexMock` serves the
@@ -399,12 +399,21 @@ export interface SkillStep {
399
399
  /** Max attempts to satisfy the gate before the step fails (default 1). */
400
400
  readonly retry?: number;
401
401
  }
402
- /** Declare a skill input (compiles to argument-hint + an Arguments entry). */
403
- export declare function input(name: string, hint: string, opts?: {
402
+ /**
403
+ * Declare a skill input (compiles to argument-hint + an Arguments entry).
404
+ *
405
+ * Deliberately NOT a standalone export — reached as `experimental_skill.input`.
406
+ * The reason is on `experimental_skill` below.
407
+ */
408
+ declare function input(name: string, hint: string, opts?: {
404
409
  required?: boolean;
405
410
  }): SkillInput;
406
- /** Declare a gated pipeline step. */
407
- export declare function step(instr: string | InstructionFragment[], opts?: {
411
+ /**
412
+ * Declare a gated pipeline step.
413
+ *
414
+ * Deliberately NOT a standalone export — reached as `experimental_skill.step`.
415
+ */
416
+ declare function step(instr: string | InstructionFragment[], opts?: {
408
417
  gate?: Gate;
409
418
  retry?: number;
410
419
  }): SkillStep;
@@ -504,9 +513,32 @@ export type SkillSpecInput<P extends AuthoredPurity | undefined, V extends ToolV
504
513
  * DIFFERENT function — a `Check<Trace>` taking an id string, asking whether a
505
514
  * skill fired. This one authors a skill; that one observes one.
506
515
  *
516
+ * Its helper vocabulary hangs off it — `experimental_skill.input(…)` and
517
+ * `experimental_skill.step(…)` — rather than being exported beside it. Both are
518
+ * used ONLY by skill specs (measured: zero uses in agent/claude specs, against
519
+ * `cmd`/`file`/`ref`/`result`, which are shared and therefore stay top-level).
520
+ * Hanging them here makes the experimental marking STRUCTURAL for the whole
521
+ * family: you cannot reach `input()` without naming `experimental_skill` first.
522
+ * The prefix convention alone could not do that — it is a habit, and it had
523
+ * already leaked once when `skill()` shipped stable-named against its own docs.
524
+ *
525
+ * Honest limit: `const { input } = experimental_skill` strips the marker again
526
+ * inside one file. What the shape actually guarantees is narrower and still
527
+ * worth having — an unmarked name never crosses the package boundary.
528
+ *
529
+ * `Object.assign` rather than `export namespace`: the latter is banned by this
530
+ * repo's own lint (`no-namespace: error`, inherited from strict-type-checked).
531
+ *
532
+ * @experimental
533
+ */
534
+ declare function skillSpec<const P extends AuthoredPurity | undefined = undefined, V extends ToolVocabulary = OpenToolVocabulary>(spec: SkillSpecInput<P, V>): SkillSpec;
535
+ /**
507
536
  * @experimental
508
537
  */
509
- export declare function experimental_skill<const P extends AuthoredPurity | undefined = undefined, V extends ToolVocabulary = OpenToolVocabulary>(spec: SkillSpecInput<P, V>): SkillSpec;
538
+ export declare const experimental_skill: typeof skillSpec & {
539
+ input: typeof input;
540
+ step: typeof step;
541
+ };
510
542
  /**
511
543
  * A subagent definition (compiles to `agents/<name>.md`). Unlike a skill —
512
544
  * reference material the model reads on activation — a subagent is a *delegated
package/dist/core/spec.js CHANGED
@@ -10,7 +10,7 @@
10
10
  * guidance() — prose only, no mechanical enforcement
11
11
  */
12
12
  Object.defineProperty(exports, "__esModule", { value: true });
13
- exports.BUILTIN_LINTERS = void 0;
13
+ exports.experimental_skill = exports.BUILTIN_LINTERS = void 0;
14
14
  exports.enforce = enforce;
15
15
  exports.guidance = guidance;
16
16
  exports.guard = guard;
@@ -24,9 +24,6 @@ exports.instructions = instructions;
24
24
  exports.experimental_effect = experimental_effect;
25
25
  exports.claude = claude;
26
26
  exports.project = project;
27
- exports.input = input;
28
- exports.step = step;
29
- exports.experimental_skill = experimental_skill;
30
27
  exports.agent = agent;
31
28
  exports.result = result;
32
29
  exports.delegate = delegate;
@@ -222,11 +219,20 @@ function claude(spec) {
222
219
  function project(role) {
223
220
  return { _ref: "role", role };
224
221
  }
225
- /** Declare a skill input (compiles to argument-hint + an Arguments entry). */
222
+ /**
223
+ * Declare a skill input (compiles to argument-hint + an Arguments entry).
224
+ *
225
+ * Deliberately NOT a standalone export — reached as `experimental_skill.input`.
226
+ * The reason is on `experimental_skill` below.
227
+ */
226
228
  function input(name, hint, opts = {}) {
227
229
  return { name, hint, required: opts.required };
228
230
  }
229
- /** Declare a gated pipeline step. */
231
+ /**
232
+ * Declare a gated pipeline step.
233
+ *
234
+ * Deliberately NOT a standalone export — reached as `experimental_skill.step`.
235
+ */
230
236
  function step(instr, opts = {}) {
231
237
  return { do: instr, gate: opts.gate, retry: opts.retry };
232
238
  }
@@ -245,11 +251,31 @@ function step(instr, opts = {}) {
245
251
  * DIFFERENT function — a `Check<Trace>` taking an id string, asking whether a
246
252
  * skill fired. This one authors a skill; that one observes one.
247
253
  *
254
+ * Its helper vocabulary hangs off it — `experimental_skill.input(…)` and
255
+ * `experimental_skill.step(…)` — rather than being exported beside it. Both are
256
+ * used ONLY by skill specs (measured: zero uses in agent/claude specs, against
257
+ * `cmd`/`file`/`ref`/`result`, which are shared and therefore stay top-level).
258
+ * Hanging them here makes the experimental marking STRUCTURAL for the whole
259
+ * family: you cannot reach `input()` without naming `experimental_skill` first.
260
+ * The prefix convention alone could not do that — it is a habit, and it had
261
+ * already leaked once when `skill()` shipped stable-named against its own docs.
262
+ *
263
+ * Honest limit: `const { input } = experimental_skill` strips the marker again
264
+ * inside one file. What the shape actually guarantees is narrower and still
265
+ * worth having — an unmarked name never crosses the package boundary.
266
+ *
267
+ * `Object.assign` rather than `export namespace`: the latter is banned by this
268
+ * repo's own lint (`no-namespace: error`, inherited from strict-type-checked).
269
+ *
248
270
  * @experimental
249
271
  */
250
- function experimental_skill(spec) {
272
+ function skillSpec(spec) {
251
273
  return { _specType: "skill", ...spec };
252
274
  }
275
+ /**
276
+ * @experimental
277
+ */
278
+ exports.experimental_skill = Object.assign(skillSpec, { input, step });
253
279
  /**
254
280
  * Define a subagent specification (compiles to `agents/<name>.md`).
255
281
  *
@@ -0,0 +1,122 @@
1
+ /**
2
+ * `vigiles/eval` — the **paid** surface: every runtime export here can call a
3
+ * model, and therefore can spend money.
4
+ *
5
+ * That single property is the whole reason this subpath exists. The package used
6
+ * to split its testing API by TEST TIER (`vigiles/unit`, `vigiles/integration`,
7
+ * `vigiles/e2e`, `vigiles/testing`); it now splits on COST. Everything free is on
8
+ * the package root, [`vigiles`](./test.ts). Everything that bills is here. The
9
+ * name rhymes with the `vigiles eval` CLI verb.
10
+ *
11
+ * ## Why every runtime export is also named `paid_`
12
+ *
13
+ * The import path warns ONCE, at the top of the file. The name warns EVERY time,
14
+ * at the call site. Reading `await judged(trace, "did it refuse?")` on line 140,
15
+ * the import line is long out of view — `await paid_judged(...)` still says what
16
+ * it costs. This is not a new idiom in this package: `vigiles/experimental`
17
+ * already pairs a quarantined subpath with an `experimental_` name prefix for
18
+ * exactly this reason. The same device, applied to a second axis.
19
+ *
20
+ * ⚠️ **The prefix slightly OVERSTATES the cost, and that is a deliberate trade
21
+ * rather than an oversight.** `paid_judged` takes an injectable judge:
22
+ * `paid_judged(rubric, { judge: myFn })` calls no model and spends nothing — only
23
+ * the DEFAULT judge bills. `paid_measureTriggerRate` and friends likewise accept
24
+ * an injected `evalDriver`. The strictly accurate prefix would be `metered_`, but
25
+ * it reads a beat slower, and a warning that is not absorbed at a glance is not a
26
+ * warning. Clarity wins; the imprecision is recorded here and on each symbol
27
+ * rather than left for a reader to discover.
28
+ *
29
+ * ## Why the split had to be cost, not tier
30
+ *
31
+ * The old `vigiles/unit` docstring promised "no `claude`, no model, no
32
+ * bubblewrap, no network" — and re-exported `judged`, whose default judge is a
33
+ * real model call (`check.ts` resolves `opts.judge ?? ((o) => runJudge(o))`). A
34
+ * test importing only from that barrel could still spend money, and the one line
35
+ * a reader would have relied on to know otherwise was the false one. `judged`
36
+ * moved HERE and became `paid_judged`, which makes that class of defect
37
+ * unrepresentable rather than apologised for: on this path, billing is the
38
+ * advertised property, in the subpath and in every name.
39
+ *
40
+ * ## What is NOT here, and why
41
+ *
42
+ * The ~two dozen helpers that read an eval RESULT — `assertImproves`,
43
+ * `assertNoRegression`, `assertReliable`, `assertSignificant`, `assertTriggerRate`,
44
+ * `reliable`, `improvement`, `significantlyBeats`, `compareArms`, `diffReports`,
45
+ * `diffToJUnit`, `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`,
46
+ * `toBaselineFile`, `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`,
47
+ * `outputTokens`, `cacheTokens`, `formatEvalReport`, `formatCheckReport`,
48
+ * `formatTriggerRateReport`, `checkReportToJUnit`, `assertRates` — spend nothing,
49
+ * so they live on the free root barrel, unprefixed, even though their subject
50
+ * matter is evals.
51
+ *
52
+ * That is the one seam the split cannot make clean: the free helpers are typed
53
+ * over report shapes DEFINED here. The coupling is real and irreducible, so both
54
+ * barrels re-export the same report types. **The types are NOT prefixed** — the
55
+ * prefix is a warning about calling something, and a type is never called;
56
+ * `paid_EvalReport` would be noise on a symbol that cannot bill. Importing
57
+ * `vigiles/eval` for a type alone should never be necessary; importing it for a
58
+ * FUNCTION means you accepted a bill.
59
+ *
60
+ * Harness-specific eval drivers are not here either — `codexEvalDriver` /
61
+ * `codexEvalRunner` / `codexEvalAgentRunner` live on `vigiles/codex`, and
62
+ * `measureSelectionMatrix` / `assertNoCollision` on `vigiles/claude-code`, because
63
+ * those surfaces are chosen by harness, not by cost.
64
+ */
65
+ /**
66
+ * Run an A/B eval across arms with a real model. Spends money.
67
+ *
68
+ * ⚠️ The `paid_` prefix overstates slightly: pass your own `evalDriver` and no
69
+ * model of ours is called. The DEFAULT path bills.
70
+ */
71
+ export { runEval as paid_runEval } from "./eval.js";
72
+ /**
73
+ * Score checks over N trials of one task with a real model. Spends money.
74
+ *
75
+ * ⚠️ `paid_` overstates slightly: with an injected `evalDriver` this drives
76
+ * whatever you supply. The DEFAULT path bills.
77
+ */
78
+ export { measure as paid_measure } from "./eval.js";
79
+ /**
80
+ * Score the same checks per arm (a hook/skill/rule on vs off) with a real model.
81
+ * Spends money.
82
+ *
83
+ * ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
84
+ * path. The DEFAULT path bills.
85
+ */
86
+ export { measureArms as paid_measureArms } from "./eval.js";
87
+ /**
88
+ * Measure whether a skill's description actually FIRES (recall + precision) by
89
+ * running real prompts against a real model. Spends money.
90
+ *
91
+ * ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
92
+ * path, and `stubSkillBodies` makes the default path much cheaper without making
93
+ * it free. The DEFAULT path bills.
94
+ */
95
+ export { measureTriggerRate as paid_measureTriggerRate } from "./eval.js";
96
+ /**
97
+ * The model-graded judge (harness-agnostic — grades text against a rubric).
98
+ * Shells out to the `claude` CLI synchronously; needs model auth. Spends money.
99
+ *
100
+ * ⚠️ Unlike the others this one has no injection seam — it always calls a model,
101
+ * so here `paid_` is exact.
102
+ */
103
+ export { judge as paid_judge } from "./judge.js";
104
+ /**
105
+ * The one member of the declarative check vocabulary that bills: its default
106
+ * judge is a real model call. Every other `Check` is deterministic and lives
107
+ * unprefixed on the free root barrel.
108
+ *
109
+ * ⚠️ The `paid_` prefix overstates slightly, and this is the symbol it overstates
110
+ * most: `paid_judged(rubric, { judge: myFn })` calls YOUR function and spends
111
+ * nothing. Only `paid_judged(rubric)` — the default — bills.
112
+ */
113
+ export { judged as paid_judged } from "./check.js";
114
+ /**
115
+ * The Claude-Code eval driver: the transport `paid_runEval` and friends use by
116
+ * default. Naming it here is how you pass it explicitly; using it spends money.
117
+ */
118
+ export { claudeEvalDriver as paid_claudeEvalDriver } from "./eval.js";
119
+ export type { Check, CheckResult, JudgeFn } from "./check.js";
120
+ export type { Trace } from "./harness-test.js";
121
+ export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
122
+ //# sourceMappingURL=eval-surface.d.ts.map
@@ -0,0 +1,130 @@
1
+ "use strict";
2
+ /**
3
+ * `vigiles/eval` — the **paid** surface: every runtime export here can call a
4
+ * model, and therefore can spend money.
5
+ *
6
+ * That single property is the whole reason this subpath exists. The package used
7
+ * to split its testing API by TEST TIER (`vigiles/unit`, `vigiles/integration`,
8
+ * `vigiles/e2e`, `vigiles/testing`); it now splits on COST. Everything free is on
9
+ * the package root, [`vigiles`](./test.ts). Everything that bills is here. The
10
+ * name rhymes with the `vigiles eval` CLI verb.
11
+ *
12
+ * ## Why every runtime export is also named `paid_`
13
+ *
14
+ * The import path warns ONCE, at the top of the file. The name warns EVERY time,
15
+ * at the call site. Reading `await judged(trace, "did it refuse?")` on line 140,
16
+ * the import line is long out of view — `await paid_judged(...)` still says what
17
+ * it costs. This is not a new idiom in this package: `vigiles/experimental`
18
+ * already pairs a quarantined subpath with an `experimental_` name prefix for
19
+ * exactly this reason. The same device, applied to a second axis.
20
+ *
21
+ * ⚠️ **The prefix slightly OVERSTATES the cost, and that is a deliberate trade
22
+ * rather than an oversight.** `paid_judged` takes an injectable judge:
23
+ * `paid_judged(rubric, { judge: myFn })` calls no model and spends nothing — only
24
+ * the DEFAULT judge bills. `paid_measureTriggerRate` and friends likewise accept
25
+ * an injected `evalDriver`. The strictly accurate prefix would be `metered_`, but
26
+ * it reads a beat slower, and a warning that is not absorbed at a glance is not a
27
+ * warning. Clarity wins; the imprecision is recorded here and on each symbol
28
+ * rather than left for a reader to discover.
29
+ *
30
+ * ## Why the split had to be cost, not tier
31
+ *
32
+ * The old `vigiles/unit` docstring promised "no `claude`, no model, no
33
+ * bubblewrap, no network" — and re-exported `judged`, whose default judge is a
34
+ * real model call (`check.ts` resolves `opts.judge ?? ((o) => runJudge(o))`). A
35
+ * test importing only from that barrel could still spend money, and the one line
36
+ * a reader would have relied on to know otherwise was the false one. `judged`
37
+ * moved HERE and became `paid_judged`, which makes that class of defect
38
+ * unrepresentable rather than apologised for: on this path, billing is the
39
+ * advertised property, in the subpath and in every name.
40
+ *
41
+ * ## What is NOT here, and why
42
+ *
43
+ * The ~two dozen helpers that read an eval RESULT — `assertImproves`,
44
+ * `assertNoRegression`, `assertReliable`, `assertSignificant`, `assertTriggerRate`,
45
+ * `reliable`, `improvement`, `significantlyBeats`, `compareArms`, `diffReports`,
46
+ * `diffToJUnit`, `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`,
47
+ * `toBaselineFile`, `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`,
48
+ * `outputTokens`, `cacheTokens`, `formatEvalReport`, `formatCheckReport`,
49
+ * `formatTriggerRateReport`, `checkReportToJUnit`, `assertRates` — spend nothing,
50
+ * so they live on the free root barrel, unprefixed, even though their subject
51
+ * matter is evals.
52
+ *
53
+ * That is the one seam the split cannot make clean: the free helpers are typed
54
+ * over report shapes DEFINED here. The coupling is real and irreducible, so both
55
+ * barrels re-export the same report types. **The types are NOT prefixed** — the
56
+ * prefix is a warning about calling something, and a type is never called;
57
+ * `paid_EvalReport` would be noise on a symbol that cannot bill. Importing
58
+ * `vigiles/eval` for a type alone should never be necessary; importing it for a
59
+ * FUNCTION means you accepted a bill.
60
+ *
61
+ * Harness-specific eval drivers are not here either — `codexEvalDriver` /
62
+ * `codexEvalRunner` / `codexEvalAgentRunner` live on `vigiles/codex`, and
63
+ * `measureSelectionMatrix` / `assertNoCollision` on `vigiles/claude-code`, because
64
+ * those surfaces are chosen by harness, not by cost.
65
+ */
66
+ Object.defineProperty(exports, "__esModule", { value: true });
67
+ exports.paid_claudeEvalDriver = exports.paid_judged = exports.paid_judge = exports.paid_measureTriggerRate = exports.paid_measureArms = exports.paid_measure = exports.paid_runEval = void 0;
68
+ // --- the runners: each one drives a real model ---
69
+ /**
70
+ * Run an A/B eval across arms with a real model. Spends money.
71
+ *
72
+ * ⚠️ The `paid_` prefix overstates slightly: pass your own `evalDriver` and no
73
+ * model of ours is called. The DEFAULT path bills.
74
+ */
75
+ var eval_js_1 = require("./eval.js");
76
+ Object.defineProperty(exports, "paid_runEval", { enumerable: true, get: function () { return eval_js_1.runEval; } });
77
+ /**
78
+ * Score checks over N trials of one task with a real model. Spends money.
79
+ *
80
+ * ⚠️ `paid_` overstates slightly: with an injected `evalDriver` this drives
81
+ * whatever you supply. The DEFAULT path bills.
82
+ */
83
+ var eval_js_2 = require("./eval.js");
84
+ Object.defineProperty(exports, "paid_measure", { enumerable: true, get: function () { return eval_js_2.measure; } });
85
+ /**
86
+ * Score the same checks per arm (a hook/skill/rule on vs off) with a real model.
87
+ * Spends money.
88
+ *
89
+ * ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
90
+ * path. The DEFAULT path bills.
91
+ */
92
+ var eval_js_3 = require("./eval.js");
93
+ Object.defineProperty(exports, "paid_measureArms", { enumerable: true, get: function () { return eval_js_3.measureArms; } });
94
+ /**
95
+ * Measure whether a skill's description actually FIRES (recall + precision) by
96
+ * running real prompts against a real model. Spends money.
97
+ *
98
+ * ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
99
+ * path, and `stubSkillBodies` makes the default path much cheaper without making
100
+ * it free. The DEFAULT path bills.
101
+ */
102
+ var eval_js_4 = require("./eval.js");
103
+ Object.defineProperty(exports, "paid_measureTriggerRate", { enumerable: true, get: function () { return eval_js_4.measureTriggerRate; } });
104
+ /**
105
+ * The model-graded judge (harness-agnostic — grades text against a rubric).
106
+ * Shells out to the `claude` CLI synchronously; needs model auth. Spends money.
107
+ *
108
+ * ⚠️ Unlike the others this one has no injection seam — it always calls a model,
109
+ * so here `paid_` is exact.
110
+ */
111
+ var judge_js_1 = require("./judge.js");
112
+ Object.defineProperty(exports, "paid_judge", { enumerable: true, get: function () { return judge_js_1.judge; } });
113
+ /**
114
+ * The one member of the declarative check vocabulary that bills: its default
115
+ * judge is a real model call. Every other `Check` is deterministic and lives
116
+ * unprefixed on the free root barrel.
117
+ *
118
+ * ⚠️ The `paid_` prefix overstates slightly, and this is the symbol it overstates
119
+ * most: `paid_judged(rubric, { judge: myFn })` calls YOUR function and spends
120
+ * nothing. Only `paid_judged(rubric)` — the default — bills.
121
+ */
122
+ var check_js_1 = require("./check.js");
123
+ Object.defineProperty(exports, "paid_judged", { enumerable: true, get: function () { return check_js_1.judged; } });
124
+ /**
125
+ * The Claude-Code eval driver: the transport `paid_runEval` and friends use by
126
+ * default. Naming it here is how you pass it explicitly; using it spends money.
127
+ */
128
+ var eval_js_5 = require("./eval.js");
129
+ Object.defineProperty(exports, "paid_claudeEvalDriver", { enumerable: true, get: function () { return eval_js_5.claudeEvalDriver; } });
130
+ //# sourceMappingURL=eval-surface.js.map
@@ -309,7 +309,7 @@ interface MatcherOutput {
309
309
  * Custom matchers compatible with both vitest and jest. Register once:
310
310
  *
311
311
  * import { expect } from "vitest"; // or "@jest/globals"
312
- * import { vigilesMatchers } from "vigiles/testing";
312
+ * import { vigilesMatchers } from "vigiles";
313
313
  * expect.extend(vigilesMatchers);
314
314
  *
315
315
  * expect(result).toHaveCreated("RESULT");
@@ -69,7 +69,7 @@ const agent_result_js_1 = require("./adapters/claude-code/agent-result.js");
69
69
  const stats_js_1 = require("./stats.js");
70
70
  const eval_baseline_js_1 = require("./eval-baseline.js");
71
71
  // Re-export the significance primitives so the whole eval-analysis surface lives
72
- // re-exported from `vigiles/testing` (there is no `vigiles/harness-assert` entry
72
+ // re-exported from the `vigiles` root (there is no `vigiles/harness-assert` entry
73
73
  // point — it was advertised in this very comment and never existed).
74
74
  var stats_js_2 = require("./stats.js");
75
75
  Object.defineProperty(exports, "compareArms", { enumerable: true, get: function () { return stats_js_2.compareArms; } });
@@ -690,7 +690,7 @@ function assertTriggerRate(report, opts) {
690
690
  * Custom matchers compatible with both vitest and jest. Register once:
691
691
  *
692
692
  * import { expect } from "vitest"; // or "@jest/globals"
693
- * import { vigilesMatchers } from "vigiles/testing";
693
+ * import { vigilesMatchers } from "vigiles";
694
694
  * expect.extend(vigilesMatchers);
695
695
  *
696
696
  * expect(result).toHaveCreated("RESULT");
@@ -8,7 +8,7 @@ import { type AnyHook } from "./core/hook-program.js";
8
8
  * points at authoring the hook as `.mjs`.
9
9
  *
10
10
  * ```js
11
- * import { loadHook, assertHookDenies } from "vigiles/testing";
11
+ * import { loadHook, assertHookDenies } from "vigiles";
12
12
  *
13
13
  * const guard = await loadHook(".vigiles/hooks/guard.mjs");
14
14
  * assertHookDenies(guard, {
package/dist/load-hook.js CHANGED
@@ -27,7 +27,7 @@ const hook_program_js_1 = require("./core/hook-program.js");
27
27
  * points at authoring the hook as `.mjs`.
28
28
  *
29
29
  * ```js
30
- * import { loadHook, assertHookDenies } from "vigiles/testing";
30
+ * import { loadHook, assertHookDenies } from "vigiles";
31
31
  *
32
32
  * const guard = await loadHook(".vigiles/hooks/guard.mjs");
33
33
  * assertHookDenies(guard, {
package/dist/run-hook.js CHANGED
@@ -11,7 +11,7 @@ const hook_protocol_js_1 = require("./adapters/claude-code/hook-protocol.js");
11
11
  const proofs_js_1 = require("./core/proofs.js");
12
12
  const hook_program_js_1 = require("./core/hook-program.js");
13
13
  const run_script_js_1 = require("./run-script.js");
14
- // The general runner this module specializes. Re-exported so `vigiles/unit`'s
14
+ // The general runner this module specializes. Re-exported so the `vigiles` root's
15
15
  // hook surface and the script surface stay one import away from each other.
16
16
  var run_script_js_2 = require("./run-script.js");
17
17
  Object.defineProperty(exports, "runScript", { enumerable: true, get: function () { return run_script_js_2.runScript; } });