vigiles 15.4.0 → 16.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/run-scripts.d.ts +4 -4
- package/dist/adapters/claude-code/run-scripts.js +6 -6
- package/dist/adapters/claude-code/typed-spec.d.ts +26 -3
- package/dist/adapters/claude-code/typed-spec.js +21 -4
- package/dist/audit-report.template.html +1 -1
- package/dist/audit-score.d.ts +1 -1
- package/dist/audit-score.js +2 -2
- package/dist/check-count.js +1 -1
- package/dist/claude-code.d.ts +1 -1
- package/dist/claude-code.js +3 -3
- package/dist/cli.js +8 -8
- package/dist/codex.d.ts +1 -1
- package/dist/codex.js +1 -1
- package/dist/core/spec.d.ts +37 -5
- package/dist/core/spec.js +33 -7
- package/dist/eval-surface.d.ts +122 -0
- package/dist/eval-surface.js +130 -0
- package/dist/harness-assert.d.ts +1 -1
- package/dist/harness-assert.js +2 -2
- package/dist/load-hook.d.ts +1 -1
- package/dist/load-hook.js +1 -1
- package/dist/run-hook.js +1 -1
- package/dist/scaffold-test.js +7 -11
- package/dist/scan-trigger-suggest.d.ts +3 -3
- package/dist/scan-trigger-suggest.js +4 -4
- package/dist/test-coverage.js +1 -1
- package/dist/test.d.ts +75 -0
- package/dist/test.js +192 -0
- package/package.json +9 -8
- package/skills/test-harness/SKILL.md +16 -14
- package/dist/e2e.d.ts +0 -18
- package/dist/e2e.js +0 -34
- package/dist/integration.d.ts +0 -29
- package/dist/integration.js +0 -58
- package/dist/testing.d.ts +0 -41
- package/dist/testing.js +0 -123
- package/dist/unit.d.ts +0 -39
- package/dist/unit.js +0 -74
package/dist/audit-score.d.ts
CHANGED
|
@@ -23,7 +23,7 @@
|
|
|
23
23
|
* design). NB the EXECUTING
|
|
24
24
|
* "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
|
|
25
25
|
* running arbitrary hooks safely needs cross-platform confinement that isn't
|
|
26
|
-
* shipped yet, so the battery lives in the `vigiles
|
|
26
|
+
* shipped yet, so the battery lives in the `vigiles` testing API via
|
|
27
27
|
* `guardrail-check`/`assertBlocksDisasters`, where you opt in explicitly.
|
|
28
28
|
*
|
|
29
29
|
* TESTED vs EVALUATED are two rings, not one, because a harness and an eval differ
|
package/dist/audit-score.js
CHANGED
|
@@ -28,7 +28,7 @@ exports.formatAuditScore = formatAuditScore;
|
|
|
28
28
|
* design). NB the EXECUTING
|
|
29
29
|
* "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
|
|
30
30
|
* running arbitrary hooks safely needs cross-platform confinement that isn't
|
|
31
|
-
* shipped yet, so the battery lives in the `vigiles
|
|
31
|
+
* shipped yet, so the battery lives in the `vigiles` testing API via
|
|
32
32
|
* `guardrail-check`/`assertBlocksDisasters`, where you opt in explicitly.
|
|
33
33
|
*
|
|
34
34
|
* TESTED vs EVALUATED are two rings, not one, because a harness and an eval differ
|
|
@@ -49,7 +49,7 @@ const score_core_js_1 = require("./score-core.js");
|
|
|
49
49
|
const W_UNTESTED = 3;
|
|
50
50
|
/** The command that answers the firing question — named, not alluded to. */
|
|
51
51
|
const MEASURE_FIRING_COMMAND = "run `npx vigiles audit` interactively to measure, or add a `*.eval.mjs` " +
|
|
52
|
-
"(`
|
|
52
|
+
"(`paid_measureTriggerRate`, vigiles/eval)";
|
|
53
53
|
/** Resolve the terse "thing(s)" plural placeholder against a count:
|
|
54
54
|
* n===1 drops the "(s)" ("1 tool"); otherwise it becomes "s" ("3 tools"). */
|
|
55
55
|
function pluralizeLabel(n, label) {
|
package/dist/check-count.js
CHANGED
|
@@ -46,7 +46,7 @@ exports.armCheckReport = armCheckReport;
|
|
|
46
46
|
* tool avoids.
|
|
47
47
|
*
|
|
48
48
|
* WHAT A MISSING COUNT MEANS: nothing at all. A script that never imports
|
|
49
|
-
* `vigiles
|
|
49
|
+
* `vigiles` cannot report, so the runner sees no file and treats it
|
|
50
50
|
* exactly as before — a plain pass. Silence is the legacy branch, never a
|
|
51
51
|
* verdict; only a count of literally zero is a finding. (The alternative —
|
|
52
52
|
* force-loading this module into every child with `node --import` so silence
|
package/dist/claude-code.d.ts
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
* `vigiles/claude-code` — the Claude Code-specific harness pieces a *different*
|
|
3
3
|
* harness would swap out: the plugin/repo loader (reads real Claude Code plugin
|
|
4
4
|
* layouts) and the scriptable Anthropic Messages mock. Split from
|
|
5
|
-
* `vigiles
|
|
5
|
+
* the `vigiles` root on purpose — the test API above is the stable surface; this
|
|
6
6
|
* is the adapter, so a future `vigiles/<other-harness>` can sit beside it.
|
|
7
7
|
*/
|
|
8
8
|
export * from "./adapters/claude-code/plugin-loader.js";
|
package/dist/claude-code.js
CHANGED
|
@@ -19,7 +19,7 @@ exports.formatSelectionReport = exports.assertNoCollision = exports.measureSelec
|
|
|
19
19
|
* `vigiles/claude-code` — the Claude Code-specific harness pieces a *different*
|
|
20
20
|
* harness would swap out: the plugin/repo loader (reads real Claude Code plugin
|
|
21
21
|
* layouts) and the scriptable Anthropic Messages mock. Split from
|
|
22
|
-
* `vigiles
|
|
22
|
+
* the `vigiles` root on purpose — the test API above is the stable surface; this
|
|
23
23
|
* is the adapter, so a future `vigiles/<other-harness>` can sit beside it.
|
|
24
24
|
*/
|
|
25
25
|
__exportStar(require("./adapters/claude-code/plugin-loader.js"), exports);
|
|
@@ -28,7 +28,7 @@ __exportStar(require("./mock-model.js"), exports);
|
|
|
28
28
|
// helpers + the `claude` capability probe. Agnostic users never need these
|
|
29
29
|
// (`runHarnessTest` defaults to the CC driver), but they're exposed here — beside
|
|
30
30
|
// the Codex driver in `vigiles/codex` — for CC-specific tests/tooling. They are
|
|
31
|
-
// deliberately NOT on the agnostic `vigiles
|
|
31
|
+
// deliberately NOT on the agnostic `vigiles` surface.
|
|
32
32
|
var harness_test_js_1 = require("./harness-test.js");
|
|
33
33
|
Object.defineProperty(exports, "claudeCodeDriver", { enumerable: true, get: function () { return harness_test_js_1.claudeCodeDriver; } });
|
|
34
34
|
Object.defineProperty(exports, "buildClaudeArgs", { enumerable: true, get: function () { return harness_test_js_1.buildClaudeArgs; } });
|
|
@@ -50,7 +50,7 @@ Object.defineProperty(exports, "agent", { enumerable: true, get: function () { r
|
|
|
50
50
|
Object.defineProperty(exports, "experimental_skill", { enumerable: true, get: function () { return typed_spec_js_1.experimental_skill; } });
|
|
51
51
|
// Selection-collision — a Claude-Code-ONLY behavioral measurement (Codex has no
|
|
52
52
|
// skill-selection event to read), so it lives on this surface, not the agnostic
|
|
53
|
-
// `vigiles
|
|
53
|
+
// `vigiles`. `measureSelectionMatrix` builds the N×N "which skill fired?"
|
|
54
54
|
// matrix (diagonal = recall, off-diagonal = collision); `assertNoCollision` gates it.
|
|
55
55
|
var scan_behavioral_js_1 = require("./scan-behavioral.js");
|
|
56
56
|
Object.defineProperty(exports, "measureSelectionMatrix", { enumerable: true, get: function () { return scan_behavioral_js_1.measureSelectionMatrix; } });
|
package/dist/cli.js
CHANGED
|
@@ -1702,7 +1702,7 @@ function formatTriggerNudge(triggerableSkills) {
|
|
|
1702
1702
|
return "";
|
|
1703
1703
|
const n = triggerableSkills;
|
|
1704
1704
|
return (`ℹ Do your ${String(n)} skill${n === 1 ? "" : "s"} actually fire? The deterministic read can't tell — ` +
|
|
1705
|
-
`run \`audit\` interactively to measure, or test with \`
|
|
1705
|
+
`run \`audit\` interactively to measure, or test with \`paid_measureTriggerRate\` (vigiles/eval).`);
|
|
1706
1706
|
}
|
|
1707
1707
|
/** A terminal summary of the rule map: the CONFIDENT lane counts + the POSSIBLE
|
|
1708
1708
|
* (review) and SKIPPED tiers, with the honest caveat that detection is a
|
|
@@ -1973,7 +1973,7 @@ function vigilesWorkflow(plan, harnesses, hasPackageJson) {
|
|
|
1973
1973
|
`
|
|
1974
1974
|
: "";
|
|
1975
1975
|
// The test pillar's harness job runs `*.harness.mjs` files that
|
|
1976
|
-
// `import "vigiles
|
|
1976
|
+
// `import "vigiles"`, so it needs vigiles RESOLVABLE — a package.json +
|
|
1977
1977
|
// install. A repo with no package.json can't run the JS test tier in CI at all
|
|
1978
1978
|
// (the import won't resolve even though the CLI itself came from npx), so the
|
|
1979
1979
|
// harness job is emitted ONLY when a package.json exists (Codex review / #111 /
|
|
@@ -2136,7 +2136,7 @@ const STARTER_HARNESS = `/**
|
|
|
2136
2136
|
*
|
|
2137
2137
|
* Guide: https://github.com/zernie/vigiles/blob/main/docs/harness-testing.md
|
|
2138
2138
|
*/
|
|
2139
|
-
import { runHook } from "vigiles
|
|
2139
|
+
import { runHook } from "vigiles";
|
|
2140
2140
|
import assert from "node:assert/strict";
|
|
2141
2141
|
|
|
2142
2142
|
// EXAMPLE — replace with one of YOUR hooks. This PreToolUse Bash guard blocks a
|
|
@@ -4384,7 +4384,7 @@ const COMMAND_HELP = {
|
|
|
4384
4384
|
usage: " vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)",
|
|
4385
4385
|
detail: [
|
|
4386
4386
|
" writes vigiles-report.html + .json (auto-gitignored; --out=<dir> · --no-html/--no-json · --no-open · --json for machine output). NOT a CI step — use `vigiles lint` in CI.",
|
|
4387
|
-
" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles
|
|
4387
|
+
" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles testing API",
|
|
4388
4388
|
" --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt",
|
|
4389
4389
|
],
|
|
4390
4390
|
},
|
|
@@ -4970,7 +4970,7 @@ function refsHookCommand() {
|
|
|
4970
4970
|
// ---------------------------------------------------------------------------
|
|
4971
4971
|
/**
|
|
4972
4972
|
* Load a compiled-hook program's default export — the SHARED loader, also the
|
|
4973
|
-
* public `vigiles
|
|
4973
|
+
* public `vigiles` `loadHook` a `.harness.mjs` test uses, so a hook that
|
|
4974
4974
|
* loads in a test loads identically here (one loader, no drift).
|
|
4975
4975
|
*/
|
|
4976
4976
|
const loadHookProgram = load_hook_js_1.loadHook;
|
|
@@ -6135,7 +6135,7 @@ async function main() {
|
|
|
6135
6135
|
// uses `vigiles lint`). The executing checks (safety battery + live MCP +
|
|
6136
6136
|
// skill firing) run only on consent: at a TTY `audit` asks once (remembered
|
|
6137
6137
|
// in `.vigilesrc.json`), headless it stays a read + a one-line nudge. There
|
|
6138
|
-
// is NO execution flag — automation tests the harness via the vigiles
|
|
6138
|
+
// is NO execution flag — automation tests the harness via the vigiles testing
|
|
6139
6139
|
// API + skills, not the report verb. See the `audit-side-effect-free` rule.
|
|
6140
6140
|
const dirs = restArgs.length > 0 ? restArgs : ["."];
|
|
6141
6141
|
const json = args.includes("--json");
|
|
@@ -6337,10 +6337,10 @@ async function main() {
|
|
|
6337
6337
|
// linter also executes it). A plain `audit` is a deterministic READ;
|
|
6338
6338
|
// these run only on consent — ASK once at a TTY (remembered); headless
|
|
6339
6339
|
// stays a read + a nudge (no execution flag — automation uses the
|
|
6340
|
-
// vigiles
|
|
6340
|
+
// vigiles testing API). Resolved AFTER the report (so the read leads) but
|
|
6341
6341
|
// BEFORE the rule map is routed, so a first-time "yes" enriches THIS run.
|
|
6342
6342
|
// (The safety battery is NOT here — it needs cross-platform confinement
|
|
6343
|
-
// that isn't shipped, so it lives in the vigiles
|
|
6343
|
+
// that isn't shipped, so it lives in the vigiles testing API.)
|
|
6344
6344
|
// `surfaces` + `execDecision` were resolved above (the ring needed them);
|
|
6345
6345
|
// this only turns an `ask` into a prompt and remembers the answer.
|
|
6346
6346
|
const { execute, note: execNote } = await resolveExecution(surfaces, execDecision, json, adapter.name);
|
package/dist/codex.d.ts
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* `vigiles/codex` — the OpenAI Codex harness adapter. Sits beside
|
|
3
|
-
* `vigiles/claude-code`: same harness-agnostic `vigiles
|
|
3
|
+
* `vigiles/claude-code`: same harness-agnostic `vigiles` testing core, a
|
|
4
4
|
* different transport (the `codex` binary + the OpenAI **Responses** SSE mock).
|
|
5
5
|
*
|
|
6
6
|
* Pillar 2 (harness testing) is proven here — `startCodexMock` serves the
|
package/dist/codex.js
CHANGED
|
@@ -16,7 +16,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
|
16
16
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
17
|
/**
|
|
18
18
|
* `vigiles/codex` — the OpenAI Codex harness adapter. Sits beside
|
|
19
|
-
* `vigiles/claude-code`: same harness-agnostic `vigiles
|
|
19
|
+
* `vigiles/claude-code`: same harness-agnostic `vigiles` testing core, a
|
|
20
20
|
* different transport (the `codex` binary + the OpenAI **Responses** SSE mock).
|
|
21
21
|
*
|
|
22
22
|
* Pillar 2 (harness testing) is proven here — `startCodexMock` serves the
|
package/dist/core/spec.d.ts
CHANGED
|
@@ -399,12 +399,21 @@ export interface SkillStep {
|
|
|
399
399
|
/** Max attempts to satisfy the gate before the step fails (default 1). */
|
|
400
400
|
readonly retry?: number;
|
|
401
401
|
}
|
|
402
|
-
/**
|
|
403
|
-
|
|
402
|
+
/**
|
|
403
|
+
* Declare a skill input (compiles to argument-hint + an Arguments entry).
|
|
404
|
+
*
|
|
405
|
+
* Deliberately NOT a standalone export — reached as `experimental_skill.input`.
|
|
406
|
+
* The reason is on `experimental_skill` below.
|
|
407
|
+
*/
|
|
408
|
+
declare function input(name: string, hint: string, opts?: {
|
|
404
409
|
required?: boolean;
|
|
405
410
|
}): SkillInput;
|
|
406
|
-
/**
|
|
407
|
-
|
|
411
|
+
/**
|
|
412
|
+
* Declare a gated pipeline step.
|
|
413
|
+
*
|
|
414
|
+
* Deliberately NOT a standalone export — reached as `experimental_skill.step`.
|
|
415
|
+
*/
|
|
416
|
+
declare function step(instr: string | InstructionFragment[], opts?: {
|
|
408
417
|
gate?: Gate;
|
|
409
418
|
retry?: number;
|
|
410
419
|
}): SkillStep;
|
|
@@ -504,9 +513,32 @@ export type SkillSpecInput<P extends AuthoredPurity | undefined, V extends ToolV
|
|
|
504
513
|
* DIFFERENT function — a `Check<Trace>` taking an id string, asking whether a
|
|
505
514
|
* skill fired. This one authors a skill; that one observes one.
|
|
506
515
|
*
|
|
516
|
+
* Its helper vocabulary hangs off it — `experimental_skill.input(…)` and
|
|
517
|
+
* `experimental_skill.step(…)` — rather than being exported beside it. Both are
|
|
518
|
+
* used ONLY by skill specs (measured: zero uses in agent/claude specs, against
|
|
519
|
+
* `cmd`/`file`/`ref`/`result`, which are shared and therefore stay top-level).
|
|
520
|
+
* Hanging them here makes the experimental marking STRUCTURAL for the whole
|
|
521
|
+
* family: you cannot reach `input()` without naming `experimental_skill` first.
|
|
522
|
+
* The prefix convention alone could not do that — it is a habit, and it had
|
|
523
|
+
* already leaked once when `skill()` shipped stable-named against its own docs.
|
|
524
|
+
*
|
|
525
|
+
* Honest limit: `const { input } = experimental_skill` strips the marker again
|
|
526
|
+
* inside one file. What the shape actually guarantees is narrower and still
|
|
527
|
+
* worth having — an unmarked name never crosses the package boundary.
|
|
528
|
+
*
|
|
529
|
+
* `Object.assign` rather than `export namespace`: the latter is banned by this
|
|
530
|
+
* repo's own lint (`no-namespace: error`, inherited from strict-type-checked).
|
|
531
|
+
*
|
|
532
|
+
* @experimental
|
|
533
|
+
*/
|
|
534
|
+
declare function skillSpec<const P extends AuthoredPurity | undefined = undefined, V extends ToolVocabulary = OpenToolVocabulary>(spec: SkillSpecInput<P, V>): SkillSpec;
|
|
535
|
+
/**
|
|
507
536
|
* @experimental
|
|
508
537
|
*/
|
|
509
|
-
export declare
|
|
538
|
+
export declare const experimental_skill: typeof skillSpec & {
|
|
539
|
+
input: typeof input;
|
|
540
|
+
step: typeof step;
|
|
541
|
+
};
|
|
510
542
|
/**
|
|
511
543
|
* A subagent definition (compiles to `agents/<name>.md`). Unlike a skill —
|
|
512
544
|
* reference material the model reads on activation — a subagent is a *delegated
|
package/dist/core/spec.js
CHANGED
|
@@ -10,7 +10,7 @@
|
|
|
10
10
|
* guidance() — prose only, no mechanical enforcement
|
|
11
11
|
*/
|
|
12
12
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
13
|
-
exports.BUILTIN_LINTERS = void 0;
|
|
13
|
+
exports.experimental_skill = exports.BUILTIN_LINTERS = void 0;
|
|
14
14
|
exports.enforce = enforce;
|
|
15
15
|
exports.guidance = guidance;
|
|
16
16
|
exports.guard = guard;
|
|
@@ -24,9 +24,6 @@ exports.instructions = instructions;
|
|
|
24
24
|
exports.experimental_effect = experimental_effect;
|
|
25
25
|
exports.claude = claude;
|
|
26
26
|
exports.project = project;
|
|
27
|
-
exports.input = input;
|
|
28
|
-
exports.step = step;
|
|
29
|
-
exports.experimental_skill = experimental_skill;
|
|
30
27
|
exports.agent = agent;
|
|
31
28
|
exports.result = result;
|
|
32
29
|
exports.delegate = delegate;
|
|
@@ -222,11 +219,20 @@ function claude(spec) {
|
|
|
222
219
|
function project(role) {
|
|
223
220
|
return { _ref: "role", role };
|
|
224
221
|
}
|
|
225
|
-
/**
|
|
222
|
+
/**
|
|
223
|
+
* Declare a skill input (compiles to argument-hint + an Arguments entry).
|
|
224
|
+
*
|
|
225
|
+
* Deliberately NOT a standalone export — reached as `experimental_skill.input`.
|
|
226
|
+
* The reason is on `experimental_skill` below.
|
|
227
|
+
*/
|
|
226
228
|
function input(name, hint, opts = {}) {
|
|
227
229
|
return { name, hint, required: opts.required };
|
|
228
230
|
}
|
|
229
|
-
/**
|
|
231
|
+
/**
|
|
232
|
+
* Declare a gated pipeline step.
|
|
233
|
+
*
|
|
234
|
+
* Deliberately NOT a standalone export — reached as `experimental_skill.step`.
|
|
235
|
+
*/
|
|
230
236
|
function step(instr, opts = {}) {
|
|
231
237
|
return { do: instr, gate: opts.gate, retry: opts.retry };
|
|
232
238
|
}
|
|
@@ -245,11 +251,31 @@ function step(instr, opts = {}) {
|
|
|
245
251
|
* DIFFERENT function — a `Check<Trace>` taking an id string, asking whether a
|
|
246
252
|
* skill fired. This one authors a skill; that one observes one.
|
|
247
253
|
*
|
|
254
|
+
* Its helper vocabulary hangs off it — `experimental_skill.input(…)` and
|
|
255
|
+
* `experimental_skill.step(…)` — rather than being exported beside it. Both are
|
|
256
|
+
* used ONLY by skill specs (measured: zero uses in agent/claude specs, against
|
|
257
|
+
* `cmd`/`file`/`ref`/`result`, which are shared and therefore stay top-level).
|
|
258
|
+
* Hanging them here makes the experimental marking STRUCTURAL for the whole
|
|
259
|
+
* family: you cannot reach `input()` without naming `experimental_skill` first.
|
|
260
|
+
* The prefix convention alone could not do that — it is a habit, and it had
|
|
261
|
+
* already leaked once when `skill()` shipped stable-named against its own docs.
|
|
262
|
+
*
|
|
263
|
+
* Honest limit: `const { input } = experimental_skill` strips the marker again
|
|
264
|
+
* inside one file. What the shape actually guarantees is narrower and still
|
|
265
|
+
* worth having — an unmarked name never crosses the package boundary.
|
|
266
|
+
*
|
|
267
|
+
* `Object.assign` rather than `export namespace`: the latter is banned by this
|
|
268
|
+
* repo's own lint (`no-namespace: error`, inherited from strict-type-checked).
|
|
269
|
+
*
|
|
248
270
|
* @experimental
|
|
249
271
|
*/
|
|
250
|
-
function
|
|
272
|
+
function skillSpec(spec) {
|
|
251
273
|
return { _specType: "skill", ...spec };
|
|
252
274
|
}
|
|
275
|
+
/**
|
|
276
|
+
* @experimental
|
|
277
|
+
*/
|
|
278
|
+
exports.experimental_skill = Object.assign(skillSpec, { input, step });
|
|
253
279
|
/**
|
|
254
280
|
* Define a subagent specification (compiles to `agents/<name>.md`).
|
|
255
281
|
*
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `vigiles/eval` — the **paid** surface: every runtime export here can call a
|
|
3
|
+
* model, and therefore can spend money.
|
|
4
|
+
*
|
|
5
|
+
* That single property is the whole reason this subpath exists. The package used
|
|
6
|
+
* to split its testing API by TEST TIER (`vigiles/unit`, `vigiles/integration`,
|
|
7
|
+
* `vigiles/e2e`, `vigiles/testing`); it now splits on COST. Everything free is on
|
|
8
|
+
* the package root, [`vigiles`](./test.ts). Everything that bills is here. The
|
|
9
|
+
* name rhymes with the `vigiles eval` CLI verb.
|
|
10
|
+
*
|
|
11
|
+
* ## Why every runtime export is also named `paid_`
|
|
12
|
+
*
|
|
13
|
+
* The import path warns ONCE, at the top of the file. The name warns EVERY time,
|
|
14
|
+
* at the call site. Reading `await judged(trace, "did it refuse?")` on line 140,
|
|
15
|
+
* the import line is long out of view — `await paid_judged(...)` still says what
|
|
16
|
+
* it costs. This is not a new idiom in this package: `vigiles/experimental`
|
|
17
|
+
* already pairs a quarantined subpath with an `experimental_` name prefix for
|
|
18
|
+
* exactly this reason. The same device, applied to a second axis.
|
|
19
|
+
*
|
|
20
|
+
* ⚠️ **The prefix slightly OVERSTATES the cost, and that is a deliberate trade
|
|
21
|
+
* rather than an oversight.** `paid_judged` takes an injectable judge:
|
|
22
|
+
* `paid_judged(rubric, { judge: myFn })` calls no model and spends nothing — only
|
|
23
|
+
* the DEFAULT judge bills. `paid_measureTriggerRate` and friends likewise accept
|
|
24
|
+
* an injected `evalDriver`. The strictly accurate prefix would be `metered_`, but
|
|
25
|
+
* it reads a beat slower, and a warning that is not absorbed at a glance is not a
|
|
26
|
+
* warning. Clarity wins; the imprecision is recorded here and on each symbol
|
|
27
|
+
* rather than left for a reader to discover.
|
|
28
|
+
*
|
|
29
|
+
* ## Why the split had to be cost, not tier
|
|
30
|
+
*
|
|
31
|
+
* The old `vigiles/unit` docstring promised "no `claude`, no model, no
|
|
32
|
+
* bubblewrap, no network" — and re-exported `judged`, whose default judge is a
|
|
33
|
+
* real model call (`check.ts` resolves `opts.judge ?? ((o) => runJudge(o))`). A
|
|
34
|
+
* test importing only from that barrel could still spend money, and the one line
|
|
35
|
+
* a reader would have relied on to know otherwise was the false one. `judged`
|
|
36
|
+
* moved HERE and became `paid_judged`, which makes that class of defect
|
|
37
|
+
* unrepresentable rather than apologised for: on this path, billing is the
|
|
38
|
+
* advertised property, in the subpath and in every name.
|
|
39
|
+
*
|
|
40
|
+
* ## What is NOT here, and why
|
|
41
|
+
*
|
|
42
|
+
* The ~two dozen helpers that read an eval RESULT — `assertImproves`,
|
|
43
|
+
* `assertNoRegression`, `assertReliable`, `assertSignificant`, `assertTriggerRate`,
|
|
44
|
+
* `reliable`, `improvement`, `significantlyBeats`, `compareArms`, `diffReports`,
|
|
45
|
+
* `diffToJUnit`, `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`,
|
|
46
|
+
* `toBaselineFile`, `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`,
|
|
47
|
+
* `outputTokens`, `cacheTokens`, `formatEvalReport`, `formatCheckReport`,
|
|
48
|
+
* `formatTriggerRateReport`, `checkReportToJUnit`, `assertRates` — spend nothing,
|
|
49
|
+
* so they live on the free root barrel, unprefixed, even though their subject
|
|
50
|
+
* matter is evals.
|
|
51
|
+
*
|
|
52
|
+
* That is the one seam the split cannot make clean: the free helpers are typed
|
|
53
|
+
* over report shapes DEFINED here. The coupling is real and irreducible, so both
|
|
54
|
+
* barrels re-export the same report types. **The types are NOT prefixed** — the
|
|
55
|
+
* prefix is a warning about calling something, and a type is never called;
|
|
56
|
+
* `paid_EvalReport` would be noise on a symbol that cannot bill. Importing
|
|
57
|
+
* `vigiles/eval` for a type alone should never be necessary; importing it for a
|
|
58
|
+
* FUNCTION means you accepted a bill.
|
|
59
|
+
*
|
|
60
|
+
* Harness-specific eval drivers are not here either — `codexEvalDriver` /
|
|
61
|
+
* `codexEvalRunner` / `codexEvalAgentRunner` live on `vigiles/codex`, and
|
|
62
|
+
* `measureSelectionMatrix` / `assertNoCollision` on `vigiles/claude-code`, because
|
|
63
|
+
* those surfaces are chosen by harness, not by cost.
|
|
64
|
+
*/
|
|
65
|
+
/**
|
|
66
|
+
* Run an A/B eval across arms with a real model. Spends money.
|
|
67
|
+
*
|
|
68
|
+
* ⚠️ The `paid_` prefix overstates slightly: pass your own `evalDriver` and no
|
|
69
|
+
* model of ours is called. The DEFAULT path bills.
|
|
70
|
+
*/
|
|
71
|
+
export { runEval as paid_runEval } from "./eval.js";
|
|
72
|
+
/**
|
|
73
|
+
* Score checks over N trials of one task with a real model. Spends money.
|
|
74
|
+
*
|
|
75
|
+
* ⚠️ `paid_` overstates slightly: with an injected `evalDriver` this drives
|
|
76
|
+
* whatever you supply. The DEFAULT path bills.
|
|
77
|
+
*/
|
|
78
|
+
export { measure as paid_measure } from "./eval.js";
|
|
79
|
+
/**
|
|
80
|
+
* Score the same checks per arm (a hook/skill/rule on vs off) with a real model.
|
|
81
|
+
* Spends money.
|
|
82
|
+
*
|
|
83
|
+
* ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
|
|
84
|
+
* path. The DEFAULT path bills.
|
|
85
|
+
*/
|
|
86
|
+
export { measureArms as paid_measureArms } from "./eval.js";
|
|
87
|
+
/**
|
|
88
|
+
* Measure whether a skill's description actually FIRES (recall + precision) by
|
|
89
|
+
* running real prompts against a real model. Spends money.
|
|
90
|
+
*
|
|
91
|
+
* ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
|
|
92
|
+
* path, and `stubSkillBodies` makes the default path much cheaper without making
|
|
93
|
+
* it free. The DEFAULT path bills.
|
|
94
|
+
*/
|
|
95
|
+
export { measureTriggerRate as paid_measureTriggerRate } from "./eval.js";
|
|
96
|
+
/**
|
|
97
|
+
* The model-graded judge (harness-agnostic — grades text against a rubric).
|
|
98
|
+
* Shells out to the `claude` CLI synchronously; needs model auth. Spends money.
|
|
99
|
+
*
|
|
100
|
+
* ⚠️ Unlike the others this one has no injection seam — it always calls a model,
|
|
101
|
+
* so here `paid_` is exact.
|
|
102
|
+
*/
|
|
103
|
+
export { judge as paid_judge } from "./judge.js";
|
|
104
|
+
/**
|
|
105
|
+
* The one member of the declarative check vocabulary that bills: its default
|
|
106
|
+
* judge is a real model call. Every other `Check` is deterministic and lives
|
|
107
|
+
* unprefixed on the free root barrel.
|
|
108
|
+
*
|
|
109
|
+
* ⚠️ The `paid_` prefix overstates slightly, and this is the symbol it overstates
|
|
110
|
+
* most: `paid_judged(rubric, { judge: myFn })` calls YOUR function and spends
|
|
111
|
+
* nothing. Only `paid_judged(rubric)` — the default — bills.
|
|
112
|
+
*/
|
|
113
|
+
export { judged as paid_judged } from "./check.js";
|
|
114
|
+
/**
|
|
115
|
+
* The Claude-Code eval driver: the transport `paid_runEval` and friends use by
|
|
116
|
+
* default. Naming it here is how you pass it explicitly; using it spends money.
|
|
117
|
+
*/
|
|
118
|
+
export { claudeEvalDriver as paid_claudeEvalDriver } from "./eval.js";
|
|
119
|
+
export type { Check, CheckResult, JudgeFn } from "./check.js";
|
|
120
|
+
export type { Trace } from "./harness-test.js";
|
|
121
|
+
export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
|
|
122
|
+
//# sourceMappingURL=eval-surface.d.ts.map
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* `vigiles/eval` — the **paid** surface: every runtime export here can call a
|
|
4
|
+
* model, and therefore can spend money.
|
|
5
|
+
*
|
|
6
|
+
* That single property is the whole reason this subpath exists. The package used
|
|
7
|
+
* to split its testing API by TEST TIER (`vigiles/unit`, `vigiles/integration`,
|
|
8
|
+
* `vigiles/e2e`, `vigiles/testing`); it now splits on COST. Everything free is on
|
|
9
|
+
* the package root, [`vigiles`](./test.ts). Everything that bills is here. The
|
|
10
|
+
* name rhymes with the `vigiles eval` CLI verb.
|
|
11
|
+
*
|
|
12
|
+
* ## Why every runtime export is also named `paid_`
|
|
13
|
+
*
|
|
14
|
+
* The import path warns ONCE, at the top of the file. The name warns EVERY time,
|
|
15
|
+
* at the call site. Reading `await judged(trace, "did it refuse?")` on line 140,
|
|
16
|
+
* the import line is long out of view — `await paid_judged(...)` still says what
|
|
17
|
+
* it costs. This is not a new idiom in this package: `vigiles/experimental`
|
|
18
|
+
* already pairs a quarantined subpath with an `experimental_` name prefix for
|
|
19
|
+
* exactly this reason. The same device, applied to a second axis.
|
|
20
|
+
*
|
|
21
|
+
* ⚠️ **The prefix slightly OVERSTATES the cost, and that is a deliberate trade
|
|
22
|
+
* rather than an oversight.** `paid_judged` takes an injectable judge:
|
|
23
|
+
* `paid_judged(rubric, { judge: myFn })` calls no model and spends nothing — only
|
|
24
|
+
* the DEFAULT judge bills. `paid_measureTriggerRate` and friends likewise accept
|
|
25
|
+
* an injected `evalDriver`. The strictly accurate prefix would be `metered_`, but
|
|
26
|
+
* it reads a beat slower, and a warning that is not absorbed at a glance is not a
|
|
27
|
+
* warning. Clarity wins; the imprecision is recorded here and on each symbol
|
|
28
|
+
* rather than left for a reader to discover.
|
|
29
|
+
*
|
|
30
|
+
* ## Why the split had to be cost, not tier
|
|
31
|
+
*
|
|
32
|
+
* The old `vigiles/unit` docstring promised "no `claude`, no model, no
|
|
33
|
+
* bubblewrap, no network" — and re-exported `judged`, whose default judge is a
|
|
34
|
+
* real model call (`check.ts` resolves `opts.judge ?? ((o) => runJudge(o))`). A
|
|
35
|
+
* test importing only from that barrel could still spend money, and the one line
|
|
36
|
+
* a reader would have relied on to know otherwise was the false one. `judged`
|
|
37
|
+
* moved HERE and became `paid_judged`, which makes that class of defect
|
|
38
|
+
* unrepresentable rather than apologised for: on this path, billing is the
|
|
39
|
+
* advertised property, in the subpath and in every name.
|
|
40
|
+
*
|
|
41
|
+
* ## What is NOT here, and why
|
|
42
|
+
*
|
|
43
|
+
* The ~two dozen helpers that read an eval RESULT — `assertImproves`,
|
|
44
|
+
* `assertNoRegression`, `assertReliable`, `assertSignificant`, `assertTriggerRate`,
|
|
45
|
+
* `reliable`, `improvement`, `significantlyBeats`, `compareArms`, `diffReports`,
|
|
46
|
+
* `diffToJUnit`, `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`,
|
|
47
|
+
* `toBaselineFile`, `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`,
|
|
48
|
+
* `outputTokens`, `cacheTokens`, `formatEvalReport`, `formatCheckReport`,
|
|
49
|
+
* `formatTriggerRateReport`, `checkReportToJUnit`, `assertRates` — spend nothing,
|
|
50
|
+
* so they live on the free root barrel, unprefixed, even though their subject
|
|
51
|
+
* matter is evals.
|
|
52
|
+
*
|
|
53
|
+
* That is the one seam the split cannot make clean: the free helpers are typed
|
|
54
|
+
* over report shapes DEFINED here. The coupling is real and irreducible, so both
|
|
55
|
+
* barrels re-export the same report types. **The types are NOT prefixed** — the
|
|
56
|
+
* prefix is a warning about calling something, and a type is never called;
|
|
57
|
+
* `paid_EvalReport` would be noise on a symbol that cannot bill. Importing
|
|
58
|
+
* `vigiles/eval` for a type alone should never be necessary; importing it for a
|
|
59
|
+
* FUNCTION means you accepted a bill.
|
|
60
|
+
*
|
|
61
|
+
* Harness-specific eval drivers are not here either — `codexEvalDriver` /
|
|
62
|
+
* `codexEvalRunner` / `codexEvalAgentRunner` live on `vigiles/codex`, and
|
|
63
|
+
* `measureSelectionMatrix` / `assertNoCollision` on `vigiles/claude-code`, because
|
|
64
|
+
* those surfaces are chosen by harness, not by cost.
|
|
65
|
+
*/
|
|
66
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
67
|
+
exports.paid_claudeEvalDriver = exports.paid_judged = exports.paid_judge = exports.paid_measureTriggerRate = exports.paid_measureArms = exports.paid_measure = exports.paid_runEval = void 0;
|
|
68
|
+
// --- the runners: each one drives a real model ---
|
|
69
|
+
/**
|
|
70
|
+
* Run an A/B eval across arms with a real model. Spends money.
|
|
71
|
+
*
|
|
72
|
+
* ⚠️ The `paid_` prefix overstates slightly: pass your own `evalDriver` and no
|
|
73
|
+
* model of ours is called. The DEFAULT path bills.
|
|
74
|
+
*/
|
|
75
|
+
var eval_js_1 = require("./eval.js");
|
|
76
|
+
Object.defineProperty(exports, "paid_runEval", { enumerable: true, get: function () { return eval_js_1.runEval; } });
|
|
77
|
+
/**
|
|
78
|
+
* Score checks over N trials of one task with a real model. Spends money.
|
|
79
|
+
*
|
|
80
|
+
* ⚠️ `paid_` overstates slightly: with an injected `evalDriver` this drives
|
|
81
|
+
* whatever you supply. The DEFAULT path bills.
|
|
82
|
+
*/
|
|
83
|
+
var eval_js_2 = require("./eval.js");
|
|
84
|
+
Object.defineProperty(exports, "paid_measure", { enumerable: true, get: function () { return eval_js_2.measure; } });
|
|
85
|
+
/**
|
|
86
|
+
* Score the same checks per arm (a hook/skill/rule on vs off) with a real model.
|
|
87
|
+
* Spends money.
|
|
88
|
+
*
|
|
89
|
+
* ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
|
|
90
|
+
* path. The DEFAULT path bills.
|
|
91
|
+
*/
|
|
92
|
+
var eval_js_3 = require("./eval.js");
|
|
93
|
+
Object.defineProperty(exports, "paid_measureArms", { enumerable: true, get: function () { return eval_js_3.measureArms; } });
|
|
94
|
+
/**
|
|
95
|
+
* Measure whether a skill's description actually FIRES (recall + precision) by
|
|
96
|
+
* running real prompts against a real model. Spends money.
|
|
97
|
+
*
|
|
98
|
+
* ⚠️ `paid_` overstates slightly: an injected `evalDriver` replaces the billed
|
|
99
|
+
* path, and `stubSkillBodies` makes the default path much cheaper without making
|
|
100
|
+
* it free. The DEFAULT path bills.
|
|
101
|
+
*/
|
|
102
|
+
var eval_js_4 = require("./eval.js");
|
|
103
|
+
Object.defineProperty(exports, "paid_measureTriggerRate", { enumerable: true, get: function () { return eval_js_4.measureTriggerRate; } });
|
|
104
|
+
/**
|
|
105
|
+
* The model-graded judge (harness-agnostic — grades text against a rubric).
|
|
106
|
+
* Shells out to the `claude` CLI synchronously; needs model auth. Spends money.
|
|
107
|
+
*
|
|
108
|
+
* ⚠️ Unlike the others this one has no injection seam — it always calls a model,
|
|
109
|
+
* so here `paid_` is exact.
|
|
110
|
+
*/
|
|
111
|
+
var judge_js_1 = require("./judge.js");
|
|
112
|
+
Object.defineProperty(exports, "paid_judge", { enumerable: true, get: function () { return judge_js_1.judge; } });
|
|
113
|
+
/**
|
|
114
|
+
* The one member of the declarative check vocabulary that bills: its default
|
|
115
|
+
* judge is a real model call. Every other `Check` is deterministic and lives
|
|
116
|
+
* unprefixed on the free root barrel.
|
|
117
|
+
*
|
|
118
|
+
* ⚠️ The `paid_` prefix overstates slightly, and this is the symbol it overstates
|
|
119
|
+
* most: `paid_judged(rubric, { judge: myFn })` calls YOUR function and spends
|
|
120
|
+
* nothing. Only `paid_judged(rubric)` — the default — bills.
|
|
121
|
+
*/
|
|
122
|
+
var check_js_1 = require("./check.js");
|
|
123
|
+
Object.defineProperty(exports, "paid_judged", { enumerable: true, get: function () { return check_js_1.judged; } });
|
|
124
|
+
/**
|
|
125
|
+
* The Claude-Code eval driver: the transport `paid_runEval` and friends use by
|
|
126
|
+
* default. Naming it here is how you pass it explicitly; using it spends money.
|
|
127
|
+
*/
|
|
128
|
+
var eval_js_5 = require("./eval.js");
|
|
129
|
+
Object.defineProperty(exports, "paid_claudeEvalDriver", { enumerable: true, get: function () { return eval_js_5.claudeEvalDriver; } });
|
|
130
|
+
//# sourceMappingURL=eval-surface.js.map
|
package/dist/harness-assert.d.ts
CHANGED
|
@@ -309,7 +309,7 @@ interface MatcherOutput {
|
|
|
309
309
|
* Custom matchers compatible with both vitest and jest. Register once:
|
|
310
310
|
*
|
|
311
311
|
* import { expect } from "vitest"; // or "@jest/globals"
|
|
312
|
-
* import { vigilesMatchers } from "vigiles
|
|
312
|
+
* import { vigilesMatchers } from "vigiles";
|
|
313
313
|
* expect.extend(vigilesMatchers);
|
|
314
314
|
*
|
|
315
315
|
* expect(result).toHaveCreated("RESULT");
|
package/dist/harness-assert.js
CHANGED
|
@@ -69,7 +69,7 @@ const agent_result_js_1 = require("./adapters/claude-code/agent-result.js");
|
|
|
69
69
|
const stats_js_1 = require("./stats.js");
|
|
70
70
|
const eval_baseline_js_1 = require("./eval-baseline.js");
|
|
71
71
|
// Re-export the significance primitives so the whole eval-analysis surface lives
|
|
72
|
-
// re-exported from `vigiles
|
|
72
|
+
// re-exported from the `vigiles` root (there is no `vigiles/harness-assert` entry
|
|
73
73
|
// point — it was advertised in this very comment and never existed).
|
|
74
74
|
var stats_js_2 = require("./stats.js");
|
|
75
75
|
Object.defineProperty(exports, "compareArms", { enumerable: true, get: function () { return stats_js_2.compareArms; } });
|
|
@@ -690,7 +690,7 @@ function assertTriggerRate(report, opts) {
|
|
|
690
690
|
* Custom matchers compatible with both vitest and jest. Register once:
|
|
691
691
|
*
|
|
692
692
|
* import { expect } from "vitest"; // or "@jest/globals"
|
|
693
|
-
* import { vigilesMatchers } from "vigiles
|
|
693
|
+
* import { vigilesMatchers } from "vigiles";
|
|
694
694
|
* expect.extend(vigilesMatchers);
|
|
695
695
|
*
|
|
696
696
|
* expect(result).toHaveCreated("RESULT");
|
package/dist/load-hook.d.ts
CHANGED
|
@@ -8,7 +8,7 @@ import { type AnyHook } from "./core/hook-program.js";
|
|
|
8
8
|
* points at authoring the hook as `.mjs`.
|
|
9
9
|
*
|
|
10
10
|
* ```js
|
|
11
|
-
* import { loadHook, assertHookDenies } from "vigiles
|
|
11
|
+
* import { loadHook, assertHookDenies } from "vigiles";
|
|
12
12
|
*
|
|
13
13
|
* const guard = await loadHook(".vigiles/hooks/guard.mjs");
|
|
14
14
|
* assertHookDenies(guard, {
|
package/dist/load-hook.js
CHANGED
|
@@ -27,7 +27,7 @@ const hook_program_js_1 = require("./core/hook-program.js");
|
|
|
27
27
|
* points at authoring the hook as `.mjs`.
|
|
28
28
|
*
|
|
29
29
|
* ```js
|
|
30
|
-
* import { loadHook, assertHookDenies } from "vigiles
|
|
30
|
+
* import { loadHook, assertHookDenies } from "vigiles";
|
|
31
31
|
*
|
|
32
32
|
* const guard = await loadHook(".vigiles/hooks/guard.mjs");
|
|
33
33
|
* assertHookDenies(guard, {
|
package/dist/run-hook.js
CHANGED
|
@@ -11,7 +11,7 @@ const hook_protocol_js_1 = require("./adapters/claude-code/hook-protocol.js");
|
|
|
11
11
|
const proofs_js_1 = require("./core/proofs.js");
|
|
12
12
|
const hook_program_js_1 = require("./core/hook-program.js");
|
|
13
13
|
const run_script_js_1 = require("./run-script.js");
|
|
14
|
-
// The general runner this module specializes. Re-exported so `vigiles
|
|
14
|
+
// The general runner this module specializes. Re-exported so the `vigiles` root's
|
|
15
15
|
// hook surface and the script surface stay one import away from each other.
|
|
16
16
|
var run_script_js_2 = require("./run-script.js");
|
|
17
17
|
Object.defineProperty(exports, "runScript", { enumerable: true, get: function () { return run_script_js_2.runScript; } });
|