vigiles 15.4.0 → 16.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapters/claude-code/run-scripts.d.ts +4 -4
- package/dist/adapters/claude-code/run-scripts.js +6 -6
- package/dist/adapters/claude-code/typed-spec.d.ts +26 -3
- package/dist/adapters/claude-code/typed-spec.js +21 -4
- package/dist/audit-report.template.html +1 -1
- package/dist/audit-score.d.ts +1 -1
- package/dist/audit-score.js +2 -2
- package/dist/check-count.js +1 -1
- package/dist/claude-code.d.ts +1 -1
- package/dist/claude-code.js +3 -3
- package/dist/cli.js +8 -8
- package/dist/codex.d.ts +1 -1
- package/dist/codex.js +1 -1
- package/dist/core/spec.d.ts +37 -5
- package/dist/core/spec.js +33 -7
- package/dist/eval-surface.d.ts +122 -0
- package/dist/eval-surface.js +130 -0
- package/dist/harness-assert.d.ts +1 -1
- package/dist/harness-assert.js +2 -2
- package/dist/load-hook.d.ts +1 -1
- package/dist/load-hook.js +1 -1
- package/dist/run-hook.js +1 -1
- package/dist/scaffold-test.js +7 -11
- package/dist/scan-trigger-suggest.d.ts +3 -3
- package/dist/scan-trigger-suggest.js +4 -4
- package/dist/test-coverage.js +1 -1
- package/dist/test.d.ts +75 -0
- package/dist/test.js +192 -0
- package/package.json +9 -8
- package/skills/test-harness/SKILL.md +16 -14
- package/dist/e2e.d.ts +0 -18
- package/dist/e2e.js +0 -34
- package/dist/integration.d.ts +0 -29
- package/dist/integration.js +0 -58
- package/dist/testing.d.ts +0 -41
- package/dist/testing.js +0 -123
- package/dist/unit.d.ts +0 -39
- package/dist/unit.js +0 -74
package/dist/scaffold-test.js
CHANGED
|
@@ -57,7 +57,7 @@ import {
|
|
|
57
57
|
verifyGuardrail,
|
|
58
58
|
formatGuardrailReport,
|
|
59
59
|
// assertBlocksDisasters, // uncomment to gate CI on the battery (see below)
|
|
60
|
-
} from "vigiles
|
|
60
|
+
} from "vigiles";
|
|
61
61
|
|
|
62
62
|
const cmd = ${JSON.stringify(cmd)};
|
|
63
63
|
|
|
@@ -89,19 +89,15 @@ function skillScaffold(input) {
|
|
|
89
89
|
? "\n// NOTE: this skill is user-invoked (disableModelInvocation). Trigger-rate\n// measures MODEL-invocable skills; either make it model-invocable or test its\n// slash-command invocation with runHarnessTest instead.\n"
|
|
90
90
|
: "";
|
|
91
91
|
return `${header(`Starter trigger-rate eval for the \`${id}\` skill (recall + precision).`, `npx vigiles eval ${suggestedPath(input)} # real model, on your subscription`)}
|
|
92
|
-
import {
|
|
93
|
-
|
|
94
|
-
formatTriggerRateReport,
|
|
95
|
-
assertTriggerRate,
|
|
96
|
-
skillResolved,
|
|
97
|
-
} from "vigiles/testing";
|
|
92
|
+
import { paid_measureTriggerRate } from "vigiles/eval"; // paid_ = a real model runs
|
|
93
|
+
import { formatTriggerRateReport, assertTriggerRate, skillResolved } from "vigiles";
|
|
98
94
|
import { fileURLToPath } from "node:url";
|
|
99
95
|
${note}
|
|
100
96
|
// TODO: point at the plugin root (the dir holding .claude-plugin/ or skills/).
|
|
101
97
|
const pluginDir = fileURLToPath(new URL("../../", import.meta.url));
|
|
102
98
|
const skill = ${JSON.stringify(id)};
|
|
103
99
|
|
|
104
|
-
const report = await
|
|
100
|
+
const report = await paid_measureTriggerRate({
|
|
105
101
|
pluginDir,
|
|
106
102
|
stubSkillBodies: true, // firing is a frontmatter property — stub bodies, pay less
|
|
107
103
|
prompts: [
|
|
@@ -152,7 +148,7 @@ function outcomeSection(input, contract) {
|
|
|
152
148
|
: "";
|
|
153
149
|
return `import assert from "node:assert/strict";
|
|
154
150
|
import { result } from "vigiles/spec";
|
|
155
|
-
import { assertAgentOk } from "vigiles
|
|
151
|
+
import { assertAgentOk } from "vigiles";
|
|
156
152
|
|
|
157
153
|
// Reconstructed from ${input.name}'s ## Output contract (its compiled .md) — the
|
|
158
154
|
// typed result() the spec wrote. assertAgentOk parses + validates the outcome
|
|
@@ -177,7 +173,7 @@ function fallbackSection(input) {
|
|
|
177
173
|
const toolHint = input.tools && input.tools.length > 0
|
|
178
174
|
? `assertToolUsed(r, ${JSON.stringify(input.tools[0])}); // its declared contract: ${input.tools.join(", ")}`
|
|
179
175
|
: `assertToolUsed(r, "Task"); // TODO: assert what the subagent should do`;
|
|
180
|
-
return `import { runHarnessTest, assertToolUsed } from "vigiles
|
|
176
|
+
return `import { runHarnessTest, assertToolUsed } from "vigiles";
|
|
181
177
|
|
|
182
178
|
// ${input.name} has no result() contract, so its outcome can't be asserted
|
|
183
179
|
// deterministically — add one (result() on its agent() spec) for a no-judge
|
|
@@ -239,7 +235,7 @@ ${checks.join("\n")}
|
|
|
239
235
|
function agentScaffold(input) {
|
|
240
236
|
const head = header(`Starter harness test for the \`${input.name}\` subagent.`, `npx vigiles test ${suggestedPath(input)}`);
|
|
241
237
|
const safetyImport = input.sideEffectingTools && input.sideEffectingTools.length > 0
|
|
242
|
-
? `import { notTool, didNotWrite, assertChecks } from "vigiles
|
|
238
|
+
? `import { notTool, didNotWrite, assertChecks } from "vigiles";\n`
|
|
243
239
|
: "";
|
|
244
240
|
const body = input.resultContract
|
|
245
241
|
? outcomeSection(input, input.resultContract)
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
|
|
8
8
|
* `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
|
|
9
9
|
* never silently executes. There is deliberately NO execution flag: automation
|
|
10
|
-
* tests the harness through the `vigiles
|
|
10
|
+
* tests the harness through the `vigiles` testing API + skills (the layered tiers),
|
|
11
11
|
* not through the report verb. The IO (prompt / run / remember) lives in the CLI;
|
|
12
12
|
* this is the pure decision + helpers.
|
|
13
13
|
*/
|
|
@@ -69,7 +69,7 @@ export interface ExecuteEnv {
|
|
|
69
69
|
* the executing checks need a human to consent:
|
|
70
70
|
* 1. nothing executable → skip "nothing" (a clean read; no nudge)
|
|
71
71
|
* 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
|
|
72
|
-
* skip "headless" (no one to ask; automation uses the `vigiles
|
|
72
|
+
* skip "headless" (no one to ask; automation uses the `vigiles` testing API)
|
|
73
73
|
* 3. sticky no → skip "remembered-no"
|
|
74
74
|
* 4. sticky yes → run
|
|
75
75
|
* 5. interactive human, no sticky choice → ask (then remember)
|
|
@@ -79,7 +79,7 @@ export declare function decideExecute(o: ExecuteEnv): ExecuteDecision;
|
|
|
79
79
|
* The one-line "executing checks not run" nudge for a skipped read (the
|
|
80
80
|
* no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
|
|
81
81
|
* gap). There is no flag to point at — `audit` runs them only interactively, and
|
|
82
|
-
* automation uses the `vigiles
|
|
82
|
+
* automation uses the `vigiles` testing API.
|
|
83
83
|
*/
|
|
84
84
|
export declare function formatExecuteSkip(reason: ExecuteSkipReason): string | null;
|
|
85
85
|
/**
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* remembers in `.vigilesrc.json` `audit.measure`); headless (an agent / `--json` /
|
|
9
9
|
* `--no-interactive` / a pipe) it stays a read + a one-line nudge — never hangs,
|
|
10
10
|
* never silently executes. There is deliberately NO execution flag: automation
|
|
11
|
-
* tests the harness through the `vigiles
|
|
11
|
+
* tests the harness through the `vigiles` testing API + skills (the layered tiers),
|
|
12
12
|
* not through the report verb. The IO (prompt / run / remember) lives in the CLI;
|
|
13
13
|
* this is the pure decision + helpers.
|
|
14
14
|
*/
|
|
@@ -45,7 +45,7 @@ function isMeteredAccess(env) {
|
|
|
45
45
|
* the executing checks need a human to consent:
|
|
46
46
|
* 1. nothing executable → skip "nothing" (a clean read; no nudge)
|
|
47
47
|
* 2. headless (`--json` / `--no-interactive` / non-TTY — an agent, a pipe, CI) →
|
|
48
|
-
* skip "headless" (no one to ask; automation uses the `vigiles
|
|
48
|
+
* skip "headless" (no one to ask; automation uses the `vigiles` testing API)
|
|
49
49
|
* 3. sticky no → skip "remembered-no"
|
|
50
50
|
* 4. sticky yes → run
|
|
51
51
|
* 5. interactive human, no sticky choice → ask (then remember)
|
|
@@ -65,7 +65,7 @@ function decideExecute(o) {
|
|
|
65
65
|
* The one-line "executing checks not run" nudge for a skipped read (the
|
|
66
66
|
* no-silent-skips corollary). Returns null for `nothing` (nothing to run — not a
|
|
67
67
|
* gap). There is no flag to point at — `audit` runs them only interactively, and
|
|
68
|
-
* automation uses the `vigiles
|
|
68
|
+
* automation uses the `vigiles` testing API.
|
|
69
69
|
*/
|
|
70
70
|
function formatExecuteSkip(reason) {
|
|
71
71
|
switch (reason) {
|
|
@@ -74,7 +74,7 @@ function formatExecuteSkip(reason) {
|
|
|
74
74
|
case "headless":
|
|
75
75
|
return ("\nℹ Executing checks (safety battery · live MCP · skill firing) skipped — " +
|
|
76
76
|
"`audit` runs them only interactively (a terminal). For automation, test the " +
|
|
77
|
-
"harness with the `vigiles
|
|
77
|
+
"harness with the `vigiles` testing API.");
|
|
78
78
|
case "remembered-no":
|
|
79
79
|
return ("\nℹ Executing checks not run (you disabled them — edit .vigilesrc.json " +
|
|
80
80
|
"`audit.measure` to re-enable).");
|
package/dist/test-coverage.js
CHANGED
|
@@ -595,7 +595,7 @@ function evalTierQuestion(kind) {
|
|
|
595
595
|
//
|
|
596
596
|
// ⚠️ It does NOT offer `{ evalDriver }` here, and that is checked rather
|
|
597
597
|
// than assumed: `runEval`/`measure`/`measureArms` all hard-wire
|
|
598
|
-
// `spawnAgent`, and `runEvalWith` is not exported from `vigiles/
|
|
598
|
+
// `spawnAgent`, and `runEvalWith` is not exported from `vigiles/eval` —
|
|
599
599
|
// `measureTriggerRate` is the only public call with that seam. Suggesting
|
|
600
600
|
// an argument the function does not take is this finding's own defect.
|
|
601
601
|
return (`A deterministic test covers it, but that test drove a SCRIPTED model. ` +
|
package/dist/test.d.ts
ADDED
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `vigiles` (the package root) — the **free** testing surface: everything you can
|
|
3
|
+
* run without a model call, and therefore without a bill.
|
|
4
|
+
*
|
|
5
|
+
* This is the front door. There is no `vigiles/test` subpath: the bare package
|
|
6
|
+
* name IS the testing surface, because a second name for the same door is the
|
|
7
|
+
* duplication this reorganisation removed (the old `.` + `./spec` pair pointed at
|
|
8
|
+
* one module under two names). A subpath is warranted when something genuinely
|
|
9
|
+
* DIFFERENT lives behind it — another runtime, another driver, another protocol.
|
|
10
|
+
* Here the one real difference is whether a call can spend money, so there are
|
|
11
|
+
* exactly two doors: this one, and [`vigiles/eval`](./eval-surface.ts).
|
|
12
|
+
*
|
|
13
|
+
* ## What replaced what
|
|
14
|
+
*
|
|
15
|
+
* The old split was by TEST TIER — `vigiles/unit`, `vigiles/integration`,
|
|
16
|
+
* `vigiles/e2e`, `vigiles/testing`. Tiers are a property of test-file layout and
|
|
17
|
+
* runner config, not of a module graph: this repo already expresses them as
|
|
18
|
+
* vitest projects (`test:unit`, `test:integration`, `test:e2e`), so the exports
|
|
19
|
+
* map was a second, competing implementation of the same idea. `src/e2e.ts` had
|
|
20
|
+
* degenerated to `export * from "./integration.js"` — a barrel adding zero
|
|
21
|
+
* symbols, which its own docstring called "the definition of a non-tier".
|
|
22
|
+
*
|
|
23
|
+
* ## Two honest costs of the new split
|
|
24
|
+
*
|
|
25
|
+
* **1. The type coupling crosses the boundary and cannot be removed.** About two
|
|
26
|
+
* dozen helpers here — `assertImproves`, `assertNoRegression`, `assertReliable`,
|
|
27
|
+
* `assertSignificant`, `assertTriggerRate`, `reliable`, `improvement`,
|
|
28
|
+
* `significantlyBeats`, `compareArms`, `diffReports`, `diffToJUnit`,
|
|
29
|
+
* `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`, `toBaselineFile`,
|
|
30
|
+
* `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`, `outputTokens`,
|
|
31
|
+
* `cacheTokens`, plus `formatEvalReport` / `formatCheckReport` /
|
|
32
|
+
* `formatTriggerRateReport` / `checkReportToJUnit` / `assertRates` — operate on
|
|
33
|
+
* eval RESULTS while spending nothing themselves. They belong on the free path.
|
|
34
|
+
* Their TYPES (`EvalReport`, `CheckReport`, `TriggerRateReport`, `Metrics`, …)
|
|
35
|
+
* are defined on the paid side. So those types are re-exported from BOTH barrels:
|
|
36
|
+
* a user of the free surface must never import `vigiles/eval` for a type alone.
|
|
37
|
+
* The duplication is deliberate and is the price of splitting on cost.
|
|
38
|
+
*
|
|
39
|
+
* **2. Free is not the same as fast.** The old "real CLI, no API key" tier
|
|
40
|
+
* (`vigiles/integration`) collapses into this barrel, so the import path no longer
|
|
41
|
+
* warns you that `runHarnessTest` spawns a real `claude` binary under bubblewrap
|
|
42
|
+
* and can take ~40 seconds. Nothing here bills you; plenty here is slow. Duration
|
|
43
|
+
* now lives where duration always lived in practice — in the runner config
|
|
44
|
+
* (`--project integration`), not in the import.
|
|
45
|
+
*
|
|
46
|
+
* **What was lost, stated plainly:** the symmetry with the CLI verbs is now
|
|
47
|
+
* one-sided. `vigiles/eval` rhymes with `vigiles eval`, but the free half answers
|
|
48
|
+
* to the plain package name rather than to `vigiles test`. That is accepted, not
|
|
49
|
+
* overlooked. `@playwright/test` — an end-to-end tool by definition — ships no
|
|
50
|
+
* subpath naming a test type at all, and keeps the whole fast/slow/expensive
|
|
51
|
+
* distinction in its config rather than in its imports.
|
|
52
|
+
*/
|
|
53
|
+
export { recordCheck } from "./check-count.js";
|
|
54
|
+
export { runScript } from "./run-script.js";
|
|
55
|
+
export type { RunScriptOptions, ScriptRunResult } from "./run-script.js";
|
|
56
|
+
export { runHook, parseHookOutput, decideHook, propertyHook, fileToolEvents, egressRoutes, } from "./run-hook.js";
|
|
57
|
+
export type { HookRunResult, RunHookOptions, HookInput, HookOutput, HookPropertyResult, FileToolEventOptions, } from "./run-hook.js";
|
|
58
|
+
export * from "./harness-assert.js";
|
|
59
|
+
export { loadHook } from "./load-hook.js";
|
|
60
|
+
export { evalChecks, assertChecks, tool, toolWith, notTool, onlyTools, skill, output, hookFired, received, turns, wrote, didNotWrite, subagent, blocked, allowed, mcp, cost, latency, tokens, inputTokens, outputTokens, cacheTokens, } from "./check.js";
|
|
61
|
+
export type { ArgMatcher, Check, CheckJSON, CheckResult, JudgeFn, } from "./check.js";
|
|
62
|
+
export { DISASTER_CATALOG, verifyGuardrail, unblockedDisasters, assertBlocksDisasters, formatGuardrailReport, } from "./guardrail-check.js";
|
|
63
|
+
export type { DisasterEvent, DisasterCategory, GuardrailResult, VerifyGuardrailOptions, } from "./guardrail-check.js";
|
|
64
|
+
export * from "./tool-stub.js";
|
|
65
|
+
export { runHarnessTest, runHarness, parseToolCalls, parseSubagents, parseResultEvent, parseOutput, parseHooks, decideSandbox, specTrusted, sandboxAvailable, } from "./harness-test.js";
|
|
66
|
+
export type { HarnessTestSpec, Trace, SubagentTrace, HarnessTestResult, RunHarnessTestOptions, ModelTurn, ModelRequest, ToolCall, HookFire, HarnessTestDriver, SandboxMode, } from "./harness-test.js";
|
|
67
|
+
export { commandsIn, mustInclude, mustNotInclude } from "./doc-commands.js";
|
|
68
|
+
export type { DocCommand } from "./doc-commands.js";
|
|
69
|
+
export { skillContract } from "./skill-contract.js";
|
|
70
|
+
export type { SkillContract, SkillContractOptions } from "./skill-contract.js";
|
|
71
|
+
export { compareContainment, formatContainment, } from "./trigger-containment.js";
|
|
72
|
+
export type { ContainmentInput, ContainmentVerdict, } from "./trigger-containment.js";
|
|
73
|
+
export { assertRates, assertPromptDiversity, checkPromptDiversity, checkReportToJUnit, formatCheckReport, formatEvalReport, formatTriggerRateReport, parseClaudeRun, stubSkillBody, } from "./eval.js";
|
|
74
|
+
export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
|
|
75
|
+
//# sourceMappingURL=test.d.ts.map
|
package/dist/test.js
ADDED
|
@@ -0,0 +1,192 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* `vigiles` (the package root) — the **free** testing surface: everything you can
|
|
4
|
+
* run without a model call, and therefore without a bill.
|
|
5
|
+
*
|
|
6
|
+
* This is the front door. There is no `vigiles/test` subpath: the bare package
|
|
7
|
+
* name IS the testing surface, because a second name for the same door is the
|
|
8
|
+
* duplication this reorganisation removed (the old `.` + `./spec` pair pointed at
|
|
9
|
+
* one module under two names). A subpath is warranted when something genuinely
|
|
10
|
+
* DIFFERENT lives behind it — another runtime, another driver, another protocol.
|
|
11
|
+
* Here the one real difference is whether a call can spend money, so there are
|
|
12
|
+
* exactly two doors: this one, and [`vigiles/eval`](./eval-surface.ts).
|
|
13
|
+
*
|
|
14
|
+
* ## What replaced what
|
|
15
|
+
*
|
|
16
|
+
* The old split was by TEST TIER — `vigiles/unit`, `vigiles/integration`,
|
|
17
|
+
* `vigiles/e2e`, `vigiles/testing`. Tiers are a property of test-file layout and
|
|
18
|
+
* runner config, not of a module graph: this repo already expresses them as
|
|
19
|
+
* vitest projects (`test:unit`, `test:integration`, `test:e2e`), so the exports
|
|
20
|
+
* map was a second, competing implementation of the same idea. `src/e2e.ts` had
|
|
21
|
+
* degenerated to `export * from "./integration.js"` — a barrel adding zero
|
|
22
|
+
* symbols, which its own docstring called "the definition of a non-tier".
|
|
23
|
+
*
|
|
24
|
+
* ## Two honest costs of the new split
|
|
25
|
+
*
|
|
26
|
+
* **1. The type coupling crosses the boundary and cannot be removed.** About two
|
|
27
|
+
* dozen helpers here — `assertImproves`, `assertNoRegression`, `assertReliable`,
|
|
28
|
+
* `assertSignificant`, `assertTriggerRate`, `reliable`, `improvement`,
|
|
29
|
+
* `significantlyBeats`, `compareArms`, `diffReports`, `diffToJUnit`,
|
|
30
|
+
* `formatBaselineDiff`, `parseBaselineFile`, `readBaseline`, `toBaselineFile`,
|
|
31
|
+
* `writeBaseline`, `cost`, `latency`, `tokens`, `inputTokens`, `outputTokens`,
|
|
32
|
+
* `cacheTokens`, plus `formatEvalReport` / `formatCheckReport` /
|
|
33
|
+
* `formatTriggerRateReport` / `checkReportToJUnit` / `assertRates` — operate on
|
|
34
|
+
* eval RESULTS while spending nothing themselves. They belong on the free path.
|
|
35
|
+
* Their TYPES (`EvalReport`, `CheckReport`, `TriggerRateReport`, `Metrics`, …)
|
|
36
|
+
* are defined on the paid side. So those types are re-exported from BOTH barrels:
|
|
37
|
+
* a user of the free surface must never import `vigiles/eval` for a type alone.
|
|
38
|
+
* The duplication is deliberate and is the price of splitting on cost.
|
|
39
|
+
*
|
|
40
|
+
* **2. Free is not the same as fast.** The old "real CLI, no API key" tier
|
|
41
|
+
* (`vigiles/integration`) collapses into this barrel, so the import path no longer
|
|
42
|
+
* warns you that `runHarnessTest` spawns a real `claude` binary under bubblewrap
|
|
43
|
+
* and can take ~40 seconds. Nothing here bills you; plenty here is slow. Duration
|
|
44
|
+
* now lives where duration always lived in practice — in the runner config
|
|
45
|
+
* (`--project integration`), not in the import.
|
|
46
|
+
*
|
|
47
|
+
* **What was lost, stated plainly:** the symmetry with the CLI verbs is now
|
|
48
|
+
* one-sided. `vigiles/eval` rhymes with `vigiles eval`, but the free half answers
|
|
49
|
+
* to the plain package name rather than to `vigiles test`. That is accepted, not
|
|
50
|
+
* overlooked. `@playwright/test` — an end-to-end tool by definition — ships no
|
|
51
|
+
* subpath naming a test type at all, and keeps the whole fast/slow/expensive
|
|
52
|
+
* distinction in its config rather than in its imports.
|
|
53
|
+
*/
|
|
54
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
55
|
+
if (k2 === undefined) k2 = k;
|
|
56
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
57
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
58
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
59
|
+
}
|
|
60
|
+
Object.defineProperty(o, k2, desc);
|
|
61
|
+
}) : (function(o, m, k, k2) {
|
|
62
|
+
if (k2 === undefined) k2 = k;
|
|
63
|
+
o[k2] = m[k];
|
|
64
|
+
}));
|
|
65
|
+
var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
66
|
+
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
67
|
+
};
|
|
68
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
69
|
+
exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.loadHook = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
|
|
70
|
+
exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
|
|
71
|
+
// --- reporting: how much did this script actually do? ---
|
|
72
|
+
// `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
|
|
73
|
+
// prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
|
|
74
|
+
// exporting an object of tests nobody calls). Call `recordCheck()` yourself when
|
|
75
|
+
// you assert some OTHER way — `node:assert`, vitest's `expect` — so those are
|
|
76
|
+
// visible to the runner too. See check-count.ts.
|
|
77
|
+
var check_count_js_1 = require("./check-count.js");
|
|
78
|
+
Object.defineProperty(exports, "recordCheck", { enumerable: true, get: function () { return check_count_js_1.recordCheck; } });
|
|
79
|
+
// --- the process primitives: runScript (any program) + runHook (plus a decision) ---
|
|
80
|
+
// `runScript` runs any program and reports what it DID (exit, both streams,
|
|
81
|
+
// writes, egress). `runHook` is that plus the hook protocol: event to stdin,
|
|
82
|
+
// exit code to allow/deny. Testing a plain helper script? Reach for runScript —
|
|
83
|
+
// its result carries no `decision`, because a script does not have one.
|
|
84
|
+
var run_script_js_1 = require("./run-script.js");
|
|
85
|
+
Object.defineProperty(exports, "runScript", { enumerable: true, get: function () { return run_script_js_1.runScript; } });
|
|
86
|
+
var run_hook_js_1 = require("./run-hook.js");
|
|
87
|
+
Object.defineProperty(exports, "runHook", { enumerable: true, get: function () { return run_hook_js_1.runHook; } });
|
|
88
|
+
Object.defineProperty(exports, "parseHookOutput", { enumerable: true, get: function () { return run_hook_js_1.parseHookOutput; } });
|
|
89
|
+
Object.defineProperty(exports, "decideHook", { enumerable: true, get: function () { return run_hook_js_1.decideHook; } });
|
|
90
|
+
Object.defineProperty(exports, "propertyHook", { enumerable: true, get: function () { return run_hook_js_1.propertyHook; } });
|
|
91
|
+
Object.defineProperty(exports, "fileToolEvents", { enumerable: true, get: function () { return run_hook_js_1.fileToolEvents; } });
|
|
92
|
+
Object.defineProperty(exports, "egressRoutes", { enumerable: true, get: function () { return run_hook_js_1.egressRoutes; } });
|
|
93
|
+
// --- assertions and the eval-RESULT analysis helpers (free; see cost #1 above) ---
|
|
94
|
+
__exportStar(require("./harness-assert.js"), exports);
|
|
95
|
+
// The compiled-hook LOADER. The in-process assertions above take the hook
|
|
96
|
+
// OBJECT, but a `.harness.mjs` test only has its PATH — without this the
|
|
97
|
+
// intended in-process test path is unreachable from the file format
|
|
98
|
+
// `vigiles test` actually runs, and authors fall back to spawning the runtime as
|
|
99
|
+
// a subprocess (the very plumbing compiled hooks exist to remove). Same loader
|
|
100
|
+
// the CLI runtime uses, so a hook that loads in a test loads identically in prod.
|
|
101
|
+
var load_hook_js_1 = require("./load-hook.js");
|
|
102
|
+
Object.defineProperty(exports, "loadHook", { enumerable: true, get: function () { return load_hook_js_1.loadHook; } });
|
|
103
|
+
// --- the declarative check vocabulary ---
|
|
104
|
+
// Enumerated rather than `export *` ON PURPOSE: `judged` also lives in check.ts
|
|
105
|
+
// and its default judge is a real model call, so it is the one member of this
|
|
106
|
+
// module that belongs on the paid barrel (as `paid_judged`). Listing the members
|
|
107
|
+
// makes "a free import that can bill you" unrepresentable here rather than merely
|
|
108
|
+
// documented — the previous barrel carried a paragraph apologising for exactly
|
|
109
|
+
// that. `hookFired` is re-exported explicitly so this `Check<Trace>` wins over the
|
|
110
|
+
// legacy boolean predicate of the same name from harness-assert.
|
|
111
|
+
var check_js_1 = require("./check.js");
|
|
112
|
+
Object.defineProperty(exports, "evalChecks", { enumerable: true, get: function () { return check_js_1.evalChecks; } });
|
|
113
|
+
Object.defineProperty(exports, "assertChecks", { enumerable: true, get: function () { return check_js_1.assertChecks; } });
|
|
114
|
+
Object.defineProperty(exports, "tool", { enumerable: true, get: function () { return check_js_1.tool; } });
|
|
115
|
+
Object.defineProperty(exports, "toolWith", { enumerable: true, get: function () { return check_js_1.toolWith; } });
|
|
116
|
+
Object.defineProperty(exports, "notTool", { enumerable: true, get: function () { return check_js_1.notTool; } });
|
|
117
|
+
Object.defineProperty(exports, "onlyTools", { enumerable: true, get: function () { return check_js_1.onlyTools; } });
|
|
118
|
+
Object.defineProperty(exports, "skill", { enumerable: true, get: function () { return check_js_1.skill; } });
|
|
119
|
+
Object.defineProperty(exports, "output", { enumerable: true, get: function () { return check_js_1.output; } });
|
|
120
|
+
Object.defineProperty(exports, "hookFired", { enumerable: true, get: function () { return check_js_1.hookFired; } });
|
|
121
|
+
Object.defineProperty(exports, "received", { enumerable: true, get: function () { return check_js_1.received; } });
|
|
122
|
+
Object.defineProperty(exports, "turns", { enumerable: true, get: function () { return check_js_1.turns; } });
|
|
123
|
+
Object.defineProperty(exports, "wrote", { enumerable: true, get: function () { return check_js_1.wrote; } });
|
|
124
|
+
Object.defineProperty(exports, "didNotWrite", { enumerable: true, get: function () { return check_js_1.didNotWrite; } });
|
|
125
|
+
Object.defineProperty(exports, "subagent", { enumerable: true, get: function () { return check_js_1.subagent; } });
|
|
126
|
+
Object.defineProperty(exports, "blocked", { enumerable: true, get: function () { return check_js_1.blocked; } });
|
|
127
|
+
Object.defineProperty(exports, "allowed", { enumerable: true, get: function () { return check_js_1.allowed; } });
|
|
128
|
+
Object.defineProperty(exports, "mcp", { enumerable: true, get: function () { return check_js_1.mcp; } });
|
|
129
|
+
Object.defineProperty(exports, "cost", { enumerable: true, get: function () { return check_js_1.cost; } });
|
|
130
|
+
Object.defineProperty(exports, "latency", { enumerable: true, get: function () { return check_js_1.latency; } });
|
|
131
|
+
Object.defineProperty(exports, "tokens", { enumerable: true, get: function () { return check_js_1.tokens; } });
|
|
132
|
+
Object.defineProperty(exports, "inputTokens", { enumerable: true, get: function () { return check_js_1.inputTokens; } });
|
|
133
|
+
Object.defineProperty(exports, "outputTokens", { enumerable: true, get: function () { return check_js_1.outputTokens; } });
|
|
134
|
+
Object.defineProperty(exports, "cacheTokens", { enumerable: true, get: function () { return check_js_1.cacheTokens; } });
|
|
135
|
+
// Guardrail verification — "prove your safety hook actually blocks" (over runHook).
|
|
136
|
+
var guardrail_check_js_1 = require("./guardrail-check.js");
|
|
137
|
+
Object.defineProperty(exports, "DISASTER_CATALOG", { enumerable: true, get: function () { return guardrail_check_js_1.DISASTER_CATALOG; } });
|
|
138
|
+
Object.defineProperty(exports, "verifyGuardrail", { enumerable: true, get: function () { return guardrail_check_js_1.verifyGuardrail; } });
|
|
139
|
+
Object.defineProperty(exports, "unblockedDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.unblockedDisasters; } });
|
|
140
|
+
Object.defineProperty(exports, "assertBlocksDisasters", { enumerable: true, get: function () { return guardrail_check_js_1.assertBlocksDisasters; } });
|
|
141
|
+
Object.defineProperty(exports, "formatGuardrailReport", { enumerable: true, get: function () { return guardrail_check_js_1.formatGuardrailReport; } });
|
|
142
|
+
// Tool stubs on PATH (rung R2): shadow a CLI tool with a recorded canned result.
|
|
143
|
+
__exportStar(require("./tool-stub.js"), exports);
|
|
144
|
+
// The assembled machine — AGNOSTIC SURFACE ONLY. The Claude-Code transport
|
|
145
|
+
// (`scriptModel`, `claudeCodeDriver`, `buildClaudeArgs`, `claudeAvailable`,
|
|
146
|
+
// `loadPlugin`, `resolveHarness`) is deliberately NOT re-exported here, so this
|
|
147
|
+
// surface stays harness-agnostic — import those from `vigiles/claude-code` (or
|
|
148
|
+
// your harness's package). See `research/code-adapter-architecture.md`.
|
|
149
|
+
// Slow but free: see cost #2 in the module doc.
|
|
150
|
+
var harness_test_js_1 = require("./harness-test.js");
|
|
151
|
+
Object.defineProperty(exports, "runHarnessTest", { enumerable: true, get: function () { return harness_test_js_1.runHarnessTest; } });
|
|
152
|
+
Object.defineProperty(exports, "runHarness", { enumerable: true, get: function () { return harness_test_js_1.runHarness; } });
|
|
153
|
+
Object.defineProperty(exports, "parseToolCalls", { enumerable: true, get: function () { return harness_test_js_1.parseToolCalls; } });
|
|
154
|
+
Object.defineProperty(exports, "parseSubagents", { enumerable: true, get: function () { return harness_test_js_1.parseSubagents; } });
|
|
155
|
+
Object.defineProperty(exports, "parseResultEvent", { enumerable: true, get: function () { return harness_test_js_1.parseResultEvent; } });
|
|
156
|
+
Object.defineProperty(exports, "parseOutput", { enumerable: true, get: function () { return harness_test_js_1.parseOutput; } });
|
|
157
|
+
Object.defineProperty(exports, "parseHooks", { enumerable: true, get: function () { return harness_test_js_1.parseHooks; } });
|
|
158
|
+
Object.defineProperty(exports, "decideSandbox", { enumerable: true, get: function () { return harness_test_js_1.decideSandbox; } });
|
|
159
|
+
Object.defineProperty(exports, "specTrusted", { enumerable: true, get: function () { return harness_test_js_1.specTrusted; } });
|
|
160
|
+
Object.defineProperty(exports, "sandboxAvailable", { enumerable: true, get: function () { return harness_test_js_1.sandboxAvailable; } });
|
|
161
|
+
// A document's own rule about its own COMMANDS, made checkable. vigiles cannot
|
|
162
|
+
// infer "always pass -g" from prose; what was missing was a cheap way to declare
|
|
163
|
+
// it — measured at ~95 lines of markdown parsing around ~5 lines of rule, which
|
|
164
|
+
// is why nobody wrote the check and the rule stayed prose.
|
|
165
|
+
var doc_commands_js_1 = require("./doc-commands.js");
|
|
166
|
+
Object.defineProperty(exports, "commandsIn", { enumerable: true, get: function () { return doc_commands_js_1.commandsIn; } });
|
|
167
|
+
Object.defineProperty(exports, "mustInclude", { enumerable: true, get: function () { return doc_commands_js_1.mustInclude; } });
|
|
168
|
+
Object.defineProperty(exports, "mustNotInclude", { enumerable: true, get: function () { return doc_commands_js_1.mustNotInclude; } });
|
|
169
|
+
// A skill's own `allowed-tools:` frontmatter, read back as checks — the wiring
|
|
170
|
+
// from a declaration to the existing check vocabulary, not new machinery.
|
|
171
|
+
var skill_contract_js_1 = require("./skill-contract.js");
|
|
172
|
+
Object.defineProperty(exports, "skillContract", { enumerable: true, get: function () { return skill_contract_js_1.skillContract; } });
|
|
173
|
+
// Is a WEAKER model a valid lower bound for trigger-rate? Pure comparison of two
|
|
174
|
+
// runs; the open question the model floor makes people ask.
|
|
175
|
+
var trigger_containment_js_1 = require("./trigger-containment.js");
|
|
176
|
+
Object.defineProperty(exports, "compareContainment", { enumerable: true, get: function () { return trigger_containment_js_1.compareContainment; } });
|
|
177
|
+
Object.defineProperty(exports, "formatContainment", { enumerable: true, get: function () { return trigger_containment_js_1.formatContainment; } });
|
|
178
|
+
// --- free analysis OVER eval results (the coupling from cost #1) ---
|
|
179
|
+
// These read a report that `vigiles/eval` produced. They spend nothing, so they
|
|
180
|
+
// live here and carry no `paid_` prefix; their argument types are defined over
|
|
181
|
+
// there, so those types are re-exported below from this barrel too.
|
|
182
|
+
var eval_js_1 = require("./eval.js");
|
|
183
|
+
Object.defineProperty(exports, "assertRates", { enumerable: true, get: function () { return eval_js_1.assertRates; } });
|
|
184
|
+
Object.defineProperty(exports, "assertPromptDiversity", { enumerable: true, get: function () { return eval_js_1.assertPromptDiversity; } });
|
|
185
|
+
Object.defineProperty(exports, "checkPromptDiversity", { enumerable: true, get: function () { return eval_js_1.checkPromptDiversity; } });
|
|
186
|
+
Object.defineProperty(exports, "checkReportToJUnit", { enumerable: true, get: function () { return eval_js_1.checkReportToJUnit; } });
|
|
187
|
+
Object.defineProperty(exports, "formatCheckReport", { enumerable: true, get: function () { return eval_js_1.formatCheckReport; } });
|
|
188
|
+
Object.defineProperty(exports, "formatEvalReport", { enumerable: true, get: function () { return eval_js_1.formatEvalReport; } });
|
|
189
|
+
Object.defineProperty(exports, "formatTriggerRateReport", { enumerable: true, get: function () { return eval_js_1.formatTriggerRateReport; } });
|
|
190
|
+
Object.defineProperty(exports, "parseClaudeRun", { enumerable: true, get: function () { return eval_js_1.parseClaudeRun; } });
|
|
191
|
+
Object.defineProperty(exports, "stubSkillBody", { enumerable: true, get: function () { return eval_js_1.stubSkillBody; } });
|
|
192
|
+
//# sourceMappingURL=test.js.map
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "16.0.0",
|
|
4
4
|
"description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|
|
@@ -35,17 +35,14 @@
|
|
|
35
35
|
"bin": {
|
|
36
36
|
"vigiles": "dist/cli.js"
|
|
37
37
|
},
|
|
38
|
-
"main": "./dist/
|
|
39
|
-
"types": "./dist/
|
|
38
|
+
"main": "./dist/test.js",
|
|
39
|
+
"types": "./dist/test.d.ts",
|
|
40
40
|
"exports": {
|
|
41
|
-
".": "./dist/
|
|
41
|
+
".": "./dist/test.js",
|
|
42
42
|
"./spec": "./dist/core/spec.js",
|
|
43
|
+
"./eval": "./dist/eval-surface.js",
|
|
43
44
|
"./linting": "./dist/linting.js",
|
|
44
|
-
"./testing": "./dist/testing.js",
|
|
45
|
-
"./unit": "./dist/unit.js",
|
|
46
45
|
"./hook": "./dist/hook.js",
|
|
47
|
-
"./integration": "./dist/integration.js",
|
|
48
|
-
"./e2e": "./dist/e2e.js",
|
|
49
46
|
"./claude-code": "./dist/claude-code.js",
|
|
50
47
|
"./codex": "./dist/codex.js",
|
|
51
48
|
"./adapter": "./dist/adapter.js",
|
|
@@ -92,7 +89,9 @@
|
|
|
92
89
|
"test:types": "npm run build && tsc --noEmit -p test/types/tsconfig.json",
|
|
93
90
|
"api:report": "npm run build && node scripts/api-extractor.mjs --local",
|
|
94
91
|
"api:check": "npm run build && node scripts/api-extractor.mjs",
|
|
92
|
+
"check": "node scripts/check.mjs",
|
|
95
93
|
"docs:check": "npm run build && node scripts/check-doc-imports.mjs . docs README.md",
|
|
94
|
+
"exports:check": "npm run build && node scripts/check-export-prefixes.mjs .",
|
|
96
95
|
"experimental:check": "npm run api:check && node scripts/check-experimental-naming.mjs",
|
|
97
96
|
"docs:api": "typedoc"
|
|
98
97
|
},
|
|
@@ -100,6 +99,7 @@
|
|
|
100
99
|
"@eslint/js": "^10.0.1",
|
|
101
100
|
"@jackchuka/mdschema": "^0.12.8",
|
|
102
101
|
"@microsoft/api-extractor": "^7.58.9",
|
|
102
|
+
"@semantic-release/commit-analyzer": "^13.0.1",
|
|
103
103
|
"@types/js-yaml": "^4.0.9",
|
|
104
104
|
"@types/markdown-it": "^14.1.2",
|
|
105
105
|
"@types/minimatch": "^5.1.2",
|
|
@@ -107,6 +107,7 @@
|
|
|
107
107
|
"@typescript-eslint/eslint-plugin": "^8.58.0",
|
|
108
108
|
"@typescript-eslint/parser": "^8.58.0",
|
|
109
109
|
"@vitest/coverage-v8": "^4.1.8",
|
|
110
|
+
"conventional-changelog-conventionalcommits": "^10.3.0",
|
|
110
111
|
"eslint": "^10.1.0",
|
|
111
112
|
"eslint-import-resolver-typescript": "^4.4.5",
|
|
112
113
|
"eslint-plugin-boundaries": "^6.0.2",
|
|
@@ -58,7 +58,7 @@ anything; every one of them ships today.
|
|
|
58
58
|
|
|
59
59
|
| The question you're actually asking | Use |
|
|
60
60
|
| ----------------------------------------------------------------------- | -------------------------------------------------------------------------------------- |
|
|
61
|
-
| Which tools did it call, and with what arguments? | `trace.toolCalls` · `tool` / `toolWith` checks · `parseToolCalls` (`vigiles
|
|
61
|
+
| Which tools did it call, and with what arguments? | `trace.toolCalls` · `tool` / `toolWith` checks · `parseToolCalls` (`vigiles`) |
|
|
62
62
|
| Did it call a tool it must not? | `notTool(name)` |
|
|
63
63
|
| Did it call **only** tools from a known set? | `onlyTools([...])` — the white-list, symmetric to `assertWroteOnly` |
|
|
64
64
|
| Did it stay inside the `allowed-tools` its own frontmatter declares? | `skillContract(dir).surface` — builds that check FROM the declaration |
|
|
@@ -80,7 +80,7 @@ the `allowed-tools:` the skill already claims and hands back ready checks, so
|
|
|
80
80
|
the claim is verified instead of restated:
|
|
81
81
|
|
|
82
82
|
```ts
|
|
83
|
-
import { skillContract, assertChecks } from "vigiles
|
|
83
|
+
import { skillContract, assertChecks } from "vigiles";
|
|
84
84
|
|
|
85
85
|
const c = skillContract(".claude/skills/my-skill");
|
|
86
86
|
assertChecks(trace, [c.activation, ...c.surface]);
|
|
@@ -157,7 +157,7 @@ Pick one concrete thing to pin down — a specific `PreToolUse` hook, a specific
|
|
|
157
157
|
**Unit (`runHook`)** — hand a hook a synthesized event, assert the decision:
|
|
158
158
|
|
|
159
159
|
```ts
|
|
160
|
-
import { runHook, assertHookBlocked } from "vigiles
|
|
160
|
+
import { runHook, assertHookBlocked } from "vigiles";
|
|
161
161
|
|
|
162
162
|
const r = runHook(hookCommand, {
|
|
163
163
|
hook_event_name: "PreToolUse",
|
|
@@ -188,9 +188,9 @@ import {
|
|
|
188
188
|
runHarnessTest,
|
|
189
189
|
assertHookFired,
|
|
190
190
|
assertRequestContains,
|
|
191
|
-
} from "vigiles
|
|
191
|
+
} from "vigiles";
|
|
192
192
|
// `scriptModel` is the Claude-Code TRANSPORT, deliberately not re-exported from
|
|
193
|
-
// the harness-agnostic
|
|
193
|
+
// the harness-agnostic root surface — import it from the harness package:
|
|
194
194
|
import { scriptModel } from "vigiles/claude-code";
|
|
195
195
|
|
|
196
196
|
const r = await runHarnessTest({
|
|
@@ -202,34 +202,36 @@ assertHookFired(r, "SessionStart");
|
|
|
202
202
|
assertRequestContains(r, "expected injected text"); // did it actually land?
|
|
203
203
|
```
|
|
204
204
|
|
|
205
|
-
**Eval — absolute (`
|
|
205
|
+
**Eval — absolute (`paid_measure` + `paid_judged`)** — testing _one_ skill, the usual case:
|
|
206
206
|
score its output directly against a rubric. No on/off baseline — this is the
|
|
207
207
|
"is it any good?" oracle (what promptfoo/DeepEval lead with), and the right
|
|
208
208
|
default when there's nothing to compare against:
|
|
209
209
|
|
|
210
210
|
```ts
|
|
211
|
-
import {
|
|
211
|
+
import { paid_measure, paid_judged } from "vigiles/eval"; // `paid_` = these call a model
|
|
212
|
+
import { skill, assertRates } from "vigiles";
|
|
212
213
|
|
|
213
|
-
const report = await
|
|
214
|
+
const report = await paid_measure({
|
|
214
215
|
pluginDir: "./",
|
|
215
216
|
task: "…a task the skill should handle…",
|
|
216
217
|
checks: [
|
|
217
218
|
skill("my-plugin:my-skill"), // it fired
|
|
218
|
-
|
|
219
|
+
paid_judged("the answer correctly does X and avoids Y"), // …and the output is good
|
|
219
220
|
],
|
|
220
221
|
trials: 6,
|
|
221
222
|
});
|
|
222
223
|
assertRates(report, { min: 0.8 }); // each check passes ≥ 80% of trials
|
|
223
224
|
```
|
|
224
225
|
|
|
225
|
-
**Eval — relative (`
|
|
226
|
+
**Eval — relative (`paid_runEval` + `assertSignificant`)** — when the question is
|
|
226
227
|
_lift over no-skill_ (regression, or proving a change isn't noise): A/B the
|
|
227
228
|
change on vs off and gate on significance, not eyeballing:
|
|
228
229
|
|
|
229
230
|
```ts
|
|
230
|
-
import {
|
|
231
|
+
import { paid_runEval } from "vigiles/eval"; // `paid_` = a real model runs
|
|
232
|
+
import { assertSignificant } from "vigiles";
|
|
231
233
|
|
|
232
|
-
const report = await
|
|
234
|
+
const report = await paid_runEval({
|
|
233
235
|
arms: { off: {}, on: { pluginDir: "./" } },
|
|
234
236
|
task: "…a task the harness change should affect…",
|
|
235
237
|
measure: (ctx) => ({ ok: /* a bare predicate over the trace */ true }),
|
|
@@ -261,7 +263,7 @@ unrepresentable:
|
|
|
261
263
|
Use **`runScript`** — it runs any command and reports what it did:
|
|
262
264
|
|
|
263
265
|
```ts
|
|
264
|
-
import { runScript } from "vigiles
|
|
266
|
+
import { runScript } from "vigiles";
|
|
265
267
|
|
|
266
268
|
const r = runScript("bash scripts/check-links.sh", { cwd: repoDir });
|
|
267
269
|
assert.equal(r.exitCode, 0);
|
|
@@ -293,7 +295,7 @@ npx vigiles eval --trials=6 # *.eval.{mjs,ts} — real model (local / night
|
|
|
293
295
|
Unit-tier `runHook` tests need no `claude` and **always run** — write and run them
|
|
294
296
|
even with no `claude` installed. A tier that genuinely can't run reports a loud
|
|
295
297
|
`⊘ SKIPPED` (tallied separately, never a fake `✓`); a standalone script emits one
|
|
296
|
-
via `skip(reason)` from `vigiles
|
|
298
|
+
via `skip(reason)` from `vigiles`. A skip passes by default, but in a CI
|
|
297
299
|
job that asserts the capability is present, run **`vigiles test --no-skip`** so a
|
|
298
300
|
skipped tier fails — a green-with-skips is untested surface. Keep unit +
|
|
299
301
|
deterministic tests in CI (free); run evals locally or on a schedule with auth.
|
package/dist/e2e.d.ts
DELETED
|
@@ -1,18 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* `vigiles/e2e` — DEPRECATED back-compat alias for [`vigiles/integration`](./integration.ts).
|
|
3
|
-
*
|
|
4
|
-
* There is no separate "e2e" tier: real **egress** is a *capability* of the
|
|
5
|
-
* harness/integration scope (`egressRoutes()` + `runHook`'s `egress: { allow }`),
|
|
6
|
-
* not a different kind of test — the old `e2e` barrel added exactly one symbol
|
|
7
|
-
* over `integration`, which is the definition of a non-tier. It now lives on
|
|
8
|
-
* `vigiles/integration`; this entry re-exports it unchanged so existing imports
|
|
9
|
-
* keep working. See `research/testing-api-design.md` Part 4 (two scopes, not
|
|
10
|
-
* four tiers). Prefer `vigiles/integration`.
|
|
11
|
-
*
|
|
12
|
-
* NOT here: **evals** (`runEval` / `measure` / `measureTriggerRate` / `judge`) —
|
|
13
|
-
* those are non-deterministic measurement, a different axis. (They live on
|
|
14
|
-
* `vigiles/testing`; `vigiles/eval` is named here in the original comment and does
|
|
15
|
-
* not exist as an entry point.)
|
|
16
|
-
*/
|
|
17
|
-
export * from "./integration.js";
|
|
18
|
-
//# sourceMappingURL=e2e.d.ts.map
|
package/dist/e2e.js
DELETED
|
@@ -1,34 +0,0 @@
|
|
|
1
|
-
"use strict";
|
|
2
|
-
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
-
if (k2 === undefined) k2 = k;
|
|
4
|
-
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
-
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
-
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
-
}
|
|
8
|
-
Object.defineProperty(o, k2, desc);
|
|
9
|
-
}) : (function(o, m, k, k2) {
|
|
10
|
-
if (k2 === undefined) k2 = k;
|
|
11
|
-
o[k2] = m[k];
|
|
12
|
-
}));
|
|
13
|
-
var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
14
|
-
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
15
|
-
};
|
|
16
|
-
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
-
/**
|
|
18
|
-
* `vigiles/e2e` — DEPRECATED back-compat alias for [`vigiles/integration`](./integration.ts).
|
|
19
|
-
*
|
|
20
|
-
* There is no separate "e2e" tier: real **egress** is a *capability* of the
|
|
21
|
-
* harness/integration scope (`egressRoutes()` + `runHook`'s `egress: { allow }`),
|
|
22
|
-
* not a different kind of test — the old `e2e` barrel added exactly one symbol
|
|
23
|
-
* over `integration`, which is the definition of a non-tier. It now lives on
|
|
24
|
-
* `vigiles/integration`; this entry re-exports it unchanged so existing imports
|
|
25
|
-
* keep working. See `research/testing-api-design.md` Part 4 (two scopes, not
|
|
26
|
-
* four tiers). Prefer `vigiles/integration`.
|
|
27
|
-
*
|
|
28
|
-
* NOT here: **evals** (`runEval` / `measure` / `measureTriggerRate` / `judge`) —
|
|
29
|
-
* those are non-deterministic measurement, a different axis. (They live on
|
|
30
|
-
* `vigiles/testing`; `vigiles/eval` is named here in the original comment and does
|
|
31
|
-
* not exist as an entry point.)
|
|
32
|
-
*/
|
|
33
|
-
__exportStar(require("./integration.js"), exports);
|
|
34
|
-
//# sourceMappingURL=e2e.js.map
|