vigiles 16.1.2 → 17.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapter-conformance.js +17 -0
- package/dist/adapters/claude-code/dialect.d.ts +19 -13
- package/dist/adapters/claude-code/dialect.js +40 -62
- package/dist/adapters/claude-code/run-scripts.d.ts +19 -3
- package/dist/adapters/claude-code/run-scripts.js +17 -7
- package/dist/adapters/claude-code/vocabulary.d.ts +133 -0
- package/dist/adapters/claude-code/vocabulary.js +208 -0
- package/dist/adapters/codex/eval.js +2 -0
- package/dist/audit-score.js +1 -1
- package/dist/cli.js +7 -2
- package/dist/core/compile.js +6 -1
- package/dist/core/dialect.d.ts +27 -0
- package/dist/core/eval-load-phase.d.ts +78 -0
- package/dist/core/eval-load-phase.js +104 -0
- package/dist/core/hook-events.d.ts +32 -15
- package/dist/core/hook-events.js +23 -29
- package/dist/core/hook-program.js +12 -4
- package/dist/core/rule-meta.js +2 -2
- package/dist/core/tool-contract.d.ts +69 -30
- package/dist/core/tool-contract.js +59 -57
- package/dist/core/vocabulary-consistency.d.ts +35 -0
- package/dist/core/vocabulary-consistency.js +81 -0
- package/dist/core/vocabulary.d.ts +138 -0
- package/dist/core/vocabulary.js +262 -0
- package/dist/eval-define.d.ts +166 -0
- package/dist/eval-define.js +182 -0
- package/dist/eval-entry.d.ts +41 -0
- package/dist/eval-entry.js +203 -0
- package/dist/eval.js +2 -0
- package/dist/judge.js +2 -0
- package/dist/scan-behavioral.js +2 -0
- package/dist/scan-core.d.ts +8 -1
- package/dist/scan-core.js +64 -6
- package/dist/scan-files.js +4 -1
- package/dist/scan.d.ts +45 -0
- package/dist/scan.js +13 -1
- package/dist/test-coverage.d.ts +45 -0
- package/dist/test-coverage.js +91 -3
- package/dist/test.d.ts +2 -0
- package/dist/test.js +8 -1
- package/package.json +1 -1
- package/skills/test-harness/SKILL.md +31 -22
package/dist/test-coverage.js
CHANGED
|
@@ -67,6 +67,7 @@ exports.suggestedTestPath = suggestedTestPath;
|
|
|
67
67
|
exports.coverageEvidenceCounts = coverageEvidenceCounts;
|
|
68
68
|
exports.skillTestNudge = skillTestNudge;
|
|
69
69
|
exports.evalTierQuestion = evalTierQuestion;
|
|
70
|
+
exports.coverageCaveats = coverageCaveats;
|
|
70
71
|
exports.formatUntestedReport = formatUntestedReport;
|
|
71
72
|
const node_fs_1 = require("node:fs");
|
|
72
73
|
const node_path_1 = require("node:path");
|
|
@@ -431,6 +432,7 @@ function findUntestedSurfaces(options = {}) {
|
|
|
431
432
|
legacyCoversFiles: tests
|
|
432
433
|
.map((t) => t.path)
|
|
433
434
|
.filter((path) => read((0, node_path_1.join)(basePath, path)).includes(LEGACY_COVERS)),
|
|
435
|
+
retiredTestNames: retiredTestNamesFor(basePath, union.untested),
|
|
434
436
|
decisions: union.decisions,
|
|
435
437
|
harness: tierOf(considered, split.harness, runIndex, "harness"),
|
|
436
438
|
evals: tierOf(considered, split.evals, runIndex, "eval"),
|
|
@@ -662,6 +664,93 @@ function staleRunNote(report) {
|
|
|
662
664
|
`coverage. Re-run \`vigiles test\` / \`vigiles eval\` to refresh it.`,
|
|
663
665
|
];
|
|
664
666
|
}
|
|
667
|
+
/** Names a default vitest/jest run collects — the suffixes vigiles will not use. */
|
|
668
|
+
const FOREIGN_RUNNER_SUFFIX = /\.(test|spec)\.(ts|mts|cts|js|mjs|cjs)$/;
|
|
669
|
+
/**
|
|
670
|
+
* The would-be colocated tests that only their NAME disqualifies.
|
|
671
|
+
*
|
|
672
|
+
* Same two questions colocation asks — is it named after the surface, is it
|
|
673
|
+
* sitting beside it — with the third answer inverted: the suffix is one a
|
|
674
|
+
* default vitest/jest run collects, which is precisely why it left
|
|
675
|
+
* {@link DEFAULT_TEST_GLOBS}. Reading the directory (rather than globbing the
|
|
676
|
+
* repo again) keeps the cost at one `readdir` per untested surface and cannot
|
|
677
|
+
* reach a file that is not beside one.
|
|
678
|
+
*/
|
|
679
|
+
function retiredTestNamesFor(basePath, untested) {
|
|
680
|
+
const out = [];
|
|
681
|
+
for (const s of untested) {
|
|
682
|
+
const dir = (0, node_path_1.dirname)(s.path);
|
|
683
|
+
let entries;
|
|
684
|
+
try {
|
|
685
|
+
entries = (0, node_fs_1.readdirSync)((0, node_path_1.join)(basePath, dir));
|
|
686
|
+
}
|
|
687
|
+
catch {
|
|
688
|
+
continue; // a surface whose directory vanished mid-scan is not our finding
|
|
689
|
+
}
|
|
690
|
+
for (const entry of entries.sort()) {
|
|
691
|
+
if (!entry.startsWith(`${s.name}.`))
|
|
692
|
+
continue;
|
|
693
|
+
if (!FOREIGN_RUNNER_SUFFIX.test(entry))
|
|
694
|
+
continue;
|
|
695
|
+
out.push({
|
|
696
|
+
path: dir === "." ? entry : `${dir}/${entry}`,
|
|
697
|
+
surface: s.path,
|
|
698
|
+
});
|
|
699
|
+
}
|
|
700
|
+
}
|
|
701
|
+
return out;
|
|
702
|
+
}
|
|
703
|
+
/**
|
|
704
|
+
* One line for the author staring at a test they already wrote.
|
|
705
|
+
*
|
|
706
|
+
* The count says untested; the directory listing says `foo.test.mjs`. Without
|
|
707
|
+
* this the two never meet: the finding prints the SURFACE path and suggests a
|
|
708
|
+
* file to add, so the reader's most likely conclusion is that the tool cannot
|
|
709
|
+
* see their test — which is true, and the reason is a suffix nobody mentioned.
|
|
710
|
+
*/
|
|
711
|
+
function retiredTestNameNote(report) {
|
|
712
|
+
const found = report.retiredTestNames ?? [];
|
|
713
|
+
if (found.length === 0)
|
|
714
|
+
return [];
|
|
715
|
+
const shown = found
|
|
716
|
+
.slice(0, 3)
|
|
717
|
+
.map((f) => f.path)
|
|
718
|
+
.join(", ");
|
|
719
|
+
const more = found.length > 3 ? ` (+${String(found.length - 3)} more)` : "";
|
|
720
|
+
return [
|
|
721
|
+
` ${String(found.length)} file(s) sit beside an untested surface and are named ` +
|
|
722
|
+
`after it, but carry a \`.test.\`/\`.spec.\` suffix (${shown}${more}) — the name a ` +
|
|
723
|
+
`default vitest/jest run collects, which is why it stopped counting in 15.x. A ` +
|
|
724
|
+
`harness test drives an agent, and under a foreign runner vigiles refuses to ` +
|
|
725
|
+
`spawn one, so that run FAILS on it rather than testing anything. Rename to ` +
|
|
726
|
+
`\`<surface>.harness.<ext>\` and it counts here without being swept up there.`,
|
|
727
|
+
];
|
|
728
|
+
}
|
|
729
|
+
/**
|
|
730
|
+
* The QUALIFIERS on a coverage number — every caveat that says "this count is
|
|
731
|
+
* not quite what it looks like", in one list, built once.
|
|
732
|
+
*
|
|
733
|
+
* 🔴 THIS EXISTS BECAUSE A CAVEAT COULD BE PRINTED BY ONE COMMAND AND NOT THE
|
|
734
|
+
* OTHER, AND WAS. Both notes below were added to `formatUntestedReport` (the
|
|
735
|
+
* `lint` renderer) and neither reached `audit`, which assembles its own fact
|
|
736
|
+
* block from the same {@link UntestedReport}. Measured 2026-08-18 on a fixture
|
|
737
|
+
* whose only harness carried the retired `vigiles:covers` marker: `lint` named
|
|
738
|
+
* the file, `audit` printed `Untested surfaces: 0` and nothing else. The
|
|
739
|
+
* migration note existed, was unit-tested, and was invisible to anyone whose
|
|
740
|
+
* habit is `audit` — which is the whole point of a note that explains a silent
|
|
741
|
+
* migration.
|
|
742
|
+
*
|
|
743
|
+
* Collecting them here is the subtraction: a caveat is no longer something a
|
|
744
|
+
* renderer can choose to carry. Adding a third one reaches both callers or
|
|
745
|
+
* neither, and "neither" is a compile error rather than a quiet omission.
|
|
746
|
+
*/
|
|
747
|
+
function coverageCaveats(report) {
|
|
748
|
+
return [
|
|
749
|
+
...staleRunNote(report),
|
|
750
|
+
...legacyCoversNote(report),
|
|
751
|
+
...retiredTestNameNote(report),
|
|
752
|
+
];
|
|
753
|
+
}
|
|
665
754
|
function formatUntestedReport(report) {
|
|
666
755
|
// Every coverage number is printed WITH its provenance. "28 covered" and
|
|
667
756
|
// "28 covered, all of it a name appearing in a file" are different facts, and
|
|
@@ -669,7 +758,7 @@ function formatUntestedReport(report) {
|
|
|
669
758
|
const provenance = (0, coverage_evidence_js_1.formatEvidence)(coverageEvidenceCounts(report));
|
|
670
759
|
if (report.untested.length === 0) {
|
|
671
760
|
const tail = report.exempt > 0 ? ` (${String(report.exempt)} exempt)` : "";
|
|
672
|
-
const legacy =
|
|
761
|
+
const legacy = coverageCaveats(report);
|
|
673
762
|
const ok = `✓ all ${String(report.total)} surface(s) have a test or eval${tail}` +
|
|
674
763
|
(provenance ? `\n ${provenance}` : "") +
|
|
675
764
|
(legacy.length ? `\n${legacy.join("\n")}` : "");
|
|
@@ -700,8 +789,7 @@ function formatUntestedReport(report) {
|
|
|
700
789
|
// What the surfaces that DID pass are resting on.
|
|
701
790
|
if (provenance)
|
|
702
791
|
lines.push(` ${provenance}`);
|
|
703
|
-
lines.push(...
|
|
704
|
-
lines.push(...legacyCoversNote(report));
|
|
792
|
+
lines.push(...coverageCaveats(report));
|
|
705
793
|
// Already testing these another way (a promptfoo suite, a home-grown evals
|
|
706
794
|
// file)? Point `testGlobs` at it so it counts toward coverage (issue #113) —
|
|
707
795
|
// and put the file NEXT TO the surface, which is the only placement that
|
package/dist/test.d.ts
CHANGED
|
@@ -70,6 +70,8 @@ export { skillContract } from "./skill-contract.js";
|
|
|
70
70
|
export type { SkillContract, SkillContractOptions } from "./skill-contract.js";
|
|
71
71
|
export { compareContainment, formatContainment, } from "./trigger-containment.js";
|
|
72
72
|
export type { ContainmentInput, ContainmentVerdict, } from "./trigger-containment.js";
|
|
73
|
+
export { defineEval } from "./eval-define.js";
|
|
74
|
+
export type { EvalDefinition, EvalDefinitionInput, EvalHooks, EvalKind, EvalMeasurements, EvalReports, SelectionMatrixSpec, } from "./eval-define.js";
|
|
73
75
|
export { assertRates, assertPromptDiversity, checkPromptDiversity, checkReportToJUnit, formatCheckReport, formatEvalReport, formatTriggerRateReport, parseClaudeRun, stubSkillBody, } from "./eval.js";
|
|
74
76
|
export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
|
|
75
77
|
//# sourceMappingURL=test.d.ts.map
|
package/dist/test.js
CHANGED
|
@@ -67,7 +67,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
|
67
67
|
};
|
|
68
68
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
69
69
|
exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.loadHook = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
|
|
70
|
-
exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
|
|
70
|
+
exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
|
|
71
71
|
// --- reporting: how much did this script actually do? ---
|
|
72
72
|
// `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
|
|
73
73
|
// prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
|
|
@@ -175,6 +175,13 @@ Object.defineProperty(exports, "skillContract", { enumerable: true, get: functio
|
|
|
175
175
|
var trigger_containment_js_1 = require("./trigger-containment.js");
|
|
176
176
|
Object.defineProperty(exports, "compareContainment", { enumerable: true, get: function () { return trigger_containment_js_1.compareContainment; } });
|
|
177
177
|
Object.defineProperty(exports, "formatContainment", { enumerable: true, get: function () { return trigger_containment_js_1.formatContainment; } });
|
|
178
|
+
// --- declaring an eval (free: a description cannot spend) ---
|
|
179
|
+
// `defineEval` is on the FREE barrel and not on `vigiles/eval`, and that is the
|
|
180
|
+
// point rather than an oversight: after this shape an eval file needs nothing
|
|
181
|
+
// that bills. It describes the run; `vigiles eval` performs it. The paid runners
|
|
182
|
+
// moved out of an eval author's vocabulary entirely.
|
|
183
|
+
var eval_define_js_1 = require("./eval-define.js");
|
|
184
|
+
Object.defineProperty(exports, "defineEval", { enumerable: true, get: function () { return eval_define_js_1.defineEval; } });
|
|
178
185
|
// --- free analysis OVER eval results (the coupling from cost #1) ---
|
|
179
186
|
// These read a report that `vigiles/eval` produced. They spend nothing, so they
|
|
180
187
|
// live here and carry no `paid_` prefix; their argument types are defined over
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "vigiles",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "17.0.0",
|
|
4
4
|
"description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"claude-code",
|
|
@@ -207,38 +207,47 @@ score its output directly against a rubric. No on/off baseline — this is the
|
|
|
207
207
|
"is it any good?" oracle (what promptfoo/DeepEval lead with), and the right
|
|
208
208
|
default when there's nothing to compare against:
|
|
209
209
|
|
|
210
|
+
An eval file **describes** its eval — it must never run one at the top level,
|
|
211
|
+
because importing such a file spends real money. Write `<name>.eval.mjs`:
|
|
212
|
+
|
|
210
213
|
```ts
|
|
211
|
-
import {
|
|
212
|
-
import {
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
214
|
+
import { defineEval, skill, assertRates } from "vigiles";
|
|
215
|
+
import { paid_judged } from "vigiles/eval"; // a Check whose default judge bills
|
|
216
|
+
|
|
217
|
+
export default defineEval({
|
|
218
|
+
measure: {
|
|
219
|
+
pluginDir: "./",
|
|
220
|
+
task: "…a task the skill should handle…",
|
|
221
|
+
checks: [
|
|
222
|
+
skill("my-plugin:my-skill"), // it fired
|
|
223
|
+
paid_judged("the answer correctly does X and avoids Y"), // …and the output is good
|
|
224
|
+
],
|
|
225
|
+
trials: 6,
|
|
226
|
+
},
|
|
227
|
+
assert: (report) => assertRates(report, { min: 0.8 }), // each check ≥ 80% of trials
|
|
222
228
|
});
|
|
223
|
-
assertRates(report, { min: 0.8 }); // each check passes ≥ 80% of trials
|
|
224
229
|
```
|
|
225
230
|
|
|
231
|
+
Run it with `npx vigiles eval <file>` — never `node <file>`, which refuses.
|
|
232
|
+
|
|
226
233
|
**Eval — relative (`paid_runEval` + `assertSignificant`)** — when the question is
|
|
227
234
|
_lift over no-skill_ (regression, or proving a change isn't noise): A/B the
|
|
228
235
|
change on vs off and gate on significance, not eyeballing:
|
|
229
236
|
|
|
230
237
|
```ts
|
|
231
|
-
import {
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
238
|
+
import { defineEval, assertSignificant } from "vigiles";
|
|
239
|
+
|
|
240
|
+
export default defineEval({
|
|
241
|
+
runEval: {
|
|
242
|
+
arms: { off: {}, on: { pluginDir: "./" } },
|
|
243
|
+
task: "…a task the harness change should affect…",
|
|
244
|
+
measure: (ctx) => ({ ok: /* a bare predicate over the trace */ true }),
|
|
245
|
+
trials: 6,
|
|
246
|
+
cache: "readwrite",
|
|
247
|
+
},
|
|
248
|
+
assert: (report) =>
|
|
249
|
+
assertSignificant(report, { baseline: "off", arm: "on", metric: "ok" }),
|
|
240
250
|
});
|
|
241
|
-
assertSignificant(report, { baseline: "off", arm: "on", metric: "ok" });
|
|
242
251
|
```
|
|
243
252
|
|
|
244
253
|
### Never hand-roll the runner — it silently eats stderr
|