vigiles 16.1.2 → 17.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/dist/adapter-conformance.js +17 -0
  2. package/dist/adapters/claude-code/dialect.d.ts +19 -13
  3. package/dist/adapters/claude-code/dialect.js +40 -62
  4. package/dist/adapters/claude-code/run-scripts.d.ts +19 -3
  5. package/dist/adapters/claude-code/run-scripts.js +17 -7
  6. package/dist/adapters/claude-code/vocabulary.d.ts +133 -0
  7. package/dist/adapters/claude-code/vocabulary.js +208 -0
  8. package/dist/adapters/codex/eval.js +2 -0
  9. package/dist/audit-score.js +1 -1
  10. package/dist/cli.js +7 -2
  11. package/dist/core/compile.js +6 -1
  12. package/dist/core/dialect.d.ts +27 -0
  13. package/dist/core/eval-load-phase.d.ts +78 -0
  14. package/dist/core/eval-load-phase.js +104 -0
  15. package/dist/core/hook-events.d.ts +32 -15
  16. package/dist/core/hook-events.js +23 -29
  17. package/dist/core/hook-program.js +12 -4
  18. package/dist/core/rule-meta.js +2 -2
  19. package/dist/core/tool-contract.d.ts +69 -30
  20. package/dist/core/tool-contract.js +59 -57
  21. package/dist/core/vocabulary-consistency.d.ts +35 -0
  22. package/dist/core/vocabulary-consistency.js +81 -0
  23. package/dist/core/vocabulary.d.ts +138 -0
  24. package/dist/core/vocabulary.js +262 -0
  25. package/dist/eval-define.d.ts +166 -0
  26. package/dist/eval-define.js +182 -0
  27. package/dist/eval-entry.d.ts +41 -0
  28. package/dist/eval-entry.js +203 -0
  29. package/dist/eval.js +2 -0
  30. package/dist/judge.js +2 -0
  31. package/dist/scan-behavioral.js +2 -0
  32. package/dist/scan-core.d.ts +8 -1
  33. package/dist/scan-core.js +64 -6
  34. package/dist/scan-files.js +4 -1
  35. package/dist/scan.d.ts +45 -0
  36. package/dist/scan.js +13 -1
  37. package/dist/test-coverage.d.ts +45 -0
  38. package/dist/test-coverage.js +91 -3
  39. package/dist/test.d.ts +2 -0
  40. package/dist/test.js +8 -1
  41. package/package.json +1 -1
  42. package/skills/test-harness/SKILL.md +31 -22
@@ -67,6 +67,7 @@ exports.suggestedTestPath = suggestedTestPath;
67
67
  exports.coverageEvidenceCounts = coverageEvidenceCounts;
68
68
  exports.skillTestNudge = skillTestNudge;
69
69
  exports.evalTierQuestion = evalTierQuestion;
70
+ exports.coverageCaveats = coverageCaveats;
70
71
  exports.formatUntestedReport = formatUntestedReport;
71
72
  const node_fs_1 = require("node:fs");
72
73
  const node_path_1 = require("node:path");
@@ -431,6 +432,7 @@ function findUntestedSurfaces(options = {}) {
431
432
  legacyCoversFiles: tests
432
433
  .map((t) => t.path)
433
434
  .filter((path) => read((0, node_path_1.join)(basePath, path)).includes(LEGACY_COVERS)),
435
+ retiredTestNames: retiredTestNamesFor(basePath, union.untested),
434
436
  decisions: union.decisions,
435
437
  harness: tierOf(considered, split.harness, runIndex, "harness"),
436
438
  evals: tierOf(considered, split.evals, runIndex, "eval"),
@@ -662,6 +664,93 @@ function staleRunNote(report) {
662
664
  `coverage. Re-run \`vigiles test\` / \`vigiles eval\` to refresh it.`,
663
665
  ];
664
666
  }
667
+ /** Names a default vitest/jest run collects — the suffixes vigiles will not use. */
668
+ const FOREIGN_RUNNER_SUFFIX = /\.(test|spec)\.(ts|mts|cts|js|mjs|cjs)$/;
669
+ /**
670
+ * The would-be colocated tests that only their NAME disqualifies.
671
+ *
672
+ * Same two questions colocation asks — is it named after the surface, is it
673
+ * sitting beside it — with the third answer inverted: the suffix is one a
674
+ * default vitest/jest run collects, which is precisely why it left
675
+ * {@link DEFAULT_TEST_GLOBS}. Reading the directory (rather than globbing the
676
+ * repo again) keeps the cost at one `readdir` per untested surface and cannot
677
+ * reach a file that is not beside one.
678
+ */
679
+ function retiredTestNamesFor(basePath, untested) {
680
+ const out = [];
681
+ for (const s of untested) {
682
+ const dir = (0, node_path_1.dirname)(s.path);
683
+ let entries;
684
+ try {
685
+ entries = (0, node_fs_1.readdirSync)((0, node_path_1.join)(basePath, dir));
686
+ }
687
+ catch {
688
+ continue; // a surface whose directory vanished mid-scan is not our finding
689
+ }
690
+ for (const entry of entries.sort()) {
691
+ if (!entry.startsWith(`${s.name}.`))
692
+ continue;
693
+ if (!FOREIGN_RUNNER_SUFFIX.test(entry))
694
+ continue;
695
+ out.push({
696
+ path: dir === "." ? entry : `${dir}/${entry}`,
697
+ surface: s.path,
698
+ });
699
+ }
700
+ }
701
+ return out;
702
+ }
703
+ /**
704
+ * One line for the author staring at a test they already wrote.
705
+ *
706
+ * The count says untested; the directory listing says `foo.test.mjs`. Without
707
+ * this the two never meet: the finding prints the SURFACE path and suggests a
708
+ * file to add, so the reader's most likely conclusion is that the tool cannot
709
+ * see their test — which is true, and the reason is a suffix nobody mentioned.
710
+ */
711
+ function retiredTestNameNote(report) {
712
+ const found = report.retiredTestNames ?? [];
713
+ if (found.length === 0)
714
+ return [];
715
+ const shown = found
716
+ .slice(0, 3)
717
+ .map((f) => f.path)
718
+ .join(", ");
719
+ const more = found.length > 3 ? ` (+${String(found.length - 3)} more)` : "";
720
+ return [
721
+ ` ${String(found.length)} file(s) sit beside an untested surface and are named ` +
722
+ `after it, but carry a \`.test.\`/\`.spec.\` suffix (${shown}${more}) — the name a ` +
723
+ `default vitest/jest run collects, which is why it stopped counting in 15.x. A ` +
724
+ `harness test drives an agent, and under a foreign runner vigiles refuses to ` +
725
+ `spawn one, so that run FAILS on it rather than testing anything. Rename to ` +
726
+ `\`<surface>.harness.<ext>\` and it counts here without being swept up there.`,
727
+ ];
728
+ }
729
+ /**
730
+ * The QUALIFIERS on a coverage number — every caveat that says "this count is
731
+ * not quite what it looks like", in one list, built once.
732
+ *
733
+ * 🔴 THIS EXISTS BECAUSE A CAVEAT COULD BE PRINTED BY ONE COMMAND AND NOT THE
734
+ * OTHER, AND WAS. Both notes below were added to `formatUntestedReport` (the
735
+ * `lint` renderer) and neither reached `audit`, which assembles its own fact
736
+ * block from the same {@link UntestedReport}. Measured 2026-08-18 on a fixture
737
+ * whose only harness carried the retired `vigiles:covers` marker: `lint` named
738
+ * the file, `audit` printed `Untested surfaces: 0` and nothing else. The
739
+ * migration note existed, was unit-tested, and was invisible to anyone whose
740
+ * habit is `audit` — which is the whole point of a note that explains a silent
741
+ * migration.
742
+ *
743
+ * Collecting them here is the subtraction: a caveat is no longer something a
744
+ * renderer can choose to carry. Adding a third one reaches both callers or
745
+ * neither, and "neither" is a compile error rather than a quiet omission.
746
+ */
747
+ function coverageCaveats(report) {
748
+ return [
749
+ ...staleRunNote(report),
750
+ ...legacyCoversNote(report),
751
+ ...retiredTestNameNote(report),
752
+ ];
753
+ }
665
754
  function formatUntestedReport(report) {
666
755
  // Every coverage number is printed WITH its provenance. "28 covered" and
667
756
  // "28 covered, all of it a name appearing in a file" are different facts, and
@@ -669,7 +758,7 @@ function formatUntestedReport(report) {
669
758
  const provenance = (0, coverage_evidence_js_1.formatEvidence)(coverageEvidenceCounts(report));
670
759
  if (report.untested.length === 0) {
671
760
  const tail = report.exempt > 0 ? ` (${String(report.exempt)} exempt)` : "";
672
- const legacy = [...staleRunNote(report), ...legacyCoversNote(report)];
761
+ const legacy = coverageCaveats(report);
673
762
  const ok = `✓ all ${String(report.total)} surface(s) have a test or eval${tail}` +
674
763
  (provenance ? `\n ${provenance}` : "") +
675
764
  (legacy.length ? `\n${legacy.join("\n")}` : "");
@@ -700,8 +789,7 @@ function formatUntestedReport(report) {
700
789
  // What the surfaces that DID pass are resting on.
701
790
  if (provenance)
702
791
  lines.push(` ${provenance}`);
703
- lines.push(...staleRunNote(report));
704
- lines.push(...legacyCoversNote(report));
792
+ lines.push(...coverageCaveats(report));
705
793
  // Already testing these another way (a promptfoo suite, a home-grown evals
706
794
  // file)? Point `testGlobs` at it so it counts toward coverage (issue #113) —
707
795
  // and put the file NEXT TO the surface, which is the only placement that
package/dist/test.d.ts CHANGED
@@ -70,6 +70,8 @@ export { skillContract } from "./skill-contract.js";
70
70
  export type { SkillContract, SkillContractOptions } from "./skill-contract.js";
71
71
  export { compareContainment, formatContainment, } from "./trigger-containment.js";
72
72
  export type { ContainmentInput, ContainmentVerdict, } from "./trigger-containment.js";
73
+ export { defineEval } from "./eval-define.js";
74
+ export type { EvalDefinition, EvalDefinitionInput, EvalHooks, EvalKind, EvalMeasurements, EvalReports, SelectionMatrixSpec, } from "./eval-define.js";
73
75
  export { assertRates, assertPromptDiversity, checkPromptDiversity, checkReportToJUnit, formatCheckReport, formatEvalReport, formatTriggerRateReport, parseClaudeRun, stubSkillBody, } from "./eval.js";
74
76
  export type { EvalArm, EvalDriver, EvalSpec, EvalReport, EvalUsage, MeasureSpec, ArmsMeasureSpec, ArmReport, ArmUsage, ArmsCheckReport, CheckRate, CheckReport, MetricStat, Metrics, ModelOutputParser, ParsedModelRun, PromptDiversityIssue, PromptTriggerStat, RunContext, RunOut, SelectionTrialResult, TriggerRateReport, TriggerRateSpec, AgentRunArgs, AgentRunner, } from "./eval.js";
75
77
  //# sourceMappingURL=test.d.ts.map
package/dist/test.js CHANGED
@@ -67,7 +67,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
67
67
  };
68
68
  Object.defineProperty(exports, "__esModule", { value: true });
69
69
  exports.mustNotInclude = exports.mustInclude = exports.commandsIn = exports.sandboxAvailable = exports.specTrusted = exports.decideSandbox = exports.parseHooks = exports.parseOutput = exports.parseResultEvent = exports.parseSubagents = exports.parseToolCalls = exports.runHarness = exports.runHarnessTest = exports.formatGuardrailReport = exports.assertBlocksDisasters = exports.unblockedDisasters = exports.verifyGuardrail = exports.DISASTER_CATALOG = exports.cacheTokens = exports.outputTokens = exports.inputTokens = exports.tokens = exports.latency = exports.cost = exports.mcp = exports.allowed = exports.blocked = exports.subagent = exports.didNotWrite = exports.wrote = exports.turns = exports.received = exports.hookFired = exports.output = exports.skill = exports.onlyTools = exports.notTool = exports.toolWith = exports.tool = exports.assertChecks = exports.evalChecks = exports.loadHook = exports.egressRoutes = exports.fileToolEvents = exports.propertyHook = exports.decideHook = exports.parseHookOutput = exports.runHook = exports.runScript = exports.recordCheck = void 0;
70
- exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
70
+ exports.stubSkillBody = exports.parseClaudeRun = exports.formatTriggerRateReport = exports.formatEvalReport = exports.formatCheckReport = exports.checkReportToJUnit = exports.checkPromptDiversity = exports.assertPromptDiversity = exports.assertRates = exports.defineEval = exports.formatContainment = exports.compareContainment = exports.skillContract = void 0;
71
71
  // --- reporting: how much did this script actually do? ---
72
72
  // `vigiles test` can otherwise see only an exit code, so a file that runs NOTHING
73
73
  // prints the same `✓` as one that ran and passed (measured 2026-08-08 on a file
@@ -175,6 +175,13 @@ Object.defineProperty(exports, "skillContract", { enumerable: true, get: functio
175
175
  var trigger_containment_js_1 = require("./trigger-containment.js");
176
176
  Object.defineProperty(exports, "compareContainment", { enumerable: true, get: function () { return trigger_containment_js_1.compareContainment; } });
177
177
  Object.defineProperty(exports, "formatContainment", { enumerable: true, get: function () { return trigger_containment_js_1.formatContainment; } });
178
+ // --- declaring an eval (free: a description cannot spend) ---
179
+ // `defineEval` is on the FREE barrel and not on `vigiles/eval`, and that is the
180
+ // point rather than an oversight: after this shape an eval file needs nothing
181
+ // that bills. It describes the run; `vigiles eval` performs it. The paid runners
182
+ // moved out of an eval author's vocabulary entirely.
183
+ var eval_define_js_1 = require("./eval-define.js");
184
+ Object.defineProperty(exports, "defineEval", { enumerable: true, get: function () { return eval_define_js_1.defineEval; } });
178
185
  // --- free analysis OVER eval results (the coupling from cost #1) ---
179
186
  // These read a report that `vigiles/eval` produced. They spend nothing, so they
180
187
  // live here and carry no `paid_` prefix; their argument types are defined over
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "vigiles",
3
- "version": "16.1.2",
3
+ "version": "17.0.0",
4
4
  "description": "Audit, test and measure the harness your AI agent runs on — grade your CLAUDE.md / AGENTS.md, skills, subagents and hooks, run them against a scripted model, and measure whether they actually fire.",
5
5
  "keywords": [
6
6
  "claude-code",
@@ -207,38 +207,47 @@ score its output directly against a rubric. No on/off baseline — this is the
207
207
  "is it any good?" oracle (what promptfoo/DeepEval lead with), and the right
208
208
  default when there's nothing to compare against:
209
209
 
210
+ An eval file **describes** its eval — it must never run one at the top level,
211
+ because importing such a file spends real money. Write `<name>.eval.mjs`:
212
+
210
213
  ```ts
211
- import { paid_measure, paid_judged } from "vigiles/eval"; // `paid_` = these call a model
212
- import { skill, assertRates } from "vigiles";
213
-
214
- const report = await paid_measure({
215
- pluginDir: "./",
216
- task: "…a task the skill should handle…",
217
- checks: [
218
- skill("my-plugin:my-skill"), // it fired
219
- paid_judged("the answer correctly does X and avoids Y"), // …and the output is good
220
- ],
221
- trials: 6,
214
+ import { defineEval, skill, assertRates } from "vigiles";
215
+ import { paid_judged } from "vigiles/eval"; // a Check whose default judge bills
216
+
217
+ export default defineEval({
218
+ measure: {
219
+ pluginDir: "./",
220
+ task: "…a task the skill should handle…",
221
+ checks: [
222
+ skill("my-plugin:my-skill"), // it fired
223
+ paid_judged("the answer correctly does X and avoids Y"), // …and the output is good
224
+ ],
225
+ trials: 6,
226
+ },
227
+ assert: (report) => assertRates(report, { min: 0.8 }), // each check ≥ 80% of trials
222
228
  });
223
- assertRates(report, { min: 0.8 }); // each check passes ≥ 80% of trials
224
229
  ```
225
230
 
231
+ Run it with `npx vigiles eval <file>` — never `node <file>`, which refuses.
232
+
226
233
  **Eval — relative (`paid_runEval` + `assertSignificant`)** — when the question is
227
234
  _lift over no-skill_ (regression, or proving a change isn't noise): A/B the
228
235
  change on vs off and gate on significance, not eyeballing:
229
236
 
230
237
  ```ts
231
- import { paid_runEval } from "vigiles/eval"; // `paid_` = a real model runs
232
- import { assertSignificant } from "vigiles";
233
-
234
- const report = await paid_runEval({
235
- arms: { off: {}, on: { pluginDir: "./" } },
236
- task: "…a task the harness change should affect…",
237
- measure: (ctx) => ({ ok: /* a bare predicate over the trace */ true }),
238
- trials: 6,
239
- cache: "readwrite",
238
+ import { defineEval, assertSignificant } from "vigiles";
239
+
240
+ export default defineEval({
241
+ runEval: {
242
+ arms: { off: {}, on: { pluginDir: "./" } },
243
+ task: "…a task the harness change should affect…",
244
+ measure: (ctx) => ({ ok: /* a bare predicate over the trace */ true }),
245
+ trials: 6,
246
+ cache: "readwrite",
247
+ },
248
+ assert: (report) =>
249
+ assertSignificant(report, { baseline: "off", arm: "on", metric: "ok" }),
240
250
  });
241
- assertSignificant(report, { baseline: "off", arm: "on", metric: "ok" });
242
251
  ```
243
252
 
244
253
  ### Never hand-roll the runner — it silently eats stderr