vigiles 8.0.0 → 9.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -26,12 +26,16 @@ const effects_js_1 = require("./core/effects.js");
26
26
  const scan_js_1 = require("./scan.js");
27
27
  const scan_trigger_suggest_js_1 = require("./scan-trigger-suggest.js");
28
28
  const dialect_drift_js_1 = require("./dialect-drift.js");
29
- const score_explainer_js_1 = require("./score-explainer.js");
30
29
  const scan_behavioral_js_1 = require("./scan-behavioral.js");
31
30
  const adapter_registry_js_1 = require("./adapter-registry.js");
32
31
  const skill_harness_js_1 = require("./skill-harness.js");
33
32
  const leaderboard_js_1 = require("./leaderboard.js");
34
33
  const optimize_js_1 = require("./optimize.js");
34
+ const audit_score_js_1 = require("./audit-score.js");
35
+ const audit_prompts_js_1 = require("./audit-prompts.js");
36
+ const audit_html_js_1 = require("./audit-html.js");
37
+ const audit_report_js_1 = require("./audit-report.js");
38
+ const adoptability_js_1 = require("./adoptability.js");
35
39
  const compile_js_1 = require("./core/compile.js");
36
40
  const proofs_js_1 = require("./core/proofs.js");
37
41
  const inline_js_1 = require("./core/inline.js");
@@ -1206,7 +1210,110 @@ function targetHasHash(absPath) {
1206
1210
  return false;
1207
1211
  }
1208
1212
  }
1209
- function init(args) {
1213
+ /**
1214
+ * Single-target spec scaffolder — the small building block behind the `init`
1215
+ * verb, NOT the wizard. Creates exactly one sibling `<target>.spec.ts`: it
1216
+ * faithfully ADOPTS an existing hand-written instruction file (non-destructive —
1217
+ * never overwrites the markdown) or writes a blank starter for a greenfield
1218
+ * target. Called directly for `vigiles init --target=<file>`, and once per
1219
+ * target by `setupPillar1`. The full onboarding (both layers, deps, CI, plugin)
1220
+ * is `setup()`.
1221
+ */
1222
+ /** Classify an adoption target by its path: a `SKILL.md` is a skill, a file under
1223
+ * an `agents/` dir is a subagent, everything else is an instruction file. Used to
1224
+ * pick the right adopt function so `init --target=skills/x/SKILL.md` (the
1225
+ * per-surface path the audit report points at) makes a `skill()`/`agent()` spec. */
1226
+ function surfaceKind(target) {
1227
+ if (/^SKILL\.md$/i.test((0, node_path_1.basename)(target)))
1228
+ return "skill";
1229
+ if (/(^|[/\\])agents[/\\]/.test(target))
1230
+ return "agent";
1231
+ return "instruction";
1232
+ }
1233
+ function logAdoptedSurface(target, specPath, label, unmappedKeys) {
1234
+ const note = unmappedKeys.length > 0
1235
+ ? ` (review the // NOTE — unmapped frontmatter: ${unmappedKeys.join(", ")})`
1236
+ : "";
1237
+ console.log(`Adopted ${label} ${target} → ${specPath}${note}. ` +
1238
+ `Run \`vigiles compile\` and review the diff.`);
1239
+ }
1240
+ /** Discover existing skill (`skills/<x>/SKILL.md`) and subagent (`agents/<x>.md`)
1241
+ * surfaces — under the bare or `.claude/` roots — that don't yet have a spec, so
1242
+ * bare `vigiles init` creates a spec for EVERY surface it can, not just the
1243
+ * instruction file. Shallow (top-level only) so it never walks node_modules or a
1244
+ * vendored plugin. CC paths are intentional here — `init` is the one composition
1245
+ * point allowed to know them (see adapter-aware-lint-rules). */
1246
+ function discoverAdoptableSurfaces(cwd) {
1247
+ const out = [];
1248
+ const unspecced = (rel) => (0, node_fs_1.existsSync)((0, node_path_1.resolve)(cwd, rel)) &&
1249
+ !(0, node_fs_1.existsSync)((0, node_path_1.resolve)(cwd, `${rel}.spec.ts`));
1250
+ for (const root of ["skills", ".claude/skills"]) {
1251
+ const abs = (0, node_path_1.resolve)(cwd, root);
1252
+ if (!(0, node_fs_1.existsSync)(abs))
1253
+ continue;
1254
+ for (const e of (0, node_fs_1.readdirSync)(abs, { withFileTypes: true })) {
1255
+ const rel = `${root}/${e.name}/SKILL.md`;
1256
+ if (e.isDirectory() && unspecced(rel))
1257
+ out.push(rel);
1258
+ }
1259
+ }
1260
+ for (const root of ["agents", ".claude/agents"]) {
1261
+ const abs = (0, node_path_1.resolve)(cwd, root);
1262
+ if (!(0, node_fs_1.existsSync)(abs))
1263
+ continue;
1264
+ for (const e of (0, node_fs_1.readdirSync)(abs, { withFileTypes: true })) {
1265
+ const rel = `${root}/${e.name}`;
1266
+ if (e.isFile() && e.name.endsWith(".md") && unspecced(rel))
1267
+ out.push(rel);
1268
+ }
1269
+ }
1270
+ return out;
1271
+ }
1272
+ /** The full adoptable-surface list `audit` reports: the instruction file (when it
1273
+ * exists hand-written, no spec) PLUS every skill/subagent surface without a spec.
1274
+ * Same notion `init` adopts; surfaced in the AuditReport + the terminal nudge so
1275
+ * the report's "Create spec" / "Create all specs" affordances have their paths.
1276
+ * Composition-root only — CC paths are intentional here (like discoverAdoptableSurfaces). */
1277
+ function discoverAdoptableForAudit(root, instructionFile) {
1278
+ const out = [];
1279
+ const instrAbs = (0, node_path_1.resolve)(root, instructionFile);
1280
+ if ((0, node_fs_1.existsSync)(instrAbs) &&
1281
+ !targetHasHash(instrAbs) &&
1282
+ !(0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, `${instructionFile}.spec.ts`))) {
1283
+ out.push(instructionFile);
1284
+ }
1285
+ out.push(...discoverAdoptableSurfaces(root));
1286
+ return out;
1287
+ }
1288
+ /** The terminal "adoptable surfaces" nudge — N un-spec'd surfaces + the create-all
1289
+ * command and up to ~5 per-surface commands (then "+K more"). "" when nothing to
1290
+ * adopt (a fully spec-managed repo says nothing). */
1291
+ function formatAdoptableNudge(surfaces) {
1292
+ if (surfaces.length === 0)
1293
+ return "";
1294
+ const n = surfaces.length;
1295
+ const lines = [
1296
+ `ℹ ${String(n)} surface${n === 1 ? "" : "s"} not yet spec-managed — create specs with \`npx vigiles init\``,
1297
+ ` (or one at a time: \`npx vigiles init --target=<path>\`)`,
1298
+ ];
1299
+ const shown = surfaces.slice(0, 5);
1300
+ for (const s of shown)
1301
+ lines.push(` • npx vigiles init --target=${s}`);
1302
+ const more = n - shown.length;
1303
+ if (more > 0)
1304
+ lines.push(` • +${String(more)} more`);
1305
+ return lines.join("\n");
1306
+ }
1307
+ /** A small, terse behavioral nudge — the deterministic read can't tell whether a
1308
+ * skill actually FIRES. "" when there are no model-invocable skills. */
1309
+ function formatTriggerNudge(triggerableSkills) {
1310
+ if (triggerableSkills <= 0)
1311
+ return "";
1312
+ const n = triggerableSkills;
1313
+ return (`ℹ Do your ${String(n)} skill${n === 1 ? "" : "s"} actually fire? The deterministic read can't tell — ` +
1314
+ `run \`audit\` interactively to measure, or test with \`measureTriggerRate\` (vigiles/testing).`);
1315
+ }
1316
+ function scaffoldSpec(args) {
1210
1317
  const targetFlag = args.find((a) => a.startsWith("--target="));
1211
1318
  const target = targetFlag ? targetFlag.split("=")[1] : "CLAUDE.md";
1212
1319
  const specPath = `${target}.spec.ts`;
@@ -1224,11 +1331,24 @@ function init(args) {
1224
1331
  const targetAbs = (0, node_path_1.resolve)(process.cwd(), target);
1225
1332
  if ((0, node_fs_1.existsSync)(targetAbs) && !targetHasHash(targetAbs)) {
1226
1333
  const md = (0, node_fs_1.readFileSync)(targetAbs, "utf-8");
1227
- const { source, tier, sectionCount } = (0, adopt_js_1.adoptMarkdown)(md, (0, node_path_1.basename)(target));
1228
1334
  (0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(specAbs), { recursive: true });
1229
- (0, node_fs_1.writeFileSync)(specAbs, source);
1230
- console.log(`Adopted ${target} → ${specPath} (${tier}, ${String(sectionCount)} section${sectionCount === 1 ? "" : "s"}). ` +
1231
- `Run \`vigiles compile\` and review the diff; the \`/strengthen\` skill upgrades prose to verified rules.`);
1335
+ const kind = surfaceKind(target);
1336
+ if (kind === "skill") {
1337
+ const { source, unmappedKeys } = (0, adopt_js_1.adoptSkill)(md, (0, node_path_1.basename)((0, node_path_1.dirname)(target)));
1338
+ (0, node_fs_1.writeFileSync)(specAbs, source);
1339
+ logAdoptedSurface(target, specPath, "skill", unmappedKeys);
1340
+ }
1341
+ else if (kind === "agent") {
1342
+ const { source, unmappedKeys } = (0, adopt_js_1.adoptAgent)(md, (0, node_path_1.basename)(target, ".md"));
1343
+ (0, node_fs_1.writeFileSync)(specAbs, source);
1344
+ logAdoptedSurface(target, specPath, "subagent", unmappedKeys);
1345
+ }
1346
+ else {
1347
+ const { source, tier, sectionCount } = (0, adopt_js_1.adoptMarkdown)(md, (0, node_path_1.basename)(target));
1348
+ (0, node_fs_1.writeFileSync)(specAbs, source);
1349
+ console.log(`Adopted ${target} → ${specPath} (${tier}, ${String(sectionCount)} section${sectionCount === 1 ? "" : "s"}). ` +
1350
+ `Run \`vigiles compile\` and review the diff; the \`/strengthen\` skill upgrades prose to verified rules.`);
1351
+ }
1232
1352
  return;
1233
1353
  }
1234
1354
  // The compiled output is derived from the spec FILE path; the spec's `target`
@@ -1454,7 +1574,7 @@ function workflowUsesStaleApi(content) {
1454
1574
  return false; // uses the Action — fine
1455
1575
  if (!/\bvigiles\b/.test(content))
1456
1576
  return false; // not a vigiles workflow
1457
- const hasModernCmd = /vigiles\s+(lint|test|eval|compile|scan|generate-types|generate-schema|init)\b/.test(content);
1577
+ const hasModernCmd = /vigiles\s+(lint|test|eval|compile|audit|generate-types|generate-schema|init)\b/.test(content);
1458
1578
  return !hasModernCmd;
1459
1579
  }
1460
1580
  /**
@@ -1465,7 +1585,7 @@ function workflowUsesStaleApi(content) {
1465
1585
  * the bare-API heuristic above, which an Action reference short-circuits.
1466
1586
  */
1467
1587
  const REMOVED_SUBCOMMANDS = {
1468
- audit: "lint", // v3 → v4 rename
1588
+ scan: "audit", // renamed: the Lighthouse report verb is `audit`
1469
1589
  };
1470
1590
  /** The first removed/renamed `vigiles <sub>` a workflow still calls, if any. */
1471
1591
  function workflowRemovedSubcommand(content) {
@@ -1475,7 +1595,7 @@ function workflowRemovedSubcommand(content) {
1475
1595
  }
1476
1596
  return null;
1477
1597
  }
1478
- /** Rewrite removed/renamed `vigiles <sub>` invocations in place (audit → lint).
1598
+ /** Rewrite removed/renamed `vigiles <sub>` invocations in place (scan → audit).
1479
1599
  * Surgical — preserves the rest of the user's workflow. */
1480
1600
  function rewriteRemovedSubcommands(content) {
1481
1601
  let out = content;
@@ -1755,11 +1875,17 @@ async function setupPillar1(detected, targetValue, harnesses) {
1755
1875
  // An explicit --target is honoured as-is; otherwise collapse a CLAUDE.md⇄
1756
1876
  // AGENTS.md mirror (symlink or synced) to one canonical spec, then redirect
1757
1877
  // into a sync tool's source slot when one would own the output.
1758
- const targets = targetValue
1878
+ const instructionTargets = targetValue
1759
1879
  ? determineTargets(detected, targetValue, harnesses)
1760
1880
  : redirectSyncToolTargets(cwd, collapseMirroredTargets(determineTargets(detected, targetValue, harnesses), (0, compose_js_1.detectInstructionMirror)(cwd)));
1881
+ // Bare `init` (no explicit --target) also adopts every existing skill +
1882
+ // subagent surface — "create all the specs it can", not just the instruction
1883
+ // file. An explicit --target stays scoped to that one surface.
1884
+ const targets = targetValue
1885
+ ? instructionTargets
1886
+ : [...instructionTargets, ...discoverAdoptableSurfaces(cwd)];
1761
1887
  // Create specs. An existing hand-written target is faithfully ADOPTED into a
1762
- // spec (init() does the convert), not clobbered with a blank one — so the
1888
+ // spec (scaffoldSpec() does the convert), not clobbered with a blank one — so the
1763
1889
  // compile below reproduces it (the user reviews the diff). A greenfield target
1764
1890
  // gets a blank starter spec.
1765
1891
  for (const target of targets) {
@@ -1772,7 +1898,7 @@ async function setupPillar1(detected, targetValue, harnesses) {
1772
1898
  console.log(`✓ ${specPath} already exists`);
1773
1899
  }
1774
1900
  else {
1775
- init(["--target=" + target]); // adopts existing content, else blank scaffold
1901
+ scaffoldSpec(["--target=" + target]); // adopts existing content, else blank scaffold
1776
1902
  written.push(specPath);
1777
1903
  if (willAdopt)
1778
1904
  adopted.push(target);
@@ -2709,14 +2835,14 @@ function flagValue(args, name) {
2709
2835
  return args.find((a) => a.startsWith(`${name}=`))?.slice(name.length + 1);
2710
2836
  }
2711
2837
  /**
2712
- * `vigiles scan <dir> --trigger` — the MODEL-GATED behavioral report on a plugin (the paid
2713
- * tier; `scan` stays free/deterministic). Loads the author-supplied per-skill prompt
2714
- * sets (`--prompts`) and reports BOTH behavioral columns: trigger-rate (does each
2715
- * skill actually FIRE — recall + precision) and the selection-collision matrix (does
2716
- * one skill HIJACK a sibling's prompt — the behavioral confirmation of the
2838
+ * The MODEL-GATED behavioral half of `vigiles audit` (the model trigger tier; the
2839
+ * deterministic core of `audit` stays free). Loads the author-supplied per-skill
2840
+ * prompt sets (`--prompts`) and reports BOTH behavioral columns: trigger-rate (does
2841
+ * each skill actually FIRE — recall + precision) and the selection-collision matrix
2842
+ * (does one skill HIJACK a sibling's prompt — the behavioral confirmation of the
2717
2843
  * deterministic `description-overlap` rule). Needs the harness CLI + model auth;
2718
2844
  * degrades honestly ("unavailable") when absent. The OSS-testing front door:
2719
- * `vigiles scan ./plugin --trigger --prompts=p.json`.
2845
+ * `vigiles audit ./plugin --prompts=p.json` (interactive — say yes when asked).
2720
2846
  */
2721
2847
  async function handleMeasure(restArgs, args) {
2722
2848
  const dir = (0, node_path_1.resolve)(restArgs[0] ?? ".");
@@ -3051,32 +3177,6 @@ function harnessFlagFrom(argv) {
3051
3177
  .find((a) => a.startsWith("--harness="))
3052
3178
  ?.slice("--harness=".length);
3053
3179
  }
3054
- /**
3055
- * `vigiles scan <dir> --explain [name]` — the deterministic WHY behind a low score (C4):
3056
- * scan a plugin and surface the structural CAUSE of a behavioral symptom + the
3057
- * one-line fix. No model — it reads the same `ScanReport` `scan` computes. An
3058
- * optional surface name narrows to one underperforming skill/agent (the
3059
- * optimizer's call). `--json` for the agent-consumable shape, `--harness=` to
3060
- * override detection.
3061
- */
3062
- function handleExplain(restArgs, args) {
3063
- const dir = (0, node_path_1.resolve)(restArgs[0] ?? ".");
3064
- const surface = restArgs[1];
3065
- const json = args.includes("--json");
3066
- const harnessFlag = harnessFlagFrom(args);
3067
- const adapter = harnessFlag
3068
- ? (0, adapter_registry_js_1.resolveAdapter)(dir, harnessFlag)
3069
- : (0, adapter_registry_js_1.detectAdapterResult)(dir).adapter;
3070
- const report = (0, scan_js_1.scanPlugin)(dir, adapter.layout, adapter.dialect);
3071
- const exps = surface ? (0, score_explainer_js_1.explainSurface)(report, surface) : (0, score_explainer_js_1.explainScore)(report);
3072
- if (json) {
3073
- console.log(JSON.stringify(exps, null, 2));
3074
- return;
3075
- }
3076
- if (surface)
3077
- console.log(`Explaining "${surface}":\n`);
3078
- console.log((0, score_explainer_js_1.formatExplanations)(exps));
3079
- }
3080
3180
  /**
3081
3181
  * Whole-harness capability lattice from a scanned plugin's agents (no `tools:` line →
3082
3182
  * inherits-all). The substrate `scan --capability-diff` diffs. Reused for both the
@@ -3227,9 +3327,9 @@ function printUsage(command) {
3227
3327
  console.log(" vigiles compile [files...] Compile .spec.ts → .md");
3228
3328
  console.log(" vigiles eject [file] Un-manage a compiled file → plain hand-owned markdown (--keep-spec)");
3229
3329
  console.log(" vigiles lint [files...] Verify references, find gaps in instruction files");
3230
- console.log(" vigiles scan [dir...] Report what a plugin ships + what's broken (free; 2+ dirs → leaderboard)");
3231
- console.log(" --trigger: do skills fire/collide? (real model) · --explain: why a surface underperforms · --fix-plan");
3232
- console.log(" with model access + a TTY, scan offers to measure firing; --no-interactive/--yes/--json hint instead (agents)");
3330
+ console.log(" vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)");
3331
+ console.log(" writes vigiles-report.html + vigiles-report.json (--no-html/--no-json) · --json for machine output. NOT a CI step — use `vigiles lint` in CI.");
3332
+ console.log(" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles/testing API");
3233
3333
  console.log(" vigiles test [files...] Run *.harness.mjs deterministic harness tests");
3234
3334
  console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
3235
3335
  console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
@@ -4030,32 +4130,175 @@ async function runHookProgramCommand(file) {
4030
4130
  }
4031
4131
  }
4032
4132
  /**
4033
- * After a single-plugin `scan` report: if the plugin ships model-invocable
4034
- * skills AND a real model is reachable, surface the real-model `--trigger` tier
4035
- * that measures whether those skills actually FIRE. A human at a TTY is offered
4036
- * setup; an agent / CI (non-TTY / `--json` / `--no-interactive` / `--yes`) gets a
4037
- * one-line, non-blocking hint — a `scan` must never hang (`great-agent-flow`).
4133
+ * Write the versioned JSON artifact (`vigiles-report.json`) — the upload/CI
4134
+ * boundary a hosted dashboard ingests. Stamps `meta.generatedAt` here (at write
4135
+ * time, not in the pure builder, so the HTML-embedded form stays deterministic).
4136
+ */
4137
+ function writeAuditJson(report) {
4138
+ const jsonPath = (0, node_path_1.resolve)(process.cwd(), "vigiles-report.json");
4139
+ const stamped = {
4140
+ ...report,
4141
+ meta: { ...report.meta, generatedAt: new Date().toISOString() },
4142
+ };
4143
+ try {
4144
+ (0, node_fs_1.writeFileSync)(jsonPath, JSON.stringify(stamped, null, 2) + "\n");
4145
+ console.log("✓ Wrote vigiles-report.json — the upload/CI artifact");
4146
+ }
4147
+ catch (e) {
4148
+ console.log(`\n⚠ could not write vigiles-report.json: ${e instanceof Error ? e.message : String(e)}`);
4149
+ }
4150
+ }
4151
+ /** Best-effort open the report in the default browser (TTY-only caller). */
4152
+ function openBestEffort(file) {
4153
+ const cmd = process.platform === "darwin"
4154
+ ? "open"
4155
+ : process.platform === "win32"
4156
+ ? "start"
4157
+ : "xdg-open";
4158
+ void import("node:child_process")
4159
+ .then(({ spawn }) => {
4160
+ const child = spawn(cmd, [file], {
4161
+ stdio: "ignore",
4162
+ detached: true,
4163
+ shell: process.platform === "win32",
4164
+ });
4165
+ child.on("error", () => undefined);
4166
+ child.unref();
4167
+ })
4168
+ .catch(() => undefined);
4169
+ }
4170
+ /**
4171
+ * Write the self-contained HTML audit report to `vigiles-report.html` (cwd) and,
4172
+ * for a human at a TTY, open it best-effort. The shareable Lighthouse artifact;
4173
+ * never spawns a browser for an agent / CI run.
4174
+ */
4175
+ function writeAuditHtml(report) {
4176
+ const htmlPath = (0, node_path_1.resolve)(process.cwd(), "vigiles-report.html");
4177
+ try {
4178
+ (0, node_fs_1.writeFileSync)(htmlPath, (0, audit_html_js_1.renderAuditHtml)(report));
4179
+ console.log("\n✓ Wrote vigiles-report.html — open it for the full report");
4180
+ if (process.stdout.isTTY)
4181
+ openBestEffort(htmlPath);
4182
+ }
4183
+ catch (e) {
4184
+ // No template (unbuilt checkout) or a write error — skip the HTML; the JSON
4185
+ // artifact + terminal report don't depend on it.
4186
+ console.log(`\n⚠ skipped vigiles-report.html: ${e instanceof Error ? e.message : String(e)}`);
4187
+ }
4188
+ }
4189
+ /**
4190
+ * Run the model trigger tier with no `--prompts`: auto-generate diverse probe
4191
+ * prompts from each skill's description and measure trigger-rate (recall +
4192
+ * precision) — zero-setup. `--prompts=<file>` (handled by handleMeasure)
4193
+ * overrides for a curated benchmark + the collision matrix. Model-gated: the
4194
+ * probes are deterministic, but RUNNING them needs the harness CLI + model auth
4195
+ * (degrades to "unavailable" otherwise).
4196
+ */
4197
+ async function runAutoTrigger(dir, report, adapter, args) {
4198
+ const json = args.includes("--json");
4199
+ const harness = adapter.name === "codex" ? "codex" : "claude-code";
4200
+ const skills = report.skills
4201
+ .filter((s) => s.hasDescription && !s.userInvoked && s.description)
4202
+ .map((s) => ({ name: s.name, description: s.description ?? "" }));
4203
+ if (skills.length === 0) {
4204
+ if (!json) {
4205
+ console.log("\nℹ no model-invocable skills with a description to measure.");
4206
+ }
4207
+ return;
4208
+ }
4209
+ const promptSet = (0, audit_prompts_js_1.autoTriggerPrompts)(skills);
4210
+ if (!json) {
4211
+ console.log("\nℹ auto-generated probe prompts from skill descriptions (pass --prompts=<file> for a curated set).");
4212
+ }
4213
+ const trigger = await (0, scan_behavioral_js_1.probePluginTriggers)(dir, promptSet, {
4214
+ minPrompts: audit_prompts_js_1.AUTO_RECALL_COUNT,
4215
+ minDistance: audit_prompts_js_1.AUTO_MIN_DISTANCE,
4216
+ model: flagValue(args, "--model"),
4217
+ harness,
4218
+ // Discover candidates with the resolved adapter's layout/dialect — a Codex
4219
+ // repo's skills live under the Codex layout, not the default CC one.
4220
+ layout: adapter.layout,
4221
+ dialect: adapter.dialect,
4222
+ });
4223
+ console.log(json
4224
+ ? JSON.stringify({ trigger }, null, 2)
4225
+ : "\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
4226
+ }
4227
+ /**
4228
+ * The ONE read-vs-run decision for a single-plugin `audit`. A plain `audit` is a
4229
+ * deterministic READ; the executing checks are opt-in via a single consent —
4230
+ * `decideExecute` resolves it (run / ask / skip). At a TTY we ASK ONCE (bundled,
4231
+ * with a confinement + cost DISCLOSURE) and remember the answer in `.vigilesrc.json`;
4232
+ * headless we stay a read (the `note` is the loud nudge, printed by the caller
4233
+ * AFTER the report). Never hangs an agent / CI run (`great-agent-flow`).
4234
+ *
4235
+ * Returns `execute` (run the executing checks?) + a `note` to print at the end.
4038
4236
  */
4039
- async function maybeSuggestTrigger(report, dir, json, args) {
4040
- const triggerable = report.skills.filter((s) => s.hasDescription && !s.userInvoked);
4041
- const decision = (0, scan_trigger_suggest_js_1.decideTriggerSuggestion)({
4042
- modelAccess: (0, scan_trigger_suggest_js_1.hasModelAccess)(process.env),
4043
- // Prompt only when BOTH streams are a terminal — `askOnce` reads stdin, so a
4044
- // TTY stdout with piped/redirected stdin (agents, shell pipelines) must take
4045
- // the non-blocking hint path, not block on a read that never gets input.
4046
- // `isTTY` is `undefined` at runtime when not a terminal (falsy → hint path).
4237
+ async function resolveExecution(s, json, args, harness) {
4238
+ const decision = (0, scan_trigger_suggest_js_1.decideExecute)({
4239
+ hasExecutable: s.hasMcp || s.triggerableSkills > 0 || s.adoptableRefs,
4240
+ // A human is "interactive" only when BOTH streams are a terminal — `askOnce`
4241
+ // reads stdin, so a TTY stdout with piped/redirected stdin (agents, shell
4242
+ // pipelines) must NOT block on a read that never gets input.
4047
4243
  isTTY: process.stdout.isTTY && process.stdin.isTTY,
4048
- triggerableSkills: triggerable.length,
4049
4244
  json,
4050
4245
  noInteractive: args.includes("--no-interactive") || args.includes("--yes"),
4246
+ remembered: (0, validate_js_1.loadConfig)().audit?.measure,
4051
4247
  });
4052
- if (decision === "none")
4053
- return;
4054
- if (decision === "hint") {
4055
- console.log("\n" + (0, scan_trigger_suggest_js_1.formatTriggerHint)(dir, triggerable.length));
4056
- return;
4248
+ if (decision.kind === "run")
4249
+ return { execute: true, note: null };
4250
+ if (decision.kind === "skip")
4251
+ return {
4252
+ execute: false,
4253
+ note: json ? null : (0, scan_trigger_suggest_js_1.formatExecuteSkip)(decision.reason),
4254
+ };
4255
+ // ask — prompt once, then remember the answer.
4256
+ const answer = await askOnce(buildExecuteDisclosure(s, harness));
4257
+ const yes = /^y(es)?$/i.test(answer); // default NO (executes your hooks / servers)
4258
+ rememberAuditMeasure(yes);
4259
+ return {
4260
+ execute: yes,
4261
+ note: yes
4262
+ ? null
4263
+ : " Skipped (remembered — edit .vigilesrc.json `audit.measure` to change).",
4264
+ };
4265
+ }
4266
+ /** The bundled consent prompt — discloses exactly what will execute (and what it
4267
+ * costs) so the yes is informed. Default NO. Harness-aware: a Codex repo measures
4268
+ * via the codex CLI (not a Claude env var), so the cost wording must not falsely
4269
+ * read "no model access" in exactly the case the prompt is meant to disclose. */
4270
+ function buildExecuteDisclosure(s, harness) {
4271
+ const lines = ["\nRun the executing checks against your harness?"];
4272
+ if (s.hasMcp)
4273
+ lines.push(" · start your MCP servers — connects to their backends");
4274
+ if (s.triggerableSkills > 0) {
4275
+ lines.push(` · measure whether skills fire (${triggerCostWording(harness)})`);
4276
+ }
4277
+ if (s.adoptableRefs) {
4278
+ lines.push(` · draft + verify your instruction file's references (${triggerCostWording(harness)})`);
4279
+ }
4280
+ lines.push("Asked once — remembered in .vigilesrc.json. [y/N] ");
4281
+ return lines.join("\n");
4282
+ }
4283
+ /** Cost/availability wording for the trigger tier, per harness. Codex runs on the
4284
+ * codex CLI (its own auth/plan), so it's never gated on a Claude env var. */
4285
+ function triggerCostWording(harness) {
4286
+ if (harness === "codex")
4287
+ return "your Codex CLI, $0 metered — skips if `codex` isn't on PATH";
4288
+ return !(0, scan_trigger_suggest_js_1.hasModelAccess)(process.env)
4289
+ ? "needs model access — none detected, will skip"
4290
+ : (0, scan_trigger_suggest_js_1.isMeteredAccess)(process.env)
4291
+ ? "⚠ spends API credits"
4292
+ : "your subscription, $0 metered";
4293
+ }
4294
+ /** Run the trigger tier: a curated `--prompts` file, else auto-generated probes. */
4295
+ async function runTriggerTier(dir, report, adapter, args) {
4296
+ if (flagValue(args, "--prompts")) {
4297
+ await handleMeasure([dir], args);
4298
+ }
4299
+ else {
4300
+ await runAutoTrigger(dir, report, adapter, args);
4057
4301
  }
4058
- await promptTriggerSetup(report, dir, triggerable.length, args);
4059
4302
  }
4060
4303
  /** Ask one question on a fresh readline, closing it after (the codebase pattern). */
4061
4304
  async function askOnce(q) {
@@ -4075,29 +4318,33 @@ async function askOnce(q) {
4075
4318
  rl.close();
4076
4319
  }
4077
4320
  }
4078
- /** Interactive (TTY) trigger-tier setup: run an existing prompts file now, or
4079
- * scaffold one to fill in. The real-model run stays an explicit confirmation so
4080
- * a plain `scan` never spends a token without a yes. */
4081
- async function promptTriggerSetup(report, dir, n, args) {
4082
- const promptsPath = (0, node_path_1.resolve)(process.cwd(), "trigger-prompts.json");
4083
- const exists = (0, node_fs_1.existsSync)(promptsPath);
4084
- const q = exists
4085
- ? `\nℹ Model access detected. Measure whether your ${String(n)} skill(s) FIRE now using trigger-prompts.json (real model)? [y/N] `
4086
- : `\nℹ ${String(n)} model-invocable skill(s) + model access detected. Scaffold trigger-prompts.json to measure firing? [y/N] `;
4087
- const answer = await askOnce(q);
4088
- if (!/^y(es)?$/i.test(answer)) {
4089
- console.log(" Skipped. " + (0, scan_trigger_suggest_js_1.formatTriggerHint)(dir, n));
4090
- return;
4321
+ /**
4322
+ * Persist the audit consent (`audit.measure`) into `.vigilesrc.json` without
4323
+ * clobbering existing keys — the "ask once, remember" sticky choice. Best-effort:
4324
+ * a malformed user config is left untouched, a write error is non-fatal (the
4325
+ * measurement already ran / was skipped; only the memory is lost).
4326
+ */
4327
+ function rememberAuditMeasure(value) {
4328
+ const configPath = (0, node_path_1.resolve)(process.cwd(), ".vigilesrc.json");
4329
+ let existing = {};
4330
+ if ((0, node_fs_1.existsSync)(configPath)) {
4331
+ try {
4332
+ existing = JSON.parse((0, node_fs_1.readFileSync)(configPath, "utf-8"));
4333
+ }
4334
+ catch {
4335
+ return; // user-owned malformed config — never clobber it
4336
+ }
4091
4337
  }
4092
- if (exists) {
4093
- await handleMeasure([dir], [...args, "--prompts=trigger-prompts.json"]);
4094
- return;
4338
+ const prevAudit = typeof existing.audit === "object" && existing.audit !== null
4339
+ ? existing.audit
4340
+ : {};
4341
+ const merged = { ...existing, audit: { ...prevAudit, measure: value } };
4342
+ try {
4343
+ (0, node_fs_1.writeFileSync)(configPath, JSON.stringify(merged, null, 2) + "\n");
4344
+ }
4345
+ catch {
4346
+ /* non-fatal — the run already happened; only the remembered choice is lost */
4095
4347
  }
4096
- (0, node_fs_1.writeFileSync)(promptsPath, (0, scan_trigger_suggest_js_1.scaffoldTriggerPrompts)(report.skills
4097
- .filter((s) => s.hasDescription && !s.userInvoked)
4098
- .map((s) => s.name)));
4099
- console.log(" ✓ Wrote trigger-prompts.json — fill in the placeholders, then run:");
4100
- console.log(` vigiles scan ${dir} --trigger --prompts=trigger-prompts.json`);
4101
4348
  }
4102
4349
  async function main() {
4103
4350
  const args = process.argv.slice(2);
@@ -4121,7 +4368,7 @@ async function main() {
4121
4368
  // still runs the full wizard (project detection + auto-targets).
4122
4369
  const hasTarget = args.some((a) => a.startsWith("--target="));
4123
4370
  if (hasTarget) {
4124
- init(args.slice(1));
4371
+ scaffoldSpec(args.slice(1));
4125
4372
  }
4126
4373
  else {
4127
4374
  await setup(args);
@@ -4184,19 +4431,15 @@ async function main() {
4184
4431
  case "eval":
4185
4432
  handleRunScripts("eval", args, restArgs);
4186
4433
  break;
4187
- case "scan": {
4188
- // Model-gated behavioral column + the deterministic diagnostic, folded into
4189
- // scan (formerly the `measure` / `explain` verbs): `--trigger` measures
4190
- // whether each skill FIRES / COLLIDES (real model), `--explain` is the
4191
- // free WHY-a-surface-underperforms + the fix.
4192
- if (args.includes("--trigger")) {
4193
- await handleMeasure(restArgs, args);
4194
- break;
4195
- }
4196
- if (args.includes("--explain")) {
4197
- handleExplain(restArgs, args);
4198
- break;
4199
- }
4434
+ case "audit": {
4435
+ // The Lighthouse run: a plain `audit` is a deterministic READ — rings, each
4436
+ // finding's fix inline, HTML/JSON report — safe + identical on every OS,
4437
+ // nothing executes. Like Lighthouse it's a LOCAL report, NOT a CI step (CI
4438
+ // uses `vigiles lint`). The executing checks (safety battery + live MCP +
4439
+ // skill firing) run only on consent: at a TTY `audit` asks once (remembered
4440
+ // in `.vigilesrc.json`), headless it stays a read + a one-line nudge. There
4441
+ // is NO execution flag — automation tests the harness via the vigiles/testing
4442
+ // API + skills, not the report verb. See the `audit-side-effect-free` rule.
4200
4443
  const dirs = restArgs.length > 0 ? restArgs : ["."];
4201
4444
  const json = args.includes("--json");
4202
4445
  // A single dir that's a marketplace (e.g. wshobson/agents' 80+ plugins
@@ -4249,19 +4492,69 @@ async function main() {
4249
4492
  }
4250
4493
  console.log("");
4251
4494
  }
4252
- if (args.includes("--fix-plan")) {
4253
- // The deterministic optimization lens on the SAME report: health score
4254
- // + the ranked free fixes to clear before measuring (the A2 spine).
4255
- const plan = (0, optimize_js_1.optimize)(report);
4256
- console.log(json ? JSON.stringify(plan, null, 2) : (0, optimize_js_1.formatOptimize)(plan));
4495
+ // The versioned AuditReport is the product boundary — the same JSON the
4496
+ // HTML renders, `--json` emits, and (later) a hosted dashboard ingests.
4497
+ // Built ONCE; the rings + fix list are read off it. Pure deterministic —
4498
+ // nothing executes to produce it.
4499
+ // Surfaces that exist but aren't spec-managed yet — the same notion
4500
+ // `init` adopts (layout-driven instruction file + skill/subagent sweep).
4501
+ // Surfaced in the AuditReport (the report's "Create spec" command-emit
4502
+ // buttons read it) and the terminal nudge below.
4503
+ const adoptableSurfaces = discoverAdoptableForAudit(root, adapter.layout.instructionFile);
4504
+ const auditReport = (0, audit_report_js_1.buildAuditReport)(report, {
4505
+ harness: adapter.name,
4506
+ vigilesVersion: getVersion(),
4507
+ adoptableSurfaces,
4508
+ });
4509
+ const sc = auditReport.score;
4510
+ const plan = (0, optimize_js_1.optimize)(report);
4511
+ if (!json) {
4512
+ // The Lighthouse rings: per-category 0–100 + the weighted overall,
4513
+ // shown before the detailed report so the headline signal leads.
4514
+ console.log((0, audit_score_js_1.formatAuditScore)(sc));
4515
+ console.log("");
4257
4516
  }
4258
- else {
4259
- console.log(json ? JSON.stringify(report, null, 2) : (0, scan_js_1.formatScanReport)(report));
4517
+ console.log(json
4518
+ ? JSON.stringify(auditReport, null, 2)
4519
+ : (0, scan_js_1.formatScanReport)(report));
4520
+ if (!json) {
4521
+ // Fold each finding's fix inline (replaces the former --fix-plan/--explain
4522
+ // flags): the deterministic, free recommendation list under the report.
4523
+ const fixes = (0, optimize_js_1.formatRecommendations)(plan);
4524
+ if (fixes)
4525
+ console.log("\n" + fixes);
4526
+ // Adoption nudge: surfaces that exist but aren't spec-managed yet, with
4527
+ // the create-all + per-surface `init` commands (the JSON carries the
4528
+ // data in `adoptable` instead — the terminal stays human-readable).
4529
+ const adoptNudge = formatAdoptableNudge(adoptableSurfaces);
4530
+ if (adoptNudge)
4531
+ console.log("\n" + adoptNudge);
4532
+ // A small behavioral nudge — the deterministic read can't tell whether
4533
+ // skills actually FIRE; point at the interactive measure + the API.
4534
+ const fireNudge = formatTriggerNudge(report.skills.filter((s) => s.hasDescription && !s.userInvoked)
4535
+ .length);
4536
+ if (fireNudge)
4537
+ console.log("\n" + fireNudge);
4260
4538
  }
4261
- if (args.includes("--verify-mcp")) {
4262
- // Opt-in LIVE MCP tool resolution: starts each declared server and checks
4263
- // the agent's mcp__server__tool refs actually exist (the dynamic check no
4264
- // static linter can do). Side-effecting (spawns servers) → opt-in only.
4539
+ // ONE read-vs-run decision for the EXECUTING checks (live MCP + skill
4540
+ // firing). A plain `audit` is a deterministic READ; these run only on
4541
+ // consent — ASK once at a TTY (remembered); headless stays a read + a
4542
+ // nudge (no execution flag — automation uses the vigiles/testing API).
4543
+ // (The safety battery is NOT here — it needs cross-platform confinement
4544
+ // that isn't shipped, so it lives in the vigiles/testing API.)
4545
+ const isForeign = root !== process.cwd();
4546
+ const surfaces = {
4547
+ hasMcp: report.mcp && !isForeign,
4548
+ triggerableSkills: report.skills.filter((s) => s.hasDescription && !s.userInvoked).length,
4549
+ adoptableRefs: adapter.name === "claude-code" &&
4550
+ (0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, adapter.layout.instructionFile)),
4551
+ };
4552
+ const { execute, note: execNote } = await resolveExecution(surfaces, json, args, adapter.name);
4553
+ // LIVE MCP tool resolution STARTS each declared MCP server — a server is
4554
+ // exactly what connects to a real Postgres / authenticates a real API on
4555
+ // boot. So it runs only under consent (`execute`) AND own-repo (never
4556
+ // spawn a stranger's server).
4557
+ if (execute && surfaces.hasMcp) {
4265
4558
  const mcpErrs = await (0, scan_js_1.verifyLiveMcpTools)(report, adapter.layout, adapter.dialect);
4266
4559
  console.log(json
4267
4560
  ? JSON.stringify({ mcpContractTools: mcpErrs }, null, 2)
@@ -4280,9 +4573,76 @@ async function main() {
4280
4573
  if (args.includes("--fail-on-widen") && diff.widened)
4281
4574
  process.exitCode = 1;
4282
4575
  }
4283
- // Nudge toward the real-model trigger tier when a model is reachable and
4284
- // the plugin ships model-invocable skills (hint for agents, offer for humans).
4285
- await maybeSuggestTrigger(report, targets[0], json, args);
4576
+ // The model trigger tier — "do your skills actually FIRE?" — runs as part
4577
+ // of the same consent (`execute`), and only when a model is reachable;
4578
+ // otherwise it's a one-line note (never a hang).
4579
+ if (execute && surfaces.triggerableSkills > 0) {
4580
+ // Model-access detection is per-harness: `hasModelAccess` reads Claude
4581
+ // env (claude CLI / ANTHROPIC_API_KEY). A Codex repo authenticates the
4582
+ // codex CLI instead, so we DON'T gate it on a Claude var — the Codex
4583
+ // probe checks `codexDriver.available()` internally and self-reports
4584
+ // unavailable. (harness-parity: never block Codex behind a CC check.)
4585
+ const modelReachable = adapter.name === "codex" || (0, scan_trigger_suggest_js_1.hasModelAccess)(process.env);
4586
+ if (modelReachable) {
4587
+ await runTriggerTier(targets[0], report, adapter, args);
4588
+ }
4589
+ else if (!json) {
4590
+ console.log("\nℹ skill firing not measured — no model access (authenticate the `claude` CLI or set ANTHROPIC_API_KEY).");
4591
+ }
4592
+ }
4593
+ // The adoption preview — "what would vigiles catch in YOUR repo?" The model
4594
+ // DRAFTS the verifiable refs in the instruction file; the cross-ref engine
4595
+ // VERIFIES each, so "M broken right now" is trustworthy though the extraction
4596
+ // is probabilistic. Same consent as the trigger tier (`surfaces.adoptableRefs`
4597
+ // makes a bare instruction-file repo consent-eligible — the prime adoption
4598
+ // target). v1: instruction-file only; drafting drives the `claude` CLI, so
4599
+ // `adoptableRefs` is Claude Code only (the Codex deferral note is printed
4600
+ // below as a LOUD harness-parity deferral, never a silent CC-only path).
4601
+ let adoptabilityResult;
4602
+ if (execute && surfaces.adoptableRefs) {
4603
+ if ((0, scan_trigger_suggest_js_1.hasModelAccess)(process.env)) {
4604
+ const instrPath = (0, node_path_1.resolve)(root, adapter.layout.instructionFile);
4605
+ adoptabilityResult = await (0, adoptability_js_1.runAdoptabilityTier)({
4606
+ instructionContent: (0, node_fs_1.readFileSync)(instrPath, "utf-8"),
4607
+ basePath: root,
4608
+ });
4609
+ if (!json)
4610
+ console.log("\n" +
4611
+ (0, adoptability_js_1.formatAdoptability)(adoptabilityResult, adapter.layout.instructionFile));
4612
+ }
4613
+ else if (!json && surfaces.triggerableSkills === 0) {
4614
+ // Only when the trigger tier didn't already print the same note.
4615
+ console.log("\nℹ adoptability not measured — no model access (authenticate the `claude` CLI or set ANTHROPIC_API_KEY).");
4616
+ }
4617
+ }
4618
+ // LOUD harness-parity deferral: adoptability drafting drives the `claude`
4619
+ // CLI, so a Codex repo with an instruction file is told it's a follow-up,
4620
+ // never silently skipped (research/adoption-gateway-preview.md, increment 4).
4621
+ if (!json &&
4622
+ adapter.name === "codex" &&
4623
+ (0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, adapter.layout.instructionFile))) {
4624
+ console.log("\nℹ adoptability preview (what vigiles would catch in your repo) is Claude Code only for now — Codex support is a follow-up.");
4625
+ }
4626
+ // The loud read-vs-run nudge for a headless / remembered-no skip — printed
4627
+ // AFTER the report so the deterministic read leads.
4628
+ if (execNote)
4629
+ console.log(execNote);
4630
+ // The final report folds in the adoptability preview (when the model-gated
4631
+ // tier ran) so the written HTML/JSON carry it; a deterministic read omits it.
4632
+ const finalReport = adoptabilityResult
4633
+ ? { ...auditReport, adoptability: adoptabilityResult }
4634
+ : auditReport;
4635
+ // The shareable HTML report — written by default (--no-html to skip), and
4636
+ // opened best-effort only for a human at a TTY (never spawn a browser for
4637
+ // an agent / CI run).
4638
+ if (!json && !args.includes("--no-html")) {
4639
+ writeAuditHtml(finalReport);
4640
+ }
4641
+ // The versioned JSON artifact — the upload/CI boundary (a hosted dashboard
4642
+ // ingests this). Written by default in the human path; --no-json to skip.
4643
+ if (!json && !args.includes("--no-json")) {
4644
+ writeAuditJson(finalReport);
4645
+ }
4286
4646
  }
4287
4647
  break;
4288
4648
  }