vigiles 9.0.0 → 10.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +144 -222
- package/dist/audit-html.d.ts +15 -4
- package/dist/audit-html.js +15 -6
- package/dist/audit-report.d.ts +94 -1
- package/dist/audit-report.js +50 -0
- package/dist/audit-report.template.html +47 -22
- package/dist/audit-score.d.ts +27 -7
- package/dist/audit-score.js +63 -50
- package/dist/audit-serve.d.ts +109 -0
- package/dist/audit-serve.js +257 -0
- package/dist/cli.js +322 -25
- package/dist/core/adopt.d.ts +28 -0
- package/dist/core/adopt.js +203 -0
- package/dist/core/compile.d.ts +5 -1
- package/dist/core/compile.js +19 -10
- package/dist/leaderboard.d.ts +32 -0
- package/dist/leaderboard.js +109 -50
- package/dist/scan-behavioral.d.ts +85 -0
- package/dist/scan-behavioral.js +225 -0
- package/dist/scan.js +23 -1
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -34,6 +34,7 @@ const optimize_js_1 = require("./optimize.js");
|
|
|
34
34
|
const audit_score_js_1 = require("./audit-score.js");
|
|
35
35
|
const audit_prompts_js_1 = require("./audit-prompts.js");
|
|
36
36
|
const audit_html_js_1 = require("./audit-html.js");
|
|
37
|
+
const audit_serve_js_1 = require("./audit-serve.js");
|
|
37
38
|
const audit_report_js_1 = require("./audit-report.js");
|
|
38
39
|
const adoptability_js_1 = require("./adoptability.js");
|
|
39
40
|
const compile_js_1 = require("./core/compile.js");
|
|
@@ -140,6 +141,14 @@ function printErrors(specFile, errors) {
|
|
|
140
141
|
console.log(`::error file=${specFile}::${err.message}`);
|
|
141
142
|
}
|
|
142
143
|
}
|
|
144
|
+
/** Non-blocking advisories — printed, but never fail the compile. */
|
|
145
|
+
function printWarnings(specFile, warnings) {
|
|
146
|
+
for (const w of warnings) {
|
|
147
|
+
const pathInfo = w.path ? ` (${w.path})` : "";
|
|
148
|
+
console.log(` ⚠ [${w.type}] ${w.message}${pathInfo}`);
|
|
149
|
+
console.log(`::warning file=${specFile}::${w.message}`);
|
|
150
|
+
}
|
|
151
|
+
}
|
|
143
152
|
// ---------------------------------------------------------------------------
|
|
144
153
|
// Commands
|
|
145
154
|
// ---------------------------------------------------------------------------
|
|
@@ -238,7 +247,7 @@ function writeInstructionMirrors(primaryOutput, harnesses) {
|
|
|
238
247
|
/** Compile a declarative SkillSpec → SKILL.md. */
|
|
239
248
|
function compileSkillToFile(spec, specPath, dialect) {
|
|
240
249
|
const outputPath = specPath.replace(/\.spec\.ts$/, "");
|
|
241
|
-
const { markdown, errors } = (0, compile_js_1.compileSkill)(spec, {
|
|
250
|
+
const { markdown, errors, warnings } = (0, compile_js_1.compileSkill)(spec, {
|
|
242
251
|
basePath: process.cwd(),
|
|
243
252
|
specFile: specPath,
|
|
244
253
|
// The SKILL.md frontmatter profile comes from the resolved harness — a Codex
|
|
@@ -248,16 +257,18 @@ function compileSkillToFile(spec, specPath, dialect) {
|
|
|
248
257
|
(0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
|
|
249
258
|
if (errors.length === 0) {
|
|
250
259
|
console.log(`\n✓ ${specPath} → ${outputPath}`);
|
|
260
|
+
printWarnings(specPath, warnings);
|
|
251
261
|
return true;
|
|
252
262
|
}
|
|
253
263
|
console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
|
|
254
264
|
printErrors(specPath, errors);
|
|
265
|
+
printWarnings(specPath, warnings);
|
|
255
266
|
return false;
|
|
256
267
|
}
|
|
257
268
|
/** Compile a subagent spec → agents/<name>.md (with its result-contract section). */
|
|
258
269
|
function compileAgentToFile(spec, specPath, dialect) {
|
|
259
270
|
const outputPath = specPath.replace(/\.spec\.ts$/, "");
|
|
260
|
-
const { markdown, errors } = (0, compile_js_1.compileAgent)(spec, {
|
|
271
|
+
const { markdown, errors, warnings } = (0, compile_js_1.compileAgent)(spec, {
|
|
261
272
|
basePath: process.cwd(),
|
|
262
273
|
specFile: specPath,
|
|
263
274
|
dialect,
|
|
@@ -265,10 +276,12 @@ function compileAgentToFile(spec, specPath, dialect) {
|
|
|
265
276
|
(0, node_fs_1.writeFileSync)((0, node_path_1.resolve)(process.cwd(), outputPath), markdown);
|
|
266
277
|
if (errors.length === 0) {
|
|
267
278
|
console.log(`\n✓ ${specPath} → ${outputPath}`);
|
|
279
|
+
printWarnings(specPath, warnings);
|
|
268
280
|
return true;
|
|
269
281
|
}
|
|
270
282
|
console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
|
|
271
283
|
printErrors(specPath, errors);
|
|
284
|
+
printWarnings(specPath, warnings);
|
|
272
285
|
return false;
|
|
273
286
|
}
|
|
274
287
|
/**
|
|
@@ -819,8 +832,26 @@ function verifyFrontmatterRules(filePath, silent, exclude, linterOptions) {
|
|
|
819
832
|
};
|
|
820
833
|
}
|
|
821
834
|
/**
|
|
822
|
-
*
|
|
823
|
-
*
|
|
835
|
+
* Frontmatter mode (Level 1 — a `vigiles:` YAML block) is DISABLED in lint:
|
|
836
|
+
* KEPT IN CODE (`src/core/frontmatter.ts`, `verifyFrontmatterRules`,
|
|
837
|
+
* `vigiles generate schema`), but INERT — lint no longer reads or verifies a
|
|
838
|
+
* `vigiles:` block, so it never fires and never fails a build.
|
|
839
|
+
*
|
|
840
|
+
* WHY disabled-not-removed: the three-rung adoption ladder (inline / frontmatter
|
|
841
|
+
* / typed spec) collapsed to TWO on-ramps — inline comments (the zero-TS floor)
|
|
842
|
+
* and the typed `.spec.ts` (the source of truth). Frontmatter mode was the
|
|
843
|
+
* weakest middle rung and an undocumented-but-live surface that muddied the
|
|
844
|
+
* spec-first story (it literally confused a review). With ~no users to break,
|
|
845
|
+
* gating it off makes lint coherent (verify compiled output + inline marks +
|
|
846
|
+
* specs, nothing else) while preserving the code so the decision is reversible:
|
|
847
|
+
* flip this to `true` to re-enable. See `research/pre-release-focus.md` and the
|
|
848
|
+
* parked note in `docs/markdown-mode.md`.
|
|
849
|
+
*/
|
|
850
|
+
const FRONTMATTER_MODE_ENABLED = false;
|
|
851
|
+
/**
|
|
852
|
+
* Verify inline `<!-- vigiles:enforce -->` comments (and, when
|
|
853
|
+
* {@link FRONTMATTER_MODE_ENABLED}, `vigiles:` YAML frontmatter) in instruction
|
|
854
|
+
* files that aren't managed by a spec.
|
|
824
855
|
*
|
|
825
856
|
* Spec mode is the source of truth when it exists, so a literal
|
|
826
857
|
* `<!-- vigiles:enforce ... -->` snippet that survived into compiled
|
|
@@ -861,9 +892,13 @@ function verifyMarkdownModeRules(files, silent, config) {
|
|
|
861
892
|
const inline = verifyInlineRules(filePath, silent, linterOptions);
|
|
862
893
|
totals.inlineErrors += inline.errorCount;
|
|
863
894
|
totals.inlineRules += inline.ruleCount;
|
|
864
|
-
|
|
865
|
-
|
|
866
|
-
|
|
895
|
+
// Frontmatter mode is DISABLED (kept in code, inert in lint) — a `vigiles:`
|
|
896
|
+
// block is ignored, never verified. See FRONTMATTER_MODE_ENABLED.
|
|
897
|
+
if (FRONTMATTER_MODE_ENABLED) {
|
|
898
|
+
const fm = verifyFrontmatterRules(filePath, silent, new Set(inline.ruleNames), linterOptions);
|
|
899
|
+
totals.frontmatterErrors += fm.errorCount;
|
|
900
|
+
totals.frontmatterRules += fm.ruleCount;
|
|
901
|
+
}
|
|
867
902
|
}
|
|
868
903
|
if (!silent &&
|
|
869
904
|
files.length > 0 &&
|
|
@@ -1219,6 +1254,100 @@ function targetHasHash(absPath) {
|
|
|
1219
1254
|
* target by `setupPillar1`. The full onboarding (both layers, deps, CI, plugin)
|
|
1220
1255
|
* is `setup()`.
|
|
1221
1256
|
*/
|
|
1257
|
+
/** Classify an adoption target by its path: a `SKILL.md` is a skill, a file under
|
|
1258
|
+
* an `agents/` dir is a subagent, everything else is an instruction file. Used to
|
|
1259
|
+
* pick the right adopt function so `init --target=skills/x/SKILL.md` (the
|
|
1260
|
+
* per-surface path the audit report points at) makes a `skill()`/`agent()` spec. */
|
|
1261
|
+
function surfaceKind(target) {
|
|
1262
|
+
if (/^SKILL\.md$/i.test((0, node_path_1.basename)(target)))
|
|
1263
|
+
return "skill";
|
|
1264
|
+
if (/(^|[/\\])agents[/\\]/.test(target))
|
|
1265
|
+
return "agent";
|
|
1266
|
+
return "instruction";
|
|
1267
|
+
}
|
|
1268
|
+
function logAdoptedSurface(target, specPath, label, unmappedKeys) {
|
|
1269
|
+
const note = unmappedKeys.length > 0
|
|
1270
|
+
? ` (review the // NOTE — unmapped frontmatter: ${unmappedKeys.join(", ")})`
|
|
1271
|
+
: "";
|
|
1272
|
+
console.log(`Adopted ${label} ${target} → ${specPath}${note}. ` +
|
|
1273
|
+
`Run \`vigiles compile\` and review the diff.`);
|
|
1274
|
+
}
|
|
1275
|
+
/** Discover existing skill (`skills/<x>/SKILL.md`) and subagent (`agents/<x>.md`)
|
|
1276
|
+
* surfaces — under the bare or `.claude/` roots — that don't yet have a spec, so
|
|
1277
|
+
* bare `vigiles init` creates a spec for EVERY surface it can, not just the
|
|
1278
|
+
* instruction file. Shallow (top-level only) so it never walks node_modules or a
|
|
1279
|
+
* vendored plugin. CC paths are intentional here — `init` is the one composition
|
|
1280
|
+
* point allowed to know them (see adapter-aware-lint-rules). */
|
|
1281
|
+
function discoverAdoptableSurfaces(cwd) {
|
|
1282
|
+
const out = [];
|
|
1283
|
+
const unspecced = (rel) => (0, node_fs_1.existsSync)((0, node_path_1.resolve)(cwd, rel)) &&
|
|
1284
|
+
!(0, node_fs_1.existsSync)((0, node_path_1.resolve)(cwd, `${rel}.spec.ts`));
|
|
1285
|
+
for (const root of ["skills", ".claude/skills"]) {
|
|
1286
|
+
const abs = (0, node_path_1.resolve)(cwd, root);
|
|
1287
|
+
if (!(0, node_fs_1.existsSync)(abs))
|
|
1288
|
+
continue;
|
|
1289
|
+
for (const e of (0, node_fs_1.readdirSync)(abs, { withFileTypes: true })) {
|
|
1290
|
+
const rel = `${root}/${e.name}/SKILL.md`;
|
|
1291
|
+
if (e.isDirectory() && unspecced(rel))
|
|
1292
|
+
out.push(rel);
|
|
1293
|
+
}
|
|
1294
|
+
}
|
|
1295
|
+
for (const root of ["agents", ".claude/agents"]) {
|
|
1296
|
+
const abs = (0, node_path_1.resolve)(cwd, root);
|
|
1297
|
+
if (!(0, node_fs_1.existsSync)(abs))
|
|
1298
|
+
continue;
|
|
1299
|
+
for (const e of (0, node_fs_1.readdirSync)(abs, { withFileTypes: true })) {
|
|
1300
|
+
const rel = `${root}/${e.name}`;
|
|
1301
|
+
if (e.isFile() && e.name.endsWith(".md") && unspecced(rel))
|
|
1302
|
+
out.push(rel);
|
|
1303
|
+
}
|
|
1304
|
+
}
|
|
1305
|
+
return out;
|
|
1306
|
+
}
|
|
1307
|
+
/** The full adoptable-surface list `audit` reports: the instruction file (when it
|
|
1308
|
+
* exists hand-written, no spec) PLUS every skill/subagent surface without a spec.
|
|
1309
|
+
* Same notion `init` adopts; surfaced in the AuditReport + the terminal nudge so
|
|
1310
|
+
* the report's "Create spec" / "Create all specs" affordances have their paths.
|
|
1311
|
+
* Composition-root only — CC paths are intentional here (like discoverAdoptableSurfaces). */
|
|
1312
|
+
function discoverAdoptableForAudit(root, instructionFile) {
|
|
1313
|
+
const out = [];
|
|
1314
|
+
const instrAbs = (0, node_path_1.resolve)(root, instructionFile);
|
|
1315
|
+
if ((0, node_fs_1.existsSync)(instrAbs) &&
|
|
1316
|
+
!targetHasHash(instrAbs) &&
|
|
1317
|
+
!(0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, `${instructionFile}.spec.ts`))) {
|
|
1318
|
+
out.push(instructionFile);
|
|
1319
|
+
}
|
|
1320
|
+
out.push(...discoverAdoptableSurfaces(root));
|
|
1321
|
+
return out;
|
|
1322
|
+
}
|
|
1323
|
+
/** The terminal "adoptable surfaces" nudge — N un-spec'd surfaces + the create-all
|
|
1324
|
+
* command and up to ~5 per-surface commands (then "+K more"). "" when nothing to
|
|
1325
|
+
* adopt (a fully spec-managed repo says nothing). */
|
|
1326
|
+
function formatAdoptableNudge(surfaces) {
|
|
1327
|
+
if (surfaces.length === 0)
|
|
1328
|
+
return "";
|
|
1329
|
+
const n = surfaces.length;
|
|
1330
|
+
const lines = [
|
|
1331
|
+
`ℹ ${String(n)} surface${n === 1 ? "" : "s"} not yet spec-managed — create specs with \`npx vigiles init\``,
|
|
1332
|
+
` (or one at a time: \`npx vigiles init --target=<path>\`)`,
|
|
1333
|
+
];
|
|
1334
|
+
const shown = surfaces.slice(0, 5);
|
|
1335
|
+
for (const s of shown)
|
|
1336
|
+
lines.push(` • npx vigiles init --target=${s}`);
|
|
1337
|
+
const more = n - shown.length;
|
|
1338
|
+
if (more > 0)
|
|
1339
|
+
lines.push(` • +${String(more)} more`);
|
|
1340
|
+
return lines.join("\n");
|
|
1341
|
+
}
|
|
1342
|
+
/** A small, terse behavioral nudge — the deterministic read can't tell whether a
|
|
1343
|
+
* skill actually FIRES. "" when there are no model-invocable skills. */
|
|
1344
|
+
function formatTriggerNudge(triggerableSkills) {
|
|
1345
|
+
if (triggerableSkills <= 0)
|
|
1346
|
+
return "";
|
|
1347
|
+
const n = triggerableSkills;
|
|
1348
|
+
return (`ℹ Do your ${String(n)} skill${n === 1 ? "" : "s"} actually fire? The deterministic read can't tell — ` +
|
|
1349
|
+
`run \`audit\` interactively to measure, or test with \`measureTriggerRate\` (vigiles/testing).`);
|
|
1350
|
+
}
|
|
1222
1351
|
function scaffoldSpec(args) {
|
|
1223
1352
|
const targetFlag = args.find((a) => a.startsWith("--target="));
|
|
1224
1353
|
const target = targetFlag ? targetFlag.split("=")[1] : "CLAUDE.md";
|
|
@@ -1237,11 +1366,24 @@ function scaffoldSpec(args) {
|
|
|
1237
1366
|
const targetAbs = (0, node_path_1.resolve)(process.cwd(), target);
|
|
1238
1367
|
if ((0, node_fs_1.existsSync)(targetAbs) && !targetHasHash(targetAbs)) {
|
|
1239
1368
|
const md = (0, node_fs_1.readFileSync)(targetAbs, "utf-8");
|
|
1240
|
-
const { source, tier, sectionCount } = (0, adopt_js_1.adoptMarkdown)(md, (0, node_path_1.basename)(target));
|
|
1241
1369
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(specAbs), { recursive: true });
|
|
1242
|
-
|
|
1243
|
-
|
|
1244
|
-
|
|
1370
|
+
const kind = surfaceKind(target);
|
|
1371
|
+
if (kind === "skill") {
|
|
1372
|
+
const { source, unmappedKeys } = (0, adopt_js_1.adoptSkill)(md, (0, node_path_1.basename)((0, node_path_1.dirname)(target)));
|
|
1373
|
+
(0, node_fs_1.writeFileSync)(specAbs, source);
|
|
1374
|
+
logAdoptedSurface(target, specPath, "skill", unmappedKeys);
|
|
1375
|
+
}
|
|
1376
|
+
else if (kind === "agent") {
|
|
1377
|
+
const { source, unmappedKeys } = (0, adopt_js_1.adoptAgent)(md, (0, node_path_1.basename)(target, ".md"));
|
|
1378
|
+
(0, node_fs_1.writeFileSync)(specAbs, source);
|
|
1379
|
+
logAdoptedSurface(target, specPath, "subagent", unmappedKeys);
|
|
1380
|
+
}
|
|
1381
|
+
else {
|
|
1382
|
+
const { source, tier, sectionCount } = (0, adopt_js_1.adoptMarkdown)(md, (0, node_path_1.basename)(target));
|
|
1383
|
+
(0, node_fs_1.writeFileSync)(specAbs, source);
|
|
1384
|
+
console.log(`Adopted ${target} → ${specPath} (${tier}, ${String(sectionCount)} section${sectionCount === 1 ? "" : "s"}). ` +
|
|
1385
|
+
`Run \`vigiles compile\` and review the diff; the \`/strengthen\` skill upgrades prose to verified rules.`);
|
|
1386
|
+
}
|
|
1245
1387
|
return;
|
|
1246
1388
|
}
|
|
1247
1389
|
// The compiled output is derived from the spec FILE path; the spec's `target`
|
|
@@ -1768,9 +1910,15 @@ async function setupPillar1(detected, targetValue, harnesses) {
|
|
|
1768
1910
|
// An explicit --target is honoured as-is; otherwise collapse a CLAUDE.md⇄
|
|
1769
1911
|
// AGENTS.md mirror (symlink or synced) to one canonical spec, then redirect
|
|
1770
1912
|
// into a sync tool's source slot when one would own the output.
|
|
1771
|
-
const
|
|
1913
|
+
const instructionTargets = targetValue
|
|
1772
1914
|
? determineTargets(detected, targetValue, harnesses)
|
|
1773
1915
|
: redirectSyncToolTargets(cwd, collapseMirroredTargets(determineTargets(detected, targetValue, harnesses), (0, compose_js_1.detectInstructionMirror)(cwd)));
|
|
1916
|
+
// Bare `init` (no explicit --target) also adopts every existing skill +
|
|
1917
|
+
// subagent surface — "create all the specs it can", not just the instruction
|
|
1918
|
+
// file. An explicit --target stays scoped to that one surface.
|
|
1919
|
+
const targets = targetValue
|
|
1920
|
+
? instructionTargets
|
|
1921
|
+
: [...instructionTargets, ...discoverAdoptableSurfaces(cwd)];
|
|
1774
1922
|
// Create specs. An existing hand-written target is faithfully ADOPTED into a
|
|
1775
1923
|
// spec (scaffoldSpec() does the convert), not clobbered with a blank one — so the
|
|
1776
1924
|
// compile below reproduces it (the user reviews the diff). A greenfield target
|
|
@@ -3217,6 +3365,7 @@ function printUsage(command) {
|
|
|
3217
3365
|
console.log(" vigiles audit [dir...] Lighthouse for your harness — a LOCAL report: rings + what's broken + fixes (a deterministic read; 2+ dirs → leaderboard)");
|
|
3218
3366
|
console.log(" writes vigiles-report.html + vigiles-report.json (--no-html/--no-json) · --json for machine output. NOT a CI step — use `vigiles lint` in CI.");
|
|
3219
3367
|
console.log(" the executing checks (run your hooks · live MCP · do skills fire?) run only interactively — `audit` asks once (remembered); automation uses the vigiles/testing API");
|
|
3368
|
+
console.log(" --serve opens a LIVE local report whose buttons create specs in one click (own repo only; loopback + token-guarded) · --no-serve to skip the prompt");
|
|
3220
3369
|
console.log(" vigiles test [files...] Run *.harness.mjs deterministic harness tests");
|
|
3221
3370
|
console.log(" vigiles eval [files...] Run *.eval.mjs real-model harness evals (--trials=N, --min=N, --no-skip)");
|
|
3222
3371
|
console.log(" vigiles scaffold-test [dir] Generate a starter test for each untested skill/agent/hook (--write, --json)");
|
|
@@ -4073,6 +4222,63 @@ function writeAuditHtml(report) {
|
|
|
4073
4222
|
console.log(`\n⚠ skipped vigiles-report.html: ${e instanceof Error ? e.message : String(e)}`);
|
|
4074
4223
|
}
|
|
4075
4224
|
}
|
|
4225
|
+
/**
|
|
4226
|
+
* Start the live (`--serve`) adoption server: render the report with a per-run
|
|
4227
|
+
* token, serve it on loopback, and run `init` in-process when a button POSTs. The
|
|
4228
|
+
* security model lives in src/audit-serve.ts (token + Origin + allowlist). Blocks
|
|
4229
|
+
* until the user stops it (Ctrl-C or the page's Done). Own-repo only — the caller
|
|
4230
|
+
* gates this via decideServeGate, so adopt always writes into the current repo.
|
|
4231
|
+
*/
|
|
4232
|
+
async function runAuditServe(report, adoptable, cliErr) {
|
|
4233
|
+
const token = (0, audit_serve_js_1.newToken)();
|
|
4234
|
+
const surfaces = new Set((adoptable?.surfaces ?? []).map((s) => s.path));
|
|
4235
|
+
let html;
|
|
4236
|
+
try {
|
|
4237
|
+
html = (0, audit_html_js_1.renderAuditHtml)(report, { token });
|
|
4238
|
+
}
|
|
4239
|
+
catch (e) {
|
|
4240
|
+
console.log(`\n⚠ can't serve the live report: ${cliErr(e)}`);
|
|
4241
|
+
return;
|
|
4242
|
+
}
|
|
4243
|
+
const adoptOne = (target) => {
|
|
4244
|
+
try {
|
|
4245
|
+
scaffoldSpec(["--target=" + target]); // in-process; writes into cwd (own repo)
|
|
4246
|
+
return Promise.resolve({
|
|
4247
|
+
ok: true,
|
|
4248
|
+
message: `created spec for ${target}`,
|
|
4249
|
+
});
|
|
4250
|
+
}
|
|
4251
|
+
catch (e) {
|
|
4252
|
+
return Promise.resolve({ ok: false, message: cliErr(e) });
|
|
4253
|
+
}
|
|
4254
|
+
};
|
|
4255
|
+
console.log("\n Live report — create specs with one click. Ctrl-C to stop.\n");
|
|
4256
|
+
await (0, audit_serve_js_1.serveAudit)({
|
|
4257
|
+
token,
|
|
4258
|
+
surfaces,
|
|
4259
|
+
html,
|
|
4260
|
+
runAdopt: adoptOne,
|
|
4261
|
+
runAdoptAll: () => {
|
|
4262
|
+
try {
|
|
4263
|
+
for (const p of surfaces)
|
|
4264
|
+
scaffoldSpec(["--target=" + p]);
|
|
4265
|
+
return Promise.resolve({
|
|
4266
|
+
ok: true,
|
|
4267
|
+
message: `created ${String(surfaces.size)} spec(s)`,
|
|
4268
|
+
});
|
|
4269
|
+
}
|
|
4270
|
+
catch (e) {
|
|
4271
|
+
return Promise.resolve({ ok: false, message: cliErr(e) });
|
|
4272
|
+
}
|
|
4273
|
+
},
|
|
4274
|
+
onListening: (url) => {
|
|
4275
|
+
console.log(` ${url}`);
|
|
4276
|
+
if (process.stdout.isTTY)
|
|
4277
|
+
openBestEffort(url);
|
|
4278
|
+
},
|
|
4279
|
+
});
|
|
4280
|
+
console.log("\n✓ live report closed");
|
|
4281
|
+
}
|
|
4076
4282
|
/**
|
|
4077
4283
|
* Run the model trigger tier with no `--prompts`: auto-generate diverse probe
|
|
4078
4284
|
* prompts from each skill's description and measure trigger-rate (recall +
|
|
@@ -4097,19 +4303,54 @@ async function runAutoTrigger(dir, report, adapter, args) {
|
|
|
4097
4303
|
if (!json) {
|
|
4098
4304
|
console.log("\nℹ auto-generated probe prompts from skill descriptions (pass --prompts=<file> for a curated set).");
|
|
4099
4305
|
}
|
|
4306
|
+
const model = flagValue(args, "--model");
|
|
4100
4307
|
const trigger = await (0, scan_behavioral_js_1.probePluginTriggers)(dir, promptSet, {
|
|
4101
4308
|
minPrompts: audit_prompts_js_1.AUTO_RECALL_COUNT,
|
|
4102
4309
|
minDistance: audit_prompts_js_1.AUTO_MIN_DISTANCE,
|
|
4103
|
-
model
|
|
4310
|
+
model,
|
|
4104
4311
|
harness,
|
|
4105
4312
|
// Discover candidates with the resolved adapter's layout/dialect — a Codex
|
|
4106
4313
|
// repo's skills live under the Codex layout, not the default CC one.
|
|
4107
4314
|
layout: adapter.layout,
|
|
4108
4315
|
dialect: adapter.dialect,
|
|
4109
4316
|
});
|
|
4110
|
-
|
|
4111
|
-
|
|
4112
|
-
|
|
4317
|
+
// Second behavioral eval (same consent): the selection-collision matrix — does
|
|
4318
|
+
// one skill HIJACK a sibling's prompt? This is the MEASURED confirmation of the
|
|
4319
|
+
// deterministic description-overlap proxy (the Triggering ring flags look-alikes;
|
|
4320
|
+
// this proves the wrong one actually fires). Only meaningful with ≥2 model-
|
|
4321
|
+
// invocable skills (a lone skill can't collide); reuses the same auto prompts.
|
|
4322
|
+
const collisions = skills.length >= 2
|
|
4323
|
+
? await (0, scan_behavioral_js_1.measurePluginSelection)(dir, promptSet, { model, harness })
|
|
4324
|
+
: null;
|
|
4325
|
+
// Third behavioral eval (same consent): adversarial-gate — do enforcement-gate
|
|
4326
|
+
// skills HOLD when the agent is told to violate them? Auto-derives its own
|
|
4327
|
+
// attacks; a no-op (no model calls) when the plugin declares no gate skills.
|
|
4328
|
+
const gates = await (0, scan_behavioral_js_1.measureGateAdversarial)(dir, {
|
|
4329
|
+
model,
|
|
4330
|
+
harness,
|
|
4331
|
+
layout: adapter.layout,
|
|
4332
|
+
dialect: adapter.dialect,
|
|
4333
|
+
});
|
|
4334
|
+
// Show the gate section when gate skills were DETECTED — even if the eval
|
|
4335
|
+
// couldn't RUN (a Codex audit, or no `claude` CLI) it returns available:false
|
|
4336
|
+
// with empty results, and `formatGateReport` renders the "unavailable" note.
|
|
4337
|
+
// The consent prompt already advertised these gate skills, so a skipped check
|
|
4338
|
+
// must be reported LOUDLY, never silently omitted as if there were none.
|
|
4339
|
+
const hasGates = gates.results.length > 0 || (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length > 0;
|
|
4340
|
+
if (json) {
|
|
4341
|
+
console.log(JSON.stringify({
|
|
4342
|
+
trigger,
|
|
4343
|
+
...(collisions ? { collisions } : {}),
|
|
4344
|
+
...(hasGates ? { gates } : {}),
|
|
4345
|
+
}, null, 2));
|
|
4346
|
+
}
|
|
4347
|
+
else {
|
|
4348
|
+
console.log("\n" + (0, scan_behavioral_js_1.formatBehavioralReport)(trigger));
|
|
4349
|
+
if (collisions)
|
|
4350
|
+
console.log("\n" + (0, scan_behavioral_js_1.formatSelectionReport)(collisions));
|
|
4351
|
+
if (hasGates)
|
|
4352
|
+
console.log("\n" + (0, scan_behavioral_js_1.formatGateReport)(gates));
|
|
4353
|
+
}
|
|
4113
4354
|
}
|
|
4114
4355
|
/**
|
|
4115
4356
|
* The ONE read-vs-run decision for a single-plugin `audit`. A plain `audit` is a
|
|
@@ -4159,7 +4400,17 @@ function buildExecuteDisclosure(s, harness) {
|
|
|
4159
4400
|
if (s.hasMcp)
|
|
4160
4401
|
lines.push(" · start your MCP servers — connects to their backends");
|
|
4161
4402
|
if (s.triggerableSkills > 0) {
|
|
4162
|
-
|
|
4403
|
+
// ≥2 model-invocable skills also get the selection-collision matrix (does one
|
|
4404
|
+
// skill hijack a sibling's prompt) — disclose it so the consent stays honest.
|
|
4405
|
+
const what = s.triggerableSkills >= 2
|
|
4406
|
+
? "measure whether skills fire and collide"
|
|
4407
|
+
: "measure whether skills fire";
|
|
4408
|
+
lines.push(` · ${what} (${triggerCostWording(harness)})`);
|
|
4409
|
+
}
|
|
4410
|
+
if (s.gateSkills > 0) {
|
|
4411
|
+
// The adversarial-gate eval runs the FULL (unstubbed) skill — the most
|
|
4412
|
+
// expensive check — so disclose it separately when gate skills are present.
|
|
4413
|
+
lines.push(` · test whether ${String(s.gateSkills)} enforcement-gate skill${s.gateSkills === 1 ? "" : "s"} hold under pressure — runs the full skill (${triggerCostWording(harness)})`);
|
|
4163
4414
|
}
|
|
4164
4415
|
if (s.adoptableRefs) {
|
|
4165
4416
|
lines.push(` · draft + verify your instruction file's references (${triggerCostWording(harness)})`);
|
|
@@ -4339,7 +4590,10 @@ async function main() {
|
|
|
4339
4590
|
// through to a misleading "empty machine / no structural issues" report
|
|
4340
4591
|
// (obra/superpowers-marketplace, anthropics/claude-plugins-community).
|
|
4341
4592
|
if (json) {
|
|
4342
|
-
console.log(JSON.stringify(
|
|
4593
|
+
console.log(JSON.stringify((0, audit_report_js_1.buildMarketplaceReport)(market, {
|
|
4594
|
+
vigilesVersion: getVersion(),
|
|
4595
|
+
dir: (0, node_path_1.resolve)(dirs[0]),
|
|
4596
|
+
}), null, 2));
|
|
4343
4597
|
}
|
|
4344
4598
|
else {
|
|
4345
4599
|
console.log(`Marketplace "${market.name}": ${String(market.total)} plugin(s), all external ` +
|
|
@@ -4355,7 +4609,12 @@ async function main() {
|
|
|
4355
4609
|
const text = args.includes("--md")
|
|
4356
4610
|
? (0, leaderboard_js_1.formatLeaderboardMarkdown)(scores)
|
|
4357
4611
|
: (0, leaderboard_js_1.formatLeaderboard)(scores);
|
|
4358
|
-
console.log(json
|
|
4612
|
+
console.log(json
|
|
4613
|
+
? JSON.stringify((0, audit_report_js_1.buildLeaderboardReport)(scores, {
|
|
4614
|
+
vigilesVersion: getVersion(),
|
|
4615
|
+
dir: (0, node_path_1.resolve)(dirs[0]),
|
|
4616
|
+
}), null, 2)
|
|
4617
|
+
: text);
|
|
4359
4618
|
}
|
|
4360
4619
|
else {
|
|
4361
4620
|
const root = (0, node_path_1.resolve)(targets[0]);
|
|
@@ -4383,9 +4642,15 @@ async function main() {
|
|
|
4383
4642
|
// HTML renders, `--json` emits, and (later) a hosted dashboard ingests.
|
|
4384
4643
|
// Built ONCE; the rings + fix list are read off it. Pure deterministic —
|
|
4385
4644
|
// nothing executes to produce it.
|
|
4645
|
+
// Surfaces that exist but aren't spec-managed yet — the same notion
|
|
4646
|
+
// `init` adopts (layout-driven instruction file + skill/subagent sweep).
|
|
4647
|
+
// Surfaced in the AuditReport (the report's "Create spec" command-emit
|
|
4648
|
+
// buttons read it) and the terminal nudge below.
|
|
4649
|
+
const adoptableSurfaces = discoverAdoptableForAudit(root, adapter.layout.instructionFile);
|
|
4386
4650
|
const auditReport = (0, audit_report_js_1.buildAuditReport)(report, {
|
|
4387
4651
|
harness: adapter.name,
|
|
4388
4652
|
vigilesVersion: getVersion(),
|
|
4653
|
+
adoptableSurfaces,
|
|
4389
4654
|
});
|
|
4390
4655
|
const sc = auditReport.score;
|
|
4391
4656
|
const plan = (0, optimize_js_1.optimize)(report);
|
|
@@ -4404,6 +4669,18 @@ async function main() {
|
|
|
4404
4669
|
const fixes = (0, optimize_js_1.formatRecommendations)(plan);
|
|
4405
4670
|
if (fixes)
|
|
4406
4671
|
console.log("\n" + fixes);
|
|
4672
|
+
// Adoption nudge: surfaces that exist but aren't spec-managed yet, with
|
|
4673
|
+
// the create-all + per-surface `init` commands (the JSON carries the
|
|
4674
|
+
// data in `adoptable` instead — the terminal stays human-readable).
|
|
4675
|
+
const adoptNudge = formatAdoptableNudge(adoptableSurfaces);
|
|
4676
|
+
if (adoptNudge)
|
|
4677
|
+
console.log("\n" + adoptNudge);
|
|
4678
|
+
// A small behavioral nudge — the deterministic read can't tell whether
|
|
4679
|
+
// skills actually FIRE; point at the interactive measure + the API.
|
|
4680
|
+
const fireNudge = formatTriggerNudge(report.skills.filter((s) => s.hasDescription && !s.userInvoked)
|
|
4681
|
+
.length);
|
|
4682
|
+
if (fireNudge)
|
|
4683
|
+
console.log("\n" + fireNudge);
|
|
4407
4684
|
}
|
|
4408
4685
|
// ONE read-vs-run decision for the EXECUTING checks (live MCP + skill
|
|
4409
4686
|
// firing). A plain `audit` is a deterministic READ; these run only on
|
|
@@ -4415,6 +4692,7 @@ async function main() {
|
|
|
4415
4692
|
const surfaces = {
|
|
4416
4693
|
hasMcp: report.mcp && !isForeign,
|
|
4417
4694
|
triggerableSkills: report.skills.filter((s) => s.hasDescription && !s.userInvoked).length,
|
|
4695
|
+
gateSkills: (0, scan_behavioral_js_1.detectGateSkills)(report.skills).length,
|
|
4418
4696
|
adoptableRefs: adapter.name === "claude-code" &&
|
|
4419
4697
|
(0, node_fs_1.existsSync)((0, node_path_1.resolve)(root, adapter.layout.instructionFile)),
|
|
4420
4698
|
};
|
|
@@ -4501,17 +4779,36 @@ async function main() {
|
|
|
4501
4779
|
const finalReport = adoptabilityResult
|
|
4502
4780
|
? { ...auditReport, adoptability: adoptabilityResult }
|
|
4503
4781
|
: auditReport;
|
|
4504
|
-
// The shareable HTML report — written by default (--no-html to skip), and
|
|
4505
|
-
// opened best-effort only for a human at a TTY (never spawn a browser for
|
|
4506
|
-
// an agent / CI run).
|
|
4507
|
-
if (!json && !args.includes("--no-html")) {
|
|
4508
|
-
writeAuditHtml(finalReport);
|
|
4509
|
-
}
|
|
4510
4782
|
// The versioned JSON artifact — the upload/CI boundary (a hosted dashboard
|
|
4511
4783
|
// ingests this). Written by default in the human path; --no-json to skip.
|
|
4512
4784
|
if (!json && !args.includes("--no-json")) {
|
|
4513
4785
|
writeAuditJson(finalReport);
|
|
4514
4786
|
}
|
|
4787
|
+
// The HTML report has two deliveries. STATIC (default): write the
|
|
4788
|
+
// shareable file whose buttons copy the `init` command. LIVE (`--serve`,
|
|
4789
|
+
// or a TTY "yes"): start a loopback server whose buttons run `init` for
|
|
4790
|
+
// you. The gate keeps the default a terminating, headless-safe read and
|
|
4791
|
+
// restricts the write-server to your own repo (decideServeGate).
|
|
4792
|
+
const serveGate = (0, audit_serve_js_1.decideServeGate)({
|
|
4793
|
+
serveFlag: args.includes("--serve"),
|
|
4794
|
+
noServeFlag: args.includes("--no-serve"),
|
|
4795
|
+
json,
|
|
4796
|
+
isTTY: process.stdout.isTTY && process.stdin.isTTY,
|
|
4797
|
+
ownRepo: !isForeign,
|
|
4798
|
+
adoptableCount: finalReport.adoptable?.surfaces.length ?? 0,
|
|
4799
|
+
});
|
|
4800
|
+
let serveLive = serveGate === "serve";
|
|
4801
|
+
if (serveGate === "ask") {
|
|
4802
|
+
const ans = (await askOnce("\nOpen the live report to create specs with one click? [y/N] ")).toLowerCase();
|
|
4803
|
+
serveLive = ans === "y" || ans === "yes";
|
|
4804
|
+
}
|
|
4805
|
+
const errMsg = (e) => e instanceof Error ? e.message : String(e);
|
|
4806
|
+
if (serveLive) {
|
|
4807
|
+
await runAuditServe(finalReport, finalReport.adoptable, errMsg);
|
|
4808
|
+
}
|
|
4809
|
+
else if (!json && !args.includes("--no-html")) {
|
|
4810
|
+
writeAuditHtml(finalReport);
|
|
4811
|
+
}
|
|
4515
4812
|
}
|
|
4516
4813
|
break;
|
|
4517
4814
|
}
|
package/dist/core/adopt.d.ts
CHANGED
|
@@ -18,6 +18,7 @@
|
|
|
18
18
|
* vs `enforce()` is deliberately NOT guessed here — that cross-referencing is
|
|
19
19
|
* `strengthen`'s separate, later job; adoption is lossless transcription.
|
|
20
20
|
*/
|
|
21
|
+
import { type SkillSpec, type AgentSpec } from "./spec.js";
|
|
21
22
|
export type AdoptTier = "structured" | "raw";
|
|
22
23
|
export interface AdoptResult {
|
|
23
24
|
/** Generated `.spec.ts` source (compiles back to ~the original file). */
|
|
@@ -62,4 +63,31 @@ export declare function adoptToSpec(markdown: string, target: string): AdoptedSp
|
|
|
62
63
|
* (the deliverable `init` writes).
|
|
63
64
|
*/
|
|
64
65
|
export declare function adoptMarkdown(markdown: string, target: string): AdoptResult;
|
|
66
|
+
export interface AdoptSurfaceResult {
|
|
67
|
+
/** Generated `.spec.ts` source. */
|
|
68
|
+
source: string;
|
|
69
|
+
kind: "skill" | "agent";
|
|
70
|
+
/**
|
|
71
|
+
* The parsed spec object the source builds — exposed so a round-trip test can
|
|
72
|
+
* feed it straight to `compileSkill`/`compileAgent` without evaluating the
|
|
73
|
+
* generated TS (the same split as `adoptToSpec`/`renderSpecSource`).
|
|
74
|
+
*/
|
|
75
|
+
spec: SkillSpec | AgentSpec;
|
|
76
|
+
/**
|
|
77
|
+
* Frontmatter keys present in the source that the typed spec has no field for
|
|
78
|
+
* — emitted as a `// NOTE:` comment in the source so nothing is lost silently.
|
|
79
|
+
*/
|
|
80
|
+
unmappedKeys: string[];
|
|
81
|
+
}
|
|
82
|
+
/**
|
|
83
|
+
* Adopt an existing SKILL.md into a `skill()` spec. The body is carried verbatim
|
|
84
|
+
* (skills are freeform markdown — `##` headings stay in the body), so a clean
|
|
85
|
+
* skill round-trips below the integrity header.
|
|
86
|
+
*
|
|
87
|
+
* @param markdown the SKILL.md content
|
|
88
|
+
* @param dirName the skill's directory name — the CC fallback for `name` when
|
|
89
|
+
* frontmatter omits it
|
|
90
|
+
*/
|
|
91
|
+
export declare function adoptSkill(markdown: string, dirName: string): AdoptSurfaceResult;
|
|
92
|
+
export declare function adoptAgent(markdown: string, fileBase: string): AdoptSurfaceResult;
|
|
65
93
|
//# sourceMappingURL=adopt.d.ts.map
|