vigiles 5.1.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -18
- package/dist/adapters/claude-code/adapter.js +1 -0
- package/dist/adapters/claude-code/agent-runtime.d.ts +45 -6
- package/dist/adapters/claude-code/agent-runtime.js +94 -8
- package/dist/adapters/claude-code/dialect.d.ts +34 -0
- package/dist/adapters/claude-code/dialect.js +51 -19
- package/dist/adapters/claude-code/effect-region.d.ts +9 -0
- package/dist/adapters/claude-code/effect-region.js +45 -0
- package/dist/adapters/claude-code/layout.js +3 -0
- package/dist/adapters/claude-code/skill-runtime.d.ts +25 -0
- package/dist/adapters/claude-code/skill-runtime.js +40 -0
- package/dist/adapters/claude-code/typed-spec.d.ts +58 -0
- package/dist/adapters/claude-code/typed-spec.js +55 -0
- package/dist/adapters/codex/adapter.js +3 -0
- package/dist/adapters/codex/layout.js +3 -0
- package/dist/adapters/opencode/adapter.js +1 -0
- package/dist/adapters/opencode/layout.js +3 -0
- package/dist/check.d.ts +8 -0
- package/dist/check.js +27 -3
- package/dist/claude-code.d.ts +1 -0
- package/dist/claude-code.js +8 -1
- package/dist/cli.js +469 -88
- package/dist/core/adapter.d.ts +10 -0
- package/dist/core/bash-effects.d.ts +41 -0
- package/dist/core/bash-effects.js +405 -0
- package/dist/core/compile.d.ts +3 -1
- package/dist/core/compile.js +176 -39
- package/dist/core/dialect.d.ts +10 -0
- package/dist/core/effects.d.ts +172 -0
- package/dist/core/effects.js +245 -0
- package/dist/core/generate-harness.d.ts +187 -0
- package/dist/core/generate-harness.js +337 -0
- package/dist/core/layout.d.ts +6 -0
- package/dist/core/mcp-tool.d.ts +1 -1
- package/dist/core/orphans.js +21 -0
- package/dist/core/spec.d.ts +432 -11
- package/dist/core/spec.js +166 -3
- package/dist/core/tool-contract.d.ts +1 -1
- package/dist/core/types.d.ts +6 -6
- package/dist/core/validate.js +4 -4
- package/dist/harness-test.d.ts +7 -0
- package/dist/harness-test.js +19 -7
- package/dist/leaderboard.d.ts +2 -0
- package/dist/leaderboard.js +2 -0
- package/dist/optimize.d.ts +74 -0
- package/dist/optimize.js +94 -0
- package/dist/scaffold-test.d.ts +58 -0
- package/dist/scaffold-test.js +263 -0
- package/dist/scan.d.ts +40 -0
- package/dist/scan.js +91 -43
- package/dist/score-explainer.d.ts +69 -0
- package/dist/score-explainer.js +169 -0
- package/dist/test-coverage.d.ts +7 -0
- package/dist/test-coverage.js +39 -24
- package/package.json +2 -1
- package/skills/{migrate-to-spec → adopt-spec}/SKILL.md +4 -4
- package/skills/edit-spec/SKILL.md +1 -1
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.scaffoldTest = scaffoldTest;
|
|
4
|
+
exports.formatScaffolds = formatScaffolds;
|
|
5
|
+
/**
|
|
6
|
+
* `vigiles scaffold-test` — the deterministic test-gen engine (B1 v0).
|
|
7
|
+
*
|
|
8
|
+
* Free-form in, a RUNNABLE starter test out. Given an existing hand-written
|
|
9
|
+
* skill / subagent / hook, emit a scaffolded `*.harness.mjs` / `*.eval.mjs` at the
|
|
10
|
+
* surface's suggested test path — the deterministic counterpart to the
|
|
11
|
+
* `test-harness` SKILL (which picks the tier with a model). The scaffold picks the
|
|
12
|
+
* cheapest meaningful tier for the kind, wires the real public API + the surface's
|
|
13
|
+
* own metadata (name, namespaced id, declared tools), and leaves TODOs only where a
|
|
14
|
+
* human/model must supply judgement (the prompts, the event input, the assertion).
|
|
15
|
+
*
|
|
16
|
+
* Pure + model-free: hand it a `ScaffoldInput`, get back a `{ path, content }`. The
|
|
17
|
+
* CLI resolves a surface path / plugin dir into inputs (reusing the scan + untested
|
|
18
|
+
* detectors) and writes the files; this module owns only the templating.
|
|
19
|
+
*/
|
|
20
|
+
const node_path_1 = require("node:path");
|
|
21
|
+
const PLUGIN_TODO = "<plugin>";
|
|
22
|
+
/**
|
|
23
|
+
* The suggested test path for a surface — mirrors `suggestedTestPath` in
|
|
24
|
+
* `test-coverage.ts` (a skill gets an `.eval.mjs`, agent/hook a `.harness.mjs`),
|
|
25
|
+
* so a generated file is colocated where the untested-surface detector looks for it
|
|
26
|
+
* and the surface stops being reported untested.
|
|
27
|
+
*/
|
|
28
|
+
function suggestedPath(input) {
|
|
29
|
+
const dir = (0, node_path_1.dirname)(input.path);
|
|
30
|
+
const ext = input.kind === "skill" ? "eval.mjs" : "harness.mjs";
|
|
31
|
+
return `${dir}/${input.name}.${ext}`;
|
|
32
|
+
}
|
|
33
|
+
function header(title, run) {
|
|
34
|
+
return [
|
|
35
|
+
"/**",
|
|
36
|
+
` * ${title}`,
|
|
37
|
+
" *",
|
|
38
|
+
" * Generated by `vigiles scaffold-test` — a STARTER, not a finished test. Fill in",
|
|
39
|
+
" * the TODOs (they're where a human/model must supply judgement), then run:",
|
|
40
|
+
` * ${run}`,
|
|
41
|
+
" */",
|
|
42
|
+
].join("\n");
|
|
43
|
+
}
|
|
44
|
+
/** A hook → the unit tier (`runHook`): free, no model, reaches every event. */
|
|
45
|
+
function hookScaffold(input) {
|
|
46
|
+
const cmd = input.hookCommand ?? `bash ${input.path}`;
|
|
47
|
+
return `${header(`Starter unit test for the \`${input.name}\` hook.`, `npx vigiles test ${suggestedPath(input)}`)}
|
|
48
|
+
import { runHook, assertHookAllowed } from "vigiles/unit";
|
|
49
|
+
|
|
50
|
+
// TODO: set the event + input your hook actually inspects (PreToolUse/Bash shown).
|
|
51
|
+
const event = {
|
|
52
|
+
hook_event_name: "PreToolUse",
|
|
53
|
+
tool_name: "Bash",
|
|
54
|
+
tool_input: { command: "echo hello" },
|
|
55
|
+
};
|
|
56
|
+
|
|
57
|
+
const r = runHook(${JSON.stringify(cmd)}, event);
|
|
58
|
+
|
|
59
|
+
// TODO: assert the decision you expect. Use assertHookBlocked(r) for the deny case
|
|
60
|
+
// (and a second runHook with the input that SHOULD be blocked).
|
|
61
|
+
assertHookAllowed(r);
|
|
62
|
+
console.log("✓ ${input.name}: hook allowed the benign event");
|
|
63
|
+
`;
|
|
64
|
+
}
|
|
65
|
+
/** A skill → the eval tier (`measureTriggerRate`): does its description FIRE? */
|
|
66
|
+
function skillScaffold(input) {
|
|
67
|
+
const id = `${input.pluginName ?? PLUGIN_TODO}:${input.name}`;
|
|
68
|
+
const note = input.userInvoked
|
|
69
|
+
? "\n// NOTE: this skill is user-invoked (disableModelInvocation). Trigger-rate\n// measures MODEL-invocable skills; either make it model-invocable or test its\n// slash-command invocation with runHarnessTest instead.\n"
|
|
70
|
+
: "";
|
|
71
|
+
return `${header(`Starter trigger-rate eval for the \`${id}\` skill (recall + precision).`, `npx vigiles eval ${suggestedPath(input)} # real model, on your subscription`)}
|
|
72
|
+
import {
|
|
73
|
+
measureTriggerRate,
|
|
74
|
+
formatTriggerRateReport,
|
|
75
|
+
assertTriggerRate,
|
|
76
|
+
skillResolved,
|
|
77
|
+
} from "vigiles/testing";
|
|
78
|
+
import { fileURLToPath } from "node:url";
|
|
79
|
+
${note}
|
|
80
|
+
// TODO: point at the plugin root (the dir holding .claude-plugin/ or skills/).
|
|
81
|
+
const pluginDir = fileURLToPath(new URL("../../", import.meta.url));
|
|
82
|
+
const skill = ${JSON.stringify(id)};
|
|
83
|
+
|
|
84
|
+
const report = await measureTriggerRate({
|
|
85
|
+
pluginDir,
|
|
86
|
+
stubSkillBodies: true, // firing is a frontmatter property — stub bodies, pay less
|
|
87
|
+
prompts: [
|
|
88
|
+
// TODO: >=5 varied prompts that SHOULD fire ${input.name} (recall).
|
|
89
|
+
"TODO: a realistic task that should trigger ${input.name}",
|
|
90
|
+
],
|
|
91
|
+
irrelevantPrompts: [
|
|
92
|
+
// TODO: >=5 unrelated prompts that should NOT fire it (precision).
|
|
93
|
+
"TODO: an unrelated coding task",
|
|
94
|
+
],
|
|
95
|
+
fired: (t) => skillResolved(t, skill),
|
|
96
|
+
trials: Number(process.env.VIGILES_TRIALS || 1),
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
console.log(formatTriggerRateReport(report));
|
|
100
|
+
assertTriggerRate(report, { min: 0.8, maxFalsePositive: 0.3 });
|
|
101
|
+
`;
|
|
102
|
+
}
|
|
103
|
+
/** A JSON value placeholder for an `OutputFieldType`, for the `vigiles:ok` block. */
|
|
104
|
+
function placeholderFor(type) {
|
|
105
|
+
switch (type) {
|
|
106
|
+
case "number":
|
|
107
|
+
return 1;
|
|
108
|
+
case "boolean":
|
|
109
|
+
return true;
|
|
110
|
+
case "string[]":
|
|
111
|
+
return ["example"];
|
|
112
|
+
default:
|
|
113
|
+
return "example";
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
/** Render a `result(ok, err)` builder call reconstructed from the parsed contract. */
|
|
117
|
+
function renderContractBuilder(contract) {
|
|
118
|
+
const shape = (fields) => `{ ${fields.map((f) => `${f.name}: ${JSON.stringify(f.type)}`).join(", ")} }`;
|
|
119
|
+
return `result(\n ${shape(contract.ok)},\n ${shape(contract.err)},\n)`;
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* The OUTCOME test, GENERATED FROM the subagent's `result()` contract: reconstruct
|
|
123
|
+
* the contract, build a matching `vigiles:ok` block, and `assertAgentOk` it —
|
|
124
|
+
* deterministic, no LLM judge. This is the typed-spec payoff a markdown
|
|
125
|
+
* `description:` cannot give you: a parseable, typed outcome a test reads directly.
|
|
126
|
+
*/
|
|
127
|
+
function outcomeSection(input, contract) {
|
|
128
|
+
const okValue = Object.fromEntries(contract.ok.map((f) => [f.name, placeholderFor(f.type)]));
|
|
129
|
+
const firstField = contract.ok[0]?.name;
|
|
130
|
+
const fieldAssertion = firstField
|
|
131
|
+
? `// TODO: assert the VALUES you expect (the shape is already validated above), e.g.:\n// assert.ok(value.${firstField}, "expected a ${firstField}");`
|
|
132
|
+
: "";
|
|
133
|
+
return `import assert from "node:assert/strict";
|
|
134
|
+
import { result } from "vigiles/spec";
|
|
135
|
+
import { assertAgentOk } from "vigiles/testing";
|
|
136
|
+
|
|
137
|
+
// Reconstructed from ${input.name}'s ## Output contract (its compiled .md) — the
|
|
138
|
+
// typed result() the spec wrote. assertAgentOk parses + validates the outcome
|
|
139
|
+
// with NO model judge; swap \`okOutput\` for a real \`runHarness\` turn (Part B in
|
|
140
|
+
// examples/harness/railway-result.harness.mjs) to assert REAL behaviour.
|
|
141
|
+
const contract = ${renderContractBuilder(contract)};
|
|
142
|
+
|
|
143
|
+
const okOutput = [
|
|
144
|
+
"${input.name} finished its task.",
|
|
145
|
+
"\`\`\`vigiles:ok",
|
|
146
|
+
${JSON.stringify(JSON.stringify(okValue))},
|
|
147
|
+
"\`\`\`",
|
|
148
|
+
].join("\\n");
|
|
149
|
+
|
|
150
|
+
const value = assertAgentOk(okOutput, contract); // deterministic — no LLM judge
|
|
151
|
+
${fieldAssertion}
|
|
152
|
+
console.log("✓ ${input.name}: result() outcome parses + validates against its typed contract");
|
|
153
|
+
`;
|
|
154
|
+
}
|
|
155
|
+
/** The fallback when the subagent has no `result()` contract — assert a tool use. */
|
|
156
|
+
function fallbackSection(input) {
|
|
157
|
+
const toolHint = input.tools && input.tools.length > 0
|
|
158
|
+
? `assertToolUsed(r, ${JSON.stringify(input.tools[0])}); // its declared contract: ${input.tools.join(", ")}`
|
|
159
|
+
: `assertToolUsed(r, "Task"); // TODO: assert what the subagent should do`;
|
|
160
|
+
return `import { runHarnessTest, assertToolUsed } from "vigiles/testing";
|
|
161
|
+
|
|
162
|
+
// ${input.name} has no result() contract, so its outcome can't be asserted
|
|
163
|
+
// deterministically — add one (result() on its agent() spec) for a no-judge
|
|
164
|
+
// outcome test. For now, assert it reaches for the right tool.
|
|
165
|
+
const r = await runHarnessTest({
|
|
166
|
+
plugin: ".", // TODO: the plugin dir holding this subagent
|
|
167
|
+
// TODO: a prompt that dispatches ${input.name} (via the Task tool).
|
|
168
|
+
prompt: "TODO: a task that should dispatch ${input.name}",
|
|
169
|
+
transcript: true,
|
|
170
|
+
model: [{ text: "on it" }],
|
|
171
|
+
});
|
|
172
|
+
|
|
173
|
+
${toolHint}
|
|
174
|
+
console.log("✓ ${input.name}: subagent test ran");
|
|
175
|
+
`;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* The SAFETY check, GENERATED FROM the subagent's side-effecting `tools`: assert it
|
|
179
|
+
* stays inside its declared write surface and never reaches for a destructive op.
|
|
180
|
+
* The `tools` allowlist + effectSurface identify the "hole"; the check asserts it —
|
|
181
|
+
* a test the typed contract writes for you (markdown can declare the tools, not test them).
|
|
182
|
+
*/
|
|
183
|
+
function safetySection(input, sideEffecting) {
|
|
184
|
+
const checks = [];
|
|
185
|
+
if (sideEffecting.includes("Bash")) {
|
|
186
|
+
checks.push(` notTool("Bash", { command: /git push|rm -rf/ }), // never a destructive op`);
|
|
187
|
+
}
|
|
188
|
+
if (sideEffecting.some((t) => t === "Write" || t === "Edit")) {
|
|
189
|
+
checks.push(` didNotWrite("secrets.env"), // TODO: the path(s) it must NOT write outside its surface`);
|
|
190
|
+
}
|
|
191
|
+
if (checks.length === 0) {
|
|
192
|
+
checks.push(` // TODO: a notTool()/didNotWrite() per side-effecting tool: ${sideEffecting.join(", ")}`);
|
|
193
|
+
}
|
|
194
|
+
return `
|
|
195
|
+
// --- Safety (deterministic) — generated from ${input.name}'s side-effecting tools: ${sideEffecting.join(", ")} ---
|
|
196
|
+
// In a real run, replace this constructed Trace with a real \`runHarness\` /
|
|
197
|
+
// \`measure\` turn (use interceptTools so a real model's attempt is DENIED, never
|
|
198
|
+
// executed — see docs/eval-architecture.md). The checks below are derived from the
|
|
199
|
+
// declared tools contract — the agent's "hole" asserted to stay in its lane.
|
|
200
|
+
{
|
|
201
|
+
const trace = {
|
|
202
|
+
output: "done",
|
|
203
|
+
turns: 1,
|
|
204
|
+
hooks: [],
|
|
205
|
+
toolCalls: [
|
|
206
|
+
// TODO: the tool calls a benign run of ${input.name} makes.
|
|
207
|
+
{ name: "Bash", input: { command: "git status" } },
|
|
208
|
+
],
|
|
209
|
+
file: () => null,
|
|
210
|
+
};
|
|
211
|
+
assertChecks(trace, [
|
|
212
|
+
${checks.join("\n")}
|
|
213
|
+
]);
|
|
214
|
+
console.log("✓ ${input.name}: stayed inside its declared side-effect surface");
|
|
215
|
+
}
|
|
216
|
+
`;
|
|
217
|
+
}
|
|
218
|
+
/** A subagent → deterministic outcome + safety tests, generated from its typed contract. */
|
|
219
|
+
function agentScaffold(input) {
|
|
220
|
+
const head = header(`Starter harness test for the \`${input.name}\` subagent.`, `npx vigiles test ${suggestedPath(input)}`);
|
|
221
|
+
const safetyImport = input.sideEffectingTools && input.sideEffectingTools.length > 0
|
|
222
|
+
? `import { notTool, didNotWrite, assertChecks } from "vigiles/testing";\n`
|
|
223
|
+
: "";
|
|
224
|
+
const body = input.resultContract
|
|
225
|
+
? outcomeSection(input, input.resultContract)
|
|
226
|
+
: fallbackSection(input);
|
|
227
|
+
const safety = input.sideEffectingTools && input.sideEffectingTools.length > 0
|
|
228
|
+
? safetySection(input, input.sideEffectingTools)
|
|
229
|
+
: "";
|
|
230
|
+
return `${head}\n${safetyImport}${body}${safety}`;
|
|
231
|
+
}
|
|
232
|
+
const TIER = {
|
|
233
|
+
hook: "unit",
|
|
234
|
+
skill: "eval",
|
|
235
|
+
agent: "harness",
|
|
236
|
+
};
|
|
237
|
+
const BUILDER = {
|
|
238
|
+
hook: hookScaffold,
|
|
239
|
+
skill: skillScaffold,
|
|
240
|
+
agent: agentScaffold,
|
|
241
|
+
};
|
|
242
|
+
/** Scaffold a starter test for one surface. Pure: path + content, no I/O. */
|
|
243
|
+
function scaffoldTest(input) {
|
|
244
|
+
return {
|
|
245
|
+
path: suggestedPath(input),
|
|
246
|
+
content: BUILDER[input.kind](input),
|
|
247
|
+
kind: input.kind,
|
|
248
|
+
tier: TIER[input.kind],
|
|
249
|
+
};
|
|
250
|
+
}
|
|
251
|
+
/** Render a set of scaffolds for the CLI (what was generated, where, which tier). */
|
|
252
|
+
function formatScaffolds(scaffolds) {
|
|
253
|
+
if (scaffolds.length === 0) {
|
|
254
|
+
return "Nothing to scaffold — every surface already has a test, or none was found.";
|
|
255
|
+
}
|
|
256
|
+
const lines = [`Scaffolded ${String(scaffolds.length)} starter test(s):`, ""];
|
|
257
|
+
for (const s of scaffolds) {
|
|
258
|
+
lines.push(` ${s.path} [${s.kind} → ${s.tier} tier]`);
|
|
259
|
+
}
|
|
260
|
+
lines.push("", "These are STARTERS — fill in the TODOs (prompts / event / assertions), then run", "them with `npx vigiles test` (deterministic) or `npx vigiles eval` (real model).");
|
|
261
|
+
return lines.join("\n");
|
|
262
|
+
}
|
|
263
|
+
//# sourceMappingURL=scaffold-test.js.map
|
package/dist/scan.d.ts
CHANGED
|
@@ -19,6 +19,7 @@ import { type McpIssue } from "./core/mcp-config.js";
|
|
|
19
19
|
import { type DescriptionOverlap } from "./core/description-overlap.js";
|
|
20
20
|
import { type McpToolIssue } from "./core/mcp-tool.js";
|
|
21
21
|
import { type McpHookIssue } from "./core/mcp-hook.js";
|
|
22
|
+
import { type PurityLevel, type EffectSurface } from "./core/effects.js";
|
|
22
23
|
/** A named writing system. The label `unexpectedScript` reports + the config's expectation parse into this. */
|
|
23
24
|
export type Script = "Latin" | "Cyrillic" | "Han" | "Japanese" | "Korean" | "Arabic" | "Hebrew" | "Greek" | "Devanagari" | "Thai";
|
|
24
25
|
export interface ScanSkill {
|
|
@@ -47,6 +48,19 @@ export interface ScanAgent {
|
|
|
47
48
|
readonly mcpToolIssues: readonly McpToolIssue[];
|
|
48
49
|
/** `disallowedTools:` block-list entries that are typos of a real tool (block nothing). */
|
|
49
50
|
readonly disallowedToolIssues: readonly ToolIssue[];
|
|
51
|
+
/**
|
|
52
|
+
* Static effect-surface purity of the agent's declared tool contract.
|
|
53
|
+
* - `"pure"` — no side-effecting tools (read-only, deterministically testable).
|
|
54
|
+
* - `"bounded"` — has side-effecting tools (Edit/Write/…) but no Bash or unknown.
|
|
55
|
+
* - `"unrestricted"` — has Bash, any MCP/unknown tool, or inherits-all (no contract).
|
|
56
|
+
* Computed by `effectSurface()` from `src/core/effects.ts` — one detector, no drift.
|
|
57
|
+
*/
|
|
58
|
+
readonly purity: PurityLevel;
|
|
59
|
+
/**
|
|
60
|
+
* The three tool buckets from `effectSurface()`: read-only, side-effecting, and
|
|
61
|
+
* unknown-effect (MCP or unrecognized) tool names in the declared contract.
|
|
62
|
+
*/
|
|
63
|
+
readonly effectBuckets: Pick<EffectSurface, "readOnly" | "sideEffecting" | "unknown">;
|
|
50
64
|
}
|
|
51
65
|
/** A skill/agent whose frontmatter is missing a required field (name / description). */
|
|
52
66
|
export interface FrontmatterIssue {
|
|
@@ -121,6 +135,32 @@ export interface ScanReport {
|
|
|
121
135
|
readonly malformedFrontmatter: readonly FrontmatterParseIssue[];
|
|
122
136
|
readonly warnings: readonly string[];
|
|
123
137
|
readonly untested: number;
|
|
138
|
+
/**
|
|
139
|
+
* Harness-level purity summary: how many scanned agents fall into each purity
|
|
140
|
+
* rung. A high `pure` count means more of the harness is statically testable
|
|
141
|
+
* (deterministic, no mocks); `unrestricted` is the blind-spot count.
|
|
142
|
+
* Computed by `effectSurface()` (one detector, no drift).
|
|
143
|
+
*/
|
|
144
|
+
readonly puritySummary: {
|
|
145
|
+
pure: number;
|
|
146
|
+
bounded: number;
|
|
147
|
+
unrestricted: number;
|
|
148
|
+
};
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* Per-kind surface classifiers, built from the harness `PluginLayout`'s
|
|
152
|
+
* `skillDir`/`agentDir`/`commandDir` — so adding a harness whose subagents live
|
|
153
|
+
* somewhere other than `agents/` (OpenCode's `.opencode/agent`) needs no change
|
|
154
|
+
* here. Each anchors on a real path boundary (start-of-path or a `/`), so a
|
|
155
|
+
* directory whose NAME merely ends in the keyword isn't misclassified — e.g. the
|
|
156
|
+
* skill `skills/dispatching-parallel-agents/SKILL.md` must NOT register as an
|
|
157
|
+
* agent named "SKILL" (the `-agents/` substring), which real plugins like
|
|
158
|
+
* obra/superpowers ship. See scan.test.ts for the regression cases.
|
|
159
|
+
*/
|
|
160
|
+
export interface SurfaceClassifier {
|
|
161
|
+
readonly isSkill: (f: string) => boolean;
|
|
162
|
+
readonly isAgent: (f: string) => boolean;
|
|
163
|
+
readonly isCommand: (f: string) => boolean;
|
|
124
164
|
}
|
|
125
165
|
/**
|
|
126
166
|
* The description's dominant alphabetic script when it DIFFERS from `expected`
|
package/dist/scan.js
CHANGED
|
@@ -34,6 +34,7 @@ const mcp_tool_js_1 = require("./core/mcp-tool.js");
|
|
|
34
34
|
const mcp_hook_js_1 = require("./core/mcp-hook.js");
|
|
35
35
|
const agent_runtime_js_1 = require("./adapters/claude-code/agent-runtime.js");
|
|
36
36
|
const test_coverage_js_1 = require("./test-coverage.js");
|
|
37
|
+
const effects_js_1 = require("./core/effects.js");
|
|
37
38
|
// ---------------------------------------------------------------------------
|
|
38
39
|
// Internals
|
|
39
40
|
// ---------------------------------------------------------------------------
|
|
@@ -51,14 +52,24 @@ function frontmatter(md) {
|
|
|
51
52
|
color: (0, frontmatter_read_js_1.frontmatterScalar)(fm, "color"),
|
|
52
53
|
};
|
|
53
54
|
}
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
//
|
|
59
|
-
const
|
|
60
|
-
const
|
|
61
|
-
const
|
|
55
|
+
function escapeRe(s) {
|
|
56
|
+
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
57
|
+
}
|
|
58
|
+
function makeClassifier(layout) {
|
|
59
|
+
// An empty dir means "this harness has no such surface" → never matches.
|
|
60
|
+
const at = (dir) => dir ? `(?:^|/)${escapeRe(dir)}/` : null;
|
|
61
|
+
const skill = at(layout.skillDir);
|
|
62
|
+
const agent = at(layout.agentDir);
|
|
63
|
+
const command = at(layout.commandDir);
|
|
64
|
+
const skillRe = skill ? new RegExp(`${skill}[^/]+/SKILL\\.md$`) : null;
|
|
65
|
+
const agentRe = agent ? new RegExp(`${agent}[^/]+\\.md$`) : null;
|
|
66
|
+
const commandRe = command ? new RegExp(`${command}.+\\.md$`) : null;
|
|
67
|
+
return {
|
|
68
|
+
isSkill: (f) => skillRe?.test(f) ?? false,
|
|
69
|
+
isAgent: (f) => (agentRe?.test(f) ?? false) && !f.endsWith(".spec.ts"),
|
|
70
|
+
isCommand: (f) => commandRe?.test(f) ?? false,
|
|
71
|
+
};
|
|
72
|
+
}
|
|
62
73
|
function skillName(path) {
|
|
63
74
|
return (path
|
|
64
75
|
.replace(/\/SKILL\.md$/, "")
|
|
@@ -137,10 +148,10 @@ function firstBodyParagraph(md) {
|
|
|
137
148
|
}
|
|
138
149
|
return para.join(" ").trim() || undefined;
|
|
139
150
|
}
|
|
140
|
-
function scanSkills(files) {
|
|
151
|
+
function scanSkills(files, cls) {
|
|
141
152
|
const out = [];
|
|
142
153
|
for (const [path, md] of Object.entries(files)) {
|
|
143
|
-
if (!isSkill(path))
|
|
154
|
+
if (!cls.isSkill(path))
|
|
144
155
|
continue;
|
|
145
156
|
const fm = frontmatter(md);
|
|
146
157
|
// A skill's trigger surface is its frontmatter `description` OR — when that's
|
|
@@ -165,10 +176,10 @@ function scanSkills(files) {
|
|
|
165
176
|
* logic as `scanSkills` (frontmatter `description` ← first body paragraph), then
|
|
166
177
|
* the NCD precision-proxy. See description-overlap.ts.
|
|
167
178
|
*/
|
|
168
|
-
function descriptionOverlapsFor(files) {
|
|
179
|
+
function descriptionOverlapsFor(files, cls) {
|
|
169
180
|
const surfaces = [];
|
|
170
181
|
for (const [path, md] of Object.entries(files)) {
|
|
171
|
-
if (!isSkill(path))
|
|
182
|
+
if (!cls.isSkill(path))
|
|
172
183
|
continue;
|
|
173
184
|
if (/^\s*disable-model-invocation:\s*true\s*$/m.test(md))
|
|
174
185
|
continue;
|
|
@@ -180,12 +191,16 @@ function descriptionOverlapsFor(files) {
|
|
|
180
191
|
}
|
|
181
192
|
return (0, description_overlap_js_1.findDescriptionOverlaps)(surfaces);
|
|
182
193
|
}
|
|
183
|
-
function scanAgents(files, dialect, declaredServers) {
|
|
194
|
+
function scanAgents(files, dialect, declaredServers, cls) {
|
|
184
195
|
const out = [];
|
|
185
196
|
for (const [path, md] of Object.entries(files)) {
|
|
186
|
-
if (!isAgent(path))
|
|
197
|
+
if (!cls.isAgent(path))
|
|
187
198
|
continue;
|
|
188
199
|
const tools = (0, agent_runtime_js_1.parseAgentTools)(md);
|
|
200
|
+
// An inherits-all agent (no `tools:` line) grants access to every tool
|
|
201
|
+
// including every side-effecting one — pass the wildcard sentinel so
|
|
202
|
+
// effectSurface correctly classifies it as `"unrestricted"`.
|
|
203
|
+
const surface = (0, effects_js_1.effectSurface)(tools ?? ["*"], dialect);
|
|
189
204
|
out.push({
|
|
190
205
|
name: (0, node_path_1.basename)(path, ".md"),
|
|
191
206
|
path,
|
|
@@ -206,21 +221,31 @@ function scanAgents(files, dialect, declaredServers) {
|
|
|
206
221
|
// The block-list mirror: a `disallowedTools:` entry that's a typo of a real
|
|
207
222
|
// tool blocks nothing (close-typo only — high-precision). See tool-contract.ts.
|
|
208
223
|
disallowedToolIssues: (0, tool_contract_js_1.disallowedToolIssues)((0, agent_runtime_js_1.parseAgentToolList)(md, "disallowedTools") ?? [], dialect),
|
|
224
|
+
purity: surface.purity,
|
|
225
|
+
effectBuckets: {
|
|
226
|
+
readOnly: surface.readOnly,
|
|
227
|
+
sideEffecting: surface.sideEffecting,
|
|
228
|
+
unknown: surface.unknown,
|
|
229
|
+
},
|
|
209
230
|
});
|
|
210
231
|
}
|
|
211
232
|
return out.sort((a, b) => a.name.localeCompare(b.name));
|
|
212
233
|
}
|
|
213
234
|
/**
|
|
214
235
|
* Resolve a hook script token to a checkable path. `loadPlugin` expands the
|
|
215
|
-
* braced `${CLAUDE_PLUGIN_ROOT}
|
|
216
|
-
* survives, so resolve
|
|
217
|
-
*
|
|
236
|
+
* braced plugin-root token (`${CLAUDE_PLUGIN_ROOT}`, Codex `${PLUGIN_ROOT}`, …);
|
|
237
|
+
* the unbraced shell form survives, so resolve BOTH forms of the HARNESS's token
|
|
238
|
+
* (from the layout, not hard-coded) against the plugin root and strip shell
|
|
239
|
+
* quotes. A token that still carries any `$VAR` after that is genuinely
|
|
240
|
+
* uncheckable.
|
|
218
241
|
*/
|
|
219
|
-
function resolveScript(token, root) {
|
|
242
|
+
function resolveScript(token, root, pluginRootToken) {
|
|
243
|
+
// "${CLAUDE_PLUGIN_ROOT}" → unbraced "$CLAUDE_PLUGIN_ROOT".
|
|
244
|
+
const unbraced = pluginRootToken.replace(/^\$\{(.+)\}$/, "$$$1");
|
|
220
245
|
const cleaned = token
|
|
221
246
|
.replace(/["']/g, "")
|
|
222
|
-
.replaceAll(
|
|
223
|
-
.replaceAll(
|
|
247
|
+
.replaceAll(pluginRootToken, root)
|
|
248
|
+
.replaceAll(unbraced, root);
|
|
224
249
|
if (cleaned.includes("$"))
|
|
225
250
|
return { script: token, status: "unresolved" };
|
|
226
251
|
// A relative hook path (`./hooks/x.sh`, `scripts/x.py`) is the plugin's own —
|
|
@@ -238,7 +263,7 @@ function resolveScript(token, root) {
|
|
|
238
263
|
// command as MISSING (a false positive caught on gmickel/flow-next's ralph-guard).
|
|
239
264
|
const EXISTENCE_GUARD = /(?:\[\[?\s*!?\s*-[efsx]\s)|(?:\btest\s+!?\s*-[efsx]\s)/;
|
|
240
265
|
/** Pull script-file hook commands out of the resolved settings; count inline ones. */
|
|
241
|
-
function scanHooks(settings, root) {
|
|
266
|
+
function scanHooks(settings, root, pluginRootToken) {
|
|
242
267
|
const text = JSON.stringify(settings.hooks ?? {});
|
|
243
268
|
const commands = [...text.matchAll(/"command":\s*"((?:[^"\\]|\\.)*)"/g)].map((m) => m[1]);
|
|
244
269
|
const byScript = new Map();
|
|
@@ -257,7 +282,7 @@ function scanHooks(settings, root) {
|
|
|
257
282
|
continue;
|
|
258
283
|
}
|
|
259
284
|
for (const tok of found) {
|
|
260
|
-
const hook = resolveScript(tok, root);
|
|
285
|
+
const hook = resolveScript(tok, root, pluginRootToken);
|
|
261
286
|
byScript.set(hook.script, hook);
|
|
262
287
|
}
|
|
263
288
|
}
|
|
@@ -276,10 +301,10 @@ function scanHooks(settings, root) {
|
|
|
276
301
|
* is a separate, behavioral concern). See https://code.claude.com/docs/en/skills
|
|
277
302
|
* and …/sub-agents.
|
|
278
303
|
*/
|
|
279
|
-
function frontmatterIssuesFor(files) {
|
|
304
|
+
function frontmatterIssuesFor(files, cls) {
|
|
280
305
|
const out = [];
|
|
281
306
|
for (const [path, md] of Object.entries(files)) {
|
|
282
|
-
if (!isAgent(path))
|
|
307
|
+
if (!cls.isAgent(path))
|
|
283
308
|
continue; // skills require no frontmatter (dir/body fallbacks)
|
|
284
309
|
const fm = frontmatter(md);
|
|
285
310
|
const missing = [];
|
|
@@ -337,12 +362,12 @@ function closeCandidate(value, candidates) {
|
|
|
337
362
|
* Agent frontmatter VALUE validity — a `model:` or `color:` that's a close typo
|
|
338
363
|
* of a real one. A bad `model:` silently falls back; a bad `color:` is ignored.
|
|
339
364
|
* High-precision (close-typo only); a full/dated model id is left alone. Folded
|
|
340
|
-
* into the `
|
|
365
|
+
* into the `subagent-frontmatter` rule. Agents only (skills have no model/color).
|
|
341
366
|
*/
|
|
342
|
-
function frontmatterValueIssuesFor(files) {
|
|
367
|
+
function frontmatterValueIssuesFor(files, cls) {
|
|
343
368
|
const out = [];
|
|
344
369
|
for (const [path, md] of Object.entries(files)) {
|
|
345
|
-
if (!isAgent(path))
|
|
370
|
+
if (!cls.isAgent(path))
|
|
346
371
|
continue;
|
|
347
372
|
const fm = frontmatter(md);
|
|
348
373
|
// A model id with a digit/hyphen is an explicit form, not an alias typo.
|
|
@@ -382,10 +407,10 @@ function frontmatterValueIssuesFor(files) {
|
|
|
382
407
|
* as an informational note (NOT a structural defect) and the lint rule is a
|
|
383
408
|
* warn, not an error. The file's other fields are still salvaged.
|
|
384
409
|
*/
|
|
385
|
-
function malformedFrontmatterFor(files) {
|
|
410
|
+
function malformedFrontmatterFor(files, cls) {
|
|
386
411
|
const out = [];
|
|
387
412
|
for (const [path, md] of Object.entries(files)) {
|
|
388
|
-
if (!isSkill(path) && !isAgent(path))
|
|
413
|
+
if (!cls.isSkill(path) && !cls.isAgent(path))
|
|
389
414
|
continue;
|
|
390
415
|
if (!(0, frontmatter_read_js_1.readFrontmatter)(md).malformed)
|
|
391
416
|
continue;
|
|
@@ -405,10 +430,10 @@ function malformedFrontmatterFor(files) {
|
|
|
405
430
|
* missing either; surfaced as a soft note in scan (NOT a structural defect, NOT
|
|
406
431
|
* scored) and gated by the `skill-frontmatter` lint rule (warn by default).
|
|
407
432
|
*/
|
|
408
|
-
function skillMetaIssuesFor(files) {
|
|
433
|
+
function skillMetaIssuesFor(files, cls) {
|
|
409
434
|
const out = [];
|
|
410
435
|
for (const [path, md] of Object.entries(files)) {
|
|
411
|
-
if (!isSkill(path))
|
|
436
|
+
if (!cls.isSkill(path))
|
|
412
437
|
continue;
|
|
413
438
|
const fm = frontmatter(md);
|
|
414
439
|
const missing = [];
|
|
@@ -457,8 +482,9 @@ function collectMcpServers(root, layout) {
|
|
|
457
482
|
/** Scan a plugin/repo directory and report its surfaces + structural issues. */
|
|
458
483
|
function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
|
|
459
484
|
const lay = layout ?? layout_js_1.claudeCodeLayout;
|
|
485
|
+
const cls = makeClassifier(lay);
|
|
460
486
|
const loaded = (0, plugin_loader_js_1.loadPlugin)(dir, lay);
|
|
461
|
-
const { hooks, inline } = scanHooks(loaded.settings, (0, node_path_1.resolve)(dir));
|
|
487
|
+
const { hooks, inline } = scanHooks(loaded.settings, (0, node_path_1.resolve)(dir), lay.pluginRootToken);
|
|
462
488
|
// Hook-event keys are a CLOSED platform set — an unrecognized one is a dead
|
|
463
489
|
// registration (the hook never fires), so flag every unknown (not just typos).
|
|
464
490
|
// ONLY for the canonical object-keyed-by-event shape: a plugin shipping a
|
|
@@ -480,26 +506,33 @@ function scanPlugin(dir, layout, dialect = dialect_js_1.claudeCodeDialect) {
|
|
|
480
506
|
: null;
|
|
481
507
|
const mcpServers = collectMcpServers((0, node_path_1.resolve)(dir), lay);
|
|
482
508
|
const declaredServers = Object.keys(mcpServers);
|
|
509
|
+
const agents = scanAgents(loaded.files, dialect, declaredServers, cls);
|
|
510
|
+
const puritySummary = agents.reduce((acc, a) => {
|
|
511
|
+
acc[a.purity]++;
|
|
512
|
+
return acc;
|
|
513
|
+
}, { pure: 0, bounded: 0, unrestricted: 0 });
|
|
483
514
|
return {
|
|
484
515
|
dir,
|
|
485
516
|
instructions,
|
|
486
|
-
skills: scanSkills(loaded.files),
|
|
487
|
-
agents
|
|
517
|
+
skills: scanSkills(loaded.files, cls),
|
|
518
|
+
agents,
|
|
488
519
|
hooks,
|
|
489
520
|
inlineHooks: inline,
|
|
490
|
-
commands: Object.keys(loaded.files).filter(isCommand).length,
|
|
521
|
+
commands: Object.keys(loaded.files).filter(cls.isCommand).length,
|
|
491
522
|
mcp: loaded.warnings.some((w) => w.includes("MCP server")),
|
|
492
523
|
danglingRefs: (0, plugin_loader_js_2.danglingRefs)((0, node_path_1.resolve)(dir), lay),
|
|
493
524
|
hookEventIssues,
|
|
494
|
-
frontmatterIssues: frontmatterIssuesFor(loaded.files),
|
|
495
|
-
frontmatterValueIssues: frontmatterValueIssuesFor(loaded.files),
|
|
496
|
-
skillMetaIssues: skillMetaIssuesFor(loaded.files),
|
|
525
|
+
frontmatterIssues: frontmatterIssuesFor(loaded.files, cls),
|
|
526
|
+
frontmatterValueIssues: frontmatterValueIssuesFor(loaded.files, cls),
|
|
527
|
+
skillMetaIssues: skillMetaIssuesFor(loaded.files, cls),
|
|
497
528
|
mcpIssues: (0, mcp_config_js_1.verifyMcpServers)(mcpServers),
|
|
498
529
|
mcpHookIssues: (0, mcp_hook_js_1.verifyMcpHookTargets)(loaded.settings.hooks, declaredServers, dialect),
|
|
499
|
-
descriptionOverlaps: descriptionOverlapsFor(loaded.files),
|
|
500
|
-
malformedFrontmatter: malformedFrontmatterFor(loaded.files),
|
|
530
|
+
descriptionOverlaps: descriptionOverlapsFor(loaded.files, cls),
|
|
531
|
+
malformedFrontmatter: malformedFrontmatterFor(loaded.files, cls),
|
|
501
532
|
warnings: loaded.warnings,
|
|
502
|
-
untested: (0, test_coverage_js_1.findUntestedSurfaces)({ basePath: dir }).untested
|
|
533
|
+
untested: (0, test_coverage_js_1.findUntestedSurfaces)({ basePath: dir, layout: lay }).untested
|
|
534
|
+
.length,
|
|
535
|
+
puritySummary,
|
|
503
536
|
};
|
|
504
537
|
}
|
|
505
538
|
/**
|
|
@@ -588,7 +621,7 @@ function skillLine(s) {
|
|
|
588
621
|
const mark = s.descriptionScript ? "⚠" : "✓";
|
|
589
622
|
return ` ${mark} ${s.name}${notes.length ? ` (${notes.join("; ")})` : ""}`;
|
|
590
623
|
}
|
|
591
|
-
/** One agent's report block: ✗ (broken contract) / ⚠ (inherits all) / ✓ + issues. */
|
|
624
|
+
/** One agent's report block: ✗ (broken contract) / ⚠ (inherits all) / ✓ + issues + purity. */
|
|
592
625
|
function agentLines(a) {
|
|
593
626
|
const tools = a.tools === null
|
|
594
627
|
? "tools: (inherits all — no contract)"
|
|
@@ -601,7 +634,15 @@ function agentLines(a) {
|
|
|
601
634
|
mark = "✗";
|
|
602
635
|
else if (a.tools === null)
|
|
603
636
|
mark = "⚠";
|
|
604
|
-
|
|
637
|
+
// Purity is an informational health signal (not a structural defect); mark it
|
|
638
|
+
// clearly so a reader knows which rung this agent is on.
|
|
639
|
+
const PURITY_TAGS = {
|
|
640
|
+
pure: "pure",
|
|
641
|
+
bounded: "bounded",
|
|
642
|
+
unrestricted: "unrestricted",
|
|
643
|
+
};
|
|
644
|
+
const purityTag = PURITY_TAGS[a.purity] ?? "unrestricted";
|
|
645
|
+
const lines = [` ${mark} ${a.name} — ${tools} [${purityTag}]`];
|
|
605
646
|
for (const issue of a.toolIssues)
|
|
606
647
|
lines.push(` ✗ ${issue.message}`);
|
|
607
648
|
for (const issue of a.mcpToolIssues)
|
|
@@ -650,6 +691,13 @@ function formatScanReport(r) {
|
|
|
650
691
|
facts.push(`Commands: ${String(r.commands)}`);
|
|
651
692
|
facts.push(`MCP servers: ${r.mcp ? "yes" : "no"}`);
|
|
652
693
|
facts.push(`Untested surfaces: ${String(r.untested)}`);
|
|
694
|
+
// Effect surface: harness-level purity summary across all scanned agents.
|
|
695
|
+
// Informational (higher pure% = more constrained, cheaper to test); shown
|
|
696
|
+
// only when there are agents to summarize (no agents → no summary line).
|
|
697
|
+
if (r.agents.length > 0) {
|
|
698
|
+
const { pure, bounded, unrestricted } = r.puritySummary;
|
|
699
|
+
facts.push(`Effect surface: ${String(pure)} pure · ${String(bounded)} bounded · ${String(unrestricted)} unrestricted`);
|
|
700
|
+
}
|
|
653
701
|
out.push(...facts, "");
|
|
654
702
|
// The dangling-ref warning is now shown as a first-class ✗ section above, so
|
|
655
703
|
// drop it from the free-text list to avoid saying the same thing twice.
|