vigiles 5.1.0 → 6.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +59 -18
- package/dist/adapters/claude-code/adapter.js +1 -0
- package/dist/adapters/claude-code/agent-runtime.d.ts +45 -6
- package/dist/adapters/claude-code/agent-runtime.js +94 -8
- package/dist/adapters/claude-code/dialect.d.ts +34 -0
- package/dist/adapters/claude-code/dialect.js +51 -19
- package/dist/adapters/claude-code/effect-region.d.ts +9 -0
- package/dist/adapters/claude-code/effect-region.js +45 -0
- package/dist/adapters/claude-code/layout.js +3 -0
- package/dist/adapters/claude-code/skill-runtime.d.ts +25 -0
- package/dist/adapters/claude-code/skill-runtime.js +40 -0
- package/dist/adapters/claude-code/typed-spec.d.ts +58 -0
- package/dist/adapters/claude-code/typed-spec.js +55 -0
- package/dist/adapters/codex/adapter.js +3 -0
- package/dist/adapters/codex/layout.js +3 -0
- package/dist/adapters/opencode/adapter.js +1 -0
- package/dist/adapters/opencode/layout.js +3 -0
- package/dist/check.d.ts +8 -0
- package/dist/check.js +27 -3
- package/dist/claude-code.d.ts +1 -0
- package/dist/claude-code.js +8 -1
- package/dist/cli.js +469 -88
- package/dist/core/adapter.d.ts +10 -0
- package/dist/core/bash-effects.d.ts +41 -0
- package/dist/core/bash-effects.js +405 -0
- package/dist/core/compile.d.ts +3 -1
- package/dist/core/compile.js +176 -39
- package/dist/core/dialect.d.ts +10 -0
- package/dist/core/effects.d.ts +172 -0
- package/dist/core/effects.js +245 -0
- package/dist/core/generate-harness.d.ts +187 -0
- package/dist/core/generate-harness.js +337 -0
- package/dist/core/layout.d.ts +6 -0
- package/dist/core/mcp-tool.d.ts +1 -1
- package/dist/core/orphans.js +21 -0
- package/dist/core/spec.d.ts +432 -11
- package/dist/core/spec.js +166 -3
- package/dist/core/tool-contract.d.ts +1 -1
- package/dist/core/types.d.ts +6 -6
- package/dist/core/validate.js +4 -4
- package/dist/harness-test.d.ts +7 -0
- package/dist/harness-test.js +19 -7
- package/dist/leaderboard.d.ts +2 -0
- package/dist/leaderboard.js +2 -0
- package/dist/optimize.d.ts +74 -0
- package/dist/optimize.js +94 -0
- package/dist/scaffold-test.d.ts +58 -0
- package/dist/scaffold-test.js +263 -0
- package/dist/scan.d.ts +40 -0
- package/dist/scan.js +91 -43
- package/dist/score-explainer.d.ts +69 -0
- package/dist/score-explainer.js +169 -0
- package/dist/test-coverage.d.ts +7 -0
- package/dist/test-coverage.js +39 -24
- package/package.json +2 -1
- package/skills/{migrate-to-spec → adopt-spec}/SKILL.md +4 -4
- package/skills/edit-spec/SKILL.md +1 -1
package/dist/core/spec.js
CHANGED
|
@@ -17,7 +17,10 @@ exports.file = file;
|
|
|
17
17
|
exports.cmd = cmd;
|
|
18
18
|
exports.symbol = symbol;
|
|
19
19
|
exports.ref = ref;
|
|
20
|
+
exports.dir = dir;
|
|
21
|
+
exports.glob = glob;
|
|
20
22
|
exports.instructions = instructions;
|
|
23
|
+
exports.effect = effect;
|
|
21
24
|
exports.claude = claude;
|
|
22
25
|
exports.project = project;
|
|
23
26
|
exports.input = input;
|
|
@@ -27,6 +30,11 @@ exports.agent = agent;
|
|
|
27
30
|
exports.result = result;
|
|
28
31
|
exports.delegate = delegate;
|
|
29
32
|
exports.railway = railway;
|
|
33
|
+
exports.needs = needs;
|
|
34
|
+
exports.pipeStep = pipeStep;
|
|
35
|
+
exports.start = start;
|
|
36
|
+
exports.andThen = andThen;
|
|
37
|
+
exports.pipe = pipe;
|
|
30
38
|
exports.defineConfig = defineConfig;
|
|
31
39
|
// ---------------------------------------------------------------------------
|
|
32
40
|
// Builder functions
|
|
@@ -105,6 +113,24 @@ function symbol(file, name) {
|
|
|
105
113
|
function ref(path) {
|
|
106
114
|
return { _ref: "skill", path: path };
|
|
107
115
|
}
|
|
116
|
+
/**
|
|
117
|
+
* Reference a directory — verified at compile time to exist AND be a directory
|
|
118
|
+
* (not a file). The "architecture floats free" fix: a spec that names `src/core/`
|
|
119
|
+
* proves the directory is really there, where a plain string in prose rots
|
|
120
|
+
* silently. Compiles to the inline form `` `path` ``.
|
|
121
|
+
*/
|
|
122
|
+
function dir(path) {
|
|
123
|
+
return { _ref: "dir", path: path };
|
|
124
|
+
}
|
|
125
|
+
/**
|
|
126
|
+
* Reference a glob pattern — verified at compile time to match at least one path,
|
|
127
|
+
* so `glob("src/*.test.ts")` proves tests actually exist where the instructions
|
|
128
|
+
* claim (the pattern supports the usual `*` / `**` syntax). Compiles to the
|
|
129
|
+
* inline form `` `pattern` ``.
|
|
130
|
+
*/
|
|
131
|
+
function glob(pattern) {
|
|
132
|
+
return { _ref: "glob", pattern: pattern };
|
|
133
|
+
}
|
|
108
134
|
/**
|
|
109
135
|
* Tagged template literal for skill instructions with typed references.
|
|
110
136
|
*
|
|
@@ -124,6 +150,33 @@ function instructions(strings, ...values) {
|
|
|
124
150
|
}
|
|
125
151
|
return result;
|
|
126
152
|
}
|
|
153
|
+
/**
|
|
154
|
+
* Tagged template literal marking a side-effect boundary — usable as an
|
|
155
|
+
* interpolated fragment inside a body / `instructions\`\``:
|
|
156
|
+
*
|
|
157
|
+
* instructions`
|
|
158
|
+
* ## Apply
|
|
159
|
+
* ${effect`
|
|
160
|
+
* Side effects are allowed ONLY here:
|
|
161
|
+
* - write ${file("CHANGELOG.md")}
|
|
162
|
+
* - ${cmd("npm publish")}
|
|
163
|
+
* `}
|
|
164
|
+
* `
|
|
165
|
+
*
|
|
166
|
+
* Returns an `EffectRegion` fragment; `compile` wraps its rendered body in
|
|
167
|
+
* `<!-- vigiles:effect -->` markers. Independent of the `doc()` authoring
|
|
168
|
+
* surface — it does not block on it.
|
|
169
|
+
*/
|
|
170
|
+
function effect(strings, ...values) {
|
|
171
|
+
const body = [];
|
|
172
|
+
for (let i = 0; i < strings.length; i++) {
|
|
173
|
+
if (strings[i])
|
|
174
|
+
body.push(strings[i]);
|
|
175
|
+
if (i < values.length)
|
|
176
|
+
body.push(values[i]);
|
|
177
|
+
}
|
|
178
|
+
return { _ref: "effect", body };
|
|
179
|
+
}
|
|
127
180
|
/**
|
|
128
181
|
* Define a CLAUDE.md specification.
|
|
129
182
|
*
|
|
@@ -154,6 +207,11 @@ function step(instr, opts = {}) {
|
|
|
154
207
|
*
|
|
155
208
|
* // skills/my-skill/SKILL.md.spec.ts
|
|
156
209
|
* export default skill({ name: "my-skill", description: "...", body: "..." });
|
|
210
|
+
*
|
|
211
|
+
* Generic over a tool `Vocabulary` (default `OpenToolVocabulary` — no
|
|
212
|
+
* constraint), exactly like `agent()`: a vocabulary-bound `skill` (e.g.
|
|
213
|
+
* `vigiles/claude-code`) makes `purity: "pure"` + a side-effecting tool a `tsc`
|
|
214
|
+
* error; the bare core `skill()` accepts any tools, as before.
|
|
157
215
|
*/
|
|
158
216
|
function skill(spec) {
|
|
159
217
|
return { _specType: "skill", ...spec };
|
|
@@ -172,6 +230,17 @@ function skill(spec) {
|
|
|
172
230
|
* "no-floating": enforce("@typescript-eslint/no-floating-promises", "Await promises."),
|
|
173
231
|
* },
|
|
174
232
|
* });
|
|
233
|
+
*
|
|
234
|
+
* Generic over a tool `Vocabulary` (default `OpenToolVocabulary` — no
|
|
235
|
+
* constraint). A harness adapter re-exports a vocabulary-bound `agent` (e.g.
|
|
236
|
+
* `vigiles/claude-code`) so `purity: "pure"` + a side-effecting tool is a `tsc`
|
|
237
|
+
* error at edit time; the bare core `agent()` accepts any tools, as before.
|
|
238
|
+
*
|
|
239
|
+
* Also generic over the result's `Ok`/`Err` shapes, inferred from `output:
|
|
240
|
+
* result(...)`. The returned value is a `TypedAgentSpec<Ok, Err>` — an
|
|
241
|
+
* `AgentSpec` that carries those shapes at the type level, so a typed `pipe`
|
|
242
|
+
* can cross-reference the handoff. With no `output` the shapes default to the
|
|
243
|
+
* erased `Shape`, and the value is still a plain `AgentSpec` — backwards-compatible.
|
|
175
244
|
*/
|
|
176
245
|
function agent(spec) {
|
|
177
246
|
return { _specType: "agent", ...spec };
|
|
@@ -186,15 +255,35 @@ function agent(spec) {
|
|
|
186
255
|
*
|
|
187
256
|
* (Distinct from a skill's `result:` postcondition gate — this types a
|
|
188
257
|
* subagent's *return value*, the success/error tracks of the railway.)
|
|
258
|
+
*
|
|
259
|
+
* The literal field shapes are PRESERVED in the return type (`const` inference),
|
|
260
|
+
* not erased to `Record<string, OutputFieldType>` — this is what lets `pipe`
|
|
261
|
+
* cross-reference one agent's `ok` against the next agent's needs at `tsc` time.
|
|
262
|
+
* The return is still an `OutputContract`, so every existing consumer (the
|
|
263
|
+
* `output:` field, `renderOutputContract`, `parseAgentResult`) is unchanged.
|
|
189
264
|
*/
|
|
190
265
|
function result(ok, err) {
|
|
191
266
|
return { _ref: "output", ok, err };
|
|
192
267
|
}
|
|
193
|
-
/**
|
|
194
|
-
|
|
195
|
-
|
|
268
|
+
/**
|
|
269
|
+
* Build a railway step that dispatches `agent`.
|
|
270
|
+
*
|
|
271
|
+
* delegate("planner") // no task, no handoff
|
|
272
|
+
* delegate("implementer", "implement the plan") // task hint only
|
|
273
|
+
* delegate("reviewer", undefined, needs({ diff: "string" })) // + handoff check
|
|
274
|
+
*
|
|
275
|
+
* The optional 3rd argument carries the step's input `needs` (built by
|
|
276
|
+
* `needs(...)`). When present, the whole-harness registry asserts that the
|
|
277
|
+
* PREVIOUS success-track step's `result().ok` SUPPLIES it — a cross-file
|
|
278
|
+
* handoff that doesn't line up is a `tsc` error naming the offending field.
|
|
279
|
+
* Omitting it (the historical 1-/2-arg call) keeps the exact string-path
|
|
280
|
+
* behavior — fully backwards-compatible.
|
|
281
|
+
*/
|
|
282
|
+
function delegate(agent, task, needsContract) {
|
|
283
|
+
const base = task === undefined
|
|
196
284
|
? { _step: "delegate", agent }
|
|
197
285
|
: { _step: "delegate", agent, task };
|
|
286
|
+
return needsContract === undefined ? base : { ...base, needs: needsContract };
|
|
198
287
|
}
|
|
199
288
|
/**
|
|
200
289
|
* Compose flat subagents into a railway (compiles to an orchestrator command).
|
|
@@ -209,6 +298,80 @@ function delegate(agent, task) {
|
|
|
209
298
|
function railway(spec) {
|
|
210
299
|
return { _specType: "railway", ...spec };
|
|
211
300
|
}
|
|
301
|
+
/**
|
|
302
|
+
* Declare the input fields a step reads from its predecessor's success payload.
|
|
303
|
+
* Pass it as `needs:` on a typed pipeline step. `needs({})` (the default) is a
|
|
304
|
+
* step with no upstream requirement — valid as the FIRST step of a pipeline.
|
|
305
|
+
*
|
|
306
|
+
* needs({ plan: "string", files: "string[]" })
|
|
307
|
+
*/
|
|
308
|
+
function needs(shape) {
|
|
309
|
+
return shape;
|
|
310
|
+
}
|
|
311
|
+
/**
|
|
312
|
+
* Pair a typed agent with the input it `needs` from the previous step. The first
|
|
313
|
+
* argument is an `agent()` VALUE (which carries its `result()` shape); the
|
|
314
|
+
* second is the `needs(...)` input contract.
|
|
315
|
+
*
|
|
316
|
+
* pipeStep(implementer, needs({ plan: "string", files: "string[]" }))
|
|
317
|
+
*/
|
|
318
|
+
function pipeStep(a, needsContract = {}) {
|
|
319
|
+
return { _step: "typed-delegate", agent: a, needs: needsContract };
|
|
320
|
+
}
|
|
321
|
+
/**
|
|
322
|
+
* Begin a typed pipeline from its first step. The first step has no upstream, so
|
|
323
|
+
* its `needs` must be empty (`needs({})` or omitted). Returns a `Pipeline`
|
|
324
|
+
* carrying that step's `ok`/`err` forward.
|
|
325
|
+
*/
|
|
326
|
+
function start(first) {
|
|
327
|
+
const step = "_step" in first ? first : pipeStep(first, {});
|
|
328
|
+
const out = step.agent.output;
|
|
329
|
+
return {
|
|
330
|
+
_specType: "pipeline",
|
|
331
|
+
agents: [step.agent.name],
|
|
332
|
+
ok: (out ? out.ok : {}),
|
|
333
|
+
err: (out ? out.err : {}),
|
|
334
|
+
railway: railway({
|
|
335
|
+
name: step.agent.name,
|
|
336
|
+
steps: [delegate(step.agent.name)],
|
|
337
|
+
}),
|
|
338
|
+
};
|
|
339
|
+
}
|
|
340
|
+
/**
|
|
341
|
+
* Append a step to a typed pipeline. The handoff is CHECKED: the constraint
|
|
342
|
+
* `Supplies<PriorOk, Needs>` must be `true`, else the `next` parameter's type
|
|
343
|
+
* collapses to a `__HANDOFF_ERROR` object and `tsc` rejects the call, naming the
|
|
344
|
+
* missing/mismatched field. Carries the new step's `ok` forward and accumulates
|
|
345
|
+
* the error track. Shallow per-call check — no recursive chain type.
|
|
346
|
+
*
|
|
347
|
+
* Named `andThen` (Wlaschin's railway `bind`/`andThen`), NOT `then`: a module
|
|
348
|
+
* exporting a function called `then` becomes a thenable, so `await import()` of
|
|
349
|
+
* any barrel re-exporting it would invoke it — a footgun the rename avoids.
|
|
350
|
+
*/
|
|
351
|
+
function andThen(prior, next) {
|
|
352
|
+
const real = next;
|
|
353
|
+
const out = real.agent.output;
|
|
354
|
+
const rw = railway({
|
|
355
|
+
name: prior.railway.name,
|
|
356
|
+
steps: [...prior.railway.steps, delegate(real.agent.name)],
|
|
357
|
+
});
|
|
358
|
+
return {
|
|
359
|
+
_specType: "pipeline",
|
|
360
|
+
agents: [...prior.agents, real.agent.name],
|
|
361
|
+
ok: (out ? out.ok : {}),
|
|
362
|
+
err: (out ? out.err : {}),
|
|
363
|
+
railway: rw,
|
|
364
|
+
};
|
|
365
|
+
}
|
|
366
|
+
function pipe(first, ...rest) {
|
|
367
|
+
// The overloads above enforce each handoff at the type level; the runtime body
|
|
368
|
+
// is the same left fold of start/andThen, untyped (the checks already happened).
|
|
369
|
+
let pipeline = start(first);
|
|
370
|
+
for (const s of rest) {
|
|
371
|
+
pipeline = andThen(pipeline, s);
|
|
372
|
+
}
|
|
373
|
+
return pipeline;
|
|
374
|
+
}
|
|
212
375
|
function defineConfig(config) {
|
|
213
376
|
return config;
|
|
214
377
|
}
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
*
|
|
8
8
|
* ONE pure detector (`one-detector-no-drift`), reused by THREE callers so they
|
|
9
9
|
* can't disagree: `compileAgent` (spec authoring), `scan` (read-only audit of a
|
|
10
|
-
* shipped plugin), and the `
|
|
10
|
+
* shipped plugin), and the `subagent-tool-contract` lint rule (the severity-gated
|
|
11
11
|
* commit gate). The dialect is injected (core ⊄ adapter) — the composition root
|
|
12
12
|
* passes `claudeCodeDialect` / `codexDialect`.
|
|
13
13
|
*
|
package/dist/core/types.d.ts
CHANGED
|
@@ -71,7 +71,7 @@ export interface OrphansConfig {
|
|
|
71
71
|
}
|
|
72
72
|
/**
|
|
73
73
|
* Shared options for the per-kind untested-* rules (`untested-skill` /
|
|
74
|
-
* `untested-
|
|
74
|
+
* `untested-subagent` / `untested-hook`). Which kinds are scanned is controlled by
|
|
75
75
|
* each rule's severity (set a rule to `false` to skip that kind), so only the
|
|
76
76
|
* test-discovery knobs live here.
|
|
77
77
|
*/
|
|
@@ -97,7 +97,7 @@ export interface RulesConfig {
|
|
|
97
97
|
/** Flag a skill (SKILL.md) that ships with no test or eval. Default: "warn". */
|
|
98
98
|
"untested-skill"?: RuleWithOptions<TestCoverageConfig>;
|
|
99
99
|
/** Flag a subagent (agents/*.md) that ships with no test or eval. Default: "warn". */
|
|
100
|
-
"untested-
|
|
100
|
+
"untested-subagent"?: RuleWithOptions<TestCoverageConfig>;
|
|
101
101
|
/** Flag a hook script that ships with no test or eval. Default: "warn". */
|
|
102
102
|
"untested-hook"?: RuleWithOptions<TestCoverageConfig>;
|
|
103
103
|
/**
|
|
@@ -115,7 +115,7 @@ export interface RulesConfig {
|
|
|
115
115
|
* plugin/MCP-provided, never flagged). Off unless set; "warn" surfaces,
|
|
116
116
|
* "error" gates CI. Same detector as `scan` + `compileAgent`.
|
|
117
117
|
*/
|
|
118
|
-
"
|
|
118
|
+
"subagent-tool-contract"?: RuleSeverity;
|
|
119
119
|
/**
|
|
120
120
|
* Flag a hook registered under an event name the harness doesn't define (a
|
|
121
121
|
* typo → the hook never fires). High-precision: close typos only, never a
|
|
@@ -128,7 +128,7 @@ export interface RulesConfig {
|
|
|
128
128
|
* `name` (to load), an agent needs `name` + `description`. A broken surface
|
|
129
129
|
* that won't register. Default "warn"; "error" gates CI. Same detector as `scan`.
|
|
130
130
|
*/
|
|
131
|
-
"
|
|
131
|
+
"subagent-frontmatter"?: RuleSeverity;
|
|
132
132
|
/**
|
|
133
133
|
* Flag a declared MCP server that can't start — neither a `command` (stdio)
|
|
134
134
|
* nor a `url` (http/sse). Default "warn"; "error" gates CI. Same detector as
|
|
@@ -146,7 +146,7 @@ export interface RulesConfig {
|
|
|
146
146
|
/**
|
|
147
147
|
* Cross-reference an `mcp__server__tool` in a subagent's contract against the
|
|
148
148
|
* plugin's declared `mcpServers` — flag a server the plugin doesn't declare
|
|
149
|
-
* (the MCP half of the tool moat; `
|
|
149
|
+
* (the MCP half of the tool moat; `subagent-tool-contract` checks the built-in
|
|
150
150
|
* half). High-precision: only flags when the plugin SHIPS a declared set,
|
|
151
151
|
* allowlists harness built-ins (`ide`), and skips the plugin-namespaced
|
|
152
152
|
* `mcp__plugin_…` form. Default "warn"; "error" gates CI. Same detector as
|
|
@@ -163,7 +163,7 @@ export interface RulesConfig {
|
|
|
163
163
|
"hook-script-exists"?: RuleSeverity;
|
|
164
164
|
/**
|
|
165
165
|
* Cross-reference a subagent's `disallowedTools:` block-list against the
|
|
166
|
-
* catalog — the deny-side mirror of `
|
|
166
|
+
* catalog — the deny-side mirror of `subagent-tool-contract`. A close typo there
|
|
167
167
|
* blocks NOTHING (you meant to deny `Bash`, wrote `Bsh`), leaving the tool
|
|
168
168
|
* available. High-precision: close-typo only (a never-available tool is
|
|
169
169
|
* harmless to list, a bare unknown is likely a plugin tool). Default "warn";
|
package/dist/core/validate.js
CHANGED
|
@@ -43,15 +43,15 @@ const DEFAULT_RULES = {
|
|
|
43
43
|
coverage: false,
|
|
44
44
|
// Per-kind surface-coverage: a skill/agent/hook must ship with a test or eval.
|
|
45
45
|
"untested-skill": "warn",
|
|
46
|
-
"untested-
|
|
46
|
+
"untested-subagent": "warn",
|
|
47
47
|
"untested-hook": "warn",
|
|
48
48
|
"unmarked-refs": "warn",
|
|
49
49
|
// High-precision (never-available + close typos only), so on by default at warn.
|
|
50
|
-
"
|
|
50
|
+
"subagent-tool-contract": "warn",
|
|
51
51
|
// High-precision (close typos only), on by default at warn.
|
|
52
52
|
"hook-events": "warn",
|
|
53
53
|
// Missing required frontmatter (name/description) — on by default at warn.
|
|
54
|
-
"
|
|
54
|
+
"subagent-frontmatter": "warn",
|
|
55
55
|
// A declared MCP server with no command/url can't start — on by default at warn.
|
|
56
56
|
"mcp-config": "warn",
|
|
57
57
|
// Best-practice nudge (skills load without frontmatter) — warn, not error.
|
|
@@ -60,7 +60,7 @@ const DEFAULT_RULES = {
|
|
|
60
60
|
"mcp-tool-resolves": "warn",
|
|
61
61
|
// A hook script referenced but missing never runs — on by default at warn.
|
|
62
62
|
"hook-script-exists": "warn",
|
|
63
|
-
// High-precision (close-typo only) deny-list mirror of
|
|
63
|
+
// High-precision (close-typo only) deny-list mirror of subagent-tool-contract.
|
|
64
64
|
"disallowed-tools-contract": "warn",
|
|
65
65
|
// Deterministic NCD precision proxy (near-identical skill descriptions) — warn.
|
|
66
66
|
"description-overlap": "warn",
|
package/dist/harness-test.d.ts
CHANGED
|
@@ -110,6 +110,13 @@ export interface SubagentTrace {
|
|
|
110
110
|
readonly name: string;
|
|
111
111
|
/** The tools the subagent invoked (events tagged with the Task's id). */
|
|
112
112
|
readonly toolCalls: readonly ToolCall[];
|
|
113
|
+
/**
|
|
114
|
+
* The subagent's RETURNED text — the dispatch tool_result the orchestrator
|
|
115
|
+
* receives back. This is where a `result()` contract's `vigiles:ok`/`vigiles:err`
|
|
116
|
+
* block lands, so `subagent(name, [output(/vigiles:ok/)])` can assert the typed
|
|
117
|
+
* outcome. "" if not captured.
|
|
118
|
+
*/
|
|
119
|
+
readonly output: string;
|
|
113
120
|
}
|
|
114
121
|
export interface HarnessTestResult extends Trace {
|
|
115
122
|
readonly exitCode: number;
|
package/dist/harness-test.js
CHANGED
|
@@ -130,6 +130,7 @@ function parseToolCalls(streamJson) {
|
|
|
130
130
|
*/
|
|
131
131
|
function parseSubagents(streamJson) {
|
|
132
132
|
const tasks = new Map(); // dispatch id → subagent name
|
|
133
|
+
const dispatchOutput = new Map(); // dispatch id → returned text
|
|
133
134
|
const byParent = new Map();
|
|
134
135
|
const groupFor = (parent) => {
|
|
135
136
|
let g = byParent.get(parent);
|
|
@@ -162,7 +163,10 @@ function parseSubagents(streamJson) {
|
|
|
162
163
|
// A subagent dispatch is any top-level tool_use whose input carries a
|
|
163
164
|
// `subagent_type` — the dispatch tool is named "Agent" on the live CLI
|
|
164
165
|
// (older docs say "Task"), so match the input field, NOT the tool name,
|
|
165
|
-
// to survive the rename. Confirmed against real claude output.
|
|
166
|
+
// to survive the rename. Confirmed against real claude output. CC NOTE:
|
|
167
|
+
// under `--plugin-dir` the value is NAMESPACED `plugin:agent` (captured
|
|
168
|
+
// "reviewer-spec:code-reviewer"); the bare agent name is matched in the
|
|
169
|
+
// `subagent()` check (src/check.ts), so the full id is preserved here.
|
|
166
170
|
const sub = b.input?.subagent_type;
|
|
167
171
|
if (typeof sub === "string")
|
|
168
172
|
tasks.set(id, sub);
|
|
@@ -170,12 +174,20 @@ function parseSubagents(streamJson) {
|
|
|
170
174
|
if (parent)
|
|
171
175
|
groupFor(parent).uses.push({ id, name: b.name, input: b.input });
|
|
172
176
|
}
|
|
173
|
-
else if (b.type === "tool_result"
|
|
177
|
+
else if (b.type === "tool_result") {
|
|
174
178
|
const id = typeof b.tool_use_id === "string" ? b.tool_use_id : "";
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
+
if (parent) {
|
|
180
|
+
groupFor(parent).results.set(id, {
|
|
181
|
+
text: contentText(b.content),
|
|
182
|
+
isError: b.is_error === true,
|
|
183
|
+
});
|
|
184
|
+
}
|
|
185
|
+
else if (id) {
|
|
186
|
+
// A top-level tool_result whose id is a subagent dispatch is the SUB's
|
|
187
|
+
// RETURN to the orchestrator (where a result() vigiles:ok/err block
|
|
188
|
+
// lands). Record it; matched to its dispatch by id below.
|
|
189
|
+
dispatchOutput.set(id, contentText(b.content));
|
|
190
|
+
}
|
|
179
191
|
}
|
|
180
192
|
}
|
|
181
193
|
}
|
|
@@ -188,7 +200,7 @@ function parseSubagents(streamJson) {
|
|
|
188
200
|
resultText: g?.results.get(u.id)?.text ?? "",
|
|
189
201
|
isError: g?.results.get(u.id)?.isError ?? false,
|
|
190
202
|
}));
|
|
191
|
-
out.push({ name, toolCalls });
|
|
203
|
+
out.push({ name, toolCalls, output: dispatchOutput.get(taskId) ?? "" });
|
|
192
204
|
}
|
|
193
205
|
return out;
|
|
194
206
|
}
|
package/dist/leaderboard.d.ts
CHANGED
|
@@ -21,6 +21,8 @@ export interface PluginScore {
|
|
|
21
21
|
readonly issues: readonly string[];
|
|
22
22
|
readonly report: ScanReport;
|
|
23
23
|
}
|
|
24
|
+
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
25
|
+
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
24
26
|
/** Deterministic structural-health score for one scanned plugin. */
|
|
25
27
|
export declare function scoreReport(r: ScanReport): {
|
|
26
28
|
score: number;
|
package/dist/leaderboard.js
CHANGED
|
@@ -12,6 +12,7 @@
|
|
|
12
12
|
* model and stack on top later; this part runs anywhere in CI for free.
|
|
13
13
|
*/
|
|
14
14
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
15
|
+
exports.gradeFor = gradeFor;
|
|
15
16
|
exports.scoreReport = scoreReport;
|
|
16
17
|
exports.rankPlugins = rankPlugins;
|
|
17
18
|
exports.formatLeaderboard = formatLeaderboard;
|
|
@@ -23,6 +24,7 @@ const W_NO_DESCRIPTION = 10; // a skill with no usable description → can't tri
|
|
|
23
24
|
const W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
24
25
|
const W_NO_CONTRACT = 5; // an agent with no `tools:` line → inherits everything
|
|
25
26
|
const W_UNTESTED = 3; // a surface with no test/eval → warning-tier
|
|
27
|
+
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
26
28
|
function gradeFor(score) {
|
|
27
29
|
if (score >= 90)
|
|
28
30
|
return "A";
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The per-repo harness optimizer's DETERMINISTIC spine — shipped as the
|
|
3
|
+
* `vigiles scan --fix-plan` lens (NOT its own `optimize` verb: until the measured
|
|
4
|
+
* A/B half lands, an "optimizer" that only re-prints scan's findings doesn't earn
|
|
5
|
+
* a separate command, so it's folded into scan as one more view on the same
|
|
6
|
+
* report; see research/roadmap.md §P2 "reconsider an `optimize` verb").
|
|
7
|
+
*
|
|
8
|
+
* A2 in the measurement-authority pivot is the ADOPTION product: measure a user's
|
|
9
|
+
* own skills/model/rules on their tasks and recommend add/drop/swap with a MEASURED
|
|
10
|
+
* delta. The measured delta is the real-model layer (gated on the Pro/Max
|
|
11
|
+
* subscription, costs tokens); this v0 ships the deterministic HALF — the free
|
|
12
|
+
* pre-filter that runs on every commit with no model.
|
|
13
|
+
*
|
|
14
|
+
* It answers "what should I fix in my harness, and why" using only the cross-ref
|
|
15
|
+
* findings the linter already computes: it reuses `scoreReport` for the headline
|
|
16
|
+
* structural-health score and `explainScore` for the per-surface cause + one-line
|
|
17
|
+
* fix (one-detector-no-drift — it never re-detects). The result is a prioritized,
|
|
18
|
+
* typed action list — the spine the measured A/B (the behavioral delta) stacks on.
|
|
19
|
+
*
|
|
20
|
+
* This is the "linting as a free pre-filter to measurement" thesis made a command:
|
|
21
|
+
* clear the structural dead-ends a model can't help with FIRST (free, certain),
|
|
22
|
+
* THEN spend tokens measuring whether the structurally-clean skills earn their keep.
|
|
23
|
+
*
|
|
24
|
+
* Distinct from `vigiles explain` (which diagnoses ONE underperforming surface a
|
|
25
|
+
* measurement flagged): `optimize` is the whole-repo adoption view — health score +
|
|
26
|
+
* the ranked fix list + the hand-off to the measured layer. Same findings, the
|
|
27
|
+
* optimization framing. See research/measurement-authority.md (A2) + roadmap §P1.
|
|
28
|
+
*/
|
|
29
|
+
import { type PluginScore } from "./leaderboard.js";
|
|
30
|
+
import { type ExplanationConfidence } from "./score-explainer.js";
|
|
31
|
+
import type { ScanReport } from "./scan.js";
|
|
32
|
+
/**
|
|
33
|
+
* The action a recommendation asks for. The deterministic detectors yield two:
|
|
34
|
+
* `"differentiate"` (a description-overlap PAIR — make the two distinct so the
|
|
35
|
+
* selector can disambiguate) and `"fix"` (every other structural dead-end —
|
|
36
|
+
* correct/add/remove the offending bit). The richer add/drop/swap vocabulary
|
|
37
|
+
* belongs to the MEASURED layer, once a behavioral delta ranks the alternatives.
|
|
38
|
+
*/
|
|
39
|
+
export type OptimizeAction = "fix" | "differentiate";
|
|
40
|
+
export interface Recommendation {
|
|
41
|
+
/** The affected surface (a skill/agent/hook name or path, or an `a ↔ b` pair). */
|
|
42
|
+
readonly surface: string;
|
|
43
|
+
readonly action: OptimizeAction;
|
|
44
|
+
/** The deterministic cause (the detector's own message — no drift). */
|
|
45
|
+
readonly rationale: string;
|
|
46
|
+
/** A single, actionable fix. */
|
|
47
|
+
readonly fix: string;
|
|
48
|
+
/** The lint rule that found it (open `docs/rules/<detector>.md`). */
|
|
49
|
+
readonly detector: string;
|
|
50
|
+
readonly confidence: ExplanationConfidence;
|
|
51
|
+
}
|
|
52
|
+
export interface OptimizeReport {
|
|
53
|
+
readonly dir: string;
|
|
54
|
+
/** Structural-health score 0–100 (the same `scoreReport` the leaderboard uses). */
|
|
55
|
+
readonly score: number;
|
|
56
|
+
readonly grade: PluginScore["grade"];
|
|
57
|
+
/** The free deterministic fixes, `likely` dead-ends first (explainScore's order). */
|
|
58
|
+
readonly recommendations: readonly Recommendation[];
|
|
59
|
+
/**
|
|
60
|
+
* No loadable surface at all (not a plugin, or a broken load) — distinct from a
|
|
61
|
+
* clean-and-loaded harness with zero findings, so the formatter doesn't call an
|
|
62
|
+
* EMPTY machine "clean". Mirrors `scoreReport`'s empty-machine case.
|
|
63
|
+
*/
|
|
64
|
+
readonly empty: boolean;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Turn a scan report into a prioritized optimization plan: the structural-health
|
|
68
|
+
* score + a typed recommendation per deterministic finding (`likely` dead-ends
|
|
69
|
+
* before `possible` proxies, via explainScore's own ordering). Pure over the report.
|
|
70
|
+
*/
|
|
71
|
+
export declare function optimize(report: ScanReport): OptimizeReport;
|
|
72
|
+
/** Render an optimization plan for the CLI. */
|
|
73
|
+
export declare function formatOptimize(rep: OptimizeReport): string;
|
|
74
|
+
//# sourceMappingURL=optimize.d.ts.map
|
package/dist/optimize.js
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.optimize = optimize;
|
|
4
|
+
exports.formatOptimize = formatOptimize;
|
|
5
|
+
/**
|
|
6
|
+
* The per-repo harness optimizer's DETERMINISTIC spine — shipped as the
|
|
7
|
+
* `vigiles scan --fix-plan` lens (NOT its own `optimize` verb: until the measured
|
|
8
|
+
* A/B half lands, an "optimizer" that only re-prints scan's findings doesn't earn
|
|
9
|
+
* a separate command, so it's folded into scan as one more view on the same
|
|
10
|
+
* report; see research/roadmap.md §P2 "reconsider an `optimize` verb").
|
|
11
|
+
*
|
|
12
|
+
* A2 in the measurement-authority pivot is the ADOPTION product: measure a user's
|
|
13
|
+
* own skills/model/rules on their tasks and recommend add/drop/swap with a MEASURED
|
|
14
|
+
* delta. The measured delta is the real-model layer (gated on the Pro/Max
|
|
15
|
+
* subscription, costs tokens); this v0 ships the deterministic HALF — the free
|
|
16
|
+
* pre-filter that runs on every commit with no model.
|
|
17
|
+
*
|
|
18
|
+
* It answers "what should I fix in my harness, and why" using only the cross-ref
|
|
19
|
+
* findings the linter already computes: it reuses `scoreReport` for the headline
|
|
20
|
+
* structural-health score and `explainScore` for the per-surface cause + one-line
|
|
21
|
+
* fix (one-detector-no-drift — it never re-detects). The result is a prioritized,
|
|
22
|
+
* typed action list — the spine the measured A/B (the behavioral delta) stacks on.
|
|
23
|
+
*
|
|
24
|
+
* This is the "linting as a free pre-filter to measurement" thesis made a command:
|
|
25
|
+
* clear the structural dead-ends a model can't help with FIRST (free, certain),
|
|
26
|
+
* THEN spend tokens measuring whether the structurally-clean skills earn their keep.
|
|
27
|
+
*
|
|
28
|
+
* Distinct from `vigiles explain` (which diagnoses ONE underperforming surface a
|
|
29
|
+
* measurement flagged): `optimize` is the whole-repo adoption view — health score +
|
|
30
|
+
* the ranked fix list + the hand-off to the measured layer. Same findings, the
|
|
31
|
+
* optimization framing. See research/measurement-authority.md (A2) + roadmap §P1.
|
|
32
|
+
*/
|
|
33
|
+
const leaderboard_js_1 = require("./leaderboard.js");
|
|
34
|
+
const score_explainer_js_1 = require("./score-explainer.js");
|
|
35
|
+
function actionFor(e) {
|
|
36
|
+
return e.symptom === "wrong-skill-fires" ? "differentiate" : "fix";
|
|
37
|
+
}
|
|
38
|
+
function isEmptyMachine(r) {
|
|
39
|
+
const surfaces = r.skills.length + r.agents.length + r.hooks.length + r.commands;
|
|
40
|
+
return surfaces === 0 && !r.mcp;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Turn a scan report into a prioritized optimization plan: the structural-health
|
|
44
|
+
* score + a typed recommendation per deterministic finding (`likely` dead-ends
|
|
45
|
+
* before `possible` proxies, via explainScore's own ordering). Pure over the report.
|
|
46
|
+
*/
|
|
47
|
+
function optimize(report) {
|
|
48
|
+
const { score } = (0, leaderboard_js_1.scoreReport)(report);
|
|
49
|
+
const recommendations = (0, score_explainer_js_1.explainScore)(report).map((e) => ({
|
|
50
|
+
surface: e.surface,
|
|
51
|
+
action: actionFor(e),
|
|
52
|
+
rationale: e.cause,
|
|
53
|
+
fix: e.fix,
|
|
54
|
+
detector: e.detector,
|
|
55
|
+
confidence: e.confidence,
|
|
56
|
+
}));
|
|
57
|
+
return {
|
|
58
|
+
dir: report.dir,
|
|
59
|
+
score,
|
|
60
|
+
grade: (0, leaderboard_js_1.gradeFor)(score),
|
|
61
|
+
recommendations,
|
|
62
|
+
empty: isEmptyMachine(report),
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
const ACTION_LABEL = {
|
|
66
|
+
fix: "FIX",
|
|
67
|
+
differentiate: "DIFFERENTIATE",
|
|
68
|
+
};
|
|
69
|
+
const measureHint = (dir) => `\`vigiles scan ${dir} --trigger\` — real-model, runs on your subscription`;
|
|
70
|
+
/** Render an optimization plan for the CLI. */
|
|
71
|
+
function formatOptimize(rep) {
|
|
72
|
+
const head = `Harness health: ${String(rep.score)}/100 (${rep.grade}) — ${rep.dir}`;
|
|
73
|
+
if (rep.empty) {
|
|
74
|
+
return `${head}\n\nNothing loaded — this isn't a plugin/harness, or the load failed. Point optimize at a dir with a CLAUDE.md/AGENTS.md, skills, agents, or hooks.`;
|
|
75
|
+
}
|
|
76
|
+
if (rep.recommendations.length === 0) {
|
|
77
|
+
return `${head}\n\nNo deterministic fixes found — the structure is clean. Whether your skills actually help is a BEHAVIORAL question; measure it with ${measureHint(rep.dir)}.`;
|
|
78
|
+
}
|
|
79
|
+
const lines = [
|
|
80
|
+
head,
|
|
81
|
+
"",
|
|
82
|
+
`${String(rep.recommendations.length)} deterministic fix(es) — free, no model. Apply these before measuring:`,
|
|
83
|
+
"",
|
|
84
|
+
];
|
|
85
|
+
for (const r of rep.recommendations) {
|
|
86
|
+
const mark = r.confidence === "likely" ? "✗" : "⚠";
|
|
87
|
+
lines.push(`${mark} [${ACTION_LABEL[r.action]}] ${r.surface}`);
|
|
88
|
+
lines.push(` why: ${r.rationale} [${r.detector}]`);
|
|
89
|
+
lines.push(` → ${r.fix}`);
|
|
90
|
+
}
|
|
91
|
+
lines.push("", `Then measure the behavioral delta of what's left (does each skill earn its keep?) with ${measureHint(rep.dir)}.`);
|
|
92
|
+
return lines.join("\n");
|
|
93
|
+
}
|
|
94
|
+
//# sourceMappingURL=optimize.js.map
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
export type SurfaceKind = "skill" | "agent" | "hook";
|
|
2
|
+
/** The cheapest meaningful tier for a surface kind (mirrors the test-harness skill). */
|
|
3
|
+
export type TestTier = "unit" | "harness" | "eval";
|
|
4
|
+
/** One field of a subagent's `result()` contract, parsed from its compiled `.md`. */
|
|
5
|
+
export interface ContractField {
|
|
6
|
+
readonly name: string;
|
|
7
|
+
/** The `OutputFieldType` literal: `"string" | "number" | "boolean" | "string[]"`. */
|
|
8
|
+
readonly type: string;
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* A subagent's typed `result(ok, err)` outcome contract — the typed-spec payoff
|
|
12
|
+
* the generator turns into a real `assertAgentOk` test (no LLM judge). Parsed from
|
|
13
|
+
* the `vigiles:ok` / `vigiles:err` blocks the compiler emits into the agent `.md`.
|
|
14
|
+
*/
|
|
15
|
+
export interface ResultContract {
|
|
16
|
+
readonly ok: readonly ContractField[];
|
|
17
|
+
readonly err: readonly ContractField[];
|
|
18
|
+
}
|
|
19
|
+
/** What the generator needs to know about a surface to scaffold its test. */
|
|
20
|
+
export interface ScaffoldInput {
|
|
21
|
+
readonly kind: SurfaceKind;
|
|
22
|
+
/** Skill dir name / agent name / hook-script basename. */
|
|
23
|
+
readonly name: string;
|
|
24
|
+
/** Repo-relative path to the surface (SKILL.md / agent .md / hook script). */
|
|
25
|
+
readonly path: string;
|
|
26
|
+
/** Plugin name for the namespaced skill id; a placeholder TODO when unknown. */
|
|
27
|
+
readonly pluginName?: string;
|
|
28
|
+
/** A user-invoked skill — trigger-rate is for model-invocable skills, so note it. */
|
|
29
|
+
readonly userInvoked?: boolean;
|
|
30
|
+
/** A subagent's declared tool contract (drives the assertion hint); null = inherits all. */
|
|
31
|
+
readonly tools?: readonly string[] | null;
|
|
32
|
+
/**
|
|
33
|
+
* The side-effecting tools in the contract (`effectSurface(tools).sideEffecting`,
|
|
34
|
+
* computed by the CLI with the resolved dialect — kept harness-agnostic here). A
|
|
35
|
+
* non-empty list drives a generated SAFETY check: the agent's "hole" is mocked/denied
|
|
36
|
+
* and asserted to stay in its lane — the typed `tools` contract writing its own test.
|
|
37
|
+
*/
|
|
38
|
+
readonly sideEffectingTools?: readonly string[];
|
|
39
|
+
/**
|
|
40
|
+
* The subagent's `result()` contract parsed from its compiled `.md`. Present →
|
|
41
|
+
* the generator emits a deterministic `assertAgentOk` outcome test (no LLM judge);
|
|
42
|
+
* the single most-concrete "the typed spec wrote your test" payoff.
|
|
43
|
+
*/
|
|
44
|
+
readonly resultContract?: ResultContract | null;
|
|
45
|
+
/** How the CLI invokes the hook (e.g. `bash hooks/pre-edit.sh`); a TODO when unknown. */
|
|
46
|
+
readonly hookCommand?: string;
|
|
47
|
+
}
|
|
48
|
+
export interface Scaffold {
|
|
49
|
+
readonly path: string;
|
|
50
|
+
readonly content: string;
|
|
51
|
+
readonly kind: SurfaceKind;
|
|
52
|
+
readonly tier: TestTier;
|
|
53
|
+
}
|
|
54
|
+
/** Scaffold a starter test for one surface. Pure: path + content, no I/O. */
|
|
55
|
+
export declare function scaffoldTest(input: ScaffoldInput): Scaffold;
|
|
56
|
+
/** Render a set of scaffolds for the CLI (what was generated, where, which tier). */
|
|
57
|
+
export declare function formatScaffolds(scaffolds: readonly Scaffold[]): string;
|
|
58
|
+
//# sourceMappingURL=scaffold-test.d.ts.map
|