vigiles 27.1.7 → 27.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapter-conformance.js +15 -3
- package/dist/adapters/claude-code/dialect.js +30 -0
- package/dist/adapters/claude-code/event-capability.d.ts +20 -0
- package/dist/adapters/claude-code/event-capability.js +81 -0
- package/dist/adapters/claude-code/hook-protocol.js +26 -3
- package/dist/adapters/claude-code/run-scripts.js +47 -8
- package/dist/adapters/codex/dialect.js +19 -0
- package/dist/cli-main.js +100 -11
- package/dist/core/adopt.js +23 -5
- package/dist/core/compile.d.ts +40 -1
- package/dist/core/compile.js +76 -2
- package/dist/core/dialect.d.ts +43 -1
- package/dist/core/event-capability.d.ts +128 -0
- package/dist/core/event-capability.js +114 -0
- package/dist/core/hook-program.d.ts +38 -0
- package/dist/core/hook-program.js +103 -0
- package/dist/core/hook-protocol.d.ts +19 -3
- package/dist/core/instruction-weight.d.ts +86 -0
- package/dist/core/instruction-weight.js +86 -0
- package/dist/core/linters.js +3 -3
- package/dist/core/test-utils.d.ts +1 -2
- package/dist/core/test-utils.js +7 -9
- package/dist/core/tmp-root.d.ts +12 -0
- package/dist/core/tmp-root.js +59 -0
- package/dist/core/vocabulary-consistency.js +10 -0
- package/dist/eval.js +6 -5
- package/dist/harness-test.js +2 -2
- package/dist/hook-install.d.ts +9 -7
- package/dist/hook-install.js +75 -16
- package/dist/hook-runtime.js +4 -1
- package/dist/posix-path.js +1 -1
- package/dist/run-script.js +3 -3
- package/dist/sandbox.js +2 -2
- package/dist/scan-behavioral.js +2 -2
- package/dist/scan-files.js +10 -3
- package/dist/scan.d.ts +9 -1
- package/dist/scan.js +96 -3
- package/dist/setup-plan.d.ts +8 -0
- package/dist/test.d.ts +1 -0
- package/dist/test.js +17 -2
- package/package.json +1 -1
- package/skills/adopt-spec/SKILL.md +7 -1
- package/skills/edit-spec/SKILL.md +13 -0
|
@@ -13,12 +13,13 @@ exports.assertAdapterLoadsHooks = assertAdapterLoadsHooks;
|
|
|
13
13
|
* in their test suite. See `docs/authoring-an-adapter.md`.
|
|
14
14
|
*/
|
|
15
15
|
const node_fs_1 = require("node:fs");
|
|
16
|
-
const node_os_1 = require("node:os");
|
|
17
16
|
const node_path_1 = require("node:path");
|
|
18
17
|
const compile_js_1 = require("./core/compile.js");
|
|
19
18
|
const spec_js_1 = require("./core/spec.js");
|
|
20
19
|
const plugin_loader_js_1 = require("./plugin-loader.js");
|
|
21
20
|
const vocabulary_consistency_js_1 = require("./core/vocabulary-consistency.js");
|
|
21
|
+
const event_capability_js_1 = require("./core/event-capability.js");
|
|
22
|
+
const tmp_root_js_1 = require("./core/tmp-root.js");
|
|
22
23
|
/** Check an adapter against the port contracts; returns the (possibly empty) failure list. */
|
|
23
24
|
function checkAdapterConformance(adapter) {
|
|
24
25
|
const failures = [];
|
|
@@ -89,7 +90,18 @@ function checkAdapterConformance(adapter) {
|
|
|
89
90
|
// deliver an inject hook?" a tested contract — the gap that let Codex's
|
|
90
91
|
// inject support sit unverified in prose. Empty would mean the harness
|
|
91
92
|
// can't inject context from a hook at all; every harness we support can.
|
|
92
|
-
|
|
93
|
+
// eslint-disable-next-line @typescript-eslint/no-deprecated -- the legacy list is the FALLBACK for an adapter with no capability table, and the thing checked for drift below
|
|
94
|
+
const declaredInject = adapter.hookProtocol.injectableEvents;
|
|
95
|
+
need((0, event_capability_js_1.injectableEventsOf)(adapter.dialect, declaredInject).length > 0, "hookProtocol.injectableEvents is empty — a shell-hook harness must declare the events that honor additionalContext injection (or it can't deliver an inject/nudge hook)");
|
|
96
|
+
// …and when an adapter declares BOTH, they must agree. Without this the
|
|
97
|
+
// table silently COVERS FOR a broken list: an adapter could ship
|
|
98
|
+
// `injectableEvents: []` and still pass, because the effective answer came
|
|
99
|
+
// from the dialect. Two sources that disagree are worse than one, and this
|
|
100
|
+
// is the assertion that keeps the deprecation honest rather than lossy.
|
|
101
|
+
if (adapter.dialect.eventCapabilities) {
|
|
102
|
+
const derived = [...(0, event_capability_js_1.injectableEventsOf)(adapter.dialect, [])].sort();
|
|
103
|
+
need(derived.join("|") === [...declaredInject].sort().join("|"), `hookProtocol.injectableEvents disagrees with dialect.eventCapabilities — the list says [${[...declaredInject].sort().join(", ")}], the table says [${derived.join(", ")}]; they describe the same fact and must match`);
|
|
104
|
+
}
|
|
93
105
|
portNames.push(["hookProtocol", adapter.hookProtocol.name]);
|
|
94
106
|
}
|
|
95
107
|
}
|
|
@@ -154,7 +166,7 @@ function assertHarnessTestable(adapter) {
|
|
|
154
166
|
* run with zero hooks. Does filesystem IO, so it's a separate opt-in assert.
|
|
155
167
|
*/
|
|
156
168
|
function assertAdapterLoadsHooks(adapter) {
|
|
157
|
-
const dir = (0,
|
|
169
|
+
const dir = (0, tmp_root_js_1.makeTmpDir)("conformance");
|
|
158
170
|
try {
|
|
159
171
|
const settingsAbs = (0, node_path_1.join)(dir, adapter.layout.settingsPath);
|
|
160
172
|
(0, node_fs_1.mkdirSync)((0, node_path_1.dirname)(settingsAbs), { recursive: true });
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
3
|
exports.claudeCodeDialect = exports.claudeCodeSideEffectingTools = exports.claudeCodeBuiltinAgentTools = void 0;
|
|
4
|
+
const event_capability_js_1 = require("./event-capability.js");
|
|
4
5
|
const vocabulary_js_1 = require("./vocabulary.js");
|
|
5
6
|
/**
|
|
6
7
|
* The Claude Code built-in subagent tool catalog as a `const` tuple, so a typed
|
|
@@ -61,6 +62,35 @@ exports.claudeCodeDialect = {
|
|
|
61
62
|
"Notification",
|
|
62
63
|
"PreCompact",
|
|
63
64
|
],
|
|
65
|
+
// What Claude Code loads unasked, and what it does with too much of it.
|
|
66
|
+
//
|
|
67
|
+
// WARNS, does not truncate: `/doctor` reports "Large file will impact
|
|
68
|
+
// performance" and the instructions still reach the model. So being over
|
|
69
|
+
// budget here costs money and attention — not rules. (Codex is the opposite;
|
|
70
|
+
// see its dialect, and see why `onExceed` is reported at all.)
|
|
71
|
+
//
|
|
72
|
+
// 🔴 `alwaysLoaded` IS THE POINT, and `.claude/rules/**` is in it on a
|
|
73
|
+
// MEASUREMENT, not a doc: a consumer repo moved 225 837 characters out of
|
|
74
|
+
// CLAUDE.md into that directory and the request cost did not move, because
|
|
75
|
+
// the harness loads it either way. A per-file check would have called that
|
|
76
|
+
// split a success.
|
|
77
|
+
instructionBudget: {
|
|
78
|
+
unit: "chars",
|
|
79
|
+
limit: 40000,
|
|
80
|
+
onExceed: "warns",
|
|
81
|
+
capturedFrom: "claude-code /doctor large-file warning; .claude/rules/** measured 2026-09-16 in a consumer repo",
|
|
82
|
+
alwaysLoaded: [
|
|
83
|
+
"CLAUDE.md",
|
|
84
|
+
"CLAUDE.local.md",
|
|
85
|
+
".claude/CLAUDE.md",
|
|
86
|
+
".claude/rules/**",
|
|
87
|
+
],
|
|
88
|
+
},
|
|
89
|
+
// The capability table — what each event CARRIES and HONOURS. Nine of the 31
|
|
90
|
+
// events, each with its basis; see ./event-capability.ts. The three flat lists
|
|
91
|
+
// above are kept for now and are ASSERTED against this table in
|
|
92
|
+
// event-capability.test.ts, so there is one source rather than two truths.
|
|
93
|
+
eventCapabilities: event_capability_js_1.claudeCodeEventCapabilities,
|
|
64
94
|
// PreToolUse is the one event whose deny needs the structured
|
|
65
95
|
// `hookSpecificOutput.permissionDecision:"deny"`; the legacy top-level
|
|
66
96
|
// `decision` field is ignored there.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Claude Code's hook-event capability table — what each event carries and what
|
|
3
|
+
* it honours. The CAPTURE half of `core/event-capability.ts`; see that module
|
|
4
|
+
* for why the table is partial and why `unknown` is fail-open.
|
|
5
|
+
*
|
|
6
|
+
* NINE of the vendor's 31 events are recorded here, and the other 22 are absent
|
|
7
|
+
* ON PURPOSE. These nine are the ones a capability is actually KNOWN for —
|
|
8
|
+
* either documented by the vendor or measured against a real binary — and every
|
|
9
|
+
* row below carries which. The 22 absent ones are not "no capabilities"; they
|
|
10
|
+
* are "we have not looked", and the classifier says so rather than guessing.
|
|
11
|
+
*
|
|
12
|
+
* Adding a row is a claim about somebody else's product, so it needs a basis in
|
|
13
|
+
* the comment beside it: `doc` (the vendor documents it) or `measured <date> on
|
|
14
|
+
* <version>` (we ran it). "Probably the same as its sibling event" is not a
|
|
15
|
+
* basis — `SubagentStop` is in here on the vendor's own wording, not on its
|
|
16
|
+
* resemblance to `Stop`.
|
|
17
|
+
*/
|
|
18
|
+
import type { EventCapabilityTable } from "../../core/event-capability.js";
|
|
19
|
+
export declare const claudeCodeEventCapabilities: EventCapabilityTable;
|
|
20
|
+
//# sourceMappingURL=event-capability.d.ts.map
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.claudeCodeEventCapabilities = void 0;
|
|
4
|
+
exports.claudeCodeEventCapabilities = {
|
|
5
|
+
capturedFrom: "code.claude.com/docs/en/hooks (claude-code 2.1.273) + inject/veto measured 2026-09-16 headless",
|
|
6
|
+
events: {
|
|
7
|
+
// doc: SessionStart context is prepended to the session. MEASURED 2026-09-16:
|
|
8
|
+
// inject lands; exit 2 is ignored entirely (it is in noEffectHookEvents).
|
|
9
|
+
SessionStart: {
|
|
10
|
+
carries: "session",
|
|
11
|
+
honours: ["inject"],
|
|
12
|
+
matcher: false,
|
|
13
|
+
denyShape: "exit-code",
|
|
14
|
+
},
|
|
15
|
+
// doc: exit 2 blocks the prompt and erases it; stdout becomes context.
|
|
16
|
+
UserPromptSubmit: {
|
|
17
|
+
carries: "prompt",
|
|
18
|
+
honours: ["veto", "inject"],
|
|
19
|
+
matcher: false,
|
|
20
|
+
denyShape: "exit-code",
|
|
21
|
+
},
|
|
22
|
+
// doc: the one event whose deny needs `hookSpecificOutput.permissionDecision`
|
|
23
|
+
// (the legacy top-level `decision` is ignored), and the one that can `ask`.
|
|
24
|
+
// MEASURED 2026-09-16: additionalContext also lands here — the reason this
|
|
25
|
+
// table exists at all, since `injectableEvents` had said otherwise on an
|
|
26
|
+
// assumption nobody had run.
|
|
27
|
+
PreToolUse: {
|
|
28
|
+
carries: "tool",
|
|
29
|
+
honours: ["veto", "ask", "inject"],
|
|
30
|
+
matcher: true,
|
|
31
|
+
denyShape: "permission-decision",
|
|
32
|
+
},
|
|
33
|
+
// doc: exit 2 feeds stderr back to the MODEL but the tool has already run —
|
|
34
|
+
// feedback, never a veto. This distinction is why `hook-block-ineffective`
|
|
35
|
+
// does not flag PostToolUse and would cry wolf if it did.
|
|
36
|
+
PostToolUse: {
|
|
37
|
+
carries: "tool",
|
|
38
|
+
honours: ["feedback", "inject"],
|
|
39
|
+
matcher: true,
|
|
40
|
+
denyShape: "exit-code",
|
|
41
|
+
},
|
|
42
|
+
// doc: exit 2 blocks the agent from stopping (the gate-until-tests-pass
|
|
43
|
+
// shape). MEASURED 2026-09-16: inject lands too.
|
|
44
|
+
Stop: {
|
|
45
|
+
carries: "stop",
|
|
46
|
+
honours: ["veto", "inject"],
|
|
47
|
+
matcher: false,
|
|
48
|
+
denyShape: "exit-code",
|
|
49
|
+
},
|
|
50
|
+
// doc: same block semantics as Stop, for a subagent. Inject NOT claimed —
|
|
51
|
+
// it was not measured, and Stop's behaviour is not evidence about it.
|
|
52
|
+
SubagentStop: {
|
|
53
|
+
carries: "stop",
|
|
54
|
+
honours: ["veto"],
|
|
55
|
+
matcher: false,
|
|
56
|
+
denyShape: "exit-code",
|
|
57
|
+
},
|
|
58
|
+
// doc + noEffectHookEvents: a block decision here is silently ignored
|
|
59
|
+
// ENTIRELY — no veto AND no model feedback. A hook on these three is inert
|
|
60
|
+
// whatever it returns, which is exactly what `honours: []` says.
|
|
61
|
+
SessionEnd: {
|
|
62
|
+
carries: "none",
|
|
63
|
+
honours: [],
|
|
64
|
+
matcher: false,
|
|
65
|
+
denyShape: "exit-code",
|
|
66
|
+
},
|
|
67
|
+
Notification: {
|
|
68
|
+
carries: "none",
|
|
69
|
+
honours: [],
|
|
70
|
+
matcher: false,
|
|
71
|
+
denyShape: "exit-code",
|
|
72
|
+
},
|
|
73
|
+
PreCompact: {
|
|
74
|
+
carries: "none",
|
|
75
|
+
honours: [],
|
|
76
|
+
matcher: false,
|
|
77
|
+
denyShape: "exit-code",
|
|
78
|
+
},
|
|
79
|
+
},
|
|
80
|
+
};
|
|
81
|
+
//# sourceMappingURL=event-capability.js.map
|
|
@@ -11,9 +11,32 @@ exports.claudeCodeHookProtocol = {
|
|
|
11
11
|
// the agent — a stronger stop than a per-call deny, and a documented one.
|
|
12
12
|
haltsTurnField: "continue",
|
|
13
13
|
// Events that honor `hookSpecificOutput.additionalContext` (developer-context
|
|
14
|
-
// injection). Covers vigiles's shipped inject hooks
|
|
15
|
-
// summary
|
|
16
|
-
|
|
14
|
+
// injection). Covers vigiles's shipped inject hooks (the SessionStart lint
|
|
15
|
+
// summary, the PostToolUse refs / eval-lock nudges) AND the two events added
|
|
16
|
+
// 2026-09-15 after a MEASUREMENT contradicted the assumption below.
|
|
17
|
+
//
|
|
18
|
+
// `PreToolUse` and `Stop` were absent because the port's doc-comment asserted
|
|
19
|
+
// that "a few (Stop, SubagentStop, PreCompact) carry no context on either"
|
|
20
|
+
// harness. That was never measured for Claude Code. Measured on 2.1.273 by
|
|
21
|
+
// emitting `additionalContext` from a hook on each event and reading the
|
|
22
|
+
// model's OWN request back (`requestContains` over a real `runHarnessTest`
|
|
23
|
+
// run, not the hook's stdout): both events DELIVER. The prior list was
|
|
24
|
+
// therefore a self-inflicted gate — the runtime refused to emit a shape the
|
|
25
|
+
// harness would have honored, and the four affected react hooks read as
|
|
26
|
+
// "undeliverable by vocabulary" when they were undeliverable by our choice.
|
|
27
|
+
//
|
|
28
|
+
// HONEST SCOPE: measured HEADLESS (`claude -p`, which is what runHarnessTest
|
|
29
|
+
// drives). Interactive is unverified, and for the `ask` channel the two
|
|
30
|
+
// plausibly differ (headless has nobody to ask) — but `ask` is a GATE
|
|
31
|
+
// channel, not this inject one, so it does not bear on this list.
|
|
32
|
+
// `SubagentStop`/`PreCompact` stay out: not measured, so not claimed.
|
|
33
|
+
injectableEvents: [
|
|
34
|
+
"SessionStart",
|
|
35
|
+
"UserPromptSubmit",
|
|
36
|
+
"PreToolUse",
|
|
37
|
+
"PostToolUse",
|
|
38
|
+
"Stop",
|
|
39
|
+
],
|
|
17
40
|
// The `if` field: a permission-rule pattern deciding whether the hook is spawned
|
|
18
41
|
// at all. See ./hook-condition.ts — without it a conditional guard was reported
|
|
19
42
|
// as blocking every disaster in the battery.
|
|
@@ -26,7 +26,6 @@ const node_os_1 = require("node:os");
|
|
|
26
26
|
const node_path_1 = require("node:path");
|
|
27
27
|
const node_url_1 = require("node:url");
|
|
28
28
|
const node_fs_1 = require("node:fs");
|
|
29
|
-
const node_os_2 = require("node:os");
|
|
30
29
|
const glob_1 = require("glob");
|
|
31
30
|
const check_count_js_1 = require("../../check-count.js");
|
|
32
31
|
/**
|
|
@@ -64,8 +63,21 @@ exports.SKIP_EXIT_CODE = 77;
|
|
|
64
63
|
function statusFor(code, checks, output) {
|
|
65
64
|
if (code === exports.SKIP_EXIT_CODE)
|
|
66
65
|
return "skip";
|
|
66
|
+
// 🔴 `checks === undefined` GUARDS THE TEXT MATCH, and it is the load-bearing
|
|
67
|
+
// half. A script that REPORTED a count executed: the counter is written by an
|
|
68
|
+
// exit handler that exists only once the module was linked and run
|
|
69
|
+
// (`check-count.ts`), so the count is a STRUCTURAL fact about the child, while
|
|
70
|
+
// `didNotLoad` is a guess about its text.
|
|
71
|
+
//
|
|
72
|
+
// MEASURED: `statusFor(1, 3, "<hook stderr: Cannot find module …>\nAssertionError")`
|
|
73
|
+
// returned `"skip"` and the run exited 0 — a harness that ran, recorded three
|
|
74
|
+
// checks and FAILED an assertion, reported as skipped. That is not an exotic
|
|
75
|
+
// input: vigiles harnesses drive hooks and print their transcripts, so a
|
|
76
|
+
// loader phrase in the output is ordinary EVIDENCE about the thing under test,
|
|
77
|
+
// not a diagnosis of the harness. Watching the child from outside cannot tell
|
|
78
|
+
// those apart. The count can, and it was already in hand.
|
|
67
79
|
if (code !== 0)
|
|
68
|
-
return didNotLoad(output) ? "skip" : "fail";
|
|
80
|
+
return checks === undefined && didNotLoad(output) ? "skip" : "fail";
|
|
69
81
|
return checks === 0 ? "vacuous" : "pass";
|
|
70
82
|
}
|
|
71
83
|
/**
|
|
@@ -83,10 +95,17 @@ function statusFor(code, checks, output) {
|
|
|
83
95
|
* 'recordCheck' not found`, and the ledger dropped from 48 records to 34 and from
|
|
84
96
|
* 47 to 33. Nothing about those surfaces had changed — the machine had.
|
|
85
97
|
*
|
|
86
|
-
* ⚠️ This does NOT make a broken environment quiet
|
|
87
|
-
*
|
|
88
|
-
*
|
|
89
|
-
* measurement taken on a machine
|
|
98
|
+
* ⚠️ This does NOT make a broken environment quiet: a script classified here
|
|
99
|
+
* fails the run by default (`cli-main.ts`, right after `anyFailed`), because a
|
|
100
|
+
* skip the AUTHOR never declared is not a skip. What this classification buys is
|
|
101
|
+
* only that a machine problem may not delete a measurement taken on a machine
|
|
102
|
+
* that worked.
|
|
103
|
+
*
|
|
104
|
+
* 🔴 That default is new, and the sentence it replaces was false. It read
|
|
105
|
+
* «`--no-skip` — which this repo's own CI passes — still fails the run».
|
|
106
|
+
* Measured: `--no-skip` appears ZERO times under `.github/`, `package.json`,
|
|
107
|
+
* `scripts/` and `.claude/`; CI passes `--min=14` and nothing else. The
|
|
108
|
+
* safety net the non-fatal classification leaned on was never strung.
|
|
90
109
|
*
|
|
91
110
|
* Deliberately literal, and only the loader's own vocabulary: these strings come
|
|
92
111
|
* from Node's module resolution, not from user code. A test that legitimately
|
|
@@ -99,6 +118,13 @@ function didNotLoad(output) {
|
|
|
99
118
|
output.includes("Cannot find package") ||
|
|
100
119
|
output.includes("Cannot find module") ||
|
|
101
120
|
/SyntaxError: Named export '[^']*' not found/.test(output) ||
|
|
121
|
+
// Same event, ESM spelling. Node phrases a missing named export one way for
|
|
122
|
+
// a CommonJS target and another for an ES module, and only the first was
|
|
123
|
+
// listed — so the 2026-08-20 class below still RETRACTED coverage whenever
|
|
124
|
+
// the dependency happened to be ESM. Measured on Node 22:
|
|
125
|
+
// CJS: SyntaxError: Named export 'recordCheck' not found. The requested module …
|
|
126
|
+
// ESM: SyntaxError: The requested module './x.mjs' does not provide an export named 'recordCheck'
|
|
127
|
+
/SyntaxError: The requested module '[^']*' does not provide an export named/.test(output) ||
|
|
102
128
|
output.includes("ERR_UNSUPPORTED_DIR_IMPORT") ||
|
|
103
129
|
output.includes("ERR_PACKAGE_PATH_NOT_EXPORTED"));
|
|
104
130
|
}
|
|
@@ -118,6 +144,7 @@ var ts_runner_caps_js_1 = require("../../ts-runner-caps.js");
|
|
|
118
144
|
Object.defineProperty(exports, "detectNodeCaps", { enumerable: true, get: function () { return ts_runner_caps_js_1.detectNodeCaps; } });
|
|
119
145
|
Object.defineProperty(exports, "canRunTypeScript", { enumerable: true, get: function () { return ts_runner_caps_js_1.canRunTypeScript; } });
|
|
120
146
|
const ts_runner_caps_js_2 = require("../../ts-runner-caps.js");
|
|
147
|
+
const tmp_root_js_1 = require("../../core/tmp-root.js");
|
|
121
148
|
/**
|
|
122
149
|
* The `node` argv (after the binary) to run a single script. Plain JS runs
|
|
123
150
|
* directly; a TypeScript script picks `tsx` when available, else Node's native
|
|
@@ -159,7 +186,13 @@ function discoverScripts(patterns, defaultGlob, cwd, ignore) {
|
|
|
159
186
|
const globs = patterns.length > 0 ? patterns : [defaultGlob];
|
|
160
187
|
const found = new Set();
|
|
161
188
|
for (const p of globs) {
|
|
162
|
-
|
|
189
|
+
// 🔴 `isFile`, not `existsSync`: a DIRECTORY exists too. `vigiles test .`
|
|
190
|
+
// therefore passed `.` through as a script, `spawn("node", ["."])` died with
|
|
191
|
+
// Node's `ERR_UNSUPPORTED_DIR_IMPORT` stack, and the classifier downstream
|
|
192
|
+
// read that stack as "did not load" — a crash reported as a skip, exit 0.
|
|
193
|
+
// A directory now contributes no files, so the caller's own loud
|
|
194
|
+
// nothing-matched path owns the message (see `cli-main.ts`).
|
|
195
|
+
if ((0, node_fs_1.statSync)((0, node_path_1.resolve)(cwd, p), { throwIfNoEntry: false })?.isFile()) {
|
|
163
196
|
found.add(p);
|
|
164
197
|
continue;
|
|
165
198
|
}
|
|
@@ -174,10 +207,16 @@ function discoverScripts(patterns, defaultGlob, cwd, ignore) {
|
|
|
174
207
|
// `test-coverage.ts` and `cli.ts` both pass `dot: true` with comments saying why, and
|
|
175
208
|
// `test-coverage.test.ts` records "glob without `dot:true` never found it and the surface
|
|
176
209
|
// looked untested". Coverage learned it; the runner did not.
|
|
210
|
+
// 🔴 `nodir: true` for the same reason as the `isFile` guard above, and the
|
|
211
|
+
// guard alone was NOT enough — caught by its own test. A pattern that names
|
|
212
|
+
// an existing directory (`sub`, `.`) skips the fast path and then comes back
|
|
213
|
+
// out of the globber, because a directory matches a glob perfectly well.
|
|
214
|
+
// Asked of the globber rather than filtered afterwards: it already knows.
|
|
177
215
|
for (const m of (0, glob_1.globSync)(p, {
|
|
178
216
|
cwd,
|
|
179
217
|
ignore: [...ignore],
|
|
180
218
|
dot: true,
|
|
219
|
+
nodir: true,
|
|
181
220
|
})) {
|
|
182
221
|
found.add(m);
|
|
183
222
|
}
|
|
@@ -219,7 +258,7 @@ function readCheckReport(path) {
|
|
|
219
258
|
*/
|
|
220
259
|
async function runScripts(files, cwd, env = {}, opts = {}) {
|
|
221
260
|
const caps = (0, ts_runner_caps_js_2.detectNodeCaps)(cwd);
|
|
222
|
-
const countDir = (0,
|
|
261
|
+
const countDir = (0, tmp_root_js_1.makeTmpDir)("checks");
|
|
223
262
|
// 🔴 THE DEFAULT IS DECIDED BY `entry`, NOT BY A FLAG, because the two commands
|
|
224
263
|
// that share this runner have OPPOSITE right answers and the caller already
|
|
225
264
|
// distinguishes them:
|
|
@@ -21,6 +21,25 @@ exports.codexDialect = {
|
|
|
21
21
|
"UserPromptSubmit",
|
|
22
22
|
"Stop",
|
|
23
23
|
],
|
|
24
|
+
// 🔴 TRUNCATES, SILENTLY — the asymmetry that makes `onExceed` worth carrying.
|
|
25
|
+
// Codex's own source: "Maximum number of bytes of the documentation that will
|
|
26
|
+
// be embedded. Larger files are *silently truncated*" (openai/codex#7138,
|
|
27
|
+
// CLOSED AS NOT PLANNED — standing behaviour, not a bug in flight). Default
|
|
28
|
+
// `project_doc_max_bytes` is 32 * 1024. Over budget on Claude Code costs
|
|
29
|
+
// money; over budget here means some of your rules DO NOT EXIST for the model
|
|
30
|
+
// and nothing in the session says which.
|
|
31
|
+
//
|
|
32
|
+
// BYTES, not chars: the two diverge on any non-ASCII instruction file, and
|
|
33
|
+
// this is the unit Codex actually counts.
|
|
34
|
+
instructionBudget: {
|
|
35
|
+
unit: "bytes",
|
|
36
|
+
limit: 32768,
|
|
37
|
+
onExceed: "truncates",
|
|
38
|
+
capturedFrom: "codex config project_doc_max_bytes default 32 * 1024; truncation quoted in openai/codex#7138",
|
|
39
|
+
// Read root-to-leaf and concatenated, so a nested AGENTS.md pays into the
|
|
40
|
+
// same budget — the sum is what gets cut, not the individual file.
|
|
41
|
+
alwaysLoaded: ["AGENTS.md", "**/AGENTS.md"],
|
|
42
|
+
},
|
|
24
43
|
instructionTargets: ["AGENTS.md"],
|
|
25
44
|
pluginRootToken: "${PLUGIN_ROOT}",
|
|
26
45
|
// Codex SKILL.md frontmatter is name + description ONLY — the CC-only keys
|
package/dist/cli-main.js
CHANGED
|
@@ -16,6 +16,7 @@ exports.discoverNestedBundles = discoverNestedBundles;
|
|
|
16
16
|
exports.handleHookRuntime = handleHookRuntime;
|
|
17
17
|
exports.main = main;
|
|
18
18
|
const node_fs_1 = require("node:fs");
|
|
19
|
+
const event_capability_js_1 = require("./core/event-capability.js");
|
|
19
20
|
const node_path_1 = require("node:path");
|
|
20
21
|
const repo_path_js_1 = require("./core/repo-path.js");
|
|
21
22
|
const node_child_process_1 = require("node:child_process");
|
|
@@ -323,7 +324,7 @@ function compileGeneratorSkillToFile(specPath, source) {
|
|
|
323
324
|
if (artifact)
|
|
324
325
|
writeArtifact(outputPath, artifact);
|
|
325
326
|
if (errors.length === 0) {
|
|
326
|
-
console.log(`\n✓ ${specPath} → ${outputPath} (generator skill)`);
|
|
327
|
+
console.log(`\n✓ ${specPath} → ${outputPath} (generator skill${artifact ? `, ${formatArtifactSize(artifact)}` : ""})`);
|
|
327
328
|
return true;
|
|
328
329
|
}
|
|
329
330
|
console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
|
|
@@ -334,7 +335,7 @@ function compileGeneratorSkillToFile(specPath, source) {
|
|
|
334
335
|
/** Compile a ClaudeSpec → its primary + any additional targets. */
|
|
335
336
|
function compileClaudeToFile(spec, specPath, config, dialect) {
|
|
336
337
|
const basePath = process.cwd();
|
|
337
|
-
const { markdown, errors, linterResults, targets } = (0, compile_js_1.compileClaude)(spec, {
|
|
338
|
+
const { markdown, errors, warnings, linterResults, targets } = (0, compile_js_1.compileClaude)(spec, {
|
|
338
339
|
basePath,
|
|
339
340
|
specFile: specPath,
|
|
340
341
|
dialect,
|
|
@@ -345,6 +346,16 @@ function compileClaudeToFile(spec, specPath, config, dialect) {
|
|
|
345
346
|
linters: config.linters,
|
|
346
347
|
});
|
|
347
348
|
const primaryOutput = specPath.replace(/\.spec\.ts$/, "");
|
|
349
|
+
// Budget findings print BEFORE the pass/fail line and never change the exit
|
|
350
|
+
// code. A number nobody prints cannot be acted on — the same reason the
|
|
351
|
+
// compiled size is printed at all — and an instruction file grows one
|
|
352
|
+
// unremarkable entry at a time, so the warning has to arrive at the moment
|
|
353
|
+
// the entry is added rather than at some later audit.
|
|
354
|
+
if (warnings.length > 0) {
|
|
355
|
+
console.log(`\nℹ ${specPath} — ${String(warnings.length)} budget warning(s)`);
|
|
356
|
+
for (const w of warnings)
|
|
357
|
+
console.log(` ${w.message}`);
|
|
358
|
+
}
|
|
348
359
|
if (errors.length > 0) {
|
|
349
360
|
console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
|
|
350
361
|
printErrors(specPath, errors);
|
|
@@ -376,7 +387,7 @@ function compileClaudeToFile(spec, specPath, config, dialect) {
|
|
|
376
387
|
outputNames.push(targetPath);
|
|
377
388
|
}
|
|
378
389
|
const linterCount = linterResults.filter((r) => r.exists).length;
|
|
379
|
-
console.log(`\n✓ ${specPath} → ${outputNames.join(", ")}`);
|
|
390
|
+
console.log(`\n✓ ${specPath} → ${outputNames.join(", ")} (${formatArtifactSize(markdown)})`);
|
|
380
391
|
console.log(` ${String(Object.keys(spec.rules).length)} rules (${String(linterCount)} linter-verified)`);
|
|
381
392
|
return true;
|
|
382
393
|
}
|
|
@@ -419,6 +430,44 @@ function writeInstructionMirrors(primaryOutput, harnesses) {
|
|
|
419
430
|
console.log(` ↳ mirrored ${primaryName} → ${target} (byte-identical)`);
|
|
420
431
|
}
|
|
421
432
|
}
|
|
433
|
+
/**
|
|
434
|
+
* The size of a compiled artifact, printed beside every `✓` on STDOUT.
|
|
435
|
+
*
|
|
436
|
+
* WHY STDOUT: every human line `compile` emits already goes there (the `✓`
|
|
437
|
+
* lines, `Compilation complete`), and stderr in this CLI is the ERROR channel.
|
|
438
|
+
* An informational number in the error stream reads as a problem in CI logs and
|
|
439
|
+
* is killed by `2>/dev/null`. This is part of the operation's RESULT, like the
|
|
440
|
+
* output path beside it — not a diagnostic about a failure.
|
|
441
|
+
*
|
|
442
|
+
* WHY BYTES AND CHARS, AND WHY THE TOKEN COUNT IS ONLY AN ESTIMATE. These are
|
|
443
|
+
* the units the HARNESSES themselves measure, and both are exact:
|
|
444
|
+
*
|
|
445
|
+
* - Claude Code warns per FILE CHARS (`/doctor`: "Large file will impact
|
|
446
|
+
* performance"), and skips a file past a hard size;
|
|
447
|
+
* - Codex caps AGENTS.md by BYTES (`project_doc_max_bytes`) and TRUNCATES —
|
|
448
|
+
* instructions past that byte simply do not exist for the model.
|
|
449
|
+
*
|
|
450
|
+
* Neither gates on tokens, so a real tokenizer would buy precision in a unit
|
|
451
|
+
* nothing decides on. It is also not available: Anthropic publishes no local
|
|
452
|
+
* tokenizer for current Claude models (the supported count is a NETWORK call
|
|
453
|
+
* with an API key, which a zero-config offline compile must not need), while
|
|
454
|
+
* OpenAI's is local — so tokenizing would make us precise for one harness and
|
|
455
|
+
* blind for the other, and the two numbers would stop being comparable.
|
|
456
|
+
*
|
|
457
|
+
* `estimateTokens` is `length / 4`, calibrated for ASCII English; it undercounts
|
|
458
|
+
* Cyrillic and CJK substantially. It is labelled `est.` for that reason and must
|
|
459
|
+
* never become the number a rule gates on — see the instruction-weight rule.
|
|
460
|
+
*/
|
|
461
|
+
function formatArtifactSize(markdown) {
|
|
462
|
+
const chars = markdown.length;
|
|
463
|
+
const bytes = Buffer.byteLength(markdown, "utf8");
|
|
464
|
+
const group = (n) => String(n).replace(/\B(?=(\d{3})+(?!\d))/g, ",");
|
|
465
|
+
const kTokens = Math.round((0, compile_js_1.estimateTokens)(markdown) / 100) / 10;
|
|
466
|
+
const size = bytes === chars
|
|
467
|
+
? `${group(chars)} chars`
|
|
468
|
+
: `${group(chars)} chars / ${group(bytes)} bytes`;
|
|
469
|
+
return `${size} · ~${String(kTokens)}k tokens est.`;
|
|
470
|
+
}
|
|
422
471
|
/** Compile a declarative SkillSpec → SKILL.md. */
|
|
423
472
|
function compileSkillToFile(spec, specPath, dialect) {
|
|
424
473
|
const outputPath = specPath.replace(/\.spec\.ts$/, "");
|
|
@@ -434,7 +483,7 @@ function compileSkillToFile(spec, specPath, dialect) {
|
|
|
434
483
|
if (artifact)
|
|
435
484
|
writeArtifact(outputPath, artifact);
|
|
436
485
|
if (errors.length === 0) {
|
|
437
|
-
console.log(`\n✓ ${specPath} → ${outputPath}`);
|
|
486
|
+
console.log(`\n✓ ${specPath} → ${outputPath}${artifact ? ` (${formatArtifactSize(artifact)})` : ""}`);
|
|
438
487
|
printWarnings(specPath, warnings);
|
|
439
488
|
return true;
|
|
440
489
|
}
|
|
@@ -456,7 +505,7 @@ function compileAgentToFile(spec, specPath, dialect) {
|
|
|
456
505
|
if (artifact)
|
|
457
506
|
writeArtifact(outputPath, artifact);
|
|
458
507
|
if (errors.length === 0) {
|
|
459
|
-
console.log(`\n✓ ${specPath} → ${outputPath}`);
|
|
508
|
+
console.log(`\n✓ ${specPath} → ${outputPath}${artifact ? ` (${formatArtifactSize(artifact)})` : ""}`);
|
|
460
509
|
printWarnings(specPath, warnings);
|
|
461
510
|
return true;
|
|
462
511
|
}
|
|
@@ -481,7 +530,7 @@ function compileRailwayToFile(spec, specPath, knownAgents) {
|
|
|
481
530
|
if (artifact)
|
|
482
531
|
writeArtifact(outputPath, artifact);
|
|
483
532
|
if (errors.length === 0) {
|
|
484
|
-
console.log(`\n✓ ${specPath} → ${outputPath}`);
|
|
533
|
+
console.log(`\n✓ ${specPath} → ${outputPath}${artifact ? ` (${formatArtifactSize(artifact)})` : ""}`);
|
|
485
534
|
return true;
|
|
486
535
|
}
|
|
487
536
|
console.log(`\n✗ ${specPath} — ${String(errors.length)} error(s)`);
|
|
@@ -4887,10 +4936,19 @@ async function handleRunScripts(kind, args, restArgs, excludes) {
|
|
|
4887
4936
|
// VALUE cannot tell them apart — only the flag's presence can. (Caught by a control:
|
|
4888
4937
|
// the first version read `minRequired === 0` and made `--min=0` do nothing.)
|
|
4889
4938
|
if (restArgs.length > 0 && minFlag === undefined) {
|
|
4890
|
-
|
|
4891
|
-
|
|
4892
|
-
|
|
4893
|
-
|
|
4939
|
+
// A DIRECTORY is the one stale-looking target whose cause we actually know,
|
|
4940
|
+
// so say it instead of listing the three guesses. `vigiles test .` used to
|
|
4941
|
+
// reach `spawn("node", ["."])` and surface Node's module-resolution stack;
|
|
4942
|
+
// the generic message above would now be true but unhelpful.
|
|
4943
|
+
const dirs = restArgs.filter((a) => (0, node_fs_1.lstatSync)(a, { throwIfNoEntry: false })?.isDirectory() === true);
|
|
4944
|
+
console.error(dirs.length > 0
|
|
4945
|
+
? `✗ vigiles ${kind}: ${dirs.join(", ")} ${dirs.length === 1 ? "is a directory" : "are directories"} — ` +
|
|
4946
|
+
`pass a file, or a glob like "${defaultGlob}".\n` +
|
|
4947
|
+
` A directory is not a script; nothing ran.`
|
|
4948
|
+
: `✗ vigiles ${kind}: ${String(restArgs.length)} target(s) given and NOTHING matched — ` +
|
|
4949
|
+
`${restArgs.join(", ")}\n` +
|
|
4950
|
+
` Nothing ran. A stale path, a wrong glob, or a moved file all look like this.\n` +
|
|
4951
|
+
` If an empty match is expected here, say so with --min=0.`);
|
|
4894
4952
|
process.exit(1);
|
|
4895
4953
|
}
|
|
4896
4954
|
console.log(`No ${defaultGlob} files found.`);
|
|
@@ -4953,6 +5011,35 @@ async function handleRunScripts(kind, args, restArgs, excludes) {
|
|
|
4953
5011
|
console.log("\n" + (0, run_scripts_js_1.formatScriptSummary)(results));
|
|
4954
5012
|
if ((0, run_scripts_js_1.anyFailed)(results))
|
|
4955
5013
|
process.exit(1);
|
|
5014
|
+
// 🔴 A SKIP THE AUTHOR NEVER DECLARED IS NOT A SKIP — and the discriminator was
|
|
5015
|
+
// already sitting in the result. `skip()` exits 77 (`SKIP_EXIT_CODE`); a script
|
|
5016
|
+
// the runtime could not evaluate exits with whatever the loader gave it, 1 in
|
|
5017
|
+
// practice. Both are classified `"skip"` so that neither RETRACTS coverage —
|
|
5018
|
+
// which is right, a file that did not run proved nothing either way — but only
|
|
5019
|
+
// the declared one is a reason to stay green.
|
|
5020
|
+
//
|
|
5021
|
+
// Reported as #243: `vigiles test .` printed a resolver stack over a `⊘`, said
|
|
5022
|
+
// `0 passed, 1 skipped`, and exited 0. Downstream a consumer's README shipped
|
|
5023
|
+
// that exact command as its first setup step, so a new reader's suite silently
|
|
5024
|
+
// never ran. `--no-skip` would have caught it and is not the default; `--min=1`
|
|
5025
|
+
// does not, because it counts files MATCHED, not scripts executed.
|
|
5026
|
+
//
|
|
5027
|
+
// This is deliberately NOT a fifth `ScriptStatus`. Coverage retraction reads the
|
|
5028
|
+
// status as a bare STRING (`executedScripts`, `coverage-artifact.ts`, whose
|
|
5029
|
+
// parameter is typed `string`), so a new member would start retracting silently
|
|
5030
|
+
// with no type error — breaking the one property the classification exists to
|
|
5031
|
+
// protect.
|
|
5032
|
+
const notEvaluated = results.filter((r) => r.status === "skip" && r.code !== run_scripts_js_1.SKIP_EXIT_CODE);
|
|
5033
|
+
if (notEvaluated.length > 0) {
|
|
5034
|
+
console.error(`\n✗ vigiles ${kind}: ${String(notEvaluated.length)} script(s) never ran — the runtime could not load them:\n` +
|
|
5035
|
+
notEvaluated
|
|
5036
|
+
.map((r) => ` ${r.file} (exit ${String(r.code)})`)
|
|
5037
|
+
.join("\n") +
|
|
5038
|
+
`\n Their previous coverage is kept, because a script that did not run retracts nothing.\n` +
|
|
5039
|
+
` If a missing dependency is expected here, import it dynamically and call skip() — ` +
|
|
5040
|
+
`a declared skip stays green.`);
|
|
5041
|
+
process.exit(1);
|
|
5042
|
+
}
|
|
4956
5043
|
// `--no-skip`: in a context that ASSERTS the capability is present (a CI job),
|
|
4957
5044
|
// a skipped tier is untested surface — fail loudly instead of passing green.
|
|
4958
5045
|
if (args.includes("--no-skip") && results.some((r) => r.status === "skip")) {
|
|
@@ -5730,7 +5817,9 @@ async function installHookFile(file, adapter, registeredProviders = []) {
|
|
|
5730
5817
|
// cross-harness and never warns.
|
|
5731
5818
|
const role = (0, hook_program_js_1.dispatchKind)(program);
|
|
5732
5819
|
const event = typeof program.on === "string" ? program.on : "";
|
|
5733
|
-
const injectable = adapter.
|
|
5820
|
+
const injectable = (0, event_capability_js_1.injectableEventsOf)(adapter.dialect,
|
|
5821
|
+
// eslint-disable-next-line @typescript-eslint/no-deprecated -- the legacy list is the FALLBACK an adapter without a capability table still relies on; reading it here is the point
|
|
5822
|
+
adapter.hookProtocol?.injectableEvents);
|
|
5734
5823
|
const matcher = (0, hook_program_js_1.hookRouting)(program).matcher;
|
|
5735
5824
|
let warning;
|
|
5736
5825
|
// A react on an event this harness does NOT inject can still call `notice()`,
|
package/dist/core/adopt.js
CHANGED
|
@@ -25,6 +25,7 @@ exports.adoptToSpec = adoptToSpec;
|
|
|
25
25
|
exports.adoptMarkdown = adoptMarkdown;
|
|
26
26
|
exports.adoptSkill = adoptSkill;
|
|
27
27
|
exports.adoptAgent = adoptAgent;
|
|
28
|
+
const compile_js_1 = require("./compile.js");
|
|
28
29
|
const integrity_js_1 = require("./integrity.js");
|
|
29
30
|
const frontmatter_read_js_1 = require("./frontmatter-read.js");
|
|
30
31
|
const spec_js_1 = require("./spec.js");
|
|
@@ -195,8 +196,15 @@ function renderSpecSource(spec, exists) {
|
|
|
195
196
|
const targetLine = spec.target !== "CLAUDE.md"
|
|
196
197
|
? `\n target: ${JSON.stringify(spec.target)},`
|
|
197
198
|
: "";
|
|
199
|
+
// The raised guard is rendered WITH the debt spelled out. A bare number reads
|
|
200
|
+
// as a considered setting; the comment says it was accepted as-is and by how
|
|
201
|
+
// much, so the next author sees a debt rather than a decision.
|
|
202
|
+
const over = (spec.maxSectionLines ?? 0) - compile_js_1.DEFAULT_MAX_SECTION_LINES;
|
|
198
203
|
const maxLine = spec.maxSectionLines !== undefined
|
|
199
|
-
? `\n
|
|
204
|
+
? `\n // Adopted as-is: the longest section is ${String(spec.maxSectionLines)} lines, ` +
|
|
205
|
+
`${String(over)} over the ${String(compile_js_1.DEFAULT_MAX_SECTION_LINES)}-line budget.` +
|
|
206
|
+
`\n // Lower this as you split that section up; it is a debt, not a setting.` +
|
|
207
|
+
`\n maxSectionLines: ${String(spec.maxSectionLines)},`
|
|
200
208
|
: "";
|
|
201
209
|
const adopted = [];
|
|
202
210
|
const entries = Object.entries(spec.sections)
|
|
@@ -269,14 +277,24 @@ function adoptToSpec(markdown, target) {
|
|
|
269
277
|
const sections = {};
|
|
270
278
|
for (const { key, content } of ordered)
|
|
271
279
|
sections[key] = content;
|
|
272
|
-
// A faithful section can legitimately be long
|
|
273
|
-
// the
|
|
274
|
-
//
|
|
280
|
+
// A faithful section can legitimately be long, and adoption must not fail on
|
|
281
|
+
// prose the user already has — so the 200-line guard is raised to EXACTLY the
|
|
282
|
+
// longest section, never above it.
|
|
283
|
+
//
|
|
284
|
+
// IT USED TO BE `longest + 50`, and that headroom is the bug this block exists
|
|
285
|
+
// to name: the guard was disarmed hardest for the population that needs it
|
|
286
|
+
// most — someone adopting a spec over a file that is already oversized — and
|
|
287
|
+
// the next fifty lines of growth were pre-approved, silently, by the tool
|
|
288
|
+
// meant to notice them. Raising to `longest` keeps adoption green on the file
|
|
289
|
+
// as it stands and makes the very next line that grows trip the gate. The
|
|
290
|
+
// override is rendered with a comment stating how far above the budget it
|
|
291
|
+
// sits, so the debt is visible in the spec rather than inferable from a number
|
|
292
|
+
// nobody reads.
|
|
275
293
|
const longest = ordered.reduce((n, s) => Math.max(n, s.content.split("\n").length), 0);
|
|
276
294
|
return {
|
|
277
295
|
target,
|
|
278
296
|
sections,
|
|
279
|
-
maxSectionLines: longest >
|
|
297
|
+
maxSectionLines: longest > compile_js_1.DEFAULT_MAX_SECTION_LINES ? longest : undefined,
|
|
280
298
|
tier: synthesizedHeading || ordered.length === 0 ? "raw" : "structured",
|
|
281
299
|
};
|
|
282
300
|
}
|