vigiles 14.1.0 → 14.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -14
- package/dist/adapters/claude-code/dialect.d.ts +1 -1
- package/dist/audit-report.template.html +14 -14
- package/dist/audit-score.d.ts +4 -2
- package/dist/audit-score.js +8 -5
- package/dist/cli.js +6 -4
- package/dist/core/rule-catalog.d.ts +1 -1
- package/dist/core/rule-catalog.js +1 -1
- package/dist/core/skill-resources.js +61 -8
- package/dist/dialect-drift.d.ts +18 -0
- package/dist/dialect-drift.js +44 -2
- package/dist/instruction-sources.d.ts +1 -1
- package/dist/instruction-sources.js +1 -1
- package/dist/leaderboard.d.ts +1 -1
- package/dist/leaderboard.js +13 -9
- package/dist/rule-inventory.js +151 -2
- package/dist/rule-routing.d.ts +16 -1
- package/dist/rule-routing.js +172 -138
- package/dist/rule-signals.d.ts +46 -0
- package/dist/rule-signals.js +49 -0
- package/dist/segment.d.ts +16 -5
- package/dist/segment.js +219 -179
- package/package.json +1 -1
package/dist/audit-score.d.ts
CHANGED
|
@@ -14,8 +14,10 @@
|
|
|
14
14
|
* (`lethalTrifectaIssues` → `report.trifectaFindings`): a unit holding all three
|
|
15
15
|
* capability legs is a prompt-injection exfil path detectable from the tool-SET
|
|
16
16
|
* alone — nothing executes, so it sidesteps the confinement blocker. A `"hard"`
|
|
17
|
-
* (explicit all-three) finding is GRADED into the overall
|
|
18
|
-
* (
|
|
17
|
+
* (explicit all-three) finding is GRADED into the overall at a REDUCED weight
|
|
18
|
+
* (`W_TRIFECTA=10`, HALF the old 20 — a DING, not a fail: it dents the grade
|
|
19
|
+
* without a catastrophic F for a pattern official plugins ship by design); a
|
|
20
|
+
* `"advisory"` (inherits-all) finding is SHOWN in the ring but not graded. NB the EXECUTING
|
|
19
21
|
* "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
|
|
20
22
|
* running arbitrary hooks safely needs cross-platform confinement that isn't
|
|
21
23
|
* shipped yet, so the battery lives in the `vigiles/testing` API via
|
package/dist/audit-score.js
CHANGED
|
@@ -18,8 +18,10 @@ exports.formatAuditScore = formatAuditScore;
|
|
|
18
18
|
* (`lethalTrifectaIssues` → `report.trifectaFindings`): a unit holding all three
|
|
19
19
|
* capability legs is a prompt-injection exfil path detectable from the tool-SET
|
|
20
20
|
* alone — nothing executes, so it sidesteps the confinement blocker. A `"hard"`
|
|
21
|
-
* (explicit all-three) finding is GRADED into the overall
|
|
22
|
-
* (
|
|
21
|
+
* (explicit all-three) finding is GRADED into the overall at a REDUCED weight
|
|
22
|
+
* (`W_TRIFECTA=10`, HALF the old 20 — a DING, not a fail: it dents the grade
|
|
23
|
+
* without a catastrophic F for a pattern official plugins ship by design); a
|
|
24
|
+
* `"advisory"` (inherits-all) finding is SHOWN in the ring but not graded. NB the EXECUTING
|
|
23
25
|
* "do your hooks actually block?" disaster-battery is STILL not an `audit` ring:
|
|
24
26
|
* running arbitrary hooks safely needs cross-platform confinement that isn't
|
|
25
27
|
* shipped yet, so the battery lives in the `vigiles/testing` API via
|
|
@@ -154,9 +156,10 @@ function structure(r) {
|
|
|
154
156
|
};
|
|
155
157
|
}
|
|
156
158
|
/**
|
|
157
|
-
* SAFETY — fed by the STATIC lethal-trifecta check (`report.trifectaFindings`)
|
|
158
|
-
* A `"hard"` finding (an explicit contract naming all three
|
|
159
|
-
*
|
|
159
|
+
* SAFETY — fed by the STATIC lethal-trifecta check (`report.trifectaFindings`), a
|
|
160
|
+
* GRADED ring. A `"hard"` finding (an explicit contract naming all three
|
|
161
|
+
* capability legs) is a prompt-injection exfil path and is GRADED at a REDUCED
|
|
162
|
+
* weight (`W_TRIFECTA=10` each — HALF the old 20, a DING not a fail; the same
|
|
160
163
|
* weight `reportDeductions` sums into the overall, so the ring and the headline
|
|
161
164
|
* agree). A `"advisory"` finding (inherits-all) is SHOWN in the ring's findings
|
|
162
165
|
* but NOT graded — aligned with the inherits-all-is-advisory stance.
|
package/dist/cli.js
CHANGED
|
@@ -1535,7 +1535,7 @@ function computeRuleInventory(root, instructionFile) {
|
|
|
1535
1535
|
* instruction files PLUS nested subdirectory-memory (`src/CLAUDE.md`,
|
|
1536
1536
|
* `research/CLAUDE.md`, …), skipping fixture/demo/build/test dirs (`isFixturePath`)
|
|
1537
1537
|
* so a repo's real memory is read without the test-fixture noise. `.claude/` rule
|
|
1538
|
-
* sources remain a future source. research/rule-
|
|
1538
|
+
* sources remain a future source. research/rule-enforcer-multilang-design.md §0. */
|
|
1539
1539
|
function gatherInstructionFiles(root, instructionFile) {
|
|
1540
1540
|
const raw = [];
|
|
1541
1541
|
const collect = (rel) => {
|
|
@@ -1696,10 +1696,12 @@ function formatRuleMapSummary(routing) {
|
|
|
1696
1696
|
counts.meta;
|
|
1697
1697
|
if (confident === 0 && possible.length === 0 && skipped.length === 0)
|
|
1698
1698
|
return "";
|
|
1699
|
+
// Lane counts, rendered from the single-source LANE_META (glyph + label).
|
|
1700
|
+
const lane = (c) => `${rule_routing_js_1.LANE_META[c].glyph} ${String(counts[c])} ${rule_routing_js_1.LANE_META[c].label}`;
|
|
1699
1701
|
const lines = [
|
|
1700
|
-
"Rule map — how your prose rules could be enforced (heuristic, precision-first):",
|
|
1701
|
-
`
|
|
1702
|
-
(counts.meta > 0 ? ` ·
|
|
1702
|
+
"Rule map [experimental] — how your prose rules could be enforced (heuristic, precision-first; misses some rules):",
|
|
1703
|
+
` ${lane("reuse")} · ${lane("hook")} · ${lane("unrouted")} · ${lane("semantic")}` +
|
|
1704
|
+
(counts.meta > 0 ? ` · ${lane("meta")}` : ""),
|
|
1703
1705
|
];
|
|
1704
1706
|
if (possible.length > 0)
|
|
1705
1707
|
lines.push(` ? ${String(possible.length)} possible — rule-ish, but below the confidence bar (review these)`);
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* / boundaries), of which ~140 are enabled — vs the old static map's ~23. That
|
|
9
9
|
* makes an architecture norm enforceable too (`boundaries/dependencies` is in the
|
|
10
10
|
* catalog), which a static map never captured. See
|
|
11
|
-
* `research/rule-
|
|
11
|
+
* `research/rule-enforcer-multilang-design.md` §0 (the spike this productizes).
|
|
12
12
|
*
|
|
13
13
|
* SAFETY — this EXECUTES the linter. Loading ESLint resolves the repo's real
|
|
14
14
|
* config (which can run plugin/config code), so `enumerateEslintCatalog` is an
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
* / boundaries), of which ~140 are enabled — vs the old static map's ~23. That
|
|
10
10
|
* makes an architecture norm enforceable too (`boundaries/dependencies` is in the
|
|
11
11
|
* catalog), which a static map never captured. See
|
|
12
|
-
* `research/rule-
|
|
12
|
+
* `research/rule-enforcer-multilang-design.md` §0 (the spike this productizes).
|
|
13
13
|
*
|
|
14
14
|
* SAFETY — this EXECUTES the linter. Loading ESLint resolves the repo's real
|
|
15
15
|
* config (which can run plugin/config code), so `enumerateEslintCatalog` is an
|
|
@@ -25,6 +25,15 @@ exports.skillResourceIssues = skillResourceIssues;
|
|
|
25
25
|
* MISSING a real ref over emitting a false positive — a noisy resource check
|
|
26
26
|
* would teach users to ignore it.
|
|
27
27
|
*
|
|
28
|
+
* The INLINE-CODE path form is weaker than a link, so it carries an extra prose
|
|
29
|
+
* gate: a skill that TEACHES how to build skills mentions bundle paths
|
|
30
|
+
* constantly as EXAMPLES of what a skill *could* ship ("a `scripts/rotate.py`
|
|
31
|
+
* would be helpful to store", "**Examples**: `references/finance.md`"). An
|
|
32
|
+
* inline path is treated as a real reference ONLY when the line DIRECTS the
|
|
33
|
+
* agent to use the file (read/run/see/…) and carries no illustrative cue
|
|
34
|
+
* (example / e.g. / such as / would be / template / →). Markdown links are
|
|
35
|
+
* unchanged — a link is already an act-on-it reference. See `inlinePathIsUsed`.
|
|
36
|
+
*
|
|
28
37
|
* Pure: the only IO is an injectable `existsSync` (default node:fs), mirroring
|
|
29
38
|
* core/refs.ts and the loader so the detector is testable with a fake.
|
|
30
39
|
*/
|
|
@@ -119,6 +128,46 @@ function isInlineBundlePath(token) {
|
|
|
119
128
|
const normalized = t.replace(/^\.\//, "");
|
|
120
129
|
return BUNDLE_PREFIX.test(normalized) && HAS_EXT.test(normalized);
|
|
121
130
|
}
|
|
131
|
+
// ---------------------------------------------------------------------------
|
|
132
|
+
// Inline-path prose gate (don't-cry-wolf on TEACHING / illustrative skills)
|
|
133
|
+
// ---------------------------------------------------------------------------
|
|
134
|
+
//
|
|
135
|
+
// An inline-code bundle path (`` `scripts/foo.py` ``) is a much WEAKER signal
|
|
136
|
+
// than a markdown link — skills that TEACH how to build skills (e.g. the
|
|
137
|
+
// official `skill-development` skill) are full of bundle paths used as
|
|
138
|
+
// EXAMPLES of what a skill *could* contain, not as references to a file the
|
|
139
|
+
// skill actually ships: "a `scripts/rotate_pdf.py` would be helpful to store",
|
|
140
|
+
// "**Examples**: `references/finance.md` …", "- **`references/patterns.md`** —
|
|
141
|
+
// Common patterns". Flagging those as "bundled resource not found" cries wolf
|
|
142
|
+
// and graded a clean, correct skill an F.
|
|
143
|
+
//
|
|
144
|
+
// So an inline path is only treated as a real reference when the surrounding
|
|
145
|
+
// prose DIRECTS the agent to ACT on the file (read/run/see/…) AND carries no
|
|
146
|
+
// illustrative/hypothetical cue. Markdown links (`[text](path)`) are unchanged
|
|
147
|
+
// — a link is already a high-confidence, follow-me reference. We bias HARD
|
|
148
|
+
// toward precision here: missing a real dead-ref is far better than a false
|
|
149
|
+
// positive on a teaching skill (the same don't-cry-wolf discipline as the
|
|
150
|
+
// loader's `danglingRefs`).
|
|
151
|
+
// Verbs that direct the agent to CONSUME an existing file. Deliberately EXCLUDES
|
|
152
|
+
// authoring verbs (store/create/add/move/write) — "a `scripts/x.py` would be
|
|
153
|
+
// helpful to STORE in the skill" is describing a resource to CREATE, exactly
|
|
154
|
+
// what a teaching skill illustrates, not a file the skill already ships.
|
|
155
|
+
const USE_DIRECTIVE = /\b(run|runs|execute|executes|read|reads|load|loads|open|opens|source|sources|import|imports|see|view|refer|follow|call|invoke|apply|consult|check)\b/i;
|
|
156
|
+
// Illustrative / hypothetical prose cues — a line carrying one is describing
|
|
157
|
+
// what a skill MIGHT contain (an example, a suggestion, a template), not
|
|
158
|
+
// pointing at a shipped file. Includes the `→`/`->` arrow used in "move detail
|
|
159
|
+
// → `references/x.md`" authoring lists.
|
|
160
|
+
const ILLUSTRATIVE_CUE = /\b(example|examples|e\.g\.?|i\.e\.?|such as|for instance|would be|helpful|useful|template|boilerplate)\b|→|->/i;
|
|
161
|
+
/**
|
|
162
|
+
* Whether an inline bundle-path on this line reads as a REAL reference (the
|
|
163
|
+
* agent is told to use the file) rather than an illustrative mention. Requires
|
|
164
|
+
* a positive use-directive and the absence of an illustrative cue — both
|
|
165
|
+
* evaluated over the whole line for simplicity (a tight, precise rule over a
|
|
166
|
+
* clever one). Only the inline-path branch consults this; markdown links do not.
|
|
167
|
+
*/
|
|
168
|
+
function inlinePathIsUsed(line) {
|
|
169
|
+
return USE_DIRECTIVE.test(line) && !ILLUSTRATIVE_CUE.test(line);
|
|
170
|
+
}
|
|
122
171
|
/** Collect candidate bundled-resource refs from one body line, skipping fences. */
|
|
123
172
|
function candidatesInLine(line, lineNo) {
|
|
124
173
|
const out = [];
|
|
@@ -129,14 +178,18 @@ function candidatesInLine(line, lineNo) {
|
|
|
129
178
|
out.push({ ref: m[1].trim(), resolved, kind: "link", line: lineNo });
|
|
130
179
|
}
|
|
131
180
|
}
|
|
132
|
-
// Inline-code path mentions: only the high-confidence bundle-dir-prefixed
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
181
|
+
// Inline-code path mentions: only the high-confidence bundle-dir-prefixed
|
|
182
|
+
// form, AND only when the surrounding prose USES the file (not an illustrative
|
|
183
|
+
// mention on a teaching skill — see `inlinePathIsUsed`).
|
|
184
|
+
if (inlinePathIsUsed(line)) {
|
|
185
|
+
for (const m of line.matchAll(INLINE_SPAN)) {
|
|
186
|
+
const token = m[1].trim();
|
|
187
|
+
if (!isInlineBundlePath(token))
|
|
188
|
+
continue;
|
|
189
|
+
const resolved = localResourceTarget(token);
|
|
190
|
+
if (resolved !== null) {
|
|
191
|
+
out.push({ ref: token, resolved, kind: "path", line: lineNo });
|
|
192
|
+
}
|
|
140
193
|
}
|
|
141
194
|
}
|
|
142
195
|
return out;
|
package/dist/dialect-drift.d.ts
CHANGED
|
@@ -43,9 +43,27 @@ export declare function eventsMissingFromBundle(bundle: string, events: readonly
|
|
|
43
43
|
* we only read files the user already installed under their own CC license.
|
|
44
44
|
*/
|
|
45
45
|
export declare function findClaudeCodePackage(): string | null;
|
|
46
|
+
/**
|
|
47
|
+
* Extract the semver core from a `claude --version` line
|
|
48
|
+
* (e.g. `"2.1.211 (Claude Code)"` → `"2.1.211"`). Pure; null when absent.
|
|
49
|
+
*/
|
|
50
|
+
export declare function parseClaudeVersion(raw: string): string | null;
|
|
51
|
+
/**
|
|
52
|
+
* The version of the `claude` binary actually on PATH — which can DIFFER from the
|
|
53
|
+
* package {@link findClaudeCodePackage} locates. In the native-binary era CC ships
|
|
54
|
+
* as a platform binary with no readable `sdk-tools.d.ts`, so a box can carry a
|
|
55
|
+
* stale leftover `@anthropic-ai/claude-code` in a global `node_modules` (readable,
|
|
56
|
+
* but months old and NOT what's running) beside the real, newer CC. Reconciling
|
|
57
|
+
* against this stops the drift alarm crying wolf on that leftover. Real IO; null
|
|
58
|
+
* when `claude` isn't runnable.
|
|
59
|
+
*/
|
|
60
|
+
export declare function onPathClaudeVersion(): string | null;
|
|
46
61
|
/** A runtime drift report: how the INSTALLED CC's tool surface compares to ours. */
|
|
47
62
|
export interface DialectDriftReport {
|
|
63
|
+
/** Version of the located `@anthropic-ai/claude-code` package (its types we read). */
|
|
48
64
|
readonly installedVersion: string;
|
|
65
|
+
/** Version of the `claude` binary on PATH, or null — the actually-running CC. */
|
|
66
|
+
readonly runningVersion: string | null;
|
|
49
67
|
readonly validatedVersion: string;
|
|
50
68
|
/** Tool-input types present in the install but not in ACKNOWLEDGED (CC added). */
|
|
51
69
|
readonly newToolTypes: string[];
|
package/dist/dialect-drift.js
CHANGED
|
@@ -5,6 +5,8 @@ exports.parseToolInputTypes = parseToolInputTypes;
|
|
|
5
5
|
exports.findClaudeCodeBundle = findClaudeCodeBundle;
|
|
6
6
|
exports.eventsMissingFromBundle = eventsMissingFromBundle;
|
|
7
7
|
exports.findClaudeCodePackage = findClaudeCodePackage;
|
|
8
|
+
exports.parseClaudeVersion = parseClaudeVersion;
|
|
9
|
+
exports.onPathClaudeVersion = onPathClaudeVersion;
|
|
8
10
|
exports.checkDialectDrift = checkDialectDrift;
|
|
9
11
|
exports.formatDialectDrift = formatDialectDrift;
|
|
10
12
|
/**
|
|
@@ -36,7 +38,13 @@ exports.formatDialectDrift = formatDialectDrift;
|
|
|
36
38
|
* NOT a readable `cli.js` JS bundle. So the old "grep hook-event string literals out
|
|
37
39
|
* of cli.js" check has no bundle to read and degrades to a LOUD SKIP (see
|
|
38
40
|
* `findClaudeCodeBundle`); `sdk-tools.d.ts` is still shipped, so the tool-type drift
|
|
39
|
-
* alarm keeps working.
|
|
41
|
+
* alarm keeps working. But the native binary means a box can carry a STALE leftover
|
|
42
|
+
* `@anthropic-ai/claude-code` JS package (readable `sdk-tools.d.ts`, months old) in a
|
|
43
|
+
* global `node_modules` while the real, newer CC runs from the native binary — so
|
|
44
|
+
* `findClaudeCodePackage` can locate a package that ISN'T what's running. The runtime
|
|
45
|
+
* alarm (`checkDialectDrift`/`formatDialectDrift`) reconciles the located package
|
|
46
|
+
* version against the on-PATH `claude --version` and SUPPRESSES the warning on a
|
|
47
|
+
* mismatch (the located types don't describe the running CC — crying wolf otherwise).
|
|
40
48
|
*
|
|
41
49
|
* Pure parsers (testable with fixtures) + a local-install locator. TWO consumers:
|
|
42
50
|
* the gated CI test in `dialect-drift.test.ts` (fails loud on tool/event drift), and
|
|
@@ -167,6 +175,31 @@ function findClaudeCodePackage() {
|
|
|
167
175
|
}
|
|
168
176
|
return null;
|
|
169
177
|
}
|
|
178
|
+
/**
|
|
179
|
+
* Extract the semver core from a `claude --version` line
|
|
180
|
+
* (e.g. `"2.1.211 (Claude Code)"` → `"2.1.211"`). Pure; null when absent.
|
|
181
|
+
*/
|
|
182
|
+
function parseClaudeVersion(raw) {
|
|
183
|
+
const m = raw.match(/\b(\d+\.\d+\.\d+)\b/);
|
|
184
|
+
return m ? m[1] : null;
|
|
185
|
+
}
|
|
186
|
+
/**
|
|
187
|
+
* The version of the `claude` binary actually on PATH — which can DIFFER from the
|
|
188
|
+
* package {@link findClaudeCodePackage} locates. In the native-binary era CC ships
|
|
189
|
+
* as a platform binary with no readable `sdk-tools.d.ts`, so a box can carry a
|
|
190
|
+
* stale leftover `@anthropic-ai/claude-code` in a global `node_modules` (readable,
|
|
191
|
+
* but months old and NOT what's running) beside the real, newer CC. Reconciling
|
|
192
|
+
* against this stops the drift alarm crying wolf on that leftover. Real IO; null
|
|
193
|
+
* when `claude` isn't runnable.
|
|
194
|
+
*/
|
|
195
|
+
function onPathClaudeVersion() {
|
|
196
|
+
try {
|
|
197
|
+
return parseClaudeVersion((0, node_child_process_1.execSync)("claude --version", { encoding: "utf-8" }));
|
|
198
|
+
}
|
|
199
|
+
catch {
|
|
200
|
+
return null;
|
|
201
|
+
}
|
|
202
|
+
}
|
|
170
203
|
/**
|
|
171
204
|
* Best-effort, read-local drift check for `scan` (and other runtime callers). Reads
|
|
172
205
|
* only the small `sdk-tools.d.ts` (fast — no `cli.js` bundle scan; events are the
|
|
@@ -191,6 +224,7 @@ function checkDialectDrift() {
|
|
|
191
224
|
}
|
|
192
225
|
return {
|
|
193
226
|
installedVersion,
|
|
227
|
+
runningVersion: onPathClaudeVersion(),
|
|
194
228
|
validatedVersion: exports.VALIDATED_CC_VERSION,
|
|
195
229
|
newToolTypes: [...installed].filter((t) => !ack.has(t)).sort(),
|
|
196
230
|
removedToolTypes: [...ack].filter((t) => !installed.has(t)).sort(),
|
|
@@ -202,7 +236,15 @@ function checkDialectDrift() {
|
|
|
202
236
|
}
|
|
203
237
|
/** A one-line freshness warning if the dialect drifted from the install, else null. */
|
|
204
238
|
function formatDialectDrift(r) {
|
|
205
|
-
if (!r
|
|
239
|
+
if (!r)
|
|
240
|
+
return null;
|
|
241
|
+
// The located `sdk-tools.d.ts` belongs to a DIFFERENT install than the CC
|
|
242
|
+
// actually running (a stale leftover npm package beside a newer native binary),
|
|
243
|
+
// so its tool set says nothing about the CC you're on — don't cry wolf. The
|
|
244
|
+
// pinned CI drift TEST is where real drift is gated.
|
|
245
|
+
if (r.runningVersion && r.runningVersion !== r.installedVersion)
|
|
246
|
+
return null;
|
|
247
|
+
if (r.newToolTypes.length === 0 && r.removedToolTypes.length === 0)
|
|
206
248
|
return null;
|
|
207
249
|
const parts = [];
|
|
208
250
|
if (r.newToolTypes.length > 0)
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
* NOISE that would flood the preview. `isFixturePath` is the precision-first
|
|
8
8
|
* discriminator (over-skip a legit `sample-service` before flooding with fixture
|
|
9
9
|
* rules). Pure + unit-tested; the fs discovery/glue lives in cli.ts
|
|
10
|
-
* (`gatherInstructionFiles`). See research/rule-
|
|
10
|
+
* (`gatherInstructionFiles`). See research/rule-enforcer-multilang-design.md §0.
|
|
11
11
|
*/
|
|
12
12
|
/** One instruction file gathered from disk, with its canonical (symlink-resolved)
|
|
13
13
|
* path so a mirror can be detected. */
|
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* NOISE that would flood the preview. `isFixturePath` is the precision-first
|
|
9
9
|
* discriminator (over-skip a legit `sample-service` before flooding with fixture
|
|
10
10
|
* rules). Pure + unit-tested; the fs discovery/glue lives in cli.ts
|
|
11
|
-
* (`gatherInstructionFiles`). See research/rule-
|
|
11
|
+
* (`gatherInstructionFiles`). See research/rule-enforcer-multilang-design.md §0.
|
|
12
12
|
*/
|
|
13
13
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
14
14
|
exports.dedupeInstructionFiles = dedupeInstructionFiles;
|
package/dist/leaderboard.d.ts
CHANGED
|
@@ -26,7 +26,7 @@ export declare const W_NO_DESCRIPTION = 10;
|
|
|
26
26
|
export declare const W_DANGLING_REF = 8;
|
|
27
27
|
export declare const W_OVERLAP = 8;
|
|
28
28
|
export declare const W_NO_CONTRACT = 5;
|
|
29
|
-
export declare const W_TRIFECTA =
|
|
29
|
+
export declare const W_TRIFECTA = 10;
|
|
30
30
|
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
31
31
|
export declare function gradeFor(score: number): PluginScore["grade"];
|
|
32
32
|
/** One deduction: a count, its per-item weight, and the label if non-zero. */
|
package/dist/leaderboard.js
CHANGED
|
@@ -49,10 +49,12 @@ exports.W_NO_DESCRIPTION = 10; // a skill with no usable description → can't t
|
|
|
49
49
|
exports.W_DANGLING_REF = 8; // a referenced intra-plugin file that's missing → broken path
|
|
50
50
|
exports.W_OVERLAP = 8; // a description collision → the wrong skill fires
|
|
51
51
|
exports.W_NO_CONTRACT = 5; // generic small-footgun weight (disallowedTools typo, invalid model/color)
|
|
52
|
-
exports.W_TRIFECTA =
|
|
52
|
+
exports.W_TRIFECTA = 10; // a HARD lethal-trifecta contract (all three legs, explicit) → a prompt-injection exfil path. HALF the old 20: a DING, not a fail — a trifecta is a real risk worth surfacing in the grade, but official plugins ship the pattern by design, so it dents the score (e.g. feature-dev's 3 hard units → −30 → C) without a catastrophic F.
|
|
53
53
|
// Two things are advisory, NOT graded penalties (shown, never scored — see scoreReport):
|
|
54
54
|
// - untested surfaces — a hardening gap, not breakage.
|
|
55
55
|
// - an agent that inherits all tools (no `tools:` line) — see reportDeductions for why.
|
|
56
|
+
// - an inherits-all (severity "advisory") trifecta finding — shown by the Safety
|
|
57
|
+
// ring but never scored; only the HARD, explicit all-three-legs finding grades.
|
|
56
58
|
/** Map a 0–100 structural-health score to its letter grade (A ≥90 … F <60). */
|
|
57
59
|
function gradeFor(score) {
|
|
58
60
|
if (score >= 90)
|
|
@@ -79,9 +81,11 @@ function reportDeductions(r) {
|
|
|
79
81
|
const deadMcpTools = r.agents.reduce((n, a) => n + a.mcpToolIssues.length, 0);
|
|
80
82
|
const deadDisallowed = r.agents.reduce((n, a) => n + a.disallowedToolIssues.length, 0);
|
|
81
83
|
// HARD lethal-trifecta findings only — an EXPLICIT contract naming all three
|
|
82
|
-
// legs (a
|
|
83
|
-
//
|
|
84
|
-
//
|
|
84
|
+
// legs (a prompt-injection exfil path). Graded at W_TRIFECTA=10 (HALF the old
|
|
85
|
+
// 20): a DING that surfaces a real risk in the grade without a catastrophic F
|
|
86
|
+
// for an accepted design pattern official plugins ship. Advisory (inherits-all)
|
|
87
|
+
// trifecta findings are surfaced but NEVER graded (aligned with the inherits-all
|
|
88
|
+
// stance), so they're excluded here.
|
|
85
89
|
const hardTrifecta = r.trifectaFindings.filter((f) => f.finding.severity === "hard").length;
|
|
86
90
|
return [
|
|
87
91
|
{
|
|
@@ -283,13 +287,13 @@ function formatLeaderboard(scores) {
|
|
|
283
287
|
const issue = s.issues.length > 0 ? ` — ${s.issues.join("; ")}` : "";
|
|
284
288
|
out.push(` ${rank} ${score} ${s.grade} ${s.name}${issue}`);
|
|
285
289
|
});
|
|
286
|
-
out.push("", "Structural health only (no model). Weights:
|
|
290
|
+
out.push("", "Structural health only (no model). Weights: missing hook -15, hard lethal-", "trifecta unit -10, no-description skill -10, broken intra-plugin ref -8, dead", "tool/MCP ref -8. Inherit-all subagents, inherits-all trifecta and untested", "surfaces are advisory — shown, not scored.");
|
|
287
291
|
return out.join("\n");
|
|
288
292
|
}
|
|
289
|
-
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model):
|
|
290
|
-
"
|
|
291
|
-
"Inherit-all subagents and untested surfaces are advisory
|
|
292
|
-
"scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
|
|
293
|
+
const LEADERBOARD_METHOD = "_Structural health only (deterministic, no model): missing hook −15, hard " +
|
|
294
|
+
"lethal-trifecta unit −10, no-description skill −10, broken intra-plugin / dead-tool ref −8. " +
|
|
295
|
+
"Inherit-all subagents, inherits-all trifecta and untested surfaces are advisory " +
|
|
296
|
+
"(shown, not scored). Behavioural columns (trigger-rate, collisions, egress) stack on top._";
|
|
293
297
|
/**
|
|
294
298
|
* Format a ranked leaderboard as a Markdown table — the PUBLISHABLE form (a README,
|
|
295
299
|
* a gist, the leaderboard site). Shows the top 2 deductions per plugin; the full
|
package/dist/rule-inventory.js
CHANGED
|
@@ -42,7 +42,7 @@ exports.buildRuleInventory = buildRuleInventory;
|
|
|
42
42
|
// the DYNAMIC available-rule catalog — enumerate the rules the repo's linter
|
|
43
43
|
// ACTUALLY has (spike: 702 for this repo vs ~23 here) and match prose against
|
|
44
44
|
// THAT, own-repo/consented since it executes the linter. Do NOT keep growing this
|
|
45
|
-
// by hand. See research/rule-
|
|
45
|
+
// by hand. See research/rule-enforcer-multilang-design.md §0.
|
|
46
46
|
exports.INTENT_MAP = [
|
|
47
47
|
{
|
|
48
48
|
intent: "no console.log / use the logger",
|
|
@@ -235,12 +235,93 @@ exports.INTENT_MAP = [
|
|
|
235
235
|
rule: "no-only-tests/no-only-tests",
|
|
236
236
|
configFix: '"no-only-tests/no-only-tests": "error"',
|
|
237
237
|
},
|
|
238
|
+
// --- Ruff (Python) — the modern default (allow-list, like ESLint). These are
|
|
239
|
+
// the NET-NEW intents Pylint does NOT already cover (imports-inside-functions,
|
|
240
|
+
// annotations, logger.exception, print, import-sorting). The SHARED Python
|
|
241
|
+
// intents (bare-except, broad-except, mutable-default, wildcard, global,
|
|
242
|
+
// too-many-*, line-length, unused, f-strings, docstrings) stay with the Pylint
|
|
243
|
+
// entries below — one intent, one linter, so keyword disjointness holds (the
|
|
244
|
+
// `rule-routing-dogfood` invariant test) and prose never double-routes. Whether
|
|
245
|
+
// shared Python NL prose should later PREFER Ruff over Pylint is an open
|
|
246
|
+
// ownership decision, deferred. ROUTE-ONLY like Pylint: config-state
|
|
247
|
+
// (buildRuleInventory) stays gated to eslint — Ruff's select/ignore config shape
|
|
248
|
+
// differs from the eslint rules-map, so the eslint-shaped check would MISLABEL;
|
|
249
|
+
// accurate Ruff enabled-state is the catalog/ConfigProbe follow-up
|
|
250
|
+
// (rule-enforcer-multilang-design.md §3). Every CODE was verified to exist
|
|
251
|
+
// against ruff 0.15.8 (`ruff rule <CODE>`); keywords are Python-diagnostic (no
|
|
252
|
+
// cross-language phrase that would mis-attribute the linter in a JS/TS doc).
|
|
253
|
+
{
|
|
254
|
+
intent: "no imports inside functions (Python)",
|
|
255
|
+
linter: "ruff",
|
|
256
|
+
keywords: [
|
|
257
|
+
"PLC0415",
|
|
258
|
+
"inline import",
|
|
259
|
+
"inline imports",
|
|
260
|
+
"imports inside functions",
|
|
261
|
+
"import inside a function",
|
|
262
|
+
"import inside functions",
|
|
263
|
+
],
|
|
264
|
+
rule: "PLC0415",
|
|
265
|
+
configFix: 'add "PLC0415" (import-outside-top-level) to [tool.ruff.lint] select',
|
|
266
|
+
},
|
|
267
|
+
{
|
|
268
|
+
// E402 flags module-level code BEFORE the imports; the "no imports inside
|
|
269
|
+
// functions" reading is PLC0415 above. Code-only keyword ON PURPOSE: the
|
|
270
|
+
// phrase "imports at the top" is ambiguous between the two AND collides with
|
|
271
|
+
// ESLint `import/first` in a JS doc, so it is NOT a keyword here (a missing
|
|
272
|
+
// route is recoverable; a wrong "enforceable" claim is not).
|
|
273
|
+
intent: "no module code before imports (Python)",
|
|
274
|
+
linter: "ruff",
|
|
275
|
+
keywords: ["E402"],
|
|
276
|
+
rule: "E402",
|
|
277
|
+
configFix: 'add "E402" (module-import-not-at-top-of-file) to [tool.ruff.lint] select',
|
|
278
|
+
},
|
|
279
|
+
{
|
|
280
|
+
intent: "require argument type annotations (Python)",
|
|
281
|
+
linter: "ruff",
|
|
282
|
+
keywords: [
|
|
283
|
+
"ANN001",
|
|
284
|
+
"type hints required",
|
|
285
|
+
"type hints are required",
|
|
286
|
+
"require type hints",
|
|
287
|
+
],
|
|
288
|
+
rule: "ANN001",
|
|
289
|
+
configFix: 'add "ANN001" (missing-type-function-argument) to [tool.ruff.lint] select',
|
|
290
|
+
},
|
|
291
|
+
{
|
|
292
|
+
intent: "require return type annotations (Python)",
|
|
293
|
+
linter: "ruff",
|
|
294
|
+
keywords: ["ANN201"],
|
|
295
|
+
rule: "ANN201",
|
|
296
|
+
configFix: 'add "ANN201" (missing-return-type-undocumented-public-function) to [tool.ruff.lint] select',
|
|
297
|
+
},
|
|
298
|
+
{
|
|
299
|
+
intent: "use logger.exception in handlers (Python)",
|
|
300
|
+
linter: "ruff",
|
|
301
|
+
keywords: ["TRY400", "logger.exception", "logging.exception"],
|
|
302
|
+
rule: "TRY400",
|
|
303
|
+
configFix: 'add "TRY400" (error-instead-of-exception) to [tool.ruff.lint] select',
|
|
304
|
+
},
|
|
305
|
+
{
|
|
306
|
+
intent: "no print statements — use logging (Python)",
|
|
307
|
+
linter: "ruff",
|
|
308
|
+
keywords: ["T201", "print statement", "print statements"],
|
|
309
|
+
rule: "T201",
|
|
310
|
+
configFix: 'add "T201" (print) to [tool.ruff.lint] select',
|
|
311
|
+
},
|
|
312
|
+
{
|
|
313
|
+
intent: "keep imports sorted (Python)",
|
|
314
|
+
linter: "ruff",
|
|
315
|
+
keywords: ["I001", "sorted imports", "isort"],
|
|
316
|
+
rule: "I001",
|
|
317
|
+
configFix: 'add "I001" (unsorted-imports) to [tool.ruff.lint] select',
|
|
318
|
+
},
|
|
238
319
|
// --- Pylint (Python) — routing basics. These feed classify() (routing → reuse);
|
|
239
320
|
// buildRuleInventory is gated to eslint (see below) because pylint is
|
|
240
321
|
// ON-BY-DEFAULT (deny-list), so the eslint-shaped config-state check would
|
|
241
322
|
// MISLABEL it (a symbol in `disable=` reads as "in-config", an absent one as
|
|
242
323
|
// "enable it"). Accurate pylint enabled-state needs the inverted-polarity
|
|
243
|
-
// ConfigProbe (research/rule-
|
|
324
|
+
// ConfigProbe (research/rule-enforcer-multilang-design.md §3), deferred —
|
|
244
325
|
// classify() needs NO enabled-state, so pylint prose still routes honestly.
|
|
245
326
|
// Keywords are code-shaped symbols + Python-UNAMBIGUOUS compounds (singular AND
|
|
246
327
|
// plural, since matchesWholeToken is boundary-exact); bare ambiguous words
|
|
@@ -351,6 +432,74 @@ exports.INTENT_MAP = [
|
|
|
351
432
|
rule: "unused-import",
|
|
352
433
|
configFix: "pylint enables unused-import (W0611) by default; keep it out of the disable list",
|
|
353
434
|
},
|
|
435
|
+
// --- Clippy (Rust) — routing basics (backlog #3: the 0%→ win for Rust
|
|
436
|
+
// rulebooks; codex/ghostty routed 0% purely because clippy was unmapped, yet
|
|
437
|
+
// codex literally says "avoid patterns that require `panic!`, `unreachable!`,
|
|
438
|
+
// or `.unwrap()`" and "make `match` exhaustive … avoid wildcard arms"). Every
|
|
439
|
+
// lint is a real clippy restriction lint. Route-only, like ruff/pylint —
|
|
440
|
+
// config-state stays eslint-gated (L639) because clippy's lint levels live in
|
|
441
|
+
// Cargo.toml `[lints.clippy]` / crate attrs, a different shape (accurate
|
|
442
|
+
// enabled-state is the ConfigProbe follow-up, rule-enforcer-multilang-design
|
|
443
|
+
// §3). Keywords are Rust-UNAMBIGUOUS — macro `!` forms, `.method` call forms,
|
|
444
|
+
// `clippy::` symbols, and Rust-only compounds ("wildcard arm(s)", NOT bare
|
|
445
|
+
// "wildcard" which collides with pylint `wildcard-import`) — so a Python/JS
|
|
446
|
+
// doc never mis-attributes clippy (the cross-language-FP test guards this).
|
|
447
|
+
{
|
|
448
|
+
intent: "no .unwrap() (Rust)",
|
|
449
|
+
linter: "clippy",
|
|
450
|
+
keywords: ["clippy::unwrap_used", "unwrap_used", ".unwrap"],
|
|
451
|
+
rule: "clippy::unwrap_used",
|
|
452
|
+
configFix: 'set unwrap_used = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::unwrap_used)]',
|
|
453
|
+
},
|
|
454
|
+
{
|
|
455
|
+
intent: "no .expect() (Rust)",
|
|
456
|
+
linter: "clippy",
|
|
457
|
+
keywords: ["clippy::expect_used", "expect_used", ".expect"],
|
|
458
|
+
rule: "clippy::expect_used",
|
|
459
|
+
configFix: 'set expect_used = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::expect_used)]',
|
|
460
|
+
},
|
|
461
|
+
{
|
|
462
|
+
intent: "no panic! (Rust)",
|
|
463
|
+
linter: "clippy",
|
|
464
|
+
keywords: ["clippy::panic", "panic!"],
|
|
465
|
+
rule: "clippy::panic",
|
|
466
|
+
configFix: 'set panic = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::panic)]',
|
|
467
|
+
},
|
|
468
|
+
{
|
|
469
|
+
intent: "no unreachable! (Rust)",
|
|
470
|
+
linter: "clippy",
|
|
471
|
+
keywords: ["clippy::unreachable", "unreachable!"],
|
|
472
|
+
rule: "clippy::unreachable",
|
|
473
|
+
configFix: 'set unreachable = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::unreachable)]',
|
|
474
|
+
},
|
|
475
|
+
{
|
|
476
|
+
intent: "no todo! (Rust)",
|
|
477
|
+
linter: "clippy",
|
|
478
|
+
keywords: ["clippy::todo", "todo!"],
|
|
479
|
+
rule: "clippy::todo",
|
|
480
|
+
configFix: 'set todo = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::todo)]',
|
|
481
|
+
},
|
|
482
|
+
{
|
|
483
|
+
intent: "no dbg! (Rust)",
|
|
484
|
+
linter: "clippy",
|
|
485
|
+
keywords: ["clippy::dbg_macro", "dbg!"],
|
|
486
|
+
rule: "clippy::dbg_macro",
|
|
487
|
+
configFix: 'set dbg_macro = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::dbg_macro)]',
|
|
488
|
+
},
|
|
489
|
+
{
|
|
490
|
+
intent: "exhaustive match — no wildcard enum arm (Rust)",
|
|
491
|
+
linter: "clippy",
|
|
492
|
+
keywords: [
|
|
493
|
+
"clippy::wildcard_enum_match_arm",
|
|
494
|
+
"wildcard_enum_match_arm",
|
|
495
|
+
"wildcard match arm",
|
|
496
|
+
"wildcard match arms",
|
|
497
|
+
"wildcard arm",
|
|
498
|
+
"wildcard arms",
|
|
499
|
+
],
|
|
500
|
+
rule: "clippy::wildcard_enum_match_arm",
|
|
501
|
+
configFix: 'set wildcard_enum_match_arm = "warn" in [lints.clippy] (Cargo.toml), or #![warn(clippy::wildcard_enum_match_arm)]',
|
|
502
|
+
},
|
|
354
503
|
];
|
|
355
504
|
/** Escape a keyword for use inside a RegExp. */
|
|
356
505
|
function escapeRe(s) {
|
package/dist/rule-routing.d.ts
CHANGED
|
@@ -41,6 +41,21 @@ import type { RuleCatalog } from "./core/rule-catalog.js";
|
|
|
41
41
|
/** How a routed rule would be enforced (a MECHANISM ladder, not a 1-10 score). */
|
|
42
42
|
export type RuleCategory = "reuse" | "hook" | "meta" | "semantic" | "unrouted";
|
|
43
43
|
export type RuleMechanism = "config-line" | "hook" | "prose" | "synthesize";
|
|
44
|
+
/**
|
|
45
|
+
* The user-facing presentation of each routing category — its glyph + lane
|
|
46
|
+
* label. The SINGLE SOURCE the terminal summary reads (and the HTML report
|
|
47
|
+
* mirrors), so the category-name → lane-label mapping lives in one place.
|
|
48
|
+
*
|
|
49
|
+
* NB the type name `unrouted` is a WIRE value (it appears in the versioned
|
|
50
|
+
* `AuditReport` JSON), which is why it isn't renamed to its lane label `custom`;
|
|
51
|
+
* this table is where the human-facing name is resolved. The category meanings
|
|
52
|
+
* are documented in the file header; the mapping is tabled in
|
|
53
|
+
* `research/rule-enforcer-design.md` §4.
|
|
54
|
+
*/
|
|
55
|
+
export declare const LANE_META: Record<RuleCategory, {
|
|
56
|
+
readonly glyph: string;
|
|
57
|
+
readonly label: string;
|
|
58
|
+
}>;
|
|
44
59
|
/** One segmented, deterministically-routed rule with provenance. */
|
|
45
60
|
export interface RoutedRule {
|
|
46
61
|
/** Normalized atomic rule text (from the segmenter). */
|
|
@@ -75,7 +90,7 @@ export interface RuleRouting {
|
|
|
75
90
|
/** The POSSIBLE tier — rule-ish bullets (medium confidence) that did NOT clear
|
|
76
91
|
* the bar, still classified so a human can review + promote them. Detection is
|
|
77
92
|
* precision-first, so this is where a declarative rule ("Every X must Y") that
|
|
78
|
-
* the confident tier misses shows up. See `research/rule-
|
|
93
|
+
* the confident tier misses shows up. See `research/rule-enforcer-design.md` §2. */
|
|
79
94
|
readonly possible: readonly RoutedRule[];
|
|
80
95
|
/** Bullets the segmenter decided were NOT rules, each with a reason — so the
|
|
81
96
|
* report is honest about what it set aside (§3). */
|