vigiles 10.0.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/plugin.json +9 -0
- package/README.md +121 -86
- package/action.yml +13 -2
- package/dist/adapter-conformance.js +6 -0
- package/dist/adapter-registry.d.ts +20 -0
- package/dist/adapter-registry.js +27 -0
- package/dist/adapters/claude-code/dialect.js +15 -0
- package/dist/adapters/claude-code/hook-protocol.js +4 -0
- package/dist/adapters/claude-code/runtime.js +12 -0
- package/dist/adapters/codex/eval.js +3 -0
- package/dist/adapters/codex/hook-protocol.d.ts +9 -1
- package/dist/adapters/codex/hook-protocol.js +10 -0
- package/dist/adapters/codex/runtime.js +10 -0
- package/dist/adapters/opencode/runtime.js +4 -0
- package/dist/audit-report.d.ts +1 -1
- package/dist/audit-report.template.html +1 -1
- package/dist/audit-score.d.ts +19 -12
- package/dist/audit-score.js +65 -11
- package/dist/cli-commands.d.ts +1 -1
- package/dist/cli-commands.js +1 -0
- package/dist/cli.js +460 -29
- package/dist/core/CLAUDE.md.spec.d.ts +3 -0
- package/dist/core/CLAUDE.md.spec.js +26 -0
- package/dist/core/delegation-trifecta.d.ts +64 -0
- package/dist/core/delegation-trifecta.js +124 -0
- package/dist/core/dialect.d.ts +18 -0
- package/dist/core/hook-block-ineffective.d.ts +62 -0
- package/dist/core/hook-block-ineffective.js +153 -0
- package/dist/core/hook-matcher.d.ts +66 -0
- package/dist/core/hook-matcher.js +182 -0
- package/dist/core/hook-normalize.d.ts +43 -0
- package/dist/core/hook-normalize.js +78 -0
- package/dist/core/hook-protocol.d.ts +15 -0
- package/dist/core/lethal-trifecta.d.ts +100 -0
- package/dist/core/lethal-trifecta.js +197 -0
- package/dist/core/plugin-dir-layout.d.ts +30 -0
- package/dist/core/plugin-dir-layout.js +73 -0
- package/dist/core/rule-meta.d.ts +82 -0
- package/dist/core/rule-meta.js +266 -0
- package/dist/core/runtime.d.ts +20 -0
- package/dist/core/skill-missing-fence.d.ts +47 -0
- package/dist/core/skill-missing-fence.js +119 -0
- package/dist/core/skill-resources.d.ts +27 -0
- package/dist/core/skill-resources.js +167 -0
- package/dist/core/types.d.ts +83 -0
- package/dist/core/validate.d.ts +1 -0
- package/dist/core/validate.js +26 -4
- package/dist/eval-cache.d.ts +6 -0
- package/dist/eval-cache.js +2 -0
- package/dist/eval-lock.d.ts +192 -0
- package/dist/eval-lock.js +286 -0
- package/dist/eval.d.ts +33 -20
- package/dist/eval.js +199 -51
- package/dist/leaderboard.d.ts +1 -0
- package/dist/leaderboard.js +42 -4
- package/dist/scan.d.ts +106 -0
- package/dist/scan.js +251 -45
- package/dist/setup-plan.d.ts +43 -3
- package/dist/setup-plan.js +78 -6
- package/hooks/eval-lock-nudge.sh +21 -0
- package/package.json +1 -1
- package/skills/test-harness/SKILL.md +27 -0
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
/**
|
|
4
|
+
* Directory-scoped guidance for working in `src/core/` (the harness-agnostic
|
|
5
|
+
* detectors + domain).
|
|
6
|
+
*
|
|
7
|
+
* The full project rule set is the ROOT `CLAUDE.md` (compiled from
|
|
8
|
+
* `CLAUDE.md.spec.ts`). This nested spec adds only the discipline that belongs
|
|
9
|
+
* next to the detectors themselves — Claude Code loads it as directory memory
|
|
10
|
+
* whenever you work in `src/core/`. Source of truth; `src/core/CLAUDE.md` is a
|
|
11
|
+
* compiled build artifact (`vigiles compile`).
|
|
12
|
+
*/
|
|
13
|
+
const spec_js_1 = require("./spec.js");
|
|
14
|
+
exports.default = (0, spec_js_1.claude)({
|
|
15
|
+
sections: {
|
|
16
|
+
scope: `Working in \`src/core/\`? This is the harness-AGNOSTIC domain (spec, compile, linters, the lint/audit detectors). The root \`CLAUDE.md\` holds the full positioning + rule set — read it first. Two invariants live closest to this code: the core must not import an adapter (\`core ⊄ adapter\`, eslint-enforced) and must not hard-code a Claude Code literal (read it from the injected layout/dialect). This file adds the rule for ADDING or CHANGING a detector.`,
|
|
17
|
+
},
|
|
18
|
+
keyFiles: {
|
|
19
|
+
"src/core/rule-meta.ts": "The RuleMeta registry — every rule's decidability bucket + severity + detector, the single source the detector-meta rule enforces.",
|
|
20
|
+
"src/core/types.ts": "RulesConfig — the rule-name keys the registry is keyed on.",
|
|
21
|
+
},
|
|
22
|
+
rules: {
|
|
23
|
+
"detector-meta": (0, spec_js_1.guidance)("A deterministic DETECTOR here is one half of a RULE — and a rule is not done until it is DECLARED. Three things move together (sibling of one-detector-no-drift + rules-docs-in-sync): (1) the pure detector function (shared by `lint` AND `audit`, never reimplemented per surface; read the layout/dialect, never a CC literal); (2) its entry in `src/core/rule-meta.ts` — the `Record<RuleName, RuleMeta>` won't typecheck without it — declaring its DECIDABILITY BUCKET (structural-closed = a type could prevent it / external-decidable = needs the world, error-capable / heuristic-behavioral = warn-or-measure-only), surface, defaultSeverity, the detector name, and any upstreamPrevention; (3) its `docs/rules/<name>.md` (the coverage test binds the registry to the docs by an EXACT set match, so a missing meta or doc fails CI). The bucket is the CEILING, not a preference — a heuristic proxy may NEVER default to `error` (it cries wolf); a structural/external fact MAY, once proven FP-safe. Before writing a new detector, CLASSIFY the defect into a bucket — that decides whether it can ever gate. The full model + the prose behind the buckets is the root `lint-rule-calibration` rule and `research/enforcement-model.md`."),
|
|
24
|
+
},
|
|
25
|
+
});
|
|
26
|
+
//# sourceMappingURL=CLAUDE.md.spec.js.map
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DELEGATION-TRIFECTA — the lethal trifecta SPLIT across a delegation edge.
|
|
3
|
+
*
|
|
4
|
+
* The per-unit {@link lethalTrifectaIssues} (`./lethal-trifecta.ts`) catches a
|
|
5
|
+
* single subagent / skill that holds all THREE capability legs at once. But the
|
|
6
|
+
* trifecta can EMERGE across a delegation (or inheritance) edge: a parent that
|
|
7
|
+
* can only read private data (leg A) delegates to a child that can ingest
|
|
8
|
+
* untrusted content AND exfiltrate (legs B+C). NEITHER unit trips the per-unit
|
|
9
|
+
* check, yet the CHAIN — the parent plus everything it can reach — forms the full
|
|
10
|
+
* trifecta. A prompt injection in the child's untrusted input can pivot back
|
|
11
|
+
* through the delegation and leak the parent's private data.
|
|
12
|
+
*
|
|
13
|
+
* This is CAPABILITY-DIFF ACROSS THE DELEGATION TREE: the EFFECTIVE (combined)
|
|
14
|
+
* capability of a unit is the union of its own tools plus the tools of every unit
|
|
15
|
+
* reachable through `delegatesTo`. We classify that effective set and flag a unit
|
|
16
|
+
* whose effective set is a full trifecta while its OWN set is not.
|
|
17
|
+
*
|
|
18
|
+
* NO DOUBLE-REPORT (one-detector-no-drift / don't-cry-wolf): a unit whose OWN
|
|
19
|
+
* tools already form a full trifecta is SKIPPED here — {@link lethalTrifectaIssues}
|
|
20
|
+
* owns it. This detector reports ONLY the EMERGENT case the per-unit check can't
|
|
21
|
+
* see.
|
|
22
|
+
*
|
|
23
|
+
* HIGH-PRECISION (FP-safe): if the effective set contains a wildcard (inherits-all)
|
|
24
|
+
* the unit reaches everything and would always "trifecta" — that maximal-blast-
|
|
25
|
+
* radius case is the per-unit ADVISORY detector's job, so we SKIP it. We flag ONLY
|
|
26
|
+
* concrete, explicit tool unions where every leg is supplied by a named tool.
|
|
27
|
+
*
|
|
28
|
+
* Pure, no IO. The delegation graph (nodes + directed `delegatesTo` edges) is the
|
|
29
|
+
* INPUT — this module does not decide where edges come from; a caller supplies them
|
|
30
|
+
* from the parsed harness. The dialect is injected (core ⊄ adapter), reused for the
|
|
31
|
+
* underlying leg classification.
|
|
32
|
+
*/
|
|
33
|
+
import type { HarnessDialect } from "./dialect.js";
|
|
34
|
+
import { type TrifectaLegs } from "./lethal-trifecta.js";
|
|
35
|
+
/** One unit (subagent/skill) in the delegation graph. */
|
|
36
|
+
export interface CapabilityNode {
|
|
37
|
+
readonly name: string;
|
|
38
|
+
readonly kind: "skill" | "agent";
|
|
39
|
+
/** This unit's OWN declared tools. [] = none declared. ["*"] = inherits-all (wildcard). */
|
|
40
|
+
readonly tools: readonly string[];
|
|
41
|
+
/** Names of units this one can delegate to / inherits capabilities from (directed edges). */
|
|
42
|
+
readonly delegatesTo: readonly string[];
|
|
43
|
+
}
|
|
44
|
+
/** A trifecta that EMERGES across delegation — present in a unit's effective set but NOT its own. */
|
|
45
|
+
export interface DelegationTrifectaFinding {
|
|
46
|
+
readonly name: string;
|
|
47
|
+
readonly kind: "skill" | "agent";
|
|
48
|
+
/** The delegated-to units (by name) that supplied at least one leg the unit lacks on its own. */
|
|
49
|
+
readonly via: readonly string[];
|
|
50
|
+
/** The tools that supplied each leg in the EFFECTIVE (combined) set. */
|
|
51
|
+
readonly legs: TrifectaLegs;
|
|
52
|
+
/** Ready-to-show, actionable message. */
|
|
53
|
+
readonly message: string;
|
|
54
|
+
}
|
|
55
|
+
/**
|
|
56
|
+
* Find units whose EFFECTIVE (own + delegated) capability set forms a full lethal
|
|
57
|
+
* trifecta that their OWN set does not — an emergent, cross-delegation exfil path.
|
|
58
|
+
*
|
|
59
|
+
* Returns findings in stable order (by node name). See the module header for the
|
|
60
|
+
* skip rules (own-set already a trifecta → owned by the per-unit detector; an
|
|
61
|
+
* effective wildcard → owned by the per-unit advisory).
|
|
62
|
+
*/
|
|
63
|
+
export declare function delegationTrifectaIssues(nodes: readonly CapabilityNode[], dialect: HarnessDialect): DelegationTrifectaFinding[];
|
|
64
|
+
//# sourceMappingURL=delegation-trifecta.d.ts.map
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.delegationTrifectaIssues = delegationTrifectaIssues;
|
|
4
|
+
const lethal_trifecta_js_1 = require("./lethal-trifecta.js");
|
|
5
|
+
// ---------------------------------------------------------------------------
|
|
6
|
+
// Internal helpers
|
|
7
|
+
// ---------------------------------------------------------------------------
|
|
8
|
+
/** A full trifecta = all three legs non-empty. */
|
|
9
|
+
function isFullTrifecta(legs) {
|
|
10
|
+
return (legs.private.length > 0 &&
|
|
11
|
+
legs.untrusted.length > 0 &&
|
|
12
|
+
legs.exfil.length > 0);
|
|
13
|
+
}
|
|
14
|
+
/** Strips a `Tool(restriction)` suffix and returns the base tool name. */
|
|
15
|
+
function baseTool(raw) {
|
|
16
|
+
return raw.split("(")[0].trim();
|
|
17
|
+
}
|
|
18
|
+
/** True for the wildcard sentinels that mean "inherits-all". */
|
|
19
|
+
function isWildcard(tool) {
|
|
20
|
+
return tool === "" || tool === "*";
|
|
21
|
+
}
|
|
22
|
+
/**
|
|
23
|
+
* The set of node names reachable from `start` over `delegatesTo`, INCLUDING
|
|
24
|
+
* `start` itself. Cycle-safe (a `visited` set). An edge naming a node not in the
|
|
25
|
+
* map is skipped — its tools can't be resolved.
|
|
26
|
+
*/
|
|
27
|
+
function effectiveReach(start, byName) {
|
|
28
|
+
const visited = new Set();
|
|
29
|
+
const stack = [start];
|
|
30
|
+
while (stack.length > 0) {
|
|
31
|
+
const name = stack.pop();
|
|
32
|
+
if (name === undefined || visited.has(name))
|
|
33
|
+
continue;
|
|
34
|
+
visited.add(name);
|
|
35
|
+
const node = byName.get(name);
|
|
36
|
+
if (node === undefined)
|
|
37
|
+
continue;
|
|
38
|
+
for (const next of node.delegatesTo) {
|
|
39
|
+
if (!visited.has(next))
|
|
40
|
+
stack.push(next);
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
return visited;
|
|
44
|
+
}
|
|
45
|
+
// ---------------------------------------------------------------------------
|
|
46
|
+
// Public API
|
|
47
|
+
// ---------------------------------------------------------------------------
|
|
48
|
+
/**
|
|
49
|
+
* Find units whose EFFECTIVE (own + delegated) capability set forms a full lethal
|
|
50
|
+
* trifecta that their OWN set does not — an emergent, cross-delegation exfil path.
|
|
51
|
+
*
|
|
52
|
+
* Returns findings in stable order (by node name). See the module header for the
|
|
53
|
+
* skip rules (own-set already a trifecta → owned by the per-unit detector; an
|
|
54
|
+
* effective wildcard → owned by the per-unit advisory).
|
|
55
|
+
*/
|
|
56
|
+
function delegationTrifectaIssues(nodes, dialect) {
|
|
57
|
+
const byName = new Map();
|
|
58
|
+
for (const node of nodes)
|
|
59
|
+
byName.set(node.name, node);
|
|
60
|
+
const findings = [];
|
|
61
|
+
for (const node of nodes) {
|
|
62
|
+
// (b) If the unit's OWN tools already form a full trifecta, the per-unit
|
|
63
|
+
// detector owns it — never double-report.
|
|
64
|
+
const ownLegs = (0, lethal_trifecta_js_1.classifyTrifectaLegs)(node.tools, dialect);
|
|
65
|
+
if (isFullTrifecta(ownLegs))
|
|
66
|
+
continue;
|
|
67
|
+
// (c) Effective tools = the de-duplicated union across the reachable set.
|
|
68
|
+
const reach = effectiveReach(node.name, byName);
|
|
69
|
+
const effSet = new Set();
|
|
70
|
+
for (const name of reach) {
|
|
71
|
+
const reached = byName.get(name);
|
|
72
|
+
if (reached === undefined)
|
|
73
|
+
continue;
|
|
74
|
+
for (const tool of reached.tools)
|
|
75
|
+
effSet.add(tool);
|
|
76
|
+
}
|
|
77
|
+
const effectiveTools = [...effSet];
|
|
78
|
+
// (d) FP-safe wildcard guard: an inherits-all unit in the reachable set
|
|
79
|
+
// would always "trifecta" — that's the per-unit advisory's job.
|
|
80
|
+
if (effectiveTools.some((t) => isWildcard(baseTool(t))))
|
|
81
|
+
continue;
|
|
82
|
+
// (e) Classify the effective set.
|
|
83
|
+
const effLegs = (0, lethal_trifecta_js_1.classifyTrifectaLegs)(effectiveTools, dialect);
|
|
84
|
+
// (f) Emit ONLY when the effective set is a full trifecta (own set wasn't).
|
|
85
|
+
if (!isFullTrifecta(effLegs))
|
|
86
|
+
continue;
|
|
87
|
+
// The tools that supplied any leg in the effective set.
|
|
88
|
+
const legTools = new Set([
|
|
89
|
+
...effLegs.private,
|
|
90
|
+
...effLegs.untrusted,
|
|
91
|
+
...effLegs.exfil,
|
|
92
|
+
]);
|
|
93
|
+
// `via` = reachable units (excluding this node) that contribute a leg tool.
|
|
94
|
+
const via = [];
|
|
95
|
+
for (const name of reach) {
|
|
96
|
+
if (name === node.name)
|
|
97
|
+
continue;
|
|
98
|
+
const reached = byName.get(name);
|
|
99
|
+
if (reached === undefined)
|
|
100
|
+
continue;
|
|
101
|
+
const contributes = reached.tools.some((t) => legTools.has(baseTool(t)));
|
|
102
|
+
if (contributes && !via.includes(name))
|
|
103
|
+
via.push(name);
|
|
104
|
+
}
|
|
105
|
+
via.sort();
|
|
106
|
+
const message = `Subagent "${node.name}" is not a data-leak risk on its own, but combined ` +
|
|
107
|
+
`with what it delegates to (${via.join(", ")}), the chain can read private ` +
|
|
108
|
+
`data (${effLegs.private.join(", ")}), ingest untrusted content ` +
|
|
109
|
+
`(${effLegs.untrusted.join(", ")}), AND exfiltrate ` +
|
|
110
|
+
`(${effLegs.exfil.join(", ")}) — a prompt injection in the untrusted input ` +
|
|
111
|
+
`could pivot through the delegation to leak data. Break the delegation or ` +
|
|
112
|
+
`drop one leg.`;
|
|
113
|
+
findings.push({
|
|
114
|
+
name: node.name,
|
|
115
|
+
kind: node.kind,
|
|
116
|
+
via,
|
|
117
|
+
legs: effLegs,
|
|
118
|
+
message,
|
|
119
|
+
});
|
|
120
|
+
}
|
|
121
|
+
findings.sort((a, b) => a.name.localeCompare(b.name));
|
|
122
|
+
return findings;
|
|
123
|
+
}
|
|
124
|
+
//# sourceMappingURL=delegation-trifecta.js.map
|
package/dist/core/dialect.d.ts
CHANGED
|
@@ -42,6 +42,24 @@ export interface HarnessDialect {
|
|
|
42
42
|
readonly knownMcpServers?: readonly string[];
|
|
43
43
|
/** Hook event names the harness fires. */
|
|
44
44
|
readonly hookEvents: readonly string[];
|
|
45
|
+
/**
|
|
46
|
+
* The subset of `hookEvents` where a block decision (`exit 2` / a deny field)
|
|
47
|
+
* is SILENTLY IGNORED ENTIRELY — no veto AND no model feedback (Claude Code's
|
|
48
|
+
* SessionStart / SessionEnd / Notification / PreCompact: exit 2 there writes
|
|
49
|
+
* stderr only to the user). The basis for the `hook-block-ineffective`
|
|
50
|
+
* "wrong-event" check, which fires ONLY on these (so it stays FP-safe and never
|
|
51
|
+
* cries wolf on a PostToolUse feedback/nudge hook). Optional (additive,
|
|
52
|
+
* non-breaking) — absent ⇒ the harness's block semantics are undeclared and the
|
|
53
|
+
* check does not run for it.
|
|
54
|
+
*/
|
|
55
|
+
readonly noEffectHookEvents?: readonly string[];
|
|
56
|
+
/**
|
|
57
|
+
* The subset of blocking events whose deny REQUIRES the structured
|
|
58
|
+
* `permissionDecision` field (e.g. Claude Code's `PreToolUse`), where the
|
|
59
|
+
* legacy top-level `decision` field is silently ignored. The basis for the
|
|
60
|
+
* `hook-block-ineffective` "wrong-field" check. Optional (additive).
|
|
61
|
+
*/
|
|
62
|
+
readonly permissionDecisionHookEvents?: readonly string[];
|
|
45
63
|
/** Instruction-file targets the harness reads (also the h1 heading). */
|
|
46
64
|
readonly instructionTargets: readonly string[];
|
|
47
65
|
/** The env token expanded to the plugin root in hook commands. */
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/** The two "looks like it blocks but doesn't" failure shapes. */
|
|
2
|
+
export type HookBlockKind = "wrong-event" | "wrong-field";
|
|
3
|
+
/** One false-confidence finding: a hook that appears to block but silently won't. */
|
|
4
|
+
export interface HookBlockFinding {
|
|
5
|
+
readonly event: string;
|
|
6
|
+
readonly kind: HookBlockKind;
|
|
7
|
+
/** The hook script that was inspected (path if resolved from a script file, else null = inline command). */
|
|
8
|
+
readonly scriptPath: string | null;
|
|
9
|
+
readonly message: string;
|
|
10
|
+
}
|
|
11
|
+
/** A single hook registration to inspect. */
|
|
12
|
+
export interface HookScriptEntry {
|
|
13
|
+
readonly event: string;
|
|
14
|
+
readonly matcher?: string;
|
|
15
|
+
/** The hook command line as registered. */
|
|
16
|
+
readonly command: string;
|
|
17
|
+
/** Resolved path to the script file the command runs, if known (else null → inspect `command`). */
|
|
18
|
+
readonly scriptPath?: string | null;
|
|
19
|
+
}
|
|
20
|
+
/** Options for {@link hookBlockIssues}. All fields are injectable for testability. */
|
|
21
|
+
export interface HookBlockOptions {
|
|
22
|
+
/**
|
|
23
|
+
* Events on this harness where a block decision (`exit 2` / a deny field) is
|
|
24
|
+
* SILENTLY IGNORED ENTIRELY — no veto AND no feedback to the model (Claude
|
|
25
|
+
* Code's `SessionStart`/`SessionEnd`/`Notification`/`PreCompact`: exit 2 there
|
|
26
|
+
* only writes stderr to the user). These are the ONLY events the "wrong-event"
|
|
27
|
+
* check fires on, so it stays FP-safe.
|
|
28
|
+
*
|
|
29
|
+
* Deliberately EXCLUDES `PostToolUse`: a `PostToolUse` exit 2 feeds stderr back
|
|
30
|
+
* to the model — a legitimate FEEDBACK channel, NOT a failed block — so flagging
|
|
31
|
+
* it would cry wolf on every nudge/lint hook (the dogfood lesson: vigiles's own
|
|
32
|
+
* `refs-nudge.sh` is exactly that shape). A hook that exits 2 on `PostToolUse`
|
|
33
|
+
* intending to BLOCK is misguided, but that intent is not deterministically
|
|
34
|
+
* distinguishable from feedback, so we don't flag it.
|
|
35
|
+
*/
|
|
36
|
+
readonly noEffectEvents: ReadonlySet<string>;
|
|
37
|
+
/**
|
|
38
|
+
* Events that require the structured `permissionDecision` field for a deny
|
|
39
|
+
* (e.g. `PreToolUse`). On these events the legacy top-level `"decision":"block"`
|
|
40
|
+
* field is ignored; only `hookSpecificOutput.permissionDecision:"deny"` works.
|
|
41
|
+
*/
|
|
42
|
+
readonly permissionDecisionEvents: ReadonlySet<string>;
|
|
43
|
+
/**
|
|
44
|
+
* Injectable file read (default: node:fs `readFileSync(p, "utf8")`).
|
|
45
|
+
* Returns `""` on any error so a missing / unreadable script doesn't crash
|
|
46
|
+
* the detector — it simply produces no findings for that entry.
|
|
47
|
+
*/
|
|
48
|
+
readonly readFileSync?: (p: string) => string;
|
|
49
|
+
}
|
|
50
|
+
/**
|
|
51
|
+
* Detect false-confidence "blocks" across a set of hook entries.
|
|
52
|
+
*
|
|
53
|
+
* Returns one {@link HookBlockFinding} per entry (at most one per entry —
|
|
54
|
+
* `wrong-event` takes precedence over `wrong-field`). Identical
|
|
55
|
+
* (event, kind, scriptPath) pairs are de-duped.
|
|
56
|
+
*
|
|
57
|
+
* @param entries - The hook registrations to inspect (event + command/script).
|
|
58
|
+
* @param opts - Injected sets of blocking/permission events, and an optional
|
|
59
|
+
* `readFileSync` (default: node:fs).
|
|
60
|
+
*/
|
|
61
|
+
export declare function hookBlockIssues(entries: readonly HookScriptEntry[], opts: HookBlockOptions): HookBlockFinding[];
|
|
62
|
+
//# sourceMappingURL=hook-block-ineffective.d.ts.map
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.hookBlockIssues = hookBlockIssues;
|
|
4
|
+
/**
|
|
5
|
+
* Hook-block-ineffective detector — the #1 verified hook pain ("false confidence").
|
|
6
|
+
*
|
|
7
|
+
* A safety hook that LOOKS like it blocks but SILENTLY DOESN'T. Two shapes:
|
|
8
|
+
*
|
|
9
|
+
* 1. **wrong-event** — the script tries to block (contains `exit 2`, a legacy
|
|
10
|
+
* `"decision":"block"` JSON, or a `"permissionDecision":"deny"`) but is
|
|
11
|
+
* registered on an event where a block is SILENTLY IGNORED ENTIRELY —
|
|
12
|
+
* SessionStart / SessionEnd / Notification / PreCompact, where exit 2 writes
|
|
13
|
+
* stderr only to the user (no veto, no model feedback). The author believes a
|
|
14
|
+
* gate is in place; nothing happens. (#19009 names the class.) NB `PostToolUse`
|
|
15
|
+
* is deliberately NOT flagged: there exit 2 feeds stderr back to the model — a
|
|
16
|
+
* legitimate FEEDBACK channel (a nudge/lint hook), not a failed block, and the
|
|
17
|
+
* block-vs-feedback intent isn't deterministically separable.
|
|
18
|
+
*
|
|
19
|
+
* 2. **wrong-field** — the hook IS on a permission-gated event (e.g. PreToolUse)
|
|
20
|
+
* but emits the LEGACY top-level `"decision":"block"` field instead of the
|
|
21
|
+
* required `hookSpecificOutput.permissionDecision:"deny"`. A copied template
|
|
22
|
+
* (PostToolUse style → PreToolUse registration) is the usual cause; the deny
|
|
23
|
+
* is silently discarded, nothing is blocked.
|
|
24
|
+
*
|
|
25
|
+
* Both shapes cause identical user-visible behaviour: the hook "works" (exits,
|
|
26
|
+
* no crash) but never actually stops anything. The only signal is "why did this
|
|
27
|
+
* run anyway?" after an incident.
|
|
28
|
+
*
|
|
29
|
+
* FP-SAFETY — conservative literal patterns only, `warn` severity by default:
|
|
30
|
+
* - `exit 2` is matched by a shell-context regex that requires surrounding
|
|
31
|
+
* whitespace/control chars to avoid false-positives on `exit 200` or a
|
|
32
|
+
* `git status --exit-code 2` argument.
|
|
33
|
+
* - JSON decision patterns are matched literally (no partial JSON walking).
|
|
34
|
+
* - Nothing is flagged on an unknown event (we only know which events CAN
|
|
35
|
+
* block because the caller injected that set).
|
|
36
|
+
*
|
|
37
|
+
* HARNESS-NEUTRAL — the sets of blocking events and permission-decision events
|
|
38
|
+
* are INJECTED from the dialect (never hard-coded here). The caller supplies the
|
|
39
|
+
* Claude Code sets; a Codex adapter supplies its own. This is the same
|
|
40
|
+
* dependency-injection pattern used by `verifyHookEvents` and `verifyToolContract`
|
|
41
|
+
* (core ⊄ adapter — one-detector-no-drift).
|
|
42
|
+
*
|
|
43
|
+
* ONE detector reused by scan + the `hook-block-ineffective` lint rule (two
|
|
44
|
+
* callers, no drift). See `research/hook-pain-points.md` for the verified corpus
|
|
45
|
+
* and `docs/compiled-hooks.md` for the authoritative fix (compiled hooks make
|
|
46
|
+
* this whole class unrepresentable).
|
|
47
|
+
*/
|
|
48
|
+
const node_fs_1 = require("node:fs");
|
|
49
|
+
// ---------------------------------------------------------------------------
|
|
50
|
+
// Block-mechanism patterns (conservative / FP-safe)
|
|
51
|
+
// ---------------------------------------------------------------------------
|
|
52
|
+
/**
|
|
53
|
+
* An `exit 2` statement.
|
|
54
|
+
*
|
|
55
|
+
* Requires a shell-context separator before `exit` (start-of-line, whitespace,
|
|
56
|
+
* `;`, `&`, `|`) and word-boundary / separator after `2` — so `exit 200` and
|
|
57
|
+
* `--exit-code 2` are NOT matched.
|
|
58
|
+
*/
|
|
59
|
+
const EXIT_2 = /(^|[\s;&|])exit\s+2(\s|;|$|['")])/m;
|
|
60
|
+
/**
|
|
61
|
+
* A status-2 exit in a NON-shell hook script (a hook file may be `.js`/`.mjs`/
|
|
62
|
+
* `.py`/`.rb`): `process.exit(2)` (Node), `sys.exit(2)` / `exit(2)` (Python),
|
|
63
|
+
* `Process.exit(2)` / `exit(2)` (Ruby), `os._exit(2)`. So a guard written in
|
|
64
|
+
* Node/Python on a no-effect event isn't shown clean.
|
|
65
|
+
*/
|
|
66
|
+
const EXIT_2_CODE = /\b(?:process\.exit|sys\.exit|os\._exit|Process\.exit|exit)\s*\(\s*2\s*\)/;
|
|
67
|
+
/**
|
|
68
|
+
* A legacy top-level `"decision":"block"` or `"decision":"deny"` JSON field.
|
|
69
|
+
* This is the OLD Claude Code hook output format. On permission-gated events
|
|
70
|
+
* (PreToolUse) it is ignored; on non-blocking events it never had any effect.
|
|
71
|
+
*/
|
|
72
|
+
const DECISION_BLOCK = /"decision"\s*:\s*"(block|deny)"/;
|
|
73
|
+
/**
|
|
74
|
+
* The CORRECT structured deny for permission-gated events:
|
|
75
|
+
* `"permissionDecision":"deny"` or `"permissionDecision":"ask"`.
|
|
76
|
+
* (Both require a structured response, as opposed to the legacy field.)
|
|
77
|
+
*/
|
|
78
|
+
const PERMISSION_DENY = /"permissionDecision"\s*:\s*"(deny|ask)"/;
|
|
79
|
+
// ---------------------------------------------------------------------------
|
|
80
|
+
// Detector
|
|
81
|
+
// ---------------------------------------------------------------------------
|
|
82
|
+
function defaultReadFile(p) {
|
|
83
|
+
try {
|
|
84
|
+
return (0, node_fs_1.readFileSync)(p, "utf8");
|
|
85
|
+
}
|
|
86
|
+
catch {
|
|
87
|
+
return "";
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
/**
|
|
91
|
+
* Detect false-confidence "blocks" across a set of hook entries.
|
|
92
|
+
*
|
|
93
|
+
* Returns one {@link HookBlockFinding} per entry (at most one per entry —
|
|
94
|
+
* `wrong-event` takes precedence over `wrong-field`). Identical
|
|
95
|
+
* (event, kind, scriptPath) pairs are de-duped.
|
|
96
|
+
*
|
|
97
|
+
* @param entries - The hook registrations to inspect (event + command/script).
|
|
98
|
+
* @param opts - Injected sets of blocking/permission events, and an optional
|
|
99
|
+
* `readFileSync` (default: node:fs).
|
|
100
|
+
*/
|
|
101
|
+
function hookBlockIssues(entries, opts) {
|
|
102
|
+
const { noEffectEvents, permissionDecisionEvents } = opts;
|
|
103
|
+
const readFile = opts.readFileSync ?? defaultReadFile;
|
|
104
|
+
const findings = [];
|
|
105
|
+
const seen = new Set();
|
|
106
|
+
for (const entry of entries) {
|
|
107
|
+
// Determine the script text and the canonical "path" label for findings.
|
|
108
|
+
const scriptPath = entry.scriptPath ?? null;
|
|
109
|
+
const text = scriptPath !== null ? readFile(scriptPath) : (entry.command ?? "");
|
|
110
|
+
// Detect block mechanisms.
|
|
111
|
+
const hasExit2 = EXIT_2.test(text) || EXIT_2_CODE.test(text);
|
|
112
|
+
const hasDecisionBlock = DECISION_BLOCK.test(text);
|
|
113
|
+
const hasPermissionDeny = PERMISSION_DENY.test(text);
|
|
114
|
+
const triesBlock = hasExit2 || hasDecisionBlock || hasPermissionDeny;
|
|
115
|
+
if (!triesBlock)
|
|
116
|
+
continue;
|
|
117
|
+
let kind = null;
|
|
118
|
+
let message = "";
|
|
119
|
+
if (noEffectEvents.has(entry.event)) {
|
|
120
|
+
// wrong-event: a block on an event where it's silently ignored entirely
|
|
121
|
+
// (no veto AND no feedback — stderr goes to the user, not the model).
|
|
122
|
+
kind = "wrong-event";
|
|
123
|
+
message =
|
|
124
|
+
`This hook tries to block (exit 2 / "decision" / "permissionDecision") ` +
|
|
125
|
+
`but on "${entry.event}" a block decision is silently ignored — it can ` +
|
|
126
|
+
`neither veto nor feed the model back (stderr goes only to the user). ` +
|
|
127
|
+
`Nothing is prevented. Move the gate to a blocking event (e.g. PreToolUse) ` +
|
|
128
|
+
`so the deny fires BEFORE the action.`;
|
|
129
|
+
}
|
|
130
|
+
else if (permissionDecisionEvents.has(entry.event) &&
|
|
131
|
+
hasDecisionBlock &&
|
|
132
|
+
!hasPermissionDeny) {
|
|
133
|
+
// wrong-field: on a permission-gated event, uses the legacy field.
|
|
134
|
+
kind = "wrong-field";
|
|
135
|
+
message =
|
|
136
|
+
`On "${entry.event}" a deny must use ` +
|
|
137
|
+
`\`hookSpecificOutput.permissionDecision:"deny"\`; this script uses the ` +
|
|
138
|
+
`legacy top-level \`"decision"\` field, which is ignored on this event, ` +
|
|
139
|
+
`so nothing is blocked. Update the JSON output to the structured form: ` +
|
|
140
|
+
`\`{"hookSpecificOutput":{"permissionDecision":"deny"}}\`.`;
|
|
141
|
+
}
|
|
142
|
+
if (kind === null)
|
|
143
|
+
continue;
|
|
144
|
+
// De-dupe identical (event, kind, scriptPath) triples.
|
|
145
|
+
const dedupeKey = `${entry.event}:${kind}:${scriptPath ?? ""}`;
|
|
146
|
+
if (seen.has(dedupeKey))
|
|
147
|
+
continue;
|
|
148
|
+
seen.add(dedupeKey);
|
|
149
|
+
findings.push({ event: entry.event, kind, scriptPath, message });
|
|
150
|
+
}
|
|
151
|
+
return findings;
|
|
152
|
+
}
|
|
153
|
+
//# sourceMappingURL=hook-block-ineffective.js.map
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Hook-matcher verification — the cross-referencing moat applied to the MATCHER
|
|
3
|
+
* string inside a hook registration. A PreToolUse hook fires only when its
|
|
4
|
+
* `matcher` equals the tool name the harness emits (or matches via glob/regex); a
|
|
5
|
+
* typo or wrong form silently prevents the hook from ever running — exactly the
|
|
6
|
+
* FALSE CONFIDENCE failure the compiled-hooks design exists to eliminate
|
|
7
|
+
* (research/hook-pain-points.md).
|
|
8
|
+
*
|
|
9
|
+
* THREE kinds of bad matcher, each verified here (one-detector-no-drift):
|
|
10
|
+
*
|
|
11
|
+
* 1. **tool-typo** — a bare token that is a CLOSE TYPO (edit distance ≤ 2) of a
|
|
12
|
+
* real built-in tool name but not an exact match (`bash` → `Bash`, `read` →
|
|
13
|
+
* `Read`). Suggests the correct casing. Reuses `closestTool` from
|
|
14
|
+
* `tool-contract.ts` — same edit-distance logic, same ≤ 2 confidence bound.
|
|
15
|
+
*
|
|
16
|
+
* 2. **mcp-form** — a token that looks MCP-ish (starts with `mcp`, case-
|
|
17
|
+
* insensitive) but is NOT the required `mcp__<server>__<tool>` double-underscore
|
|
18
|
+
* shape (single underscores, a hyphen, a trailing `*`…). Suggests the corrected
|
|
19
|
+
* form when the server/tool segments can be recovered.
|
|
20
|
+
*
|
|
21
|
+
* 3. **mcp-undeclared** — a correctly-formed `mcp__<server>__…` token whose server
|
|
22
|
+
* is NOT in the plugin's declared MCP servers. Gated EXACTLY like
|
|
23
|
+
* `mcp-tool-resolves`: (a) no declared set → skip (reaches global/project
|
|
24
|
+
* servers); (b) built-ins allowlisted via `dialect.knownMcpServers`; (c) the
|
|
25
|
+
* plugin-namespaced `mcp__plugin_…__…` form is skipped. Reuses `mcpToolServer`
|
|
26
|
+
* from `mcp-tool.ts` for the extraction — one parser, no drift.
|
|
27
|
+
*
|
|
28
|
+
* FP-SAFE: only a SINGLE bare token is inspected. A matcher that is empty, a pure
|
|
29
|
+
* wildcard (`*` / `.*`), or contains alternation (`|`) or other regex meta-
|
|
30
|
+
* characters is skipped — it is a pattern/glob with legitimate broad matching, not
|
|
31
|
+
* a tool name. Same don't-cry-wolf discipline as every other vigiles detector.
|
|
32
|
+
*
|
|
33
|
+
* Pure + ONE detector reused by `scan` + the `hook-matcher` lint rule
|
|
34
|
+
* (one-detector-no-drift). The dialect is injected (core ⊄ adapter).
|
|
35
|
+
*/
|
|
36
|
+
import type { HarnessDialect } from "./dialect.js";
|
|
37
|
+
/** Which matching failure was detected in the hook matcher string. */
|
|
38
|
+
export type HookMatcherKind = "tool-typo" | "mcp-form" | "mcp-undeclared";
|
|
39
|
+
/** One finding for a hook matcher that will silently never fire. */
|
|
40
|
+
export interface HookMatcherFinding {
|
|
41
|
+
/** The matcher string exactly as written. */
|
|
42
|
+
readonly matcher: string;
|
|
43
|
+
/** Which class of error was detected. */
|
|
44
|
+
readonly kind: HookMatcherKind;
|
|
45
|
+
/**
|
|
46
|
+
* The corrected matcher when the intent is recoverable (e.g. `Bash` for
|
|
47
|
+
* `bash`, `mcp__memory__.*` for `mcp_memory_*`). Absent when the server
|
|
48
|
+
* segment can't be recovered from a malformed MCP form.
|
|
49
|
+
*/
|
|
50
|
+
readonly suggestion?: string;
|
|
51
|
+
/** A ready-to-show, actionable message. */
|
|
52
|
+
readonly message: string;
|
|
53
|
+
}
|
|
54
|
+
/** A single hook registration entry — its event and matcher string. */
|
|
55
|
+
export interface HookMatcherEntry {
|
|
56
|
+
readonly event: string;
|
|
57
|
+
readonly matcher: string;
|
|
58
|
+
}
|
|
59
|
+
/**
|
|
60
|
+
* Verify hook-matcher strings for the three forms that silently never fire.
|
|
61
|
+
* Returns one {@link HookMatcherFinding} per offending entry. De-duplicates
|
|
62
|
+
* repeated matchers. Returns `[]` when all matchers are FP-safe to skip or
|
|
63
|
+
* are correct.
|
|
64
|
+
*/
|
|
65
|
+
export declare function hookMatcherIssues(entries: readonly HookMatcherEntry[], declaredServers: readonly string[], dialect: HarnessDialect): HookMatcherFinding[];
|
|
66
|
+
//# sourceMappingURL=hook-matcher.d.ts.map
|