agent-sanitizer 2.35.0 → 2.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -31
- package/THREAT-MODEL.md +16 -0
- package/claude-hooks/lib/hook-fault.mjs +1 -0
- package/claude-hooks/lib/hook-io.mjs +1 -0
- package/claude-hooks/lib/invisible-alert.mjs +146 -2
- package/claude-hooks/lib/invisible-report.mjs +51 -0
- package/claude-hooks/lib/trace.mjs +1 -0
- package/claude-hooks/plugin-hooks.mjs +15 -5
- package/claude-hooks/pretooluse-sanitize.mjs +9 -0
- package/claude-hooks/scan-invisible-chars.mjs +63 -71
- package/claude-hooks/scan-loaded-instructions.mjs +265 -0
- package/package.json +5 -1
- package/src/claude-context.mjs +140 -24
- package/src/instructions.mjs +8 -2
- package/types/claude-context.d.mts +74 -30
- package/types/claude-hooks/lib/hook-io.d.mts +1 -0
- package/types/claude-hooks/lib/invisible-alert.d.mts +78 -0
- package/types/claude-hooks/lib/invisible-report.d.mts +26 -0
- package/types/claude-hooks/lib/trace.d.mts +1 -0
- package/types/claude-hooks/scan-invisible-chars.d.mts +18 -28
- package/types/claude-hooks/scan-loaded-instructions.d.mts +66 -0
- package/types/instructions.d.mts +1 -1
- package/types/src/claude-context.d.mts +74 -30
package/src/claude-context.mjs
CHANGED
|
@@ -1,27 +1,44 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* WHICH files an agent loads as model context, as data: the glob
|
|
2
|
+
* WHICH files an agent loads as model context, as data: the glob sets and the
|
|
3
3
|
* walk-pruning predicate that together define "everything Claude Code reads as
|
|
4
4
|
* instructions, and nothing else".
|
|
5
5
|
*
|
|
6
|
-
*
|
|
6
|
+
* TWO scopes live here, because Claude Code has two load moments and scanning
|
|
7
|
+
* them at one moment is what made session start unusable:
|
|
8
|
+
*
|
|
9
|
+
* - {@link CLAUDE_LAUNCH_GLOBS} + {@link ancestorInstructionFiles} — what
|
|
10
|
+
* loads AT LAUNCH: the working directory's own instruction files, the same
|
|
11
|
+
* files in every directory above it, and the root `.claude` context tree.
|
|
12
|
+
* Rooted at the scan root, so it costs one shallow glob and a walk up the
|
|
13
|
+
* parent chain no matter how large the tree below is.
|
|
14
|
+
* - {@link CLAUDE_INSTRUCTION_GLOBS} — every instruction file ANYWHERE in a
|
|
15
|
+
* tree, `**`-rooted. A library caller asking "scan this project" wants
|
|
16
|
+
* this; a SessionStart hook must not, because Claude Code loads a
|
|
17
|
+
* subdirectory's `CLAUDE.md` only when it reads a file in that
|
|
18
|
+
* subdirectory, and a launch in `$HOME` charges the whole home tree —
|
|
19
|
+
* ~100 seconds of blocked startup — for files that mostly never load.
|
|
20
|
+
* Those lazily-loaded files are scanned by the InstructionsLoaded hook, at
|
|
21
|
+
* the moment they load.
|
|
22
|
+
*
|
|
23
|
+
* This is the SINGLE SOURCE for both. It used to live inside
|
|
7
24
|
* `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
|
|
8
25
|
* knew the answer and nobody else did: `src/instructions.mjs` takes
|
|
9
26
|
* caller-supplied globs by design (no agent's convention is baked into the
|
|
10
27
|
* engine), so the CLI, the Python port and every downstream fork spelled their
|
|
11
28
|
* own approximation of this list — and an approximation that drifts either
|
|
12
|
-
* scans bulk data that can never reach the model
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
* and a session that loads it.
|
|
29
|
+
* scans bulk data that can never reach the model or MISSES a context directory
|
|
30
|
+
* entirely, which is a silent hole in the one scan standing between a poisoned
|
|
31
|
+
* instruction file and a session that loads it.
|
|
16
32
|
*
|
|
17
|
-
* It is a standalone
|
|
18
|
-
* two reasons: `src/instructions.mjs` re-exports it as the
|
|
19
|
-
* door, and the hook imports it RELATIVELY — deliberately not
|
|
20
|
-
* `agent-sanitizer` specifier the plugin bundle pins to a published
|
|
21
|
-
* This scope is hook POLICY, not engine behavior: it must ship and move
|
|
22
|
-
* hook that walks it, or a plugin built against an older pin would
|
|
23
|
-
* wrong directories while believing it had scanned everything.
|
|
33
|
+
* It is a standalone DATA module carrying no package dependency (like
|
|
34
|
+
* ./cf-charset.mjs) for two reasons: `src/instructions.mjs` re-exports it as the
|
|
35
|
+
* library's public door, and the hook imports it RELATIVELY — deliberately not
|
|
36
|
+
* through the `agent-sanitizer` specifier the plugin bundle pins to a published
|
|
37
|
+
* engine. This scope is hook POLICY, not engine behavior: it must ship and move
|
|
38
|
+
* with the hook that walks it, or a plugin built against an older pin would
|
|
39
|
+
* prune the wrong directories while believing it had scanned everything.
|
|
24
40
|
*/
|
|
41
|
+
import { dirname, isAbsolute, join, relative, resolve } from "node:path";
|
|
25
42
|
|
|
26
43
|
/**
|
|
27
44
|
* The `.claude/` subdirectories whose markdown Claude Code loads as model
|
|
@@ -44,11 +61,33 @@ export const CLAUDE_CONTEXT_SUBDIRS = Object.freeze([
|
|
|
44
61
|
"agents",
|
|
45
62
|
"commands",
|
|
46
63
|
"output-styles",
|
|
64
|
+
"rules",
|
|
47
65
|
"skills",
|
|
48
66
|
]);
|
|
49
67
|
|
|
50
|
-
|
|
51
|
-
|
|
68
|
+
/**
|
|
69
|
+
* Claude Code's own per-directory memory files. Their own list because the
|
|
70
|
+
* parent-chain load ({@link ancestorInstructionFiles}) is Claude Code's rule and
|
|
71
|
+
* covers exactly these two.
|
|
72
|
+
*/
|
|
73
|
+
export const CLAUDE_MEMORY_FILES = Object.freeze([
|
|
74
|
+
"CLAUDE.md",
|
|
75
|
+
"CLAUDE.local.md",
|
|
76
|
+
]);
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* Every per-directory instruction file: Claude Code's memory files plus
|
|
80
|
+
* `AGENTS.md`, the cross-agent convention Claude Code does not read itself, kept
|
|
81
|
+
* because this package guards agents generally and the file is loaded as
|
|
82
|
+
* instructions by the ones that do.
|
|
83
|
+
*/
|
|
84
|
+
export const CLAUDE_DIR_INSTRUCTION_FILES = Object.freeze([
|
|
85
|
+
...CLAUDE_MEMORY_FILES,
|
|
86
|
+
"AGENTS.md",
|
|
87
|
+
]);
|
|
88
|
+
|
|
89
|
+
// The glob patterns for one `.claude` tree at `prefix` (empty for the scan root,
|
|
90
|
+
// a doubled-star segment for nested ones): its top-level markdown, plus the
|
|
52
91
|
// whitelisted context subdirectories. Built once, from the one list above.
|
|
53
92
|
/** @param {string} prefix @returns {string[]} */
|
|
54
93
|
function claudeDirPatterns(prefix) {
|
|
@@ -59,12 +98,16 @@ function claudeDirPatterns(prefix) {
|
|
|
59
98
|
}
|
|
60
99
|
|
|
61
100
|
/**
|
|
62
|
-
* Every glob whose matches Claude Code loads as model context
|
|
63
|
-
* per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
|
|
64
|
-
* the whitelisted `.claude/` markdown. Claude Code loads these on
|
|
65
|
-
* containing directory — a load path that bypasses the PostToolUse
|
|
66
|
-
* so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
|
|
67
|
-
* uncleaned unless something scans it
|
|
101
|
+
* Every glob whose matches Claude Code loads as model context ANYWHERE in a
|
|
102
|
+
* tree: the per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
|
|
103
|
+
* AGENTS.md) and the whitelisted `.claude/` markdown. Claude Code loads these on
|
|
104
|
+
* entry to their containing directory — a load path that bypasses the PostToolUse
|
|
105
|
+
* sanitizer — so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
|
|
106
|
+
* model uncleaned unless something scans it.
|
|
107
|
+
*
|
|
108
|
+
* This is the WHOLE-TREE scope, for a caller scanning a project on demand (the
|
|
109
|
+
* CLI, the Python port). It is not what a SessionStart hook walks — see
|
|
110
|
+
* {@link CLAUDE_LAUNCH_GLOBS} for why, and for what does.
|
|
68
111
|
*
|
|
69
112
|
* `**` does not descend into dot directories, so NESTED `.claude/` trees need
|
|
70
113
|
* their own doubled-star-prefixed patterns: without them a directory-scoped
|
|
@@ -77,12 +120,85 @@ function claudeDirPatterns(prefix) {
|
|
|
77
120
|
* MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
|
|
78
121
|
*/
|
|
79
122
|
export const CLAUDE_INSTRUCTION_GLOBS = Object.freeze([
|
|
80
|
-
|
|
81
|
-
"**/CLAUDE.local.md",
|
|
82
|
-
"**/AGENTS.md",
|
|
123
|
+
...CLAUDE_DIR_INSTRUCTION_FILES.map((name) => `**/${name}`),
|
|
83
124
|
...claudeDirPatterns("**/"),
|
|
84
125
|
]);
|
|
85
126
|
|
|
127
|
+
/**
|
|
128
|
+
* Every glob whose matches Claude Code loads AT LAUNCH from the scan root
|
|
129
|
+
* itself: the root's own instruction files and its `.claude` context tree. Same
|
|
130
|
+
* patterns as {@link CLAUDE_INSTRUCTION_GLOBS} without the doubled-star root, built
|
|
131
|
+
* from the same two lists so the pair cannot drift.
|
|
132
|
+
*
|
|
133
|
+
* Deliberately NOT recursive. A subdirectory's `CLAUDE.md` is loaded when Claude
|
|
134
|
+
* Code reads a file in that subdirectory, not at launch, so globbing for it at
|
|
135
|
+
* session start pays a whole-tree walk (the entire home directory, when the
|
|
136
|
+
* session is launched there) to pre-scan files that mostly never load. The
|
|
137
|
+
* InstructionsLoaded hook scans each of those at the moment it loads instead —
|
|
138
|
+
* which is also the only moment that catches one created mid-session.
|
|
139
|
+
*
|
|
140
|
+
* Pair with {@link ancestorInstructionFiles} for the other half of the launch
|
|
141
|
+
* set, and with {@link excludeFromContextScan} to prune the `.claude` walk.
|
|
142
|
+
*/
|
|
143
|
+
export const CLAUDE_LAUNCH_GLOBS = Object.freeze([
|
|
144
|
+
...CLAUDE_DIR_INSTRUCTION_FILES,
|
|
145
|
+
...claudeDirPatterns(""),
|
|
146
|
+
]);
|
|
147
|
+
|
|
148
|
+
/**
|
|
149
|
+
* The instruction files Claude Code loads from the directories ABOVE `dir`:
|
|
150
|
+
* walking up to the filesystem root, `CLAUDE.md` and `CLAUDE.local.md` in each
|
|
151
|
+
* parent are loaded IN FULL at launch, so a payload planted in a parent
|
|
152
|
+
* directory reaches the model exactly like one in the project's own file.
|
|
153
|
+
*
|
|
154
|
+
* `AGENTS.md` is absent by design: the parent-chain load is Claude Code's rule,
|
|
155
|
+
* and Claude Code does not read `AGENTS.md`.
|
|
156
|
+
*
|
|
157
|
+
* Returns CANDIDATES — absolute paths, existing or not, because this module
|
|
158
|
+
* touches no filesystem. Most parents of any directory hold neither file, so a
|
|
159
|
+
* caller that buckets its misses as "absent" should filter first: ~10 phantom
|
|
160
|
+
* entries per session would drown the one signal that bucket carries, a target
|
|
161
|
+
* that existed when the scan listed it and vanished before the read.
|
|
162
|
+
* @param {string} dir the scan root; its own files are NOT included
|
|
163
|
+
* @returns {string[]}
|
|
164
|
+
*/
|
|
165
|
+
export function ancestorInstructionFiles(dir) {
|
|
166
|
+
/** @type {string[]} */
|
|
167
|
+
const files = [];
|
|
168
|
+
let current = resolve(dir);
|
|
169
|
+
// `dirname` is its own fixed point at the filesystem root, which is what ends
|
|
170
|
+
// the walk on every platform without spelling a root path here.
|
|
171
|
+
for (
|
|
172
|
+
let parent = dirname(current);
|
|
173
|
+
parent !== current;
|
|
174
|
+
parent = dirname(current)
|
|
175
|
+
) {
|
|
176
|
+
current = parent;
|
|
177
|
+
for (const name of CLAUDE_MEMORY_FILES) files.push(join(current, name));
|
|
178
|
+
}
|
|
179
|
+
return files;
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* Whether `file` lives inside `dir`. Lexical, on already-absolute paths, and the
|
|
184
|
+
* bound on where an instruction-file scanner may REWRITE: a file above the scan
|
|
185
|
+
* root is shared with every other project beneath that root, so it is reported
|
|
186
|
+
* rather than silently edited. Both scanners ask this — one for a target it
|
|
187
|
+
* globbed, one for a path an event handed it — and a copy each is a copy that
|
|
188
|
+
* can drift into rewriting a file the other would not.
|
|
189
|
+
*
|
|
190
|
+
* Symlinks are deliberately not resolved: the guard that stops a link from
|
|
191
|
+
* redirecting the write has to live at the write itself (cleanFile opens
|
|
192
|
+
* O_NOFOLLOW), and resolving here would only duplicate it a check too early.
|
|
193
|
+
* @param {string} dir
|
|
194
|
+
* @param {string} file
|
|
195
|
+
* @returns {boolean}
|
|
196
|
+
*/
|
|
197
|
+
export function isInsideDir(dir, file) {
|
|
198
|
+
const rel = relative(dir, file);
|
|
199
|
+
return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
|
|
200
|
+
}
|
|
201
|
+
|
|
86
202
|
/**
|
|
87
203
|
* The one directory no instruction-file walk ever descends into. Its own
|
|
88
204
|
* function so the name is spelled once, and so the two predicates that need it
|
package/src/instructions.mjs
CHANGED
|
@@ -13,8 +13,10 @@
|
|
|
13
13
|
* instruction files live under (e.g. `["CLAUDE.md", "AGENTS.md",
|
|
14
14
|
* ".claude/**\/*.md", "**\/SKILL.md"]`), so no agent's convention is baked in.
|
|
15
15
|
* Claude Code's own convention is re-exported below
|
|
16
|
-
* ({@link CLAUDE_INSTRUCTION_GLOBS}
|
|
17
|
-
*
|
|
16
|
+
* ({@link CLAUDE_INSTRUCTION_GLOBS} for a whole tree,
|
|
17
|
+
* {@link CLAUDE_LAUNCH_GLOBS} + {@link ancestorInstructionFiles} for just what a
|
|
18
|
+
* session loads at launch, {@link excludeFromContextScan} to prune either walk)
|
|
19
|
+
* so a caller that wants it takes the hooks' exact scope rather than
|
|
18
20
|
* approximating it — see ./claude-context.mjs.
|
|
19
21
|
*/
|
|
20
22
|
import {
|
|
@@ -48,8 +50,12 @@ import { excludeNodeModules } from "./claude-context.mjs";
|
|
|
48
50
|
// makes `agent-sanitizer/instructions` the single place a CLI, a port or a fork
|
|
49
51
|
// reads that scope from instead of re-spelling it.
|
|
50
52
|
export {
|
|
53
|
+
ancestorInstructionFiles,
|
|
51
54
|
CLAUDE_CONTEXT_SUBDIRS,
|
|
55
|
+
CLAUDE_DIR_INSTRUCTION_FILES,
|
|
52
56
|
CLAUDE_INSTRUCTION_GLOBS,
|
|
57
|
+
CLAUDE_LAUNCH_GLOBS,
|
|
58
|
+
CLAUDE_MEMORY_FILES,
|
|
53
59
|
excludeFromContextScan,
|
|
54
60
|
} from "./claude-context.mjs";
|
|
55
61
|
|
|
@@ -1,3 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The instruction files Claude Code loads from the directories ABOVE `dir`:
|
|
3
|
+
* walking up to the filesystem root, `CLAUDE.md` and `CLAUDE.local.md` in each
|
|
4
|
+
* parent are loaded IN FULL at launch, so a payload planted in a parent
|
|
5
|
+
* directory reaches the model exactly like one in the project's own file.
|
|
6
|
+
*
|
|
7
|
+
* `AGENTS.md` is absent by design: the parent-chain load is Claude Code's rule,
|
|
8
|
+
* and Claude Code does not read `AGENTS.md`.
|
|
9
|
+
*
|
|
10
|
+
* Returns CANDIDATES — absolute paths, existing or not, because this module
|
|
11
|
+
* touches no filesystem. Most parents of any directory hold neither file, so a
|
|
12
|
+
* caller that buckets its misses as "absent" should filter first: ~10 phantom
|
|
13
|
+
* entries per session would drown the one signal that bucket carries, a target
|
|
14
|
+
* that existed when the scan listed it and vanished before the read.
|
|
15
|
+
* @param {string} dir the scan root; its own files are NOT included
|
|
16
|
+
* @returns {string[]}
|
|
17
|
+
*/
|
|
18
|
+
export function ancestorInstructionFiles(dir: string): string[];
|
|
19
|
+
/**
|
|
20
|
+
* Whether `file` lives inside `dir`. Lexical, on already-absolute paths, and the
|
|
21
|
+
* bound on where an instruction-file scanner may REWRITE: a file above the scan
|
|
22
|
+
* root is shared with every other project beneath that root, so it is reported
|
|
23
|
+
* rather than silently edited. Both scanners ask this — one for a target it
|
|
24
|
+
* globbed, one for a path an event handed it — and a copy each is a copy that
|
|
25
|
+
* can drift into rewriting a file the other would not.
|
|
26
|
+
*
|
|
27
|
+
* Symlinks are deliberately not resolved: the guard that stops a link from
|
|
28
|
+
* redirecting the write has to live at the write itself (cleanFile opens
|
|
29
|
+
* O_NOFOLLOW), and resolving here would only duplicate it a check too early.
|
|
30
|
+
* @param {string} dir
|
|
31
|
+
* @param {string} file
|
|
32
|
+
* @returns {boolean}
|
|
33
|
+
*/
|
|
34
|
+
export function isInsideDir(dir: string, file: string): boolean;
|
|
1
35
|
/**
|
|
2
36
|
* The one directory no instruction-file walk ever descends into. Its own
|
|
3
37
|
* function so the name is spelled once, and so the two predicates that need it
|
|
@@ -25,30 +59,6 @@ export function excludeNodeModules(entry: string): boolean;
|
|
|
25
59
|
* @returns {boolean}
|
|
26
60
|
*/
|
|
27
61
|
export function excludeFromContextScan(entry: string): boolean;
|
|
28
|
-
/**
|
|
29
|
-
* WHICH files an agent loads as model context, as data: the glob set and the
|
|
30
|
-
* walk-pruning predicate that together define "everything Claude Code reads as
|
|
31
|
-
* instructions, and nothing else".
|
|
32
|
-
*
|
|
33
|
-
* This is the SINGLE SOURCE for that scope. It used to live inside
|
|
34
|
-
* `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
|
|
35
|
-
* knew the answer and nobody else did: `src/instructions.mjs` takes
|
|
36
|
-
* caller-supplied globs by design (no agent's convention is baked into the
|
|
37
|
-
* engine), so the CLI, the Python port and every downstream fork spelled their
|
|
38
|
-
* own approximation of this list — and an approximation that drifts either
|
|
39
|
-
* scans bulk data that can never reach the model (the 30-second session start
|
|
40
|
-
* this whitelist exists to fix) or MISSES a context directory entirely, which
|
|
41
|
-
* is a silent hole in the one scan standing between a poisoned instruction file
|
|
42
|
-
* and a session that loads it.
|
|
43
|
-
*
|
|
44
|
-
* It is a standalone, dependency-free DATA module (like ./cf-charset.mjs) for
|
|
45
|
-
* two reasons: `src/instructions.mjs` re-exports it as the library's public
|
|
46
|
-
* door, and the hook imports it RELATIVELY — deliberately not through the
|
|
47
|
-
* `agent-sanitizer` specifier the plugin bundle pins to a published engine.
|
|
48
|
-
* This scope is hook POLICY, not engine behavior: it must ship and move with the
|
|
49
|
-
* hook that walks it, or a plugin built against an older pin would prune the
|
|
50
|
-
* wrong directories while believing it had scanned everything.
|
|
51
|
-
*/
|
|
52
62
|
/**
|
|
53
63
|
* The `.claude/` subdirectories whose markdown Claude Code loads as model
|
|
54
64
|
* context. This is a WHITELIST, and that is the point: `.claude/` is also where
|
|
@@ -68,12 +78,29 @@ export function excludeFromContextScan(entry: string): boolean;
|
|
|
68
78
|
*/
|
|
69
79
|
export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
70
80
|
/**
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
81
|
+
* Claude Code's own per-directory memory files. Their own list because the
|
|
82
|
+
* parent-chain load ({@link ancestorInstructionFiles}) is Claude Code's rule and
|
|
83
|
+
* covers exactly these two.
|
|
84
|
+
*/
|
|
85
|
+
export const CLAUDE_MEMORY_FILES: readonly string[];
|
|
86
|
+
/**
|
|
87
|
+
* Every per-directory instruction file: Claude Code's memory files plus
|
|
88
|
+
* `AGENTS.md`, the cross-agent convention Claude Code does not read itself, kept
|
|
89
|
+
* because this package guards agents generally and the file is loaded as
|
|
90
|
+
* instructions by the ones that do.
|
|
91
|
+
*/
|
|
92
|
+
export const CLAUDE_DIR_INSTRUCTION_FILES: readonly string[];
|
|
93
|
+
/**
|
|
94
|
+
* Every glob whose matches Claude Code loads as model context ANYWHERE in a
|
|
95
|
+
* tree: the per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
|
|
96
|
+
* AGENTS.md) and the whitelisted `.claude/` markdown. Claude Code loads these on
|
|
97
|
+
* entry to their containing directory — a load path that bypasses the PostToolUse
|
|
98
|
+
* sanitizer — so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
|
|
99
|
+
* model uncleaned unless something scans it.
|
|
100
|
+
*
|
|
101
|
+
* This is the WHOLE-TREE scope, for a caller scanning a project on demand (the
|
|
102
|
+
* CLI, the Python port). It is not what a SessionStart hook walks — see
|
|
103
|
+
* {@link CLAUDE_LAUNCH_GLOBS} for why, and for what does.
|
|
77
104
|
*
|
|
78
105
|
* `**` does not descend into dot directories, so NESTED `.claude/` trees need
|
|
79
106
|
* their own doubled-star-prefixed patterns: without them a directory-scoped
|
|
@@ -86,3 +113,20 @@ export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
|
86
113
|
* MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
|
|
87
114
|
*/
|
|
88
115
|
export const CLAUDE_INSTRUCTION_GLOBS: readonly string[];
|
|
116
|
+
/**
|
|
117
|
+
* Every glob whose matches Claude Code loads AT LAUNCH from the scan root
|
|
118
|
+
* itself: the root's own instruction files and its `.claude` context tree. Same
|
|
119
|
+
* patterns as {@link CLAUDE_INSTRUCTION_GLOBS} without the doubled-star root, built
|
|
120
|
+
* from the same two lists so the pair cannot drift.
|
|
121
|
+
*
|
|
122
|
+
* Deliberately NOT recursive. A subdirectory's `CLAUDE.md` is loaded when Claude
|
|
123
|
+
* Code reads a file in that subdirectory, not at launch, so globbing for it at
|
|
124
|
+
* session start pays a whole-tree walk (the entire home directory, when the
|
|
125
|
+
* session is launched there) to pre-scan files that mostly never load. The
|
|
126
|
+
* InstructionsLoaded hook scans each of those at the moment it loads instead —
|
|
127
|
+
* which is also the only moment that catches one created mid-session.
|
|
128
|
+
*
|
|
129
|
+
* Pair with {@link ancestorInstructionFiles} for the other half of the launch
|
|
130
|
+
* set, and with {@link excludeFromContextScan} to prune the `.claude` walk.
|
|
131
|
+
*/
|
|
132
|
+
export const CLAUDE_LAUNCH_GLOBS: readonly string[];
|
|
@@ -425,6 +425,7 @@ export const HookEvent: Readonly<{
|
|
|
425
425
|
POST_TOOL_USE: "PostToolUse";
|
|
426
426
|
USER_PROMPT_SUBMIT: "UserPromptSubmit";
|
|
427
427
|
SESSION_START: "SessionStart";
|
|
428
|
+
INSTRUCTIONS_LOADED: "InstructionsLoaded";
|
|
428
429
|
}>;
|
|
429
430
|
/** Claude Code permissionDecision verdicts. */
|
|
430
431
|
export const PermissionDecision: Readonly<{
|
|
@@ -1,3 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Marker the InstructionsLoaded scanner writes on every fire, so another hook
|
|
3
|
+
* can tell whether that event is being scanned at all this session.
|
|
4
|
+
*
|
|
5
|
+
* Keyed by the SESSION, and never cleared at SessionStart like the alert pair
|
|
6
|
+
* above (a later session sweeps it once it is older than the TTL):
|
|
7
|
+
* nothing pins the order of SessionStart against the InstructionsLoaded events
|
|
8
|
+
* Claude Code fires for the files it loads at launch, so a clear could erase a
|
|
9
|
+
* marker written moments earlier and produce the notice on a session that IS
|
|
10
|
+
* covered. Session-keyed, the question each session asks is answered by that
|
|
11
|
+
* session's own file and no ordering matters. A host that exports no session id
|
|
12
|
+
* falls back to one shared name — where the marker can outlive its session, and
|
|
13
|
+
* a later session on a host that stopped emitting the event stays quiet.
|
|
14
|
+
* @param {string} [sessionId]
|
|
15
|
+
* @returns {string}
|
|
16
|
+
*/
|
|
17
|
+
export function instructionsLoadedFile(sessionId?: string): string;
|
|
18
|
+
/**
|
|
19
|
+
* Companion marker: the notice below has been surfaced this session.
|
|
20
|
+
* @param {string} [sessionId]
|
|
21
|
+
* @returns {string}
|
|
22
|
+
*/
|
|
23
|
+
export function instructionsLoadedNoticeFile(sessionId?: string): string;
|
|
24
|
+
/**
|
|
25
|
+
* Whether the InstructionsLoaded scanner has run this session — i.e. whether the
|
|
26
|
+
* lazily-loaded instruction files are being scanned at all. Ownership-validated
|
|
27
|
+
* like every other marker here: a co-tenant could otherwise plant the
|
|
28
|
+
* predictable path and suppress the notice below, which is the whole signal that
|
|
29
|
+
* nested files are going unscanned.
|
|
30
|
+
* @param {string} [sessionId]
|
|
31
|
+
* @returns {boolean}
|
|
32
|
+
*/
|
|
33
|
+
export function instructionsLoadedSeen(sessionId?: string): boolean;
|
|
34
|
+
/**
|
|
35
|
+
* Record that the InstructionsLoaded scanner engaged. Symlink-safe presence
|
|
36
|
+
* write (see writeSentinelFile) at a predictable $TMPDIR path.
|
|
37
|
+
*
|
|
38
|
+
* The event fires once per instruction file loaded, so the already-recorded case
|
|
39
|
+
* returns without a write — and the stale-marker sweep rides the FIRST fire of a
|
|
40
|
+
* session, where one readdir is paid once rather than per loaded file.
|
|
41
|
+
* @param {string} [sessionId]
|
|
42
|
+
* @returns {void}
|
|
43
|
+
*/
|
|
44
|
+
export function recordInstructionsLoaded(sessionId?: string): void;
|
|
45
|
+
/**
|
|
46
|
+
* The one-time context line for a session where no InstructionsLoaded scan ran,
|
|
47
|
+
* or null when the scan has been seen or the notice was already surfaced this
|
|
48
|
+
* session. Records the notice as it hands it out, so it rides on ONE tool call
|
|
49
|
+
* rather than every one — the per-call repeat is what trains a reader to skip it.
|
|
50
|
+
*
|
|
51
|
+
* The loss it names is real and otherwise invisible: SessionStart scans the
|
|
52
|
+
* instruction files that load at launch, and everything a subdirectory loads
|
|
53
|
+
* later is scanned by the event. No scan, and nothing says so.
|
|
54
|
+
*
|
|
55
|
+
* The notice names the OBSERVABLE — no scan ran — and both of its causes, because
|
|
56
|
+
* the marker cannot tell a host that never emits the event from an operator who
|
|
57
|
+
* switched the hook off in AGENT_SANITIZER_DISABLED_HOOKS, and asserting the
|
|
58
|
+
* first would send an operator who chose the second to the wrong fix.
|
|
59
|
+
* @param {string} [sessionId] the harness's session identity, so the answer
|
|
60
|
+
* belongs to THIS session (see instructionsLoadedFile)
|
|
61
|
+
* @returns {string | null}
|
|
62
|
+
*/
|
|
63
|
+
export function instructionsLoadedGapNotice(sessionId?: string): string | null;
|
|
1
64
|
/**
|
|
2
65
|
* The alert findings if invisible-char injection was detected in instruction
|
|
3
66
|
* files and couldn't be auto-cleaned, else null. ALERT_FILE lives at a predictable,
|
|
@@ -9,6 +72,21 @@
|
|
|
9
72
|
* @returns {string | null}
|
|
10
73
|
*/
|
|
11
74
|
export function invisibleCharAlert(): string | null;
|
|
75
|
+
/**
|
|
76
|
+
* Add `text` to the alert the PreToolUse gate surfaces, keeping whatever is
|
|
77
|
+
* already there.
|
|
78
|
+
*
|
|
79
|
+
* Appending, where the SessionStart scanner TRUNCATES: that scan runs once and
|
|
80
|
+
* owns the session's reset, while an instruction file loaded mid-session is one
|
|
81
|
+
* more finding on top of whatever the launch scan left — a truncating write here
|
|
82
|
+
* would silently drop the earlier report. Symlink-refusing (writeFileNoFollow)
|
|
83
|
+
* and ownership-checked on read, because ALERT_FILE sits at a predictable,
|
|
84
|
+
* world-visible $TMPDIR path; a foreign or squatted file reads as empty and is
|
|
85
|
+
* replaced rather than appended to.
|
|
86
|
+
* @param {string} text
|
|
87
|
+
* @returns {void}
|
|
88
|
+
*/
|
|
89
|
+
export function appendAlert(text: string): void;
|
|
12
90
|
/**
|
|
13
91
|
* True once the gate has surfaced its blocking ask this session. Validates
|
|
14
92
|
* ownership (not mere existence): a co-tenant could pre-create ALERT_ACK_FILE at its
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The operator-facing report for hidden-Unicode findings in instruction files.
|
|
3
|
+
*
|
|
4
|
+
* Its own module because two hooks render it — the SessionStart scan of the
|
|
5
|
+
* files that load at launch, and the InstructionsLoaded scan of every file
|
|
6
|
+
* loaded after that — and a second copy would drift in exactly the way that
|
|
7
|
+
* matters: the framing that keeps a decoded payload from reading as an
|
|
8
|
+
* instruction (see decodeRun's `untrusted data` prefix) is part of the report,
|
|
9
|
+
* not decoration on it.
|
|
10
|
+
*/
|
|
11
|
+
/**
|
|
12
|
+
* @param {Array<{
|
|
13
|
+
* file: string,
|
|
14
|
+
* findings: Array<{ line: number | null, charCount: number, method: string, decoded: string }>,
|
|
15
|
+
* }>} allFindings
|
|
16
|
+
* @returns {string}
|
|
17
|
+
*/
|
|
18
|
+
export function formatReport(allFindings: Array<{
|
|
19
|
+
file: string;
|
|
20
|
+
findings: Array<{
|
|
21
|
+
line: number | null;
|
|
22
|
+
charCount: number;
|
|
23
|
+
method: string;
|
|
24
|
+
decoded: string;
|
|
25
|
+
}>;
|
|
26
|
+
}>): string;
|
|
@@ -47,6 +47,7 @@ export function bestEffortTrace(sink: TraceFn): TraceFn;
|
|
|
47
47
|
export const TraceEvent: Readonly<{
|
|
48
48
|
HOOK_RAN: "hook_ran";
|
|
49
49
|
SCAN_INVISIBLE_CHARS_RAN: "scan_invisible_chars_ran";
|
|
50
|
+
SCAN_LOADED_INSTRUCTIONS_RAN: "scan_loaded_instructions_ran";
|
|
50
51
|
}>;
|
|
51
52
|
/**
|
|
52
53
|
* The sink shape a hook emits through: the event name, its metadata fields, and
|
|
@@ -71,6 +71,7 @@ export function cliMain(opts?: {
|
|
|
71
71
|
export function scanFile(filePath: string): ReturnType<typeof import("agent-sanitizer/instructions").scanText>;
|
|
72
72
|
import { CLAUDE_CONTEXT_SUBDIRS } from "../src/claude-context.mjs";
|
|
73
73
|
import { CLAUDE_INSTRUCTION_GLOBS } from "../src/claude-context.mjs";
|
|
74
|
+
import { CLAUDE_LAUNCH_GLOBS } from "../src/claude-context.mjs";
|
|
74
75
|
/**
|
|
75
76
|
* The SSOT decoder, re-exported through a lazy-bound wrapper (the binding is
|
|
76
77
|
* `let` and may be re-bound by the cold-start reload, so the export must read
|
|
@@ -86,18 +87,23 @@ export function decodeRun(run: string): {
|
|
|
86
87
|
decoded: string;
|
|
87
88
|
};
|
|
88
89
|
/**
|
|
89
|
-
* Every file
|
|
90
|
-
*
|
|
91
|
-
*
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
* uncleaned unless it is scanned here.
|
|
90
|
+
* Every file Claude Code loads as model context AT LAUNCH: `dir`'s own
|
|
91
|
+
* instruction files and its `.claude/` context tree, plus the CLAUDE.md /
|
|
92
|
+
* CLAUDE.local.md of every directory above it (loaded in full at launch, and
|
|
93
|
+
* until now never scanned by anything). These load before the session's first
|
|
94
|
+
* tool call — a path that bypasses the PostToolUse sanitizer — so a payload in
|
|
95
|
+
* one of them reaches the model uncleaned unless it is scanned here.
|
|
95
96
|
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
*
|
|
99
|
-
*
|
|
100
|
-
*
|
|
97
|
+
* Bounded on purpose: one shallow glob plus a walk up the parent chain. The
|
|
98
|
+
* `**`-rooted scope ({@link CLAUDE_INSTRUCTION_GLOBS}) walks the entire tree
|
|
99
|
+
* below `dir`, which for a session launched in a home directory is ~100 seconds
|
|
100
|
+
* of blocked startup spent on files Claude Code does not load at launch. Those
|
|
101
|
+
* files load when a tool reads their directory, and scan-loaded-instructions
|
|
102
|
+
* scans each one at that moment.
|
|
103
|
+
*
|
|
104
|
+
* The scope itself — which globs, and which directories the walk must prune — is
|
|
105
|
+
* the library's (see src/claude-context.mjs for why it is imported relatively
|
|
106
|
+
* rather than through the `agent-sanitizer` specifier the plugin bundle pins).
|
|
101
107
|
* @param {string} dir
|
|
102
108
|
* @returns {string[]}
|
|
103
109
|
*/
|
|
@@ -107,20 +113,4 @@ import { ALERT_ACK_FILE } from "./lib/invisible-alert.mjs";
|
|
|
107
113
|
export let LONG_RUN_RE: RegExp;
|
|
108
114
|
export let LONG_RUN_THRESHOLD: 10;
|
|
109
115
|
export let TOTAL_INVISIBLE_THRESHOLD: 30;
|
|
110
|
-
|
|
111
|
-
* @param {Array<{
|
|
112
|
-
* file: string,
|
|
113
|
-
* findings: Array<{ line: number | null, charCount: number, method: string, decoded: string }>,
|
|
114
|
-
* }>} allFindings
|
|
115
|
-
* @returns {string}
|
|
116
|
-
*/
|
|
117
|
-
export function formatReport(allFindings: Array<{
|
|
118
|
-
file: string;
|
|
119
|
-
findings: Array<{
|
|
120
|
-
line: number | null;
|
|
121
|
-
charCount: number;
|
|
122
|
-
method: string;
|
|
123
|
-
decoded: string;
|
|
124
|
-
}>;
|
|
125
|
-
}>): string;
|
|
126
|
-
export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, ALERT_FILE, ALERT_ACK_FILE };
|
|
116
|
+
export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, CLAUDE_LAUNCH_GLOBS, ALERT_FILE, ALERT_ACK_FILE };
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The payload fields this hook reads, validated. A payload missing either is
|
|
3
|
+
* harness-contract drift, not a clean file: reporting "no findings" for bytes we
|
|
4
|
+
* never saw is the one answer that must never be reachable, so this throws into
|
|
5
|
+
* the declared fault posture instead.
|
|
6
|
+
* @param {unknown} payload
|
|
7
|
+
* @returns {{ filePath: string, content: string, loadReason: string }}
|
|
8
|
+
*/
|
|
9
|
+
export function readLoadedFile(payload: unknown): {
|
|
10
|
+
filePath: string;
|
|
11
|
+
content: string;
|
|
12
|
+
loadReason: string;
|
|
13
|
+
};
|
|
14
|
+
/**
|
|
15
|
+
* Scan one loaded instruction file. Returns the report and what to do with it:
|
|
16
|
+
* `cleaned` says the payload is gone from disk, `alert` carries the text that
|
|
17
|
+
* must arm the PreToolUse gate (empty when the clean succeeded).
|
|
18
|
+
*
|
|
19
|
+
* The scan runs on the payload's bytes, never a re-read of the path: those are
|
|
20
|
+
* the bytes that reached the model, and a file rewritten between the load and
|
|
21
|
+
* this hook would otherwise be scanned in a state the model never saw.
|
|
22
|
+
* @param {{ filePath: string, content: string }} loaded
|
|
23
|
+
* @param {{ projectDir?: string, clean?: typeof cleanFile }} [opts] injectable
|
|
24
|
+
* for tests; the default cleans through the SSOT's guarded rewrite
|
|
25
|
+
* @returns {{ report: string, cleaned: boolean, reason: string | null } | null}
|
|
26
|
+
* null when the file is clean
|
|
27
|
+
*/
|
|
28
|
+
export function scanLoadedFile({ filePath, content }: {
|
|
29
|
+
filePath: string;
|
|
30
|
+
content: string;
|
|
31
|
+
}, { projectDir, clean }?: {
|
|
32
|
+
projectDir?: string;
|
|
33
|
+
clean?: typeof cleanFile;
|
|
34
|
+
}): {
|
|
35
|
+
report: string;
|
|
36
|
+
cleaned: boolean;
|
|
37
|
+
reason: string | null;
|
|
38
|
+
} | null;
|
|
39
|
+
/**
|
|
40
|
+
* The operator- and model-facing text for a scanned file. Both channels carry
|
|
41
|
+
* it: the bytes are already in context, so the model is told to distrust what it
|
|
42
|
+
* just read, and the user is told what changed on disk.
|
|
43
|
+
* @param {{ report: string, cleaned: boolean, reason: string | null }} result
|
|
44
|
+
* @param {string} filePath
|
|
45
|
+
* @returns {string}
|
|
46
|
+
*/
|
|
47
|
+
export function loadedFileMessage({ report, cleaned, reason }: {
|
|
48
|
+
report: string;
|
|
49
|
+
cleaned: boolean;
|
|
50
|
+
reason: string | null;
|
|
51
|
+
}, filePath: string): string;
|
|
52
|
+
/**
|
|
53
|
+
* The hook's CLI: read the event, scan the loaded bytes, clean and report.
|
|
54
|
+
* Exported so a bundle entry (which must claim the CLI slot before this module
|
|
55
|
+
* loads) can run the exact same wiring instead of duplicating it.
|
|
56
|
+
* @param {{ trace?: import("./lib/trace.mjs").TraceFn }} [opts] `trace` is
|
|
57
|
+
* where this scan announces engagement; a host with its own trace channel
|
|
58
|
+
* passes its sink (see lib/trace.mjs)
|
|
59
|
+
* @returns {Promise<void>}
|
|
60
|
+
*/
|
|
61
|
+
export function cliMain({ trace: sink }?: {
|
|
62
|
+
trace?: import("./lib/trace.mjs").TraceFn;
|
|
63
|
+
}): Promise<void>;
|
|
64
|
+
declare const cleanFile: typeof import("agent-sanitizer/instructions").cleanFile;
|
|
65
|
+
export const HOOK_NAME: "scan-loaded-instructions";
|
|
66
|
+
export {};
|
package/types/instructions.d.mts
CHANGED
|
@@ -148,4 +148,4 @@ export function atomicReplaceFile(absPath: string, data: string, mode: number, t
|
|
|
148
148
|
* @returns {boolean}
|
|
149
149
|
*/
|
|
150
150
|
export function cleanFile(absPath: string, lstat?: (path: string) => import("node:fs").Stats): boolean;
|
|
151
|
-
export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, excludeFromContextScan } from "./claude-context.mjs";
|
|
151
|
+
export { ancestorInstructionFiles, CLAUDE_CONTEXT_SUBDIRS, CLAUDE_DIR_INSTRUCTION_FILES, CLAUDE_INSTRUCTION_GLOBS, CLAUDE_LAUNCH_GLOBS, CLAUDE_MEMORY_FILES, excludeFromContextScan } from "./claude-context.mjs";
|