agent-sanitizer 2.35.0 → 2.36.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +41 -31
- package/THREAT-MODEL.md +16 -0
- package/claude-hooks/lib/hook-fault.mjs +1 -0
- package/claude-hooks/lib/hook-io.mjs +1 -0
- package/claude-hooks/lib/invisible-alert.mjs +146 -2
- package/claude-hooks/lib/invisible-report.mjs +51 -0
- package/claude-hooks/lib/trace.mjs +1 -0
- package/claude-hooks/plugin-hooks.mjs +15 -5
- package/claude-hooks/pretooluse-sanitize.mjs +9 -0
- package/claude-hooks/scan-invisible-chars.mjs +63 -71
- package/claude-hooks/scan-loaded-instructions.mjs +265 -0
- package/package.json +5 -1
- package/src/claude-context.mjs +140 -24
- package/src/instructions.mjs +8 -2
- package/types/claude-context.d.mts +74 -30
- package/types/claude-hooks/lib/hook-io.d.mts +1 -0
- package/types/claude-hooks/lib/invisible-alert.d.mts +78 -0
- package/types/claude-hooks/lib/invisible-report.d.mts +26 -0
- package/types/claude-hooks/lib/trace.d.mts +1 -0
- package/types/claude-hooks/scan-invisible-chars.d.mts +18 -28
- package/types/claude-hooks/scan-loaded-instructions.d.mts +66 -0
- package/types/instructions.d.mts +1 -1
- package/types/src/claude-context.d.mts +74 -30
|
@@ -1,3 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The instruction files Claude Code loads from the directories ABOVE `dir`:
|
|
3
|
+
* walking up to the filesystem root, `CLAUDE.md` and `CLAUDE.local.md` in each
|
|
4
|
+
* parent are loaded IN FULL at launch, so a payload planted in a parent
|
|
5
|
+
* directory reaches the model exactly like one in the project's own file.
|
|
6
|
+
*
|
|
7
|
+
* `AGENTS.md` is absent by design: the parent-chain load is Claude Code's rule,
|
|
8
|
+
* and Claude Code does not read `AGENTS.md`.
|
|
9
|
+
*
|
|
10
|
+
* Returns CANDIDATES — absolute paths, existing or not, because this module
|
|
11
|
+
* touches no filesystem. Most parents of any directory hold neither file, so a
|
|
12
|
+
* caller that buckets its misses as "absent" should filter first: ~10 phantom
|
|
13
|
+
* entries per session would drown the one signal that bucket carries, a target
|
|
14
|
+
* that existed when the scan listed it and vanished before the read.
|
|
15
|
+
* @param {string} dir the scan root; its own files are NOT included
|
|
16
|
+
* @returns {string[]}
|
|
17
|
+
*/
|
|
18
|
+
export function ancestorInstructionFiles(dir: string): string[];
|
|
19
|
+
/**
|
|
20
|
+
* Whether `file` lives inside `dir`. Lexical, on already-absolute paths, and the
|
|
21
|
+
* bound on where an instruction-file scanner may REWRITE: a file above the scan
|
|
22
|
+
* root is shared with every other project beneath that root, so it is reported
|
|
23
|
+
* rather than silently edited. Both scanners ask this — one for a target it
|
|
24
|
+
* globbed, one for a path an event handed it — and a copy each is a copy that
|
|
25
|
+
* can drift into rewriting a file the other would not.
|
|
26
|
+
*
|
|
27
|
+
* Symlinks are deliberately not resolved: the guard that stops a link from
|
|
28
|
+
* redirecting the write has to live at the write itself (cleanFile opens
|
|
29
|
+
* O_NOFOLLOW), and resolving here would only duplicate it a check too early.
|
|
30
|
+
* @param {string} dir
|
|
31
|
+
* @param {string} file
|
|
32
|
+
* @returns {boolean}
|
|
33
|
+
*/
|
|
34
|
+
export function isInsideDir(dir: string, file: string): boolean;
|
|
1
35
|
/**
|
|
2
36
|
* The one directory no instruction-file walk ever descends into. Its own
|
|
3
37
|
* function so the name is spelled once, and so the two predicates that need it
|
|
@@ -25,30 +59,6 @@ export function excludeNodeModules(entry: string): boolean;
|
|
|
25
59
|
* @returns {boolean}
|
|
26
60
|
*/
|
|
27
61
|
export function excludeFromContextScan(entry: string): boolean;
|
|
28
|
-
/**
|
|
29
|
-
* WHICH files an agent loads as model context, as data: the glob set and the
|
|
30
|
-
* walk-pruning predicate that together define "everything Claude Code reads as
|
|
31
|
-
* instructions, and nothing else".
|
|
32
|
-
*
|
|
33
|
-
* This is the SINGLE SOURCE for that scope. It used to live inside
|
|
34
|
-
* `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
|
|
35
|
-
* knew the answer and nobody else did: `src/instructions.mjs` takes
|
|
36
|
-
* caller-supplied globs by design (no agent's convention is baked into the
|
|
37
|
-
* engine), so the CLI, the Python port and every downstream fork spelled their
|
|
38
|
-
* own approximation of this list — and an approximation that drifts either
|
|
39
|
-
* scans bulk data that can never reach the model (the 30-second session start
|
|
40
|
-
* this whitelist exists to fix) or MISSES a context directory entirely, which
|
|
41
|
-
* is a silent hole in the one scan standing between a poisoned instruction file
|
|
42
|
-
* and a session that loads it.
|
|
43
|
-
*
|
|
44
|
-
* It is a standalone, dependency-free DATA module (like ./cf-charset.mjs) for
|
|
45
|
-
* two reasons: `src/instructions.mjs` re-exports it as the library's public
|
|
46
|
-
* door, and the hook imports it RELATIVELY — deliberately not through the
|
|
47
|
-
* `agent-sanitizer` specifier the plugin bundle pins to a published engine.
|
|
48
|
-
* This scope is hook POLICY, not engine behavior: it must ship and move with the
|
|
49
|
-
* hook that walks it, or a plugin built against an older pin would prune the
|
|
50
|
-
* wrong directories while believing it had scanned everything.
|
|
51
|
-
*/
|
|
52
62
|
/**
|
|
53
63
|
* The `.claude/` subdirectories whose markdown Claude Code loads as model
|
|
54
64
|
* context. This is a WHITELIST, and that is the point: `.claude/` is also where
|
|
@@ -68,12 +78,29 @@ export function excludeFromContextScan(entry: string): boolean;
|
|
|
68
78
|
*/
|
|
69
79
|
export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
70
80
|
/**
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
*
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
81
|
+
* Claude Code's own per-directory memory files. Their own list because the
|
|
82
|
+
* parent-chain load ({@link ancestorInstructionFiles}) is Claude Code's rule and
|
|
83
|
+
* covers exactly these two.
|
|
84
|
+
*/
|
|
85
|
+
export const CLAUDE_MEMORY_FILES: readonly string[];
|
|
86
|
+
/**
|
|
87
|
+
* Every per-directory instruction file: Claude Code's memory files plus
|
|
88
|
+
* `AGENTS.md`, the cross-agent convention Claude Code does not read itself, kept
|
|
89
|
+
* because this package guards agents generally and the file is loaded as
|
|
90
|
+
* instructions by the ones that do.
|
|
91
|
+
*/
|
|
92
|
+
export const CLAUDE_DIR_INSTRUCTION_FILES: readonly string[];
|
|
93
|
+
/**
|
|
94
|
+
* Every glob whose matches Claude Code loads as model context ANYWHERE in a
|
|
95
|
+
* tree: the per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
|
|
96
|
+
* AGENTS.md) and the whitelisted `.claude/` markdown. Claude Code loads these on
|
|
97
|
+
* entry to their containing directory — a load path that bypasses the PostToolUse
|
|
98
|
+
* sanitizer — so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
|
|
99
|
+
* model uncleaned unless something scans it.
|
|
100
|
+
*
|
|
101
|
+
* This is the WHOLE-TREE scope, for a caller scanning a project on demand (the
|
|
102
|
+
* CLI, the Python port). It is not what a SessionStart hook walks — see
|
|
103
|
+
* {@link CLAUDE_LAUNCH_GLOBS} for why, and for what does.
|
|
77
104
|
*
|
|
78
105
|
* `**` does not descend into dot directories, so NESTED `.claude/` trees need
|
|
79
106
|
* their own doubled-star-prefixed patterns: without them a directory-scoped
|
|
@@ -86,3 +113,20 @@ export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
|
|
|
86
113
|
* MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
|
|
87
114
|
*/
|
|
88
115
|
export const CLAUDE_INSTRUCTION_GLOBS: readonly string[];
|
|
116
|
+
/**
|
|
117
|
+
* Every glob whose matches Claude Code loads AT LAUNCH from the scan root
|
|
118
|
+
* itself: the root's own instruction files and its `.claude` context tree. Same
|
|
119
|
+
* patterns as {@link CLAUDE_INSTRUCTION_GLOBS} without the doubled-star root, built
|
|
120
|
+
* from the same two lists so the pair cannot drift.
|
|
121
|
+
*
|
|
122
|
+
* Deliberately NOT recursive. A subdirectory's `CLAUDE.md` is loaded when Claude
|
|
123
|
+
* Code reads a file in that subdirectory, not at launch, so globbing for it at
|
|
124
|
+
* session start pays a whole-tree walk (the entire home directory, when the
|
|
125
|
+
* session is launched there) to pre-scan files that mostly never load. The
|
|
126
|
+
* InstructionsLoaded hook scans each of those at the moment it loads instead —
|
|
127
|
+
* which is also the only moment that catches one created mid-session.
|
|
128
|
+
*
|
|
129
|
+
* Pair with {@link ancestorInstructionFiles} for the other half of the launch
|
|
130
|
+
* set, and with {@link excludeFromContextScan} to prune the `.claude` walk.
|
|
131
|
+
*/
|
|
132
|
+
export const CLAUDE_LAUNCH_GLOBS: readonly string[];
|