agent-sanitizer 2.35.0 → 2.36.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,27 +1,44 @@
1
1
  /**
2
- * WHICH files an agent loads as model context, as data: the glob set and the
2
+ * WHICH files an agent loads as model context, as data: the glob sets and the
3
3
  * walk-pruning predicate that together define "everything Claude Code reads as
4
4
  * instructions, and nothing else".
5
5
  *
6
- * This is the SINGLE SOURCE for that scope. It used to live inside
6
+ * TWO scopes live here, because Claude Code has two load moments and scanning
7
+ * them at one moment is what made session start unusable:
8
+ *
9
+ * - {@link CLAUDE_LAUNCH_GLOBS} + {@link ancestorInstructionFiles} — what
10
+ * loads AT LAUNCH: the working directory's own instruction files, the same
11
+ * files in every directory above it, and the root `.claude` context tree.
12
+ * Rooted at the scan root, so it costs one shallow glob and a walk up the
13
+ * parent chain no matter how large the tree below is.
14
+ * - {@link CLAUDE_INSTRUCTION_GLOBS} — every instruction file ANYWHERE in a
15
+ * tree, `**`-rooted. A library caller asking "scan this project" wants
16
+ * this; a SessionStart hook must not, because Claude Code loads a
17
+ * subdirectory's `CLAUDE.md` only when it reads a file in that
18
+ * subdirectory, and a launch in `$HOME` charges the whole home tree —
19
+ * ~100 seconds of blocked startup — for files that mostly never load.
20
+ * Those lazily-loaded files are scanned by the InstructionsLoaded hook, at
21
+ * the moment they load.
22
+ *
23
+ * This is the SINGLE SOURCE for both. It used to live inside
7
24
  * `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
8
25
  * knew the answer and nobody else did: `src/instructions.mjs` takes
9
26
  * caller-supplied globs by design (no agent's convention is baked into the
10
27
  * engine), so the CLI, the Python port and every downstream fork spelled their
11
28
  * own approximation of this list — and an approximation that drifts either
12
- * scans bulk data that can never reach the model (the 30-second session start
13
- * this whitelist exists to fix) or MISSES a context directory entirely, which
14
- * is a silent hole in the one scan standing between a poisoned instruction file
15
- * and a session that loads it.
29
+ * scans bulk data that can never reach the model or MISSES a context directory
30
+ * entirely, which is a silent hole in the one scan standing between a poisoned
31
+ * instruction file and a session that loads it.
16
32
  *
17
- * It is a standalone, dependency-free DATA module (like ./cf-charset.mjs) for
18
- * two reasons: `src/instructions.mjs` re-exports it as the library's public
19
- * door, and the hook imports it RELATIVELY — deliberately not through the
20
- * `agent-sanitizer` specifier the plugin bundle pins to a published engine.
21
- * This scope is hook POLICY, not engine behavior: it must ship and move with the
22
- * hook that walks it, or a plugin built against an older pin would prune the
23
- * wrong directories while believing it had scanned everything.
33
+ * It is a standalone DATA module carrying no package dependency (like
34
+ * ./cf-charset.mjs) for two reasons: `src/instructions.mjs` re-exports it as the
35
+ * library's public door, and the hook imports it RELATIVELY — deliberately not
36
+ * through the `agent-sanitizer` specifier the plugin bundle pins to a published
37
+ * engine. This scope is hook POLICY, not engine behavior: it must ship and move
38
+ * with the hook that walks it, or a plugin built against an older pin would
39
+ * prune the wrong directories while believing it had scanned everything.
24
40
  */
41
+ import { dirname, isAbsolute, join, relative, resolve } from "node:path";
25
42
 
26
43
  /**
27
44
  * The `.claude/` subdirectories whose markdown Claude Code loads as model
@@ -44,11 +61,33 @@ export const CLAUDE_CONTEXT_SUBDIRS = Object.freeze([
44
61
  "agents",
45
62
  "commands",
46
63
  "output-styles",
64
+ "rules",
47
65
  "skills",
48
66
  ]);
49
67
 
50
- // The glob patterns for one `.claude` tree at `prefix` (empty for the project
51
- // root, a doubled-star segment for nested ones): its top-level markdown, plus the
68
+ /**
69
+ * Claude Code's own per-directory memory files. Their own list because the
70
+ * parent-chain load ({@link ancestorInstructionFiles}) is Claude Code's rule and
71
+ * covers exactly these two.
72
+ */
73
+ export const CLAUDE_MEMORY_FILES = Object.freeze([
74
+ "CLAUDE.md",
75
+ "CLAUDE.local.md",
76
+ ]);
77
+
78
+ /**
79
+ * Every per-directory instruction file: Claude Code's memory files plus
80
+ * `AGENTS.md`, the cross-agent convention Claude Code does not read itself, kept
81
+ * because this package guards agents generally and the file is loaded as
82
+ * instructions by the ones that do.
83
+ */
84
+ export const CLAUDE_DIR_INSTRUCTION_FILES = Object.freeze([
85
+ ...CLAUDE_MEMORY_FILES,
86
+ "AGENTS.md",
87
+ ]);
88
+
89
+ // The glob patterns for one `.claude` tree at `prefix` (empty for the scan root,
90
+ // a doubled-star segment for nested ones): its top-level markdown, plus the
52
91
  // whitelisted context subdirectories. Built once, from the one list above.
53
92
  /** @param {string} prefix @returns {string[]} */
54
93
  function claudeDirPatterns(prefix) {
@@ -59,12 +98,16 @@ function claudeDirPatterns(prefix) {
59
98
  }
60
99
 
61
100
  /**
62
- * Every glob whose matches Claude Code loads as model context: the
63
- * per-directory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and
64
- * the whitelisted `.claude/` markdown. Claude Code loads these on entry to their
65
- * containing directory — a load path that bypasses the PostToolUse sanitizer —
66
- * so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model
67
- * uncleaned unless something scans it here.
101
+ * Every glob whose matches Claude Code loads as model context ANYWHERE in a
102
+ * tree: the per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
103
+ * AGENTS.md) and the whitelisted `.claude/` markdown. Claude Code loads these on
104
+ * entry to their containing directory — a load path that bypasses the PostToolUse
105
+ * sanitizer — so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
106
+ * model uncleaned unless something scans it.
107
+ *
108
+ * This is the WHOLE-TREE scope, for a caller scanning a project on demand (the
109
+ * CLI, the Python port). It is not what a SessionStart hook walks — see
110
+ * {@link CLAUDE_LAUNCH_GLOBS} for why, and for what does.
68
111
  *
69
112
  * `**` does not descend into dot directories, so NESTED `.claude/` trees need
70
113
  * their own doubled-star-prefixed patterns: without them a directory-scoped
@@ -77,12 +120,85 @@ function claudeDirPatterns(prefix) {
77
120
  * MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
78
121
  */
79
122
  export const CLAUDE_INSTRUCTION_GLOBS = Object.freeze([
80
- "**/CLAUDE.md",
81
- "**/CLAUDE.local.md",
82
- "**/AGENTS.md",
123
+ ...CLAUDE_DIR_INSTRUCTION_FILES.map((name) => `**/${name}`),
83
124
  ...claudeDirPatterns("**/"),
84
125
  ]);
85
126
 
127
+ /**
128
+ * Every glob whose matches Claude Code loads AT LAUNCH from the scan root
129
+ * itself: the root's own instruction files and its `.claude` context tree. Same
130
+ * patterns as {@link CLAUDE_INSTRUCTION_GLOBS} without the doubled-star root, built
131
+ * from the same two lists so the pair cannot drift.
132
+ *
133
+ * Deliberately NOT recursive. A subdirectory's `CLAUDE.md` is loaded when Claude
134
+ * Code reads a file in that subdirectory, not at launch, so globbing for it at
135
+ * session start pays a whole-tree walk (the entire home directory, when the
136
+ * session is launched there) to pre-scan files that mostly never load. The
137
+ * InstructionsLoaded hook scans each of those at the moment it loads instead —
138
+ * which is also the only moment that catches one created mid-session.
139
+ *
140
+ * Pair with {@link ancestorInstructionFiles} for the other half of the launch
141
+ * set, and with {@link excludeFromContextScan} to prune the `.claude` walk.
142
+ */
143
+ export const CLAUDE_LAUNCH_GLOBS = Object.freeze([
144
+ ...CLAUDE_DIR_INSTRUCTION_FILES,
145
+ ...claudeDirPatterns(""),
146
+ ]);
147
+
148
+ /**
149
+ * The instruction files Claude Code loads from the directories ABOVE `dir`:
150
+ * walking up to the filesystem root, `CLAUDE.md` and `CLAUDE.local.md` in each
151
+ * parent are loaded IN FULL at launch, so a payload planted in a parent
152
+ * directory reaches the model exactly like one in the project's own file.
153
+ *
154
+ * `AGENTS.md` is absent by design: the parent-chain load is Claude Code's rule,
155
+ * and Claude Code does not read `AGENTS.md`.
156
+ *
157
+ * Returns CANDIDATES — absolute paths, existing or not, because this module
158
+ * touches no filesystem. Most parents of any directory hold neither file, so a
159
+ * caller that buckets its misses as "absent" should filter first: ~10 phantom
160
+ * entries per session would drown the one signal that bucket carries, a target
161
+ * that existed when the scan listed it and vanished before the read.
162
+ * @param {string} dir the scan root; its own files are NOT included
163
+ * @returns {string[]}
164
+ */
165
+ export function ancestorInstructionFiles(dir) {
166
+ /** @type {string[]} */
167
+ const files = [];
168
+ let current = resolve(dir);
169
+ // `dirname` is its own fixed point at the filesystem root, which is what ends
170
+ // the walk on every platform without spelling a root path here.
171
+ for (
172
+ let parent = dirname(current);
173
+ parent !== current;
174
+ parent = dirname(current)
175
+ ) {
176
+ current = parent;
177
+ for (const name of CLAUDE_MEMORY_FILES) files.push(join(current, name));
178
+ }
179
+ return files;
180
+ }
181
+
182
+ /**
183
+ * Whether `file` lives inside `dir`. Lexical, on already-absolute paths, and the
184
+ * bound on where an instruction-file scanner may REWRITE: a file above the scan
185
+ * root is shared with every other project beneath that root, so it is reported
186
+ * rather than silently edited. Both scanners ask this — one for a target it
187
+ * globbed, one for a path an event handed it — and a copy each is a copy that
188
+ * can drift into rewriting a file the other would not.
189
+ *
190
+ * Symlinks are deliberately not resolved: the guard that stops a link from
191
+ * redirecting the write has to live at the write itself (cleanFile opens
192
+ * O_NOFOLLOW), and resolving here would only duplicate it a check too early.
193
+ * @param {string} dir
194
+ * @param {string} file
195
+ * @returns {boolean}
196
+ */
197
+ export function isInsideDir(dir, file) {
198
+ const rel = relative(dir, file);
199
+ return rel !== "" && !rel.startsWith("..") && !isAbsolute(rel);
200
+ }
201
+
86
202
  /**
87
203
  * The one directory no instruction-file walk ever descends into. Its own
88
204
  * function so the name is spelled once, and so the two predicates that need it
@@ -13,8 +13,10 @@
13
13
  * instruction files live under (e.g. `["CLAUDE.md", "AGENTS.md",
14
14
  * ".claude/**\/*.md", "**\/SKILL.md"]`), so no agent's convention is baked in.
15
15
  * Claude Code's own convention is re-exported below
16
- * ({@link CLAUDE_INSTRUCTION_GLOBS} / {@link excludeFromContextScan}) so a
17
- * caller that wants it takes the SessionStart hook's exact scope rather than
16
+ * ({@link CLAUDE_INSTRUCTION_GLOBS} for a whole tree,
17
+ * {@link CLAUDE_LAUNCH_GLOBS} + {@link ancestorInstructionFiles} for just what a
18
+ * session loads at launch, {@link excludeFromContextScan} to prune either walk)
19
+ * so a caller that wants it takes the hooks' exact scope rather than
18
20
  * approximating it — see ./claude-context.mjs.
19
21
  */
20
22
  import {
@@ -48,8 +50,12 @@ import { excludeNodeModules } from "./claude-context.mjs";
48
50
  // makes `agent-sanitizer/instructions` the single place a CLI, a port or a fork
49
51
  // reads that scope from instead of re-spelling it.
50
52
  export {
53
+ ancestorInstructionFiles,
51
54
  CLAUDE_CONTEXT_SUBDIRS,
55
+ CLAUDE_DIR_INSTRUCTION_FILES,
52
56
  CLAUDE_INSTRUCTION_GLOBS,
57
+ CLAUDE_LAUNCH_GLOBS,
58
+ CLAUDE_MEMORY_FILES,
53
59
  excludeFromContextScan,
54
60
  } from "./claude-context.mjs";
55
61
 
@@ -1,3 +1,37 @@
1
+ /**
2
+ * The instruction files Claude Code loads from the directories ABOVE `dir`:
3
+ * walking up to the filesystem root, `CLAUDE.md` and `CLAUDE.local.md` in each
4
+ * parent are loaded IN FULL at launch, so a payload planted in a parent
5
+ * directory reaches the model exactly like one in the project's own file.
6
+ *
7
+ * `AGENTS.md` is absent by design: the parent-chain load is Claude Code's rule,
8
+ * and Claude Code does not read `AGENTS.md`.
9
+ *
10
+ * Returns CANDIDATES — absolute paths, existing or not, because this module
11
+ * touches no filesystem. Most parents of any directory hold neither file, so a
12
+ * caller that buckets its misses as "absent" should filter first: ~10 phantom
13
+ * entries per session would drown the one signal that bucket carries, a target
14
+ * that existed when the scan listed it and vanished before the read.
15
+ * @param {string} dir the scan root; its own files are NOT included
16
+ * @returns {string[]}
17
+ */
18
+ export function ancestorInstructionFiles(dir: string): string[];
19
+ /**
20
+ * Whether `file` lives inside `dir`. Lexical, on already-absolute paths, and the
21
+ * bound on where an instruction-file scanner may REWRITE: a file above the scan
22
+ * root is shared with every other project beneath that root, so it is reported
23
+ * rather than silently edited. Both scanners ask this — one for a target it
24
+ * globbed, one for a path an event handed it — and a copy each is a copy that
25
+ * can drift into rewriting a file the other would not.
26
+ *
27
+ * Symlinks are deliberately not resolved: the guard that stops a link from
28
+ * redirecting the write has to live at the write itself (cleanFile opens
29
+ * O_NOFOLLOW), and resolving here would only duplicate it a check too early.
30
+ * @param {string} dir
31
+ * @param {string} file
32
+ * @returns {boolean}
33
+ */
34
+ export function isInsideDir(dir: string, file: string): boolean;
1
35
  /**
2
36
  * The one directory no instruction-file walk ever descends into. Its own
3
37
  * function so the name is spelled once, and so the two predicates that need it
@@ -25,30 +59,6 @@ export function excludeNodeModules(entry: string): boolean;
25
59
  * @returns {boolean}
26
60
  */
27
61
  export function excludeFromContextScan(entry: string): boolean;
28
- /**
29
- * WHICH files an agent loads as model context, as data: the glob set and the
30
- * walk-pruning predicate that together define "everything Claude Code reads as
31
- * instructions, and nothing else".
32
- *
33
- * This is the SINGLE SOURCE for that scope. It used to live inside
34
- * `claude-hooks/scan-invisible-chars.mjs`, which meant the SessionStart hook
35
- * knew the answer and nobody else did: `src/instructions.mjs` takes
36
- * caller-supplied globs by design (no agent's convention is baked into the
37
- * engine), so the CLI, the Python port and every downstream fork spelled their
38
- * own approximation of this list — and an approximation that drifts either
39
- * scans bulk data that can never reach the model (the 30-second session start
40
- * this whitelist exists to fix) or MISSES a context directory entirely, which
41
- * is a silent hole in the one scan standing between a poisoned instruction file
42
- * and a session that loads it.
43
- *
44
- * It is a standalone, dependency-free DATA module (like ./cf-charset.mjs) for
45
- * two reasons: `src/instructions.mjs` re-exports it as the library's public
46
- * door, and the hook imports it RELATIVELY — deliberately not through the
47
- * `agent-sanitizer` specifier the plugin bundle pins to a published engine.
48
- * This scope is hook POLICY, not engine behavior: it must ship and move with the
49
- * hook that walks it, or a plugin built against an older pin would prune the
50
- * wrong directories while believing it had scanned everything.
51
- */
52
62
  /**
53
63
  * The `.claude/` subdirectories whose markdown Claude Code loads as model
54
64
  * context. This is a WHITELIST, and that is the point: `.claude/` is also where
@@ -68,12 +78,29 @@ export function excludeFromContextScan(entry: string): boolean;
68
78
  */
69
79
  export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
70
80
  /**
71
- * Every glob whose matches Claude Code loads as model context: the
72
- * per-directory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and
73
- * the whitelisted `.claude/` markdown. Claude Code loads these on entry to their
74
- * containing directory — a load path that bypasses the PostToolUse sanitizer —
75
- * so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model
76
- * uncleaned unless something scans it here.
81
+ * Claude Code's own per-directory memory files. Their own list because the
82
+ * parent-chain load ({@link ancestorInstructionFiles}) is Claude Code's rule and
83
+ * covers exactly these two.
84
+ */
85
+ export const CLAUDE_MEMORY_FILES: readonly string[];
86
+ /**
87
+ * Every per-directory instruction file: Claude Code's memory files plus
88
+ * `AGENTS.md`, the cross-agent convention Claude Code does not read itself, kept
89
+ * because this package guards agents generally and the file is loaded as
90
+ * instructions by the ones that do.
91
+ */
92
+ export const CLAUDE_DIR_INSTRUCTION_FILES: readonly string[];
93
+ /**
94
+ * Every glob whose matches Claude Code loads as model context ANYWHERE in a
95
+ * tree: the per-directory instruction files (CLAUDE.md, CLAUDE.local.md,
96
+ * AGENTS.md) and the whitelisted `.claude/` markdown. Claude Code loads these on
97
+ * entry to their containing directory — a load path that bypasses the PostToolUse
98
+ * sanitizer — so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the
99
+ * model uncleaned unless something scans it.
100
+ *
101
+ * This is the WHOLE-TREE scope, for a caller scanning a project on demand (the
102
+ * CLI, the Python port). It is not what a SessionStart hook walks — see
103
+ * {@link CLAUDE_LAUNCH_GLOBS} for why, and for what does.
77
104
  *
78
105
  * `**` does not descend into dot directories, so NESTED `.claude/` trees need
79
106
  * their own doubled-star-prefixed patterns: without them a directory-scoped
@@ -86,3 +113,20 @@ export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
86
113
  * MATCH a bulk directory, but only pruning the WALK avoids paying to read it.
87
114
  */
88
115
  export const CLAUDE_INSTRUCTION_GLOBS: readonly string[];
116
+ /**
117
+ * Every glob whose matches Claude Code loads AT LAUNCH from the scan root
118
+ * itself: the root's own instruction files and its `.claude` context tree. Same
119
+ * patterns as {@link CLAUDE_INSTRUCTION_GLOBS} without the doubled-star root, built
120
+ * from the same two lists so the pair cannot drift.
121
+ *
122
+ * Deliberately NOT recursive. A subdirectory's `CLAUDE.md` is loaded when Claude
123
+ * Code reads a file in that subdirectory, not at launch, so globbing for it at
124
+ * session start pays a whole-tree walk (the entire home directory, when the
125
+ * session is launched there) to pre-scan files that mostly never load. The
126
+ * InstructionsLoaded hook scans each of those at the moment it loads instead —
127
+ * which is also the only moment that catches one created mid-session.
128
+ *
129
+ * Pair with {@link ancestorInstructionFiles} for the other half of the launch
130
+ * set, and with {@link excludeFromContextScan} to prune the `.claude` walk.
131
+ */
132
+ export const CLAUDE_LAUNCH_GLOBS: readonly string[];
@@ -425,6 +425,7 @@ export const HookEvent: Readonly<{
425
425
  POST_TOOL_USE: "PostToolUse";
426
426
  USER_PROMPT_SUBMIT: "UserPromptSubmit";
427
427
  SESSION_START: "SessionStart";
428
+ INSTRUCTIONS_LOADED: "InstructionsLoaded";
428
429
  }>;
429
430
  /** Claude Code permissionDecision verdicts. */
430
431
  export const PermissionDecision: Readonly<{
@@ -1,3 +1,66 @@
1
+ /**
2
+ * Marker the InstructionsLoaded scanner writes on every fire, so another hook
3
+ * can tell whether that event is being scanned at all this session.
4
+ *
5
+ * Keyed by the SESSION, and never cleared at SessionStart like the alert pair
6
+ * above (a later session sweeps it once it is older than the TTL):
7
+ * nothing pins the order of SessionStart against the InstructionsLoaded events
8
+ * Claude Code fires for the files it loads at launch, so a clear could erase a
9
+ * marker written moments earlier and produce the notice on a session that IS
10
+ * covered. Session-keyed, the question each session asks is answered by that
11
+ * session's own file and no ordering matters. A host that exports no session id
12
+ * falls back to one shared name — where the marker can outlive its session, and
13
+ * a later session on a host that stopped emitting the event stays quiet.
14
+ * @param {string} [sessionId]
15
+ * @returns {string}
16
+ */
17
+ export function instructionsLoadedFile(sessionId?: string): string;
18
+ /**
19
+ * Companion marker: the notice below has been surfaced this session.
20
+ * @param {string} [sessionId]
21
+ * @returns {string}
22
+ */
23
+ export function instructionsLoadedNoticeFile(sessionId?: string): string;
24
+ /**
25
+ * Whether the InstructionsLoaded scanner has run this session — i.e. whether the
26
+ * lazily-loaded instruction files are being scanned at all. Ownership-validated
27
+ * like every other marker here: a co-tenant could otherwise plant the
28
+ * predictable path and suppress the notice below, which is the whole signal that
29
+ * nested files are going unscanned.
30
+ * @param {string} [sessionId]
31
+ * @returns {boolean}
32
+ */
33
+ export function instructionsLoadedSeen(sessionId?: string): boolean;
34
+ /**
35
+ * Record that the InstructionsLoaded scanner engaged. Symlink-safe presence
36
+ * write (see writeSentinelFile) at a predictable $TMPDIR path.
37
+ *
38
+ * The event fires once per instruction file loaded, so the already-recorded case
39
+ * returns without a write — and the stale-marker sweep rides the FIRST fire of a
40
+ * session, where one readdir is paid once rather than per loaded file.
41
+ * @param {string} [sessionId]
42
+ * @returns {void}
43
+ */
44
+ export function recordInstructionsLoaded(sessionId?: string): void;
45
+ /**
46
+ * The one-time context line for a session where no InstructionsLoaded scan ran,
47
+ * or null when the scan has been seen or the notice was already surfaced this
48
+ * session. Records the notice as it hands it out, so it rides on ONE tool call
49
+ * rather than every one — the per-call repeat is what trains a reader to skip it.
50
+ *
51
+ * The loss it names is real and otherwise invisible: SessionStart scans the
52
+ * instruction files that load at launch, and everything a subdirectory loads
53
+ * later is scanned by the event. No scan, and nothing says so.
54
+ *
55
+ * The notice names the OBSERVABLE — no scan ran — and both of its causes, because
56
+ * the marker cannot tell a host that never emits the event from an operator who
57
+ * switched the hook off in AGENT_SANITIZER_DISABLED_HOOKS, and asserting the
58
+ * first would send an operator who chose the second to the wrong fix.
59
+ * @param {string} [sessionId] the harness's session identity, so the answer
60
+ * belongs to THIS session (see instructionsLoadedFile)
61
+ * @returns {string | null}
62
+ */
63
+ export function instructionsLoadedGapNotice(sessionId?: string): string | null;
1
64
  /**
2
65
  * The alert findings if invisible-char injection was detected in instruction
3
66
  * files and couldn't be auto-cleaned, else null. ALERT_FILE lives at a predictable,
@@ -9,6 +72,21 @@
9
72
  * @returns {string | null}
10
73
  */
11
74
  export function invisibleCharAlert(): string | null;
75
+ /**
76
+ * Add `text` to the alert the PreToolUse gate surfaces, keeping whatever is
77
+ * already there.
78
+ *
79
+ * Appending, where the SessionStart scanner TRUNCATES: that scan runs once and
80
+ * owns the session's reset, while an instruction file loaded mid-session is one
81
+ * more finding on top of whatever the launch scan left — a truncating write here
82
+ * would silently drop the earlier report. Symlink-refusing (writeFileNoFollow)
83
+ * and ownership-checked on read, because ALERT_FILE sits at a predictable,
84
+ * world-visible $TMPDIR path; a foreign or squatted file reads as empty and is
85
+ * replaced rather than appended to.
86
+ * @param {string} text
87
+ * @returns {void}
88
+ */
89
+ export function appendAlert(text: string): void;
12
90
  /**
13
91
  * True once the gate has surfaced its blocking ask this session. Validates
14
92
  * ownership (not mere existence): a co-tenant could pre-create ALERT_ACK_FILE at its
@@ -0,0 +1,26 @@
1
+ /**
2
+ * The operator-facing report for hidden-Unicode findings in instruction files.
3
+ *
4
+ * Its own module because two hooks render it — the SessionStart scan of the
5
+ * files that load at launch, and the InstructionsLoaded scan of every file
6
+ * loaded after that — and a second copy would drift in exactly the way that
7
+ * matters: the framing that keeps a decoded payload from reading as an
8
+ * instruction (see decodeRun's `untrusted data` prefix) is part of the report,
9
+ * not decoration on it.
10
+ */
11
+ /**
12
+ * @param {Array<{
13
+ * file: string,
14
+ * findings: Array<{ line: number | null, charCount: number, method: string, decoded: string }>,
15
+ * }>} allFindings
16
+ * @returns {string}
17
+ */
18
+ export function formatReport(allFindings: Array<{
19
+ file: string;
20
+ findings: Array<{
21
+ line: number | null;
22
+ charCount: number;
23
+ method: string;
24
+ decoded: string;
25
+ }>;
26
+ }>): string;
@@ -47,6 +47,7 @@ export function bestEffortTrace(sink: TraceFn): TraceFn;
47
47
  export const TraceEvent: Readonly<{
48
48
  HOOK_RAN: "hook_ran";
49
49
  SCAN_INVISIBLE_CHARS_RAN: "scan_invisible_chars_ran";
50
+ SCAN_LOADED_INSTRUCTIONS_RAN: "scan_loaded_instructions_ran";
50
51
  }>;
51
52
  /**
52
53
  * The sink shape a hook emits through: the event name, its metadata fields, and
@@ -71,6 +71,7 @@ export function cliMain(opts?: {
71
71
  export function scanFile(filePath: string): ReturnType<typeof import("agent-sanitizer/instructions").scanText>;
72
72
  import { CLAUDE_CONTEXT_SUBDIRS } from "../src/claude-context.mjs";
73
73
  import { CLAUDE_INSTRUCTION_GLOBS } from "../src/claude-context.mjs";
74
+ import { CLAUDE_LAUNCH_GLOBS } from "../src/claude-context.mjs";
74
75
  /**
75
76
  * The SSOT decoder, re-exported through a lazy-bound wrapper (the binding is
76
77
  * `let` and may be re-bound by the cold-start reload, so the export must read
@@ -86,18 +87,23 @@ export function decodeRun(run: string): {
86
87
  decoded: string;
87
88
  };
88
89
  /**
89
- * Every file under `dir` that Claude Code loads as model context: the
90
- * per-directory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and
91
- * the whitelisted `.claude/` markdown. Claude Code loads these on entry to their
92
- * containing directory — a load path that bypasses the PostToolUse sanitizer —
93
- * so a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model
94
- * uncleaned unless it is scanned here.
90
+ * Every file Claude Code loads as model context AT LAUNCH: `dir`'s own
91
+ * instruction files and its `.claude/` context tree, plus the CLAUDE.md /
92
+ * CLAUDE.local.md of every directory above it (loaded in full at launch, and
93
+ * until now never scanned by anything). These load before the session's first
94
+ * tool call — a path that bypasses the PostToolUse sanitizer — so a payload in
95
+ * one of them reaches the model uncleaned unless it is scanned here.
95
96
  *
96
- * The scope itself — which globs, and which directories the walk must prune —
97
- * is the library's {@link CLAUDE_INSTRUCTION_GLOBS} /
98
- * {@link excludeFromContextScan}, so this hook and every other consumer read one
99
- * list (see src/claude-context.mjs for why it is imported relatively rather than
100
- * through the `agent-sanitizer` specifier the plugin bundle pins).
97
+ * Bounded on purpose: one shallow glob plus a walk up the parent chain. The
98
+ * `**`-rooted scope ({@link CLAUDE_INSTRUCTION_GLOBS}) walks the entire tree
99
+ * below `dir`, which for a session launched in a home directory is ~100 seconds
100
+ * of blocked startup spent on files Claude Code does not load at launch. Those
101
+ * files load when a tool reads their directory, and scan-loaded-instructions
102
+ * scans each one at that moment.
103
+ *
104
+ * The scope itself — which globs, and which directories the walk must prune — is
105
+ * the library's (see src/claude-context.mjs for why it is imported relatively
106
+ * rather than through the `agent-sanitizer` specifier the plugin bundle pins).
101
107
  * @param {string} dir
102
108
  * @returns {string[]}
103
109
  */
@@ -107,20 +113,4 @@ import { ALERT_ACK_FILE } from "./lib/invisible-alert.mjs";
107
113
  export let LONG_RUN_RE: RegExp;
108
114
  export let LONG_RUN_THRESHOLD: 10;
109
115
  export let TOTAL_INVISIBLE_THRESHOLD: 30;
110
- /**
111
- * @param {Array<{
112
- * file: string,
113
- * findings: Array<{ line: number | null, charCount: number, method: string, decoded: string }>,
114
- * }>} allFindings
115
- * @returns {string}
116
- */
117
- export function formatReport(allFindings: Array<{
118
- file: string;
119
- findings: Array<{
120
- line: number | null;
121
- charCount: number;
122
- method: string;
123
- decoded: string;
124
- }>;
125
- }>): string;
126
- export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, ALERT_FILE, ALERT_ACK_FILE };
116
+ export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, CLAUDE_LAUNCH_GLOBS, ALERT_FILE, ALERT_ACK_FILE };
@@ -0,0 +1,66 @@
1
+ /**
2
+ * The payload fields this hook reads, validated. A payload missing either is
3
+ * harness-contract drift, not a clean file: reporting "no findings" for bytes we
4
+ * never saw is the one answer that must never be reachable, so this throws into
5
+ * the declared fault posture instead.
6
+ * @param {unknown} payload
7
+ * @returns {{ filePath: string, content: string, loadReason: string }}
8
+ */
9
+ export function readLoadedFile(payload: unknown): {
10
+ filePath: string;
11
+ content: string;
12
+ loadReason: string;
13
+ };
14
+ /**
15
+ * Scan one loaded instruction file. Returns the report and what to do with it:
16
+ * `cleaned` says the payload is gone from disk, `alert` carries the text that
17
+ * must arm the PreToolUse gate (empty when the clean succeeded).
18
+ *
19
+ * The scan runs on the payload's bytes, never a re-read of the path: those are
20
+ * the bytes that reached the model, and a file rewritten between the load and
21
+ * this hook would otherwise be scanned in a state the model never saw.
22
+ * @param {{ filePath: string, content: string }} loaded
23
+ * @param {{ projectDir?: string, clean?: typeof cleanFile }} [opts] injectable
24
+ * for tests; the default cleans through the SSOT's guarded rewrite
25
+ * @returns {{ report: string, cleaned: boolean, reason: string | null } | null}
26
+ * null when the file is clean
27
+ */
28
+ export function scanLoadedFile({ filePath, content }: {
29
+ filePath: string;
30
+ content: string;
31
+ }, { projectDir, clean }?: {
32
+ projectDir?: string;
33
+ clean?: typeof cleanFile;
34
+ }): {
35
+ report: string;
36
+ cleaned: boolean;
37
+ reason: string | null;
38
+ } | null;
39
+ /**
40
+ * The operator- and model-facing text for a scanned file. Both channels carry
41
+ * it: the bytes are already in context, so the model is told to distrust what it
42
+ * just read, and the user is told what changed on disk.
43
+ * @param {{ report: string, cleaned: boolean, reason: string | null }} result
44
+ * @param {string} filePath
45
+ * @returns {string}
46
+ */
47
+ export function loadedFileMessage({ report, cleaned, reason }: {
48
+ report: string;
49
+ cleaned: boolean;
50
+ reason: string | null;
51
+ }, filePath: string): string;
52
+ /**
53
+ * The hook's CLI: read the event, scan the loaded bytes, clean and report.
54
+ * Exported so a bundle entry (which must claim the CLI slot before this module
55
+ * loads) can run the exact same wiring instead of duplicating it.
56
+ * @param {{ trace?: import("./lib/trace.mjs").TraceFn }} [opts] `trace` is
57
+ * where this scan announces engagement; a host with its own trace channel
58
+ * passes its sink (see lib/trace.mjs)
59
+ * @returns {Promise<void>}
60
+ */
61
+ export function cliMain({ trace: sink }?: {
62
+ trace?: import("./lib/trace.mjs").TraceFn;
63
+ }): Promise<void>;
64
+ declare const cleanFile: typeof import("agent-sanitizer/instructions").cleanFile;
65
+ export const HOOK_NAME: "scan-loaded-instructions";
66
+ export {};
@@ -148,4 +148,4 @@ export function atomicReplaceFile(absPath: string, data: string, mode: number, t
148
148
  * @returns {boolean}
149
149
  */
150
150
  export function cleanFile(absPath: string, lstat?: (path: string) => import("node:fs").Stats): boolean;
151
- export { CLAUDE_CONTEXT_SUBDIRS, CLAUDE_INSTRUCTION_GLOBS, excludeFromContextScan } from "./claude-context.mjs";
151
+ export { ancestorInstructionFiles, CLAUDE_CONTEXT_SUBDIRS, CLAUDE_DIR_INSTRUCTION_FILES, CLAUDE_INSTRUCTION_GLOBS, CLAUDE_LAUNCH_GLOBS, CLAUDE_MEMORY_FILES, excludeFromContextScan } from "./claude-context.mjs";