agent-sanitizer 2.23.2 → 2.24.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -67,10 +67,28 @@ export function formatSkipped(skipped: Array<{
67
67
  * untested fault path is how a posture goes missing in the first place.
68
68
  * @returns {Promise<void>}
69
69
  */
70
- export function cliMain({ trace: sink, scan: runScan }?: {
70
+ export function cliMain(opts?: {
71
71
  trace?: import("./lib/trace.mjs").TraceFn;
72
72
  scan?: () => ReturnType<typeof scanProject>;
73
73
  }): Promise<void>;
74
+ /**
75
+ * The `.claude/` subdirectories whose markdown Claude Code loads as model
76
+ * context. This is a WHITELIST, and that is the point: `.claude/` is also where
77
+ * tooling parks bulk data that is never loaded as context — `worktrees/`
78
+ * (entire checked-out copies of the repo), plus caches, transcripts and
79
+ * snapshots — and globbing `.claude/**` swept all of it in. On a repo with a few
80
+ * populated worktrees that is thousands of files READ at every session start:
81
+ * one report put it at 30 seconds of blocked startup, paid for scanning files
82
+ * that cannot reach the model.
83
+ *
84
+ * A whitelist, not a `worktrees` denylist, because the failure modes are not
85
+ * symmetric: an unlisted context directory costs a scan this hook was never
86
+ * asked for anyway (the PostToolUse sanitizer still cleans those bytes when a
87
+ * tool reads them), while an unlisted BULK directory silently costs every future
88
+ * session its startup. Add an entry here when Claude Code starts loading a new
89
+ * `.claude/` subdirectory as context.
90
+ */
91
+ export const CLAUDE_CONTEXT_SUBDIRS: readonly string[];
74
92
  /**
75
93
  * @param {string} filePath
76
94
  * @returns {Array<{ line: number, charCount: number, method: string, decoded: string }>}
@@ -90,21 +108,20 @@ export function decodeRun(run: string): {
90
108
  decoded: string;
91
109
  };
92
110
  /**
93
- * @param {string} dir
94
- * @returns {string[]}
95
- */
96
- export function findMdFiles(dir: string): string[];
97
- /**
98
- * Every subdirectory instruction file (CLAUDE.md, CLAUDE.local.md, AGENTS.md)
99
- * under `dir`. Claude Code loads these as project instructions on entry to their
100
- * containing directory — a load path that bypasses the PostToolUse sanitizer — so
101
- * a payload planted in e.g. `packages/foo/CLAUDE.md` reaches the model uncleaned
102
- * unless it is scanned here. Skips node_modules.
111
+ * Every file under `dir` that Claude Code loads as model context: the
112
+ * subdirectory instruction files (CLAUDE.md, CLAUDE.local.md, AGENTS.md) and the
113
+ * whitelisted `.claude/` markdown (see {@link CLAUDE_CONTEXT_SUBDIRS}). Claude
114
+ * Code loads these on entry to their containing directory — a load path that
115
+ * bypasses the PostToolUse sanitizer — so a payload planted in e.g.
116
+ * `packages/foo/CLAUDE.md` reaches the model uncleaned unless it is scanned
117
+ * here. Skips node_modules.
103
118
  *
104
119
  * `**` does not descend into dot directories, so NESTED `.claude/` trees need
105
- * their own pattern: the caller scans only the project-root `.claude`, which
106
- * would leave a directory-scoped skill at `packages/foo/.claude/skills/x/SKILL.md`
107
- * — model context by the same load path — never scanned.
120
+ * their own doubled-star-prefixed patterns: without them a directory-scoped skill at
121
+ * `packages/foo/.claude/skills/x/SKILL.md` — model context by the same load
122
+ * path — is never scanned. That same rule is why the root `.claude` needs no
123
+ * separate walk: a leading doubled star matches zero segments, so the nested
124
+ * patterns cover the root tree too.
108
125
  * @param {string} dir
109
126
  * @returns {string[]}
110
127
  */
package/types/index.d.mts CHANGED
@@ -37,6 +37,6 @@ export function sanitize(text: string, options?: {
37
37
  found: string[];
38
38
  warnings: string[];
39
39
  }>;
40
- export { applyLayer1, stripAnsiFully, LONE_SURROGATE_RE } from "./layer1.mjs";
40
+ export { applyLayer1, isBenignAnsi, isBenignAnsiKinds, stripAnsiFully, LONE_SURROGATE_RE } from "./layer1.mjs";
41
41
  export { stripInvisible, stripInvisibleWithReport, isSgrOnly, STRIP, SGR_RE, CHECKS, CATEGORY, CATEGORY_LABELS, LINGUISTIC_SCRIPTS, VS, BLANK_NON_CF, LONG_RUN_RE, LONG_RUN_THRESHOLD, SCATTERED_THRESHOLD } from "./invisible.mjs";
42
42
  export { HTML_TAG_PRESENT, MD_LINK_HINT, SECRET_HINT, SECRET_HINT_EXT, matchesSecretHint } from "./gates.mjs";
@@ -11,9 +11,54 @@
11
11
  * survives here. Past the bound a reconstituted sequence therefore degrades to
12
12
  * VISIBLE text rather than a hidden control, which is the fail-open direction.
13
13
  * @param {string} input
14
+ * @param {Set<string>} [kinds] see {@link stripAnsiOnce}; accumulates across passes
14
15
  * @returns {string}
15
16
  */
16
- export function stripAnsiFully(input: string): string;
17
+ export function stripAnsiFully(input: string, kinds?: Set<string>): string;
18
+ /**
19
+ * True when the ANSI a Layer-1 strip removed was INERT: every removed sequence
20
+ * was either a display-only SGR colour token or a LONE 7-bit `ESC` that opened
21
+ * nothing at all (a stray byte in a file, a truncated write, a log fragment cut
22
+ * mid-escape).
23
+ *
24
+ * The two other orphan kinds are deliberately NOT inert. A raw C1 orphan
25
+ * (TOKEN_KIND.ORPHAN_C1): legit UTF-8 text does not carry raw C1 bytes, and the
26
+ * block holds the DCS/SOS/PM/APC string introducers this grammar does not
27
+ * consume — so an unrecognized one means a terminal would have eaten the
28
+ * following text as a control payload. An incomplete CSI (TOKEN_KIND.ORPHAN_CSI)
29
+ * for the same reason at 7 bits: the CSI parser keeps consuming until a final
30
+ * byte, so `hello ESC[12 world` hides ` w` from the human while the model reads
31
+ * the whole prompt.
32
+ *
33
+ * This draws a severity line, not a presence line: the bytes are stripped
34
+ * either way, so all that rides on the answer is whether the operator sees a
35
+ * WARNING or a terse note. An orphan introducer cannot move the cursor, erase
36
+ * the screen, relabel a window, or open an OSC string — every one of those needs
37
+ * a COMPLETE token, which {@link scanAnsi} classifies as CSI or OSC and this
38
+ * rejects. Warning on a lone `ESC` is the false positive that costs the most: one
39
+ * pre-existing `ESC` in a markdown file, echoed back in an Edit result, raises
40
+ * the same alarm as a cursor-spoofing payload, and an alarm that fires on inert
41
+ * bytes is the one operators learn to scroll past.
42
+ *
43
+ * It takes the kinds the STRIP recorded, never a fresh scan of the raw text,
44
+ * and that is the whole point: a scan of the raw text answers about sequences
45
+ * that have not been reconstituted yet, so `ESC` + `ESC[m` + `[2J` (a bare ESC,
46
+ * an SGR, then plain text) reads as orphan-only there while the strip's second
47
+ * pass actually removes a CSI erase. Recording what each pass removed reports
48
+ * the sequences that really existed at Layer 1's fixed point.
49
+ * @param {readonly string[] | Set<string>} kinds {@link TOKEN_KIND} values removed
50
+ * @returns {boolean}
51
+ */
52
+ export function isBenignAnsiKinds(kinds: readonly string[] | Set<string>): boolean;
53
+ /**
54
+ * {@link isBenignAnsiKinds} for callers that hold only the text — it runs the
55
+ * full Layer-1 composition to get the fixed-point view. Callers that already
56
+ * ran {@link applyLayer1} must read its `ansiKinds` instead of paying for a
57
+ * second strip.
58
+ * @param {string} text
59
+ * @returns {boolean}
60
+ */
61
+ export function isBenignAnsi(text: string): boolean;
17
62
  /**
18
63
  * Layer 1: ANSI + invisible-char strip with a result guaranteed free of every
19
64
  * raw ANSI control introducer (7-bit ESC U+001B and the whole 8-bit C1 control
@@ -38,12 +83,19 @@ export function stripAnsiFully(input: string): string;
38
83
  *
39
84
  * `deAnsi` is the ANSI strip of the ORIGINAL text (invisible runs intact), the
40
85
  * scope a LONG_RUN payload check needs — not an intermediate of the loop.
86
+ *
87
+ * `ansiKinds` is the {@link TOKEN_KIND} of every ANSI sequence the composition
88
+ * removed, deduped — the severity detail `found`'s single ANSI category cannot
89
+ * carry (see {@link isBenignAnsiKinds}). It is also what DERIVES that category:
90
+ * a kind is recorded exactly when bytes were removed, so "we reported ANSI" and
91
+ * "here is what the ANSI was" can no longer disagree.
41
92
  * @param {string} text
42
- * @returns {{ cleaned: string, deAnsi: string, found: string[] }}
93
+ * @returns {{ cleaned: string, deAnsi: string, found: string[], ansiKinds: string[] }}
43
94
  */
44
95
  export function applyLayer1(text: string): {
45
96
  cleaned: string;
46
97
  deAnsi: string;
47
98
  found: string[];
99
+ ansiKinds: string[];
48
100
  };
49
101
  export const LONE_SURROGATE_RE: RegExp;