agent-sanitizer 2.19.3 → 2.19.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THREAT-MODEL.md +18 -6
- package/claude-hooks/lib/hook-fault.mjs +224 -0
- package/claude-hooks/lib/layer-pipeline.mjs +147 -0
- package/claude-hooks/plugin-hooks.mjs +45 -9
- package/claude-hooks/pretooluse-sanitize.mjs +116 -41
- package/claude-hooks/sanitize-output.mjs +66 -19
- package/claude-hooks/sanitize-user-prompt.mjs +31 -23
- package/claude-hooks/scan-invisible-chars.mjs +235 -47
- package/package.json +1 -1
- package/src/ansi.mjs +207 -0
- package/src/confusables.mjs +6 -2
- package/src/html.mjs +230 -64
- package/src/invisible.mjs +202 -159
- package/src/layer1.mjs +101 -116
- package/src/prompt.mjs +9 -6
- package/types/ansi.d.mts +66 -0
- package/types/claude-hooks/lib/hook-fault.d.mts +104 -0
- package/types/claude-hooks/lib/layer-pipeline.d.mts +113 -0
- package/types/claude-hooks/pretooluse-sanitize.d.mts +16 -0
- package/types/claude-hooks/scan-invisible-chars.d.mts +74 -14
- package/types/confusables.d.mts +6 -2
- package/types/html.d.mts +2 -0
- package/types/invisible.d.mts +11 -2
- package/types/layer1.d.mts +19 -10
package/src/ansi.mjs
ADDED
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The ONE ANSI grammar: the raw control-introducer charset and the tokenizer
|
|
3
|
+
* every consumer scans with.
|
|
4
|
+
*
|
|
5
|
+
* Two modules need this grammar and they cannot import each other —
|
|
6
|
+
* `layer1.mjs` imports `invisible.mjs`, so `invisible.mjs` (which owns the
|
|
7
|
+
* public `isSgrOnly` / `SGR_RE`) must not import back. Before this module the
|
|
8
|
+
* grammar was therefore written out twice with DIFFERENT param rules
|
|
9
|
+
* (`invisible.mjs`'s SGR regex accepted any digit run, `layer1.mjs`'s CSI
|
|
10
|
+
* branch capped each parameter at four digits), and the introducer charset
|
|
11
|
+
* three times. The looser copy suppressed the operator warning for a sequence
|
|
12
|
+
* the stripper could not match: `ESC[12345m` read as "display-only colour"
|
|
13
|
+
* while `[12345m` was spliced into the model's view as visible text. One
|
|
14
|
+
* tokenizer, one charset, consumed by both — the disagreement cannot recur.
|
|
15
|
+
*
|
|
16
|
+
* Same precedent (and same reason) as `cf-charset.mjs`: a dependency-free leaf
|
|
17
|
+
* module both layers read from.
|
|
18
|
+
*/
|
|
19
|
+
|
|
20
|
+
// Raw control introducers that must not survive Layer 1: 7-bit ESC (U+001B) and
|
|
21
|
+
// the entire 8-bit C1 control block (U+0080-U+009F) — which includes CSI
|
|
22
|
+
// (U+009B), the string introducers DCS/SOS/OSC/PM/APC
|
|
23
|
+
// (U+0090/0098/009D/009E/009F), and ST (U+009C). Gating the whole C1 block, not
|
|
24
|
+
// just the introducers the sequence grammar below names, fails closed: a
|
|
25
|
+
// DCS/SOS/PM/APC string the grammar does not consume still loses its introducer
|
|
26
|
+
// and terminator, so no terminal can hide-render its body as a control payload.
|
|
27
|
+
//
|
|
28
|
+
// A SOURCE STRING, not a literal: three call sites need it with different flags
|
|
29
|
+
// (`g` for the Layer-1 sweep, unflagged for the prompt gate, `g` again to drive
|
|
30
|
+
// the scan below), and spelling the class out at each site is how the three
|
|
31
|
+
// copies came to spell the same byte two different ways — which defeats a
|
|
32
|
+
// grep-based drift check as well. Building from `\uXXXX` escapes keeps every
|
|
33
|
+
// raw control byte out of the source (no `no-control-regex` disable needed).
|
|
34
|
+
export const CONTROL_INTRODUCER_SOURCE = "[\\u001b\\u0080-\\u009f]";
|
|
35
|
+
|
|
36
|
+
// SGR (Select Graphic Rendition): colors, bold, reset. The grammar is closed:
|
|
37
|
+
// params are [0-9;:]* and the final byte is `m`, so a match can only restyle
|
|
38
|
+
// text — never reposition the cursor, erase, or smuggle an OSC string. `:` is
|
|
39
|
+
// included alongside `;` because ITU T.416 colon-separated SGR sub-parameters
|
|
40
|
+
// (truecolor `ESC[38:2:255:0:0m`, as emitted by tmux/kitty/mintty) are pure
|
|
41
|
+
// display-only SGR too. A SGR sequence has TWO encodings: the 7-bit `ESC [ … m`
|
|
42
|
+
// and the 8-bit C1 form where a single U+009B (CSI) replaces `ESC [`; both must
|
|
43
|
+
// be recognized, or a C1-introduced `U+009B 31m … 0m` is pure color yet is
|
|
44
|
+
// misread as a non-SGR payload.
|
|
45
|
+
const SGR_SOURCE = "(?:\\u001b\\[|\\u009b)[0-9;:]*m";
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Public alias kept for compatibility (re-exported by `invisible.mjs` and the
|
|
49
|
+
* package root). It is now DERIVED: {@link scanAnsi} classifies a token as SGR
|
|
50
|
+
* by testing the token's own text against this exact source, so the predicate
|
|
51
|
+
* and the regex can no longer describe different languages.
|
|
52
|
+
*/
|
|
53
|
+
export const SGR_RE = new RegExp(SGR_SOURCE, "g");
|
|
54
|
+
|
|
55
|
+
// The same language, anchored — the SGR/CSI discriminator for a token the
|
|
56
|
+
// scanner has already delimited.
|
|
57
|
+
const SGR_ANCHORED_RE = new RegExp(`^${SGR_SOURCE}$`);
|
|
58
|
+
|
|
59
|
+
// Private parameter-prefix and intermediate bytes that may follow an
|
|
60
|
+
// introducer before the parameters (`ESC[?25h`, `ESC(B`, `ESC#8`). Also covers
|
|
61
|
+
// the 7-bit `ESC [` CSI introducer's bracket itself.
|
|
62
|
+
const CSI_INTRO_RE = /[[()#;?]/;
|
|
63
|
+
|
|
64
|
+
// ECMA-48 parameter bytes.
|
|
65
|
+
const CSI_PARAM_RE = /[0-9;:]/;
|
|
66
|
+
|
|
67
|
+
// ECMA-48 final bytes, minus the ones a terminal never accepts here. Digits are
|
|
68
|
+
// PARAMETER bytes and can never terminate a sequence — an unterminated `ESC[`
|
|
69
|
+
// must not eat trailing visible digits (`ESC[2024 report` is NOT `ESC[` +
|
|
70
|
+
// final-byte `2`; it is an incomplete intro whose ESC the residual sweep
|
|
71
|
+
// removes, leaving "2024 report" intact). `<=>?` (0x3C-0x3F) are private
|
|
72
|
+
// PARAMETER-prefix bytes per ECMA-48 § 5.4, not finals — including them let a
|
|
73
|
+
// private-marker sequence terminate one byte too early. `~` (0x7E) IS a real
|
|
74
|
+
// final byte (vt220 function keys, `ESC[3~` for Delete) and is kept.
|
|
75
|
+
const CSI_FINAL_RE = /[A-PR-TZcf-nqrty~]/;
|
|
76
|
+
|
|
77
|
+
const ESC = 0x1b;
|
|
78
|
+
const CSI_C1 = 0x9b;
|
|
79
|
+
const ST_C1 = 0x9c;
|
|
80
|
+
const OSC_C1 = 0x9d;
|
|
81
|
+
const BEL = 0x07;
|
|
82
|
+
|
|
83
|
+
/** The four things an introducer can turn out to be. */
|
|
84
|
+
export const TOKEN_KIND = Object.freeze({
|
|
85
|
+
/** A display-only `ESC[…m` / `U+009B…m` colour sequence. */
|
|
86
|
+
SGR: "sgr",
|
|
87
|
+
/** Any other complete CSI / two-byte escape (cursor move, erase, charset). */
|
|
88
|
+
CSI: "csi",
|
|
89
|
+
/** An OSC string: introducer, body and terminator as one unit. */
|
|
90
|
+
OSC: "osc",
|
|
91
|
+
/** An introducer that starts no sequence the grammar recognizes. */
|
|
92
|
+
ORPHAN: "orphan-introducer",
|
|
93
|
+
});
|
|
94
|
+
|
|
95
|
+
/**
|
|
96
|
+
* @typedef {object} AnsiToken
|
|
97
|
+
* @property {number} start Index of the introducer.
|
|
98
|
+
* @property {number} end Index one past the last character of the token.
|
|
99
|
+
* @property {string} kind One of {@link TOKEN_KIND}.
|
|
100
|
+
*/
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* End index of the OSC string starting at `start`, or -1 if no OSC introducer
|
|
104
|
+
* is there.
|
|
105
|
+
*
|
|
106
|
+
* An OSC (Operating System Command) string is `<introducer> body <terminator>`:
|
|
107
|
+
* a title, a clickable-hyperlink URL, a clipboard write — i.e. attacker-
|
|
108
|
+
* controlled PAYLOAD TEXT. Consuming the introducer alone would leave that
|
|
109
|
+
* payload in the model's view, so the whole string is one token. Three ways it
|
|
110
|
+
* can end:
|
|
111
|
+
* 1. a real terminator — ST (`ESC\` or the 8-bit C1 ST U+009C) or the legacy
|
|
112
|
+
* BEL — which is consumed with the body.
|
|
113
|
+
* 2. an ABORT: per ECMA-48/xterm a bare ESC (one not forming ST) drops the
|
|
114
|
+
* terminal out of the OSC string, and a nested C1 OSC introducer likewise
|
|
115
|
+
* starts something new. The token ends BEFORE that byte so the scan
|
|
116
|
+
* re-reads it as its own sequence — without this, an interior ESC deleted
|
|
117
|
+
* the rest of the document via case 3.
|
|
118
|
+
* 3. end of input, for a genuinely unterminated string: fail closed and drop
|
|
119
|
+
* everything from the introducer on, so no OSC body survives.
|
|
120
|
+
* @param {string} text
|
|
121
|
+
* @param {number} start
|
|
122
|
+
* @returns {number}
|
|
123
|
+
*/
|
|
124
|
+
function scanOsc(text, start) {
|
|
125
|
+
const code = text.charCodeAt(start);
|
|
126
|
+
let i = -1;
|
|
127
|
+
if (code === OSC_C1) i = start + 1;
|
|
128
|
+
if (code === ESC && text[start + 1] === "]") i = start + 2;
|
|
129
|
+
if (i < 0) return -1;
|
|
130
|
+
for (; i < text.length; i++) {
|
|
131
|
+
const byte = text.charCodeAt(i);
|
|
132
|
+
if (byte === BEL || byte === ST_C1) return i + 1;
|
|
133
|
+
if (byte === ESC) return text[i + 1] === "\\" ? i + 2 : i;
|
|
134
|
+
if (byte === OSC_C1) return i;
|
|
135
|
+
}
|
|
136
|
+
return text.length;
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* End index of the CSI / two-byte escape sequence starting at `start`, or -1
|
|
141
|
+
* when the introducer completes no sequence.
|
|
142
|
+
*
|
|
143
|
+
* Single-pass and greedy: intro bytes, then parameter bytes, then exactly one
|
|
144
|
+
* final byte. The previous regex form had to BOUND the intro run ({0,12})
|
|
145
|
+
* because `;` lives in both the intro and parameter classes, so an unbounded
|
|
146
|
+
* run let a `;#;#…` string be split between the two quantifiers — quadratic
|
|
147
|
+
* backtracking (CodeQL js/polynomial-redos). A hand-written scanner never
|
|
148
|
+
* backtracks, so the bound is gone and the scan is linear by construction.
|
|
149
|
+
* @param {string} text
|
|
150
|
+
* @param {number} start
|
|
151
|
+
* @returns {number}
|
|
152
|
+
*/
|
|
153
|
+
function scanCsi(text, start) {
|
|
154
|
+
const code = text.charCodeAt(start);
|
|
155
|
+
if (code !== ESC && code !== CSI_C1) return -1;
|
|
156
|
+
let i = start + 1;
|
|
157
|
+
while (i < text.length && CSI_INTRO_RE.test(text[i])) i++;
|
|
158
|
+
while (i < text.length && CSI_PARAM_RE.test(text[i])) i++;
|
|
159
|
+
if (i < text.length && CSI_FINAL_RE.test(text[i])) return i + 1;
|
|
160
|
+
return -1;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
// Drives the scan: jumping introducer-to-introducer keeps the common case (text
|
|
164
|
+
// with no escapes at all) a single native regex scan rather than a per-character
|
|
165
|
+
// JS loop.
|
|
166
|
+
const INTRODUCER_SCAN_RE = new RegExp(CONTROL_INTRODUCER_SOURCE, "g");
|
|
167
|
+
|
|
168
|
+
/**
|
|
169
|
+
* Tokenize every raw control introducer in `text`.
|
|
170
|
+
*
|
|
171
|
+
* Every introducer yields exactly one token — an ORPHAN when it starts nothing
|
|
172
|
+
* the grammar recognizes — so "which introducers are in this text" and "which
|
|
173
|
+
* sequences are in this text" are answered by the same scan. That is what lets
|
|
174
|
+
* the stripper (splice every non-orphan token, then sweep) and the SGR-only
|
|
175
|
+
* predicate (every token is SGR) agree by construction.
|
|
176
|
+
*
|
|
177
|
+
* Tokens are disjoint and ordered by `start`; each `end` is strictly greater
|
|
178
|
+
* than its `start`, so the scan always advances.
|
|
179
|
+
* @param {string} text
|
|
180
|
+
* @returns {AnsiToken[]}
|
|
181
|
+
*/
|
|
182
|
+
export function scanAnsi(text) {
|
|
183
|
+
/** @type {AnsiToken[]} */
|
|
184
|
+
const tokens = [];
|
|
185
|
+
INTRODUCER_SCAN_RE.lastIndex = 0;
|
|
186
|
+
let match;
|
|
187
|
+
while ((match = INTRODUCER_SCAN_RE.exec(text)) !== null) {
|
|
188
|
+
const start = match.index;
|
|
189
|
+
const oscEnd = scanOsc(text, start);
|
|
190
|
+
const csiEnd = oscEnd < 0 ? scanCsi(text, start) : -1;
|
|
191
|
+
let end = start + 1;
|
|
192
|
+
/** @type {string} */
|
|
193
|
+
let kind = TOKEN_KIND.ORPHAN;
|
|
194
|
+
if (oscEnd >= 0) {
|
|
195
|
+
end = oscEnd;
|
|
196
|
+
kind = TOKEN_KIND.OSC;
|
|
197
|
+
} else if (csiEnd >= 0) {
|
|
198
|
+
end = csiEnd;
|
|
199
|
+
kind = SGR_ANCHORED_RE.test(text.slice(start, csiEnd))
|
|
200
|
+
? TOKEN_KIND.SGR
|
|
201
|
+
: TOKEN_KIND.CSI;
|
|
202
|
+
}
|
|
203
|
+
tokens.push({ start, end, kind });
|
|
204
|
+
INTRODUCER_SCAN_RE.lastIndex = end;
|
|
205
|
+
}
|
|
206
|
+
return tokens;
|
|
207
|
+
}
|
package/src/confusables.mjs
CHANGED
|
@@ -38,8 +38,12 @@
|
|
|
38
38
|
*
|
|
39
39
|
* ORDERING: the soundness argument assumes no later layer erases code points
|
|
40
40
|
* from the same field, which would let an unmapped glyph the gate relied on
|
|
41
|
-
* disappear after the decision
|
|
42
|
-
*
|
|
41
|
+
* disappear after the decision — a zero-width run padded into a token suppresses
|
|
42
|
+
* the fold, and the erasing layer then removes the very evidence for skipping it.
|
|
43
|
+
* This fold does NOT run last: on Bash.command the invisible-char strip follows
|
|
44
|
+
* it. A caller that composes the two is therefore responsible for re-running
|
|
45
|
+
* this fold on the post-erasure text until it reports nothing, which is what the
|
|
46
|
+
* hook driver in claude-hooks/lib/layer-pipeline.mjs does.
|
|
43
47
|
*
|
|
44
48
|
* Genuine non-confusable non-ASCII (accented Latin, CJK, emoji) is untouched
|
|
45
49
|
* regardless, since a faithful scanner does not flag it.
|