agent-sanitizer 2.19.3 → 2.19.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/ansi.mjs ADDED
@@ -0,0 +1,207 @@
1
+ /**
2
+ * The ONE ANSI grammar: the raw control-introducer charset and the tokenizer
3
+ * every consumer scans with.
4
+ *
5
+ * Two modules need this grammar and they cannot import each other —
6
+ * `layer1.mjs` imports `invisible.mjs`, so `invisible.mjs` (which owns the
7
+ * public `isSgrOnly` / `SGR_RE`) must not import back. Before this module the
8
+ * grammar was therefore written out twice with DIFFERENT param rules
9
+ * (`invisible.mjs`'s SGR regex accepted any digit run, `layer1.mjs`'s CSI
10
+ * branch capped each parameter at four digits), and the introducer charset
11
+ * three times. The looser copy suppressed the operator warning for a sequence
12
+ * the stripper could not match: `ESC[12345m` read as "display-only colour"
13
+ * while `[12345m` was spliced into the model's view as visible text. One
14
+ * tokenizer, one charset, consumed by both — the disagreement cannot recur.
15
+ *
16
+ * Same precedent (and same reason) as `cf-charset.mjs`: a dependency-free leaf
17
+ * module both layers read from.
18
+ */
19
+
20
+ // Raw control introducers that must not survive Layer 1: 7-bit ESC (U+001B) and
21
+ // the entire 8-bit C1 control block (U+0080-U+009F) — which includes CSI
22
+ // (U+009B), the string introducers DCS/SOS/OSC/PM/APC
23
+ // (U+0090/0098/009D/009E/009F), and ST (U+009C). Gating the whole C1 block, not
24
+ // just the introducers the sequence grammar below names, fails closed: a
25
+ // DCS/SOS/PM/APC string the grammar does not consume still loses its introducer
26
+ // and terminator, so no terminal can hide-render its body as a control payload.
27
+ //
28
+ // A SOURCE STRING, not a literal: three call sites need it with different flags
29
+ // (`g` for the Layer-1 sweep, unflagged for the prompt gate, `g` again to drive
30
+ // the scan below), and spelling the class out at each site is how the three
31
+ // copies came to spell the same byte two different ways — which defeats a
32
+ // grep-based drift check as well. Building from `\uXXXX` escapes keeps every
33
+ // raw control byte out of the source (no `no-control-regex` disable needed).
34
+ export const CONTROL_INTRODUCER_SOURCE = "[\\u001b\\u0080-\\u009f]";
35
+
36
+ // SGR (Select Graphic Rendition): colors, bold, reset. The grammar is closed:
37
+ // params are [0-9;:]* and the final byte is `m`, so a match can only restyle
38
+ // text — never reposition the cursor, erase, or smuggle an OSC string. `:` is
39
+ // included alongside `;` because ITU T.416 colon-separated SGR sub-parameters
40
+ // (truecolor `ESC[38:2:255:0:0m`, as emitted by tmux/kitty/mintty) are pure
41
+ // display-only SGR too. A SGR sequence has TWO encodings: the 7-bit `ESC [ … m`
42
+ // and the 8-bit C1 form where a single U+009B (CSI) replaces `ESC [`; both must
43
+ // be recognized, or a C1-introduced `U+009B 31m … 0m` is pure color yet is
44
+ // misread as a non-SGR payload.
45
+ const SGR_SOURCE = "(?:\\u001b\\[|\\u009b)[0-9;:]*m";
46
+
47
+ /**
48
+ * Public alias kept for compatibility (re-exported by `invisible.mjs` and the
49
+ * package root). It is now DERIVED: {@link scanAnsi} classifies a token as SGR
50
+ * by testing the token's own text against this exact source, so the predicate
51
+ * and the regex can no longer describe different languages.
52
+ */
53
+ export const SGR_RE = new RegExp(SGR_SOURCE, "g");
54
+
55
+ // The same language, anchored — the SGR/CSI discriminator for a token the
56
+ // scanner has already delimited.
57
+ const SGR_ANCHORED_RE = new RegExp(`^${SGR_SOURCE}$`);
58
+
59
+ // Private parameter-prefix and intermediate bytes that may follow an
60
+ // introducer before the parameters (`ESC[?25h`, `ESC(B`, `ESC#8`). Also covers
61
+ // the 7-bit `ESC [` CSI introducer's bracket itself.
62
+ const CSI_INTRO_RE = /[[()#;?]/;
63
+
64
+ // ECMA-48 parameter bytes.
65
+ const CSI_PARAM_RE = /[0-9;:]/;
66
+
67
+ // ECMA-48 final bytes, minus the ones a terminal never accepts here. Digits are
68
+ // PARAMETER bytes and can never terminate a sequence — an unterminated `ESC[`
69
+ // must not eat trailing visible digits (`ESC[2024 report` is NOT `ESC[` +
70
+ // final-byte `2`; it is an incomplete intro whose ESC the residual sweep
71
+ // removes, leaving "2024 report" intact). `<=>?` (0x3C-0x3F) are private
72
+ // PARAMETER-prefix bytes per ECMA-48 § 5.4, not finals — including them let a
73
+ // private-marker sequence terminate one byte too early. `~` (0x7E) IS a real
74
+ // final byte (vt220 function keys, `ESC[3~` for Delete) and is kept.
75
+ const CSI_FINAL_RE = /[A-PR-TZcf-nqrty~]/;
76
+
77
+ const ESC = 0x1b;
78
+ const CSI_C1 = 0x9b;
79
+ const ST_C1 = 0x9c;
80
+ const OSC_C1 = 0x9d;
81
+ const BEL = 0x07;
82
+
83
+ /** The four things an introducer can turn out to be. */
84
+ export const TOKEN_KIND = Object.freeze({
85
+ /** A display-only `ESC[…m` / `U+009B…m` colour sequence. */
86
+ SGR: "sgr",
87
+ /** Any other complete CSI / two-byte escape (cursor move, erase, charset). */
88
+ CSI: "csi",
89
+ /** An OSC string: introducer, body and terminator as one unit. */
90
+ OSC: "osc",
91
+ /** An introducer that starts no sequence the grammar recognizes. */
92
+ ORPHAN: "orphan-introducer",
93
+ });
94
+
95
+ /**
96
+ * @typedef {object} AnsiToken
97
+ * @property {number} start Index of the introducer.
98
+ * @property {number} end Index one past the last character of the token.
99
+ * @property {string} kind One of {@link TOKEN_KIND}.
100
+ */
101
+
102
+ /**
103
+ * End index of the OSC string starting at `start`, or -1 if no OSC introducer
104
+ * is there.
105
+ *
106
+ * An OSC (Operating System Command) string is `<introducer> body <terminator>`:
107
+ * a title, a clickable-hyperlink URL, a clipboard write — i.e. attacker-
108
+ * controlled PAYLOAD TEXT. Consuming the introducer alone would leave that
109
+ * payload in the model's view, so the whole string is one token. Three ways it
110
+ * can end:
111
+ * 1. a real terminator — ST (`ESC\` or the 8-bit C1 ST U+009C) or the legacy
112
+ * BEL — which is consumed with the body.
113
+ * 2. an ABORT: per ECMA-48/xterm a bare ESC (one not forming ST) drops the
114
+ * terminal out of the OSC string, and a nested C1 OSC introducer likewise
115
+ * starts something new. The token ends BEFORE that byte so the scan
116
+ * re-reads it as its own sequence — without this, an interior ESC deleted
117
+ * the rest of the document via case 3.
118
+ * 3. end of input, for a genuinely unterminated string: fail closed and drop
119
+ * everything from the introducer on, so no OSC body survives.
120
+ * @param {string} text
121
+ * @param {number} start
122
+ * @returns {number}
123
+ */
124
+ function scanOsc(text, start) {
125
+ const code = text.charCodeAt(start);
126
+ let i = -1;
127
+ if (code === OSC_C1) i = start + 1;
128
+ if (code === ESC && text[start + 1] === "]") i = start + 2;
129
+ if (i < 0) return -1;
130
+ for (; i < text.length; i++) {
131
+ const byte = text.charCodeAt(i);
132
+ if (byte === BEL || byte === ST_C1) return i + 1;
133
+ if (byte === ESC) return text[i + 1] === "\\" ? i + 2 : i;
134
+ if (byte === OSC_C1) return i;
135
+ }
136
+ return text.length;
137
+ }
138
+
139
+ /**
140
+ * End index of the CSI / two-byte escape sequence starting at `start`, or -1
141
+ * when the introducer completes no sequence.
142
+ *
143
+ * Single-pass and greedy: intro bytes, then parameter bytes, then exactly one
144
+ * final byte. The previous regex form had to BOUND the intro run ({0,12})
145
+ * because `;` lives in both the intro and parameter classes, so an unbounded
146
+ * run let a `;#;#…` string be split between the two quantifiers — quadratic
147
+ * backtracking (CodeQL js/polynomial-redos). A hand-written scanner never
148
+ * backtracks, so the bound is gone and the scan is linear by construction.
149
+ * @param {string} text
150
+ * @param {number} start
151
+ * @returns {number}
152
+ */
153
+ function scanCsi(text, start) {
154
+ const code = text.charCodeAt(start);
155
+ if (code !== ESC && code !== CSI_C1) return -1;
156
+ let i = start + 1;
157
+ while (i < text.length && CSI_INTRO_RE.test(text[i])) i++;
158
+ while (i < text.length && CSI_PARAM_RE.test(text[i])) i++;
159
+ if (i < text.length && CSI_FINAL_RE.test(text[i])) return i + 1;
160
+ return -1;
161
+ }
162
+
163
+ // Drives the scan: jumping introducer-to-introducer keeps the common case (text
164
+ // with no escapes at all) a single native regex scan rather than a per-character
165
+ // JS loop.
166
+ const INTRODUCER_SCAN_RE = new RegExp(CONTROL_INTRODUCER_SOURCE, "g");
167
+
168
+ /**
169
+ * Tokenize every raw control introducer in `text`.
170
+ *
171
+ * Every introducer yields exactly one token — an ORPHAN when it starts nothing
172
+ * the grammar recognizes — so "which introducers are in this text" and "which
173
+ * sequences are in this text" are answered by the same scan. That is what lets
174
+ * the stripper (splice every non-orphan token, then sweep) and the SGR-only
175
+ * predicate (every token is SGR) agree by construction.
176
+ *
177
+ * Tokens are disjoint and ordered by `start`; each `end` is strictly greater
178
+ * than its `start`, so the scan always advances.
179
+ * @param {string} text
180
+ * @returns {AnsiToken[]}
181
+ */
182
+ export function scanAnsi(text) {
183
+ /** @type {AnsiToken[]} */
184
+ const tokens = [];
185
+ INTRODUCER_SCAN_RE.lastIndex = 0;
186
+ let match;
187
+ while ((match = INTRODUCER_SCAN_RE.exec(text)) !== null) {
188
+ const start = match.index;
189
+ const oscEnd = scanOsc(text, start);
190
+ const csiEnd = oscEnd < 0 ? scanCsi(text, start) : -1;
191
+ let end = start + 1;
192
+ /** @type {string} */
193
+ let kind = TOKEN_KIND.ORPHAN;
194
+ if (oscEnd >= 0) {
195
+ end = oscEnd;
196
+ kind = TOKEN_KIND.OSC;
197
+ } else if (csiEnd >= 0) {
198
+ end = csiEnd;
199
+ kind = SGR_ANCHORED_RE.test(text.slice(start, csiEnd))
200
+ ? TOKEN_KIND.SGR
201
+ : TOKEN_KIND.CSI;
202
+ }
203
+ tokens.push({ start, end, kind });
204
+ INTRODUCER_SCAN_RE.lastIndex = end;
205
+ }
206
+ return tokens;
207
+ }
@@ -38,8 +38,12 @@
38
38
  *
39
39
  * ORDERING: the soundness argument assumes no later layer erases code points
40
40
  * from the same field, which would let an unmapped glyph the gate relied on
41
- * disappear after the decision. Layer 4 runs before sanitizeAuthoredContent on
42
- * Bash.command; keep it there.
41
+ * disappear after the decision — a zero-width run padded into a token suppresses
42
+ * the fold, and the erasing layer then removes the very evidence for skipping it.
43
+ * This fold does NOT run last: on Bash.command the invisible-char strip follows
44
+ * it. A caller that composes the two is therefore responsible for re-running
45
+ * this fold on the post-erasure text until it reports nothing, which is what the
46
+ * hook driver in claude-hooks/lib/layer-pipeline.mjs does.
43
47
  *
44
48
  * Genuine non-confusable non-ASCII (accented Latin, CJK, emoji) is untouched
45
49
  * regardless, since a faithful scanner does not flag it.
package/src/html.mjs CHANGED
@@ -57,6 +57,13 @@ export {
57
57
  // hand-rolled parser kept re-introducing simply cannot arise. Ambiguity still
58
58
  // fails OPEN (treated as visible): an unresolved unit, `calc()`, or `var()`
59
59
  // never counts as hidden.
60
+ //
61
+ // Every ident in a value — keyword, function name, dimension unit — is
62
+ // escape-decoded and lowercased ONCE at the parse boundary (see
63
+ // {@link canonicalizeValue}), so the detectors below compare canonical tokens
64
+ // against literals. Doing it per-site is what let `left:-9999PX` through: CSS
65
+ // units are ASCII case-insensitive, and every site that forgot `.toLowerCase()`
66
+ // was a one-keystroke bypass of the whole layer.
60
67
 
61
68
  // A length/opacity/size is "near zero" when its magnitude is below this — a
62
69
  // browser renders 0.0001px text or 0.001 opacity as effectively invisible, so
@@ -206,7 +213,9 @@ function isHidingTransform(node) {
206
213
  if (!node) return false;
207
214
  for (const fn of valueTokens(node)) {
208
215
  if (fn.type !== "Function") continue;
209
- const name = fn.name.toLowerCase();
216
+ // Function names, units and idents arrive lowercased and escape-decoded
217
+ // from parseDeclarations, so a literal compare is correct by construction.
218
+ const name = fn.name;
210
219
  const args = valueTokens(fn);
211
220
  if (/^(?:scale|scale3d|scalex|scaley|matrix|matrix3d)$/.test(name)) {
212
221
  // scale/matrix collapse to nothing when EITHER axis factor is (near-)zero —
@@ -240,12 +249,8 @@ function isHidingTransform(node) {
240
249
  // normalizes deg/grad/rad/turn to [0,360), and a near-90/270 band absorbs
241
250
  // the float drift of rad→deg.
242
251
  const a = args[0];
243
- if (
244
- a &&
245
- a.type === "Dimension" &&
246
- ANGLE_UNITS.has(a.unit.toLowerCase())
247
- ) {
248
- const degrees = hueDegrees(`${a.value}${a.unit}`.toLowerCase());
252
+ if (a && a.type === "Dimension" && ANGLE_UNITS.has(a.unit)) {
253
+ const degrees = hueDegrees(`${a.value}${a.unit}`);
249
254
  if (
250
255
  degrees !== null &&
251
256
  (Math.abs(degrees - 90) < NEAR_ZERO_EPSILON ||
@@ -279,7 +284,7 @@ function isHidingTransform(node) {
279
284
  function isHidingFilter(node) {
280
285
  if (!node) return false;
281
286
  for (const fn of valueTokens(node)) {
282
- if (fn.type !== "Function" || fn.name.toLowerCase() !== "opacity") continue;
287
+ if (fn.type !== "Function" || fn.name !== "opacity") continue;
283
288
  const amount = valueTokens(fn)[0];
284
289
  if (!amount) continue;
285
290
  if (
@@ -321,7 +326,7 @@ function clipEdge(token) {
321
326
  function isClipRectHidden(node) {
322
327
  if (!node) return false;
323
328
  const rect = valueTokens(node).find(
324
- (t) => t.type === "Function" && t.name.toLowerCase() === "rect",
329
+ (t) => t.type === "Function" && t.name === "rect",
325
330
  );
326
331
  if (!rect) return false;
327
332
  const edges = valueTokens(rect).map(clipEdge);
@@ -686,18 +691,50 @@ function canonicalizeColor(raw) {
686
691
  return canonicalizeColorFunction(value) ?? value;
687
692
  }
688
693
 
694
+ // True when a value node paints — or cannot be proven NOT to paint — a
695
+ // background IMAGE layer: a `url()`, a `*-gradient()` or an `image-set()`
696
+ // anywhere in the value. The value AST is walked rather than a re-serialized
697
+ // string regexed: an escaped function name (`\49 mage-set(…)`, which a browser
698
+ // reads as `image-set(…)`) survives serialization escaped and slipped past the
699
+ // regex, so the image layer went unseen and same-colored text painted over an
700
+ // image was spliced as hidden. A `Raw` node — a value css-tree could not parse,
701
+ // which is also what an escaped `url(` degrades to — is unresolvable and so
702
+ // counts as an image layer (fail OPEN: no same-color hide).
703
+ /** @param {any} node value node, or null @returns {boolean} */
704
+ function paintsImageLayer(node) {
705
+ if (!node) return false;
706
+ let found = false;
707
+ csstree.walk(node, {
708
+ enter(/** @type {any} */ child) {
709
+ if (child.type === "Url" || child.type === "Raw") found = true;
710
+ // Names are canonicalized at the parse boundary, so a suffix test covers
711
+ // every gradient (`linear-`/`radial-`/`conic-`/`repeating-`/`-webkit-`)
712
+ // and both `image-set` spellings. `url` appears as a Function (not a Url)
713
+ // node when its name carried an escape — the very case the old regex on
714
+ // the re-serialized text missed.
715
+ if (
716
+ child.type === "Function" &&
717
+ (child.name === "url" ||
718
+ child.name.endsWith("gradient") ||
719
+ child.name.endsWith("image-set"))
720
+ )
721
+ found = true;
722
+ },
723
+ });
724
+ return found;
725
+ }
726
+
689
727
  // The leading color token of a `background` shorthand (the first token that
690
728
  // canonicalizes to a real color), so `background:#fff` still compares. Returns
691
- // "" (fail open, no same-color hide) when the shorthand carries an IMAGE layer
692
- // — `url(...)`, a gradient, or `image-set(...)`: the painted image can make
693
- // same-colored text perfectly readable over it (and if it fails to load the
694
- // element's own background shows through), so the flat color token is not
695
- // provably the rendered backdrop.
696
- /** @param {string} shorthand @returns {string} */
697
- function backgroundColor(shorthand) {
698
- if (/\burl\(|gradient\(|image-set\(/i.test(shorthand)) return "";
699
- for (const token of shorthand.split(/\s+/)) {
700
- const color = canonicalizeColor(token);
729
+ // "" (fail open, no same-color hide) when the shorthand carries an IMAGE layer:
730
+ // the painted image can make same-colored text perfectly readable over it (and
731
+ // if it fails to load the element's own background shows through), so the flat
732
+ // color token is not provably the rendered backdrop.
733
+ /** @param {any} node value node for `background`, or null @returns {string} */
734
+ function backgroundColor(node) {
735
+ if (!node || paintsImageLayer(node)) return "";
736
+ for (const token of valueTokens(node)) {
737
+ const color = canonicalizeColor(tokenText(token));
701
738
  if (color && (color.startsWith("#") || color === "transparent"))
702
739
  return color;
703
740
  }
@@ -727,8 +764,7 @@ function insetEdges(fn) {
727
764
  /** @type {any[]} */
728
765
  const edges = [];
729
766
  for (const token of valueTokens(fn)) {
730
- if (token.type === "Identifier" && token.name.toLowerCase() === "round")
731
- break;
767
+ if (token.type === "Identifier" && token.name === "round") break;
732
768
  edges.push(token);
733
769
  }
734
770
  return edges;
@@ -748,7 +784,7 @@ function isClipPathHidden(node) {
748
784
  if (!node) return false;
749
785
  for (const fn of valueTokens(node)) {
750
786
  if (fn.type !== "Function") continue;
751
- const name = fn.name.toLowerCase();
787
+ const name = fn.name;
752
788
  if (name === "circle") {
753
789
  const radius = valueTokens(fn)[0];
754
790
  if (
@@ -808,6 +844,13 @@ function isTextPaintedVisible(val) {
808
844
  return isConcreteColor(stroke) && stroke !== "transparent";
809
845
  }
810
846
 
847
+ /** True when a value is exactly the keyword `none`.
848
+ * @param {any} node @returns {boolean} */
849
+ function isNoneKeyword(node) {
850
+ const token = soleToken(node);
851
+ return Boolean(token && token.type === "Identifier" && token.name === "none");
852
+ }
853
+
811
854
  // True when the element paints a background IMAGE layer — a `background-image`
812
855
  // longhand set to anything but `none`, or a `background` shorthand carrying
813
856
  // `url(...)`, a gradient, or `image-set(...)`. A same-color text/background hide
@@ -819,11 +862,13 @@ function isTextPaintedVisible(val) {
819
862
  // only the flat color and missed a co-declared `background-image`, splicing
820
863
  // visible text. `background-clip:text` is NOT an image layer here — it paints
821
864
  // the background THROUGH the glyphs and is handled by {@link isTextPaintedVisible}.
822
- /** @param {(key: string) => string} textOf @returns {boolean} */
823
- function hasImageLayer(textOf) {
824
- const img = textOf("background-image");
825
- if (img && img !== "none") return true;
826
- return /\burl\(|gradient\(|image-set\(/i.test(textOf("background"));
865
+ /** @param {(key: string) => any} nodeOf @returns {boolean} */
866
+ function hasImageLayer(nodeOf) {
867
+ const img = nodeOf("background-image");
868
+ // Only the single keyword `none` proves the longhand paints nothing; any
869
+ // other value (an unresolvable `var()` included) counts as a layer.
870
+ if (img && !isNoneKeyword(img)) return true;
871
+ return paintsImageLayer(nodeOf("background"));
827
872
  }
828
873
 
829
874
  /**
@@ -871,12 +916,23 @@ function isFontShorthandHidden(node) {
871
916
  const CSS_PROPERTY_IDENT_RE = /^-{0,2}[A-Za-z_][A-Za-z0-9_-]*$/;
872
917
 
873
918
  /**
874
- * Reconstruct a declaration's decoded value as a string for keyword/color
875
- * comparisons. Identifier tokens are escape-decoded through css-tree's ident
876
- * decoder (so `no\6e e`/`hi\64 den` read as `none`/`hidden`, with FF/CR/CRLF
877
- * terminators and invalid codepoints handled per the CSS spec); every other
878
- * token is re-serialized. A whole-value `Raw` (an unparsed value) is returned
879
- * verbatim — it never matches a hiding keyword, so it fails open.
919
+ * One value token as text. An Identifier's `name` is already the decoded,
920
+ * lowercased ident (see {@link canonicalizeValue}), so it is used verbatim
921
+ * rather than round-tripped through the serializer; every other token is
922
+ * re-serialized from its (canonicalized) node fields.
923
+ * @param {any} token
924
+ * @returns {string}
925
+ */
926
+ function tokenText(token) {
927
+ return token.type === "Identifier" ? token.name : csstree.generate(token);
928
+ }
929
+
930
+ /**
931
+ * Reconstruct a declaration's canonicalized value as a string for keyword/color
932
+ * comparisons. Escapes and letter case were resolved at the parse boundary, so
933
+ * `no\6e e`, `NONE` and `none` all render as `none` here. A whole-value `Raw`
934
+ * (an unparsed value) is returned verbatim — it never matches a hiding keyword,
935
+ * so it fails open.
880
936
  * @param {any} valueNode
881
937
  * @returns {string}
882
938
  */
@@ -887,25 +943,50 @@ function declText(valueNode) {
887
943
  const parts = [];
888
944
  if (valueNode.children)
889
945
  valueNode.children.forEach((/** @type {any} */ child) =>
890
- parts.push(
891
- child.type === "Identifier"
892
- ? csstree.ident.decode(child.name)
893
- : csstree.generate(child),
894
- ),
946
+ parts.push(tokenText(child)),
895
947
  );
896
948
  return parts.join(" ");
897
949
  }
898
950
 
951
+ /**
952
+ * Canonicalize a parsed value subtree IN PLACE so every downstream comparison
953
+ * against a literal (`ABSOLUTE_UNITS`, `"rect"`, `"none"`) is correct by
954
+ * construction instead of depending on each call site remembering to decode and
955
+ * lowercase. CSS idents — keywords, function names and dimension units — are
956
+ * escape-decodable and ASCII case-insensitive for every keyword this module
957
+ * matches, so a browser reads `left:-9999PX`, `left:-9999p\78` and
958
+ * `left:-9999px` identically; the ad-hoc per-site `.toLowerCase()` did not, and
959
+ * a single uppercased unit walked straight past the detector.
960
+ *
961
+ * `Url.value` is deliberately NOT touched: css-tree already decodes url escapes,
962
+ * and a second decode would eat a legitimately backslash-bearing path.
963
+ * @param {any} valueNode
964
+ * @returns {void}
965
+ */
966
+ function canonicalizeValue(valueNode) {
967
+ csstree.walk(valueNode, {
968
+ enter(/** @type {any} */ node) {
969
+ // ident.decode is pure string iteration and cannot throw on a token the
970
+ // tokenizer already produced.
971
+ if (node.type === "Identifier" || node.type === "Function")
972
+ node.name = csstree.ident.decode(node.name).toLowerCase();
973
+ else if (node.type === "Dimension")
974
+ node.unit = csstree.ident.decode(node.unit).toLowerCase();
975
+ },
976
+ });
977
+ }
978
+
899
979
  /**
900
980
  * Parse a style string into a map of decoded lowercase property name -> parsed
901
- * value node, via css-tree's tolerant declaration-list parser. This replaces the
902
- * hand-rolled declaration splitter, per-declaration salvage, escape decoder, and
903
- * `!important` stripper in one pass: css-tree recovers per-declaration exactly
904
- * as a browser does (a bogus declaration is dropped, the rest kept), keeps a `;`
905
- * inside a string/`url()`/paren as part of the value, and exposes `!important`
906
- * as `node.important` (so an escaped spelling `none!\69mportant` is stripped for
907
- * free). Property names are escape-decoded and gated to real CSS idents;
908
- * anything else is dropped (fail open). Later declarations win, per the cascade.
981
+ * and canonicalized value node, via css-tree's tolerant declaration-list
982
+ * parser. This replaces the hand-rolled declaration splitter, per-declaration
983
+ * salvage, escape decoder, and `!important` stripper in one pass: css-tree
984
+ * recovers per-declaration exactly as a browser does (a bogus declaration is
985
+ * dropped, the rest kept), keeps a `;` inside a string/`url()`/paren as part of
986
+ * the value, and exposes `!important` as `node.important` (so an escaped
987
+ * spelling `none!\69mportant` is stripped for free). Property names are
988
+ * escape-decoded and gated to real CSS idents; anything else is dropped (fail
989
+ * open). Later declarations win, per the cascade.
909
990
  * @param {string} styleStr
910
991
  * @returns {Map<string, any>}
911
992
  */
@@ -934,6 +1015,7 @@ function parseDeclarations(styleStr) {
934
1015
  // token; property is escape-decoded then gated to a clean CSS ident.
935
1016
  const property = csstree.ident.decode(node.property).trim().toLowerCase();
936
1017
  if (!CSS_PROPERTY_IDENT_RE.test(property)) return;
1018
+ canonicalizeValue(node.value);
937
1019
  decls.set(property, node.value);
938
1020
  },
939
1021
  });
@@ -1015,7 +1097,7 @@ export function isHiddenStyle(styleStr) {
1015
1097
  return true;
1016
1098
  const background =
1017
1099
  canonicalizeColor(textOf("background-color")) ||
1018
- backgroundColor(textOf("background"));
1100
+ backgroundColor(nodeOf("background"));
1019
1101
  // Only flag same-color when BOTH sides resolve to a concrete color (`#rrggbb`
1020
1102
  // or `transparent`), AND no background IMAGE layer is present (an image can
1021
1103
  // make same-colored text readable). `var(--x)`, `inherit`, and `currentColor`
@@ -1027,7 +1109,7 @@ export function isHiddenStyle(styleStr) {
1027
1109
  effectiveColor &&
1028
1110
  effectiveColor === background &&
1029
1111
  isConcreteColor(effectiveColor) &&
1030
- !hasImageLayer(textOf)
1112
+ !hasImageLayer(nodeOf)
1031
1113
  )
1032
1114
  return true;
1033
1115