agent-sanitizer 2.19.3 → 2.19.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/THREAT-MODEL.md +18 -6
- package/claude-hooks/lib/hook-fault.mjs +224 -0
- package/claude-hooks/lib/layer-pipeline.mjs +147 -0
- package/claude-hooks/plugin-hooks.mjs +45 -9
- package/claude-hooks/pretooluse-sanitize.mjs +116 -41
- package/claude-hooks/sanitize-output.mjs +66 -19
- package/claude-hooks/sanitize-user-prompt.mjs +31 -23
- package/claude-hooks/scan-invisible-chars.mjs +235 -47
- package/package.json +1 -1
- package/src/ansi.mjs +207 -0
- package/src/confusables.mjs +6 -2
- package/src/html.mjs +230 -64
- package/src/invisible.mjs +202 -159
- package/src/layer1.mjs +101 -116
- package/src/prompt.mjs +9 -6
- package/types/ansi.d.mts +66 -0
- package/types/claude-hooks/lib/hook-fault.d.mts +104 -0
- package/types/claude-hooks/lib/layer-pipeline.d.mts +113 -0
- package/types/claude-hooks/pretooluse-sanitize.d.mts +16 -0
- package/types/claude-hooks/scan-invisible-chars.d.mts +74 -14
- package/types/confusables.d.mts +6 -2
- package/types/html.d.mts +2 -0
- package/types/invisible.d.mts +11 -2
- package/types/layer1.d.mts +19 -10
package/src/html.mjs
CHANGED
|
@@ -57,6 +57,13 @@ export {
|
|
|
57
57
|
// hand-rolled parser kept re-introducing simply cannot arise. Ambiguity still
|
|
58
58
|
// fails OPEN (treated as visible): an unresolved unit, `calc()`, or `var()`
|
|
59
59
|
// never counts as hidden.
|
|
60
|
+
//
|
|
61
|
+
// Every ident in a value — keyword, function name, dimension unit — is
|
|
62
|
+
// escape-decoded and lowercased ONCE at the parse boundary (see
|
|
63
|
+
// {@link canonicalizeValue}), so the detectors below compare canonical tokens
|
|
64
|
+
// against literals. Doing it per-site is what let `left:-9999PX` through: CSS
|
|
65
|
+
// units are ASCII case-insensitive, and every site that forgot `.toLowerCase()`
|
|
66
|
+
// was a one-keystroke bypass of the whole layer.
|
|
60
67
|
|
|
61
68
|
// A length/opacity/size is "near zero" when its magnitude is below this — a
|
|
62
69
|
// browser renders 0.0001px text or 0.001 opacity as effectively invisible, so
|
|
@@ -206,7 +213,9 @@ function isHidingTransform(node) {
|
|
|
206
213
|
if (!node) return false;
|
|
207
214
|
for (const fn of valueTokens(node)) {
|
|
208
215
|
if (fn.type !== "Function") continue;
|
|
209
|
-
|
|
216
|
+
// Function names, units and idents arrive lowercased and escape-decoded
|
|
217
|
+
// from parseDeclarations, so a literal compare is correct by construction.
|
|
218
|
+
const name = fn.name;
|
|
210
219
|
const args = valueTokens(fn);
|
|
211
220
|
if (/^(?:scale|scale3d|scalex|scaley|matrix|matrix3d)$/.test(name)) {
|
|
212
221
|
// scale/matrix collapse to nothing when EITHER axis factor is (near-)zero —
|
|
@@ -240,12 +249,8 @@ function isHidingTransform(node) {
|
|
|
240
249
|
// normalizes deg/grad/rad/turn to [0,360), and a near-90/270 band absorbs
|
|
241
250
|
// the float drift of rad→deg.
|
|
242
251
|
const a = args[0];
|
|
243
|
-
if (
|
|
244
|
-
a
|
|
245
|
-
a.type === "Dimension" &&
|
|
246
|
-
ANGLE_UNITS.has(a.unit.toLowerCase())
|
|
247
|
-
) {
|
|
248
|
-
const degrees = hueDegrees(`${a.value}${a.unit}`.toLowerCase());
|
|
252
|
+
if (a && a.type === "Dimension" && ANGLE_UNITS.has(a.unit)) {
|
|
253
|
+
const degrees = hueDegrees(`${a.value}${a.unit}`);
|
|
249
254
|
if (
|
|
250
255
|
degrees !== null &&
|
|
251
256
|
(Math.abs(degrees - 90) < NEAR_ZERO_EPSILON ||
|
|
@@ -279,7 +284,7 @@ function isHidingTransform(node) {
|
|
|
279
284
|
function isHidingFilter(node) {
|
|
280
285
|
if (!node) return false;
|
|
281
286
|
for (const fn of valueTokens(node)) {
|
|
282
|
-
if (fn.type !== "Function" || fn.name
|
|
287
|
+
if (fn.type !== "Function" || fn.name !== "opacity") continue;
|
|
283
288
|
const amount = valueTokens(fn)[0];
|
|
284
289
|
if (!amount) continue;
|
|
285
290
|
if (
|
|
@@ -321,7 +326,7 @@ function clipEdge(token) {
|
|
|
321
326
|
function isClipRectHidden(node) {
|
|
322
327
|
if (!node) return false;
|
|
323
328
|
const rect = valueTokens(node).find(
|
|
324
|
-
(t) => t.type === "Function" && t.name
|
|
329
|
+
(t) => t.type === "Function" && t.name === "rect",
|
|
325
330
|
);
|
|
326
331
|
if (!rect) return false;
|
|
327
332
|
const edges = valueTokens(rect).map(clipEdge);
|
|
@@ -686,18 +691,50 @@ function canonicalizeColor(raw) {
|
|
|
686
691
|
return canonicalizeColorFunction(value) ?? value;
|
|
687
692
|
}
|
|
688
693
|
|
|
694
|
+
// True when a value node paints — or cannot be proven NOT to paint — a
|
|
695
|
+
// background IMAGE layer: a `url()`, a `*-gradient()` or an `image-set()`
|
|
696
|
+
// anywhere in the value. The value AST is walked rather than a re-serialized
|
|
697
|
+
// string regexed: an escaped function name (`\49 mage-set(…)`, which a browser
|
|
698
|
+
// reads as `image-set(…)`) survives serialization escaped and slipped past the
|
|
699
|
+
// regex, so the image layer went unseen and same-colored text painted over an
|
|
700
|
+
// image was spliced as hidden. A `Raw` node — a value css-tree could not parse,
|
|
701
|
+
// which is also what an escaped `url(` degrades to — is unresolvable and so
|
|
702
|
+
// counts as an image layer (fail OPEN: no same-color hide).
|
|
703
|
+
/** @param {any} node value node, or null @returns {boolean} */
|
|
704
|
+
function paintsImageLayer(node) {
|
|
705
|
+
if (!node) return false;
|
|
706
|
+
let found = false;
|
|
707
|
+
csstree.walk(node, {
|
|
708
|
+
enter(/** @type {any} */ child) {
|
|
709
|
+
if (child.type === "Url" || child.type === "Raw") found = true;
|
|
710
|
+
// Names are canonicalized at the parse boundary, so a suffix test covers
|
|
711
|
+
// every gradient (`linear-`/`radial-`/`conic-`/`repeating-`/`-webkit-`)
|
|
712
|
+
// and both `image-set` spellings. `url` appears as a Function (not a Url)
|
|
713
|
+
// node when its name carried an escape — the very case the old regex on
|
|
714
|
+
// the re-serialized text missed.
|
|
715
|
+
if (
|
|
716
|
+
child.type === "Function" &&
|
|
717
|
+
(child.name === "url" ||
|
|
718
|
+
child.name.endsWith("gradient") ||
|
|
719
|
+
child.name.endsWith("image-set"))
|
|
720
|
+
)
|
|
721
|
+
found = true;
|
|
722
|
+
},
|
|
723
|
+
});
|
|
724
|
+
return found;
|
|
725
|
+
}
|
|
726
|
+
|
|
689
727
|
// The leading color token of a `background` shorthand (the first token that
|
|
690
728
|
// canonicalizes to a real color), so `background:#fff` still compares. Returns
|
|
691
|
-
// "" (fail open, no same-color hide) when the shorthand carries an IMAGE layer
|
|
692
|
-
//
|
|
693
|
-
//
|
|
694
|
-
//
|
|
695
|
-
|
|
696
|
-
|
|
697
|
-
|
|
698
|
-
|
|
699
|
-
|
|
700
|
-
const color = canonicalizeColor(token);
|
|
729
|
+
// "" (fail open, no same-color hide) when the shorthand carries an IMAGE layer:
|
|
730
|
+
// the painted image can make same-colored text perfectly readable over it (and
|
|
731
|
+
// if it fails to load the element's own background shows through), so the flat
|
|
732
|
+
// color token is not provably the rendered backdrop.
|
|
733
|
+
/** @param {any} node value node for `background`, or null @returns {string} */
|
|
734
|
+
function backgroundColor(node) {
|
|
735
|
+
if (!node || paintsImageLayer(node)) return "";
|
|
736
|
+
for (const token of valueTokens(node)) {
|
|
737
|
+
const color = canonicalizeColor(tokenText(token));
|
|
701
738
|
if (color && (color.startsWith("#") || color === "transparent"))
|
|
702
739
|
return color;
|
|
703
740
|
}
|
|
@@ -727,8 +764,7 @@ function insetEdges(fn) {
|
|
|
727
764
|
/** @type {any[]} */
|
|
728
765
|
const edges = [];
|
|
729
766
|
for (const token of valueTokens(fn)) {
|
|
730
|
-
if (token.type === "Identifier" && token.name
|
|
731
|
-
break;
|
|
767
|
+
if (token.type === "Identifier" && token.name === "round") break;
|
|
732
768
|
edges.push(token);
|
|
733
769
|
}
|
|
734
770
|
return edges;
|
|
@@ -748,7 +784,7 @@ function isClipPathHidden(node) {
|
|
|
748
784
|
if (!node) return false;
|
|
749
785
|
for (const fn of valueTokens(node)) {
|
|
750
786
|
if (fn.type !== "Function") continue;
|
|
751
|
-
const name = fn.name
|
|
787
|
+
const name = fn.name;
|
|
752
788
|
if (name === "circle") {
|
|
753
789
|
const radius = valueTokens(fn)[0];
|
|
754
790
|
if (
|
|
@@ -808,6 +844,13 @@ function isTextPaintedVisible(val) {
|
|
|
808
844
|
return isConcreteColor(stroke) && stroke !== "transparent";
|
|
809
845
|
}
|
|
810
846
|
|
|
847
|
+
/** True when a value is exactly the keyword `none`.
|
|
848
|
+
* @param {any} node @returns {boolean} */
|
|
849
|
+
function isNoneKeyword(node) {
|
|
850
|
+
const token = soleToken(node);
|
|
851
|
+
return Boolean(token && token.type === "Identifier" && token.name === "none");
|
|
852
|
+
}
|
|
853
|
+
|
|
811
854
|
// True when the element paints a background IMAGE layer — a `background-image`
|
|
812
855
|
// longhand set to anything but `none`, or a `background` shorthand carrying
|
|
813
856
|
// `url(...)`, a gradient, or `image-set(...)`. A same-color text/background hide
|
|
@@ -819,11 +862,13 @@ function isTextPaintedVisible(val) {
|
|
|
819
862
|
// only the flat color and missed a co-declared `background-image`, splicing
|
|
820
863
|
// visible text. `background-clip:text` is NOT an image layer here — it paints
|
|
821
864
|
// the background THROUGH the glyphs and is handled by {@link isTextPaintedVisible}.
|
|
822
|
-
/** @param {(key: string) =>
|
|
823
|
-
function hasImageLayer(
|
|
824
|
-
const img =
|
|
825
|
-
|
|
826
|
-
|
|
865
|
+
/** @param {(key: string) => any} nodeOf @returns {boolean} */
|
|
866
|
+
function hasImageLayer(nodeOf) {
|
|
867
|
+
const img = nodeOf("background-image");
|
|
868
|
+
// Only the single keyword `none` proves the longhand paints nothing; any
|
|
869
|
+
// other value (an unresolvable `var()` included) counts as a layer.
|
|
870
|
+
if (img && !isNoneKeyword(img)) return true;
|
|
871
|
+
return paintsImageLayer(nodeOf("background"));
|
|
827
872
|
}
|
|
828
873
|
|
|
829
874
|
/**
|
|
@@ -871,12 +916,23 @@ function isFontShorthandHidden(node) {
|
|
|
871
916
|
const CSS_PROPERTY_IDENT_RE = /^-{0,2}[A-Za-z_][A-Za-z0-9_-]*$/;
|
|
872
917
|
|
|
873
918
|
/**
|
|
874
|
-
*
|
|
875
|
-
*
|
|
876
|
-
*
|
|
877
|
-
*
|
|
878
|
-
*
|
|
879
|
-
*
|
|
919
|
+
* One value token as text. An Identifier's `name` is already the decoded,
|
|
920
|
+
* lowercased ident (see {@link canonicalizeValue}), so it is used verbatim
|
|
921
|
+
* rather than round-tripped through the serializer; every other token is
|
|
922
|
+
* re-serialized from its (canonicalized) node fields.
|
|
923
|
+
* @param {any} token
|
|
924
|
+
* @returns {string}
|
|
925
|
+
*/
|
|
926
|
+
function tokenText(token) {
|
|
927
|
+
return token.type === "Identifier" ? token.name : csstree.generate(token);
|
|
928
|
+
}
|
|
929
|
+
|
|
930
|
+
/**
|
|
931
|
+
* Reconstruct a declaration's canonicalized value as a string for keyword/color
|
|
932
|
+
* comparisons. Escapes and letter case were resolved at the parse boundary, so
|
|
933
|
+
* `no\6e e`, `NONE` and `none` all render as `none` here. A whole-value `Raw`
|
|
934
|
+
* (an unparsed value) is returned verbatim — it never matches a hiding keyword,
|
|
935
|
+
* so it fails open.
|
|
880
936
|
* @param {any} valueNode
|
|
881
937
|
* @returns {string}
|
|
882
938
|
*/
|
|
@@ -887,25 +943,50 @@ function declText(valueNode) {
|
|
|
887
943
|
const parts = [];
|
|
888
944
|
if (valueNode.children)
|
|
889
945
|
valueNode.children.forEach((/** @type {any} */ child) =>
|
|
890
|
-
parts.push(
|
|
891
|
-
child.type === "Identifier"
|
|
892
|
-
? csstree.ident.decode(child.name)
|
|
893
|
-
: csstree.generate(child),
|
|
894
|
-
),
|
|
946
|
+
parts.push(tokenText(child)),
|
|
895
947
|
);
|
|
896
948
|
return parts.join(" ");
|
|
897
949
|
}
|
|
898
950
|
|
|
951
|
+
/**
|
|
952
|
+
* Canonicalize a parsed value subtree IN PLACE so every downstream comparison
|
|
953
|
+
* against a literal (`ABSOLUTE_UNITS`, `"rect"`, `"none"`) is correct by
|
|
954
|
+
* construction instead of depending on each call site remembering to decode and
|
|
955
|
+
* lowercase. CSS idents — keywords, function names and dimension units — are
|
|
956
|
+
* escape-decodable and ASCII case-insensitive for every keyword this module
|
|
957
|
+
* matches, so a browser reads `left:-9999PX`, `left:-9999p\78` and
|
|
958
|
+
* `left:-9999px` identically; the ad-hoc per-site `.toLowerCase()` did not, and
|
|
959
|
+
* a single uppercased unit walked straight past the detector.
|
|
960
|
+
*
|
|
961
|
+
* `Url.value` is deliberately NOT touched: css-tree already decodes url escapes,
|
|
962
|
+
* and a second decode would eat a legitimately backslash-bearing path.
|
|
963
|
+
* @param {any} valueNode
|
|
964
|
+
* @returns {void}
|
|
965
|
+
*/
|
|
966
|
+
function canonicalizeValue(valueNode) {
|
|
967
|
+
csstree.walk(valueNode, {
|
|
968
|
+
enter(/** @type {any} */ node) {
|
|
969
|
+
// ident.decode is pure string iteration and cannot throw on a token the
|
|
970
|
+
// tokenizer already produced.
|
|
971
|
+
if (node.type === "Identifier" || node.type === "Function")
|
|
972
|
+
node.name = csstree.ident.decode(node.name).toLowerCase();
|
|
973
|
+
else if (node.type === "Dimension")
|
|
974
|
+
node.unit = csstree.ident.decode(node.unit).toLowerCase();
|
|
975
|
+
},
|
|
976
|
+
});
|
|
977
|
+
}
|
|
978
|
+
|
|
899
979
|
/**
|
|
900
980
|
* Parse a style string into a map of decoded lowercase property name -> parsed
|
|
901
|
-
* value node, via css-tree's tolerant declaration-list
|
|
902
|
-
* hand-rolled declaration splitter, per-declaration
|
|
903
|
-
* `!important` stripper in one pass: css-tree
|
|
904
|
-
* as a browser does (a bogus declaration is
|
|
905
|
-
* inside a string/`url()`/paren as part of
|
|
906
|
-
* as `node.important` (so an escaped
|
|
907
|
-
* free). Property names are
|
|
908
|
-
*
|
|
981
|
+
* and canonicalized value node, via css-tree's tolerant declaration-list
|
|
982
|
+
* parser. This replaces the hand-rolled declaration splitter, per-declaration
|
|
983
|
+
* salvage, escape decoder, and `!important` stripper in one pass: css-tree
|
|
984
|
+
* recovers per-declaration exactly as a browser does (a bogus declaration is
|
|
985
|
+
* dropped, the rest kept), keeps a `;` inside a string/`url()`/paren as part of
|
|
986
|
+
* the value, and exposes `!important` as `node.important` (so an escaped
|
|
987
|
+
* spelling `none!\69mportant` is stripped for free). Property names are
|
|
988
|
+
* escape-decoded and gated to real CSS idents; anything else is dropped (fail
|
|
989
|
+
* open). Later declarations win, per the cascade.
|
|
909
990
|
* @param {string} styleStr
|
|
910
991
|
* @returns {Map<string, any>}
|
|
911
992
|
*/
|
|
@@ -934,6 +1015,7 @@ function parseDeclarations(styleStr) {
|
|
|
934
1015
|
// token; property is escape-decoded then gated to a clean CSS ident.
|
|
935
1016
|
const property = csstree.ident.decode(node.property).trim().toLowerCase();
|
|
936
1017
|
if (!CSS_PROPERTY_IDENT_RE.test(property)) return;
|
|
1018
|
+
canonicalizeValue(node.value);
|
|
937
1019
|
decls.set(property, node.value);
|
|
938
1020
|
},
|
|
939
1021
|
});
|
|
@@ -1015,7 +1097,7 @@ export function isHiddenStyle(styleStr) {
|
|
|
1015
1097
|
return true;
|
|
1016
1098
|
const background =
|
|
1017
1099
|
canonicalizeColor(textOf("background-color")) ||
|
|
1018
|
-
backgroundColor(
|
|
1100
|
+
backgroundColor(nodeOf("background"));
|
|
1019
1101
|
// Only flag same-color when BOTH sides resolve to a concrete color (`#rrggbb`
|
|
1020
1102
|
// or `transparent`), AND no background IMAGE layer is present (an image can
|
|
1021
1103
|
// make same-colored text readable). `var(--x)`, `inherit`, and `currentColor`
|
|
@@ -1027,7 +1109,7 @@ export function isHiddenStyle(styleStr) {
|
|
|
1027
1109
|
effectiveColor &&
|
|
1028
1110
|
effectiveColor === background &&
|
|
1029
1111
|
isConcreteColor(effectiveColor) &&
|
|
1030
|
-
!hasImageLayer(
|
|
1112
|
+
!hasImageLayer(nodeOf)
|
|
1031
1113
|
)
|
|
1032
1114
|
return true;
|
|
1033
1115
|
|
|
@@ -1137,12 +1219,26 @@ function hasDataSrc(el) {
|
|
|
1137
1219
|
);
|
|
1138
1220
|
}
|
|
1139
1221
|
|
|
1222
|
+
// One shared fragment parser for every HTML parse in this module (mirroring
|
|
1223
|
+
// `mdParser` below): all of them must agree on the tokenizer's verdict, so
|
|
1224
|
+
// there is exactly one parser configuration to reason about.
|
|
1225
|
+
const htmlParser = unified().use(rehypeParse, { fragment: true });
|
|
1226
|
+
|
|
1227
|
+
/**
|
|
1228
|
+
* Parse `html` as an HTML fragment with the real tokenizer (parse5, via rehype).
|
|
1229
|
+
* @param {string} html
|
|
1230
|
+
* @returns {any}
|
|
1231
|
+
*/
|
|
1232
|
+
function parseFragment(html) {
|
|
1233
|
+
return htmlParser.parse(html);
|
|
1234
|
+
}
|
|
1235
|
+
|
|
1140
1236
|
/**
|
|
1141
1237
|
* @param {string} htmlValue
|
|
1142
1238
|
* @returns {any}
|
|
1143
1239
|
*/
|
|
1144
1240
|
function parseHtmlTag(htmlValue) {
|
|
1145
|
-
const tree =
|
|
1241
|
+
const tree = parseFragment(htmlValue);
|
|
1146
1242
|
/** @type {any} */
|
|
1147
1243
|
let firstElement = null;
|
|
1148
1244
|
visit(tree, "element", (node) => {
|
|
@@ -1276,7 +1372,19 @@ function hasWarned(warned) {
|
|
|
1276
1372
|
* @returns {{ ranges: Array<{start: number, end: number, kind: "comment" | "hidden"}>, warned: ReturnType<typeof newWarned> }}
|
|
1277
1373
|
*/
|
|
1278
1374
|
export function scanHtmlFragment(html) {
|
|
1279
|
-
|
|
1375
|
+
return scanFragmentTree(html, parseFragment(html));
|
|
1376
|
+
}
|
|
1377
|
+
|
|
1378
|
+
/**
|
|
1379
|
+
* `scanHtmlFragment` for a caller that already has the fragment tree — the
|
|
1380
|
+
* dispatch in `sanitizeHtml` parses to decide the branch, so re-parsing there
|
|
1381
|
+
* would tokenize the same input twice. `tree` MUST be the parse of `html`;
|
|
1382
|
+
* the ranges are offsets into `html`, read from that tree's positions.
|
|
1383
|
+
* @param {string} html
|
|
1384
|
+
* @param {any} tree
|
|
1385
|
+
* @returns {{ ranges: Array<{start: number, end: number, kind: "comment" | "hidden"}>, warned: ReturnType<typeof newWarned> }}
|
|
1386
|
+
*/
|
|
1387
|
+
function scanFragmentTree(html, tree) {
|
|
1280
1388
|
/** @type {Array<{start: number, end: number, kind: "comment" | "hidden"}>} */
|
|
1281
1389
|
const ranges = [];
|
|
1282
1390
|
const warned = newWarned();
|
|
@@ -1361,7 +1469,7 @@ function foldAbsorb(absorbing, raw) {
|
|
|
1361
1469
|
* @returns {Map<number, number>}
|
|
1362
1470
|
*/
|
|
1363
1471
|
function commentSpans(value) {
|
|
1364
|
-
const tree =
|
|
1472
|
+
const tree = parseFragment(value);
|
|
1365
1473
|
/** @type {Map<number, number>} */
|
|
1366
1474
|
const spans = new Map();
|
|
1367
1475
|
visit(tree, "comment", (/** @type {any} */ node) => {
|
|
@@ -1629,20 +1737,77 @@ function scanMarkdown(text) {
|
|
|
1629
1737
|
return { ranges, warned };
|
|
1630
1738
|
}
|
|
1631
1739
|
|
|
1632
|
-
// 30%-of-lines heuristic: HTML *source* gets scanned as one rehype fragment;
|
|
1633
|
-
// inline tags scattered in prose go through the markdown branch instead.
|
|
1634
1740
|
/**
|
|
1741
|
+
* True when remark finds a code block — fenced or indented — in `text`.
|
|
1742
|
+
*
|
|
1743
|
+
* Asked before the HTML tokenizer because "no character data outside the
|
|
1744
|
+
* markup" cannot see an INDENTED code block: its four leading spaces are
|
|
1745
|
+
* whitespace, so a document that is nothing but one indented block
|
|
1746
|
+
* (`" <div hidden>x</div>\n"`) satisfies the rule and takes the source
|
|
1747
|
+
* branch, and the hidden element gets spliced out of a block the renderer
|
|
1748
|
+
* displays as literal text. A fence escapes only incidentally, because the
|
|
1749
|
+
* backticks are non-whitespace character data. Code blocks are markdown-ONLY
|
|
1750
|
+
* syntax, so their presence settles the question the same way the tokenizer
|
|
1751
|
+
* does — by parsing, not by counting.
|
|
1635
1752
|
* @param {string} text
|
|
1636
1753
|
* @returns {boolean}
|
|
1637
1754
|
*/
|
|
1638
|
-
|
|
1639
|
-
|
|
1640
|
-
|
|
1641
|
-
|
|
1642
|
-
|
|
1643
|
-
|
|
1755
|
+
function hasMarkdownCode(text) {
|
|
1756
|
+
let found = false;
|
|
1757
|
+
visit(mdParser.parse(text), "code", () => {
|
|
1758
|
+
found = true;
|
|
1759
|
+
return EXIT;
|
|
1760
|
+
});
|
|
1761
|
+
return found;
|
|
1762
|
+
}
|
|
1763
|
+
|
|
1764
|
+
/**
|
|
1765
|
+
* The parsed fragment tree for `text` when `text` is HTML *source*, else null.
|
|
1766
|
+
*
|
|
1767
|
+
* "HTML source" means the markup accounts for the WHOLE document: the real
|
|
1768
|
+
* tokenizer (parse5, via rehype) places every element there is, and the only
|
|
1769
|
+
* character data it leaves OUTSIDE all of them is whitespace. That is exactly
|
|
1770
|
+
* the property the source branch needs — it hands the whole input to
|
|
1771
|
+
* `scanHtmlFragment` as one fragment, which is faithful only when there is no
|
|
1772
|
+
* non-HTML syntax around the markup for that parse to misread.
|
|
1773
|
+
*
|
|
1774
|
+
* Everything else fails OPEN to the markdown branch, which parses with remark
|
|
1775
|
+
* and scans only the spans remark itself calls HTML. That is the conservative
|
|
1776
|
+
* direction: markdown-only constructs (fenced/indented code, tables, lists)
|
|
1777
|
+
* keep their meaning, so an HTML sample inside a code fence is displayed
|
|
1778
|
+
* rather than spliced. The dispatch this replaces counted tag-shaped LINES and
|
|
1779
|
+
* took the source branch above 30% of them, which got that case wrong — it
|
|
1780
|
+
* spliced hidden-element examples out of documentation code blocks.
|
|
1781
|
+
*
|
|
1782
|
+
* Character data is judged by its DECODED value, as a renderer sees it: the
|
|
1783
|
+
* ignored `<html>`/`<head>`/`<body>` tags of a full page leave their
|
|
1784
|
+
* surrounding newlines merged into one text node, and ` ` between two
|
|
1785
|
+
* elements is whitespace on the page.
|
|
1786
|
+
* @param {string} text
|
|
1787
|
+
* @returns {any}
|
|
1788
|
+
*/
|
|
1789
|
+
function htmlSourceTree(text) {
|
|
1790
|
+
if (hasMarkdownCode(text)) return null;
|
|
1791
|
+
const tree = parseFragment(text);
|
|
1792
|
+
let sawElement = false;
|
|
1793
|
+
// Only ROOT children can hold character data outside an element; everything
|
|
1794
|
+
// deeper is by construction inside one.
|
|
1795
|
+
for (const node of tree.children) {
|
|
1796
|
+
if (node.type === "element") sawElement = true;
|
|
1797
|
+
// Comments and the doctype are markup, not character data.
|
|
1798
|
+
else if (node.type === "text" && node.value.trim() !== "") return null;
|
|
1644
1799
|
}
|
|
1645
|
-
return
|
|
1800
|
+
return sawElement ? tree : null;
|
|
1801
|
+
}
|
|
1802
|
+
|
|
1803
|
+
/**
|
|
1804
|
+
* True when `text` is HTML source rather than markdown that merely contains
|
|
1805
|
+
* tags — see `htmlSourceTree` for the definition and the fail-open rationale.
|
|
1806
|
+
* @param {string} text
|
|
1807
|
+
* @returns {boolean}
|
|
1808
|
+
*/
|
|
1809
|
+
export function looksLikeHtmlSource(text) {
|
|
1810
|
+
return htmlSourceTree(text) !== null;
|
|
1646
1811
|
}
|
|
1647
1812
|
|
|
1648
1813
|
/**
|
|
@@ -1658,9 +1823,10 @@ export function sanitizeHtml(text) {
|
|
|
1658
1823
|
/** @type {{ ranges: Array<{start: number, end: number, kind: "comment" | "hidden"}>, warned: ReturnType<typeof newWarned> }} */
|
|
1659
1824
|
let scan;
|
|
1660
1825
|
try {
|
|
1661
|
-
|
|
1662
|
-
|
|
1663
|
-
|
|
1826
|
+
// One parse decides the branch AND feeds it, so the source branch does not
|
|
1827
|
+
// tokenize the input twice.
|
|
1828
|
+
const sourceTree = htmlSourceTree(text);
|
|
1829
|
+
scan = sourceTree ? scanFragmentTree(text, sourceTree) : scanMarkdown(text);
|
|
1664
1830
|
} catch {
|
|
1665
1831
|
// The parse/visit blew up (stack overflow on pathological nesting, or any
|
|
1666
1832
|
// other parser error). Fail CLOSED at this boundary so `sanitize`/
|
|
@@ -2151,7 +2317,7 @@ function multiUrlAttr(value) {
|
|
|
2151
2317
|
* @returns {Array<{ url: string, isImage: boolean, context: "resource" | "form" | "refresh" }>}
|
|
2152
2318
|
*/
|
|
2153
2319
|
function extractHtmlUrls(text) {
|
|
2154
|
-
const tree =
|
|
2320
|
+
const tree = parseFragment(text);
|
|
2155
2321
|
/** @type {Array<{ url: string, isImage: boolean, context: "resource" | "form" | "refresh" }>} */
|
|
2156
2322
|
const urls = [];
|
|
2157
2323
|
visit(tree, "element", (/** @type {any} */ node) => {
|