sanitize-html 2.17.7 → 2.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +14 -0
- package/index.js +195 -7
- package/package.json +14 -14
package/README.md
CHANGED
|
@@ -810,6 +810,20 @@ nestingLimit: 6
|
|
|
810
810
|
|
|
811
811
|
This will prevent the user from nesting tags more than 6 levels deep. Tags deeper than that are stripped out exactly as if they were disallowed. Note that this means text is preserved in the usual ways where appropriate.
|
|
812
812
|
|
|
813
|
+
### Routing warnings to your own logger
|
|
814
|
+
|
|
815
|
+
sanitize-html writes its own diagnostics - the vulnerable tag notice above, and the style parsing notice in the browser - to the console. If your application has a logging pipeline of its own, pass any console-shaped object as the `logger` option and they are delivered to it instead:
|
|
816
|
+
|
|
817
|
+
```javascript
|
|
818
|
+
sanitizeHtml(dirty, {
|
|
819
|
+
logger: myLogger // an object with debug, info, warn and error methods
|
|
820
|
+
});
|
|
821
|
+
```
|
|
822
|
+
|
|
823
|
+
Any of the four methods that your object does not provide falls back to the console, and without the option at all the console is still the destination, exactly as before.
|
|
824
|
+
|
|
825
|
+
This matters for applications that write structured logs: a raw `console.warn` bypasses their filtering and formatting, and puts unparseable text on a stream that is expected to be one JSON object per line.
|
|
826
|
+
|
|
813
827
|
### Advanced filtering
|
|
814
828
|
|
|
815
829
|
For more advanced filtering you can hook directly into the parsing process using tag open and tag close events.
|
package/index.js
CHANGED
|
@@ -24,6 +24,22 @@ const svgAnimationTags = [
|
|
|
24
24
|
// `href` are covered.
|
|
25
25
|
const alwaysUrlAttributes = [ 'href' ];
|
|
26
26
|
|
|
27
|
+
// Our own diagnostics. A console-shaped `logger` option takes them instead of
|
|
28
|
+
// the console; missing methods fall back to it.
|
|
29
|
+
const severities = [ 'debug', 'info', 'warn', 'error' ];
|
|
30
|
+
|
|
31
|
+
function loggerFor(options) {
|
|
32
|
+
const source = (options && options.logger) || console;
|
|
33
|
+
const logger = {};
|
|
34
|
+
for (const severity of severities) {
|
|
35
|
+
logger[severity] = typeof source[severity] === 'function'
|
|
36
|
+
? (...args) => source[severity](...args)
|
|
37
|
+
// eslint-disable-next-line no-console
|
|
38
|
+
: (...args) => console[severity](...args);
|
|
39
|
+
}
|
|
40
|
+
return logger;
|
|
41
|
+
}
|
|
42
|
+
|
|
27
43
|
function each(obj, cb) {
|
|
28
44
|
if (obj) {
|
|
29
45
|
Object.keys(obj).forEach(function (key) {
|
|
@@ -131,6 +147,8 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
131
147
|
options = Object.assign({}, sanitizeHtml.defaults, options);
|
|
132
148
|
options.parser = Object.assign({}, htmlParserDefaults, options.parser);
|
|
133
149
|
|
|
150
|
+
const logger = loggerFor(options);
|
|
151
|
+
|
|
134
152
|
const tagAllowed = function (name) {
|
|
135
153
|
return options.allowedTags === false ||
|
|
136
154
|
(options.allowedTags || []).indexOf(name) > -1;
|
|
@@ -139,7 +157,12 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
139
157
|
// vulnerableTags
|
|
140
158
|
vulnerableTags.forEach(function (tag) {
|
|
141
159
|
if (tagAllowed(tag) && !options.allowVulnerableTags) {
|
|
142
|
-
|
|
160
|
+
logger.warn(
|
|
161
|
+
`Your \`allowedTags\` option includes \`${tag}\`, which is inherently ` +
|
|
162
|
+
'vulnerable to XSS attacks. Please remove it from `allowedTags`, or, ' +
|
|
163
|
+
'to disable this warning, add the `allowVulnerableTags` option and ' +
|
|
164
|
+
'ensure you are accounting for this risk.'
|
|
165
|
+
);
|
|
143
166
|
}
|
|
144
167
|
});
|
|
145
168
|
|
|
@@ -233,6 +256,15 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
233
256
|
let transformMap;
|
|
234
257
|
let skipText;
|
|
235
258
|
let skipTextDepth;
|
|
259
|
+
// Browsers (with scripting enabled) parse <noscript> content as raw text up
|
|
260
|
+
// to the first `</noscript`, but htmlparser2 parses it as markup, so an end
|
|
261
|
+
// tag for an ancestor can make htmlparser2 close the <noscript> implicitly
|
|
262
|
+
// much earlier. `rawTextEnd` is the source offset where the browser ends the
|
|
263
|
+
// <noscript> being discarded, and `skipRawText` is set while htmlparser2 has
|
|
264
|
+
// already closed it but the browser has not, so that we keep discarding
|
|
265
|
+
// until we reach that offset (GHSA-x3q4-9hxx-gx8m).
|
|
266
|
+
let rawTextEnd;
|
|
267
|
+
let skipRawText;
|
|
236
268
|
let addedText = false;
|
|
237
269
|
|
|
238
270
|
initializeState();
|
|
@@ -242,6 +274,7 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
242
274
|
if (options.onOpenTag) {
|
|
243
275
|
options.onOpenTag(name, attribs);
|
|
244
276
|
}
|
|
277
|
+
updateRawTextRegion();
|
|
245
278
|
|
|
246
279
|
// If `enforceHtmlBoundary` is `true` and this has found the opening
|
|
247
280
|
// `html` tag, reset the state.
|
|
@@ -249,6 +282,9 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
249
282
|
initializeState();
|
|
250
283
|
}
|
|
251
284
|
|
|
285
|
+
if (skipRawText) {
|
|
286
|
+
return;
|
|
287
|
+
}
|
|
252
288
|
if (skipText) {
|
|
253
289
|
skipTextDepth++;
|
|
254
290
|
return;
|
|
@@ -290,6 +326,9 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
290
326
|
if (nonTextTagsArray.indexOf(name) !== -1) {
|
|
291
327
|
skipText = true;
|
|
292
328
|
skipTextDepth = 1;
|
|
329
|
+
if (frame.tag.toLowerCase() === 'noscript') {
|
|
330
|
+
rawTextEnd = findRawTextEnd('noscript', parser.endIndex + 1);
|
|
331
|
+
}
|
|
293
332
|
}
|
|
294
333
|
}
|
|
295
334
|
}
|
|
@@ -389,6 +428,14 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
389
428
|
}
|
|
390
429
|
}
|
|
391
430
|
|
|
431
|
+
// `<meta http-equiv="refresh" content="0;url=...">` navigates to a
|
|
432
|
+
// URL embedded in `content`, so scheme check that URL too
|
|
433
|
+
// (GHSA-cv27-6wvh-8x7j). Other meta `content` values are left alone.
|
|
434
|
+
if (name === 'meta' && a.toLowerCase() === 'content' && isRefresh(attribs) && naughtyRefresh(value)) {
|
|
435
|
+
delete frame.attribs[a];
|
|
436
|
+
return;
|
|
437
|
+
}
|
|
438
|
+
|
|
392
439
|
if (name === 'script' && a === 'src') {
|
|
393
440
|
|
|
394
441
|
let allowed = true;
|
|
@@ -455,7 +502,7 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
455
502
|
try {
|
|
456
503
|
let parsed = parseSrcset(value);
|
|
457
504
|
parsed.forEach(function(value) {
|
|
458
|
-
if (naughtyHref(
|
|
505
|
+
if (naughtyHref(name, value.url)) {
|
|
459
506
|
value.evil = true;
|
|
460
507
|
}
|
|
461
508
|
});
|
|
@@ -527,7 +574,14 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
527
574
|
}
|
|
528
575
|
} catch (e) {
|
|
529
576
|
if (typeof window !== 'undefined') {
|
|
530
|
-
|
|
577
|
+
logger.warn(
|
|
578
|
+
`Failed to parse "${name} {${value}}". If you are ` +
|
|
579
|
+
'running this in a browser, we recommend disabling ' +
|
|
580
|
+
'style parsing with the parseStyleAttributes option, ' +
|
|
581
|
+
'since it only works in a node environment due to a ' +
|
|
582
|
+
'postcss dependency. More info: ' +
|
|
583
|
+
'https://github.com/apostrophecms/sanitize-html/issues/547'
|
|
584
|
+
);
|
|
531
585
|
}
|
|
532
586
|
delete frame.attribs[a];
|
|
533
587
|
return;
|
|
@@ -568,7 +622,8 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
568
622
|
frame.openingTagLength = result.length - frame.tagPosition;
|
|
569
623
|
},
|
|
570
624
|
ontext: function(text) {
|
|
571
|
-
|
|
625
|
+
updateRawTextRegion();
|
|
626
|
+
if (skipText || skipRawText) {
|
|
572
627
|
return;
|
|
573
628
|
}
|
|
574
629
|
const lastFrame = stack[stack.length - 1];
|
|
@@ -622,6 +677,16 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
622
677
|
// htmlparser2, so their contents are decoded and must be escaped below
|
|
623
678
|
// like any other text (important to prevent XSS via entity-encoded
|
|
624
679
|
// payloads such as <option><script>...</script></option>).
|
|
680
|
+
} else if (
|
|
681
|
+
// htmlparser2 treats <iframe> as a raw-text element, so markup inside
|
|
682
|
+
// (including after an unclosed <iframe>) arrives as a single text node.
|
|
683
|
+
// When the iframe is discarded, re-sanitize that fallback markup as HTML
|
|
684
|
+
// instead of escaping it as plain text (issue #5550).
|
|
685
|
+
tag === 'iframe' &&
|
|
686
|
+
!tagAllowed(tag) &&
|
|
687
|
+
options.disallowedTagsMode === 'discard'
|
|
688
|
+
) {
|
|
689
|
+
result += sanitizeHtml(text, options);
|
|
625
690
|
} else if (!addedText) {
|
|
626
691
|
const escaped = escapeHtml(text, false);
|
|
627
692
|
if (options.textFilter) {
|
|
@@ -640,10 +705,24 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
640
705
|
options.onCloseTag(name, isImplied);
|
|
641
706
|
}
|
|
642
707
|
|
|
643
|
-
|
|
708
|
+
updateRawTextRegion();
|
|
709
|
+
if (skipRawText) {
|
|
710
|
+
// Still inside the browser's raw text: only close elements that were
|
|
711
|
+
// opened before the discarded region, so the output stays balanced.
|
|
712
|
+
const lastFrame = stack[stack.length - 1];
|
|
713
|
+
if (!lastFrame || lastFrame.tag !== name) {
|
|
714
|
+
return;
|
|
715
|
+
}
|
|
716
|
+
} else if (skipText) {
|
|
644
717
|
skipTextDepth--;
|
|
645
718
|
if (!skipTextDepth) {
|
|
646
719
|
skipText = false;
|
|
720
|
+
if (rawTextEnd !== null) {
|
|
721
|
+
// htmlparser2 closed the element implicitly (e.g. an ancestor's
|
|
722
|
+
// end tag) before the browser would. Close its frame below, but
|
|
723
|
+
// keep discarding up to the browser's end tag.
|
|
724
|
+
skipRawText = true;
|
|
725
|
+
}
|
|
647
726
|
} else {
|
|
648
727
|
return;
|
|
649
728
|
}
|
|
@@ -746,6 +825,27 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
746
825
|
transformMap = {};
|
|
747
826
|
skipText = false;
|
|
748
827
|
skipTextDepth = 0;
|
|
828
|
+
rawTextEnd = null;
|
|
829
|
+
skipRawText = false;
|
|
830
|
+
}
|
|
831
|
+
|
|
832
|
+
// Leave the raw text region once the parser reaches the offset where the
|
|
833
|
+
// browser ends it.
|
|
834
|
+
function updateRawTextRegion() {
|
|
835
|
+
if (rawTextEnd !== null && parser.startIndex >= rawTextEnd) {
|
|
836
|
+
rawTextEnd = null;
|
|
837
|
+
skipRawText = false;
|
|
838
|
+
}
|
|
839
|
+
}
|
|
840
|
+
|
|
841
|
+
// Offset of the end tag that ends a raw text element in a browser: the
|
|
842
|
+
// first case-insensitive `</name` followed by HTML whitespace, `/` or `>`.
|
|
843
|
+
// With no such end tag the element runs to the end of the input.
|
|
844
|
+
function findRawTextEnd(tagName, from) {
|
|
845
|
+
const re = new RegExp('</' + tagName + '[\\t\\n\\f\\r />]', 'ig');
|
|
846
|
+
re.lastIndex = from;
|
|
847
|
+
const match = re.exec(html);
|
|
848
|
+
return match ? match.index : Infinity;
|
|
749
849
|
}
|
|
750
850
|
|
|
751
851
|
function escapeHtml(s, quote) {
|
|
@@ -784,6 +884,86 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
784
884
|
});
|
|
785
885
|
}
|
|
786
886
|
|
|
887
|
+
function isRefresh(attribs) {
|
|
888
|
+
return Object.keys(attribs).some(function(a) {
|
|
889
|
+
return a.toLowerCase() === 'http-equiv' &&
|
|
890
|
+
String(attribs[a]).trim().toLowerCase() === 'refresh';
|
|
891
|
+
});
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
// True if the `content` of a `<meta http-equiv="refresh">` must be dropped:
|
|
895
|
+
// its destination URL fails the scheme policy, or it cannot be parsed as a
|
|
896
|
+
// refresh at all (a browser would ignore it then, so nothing is lost).
|
|
897
|
+
// Extracts the URL the way the HTML standard's "shared declarative refresh
|
|
898
|
+
// steps" do, so that spelling, separator, quoting and case variations of
|
|
899
|
+
// `url=`, or no `url=` at all, all yield the URL a browser would navigate to.
|
|
900
|
+
function naughtyRefresh(content) {
|
|
901
|
+
const input = String(content);
|
|
902
|
+
const isWhitespace = function(c) {
|
|
903
|
+
return c === ' ' || c === '\t' || c === '\n' || c === '\f' || c === '\r';
|
|
904
|
+
};
|
|
905
|
+
let position = 0;
|
|
906
|
+
const skipWhitespace = function() {
|
|
907
|
+
while (position < input.length && isWhitespace(input[position])) {
|
|
908
|
+
position++;
|
|
909
|
+
}
|
|
910
|
+
};
|
|
911
|
+
const lowerAt = function(i) {
|
|
912
|
+
return (input[i] || '').toLowerCase();
|
|
913
|
+
};
|
|
914
|
+
skipWhitespace();
|
|
915
|
+
const timeStart = position;
|
|
916
|
+
while (position < input.length && /[0-9.]/.test(input[position])) {
|
|
917
|
+
position++;
|
|
918
|
+
}
|
|
919
|
+
if (position === timeStart) {
|
|
920
|
+
return true;
|
|
921
|
+
}
|
|
922
|
+
if (position < input.length) {
|
|
923
|
+
const c = input[position];
|
|
924
|
+
if (c !== ';' && c !== ',' && !isWhitespace(c)) {
|
|
925
|
+
return true;
|
|
926
|
+
}
|
|
927
|
+
skipWhitespace();
|
|
928
|
+
if (input[position] === ';' || input[position] === ',') {
|
|
929
|
+
position++;
|
|
930
|
+
}
|
|
931
|
+
skipWhitespace();
|
|
932
|
+
}
|
|
933
|
+
if (position >= input.length) {
|
|
934
|
+
// No URL: refreshes the current document
|
|
935
|
+
return false;
|
|
936
|
+
}
|
|
937
|
+
let url = input.slice(position);
|
|
938
|
+
let quoted = true;
|
|
939
|
+
if (lowerAt(position) === 'u') {
|
|
940
|
+
quoted = false;
|
|
941
|
+
if (lowerAt(position + 1) === 'r' && lowerAt(position + 2) === 'l') {
|
|
942
|
+
position += 3;
|
|
943
|
+
skipWhitespace();
|
|
944
|
+
if (input[position] === '=') {
|
|
945
|
+
position++;
|
|
946
|
+
skipWhitespace();
|
|
947
|
+
quoted = true;
|
|
948
|
+
}
|
|
949
|
+
}
|
|
950
|
+
}
|
|
951
|
+
if (quoted) {
|
|
952
|
+
const quote = input[position];
|
|
953
|
+
if (quote === '"' || quote === '\'') {
|
|
954
|
+
position++;
|
|
955
|
+
}
|
|
956
|
+
url = input.slice(position);
|
|
957
|
+
if (quote === '"' || quote === '\'') {
|
|
958
|
+
const end = url.indexOf(quote);
|
|
959
|
+
if (end !== -1) {
|
|
960
|
+
url = url.slice(0, end);
|
|
961
|
+
}
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
return naughtyHref('meta', url);
|
|
965
|
+
}
|
|
966
|
+
|
|
787
967
|
// True if this is an SVG SMIL animation element that animates a URL-bearing
|
|
788
968
|
// attribute, e.g. `<animate attributeName="href" values="#safe;javascript:...">`.
|
|
789
969
|
//
|
|
@@ -799,12 +979,14 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
799
979
|
// animation on the strength of its target instead. Animations of attributes
|
|
800
980
|
// that are not URL sinks, such as `fill` or `opacity`, are unaffected.
|
|
801
981
|
function animatesUrlAttribute(name, attribs) {
|
|
802
|
-
|
|
982
|
+
// In an XML serialization a prefixed name such as `svg:animate` is the same
|
|
983
|
+
// element as `animate`, so match on the local name (GHSA-374f-7chj-9948).
|
|
984
|
+
if (svgAnimationTags.indexOf(localPart(name)) === -1) {
|
|
803
985
|
return false;
|
|
804
986
|
}
|
|
805
987
|
const schemeCheckedAttributes = options.allowedSchemesAppliedToAttributes || [];
|
|
806
988
|
return Object.keys(attribs || {}).some(function(attributeName) {
|
|
807
|
-
if (attributeName
|
|
989
|
+
if (localPart(attributeName) !== 'attributename') {
|
|
808
990
|
return false;
|
|
809
991
|
}
|
|
810
992
|
const target = (attribs[attributeName] || '').trim().toLowerCase();
|
|
@@ -817,6 +999,12 @@ function sanitizeHtml(html, options, _recursing) {
|
|
|
817
999
|
});
|
|
818
1000
|
}
|
|
819
1001
|
|
|
1002
|
+
// Lowercased name with any namespace prefix removed.
|
|
1003
|
+
function localPart(name) {
|
|
1004
|
+
const lower = name.toLowerCase();
|
|
1005
|
+
return lower.slice(lower.lastIndexOf(':') + 1);
|
|
1006
|
+
}
|
|
1007
|
+
|
|
820
1008
|
function parseUrl(value) {
|
|
821
1009
|
value = value.replace(/^(\w+:)?\s*[\\/]\s*[\\/]/, '$1//');
|
|
822
1010
|
if (value.startsWith('relative:')) {
|
package/package.json
CHANGED
|
@@ -1,12 +1,16 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sanitize-html",
|
|
3
|
-
"version": "2.
|
|
3
|
+
"version": "2.18.0",
|
|
4
4
|
"description": "Clean up user-submitted HTML, preserving allowlisted elements and allowlisted attributes on a per-element basis",
|
|
5
5
|
"sideEffects": false,
|
|
6
6
|
"main": "index.js",
|
|
7
7
|
"files": [
|
|
8
8
|
"index.js"
|
|
9
9
|
],
|
|
10
|
+
"scripts": {
|
|
11
|
+
"test": "npm run lint && mocha",
|
|
12
|
+
"lint": "eslint ."
|
|
13
|
+
},
|
|
10
14
|
"repository": {
|
|
11
15
|
"type": "git",
|
|
12
16
|
"url": "https://github.com/apostrophecms/apostrophe.git",
|
|
@@ -25,25 +29,21 @@
|
|
|
25
29
|
"node": ">=22.12.0"
|
|
26
30
|
},
|
|
27
31
|
"dependencies": {
|
|
28
|
-
"deepmerge": "^4.
|
|
32
|
+
"deepmerge": "^4.3.1",
|
|
29
33
|
"escape-string-regexp": "^4.0.0",
|
|
30
34
|
"htmlparser2": "^12.0.0",
|
|
31
|
-
"is-plain-object": "^5.
|
|
35
|
+
"is-plain-object": "^5.1.0",
|
|
36
|
+
"launder": "^1.7.2",
|
|
32
37
|
"parse-srcset": "^1.0.2",
|
|
33
|
-
"postcss": "^8.
|
|
34
|
-
"launder": "^1.7.1"
|
|
38
|
+
"postcss": "^8.5.27"
|
|
35
39
|
},
|
|
36
40
|
"devDependencies": {
|
|
37
|
-
"eslint": "^9.39.
|
|
38
|
-
"
|
|
39
|
-
"
|
|
40
|
-
"
|
|
41
|
+
"eslint": "^9.39.5",
|
|
42
|
+
"eslint-config-apostrophe": "^6.1.0",
|
|
43
|
+
"mocha": "^11.8.0",
|
|
44
|
+
"sinon": "^9.2.4"
|
|
41
45
|
},
|
|
42
46
|
"apostropheTestConfig": {
|
|
43
47
|
"requiresMongo": false
|
|
44
|
-
},
|
|
45
|
-
"scripts": {
|
|
46
|
-
"test": "npm run lint && mocha",
|
|
47
|
-
"lint": "eslint ."
|
|
48
48
|
}
|
|
49
|
-
}
|
|
49
|
+
}
|