sanitize-html 2.17.6 → 2.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +27 -0
  2. package/index.js +238 -6
  3. package/package.json +14 -14
package/README.md CHANGED
@@ -696,6 +696,19 @@ And you can forbid the use of protocol-relative URLs (starting with `//`) to acc
696
696
  allowProtocolRelative: false
697
697
  ```
698
698
 
699
+ ### SVG animations of URL attributes are discarded
700
+
701
+ For security reasons, if your `allowedTags` includes the SVG animation elements (`animate`, `animateColor`, `animateMotion`, `animateTransform` and `set`), note that an animation which targets a URL attribute is always discarded:
702
+
703
+ ```html
704
+ <!-- Discarded: this would set the link's href to javascript: after sanitization -->
705
+ <animate attributeName="href" values="#safe;javascript:alert(1)" dur=".01s" fill="freeze">
706
+ ```
707
+
708
+ An animation element carries no URL itself. It names the attribute it animates with `attributeName` and supplies the new value in `values`, `from`, `to` or `by`, which the browser copies into the target attribute after sanitization. Scheme checking those values is not sufficient, because `values` is a semicolon-separated *list* of destinations, so we discard the animation instead whenever `attributeName` selects `href`, `xlink:href` or any other attribute listed in `allowedSchemesAppliedToAttributes`.
709
+
710
+ Animations of attributes that are not URLs, such as `fill` or `opacity`, are unaffected.
711
+
699
712
  ### Discarding the entire contents of a disallowed tag
700
713
 
701
714
  Normally, with a few exceptions, if a tag is not allowed, all of the text within it is preserved, and so are any allowed tags within it.
@@ -797,6 +810,20 @@ nestingLimit: 6
797
810
 
798
811
  This will prevent the user from nesting tags more than 6 levels deep. Tags deeper than that are stripped out exactly as if they were disallowed. Note that this means text is preserved in the usual ways where appropriate.
799
812
 
813
+ ### Routing warnings to your own logger
814
+
815
+ sanitize-html writes its own diagnostics - the vulnerable tag notice above, and the style parsing notice in the browser - to the console. If your application has a logging pipeline of its own, pass any console-shaped object as the `logger` option and they are delivered to it instead:
816
+
817
+ ```javascript
818
+ sanitizeHtml(dirty, {
819
+ logger: myLogger // an object with debug, info, warn and error methods
820
+ });
821
+ ```
822
+
823
+ Any of the four methods that your object does not provide falls back to the console, and without the option at all the console is still the destination, exactly as before.
824
+
825
+ This matters for applications that write structured logs: a raw `console.warn` bypasses their filtering and formatting, and puts unparseable text on a stream that is expected to be one JSON object per line.
826
+
800
827
  ### Advanced filtering
801
828
 
802
829
  For more advanced filtering you can hook directly into the parsing process using tag open and tag close events.
package/index.js CHANGED
@@ -12,6 +12,33 @@ const mediaTags = [
12
12
  ];
13
13
  // Tags that are inherently vulnerable to being used in XSS attacks.
14
14
  const vulnerableTags = [ 'script', 'style' ];
15
+ // SVG SMIL animation elements. These do not carry a URL themselves: they
16
+ // retarget an attribute of another element, naming it with `attributeName`
17
+ // and supplying the new value(s) in `values`, `from`, `to` and `by`.
18
+ const svgAnimationTags = [
19
+ 'animate', 'animatecolor', 'animatemotion', 'animatetransform', 'set'
20
+ ];
21
+ // Attribute names that always name a URL sink, whatever
22
+ // `allowedSchemesAppliedToAttributes` has been narrowed to. A namespace prefix
23
+ // is ignored when matching, so `xlink:href` and any other prefixed spelling of
24
+ // `href` are covered.
25
+ const alwaysUrlAttributes = [ 'href' ];
26
+
27
+ // Our own diagnostics. A console-shaped `logger` option takes them instead of
28
+ // the console; missing methods fall back to it.
29
+ const severities = [ 'debug', 'info', 'warn', 'error' ];
30
+
31
+ function loggerFor(options) {
32
+ const source = (options && options.logger) || console;
33
+ const logger = {};
34
+ for (const severity of severities) {
35
+ logger[severity] = typeof source[severity] === 'function'
36
+ ? (...args) => source[severity](...args)
37
+ // eslint-disable-next-line no-console
38
+ : (...args) => console[severity](...args);
39
+ }
40
+ return logger;
41
+ }
15
42
 
16
43
  function each(obj, cb) {
17
44
  if (obj) {
@@ -120,6 +147,8 @@ function sanitizeHtml(html, options, _recursing) {
120
147
  options = Object.assign({}, sanitizeHtml.defaults, options);
121
148
  options.parser = Object.assign({}, htmlParserDefaults, options.parser);
122
149
 
150
+ const logger = loggerFor(options);
151
+
123
152
  const tagAllowed = function (name) {
124
153
  return options.allowedTags === false ||
125
154
  (options.allowedTags || []).indexOf(name) > -1;
@@ -128,7 +157,12 @@ function sanitizeHtml(html, options, _recursing) {
128
157
  // vulnerableTags
129
158
  vulnerableTags.forEach(function (tag) {
130
159
  if (tagAllowed(tag) && !options.allowVulnerableTags) {
131
- console.warn(`\n\n⚠️ Your \`allowedTags\` option includes, \`${tag}\`, which is inherently\nvulnerable to XSS attacks. Please remove it from \`allowedTags\`.\nOr, to disable this warning, add the \`allowVulnerableTags\` option\nand ensure you are accounting for this risk.\n\n`);
160
+ logger.warn(
161
+ `Your \`allowedTags\` option includes \`${tag}\`, which is inherently ` +
162
+ 'vulnerable to XSS attacks. Please remove it from `allowedTags`, or, ' +
163
+ 'to disable this warning, add the `allowVulnerableTags` option and ' +
164
+ 'ensure you are accounting for this risk.'
165
+ );
132
166
  }
133
167
  });
134
168
 
@@ -222,6 +256,15 @@ function sanitizeHtml(html, options, _recursing) {
222
256
  let transformMap;
223
257
  let skipText;
224
258
  let skipTextDepth;
259
+ // Browsers (with scripting enabled) parse <noscript> content as raw text up
260
+ // to the first `</noscript`, but htmlparser2 parses it as markup, so an end
261
+ // tag for an ancestor can make htmlparser2 close the <noscript> implicitly
262
+ // much earlier. `rawTextEnd` is the source offset where the browser ends the
263
+ // <noscript> being discarded, and `skipRawText` is set while htmlparser2 has
264
+ // already closed it but the browser has not, so that we keep discarding
265
+ // until we reach that offset (GHSA-x3q4-9hxx-gx8m).
266
+ let rawTextEnd;
267
+ let skipRawText;
225
268
  let addedText = false;
226
269
 
227
270
  initializeState();
@@ -231,6 +274,7 @@ function sanitizeHtml(html, options, _recursing) {
231
274
  if (options.onOpenTag) {
232
275
  options.onOpenTag(name, attribs);
233
276
  }
277
+ updateRawTextRegion();
234
278
 
235
279
  // If `enforceHtmlBoundary` is `true` and this has found the opening
236
280
  // `html` tag, reset the state.
@@ -238,6 +282,9 @@ function sanitizeHtml(html, options, _recursing) {
238
282
  initializeState();
239
283
  }
240
284
 
285
+ if (skipRawText) {
286
+ return;
287
+ }
241
288
  if (skipText) {
242
289
  skipTextDepth++;
243
290
  return;
@@ -272,13 +319,16 @@ function sanitizeHtml(html, options, _recursing) {
272
319
  }
273
320
  }
274
321
 
275
- if (!tagAllowed(name) || (options.disallowedTagsMode === 'recursiveEscape' && !isEmptyObject(skipMap)) || (options.nestingLimit != null && depth >= options.nestingLimit)) {
322
+ if (!tagAllowed(name) || animatesUrlAttribute(name, attribs) || (options.disallowedTagsMode === 'recursiveEscape' && !isEmptyObject(skipMap)) || (options.nestingLimit != null && depth >= options.nestingLimit)) {
276
323
  skip = true;
277
324
  skipMap[depth] = true;
278
325
  if (options.disallowedTagsMode === 'discard' || options.disallowedTagsMode === 'completelyDiscard') {
279
326
  if (nonTextTagsArray.indexOf(name) !== -1) {
280
327
  skipText = true;
281
328
  skipTextDepth = 1;
329
+ if (frame.tag.toLowerCase() === 'noscript') {
330
+ rawTextEnd = findRawTextEnd('noscript', parser.endIndex + 1);
331
+ }
282
332
  }
283
333
  }
284
334
  }
@@ -378,6 +428,14 @@ function sanitizeHtml(html, options, _recursing) {
378
428
  }
379
429
  }
380
430
 
431
+ // `<meta http-equiv="refresh" content="0;url=...">` navigates to a
432
+ // URL embedded in `content`, so scheme check that URL too
433
+ // (GHSA-cv27-6wvh-8x7j). Other meta `content` values are left alone.
434
+ if (name === 'meta' && a.toLowerCase() === 'content' && isRefresh(attribs) && naughtyRefresh(value)) {
435
+ delete frame.attribs[a];
436
+ return;
437
+ }
438
+
381
439
  if (name === 'script' && a === 'src') {
382
440
 
383
441
  let allowed = true;
@@ -444,7 +502,7 @@ function sanitizeHtml(html, options, _recursing) {
444
502
  try {
445
503
  let parsed = parseSrcset(value);
446
504
  parsed.forEach(function(value) {
447
- if (naughtyHref(a, value.url)) {
505
+ if (naughtyHref(name, value.url)) {
448
506
  value.evil = true;
449
507
  }
450
508
  });
@@ -516,7 +574,14 @@ function sanitizeHtml(html, options, _recursing) {
516
574
  }
517
575
  } catch (e) {
518
576
  if (typeof window !== 'undefined') {
519
- console.warn('Failed to parse "' + name + ' {' + value + '}' + '", If you\'re running this in a browser, we recommend to disable style parsing: options.parseStyleAttributes: false, since this only works in a node environment due to a postcss dependency, More info: https://github.com/apostrophecms/sanitize-html/issues/547');
577
+ logger.warn(
578
+ `Failed to parse "${name} {${value}}". If you are ` +
579
+ 'running this in a browser, we recommend disabling ' +
580
+ 'style parsing with the parseStyleAttributes option, ' +
581
+ 'since it only works in a node environment due to a ' +
582
+ 'postcss dependency. More info: ' +
583
+ 'https://github.com/apostrophecms/sanitize-html/issues/547'
584
+ );
520
585
  }
521
586
  delete frame.attribs[a];
522
587
  return;
@@ -557,7 +622,8 @@ function sanitizeHtml(html, options, _recursing) {
557
622
  frame.openingTagLength = result.length - frame.tagPosition;
558
623
  },
559
624
  ontext: function(text) {
560
- if (skipText) {
625
+ updateRawTextRegion();
626
+ if (skipText || skipRawText) {
561
627
  return;
562
628
  }
563
629
  const lastFrame = stack[stack.length - 1];
@@ -611,6 +677,16 @@ function sanitizeHtml(html, options, _recursing) {
611
677
  // htmlparser2, so their contents are decoded and must be escaped below
612
678
  // like any other text (important to prevent XSS via entity-encoded
613
679
  // payloads such as <option>&lt;script&gt;...&lt;/script&gt;</option>).
680
+ } else if (
681
+ // htmlparser2 treats <iframe> as a raw-text element, so markup inside
682
+ // (including after an unclosed <iframe>) arrives as a single text node.
683
+ // When the iframe is discarded, re-sanitize that fallback markup as HTML
684
+ // instead of escaping it as plain text (issue #5550).
685
+ tag === 'iframe' &&
686
+ !tagAllowed(tag) &&
687
+ options.disallowedTagsMode === 'discard'
688
+ ) {
689
+ result += sanitizeHtml(text, options);
614
690
  } else if (!addedText) {
615
691
  const escaped = escapeHtml(text, false);
616
692
  if (options.textFilter) {
@@ -629,10 +705,24 @@ function sanitizeHtml(html, options, _recursing) {
629
705
  options.onCloseTag(name, isImplied);
630
706
  }
631
707
 
632
- if (skipText) {
708
+ updateRawTextRegion();
709
+ if (skipRawText) {
710
+ // Still inside the browser's raw text: only close elements that were
711
+ // opened before the discarded region, so the output stays balanced.
712
+ const lastFrame = stack[stack.length - 1];
713
+ if (!lastFrame || lastFrame.tag !== name) {
714
+ return;
715
+ }
716
+ } else if (skipText) {
633
717
  skipTextDepth--;
634
718
  if (!skipTextDepth) {
635
719
  skipText = false;
720
+ if (rawTextEnd !== null) {
721
+ // htmlparser2 closed the element implicitly (e.g. an ancestor's
722
+ // end tag) before the browser would. Close its frame below, but
723
+ // keep discarding up to the browser's end tag.
724
+ skipRawText = true;
725
+ }
636
726
  } else {
637
727
  return;
638
728
  }
@@ -735,6 +825,27 @@ function sanitizeHtml(html, options, _recursing) {
735
825
  transformMap = {};
736
826
  skipText = false;
737
827
  skipTextDepth = 0;
828
+ rawTextEnd = null;
829
+ skipRawText = false;
830
+ }
831
+
832
+ // Leave the raw text region once the parser reaches the offset where the
833
+ // browser ends it.
834
+ function updateRawTextRegion() {
835
+ if (rawTextEnd !== null && parser.startIndex >= rawTextEnd) {
836
+ rawTextEnd = null;
837
+ skipRawText = false;
838
+ }
839
+ }
840
+
841
+ // Offset of the end tag that ends a raw text element in a browser: the
842
+ // first case-insensitive `</name` followed by HTML whitespace, `/` or `>`.
843
+ // With no such end tag the element runs to the end of the input.
844
+ function findRawTextEnd(tagName, from) {
845
+ const re = new RegExp('</' + tagName + '[\\t\\n\\f\\r />]', 'ig');
846
+ re.lastIndex = from;
847
+ const match = re.exec(html);
848
+ return match ? match.index : Infinity;
738
849
  }
739
850
 
740
851
  function escapeHtml(s, quote) {
@@ -773,6 +884,127 @@ function sanitizeHtml(html, options, _recursing) {
773
884
  });
774
885
  }
775
886
 
887
+ function isRefresh(attribs) {
888
+ return Object.keys(attribs).some(function(a) {
889
+ return a.toLowerCase() === 'http-equiv' &&
890
+ String(attribs[a]).trim().toLowerCase() === 'refresh';
891
+ });
892
+ }
893
+
894
+ // True if the `content` of a `<meta http-equiv="refresh">` must be dropped:
895
+ // its destination URL fails the scheme policy, or it cannot be parsed as a
896
+ // refresh at all (a browser would ignore it then, so nothing is lost).
897
+ // Extracts the URL the way the HTML standard's "shared declarative refresh
898
+ // steps" do, so that spelling, separator, quoting and case variations of
899
+ // `url=`, or no `url=` at all, all yield the URL a browser would navigate to.
900
+ function naughtyRefresh(content) {
901
+ const input = String(content);
902
+ const isWhitespace = function(c) {
903
+ return c === ' ' || c === '\t' || c === '\n' || c === '\f' || c === '\r';
904
+ };
905
+ let position = 0;
906
+ const skipWhitespace = function() {
907
+ while (position < input.length && isWhitespace(input[position])) {
908
+ position++;
909
+ }
910
+ };
911
+ const lowerAt = function(i) {
912
+ return (input[i] || '').toLowerCase();
913
+ };
914
+ skipWhitespace();
915
+ const timeStart = position;
916
+ while (position < input.length && /[0-9.]/.test(input[position])) {
917
+ position++;
918
+ }
919
+ if (position === timeStart) {
920
+ return true;
921
+ }
922
+ if (position < input.length) {
923
+ const c = input[position];
924
+ if (c !== ';' && c !== ',' && !isWhitespace(c)) {
925
+ return true;
926
+ }
927
+ skipWhitespace();
928
+ if (input[position] === ';' || input[position] === ',') {
929
+ position++;
930
+ }
931
+ skipWhitespace();
932
+ }
933
+ if (position >= input.length) {
934
+ // No URL: refreshes the current document
935
+ return false;
936
+ }
937
+ let url = input.slice(position);
938
+ let quoted = true;
939
+ if (lowerAt(position) === 'u') {
940
+ quoted = false;
941
+ if (lowerAt(position + 1) === 'r' && lowerAt(position + 2) === 'l') {
942
+ position += 3;
943
+ skipWhitespace();
944
+ if (input[position] === '=') {
945
+ position++;
946
+ skipWhitespace();
947
+ quoted = true;
948
+ }
949
+ }
950
+ }
951
+ if (quoted) {
952
+ const quote = input[position];
953
+ if (quote === '"' || quote === '\'') {
954
+ position++;
955
+ }
956
+ url = input.slice(position);
957
+ if (quote === '"' || quote === '\'') {
958
+ const end = url.indexOf(quote);
959
+ if (end !== -1) {
960
+ url = url.slice(0, end);
961
+ }
962
+ }
963
+ }
964
+ return naughtyHref('meta', url);
965
+ }
966
+
967
+ // True if this is an SVG SMIL animation element that animates a URL-bearing
968
+ // attribute, e.g. `<animate attributeName="href" values="#safe;javascript:...">`.
969
+ //
970
+ // Such an element carries no URL of its own: the browser copies the animation
971
+ // values into the target attribute *after* sanitization, so a `javascript:`
972
+ // destination reaches a live link sink without ever being scheme checked.
973
+ // `values` compounds this, because it is a semicolon-separated LIST of
974
+ // destinations: checking it as one flat URL only validates its first entry, so
975
+ // a leading `#safe` fragment carries the rest of the list past the policy.
976
+ //
977
+ // Re-checking each entry of each value attribute would leave the safety of the
978
+ // output resting on our imitation of SMIL list parsing, so we reject the
979
+ // animation on the strength of its target instead. Animations of attributes
980
+ // that are not URL sinks, such as `fill` or `opacity`, are unaffected.
981
+ function animatesUrlAttribute(name, attribs) {
982
+ // In an XML serialization a prefixed name such as `svg:animate` is the same
983
+ // element as `animate`, so match on the local name (GHSA-374f-7chj-9948).
984
+ if (svgAnimationTags.indexOf(localPart(name)) === -1) {
985
+ return false;
986
+ }
987
+ const schemeCheckedAttributes = options.allowedSchemesAppliedToAttributes || [];
988
+ return Object.keys(attribs || {}).some(function(attributeName) {
989
+ if (localPart(attributeName) !== 'attributename') {
990
+ return false;
991
+ }
992
+ const target = (attribs[attributeName] || '').trim().toLowerCase();
993
+ // The target may be namespace prefixed (`xlink:href`). Which prefixes are
994
+ // in scope depends on the document, so consider the local name too.
995
+ const localName = target.slice(target.lastIndexOf(':') + 1);
996
+ return alwaysUrlAttributes.indexOf(localName) !== -1 ||
997
+ schemeCheckedAttributes.indexOf(target) !== -1 ||
998
+ schemeCheckedAttributes.indexOf(localName) !== -1;
999
+ });
1000
+ }
1001
+
1002
+ // Lowercased name with any namespace prefix removed.
1003
+ function localPart(name) {
1004
+ const lower = name.toLowerCase();
1005
+ return lower.slice(lower.lastIndexOf(':') + 1);
1006
+ }
1007
+
776
1008
  function parseUrl(value) {
777
1009
  value = value.replace(/^(\w+:)?\s*[\\/]\s*[\\/]/, '$1//');
778
1010
  if (value.startsWith('relative:')) {
package/package.json CHANGED
@@ -1,12 +1,16 @@
1
1
  {
2
2
  "name": "sanitize-html",
3
- "version": "2.17.6",
3
+ "version": "2.18.0",
4
4
  "description": "Clean up user-submitted HTML, preserving allowlisted elements and allowlisted attributes on a per-element basis",
5
5
  "sideEffects": false,
6
6
  "main": "index.js",
7
7
  "files": [
8
8
  "index.js"
9
9
  ],
10
+ "scripts": {
11
+ "test": "npm run lint && mocha",
12
+ "lint": "eslint ."
13
+ },
10
14
  "repository": {
11
15
  "type": "git",
12
16
  "url": "https://github.com/apostrophecms/apostrophe.git",
@@ -25,25 +29,21 @@
25
29
  "node": ">=22.12.0"
26
30
  },
27
31
  "dependencies": {
28
- "deepmerge": "^4.2.2",
32
+ "deepmerge": "^4.3.1",
29
33
  "escape-string-regexp": "^4.0.0",
30
34
  "htmlparser2": "^12.0.0",
31
- "is-plain-object": "^5.0.0",
35
+ "is-plain-object": "^5.1.0",
36
+ "launder": "^1.7.2",
32
37
  "parse-srcset": "^1.0.2",
33
- "postcss": "^8.3.11",
34
- "launder": "^1.7.1"
38
+ "postcss": "^8.5.27"
35
39
  },
36
40
  "devDependencies": {
37
- "eslint": "^9.39.1",
38
- "mocha": "^11.7.5",
39
- "sinon": "^9.0.2",
40
- "eslint-config-apostrophe": "^6.0.2"
41
+ "eslint": "^9.39.5",
42
+ "eslint-config-apostrophe": "^6.1.0",
43
+ "mocha": "^11.8.0",
44
+ "sinon": "^9.2.4"
41
45
  },
42
46
  "apostropheTestConfig": {
43
47
  "requiresMongo": false
44
- },
45
- "scripts": {
46
- "test": "npm run lint && mocha",
47
- "lint": "eslint ."
48
48
  }
49
- }
49
+ }