sanitize-html 2.17.7 → 2.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (3) hide show
  1. package/README.md +14 -0
  2. package/index.js +195 -7
  3. package/package.json +14 -14
package/README.md CHANGED
@@ -810,6 +810,20 @@ nestingLimit: 6
810
810
 
811
811
  This will prevent the user from nesting tags more than 6 levels deep. Tags deeper than that are stripped out exactly as if they were disallowed. Note that this means text is preserved in the usual ways where appropriate.
812
812
 
813
+ ### Routing warnings to your own logger
814
+
815
+ sanitize-html writes its own diagnostics - the vulnerable tag notice above, and the style parsing notice in the browser - to the console. If your application has a logging pipeline of its own, pass any console-shaped object as the `logger` option and they are delivered to it instead:
816
+
817
+ ```javascript
818
+ sanitizeHtml(dirty, {
819
+ logger: myLogger // an object with debug, info, warn and error methods
820
+ });
821
+ ```
822
+
823
+ Any of the four methods that your object does not provide falls back to the console, and without the option at all the console is still the destination, exactly as before.
824
+
825
+ This matters for applications that write structured logs: a raw `console.warn` bypasses their filtering and formatting, and puts unparseable text on a stream that is expected to be one JSON object per line.
826
+
813
827
  ### Advanced filtering
814
828
 
815
829
  For more advanced filtering you can hook directly into the parsing process using tag open and tag close events.
package/index.js CHANGED
@@ -24,6 +24,22 @@ const svgAnimationTags = [
24
24
  // `href` are covered.
25
25
  const alwaysUrlAttributes = [ 'href' ];
26
26
 
27
+ // Our own diagnostics. A console-shaped `logger` option takes them instead of
28
+ // the console; missing methods fall back to it.
29
+ const severities = [ 'debug', 'info', 'warn', 'error' ];
30
+
31
+ function loggerFor(options) {
32
+ const source = (options && options.logger) || console;
33
+ const logger = {};
34
+ for (const severity of severities) {
35
+ logger[severity] = typeof source[severity] === 'function'
36
+ ? (...args) => source[severity](...args)
37
+ // eslint-disable-next-line no-console
38
+ : (...args) => console[severity](...args);
39
+ }
40
+ return logger;
41
+ }
42
+
27
43
  function each(obj, cb) {
28
44
  if (obj) {
29
45
  Object.keys(obj).forEach(function (key) {
@@ -131,6 +147,8 @@ function sanitizeHtml(html, options, _recursing) {
131
147
  options = Object.assign({}, sanitizeHtml.defaults, options);
132
148
  options.parser = Object.assign({}, htmlParserDefaults, options.parser);
133
149
 
150
+ const logger = loggerFor(options);
151
+
134
152
  const tagAllowed = function (name) {
135
153
  return options.allowedTags === false ||
136
154
  (options.allowedTags || []).indexOf(name) > -1;
@@ -139,7 +157,12 @@ function sanitizeHtml(html, options, _recursing) {
139
157
  // vulnerableTags
140
158
  vulnerableTags.forEach(function (tag) {
141
159
  if (tagAllowed(tag) && !options.allowVulnerableTags) {
142
- console.warn(`\n\n⚠️ Your \`allowedTags\` option includes, \`${tag}\`, which is inherently\nvulnerable to XSS attacks. Please remove it from \`allowedTags\`.\nOr, to disable this warning, add the \`allowVulnerableTags\` option\nand ensure you are accounting for this risk.\n\n`);
160
+ logger.warn(
161
+ `Your \`allowedTags\` option includes \`${tag}\`, which is inherently ` +
162
+ 'vulnerable to XSS attacks. Please remove it from `allowedTags`, or, ' +
163
+ 'to disable this warning, add the `allowVulnerableTags` option and ' +
164
+ 'ensure you are accounting for this risk.'
165
+ );
143
166
  }
144
167
  });
145
168
 
@@ -233,6 +256,15 @@ function sanitizeHtml(html, options, _recursing) {
233
256
  let transformMap;
234
257
  let skipText;
235
258
  let skipTextDepth;
259
+ // Browsers (with scripting enabled) parse <noscript> content as raw text up
260
+ // to the first `</noscript`, but htmlparser2 parses it as markup, so an end
261
+ // tag for an ancestor can make htmlparser2 close the <noscript> implicitly
262
+ // much earlier. `rawTextEnd` is the source offset where the browser ends the
263
+ // <noscript> being discarded, and `skipRawText` is set while htmlparser2 has
264
+ // already closed it but the browser has not, so that we keep discarding
265
+ // until we reach that offset (GHSA-x3q4-9hxx-gx8m).
266
+ let rawTextEnd;
267
+ let skipRawText;
236
268
  let addedText = false;
237
269
 
238
270
  initializeState();
@@ -242,6 +274,7 @@ function sanitizeHtml(html, options, _recursing) {
242
274
  if (options.onOpenTag) {
243
275
  options.onOpenTag(name, attribs);
244
276
  }
277
+ updateRawTextRegion();
245
278
 
246
279
  // If `enforceHtmlBoundary` is `true` and this has found the opening
247
280
  // `html` tag, reset the state.
@@ -249,6 +282,9 @@ function sanitizeHtml(html, options, _recursing) {
249
282
  initializeState();
250
283
  }
251
284
 
285
+ if (skipRawText) {
286
+ return;
287
+ }
252
288
  if (skipText) {
253
289
  skipTextDepth++;
254
290
  return;
@@ -290,6 +326,9 @@ function sanitizeHtml(html, options, _recursing) {
290
326
  if (nonTextTagsArray.indexOf(name) !== -1) {
291
327
  skipText = true;
292
328
  skipTextDepth = 1;
329
+ if (frame.tag.toLowerCase() === 'noscript') {
330
+ rawTextEnd = findRawTextEnd('noscript', parser.endIndex + 1);
331
+ }
293
332
  }
294
333
  }
295
334
  }
@@ -389,6 +428,14 @@ function sanitizeHtml(html, options, _recursing) {
389
428
  }
390
429
  }
391
430
 
431
+ // `<meta http-equiv="refresh" content="0;url=...">` navigates to a
432
+ // URL embedded in `content`, so scheme check that URL too
433
+ // (GHSA-cv27-6wvh-8x7j). Other meta `content` values are left alone.
434
+ if (name === 'meta' && a.toLowerCase() === 'content' && isRefresh(attribs) && naughtyRefresh(value)) {
435
+ delete frame.attribs[a];
436
+ return;
437
+ }
438
+
392
439
  if (name === 'script' && a === 'src') {
393
440
 
394
441
  let allowed = true;
@@ -455,7 +502,7 @@ function sanitizeHtml(html, options, _recursing) {
455
502
  try {
456
503
  let parsed = parseSrcset(value);
457
504
  parsed.forEach(function(value) {
458
- if (naughtyHref(a, value.url)) {
505
+ if (naughtyHref(name, value.url)) {
459
506
  value.evil = true;
460
507
  }
461
508
  });
@@ -527,7 +574,14 @@ function sanitizeHtml(html, options, _recursing) {
527
574
  }
528
575
  } catch (e) {
529
576
  if (typeof window !== 'undefined') {
530
- console.warn('Failed to parse "' + name + ' {' + value + '}' + '", If you\'re running this in a browser, we recommend to disable style parsing: options.parseStyleAttributes: false, since this only works in a node environment due to a postcss dependency, More info: https://github.com/apostrophecms/sanitize-html/issues/547');
577
+ logger.warn(
578
+ `Failed to parse "${name} {${value}}". If you are ` +
579
+ 'running this in a browser, we recommend disabling ' +
580
+ 'style parsing with the parseStyleAttributes option, ' +
581
+ 'since it only works in a node environment due to a ' +
582
+ 'postcss dependency. More info: ' +
583
+ 'https://github.com/apostrophecms/sanitize-html/issues/547'
584
+ );
531
585
  }
532
586
  delete frame.attribs[a];
533
587
  return;
@@ -568,7 +622,8 @@ function sanitizeHtml(html, options, _recursing) {
568
622
  frame.openingTagLength = result.length - frame.tagPosition;
569
623
  },
570
624
  ontext: function(text) {
571
- if (skipText) {
625
+ updateRawTextRegion();
626
+ if (skipText || skipRawText) {
572
627
  return;
573
628
  }
574
629
  const lastFrame = stack[stack.length - 1];
@@ -622,6 +677,16 @@ function sanitizeHtml(html, options, _recursing) {
622
677
  // htmlparser2, so their contents are decoded and must be escaped below
623
678
  // like any other text (important to prevent XSS via entity-encoded
624
679
  // payloads such as <option>&lt;script&gt;...&lt;/script&gt;</option>).
680
+ } else if (
681
+ // htmlparser2 treats <iframe> as a raw-text element, so markup inside
682
+ // (including after an unclosed <iframe>) arrives as a single text node.
683
+ // When the iframe is discarded, re-sanitize that fallback markup as HTML
684
+ // instead of escaping it as plain text (issue #5550).
685
+ tag === 'iframe' &&
686
+ !tagAllowed(tag) &&
687
+ options.disallowedTagsMode === 'discard'
688
+ ) {
689
+ result += sanitizeHtml(text, options);
625
690
  } else if (!addedText) {
626
691
  const escaped = escapeHtml(text, false);
627
692
  if (options.textFilter) {
@@ -640,10 +705,24 @@ function sanitizeHtml(html, options, _recursing) {
640
705
  options.onCloseTag(name, isImplied);
641
706
  }
642
707
 
643
- if (skipText) {
708
+ updateRawTextRegion();
709
+ if (skipRawText) {
710
+ // Still inside the browser's raw text: only close elements that were
711
+ // opened before the discarded region, so the output stays balanced.
712
+ const lastFrame = stack[stack.length - 1];
713
+ if (!lastFrame || lastFrame.tag !== name) {
714
+ return;
715
+ }
716
+ } else if (skipText) {
644
717
  skipTextDepth--;
645
718
  if (!skipTextDepth) {
646
719
  skipText = false;
720
+ if (rawTextEnd !== null) {
721
+ // htmlparser2 closed the element implicitly (e.g. an ancestor's
722
+ // end tag) before the browser would. Close its frame below, but
723
+ // keep discarding up to the browser's end tag.
724
+ skipRawText = true;
725
+ }
647
726
  } else {
648
727
  return;
649
728
  }
@@ -746,6 +825,27 @@ function sanitizeHtml(html, options, _recursing) {
746
825
  transformMap = {};
747
826
  skipText = false;
748
827
  skipTextDepth = 0;
828
+ rawTextEnd = null;
829
+ skipRawText = false;
830
+ }
831
+
832
+ // Leave the raw text region once the parser reaches the offset where the
833
+ // browser ends it.
834
+ function updateRawTextRegion() {
835
+ if (rawTextEnd !== null && parser.startIndex >= rawTextEnd) {
836
+ rawTextEnd = null;
837
+ skipRawText = false;
838
+ }
839
+ }
840
+
841
+ // Offset of the end tag that ends a raw text element in a browser: the
842
+ // first case-insensitive `</name` followed by HTML whitespace, `/` or `>`.
843
+ // With no such end tag the element runs to the end of the input.
844
+ function findRawTextEnd(tagName, from) {
845
+ const re = new RegExp('</' + tagName + '[\\t\\n\\f\\r />]', 'ig');
846
+ re.lastIndex = from;
847
+ const match = re.exec(html);
848
+ return match ? match.index : Infinity;
749
849
  }
750
850
 
751
851
  function escapeHtml(s, quote) {
@@ -784,6 +884,86 @@ function sanitizeHtml(html, options, _recursing) {
784
884
  });
785
885
  }
786
886
 
887
+ function isRefresh(attribs) {
888
+ return Object.keys(attribs).some(function(a) {
889
+ return a.toLowerCase() === 'http-equiv' &&
890
+ String(attribs[a]).trim().toLowerCase() === 'refresh';
891
+ });
892
+ }
893
+
894
+ // True if the `content` of a `<meta http-equiv="refresh">` must be dropped:
895
+ // its destination URL fails the scheme policy, or it cannot be parsed as a
896
+ // refresh at all (a browser would ignore it then, so nothing is lost).
897
+ // Extracts the URL the way the HTML standard's "shared declarative refresh
898
+ // steps" do, so that spelling, separator, quoting and case variations of
899
+ // `url=`, or no `url=` at all, all yield the URL a browser would navigate to.
900
+ function naughtyRefresh(content) {
901
+ const input = String(content);
902
+ const isWhitespace = function(c) {
903
+ return c === ' ' || c === '\t' || c === '\n' || c === '\f' || c === '\r';
904
+ };
905
+ let position = 0;
906
+ const skipWhitespace = function() {
907
+ while (position < input.length && isWhitespace(input[position])) {
908
+ position++;
909
+ }
910
+ };
911
+ const lowerAt = function(i) {
912
+ return (input[i] || '').toLowerCase();
913
+ };
914
+ skipWhitespace();
915
+ const timeStart = position;
916
+ while (position < input.length && /[0-9.]/.test(input[position])) {
917
+ position++;
918
+ }
919
+ if (position === timeStart) {
920
+ return true;
921
+ }
922
+ if (position < input.length) {
923
+ const c = input[position];
924
+ if (c !== ';' && c !== ',' && !isWhitespace(c)) {
925
+ return true;
926
+ }
927
+ skipWhitespace();
928
+ if (input[position] === ';' || input[position] === ',') {
929
+ position++;
930
+ }
931
+ skipWhitespace();
932
+ }
933
+ if (position >= input.length) {
934
+ // No URL: refreshes the current document
935
+ return false;
936
+ }
937
+ let url = input.slice(position);
938
+ let quoted = true;
939
+ if (lowerAt(position) === 'u') {
940
+ quoted = false;
941
+ if (lowerAt(position + 1) === 'r' && lowerAt(position + 2) === 'l') {
942
+ position += 3;
943
+ skipWhitespace();
944
+ if (input[position] === '=') {
945
+ position++;
946
+ skipWhitespace();
947
+ quoted = true;
948
+ }
949
+ }
950
+ }
951
+ if (quoted) {
952
+ const quote = input[position];
953
+ if (quote === '"' || quote === '\'') {
954
+ position++;
955
+ }
956
+ url = input.slice(position);
957
+ if (quote === '"' || quote === '\'') {
958
+ const end = url.indexOf(quote);
959
+ if (end !== -1) {
960
+ url = url.slice(0, end);
961
+ }
962
+ }
963
+ }
964
+ return naughtyHref('meta', url);
965
+ }
966
+
787
967
  // True if this is an SVG SMIL animation element that animates a URL-bearing
788
968
  // attribute, e.g. `<animate attributeName="href" values="#safe;javascript:...">`.
789
969
  //
@@ -799,12 +979,14 @@ function sanitizeHtml(html, options, _recursing) {
799
979
  // animation on the strength of its target instead. Animations of attributes
800
980
  // that are not URL sinks, such as `fill` or `opacity`, are unaffected.
801
981
  function animatesUrlAttribute(name, attribs) {
802
- if (svgAnimationTags.indexOf(name.toLowerCase()) === -1) {
982
+ // In an XML serialization a prefixed name such as `svg:animate` is the same
983
+ // element as `animate`, so match on the local name (GHSA-374f-7chj-9948).
984
+ if (svgAnimationTags.indexOf(localPart(name)) === -1) {
803
985
  return false;
804
986
  }
805
987
  const schemeCheckedAttributes = options.allowedSchemesAppliedToAttributes || [];
806
988
  return Object.keys(attribs || {}).some(function(attributeName) {
807
- if (attributeName.toLowerCase() !== 'attributename') {
989
+ if (localPart(attributeName) !== 'attributename') {
808
990
  return false;
809
991
  }
810
992
  const target = (attribs[attributeName] || '').trim().toLowerCase();
@@ -817,6 +999,12 @@ function sanitizeHtml(html, options, _recursing) {
817
999
  });
818
1000
  }
819
1001
 
1002
+ // Lowercased name with any namespace prefix removed.
1003
+ function localPart(name) {
1004
+ const lower = name.toLowerCase();
1005
+ return lower.slice(lower.lastIndexOf(':') + 1);
1006
+ }
1007
+
820
1008
  function parseUrl(value) {
821
1009
  value = value.replace(/^(\w+:)?\s*[\\/]\s*[\\/]/, '$1//');
822
1010
  if (value.startsWith('relative:')) {
package/package.json CHANGED
@@ -1,12 +1,16 @@
1
1
  {
2
2
  "name": "sanitize-html",
3
- "version": "2.17.7",
3
+ "version": "2.18.0",
4
4
  "description": "Clean up user-submitted HTML, preserving allowlisted elements and allowlisted attributes on a per-element basis",
5
5
  "sideEffects": false,
6
6
  "main": "index.js",
7
7
  "files": [
8
8
  "index.js"
9
9
  ],
10
+ "scripts": {
11
+ "test": "npm run lint && mocha",
12
+ "lint": "eslint ."
13
+ },
10
14
  "repository": {
11
15
  "type": "git",
12
16
  "url": "https://github.com/apostrophecms/apostrophe.git",
@@ -25,25 +29,21 @@
25
29
  "node": ">=22.12.0"
26
30
  },
27
31
  "dependencies": {
28
- "deepmerge": "^4.2.2",
32
+ "deepmerge": "^4.3.1",
29
33
  "escape-string-regexp": "^4.0.0",
30
34
  "htmlparser2": "^12.0.0",
31
- "is-plain-object": "^5.0.0",
35
+ "is-plain-object": "^5.1.0",
36
+ "launder": "^1.7.2",
32
37
  "parse-srcset": "^1.0.2",
33
- "postcss": "^8.3.11",
34
- "launder": "^1.7.1"
38
+ "postcss": "^8.5.27"
35
39
  },
36
40
  "devDependencies": {
37
- "eslint": "^9.39.1",
38
- "mocha": "^11.7.5",
39
- "sinon": "^9.0.2",
40
- "eslint-config-apostrophe": "^6.0.2"
41
+ "eslint": "^9.39.5",
42
+ "eslint-config-apostrophe": "^6.1.0",
43
+ "mocha": "^11.8.0",
44
+ "sinon": "^9.2.4"
41
45
  },
42
46
  "apostropheTestConfig": {
43
47
  "requiresMongo": false
44
- },
45
- "scripts": {
46
- "test": "npm run lint && mocha",
47
- "lint": "eslint ."
48
48
  }
49
- }
49
+ }