html-minifier-next 7.6.0 → 8.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/htmlparser.js CHANGED
@@ -5,7 +5,7 @@
5
5
  * http://erik.eae.net/simplehtmlparser/simplehtmlparser.js
6
6
  */
7
7
 
8
- import { isThenable } from './lib/utils.js';
8
+ import { isThenable, embedSource } from './lib/utils.js';
9
9
 
10
10
  /** @import { HTMLAttribute } from './lib/attributes.js' */
11
11
 
@@ -141,9 +141,9 @@ function buildAttrRegex(handler) {
141
141
  throw new Error('`customAttrSurround` entries must be `[RegExp, RegExp]` pairs');
142
142
  }
143
143
  attrClauses[i] = '(?:' +
144
- '(' + pair[0].source + ')\\s*' +
144
+ '(' + embedSource(pair[0]) + ')\\s*' +
145
145
  pattern +
146
- '\\s*(' + pair[1].source + ')' +
146
+ '\\s*(' + embedSource(pair[1]) + ')' +
147
147
  ')';
148
148
  }
149
149
  attrClauses.push('(?:' + pattern + ')');
@@ -180,7 +180,7 @@ function joinSingleAttrAssigns(handler) {
180
180
  return singleAttrAssigns.concat(
181
181
  handler.customAttrAssign || []
182
182
  ).map(function (assign) {
183
- return '(?:' + assign.source + ')';
183
+ return '(?:' + embedSource(assign) + ')';
184
184
  }).join('|');
185
185
  }
186
186
 
@@ -13,10 +13,12 @@ import {
13
13
  isBooleanValue,
14
14
  collapsibleValues,
15
15
  srcsetElements,
16
- reEmptyAttribute
16
+ RE_EMPTY_ATTRIBUTE,
17
+ RE_STYLE_ELEMENT
17
18
  } from './constants.js';
18
19
  import { trimWhitespace, collapseWhitespaceAll } from './whitespace.js';
19
20
  import { shouldMinifyInnerHTML } from './options.js';
21
+ import { collectUsedSymbols } from './unused-css.js';
20
22
  import { identity, isThenable } from './utils.js';
21
23
 
22
24
  /** @import { ProcessedOptions } from './options.js' */
@@ -25,10 +27,11 @@ import { identity, isThenable } from './utils.js';
25
27
 
26
28
  /**
27
29
  * @typedef {{ name: string, value?: string | undefined, quote?: string, customAssign?: string, customOpen?: string, customClose?: string }} HTMLAttribute
28
- * Internal counterpart of the public typedef in `htmlminifier.js`—keep in sync.
30
+ * Internal counterpart of the public typedef in htmlminifier.js—keep in sync.
29
31
  */
30
32
 
31
- // Lazy-load entities (used for `decodeEntities` and event-handler attribute decode before `minifyJS`)
33
+ // Lazy-load entities (used for `decodeEntities`, event-handler attribute
34
+ // decoding before `minifyJS`, and `srcdoc` decoding/re-encoding)
32
35
 
33
36
  /** @type {Promise<Function> | undefined} */
34
37
  let decodeHTMLStrictPromise;
@@ -39,6 +42,15 @@ async function getDecodeHTMLStrict() {
39
42
  return decodeHTMLStrictPromise;
40
43
  }
41
44
 
45
+ /** @type {Promise<Function> | undefined} */
46
+ let escapeAttributePromise;
47
+ async function getEscapeAttribute() {
48
+ if (!escapeAttributePromise) {
49
+ escapeAttributePromise = import('entities').then(m => m.escapeAttribute);
50
+ }
51
+ return escapeAttributePromise;
52
+ }
53
+
42
54
  // Validators
43
55
 
44
56
  /**
@@ -353,7 +365,7 @@ function canDeleteEmptyAttribute(tag, attrName, attrValue, options) {
353
365
  if (typeof options.removeEmptyAttributes === 'function') {
354
366
  return options.removeEmptyAttributes(attrName, tag);
355
367
  }
356
- return (tag === 'input' && attrName === 'value') || reEmptyAttribute.test(attrName);
368
+ return (tag === 'input' && attrName === 'value') || RE_EMPTY_ATTRIBUTE.test(attrName);
357
369
  }
358
370
 
359
371
  /**
@@ -593,17 +605,62 @@ function cleanAttributeValue(tag, attrName, attrValue, options, attrs, minifyHTM
593
605
  }
594
606
 
595
607
  if (tag === 'iframe' && attrName === 'srcdoc') {
596
- // Recursively minify HTML content within `srcdoc` attribute
597
608
  // Fast-path: Skip if nothing would change
598
609
  if (!shouldMinifyInnerHTML(options)) {
599
610
  return attrValue;
600
611
  }
601
- return minifyHTMLSelf(attrValue, options, true);
612
+ return minifySrcdoc(attrValue, options, minifyHTMLSelf);
602
613
  }
603
614
 
604
615
  return attrValue;
605
616
  }
606
617
 
618
+ /**
619
+ * Recursively minify the document an `iframe srcdoc` attribute holds.
620
+ *
621
+ * Browsers resolve character references before parsing `srcdoc`, so an
622
+ * entity-encoded document is decoded first (unless `decodeEntities` already
623
+ * did) and re-encoded afterwards—but only then, so a literal value passes
624
+ * through byte-identical.
625
+ *
626
+ * @param {string} attrValue
627
+ * @param {ProcessedOptions} options
628
+ * @param {Function} minifyHTMLSelf
629
+ * @returns {Promise<string>}
630
+ */
631
+ async function minifySrcdoc(attrValue, options, minifyHTMLSelf) {
632
+ let markup = attrValue;
633
+ let wasEncoded = false;
634
+ if (!options.decodeEntities && attrValue.indexOf('&') !== -1) {
635
+ const decode = await getDecodeHTMLStrict();
636
+ markup = decode(attrValue);
637
+ wasEncoded = markup !== attrValue;
638
+ }
639
+
640
+ let srcdocOptions = options;
641
+ // The inner document sees none of the parent’s markup, so its style sheets
642
+ // minify against the symbols the `srcdoc` content references itself
643
+ if (options.removeUnusedCSS && RE_STYLE_ELEMENT.test(markup)) {
644
+ const decode = markup.indexOf('&') !== -1 ? await getDecodeHTMLStrict() : undefined;
645
+ srcdocOptions = {
646
+ ...options,
647
+ cssContext: {
648
+ warned: options.cssContext ? options.cssContext.warned : new Set(),
649
+ usedSymbols: collectUsedSymbols(markup, options.removeUnusedCSS.scripts, /** @type {((text: string) => string) | undefined} */ (decode))
650
+ }
651
+ };
652
+ }
653
+
654
+ try {
655
+ const minified = await minifyHTMLSelf(markup, srcdocOptions, true);
656
+ return wasEncoded ? (await getEscapeAttribute())(minified) : minified;
657
+ } catch (err) {
658
+ if (!options.continueOnMinifyError) throw err;
659
+ options.log && options.log(/** @type {Error} */ (err));
660
+ return attrValue;
661
+ }
662
+ }
663
+
607
664
  /**
608
665
  * Choose appropriate quote character for an attribute value
609
666
  * @param {string} attrValue - The attribute value
@@ -17,6 +17,9 @@ const RE_ATTR_WS_CHECK = /[ \n\r\t\f]/;
17
17
  const RE_ATTR_WS_COLLAPSE = /[ \n\r\t\f]+/g;
18
18
  const RE_ATTR_WS_TRIM = /^[ \n\r\t\f]+|[ \n\r\t\f]+$/g;
19
19
  const RE_STYLE_ELEMENT = /<style[\s/>]/i;
20
+ const RE_EMPTY_ATTRIBUTE = new RegExp(
21
+ '^(?:class|id|style|title|lang|dir|on(?:focus|blur|change|click|dblclick|mouse(' +
22
+ '?:down|up|over|move|out)|key(?:press|down|up)))$');
20
23
 
21
24
  // Inline element sets for whitespace handling
22
25
 
@@ -163,12 +166,6 @@ const trailingElements = new Set(['dt', 'thead']);
163
166
 
164
167
  const htmlElements = new Set(['a', 'abbr', 'acronym', 'address', 'applet', 'area', 'article', 'aside', 'audio', 'b', 'base', 'basefont', 'bdi', 'bdo', 'bgsound', 'big', 'blink', 'blockquote', 'body', 'br', 'button', 'canvas', 'caption', 'center', 'cite', 'code', 'col', 'colgroup', 'command', 'content', 'data', 'datalist', 'dd', 'del', 'details', 'dfn', 'dialog', 'dir', 'div', 'dl', 'dt', 'element', 'em', 'embed', 'fieldset', 'figcaption', 'figure', 'font', 'footer', 'form', 'frame', 'frameset', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'head', 'header', 'hgroup', 'hr', 'html', 'i', 'iframe', 'image', 'img', 'input', 'ins', 'isindex', 'kbd', 'keygen', 'label', 'legend', 'li', 'link', 'listing', 'main', 'map', 'mark', 'marquee', 'menu', 'menuitem', 'meta', 'meter', 'multicol', 'nav', 'nobr', 'noembed', 'noframes', 'noscript', 'object', 'ol', 'optgroup', 'option', 'output', 'p', 'param', 'picture', 'plaintext', 'pre', 'progress', 'q', 'rb', 'rp', 'rt', 'rtc', 'ruby', 's', 'samp', 'script', 'search', 'section', 'select', 'selectedcontent', 'shadow', 'small', 'source', 'spacer', 'span', 'strike', 'strong', 'style', 'sub', 'summary', 'sup', 'table', 'tbody', 'td', 'template', 'textarea', 'tfoot', 'th', 'thead', 'time', 'title', 'tr', 'track', 'tt', 'u', 'ul', 'var', 'video', 'wbr', 'xmp']);
165
168
 
166
- // Empty attribute regex
167
-
168
- const reEmptyAttribute = new RegExp(
169
- '^(?:class|id|style|title|lang|dir|on(?:focus|blur|change|click|dblclick|mouse(' +
170
- '?:down|up|over|move|out)|key(?:press|down|up)))$');
171
-
172
169
  // Special content elements
173
170
 
174
171
  const specialContentElements = new Set(['script', 'style']);
@@ -194,6 +191,8 @@ export {
194
191
  RE_ATTR_WS_COLLAPSE,
195
192
  RE_ATTR_WS_TRIM,
196
193
  RE_STYLE_ELEMENT,
194
+ RE_EMPTY_ATTRIBUTE,
195
+
197
196
  // Inline element sets
198
197
  inlineElementsToKeepWhitespaceAround,
199
198
  inlineElementsToKeepWhitespaceWithin,
@@ -236,7 +235,6 @@ export {
236
235
  trailingElements,
237
236
  htmlElements,
238
237
 
239
- // Regex
240
- reEmptyAttribute,
238
+ // Special content elements
241
239
  specialContentElements
242
240
  };
@@ -0,0 +1,274 @@
1
+ /**
2
+ * Custom fragment matching
3
+ *
4
+ * `ignoreCustomFragments` patterns nearly always describe the same shape: a literal
5
+ * opening delimiter, an any-character or negated-class body, and a literal closing
6
+ * delimiter, as in `<%[\s\S]*?%>` or `\{\{[^}]*?\}\}`. Such a fragment can be found
7
+ * with `indexOf` in linear time, where running the pattern as a regex costs O(n²)
8
+ * on input that opens fragments it never closes—and can cost far more than that when
9
+ * the pattern itself backtracks. Patterns of other shapes keep running as regexes,
10
+ * one per pattern, so each keeps its own flags.
11
+ */
12
+
13
+ const RE_WHITESPACE = /\s/;
14
+
15
+ // Any-character and negated-class bodies; without the `s` flag, `.` is itself a
16
+ // negated class, excluding line terminators
17
+ const RE_DELIMITED = /^(.*?)(?:\[\\s\\S\]|\[\\S\\s\]|\[\^\]|(\.)|\[\^((?:\\[^]|[^\]\\])+)\])(?:([*+])|\{(\d+)(?:,(\d*))?\})\?(.*)$/;
18
+
19
+ // Flags that leave literal matching alone, so a pattern carrying them can still be
20
+ // scanned for; `i` in particular cannot, since case folding moves character indexes
21
+ const RE_LITERAL_FLAGS = /^[gds]*$/;
22
+
23
+ /**
24
+ * @typedef {{open: string, close: string, min: number, max: number, excluded: RegExp | null}} DelimitedFragment
25
+ * Literal delimiters around a body, found by scanning; `excluded` holds the
26
+ * characters a negated-class body cannot cross, null when the body spans every
27
+ * character
28
+ * @typedef {{search: RegExp, anchored: RegExp}} PatternFragment
29
+ * Everything else, found by running the pattern itself
30
+ */
31
+
32
+ /**
33
+ * Read a regex source as a literal string, so it can be matched with `indexOf`
34
+ * @param {string} source
35
+ * @returns {string | null} Null when the source is more than literal characters
36
+ */
37
+ function toLiteral(source) {
38
+ let literal = '';
39
+
40
+ for (let i = 0; i < source.length; i++) {
41
+ const char = source[i] ?? '';
42
+ if (char === '\\') {
43
+ const escaped = source[++i];
44
+ // A trailing backslash is half an escape the body token was split out of,
45
+ // as in `<%\.*?%>`, where the `.` is literal and not a body
46
+ if (escaped === undefined) return null;
47
+ // `\n` and friends are literal characters, `\s` and `\1` are not
48
+ if (escaped === 'n') literal += '\n';
49
+ else if (escaped === 't') literal += '\t';
50
+ else if (escaped === 'r') literal += '\r';
51
+ else if (escaped === 'f') literal += '\f';
52
+ else if (escaped === 'v') literal += '\v';
53
+ else if (/[A-Za-z0-9]/.test(escaped)) return null;
54
+ else literal += escaped;
55
+ } else if ('.*+?()[]{}|^$'.includes(char)) {
56
+ return null;
57
+ } else {
58
+ literal += char;
59
+ }
60
+ }
61
+
62
+ return literal;
63
+ }
64
+
65
+ /**
66
+ * Describe a fragment pattern as literal delimiters around an any-character body
67
+ * @param {RegExp} pattern
68
+ * @returns {DelimitedFragment | null} Null when the pattern has another shape, which
69
+ * the caller answers by running it as a regex
70
+ */
71
+ function toDelimitedFragment(pattern) {
72
+ if (!RE_LITERAL_FLAGS.test(pattern.flags)) return null;
73
+
74
+ const match = RE_DELIMITED.exec(pattern.source);
75
+ if (!match) return null;
76
+
77
+ const [, rawOpen, dot, negated, simple, exact, upper, rawClose] = match;
78
+ const open = toLiteral(rawOpen ?? '');
79
+ const close = toLiteral(rawClose ?? '');
80
+ // Both delimiters have to be there: without them a match has no boundary to scan to
81
+ if (!open || !close) return null;
82
+
83
+ // The class content doubles as the search for characters the body cannot cross;
84
+ // a `.` body spans every character under the `s` flag, and excludes line
85
+ // terminators without it
86
+ const inner = dot ? (pattern.flags.includes('s') ? null : '\\n\\r\\u2028\\u2029') : negated ?? null;
87
+
88
+ const min = simple ? (simple === '+' ? 1 : 0) : Number(exact);
89
+ const max = simple || upper === '' ? Infinity : Number(upper ?? exact);
90
+
91
+ return { open, close, min, max, excluded: inner === null ? null : new RegExp('[' + inner + ']', 'g') };
92
+ }
93
+
94
+ /**
95
+ * Prepare a fragment pattern for matching, by scanning where the shape allows it
96
+ * @param {RegExp} pattern
97
+ * @returns {DelimitedFragment | PatternFragment}
98
+ */
99
+ function toFragment(pattern) {
100
+ const delimited = toDelimitedFragment(pattern);
101
+ if (delimited) return delimited;
102
+
103
+ // `g` and `y` are ours to set, the rest belong to the pattern
104
+ const flags = pattern.flags.replace(/[gy]/g, '');
105
+ return { search: new RegExp(pattern.source, flags + 'g'), anchored: new RegExp(pattern.source, flags + 'y') };
106
+ }
107
+
108
+ /**
109
+ * Replace runs of custom fragments, and the whitespace padding them
110
+ * @param {string} value - Document to scan
111
+ * @param {(DelimitedFragment | PatternFragment)[]} fragments - In the order the patterns
112
+ * were given, since the earliest match wins and ties go to the pattern listed first
113
+ * @param {(match: string) => string} replacer - Called with each match, as `replace` would
114
+ * @returns {string}
115
+ */
116
+ function replaceCustomFragments(value, fragments, replacer) {
117
+ // Where each fragment's next match may start, its next closing delimiter may be
118
+ // found, and its next excluded character sits; all only ever move forward, which
119
+ // is what keeps the whole scan linear
120
+ const found = fragments.map(() => /** @type {{start: number, end: number} | null} */ (null));
121
+ const closesAt = fragments.map(() => -1);
122
+ const excludedAt = fragments.map(() => -1);
123
+ const exhausted = fragments.map(() => false);
124
+
125
+ /**
126
+ * Whether a body region holds a character its class excludes
127
+ * @param {DelimitedFragment} fragment
128
+ * @param {number} index - Which fragment, for the forward-only bookkeeping
129
+ * @param {number} bodyStart
130
+ * @param {number} close
131
+ * @returns {boolean}
132
+ */
133
+ const bodyBlocked = (fragment, index, bodyStart, close) => {
134
+ if (!fragment.excluded) return false;
135
+ if (/** @type {number} */ (excludedAt[index]) < bodyStart) {
136
+ fragment.excluded.lastIndex = bodyStart;
137
+ const blocked = fragment.excluded.exec(value);
138
+ excludedAt[index] = blocked ? blocked.index : Infinity;
139
+ }
140
+ return /** @type {number} */ (excludedAt[index]) < close;
141
+ };
142
+
143
+ /**
144
+ * Earliest match of one delimited fragment at or after `from`
145
+ * @param {DelimitedFragment} fragment
146
+ * @param {number} index - Which fragment, for the forward-only bookkeeping
147
+ * @param {number} from
148
+ * @param {boolean} anchored - Whether the match has to start exactly at `from`
149
+ * @returns {{start: number, end: number} | null}
150
+ */
151
+ const scan = (fragment, index, from, anchored) => {
152
+ let openFrom = from;
153
+
154
+ for (;;) {
155
+ const open = anchored
156
+ ? (value.startsWith(fragment.open, from) ? from : -1)
157
+ : value.indexOf(fragment.open, openFrom);
158
+ if (open === -1) {
159
+ if (!anchored) exhausted[index] = true;
160
+ return null;
161
+ }
162
+
163
+ const bodyStart = open + fragment.open.length;
164
+ if (/** @type {number} */ (closesAt[index]) < bodyStart + fragment.min) {
165
+ closesAt[index] = value.indexOf(fragment.close, bodyStart + fragment.min);
166
+ }
167
+ const close = /** @type {number} */ (closesAt[index]);
168
+ if (close === -1) {
169
+ exhausted[index] = true;
170
+ return null;
171
+ }
172
+
173
+ if (close - bodyStart <= fragment.max && !bodyBlocked(fragment, index, bodyStart, close)) {
174
+ return { start: open, end: close + fragment.close.length };
175
+ }
176
+ // The body is longer than the pattern allows, or holds a character its class
177
+ // excludes, so the match has to start later
178
+ if (anchored) return null;
179
+ openFrom = open + 1;
180
+ }
181
+ };
182
+
183
+ /**
184
+ * Earliest match of one regex fragment at or after `from`
185
+ * @param {PatternFragment} fragment
186
+ * @param {number} from
187
+ * @param {boolean} anchored
188
+ * @returns {{start: number, end: number} | null}
189
+ */
190
+ const run = (fragment, from, anchored) => {
191
+ const pattern = anchored ? fragment.anchored : fragment.search;
192
+ pattern.lastIndex = from;
193
+
194
+ for (;;) {
195
+ const match = pattern.exec(value);
196
+ if (!match) return null;
197
+ // A pattern that matches nothing would leave the run in place forever
198
+ if (match[0].length > 0) return { start: match.index, end: match.index + match[0].length };
199
+ if (anchored) return null;
200
+ pattern.lastIndex = match.index + 1;
201
+ }
202
+ };
203
+
204
+ /**
205
+ * @param {number} from
206
+ * @param {boolean} anchored
207
+ * @returns {{start: number, end: number} | null}
208
+ */
209
+ const find = (from, anchored) => {
210
+ /** @type {{start: number, end: number} | null} */
211
+ let earliest = null;
212
+
213
+ for (let i = 0; i < fragments.length; i++) {
214
+ if (exhausted[i]) continue;
215
+ const fragment = /** @type {DelimitedFragment | PatternFragment} */ (fragments[i]);
216
+
217
+ /** @type {{start: number, end: number} | null} */
218
+ let match;
219
+ if (anchored) {
220
+ match = 'open' in fragment ? scan(fragment, i, from, true) : run(fragment, from, true);
221
+ } else {
222
+ // Matches only ever move forward, so the last one found still stands
223
+ const previous = found[i];
224
+ match = previous && previous.start >= from
225
+ ? previous
226
+ : ('open' in fragment ? scan(fragment, i, from, false) : run(fragment, from, false));
227
+ found[i] = match;
228
+ // Nothing ahead now means nothing ahead later, either
229
+ if (!match) exhausted[i] = true;
230
+ }
231
+
232
+ // Ties go to the fragment listed first, the way alternation would resolve them
233
+ if (match && (!earliest || match.start < earliest.start)) earliest = match;
234
+ }
235
+
236
+ return earliest;
237
+ };
238
+
239
+ let out = '';
240
+ let copied = 0;
241
+ let search = 0;
242
+
243
+ while (search <= value.length) {
244
+ const first = find(search, false);
245
+ if (!first) break;
246
+
247
+ // Fragments running straight into each other are one match
248
+ let end = first.end;
249
+ for (;;) {
250
+ const next = find(end, true);
251
+ if (!next) break;
252
+ end = next.end;
253
+ }
254
+
255
+ // …as is the whitespace on either side, back to where the last match left off
256
+ let start = first.start;
257
+ while (start > copied && RE_WHITESPACE.test(value[start - 1] ?? '')) start--;
258
+ while (end < value.length && RE_WHITESPACE.test(value[end] ?? '')) end++;
259
+
260
+ out += value.slice(copied, start) + replacer(value.slice(start, end));
261
+ copied = end;
262
+ search = end;
263
+ }
264
+
265
+ return copied === 0 ? value : out + value.slice(copied);
266
+ }
267
+
268
+ // Exports
269
+
270
+ export {
271
+ toDelimitedFragment,
272
+ toFragment,
273
+ replaceCustomFragments
274
+ };
@@ -1,5 +1,13 @@
1
1
  // Single source of truth for minifier option names, descriptions, types, and shared defaults
2
2
 
3
+ /**
4
+ * @typedef {object} OptionDefinition
5
+ * @property {string} description Help text, phrased for the option’s primary CLI form
6
+ * @property {string} [descriptionAffirmative] Help text for what enabling the option does, where `description` describes the negated form
7
+ * @property {string} type Key into the parser and JSON Schema type maps in cli.js and scripts/build-schema.js
8
+ */
9
+
10
+ /** @type {Record<string, OptionDefinition>} */
3
11
  const optionDefinitions = {
4
12
  cacheCSS: {
5
13
  description: 'Set CSS minification cache size (number of entries, default: 500)',
@@ -62,10 +70,6 @@ const optionDefinitions = {
62
70
  description: 'Array of regexes that allow to support custom event attributes for minifyJS (e.g., `ng-click`)',
63
71
  type: 'regexpArray'
64
72
  },
65
- customFragmentQuantifierLimit: {
66
- description: 'Set maximum quantifier limit for custom fragments to prevent ReDoS attacks (default: 200)',
67
- type: 'int'
68
- },
69
73
  decodeEntities: {
70
74
  description: 'Use direct Unicode characters whenever possible',
71
75
  type: 'boolean'
@@ -194,6 +198,10 @@ const optionDefinitions = {
194
198
  description: 'Trim whitespace around custom fragments (`--ignore-custom-fragments`)',
195
199
  type: 'boolean'
196
200
  },
201
+ strictCustomFragments: {
202
+ description: 'Reject `ignoreCustomFragments` patterns that risk catastrophic backtracking (rather than warning about them)',
203
+ type: 'boolean'
204
+ },
197
205
  useShortDoctype: {
198
206
  description: 'Replaces the doctype with the short HTML doctype',
199
207
  type: 'boolean'
@@ -1,5 +1,5 @@
1
1
  import { createUrlMinifier } from './urls.js';
2
- import { LRU, MAX_CACHE_ENTRY_SIZE, stableStringify, hashContent, identity, lowercase, replaceAsync, parseRegExp } from './utils.js';
2
+ import { LRU, MAX_CACHE_ENTRY_SIZE, stableStringify, hashContent, identity, lowercase, replaceAsync, parseRegExp, describeQuantifierRisk, lostFlag } from './utils.js';
3
3
  import { RE_TRAILING_SEMICOLON } from './constants.js';
4
4
  import { canCollapseWhitespace, canTrimWhitespace } from './whitespace.js';
5
5
  import { wrapCSS, unwrapCSS } from './content.js';
@@ -28,7 +28,7 @@ import { optionDefinitions, optionDefaults } from './option-definitions.js';
28
28
 
29
29
  /**
30
30
  * Options object produced by `processOptions` and consumed by `minifyHTML` and
31
- * the `lib/` helpers; normalization guarantees that the function-valued options
31
+ * the lib/ helpers; normalization guarantees that the function-valued options
32
32
  * below are always present (defaulting to identity/built-in functions), and
33
33
  * minification adds writable internal state on top of the public options
34
34
  * (set on prototype-chain forks during SVG/MathML namespace transitions)
@@ -97,9 +97,8 @@ const optionKeysExtra = new Set(['preset', 'log', 'canCollapseWhitespace', 'canT
97
97
  // key per process, so repeated `minify` calls (e.g., batch runs) don’t flood STDERR
98
98
  const optionKeysWarned = new Set();
99
99
  const presetNamesWarned = new Set();
100
- // The custom-fragment ReDoS warning is security-relevant, so it reaches the
101
- // console even without a `log` hook—once per process, like the warnings above
102
- let customFragmentQuantifierWarned = false;
100
+ // Custom fragments whose shape risks ReDoS, warned about once per pattern per process
101
+ const customFragmentsWarned = new Set();
103
102
  const unusedCSSWarned = new Set();
104
103
  // Object-valued options handed a string, warned about once per distinct value
105
104
  const stringValuesWarned = new Set();
@@ -191,7 +190,7 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
191
190
  // A string carries no configuration for these options. The CLI parses config
192
191
  // values as JSON first, so only a value that is not JSON reaches this from
193
192
  // there. (`minifyURLs` is deliberately excluded—there, a string names the site.)
194
- const definition = /** @type {Record<string, {type?: string}>} */ (optionDefinitions)[key];
193
+ const definition = optionDefinitions[key];
195
194
  if (typeof option === 'string' && definition?.type === 'jsonObject') {
196
195
  const message = `HTML Minifier Next: Ignoring \`${key}\`—it takes a boolean or an object, not a string (“${option}”)`;
197
196
  if (!stringValuesWarned.has(message)) {
@@ -598,22 +597,50 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
598
597
  } else if (['customAttrAssign', 'customEventAttributes', 'ignoreCustomComments', 'ignoreCustomFragments'].includes(key)) {
599
598
  // Array of regex patterns
600
599
  optionsDynamic[key] = parseRegExpArray(option);
601
- // Warn about potential ReDoS when user-provided fragments use unlimited
602
- // quantifiers; only explicitly passed fragments are checked
603
- if (key === 'ignoreCustomFragments' && !customFragmentQuantifierWarned) {
604
- for (const re of /** @type {RegExp[]} */ (optionsDynamic[key])) {
605
- if (/[*+]/.test(re.source)) {
606
- customFragmentQuantifierWarned = true;
607
- warn('HTML Minifier Next: Custom fragment contains unlimited quantifiers (“*” or “+”) which may cause ReDoS vulnerability');
608
- break;
609
- }
610
- }
611
- }
612
600
  } else {
613
601
  optionsDynamic[key] = option;
614
602
  }
615
603
  });
616
604
 
605
+ // The parser merges these into one attribute pattern, and a merged pattern
606
+ // carries no flags of its own. `i` and `s` are written into the source instead,
607
+ // but `u` and `v` change how a source reads and cannot be: dropped, `\p{L}`
608
+ // stops being a property escape and matches the literal text `p{L}`, silently
609
+ // and without failing to compile. Refusing them beats matching the wrong thing.
610
+ /** @type {[string, RegExp[]][]} */
611
+ const merged = [
612
+ ['customAttrAssign', options.customAttrAssign || []],
613
+ ['customAttrSurround', (options.customAttrSurround || []).flat()]
614
+ ];
615
+ for (const [key, patterns] of merged) {
616
+ for (const re of patterns) {
617
+ if (!(re instanceof RegExp)) continue;
618
+ const flag = lostFlag(re);
619
+ if (!flag) continue;
620
+ throw new Error(`HTML Minifier Next: \`${key}\` pattern \`/${re.source}/${re.flags}\` carries \`${flag}\`, which the merged attribute pattern cannot carry—rewrite it without \`${flag}\``);
621
+ }
622
+ }
623
+
624
+ // Fragments that compound quantifiers or alternation under unbounded repetition
625
+ // are the shapes that backtrack catastrophically, and they are also the ones a
626
+ // linear scan cannot stand in for; so are patterns too long or too deeply
627
+ // nested to read, which are refused for that rather than for a shape. A lone
628
+ // `[\s\S]*?` up to a literal terminator is linear and passes; HMN’s default
629
+ // fragments have exactly that shape, so the check flagging it would mean
630
+ // warning about the defaults themselves.
631
+ for (const re of options.ignoreCustomFragments || []) {
632
+ const risk = describeQuantifierRisk(re);
633
+ if (!risk) continue;
634
+ const problem = `Custom fragment \`/${re.source}/\` ${risk}`;
635
+ if (options.strictCustomFragments) {
636
+ throw new Error(`HTML Minifier Next: ${problem}`);
637
+ }
638
+ if (!customFragmentsWarned.has(re.source)) {
639
+ customFragmentsWarned.add(re.source);
640
+ warn(`HTML Minifier Next: ${problem}`);
641
+ }
642
+ }
643
+
617
644
  // Unused-CSS removal rides along with Lightning CSS, so it silently does nothing
618
645
  // when `minifyCSS` is off or replaced by a function—say so rather than let it pass
619
646
  if (options.removeUnusedCSS) {
@@ -37,9 +37,6 @@ const fragmentReferenceAttributes = new Set([
37
37
  'xlink:href'
38
38
  ]);
39
39
 
40
- // `srcdoc` can nest, and each level costs another scan of its own markup
41
- const SRCDOC_MAX_DEPTH = 3;
42
-
43
40
  const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
44
41
  const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
45
42
  // SVG paints and filters reach elements by ID through `url(#gradient)`, in
@@ -173,16 +170,17 @@ function findRawTextElements(haystack, tagName) {
173
170
  * Collect the class names and IDs a document references.
174
171
  *
175
172
  * Style sheet contents are excluded, so that a style sheet never counts as
176
- * evidence for its own selectors. Over-collecting is safe here (a symbol wrongly
177
- * considered used is merely kept), under-collecting is not.
173
+ * evidence for its own selectors. An `iframe srcdoc` value is not scanned,
174
+ * either: It holds a document of its own, minified against its own symbol set.
175
+ * Over-collecting is safe here (a symbol wrongly considered used is merely
176
+ * kept), under-collecting is not.
178
177
  *
179
178
  * @param {string} html - Raw document markup
180
179
  * @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
181
180
  * @param {((text: string) => string)} [decode] - Resolves character references in attribute values
182
- * @param {number} [depth] - Nesting level, counted through `srcdoc`
183
181
  * @returns {Set<string>} Symbols to keep
184
182
  */
185
- function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
183
+ function collectUsedSymbols(html, includeScripts, decode) {
186
184
  const used = new Set();
187
185
  const haystack = foldCase(html);
188
186
 
@@ -222,9 +220,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
222
220
  return element !== undefined && index >= element.start;
223
221
  };
224
222
 
225
- /** @type {string[]} */
226
- const nested = [];
227
-
228
223
  attributePattern.lastIndex = 0;
229
224
  let match;
230
225
  while ((match = attributePattern.exec(html))) {
@@ -244,8 +239,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
244
239
  } else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
245
240
  // Only a leading `#`: `href="/page#sec"` names a section of another document
246
241
  addTokens(value.slice(1));
247
- } else if (name === 'srcdoc' && depth < SRCDOC_MAX_DEPTH) {
248
- nested.push(value);
249
242
  }
250
243
  if (value.indexOf('(') !== -1) {
251
244
  fragmentURLPattern.lastIndex = 0;
@@ -256,16 +249,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
256
249
  }
257
250
  }
258
251
 
259
- // `iframe srcdoc` holds a document of its own, whose style sheets are minified
260
- // against this very set—so what it references has to be in it. The scan runs here
261
- // rather than at the attribute: The patterns it shares with this one are
262
- // module-level, so no nested scan may start while one of their loops is open.
263
- for (const markup of nested) {
264
- for (const symbol of collectUsedSymbols(markup, includeScripts, decode, depth + 1)) {
265
- used.add(symbol);
266
- }
267
- }
268
-
269
252
  if (includeScripts) {
270
253
  // Script contents are raw text, so character references stay literal;
271
254
  // an unclosed `script` runs to the end of the document, as it does in a browser