html-minifier-next 7.6.0 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/lib/utils.js CHANGED
@@ -126,7 +126,8 @@ async function replaceAsync(str, regex, asyncFn) {
126
126
  });
127
127
 
128
128
  const data = await Promise.all(promises);
129
- return str.replace(regex, () => data.shift() ?? '');
129
+ let next = 0;
130
+ return str.replace(regex, () => data[next++] ?? '');
130
131
  }
131
132
 
132
133
  // String patterns to RegExp conversion (for JSON config support)
@@ -144,7 +145,307 @@ function parseRegExp(value) {
144
145
  return value;
145
146
  }
146
147
 
147
- // Exports
148
+ // ReDoS risk detection for user-supplied patterns
149
+
150
+ // Quantifier following an atom, with its optional lazy `?`; the captures hold
151
+ // the bounds of a `{n,m}` form, the upper one empty when the form is `{n,}`
152
+ const RE_QUANTIFIER = /(?:[*+?]|\{(\d+)(?:,(\d*))?\})\??/y;
153
+
154
+ // Bounds for the walk below: Patterns beyond either are judged risky unread,
155
+ // which keeps a pathological source from nesting the analysis into a stack
156
+ // overflow or making it rescan its groups once per level
157
+ const MAX_PATTERN_LENGTH = 10000;
158
+ const MAX_PATTERN_DEPTH = 50;
159
+
160
+ // A group that only wraps its body backtracks the way that body does, so
161
+ // `(?:a)` and `a` count as the same atom; lookarounds are left alone
162
+ const RE_TRANSPARENT_GROUP = /^\((?:\?:|\?<[^>=!][^>]*>)?([\s\S]*)\)$/;
163
+
164
+ // A quantifier of exactly one repetition, which is notation rather than shape
165
+ const RE_EXACT_ONE = /\{1(?:,1)?\}\??$/;
166
+
167
+ /** @param {string} source @param {number} index - Index of the opening `[` */
168
+ function skipCharacterClass(source, index) {
169
+ let i = index + 1;
170
+ while (i < source.length && source[i] !== ']') {
171
+ i += source[i] === '\\' ? 2 : 1;
172
+ }
173
+ return i + 1;
174
+ }
175
+
176
+ /** @param {string} atom - Atom as it reads in the source */
177
+ function unwrapAtom(atom) {
178
+ let inner = atom;
179
+ for (let level = 0; level < MAX_PATTERN_DEPTH; level++) {
180
+ const next = inner.replace(RE_EXACT_ONE, '').replace(RE_TRANSPARENT_GROUP, '$1');
181
+ if (next === inner) break;
182
+ inner = next;
183
+ }
184
+ return inner;
185
+ }
186
+
187
+ // Embedding a pattern in a larger regex drops the flags it carried, so the two
188
+ // a source can carry on its own are rewritten into it
189
+ //
190
+ // @@ Replace with inline `(?i:…)` and `(?s:…)` modifiers once Node floor reaches 24
191
+
192
+ const RE_ASCII_LETTER = /^[a-zA-Z]$/;
193
+ const RE_HEX_PAIR = /^[0-9a-fA-F]{2}$/;
194
+ const RE_HEX_QUAD = /^[0-9a-fA-F]{4}$/;
195
+
196
+ /** @param {string} char @returns {boolean} Whether the character has a single-character counterpart */
197
+ function foldsCase(char) {
198
+ const lower = char.toLowerCase();
199
+ const upper = char.toUpperCase();
200
+ return lower !== upper && lower.length === 1 && upper.length === 1;
201
+ }
202
+
203
+ /** @param {string} char */
204
+ function isLower(char) {
205
+ return char === char.toLowerCase();
206
+ }
207
+
208
+ /**
209
+ * Length of the token at `index`, so that a multi-character escape is read whole
210
+ * rather than leaving its tail to be mistaken for literals.
211
+ * @param {string} source @param {number} index
212
+ */
213
+ function tokenLength(source, index) {
214
+ if (source[index] !== '\\') return 1;
215
+ const next = source[index + 1];
216
+ if (next === 'x' && RE_HEX_PAIR.test(source.slice(index + 2, index + 4))) return 4;
217
+ if (next === 'u') {
218
+ if (source[index + 2] === '{') {
219
+ const close = source.indexOf('}', index + 3);
220
+ if (close !== -1) return close - index + 1;
221
+ }
222
+ if (RE_HEX_QUAD.test(source.slice(index + 2, index + 6))) return 6;
223
+ }
224
+ if (next === 'c' && RE_ASCII_LETTER.test(source[index + 2] ?? '')) return 3;
225
+ // A property name or a group name is syntax, not text to fold
226
+ if ((next === 'p' || next === 'P') && source[index + 2] === '{') {
227
+ const close = source.indexOf('}', index + 3);
228
+ if (close !== -1) return close - index + 1;
229
+ }
230
+ if (next === 'k' && source[index + 2] === '<') {
231
+ const close = source.indexOf('>', index + 3);
232
+ if (close !== -1) return close - index + 1;
233
+ }
234
+ return 2;
235
+ }
236
+
237
+ /** @param {string} source - Regex source, assumed syntactically valid */
238
+ function expandDotAll(source) {
239
+ let out = '';
240
+ let i = 0;
241
+ while (i < source.length) {
242
+ const char = source[i];
243
+ if (char === '\\') {
244
+ const length = tokenLength(source, i);
245
+ out += source.slice(i, i + length);
246
+ i += length;
247
+ } else if (char === '[') {
248
+ const end = skipCharacterClass(source, i);
249
+ out += source.slice(i, end);
250
+ i = end;
251
+ } else if (char === '.') {
252
+ out += '[\\s\\S]';
253
+ i++;
254
+ } else {
255
+ out += char;
256
+ i++;
257
+ }
258
+ }
259
+ return out;
260
+ }
261
+
262
+ /** @param {string} source - Regex source, assumed syntactically valid */
263
+ function foldCase(source) {
264
+ let out = '';
265
+ let i = 0;
266
+ let inClass = false;
267
+
268
+ while (i < source.length) {
269
+ const char = source[i] ?? '';
270
+
271
+ if (!inClass) {
272
+ if (char === '\\') {
273
+ const length = tokenLength(source, i);
274
+ out += source.slice(i, i + length);
275
+ i += length;
276
+ } else if (char === '[') {
277
+ inClass = true;
278
+ out += char;
279
+ i++;
280
+ } else if (char === '(' && source.startsWith('(?<', i) &&
281
+ source[i + 3] !== '=' && source[i + 3] !== '!') {
282
+ // A capture group’s name is syntax, and folding it is a syntax error
283
+ const close = source.indexOf('>', i + 3);
284
+ const end = close === -1 ? i + 3 : close + 1;
285
+ out += source.slice(i, end);
286
+ i = end;
287
+ } else {
288
+ out += foldsCase(char) ? '[' + char.toLowerCase() + char.toUpperCase() + ']' : char;
289
+ i++;
290
+ }
291
+ continue;
292
+ }
293
+
294
+ if (char === ']') {
295
+ inClass = false;
296
+ out += char;
297
+ i++;
298
+ continue;
299
+ }
300
+
301
+ const fromLength = tokenLength(source, i);
302
+ const from = source.slice(i, i + fromLength);
303
+ const afterFrom = i + fromLength;
304
+
305
+ // A range needs the other case’s range beside it—only where both ends are
306
+ // ASCII letters, the one span whose two cases are contiguous and parallel
307
+ // (`ÿ` uppercases to `Ÿ`, 150 code points past where its range ends)
308
+ if (source[afterFrom] === '-' && afterFrom + 1 < source.length && source[afterFrom + 1] !== ']') {
309
+ const toLength = tokenLength(source, afterFrom + 1);
310
+ const to = source.slice(afterFrom + 1, afterFrom + 1 + toLength);
311
+ out += RE_ASCII_LETTER.test(from) && RE_ASCII_LETTER.test(to) && isLower(from) === isLower(to)
312
+ ? from + '-' + to + (isLower(from)
313
+ ? from.toUpperCase() + '-' + to.toUpperCase()
314
+ : from.toLowerCase() + '-' + to.toLowerCase())
315
+ : from + '-' + to;
316
+ i = afterFrom + 1 + toLength;
317
+ continue;
318
+ }
319
+
320
+ out += fromLength === 1 && foldsCase(from) ? from.toLowerCase() + from.toUpperCase() : from;
321
+ i = afterFrom;
322
+ }
323
+
324
+ return out;
325
+ }
326
+
327
+ /**
328
+ * A pattern’s source, rewritten to match the way its own flags make it match, for
329
+ * embedding in a larger regex that cannot carry them. `i` and `s` fit into a
330
+ * source; `m`, `u`, and `v` do not, and are left to the pattern it joins. Where
331
+ * `i` cannot be written in—a backreference, a range outside ASCII, a range that
332
+ * spans letters without being one—the source stands as it is, matching less than
333
+ * the pattern would rather than more.
334
+ * @param {RegExp} pattern
335
+ * @returns {string}
336
+ */
337
+ function embedSource(pattern) {
338
+ let source = pattern.source;
339
+ if (pattern.dotAll) source = expandDotAll(source);
340
+ if (pattern.ignoreCase) source = foldCase(source);
341
+ return source;
342
+ }
343
+
344
+ /**
345
+ * Walk a regex source for the shapes whose backtracking blows up: an unlimited
346
+ * quantifier over a group that itself contains a variable quantifier (`(a+)+`,
347
+ * `(a?)+`) or alternates, and the same atom repeated unboundedly twice in a
348
+ * row. A lone unlimited quantifier stays linear, so `[\s\S]*?` up to a literal
349
+ * terminator passes.
350
+ * @param {string} source - Regex source, assumed syntactically valid
351
+ * @param {number} [depth] - Group nesting level of this call
352
+ * @returns {{risky: boolean, varies: boolean, alternates: boolean, deep: boolean}}
353
+ */
354
+ function analyzeQuantifiers(source, depth = 0) {
355
+ if (depth > MAX_PATTERN_DEPTH) return { risky: true, varies: true, alternates: true, deep: true };
356
+
357
+ let risky = false;
358
+ let varies = false;
359
+ let alternates = false;
360
+ let deep = false;
361
+ /** @type {{text: string, repeats: boolean} | null} */
362
+ let previous = null;
363
+ let i = 0;
364
+
365
+ while (i < source.length) {
366
+ const start = i;
367
+ const char = source[i];
368
+ /** @type {ReturnType<typeof analyzeQuantifiers> | null} */
369
+ let group = null;
370
+
371
+ if (char === '|') {
372
+ // Alternatives are separate expressions, so nothing carries across
373
+ alternates = true;
374
+ previous = null;
375
+ i++;
376
+ continue;
377
+ } else if (char === '\\') {
378
+ i += 2;
379
+ } else if (char === '[') {
380
+ i = skipCharacterClass(source, i);
381
+ } else if (char === '(') {
382
+ let open = 1;
383
+ i++;
384
+ while (i < source.length && open > 0) {
385
+ const inner = source[i];
386
+ if (inner === '\\') {
387
+ i += 2;
388
+ } else if (inner === '[') {
389
+ i = skipCharacterClass(source, i);
390
+ } else {
391
+ if (inner === '(') open++;
392
+ else if (inner === ')') open--;
393
+ i++;
394
+ }
395
+ }
396
+ // Drop the group prefix (`?:`, `?=`, `?<name>`, …) before reading the body
397
+ group = analyzeQuantifiers(source.slice(start + 1, i - 1).replace(/^\?(?:[:=!]|<[=!]|<[^>]*>)/, ''), depth + 1);
398
+ } else {
399
+ i++;
400
+ }
401
+
402
+ const atom = source.slice(start, i);
403
+ RE_QUANTIFIER.lastIndex = i;
404
+ const quantifier = RE_QUANTIFIER.exec(source);
405
+ const upper = quantifier?.[2];
406
+ const repeats = !!quantifier && (quantifier[0][0] === '*' || quantifier[0][0] === '+' || upper === '');
407
+ // A count that can vary—anything but `{n}` and its `{n,n}` spelling—makes
408
+ // the group ambiguous about how much it consumes, which multiplies under an
409
+ // unlimited repeat
410
+ const exact = !!quantifier && quantifier[0][0] === '{' &&
411
+ (upper === undefined || (upper !== '' && Number(upper) === Number(quantifier[1])));
412
+ const variable = !!quantifier && !exact;
413
+ if (quantifier) i = RE_QUANTIFIER.lastIndex;
414
+
415
+ if (group) {
416
+ if (group.risky || (repeats && (group.varies || group.alternates))) risky = true;
417
+ if (group.varies) varies = true;
418
+ // A group alternates whether the `|` sits at its top level or deeper
419
+ if (group.alternates) alternates = true;
420
+ if (group.deep) deep = true;
421
+ }
422
+ if (variable) varies = true;
423
+ const text = unwrapAtom(atom);
424
+ if (repeats && previous?.repeats && previous.text === text) risky = true;
425
+ previous = { text, repeats };
426
+ }
427
+
428
+ return { risky, varies, alternates, deep };
429
+ }
430
+
431
+ /**
432
+ * @param {string} source - Regex source to judge
433
+ * @returns {string | null} What makes the pattern a backtracking risk, phrased
434
+ * to follow the pattern itself, or `null` where it is none
435
+ */
436
+ function describeQuantifierRisk(source) {
437
+ // A pattern too big to read is refused for that, not for a shape nobody saw
438
+ if (source.length > MAX_PATTERN_LENGTH) {
439
+ return `runs past ${MAX_PATTERN_LENGTH.toLocaleString()} characters, too long to analyze for catastrophic backtracking—shorten it, or split it into several patterns`;
440
+ }
441
+ const analysis = analyzeQuantifiers(source);
442
+ if (analysis.deep) {
443
+ return `nests groups more than ${MAX_PATTERN_DEPTH.toLocaleString()} deep, too deep to analyze for catastrophic backtracking—flatten it, or split it into several patterns`;
444
+ }
445
+ return analysis.risky
446
+ ? 'compounds quantifiers or alternation in a way that may cause ReDoS—bound the repetition (e.g., `{0,1000}`) instead'
447
+ : null;
448
+ }
148
449
 
149
450
  /**
150
451
  * Find the index of the `>` that closes an opening tag, correctly skipping
@@ -168,6 +469,8 @@ function findTagEnd(html, pos) {
168
469
  return -1;
169
470
  }
170
471
 
472
+ // Exports
473
+
171
474
  export {
172
475
  stableStringify,
173
476
  findTagEnd,
@@ -179,5 +482,7 @@ export {
179
482
  isThenable,
180
483
  lowercase,
181
484
  replaceAsync,
182
- parseRegExp
485
+ parseRegExp,
486
+ embedSource,
487
+ describeQuantifierRisk
183
488
  };
package/src/tokenchain.js CHANGED
@@ -1,7 +1,9 @@
1
1
  class Sorter {
2
2
  constructor() {
3
- /** @type {string[]} */
4
- this.keys = [];
3
+ // Rank per token, in the order the keys were added—looked up per token, so a
4
+ // token list sorts in one pass instead of one scan per known key
5
+ /** @type {Map<string, number>} */
6
+ this.rank = new Map();
5
7
  /** @type {Map<string, Sorter>} */
6
8
  this.sorterMap = new Map();
7
9
  }
@@ -12,36 +14,41 @@ class Sorter {
12
14
  * @returns {string[]}
13
15
  */
14
16
  sort(tokens, fromIndex = 0) {
15
- for (const token of this.keys) {
16
-
17
- // Single pass: Count matches and collect non-matches
18
- let matchCount = 0;
19
- const others = [];
20
-
21
- for (let j = fromIndex; j < tokens.length; j++) {
22
- const t = /** @type {string} */ (tokens[j]);
23
- if (t === token) {
24
- matchCount++;
25
- } else {
26
- others.push(t);
27
- }
17
+ // The present token with the lowest rank comes first—the same choice scanning
18
+ // the keys in order would make
19
+ let best = null;
20
+ let bestRank = Infinity;
21
+ for (let j = fromIndex; j < tokens.length; j++) {
22
+ const rank = this.rank.get(/** @type {string} */ (tokens[j]));
23
+ if (rank !== undefined && rank < bestRank) {
24
+ bestRank = rank;
25
+ best = /** @type {string} */ (tokens[j]);
28
26
  }
29
-
30
- if (matchCount > 0) {
31
- // Rebuild: `matchCount` instances of token first, then others
32
- let writeIdx = fromIndex;
33
- for (let j = 0; j < matchCount; j++) {
34
- tokens[writeIdx++] = token;
35
- }
36
- for (const other of others) {
37
- tokens[writeIdx++] = other;
38
- }
39
-
40
- const newFromIndex = fromIndex + matchCount;
41
- return this.sorterMap.get(token)?.sort(tokens, newFromIndex) ?? tokens;
27
+ }
28
+ if (best === null) return tokens;
29
+
30
+ // Single pass: Count matches and collect non-matches
31
+ let matchCount = 0;
32
+ const others = [];
33
+ for (let j = fromIndex; j < tokens.length; j++) {
34
+ const t = /** @type {string} */ (tokens[j]);
35
+ if (t === best) {
36
+ matchCount++;
37
+ } else {
38
+ others.push(t);
42
39
  }
43
40
  }
44
- return tokens;
41
+
42
+ // Rebuild: `matchCount` instances of the best token first, then others
43
+ let writeIdx = fromIndex;
44
+ for (let j = 0; j < matchCount; j++) {
45
+ tokens[writeIdx++] = best;
46
+ }
47
+ for (const other of others) {
48
+ tokens[writeIdx++] = other;
49
+ }
50
+
51
+ return this.sorterMap.get(best)?.sort(tokens, fromIndex + matchCount) ?? tokens;
45
52
  }
46
53
  }
47
54
 
@@ -104,7 +111,7 @@ class TokenChain {
104
111
  }
105
112
  });
106
113
 
107
- sorter.keys.push(token);
114
+ sorter.rank.set(token, sorter.rank.size);
108
115
  sorter.sorterMap.set(token, chain.createSorter());
109
116
  }
110
117
  });