html-minifier-next 7.6.0 → 8.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/lib/utils.js CHANGED
@@ -126,7 +126,8 @@ async function replaceAsync(str, regex, asyncFn) {
126
126
  });
127
127
 
128
128
  const data = await Promise.all(promises);
129
- return str.replace(regex, () => data.shift() ?? '');
129
+ let next = 0;
130
+ return str.replace(regex, () => data[next++] ?? '');
130
131
  }
131
132
 
132
133
  // String patterns to RegExp conversion (for JSON config support)
@@ -144,7 +145,669 @@ function parseRegExp(value) {
144
145
  return value;
145
146
  }
146
147
 
147
- // Exports
148
+ // Regex source reading, shared by the analysis and the rewriting below
149
+
150
+ const RE_ASCII_LETTER = /^[a-zA-Z]$/;
151
+ const RE_HEX_PAIR = /^[0-9a-fA-F]{2}$/;
152
+ const RE_HEX_QUAD = /^[0-9a-fA-F]{4}$/;
153
+
154
+ /**
155
+ * @param {string} source @param {number} index - Index of the opening `[`
156
+ * @param {boolean} [nested] - Whether the pattern carries `v`, where a class nests
157
+ * inside a class and the first `]` need not be the one that closes it
158
+ * @returns {number} Index just past the closing `]`
159
+ */
160
+ function skipCharacterClass(source, index, nested = false) {
161
+ let i = index + 1;
162
+ let open = 1;
163
+ while (i < source.length && open > 0) {
164
+ const char = source[i];
165
+ if (char === '\\') i += tokenLength(source, i);
166
+ else {
167
+ if (nested && char === '[') open++;
168
+ else if (char === ']') open--;
169
+ i++;
170
+ }
171
+ }
172
+ return i;
173
+ }
174
+
175
+ /**
176
+ * Length of the token at `index`, so that a multi-character escape is read whole
177
+ * rather than leaving its tail to be mistaken for literals.
178
+ * @param {string} source @param {number} index
179
+ */
180
+ function tokenLength(source, index) {
181
+ if (source[index] !== '\\') return 1;
182
+ const next = source[index + 1];
183
+ if (next === 'x' && RE_HEX_PAIR.test(source.slice(index + 2, index + 4))) return 4;
184
+ if (next === 'u') {
185
+ if (source[index + 2] === '{') {
186
+ const close = source.indexOf('}', index + 3);
187
+ if (close !== -1) return close - index + 1;
188
+ }
189
+ if (RE_HEX_QUAD.test(source.slice(index + 2, index + 6))) return 6;
190
+ }
191
+ if (next === 'c' && RE_ASCII_LETTER.test(source[index + 2] ?? '')) return 3;
192
+ // A `v` class holds string literals, which fold as strings or not at all
193
+ if (next === 'q' && source[index + 2] === '{') {
194
+ const close = source.indexOf('}', index + 3);
195
+ if (close !== -1) return close - index + 1;
196
+ }
197
+ // A property name or a group name is syntax, not text to fold
198
+ if ((next === 'p' || next === 'P') && source[index + 2] === '{') {
199
+ const close = source.indexOf('}', index + 3);
200
+ if (close !== -1) return close - index + 1;
201
+ }
202
+ if (next === 'k' && source[index + 2] === '<') {
203
+ const close = source.indexOf('>', index + 3);
204
+ if (close !== -1) return close - index + 1;
205
+ }
206
+ return 2;
207
+ }
208
+
209
+ // ReDoS risk detection for user-supplied patterns
210
+
211
+ // Quantifier following an atom, with its optional lazy `?`; the captures hold
212
+ // the bounds of a `{n,m}` form, the upper one empty when the form is `{n,}`
213
+ const RE_QUANTIFIER = /(?:[*+?]|\{(\d+)(?:,(\d*))?\})\??/y;
214
+
215
+ // Bounds for the walk below: Patterns beyond either are judged risky unread,
216
+ // which keeps a pathological source from nesting the analysis into a stack
217
+ // overflow or making it rescan its groups once per level
218
+ const MAX_PATTERN_LENGTH = 10000;
219
+ const MAX_PATTERN_DEPTH = 50;
220
+
221
+ // A group that only wraps its body backtracks the way that body does, so
222
+ // `(?:a)` and `a` count as the same atom; lookarounds are left alone
223
+ const RE_TRANSPARENT_GROUP = /^\((?:\?:|\?<[^>=!][^>]*>)?([\s\S]*)\)$/;
224
+
225
+ // A quantifier of exactly one repetition, which is notation rather than shape
226
+ const RE_EXACT_ONE = /\{1(?:,1)?\}\??$/;
227
+
228
+ // An atom that matches without consuming, so what precedes it stays adjacent to
229
+ // what follows; a group that can match empty is left out, being rare enough that
230
+ // missing it costs less than reading every body for it
231
+ const RE_ZERO_WIDTH = /^(?:[$^]|\\[bB]|\((?:\?[=!]|\?<[=!]))/;
232
+
233
+ // How many repeats an atom may still be adjacent to across atoms matching empty;
234
+ // comparing against every one of them is what the bound keeps from turning
235
+ // quadratic on a pattern that is nothing but repeats
236
+ const MAX_REACHABLE = 50;
237
+
238
+ /** @param {string} atom - Atom as it reads in the source */
239
+ function unwrapAtom(atom) {
240
+ let inner = atom;
241
+ for (let level = 0; level < MAX_PATTERN_DEPTH; level++) {
242
+ const next = inner.replace(RE_EXACT_ONE, '').replace(RE_TRANSPARENT_GROUP, '$1');
243
+ if (next === inner) break;
244
+ inner = next;
245
+ }
246
+ return inner;
247
+ }
248
+
249
+ // Two atoms that repeat unboundedly side by side split a run between them in as
250
+ // many ways as the run is long, whenever both can consume the same character.
251
+ // Comparing what they match, rather than how they are spelled, is what catches
252
+ // `[a]*a*` and `\w*\d*` alongside `a*a*`.
253
+
254
+ const MAX_CODE_POINT = 0x10FFFF;
255
+
256
+ /** @type {Record<string, [number, number][]>} */
257
+ const CLASS_ESCAPE_RANGES = {
258
+ d: [[0x30, 0x39]],
259
+ w: [[0x30, 0x39], [0x41, 0x5A], [0x5F, 0x5F], [0x61, 0x7A]],
260
+ s: [[0x09, 0x0D], [0x20, 0x20], [0xA0, 0xA0], [0x1680, 0x1680], [0x2000, 0x200A],
261
+ [0x2028, 0x2029], [0x202F, 0x202F], [0x205F, 0x205F], [0x3000, 0x3000], [0xFEFF, 0xFEFF]]
262
+ };
263
+
264
+ /** @type {Record<string, number>} */
265
+ const CONTROL_ESCAPE_CODES = { 0: 0x00, f: 0x0C, n: 0x0A, r: 0x0D, t: 0x09, v: 0x0B };
266
+
267
+ // An escape that stands for something other than one character of text
268
+ const RE_NON_CHARACTER_ESCAPE = /[bBdDkpPsSwW1-9]/;
269
+
270
+ /** @param {[number, number][]} ranges */
271
+ function complement(ranges) {
272
+ const sorted = [...ranges].sort((one, other) => one[0] - other[0]);
273
+ /** @type {[number, number][]} */
274
+ const out = [];
275
+ let next = 0;
276
+ for (const [low, high] of sorted) {
277
+ if (low > next) out.push([next, low - 1]);
278
+ next = Math.max(next, high + 1);
279
+ }
280
+ if (next <= MAX_CODE_POINT) out.push([next, MAX_CODE_POINT]);
281
+ return out;
282
+ }
283
+
284
+ // `.` as a bare source reads it, without the `s` flag the source cannot carry
285
+ const DOT_RANGES = complement([[0x0A, 0x0A], [0x0D, 0x0D], [0x2028, 0x2029]]);
286
+
287
+ /**
288
+ * @param {string} token - One token, as `tokenLength` measures it
289
+ * @param {boolean} [inClass] - Whether the token sits inside a character class
290
+ * @returns {[number, number][] | null} What it matches, or `null` where it is not
291
+ * one character of text
292
+ */
293
+ function tokenRanges(token, inClass = false) {
294
+ if (token[0] !== '\\') {
295
+ const code = token.codePointAt(0) ?? 0;
296
+ return String.fromCodePoint(code) === token ? [[code, code]] : null;
297
+ }
298
+
299
+ const kind = token[1] ?? '';
300
+ const named = CLASS_ESCAPE_RANGES[kind.toLowerCase()];
301
+ if (named) return kind === kind.toLowerCase() ? named : complement(named);
302
+
303
+ const control = CONTROL_ESCAPE_CODES[kind];
304
+ if (control !== undefined) return [[control, control]];
305
+
306
+ if (kind === 'x' || kind === 'u') {
307
+ const code = Number.parseInt(token[2] === '{' ? token.slice(3, -1) : token.slice(2), 16);
308
+ return Number.isNaN(code) || code > MAX_CODE_POINT ? null : [[code, code]];
309
+ }
310
+ if (kind === 'c') {
311
+ const code = token.charCodeAt(2) % 32;
312
+ return [[code, code]];
313
+ }
314
+ // `\b` asserts a word boundary on its own, and is a backspace inside a class
315
+ if (kind === 'b' && inClass) return [[0x08, 0x08]];
316
+ // `\q{…}` stands for whole strings, not for a character
317
+ if (token.startsWith('\\q{')) return null;
318
+ if (RE_NON_CHARACTER_ESCAPE.test(kind)) return null;
319
+
320
+ // The rest is punctuation escaped to be read as itself
321
+ const code = token.codePointAt(1) ?? 0;
322
+ return String.fromCodePoint(code) === token.slice(1) ? [[code, code]] : null;
323
+ }
324
+
325
+ /** @param {[number, number][] | null} ranges @returns {number | null} The one code point it holds */
326
+ function singleCode(ranges) {
327
+ const [range] = ranges ?? [];
328
+ return ranges?.length === 1 && range && range[0] === range[1] ? range[0] : null;
329
+ }
330
+
331
+ /**
332
+ * @param {string} atom - Atom as it reads in the source, already unwrapped
333
+ * @param {boolean} [nested] - Whether the pattern carries `v`
334
+ * @param {number} [depth] - Class nesting level of this call
335
+ * @returns {[number, number][] | null} What it matches, or `null` where it is not
336
+ * one character of text, or one this does not read
337
+ */
338
+ function atomRanges(atom, nested = false, depth = 0) {
339
+ if (depth > MAX_PATTERN_DEPTH) return null;
340
+ if (atom === '.') return DOT_RANGES;
341
+ if (atom[0] !== '[') return tokenRanges(atom);
342
+ if (!atom.endsWith(']')) return null;
343
+
344
+ let i = atom[1] === '^' ? 2 : 1;
345
+ const negated = i === 2;
346
+ const end = atom.length - 1;
347
+ /** @type {[number, number][]} */
348
+ const members = [];
349
+
350
+ while (i < end) {
351
+ // A `v` class nests, and nesting alone is a union to read through
352
+ if (nested && atom[i] === '[') {
353
+ const close = skipCharacterClass(atom, i, true);
354
+ const inner = atomRanges(atom.slice(i, close), true, depth + 1);
355
+ if (!inner) return null;
356
+ members.push(...inner);
357
+ i = close;
358
+ continue;
359
+ }
360
+ // Subtraction and intersection are not unions, and are left unread
361
+ //
362
+ // @@ Read `--` and `&&` as set difference and intersection
363
+ // (so that a `v` class built is compared rather than passed unjudged)
364
+ if (nested && (atom.startsWith('--', i) || atom.startsWith('&&', i))) return null;
365
+
366
+ const fromLength = tokenLength(atom, i);
367
+ const from = tokenRanges(atom.slice(i, i + fromLength), true);
368
+ if (!from) return null;
369
+ i += fromLength;
370
+
371
+ // A dash right before the closing `]` is a member, not the start of a range
372
+ if (atom[i] !== '-' || i + 1 >= end) {
373
+ members.push(...from);
374
+ continue;
375
+ }
376
+ const toLength = tokenLength(atom, i + 1);
377
+ // Only single characters bound a range; `\d-z` is not a range at all
378
+ const low = singleCode(from);
379
+ const high = singleCode(tokenRanges(atom.slice(i + 1, i + 1 + toLength), true));
380
+ if (low === null || high === null) return null;
381
+ members.push([low, high]);
382
+ i += 1 + toLength;
383
+ }
384
+
385
+ return negated ? complement(members) : members;
386
+ }
387
+
388
+ /** @param {[number, number][]} ranges @param {[number, number][]} other */
389
+ function rangesIntersect(ranges, other) {
390
+ return ranges.some(([low, high]) => other.some(([otherLow, otherHigh]) =>
391
+ low <= otherHigh && otherLow <= high));
392
+ }
393
+
394
+ // Embedding a pattern in a larger regex drops the flags it carried, so the two
395
+ // a source can carry on its own are rewritten into it
396
+ //
397
+ // @@ Replace with inline `(?i:…)` and `(?s:…)` modifiers once Node floor reaches 24
398
+
399
+ /** @param {string} char @returns {boolean} Whether the character has a single-character counterpart */
400
+ function foldsCase(char) {
401
+ const lower = char.toLowerCase();
402
+ const upper = char.toUpperCase();
403
+ return lower !== upper && lower.length === 1 && upper.length === 1;
404
+ }
405
+
406
+ /** @param {string} char */
407
+ function isLower(char) {
408
+ return char === char.toLowerCase();
409
+ }
410
+
411
+ /**
412
+ * @param {string} source - Regex source, assumed syntactically valid
413
+ * @param {boolean} [nested] - Whether the pattern carries `v`
414
+ */
415
+ function expandDotAll(source, nested = false) {
416
+ let out = '';
417
+ let i = 0;
418
+ while (i < source.length) {
419
+ const char = source[i];
420
+ if (char === '\\') {
421
+ const length = tokenLength(source, i);
422
+ out += source.slice(i, i + length);
423
+ i += length;
424
+ } else if (char === '[') {
425
+ const end = skipCharacterClass(source, i, nested);
426
+ out += source.slice(i, end);
427
+ i = end;
428
+ } else if (char === '.') {
429
+ out += '[\\s\\S]';
430
+ i++;
431
+ } else {
432
+ out += char;
433
+ i++;
434
+ }
435
+ }
436
+ return out;
437
+ }
438
+
439
+ /**
440
+ * @param {string} source - Regex source, assumed syntactically valid
441
+ * @param {boolean} [nested] - Whether the pattern carries `v`
442
+ */
443
+ function foldCase(source, nested = false) {
444
+ let out = '';
445
+ let i = 0;
446
+ let open = 0;
447
+
448
+ while (i < source.length) {
449
+ const char = source[i] ?? '';
450
+
451
+ if (open === 0) {
452
+ if (char === '\\') {
453
+ const length = tokenLength(source, i);
454
+ out += source.slice(i, i + length);
455
+ i += length;
456
+ } else if (char === '[') {
457
+ open = 1;
458
+ out += char;
459
+ i++;
460
+ } else if (char === '(' && source.startsWith('(?<', i) &&
461
+ source[i + 3] !== '=' && source[i + 3] !== '!') {
462
+ // A capture group’s name is syntax, and folding it is a syntax error
463
+ const close = source.indexOf('>', i + 3);
464
+ const end = close === -1 ? i + 3 : close + 1;
465
+ out += source.slice(i, end);
466
+ i = end;
467
+ } else {
468
+ out += foldsCase(char) ? '[' + char.toLowerCase() + char.toUpperCase() + ']' : char;
469
+ i++;
470
+ }
471
+ continue;
472
+ }
473
+
474
+ if (nested && char === '[') {
475
+ open++;
476
+ out += char;
477
+ i++;
478
+ continue;
479
+ }
480
+
481
+ if (char === ']') {
482
+ open--;
483
+ out += char;
484
+ i++;
485
+ continue;
486
+ }
487
+
488
+ const fromLength = tokenLength(source, i);
489
+ const from = source.slice(i, i + fromLength);
490
+ const afterFrom = i + fromLength;
491
+
492
+ // A range needs the other case’s range beside it—only where both ends are
493
+ // ASCII letters, the one span whose two cases are contiguous and parallel
494
+ // (`ÿ` uppercases to `Ÿ`, 150 code points past where its range ends)
495
+ if (source[afterFrom] === '-' && afterFrom + 1 < source.length && source[afterFrom + 1] !== ']') {
496
+ const toLength = tokenLength(source, afterFrom + 1);
497
+ const to = source.slice(afterFrom + 1, afterFrom + 1 + toLength);
498
+ out += RE_ASCII_LETTER.test(from) && RE_ASCII_LETTER.test(to) && isLower(from) === isLower(to)
499
+ ? from + '-' + to + (isLower(from)
500
+ ? from.toUpperCase() + '-' + to.toUpperCase()
501
+ : from.toLowerCase() + '-' + to.toLowerCase())
502
+ : from + '-' + to;
503
+ i = afterFrom + 1 + toLength;
504
+ continue;
505
+ }
506
+
507
+ out += fromLength === 1 && foldsCase(from) ? from.toLowerCase() + from.toUpperCase() : from;
508
+ i = afterFrom;
509
+ }
510
+
511
+ return out;
512
+ }
513
+
514
+ /**
515
+ * Whether a source anchors anywhere, so that `m` would move where it matches
516
+ * @param {string} source @param {boolean} [nested] - Whether the pattern carries `v`
517
+ */
518
+ function hasAnchor(source, nested = false) {
519
+ let i = 0;
520
+ while (i < source.length) {
521
+ if (source[i] === '\\') i += tokenLength(source, i);
522
+ // Inside a class `^` negates and `$` is a member, so neither anchors there
523
+ else if (source[i] === '[') i = skipCharacterClass(source, i, nested);
524
+ else if (source[i] === '^' || source[i] === '$') return true;
525
+ else i++;
526
+ }
527
+ return false;
528
+ }
529
+
530
+ // A property escape or a code point escape, both of which read as literal text
531
+ // where the flag that gives them meaning is gone
532
+ const RE_UNICODE_ESCAPE = /\\[pP]\{|\\u\{/;
533
+
534
+ // The characters `u` folds by Unicode rules and a source without it does not:
535
+ // `/s/iu` matches `\u017F` and `/k/iu` matches `\u212A`, where neither does on its
536
+ // own. Characters past the BMP fold this way, too, and are caught as astral first.
537
+ const RE_UNICODE_FOLDING = /[\u004B\u0053\u006B\u0073\u00C5\u00DF\u00E5\u017F\u0398\u03A9\u03B8\u03C9\u03D1\u03F4\u1E9E\u1F80-\u1FAF\u1FB3\u1FBC\u1FC3\u1FCC\u1FF3\u1FFC\u2126\u212A\u212B]/;
538
+
539
+ /**
540
+ * Which flag changes what a source matches, so that embedding the source where the
541
+ * flag cannot follow would silently match something else. `u` and `v` only narrow
542
+ * what syntax is legal, so a source valid under either stays valid without it—the
543
+ * difference never surfaces as a syntax error, and this stands in for the one that
544
+ * would otherwise be raised.
545
+ * @param {RegExp} pattern
546
+ * @returns {'u' | 'v' | 'm' | null} The flag the source depends on, or `null` where
547
+ * every flag it carries would leave the source matching the same
548
+ */
549
+ function lostFlag(pattern) {
550
+ const { source, unicode, unicodeSets } = pattern;
551
+ // `m` moves where `^` and `$` match, and only matters where the source has one
552
+ if (pattern.multiline && hasAnchor(source, unicodeSets)) return 'm';
553
+ if (!unicode && !unicodeSets) return null;
554
+ const flag = unicodeSets ? 'v' : 'u';
555
+ if (RE_UNICODE_ESCAPE.test(source)) return flag;
556
+ // Case folding under `i` follows Unicode rules only while the flag is there. A
557
+ // character can be written literally or as an escape, so the source is read
558
+ // token by token rather than scanned for the characters themselves; `singleCode`
559
+ // leaves out `\d` and friends, which fold alike with the flag and without it.
560
+ if (pattern.ignoreCase) {
561
+ for (let i = 0; i < source.length;) {
562
+ const length = tokenLength(source, i);
563
+ const code = singleCode(tokenRanges(source.slice(i, i + length), true));
564
+ if (code !== null && RE_UNICODE_FOLDING.test(String.fromCodePoint(code))) return flag;
565
+ i += length;
566
+ }
567
+ }
568
+ // Past the BMP a character is one code point with the flag, two units without,
569
+ // which is what a quantifier or a class beside it would go on to read wrongly
570
+ for (const char of source) {
571
+ if ((char.codePointAt(0) ?? 0) > 0xFFFF) return flag;
572
+ }
573
+ if (!unicodeSets) return null;
574
+
575
+ // `v` alone lets a class nest, subtract, intersect, and hold whole strings
576
+ let i = 0;
577
+ while (i < source.length) {
578
+ if (source[i] === '\\') {
579
+ i += tokenLength(source, i);
580
+ continue;
581
+ }
582
+ if (source[i] !== '[') {
583
+ i++;
584
+ continue;
585
+ }
586
+ const end = skipCharacterClass(source, i, true);
587
+ for (let j = i + 1; j < end - 1;) {
588
+ if (source[j] === '\\') {
589
+ if (source.startsWith('\\q{', j)) return 'v';
590
+ j += tokenLength(source, j);
591
+ continue;
592
+ }
593
+ if (source[j] === '[' || source.startsWith('--', j) || source.startsWith('&&', j)) return 'v';
594
+ j++;
595
+ }
596
+ i = end;
597
+ }
598
+ return null;
599
+ }
600
+
601
+ /**
602
+ * A pattern’s source, rewritten to match the way its own flags make it match, for
603
+ * embedding in a larger regex that cannot carry them. `i` and `s` fit into a
604
+ * source; `m`, `u`, and `v` do not, and are left to the pattern it joins. Where
605
+ * `i` cannot be written in—a backreference, a range outside ASCII, a range that
606
+ * spans letters without being one—the source stands as it is, matching less than
607
+ * the pattern would rather than more.
608
+ * @param {RegExp} pattern
609
+ * @returns {string}
610
+ */
611
+ function embedSource(pattern) {
612
+ let source = pattern.source;
613
+ if (pattern.dotAll) source = expandDotAll(source, pattern.unicodeSets);
614
+ if (pattern.ignoreCase) source = foldCase(source, pattern.unicodeSets);
615
+ return source;
616
+ }
617
+
618
+ /**
619
+ * @typedef {{text: string, ranges: [number, number][] | null}} Repeat - An unbounded
620
+ * repeat, as it reads and as the set of characters it matches
621
+ */
622
+
623
+ /**
624
+ * @param {Repeat[]} reachable - Repeats that could still sit adjacent to what comes next
625
+ * @param {Repeat} repeat
626
+ * @returns {boolean} Whether the two can consume the same character, and so split
627
+ * the same run of input between them
628
+ */
629
+ function splitsWith(reachable, repeat) {
630
+ return reachable.some(earlier => earlier.text === repeat.text ||
631
+ (!!repeat.ranges && !!earlier.ranges && rangesIntersect(repeat.ranges, earlier.ranges)));
632
+ }
633
+
634
+ /**
635
+ * @param {Repeat[]} into
636
+ * @param {Repeat[]} repeats
637
+ * @returns {void} Adds the repeats, keeping only the most recent ones the walk
638
+ * still compares against
639
+ */
640
+ function collectReachable(into, repeats) {
641
+ for (const repeat of repeats) {
642
+ into.push(repeat);
643
+ if (into.length > MAX_REACHABLE) into.shift();
644
+ }
645
+ }
646
+
647
+ /**
648
+ * Walk a regex source for the shapes whose backtracking blows up: an unlimited
649
+ * quantifier over a group that itself contains a variable quantifier (`(a+)+`,
650
+ * `(a?)+`) or alternates, and two unbounded repeats that can consume the same
651
+ * character with only atoms matching empty between them (`.*.*`, `[a]*a*`,
652
+ * `\w*\d*`, `a*b*a*`, `a*(b*)a*`, `(a*)a*`). A lone unlimited quantifier stays
653
+ * linear, so `[\s\S]*?` up to a literal terminator passes.
654
+ * @param {string} source - Regex source, assumed syntactically valid
655
+ * @param {number} [depth] - Group nesting level of this call
656
+ * @param {boolean} [nested] - Whether the pattern carries `v`
657
+ * @returns {{risky: boolean, varies: boolean, alternates: boolean, deep: boolean,
658
+ * empty: boolean, leading: Repeat[], trailing: Repeat[]}} What the walk found,
659
+ * with the repeats that reach the source’s start and end for a caller to compare
660
+ * against what stands beside it
661
+ */
662
+ function analyzeQuantifiers(source, depth = 0, nested = false) {
663
+ if (depth > MAX_PATTERN_DEPTH) return { risky: true, varies: true, alternates: true, deep: true, empty: true, leading: [], trailing: [] };
664
+
665
+ let risky = false;
666
+ let varies = false;
667
+ let alternates = false;
668
+ let deep = false;
669
+ // Whether every atom of the branch being read matches empty, and whether any
670
+ // branch before it did
671
+ let branchEmpty = true;
672
+ let anyBranchEmpty = false;
673
+ // The unbounded repeats that could still sit adjacent to what comes next, most
674
+ // recent last; an atom matching empty leaves the ones before it reachable
675
+ /** @type {Repeat[]} */
676
+ const reachable = [];
677
+ // The repeats a caller can reach from either end, gathered across branches
678
+ /** @type {Repeat[]} */
679
+ const leading = [];
680
+ /** @type {Repeat[]} */
681
+ const trailing = [];
682
+ /** @type {Repeat[]} */
683
+ let branchLeading = [];
684
+ let i = 0;
685
+
686
+ while (i < source.length) {
687
+ const start = i;
688
+ const char = source[i];
689
+ /** @type {ReturnType<typeof analyzeQuantifiers> | null} */
690
+ let group = null;
691
+
692
+ if (char === '|') {
693
+ // Alternatives are separate expressions, so nothing carries across
694
+ alternates = true;
695
+ anyBranchEmpty ||= branchEmpty;
696
+ branchEmpty = true;
697
+ collectReachable(leading, branchLeading);
698
+ collectReachable(trailing, reachable);
699
+ branchLeading = [];
700
+ reachable.length = 0;
701
+ i++;
702
+ continue;
703
+ } else if (char === '\\') {
704
+ i += tokenLength(source, i);
705
+ } else if (char === '[') {
706
+ i = skipCharacterClass(source, i, nested);
707
+ } else if (char === '(') {
708
+ let open = 1;
709
+ i++;
710
+ while (i < source.length && open > 0) {
711
+ const inner = source[i];
712
+ if (inner === '\\') {
713
+ i += tokenLength(source, i);
714
+ } else if (inner === '[') {
715
+ i = skipCharacterClass(source, i, nested);
716
+ } else {
717
+ if (inner === '(') open++;
718
+ else if (inner === ')') open--;
719
+ i++;
720
+ }
721
+ }
722
+ // Drop the group prefix (`?:`, `?=`, `?<name>`, …) before reading the body
723
+ group = analyzeQuantifiers(source.slice(start + 1, i - 1).replace(/^\?(?:[:=!]|<[=!]|<[^>]*>)/, ''), depth + 1, nested);
724
+ } else {
725
+ i++;
726
+ }
727
+
728
+ const atom = source.slice(start, i);
729
+ RE_QUANTIFIER.lastIndex = i;
730
+ const quantifier = RE_QUANTIFIER.exec(source);
731
+ const upper = quantifier?.[2];
732
+ const repeats = !!quantifier && (quantifier[0][0] === '*' || quantifier[0][0] === '+' || upper === '');
733
+ // A count that can vary—anything but `{n}` and its `{n,n}` spelling—makes
734
+ // the group ambiguous about how much it consumes, which multiplies under an
735
+ // unlimited repeat
736
+ const exact = !!quantifier && quantifier[0][0] === '{' &&
737
+ (upper === undefined || (upper !== '' && Number(upper) === Number(quantifier[1])));
738
+ const variable = !!quantifier && !exact;
739
+ // An atom matches empty where it is zero-width, where its quantifier may
740
+ // repeat none, or where a group body matches empty however often it repeats—
741
+ // `(b*)`, carrying the quantifier inside the group rather than on it
742
+ const least = quantifier && (quantifier[0][0] === '+' ? 1
743
+ : quantifier[0][0] === '{' ? Number(quantifier[1]) : 0);
744
+ const body = RE_ZERO_WIDTH.test(atom) || (!!group && group.empty);
745
+ const nullable = body || (!!quantifier && least === 0);
746
+ if (quantifier) i = RE_QUANTIFIER.lastIndex;
747
+
748
+ // A lookaround is atomic: It backtracks nothing, so what it holds neither
749
+ // reaches out of it nor is reached into
750
+ const opaque = RE_ZERO_WIDTH.test(atom);
751
+ if (group) {
752
+ if (group.risky || (repeats && (group.varies || group.alternates))) risky = true;
753
+ if (group.varies) varies = true;
754
+ // A group alternates whether the `|` sits at its top level or deeper
755
+ if (group.alternates) alternates = true;
756
+ if (group.deep) deep = true;
757
+ // A group is no wall: What it repeats at its start splits the same run as
758
+ // what stands unbounded before it
759
+ if (!opaque && group.leading.some(repeat => splitsWith(reachable, repeat))) risky = true;
760
+ }
761
+ if (variable) varies = true;
762
+ const text = unwrapAtom(atom);
763
+ const ranges = atomRanges(text, nested);
764
+ if (repeats && splitsWith(reachable, { text, ranges })) risky = true;
765
+
766
+ // Whatever a caller could reach before this atom, it reaches this one, too
767
+ if (branchEmpty) {
768
+ if (repeats) collectReachable(branchLeading, [{ text, ranges }]);
769
+ if (group && !opaque) collectReachable(branchLeading, group.leading);
770
+ }
771
+ // Nothing reaches past an atom that has to consume something
772
+ if (!nullable) {
773
+ reachable.length = 0;
774
+ branchEmpty = false;
775
+ }
776
+ if (group && !opaque) collectReachable(reachable, group.trailing);
777
+ if (repeats) collectReachable(reachable, [{ text, ranges }]);
778
+ }
779
+
780
+ collectReachable(leading, branchLeading);
781
+ collectReachable(trailing, reachable);
782
+
783
+ return { risky, varies, alternates, deep, empty: anyBranchEmpty || branchEmpty, leading, trailing };
784
+ }
785
+
786
+ /**
787
+ * @param {RegExp | string} pattern - Pattern to judge, or a bare source to read
788
+ * as it stands
789
+ * @returns {string | null} What makes the pattern a backtracking risk, phrased
790
+ * to follow the pattern itself, or `null` where it is none
791
+ */
792
+ function describeQuantifierRisk(pattern) {
793
+ const source = typeof pattern === 'string' ? pattern : pattern.source;
794
+ // A pattern too big to read is refused for that, not for a shape nobody saw;
795
+ // its own length is what counts, not the length folding case inflates it to
796
+ if (source.length > MAX_PATTERN_LENGTH) {
797
+ return `runs past ${MAX_PATTERN_LENGTH.toLocaleString()} characters, too long to analyze for catastrophic backtracking—shorten it, or split it into several patterns`;
798
+ }
799
+ // `i` and `s` change what the source matches, so they are written into it
800
+ // before the shapes are read
801
+ const analysis = typeof pattern === 'string'
802
+ ? analyzeQuantifiers(source)
803
+ : analyzeQuantifiers(embedSource(pattern), 0, pattern.unicodeSets);
804
+ if (analysis.deep) {
805
+ return `nests groups more than ${MAX_PATTERN_DEPTH.toLocaleString()} deep, too deep to analyze for catastrophic backtracking—flatten it, or split it into several patterns`;
806
+ }
807
+ return analysis.risky
808
+ ? 'compounds quantifiers or alternation in a way that may cause ReDoS—bound the repetition (e.g., `{0,1000}`) instead'
809
+ : null;
810
+ }
148
811
 
149
812
  /**
150
813
  * Find the index of the `>` that closes an opening tag, correctly skipping
@@ -168,6 +831,8 @@ function findTagEnd(html, pos) {
168
831
  return -1;
169
832
  }
170
833
 
834
+ // Exports
835
+
171
836
  export {
172
837
  stableStringify,
173
838
  findTagEnd,
@@ -179,5 +844,8 @@ export {
179
844
  isThenable,
180
845
  lowercase,
181
846
  replaceAsync,
182
- parseRegExp
847
+ parseRegExp,
848
+ embedSource,
849
+ lostFlag,
850
+ describeQuantifierRisk
183
851
  };