html-minifier-next 7.6.0 → 8.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -21
- package/cli.js +152 -50
- package/dist/types/htmlminifier.d.ts +6 -4
- package/dist/types/htmlminifier.d.ts.map +1 -1
- package/dist/types/lib/attributes.d.ts +2 -2
- package/dist/types/lib/attributes.d.ts.map +1 -1
- package/dist/types/lib/constants.d.ts +2 -2
- package/dist/types/lib/constants.d.ts.map +1 -1
- package/dist/types/lib/fragments.d.ts +46 -0
- package/dist/types/lib/fragments.d.ts.map +1 -0
- package/dist/types/lib/option-definitions.d.ts +21 -198
- package/dist/types/lib/option-definitions.d.ts.map +1 -1
- package/dist/types/lib/options.d.ts +1 -1
- package/dist/types/lib/options.d.ts.map +1 -1
- package/dist/types/lib/unused-css.d.ts +5 -4
- package/dist/types/lib/unused-css.d.ts.map +1 -1
- package/dist/types/lib/utils.d.ts +34 -1
- package/dist/types/lib/utils.d.ts.map +1 -1
- package/dist/types/tokenchain.d.ts +2 -2
- package/dist/types/tokenchain.d.ts.map +1 -1
- package/html-minifier-next.schema.json +4 -5
- package/package.json +1 -1
- package/src/htmlminifier.js +49 -64
- package/src/htmlparser.js +4 -4
- package/src/lib/attributes.js +63 -6
- package/src/lib/constants.js +6 -8
- package/src/lib/fragments.js +274 -0
- package/src/lib/option-definitions.js +12 -4
- package/src/lib/options.js +44 -17
- package/src/lib/unused-css.js +5 -22
- package/src/lib/utils.js +671 -3
- package/src/tokenchain.js +37 -30
package/src/lib/utils.js
CHANGED
|
@@ -126,7 +126,8 @@ async function replaceAsync(str, regex, asyncFn) {
|
|
|
126
126
|
});
|
|
127
127
|
|
|
128
128
|
const data = await Promise.all(promises);
|
|
129
|
-
|
|
129
|
+
let next = 0;
|
|
130
|
+
return str.replace(regex, () => data[next++] ?? '');
|
|
130
131
|
}
|
|
131
132
|
|
|
132
133
|
// String patterns to RegExp conversion (for JSON config support)
|
|
@@ -144,7 +145,669 @@ function parseRegExp(value) {
|
|
|
144
145
|
return value;
|
|
145
146
|
}
|
|
146
147
|
|
|
147
|
-
//
|
|
148
|
+
// Regex source reading, shared by the analysis and the rewriting below
|
|
149
|
+
|
|
150
|
+
const RE_ASCII_LETTER = /^[a-zA-Z]$/;
|
|
151
|
+
const RE_HEX_PAIR = /^[0-9a-fA-F]{2}$/;
|
|
152
|
+
const RE_HEX_QUAD = /^[0-9a-fA-F]{4}$/;
|
|
153
|
+
|
|
154
|
+
/**
|
|
155
|
+
* @param {string} source @param {number} index - Index of the opening `[`
|
|
156
|
+
* @param {boolean} [nested] - Whether the pattern carries `v`, where a class nests
|
|
157
|
+
* inside a class and the first `]` need not be the one that closes it
|
|
158
|
+
* @returns {number} Index just past the closing `]`
|
|
159
|
+
*/
|
|
160
|
+
function skipCharacterClass(source, index, nested = false) {
|
|
161
|
+
let i = index + 1;
|
|
162
|
+
let open = 1;
|
|
163
|
+
while (i < source.length && open > 0) {
|
|
164
|
+
const char = source[i];
|
|
165
|
+
if (char === '\\') i += tokenLength(source, i);
|
|
166
|
+
else {
|
|
167
|
+
if (nested && char === '[') open++;
|
|
168
|
+
else if (char === ']') open--;
|
|
169
|
+
i++;
|
|
170
|
+
}
|
|
171
|
+
}
|
|
172
|
+
return i;
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* Length of the token at `index`, so that a multi-character escape is read whole
|
|
177
|
+
* rather than leaving its tail to be mistaken for literals.
|
|
178
|
+
* @param {string} source @param {number} index
|
|
179
|
+
*/
|
|
180
|
+
function tokenLength(source, index) {
|
|
181
|
+
if (source[index] !== '\\') return 1;
|
|
182
|
+
const next = source[index + 1];
|
|
183
|
+
if (next === 'x' && RE_HEX_PAIR.test(source.slice(index + 2, index + 4))) return 4;
|
|
184
|
+
if (next === 'u') {
|
|
185
|
+
if (source[index + 2] === '{') {
|
|
186
|
+
const close = source.indexOf('}', index + 3);
|
|
187
|
+
if (close !== -1) return close - index + 1;
|
|
188
|
+
}
|
|
189
|
+
if (RE_HEX_QUAD.test(source.slice(index + 2, index + 6))) return 6;
|
|
190
|
+
}
|
|
191
|
+
if (next === 'c' && RE_ASCII_LETTER.test(source[index + 2] ?? '')) return 3;
|
|
192
|
+
// A `v` class holds string literals, which fold as strings or not at all
|
|
193
|
+
if (next === 'q' && source[index + 2] === '{') {
|
|
194
|
+
const close = source.indexOf('}', index + 3);
|
|
195
|
+
if (close !== -1) return close - index + 1;
|
|
196
|
+
}
|
|
197
|
+
// A property name or a group name is syntax, not text to fold
|
|
198
|
+
if ((next === 'p' || next === 'P') && source[index + 2] === '{') {
|
|
199
|
+
const close = source.indexOf('}', index + 3);
|
|
200
|
+
if (close !== -1) return close - index + 1;
|
|
201
|
+
}
|
|
202
|
+
if (next === 'k' && source[index + 2] === '<') {
|
|
203
|
+
const close = source.indexOf('>', index + 3);
|
|
204
|
+
if (close !== -1) return close - index + 1;
|
|
205
|
+
}
|
|
206
|
+
return 2;
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
// ReDoS risk detection for user-supplied patterns
|
|
210
|
+
|
|
211
|
+
// Quantifier following an atom, with its optional lazy `?`; the captures hold
|
|
212
|
+
// the bounds of a `{n,m}` form, the upper one empty when the form is `{n,}`
|
|
213
|
+
const RE_QUANTIFIER = /(?:[*+?]|\{(\d+)(?:,(\d*))?\})\??/y;
|
|
214
|
+
|
|
215
|
+
// Bounds for the walk below: Patterns beyond either are judged risky unread,
|
|
216
|
+
// which keeps a pathological source from nesting the analysis into a stack
|
|
217
|
+
// overflow or making it rescan its groups once per level
|
|
218
|
+
const MAX_PATTERN_LENGTH = 10000;
|
|
219
|
+
const MAX_PATTERN_DEPTH = 50;
|
|
220
|
+
|
|
221
|
+
// A group that only wraps its body backtracks the way that body does, so
|
|
222
|
+
// `(?:a)` and `a` count as the same atom; lookarounds are left alone
|
|
223
|
+
const RE_TRANSPARENT_GROUP = /^\((?:\?:|\?<[^>=!][^>]*>)?([\s\S]*)\)$/;
|
|
224
|
+
|
|
225
|
+
// A quantifier of exactly one repetition, which is notation rather than shape
|
|
226
|
+
const RE_EXACT_ONE = /\{1(?:,1)?\}\??$/;
|
|
227
|
+
|
|
228
|
+
// An atom that matches without consuming, so what precedes it stays adjacent to
|
|
229
|
+
// what follows; a group that can match empty is left out, being rare enough that
|
|
230
|
+
// missing it costs less than reading every body for it
|
|
231
|
+
const RE_ZERO_WIDTH = /^(?:[$^]|\\[bB]|\((?:\?[=!]|\?<[=!]))/;
|
|
232
|
+
|
|
233
|
+
// How many repeats an atom may still be adjacent to across atoms matching empty;
|
|
234
|
+
// comparing against every one of them is what the bound keeps from turning
|
|
235
|
+
// quadratic on a pattern that is nothing but repeats
|
|
236
|
+
const MAX_REACHABLE = 50;
|
|
237
|
+
|
|
238
|
+
/** @param {string} atom - Atom as it reads in the source */
|
|
239
|
+
function unwrapAtom(atom) {
|
|
240
|
+
let inner = atom;
|
|
241
|
+
for (let level = 0; level < MAX_PATTERN_DEPTH; level++) {
|
|
242
|
+
const next = inner.replace(RE_EXACT_ONE, '').replace(RE_TRANSPARENT_GROUP, '$1');
|
|
243
|
+
if (next === inner) break;
|
|
244
|
+
inner = next;
|
|
245
|
+
}
|
|
246
|
+
return inner;
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
// Two atoms that repeat unboundedly side by side split a run between them in as
|
|
250
|
+
// many ways as the run is long, whenever both can consume the same character.
|
|
251
|
+
// Comparing what they match, rather than how they are spelled, is what catches
|
|
252
|
+
// `[a]*a*` and `\w*\d*` alongside `a*a*`.
|
|
253
|
+
|
|
254
|
+
const MAX_CODE_POINT = 0x10FFFF;
|
|
255
|
+
|
|
256
|
+
/** @type {Record<string, [number, number][]>} */
|
|
257
|
+
const CLASS_ESCAPE_RANGES = {
|
|
258
|
+
d: [[0x30, 0x39]],
|
|
259
|
+
w: [[0x30, 0x39], [0x41, 0x5A], [0x5F, 0x5F], [0x61, 0x7A]],
|
|
260
|
+
s: [[0x09, 0x0D], [0x20, 0x20], [0xA0, 0xA0], [0x1680, 0x1680], [0x2000, 0x200A],
|
|
261
|
+
[0x2028, 0x2029], [0x202F, 0x202F], [0x205F, 0x205F], [0x3000, 0x3000], [0xFEFF, 0xFEFF]]
|
|
262
|
+
};
|
|
263
|
+
|
|
264
|
+
/** @type {Record<string, number>} */
|
|
265
|
+
const CONTROL_ESCAPE_CODES = { 0: 0x00, f: 0x0C, n: 0x0A, r: 0x0D, t: 0x09, v: 0x0B };
|
|
266
|
+
|
|
267
|
+
// An escape that stands for something other than one character of text
|
|
268
|
+
const RE_NON_CHARACTER_ESCAPE = /[bBdDkpPsSwW1-9]/;
|
|
269
|
+
|
|
270
|
+
/** @param {[number, number][]} ranges */
|
|
271
|
+
function complement(ranges) {
|
|
272
|
+
const sorted = [...ranges].sort((one, other) => one[0] - other[0]);
|
|
273
|
+
/** @type {[number, number][]} */
|
|
274
|
+
const out = [];
|
|
275
|
+
let next = 0;
|
|
276
|
+
for (const [low, high] of sorted) {
|
|
277
|
+
if (low > next) out.push([next, low - 1]);
|
|
278
|
+
next = Math.max(next, high + 1);
|
|
279
|
+
}
|
|
280
|
+
if (next <= MAX_CODE_POINT) out.push([next, MAX_CODE_POINT]);
|
|
281
|
+
return out;
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
// `.` as a bare source reads it, without the `s` flag the source cannot carry
|
|
285
|
+
const DOT_RANGES = complement([[0x0A, 0x0A], [0x0D, 0x0D], [0x2028, 0x2029]]);
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* @param {string} token - One token, as `tokenLength` measures it
|
|
289
|
+
* @param {boolean} [inClass] - Whether the token sits inside a character class
|
|
290
|
+
* @returns {[number, number][] | null} What it matches, or `null` where it is not
|
|
291
|
+
* one character of text
|
|
292
|
+
*/
|
|
293
|
+
function tokenRanges(token, inClass = false) {
|
|
294
|
+
if (token[0] !== '\\') {
|
|
295
|
+
const code = token.codePointAt(0) ?? 0;
|
|
296
|
+
return String.fromCodePoint(code) === token ? [[code, code]] : null;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const kind = token[1] ?? '';
|
|
300
|
+
const named = CLASS_ESCAPE_RANGES[kind.toLowerCase()];
|
|
301
|
+
if (named) return kind === kind.toLowerCase() ? named : complement(named);
|
|
302
|
+
|
|
303
|
+
const control = CONTROL_ESCAPE_CODES[kind];
|
|
304
|
+
if (control !== undefined) return [[control, control]];
|
|
305
|
+
|
|
306
|
+
if (kind === 'x' || kind === 'u') {
|
|
307
|
+
const code = Number.parseInt(token[2] === '{' ? token.slice(3, -1) : token.slice(2), 16);
|
|
308
|
+
return Number.isNaN(code) || code > MAX_CODE_POINT ? null : [[code, code]];
|
|
309
|
+
}
|
|
310
|
+
if (kind === 'c') {
|
|
311
|
+
const code = token.charCodeAt(2) % 32;
|
|
312
|
+
return [[code, code]];
|
|
313
|
+
}
|
|
314
|
+
// `\b` asserts a word boundary on its own, and is a backspace inside a class
|
|
315
|
+
if (kind === 'b' && inClass) return [[0x08, 0x08]];
|
|
316
|
+
// `\q{…}` stands for whole strings, not for a character
|
|
317
|
+
if (token.startsWith('\\q{')) return null;
|
|
318
|
+
if (RE_NON_CHARACTER_ESCAPE.test(kind)) return null;
|
|
319
|
+
|
|
320
|
+
// The rest is punctuation escaped to be read as itself
|
|
321
|
+
const code = token.codePointAt(1) ?? 0;
|
|
322
|
+
return String.fromCodePoint(code) === token.slice(1) ? [[code, code]] : null;
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/** @param {[number, number][] | null} ranges @returns {number | null} The one code point it holds */
|
|
326
|
+
function singleCode(ranges) {
|
|
327
|
+
const [range] = ranges ?? [];
|
|
328
|
+
return ranges?.length === 1 && range && range[0] === range[1] ? range[0] : null;
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* @param {string} atom - Atom as it reads in the source, already unwrapped
|
|
333
|
+
* @param {boolean} [nested] - Whether the pattern carries `v`
|
|
334
|
+
* @param {number} [depth] - Class nesting level of this call
|
|
335
|
+
* @returns {[number, number][] | null} What it matches, or `null` where it is not
|
|
336
|
+
* one character of text, or one this does not read
|
|
337
|
+
*/
|
|
338
|
+
function atomRanges(atom, nested = false, depth = 0) {
|
|
339
|
+
if (depth > MAX_PATTERN_DEPTH) return null;
|
|
340
|
+
if (atom === '.') return DOT_RANGES;
|
|
341
|
+
if (atom[0] !== '[') return tokenRanges(atom);
|
|
342
|
+
if (!atom.endsWith(']')) return null;
|
|
343
|
+
|
|
344
|
+
let i = atom[1] === '^' ? 2 : 1;
|
|
345
|
+
const negated = i === 2;
|
|
346
|
+
const end = atom.length - 1;
|
|
347
|
+
/** @type {[number, number][]} */
|
|
348
|
+
const members = [];
|
|
349
|
+
|
|
350
|
+
while (i < end) {
|
|
351
|
+
// A `v` class nests, and nesting alone is a union to read through
|
|
352
|
+
if (nested && atom[i] === '[') {
|
|
353
|
+
const close = skipCharacterClass(atom, i, true);
|
|
354
|
+
const inner = atomRanges(atom.slice(i, close), true, depth + 1);
|
|
355
|
+
if (!inner) return null;
|
|
356
|
+
members.push(...inner);
|
|
357
|
+
i = close;
|
|
358
|
+
continue;
|
|
359
|
+
}
|
|
360
|
+
// Subtraction and intersection are not unions, and are left unread
|
|
361
|
+
//
|
|
362
|
+
// @@ Read `--` and `&&` as set difference and intersection
|
|
363
|
+
// (so that a `v` class built is compared rather than passed unjudged)
|
|
364
|
+
if (nested && (atom.startsWith('--', i) || atom.startsWith('&&', i))) return null;
|
|
365
|
+
|
|
366
|
+
const fromLength = tokenLength(atom, i);
|
|
367
|
+
const from = tokenRanges(atom.slice(i, i + fromLength), true);
|
|
368
|
+
if (!from) return null;
|
|
369
|
+
i += fromLength;
|
|
370
|
+
|
|
371
|
+
// A dash right before the closing `]` is a member, not the start of a range
|
|
372
|
+
if (atom[i] !== '-' || i + 1 >= end) {
|
|
373
|
+
members.push(...from);
|
|
374
|
+
continue;
|
|
375
|
+
}
|
|
376
|
+
const toLength = tokenLength(atom, i + 1);
|
|
377
|
+
// Only single characters bound a range; `\d-z` is not a range at all
|
|
378
|
+
const low = singleCode(from);
|
|
379
|
+
const high = singleCode(tokenRanges(atom.slice(i + 1, i + 1 + toLength), true));
|
|
380
|
+
if (low === null || high === null) return null;
|
|
381
|
+
members.push([low, high]);
|
|
382
|
+
i += 1 + toLength;
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
return negated ? complement(members) : members;
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
/** @param {[number, number][]} ranges @param {[number, number][]} other */
|
|
389
|
+
function rangesIntersect(ranges, other) {
|
|
390
|
+
return ranges.some(([low, high]) => other.some(([otherLow, otherHigh]) =>
|
|
391
|
+
low <= otherHigh && otherLow <= high));
|
|
392
|
+
}
|
|
393
|
+
|
|
394
|
+
// Embedding a pattern in a larger regex drops the flags it carried, so the two
|
|
395
|
+
// a source can carry on its own are rewritten into it
|
|
396
|
+
//
|
|
397
|
+
// @@ Replace with inline `(?i:…)` and `(?s:…)` modifiers once Node floor reaches 24
|
|
398
|
+
|
|
399
|
+
/** @param {string} char @returns {boolean} Whether the character has a single-character counterpart */
|
|
400
|
+
function foldsCase(char) {
|
|
401
|
+
const lower = char.toLowerCase();
|
|
402
|
+
const upper = char.toUpperCase();
|
|
403
|
+
return lower !== upper && lower.length === 1 && upper.length === 1;
|
|
404
|
+
}
|
|
405
|
+
|
|
406
|
+
/** @param {string} char */
|
|
407
|
+
function isLower(char) {
|
|
408
|
+
return char === char.toLowerCase();
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
/**
|
|
412
|
+
* @param {string} source - Regex source, assumed syntactically valid
|
|
413
|
+
* @param {boolean} [nested] - Whether the pattern carries `v`
|
|
414
|
+
*/
|
|
415
|
+
function expandDotAll(source, nested = false) {
|
|
416
|
+
let out = '';
|
|
417
|
+
let i = 0;
|
|
418
|
+
while (i < source.length) {
|
|
419
|
+
const char = source[i];
|
|
420
|
+
if (char === '\\') {
|
|
421
|
+
const length = tokenLength(source, i);
|
|
422
|
+
out += source.slice(i, i + length);
|
|
423
|
+
i += length;
|
|
424
|
+
} else if (char === '[') {
|
|
425
|
+
const end = skipCharacterClass(source, i, nested);
|
|
426
|
+
out += source.slice(i, end);
|
|
427
|
+
i = end;
|
|
428
|
+
} else if (char === '.') {
|
|
429
|
+
out += '[\\s\\S]';
|
|
430
|
+
i++;
|
|
431
|
+
} else {
|
|
432
|
+
out += char;
|
|
433
|
+
i++;
|
|
434
|
+
}
|
|
435
|
+
}
|
|
436
|
+
return out;
|
|
437
|
+
}
|
|
438
|
+
|
|
439
|
+
/**
|
|
440
|
+
* @param {string} source - Regex source, assumed syntactically valid
|
|
441
|
+
* @param {boolean} [nested] - Whether the pattern carries `v`
|
|
442
|
+
*/
|
|
443
|
+
function foldCase(source, nested = false) {
|
|
444
|
+
let out = '';
|
|
445
|
+
let i = 0;
|
|
446
|
+
let open = 0;
|
|
447
|
+
|
|
448
|
+
while (i < source.length) {
|
|
449
|
+
const char = source[i] ?? '';
|
|
450
|
+
|
|
451
|
+
if (open === 0) {
|
|
452
|
+
if (char === '\\') {
|
|
453
|
+
const length = tokenLength(source, i);
|
|
454
|
+
out += source.slice(i, i + length);
|
|
455
|
+
i += length;
|
|
456
|
+
} else if (char === '[') {
|
|
457
|
+
open = 1;
|
|
458
|
+
out += char;
|
|
459
|
+
i++;
|
|
460
|
+
} else if (char === '(' && source.startsWith('(?<', i) &&
|
|
461
|
+
source[i + 3] !== '=' && source[i + 3] !== '!') {
|
|
462
|
+
// A capture group’s name is syntax, and folding it is a syntax error
|
|
463
|
+
const close = source.indexOf('>', i + 3);
|
|
464
|
+
const end = close === -1 ? i + 3 : close + 1;
|
|
465
|
+
out += source.slice(i, end);
|
|
466
|
+
i = end;
|
|
467
|
+
} else {
|
|
468
|
+
out += foldsCase(char) ? '[' + char.toLowerCase() + char.toUpperCase() + ']' : char;
|
|
469
|
+
i++;
|
|
470
|
+
}
|
|
471
|
+
continue;
|
|
472
|
+
}
|
|
473
|
+
|
|
474
|
+
if (nested && char === '[') {
|
|
475
|
+
open++;
|
|
476
|
+
out += char;
|
|
477
|
+
i++;
|
|
478
|
+
continue;
|
|
479
|
+
}
|
|
480
|
+
|
|
481
|
+
if (char === ']') {
|
|
482
|
+
open--;
|
|
483
|
+
out += char;
|
|
484
|
+
i++;
|
|
485
|
+
continue;
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
const fromLength = tokenLength(source, i);
|
|
489
|
+
const from = source.slice(i, i + fromLength);
|
|
490
|
+
const afterFrom = i + fromLength;
|
|
491
|
+
|
|
492
|
+
// A range needs the other case’s range beside it—only where both ends are
|
|
493
|
+
// ASCII letters, the one span whose two cases are contiguous and parallel
|
|
494
|
+
// (`ÿ` uppercases to `Ÿ`, 150 code points past where its range ends)
|
|
495
|
+
if (source[afterFrom] === '-' && afterFrom + 1 < source.length && source[afterFrom + 1] !== ']') {
|
|
496
|
+
const toLength = tokenLength(source, afterFrom + 1);
|
|
497
|
+
const to = source.slice(afterFrom + 1, afterFrom + 1 + toLength);
|
|
498
|
+
out += RE_ASCII_LETTER.test(from) && RE_ASCII_LETTER.test(to) && isLower(from) === isLower(to)
|
|
499
|
+
? from + '-' + to + (isLower(from)
|
|
500
|
+
? from.toUpperCase() + '-' + to.toUpperCase()
|
|
501
|
+
: from.toLowerCase() + '-' + to.toLowerCase())
|
|
502
|
+
: from + '-' + to;
|
|
503
|
+
i = afterFrom + 1 + toLength;
|
|
504
|
+
continue;
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
out += fromLength === 1 && foldsCase(from) ? from.toLowerCase() + from.toUpperCase() : from;
|
|
508
|
+
i = afterFrom;
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
return out;
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
/**
|
|
515
|
+
* Whether a source anchors anywhere, so that `m` would move where it matches
|
|
516
|
+
* @param {string} source @param {boolean} [nested] - Whether the pattern carries `v`
|
|
517
|
+
*/
|
|
518
|
+
function hasAnchor(source, nested = false) {
|
|
519
|
+
let i = 0;
|
|
520
|
+
while (i < source.length) {
|
|
521
|
+
if (source[i] === '\\') i += tokenLength(source, i);
|
|
522
|
+
// Inside a class `^` negates and `$` is a member, so neither anchors there
|
|
523
|
+
else if (source[i] === '[') i = skipCharacterClass(source, i, nested);
|
|
524
|
+
else if (source[i] === '^' || source[i] === '$') return true;
|
|
525
|
+
else i++;
|
|
526
|
+
}
|
|
527
|
+
return false;
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
// A property escape or a code point escape, both of which read as literal text
|
|
531
|
+
// where the flag that gives them meaning is gone
|
|
532
|
+
const RE_UNICODE_ESCAPE = /\\[pP]\{|\\u\{/;
|
|
533
|
+
|
|
534
|
+
// The characters `u` folds by Unicode rules and a source without it does not:
|
|
535
|
+
// `/s/iu` matches `\u017F` and `/k/iu` matches `\u212A`, where neither does on its
|
|
536
|
+
// own. Characters past the BMP fold this way, too, and are caught as astral first.
|
|
537
|
+
const RE_UNICODE_FOLDING = /[\u004B\u0053\u006B\u0073\u00C5\u00DF\u00E5\u017F\u0398\u03A9\u03B8\u03C9\u03D1\u03F4\u1E9E\u1F80-\u1FAF\u1FB3\u1FBC\u1FC3\u1FCC\u1FF3\u1FFC\u2126\u212A\u212B]/;
|
|
538
|
+
|
|
539
|
+
/**
|
|
540
|
+
* Which flag changes what a source matches, so that embedding the source where the
|
|
541
|
+
* flag cannot follow would silently match something else. `u` and `v` only narrow
|
|
542
|
+
* what syntax is legal, so a source valid under either stays valid without it—the
|
|
543
|
+
* difference never surfaces as a syntax error, and this stands in for the one that
|
|
544
|
+
* would otherwise be raised.
|
|
545
|
+
* @param {RegExp} pattern
|
|
546
|
+
* @returns {'u' | 'v' | 'm' | null} The flag the source depends on, or `null` where
|
|
547
|
+
* every flag it carries would leave the source matching the same
|
|
548
|
+
*/
|
|
549
|
+
function lostFlag(pattern) {
|
|
550
|
+
const { source, unicode, unicodeSets } = pattern;
|
|
551
|
+
// `m` moves where `^` and `$` match, and only matters where the source has one
|
|
552
|
+
if (pattern.multiline && hasAnchor(source, unicodeSets)) return 'm';
|
|
553
|
+
if (!unicode && !unicodeSets) return null;
|
|
554
|
+
const flag = unicodeSets ? 'v' : 'u';
|
|
555
|
+
if (RE_UNICODE_ESCAPE.test(source)) return flag;
|
|
556
|
+
// Case folding under `i` follows Unicode rules only while the flag is there. A
|
|
557
|
+
// character can be written literally or as an escape, so the source is read
|
|
558
|
+
// token by token rather than scanned for the characters themselves; `singleCode`
|
|
559
|
+
// leaves out `\d` and friends, which fold alike with the flag and without it.
|
|
560
|
+
if (pattern.ignoreCase) {
|
|
561
|
+
for (let i = 0; i < source.length;) {
|
|
562
|
+
const length = tokenLength(source, i);
|
|
563
|
+
const code = singleCode(tokenRanges(source.slice(i, i + length), true));
|
|
564
|
+
if (code !== null && RE_UNICODE_FOLDING.test(String.fromCodePoint(code))) return flag;
|
|
565
|
+
i += length;
|
|
566
|
+
}
|
|
567
|
+
}
|
|
568
|
+
// Past the BMP a character is one code point with the flag, two units without,
|
|
569
|
+
// which is what a quantifier or a class beside it would go on to read wrongly
|
|
570
|
+
for (const char of source) {
|
|
571
|
+
if ((char.codePointAt(0) ?? 0) > 0xFFFF) return flag;
|
|
572
|
+
}
|
|
573
|
+
if (!unicodeSets) return null;
|
|
574
|
+
|
|
575
|
+
// `v` alone lets a class nest, subtract, intersect, and hold whole strings
|
|
576
|
+
let i = 0;
|
|
577
|
+
while (i < source.length) {
|
|
578
|
+
if (source[i] === '\\') {
|
|
579
|
+
i += tokenLength(source, i);
|
|
580
|
+
continue;
|
|
581
|
+
}
|
|
582
|
+
if (source[i] !== '[') {
|
|
583
|
+
i++;
|
|
584
|
+
continue;
|
|
585
|
+
}
|
|
586
|
+
const end = skipCharacterClass(source, i, true);
|
|
587
|
+
for (let j = i + 1; j < end - 1;) {
|
|
588
|
+
if (source[j] === '\\') {
|
|
589
|
+
if (source.startsWith('\\q{', j)) return 'v';
|
|
590
|
+
j += tokenLength(source, j);
|
|
591
|
+
continue;
|
|
592
|
+
}
|
|
593
|
+
if (source[j] === '[' || source.startsWith('--', j) || source.startsWith('&&', j)) return 'v';
|
|
594
|
+
j++;
|
|
595
|
+
}
|
|
596
|
+
i = end;
|
|
597
|
+
}
|
|
598
|
+
return null;
|
|
599
|
+
}
|
|
600
|
+
|
|
601
|
+
/**
|
|
602
|
+
* A pattern’s source, rewritten to match the way its own flags make it match, for
|
|
603
|
+
* embedding in a larger regex that cannot carry them. `i` and `s` fit into a
|
|
604
|
+
* source; `m`, `u`, and `v` do not, and are left to the pattern it joins. Where
|
|
605
|
+
* `i` cannot be written in—a backreference, a range outside ASCII, a range that
|
|
606
|
+
* spans letters without being one—the source stands as it is, matching less than
|
|
607
|
+
* the pattern would rather than more.
|
|
608
|
+
* @param {RegExp} pattern
|
|
609
|
+
* @returns {string}
|
|
610
|
+
*/
|
|
611
|
+
function embedSource(pattern) {
|
|
612
|
+
let source = pattern.source;
|
|
613
|
+
if (pattern.dotAll) source = expandDotAll(source, pattern.unicodeSets);
|
|
614
|
+
if (pattern.ignoreCase) source = foldCase(source, pattern.unicodeSets);
|
|
615
|
+
return source;
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
/**
|
|
619
|
+
* @typedef {{text: string, ranges: [number, number][] | null}} Repeat - An unbounded
|
|
620
|
+
* repeat, as it reads and as the set of characters it matches
|
|
621
|
+
*/
|
|
622
|
+
|
|
623
|
+
/**
|
|
624
|
+
* @param {Repeat[]} reachable - Repeats that could still sit adjacent to what comes next
|
|
625
|
+
* @param {Repeat} repeat
|
|
626
|
+
* @returns {boolean} Whether the two can consume the same character, and so split
|
|
627
|
+
* the same run of input between them
|
|
628
|
+
*/
|
|
629
|
+
function splitsWith(reachable, repeat) {
|
|
630
|
+
return reachable.some(earlier => earlier.text === repeat.text ||
|
|
631
|
+
(!!repeat.ranges && !!earlier.ranges && rangesIntersect(repeat.ranges, earlier.ranges)));
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
/**
|
|
635
|
+
* @param {Repeat[]} into
|
|
636
|
+
* @param {Repeat[]} repeats
|
|
637
|
+
* @returns {void} Adds the repeats, keeping only the most recent ones the walk
|
|
638
|
+
* still compares against
|
|
639
|
+
*/
|
|
640
|
+
function collectReachable(into, repeats) {
|
|
641
|
+
for (const repeat of repeats) {
|
|
642
|
+
into.push(repeat);
|
|
643
|
+
if (into.length > MAX_REACHABLE) into.shift();
|
|
644
|
+
}
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
/**
|
|
648
|
+
* Walk a regex source for the shapes whose backtracking blows up: an unlimited
|
|
649
|
+
* quantifier over a group that itself contains a variable quantifier (`(a+)+`,
|
|
650
|
+
* `(a?)+`) or alternates, and two unbounded repeats that can consume the same
|
|
651
|
+
* character with only atoms matching empty between them (`.*.*`, `[a]*a*`,
|
|
652
|
+
* `\w*\d*`, `a*b*a*`, `a*(b*)a*`, `(a*)a*`). A lone unlimited quantifier stays
|
|
653
|
+
* linear, so `[\s\S]*?` up to a literal terminator passes.
|
|
654
|
+
* @param {string} source - Regex source, assumed syntactically valid
|
|
655
|
+
* @param {number} [depth] - Group nesting level of this call
|
|
656
|
+
* @param {boolean} [nested] - Whether the pattern carries `v`
|
|
657
|
+
* @returns {{risky: boolean, varies: boolean, alternates: boolean, deep: boolean,
|
|
658
|
+
* empty: boolean, leading: Repeat[], trailing: Repeat[]}} What the walk found,
|
|
659
|
+
* with the repeats that reach the source’s start and end for a caller to compare
|
|
660
|
+
* against what stands beside it
|
|
661
|
+
*/
|
|
662
|
+
function analyzeQuantifiers(source, depth = 0, nested = false) {
|
|
663
|
+
if (depth > MAX_PATTERN_DEPTH) return { risky: true, varies: true, alternates: true, deep: true, empty: true, leading: [], trailing: [] };
|
|
664
|
+
|
|
665
|
+
let risky = false;
|
|
666
|
+
let varies = false;
|
|
667
|
+
let alternates = false;
|
|
668
|
+
let deep = false;
|
|
669
|
+
// Whether every atom of the branch being read matches empty, and whether any
|
|
670
|
+
// branch before it did
|
|
671
|
+
let branchEmpty = true;
|
|
672
|
+
let anyBranchEmpty = false;
|
|
673
|
+
// The unbounded repeats that could still sit adjacent to what comes next, most
|
|
674
|
+
// recent last; an atom matching empty leaves the ones before it reachable
|
|
675
|
+
/** @type {Repeat[]} */
|
|
676
|
+
const reachable = [];
|
|
677
|
+
// The repeats a caller can reach from either end, gathered across branches
|
|
678
|
+
/** @type {Repeat[]} */
|
|
679
|
+
const leading = [];
|
|
680
|
+
/** @type {Repeat[]} */
|
|
681
|
+
const trailing = [];
|
|
682
|
+
/** @type {Repeat[]} */
|
|
683
|
+
let branchLeading = [];
|
|
684
|
+
let i = 0;
|
|
685
|
+
|
|
686
|
+
while (i < source.length) {
|
|
687
|
+
const start = i;
|
|
688
|
+
const char = source[i];
|
|
689
|
+
/** @type {ReturnType<typeof analyzeQuantifiers> | null} */
|
|
690
|
+
let group = null;
|
|
691
|
+
|
|
692
|
+
if (char === '|') {
|
|
693
|
+
// Alternatives are separate expressions, so nothing carries across
|
|
694
|
+
alternates = true;
|
|
695
|
+
anyBranchEmpty ||= branchEmpty;
|
|
696
|
+
branchEmpty = true;
|
|
697
|
+
collectReachable(leading, branchLeading);
|
|
698
|
+
collectReachable(trailing, reachable);
|
|
699
|
+
branchLeading = [];
|
|
700
|
+
reachable.length = 0;
|
|
701
|
+
i++;
|
|
702
|
+
continue;
|
|
703
|
+
} else if (char === '\\') {
|
|
704
|
+
i += tokenLength(source, i);
|
|
705
|
+
} else if (char === '[') {
|
|
706
|
+
i = skipCharacterClass(source, i, nested);
|
|
707
|
+
} else if (char === '(') {
|
|
708
|
+
let open = 1;
|
|
709
|
+
i++;
|
|
710
|
+
while (i < source.length && open > 0) {
|
|
711
|
+
const inner = source[i];
|
|
712
|
+
if (inner === '\\') {
|
|
713
|
+
i += tokenLength(source, i);
|
|
714
|
+
} else if (inner === '[') {
|
|
715
|
+
i = skipCharacterClass(source, i, nested);
|
|
716
|
+
} else {
|
|
717
|
+
if (inner === '(') open++;
|
|
718
|
+
else if (inner === ')') open--;
|
|
719
|
+
i++;
|
|
720
|
+
}
|
|
721
|
+
}
|
|
722
|
+
// Drop the group prefix (`?:`, `?=`, `?<name>`, …) before reading the body
|
|
723
|
+
group = analyzeQuantifiers(source.slice(start + 1, i - 1).replace(/^\?(?:[:=!]|<[=!]|<[^>]*>)/, ''), depth + 1, nested);
|
|
724
|
+
} else {
|
|
725
|
+
i++;
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
const atom = source.slice(start, i);
|
|
729
|
+
RE_QUANTIFIER.lastIndex = i;
|
|
730
|
+
const quantifier = RE_QUANTIFIER.exec(source);
|
|
731
|
+
const upper = quantifier?.[2];
|
|
732
|
+
const repeats = !!quantifier && (quantifier[0][0] === '*' || quantifier[0][0] === '+' || upper === '');
|
|
733
|
+
// A count that can vary—anything but `{n}` and its `{n,n}` spelling—makes
|
|
734
|
+
// the group ambiguous about how much it consumes, which multiplies under an
|
|
735
|
+
// unlimited repeat
|
|
736
|
+
const exact = !!quantifier && quantifier[0][0] === '{' &&
|
|
737
|
+
(upper === undefined || (upper !== '' && Number(upper) === Number(quantifier[1])));
|
|
738
|
+
const variable = !!quantifier && !exact;
|
|
739
|
+
// An atom matches empty where it is zero-width, where its quantifier may
|
|
740
|
+
// repeat none, or where a group body matches empty however often it repeats—
|
|
741
|
+
// `(b*)`, carrying the quantifier inside the group rather than on it
|
|
742
|
+
const least = quantifier && (quantifier[0][0] === '+' ? 1
|
|
743
|
+
: quantifier[0][0] === '{' ? Number(quantifier[1]) : 0);
|
|
744
|
+
const body = RE_ZERO_WIDTH.test(atom) || (!!group && group.empty);
|
|
745
|
+
const nullable = body || (!!quantifier && least === 0);
|
|
746
|
+
if (quantifier) i = RE_QUANTIFIER.lastIndex;
|
|
747
|
+
|
|
748
|
+
// A lookaround is atomic: It backtracks nothing, so what it holds neither
|
|
749
|
+
// reaches out of it nor is reached into
|
|
750
|
+
const opaque = RE_ZERO_WIDTH.test(atom);
|
|
751
|
+
if (group) {
|
|
752
|
+
if (group.risky || (repeats && (group.varies || group.alternates))) risky = true;
|
|
753
|
+
if (group.varies) varies = true;
|
|
754
|
+
// A group alternates whether the `|` sits at its top level or deeper
|
|
755
|
+
if (group.alternates) alternates = true;
|
|
756
|
+
if (group.deep) deep = true;
|
|
757
|
+
// A group is no wall: What it repeats at its start splits the same run as
|
|
758
|
+
// what stands unbounded before it
|
|
759
|
+
if (!opaque && group.leading.some(repeat => splitsWith(reachable, repeat))) risky = true;
|
|
760
|
+
}
|
|
761
|
+
if (variable) varies = true;
|
|
762
|
+
const text = unwrapAtom(atom);
|
|
763
|
+
const ranges = atomRanges(text, nested);
|
|
764
|
+
if (repeats && splitsWith(reachable, { text, ranges })) risky = true;
|
|
765
|
+
|
|
766
|
+
// Whatever a caller could reach before this atom, it reaches this one, too
|
|
767
|
+
if (branchEmpty) {
|
|
768
|
+
if (repeats) collectReachable(branchLeading, [{ text, ranges }]);
|
|
769
|
+
if (group && !opaque) collectReachable(branchLeading, group.leading);
|
|
770
|
+
}
|
|
771
|
+
// Nothing reaches past an atom that has to consume something
|
|
772
|
+
if (!nullable) {
|
|
773
|
+
reachable.length = 0;
|
|
774
|
+
branchEmpty = false;
|
|
775
|
+
}
|
|
776
|
+
if (group && !opaque) collectReachable(reachable, group.trailing);
|
|
777
|
+
if (repeats) collectReachable(reachable, [{ text, ranges }]);
|
|
778
|
+
}
|
|
779
|
+
|
|
780
|
+
collectReachable(leading, branchLeading);
|
|
781
|
+
collectReachable(trailing, reachable);
|
|
782
|
+
|
|
783
|
+
return { risky, varies, alternates, deep, empty: anyBranchEmpty || branchEmpty, leading, trailing };
|
|
784
|
+
}
|
|
785
|
+
|
|
786
|
+
/**
|
|
787
|
+
* @param {RegExp | string} pattern - Pattern to judge, or a bare source to read
|
|
788
|
+
* as it stands
|
|
789
|
+
* @returns {string | null} What makes the pattern a backtracking risk, phrased
|
|
790
|
+
* to follow the pattern itself, or `null` where it is none
|
|
791
|
+
*/
|
|
792
|
+
function describeQuantifierRisk(pattern) {
|
|
793
|
+
const source = typeof pattern === 'string' ? pattern : pattern.source;
|
|
794
|
+
// A pattern too big to read is refused for that, not for a shape nobody saw;
|
|
795
|
+
// its own length is what counts, not the length folding case inflates it to
|
|
796
|
+
if (source.length > MAX_PATTERN_LENGTH) {
|
|
797
|
+
return `runs past ${MAX_PATTERN_LENGTH.toLocaleString()} characters, too long to analyze for catastrophic backtracking—shorten it, or split it into several patterns`;
|
|
798
|
+
}
|
|
799
|
+
// `i` and `s` change what the source matches, so they are written into it
|
|
800
|
+
// before the shapes are read
|
|
801
|
+
const analysis = typeof pattern === 'string'
|
|
802
|
+
? analyzeQuantifiers(source)
|
|
803
|
+
: analyzeQuantifiers(embedSource(pattern), 0, pattern.unicodeSets);
|
|
804
|
+
if (analysis.deep) {
|
|
805
|
+
return `nests groups more than ${MAX_PATTERN_DEPTH.toLocaleString()} deep, too deep to analyze for catastrophic backtracking—flatten it, or split it into several patterns`;
|
|
806
|
+
}
|
|
807
|
+
return analysis.risky
|
|
808
|
+
? 'compounds quantifiers or alternation in a way that may cause ReDoS—bound the repetition (e.g., `{0,1000}`) instead'
|
|
809
|
+
: null;
|
|
810
|
+
}
|
|
148
811
|
|
|
149
812
|
/**
|
|
150
813
|
* Find the index of the `>` that closes an opening tag, correctly skipping
|
|
@@ -168,6 +831,8 @@ function findTagEnd(html, pos) {
|
|
|
168
831
|
return -1;
|
|
169
832
|
}
|
|
170
833
|
|
|
834
|
+
// Exports
|
|
835
|
+
|
|
171
836
|
export {
|
|
172
837
|
stableStringify,
|
|
173
838
|
findTagEnd,
|
|
@@ -179,5 +844,8 @@ export {
|
|
|
179
844
|
isThenable,
|
|
180
845
|
lowercase,
|
|
181
846
|
replaceAsync,
|
|
182
|
-
parseRegExp
|
|
847
|
+
parseRegExp,
|
|
848
|
+
embedSource,
|
|
849
|
+
lostFlag,
|
|
850
|
+
describeQuantifierRisk
|
|
183
851
|
};
|