html-minifier-next 8.3.0 → 8.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +12 -6
- package/cli.js +137 -31
- package/dist/types/htmlminifier.d.ts +3 -3
- package/dist/types/htmlminifier.d.ts.map +1 -1
- package/dist/types/htmlparser.d.ts.map +1 -1
- package/dist/types/lib/attributes.d.ts +3 -0
- package/dist/types/lib/attributes.d.ts.map +1 -1
- package/dist/types/lib/constants.d.ts +3 -0
- package/dist/types/lib/constants.d.ts.map +1 -1
- package/dist/types/lib/content.d.ts +15 -1
- package/dist/types/lib/content.d.ts.map +1 -1
- package/dist/types/lib/elements.d.ts +3 -0
- package/dist/types/lib/elements.d.ts.map +1 -1
- package/dist/types/lib/option-definitions.d.ts +3 -0
- package/dist/types/lib/option-definitions.d.ts.map +1 -1
- package/dist/types/lib/options.d.ts +5 -0
- package/dist/types/lib/options.d.ts.map +1 -1
- package/dist/types/lib/unused-css.d.ts +3 -0
- package/dist/types/lib/unused-css.d.ts.map +1 -1
- package/dist/types/lib/utils.d.ts +3 -0
- package/dist/types/lib/utils.d.ts.map +1 -1
- package/dist/types/lib/whitespace.d.ts +3 -0
- package/dist/types/lib/whitespace.d.ts.map +1 -1
- package/package.json +7 -7
- package/src/htmlminifier.js +65 -11
- package/src/htmlparser.js +184 -137
- package/src/lib/attributes.js +4 -0
- package/src/lib/constants.js +3 -1
- package/src/lib/content.js +77 -0
- package/src/lib/elements.js +4 -0
- package/src/lib/file-pool.js +177 -0
- package/src/lib/option-definitions.js +3 -1
- package/src/lib/options.js +12 -0
- package/src/lib/unused-css.js +3 -1
- package/src/lib/utils.js +3 -1
- package/src/lib/whitespace.js +4 -0
package/src/htmlminifier.js
CHANGED
|
@@ -58,6 +58,7 @@ import {
|
|
|
58
58
|
|
|
59
59
|
import {
|
|
60
60
|
hasJsonScriptType,
|
|
61
|
+
extractScriptBodies,
|
|
61
62
|
processScript
|
|
62
63
|
} from './lib/content.js';
|
|
63
64
|
|
|
@@ -103,7 +104,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
|
|
|
103
104
|
* improve performance for inputs with repeated CSS (e.g., batch processing).
|
|
104
105
|
* - Cache is created on first `minify()` call and persists for the process lifetime
|
|
105
106
|
* - Cache size is locked after first call—subsequent calls reuse the same cache
|
|
106
|
-
* -
|
|
107
|
+
* - `0` switches the cache off; negative and non-finite values fall back to the default size
|
|
107
108
|
*
|
|
108
109
|
* Default: `500`
|
|
109
110
|
*
|
|
@@ -112,7 +113,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
|
|
|
112
113
|
* values improve performance for inputs with repeated JavaScript.
|
|
113
114
|
* - Cache is created on first `minify()` call and persists for the process lifetime
|
|
114
115
|
* - Cache size is locked after first call—subsequent calls reuse the same cache
|
|
115
|
-
* -
|
|
116
|
+
* - `0` switches the cache off; negative and non-finite values fall back to the default size
|
|
116
117
|
*
|
|
117
118
|
* Default: `500`
|
|
118
119
|
*
|
|
@@ -121,7 +122,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
|
|
|
121
122
|
* values improve performance for inputs with repeated SVG content.
|
|
122
123
|
* - Cache is created on first `minify()` call and persists for the process lifetime
|
|
123
124
|
* - Cache size is locked after first call—subsequent calls reuse the same cache
|
|
124
|
-
* -
|
|
125
|
+
* - `0` switches the cache off; negative and non-finite values fall back to the default size
|
|
125
126
|
*
|
|
126
127
|
* Default: `500`
|
|
127
128
|
*
|
|
@@ -956,6 +957,27 @@ async function createSortFns(value, options, uidIgnore, uidAttr, ignoredMarkupCh
|
|
|
956
957
|
}
|
|
957
958
|
}
|
|
958
959
|
|
|
960
|
+
// Match `/^<\/([\w:-]+)>$/` by hand—the regex engine’s per-call setup costs more
|
|
961
|
+
// than the scan on these short buffer entries
|
|
962
|
+
/**
|
|
963
|
+
* @param {string} str
|
|
964
|
+
* @returns {string | null} The tag name, or `null` where the string is not a plain end tag
|
|
965
|
+
*/
|
|
966
|
+
function matchPlainEndTag(str) {
|
|
967
|
+
const end = str.length;
|
|
968
|
+
if (end < 4 || str.charCodeAt(0) !== 60 /* < */ || str.charCodeAt(1) !== 47 /* / */ || str.charCodeAt(end - 1) !== 62 /* > */) {
|
|
969
|
+
return null;
|
|
970
|
+
}
|
|
971
|
+
for (let i = 2; i < end - 1; i++) {
|
|
972
|
+
const code = str.charCodeAt(i);
|
|
973
|
+
// `[\w:-]`: Letters, digits, `_`, `:`, `-`
|
|
974
|
+
if (!((code >= 48 && code <= 57) || (code >= 65 && code <= 90) || (code >= 97 && code <= 122) || code === 95 || code === 58 || code === 45)) {
|
|
975
|
+
return null;
|
|
976
|
+
}
|
|
977
|
+
}
|
|
978
|
+
return str.slice(2, end - 1);
|
|
979
|
+
}
|
|
980
|
+
|
|
959
981
|
/**
|
|
960
982
|
* @param {string} value - HTML content to minify
|
|
961
983
|
* @param {ProcessedOptions} options - Normalized minification options
|
|
@@ -1198,10 +1220,11 @@ async function minifyHTML(value, options, partialMarkup) {
|
|
|
1198
1220
|
function trimTrailingWhitespace(/** @type {number} */ index, /** @type {string} */ nextTag) {
|
|
1199
1221
|
for (let prevTag = ''; index >= 0 && canTrimWhitespace(prevTag, emptyAttrs); index--) {
|
|
1200
1222
|
const str = buffer[index] ?? '';
|
|
1201
|
-
const
|
|
1202
|
-
if (
|
|
1203
|
-
prevTag =
|
|
1204
|
-
} else if (
|
|
1223
|
+
const endTagName = matchPlainEndTag(str);
|
|
1224
|
+
if (endTagName !== null) {
|
|
1225
|
+
prevTag = endTagName;
|
|
1226
|
+
} else if (str.charCodeAt(str.length - 1) === 62 /* > */ ||
|
|
1227
|
+
(buffer[index] = collapseWhitespaceSmart(str, '', nextTag, emptyAttrs, emptyAttrs, options, inlineElements, inlineTextSet))) {
|
|
1205
1228
|
break;
|
|
1206
1229
|
}
|
|
1207
1230
|
}
|
|
@@ -1504,6 +1527,35 @@ async function minifyHTML(value, options, partialMarkup) {
|
|
|
1504
1527
|
buffer.push(comment);
|
|
1505
1528
|
}
|
|
1506
1529
|
|
|
1530
|
+
// Inline scripts are minified one at a time as the parse reaches them, which serializes
|
|
1531
|
+
// work the minifier could overlap. Dispatching them all up front turns that into a single
|
|
1532
|
+
// wave the parse then reads results from. Bodies are matched by exact content, so one the
|
|
1533
|
+
// parse asks for in another form simply misses and takes the original path.
|
|
1534
|
+
/** @type {Map<string, Promise<string>> | null} */
|
|
1535
|
+
let jsPrewarmed = null;
|
|
1536
|
+
if (options.parallelJS && options.minifyJS !== identity && !options.processScripts?.length) {
|
|
1537
|
+
const bodies = extractScriptBodies(value);
|
|
1538
|
+
if (bodies.length > 1) {
|
|
1539
|
+
jsPrewarmed = new Map();
|
|
1540
|
+
// `collapseWhitespace` reaches script bodies through the whole-document pass above,
|
|
1541
|
+
// so the parse sees them trimmed; other whitespace modes leave them as they are
|
|
1542
|
+
const trims = Boolean(options.collapseWhitespace) &&
|
|
1543
|
+
!options.conservativeCollapse && !options.preserveLineBreaks;
|
|
1544
|
+
for (const { code, isModule } of bodies) {
|
|
1545
|
+
const body = trims ? code.trim() : code;
|
|
1546
|
+
const key = (isModule ? 'm|' : '|') + body;
|
|
1547
|
+
if (jsPrewarmed.has(key)) {
|
|
1548
|
+
continue;
|
|
1549
|
+
}
|
|
1550
|
+
const pending = Promise.resolve(options.minifyJS(body, false, isModule));
|
|
1551
|
+
// The awaiting consumer reports the failure; a body the parse never asks for
|
|
1552
|
+
// would not have been minified at all, so its failure is not ours to raise
|
|
1553
|
+
pending.catch(() => {});
|
|
1554
|
+
jsPrewarmed.set(key, pending);
|
|
1555
|
+
}
|
|
1556
|
+
}
|
|
1557
|
+
}
|
|
1558
|
+
|
|
1507
1559
|
// SVG subtree capture: When SVGO is active, record buffer positions for post-processing
|
|
1508
1560
|
/** @type {Array<{start: number, end: number}>} */
|
|
1509
1561
|
const svgBlocks = []; // Array of { start, end } buffer indices
|
|
@@ -1854,7 +1906,8 @@ async function minifyHTML(value, options, partialMarkup) {
|
|
|
1854
1906
|
text = await processScript(text, options, currentAttrs, minifyHTML);
|
|
1855
1907
|
}
|
|
1856
1908
|
if (needsMinifyJS) {
|
|
1857
|
-
|
|
1909
|
+
const prewarmed = jsPrewarmed?.get((isModuleScript ? 'm|' : '|') + text);
|
|
1910
|
+
text = prewarmed ? await prewarmed : await options.minifyJS(text, false, isModuleScript);
|
|
1858
1911
|
}
|
|
1859
1912
|
if (needsMinifyCSS) {
|
|
1860
1913
|
text = await options.minifyCSS(text, undefined, options.cssContext);
|
|
@@ -2001,7 +2054,7 @@ function joinResultSegments(results, options, restoreCustom, restoreIgnore) {
|
|
|
2001
2054
|
* - Cache sizes are locked after first initialization—subsequent calls use the same caches
|
|
2002
2055
|
* even if different `cacheCSS`/`cacheJS`/`cacheSVG` options are provided
|
|
2003
2056
|
* - The first call’s options determine the cache sizes for subsequent calls
|
|
2004
|
-
* - Invalid values (NaN, Infinity) fall back to the default size (500)
|
|
2057
|
+
* - Invalid values (NaN, Infinity, negative) fall back to the default size (500)
|
|
2005
2058
|
*/
|
|
2006
2059
|
/** @param {MinifierOptions} options */
|
|
2007
2060
|
function initCaches(options) {
|
|
@@ -2019,8 +2072,9 @@ function initCaches(options) {
|
|
|
2019
2072
|
return parsed;
|
|
2020
2073
|
};
|
|
2021
2074
|
|
|
2022
|
-
//
|
|
2023
|
-
|
|
2075
|
+
// Non-finite and negative sizes fall back to `defaultSize`, as they do for the
|
|
2076
|
+
// environment variables above
|
|
2077
|
+
const sanitizeSize = (/** @type {number} */ size) => Number.isFinite(size) && size >= 0 ? Math.floor(size) : defaultSize;
|
|
2024
2078
|
|
|
2025
2079
|
// Get cache sizes with precedence: Options > env > default
|
|
2026
2080
|
const cssSize = options.cacheCSS !== undefined ? options.cacheCSS
|
package/src/htmlparser.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
/*
|
|
2
2
|
* HTML Parser By John Resig (ejohn.org)
|
|
3
|
-
* Modified by Juriy “kangax” Zaytsev
|
|
3
|
+
* Modified by Juriy “kangax” Zaytsev and Jens Oliver Meiert
|
|
4
4
|
* Original code by Erik Arvidsson, Mozilla Public License
|
|
5
5
|
* http://erik.eae.net/simplehtmlparser/simplehtmlparser.js
|
|
6
6
|
*/
|
|
@@ -103,6 +103,31 @@ const nonPhrasing = new Set(['address', 'article', 'aside', 'base', 'blockquote'
|
|
|
103
103
|
const endsTagName = (/** @type {string} */ character) =>
|
|
104
104
|
character === '' || character === '/' || character === '>' || /\s/.test(character);
|
|
105
105
|
|
|
106
|
+
// Fast path for tag names
|
|
107
|
+
/**
|
|
108
|
+
* @param {string} html
|
|
109
|
+
* @param {number} start - Position of the first name character (after `<` or `</`)
|
|
110
|
+
* @returns {number} Name length from `start`, or -1 to fall back to the qname regex
|
|
111
|
+
*/
|
|
112
|
+
function scanAsciiTagName(html, start) {
|
|
113
|
+
const first = html.charCodeAt(start);
|
|
114
|
+
// First char: Letter or `_` (as for ncname)
|
|
115
|
+
if (!((first >= 65 && first <= 90) || (first >= 97 && first <= 122) || first === 95)) {
|
|
116
|
+
return -1;
|
|
117
|
+
}
|
|
118
|
+
let i = start + 1;
|
|
119
|
+
for (;;) {
|
|
120
|
+
const code = html.charCodeAt(i);
|
|
121
|
+
if ((code >= 65 && code <= 90) || (code >= 97 && code <= 122) || (code >= 48 && code <= 57) || code === 95 || code === 45 || code === 46) {
|
|
122
|
+
i++;
|
|
123
|
+
} else if (code === 58 || code >= 128) { // `:` or non-ASCII
|
|
124
|
+
return -1;
|
|
125
|
+
} else {
|
|
126
|
+
return i - start;
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
}
|
|
130
|
+
|
|
106
131
|
// The characters `\s` matches, by code: Tab through carriage return, space, and the
|
|
107
132
|
// Unicode whitespace and line terminators
|
|
108
133
|
const isWhitespaceCode = (/** @type {number} */ code) =>
|
|
@@ -374,8 +399,7 @@ export class HTMLParser {
|
|
|
374
399
|
cachedNextStartTag = null;
|
|
375
400
|
cachedNextEndTag = null;
|
|
376
401
|
advance(startTagMatch.advance);
|
|
377
|
-
await handleStartTag(startTagMatch);
|
|
378
|
-
prevTag = startTagMatch.tagName.toLowerCase();
|
|
402
|
+
prevTag = await handleStartTag(startTagMatch);
|
|
379
403
|
continue;
|
|
380
404
|
}
|
|
381
405
|
if (cachedNextEndTag && cachedNextEndTag.pos === pos) {
|
|
@@ -383,8 +407,7 @@ export class HTMLParser {
|
|
|
383
407
|
cachedNextStartTag = null;
|
|
384
408
|
cachedNextEndTag = null;
|
|
385
409
|
advance(endTagMatch.text.length);
|
|
386
|
-
await parseEndTag(endTagMatch.text, endTagMatch.name);
|
|
387
|
-
prevTag = '/' + endTagMatch.name.toLowerCase();
|
|
410
|
+
prevTag = '/' + await parseEndTag(endTagMatch.text, endTagMatch.name);
|
|
388
411
|
prevAttrs = [];
|
|
389
412
|
continue;
|
|
390
413
|
}
|
|
@@ -452,8 +475,7 @@ export class HTMLParser {
|
|
|
452
475
|
const endTagMatch = matchEndTag(pos);
|
|
453
476
|
if (endTagMatch) {
|
|
454
477
|
advance(endTagMatch.text.length);
|
|
455
|
-
await parseEndTag(endTagMatch.text, endTagMatch.name);
|
|
456
|
-
prevTag = '/' + endTagMatch.name.toLowerCase();
|
|
478
|
+
prevTag = '/' + await parseEndTag(endTagMatch.text, endTagMatch.name);
|
|
457
479
|
prevAttrs = [];
|
|
458
480
|
continue;
|
|
459
481
|
}
|
|
@@ -463,8 +485,7 @@ export class HTMLParser {
|
|
|
463
485
|
const startTagMatch = parseStartTag(pos);
|
|
464
486
|
if (startTagMatch) {
|
|
465
487
|
advance(startTagMatch.advance);
|
|
466
|
-
await handleStartTag(startTagMatch);
|
|
467
|
-
prevTag = startTagMatch.tagName.toLowerCase();
|
|
488
|
+
prevTag = await handleStartTag(startTagMatch);
|
|
468
489
|
continue;
|
|
469
490
|
}
|
|
470
491
|
}
|
|
@@ -614,17 +635,29 @@ export class HTMLParser {
|
|
|
614
635
|
* where none is complete
|
|
615
636
|
*/
|
|
616
637
|
function matchEndTag(startPos) {
|
|
617
|
-
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
638
|
+
let name;
|
|
639
|
+
let nameEnd;
|
|
640
|
+
const scanned = scanAsciiTagName(fullHtml, startPos + 2);
|
|
641
|
+
if (scanned !== -1) {
|
|
642
|
+
nameEnd = startPos + 2 + scanned;
|
|
643
|
+
name = fullHtml.slice(startPos + 2, nameEnd);
|
|
644
|
+
} else {
|
|
645
|
+
endTagOpenY.lastIndex = startPos;
|
|
646
|
+
const open = endTagOpenY.exec(fullHtml);
|
|
647
|
+
if (!open) {
|
|
648
|
+
return null;
|
|
649
|
+
}
|
|
650
|
+
nameEnd = startPos + open[0].length;
|
|
651
|
+
name = open[1] ?? '';
|
|
621
652
|
}
|
|
622
|
-
const close = findEndTagClose(fullHtml,
|
|
623
|
-
return close === -1 ? null : { text: fullHtml.slice(startPos, close + 1), name
|
|
653
|
+
const close = findEndTagClose(fullHtml, nameEnd);
|
|
654
|
+
return close === -1 ? null : { text: fullHtml.slice(startPos, close + 1), name };
|
|
624
655
|
}
|
|
625
656
|
|
|
626
657
|
// Where a tag ends: After optional whitespace, at `>` or `/>`. Scanned by hand in a
|
|
627
|
-
// single pass, where a regex did the same with one execution per attribute.
|
|
658
|
+
// single pass, where a regex did the same with one execution per attribute. The
|
|
659
|
+
// result packs length and slash into one number (`length << 1 | hasSlash`, 0 for no
|
|
660
|
+
// match) so the per-attribute loop allocates nothing.
|
|
628
661
|
function matchTagClose(/** @type {number} */ currentPos) {
|
|
629
662
|
let scan = currentPos;
|
|
630
663
|
let code = fullHtml.charCodeAt(scan);
|
|
@@ -632,132 +665,89 @@ export class HTMLParser {
|
|
|
632
665
|
code = fullHtml.charCodeAt(++scan);
|
|
633
666
|
}
|
|
634
667
|
if (code === 62) { // `>`
|
|
635
|
-
return
|
|
668
|
+
return (scan - currentPos + 1) << 1;
|
|
636
669
|
}
|
|
637
670
|
if (code === 47 && fullHtml.charCodeAt(scan + 1) === 62) { // `/>`
|
|
638
|
-
return
|
|
671
|
+
return ((scan - currentPos + 2) << 1) | 1;
|
|
639
672
|
}
|
|
640
|
-
return
|
|
673
|
+
return 0;
|
|
641
674
|
}
|
|
642
675
|
|
|
643
676
|
function parseStartTag(/** @type {number} */ startPos) {
|
|
644
|
-
|
|
645
|
-
|
|
646
|
-
|
|
647
|
-
|
|
648
|
-
|
|
649
|
-
|
|
650
|
-
|
|
651
|
-
|
|
652
|
-
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
|
|
663
|
-
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
|
|
667
|
-
|
|
668
|
-
|
|
669
|
-
|
|
670
|
-
|
|
677
|
+
let tagName;
|
|
678
|
+
let consumed;
|
|
679
|
+
const scanned = scanAsciiTagName(fullHtml, startPos + 1);
|
|
680
|
+
if (scanned !== -1) {
|
|
681
|
+
tagName = fullHtml.slice(startPos + 1, startPos + 1 + scanned);
|
|
682
|
+
consumed = 1 + scanned;
|
|
683
|
+
} else {
|
|
684
|
+
startTagOpenY.lastIndex = startPos;
|
|
685
|
+
const start = startTagOpenY.exec(fullHtml);
|
|
686
|
+
if (!start) {
|
|
687
|
+
return undefined;
|
|
688
|
+
}
|
|
689
|
+
tagName = start[1] ?? '';
|
|
690
|
+
consumed = start[0].length;
|
|
691
|
+
}
|
|
692
|
+
/** @type {{tagName: string, attrs: Array<Array<string|undefined>>, advance: number, unarySlash?: string}} */
|
|
693
|
+
const match = {
|
|
694
|
+
tagName,
|
|
695
|
+
attrs: [],
|
|
696
|
+
advance: 0
|
|
697
|
+
};
|
|
698
|
+
let currentPos = startPos + consumed;
|
|
699
|
+
let close, attr;
|
|
700
|
+
|
|
701
|
+
// Safety limit: Max length of input to check for attributes
|
|
702
|
+
// Protects against catastrophic backtracking on massive attribute values
|
|
703
|
+
const MAX_ATTR_PARSE_LENGTH = 20000; // 20 KB should be enough for any reasonable tag
|
|
704
|
+
|
|
705
|
+
while (true) {
|
|
706
|
+
// Check for closing tag first (manual scan, regex only for the unusual)
|
|
707
|
+
close = matchTagClose(currentPos);
|
|
708
|
+
if (close) {
|
|
709
|
+
break;
|
|
710
|
+
}
|
|
671
711
|
|
|
672
|
-
|
|
673
|
-
|
|
674
|
-
|
|
675
|
-
attr = attributeY.exec(fullHtml);
|
|
676
|
-
} else {
|
|
677
|
-
const extractEndPos = currentPos + MAX_ATTR_PARSE_LENGTH;
|
|
678
|
-
|
|
679
|
-
// Create a temporary substring only for attribute parsing (limited for safety)
|
|
680
|
-
const searchStr = fullHtml.substring(currentPos, extractEndPos);
|
|
681
|
-
attr = searchStr.match(attribute);
|
|
682
|
-
|
|
683
|
-
// If input was limited and there’s a match, check if the value might be truncated
|
|
684
|
-
if (attr) {
|
|
685
|
-
// Check if the attribute value extends beyond our search window
|
|
686
|
-
const attrEnd = attr[0].length;
|
|
687
|
-
// If the match ends near the limit, the value might be truncated
|
|
688
|
-
if (attrEnd > MAX_ATTR_PARSE_LENGTH - 100) {
|
|
689
|
-
// Manually extract this attribute to handle potentially huge value
|
|
690
|
-
const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
|
|
691
|
-
if (manualMatch) {
|
|
692
|
-
const quoteChar = searchStr[manualMatch[0].length];
|
|
693
|
-
if (quoteChar === '"' || quoteChar === "'") {
|
|
694
|
-
const closeQuote = searchStr.indexOf(quoteChar, manualMatch[0].length + 1);
|
|
695
|
-
if (closeQuote !== -1) {
|
|
696
|
-
const fullAttrLen = closeQuote + 1;
|
|
697
|
-
const numCustomParts = handler.customAttrSurround
|
|
698
|
-
? handler.customAttrSurround.length * NCP
|
|
699
|
-
: 0;
|
|
700
|
-
const baseIndex = 1 + numCustomParts;
|
|
701
|
-
|
|
702
|
-
attr = [];
|
|
703
|
-
attr[0] = searchStr.substring(0, fullAttrLen);
|
|
704
|
-
attr[baseIndex] = manualMatch[1]; // Attribute name
|
|
705
|
-
attr[baseIndex + 1] = '='; // `customAssign` (falls back to "=" for huge attributes)
|
|
706
|
-
const value = searchStr.substring(manualMatch[0].length + 1, closeQuote);
|
|
707
|
-
// Place value at correct index based on quote type
|
|
708
|
-
if (quoteChar === '"') {
|
|
709
|
-
attr[baseIndex + 2] = value; // Double-quoted value
|
|
710
|
-
} else {
|
|
711
|
-
attr[baseIndex + 3] = value; // Single-quoted value
|
|
712
|
-
}
|
|
713
|
-
currentPos += fullAttrLen;
|
|
714
|
-
consumed += fullAttrLen;
|
|
715
|
-
match.attrs.push(attr);
|
|
716
|
-
continue;
|
|
717
|
-
}
|
|
718
|
-
}
|
|
719
|
-
// Note: Unquoted attribute values are intentionally not handled here
|
|
720
|
-
// Per HTML spec, unquoted values cannot contain spaces or special chars,
|
|
721
|
-
// making a 20 KB+ unquoted value practically impossible; if encountered,
|
|
722
|
-
// it’s malformed HTML and using the truncated regex match is acceptable
|
|
723
|
-
}
|
|
724
|
-
} else {
|
|
725
|
-
// If attr has no value assign but `=` follows in `fullHtml`,
|
|
726
|
-
// the value would be cut off—reset to trigger manual extraction below
|
|
727
|
-
const numCustomParts = handler.customAttrSurround
|
|
728
|
-
? handler.customAttrSurround.length * NCP
|
|
729
|
-
: 0;
|
|
730
|
-
const baseIndex = 1 + numCustomParts;
|
|
731
|
-
if (attr[baseIndex + 1] === undefined) {
|
|
732
|
-
const posAfterName = currentPos + attrEnd;
|
|
733
|
-
if (/^\s*=/.test(fullHtml.slice(posAfterName, posAfterName + 50))) {
|
|
734
|
-
attr = null;
|
|
735
|
-
}
|
|
736
|
-
}
|
|
737
|
-
}
|
|
738
|
-
}
|
|
712
|
+
// Limit the input length to pass to the regex to prevent catastrophic backtracking
|
|
713
|
+
const remainingLen = fullLength - currentPos;
|
|
714
|
+
const isLimited = remainingLen > MAX_ATTR_PARSE_LENGTH;
|
|
739
715
|
|
|
740
|
-
|
|
741
|
-
|
|
742
|
-
|
|
716
|
+
if (!isLimited) {
|
|
717
|
+
// Common case: Use sticky regex directly on full string (no slicing)
|
|
718
|
+
attributeY.lastIndex = currentPos;
|
|
719
|
+
attr = attributeY.exec(fullHtml);
|
|
720
|
+
} else {
|
|
721
|
+
const extractEndPos = currentPos + MAX_ATTR_PARSE_LENGTH;
|
|
722
|
+
|
|
723
|
+
// Create a temporary substring only for attribute parsing (limited for safety)
|
|
724
|
+
const searchStr = fullHtml.substring(currentPos, extractEndPos);
|
|
725
|
+
attr = searchStr.match(attribute);
|
|
726
|
+
|
|
727
|
+
// If input was limited and there’s a match, check if the value might be truncated
|
|
728
|
+
if (attr) {
|
|
729
|
+
// Check if the attribute value extends beyond our search window
|
|
730
|
+
const attrEnd = attr[0].length;
|
|
731
|
+
// If the match ends near the limit, the value might be truncated
|
|
732
|
+
if (attrEnd > MAX_ATTR_PARSE_LENGTH - 100) {
|
|
733
|
+
// Manually extract this attribute to handle potentially huge value
|
|
743
734
|
const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
|
|
744
735
|
if (manualMatch) {
|
|
745
736
|
const quoteChar = searchStr[manualMatch[0].length];
|
|
746
737
|
if (quoteChar === '"' || quoteChar === "'") {
|
|
747
|
-
|
|
748
|
-
const closeQuote = fullHtml.indexOf(quoteChar, currentPos + manualMatch[0].length + 1);
|
|
738
|
+
const closeQuote = searchStr.indexOf(quoteChar, manualMatch[0].length + 1);
|
|
749
739
|
if (closeQuote !== -1) {
|
|
750
|
-
const fullAttrLen = closeQuote
|
|
740
|
+
const fullAttrLen = closeQuote + 1;
|
|
751
741
|
const numCustomParts = handler.customAttrSurround
|
|
752
742
|
? handler.customAttrSurround.length * NCP
|
|
753
743
|
: 0;
|
|
754
744
|
const baseIndex = 1 + numCustomParts;
|
|
755
745
|
|
|
756
746
|
attr = [];
|
|
757
|
-
attr[0] =
|
|
747
|
+
attr[0] = searchStr.substring(0, fullAttrLen);
|
|
758
748
|
attr[baseIndex] = manualMatch[1]; // Attribute name
|
|
759
|
-
attr[baseIndex + 1] = '='; // customAssign
|
|
760
|
-
const value =
|
|
749
|
+
attr[baseIndex + 1] = '='; // `customAssign` (falls back to "=" for huge attributes)
|
|
750
|
+
const value = searchStr.substring(manualMatch[0].length + 1, closeQuote);
|
|
761
751
|
// Place value at correct index based on quote type
|
|
762
752
|
if (quoteChar === '"') {
|
|
763
753
|
attr[baseIndex + 2] = value; // Double-quoted value
|
|
@@ -770,30 +760,83 @@ export class HTMLParser {
|
|
|
770
760
|
continue;
|
|
771
761
|
}
|
|
772
762
|
}
|
|
763
|
+
// Note: Unquoted attribute values are intentionally not handled here
|
|
764
|
+
// Per HTML spec, unquoted values cannot contain spaces or special chars,
|
|
765
|
+
// making a 20 KB+ unquoted value practically impossible; if encountered,
|
|
766
|
+
// it’s malformed HTML and using the truncated regex match is acceptable
|
|
767
|
+
}
|
|
768
|
+
} else {
|
|
769
|
+
// If attr has no value assign but `=` follows in `fullHtml`,
|
|
770
|
+
// the value would be cut off—reset to trigger manual extraction below
|
|
771
|
+
const numCustomParts = handler.customAttrSurround
|
|
772
|
+
? handler.customAttrSurround.length * NCP
|
|
773
|
+
: 0;
|
|
774
|
+
const baseIndex = 1 + numCustomParts;
|
|
775
|
+
if (attr[baseIndex + 1] === undefined) {
|
|
776
|
+
const posAfterName = currentPos + attrEnd;
|
|
777
|
+
if (/^\s*=/.test(fullHtml.slice(posAfterName, posAfterName + 50))) {
|
|
778
|
+
attr = null;
|
|
779
|
+
}
|
|
773
780
|
}
|
|
774
781
|
}
|
|
775
782
|
}
|
|
776
783
|
|
|
777
784
|
if (!attr) {
|
|
778
|
-
|
|
785
|
+
// If input was limited and there’s no match, try manual extraction
|
|
786
|
+
// This handles cases where quoted attributes exceed `MAX_ATTR_PARSE_LENGTH`
|
|
787
|
+
const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
|
|
788
|
+
if (manualMatch) {
|
|
789
|
+
const quoteChar = searchStr[manualMatch[0].length];
|
|
790
|
+
if (quoteChar === '"' || quoteChar === "'") {
|
|
791
|
+
// Search in the full HTML (not limited substring) for closing quote
|
|
792
|
+
const closeQuote = fullHtml.indexOf(quoteChar, currentPos + manualMatch[0].length + 1);
|
|
793
|
+
if (closeQuote !== -1) {
|
|
794
|
+
const fullAttrLen = closeQuote - currentPos + 1;
|
|
795
|
+
const numCustomParts = handler.customAttrSurround
|
|
796
|
+
? handler.customAttrSurround.length * NCP
|
|
797
|
+
: 0;
|
|
798
|
+
const baseIndex = 1 + numCustomParts;
|
|
799
|
+
|
|
800
|
+
attr = [];
|
|
801
|
+
attr[0] = fullHtml.substring(currentPos, closeQuote + 1);
|
|
802
|
+
attr[baseIndex] = manualMatch[1]; // Attribute name
|
|
803
|
+
attr[baseIndex + 1] = '='; // customAssign
|
|
804
|
+
const value = fullHtml.substring(currentPos + manualMatch[0].length + 1, closeQuote);
|
|
805
|
+
// Place value at correct index based on quote type
|
|
806
|
+
if (quoteChar === '"') {
|
|
807
|
+
attr[baseIndex + 2] = value; // Double-quoted value
|
|
808
|
+
} else {
|
|
809
|
+
attr[baseIndex + 3] = value; // Single-quoted value
|
|
810
|
+
}
|
|
811
|
+
currentPos += fullAttrLen;
|
|
812
|
+
consumed += fullAttrLen;
|
|
813
|
+
match.attrs.push(attr);
|
|
814
|
+
continue;
|
|
815
|
+
}
|
|
816
|
+
}
|
|
817
|
+
}
|
|
779
818
|
}
|
|
780
|
-
|
|
781
|
-
const attrLen = attr[0].length;
|
|
782
|
-
currentPos += attrLen;
|
|
783
|
-
consumed += attrLen;
|
|
784
|
-
match.attrs.push(attr);
|
|
785
819
|
}
|
|
786
820
|
|
|
787
|
-
|
|
788
|
-
|
|
789
|
-
close = matchTagClose(currentPos);
|
|
790
|
-
}
|
|
791
|
-
if (close) {
|
|
792
|
-
match.unarySlash = close.slash;
|
|
793
|
-
consumed += close.len;
|
|
794
|
-
match.advance = consumed;
|
|
795
|
-
return match;
|
|
821
|
+
if (!attr) {
|
|
822
|
+
break;
|
|
796
823
|
}
|
|
824
|
+
|
|
825
|
+
const attrLen = attr[0].length;
|
|
826
|
+
currentPos += attrLen;
|
|
827
|
+
consumed += attrLen;
|
|
828
|
+
match.attrs.push(attr);
|
|
829
|
+
}
|
|
830
|
+
|
|
831
|
+
// Check for closing tag (manual scan, regex only for the unusual)
|
|
832
|
+
if (!close) {
|
|
833
|
+
close = matchTagClose(currentPos);
|
|
834
|
+
}
|
|
835
|
+
if (close) {
|
|
836
|
+
match.unarySlash = (close & 1) ? '/' : '';
|
|
837
|
+
consumed += close >> 1;
|
|
838
|
+
match.advance = consumed;
|
|
839
|
+
return match;
|
|
797
840
|
}
|
|
798
841
|
return undefined;
|
|
799
842
|
}
|
|
@@ -964,6 +1007,8 @@ export class HTMLParser {
|
|
|
964
1007
|
if (handler.start) {
|
|
965
1008
|
await handler.start(tagName, attrs, unary, unarySlash);
|
|
966
1009
|
}
|
|
1010
|
+
// Returned so the parse loop can skip lowercasing the name again
|
|
1011
|
+
return lowerTagName;
|
|
967
1012
|
}
|
|
968
1013
|
|
|
969
1014
|
// `needle` must already be lowercase
|
|
@@ -1017,6 +1062,8 @@ export class HTMLParser {
|
|
|
1017
1062
|
handler.end(tagName, []);
|
|
1018
1063
|
}
|
|
1019
1064
|
}
|
|
1065
|
+
// Returned so the parse loop can skip lowercasing the name again
|
|
1066
|
+
return lowerTagName;
|
|
1020
1067
|
}
|
|
1021
1068
|
}
|
|
1022
1069
|
}
|
package/src/lib/attributes.js
CHANGED