html-minifier-next 8.3.0 → 8.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/README.md +12 -6
  2. package/cli.js +137 -31
  3. package/dist/types/htmlminifier.d.ts +3 -3
  4. package/dist/types/htmlminifier.d.ts.map +1 -1
  5. package/dist/types/htmlparser.d.ts.map +1 -1
  6. package/dist/types/lib/attributes.d.ts +3 -0
  7. package/dist/types/lib/attributes.d.ts.map +1 -1
  8. package/dist/types/lib/constants.d.ts +3 -0
  9. package/dist/types/lib/constants.d.ts.map +1 -1
  10. package/dist/types/lib/content.d.ts +15 -1
  11. package/dist/types/lib/content.d.ts.map +1 -1
  12. package/dist/types/lib/elements.d.ts +3 -0
  13. package/dist/types/lib/elements.d.ts.map +1 -1
  14. package/dist/types/lib/option-definitions.d.ts +3 -0
  15. package/dist/types/lib/option-definitions.d.ts.map +1 -1
  16. package/dist/types/lib/options.d.ts +5 -0
  17. package/dist/types/lib/options.d.ts.map +1 -1
  18. package/dist/types/lib/unused-css.d.ts +3 -0
  19. package/dist/types/lib/unused-css.d.ts.map +1 -1
  20. package/dist/types/lib/utils.d.ts +3 -0
  21. package/dist/types/lib/utils.d.ts.map +1 -1
  22. package/dist/types/lib/whitespace.d.ts +3 -0
  23. package/dist/types/lib/whitespace.d.ts.map +1 -1
  24. package/package.json +7 -7
  25. package/src/htmlminifier.js +65 -11
  26. package/src/htmlparser.js +184 -137
  27. package/src/lib/attributes.js +4 -0
  28. package/src/lib/constants.js +3 -1
  29. package/src/lib/content.js +77 -0
  30. package/src/lib/elements.js +4 -0
  31. package/src/lib/file-pool.js +177 -0
  32. package/src/lib/option-definitions.js +3 -1
  33. package/src/lib/options.js +12 -0
  34. package/src/lib/unused-css.js +3 -1
  35. package/src/lib/utils.js +3 -1
  36. package/src/lib/whitespace.js +4 -0
@@ -58,6 +58,7 @@ import {
58
58
 
59
59
  import {
60
60
  hasJsonScriptType,
61
+ extractScriptBodies,
61
62
  processScript
62
63
  } from './lib/content.js';
63
64
 
@@ -103,7 +104,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
103
104
  * improve performance for inputs with repeated CSS (e.g., batch processing).
104
105
  * - Cache is created on first `minify()` call and persists for the process lifetime
105
106
  * - Cache size is locked after first call—subsequent calls reuse the same cache
106
- * - Explicit `0` values are coerced to `1` (minimum functional cache size)
107
+ * - `0` switches the cache off; negative and non-finite values fall back to the default size
107
108
  *
108
109
  * Default: `500`
109
110
  *
@@ -112,7 +113,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
112
113
  * values improve performance for inputs with repeated JavaScript.
113
114
  * - Cache is created on first `minify()` call and persists for the process lifetime
114
115
  * - Cache size is locked after first call—subsequent calls reuse the same cache
115
- * - Explicit `0` values are coerced to `1` (minimum functional cache size)
116
+ * - `0` switches the cache off; negative and non-finite values fall back to the default size
116
117
  *
117
118
  * Default: `500`
118
119
  *
@@ -121,7 +122,7 @@ import { toFragment, replaceCustomFragments } from './lib/fragments.js';
121
122
  * values improve performance for inputs with repeated SVG content.
122
123
  * - Cache is created on first `minify()` call and persists for the process lifetime
123
124
  * - Cache size is locked after first call—subsequent calls reuse the same cache
124
- * - Explicit `0` values are coerced to `1` (minimum functional cache size)
125
+ * - `0` switches the cache off; negative and non-finite values fall back to the default size
125
126
  *
126
127
  * Default: `500`
127
128
  *
@@ -956,6 +957,27 @@ async function createSortFns(value, options, uidIgnore, uidAttr, ignoredMarkupCh
956
957
  }
957
958
  }
958
959
 
960
+ // Match `/^<\/([\w:-]+)>$/` by hand—the regex engine’s per-call setup costs more
961
+ // than the scan on these short buffer entries
962
+ /**
963
+ * @param {string} str
964
+ * @returns {string | null} The tag name, or `null` where the string is not a plain end tag
965
+ */
966
+ function matchPlainEndTag(str) {
967
+ const end = str.length;
968
+ if (end < 4 || str.charCodeAt(0) !== 60 /* < */ || str.charCodeAt(1) !== 47 /* / */ || str.charCodeAt(end - 1) !== 62 /* > */) {
969
+ return null;
970
+ }
971
+ for (let i = 2; i < end - 1; i++) {
972
+ const code = str.charCodeAt(i);
973
+ // `[\w:-]`: Letters, digits, `_`, `:`, `-`
974
+ if (!((code >= 48 && code <= 57) || (code >= 65 && code <= 90) || (code >= 97 && code <= 122) || code === 95 || code === 58 || code === 45)) {
975
+ return null;
976
+ }
977
+ }
978
+ return str.slice(2, end - 1);
979
+ }
980
+
959
981
  /**
960
982
  * @param {string} value - HTML content to minify
961
983
  * @param {ProcessedOptions} options - Normalized minification options
@@ -1198,10 +1220,11 @@ async function minifyHTML(value, options, partialMarkup) {
1198
1220
  function trimTrailingWhitespace(/** @type {number} */ index, /** @type {string} */ nextTag) {
1199
1221
  for (let prevTag = ''; index >= 0 && canTrimWhitespace(prevTag, emptyAttrs); index--) {
1200
1222
  const str = buffer[index] ?? '';
1201
- const match = str.match(/^<\/([\w:-]+)>$/);
1202
- if (match) {
1203
- prevTag = match[1] ?? '';
1204
- } else if (/>$/.test(str) || (buffer[index] = collapseWhitespaceSmart(str, '', nextTag, emptyAttrs, emptyAttrs, options, inlineElements, inlineTextSet))) {
1223
+ const endTagName = matchPlainEndTag(str);
1224
+ if (endTagName !== null) {
1225
+ prevTag = endTagName;
1226
+ } else if (str.charCodeAt(str.length - 1) === 62 /* > */ ||
1227
+ (buffer[index] = collapseWhitespaceSmart(str, '', nextTag, emptyAttrs, emptyAttrs, options, inlineElements, inlineTextSet))) {
1205
1228
  break;
1206
1229
  }
1207
1230
  }
@@ -1504,6 +1527,35 @@ async function minifyHTML(value, options, partialMarkup) {
1504
1527
  buffer.push(comment);
1505
1528
  }
1506
1529
 
1530
+ // Inline scripts are minified one at a time as the parse reaches them, which serializes
1531
+ // work the minifier could overlap. Dispatching them all up front turns that into a single
1532
+ // wave the parse then reads results from. Bodies are matched by exact content, so one the
1533
+ // parse asks for in another form simply misses and takes the original path.
1534
+ /** @type {Map<string, Promise<string>> | null} */
1535
+ let jsPrewarmed = null;
1536
+ if (options.parallelJS && options.minifyJS !== identity && !options.processScripts?.length) {
1537
+ const bodies = extractScriptBodies(value);
1538
+ if (bodies.length > 1) {
1539
+ jsPrewarmed = new Map();
1540
+ // `collapseWhitespace` reaches script bodies through the whole-document pass above,
1541
+ // so the parse sees them trimmed; other whitespace modes leave them as they are
1542
+ const trims = Boolean(options.collapseWhitespace) &&
1543
+ !options.conservativeCollapse && !options.preserveLineBreaks;
1544
+ for (const { code, isModule } of bodies) {
1545
+ const body = trims ? code.trim() : code;
1546
+ const key = (isModule ? 'm|' : '|') + body;
1547
+ if (jsPrewarmed.has(key)) {
1548
+ continue;
1549
+ }
1550
+ const pending = Promise.resolve(options.minifyJS(body, false, isModule));
1551
+ // The awaiting consumer reports the failure; a body the parse never asks for
1552
+ // would not have been minified at all, so its failure is not ours to raise
1553
+ pending.catch(() => {});
1554
+ jsPrewarmed.set(key, pending);
1555
+ }
1556
+ }
1557
+ }
1558
+
1507
1559
  // SVG subtree capture: When SVGO is active, record buffer positions for post-processing
1508
1560
  /** @type {Array<{start: number, end: number}>} */
1509
1561
  const svgBlocks = []; // Array of { start, end } buffer indices
@@ -1854,7 +1906,8 @@ async function minifyHTML(value, options, partialMarkup) {
1854
1906
  text = await processScript(text, options, currentAttrs, minifyHTML);
1855
1907
  }
1856
1908
  if (needsMinifyJS) {
1857
- text = await options.minifyJS(text, false, isModuleScript);
1909
+ const prewarmed = jsPrewarmed?.get((isModuleScript ? 'm|' : '|') + text);
1910
+ text = prewarmed ? await prewarmed : await options.minifyJS(text, false, isModuleScript);
1858
1911
  }
1859
1912
  if (needsMinifyCSS) {
1860
1913
  text = await options.minifyCSS(text, undefined, options.cssContext);
@@ -2001,7 +2054,7 @@ function joinResultSegments(results, options, restoreCustom, restoreIgnore) {
2001
2054
  * - Cache sizes are locked after first initialization—subsequent calls use the same caches
2002
2055
  * even if different `cacheCSS`/`cacheJS`/`cacheSVG` options are provided
2003
2056
  * - The first call’s options determine the cache sizes for subsequent calls
2004
- * - Invalid values (NaN, Infinity) fall back to the default size (500); values below `1` are clamped to `1`
2057
+ * - Invalid values (NaN, Infinity, negative) fall back to the default size (500)
2005
2058
  */
2006
2059
  /** @param {MinifierOptions} options */
2007
2060
  function initCaches(options) {
@@ -2019,8 +2072,9 @@ function initCaches(options) {
2019
2072
  return parsed;
2020
2073
  };
2021
2074
 
2022
- // Sanitize a cache size: Non-finite/NaN falls back to `defaultSize`; otherwise clamped to min 1 and floored
2023
- const sanitizeSize = (/** @type {number} */ size) => Number.isFinite(size) ? Math.max(1, Math.floor(size)) : defaultSize;
2075
+ // Non-finite and negative sizes fall back to `defaultSize`, as they do for the
2076
+ // environment variables above
2077
+ const sanitizeSize = (/** @type {number} */ size) => Number.isFinite(size) && size >= 0 ? Math.floor(size) : defaultSize;
2024
2078
 
2025
2079
  // Get cache sizes with precedence: Options > env > default
2026
2080
  const cssSize = options.cacheCSS !== undefined ? options.cacheCSS
package/src/htmlparser.js CHANGED
@@ -1,6 +1,6 @@
1
1
  /*
2
2
  * HTML Parser By John Resig (ejohn.org)
3
- * Modified by Juriy “kangax” Zaytsev
3
+ * Modified by Juriy “kangax” Zaytsev and Jens Oliver Meiert
4
4
  * Original code by Erik Arvidsson, Mozilla Public License
5
5
  * http://erik.eae.net/simplehtmlparser/simplehtmlparser.js
6
6
  */
@@ -103,6 +103,31 @@ const nonPhrasing = new Set(['address', 'article', 'aside', 'base', 'blockquote'
103
103
  const endsTagName = (/** @type {string} */ character) =>
104
104
  character === '' || character === '/' || character === '>' || /\s/.test(character);
105
105
 
106
+ // Fast path for tag names
107
+ /**
108
+ * @param {string} html
109
+ * @param {number} start - Position of the first name character (after `<` or `</`)
110
+ * @returns {number} Name length from `start`, or -1 to fall back to the qname regex
111
+ */
112
+ function scanAsciiTagName(html, start) {
113
+ const first = html.charCodeAt(start);
114
+ // First char: Letter or `_` (as for ncname)
115
+ if (!((first >= 65 && first <= 90) || (first >= 97 && first <= 122) || first === 95)) {
116
+ return -1;
117
+ }
118
+ let i = start + 1;
119
+ for (;;) {
120
+ const code = html.charCodeAt(i);
121
+ if ((code >= 65 && code <= 90) || (code >= 97 && code <= 122) || (code >= 48 && code <= 57) || code === 95 || code === 45 || code === 46) {
122
+ i++;
123
+ } else if (code === 58 || code >= 128) { // `:` or non-ASCII
124
+ return -1;
125
+ } else {
126
+ return i - start;
127
+ }
128
+ }
129
+ }
130
+
106
131
  // The characters `\s` matches, by code: Tab through carriage return, space, and the
107
132
  // Unicode whitespace and line terminators
108
133
  const isWhitespaceCode = (/** @type {number} */ code) =>
@@ -374,8 +399,7 @@ export class HTMLParser {
374
399
  cachedNextStartTag = null;
375
400
  cachedNextEndTag = null;
376
401
  advance(startTagMatch.advance);
377
- await handleStartTag(startTagMatch);
378
- prevTag = startTagMatch.tagName.toLowerCase();
402
+ prevTag = await handleStartTag(startTagMatch);
379
403
  continue;
380
404
  }
381
405
  if (cachedNextEndTag && cachedNextEndTag.pos === pos) {
@@ -383,8 +407,7 @@ export class HTMLParser {
383
407
  cachedNextStartTag = null;
384
408
  cachedNextEndTag = null;
385
409
  advance(endTagMatch.text.length);
386
- await parseEndTag(endTagMatch.text, endTagMatch.name);
387
- prevTag = '/' + endTagMatch.name.toLowerCase();
410
+ prevTag = '/' + await parseEndTag(endTagMatch.text, endTagMatch.name);
388
411
  prevAttrs = [];
389
412
  continue;
390
413
  }
@@ -452,8 +475,7 @@ export class HTMLParser {
452
475
  const endTagMatch = matchEndTag(pos);
453
476
  if (endTagMatch) {
454
477
  advance(endTagMatch.text.length);
455
- await parseEndTag(endTagMatch.text, endTagMatch.name);
456
- prevTag = '/' + endTagMatch.name.toLowerCase();
478
+ prevTag = '/' + await parseEndTag(endTagMatch.text, endTagMatch.name);
457
479
  prevAttrs = [];
458
480
  continue;
459
481
  }
@@ -463,8 +485,7 @@ export class HTMLParser {
463
485
  const startTagMatch = parseStartTag(pos);
464
486
  if (startTagMatch) {
465
487
  advance(startTagMatch.advance);
466
- await handleStartTag(startTagMatch);
467
- prevTag = startTagMatch.tagName.toLowerCase();
488
+ prevTag = await handleStartTag(startTagMatch);
468
489
  continue;
469
490
  }
470
491
  }
@@ -614,17 +635,29 @@ export class HTMLParser {
614
635
  * where none is complete
615
636
  */
616
637
  function matchEndTag(startPos) {
617
- endTagOpenY.lastIndex = startPos;
618
- const open = endTagOpenY.exec(fullHtml);
619
- if (!open) {
620
- return null;
638
+ let name;
639
+ let nameEnd;
640
+ const scanned = scanAsciiTagName(fullHtml, startPos + 2);
641
+ if (scanned !== -1) {
642
+ nameEnd = startPos + 2 + scanned;
643
+ name = fullHtml.slice(startPos + 2, nameEnd);
644
+ } else {
645
+ endTagOpenY.lastIndex = startPos;
646
+ const open = endTagOpenY.exec(fullHtml);
647
+ if (!open) {
648
+ return null;
649
+ }
650
+ nameEnd = startPos + open[0].length;
651
+ name = open[1] ?? '';
621
652
  }
622
- const close = findEndTagClose(fullHtml, startPos + open[0].length);
623
- return close === -1 ? null : { text: fullHtml.slice(startPos, close + 1), name: open[1] ?? '' };
653
+ const close = findEndTagClose(fullHtml, nameEnd);
654
+ return close === -1 ? null : { text: fullHtml.slice(startPos, close + 1), name };
624
655
  }
625
656
 
626
657
  // Where a tag ends: After optional whitespace, at `>` or `/>`. Scanned by hand in a
627
- // single pass, where a regex did the same with one execution per attribute.
658
+ // single pass, where a regex did the same with one execution per attribute. The
659
+ // result packs length and slash into one number (`length << 1 | hasSlash`, 0 for no
660
+ // match) so the per-attribute loop allocates nothing.
628
661
  function matchTagClose(/** @type {number} */ currentPos) {
629
662
  let scan = currentPos;
630
663
  let code = fullHtml.charCodeAt(scan);
@@ -632,132 +665,89 @@ export class HTMLParser {
632
665
  code = fullHtml.charCodeAt(++scan);
633
666
  }
634
667
  if (code === 62) { // `>`
635
- return { len: scan - currentPos + 1, slash: '' };
668
+ return (scan - currentPos + 1) << 1;
636
669
  }
637
670
  if (code === 47 && fullHtml.charCodeAt(scan + 1) === 62) { // `/>`
638
- return { len: scan - currentPos + 2, slash: '/' };
671
+ return ((scan - currentPos + 2) << 1) | 1;
639
672
  }
640
- return null;
673
+ return 0;
641
674
  }
642
675
 
643
676
  function parseStartTag(/** @type {number} */ startPos) {
644
- startTagOpenY.lastIndex = startPos;
645
- const start = startTagOpenY.exec(fullHtml);
646
- if (start) {
647
- /** @type {{tagName: string, attrs: Array<Array<string|undefined>>, advance: number, unarySlash?: string}} */
648
- const match = {
649
- tagName: start[1] ?? '',
650
- attrs: [],
651
- advance: 0
652
- };
653
- let consumed = start[0].length;
654
- let currentPos = startPos + consumed;
655
- let close, attr;
656
-
657
- // Safety limit: Max length of input to check for attributes
658
- // Protects against catastrophic backtracking on massive attribute values
659
- const MAX_ATTR_PARSE_LENGTH = 20000; // 20 KB should be enough for any reasonable tag
660
-
661
- while (true) {
662
- // Check for closing tag first (manual scan, regex only for the unusual)
663
- close = matchTagClose(currentPos);
664
- if (close) {
665
- break;
666
- }
667
-
668
- // Limit the input length to pass to the regex to prevent catastrophic backtracking
669
- const remainingLen = fullLength - currentPos;
670
- const isLimited = remainingLen > MAX_ATTR_PARSE_LENGTH;
677
+ let tagName;
678
+ let consumed;
679
+ const scanned = scanAsciiTagName(fullHtml, startPos + 1);
680
+ if (scanned !== -1) {
681
+ tagName = fullHtml.slice(startPos + 1, startPos + 1 + scanned);
682
+ consumed = 1 + scanned;
683
+ } else {
684
+ startTagOpenY.lastIndex = startPos;
685
+ const start = startTagOpenY.exec(fullHtml);
686
+ if (!start) {
687
+ return undefined;
688
+ }
689
+ tagName = start[1] ?? '';
690
+ consumed = start[0].length;
691
+ }
692
+ /** @type {{tagName: string, attrs: Array<Array<string|undefined>>, advance: number, unarySlash?: string}} */
693
+ const match = {
694
+ tagName,
695
+ attrs: [],
696
+ advance: 0
697
+ };
698
+ let currentPos = startPos + consumed;
699
+ let close, attr;
700
+
701
+ // Safety limit: Max length of input to check for attributes
702
+ // Protects against catastrophic backtracking on massive attribute values
703
+ const MAX_ATTR_PARSE_LENGTH = 20000; // 20 KB should be enough for any reasonable tag
704
+
705
+ while (true) {
706
+ // Check for closing tag first (manual scan, regex only for the unusual)
707
+ close = matchTagClose(currentPos);
708
+ if (close) {
709
+ break;
710
+ }
671
711
 
672
- if (!isLimited) {
673
- // Common case: Use sticky regex directly on full string (no slicing)
674
- attributeY.lastIndex = currentPos;
675
- attr = attributeY.exec(fullHtml);
676
- } else {
677
- const extractEndPos = currentPos + MAX_ATTR_PARSE_LENGTH;
678
-
679
- // Create a temporary substring only for attribute parsing (limited for safety)
680
- const searchStr = fullHtml.substring(currentPos, extractEndPos);
681
- attr = searchStr.match(attribute);
682
-
683
- // If input was limited and there’s a match, check if the value might be truncated
684
- if (attr) {
685
- // Check if the attribute value extends beyond our search window
686
- const attrEnd = attr[0].length;
687
- // If the match ends near the limit, the value might be truncated
688
- if (attrEnd > MAX_ATTR_PARSE_LENGTH - 100) {
689
- // Manually extract this attribute to handle potentially huge value
690
- const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
691
- if (manualMatch) {
692
- const quoteChar = searchStr[manualMatch[0].length];
693
- if (quoteChar === '"' || quoteChar === "'") {
694
- const closeQuote = searchStr.indexOf(quoteChar, manualMatch[0].length + 1);
695
- if (closeQuote !== -1) {
696
- const fullAttrLen = closeQuote + 1;
697
- const numCustomParts = handler.customAttrSurround
698
- ? handler.customAttrSurround.length * NCP
699
- : 0;
700
- const baseIndex = 1 + numCustomParts;
701
-
702
- attr = [];
703
- attr[0] = searchStr.substring(0, fullAttrLen);
704
- attr[baseIndex] = manualMatch[1]; // Attribute name
705
- attr[baseIndex + 1] = '='; // `customAssign` (falls back to "=" for huge attributes)
706
- const value = searchStr.substring(manualMatch[0].length + 1, closeQuote);
707
- // Place value at correct index based on quote type
708
- if (quoteChar === '"') {
709
- attr[baseIndex + 2] = value; // Double-quoted value
710
- } else {
711
- attr[baseIndex + 3] = value; // Single-quoted value
712
- }
713
- currentPos += fullAttrLen;
714
- consumed += fullAttrLen;
715
- match.attrs.push(attr);
716
- continue;
717
- }
718
- }
719
- // Note: Unquoted attribute values are intentionally not handled here
720
- // Per HTML spec, unquoted values cannot contain spaces or special chars,
721
- // making a 20 KB+ unquoted value practically impossible; if encountered,
722
- // it’s malformed HTML and using the truncated regex match is acceptable
723
- }
724
- } else {
725
- // If attr has no value assign but `=` follows in `fullHtml`,
726
- // the value would be cut off—reset to trigger manual extraction below
727
- const numCustomParts = handler.customAttrSurround
728
- ? handler.customAttrSurround.length * NCP
729
- : 0;
730
- const baseIndex = 1 + numCustomParts;
731
- if (attr[baseIndex + 1] === undefined) {
732
- const posAfterName = currentPos + attrEnd;
733
- if (/^\s*=/.test(fullHtml.slice(posAfterName, posAfterName + 50))) {
734
- attr = null;
735
- }
736
- }
737
- }
738
- }
712
+ // Limit the input length to pass to the regex to prevent catastrophic backtracking
713
+ const remainingLen = fullLength - currentPos;
714
+ const isLimited = remainingLen > MAX_ATTR_PARSE_LENGTH;
739
715
 
740
- if (!attr) {
741
- // If input was limited and there’s no match, try manual extraction
742
- // This handles cases where quoted attributes exceed `MAX_ATTR_PARSE_LENGTH`
716
+ if (!isLimited) {
717
+ // Common case: Use sticky regex directly on full string (no slicing)
718
+ attributeY.lastIndex = currentPos;
719
+ attr = attributeY.exec(fullHtml);
720
+ } else {
721
+ const extractEndPos = currentPos + MAX_ATTR_PARSE_LENGTH;
722
+
723
+ // Create a temporary substring only for attribute parsing (limited for safety)
724
+ const searchStr = fullHtml.substring(currentPos, extractEndPos);
725
+ attr = searchStr.match(attribute);
726
+
727
+ // If input was limited and there’s a match, check if the value might be truncated
728
+ if (attr) {
729
+ // Check if the attribute value extends beyond our search window
730
+ const attrEnd = attr[0].length;
731
+ // If the match ends near the limit, the value might be truncated
732
+ if (attrEnd > MAX_ATTR_PARSE_LENGTH - 100) {
733
+ // Manually extract this attribute to handle potentially huge value
743
734
  const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
744
735
  if (manualMatch) {
745
736
  const quoteChar = searchStr[manualMatch[0].length];
746
737
  if (quoteChar === '"' || quoteChar === "'") {
747
- // Search in the full HTML (not limited substring) for closing quote
748
- const closeQuote = fullHtml.indexOf(quoteChar, currentPos + manualMatch[0].length + 1);
738
+ const closeQuote = searchStr.indexOf(quoteChar, manualMatch[0].length + 1);
749
739
  if (closeQuote !== -1) {
750
- const fullAttrLen = closeQuote - currentPos + 1;
740
+ const fullAttrLen = closeQuote + 1;
751
741
  const numCustomParts = handler.customAttrSurround
752
742
  ? handler.customAttrSurround.length * NCP
753
743
  : 0;
754
744
  const baseIndex = 1 + numCustomParts;
755
745
 
756
746
  attr = [];
757
- attr[0] = fullHtml.substring(currentPos, closeQuote + 1);
747
+ attr[0] = searchStr.substring(0, fullAttrLen);
758
748
  attr[baseIndex] = manualMatch[1]; // Attribute name
759
- attr[baseIndex + 1] = '='; // customAssign
760
- const value = fullHtml.substring(currentPos + manualMatch[0].length + 1, closeQuote);
749
+ attr[baseIndex + 1] = '='; // `customAssign` (falls back to "=" for huge attributes)
750
+ const value = searchStr.substring(manualMatch[0].length + 1, closeQuote);
761
751
  // Place value at correct index based on quote type
762
752
  if (quoteChar === '"') {
763
753
  attr[baseIndex + 2] = value; // Double-quoted value
@@ -770,30 +760,83 @@ export class HTMLParser {
770
760
  continue;
771
761
  }
772
762
  }
763
+ // Note: Unquoted attribute values are intentionally not handled here
764
+ // Per HTML spec, unquoted values cannot contain spaces or special chars,
765
+ // making a 20 KB+ unquoted value practically impossible; if encountered,
766
+ // it’s malformed HTML and using the truncated regex match is acceptable
767
+ }
768
+ } else {
769
+ // If attr has no value assign but `=` follows in `fullHtml`,
770
+ // the value would be cut off—reset to trigger manual extraction below
771
+ const numCustomParts = handler.customAttrSurround
772
+ ? handler.customAttrSurround.length * NCP
773
+ : 0;
774
+ const baseIndex = 1 + numCustomParts;
775
+ if (attr[baseIndex + 1] === undefined) {
776
+ const posAfterName = currentPos + attrEnd;
777
+ if (/^\s*=/.test(fullHtml.slice(posAfterName, posAfterName + 50))) {
778
+ attr = null;
779
+ }
773
780
  }
774
781
  }
775
782
  }
776
783
 
777
784
  if (!attr) {
778
- break;
785
+ // If input was limited and there’s no match, try manual extraction
786
+ // This handles cases where quoted attributes exceed `MAX_ATTR_PARSE_LENGTH`
787
+ const manualMatch = searchStr.match(/^\s*([^\s"'<>/=]+)\s*=\s*/);
788
+ if (manualMatch) {
789
+ const quoteChar = searchStr[manualMatch[0].length];
790
+ if (quoteChar === '"' || quoteChar === "'") {
791
+ // Search in the full HTML (not limited substring) for closing quote
792
+ const closeQuote = fullHtml.indexOf(quoteChar, currentPos + manualMatch[0].length + 1);
793
+ if (closeQuote !== -1) {
794
+ const fullAttrLen = closeQuote - currentPos + 1;
795
+ const numCustomParts = handler.customAttrSurround
796
+ ? handler.customAttrSurround.length * NCP
797
+ : 0;
798
+ const baseIndex = 1 + numCustomParts;
799
+
800
+ attr = [];
801
+ attr[0] = fullHtml.substring(currentPos, closeQuote + 1);
802
+ attr[baseIndex] = manualMatch[1]; // Attribute name
803
+ attr[baseIndex + 1] = '='; // customAssign
804
+ const value = fullHtml.substring(currentPos + manualMatch[0].length + 1, closeQuote);
805
+ // Place value at correct index based on quote type
806
+ if (quoteChar === '"') {
807
+ attr[baseIndex + 2] = value; // Double-quoted value
808
+ } else {
809
+ attr[baseIndex + 3] = value; // Single-quoted value
810
+ }
811
+ currentPos += fullAttrLen;
812
+ consumed += fullAttrLen;
813
+ match.attrs.push(attr);
814
+ continue;
815
+ }
816
+ }
817
+ }
779
818
  }
780
-
781
- const attrLen = attr[0].length;
782
- currentPos += attrLen;
783
- consumed += attrLen;
784
- match.attrs.push(attr);
785
819
  }
786
820
 
787
- // Check for closing tag (manual scan, regex only for the unusual)
788
- if (!close) {
789
- close = matchTagClose(currentPos);
790
- }
791
- if (close) {
792
- match.unarySlash = close.slash;
793
- consumed += close.len;
794
- match.advance = consumed;
795
- return match;
821
+ if (!attr) {
822
+ break;
796
823
  }
824
+
825
+ const attrLen = attr[0].length;
826
+ currentPos += attrLen;
827
+ consumed += attrLen;
828
+ match.attrs.push(attr);
829
+ }
830
+
831
+ // Check for closing tag (manual scan, regex only for the unusual)
832
+ if (!close) {
833
+ close = matchTagClose(currentPos);
834
+ }
835
+ if (close) {
836
+ match.unarySlash = (close & 1) ? '/' : '';
837
+ consumed += close >> 1;
838
+ match.advance = consumed;
839
+ return match;
797
840
  }
798
841
  return undefined;
799
842
  }
@@ -964,6 +1007,8 @@ export class HTMLParser {
964
1007
  if (handler.start) {
965
1008
  await handler.start(tagName, attrs, unary, unarySlash);
966
1009
  }
1010
+ // Returned so the parse loop can skip lowercasing the name again
1011
+ return lowerTagName;
967
1012
  }
968
1013
 
969
1014
  // `needle` must already be lowercase
@@ -1017,6 +1062,8 @@ export class HTMLParser {
1017
1062
  handler.end(tagName, []);
1018
1063
  }
1019
1064
  }
1065
+ // Returned so the parse loop can skip lowercasing the name again
1066
+ return lowerTagName;
1020
1067
  }
1021
1068
  }
1022
1069
  }
@@ -1,3 +1,7 @@
1
+ /**
2
+ * HTML attribute parsing, decoding, and minification
3
+ */
4
+
1
5
  import {
2
6
  RE_EVENT_ATTR_DEFAULT,
3
7
  RE_CAN_REMOVE_ATTR_QUOTES,
@@ -1,4 +1,6 @@
1
- // Regex patterns (to avoid repeated allocations in hot paths)
1
+ /**
2
+ * Regex patterns, element sets, and default configuration
3
+ */
2
4
 
3
5
  const RE_WS_START = /^[ \n\r\t\f]+/;
4
6
  const RE_WS_END = /[ \n\r\t\f]+$/;