html-minifier-next 7.3.0 → 7.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -564,6 +564,10 @@ const RE_LAST_HTML_TAG = /[\s\S]*<(\/?[a-zA-Z][\w:-]*)/;
564
564
  // HTML encoding types for annotation-xml (MathML)
565
565
  const RE_HTML_ENCODING = /^(text\/html|application\/xhtml\+xml)$/i;
566
566
 
567
+ // Probe for SVG/MathML elements—when the input has none (the common case),
568
+ // the per-tag foreign-content machinery can be skipped entirely
569
+ const RE_FOREIGN_ELEMENT_PROBE = /<(?:svg|math)[\s/>]/i;
570
+
567
571
  // Script merging
568
572
 
569
573
  /**
@@ -599,6 +603,13 @@ function findTagEnd(html, pos) {
599
603
  * `</script>` strings inside script content are handled correctly per the HTML
600
604
  * spec (raw text ends at the first `</script>`).
601
605
  *
606
+ * Single forward pass: Each script absorbs all directly following compatible
607
+ * scripts, and the output is assembled from segments—this avoids rebuilding
608
+ * the full string and restarting the scan after every merge (O(n²) on inputs
609
+ * with many merges). Merges can’t cascade backwards: compatibility depends
610
+ * only on the attributes of the first script in a chain, which merging
611
+ * preserves, so one pass finds every merge the iterated rescan found.
612
+ *
602
613
  * @param {string} html - The HTML string to process
603
614
  * @returns {string} HTML with consecutive scripts merged
604
615
  */
@@ -617,92 +628,68 @@ function mergeConsecutiveScripts(html) {
617
628
  return attrs;
618
629
  };
619
630
 
620
- let changed = true;
631
+ /** @type {string[]} */
632
+ const segments = [];
633
+ let consumedPos = 0; // Start of the input region not yet copied to `segments`
634
+
635
+ RE_SCRIPT_OPEN.lastIndex = 0;
636
+ let m1;
621
637
 
622
- // Keep merging until no more changes (handles chains of 3+ scripts)
623
- while (changed) {
624
- changed = false;
625
- RE_SCRIPT_OPEN.lastIndex = 0;
626
- let m1;
638
+ while ((m1 = RE_SCRIPT_OPEN.exec(html)) !== null) {
639
+ // Use findTagEnd() to get the real closing '>', skipping quoted attribute values
640
+ const tagEnd1 = findTagEnd(html, m1.index + 7);
641
+ if (tagEnd1 === -1) break;
627
642
 
628
- while ((m1 = RE_SCRIPT_OPEN.exec(html)) !== null) {
629
- // Use findTagEnd() to get the real closing '>', skipping quoted attribute values
630
- const tagEnd1 = findTagEnd(html, m1.index + 7);
631
- if (tagEnd1 === -1) break;
643
+ const attrs1Str = html.slice(m1.index + 7, tagEnd1);
632
644
 
633
- const attrs1Str = html.slice(m1.index + 7, tagEnd1);
634
- const contentStart1 = tagEnd1 + 1;
645
+ // Find end of this script’s content (first `</script>`—per HTML spec, raw text ends here)
646
+ RE_SCRIPT_CLOSE.lastIndex = tagEnd1 + 1;
647
+ const close1 = RE_SCRIPT_CLOSE.exec(html);
648
+ if (!close1) break;
635
649
 
636
- // Find end of this script’s content (first `</script>`—per HTML spec, raw text ends here)
637
- RE_SCRIPT_CLOSE.lastIndex = contentStart1;
638
- const close1 = RE_SCRIPT_CLOSE.exec(html);
639
- if (!close1) break;
650
+ let mergedContent = html.slice(tagEnd1 + 1, close1.index);
651
+ let afterClose = close1.index + close1[0].length;
640
652
 
641
- const content1 = html.slice(contentStart1, close1.index);
642
- const afterClose1 = close1.index + close1[0].length;
653
+ const a1 = parseAttrs(attrs1Str);
654
+ // `src` (external) and non-default types (modules, JSON, etc.) must not be merged:
655
+ // module scripts have per-script lexical scope, and non-JS content (e.g., JSON)
656
+ // is not concatenable; even identical non-JS types are incompatible
657
+ const mergeable = !('src' in a1) && DEFAULT_JS_TYPES.has((a1.type || '').toLowerCase());
658
+ let mergedAny = false;
643
659
 
660
+ // Absorb all directly following compatible scripts
661
+ while (mergeable) {
644
662
  // Skip optional whitespace and check for a consecutive <script> tag
645
- let i = afterClose1;
663
+ let i = afterClose;
646
664
  while (i < html.length && (html[i] === ' ' || html[i] === '\t' || html[i] === '\n' || html[i] === '\r' || html[i] === '\f')) i++;
647
- if (html.slice(i, i + 7).toLowerCase() !== '<script' || (html.charAt(i + 7) !== '>' && !/\s/.test(html.charAt(i + 7)))) {
648
- RE_SCRIPT_OPEN.lastIndex = afterClose1;
649
- continue;
650
- }
665
+ if (html.slice(i, i + 7).toLowerCase() !== '<script' || (html.charAt(i + 7) !== '>' && !/\s/.test(html.charAt(i + 7)))) break;
651
666
 
652
- const tagStart2 = i;
653
- const tagEnd2 = findTagEnd(html, tagStart2 + 7);
667
+ const tagEnd2 = findTagEnd(html, i + 7);
654
668
  if (tagEnd2 === -1) break;
655
669
 
656
- const attrs2Str = html.slice(tagStart2 + 7, tagEnd2);
657
- const contentStart2 = tagEnd2 + 1;
670
+ const attrs2Str = html.slice(i + 7, tagEnd2);
658
671
 
659
- // Find end of second script’s content
660
- RE_SCRIPT_CLOSE.lastIndex = contentStart2;
672
+ // Find end of the following script’s content
673
+ RE_SCRIPT_CLOSE.lastIndex = tagEnd2 + 1;
661
674
  const close2 = RE_SCRIPT_CLOSE.exec(html);
662
675
  if (!close2) break;
663
676
 
664
- const content2 = html.slice(contentStart2, close2.index);
665
- const afterClose2 = close2.index + close2[0].length;
666
-
667
- const a1 = parseAttrs(attrs1Str);
668
677
  const a2 = parseAttrs(attrs2Str);
669
678
 
670
- // Check for `src`—cannot merge external scripts
671
- if ('src' in a1 || 'src' in a2) {
672
- RE_SCRIPT_OPEN.lastIndex = afterClose1;
673
- continue;
674
- }
675
-
676
- // Check `type` compatibility (both must be default JS)
677
- // Non-JS types (modules, JSON, etc.) must not be merged:
678
- // Module scripts have per-script lexical scope, and non-JS content (e.g., JSON)
679
- // is not concatenable; even identical non-JS types are incompatible
680
- const type1 = (a1.type || '').toLowerCase();
681
- const type2 = (a2.type || '').toLowerCase();
682
- if (!DEFAULT_JS_TYPES.has(type1) || !DEFAULT_JS_TYPES.has(type2)) {
683
- RE_SCRIPT_OPEN.lastIndex = afterClose1;
684
- continue;
685
- }
686
-
687
- // Check for conflicting boolean attributes
679
+ // Check compatibility: No `src`, default JS type, no conflicting boolean
680
+ // attributes, and `nonce` must be same or both absent
681
+ if ('src' in a2 || !DEFAULT_JS_TYPES.has((a2.type || '').toLowerCase())) break;
688
682
  let boolConflict = false;
689
683
  for (const attr of SCRIPT_BOOL_ATTRS) {
690
684
  if ((attr in a1) !== (attr in a2)) { boolConflict = true; break; }
691
685
  }
686
+ if (boolConflict || a1.nonce !== a2.nonce) break;
692
687
 
693
- // Check `nonce`—must be same or both absent
694
- if (boolConflict || a1.nonce !== a2.nonce) {
695
- RE_SCRIPT_OPEN.lastIndex = afterClose1;
696
- continue;
697
- }
698
-
699
- // Scripts are compatible—merge them
700
- changed = true;
701
-
702
- // Combine content—use semicolon normally, newline only for trailing `//` comments
703
- const c1 = content1.trim();
688
+ // Scripts are compatible—combine content: Use semicolon normally,
689
+ // newline only for trailing `//` comments
690
+ const content2 = html.slice(tagEnd2 + 1, close2.index);
691
+ const c1 = mergedContent.trim();
704
692
  const c2 = content2.trim();
705
- let mergedContent;
706
693
  if (c1 && c2) {
707
694
  // Check if last line of c1 contains `//` (single-line comment)
708
695
  // If so, use newline to terminate it; otherwise use semicolon (if not already present)
@@ -713,13 +700,23 @@ function mergeConsecutiveScripts(html) {
713
700
  mergedContent = c1 || c2;
714
701
  }
715
702
 
716
- // Use first script’s attributes (they should be compatible)
717
- html = html.slice(0, m1.index) + `<script${attrs1Str}>${mergedContent}</script>` + html.slice(afterClose2);
718
- break; // Restart scanning (outer while loop)
703
+ afterClose = close2.index + close2[0].length;
704
+ mergedAny = true;
705
+ }
706
+
707
+ if (mergedAny) {
708
+ // Use first script’s attributes (they are compatible)
709
+ segments.push(html.slice(consumedPos, m1.index), `<script${attrs1Str}>${mergedContent}</script>`);
710
+ consumedPos = afterClose;
719
711
  }
712
+
713
+ // Continue scanning after this script (skips `<script` occurrences inside content)
714
+ RE_SCRIPT_OPEN.lastIndex = afterClose;
720
715
  }
721
716
 
722
- return html;
717
+ if (!segments.length) return html;
718
+ segments.push(html.slice(consumedPos));
719
+ return segments.join('');
723
720
  }
724
721
 
725
722
  /**
@@ -1070,14 +1067,6 @@ async function minifyHTML(value, options, partialMarkup) {
1070
1067
  return re.source;
1071
1068
  });
1072
1069
  if (customFragments.length) {
1073
- // Warn about potential ReDoS if custom fragments use unlimited quantifiers
1074
- for (const fragment of customFragments) {
1075
- if (/[*+]/.test(fragment)) {
1076
- options.log?.('Warning: Custom fragment contains unlimited quantifiers (“*” or “+”) which may cause ReDoS vulnerability');
1077
- break;
1078
- }
1079
- }
1080
-
1081
1070
  // Safe approach: Use bounded quantifiers instead of unlimited ones to prevent ReDoS
1082
1071
  const maxQuantifier = options.customFragmentQuantifierLimit || 200;
1083
1072
  const whitespacePattern = `\\s{0,${maxQuantifier}}`;
@@ -1184,12 +1173,282 @@ async function minifyHTML(value, options, partialMarkup) {
1184
1173
  trimTrailingWhitespace(charsIndex, nextTag);
1185
1174
  }
1186
1175
 
1176
+ // Per-text-node context for `charsCollapse`/`charsFinalize`, set by the `chars`
1177
+ // handler before either phase runs; shared variables (rather than parameters)
1178
+ // let both phases live at this scope—`charsCollapse` reassigns `textPrevTag`
1179
+ // and `charsFinalize` must see that change. Defining the phases inside the
1180
+ // `chars` handler instead would allocate two closures per text node.
1181
+ /** @type {string} */
1182
+ let textPrevTag = '';
1183
+ /** @type {string} */
1184
+ let textNextTag = '';
1185
+ /** @type {HTMLAttribute[]} */
1186
+ let textPrevAttrs = emptyAttrs;
1187
+ /** @type {HTMLAttribute[]} */
1188
+ let textNextAttrs = emptyAttrs;
1189
+
1190
+ // Whitespace collapsing phase (sync)
1191
+ function charsCollapse(/** @type {string} */ text) {
1192
+ // Trim outermost newline-based whitespace inside `pre`/`textarea` elements
1193
+ // This removes trailing newlines often added by template engines before closing tags
1194
+ // Only trims single trailing newlines (multiple newlines are likely intentional formatting)
1195
+ if (options.collapseWhitespace && stackNoTrimWhitespace.length) {
1196
+ const topTag = stackNoTrimWhitespace[stackNoTrimWhitespace.length - 1];
1197
+ if (preTextareaDepth > 0) {
1198
+ // Trim trailing whitespace only if it ends with a single newline (not multiple)
1199
+ // Multiple newlines are likely intentional formatting, single newline is often a template artifact
1200
+ // Treat CRLF (`\r\n`), CR (`\r`), and LF (`\n`) as single line-ending units
1201
+ if (textNextTag && textNextTag === '/' + topTag && /[^\r\n](?:\r\n|\r|\n)[ \t]*$/.test(text)) {
1202
+ text = text.replace(/(?:\r\n|\r|\n)[ \t]*$/, '');
1203
+ }
1204
+ }
1205
+ }
1206
+ if (options.collapseWhitespace) {
1207
+ if (!stackNoTrimWhitespace.length) {
1208
+ // When the prev item is a UID placeholder, compute its effective tag name for whitespace decisions;
1209
+ // this is only used in `collapseWhitespaceSmart`—`textPrevTag` itself is not modified,
1210
+ // to avoid side effects on the `inlineTextSet` branch below
1211
+ let effectivePrevTag = textPrevTag;
1212
+ if (textPrevTag === 'comment') {
1213
+ const prevComment = buffer[buffer.length - 1] ?? '';
1214
+ if (!uidIgnore || prevComment.indexOf(uidIgnore) === -1) {
1215
+ if (!prevComment) {
1216
+ textPrevTag = charsPrevTag;
1217
+ effectivePrevTag = textPrevTag;
1218
+ }
1219
+ if (buffer.length > 1 && (!prevComment || (!options.conservativeCollapse && / $/.test(currentChars)))) {
1220
+ const charsIndex = buffer.length - 2;
1221
+ buffer[charsIndex] = (buffer[charsIndex] ?? '').replace(/\s+$/, function (trailingSpaces) {
1222
+ text = trailingSpaces + text;
1223
+ return '';
1224
+ });
1225
+ }
1226
+ } else if (uidIgnorePlaceholderPattern && textNextTag !== 'comment') {
1227
+ // UID placeholder followed by a real element—derive the effective `prevTag` from the
1228
+ // placeholder’s last HTML tag so `collapseWhitespaceSmart` can make the right call;
1229
+ // when `textNextTag` is `comment` (another UID placeholder), `commentFinalize` handles it
1230
+ const match = prevComment.match(uidIgnorePlaceholderPattern);
1231
+ if (match) {
1232
+ const idx = +(match[1] ?? '');
1233
+ if (idx < ignoredMarkupChunks.length) {
1234
+ const content = ignoredMarkupChunks[idx];
1235
+ const lastTagMatch = content && RE_LAST_HTML_TAG.exec(content);
1236
+ if (lastTagMatch) {
1237
+ const group = lastTagMatch[1] ?? '';
1238
+ const isClose = group.charAt(0) === '/';
1239
+ const tagName = resolveName(isClose ? group.slice(1) : group);
1240
+ effectivePrevTag = isClose ? '/' + tagName : tagName;
1241
+ }
1242
+ }
1243
+ }
1244
+ }
1245
+ }
1246
+ if (textPrevTag) {
1247
+ if (textPrevTag === '/nobr' || textPrevTag === 'wbr') {
1248
+ if (/^\s/.test(text)) {
1249
+ let tagIndex = buffer.length - 1;
1250
+ while (tagIndex > 0 && (buffer[tagIndex] ?? '').lastIndexOf('<' + textPrevTag) !== 0) {
1251
+ tagIndex--;
1252
+ }
1253
+ trimTrailingWhitespace(tagIndex - 1, 'br');
1254
+ }
1255
+ } else if (inlineTextSet.has(textPrevTag.charAt(0) === '/' ? textPrevTag.slice(1) : textPrevTag)) {
1256
+ text = collapseWhitespace(text, options, /(?:^|\s)$/.test(currentChars), false);
1257
+ }
1258
+ }
1259
+ if (textPrevTag || textNextTag) {
1260
+ text = collapseWhitespaceSmart(text, effectivePrevTag, textNextTag, textPrevAttrs, textNextAttrs, options, inlineElements, inlineTextSet);
1261
+ } else {
1262
+ text = collapseWhitespace(text, options, true, true);
1263
+ }
1264
+ if (!text && /\s$/.test(currentChars) && textPrevTag && textPrevTag.charAt(0) === '/') {
1265
+ trimTrailingWhitespace(buffer.length - 1, textNextTag);
1266
+ }
1267
+ }
1268
+ if (!stackNoCollapseWhitespace.length && textNextTag !== 'html' && !(textPrevTag && textNextTag)) {
1269
+ text = collapseWhitespace(text, options, false, false, true);
1270
+ }
1271
+ }
1272
+ return text;
1273
+ }
1274
+
1275
+ // Finalization phase (sync): Optional tag handling, entity re-encoding, buffer push
1276
+ function charsFinalize(/** @type {string} */ text) {
1277
+ if (options.removeOptionalTags && text) {
1278
+ // UID-attr tokens are padded with `\t`, which would falsely look like leading whitespace;
1279
+ // resolve single-token text to its actual content for the space/comment checks below
1280
+ let effectiveText = text;
1281
+ if (uidAttrLeadingPattern && uidAttr && text.includes(uidAttr)) {
1282
+ const uidMatch = uidAttrLeadingPattern.exec(text);
1283
+ if (uidMatch) {
1284
+ const idx = +(uidMatch[1] ?? 0);
1285
+ const chunks = idx < ignoredCustomMarkupChunks.length ? ignoredCustomMarkupChunks[idx] : null;
1286
+ if (chunks != null) {
1287
+ effectiveText = chunks[0] ?? '';
1288
+ }
1289
+ }
1290
+ }
1291
+ // `<html>` may be omitted if first thing inside is not a comment
1292
+ // `<body>` may be omitted if first thing inside is not space, comment, `<meta>`, `<link>`, `<script>`, `<style>`, or `<template>`
1293
+ if (optionalStartTag === 'html' || (optionalStartTag === 'body' && !/^\s/.test(effectiveText))) {
1294
+ removeStartTag();
1295
+ }
1296
+ optionalStartTag = '';
1297
+ // `</html>` or `</body>` may be omitted if not followed by comment
1298
+ // `</head>`, `</colgroup>`, or `</caption>` may be omitted if not followed by space or comment
1299
+ if (optionalEndTagEmitted && (compactElements.has(optionalEndTag) || (looseElements.has(optionalEndTag) && !/^\s/.test(effectiveText)))) {
1300
+ removeEndTag();
1301
+ }
1302
+ // Don’t reset `optionalEndTag` if text is only whitespace and will be collapsed (not conservatively)
1303
+ if (!/^\s+$/.test(text) || !options.collapseWhitespace || options.conservativeCollapse) {
1304
+ optionalEndTag = '';
1305
+ optionalEndTagEmitted = false;
1306
+ }
1307
+ }
1308
+ charsPrevTag = /^\s*$/.test(text) ? textPrevTag : 'comment';
1309
+ if (options.decodeEntities && text && !specialContentElements.has(currentTag)) {
1310
+ // Escape any `&` symbols that start either:
1311
+ // 1. a legacy-named character reference (i.e., one that doesn’t end with `;`)
1312
+ // 2. or any other character reference (i.e., one that does end with `;`)
1313
+ // Note that `&` can be escaped as `&amp`, without the semicolon.
1314
+ // https://mathiasbynens.be/notes/ambiguous-ampersands
1315
+ if (text.indexOf('&') !== -1) {
1316
+ text = text.replace(RE_LEGACY_ENTITIES, '&amp$1');
1317
+ }
1318
+ if (text.indexOf('<') !== -1) {
1319
+ text = text.replace(RE_ESCAPE_LT, '&lt;');
1320
+ }
1321
+ }
1322
+ if (uidPattern && options.collapseWhitespace && stackNoTrimWhitespace.length) {
1323
+ text = text.replace(/** @type {RegExp} */ (uidPattern), function (/** @type {string} */ match, /** @type {string} */ _prefix, /** @type {string} */ index) {
1324
+ return ignoredCustomMarkupChunks[+index]?.[0] ?? match;
1325
+ });
1326
+ }
1327
+ currentChars += text;
1328
+ if (text) {
1329
+ hasChars = true;
1330
+ }
1331
+ buffer.push(text);
1332
+ }
1333
+
1334
+ // Comment finalization (sync): Optional tag handling, `htmlmin:ignore` whitespace collapsing, buffer push
1335
+ function commentFinalize(/** @type {string} */ comment) {
1336
+ if (options.removeOptionalTags && comment) {
1337
+ if (uidIgnorePlaceholderPattern) {
1338
+ const match = uidIgnorePlaceholderPattern.exec(comment);
1339
+ if (match) {
1340
+ // UID placeholders represent real HTML content, not true HTML comments;
1341
+ // if there’s a pending optional end tag and the ignored content isn’t itself
1342
+ // a comment (which per the HTML spec prevents omission), resolve it now,
1343
+ // before the UID is pushed to the buffer
1344
+ const idx = +(match[1] ?? 0);
1345
+ const content = idx < ignoredMarkupChunks.length ? ignoredMarkupChunks[idx] : null;
1346
+ if (optionalEndTag && optionalEndTagEmitted && content != null && !/^\s*<!--/.test(content)) {
1347
+ const firstTagMatch = content.match(/^\s*<([a-zA-Z][^\s/>]*)/);
1348
+ const firstTagGroup = firstTagMatch?.[1] ?? '';
1349
+ const firstTag = firstTagGroup ? resolveName(firstTagGroup) : '';
1350
+ if (canRemovePrecedingTag(optionalEndTag, firstTag)) {
1351
+ removeEndTag();
1352
+ }
1353
+ }
1354
+ }
1355
+ }
1356
+ // Comments (real or placeholder) always suppress optional start tag omissions
1357
+ optionalStartTag = '';
1358
+ optionalEndTag = '';
1359
+ optionalEndTagEmitted = false;
1360
+ }
1361
+
1362
+ // Optimize whitespace collapsing between consecutive `htmlmin:ignore` placeholder comments
1363
+ if (options.collapseWhitespace && comment && uidIgnorePlaceholderPattern) {
1364
+ if (uidIgnorePlaceholderPattern.test(comment)) {
1365
+ // Check if previous buffer items are: [ignore-placeholder, whitespace-only text]
1366
+ if (buffer.length >= 2) {
1367
+ const prevText = buffer[buffer.length - 1];
1368
+ const prevComment = buffer[buffer.length - 2];
1369
+
1370
+ // Check if previous item is whitespace-only and item before that is ignore-placeholder
1371
+ if (prevText && /^\s+$/.test(prevText) && prevComment && uidIgnorePlaceholderPattern.test(prevComment)) {
1372
+ // Extract the index from both placeholders to check their content
1373
+ const currentMatch = comment.match(uidIgnorePlaceholderPattern);
1374
+ const prevMatch = prevComment.match(uidIgnorePlaceholderPattern);
1375
+
1376
+ if (currentMatch && prevMatch) {
1377
+ const currentIndex = +(currentMatch[1] ?? 0);
1378
+ const prevIndex = +(prevMatch[1] ?? 0);
1379
+
1380
+ // Defensive bounds check to ensure indices are valid
1381
+ if (currentIndex < ignoredMarkupChunks.length && prevIndex < ignoredMarkupChunks.length) {
1382
+ const currentContent = ignoredMarkupChunks[currentIndex];
1383
+ const prevContent = ignoredMarkupChunks[prevIndex];
1384
+
1385
+ // Only collapse whitespace if both blocks contain HTML (start with `<`)
1386
+ // Don’t collapse if either contains plain text, as that would change meaning
1387
+ if (currentContent && prevContent && /^\s*</.test(currentContent) && /^\s*</.test(prevContent)) {
1388
+ // Extract tag names from the HTML content
1389
+ const currentTagMatch = currentContent.match(/^\s*<([a-zA-Z][\w:-]*)/);
1390
+ const prevTagMatch = prevContent.match(/^\s*<([a-zA-Z][\w:-]*)/);
1391
+ // HTML comments are invisible (no block/inline nature), treat as non-inline
1392
+ const prevIsHtmlComment = !prevTagMatch && RE_HTML_COMMENT_START.test(prevContent);
1393
+ const currentIsHtmlComment = !currentTagMatch && RE_HTML_COMMENT_START.test(currentContent);
1394
+ // Closing tags (e.g., `</div>`)—inline-ness determines whether to collapse
1395
+ const prevClosingTagMatch = !prevTagMatch && RE_CLOSING_TAG_START.exec(prevContent);
1396
+ const currentClosingTagMatch = !currentTagMatch && RE_CLOSING_TAG_START.exec(currentContent);
1397
+
1398
+ // Collapse if both sides are element/closing tags or HTML comments, and neither is inline
1399
+ if ((currentTagMatch || currentIsHtmlComment || currentClosingTagMatch) &&
1400
+ (prevTagMatch || prevIsHtmlComment || prevClosingTagMatch)) {
1401
+ const currentTag = currentTagMatch ? resolveName(currentTagMatch[1] ?? '')
1402
+ : currentClosingTagMatch ? resolveName(currentClosingTagMatch[1] ?? '') : null;
1403
+ const prevTag = prevTagMatch ? resolveName(prevTagMatch[1] ?? '')
1404
+ : prevClosingTagMatch ? resolveName(prevClosingTagMatch[1] ?? '') : null;
1405
+
1406
+ // Don’t collapse between inline elements (HTML comments count as non-inline)
1407
+ if (!inlineElements.has(currentTag ?? '') && !inlineElements.has(prevTag ?? '')) {
1408
+ // Collapse whitespace respecting context rules
1409
+ let collapsedText = prevText;
1410
+
1411
+ // Apply `collapseWhitespace` with appropriate context
1412
+ if (!stackNoTrimWhitespace.length && !stackNoCollapseWhitespace.length) {
1413
+ // Not in pre or other no-collapse context
1414
+ if (options.preserveLineBreaks && /[\n\r]/.test(prevText)) {
1415
+ // Preserve line break as single newline
1416
+ collapsedText = '\n';
1417
+ } else if (options.conservativeCollapse) {
1418
+ // Conservative mode: Keep single space
1419
+ collapsedText = ' ';
1420
+ } else {
1421
+ // Aggressive mode: Remove all whitespace
1422
+ collapsedText = '';
1423
+ }
1424
+ }
1425
+
1426
+ // Replace the whitespace in buffer
1427
+ buffer[buffer.length - 1] = collapsedText;
1428
+ }
1429
+ }
1430
+ }
1431
+ }
1432
+ }
1433
+ }
1434
+ }
1435
+ }
1436
+ }
1437
+
1438
+ buffer.push(comment);
1439
+ }
1440
+
1187
1441
  // SVG subtree capture: When SVGO is active, record buffer positions for post-processing
1188
1442
  /** @type {Array<{start: number, end: number}>} */
1189
1443
  const svgBlocks = []; // Array of { start, end } buffer indices
1190
1444
  let svgBufferStartIndex = -1;
1191
1445
  let svgDepth = 0;
1192
1446
 
1447
+ // One-time probe: If the input contains no SVG/MathML elements and the call
1448
+ // isn’t already inside foreign content (recursive calls inherit options),
1449
+ // the per-tag foreign-content handling in `start`/`end` can be skipped
1450
+ const hasForeignContext = Boolean(options.insideForeignContent) || RE_FOREIGN_ELEMENT_PROBE.test(value);
1451
+
1193
1452
  const parser = new HTMLParser(value, {
1194
1453
  partialMarkup: partialMarkup ?? options.partialMarkup,
1195
1454
  continueOnParseError: options.continueOnParseError,
@@ -1199,44 +1458,49 @@ async function minifyHTML(value, options, partialMarkup) {
1199
1458
  wantsNextTag: !!(options.collapseWhitespace || options.collapseInlineTagWhitespace || options.conservativeCollapse),
1200
1459
 
1201
1460
  start: async function (/** @type {string} */ tag, /** @type {HTMLAttribute[]} */ attrs, /** @type {boolean} */ unary, /** @type {string} */ unarySlash, /** @type {boolean} */ autoGenerated) {
1202
- const lowerTag = tag.toLowerCase();
1203
- if (lowerTag === 'svg' || lowerTag === 'math') {
1204
- // Preserve the surrounding HTML context’s name function (e.g., `identity`
1205
- // under `caseSensitive`) so `foreignObject`/`annotation-xml` can restore it
1206
- const nameHTML = options.name;
1207
- options = Object.create(options);
1208
- options.caseSensitive = true;
1209
- options.keepClosingSlash = true;
1210
- options.name = identity;
1211
- options.nameHTML = nameHTML;
1212
- options.insideSVG = lowerTag === 'svg';
1213
- options.insideForeignContent = true;
1214
- // Disable HTML-specific options that produce invalid XML:
1215
- // SVG with `minifySVG` enabled is passed to SVGO, which requires valid XML input;
1216
- // MathML is never processed by SVGO, so these restrictions never apply to it
1217
- if (lowerTag === 'svg' && options.minifySVG) {
1218
- options.removeAttributeQuotes = false;
1219
- options.decodeEntities = false;
1220
- }
1221
- options.removeTagWhitespace = false;
1222
- }
1223
- // `foreignObject` in SVG and `annotation-xml` in MathML contain HTML content
1224
- // Note: The element itself is in SVG/MathML namespace, only its children are HTML
1461
+ // `lowerTag` stays '' when no foreign content is around—the SVG/MathML
1462
+ // checks below can then never match, and no per-tag lowercasing is needed
1463
+ let lowerTag = '';
1225
1464
  let useNameParentForTag = false;
1226
- if (options.insideForeignContent && (lowerTag === 'foreignobject' ||
1227
- (lowerTag === 'annotation-xml' && attrs.some((/** @type {HTMLAttribute} */ a) => a.name.toLowerCase() === 'encoding' &&
1228
- RE_HTML_ENCODING.test(a.value ?? ''))))) {
1229
- const nameParent = options.name;
1230
- options = Object.create(options);
1231
- options.caseSensitive = false;
1232
- options.keepClosingSlash = false;
1233
- options.nameParent = nameParent; // Preserve for the element tag itself
1234
- options.name = options.nameHTML ?? lowercase;
1235
- options.insideForeignContent = false;
1236
- // Note: `removeAttributeQuotes`, `removeTagWhitespace`, and `decodeEntities`
1237
- // stay disabled (inherited from SVG context) because the entire SVG block
1238
- // must be valid XML for SVGO processing
1239
- useNameParentForTag = true;
1465
+ if (hasForeignContext) {
1466
+ lowerTag = tag.toLowerCase();
1467
+ if (lowerTag === 'svg' || lowerTag === 'math') {
1468
+ // Preserve the surrounding HTML context’s name function (e.g., `identity`
1469
+ // under `caseSensitive`) so `foreignObject`/`annotation-xml` can restore it
1470
+ const nameHTML = options.name;
1471
+ options = Object.create(options);
1472
+ options.caseSensitive = true;
1473
+ options.keepClosingSlash = true;
1474
+ options.name = identity;
1475
+ options.nameHTML = nameHTML;
1476
+ options.insideSVG = lowerTag === 'svg';
1477
+ options.insideForeignContent = true;
1478
+ // Disable HTML-specific options that produce invalid XML:
1479
+ // SVG with `minifySVG` enabled is passed to SVGO, which requires valid XML input;
1480
+ // MathML is never processed by SVGO, so these restrictions never apply to it
1481
+ if (lowerTag === 'svg' && options.minifySVG) {
1482
+ options.removeAttributeQuotes = false;
1483
+ options.decodeEntities = false;
1484
+ }
1485
+ options.removeTagWhitespace = false;
1486
+ }
1487
+ // `foreignObject` in SVG and `annotation-xml` in MathML contain HTML content
1488
+ // Note: The element itself is in SVG/MathML namespace, only its children are HTML
1489
+ if (options.insideForeignContent && (lowerTag === 'foreignobject' ||
1490
+ (lowerTag === 'annotation-xml' && attrs.some((/** @type {HTMLAttribute} */ a) => a.name.toLowerCase() === 'encoding' &&
1491
+ RE_HTML_ENCODING.test(a.value ?? ''))))) {
1492
+ const nameParent = options.name;
1493
+ options = Object.create(options);
1494
+ options.caseSensitive = false;
1495
+ options.keepClosingSlash = false;
1496
+ options.nameParent = nameParent; // Preserve for the element tag itself
1497
+ options.name = options.nameHTML ?? lowercase;
1498
+ options.insideForeignContent = false;
1499
+ // Note: `removeAttributeQuotes`, `removeTagWhitespace`, and `decodeEntities`
1500
+ // stay disabled (inherited from SVG context) because the entire SVG block
1501
+ // must be valid XML for SVGO processing
1502
+ useNameParentForTag = true;
1503
+ }
1240
1504
  }
1241
1505
  // `nameParent` is always set when `useNameParentForTag` is true; the extra check only narrows the type
1242
1506
  tag = (useNameParentForTag && options.nameParent ? options.nameParent : options.name)(tag);
@@ -1312,22 +1576,34 @@ async function minifyHTML(value, options, partialMarkup) {
1312
1576
  /** @type {(tag: string, attrs: HTMLAttribute[]) => void} */ (options.sortAttributes)(tag, attrs);
1313
1577
  }
1314
1578
 
1315
- const attrResults = attrs.map(attr => normalizeAttr(attr, attrs, tag, options, minifyHTML));
1579
+ const attrResults = new Array(attrs.length);
1580
+ let anyThenable = false;
1581
+ for (let i = 0; i < attrs.length; i++) {
1582
+ const result = normalizeAttr(/** @type {HTMLAttribute} */ (attrs[i]), attrs, tag, options, minifyHTML);
1583
+ if (!anyThenable && isThenable(result)) {
1584
+ anyThenable = true;
1585
+ }
1586
+ attrResults[i] = result;
1587
+ }
1316
1588
  // The `isThenable` probe guarantees the sync branch holds no promises
1317
- const normalizedAttrs = /** @type {Array<{name: string, value: string | undefined, attr: HTMLAttribute} | undefined>} */ (attrResults.some(isThenable) ? await Promise.all(attrResults) : attrResults);
1318
- const parts = [];
1319
- let isLast = true;
1589
+ const normalizedAttrs = /** @type {Array<{name: string, value: string | undefined, attr: HTMLAttribute} | undefined>} */ (anyThenable ? await Promise.all(attrResults) : attrResults);
1590
+ // Find the last kept attribute, then emit in order—avoids the
1591
+ // intermediate parts array and reverse of the previous approach
1592
+ let lastKeptIndex = -1;
1320
1593
  for (let i = normalizedAttrs.length - 1; i >= 0; i--) {
1321
- const normalizedAttr = normalizedAttrs[i];
1322
- if (normalizedAttr) {
1323
- parts.push(buildAttr(normalizedAttr, hasUnarySlash, options, isLast, uidAttr));
1324
- isLast = false;
1594
+ if (normalizedAttrs[i]) {
1595
+ lastKeptIndex = i;
1596
+ break;
1325
1597
  }
1326
1598
  }
1327
- parts.reverse();
1328
- if (parts.length > 0) {
1599
+ if (lastKeptIndex !== -1) {
1329
1600
  buffer.push(' ');
1330
- buffer.push.apply(buffer, parts);
1601
+ for (let i = 0; i <= lastKeptIndex; i++) {
1602
+ const normalizedAttr = normalizedAttrs[i];
1603
+ if (normalizedAttr) {
1604
+ buffer.push(buildAttr(normalizedAttr, hasUnarySlash, options, i === lastKeptIndex, uidAttr));
1605
+ }
1606
+ }
1331
1607
  } else if (optional && optionalStartTags.has(tag)) {
1332
1608
  // Start tag must never be omitted if it has any attributes
1333
1609
  optionalStartTag = tag;
@@ -1342,13 +1618,17 @@ async function minifyHTML(value, options, partialMarkup) {
1342
1618
  }
1343
1619
  },
1344
1620
  end: function (/** @type {string} */ tag, /** @type {HTMLAttribute[]} */ attrs, /** @type {boolean} */ autoGenerated) {
1345
- const lowerTag = tag.toLowerCase();
1346
- // Restore parent context when exiting SVG/MathML or HTML-in-foreign-content elements
1347
- if (lowerTag === 'svg' || lowerTag === 'math') {
1348
- options = Object.getPrototypeOf(options);
1349
- } else if ((lowerTag === 'foreignobject' || lowerTag === 'annotation-xml') &&
1350
- !options.insideForeignContent && Object.getPrototypeOf(options).insideForeignContent) {
1351
- options = Object.getPrototypeOf(options);
1621
+ // As in `start`: `lowerTag` stays '' when no foreign content is around
1622
+ let lowerTag = '';
1623
+ if (hasForeignContext) {
1624
+ lowerTag = tag.toLowerCase();
1625
+ // Restore parent context when exiting SVG/MathML or HTML-in-foreign-content elements
1626
+ if (lowerTag === 'svg' || lowerTag === 'math') {
1627
+ options = Object.getPrototypeOf(options);
1628
+ } else if ((lowerTag === 'foreignobject' || lowerTag === 'annotation-xml') &&
1629
+ !options.insideForeignContent && Object.getPrototypeOf(options).insideForeignContent) {
1630
+ options = Object.getPrototypeOf(options);
1631
+ }
1352
1632
  }
1353
1633
  tag = options.name(tag);
1354
1634
 
@@ -1444,10 +1724,11 @@ async function minifyHTML(value, options, partialMarkup) {
1444
1724
  }
1445
1725
  },
1446
1726
  chars: function (/** @type {string} */ text, /** @type {string} */ prevTag, /** @type {string} */ nextTag, /** @type {HTMLAttribute[]} */ prevAttrs, /** @type {HTMLAttribute[]} */ nextAttrs) {
1447
- prevTag = prevTag === '' ? 'comment' : prevTag;
1448
- nextTag = nextTag === '' ? 'comment' : nextTag;
1449
- prevAttrs = prevAttrs || [];
1450
- nextAttrs = nextAttrs || [];
1727
+ // Publish this node’s context for the `charsCollapse`/`charsFinalize` phases
1728
+ textPrevTag = prevTag === '' ? 'comment' : prevTag;
1729
+ textNextTag = nextTag === '' ? 'comment' : nextTag;
1730
+ textPrevAttrs = prevAttrs || [];
1731
+ textNextAttrs = nextAttrs || [];
1451
1732
 
1452
1733
  // Detect whether any async work is actually needed for this text node
1453
1734
  const needsDecode = options.decodeEntities && text && !specialContentElements.has(currentTag) && text.indexOf('&') !== -1;
@@ -1458,150 +1739,6 @@ async function minifyHTML(value, options, partialMarkup) {
1458
1739
  );
1459
1740
  const needsMinifyCSS = options.minifyCSS !== identity && isStyleElement(currentTag, currentAttrs);
1460
1741
 
1461
- // Whitespace collapsing phase (sync); captures `prevTag`/`nextTag`/`prevAttrs`/`nextAttrs` from outer scope
1462
- function charsCollapse(/** @type {string} */ text) {
1463
- // Trim outermost newline-based whitespace inside `pre`/`textarea` elements
1464
- // This removes trailing newlines often added by template engines before closing tags
1465
- // Only trims single trailing newlines (multiple newlines are likely intentional formatting)
1466
- if (options.collapseWhitespace && stackNoTrimWhitespace.length) {
1467
- const topTag = stackNoTrimWhitespace[stackNoTrimWhitespace.length - 1];
1468
- if (preTextareaDepth > 0) {
1469
- // Trim trailing whitespace only if it ends with a single newline (not multiple)
1470
- // Multiple newlines are likely intentional formatting, single newline is often a template artifact
1471
- // Treat CRLF (`\r\n`), CR (`\r`), and LF (`\n`) as single line-ending units
1472
- if (nextTag && nextTag === '/' + topTag && /[^\r\n](?:\r\n|\r|\n)[ \t]*$/.test(text)) {
1473
- text = text.replace(/(?:\r\n|\r|\n)[ \t]*$/, '');
1474
- }
1475
- }
1476
- }
1477
- if (options.collapseWhitespace) {
1478
- if (!stackNoTrimWhitespace.length) {
1479
- // When the prev item is a UID placeholder, compute its effective tag name for whitespace decisions;
1480
- // this is only used in `collapseWhitespaceSmart`—`prevTag` itself is not modified,
1481
- // to avoid side effects on the `inlineTextSet` branch below
1482
- let effectivePrevTag = prevTag;
1483
- if (prevTag === 'comment') {
1484
- const prevComment = buffer[buffer.length - 1] ?? '';
1485
- if (!uidIgnore || prevComment.indexOf(uidIgnore) === -1) {
1486
- if (!prevComment) {
1487
- prevTag = charsPrevTag;
1488
- effectivePrevTag = prevTag;
1489
- }
1490
- if (buffer.length > 1 && (!prevComment || (!options.conservativeCollapse && / $/.test(currentChars)))) {
1491
- const charsIndex = buffer.length - 2;
1492
- buffer[charsIndex] = (buffer[charsIndex] ?? '').replace(/\s+$/, function (trailingSpaces) {
1493
- text = trailingSpaces + text;
1494
- return '';
1495
- });
1496
- }
1497
- } else if (uidIgnorePlaceholderPattern && nextTag !== 'comment') {
1498
- // UID placeholder followed by a real element—derive the effective `prevTag` from the
1499
- // placeholder’s last HTML tag so `collapseWhitespaceSmart` can make the right call;
1500
- // when `nextTag` is `comment` (another UID placeholder), `commentFinalize` handles it
1501
- const match = prevComment.match(uidIgnorePlaceholderPattern);
1502
- if (match) {
1503
- const idx = +(match[1] ?? '');
1504
- if (idx < ignoredMarkupChunks.length) {
1505
- const content = ignoredMarkupChunks[idx];
1506
- const lastTagMatch = content && RE_LAST_HTML_TAG.exec(content);
1507
- if (lastTagMatch) {
1508
- const group = lastTagMatch[1] ?? '';
1509
- const isClose = group.charAt(0) === '/';
1510
- const tagName = resolveName(isClose ? group.slice(1) : group);
1511
- effectivePrevTag = isClose ? '/' + tagName : tagName;
1512
- }
1513
- }
1514
- }
1515
- }
1516
- }
1517
- if (prevTag) {
1518
- if (prevTag === '/nobr' || prevTag === 'wbr') {
1519
- if (/^\s/.test(text)) {
1520
- let tagIndex = buffer.length - 1;
1521
- while (tagIndex > 0 && (buffer[tagIndex] ?? '').lastIndexOf('<' + prevTag) !== 0) {
1522
- tagIndex--;
1523
- }
1524
- trimTrailingWhitespace(tagIndex - 1, 'br');
1525
- }
1526
- } else if (inlineTextSet.has(prevTag.charAt(0) === '/' ? prevTag.slice(1) : prevTag)) {
1527
- text = collapseWhitespace(text, options, /(?:^|\s)$/.test(currentChars), false);
1528
- }
1529
- }
1530
- if (prevTag || nextTag) {
1531
- text = collapseWhitespaceSmart(text, effectivePrevTag, nextTag, prevAttrs, nextAttrs, options, inlineElements, inlineTextSet);
1532
- } else {
1533
- text = collapseWhitespace(text, options, true, true);
1534
- }
1535
- if (!text && /\s$/.test(currentChars) && prevTag && prevTag.charAt(0) === '/') {
1536
- trimTrailingWhitespace(buffer.length - 1, nextTag);
1537
- }
1538
- }
1539
- if (!stackNoCollapseWhitespace.length && nextTag !== 'html' && !(prevTag && nextTag)) {
1540
- text = collapseWhitespace(text, options, false, false, true);
1541
- }
1542
- }
1543
- return text;
1544
- }
1545
-
1546
- // Finalization phase (sync): Optional tag handling, entity re-encoding, buffer push
1547
- function charsFinalize(/** @type {string} */ text) {
1548
- if (options.removeOptionalTags && text) {
1549
- // UID-attr tokens are padded with `\t`, which would falsely look like leading whitespace;
1550
- // resolve single-token text to its actual content for the space/comment checks below
1551
- let effectiveText = text;
1552
- if (uidAttrLeadingPattern && uidAttr && text.includes(uidAttr)) {
1553
- const uidMatch = uidAttrLeadingPattern.exec(text);
1554
- if (uidMatch) {
1555
- const idx = +(uidMatch[1] ?? 0);
1556
- const chunks = idx < ignoredCustomMarkupChunks.length ? ignoredCustomMarkupChunks[idx] : null;
1557
- if (chunks != null) {
1558
- effectiveText = chunks[0] ?? '';
1559
- }
1560
- }
1561
- }
1562
- // `<html>` may be omitted if first thing inside is not a comment
1563
- // `<body>` may be omitted if first thing inside is not space, comment, `<meta>`, `<link>`, `<script>`, `<style>`, or `<template>`
1564
- if (optionalStartTag === 'html' || (optionalStartTag === 'body' && !/^\s/.test(effectiveText))) {
1565
- removeStartTag();
1566
- }
1567
- optionalStartTag = '';
1568
- // `</html>` or `</body>` may be omitted if not followed by comment
1569
- // `</head>`, `</colgroup>`, or `</caption>` may be omitted if not followed by space or comment
1570
- if (optionalEndTagEmitted && (compactElements.has(optionalEndTag) || (looseElements.has(optionalEndTag) && !/^\s/.test(effectiveText)))) {
1571
- removeEndTag();
1572
- }
1573
- // Don’t reset `optionalEndTag` if text is only whitespace and will be collapsed (not conservatively)
1574
- if (!/^\s+$/.test(text) || !options.collapseWhitespace || options.conservativeCollapse) {
1575
- optionalEndTag = '';
1576
- optionalEndTagEmitted = false;
1577
- }
1578
- }
1579
- charsPrevTag = /^\s*$/.test(text) ? prevTag : 'comment';
1580
- if (options.decodeEntities && text && !specialContentElements.has(currentTag)) {
1581
- // Escape any `&` symbols that start either:
1582
- // 1. a legacy-named character reference (i.e., one that doesn’t end with `;`)
1583
- // 2. or any other character reference (i.e., one that does end with `;`)
1584
- // Note that `&` can be escaped as `&amp`, without the semicolon.
1585
- // https://mathiasbynens.be/notes/ambiguous-ampersands
1586
- if (text.indexOf('&') !== -1) {
1587
- text = text.replace(RE_LEGACY_ENTITIES, '&amp$1');
1588
- }
1589
- if (text.indexOf('<') !== -1) {
1590
- text = text.replace(RE_ESCAPE_LT, '&lt;');
1591
- }
1592
- }
1593
- if (uidPattern && options.collapseWhitespace && stackNoTrimWhitespace.length) {
1594
- text = text.replace(/** @type {RegExp} */ (uidPattern), function (/** @type {string} */ match, /** @type {string} */ _prefix, /** @type {string} */ index) {
1595
- return ignoredCustomMarkupChunks[+index]?.[0] ?? match;
1596
- });
1597
- }
1598
- currentChars += text;
1599
- if (text) {
1600
- hasChars = true;
1601
- }
1602
- buffer.push(text);
1603
- }
1604
-
1605
1742
  // Fast path: All work is sync—skip async machinery entirely
1606
1743
  if (!needsDecode && !needsProcessScript && !needsMinifyJS && !needsMinifyCSS) {
1607
1744
  charsFinalize(charsCollapse(text));
@@ -1630,113 +1767,6 @@ async function minifyHTML(value, options, partialMarkup) {
1630
1767
  const prefix = nonStandard ? '<!' : '<!--';
1631
1768
  const suffix = nonStandard ? '>' : '-->';
1632
1769
 
1633
- // Finalization phase (sync): Optional tag handling, `htmlmin:ignore` whitespace collapsing, buffer push
1634
- function commentFinalize(/** @type {string} */ comment) {
1635
- if (options.removeOptionalTags && comment) {
1636
- if (uidIgnorePlaceholderPattern) {
1637
- const match = uidIgnorePlaceholderPattern.exec(comment);
1638
- if (match) {
1639
- // UID placeholders represent real HTML content, not true HTML comments;
1640
- // if there’s a pending optional end tag and the ignored content isn’t itself
1641
- // a comment (which per the HTML spec prevents omission), resolve it now,
1642
- // before the UID is pushed to the buffer
1643
- const idx = +(match[1] ?? 0);
1644
- const content = idx < ignoredMarkupChunks.length ? ignoredMarkupChunks[idx] : null;
1645
- if (optionalEndTag && optionalEndTagEmitted && content != null && !/^\s*<!--/.test(content)) {
1646
- const firstTagMatch = content.match(/^\s*<([a-zA-Z][^\s/>]*)/);
1647
- const firstTagGroup = firstTagMatch?.[1] ?? '';
1648
- const firstTag = firstTagGroup ? resolveName(firstTagGroup) : '';
1649
- if (canRemovePrecedingTag(optionalEndTag, firstTag)) {
1650
- removeEndTag();
1651
- }
1652
- }
1653
- }
1654
- }
1655
- // Comments (real or placeholder) always suppress optional start tag omissions
1656
- optionalStartTag = '';
1657
- optionalEndTag = '';
1658
- optionalEndTagEmitted = false;
1659
- }
1660
-
1661
- // Optimize whitespace collapsing between consecutive `htmlmin:ignore` placeholder comments
1662
- if (options.collapseWhitespace && comment && uidIgnorePlaceholderPattern) {
1663
- if (uidIgnorePlaceholderPattern.test(comment)) {
1664
- // Check if previous buffer items are: [ignore-placeholder, whitespace-only text]
1665
- if (buffer.length >= 2) {
1666
- const prevText = buffer[buffer.length - 1];
1667
- const prevComment = buffer[buffer.length - 2];
1668
-
1669
- // Check if previous item is whitespace-only and item before that is ignore-placeholder
1670
- if (prevText && /^\s+$/.test(prevText) && prevComment && uidIgnorePlaceholderPattern.test(prevComment)) {
1671
- // Extract the index from both placeholders to check their content
1672
- const currentMatch = comment.match(uidIgnorePlaceholderPattern);
1673
- const prevMatch = prevComment.match(uidIgnorePlaceholderPattern);
1674
-
1675
- if (currentMatch && prevMatch) {
1676
- const currentIndex = +(currentMatch[1] ?? 0);
1677
- const prevIndex = +(prevMatch[1] ?? 0);
1678
-
1679
- // Defensive bounds check to ensure indices are valid
1680
- if (currentIndex < ignoredMarkupChunks.length && prevIndex < ignoredMarkupChunks.length) {
1681
- const currentContent = ignoredMarkupChunks[currentIndex];
1682
- const prevContent = ignoredMarkupChunks[prevIndex];
1683
-
1684
- // Only collapse whitespace if both blocks contain HTML (start with `<`)
1685
- // Don’t collapse if either contains plain text, as that would change meaning
1686
- if (currentContent && prevContent && /^\s*</.test(currentContent) && /^\s*</.test(prevContent)) {
1687
- // Extract tag names from the HTML content
1688
- const currentTagMatch = currentContent.match(/^\s*<([a-zA-Z][\w:-]*)/);
1689
- const prevTagMatch = prevContent.match(/^\s*<([a-zA-Z][\w:-]*)/);
1690
- // HTML comments are invisible (no block/inline nature), treat as non-inline
1691
- const prevIsHtmlComment = !prevTagMatch && RE_HTML_COMMENT_START.test(prevContent);
1692
- const currentIsHtmlComment = !currentTagMatch && RE_HTML_COMMENT_START.test(currentContent);
1693
- // Closing tags (e.g., `</div>`)—inline-ness determines whether to collapse
1694
- const prevClosingTagMatch = !prevTagMatch && RE_CLOSING_TAG_START.exec(prevContent);
1695
- const currentClosingTagMatch = !currentTagMatch && RE_CLOSING_TAG_START.exec(currentContent);
1696
-
1697
- // Collapse if both sides are element/closing tags or HTML comments, and neither is inline
1698
- if ((currentTagMatch || currentIsHtmlComment || currentClosingTagMatch) &&
1699
- (prevTagMatch || prevIsHtmlComment || prevClosingTagMatch)) {
1700
- const currentTag = currentTagMatch ? resolveName(currentTagMatch[1] ?? '')
1701
- : currentClosingTagMatch ? resolveName(currentClosingTagMatch[1] ?? '') : null;
1702
- const prevTag = prevTagMatch ? resolveName(prevTagMatch[1] ?? '')
1703
- : prevClosingTagMatch ? resolveName(prevClosingTagMatch[1] ?? '') : null;
1704
-
1705
- // Don’t collapse between inline elements (HTML comments count as non-inline)
1706
- if (!inlineElements.has(currentTag ?? '') && !inlineElements.has(prevTag ?? '')) {
1707
- // Collapse whitespace respecting context rules
1708
- let collapsedText = prevText;
1709
-
1710
- // Apply `collapseWhitespace` with appropriate context
1711
- if (!stackNoTrimWhitespace.length && !stackNoCollapseWhitespace.length) {
1712
- // Not in pre or other no-collapse context
1713
- if (options.preserveLineBreaks && /[\n\r]/.test(prevText)) {
1714
- // Preserve line break as single newline
1715
- collapsedText = '\n';
1716
- } else if (options.conservativeCollapse) {
1717
- // Conservative mode: Keep single space
1718
- collapsedText = ' ';
1719
- } else {
1720
- // Aggressive mode: Remove all whitespace
1721
- collapsedText = '';
1722
- }
1723
- }
1724
-
1725
- // Replace the whitespace in buffer
1726
- buffer[buffer.length - 1] = collapsedText;
1727
- }
1728
- }
1729
- }
1730
- }
1731
- }
1732
- }
1733
- }
1734
- }
1735
- }
1736
-
1737
- buffer.push(comment);
1738
- }
1739
-
1740
1770
  if (options.removeComments) {
1741
1771
  if (isIgnoredComment(text, options)) {
1742
1772
  text = prefix + text + suffix;
@@ -1829,18 +1859,22 @@ function joinResultSegments(results, options, restoreCustom, restoreIgnore) {
1829
1859
 
1830
1860
  if (maxLineLength) {
1831
1861
  let line = ''; const lines = [];
1832
- while (results.length) {
1862
+ // Index-based scan—`Array.shift()` is O(n) per call, which would make this
1863
+ // loop quadratic on documents with many segments
1864
+ let index = 0;
1865
+ while (index < results.length) {
1833
1866
  const len = line.length;
1834
- const cur = results[0] ?? '';
1867
+ const cur = results[index] ?? '';
1835
1868
  const end = cur.indexOf('\n');
1836
1869
  const isClosingTag = Boolean(cur.match(endTag));
1837
1870
  const shouldKeepSameLine = noNewlinesBeforeTagClose && isClosingTag;
1838
1871
 
1839
1872
  if (end < 0) {
1840
- line += restoreIgnore(restoreCustom(results.shift() ?? ''));
1873
+ line += restoreIgnore(restoreCustom(cur));
1874
+ index++;
1841
1875
  } else {
1842
1876
  line += restoreIgnore(restoreCustom(cur.slice(0, end)));
1843
- results[0] = cur.slice(end + 1);
1877
+ results[index] = cur.slice(end + 1);
1844
1878
  }
1845
1879
  if (len > 0 && line.length > maxLineLength && !shouldKeepSameLine) {
1846
1880
  lines.push(line.slice(0, len));
@@ -1926,6 +1960,95 @@ export function getCacheStats() {
1926
1960
  };
1927
1961
  }
1928
1962
 
1963
+ // Memoized options processing: Batch runs typically pass one options object to
1964
+ // many `minify()` calls, and full processing (regex parsing, closure creation,
1965
+ // option-signature stringification) is comparatively expensive. The cache is
1966
+ // keyed on the options object and verified against a deep snapshot, so any
1967
+ // mutation of a reused object—top-level or nested (e.g., engine options)—still
1968
+ // triggers reprocessing. Each call works on a shallow copy of the processed
1969
+ // options, since `minifyHTML` reassigns top-level keys (wrapped minifiers,
1970
+ // sorters, UID comment patterns) that must not leak across calls.
1971
+ /** @type {MinifierOptions} */
1972
+ const EMPTY_OPTIONS = {};
1973
+ /** @type {WeakMap<object, {snapshot: unknown, processed: ProcessedOptions}>} */
1974
+ const processedOptionsCache = new WeakMap();
1975
+
1976
+ // A “plain” value is an array or plain object, which the snapshot copies and
1977
+ // compares structurally; anything else (functions, RegExps, class instances)
1978
+ // is kept and compared by identity—a false mismatch merely reprocesses, which
1979
+ // is the pre-memoization behavior
1980
+ /** @param {unknown} value */
1981
+ function isPlainValue(value) {
1982
+ if (Array.isArray(value)) return true;
1983
+ if (value === null || typeof value !== 'object') return false;
1984
+ const proto = Object.getPrototypeOf(value);
1985
+ return proto === Object.prototype || proto === null;
1986
+ }
1987
+
1988
+ // Budget for `snapshotValue`: Bounds the work spent on pathological options
1989
+ // objects; when the budget runs out, the object isn’t memoized and falls
1990
+ // back to per-call processing
1991
+ const SNAPSHOT_MAX_NODES = 1024;
1992
+ const SNAPSHOT_UNSAFE = Symbol('snapshot-unsafe');
1993
+ // Skipped memoization is fail-safe but silent—warn once per process
1994
+ let snapshotBudgetWarned = false;
1995
+
1996
+ /**
1997
+ * @param {unknown} value
1998
+ * @param {{budget: number}} state
1999
+ * @returns {unknown} The copy, or `SNAPSHOT_UNSAFE` once the budget is exhausted
2000
+ */
2001
+ function snapshotValue(value, state) {
2002
+ if (!isPlainValue(value)) return value;
2003
+ if (--state.budget < 0) return SNAPSHOT_UNSAFE;
2004
+ if (Array.isArray(value)) {
2005
+ const copy = new Array(value.length);
2006
+ for (let i = 0; i < value.length; i++) {
2007
+ const child = snapshotValue(value[i], state);
2008
+ if (child === SNAPSHOT_UNSAFE) return SNAPSHOT_UNSAFE;
2009
+ copy[i] = child;
2010
+ }
2011
+ return copy;
2012
+ }
2013
+ /** @type {Record<string, unknown>} */
2014
+ const copy = {};
2015
+ const source = /** @type {Record<string, unknown>} */ (value);
2016
+ for (const key of Object.keys(source)) {
2017
+ const child = snapshotValue(source[key], state);
2018
+ if (child === SNAPSHOT_UNSAFE) return SNAPSHOT_UNSAFE;
2019
+ copy[key] = child;
2020
+ }
2021
+ return copy;
2022
+ }
2023
+
2024
+ /**
2025
+ * @param {unknown} snapshot
2026
+ * @param {unknown} current
2027
+ * @returns {boolean}
2028
+ */
2029
+ function valueUnchanged(snapshot, current) {
2030
+ if (snapshot === current) return true;
2031
+ if (!isPlainValue(snapshot) || !isPlainValue(current)) return false;
2032
+ if (Array.isArray(snapshot)) {
2033
+ if (!Array.isArray(current) || snapshot.length !== current.length) return false;
2034
+ for (let i = 0; i < snapshot.length; i++) {
2035
+ if (!valueUnchanged(snapshot[i], current[i])) return false;
2036
+ }
2037
+ return true;
2038
+ }
2039
+ if (Array.isArray(current)) return false;
2040
+ const snapshotObj = /** @type {Record<string, unknown>} */ (snapshot);
2041
+ const currentObj = /** @type {Record<string, unknown>} */ (current);
2042
+ const snapshotKeys = Object.keys(snapshotObj);
2043
+ if (snapshotKeys.length !== Object.keys(currentObj).length) return false;
2044
+ for (const key of snapshotKeys) {
2045
+ if (!Object.hasOwn(currentObj, key) || !valueUnchanged(snapshotObj[key], currentObj[key])) {
2046
+ return false;
2047
+ }
2048
+ }
2049
+ return true;
2050
+ }
2051
+
1929
2052
  /**
1930
2053
  * @param {string} value
1931
2054
  * @param {MinifierOptions} [options]
@@ -1934,18 +2057,43 @@ export function getCacheStats() {
1934
2057
  export const minify = async function (value, options) {
1935
2058
  const start = Date.now();
1936
2059
 
2060
+ const inputOptions = options || EMPTY_OPTIONS;
2061
+
1937
2062
  // Initialize caches on first use with configurable sizes
1938
- const caches = initCaches(options || {});
1939
-
1940
- const processedOptions = processOptions(options || {}, {
1941
- getLightningCSS,
1942
- getTerser,
1943
- getSwc,
1944
- getSvgo,
1945
- cssMinifyCache: caches.cssMinifyCache ?? undefined,
1946
- jsMinifyCache: caches.jsMinifyCache ?? undefined,
1947
- svgMinifyCache: caches.svgMinifyCache ?? undefined
1948
- });
2063
+ const caches = initCaches(inputOptions);
2064
+
2065
+ // WeakMap keys must be objects
2066
+ const canMemoize = typeof inputOptions === 'object' && inputOptions !== null;
2067
+ const cached = canMemoize ? processedOptionsCache.get(inputOptions) : undefined;
2068
+ /** @type {ProcessedOptions} */
2069
+ let processedBase;
2070
+ // `valueUnchanged` needs no budget of its own: Recursion is bounded by the
2071
+ // stored snapshot, which is a finite tree of at most `SNAPSHOT_MAX_NODES`
2072
+ if (cached && valueUnchanged(cached.snapshot, inputOptions)) {
2073
+ processedBase = cached.processed;
2074
+ } else {
2075
+ processedBase = processOptions(inputOptions, {
2076
+ getLightningCSS,
2077
+ getTerser,
2078
+ getSwc,
2079
+ getSvgo,
2080
+ cssMinifyCache: caches.cssMinifyCache ?? undefined,
2081
+ jsMinifyCache: caches.jsMinifyCache ?? undefined,
2082
+ svgMinifyCache: caches.svgMinifyCache ?? undefined
2083
+ });
2084
+ if (canMemoize) {
2085
+ const snapshot = snapshotValue(inputOptions, { budget: SNAPSHOT_MAX_NODES });
2086
+ if (snapshot !== SNAPSHOT_UNSAFE) {
2087
+ processedOptionsCache.set(inputOptions, { snapshot, processed: processedBase });
2088
+ } else if (!snapshotBudgetWarned) {
2089
+ snapshotBudgetWarned = true;
2090
+ const warn = typeof inputOptions.log === 'function' ? inputOptions.log : console.warn;
2091
+ warn('HTML Minifier Next: Options object exceeds the memoization complexity limit; options will be processed on every call');
2092
+ }
2093
+ }
2094
+ }
2095
+ // Work on a shallow copy so per-call reassignments don’t reach the cached base
2096
+ const processedOptions = /** @type {ProcessedOptions} */ ({ ...processedBase });
1949
2097
  let result = await minifyHTML(value, processedOptions);
1950
2098
 
1951
2099
  // Post-processing: Merge consecutive inline scripts if enabled