html-minifier-next 7.6.0 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -21
- package/cli.js +152 -50
- package/dist/types/htmlminifier.d.ts +6 -4
- package/dist/types/htmlminifier.d.ts.map +1 -1
- package/dist/types/lib/attributes.d.ts +2 -2
- package/dist/types/lib/attributes.d.ts.map +1 -1
- package/dist/types/lib/constants.d.ts +2 -2
- package/dist/types/lib/constants.d.ts.map +1 -1
- package/dist/types/lib/fragments.d.ts +46 -0
- package/dist/types/lib/fragments.d.ts.map +1 -0
- package/dist/types/lib/option-definitions.d.ts +21 -198
- package/dist/types/lib/option-definitions.d.ts.map +1 -1
- package/dist/types/lib/options.d.ts +1 -1
- package/dist/types/lib/options.d.ts.map +1 -1
- package/dist/types/lib/unused-css.d.ts +5 -4
- package/dist/types/lib/unused-css.d.ts.map +1 -1
- package/dist/types/lib/utils.d.ts +18 -1
- package/dist/types/lib/utils.d.ts.map +1 -1
- package/dist/types/tokenchain.d.ts +2 -2
- package/dist/types/tokenchain.d.ts.map +1 -1
- package/html-minifier-next.schema.json +4 -5
- package/package.json +1 -1
- package/src/htmlminifier.js +49 -64
- package/src/htmlparser.js +4 -4
- package/src/lib/attributes.js +63 -6
- package/src/lib/constants.js +6 -8
- package/src/lib/fragments.js +274 -0
- package/src/lib/option-definitions.js +12 -4
- package/src/lib/options.js +25 -17
- package/src/lib/unused-css.js +5 -22
- package/src/lib/utils.js +308 -3
- package/src/tokenchain.js +37 -30
package/src/lib/attributes.js
CHANGED
|
@@ -13,10 +13,12 @@ import {
|
|
|
13
13
|
isBooleanValue,
|
|
14
14
|
collapsibleValues,
|
|
15
15
|
srcsetElements,
|
|
16
|
-
|
|
16
|
+
RE_EMPTY_ATTRIBUTE,
|
|
17
|
+
RE_STYLE_ELEMENT
|
|
17
18
|
} from './constants.js';
|
|
18
19
|
import { trimWhitespace, collapseWhitespaceAll } from './whitespace.js';
|
|
19
20
|
import { shouldMinifyInnerHTML } from './options.js';
|
|
21
|
+
import { collectUsedSymbols } from './unused-css.js';
|
|
20
22
|
import { identity, isThenable } from './utils.js';
|
|
21
23
|
|
|
22
24
|
/** @import { ProcessedOptions } from './options.js' */
|
|
@@ -25,10 +27,11 @@ import { identity, isThenable } from './utils.js';
|
|
|
25
27
|
|
|
26
28
|
/**
|
|
27
29
|
* @typedef {{ name: string, value?: string | undefined, quote?: string, customAssign?: string, customOpen?: string, customClose?: string }} HTMLAttribute
|
|
28
|
-
* Internal counterpart of the public typedef in
|
|
30
|
+
* Internal counterpart of the public typedef in htmlminifier.js—keep in sync.
|
|
29
31
|
*/
|
|
30
32
|
|
|
31
|
-
// Lazy-load entities (used for `decodeEntities
|
|
33
|
+
// Lazy-load entities (used for `decodeEntities`, event-handler attribute
|
|
34
|
+
// decoding before `minifyJS`, and `srcdoc` decoding/re-encoding)
|
|
32
35
|
|
|
33
36
|
/** @type {Promise<Function> | undefined} */
|
|
34
37
|
let decodeHTMLStrictPromise;
|
|
@@ -39,6 +42,15 @@ async function getDecodeHTMLStrict() {
|
|
|
39
42
|
return decodeHTMLStrictPromise;
|
|
40
43
|
}
|
|
41
44
|
|
|
45
|
+
/** @type {Promise<Function> | undefined} */
|
|
46
|
+
let escapeAttributePromise;
|
|
47
|
+
async function getEscapeAttribute() {
|
|
48
|
+
if (!escapeAttributePromise) {
|
|
49
|
+
escapeAttributePromise = import('entities').then(m => m.escapeAttribute);
|
|
50
|
+
}
|
|
51
|
+
return escapeAttributePromise;
|
|
52
|
+
}
|
|
53
|
+
|
|
42
54
|
// Validators
|
|
43
55
|
|
|
44
56
|
/**
|
|
@@ -353,7 +365,7 @@ function canDeleteEmptyAttribute(tag, attrName, attrValue, options) {
|
|
|
353
365
|
if (typeof options.removeEmptyAttributes === 'function') {
|
|
354
366
|
return options.removeEmptyAttributes(attrName, tag);
|
|
355
367
|
}
|
|
356
|
-
return (tag === 'input' && attrName === 'value') ||
|
|
368
|
+
return (tag === 'input' && attrName === 'value') || RE_EMPTY_ATTRIBUTE.test(attrName);
|
|
357
369
|
}
|
|
358
370
|
|
|
359
371
|
/**
|
|
@@ -593,17 +605,62 @@ function cleanAttributeValue(tag, attrName, attrValue, options, attrs, minifyHTM
|
|
|
593
605
|
}
|
|
594
606
|
|
|
595
607
|
if (tag === 'iframe' && attrName === 'srcdoc') {
|
|
596
|
-
// Recursively minify HTML content within `srcdoc` attribute
|
|
597
608
|
// Fast-path: Skip if nothing would change
|
|
598
609
|
if (!shouldMinifyInnerHTML(options)) {
|
|
599
610
|
return attrValue;
|
|
600
611
|
}
|
|
601
|
-
return
|
|
612
|
+
return minifySrcdoc(attrValue, options, minifyHTMLSelf);
|
|
602
613
|
}
|
|
603
614
|
|
|
604
615
|
return attrValue;
|
|
605
616
|
}
|
|
606
617
|
|
|
618
|
+
/**
|
|
619
|
+
* Recursively minify the document an `iframe srcdoc` attribute holds.
|
|
620
|
+
*
|
|
621
|
+
* Browsers resolve character references before parsing `srcdoc`, so an
|
|
622
|
+
* entity-encoded document is decoded first (unless `decodeEntities` already
|
|
623
|
+
* did) and re-encoded afterwards—but only then, so a literal value passes
|
|
624
|
+
* through byte-identical.
|
|
625
|
+
*
|
|
626
|
+
* @param {string} attrValue
|
|
627
|
+
* @param {ProcessedOptions} options
|
|
628
|
+
* @param {Function} minifyHTMLSelf
|
|
629
|
+
* @returns {Promise<string>}
|
|
630
|
+
*/
|
|
631
|
+
async function minifySrcdoc(attrValue, options, minifyHTMLSelf) {
|
|
632
|
+
let markup = attrValue;
|
|
633
|
+
let wasEncoded = false;
|
|
634
|
+
if (!options.decodeEntities && attrValue.indexOf('&') !== -1) {
|
|
635
|
+
const decode = await getDecodeHTMLStrict();
|
|
636
|
+
markup = decode(attrValue);
|
|
637
|
+
wasEncoded = markup !== attrValue;
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
let srcdocOptions = options;
|
|
641
|
+
// The inner document sees none of the parent’s markup, so its style sheets
|
|
642
|
+
// minify against the symbols the `srcdoc` content references itself
|
|
643
|
+
if (options.removeUnusedCSS && RE_STYLE_ELEMENT.test(markup)) {
|
|
644
|
+
const decode = markup.indexOf('&') !== -1 ? await getDecodeHTMLStrict() : undefined;
|
|
645
|
+
srcdocOptions = {
|
|
646
|
+
...options,
|
|
647
|
+
cssContext: {
|
|
648
|
+
warned: options.cssContext ? options.cssContext.warned : new Set(),
|
|
649
|
+
usedSymbols: collectUsedSymbols(markup, options.removeUnusedCSS.scripts, /** @type {((text: string) => string) | undefined} */ (decode))
|
|
650
|
+
}
|
|
651
|
+
};
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
try {
|
|
655
|
+
const minified = await minifyHTMLSelf(markup, srcdocOptions, true);
|
|
656
|
+
return wasEncoded ? (await getEscapeAttribute())(minified) : minified;
|
|
657
|
+
} catch (err) {
|
|
658
|
+
if (!options.continueOnMinifyError) throw err;
|
|
659
|
+
options.log && options.log(/** @type {Error} */ (err));
|
|
660
|
+
return attrValue;
|
|
661
|
+
}
|
|
662
|
+
}
|
|
663
|
+
|
|
607
664
|
/**
|
|
608
665
|
* Choose appropriate quote character for an attribute value
|
|
609
666
|
* @param {string} attrValue - The attribute value
|
package/src/lib/constants.js
CHANGED
|
@@ -17,6 +17,9 @@ const RE_ATTR_WS_CHECK = /[ \n\r\t\f]/;
|
|
|
17
17
|
const RE_ATTR_WS_COLLAPSE = /[ \n\r\t\f]+/g;
|
|
18
18
|
const RE_ATTR_WS_TRIM = /^[ \n\r\t\f]+|[ \n\r\t\f]+$/g;
|
|
19
19
|
const RE_STYLE_ELEMENT = /<style[\s/>]/i;
|
|
20
|
+
const RE_EMPTY_ATTRIBUTE = new RegExp(
|
|
21
|
+
'^(?:class|id|style|title|lang|dir|on(?:focus|blur|change|click|dblclick|mouse(' +
|
|
22
|
+
'?:down|up|over|move|out)|key(?:press|down|up)))$');
|
|
20
23
|
|
|
21
24
|
// Inline element sets for whitespace handling
|
|
22
25
|
|
|
@@ -163,12 +166,6 @@ const trailingElements = new Set(['dt', 'thead']);
|
|
|
163
166
|
|
|
164
167
|
const htmlElements = new Set(['a', 'abbr', 'acronym', 'address', 'applet', 'area', 'article', 'aside', 'audio', 'b', 'base', 'basefont', 'bdi', 'bdo', 'bgsound', 'big', 'blink', 'blockquote', 'body', 'br', 'button', 'canvas', 'caption', 'center', 'cite', 'code', 'col', 'colgroup', 'command', 'content', 'data', 'datalist', 'dd', 'del', 'details', 'dfn', 'dialog', 'dir', 'div', 'dl', 'dt', 'element', 'em', 'embed', 'fieldset', 'figcaption', 'figure', 'font', 'footer', 'form', 'frame', 'frameset', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6', 'head', 'header', 'hgroup', 'hr', 'html', 'i', 'iframe', 'image', 'img', 'input', 'ins', 'isindex', 'kbd', 'keygen', 'label', 'legend', 'li', 'link', 'listing', 'main', 'map', 'mark', 'marquee', 'menu', 'menuitem', 'meta', 'meter', 'multicol', 'nav', 'nobr', 'noembed', 'noframes', 'noscript', 'object', 'ol', 'optgroup', 'option', 'output', 'p', 'param', 'picture', 'plaintext', 'pre', 'progress', 'q', 'rb', 'rp', 'rt', 'rtc', 'ruby', 's', 'samp', 'script', 'search', 'section', 'select', 'selectedcontent', 'shadow', 'small', 'source', 'spacer', 'span', 'strike', 'strong', 'style', 'sub', 'summary', 'sup', 'table', 'tbody', 'td', 'template', 'textarea', 'tfoot', 'th', 'thead', 'time', 'title', 'tr', 'track', 'tt', 'u', 'ul', 'var', 'video', 'wbr', 'xmp']);
|
|
165
168
|
|
|
166
|
-
// Empty attribute regex
|
|
167
|
-
|
|
168
|
-
const reEmptyAttribute = new RegExp(
|
|
169
|
-
'^(?:class|id|style|title|lang|dir|on(?:focus|blur|change|click|dblclick|mouse(' +
|
|
170
|
-
'?:down|up|over|move|out)|key(?:press|down|up)))$');
|
|
171
|
-
|
|
172
169
|
// Special content elements
|
|
173
170
|
|
|
174
171
|
const specialContentElements = new Set(['script', 'style']);
|
|
@@ -194,6 +191,8 @@ export {
|
|
|
194
191
|
RE_ATTR_WS_COLLAPSE,
|
|
195
192
|
RE_ATTR_WS_TRIM,
|
|
196
193
|
RE_STYLE_ELEMENT,
|
|
194
|
+
RE_EMPTY_ATTRIBUTE,
|
|
195
|
+
|
|
197
196
|
// Inline element sets
|
|
198
197
|
inlineElementsToKeepWhitespaceAround,
|
|
199
198
|
inlineElementsToKeepWhitespaceWithin,
|
|
@@ -236,7 +235,6 @@ export {
|
|
|
236
235
|
trailingElements,
|
|
237
236
|
htmlElements,
|
|
238
237
|
|
|
239
|
-
//
|
|
240
|
-
reEmptyAttribute,
|
|
238
|
+
// Special content elements
|
|
241
239
|
specialContentElements
|
|
242
240
|
};
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Custom fragment matching
|
|
3
|
+
*
|
|
4
|
+
* `ignoreCustomFragments` patterns nearly always describe the same shape: a literal
|
|
5
|
+
* opening delimiter, an any-character or negated-class body, and a literal closing
|
|
6
|
+
* delimiter, as in `<%[\s\S]*?%>` or `\{\{[^}]*?\}\}`. Such a fragment can be found
|
|
7
|
+
* with `indexOf` in linear time, where running the pattern as a regex costs O(n²)
|
|
8
|
+
* on input that opens fragments it never closes—and can cost far more than that when
|
|
9
|
+
* the pattern itself backtracks. Patterns of other shapes keep running as regexes,
|
|
10
|
+
* one per pattern, so each keeps its own flags.
|
|
11
|
+
*/
|
|
12
|
+
|
|
13
|
+
const RE_WHITESPACE = /\s/;
|
|
14
|
+
|
|
15
|
+
// Any-character and negated-class bodies; without the `s` flag, `.` is itself a
|
|
16
|
+
// negated class, excluding line terminators
|
|
17
|
+
const RE_DELIMITED = /^(.*?)(?:\[\\s\\S\]|\[\\S\\s\]|\[\^\]|(\.)|\[\^((?:\\[^]|[^\]\\])+)\])(?:([*+])|\{(\d+)(?:,(\d*))?\})\?(.*)$/;
|
|
18
|
+
|
|
19
|
+
// Flags that leave literal matching alone, so a pattern carrying them can still be
|
|
20
|
+
// scanned for; `i` in particular cannot, since case folding moves character indexes
|
|
21
|
+
const RE_LITERAL_FLAGS = /^[gds]*$/;
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* @typedef {{open: string, close: string, min: number, max: number, excluded: RegExp | null}} DelimitedFragment
|
|
25
|
+
* Literal delimiters around a body, found by scanning; `excluded` holds the
|
|
26
|
+
* characters a negated-class body cannot cross, null when the body spans every
|
|
27
|
+
* character
|
|
28
|
+
* @typedef {{search: RegExp, anchored: RegExp}} PatternFragment
|
|
29
|
+
* Everything else, found by running the pattern itself
|
|
30
|
+
*/
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Read a regex source as a literal string, so it can be matched with `indexOf`
|
|
34
|
+
* @param {string} source
|
|
35
|
+
* @returns {string | null} Null when the source is more than literal characters
|
|
36
|
+
*/
|
|
37
|
+
function toLiteral(source) {
|
|
38
|
+
let literal = '';
|
|
39
|
+
|
|
40
|
+
for (let i = 0; i < source.length; i++) {
|
|
41
|
+
const char = source[i] ?? '';
|
|
42
|
+
if (char === '\\') {
|
|
43
|
+
const escaped = source[++i];
|
|
44
|
+
// A trailing backslash is half an escape the body token was split out of,
|
|
45
|
+
// as in `<%\.*?%>`, where the `.` is literal and not a body
|
|
46
|
+
if (escaped === undefined) return null;
|
|
47
|
+
// `\n` and friends are literal characters, `\s` and `\1` are not
|
|
48
|
+
if (escaped === 'n') literal += '\n';
|
|
49
|
+
else if (escaped === 't') literal += '\t';
|
|
50
|
+
else if (escaped === 'r') literal += '\r';
|
|
51
|
+
else if (escaped === 'f') literal += '\f';
|
|
52
|
+
else if (escaped === 'v') literal += '\v';
|
|
53
|
+
else if (/[A-Za-z0-9]/.test(escaped)) return null;
|
|
54
|
+
else literal += escaped;
|
|
55
|
+
} else if ('.*+?()[]{}|^$'.includes(char)) {
|
|
56
|
+
return null;
|
|
57
|
+
} else {
|
|
58
|
+
literal += char;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
return literal;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* Describe a fragment pattern as literal delimiters around an any-character body
|
|
67
|
+
* @param {RegExp} pattern
|
|
68
|
+
* @returns {DelimitedFragment | null} Null when the pattern has another shape, which
|
|
69
|
+
* the caller answers by running it as a regex
|
|
70
|
+
*/
|
|
71
|
+
function toDelimitedFragment(pattern) {
|
|
72
|
+
if (!RE_LITERAL_FLAGS.test(pattern.flags)) return null;
|
|
73
|
+
|
|
74
|
+
const match = RE_DELIMITED.exec(pattern.source);
|
|
75
|
+
if (!match) return null;
|
|
76
|
+
|
|
77
|
+
const [, rawOpen, dot, negated, simple, exact, upper, rawClose] = match;
|
|
78
|
+
const open = toLiteral(rawOpen ?? '');
|
|
79
|
+
const close = toLiteral(rawClose ?? '');
|
|
80
|
+
// Both delimiters have to be there: without them a match has no boundary to scan to
|
|
81
|
+
if (!open || !close) return null;
|
|
82
|
+
|
|
83
|
+
// The class content doubles as the search for characters the body cannot cross;
|
|
84
|
+
// a `.` body spans every character under the `s` flag, and excludes line
|
|
85
|
+
// terminators without it
|
|
86
|
+
const inner = dot ? (pattern.flags.includes('s') ? null : '\\n\\r\\u2028\\u2029') : negated ?? null;
|
|
87
|
+
|
|
88
|
+
const min = simple ? (simple === '+' ? 1 : 0) : Number(exact);
|
|
89
|
+
const max = simple || upper === '' ? Infinity : Number(upper ?? exact);
|
|
90
|
+
|
|
91
|
+
return { open, close, min, max, excluded: inner === null ? null : new RegExp('[' + inner + ']', 'g') };
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Prepare a fragment pattern for matching, by scanning where the shape allows it
|
|
96
|
+
* @param {RegExp} pattern
|
|
97
|
+
* @returns {DelimitedFragment | PatternFragment}
|
|
98
|
+
*/
|
|
99
|
+
function toFragment(pattern) {
|
|
100
|
+
const delimited = toDelimitedFragment(pattern);
|
|
101
|
+
if (delimited) return delimited;
|
|
102
|
+
|
|
103
|
+
// `g` and `y` are ours to set, the rest belong to the pattern
|
|
104
|
+
const flags = pattern.flags.replace(/[gy]/g, '');
|
|
105
|
+
return { search: new RegExp(pattern.source, flags + 'g'), anchored: new RegExp(pattern.source, flags + 'y') };
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
/**
|
|
109
|
+
* Replace runs of custom fragments, and the whitespace padding them
|
|
110
|
+
* @param {string} value - Document to scan
|
|
111
|
+
* @param {(DelimitedFragment | PatternFragment)[]} fragments - In the order the patterns
|
|
112
|
+
* were given, since the earliest match wins and ties go to the pattern listed first
|
|
113
|
+
* @param {(match: string) => string} replacer - Called with each match, as `replace` would
|
|
114
|
+
* @returns {string}
|
|
115
|
+
*/
|
|
116
|
+
function replaceCustomFragments(value, fragments, replacer) {
|
|
117
|
+
// Where each fragment's next match may start, its next closing delimiter may be
|
|
118
|
+
// found, and its next excluded character sits; all only ever move forward, which
|
|
119
|
+
// is what keeps the whole scan linear
|
|
120
|
+
const found = fragments.map(() => /** @type {{start: number, end: number} | null} */ (null));
|
|
121
|
+
const closesAt = fragments.map(() => -1);
|
|
122
|
+
const excludedAt = fragments.map(() => -1);
|
|
123
|
+
const exhausted = fragments.map(() => false);
|
|
124
|
+
|
|
125
|
+
/**
|
|
126
|
+
* Whether a body region holds a character its class excludes
|
|
127
|
+
* @param {DelimitedFragment} fragment
|
|
128
|
+
* @param {number} index - Which fragment, for the forward-only bookkeeping
|
|
129
|
+
* @param {number} bodyStart
|
|
130
|
+
* @param {number} close
|
|
131
|
+
* @returns {boolean}
|
|
132
|
+
*/
|
|
133
|
+
const bodyBlocked = (fragment, index, bodyStart, close) => {
|
|
134
|
+
if (!fragment.excluded) return false;
|
|
135
|
+
if (/** @type {number} */ (excludedAt[index]) < bodyStart) {
|
|
136
|
+
fragment.excluded.lastIndex = bodyStart;
|
|
137
|
+
const blocked = fragment.excluded.exec(value);
|
|
138
|
+
excludedAt[index] = blocked ? blocked.index : Infinity;
|
|
139
|
+
}
|
|
140
|
+
return /** @type {number} */ (excludedAt[index]) < close;
|
|
141
|
+
};
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* Earliest match of one delimited fragment at or after `from`
|
|
145
|
+
* @param {DelimitedFragment} fragment
|
|
146
|
+
* @param {number} index - Which fragment, for the forward-only bookkeeping
|
|
147
|
+
* @param {number} from
|
|
148
|
+
* @param {boolean} anchored - Whether the match has to start exactly at `from`
|
|
149
|
+
* @returns {{start: number, end: number} | null}
|
|
150
|
+
*/
|
|
151
|
+
const scan = (fragment, index, from, anchored) => {
|
|
152
|
+
let openFrom = from;
|
|
153
|
+
|
|
154
|
+
for (;;) {
|
|
155
|
+
const open = anchored
|
|
156
|
+
? (value.startsWith(fragment.open, from) ? from : -1)
|
|
157
|
+
: value.indexOf(fragment.open, openFrom);
|
|
158
|
+
if (open === -1) {
|
|
159
|
+
if (!anchored) exhausted[index] = true;
|
|
160
|
+
return null;
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
const bodyStart = open + fragment.open.length;
|
|
164
|
+
if (/** @type {number} */ (closesAt[index]) < bodyStart + fragment.min) {
|
|
165
|
+
closesAt[index] = value.indexOf(fragment.close, bodyStart + fragment.min);
|
|
166
|
+
}
|
|
167
|
+
const close = /** @type {number} */ (closesAt[index]);
|
|
168
|
+
if (close === -1) {
|
|
169
|
+
exhausted[index] = true;
|
|
170
|
+
return null;
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
if (close - bodyStart <= fragment.max && !bodyBlocked(fragment, index, bodyStart, close)) {
|
|
174
|
+
return { start: open, end: close + fragment.close.length };
|
|
175
|
+
}
|
|
176
|
+
// The body is longer than the pattern allows, or holds a character its class
|
|
177
|
+
// excludes, so the match has to start later
|
|
178
|
+
if (anchored) return null;
|
|
179
|
+
openFrom = open + 1;
|
|
180
|
+
}
|
|
181
|
+
};
|
|
182
|
+
|
|
183
|
+
/**
|
|
184
|
+
* Earliest match of one regex fragment at or after `from`
|
|
185
|
+
* @param {PatternFragment} fragment
|
|
186
|
+
* @param {number} from
|
|
187
|
+
* @param {boolean} anchored
|
|
188
|
+
* @returns {{start: number, end: number} | null}
|
|
189
|
+
*/
|
|
190
|
+
const run = (fragment, from, anchored) => {
|
|
191
|
+
const pattern = anchored ? fragment.anchored : fragment.search;
|
|
192
|
+
pattern.lastIndex = from;
|
|
193
|
+
|
|
194
|
+
for (;;) {
|
|
195
|
+
const match = pattern.exec(value);
|
|
196
|
+
if (!match) return null;
|
|
197
|
+
// A pattern that matches nothing would leave the run in place forever
|
|
198
|
+
if (match[0].length > 0) return { start: match.index, end: match.index + match[0].length };
|
|
199
|
+
if (anchored) return null;
|
|
200
|
+
pattern.lastIndex = match.index + 1;
|
|
201
|
+
}
|
|
202
|
+
};
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* @param {number} from
|
|
206
|
+
* @param {boolean} anchored
|
|
207
|
+
* @returns {{start: number, end: number} | null}
|
|
208
|
+
*/
|
|
209
|
+
const find = (from, anchored) => {
|
|
210
|
+
/** @type {{start: number, end: number} | null} */
|
|
211
|
+
let earliest = null;
|
|
212
|
+
|
|
213
|
+
for (let i = 0; i < fragments.length; i++) {
|
|
214
|
+
if (exhausted[i]) continue;
|
|
215
|
+
const fragment = /** @type {DelimitedFragment | PatternFragment} */ (fragments[i]);
|
|
216
|
+
|
|
217
|
+
/** @type {{start: number, end: number} | null} */
|
|
218
|
+
let match;
|
|
219
|
+
if (anchored) {
|
|
220
|
+
match = 'open' in fragment ? scan(fragment, i, from, true) : run(fragment, from, true);
|
|
221
|
+
} else {
|
|
222
|
+
// Matches only ever move forward, so the last one found still stands
|
|
223
|
+
const previous = found[i];
|
|
224
|
+
match = previous && previous.start >= from
|
|
225
|
+
? previous
|
|
226
|
+
: ('open' in fragment ? scan(fragment, i, from, false) : run(fragment, from, false));
|
|
227
|
+
found[i] = match;
|
|
228
|
+
// Nothing ahead now means nothing ahead later either
|
|
229
|
+
if (!match) exhausted[i] = true;
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
// Ties go to the fragment listed first, the way alternation would resolve them
|
|
233
|
+
if (match && (!earliest || match.start < earliest.start)) earliest = match;
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
return earliest;
|
|
237
|
+
};
|
|
238
|
+
|
|
239
|
+
let out = '';
|
|
240
|
+
let copied = 0;
|
|
241
|
+
let search = 0;
|
|
242
|
+
|
|
243
|
+
while (search <= value.length) {
|
|
244
|
+
const first = find(search, false);
|
|
245
|
+
if (!first) break;
|
|
246
|
+
|
|
247
|
+
// Fragments running straight into each other are one match
|
|
248
|
+
let end = first.end;
|
|
249
|
+
for (;;) {
|
|
250
|
+
const next = find(end, true);
|
|
251
|
+
if (!next) break;
|
|
252
|
+
end = next.end;
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// …as is the whitespace on either side, back to where the last match left off
|
|
256
|
+
let start = first.start;
|
|
257
|
+
while (start > copied && RE_WHITESPACE.test(value[start - 1] ?? '')) start--;
|
|
258
|
+
while (end < value.length && RE_WHITESPACE.test(value[end] ?? '')) end++;
|
|
259
|
+
|
|
260
|
+
out += value.slice(copied, start) + replacer(value.slice(start, end));
|
|
261
|
+
copied = end;
|
|
262
|
+
search = end;
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
return copied === 0 ? value : out + value.slice(copied);
|
|
266
|
+
}
|
|
267
|
+
|
|
268
|
+
// Exports
|
|
269
|
+
|
|
270
|
+
export {
|
|
271
|
+
toDelimitedFragment,
|
|
272
|
+
toFragment,
|
|
273
|
+
replaceCustomFragments
|
|
274
|
+
};
|
|
@@ -1,5 +1,13 @@
|
|
|
1
1
|
// Single source of truth for minifier option names, descriptions, types, and shared defaults
|
|
2
2
|
|
|
3
|
+
/**
|
|
4
|
+
* @typedef {object} OptionDefinition
|
|
5
|
+
* @property {string} description Help text, phrased for the option’s primary CLI form
|
|
6
|
+
* @property {string} [descriptionAffirmative] Help text for what enabling the option does, where `description` describes the negated form
|
|
7
|
+
* @property {string} type Key into the parser and JSON Schema type maps in cli.js and scripts/build-schema.js
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
/** @type {Record<string, OptionDefinition>} */
|
|
3
11
|
const optionDefinitions = {
|
|
4
12
|
cacheCSS: {
|
|
5
13
|
description: 'Set CSS minification cache size (number of entries, default: 500)',
|
|
@@ -62,10 +70,6 @@ const optionDefinitions = {
|
|
|
62
70
|
description: 'Array of regexes that allow to support custom event attributes for minifyJS (e.g., `ng-click`)',
|
|
63
71
|
type: 'regexpArray'
|
|
64
72
|
},
|
|
65
|
-
customFragmentQuantifierLimit: {
|
|
66
|
-
description: 'Set maximum quantifier limit for custom fragments to prevent ReDoS attacks (default: 200)',
|
|
67
|
-
type: 'int'
|
|
68
|
-
},
|
|
69
73
|
decodeEntities: {
|
|
70
74
|
description: 'Use direct Unicode characters whenever possible',
|
|
71
75
|
type: 'boolean'
|
|
@@ -194,6 +198,10 @@ const optionDefinitions = {
|
|
|
194
198
|
description: 'Trim whitespace around custom fragments (`--ignore-custom-fragments`)',
|
|
195
199
|
type: 'boolean'
|
|
196
200
|
},
|
|
201
|
+
strictCustomFragments: {
|
|
202
|
+
description: 'Reject `ignoreCustomFragments` patterns that risk catastrophic backtracking (rather than warning about them)',
|
|
203
|
+
type: 'boolean'
|
|
204
|
+
},
|
|
197
205
|
useShortDoctype: {
|
|
198
206
|
description: 'Replaces the doctype with the short HTML doctype',
|
|
199
207
|
type: 'boolean'
|
package/src/lib/options.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { createUrlMinifier } from './urls.js';
|
|
2
|
-
import { LRU, MAX_CACHE_ENTRY_SIZE, stableStringify, hashContent, identity, lowercase, replaceAsync, parseRegExp } from './utils.js';
|
|
2
|
+
import { LRU, MAX_CACHE_ENTRY_SIZE, stableStringify, hashContent, identity, lowercase, replaceAsync, parseRegExp, describeQuantifierRisk } from './utils.js';
|
|
3
3
|
import { RE_TRAILING_SEMICOLON } from './constants.js';
|
|
4
4
|
import { canCollapseWhitespace, canTrimWhitespace } from './whitespace.js';
|
|
5
5
|
import { wrapCSS, unwrapCSS } from './content.js';
|
|
@@ -28,7 +28,7 @@ import { optionDefinitions, optionDefaults } from './option-definitions.js';
|
|
|
28
28
|
|
|
29
29
|
/**
|
|
30
30
|
* Options object produced by `processOptions` and consumed by `minifyHTML` and
|
|
31
|
-
* the
|
|
31
|
+
* the lib/ helpers; normalization guarantees that the function-valued options
|
|
32
32
|
* below are always present (defaulting to identity/built-in functions), and
|
|
33
33
|
* minification adds writable internal state on top of the public options
|
|
34
34
|
* (set on prototype-chain forks during SVG/MathML namespace transitions)
|
|
@@ -97,9 +97,8 @@ const optionKeysExtra = new Set(['preset', 'log', 'canCollapseWhitespace', 'canT
|
|
|
97
97
|
// key per process, so repeated `minify` calls (e.g., batch runs) don’t flood STDERR
|
|
98
98
|
const optionKeysWarned = new Set();
|
|
99
99
|
const presetNamesWarned = new Set();
|
|
100
|
-
//
|
|
101
|
-
|
|
102
|
-
let customFragmentQuantifierWarned = false;
|
|
100
|
+
// Custom fragments whose shape risks ReDoS, warned about once per pattern per process
|
|
101
|
+
const customFragmentsWarned = new Set();
|
|
103
102
|
const unusedCSSWarned = new Set();
|
|
104
103
|
// Object-valued options handed a string, warned about once per distinct value
|
|
105
104
|
const stringValuesWarned = new Set();
|
|
@@ -191,7 +190,7 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
|
|
|
191
190
|
// A string carries no configuration for these options. The CLI parses config
|
|
192
191
|
// values as JSON first, so only a value that is not JSON reaches this from
|
|
193
192
|
// there. (`minifyURLs` is deliberately excluded—there, a string names the site.)
|
|
194
|
-
const definition =
|
|
193
|
+
const definition = optionDefinitions[key];
|
|
195
194
|
if (typeof option === 'string' && definition?.type === 'jsonObject') {
|
|
196
195
|
const message = `HTML Minifier Next: Ignoring \`${key}\`—it takes a boolean or an object, not a string (“${option}”)`;
|
|
197
196
|
if (!stringValuesWarned.has(message)) {
|
|
@@ -598,22 +597,31 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
|
|
|
598
597
|
} else if (['customAttrAssign', 'customEventAttributes', 'ignoreCustomComments', 'ignoreCustomFragments'].includes(key)) {
|
|
599
598
|
// Array of regex patterns
|
|
600
599
|
optionsDynamic[key] = parseRegExpArray(option);
|
|
601
|
-
// Warn about potential ReDoS when user-provided fragments use unlimited
|
|
602
|
-
// quantifiers; only explicitly passed fragments are checked
|
|
603
|
-
if (key === 'ignoreCustomFragments' && !customFragmentQuantifierWarned) {
|
|
604
|
-
for (const re of /** @type {RegExp[]} */ (optionsDynamic[key])) {
|
|
605
|
-
if (/[*+]/.test(re.source)) {
|
|
606
|
-
customFragmentQuantifierWarned = true;
|
|
607
|
-
warn('HTML Minifier Next: Custom fragment contains unlimited quantifiers (“*” or “+”) which may cause ReDoS vulnerability');
|
|
608
|
-
break;
|
|
609
|
-
}
|
|
610
|
-
}
|
|
611
|
-
}
|
|
612
600
|
} else {
|
|
613
601
|
optionsDynamic[key] = option;
|
|
614
602
|
}
|
|
615
603
|
});
|
|
616
604
|
|
|
605
|
+
// Fragments that compound quantifiers or alternation under unbounded repetition
|
|
606
|
+
// are the shapes that backtrack catastrophically, and they are also the ones a
|
|
607
|
+
// linear scan cannot stand in for; so are patterns too long or too deeply
|
|
608
|
+
// nested to read, which are refused for that rather than for a shape. A lone
|
|
609
|
+
// `[\s\S]*?` up to a literal terminator is linear and passes; HMN’s default
|
|
610
|
+
// fragments have exactly that shape, so the check flagging it would mean
|
|
611
|
+
// warning about the defaults themselves.
|
|
612
|
+
for (const re of options.ignoreCustomFragments || []) {
|
|
613
|
+
const risk = describeQuantifierRisk(re.source);
|
|
614
|
+
if (!risk) continue;
|
|
615
|
+
const problem = `Custom fragment \`/${re.source}/\` ${risk}`;
|
|
616
|
+
if (options.strictCustomFragments) {
|
|
617
|
+
throw new Error(`HTML Minifier Next: ${problem}`);
|
|
618
|
+
}
|
|
619
|
+
if (!customFragmentsWarned.has(re.source)) {
|
|
620
|
+
customFragmentsWarned.add(re.source);
|
|
621
|
+
warn(`HTML Minifier Next: ${problem}`);
|
|
622
|
+
}
|
|
623
|
+
}
|
|
624
|
+
|
|
617
625
|
// Unused-CSS removal rides along with Lightning CSS, so it silently does nothing
|
|
618
626
|
// when `minifyCSS` is off or replaced by a function—say so rather than let it pass
|
|
619
627
|
if (options.removeUnusedCSS) {
|
package/src/lib/unused-css.js
CHANGED
|
@@ -37,9 +37,6 @@ const fragmentReferenceAttributes = new Set([
|
|
|
37
37
|
'xlink:href'
|
|
38
38
|
]);
|
|
39
39
|
|
|
40
|
-
// `srcdoc` can nest, and each level costs another scan of its own markup
|
|
41
|
-
const SRCDOC_MAX_DEPTH = 3;
|
|
42
|
-
|
|
43
40
|
const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
|
|
44
41
|
const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
|
|
45
42
|
// SVG paints and filters reach elements by ID through `url(#gradient)`, in
|
|
@@ -173,16 +170,17 @@ function findRawTextElements(haystack, tagName) {
|
|
|
173
170
|
* Collect the class names and IDs a document references.
|
|
174
171
|
*
|
|
175
172
|
* Style sheet contents are excluded, so that a style sheet never counts as
|
|
176
|
-
* evidence for its own selectors.
|
|
177
|
-
*
|
|
173
|
+
* evidence for its own selectors. An `iframe srcdoc` value is not scanned
|
|
174
|
+
* either: It holds a document of its own, minified against its own symbol set.
|
|
175
|
+
* Over-collecting is safe here (a symbol wrongly considered used is merely
|
|
176
|
+
* kept), under-collecting is not.
|
|
178
177
|
*
|
|
179
178
|
* @param {string} html - Raw document markup
|
|
180
179
|
* @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
|
|
181
180
|
* @param {((text: string) => string)} [decode] - Resolves character references in attribute values
|
|
182
|
-
* @param {number} [depth] - Nesting level, counted through `srcdoc`
|
|
183
181
|
* @returns {Set<string>} Symbols to keep
|
|
184
182
|
*/
|
|
185
|
-
function collectUsedSymbols(html, includeScripts, decode
|
|
183
|
+
function collectUsedSymbols(html, includeScripts, decode) {
|
|
186
184
|
const used = new Set();
|
|
187
185
|
const haystack = foldCase(html);
|
|
188
186
|
|
|
@@ -222,9 +220,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
|
|
|
222
220
|
return element !== undefined && index >= element.start;
|
|
223
221
|
};
|
|
224
222
|
|
|
225
|
-
/** @type {string[]} */
|
|
226
|
-
const nested = [];
|
|
227
|
-
|
|
228
223
|
attributePattern.lastIndex = 0;
|
|
229
224
|
let match;
|
|
230
225
|
while ((match = attributePattern.exec(html))) {
|
|
@@ -244,8 +239,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
|
|
|
244
239
|
} else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
|
|
245
240
|
// Only a leading `#`: `href="/page#sec"` names a section of another document
|
|
246
241
|
addTokens(value.slice(1));
|
|
247
|
-
} else if (name === 'srcdoc' && depth < SRCDOC_MAX_DEPTH) {
|
|
248
|
-
nested.push(value);
|
|
249
242
|
}
|
|
250
243
|
if (value.indexOf('(') !== -1) {
|
|
251
244
|
fragmentURLPattern.lastIndex = 0;
|
|
@@ -256,16 +249,6 @@ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
|
|
|
256
249
|
}
|
|
257
250
|
}
|
|
258
251
|
|
|
259
|
-
// `iframe srcdoc` holds a document of its own, whose style sheets are minified
|
|
260
|
-
// against this very set—so what it references has to be in it. The scan runs here
|
|
261
|
-
// rather than at the attribute: The patterns it shares with this one are
|
|
262
|
-
// module-level, so no nested scan may start while one of their loops is open.
|
|
263
|
-
for (const markup of nested) {
|
|
264
|
-
for (const symbol of collectUsedSymbols(markup, includeScripts, decode, depth + 1)) {
|
|
265
|
-
used.add(symbol);
|
|
266
|
-
}
|
|
267
|
-
}
|
|
268
|
-
|
|
269
252
|
if (includeScripts) {
|
|
270
253
|
// Script contents are raw text, so character references stay literal;
|
|
271
254
|
// an unclosed `script` runs to the end of the document, as it does in a browser
|