html-minifier-next 7.5.3 → 8.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +60 -24
- package/cli.js +178 -55
- package/dist/types/htmlminifier.d.ts +24 -4
- package/dist/types/htmlminifier.d.ts.map +1 -1
- package/dist/types/lib/attributes.d.ts +2 -2
- package/dist/types/lib/attributes.d.ts.map +1 -1
- package/dist/types/lib/constants.d.ts +3 -2
- package/dist/types/lib/constants.d.ts.map +1 -1
- package/dist/types/lib/elements.d.ts +1 -1
- package/dist/types/lib/fragments.d.ts +46 -0
- package/dist/types/lib/fragments.d.ts.map +1 -0
- package/dist/types/lib/option-definitions.d.ts +21 -194
- package/dist/types/lib/option-definitions.d.ts.map +1 -1
- package/dist/types/lib/options.d.ts +33 -5
- package/dist/types/lib/options.d.ts.map +1 -1
- package/dist/types/lib/unused-css.d.ts +43 -0
- package/dist/types/lib/unused-css.d.ts.map +1 -0
- package/dist/types/lib/utils.d.ts +26 -1
- package/dist/types/lib/utils.d.ts.map +1 -1
- package/dist/types/tokenchain.d.ts +2 -2
- package/dist/types/tokenchain.d.ts.map +1 -1
- package/html-minifier-next.schema.json +11 -8
- package/package.json +2 -2
- package/src/htmlminifier.js +86 -88
- package/src/htmlparser.js +4 -4
- package/src/lib/attributes.js +65 -8
- package/src/lib/constants.js +8 -8
- package/src/lib/elements.js +3 -3
- package/src/lib/fragments.js +274 -0
- package/src/lib/option-definitions.js +19 -7
- package/src/lib/options.js +151 -28
- package/src/lib/unused-css.js +377 -0
- package/src/lib/utils.js +330 -2
- package/src/tokenchain.js +37 -30
|
@@ -0,0 +1,377 @@
|
|
|
1
|
+
// Unused-CSS removal
|
|
2
|
+
|
|
3
|
+
import { findTagEnd } from './utils.js';
|
|
4
|
+
|
|
5
|
+
// Attributes whose values name elements by ID or hold space-separated ID lists
|
|
6
|
+
const idReferenceAttributes = new Set([
|
|
7
|
+
'aria-activedescendant',
|
|
8
|
+
'aria-controls',
|
|
9
|
+
'aria-describedby',
|
|
10
|
+
'aria-details',
|
|
11
|
+
'aria-errormessage',
|
|
12
|
+
'aria-flowto',
|
|
13
|
+
'aria-labelledby',
|
|
14
|
+
'aria-owns',
|
|
15
|
+
'commandfor',
|
|
16
|
+
'contextmenu',
|
|
17
|
+
'for',
|
|
18
|
+
'form',
|
|
19
|
+
'headers',
|
|
20
|
+
'itemref',
|
|
21
|
+
'list',
|
|
22
|
+
'popovertarget'
|
|
23
|
+
]);
|
|
24
|
+
|
|
25
|
+
// Attributes whose value may be a same-document fragment URL (`#main`). `:target`
|
|
26
|
+
// rules and SVG sprite references (`<use href="#icon">`) rest on these, so the ID
|
|
27
|
+
// they name counts as used.
|
|
28
|
+
const fragmentReferenceAttributes = new Set([
|
|
29
|
+
'action',
|
|
30
|
+
'cite',
|
|
31
|
+
'data',
|
|
32
|
+
'formaction',
|
|
33
|
+
'href',
|
|
34
|
+
'poster',
|
|
35
|
+
'src',
|
|
36
|
+
'usemap',
|
|
37
|
+
'xlink:href'
|
|
38
|
+
]);
|
|
39
|
+
|
|
40
|
+
const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
|
|
41
|
+
const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
|
|
42
|
+
// SVG paints and filters reach elements by ID through `url(#gradient)`, in
|
|
43
|
+
// presentation attributes as well as in `style`
|
|
44
|
+
const fragmentURLPattern = /url\(\s*['"]?#([^)'"\s]+)/gi;
|
|
45
|
+
// Class names routinely carry characters that end a CSS identifier (`md:flex`,
|
|
46
|
+
// `w-1/2`, `p-[3px]`), so scripts and `data-*` values contribute whole tokens
|
|
47
|
+
// besides identifiers; quoted strings are where scripts keep such names
|
|
48
|
+
const stringLiteralPattern = /'((?:[^'\\\n]|\\.)*)'|"((?:[^"\\\n]|\\.)*)"|`((?:[^`\\]|\\.)*)`/g;
|
|
49
|
+
|
|
50
|
+
// CSS identifiers may contain escapes (`.md\:flex`, `.w-1\/2`), which have to be
|
|
51
|
+
// resolved before comparing them against the plain tokens found in the markup
|
|
52
|
+
const cssIdentifierPattern = /(?<![\w\\-])[.#]((?:[-_a-zA-Z]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])(?:[-\w]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])*)/g;
|
|
53
|
+
// `unusedSymbols` also drops `@keyframes` and `@counter-style` rules by name, so any
|
|
54
|
+
// name used there is off limits even when no element carries it as a class or ID
|
|
55
|
+
const reservedAtRulePattern = /@(?:-\w+-)?(?:keyframes|counter-style)\s+(-?[_a-zA-Z][\w-]*)/gi;
|
|
56
|
+
const escapePattern = /\\(?:([0-9a-fA-F]{1,6})[ \t\n]?|(.))/g;
|
|
57
|
+
|
|
58
|
+
/**
|
|
59
|
+
* Resolve CSS escape sequences in an identifier.
|
|
60
|
+
* @param {string} identifier
|
|
61
|
+
* @returns {string}
|
|
62
|
+
*/
|
|
63
|
+
function unescapeIdentifier(identifier) {
|
|
64
|
+
if (identifier.indexOf('\\') === -1) {
|
|
65
|
+
return identifier;
|
|
66
|
+
}
|
|
67
|
+
return identifier.replace(escapePattern, (_match, hex, literal) => {
|
|
68
|
+
if (!hex) {
|
|
69
|
+
return literal;
|
|
70
|
+
}
|
|
71
|
+
// Per CSS Syntax spec, a null, surrogate, or out-of-range escape becomes U+FFFD
|
|
72
|
+
const code = parseInt(hex, 16);
|
|
73
|
+
return (code === 0 || code > 0x10FFFF || (code >= 0xD800 && code <= 0xDFFF))
|
|
74
|
+
? '\uFFFD'
|
|
75
|
+
: String.fromCodePoint(code);
|
|
76
|
+
});
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/**
|
|
80
|
+
* Lowercase for matching without disturbing offsets.
|
|
81
|
+
*
|
|
82
|
+
* `toLowerCase()` can change a string's length—U+0130 becomes two code units—which
|
|
83
|
+
* would misalign every offset taken from the result. Folding just ASCII preserves
|
|
84
|
+
* length, and tag names are ASCII anyway.
|
|
85
|
+
*
|
|
86
|
+
* @param {string} text
|
|
87
|
+
* @returns {string} Same length as `text`
|
|
88
|
+
*/
|
|
89
|
+
function foldCase(text) {
|
|
90
|
+
const lowercased = text.toLowerCase();
|
|
91
|
+
return lowercased.length === text.length
|
|
92
|
+
? lowercased
|
|
93
|
+
: text.replace(/[A-Z]+/g, uppercase => uppercase.toLowerCase());
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Locate the bodies of a raw-text element (`style`, `script`).
|
|
98
|
+
*
|
|
99
|
+
* Scanning beats one regular expression here: A pattern permissive enough for the
|
|
100
|
+
* end tags browsers accept (`</script\t\n bar>`, `</script/>`) backtracks
|
|
101
|
+
* quadratically over a document full of near-matches, and bounding it would make
|
|
102
|
+
* long start tags go unrecognized.
|
|
103
|
+
*
|
|
104
|
+
* @param {string} haystack - Case-folded markup, as returned by `foldCase`
|
|
105
|
+
* @param {string} tagName - Lowercase element name
|
|
106
|
+
* @returns {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>}
|
|
107
|
+
*/
|
|
108
|
+
function findRawTextElements(haystack, tagName) {
|
|
109
|
+
/** @type {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>} */
|
|
110
|
+
const found = [];
|
|
111
|
+
const openTag = '<' + tagName;
|
|
112
|
+
const closeTag = '</' + tagName;
|
|
113
|
+
// A tag name ends at whitespace, a slash, or the closing bracket—so `<styles>`
|
|
114
|
+
// and `</scriptfoo>` name different elements and must not match
|
|
115
|
+
const endsName = (/** @type {string} */ character) =>
|
|
116
|
+
character === '' || character === '/' || character === '>' || /\s/.test(character);
|
|
117
|
+
|
|
118
|
+
let cursor = 0;
|
|
119
|
+
for (;;) {
|
|
120
|
+
const start = haystack.indexOf(openTag, cursor);
|
|
121
|
+
if (start === -1) {
|
|
122
|
+
break;
|
|
123
|
+
}
|
|
124
|
+
if (!endsName(haystack.charAt(start + openTag.length))) {
|
|
125
|
+
cursor = start + openTag.length;
|
|
126
|
+
continue;
|
|
127
|
+
}
|
|
128
|
+
// A quoted attribute value may hold a `>`, so the tag ends where the parser
|
|
129
|
+
// says it does, not at the next bracket
|
|
130
|
+
const startTagEnd = findTagEnd(haystack, start + openTag.length);
|
|
131
|
+
if (startTagEnd === -1) {
|
|
132
|
+
break;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
const bodyStart = startTagEnd + 1;
|
|
136
|
+
let search = bodyStart;
|
|
137
|
+
let bodyEnd = -1;
|
|
138
|
+
let end = -1;
|
|
139
|
+
for (;;) {
|
|
140
|
+
const candidate = haystack.indexOf(closeTag, search);
|
|
141
|
+
if (candidate === -1) {
|
|
142
|
+
break;
|
|
143
|
+
}
|
|
144
|
+
if (endsName(haystack.charAt(candidate + closeTag.length))) {
|
|
145
|
+
const closeEnd = findTagEnd(haystack, candidate + closeTag.length);
|
|
146
|
+
if (closeEnd !== -1) {
|
|
147
|
+
bodyEnd = candidate;
|
|
148
|
+
end = closeEnd + 1;
|
|
149
|
+
}
|
|
150
|
+
break;
|
|
151
|
+
}
|
|
152
|
+
search = candidate + closeTag.length;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
const closed = bodyEnd !== -1;
|
|
156
|
+
found.push({
|
|
157
|
+
start,
|
|
158
|
+
bodyStart,
|
|
159
|
+
bodyEnd: closed ? bodyEnd : haystack.length,
|
|
160
|
+
end: closed ? end : haystack.length,
|
|
161
|
+
closed
|
|
162
|
+
});
|
|
163
|
+
cursor = closed ? end : haystack.length;
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
return found;
|
|
167
|
+
}
|
|
168
|
+
|
|
169
|
+
/**
|
|
170
|
+
* Collect the class names and IDs a document references.
|
|
171
|
+
*
|
|
172
|
+
* Style sheet contents are excluded, so that a style sheet never counts as
|
|
173
|
+
* evidence for its own selectors. An `iframe srcdoc` value is not scanned
|
|
174
|
+
* either: It holds a document of its own, minified against its own symbol set.
|
|
175
|
+
* Over-collecting is safe here (a symbol wrongly considered used is merely
|
|
176
|
+
* kept), under-collecting is not.
|
|
177
|
+
*
|
|
178
|
+
* @param {string} html - Raw document markup
|
|
179
|
+
* @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
|
|
180
|
+
* @param {((text: string) => string)} [decode] - Resolves character references in attribute values
|
|
181
|
+
* @returns {Set<string>} Symbols to keep
|
|
182
|
+
*/
|
|
183
|
+
function collectUsedSymbols(html, includeScripts, decode) {
|
|
184
|
+
const used = new Set();
|
|
185
|
+
const haystack = foldCase(html);
|
|
186
|
+
|
|
187
|
+
// Style sheets are skipped rather than cut out, so both scans share one folded
|
|
188
|
+
// copy and offsets keep pointing into `html`. Only the body is skipped—the start
|
|
189
|
+
// tag carries ordinary attributes, and `<style id="theme">` could be what
|
|
190
|
+
// `#theme` refers to. An unclosed element is not skipped: Reading its contents
|
|
191
|
+
// as markup can only add symbols, whereas ignoring the rest of the document
|
|
192
|
+
// would lose them.
|
|
193
|
+
const skipped = findRawTextElements(haystack, 'style')
|
|
194
|
+
.filter(element => element.closed)
|
|
195
|
+
.map(element => ({ start: element.bodyStart, end: element.bodyEnd }));
|
|
196
|
+
|
|
197
|
+
const addIdentifiers = (/** @type {string} */ text) => {
|
|
198
|
+
identifierPattern.lastIndex = 0;
|
|
199
|
+
let identifier;
|
|
200
|
+
while ((identifier = identifierPattern.exec(text))) {
|
|
201
|
+
used.add(identifier[0]);
|
|
202
|
+
}
|
|
203
|
+
};
|
|
204
|
+
|
|
205
|
+
const addTokens = (/** @type {string} */ text) => {
|
|
206
|
+
for (const token of text.split(/\s+/)) {
|
|
207
|
+
if (token) {
|
|
208
|
+
used.add(token);
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
};
|
|
212
|
+
|
|
213
|
+
// Both loops below walk forward, so one cursor over the skipped ranges suffices
|
|
214
|
+
let skipCursor = 0;
|
|
215
|
+
const isSkipped = (/** @type {number} */ index) => {
|
|
216
|
+
while (skipCursor < skipped.length && (skipped[skipCursor]?.end ?? 0) <= index) {
|
|
217
|
+
skipCursor++;
|
|
218
|
+
}
|
|
219
|
+
const element = skipped[skipCursor];
|
|
220
|
+
return element !== undefined && index >= element.start;
|
|
221
|
+
};
|
|
222
|
+
|
|
223
|
+
attributePattern.lastIndex = 0;
|
|
224
|
+
let match;
|
|
225
|
+
while ((match = attributePattern.exec(html))) {
|
|
226
|
+
const raw = match[2] ?? match[3] ?? match[4] ?? '';
|
|
227
|
+
if (!raw || isSkipped(match.index)) {
|
|
228
|
+
continue;
|
|
229
|
+
}
|
|
230
|
+
const name = (match[1] ?? '').toLowerCase();
|
|
231
|
+
// `class="used"` names the class `used`, so compare against the decoded value
|
|
232
|
+
const value = (decode && raw.indexOf('&') !== -1) ? decode(raw) : raw;
|
|
233
|
+
if (name === 'class' || name === 'id' || idReferenceAttributes.has(name)) {
|
|
234
|
+
addTokens(value);
|
|
235
|
+
} else if (name.startsWith('data-')) {
|
|
236
|
+
// Class names are commonly parked in `data-*` attributes for scripts to apply later
|
|
237
|
+
addIdentifiers(value);
|
|
238
|
+
addTokens(value);
|
|
239
|
+
} else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
|
|
240
|
+
// Only a leading `#`: `href="/page#sec"` names a section of another document
|
|
241
|
+
addTokens(value.slice(1));
|
|
242
|
+
}
|
|
243
|
+
if (value.indexOf('(') !== -1) {
|
|
244
|
+
fragmentURLPattern.lastIndex = 0;
|
|
245
|
+
let reference;
|
|
246
|
+
while ((reference = fragmentURLPattern.exec(value))) {
|
|
247
|
+
addTokens(reference[1] ?? '');
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
if (includeScripts) {
|
|
253
|
+
// Script contents are raw text, so character references stay literal;
|
|
254
|
+
// an unclosed `script` runs to the end of the document, as it does in a browser
|
|
255
|
+
skipCursor = 0;
|
|
256
|
+
for (const element of findRawTextElements(haystack, 'script')) {
|
|
257
|
+
if (isSkipped(element.start)) {
|
|
258
|
+
continue;
|
|
259
|
+
}
|
|
260
|
+
const body = html.slice(element.bodyStart, element.bodyEnd);
|
|
261
|
+
addIdentifiers(body);
|
|
262
|
+
stringLiteralPattern.lastIndex = 0;
|
|
263
|
+
let literal;
|
|
264
|
+
while ((literal = stringLiteralPattern.exec(body))) {
|
|
265
|
+
addTokens(literal[1] ?? literal[2] ?? literal[3] ?? '');
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
return used;
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Determine which class/ID symbols a style sheet defines but the document never references.
|
|
275
|
+
* @param {string} css - Style sheet contents
|
|
276
|
+
* @param {Set<string>} used - Symbols the document references
|
|
277
|
+
* @param {Array<string | RegExp>} safelist - Symbols to keep regardless
|
|
278
|
+
* @returns {string[]} Symbols safe to remove
|
|
279
|
+
*/
|
|
280
|
+
function findUnusedSymbols(css, used, safelist) {
|
|
281
|
+
const reserved = new Set();
|
|
282
|
+
reservedAtRulePattern.lastIndex = 0;
|
|
283
|
+
let match;
|
|
284
|
+
while ((match = reservedAtRulePattern.exec(css))) {
|
|
285
|
+
if (match[1]) {
|
|
286
|
+
reserved.add(match[1]);
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
|
|
290
|
+
const unused = [];
|
|
291
|
+
const seen = new Set();
|
|
292
|
+
cssIdentifierPattern.lastIndex = 0;
|
|
293
|
+
while ((match = cssIdentifierPattern.exec(css))) {
|
|
294
|
+
const symbol = unescapeIdentifier(match[1] ?? '');
|
|
295
|
+
if (seen.has(symbol)) {
|
|
296
|
+
continue;
|
|
297
|
+
}
|
|
298
|
+
seen.add(symbol);
|
|
299
|
+
if (used.has(symbol) || reserved.has(symbol)) {
|
|
300
|
+
continue;
|
|
301
|
+
}
|
|
302
|
+
if (safelist.some(entry => {
|
|
303
|
+
if (!(entry instanceof RegExp)) {
|
|
304
|
+
return entry === symbol;
|
|
305
|
+
}
|
|
306
|
+
// `test()` advances `lastIndex` on global and sticky patterns, which would
|
|
307
|
+
// make a safelist entry match only every other symbol
|
|
308
|
+
entry.lastIndex = 0;
|
|
309
|
+
return entry.test(symbol);
|
|
310
|
+
})) {
|
|
311
|
+
continue;
|
|
312
|
+
}
|
|
313
|
+
unused.push(symbol);
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
return unused;
|
|
317
|
+
}
|
|
318
|
+
|
|
319
|
+
const unusedCSSKeys = new Set(['safelist', 'scripts']);
|
|
320
|
+
|
|
321
|
+
/**
|
|
322
|
+
* Normalize the `removeUnusedCSS` option into a settled configuration.
|
|
323
|
+
*
|
|
324
|
+
* A misspelled key or a safelist that is not an array would otherwise protect
|
|
325
|
+
* nothing, and that only surfaces as a missing rule much later—so every value
|
|
326
|
+
* that gets dropped is reported.
|
|
327
|
+
*
|
|
328
|
+
* @param {boolean | {safelist?: Array<string | RegExp>, scripts?: boolean} | undefined} option
|
|
329
|
+
* @param {(message: string) => unknown} [warn] - Receives one message per ignored value
|
|
330
|
+
* @returns {{safelist: Array<string | RegExp>, scripts: boolean} | null} Null when disabled
|
|
331
|
+
*/
|
|
332
|
+
function normalizeUnusedCSSOptions(option, warn) {
|
|
333
|
+
if (!option) {
|
|
334
|
+
return null;
|
|
335
|
+
}
|
|
336
|
+
const report = warn ?? (() => {});
|
|
337
|
+
const config = /** @type {Record<string, any>} */ (typeof option === 'object' ? option : {});
|
|
338
|
+
|
|
339
|
+
for (const key of Object.keys(config)) {
|
|
340
|
+
if (!unusedCSSKeys.has(key)) {
|
|
341
|
+
report(`Ignoring unknown \`removeUnusedCSS\` key \`${key}\`—expected \`safelist\` or \`scripts\``);
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
/** @type {Array<string | RegExp>} */
|
|
346
|
+
let safelist = [];
|
|
347
|
+
if (config.safelist !== undefined) {
|
|
348
|
+
if (!Array.isArray(config.safelist)) {
|
|
349
|
+
report('Ignoring `removeUnusedCSS.safelist`—it takes an array of strings and regular expressions');
|
|
350
|
+
} else {
|
|
351
|
+
safelist = config.safelist.filter((/** @type {unknown} */ entry) => {
|
|
352
|
+
if (typeof entry === 'string' || entry instanceof RegExp) {
|
|
353
|
+
return true;
|
|
354
|
+
}
|
|
355
|
+
report(`Ignoring \`removeUnusedCSS.safelist\` entry of type ${typeof entry}—entries must be strings or regular expressions`);
|
|
356
|
+
return false;
|
|
357
|
+
});
|
|
358
|
+
}
|
|
359
|
+
}
|
|
360
|
+
|
|
361
|
+
if (config.scripts !== undefined && typeof config.scripts !== 'boolean') {
|
|
362
|
+
report('Ignoring `removeUnusedCSS.scripts`—it takes a boolean');
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
return {
|
|
366
|
+
safelist,
|
|
367
|
+
// Keeping identifiers seen in inline scripts costs a little of the reduction
|
|
368
|
+
// but avoids the most common breakage, so it is the default
|
|
369
|
+
scripts: typeof config.scripts === 'boolean' ? config.scripts : true
|
|
370
|
+
};
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
export {
|
|
374
|
+
collectUsedSymbols,
|
|
375
|
+
findUnusedSymbols,
|
|
376
|
+
normalizeUnusedCSSOptions
|
|
377
|
+
};
|