html-minifier-next 7.5.3 → 7.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,394 @@
1
+ // Unused-CSS removal
2
+
3
+ import { findTagEnd } from './utils.js';
4
+
5
+ // Attributes whose values name elements by ID or hold space-separated ID lists
6
+ const idReferenceAttributes = new Set([
7
+ 'aria-activedescendant',
8
+ 'aria-controls',
9
+ 'aria-describedby',
10
+ 'aria-details',
11
+ 'aria-errormessage',
12
+ 'aria-flowto',
13
+ 'aria-labelledby',
14
+ 'aria-owns',
15
+ 'commandfor',
16
+ 'contextmenu',
17
+ 'for',
18
+ 'form',
19
+ 'headers',
20
+ 'itemref',
21
+ 'list',
22
+ 'popovertarget'
23
+ ]);
24
+
25
+ // Attributes whose value may be a same-document fragment URL (`#main`). `:target`
26
+ // rules and SVG sprite references (`<use href="#icon">`) rest on these, so the ID
27
+ // they name counts as used.
28
+ const fragmentReferenceAttributes = new Set([
29
+ 'action',
30
+ 'cite',
31
+ 'data',
32
+ 'formaction',
33
+ 'href',
34
+ 'poster',
35
+ 'src',
36
+ 'usemap',
37
+ 'xlink:href'
38
+ ]);
39
+
40
+ // `srcdoc` can nest, and each level costs another scan of its own markup
41
+ const SRCDOC_MAX_DEPTH = 3;
42
+
43
+ const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
44
+ const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
45
+ // SVG paints and filters reach elements by ID through `url(#gradient)`, in
46
+ // presentation attributes as well as in `style`
47
+ const fragmentURLPattern = /url\(\s*['"]?#([^)'"\s]+)/gi;
48
+ // Class names routinely carry characters that end a CSS identifier (`md:flex`,
49
+ // `w-1/2`, `p-[3px]`), so scripts and `data-*` values contribute whole tokens
50
+ // besides identifiers; quoted strings are where scripts keep such names
51
+ const stringLiteralPattern = /'((?:[^'\\\n]|\\.)*)'|"((?:[^"\\\n]|\\.)*)"|`((?:[^`\\]|\\.)*)`/g;
52
+
53
+ // CSS identifiers may contain escapes (`.md\:flex`, `.w-1\/2`), which have to be
54
+ // resolved before comparing them against the plain tokens found in the markup
55
+ const cssIdentifierPattern = /(?<![\w\\-])[.#]((?:[-_a-zA-Z]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])(?:[-\w]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])*)/g;
56
+ // `unusedSymbols` also drops `@keyframes` and `@counter-style` rules by name, so any
57
+ // name used there is off limits even when no element carries it as a class or ID
58
+ const reservedAtRulePattern = /@(?:-\w+-)?(?:keyframes|counter-style)\s+(-?[_a-zA-Z][\w-]*)/gi;
59
+ const escapePattern = /\\(?:([0-9a-fA-F]{1,6})[ \t\n]?|(.))/g;
60
+
61
+ /**
62
+ * Resolve CSS escape sequences in an identifier.
63
+ * @param {string} identifier
64
+ * @returns {string}
65
+ */
66
+ function unescapeIdentifier(identifier) {
67
+ if (identifier.indexOf('\\') === -1) {
68
+ return identifier;
69
+ }
70
+ return identifier.replace(escapePattern, (_match, hex, literal) => {
71
+ if (!hex) {
72
+ return literal;
73
+ }
74
+ // Per CSS Syntax spec, a null, surrogate, or out-of-range escape becomes U+FFFD
75
+ const code = parseInt(hex, 16);
76
+ return (code === 0 || code > 0x10FFFF || (code >= 0xD800 && code <= 0xDFFF))
77
+ ? '\uFFFD'
78
+ : String.fromCodePoint(code);
79
+ });
80
+ }
81
+
82
+ /**
83
+ * Lowercase for matching without disturbing offsets.
84
+ *
85
+ * `toLowerCase()` can change a string's length—U+0130 becomes two code units—which
86
+ * would misalign every offset taken from the result. Folding just ASCII preserves
87
+ * length, and tag names are ASCII anyway.
88
+ *
89
+ * @param {string} text
90
+ * @returns {string} Same length as `text`
91
+ */
92
+ function foldCase(text) {
93
+ const lowercased = text.toLowerCase();
94
+ return lowercased.length === text.length
95
+ ? lowercased
96
+ : text.replace(/[A-Z]+/g, uppercase => uppercase.toLowerCase());
97
+ }
98
+
99
+ /**
100
+ * Locate the bodies of a raw-text element (`style`, `script`).
101
+ *
102
+ * Scanning beats one regular expression here: A pattern permissive enough for the
103
+ * end tags browsers accept (`</script\t\n bar>`, `</script/>`) backtracks
104
+ * quadratically over a document full of near-matches, and bounding it would make
105
+ * long start tags go unrecognized.
106
+ *
107
+ * @param {string} haystack - Case-folded markup, as returned by `foldCase`
108
+ * @param {string} tagName - Lowercase element name
109
+ * @returns {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>}
110
+ */
111
+ function findRawTextElements(haystack, tagName) {
112
+ /** @type {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>} */
113
+ const found = [];
114
+ const openTag = '<' + tagName;
115
+ const closeTag = '</' + tagName;
116
+ // A tag name ends at whitespace, a slash, or the closing bracket—so `<styles>`
117
+ // and `</scriptfoo>` name different elements and must not match
118
+ const endsName = (/** @type {string} */ character) =>
119
+ character === '' || character === '/' || character === '>' || /\s/.test(character);
120
+
121
+ let cursor = 0;
122
+ for (;;) {
123
+ const start = haystack.indexOf(openTag, cursor);
124
+ if (start === -1) {
125
+ break;
126
+ }
127
+ if (!endsName(haystack.charAt(start + openTag.length))) {
128
+ cursor = start + openTag.length;
129
+ continue;
130
+ }
131
+ // A quoted attribute value may hold a `>`, so the tag ends where the parser
132
+ // says it does, not at the next bracket
133
+ const startTagEnd = findTagEnd(haystack, start + openTag.length);
134
+ if (startTagEnd === -1) {
135
+ break;
136
+ }
137
+
138
+ const bodyStart = startTagEnd + 1;
139
+ let search = bodyStart;
140
+ let bodyEnd = -1;
141
+ let end = -1;
142
+ for (;;) {
143
+ const candidate = haystack.indexOf(closeTag, search);
144
+ if (candidate === -1) {
145
+ break;
146
+ }
147
+ if (endsName(haystack.charAt(candidate + closeTag.length))) {
148
+ const closeEnd = findTagEnd(haystack, candidate + closeTag.length);
149
+ if (closeEnd !== -1) {
150
+ bodyEnd = candidate;
151
+ end = closeEnd + 1;
152
+ }
153
+ break;
154
+ }
155
+ search = candidate + closeTag.length;
156
+ }
157
+
158
+ const closed = bodyEnd !== -1;
159
+ found.push({
160
+ start,
161
+ bodyStart,
162
+ bodyEnd: closed ? bodyEnd : haystack.length,
163
+ end: closed ? end : haystack.length,
164
+ closed
165
+ });
166
+ cursor = closed ? end : haystack.length;
167
+ }
168
+
169
+ return found;
170
+ }
171
+
172
+ /**
173
+ * Collect the class names and IDs a document references.
174
+ *
175
+ * Style sheet contents are excluded, so that a style sheet never counts as
176
+ * evidence for its own selectors. Over-collecting is safe here (a symbol wrongly
177
+ * considered used is merely kept), under-collecting is not.
178
+ *
179
+ * @param {string} html - Raw document markup
180
+ * @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
181
+ * @param {((text: string) => string)} [decode] - Resolves character references in attribute values
182
+ * @param {number} [depth] - Nesting level, counted through `srcdoc`
183
+ * @returns {Set<string>} Symbols to keep
184
+ */
185
+ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
186
+ const used = new Set();
187
+ const haystack = foldCase(html);
188
+
189
+ // Style sheets are skipped rather than cut out, so both scans share one folded
190
+ // copy and offsets keep pointing into `html`. Only the body is skipped—the start
191
+ // tag carries ordinary attributes, and `<style id="theme">` could be what
192
+ // `#theme` refers to. An unclosed element is not skipped: Reading its contents
193
+ // as markup can only add symbols, whereas ignoring the rest of the document
194
+ // would lose them.
195
+ const skipped = findRawTextElements(haystack, 'style')
196
+ .filter(element => element.closed)
197
+ .map(element => ({ start: element.bodyStart, end: element.bodyEnd }));
198
+
199
+ const addIdentifiers = (/** @type {string} */ text) => {
200
+ identifierPattern.lastIndex = 0;
201
+ let identifier;
202
+ while ((identifier = identifierPattern.exec(text))) {
203
+ used.add(identifier[0]);
204
+ }
205
+ };
206
+
207
+ const addTokens = (/** @type {string} */ text) => {
208
+ for (const token of text.split(/\s+/)) {
209
+ if (token) {
210
+ used.add(token);
211
+ }
212
+ }
213
+ };
214
+
215
+ // Both loops below walk forward, so one cursor over the skipped ranges suffices
216
+ let skipCursor = 0;
217
+ const isSkipped = (/** @type {number} */ index) => {
218
+ while (skipCursor < skipped.length && (skipped[skipCursor]?.end ?? 0) <= index) {
219
+ skipCursor++;
220
+ }
221
+ const element = skipped[skipCursor];
222
+ return element !== undefined && index >= element.start;
223
+ };
224
+
225
+ /** @type {string[]} */
226
+ const nested = [];
227
+
228
+ attributePattern.lastIndex = 0;
229
+ let match;
230
+ while ((match = attributePattern.exec(html))) {
231
+ const raw = match[2] ?? match[3] ?? match[4] ?? '';
232
+ if (!raw || isSkipped(match.index)) {
233
+ continue;
234
+ }
235
+ const name = (match[1] ?? '').toLowerCase();
236
+ // `class="us&#101;d"` names the class `used`, so compare against the decoded value
237
+ const value = (decode && raw.indexOf('&') !== -1) ? decode(raw) : raw;
238
+ if (name === 'class' || name === 'id' || idReferenceAttributes.has(name)) {
239
+ addTokens(value);
240
+ } else if (name.startsWith('data-')) {
241
+ // Class names are commonly parked in `data-*` attributes for scripts to apply later
242
+ addIdentifiers(value);
243
+ addTokens(value);
244
+ } else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
245
+ // Only a leading `#`: `href="/page#sec"` names a section of another document
246
+ addTokens(value.slice(1));
247
+ } else if (name === 'srcdoc' && depth < SRCDOC_MAX_DEPTH) {
248
+ nested.push(value);
249
+ }
250
+ if (value.indexOf('(') !== -1) {
251
+ fragmentURLPattern.lastIndex = 0;
252
+ let reference;
253
+ while ((reference = fragmentURLPattern.exec(value))) {
254
+ addTokens(reference[1] ?? '');
255
+ }
256
+ }
257
+ }
258
+
259
+ // `iframe srcdoc` holds a document of its own, whose style sheets are minified
260
+ // against this very set—so what it references has to be in it. The scan runs here
261
+ // rather than at the attribute: The patterns it shares with this one are
262
+ // module-level, so no nested scan may start while one of their loops is open.
263
+ for (const markup of nested) {
264
+ for (const symbol of collectUsedSymbols(markup, includeScripts, decode, depth + 1)) {
265
+ used.add(symbol);
266
+ }
267
+ }
268
+
269
+ if (includeScripts) {
270
+ // Script contents are raw text, so character references stay literal;
271
+ // an unclosed `script` runs to the end of the document, as it does in a browser
272
+ skipCursor = 0;
273
+ for (const element of findRawTextElements(haystack, 'script')) {
274
+ if (isSkipped(element.start)) {
275
+ continue;
276
+ }
277
+ const body = html.slice(element.bodyStart, element.bodyEnd);
278
+ addIdentifiers(body);
279
+ stringLiteralPattern.lastIndex = 0;
280
+ let literal;
281
+ while ((literal = stringLiteralPattern.exec(body))) {
282
+ addTokens(literal[1] ?? literal[2] ?? literal[3] ?? '');
283
+ }
284
+ }
285
+ }
286
+
287
+ return used;
288
+ }
289
+
290
+ /**
291
+ * Determine which class/ID symbols a style sheet defines but the document never references.
292
+ * @param {string} css - Style sheet contents
293
+ * @param {Set<string>} used - Symbols the document references
294
+ * @param {Array<string | RegExp>} safelist - Symbols to keep regardless
295
+ * @returns {string[]} Symbols safe to remove
296
+ */
297
+ function findUnusedSymbols(css, used, safelist) {
298
+ const reserved = new Set();
299
+ reservedAtRulePattern.lastIndex = 0;
300
+ let match;
301
+ while ((match = reservedAtRulePattern.exec(css))) {
302
+ if (match[1]) {
303
+ reserved.add(match[1]);
304
+ }
305
+ }
306
+
307
+ const unused = [];
308
+ const seen = new Set();
309
+ cssIdentifierPattern.lastIndex = 0;
310
+ while ((match = cssIdentifierPattern.exec(css))) {
311
+ const symbol = unescapeIdentifier(match[1] ?? '');
312
+ if (seen.has(symbol)) {
313
+ continue;
314
+ }
315
+ seen.add(symbol);
316
+ if (used.has(symbol) || reserved.has(symbol)) {
317
+ continue;
318
+ }
319
+ if (safelist.some(entry => {
320
+ if (!(entry instanceof RegExp)) {
321
+ return entry === symbol;
322
+ }
323
+ // `test()` advances `lastIndex` on global and sticky patterns, which would
324
+ // make a safelist entry match only every other symbol
325
+ entry.lastIndex = 0;
326
+ return entry.test(symbol);
327
+ })) {
328
+ continue;
329
+ }
330
+ unused.push(symbol);
331
+ }
332
+
333
+ return unused;
334
+ }
335
+
336
+ const unusedCSSKeys = new Set(['safelist', 'scripts']);
337
+
338
+ /**
339
+ * Normalize the `removeUnusedCSS` option into a settled configuration.
340
+ *
341
+ * A misspelled key or a safelist that is not an array would otherwise protect
342
+ * nothing, and that only surfaces as a missing rule much later—so every value
343
+ * that gets dropped is reported.
344
+ *
345
+ * @param {boolean | {safelist?: Array<string | RegExp>, scripts?: boolean} | undefined} option
346
+ * @param {(message: string) => unknown} [warn] - Receives one message per ignored value
347
+ * @returns {{safelist: Array<string | RegExp>, scripts: boolean} | null} Null when disabled
348
+ */
349
+ function normalizeUnusedCSSOptions(option, warn) {
350
+ if (!option) {
351
+ return null;
352
+ }
353
+ const report = warn ?? (() => {});
354
+ const config = /** @type {Record<string, any>} */ (typeof option === 'object' ? option : {});
355
+
356
+ for (const key of Object.keys(config)) {
357
+ if (!unusedCSSKeys.has(key)) {
358
+ report(`Ignoring unknown \`removeUnusedCSS\` key \`${key}\`—expected \`safelist\` or \`scripts\``);
359
+ }
360
+ }
361
+
362
+ /** @type {Array<string | RegExp>} */
363
+ let safelist = [];
364
+ if (config.safelist !== undefined) {
365
+ if (!Array.isArray(config.safelist)) {
366
+ report('Ignoring `removeUnusedCSS.safelist`—it takes an array of strings and regular expressions');
367
+ } else {
368
+ safelist = config.safelist.filter((/** @type {unknown} */ entry) => {
369
+ if (typeof entry === 'string' || entry instanceof RegExp) {
370
+ return true;
371
+ }
372
+ report(`Ignoring \`removeUnusedCSS.safelist\` entry of type ${typeof entry}—entries must be strings or regular expressions`);
373
+ return false;
374
+ });
375
+ }
376
+ }
377
+
378
+ if (config.scripts !== undefined && typeof config.scripts !== 'boolean') {
379
+ report('Ignoring `removeUnusedCSS.scripts`—it takes a boolean');
380
+ }
381
+
382
+ return {
383
+ safelist,
384
+ // Keeping identifiers seen in inline scripts costs a little of the reduction
385
+ // but avoids the most common breakage, so it is the default
386
+ scripts: typeof config.scripts === 'boolean' ? config.scripts : true
387
+ };
388
+ }
389
+
390
+ export {
391
+ collectUsedSymbols,
392
+ findUnusedSymbols,
393
+ normalizeUnusedCSSOptions
394
+ };
package/src/lib/utils.js CHANGED
@@ -146,8 +146,31 @@ function parseRegExp(value) {
146
146
 
147
147
  // Exports
148
148
 
149
+ /**
150
+ * Find the index of the `>` that closes an opening tag, correctly skipping
151
+ * over quoted attribute values (which may contain `>`).
152
+ * @param {string} html
153
+ * @param {number} pos - Start position (just after the tag name)
154
+ * @returns {number} Index of the closing `>`, or -1 if not found
155
+ */
156
+ function findTagEnd(html, pos) {
157
+ let i = pos;
158
+ while (i < html.length) {
159
+ const ch = html[i];
160
+ if (ch === '>') return i;
161
+ if (ch === '"' || ch === "'") {
162
+ const q = ch;
163
+ i++;
164
+ while (i < html.length && html[i] !== q) i++;
165
+ }
166
+ i++;
167
+ }
168
+ return -1;
169
+ }
170
+
149
171
  export {
150
172
  stableStringify,
173
+ findTagEnd,
151
174
  LRU,
152
175
  MAX_CACHE_ENTRY_SIZE,
153
176
  hashContent,