html-minifier-next 7.5.3 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,377 @@
1
+ // Unused-CSS removal
2
+
3
+ import { findTagEnd } from './utils.js';
4
+
5
+ // Attributes whose values name elements by ID or hold space-separated ID lists
6
+ const idReferenceAttributes = new Set([
7
+ 'aria-activedescendant',
8
+ 'aria-controls',
9
+ 'aria-describedby',
10
+ 'aria-details',
11
+ 'aria-errormessage',
12
+ 'aria-flowto',
13
+ 'aria-labelledby',
14
+ 'aria-owns',
15
+ 'commandfor',
16
+ 'contextmenu',
17
+ 'for',
18
+ 'form',
19
+ 'headers',
20
+ 'itemref',
21
+ 'list',
22
+ 'popovertarget'
23
+ ]);
24
+
25
+ // Attributes whose value may be a same-document fragment URL (`#main`). `:target`
26
+ // rules and SVG sprite references (`<use href="#icon">`) rest on these, so the ID
27
+ // they name counts as used.
28
+ const fragmentReferenceAttributes = new Set([
29
+ 'action',
30
+ 'cite',
31
+ 'data',
32
+ 'formaction',
33
+ 'href',
34
+ 'poster',
35
+ 'src',
36
+ 'usemap',
37
+ 'xlink:href'
38
+ ]);
39
+
40
+ const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
41
+ const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
42
+ // SVG paints and filters reach elements by ID through `url(#gradient)`, in
43
+ // presentation attributes as well as in `style`
44
+ const fragmentURLPattern = /url\(\s*['"]?#([^)'"\s]+)/gi;
45
+ // Class names routinely carry characters that end a CSS identifier (`md:flex`,
46
+ // `w-1/2`, `p-[3px]`), so scripts and `data-*` values contribute whole tokens
47
+ // besides identifiers; quoted strings are where scripts keep such names
48
+ const stringLiteralPattern = /'((?:[^'\\\n]|\\.)*)'|"((?:[^"\\\n]|\\.)*)"|`((?:[^`\\]|\\.)*)`/g;
49
+
50
+ // CSS identifiers may contain escapes (`.md\:flex`, `.w-1\/2`), which have to be
51
+ // resolved before comparing them against the plain tokens found in the markup
52
+ const cssIdentifierPattern = /(?<![\w\\-])[.#]((?:[-_a-zA-Z]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])(?:[-\w]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])*)/g;
53
+ // `unusedSymbols` also drops `@keyframes` and `@counter-style` rules by name, so any
54
+ // name used there is off limits even when no element carries it as a class or ID
55
+ const reservedAtRulePattern = /@(?:-\w+-)?(?:keyframes|counter-style)\s+(-?[_a-zA-Z][\w-]*)/gi;
56
+ const escapePattern = /\\(?:([0-9a-fA-F]{1,6})[ \t\n]?|(.))/g;
57
+
58
+ /**
59
+ * Resolve CSS escape sequences in an identifier.
60
+ * @param {string} identifier
61
+ * @returns {string}
62
+ */
63
+ function unescapeIdentifier(identifier) {
64
+ if (identifier.indexOf('\\') === -1) {
65
+ return identifier;
66
+ }
67
+ return identifier.replace(escapePattern, (_match, hex, literal) => {
68
+ if (!hex) {
69
+ return literal;
70
+ }
71
+ // Per CSS Syntax spec, a null, surrogate, or out-of-range escape becomes U+FFFD
72
+ const code = parseInt(hex, 16);
73
+ return (code === 0 || code > 0x10FFFF || (code >= 0xD800 && code <= 0xDFFF))
74
+ ? '\uFFFD'
75
+ : String.fromCodePoint(code);
76
+ });
77
+ }
78
+
79
+ /**
80
+ * Lowercase for matching without disturbing offsets.
81
+ *
82
+ * `toLowerCase()` can change a string's length—U+0130 becomes two code units—which
83
+ * would misalign every offset taken from the result. Folding just ASCII preserves
84
+ * length, and tag names are ASCII anyway.
85
+ *
86
+ * @param {string} text
87
+ * @returns {string} Same length as `text`
88
+ */
89
+ function foldCase(text) {
90
+ const lowercased = text.toLowerCase();
91
+ return lowercased.length === text.length
92
+ ? lowercased
93
+ : text.replace(/[A-Z]+/g, uppercase => uppercase.toLowerCase());
94
+ }
95
+
96
+ /**
97
+ * Locate the bodies of a raw-text element (`style`, `script`).
98
+ *
99
+ * Scanning beats one regular expression here: A pattern permissive enough for the
100
+ * end tags browsers accept (`</script\t\n bar>`, `</script/>`) backtracks
101
+ * quadratically over a document full of near-matches, and bounding it would make
102
+ * long start tags go unrecognized.
103
+ *
104
+ * @param {string} haystack - Case-folded markup, as returned by `foldCase`
105
+ * @param {string} tagName - Lowercase element name
106
+ * @returns {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>}
107
+ */
108
+ function findRawTextElements(haystack, tagName) {
109
+ /** @type {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>} */
110
+ const found = [];
111
+ const openTag = '<' + tagName;
112
+ const closeTag = '</' + tagName;
113
+ // A tag name ends at whitespace, a slash, or the closing bracket—so `<styles>`
114
+ // and `</scriptfoo>` name different elements and must not match
115
+ const endsName = (/** @type {string} */ character) =>
116
+ character === '' || character === '/' || character === '>' || /\s/.test(character);
117
+
118
+ let cursor = 0;
119
+ for (;;) {
120
+ const start = haystack.indexOf(openTag, cursor);
121
+ if (start === -1) {
122
+ break;
123
+ }
124
+ if (!endsName(haystack.charAt(start + openTag.length))) {
125
+ cursor = start + openTag.length;
126
+ continue;
127
+ }
128
+ // A quoted attribute value may hold a `>`, so the tag ends where the parser
129
+ // says it does, not at the next bracket
130
+ const startTagEnd = findTagEnd(haystack, start + openTag.length);
131
+ if (startTagEnd === -1) {
132
+ break;
133
+ }
134
+
135
+ const bodyStart = startTagEnd + 1;
136
+ let search = bodyStart;
137
+ let bodyEnd = -1;
138
+ let end = -1;
139
+ for (;;) {
140
+ const candidate = haystack.indexOf(closeTag, search);
141
+ if (candidate === -1) {
142
+ break;
143
+ }
144
+ if (endsName(haystack.charAt(candidate + closeTag.length))) {
145
+ const closeEnd = findTagEnd(haystack, candidate + closeTag.length);
146
+ if (closeEnd !== -1) {
147
+ bodyEnd = candidate;
148
+ end = closeEnd + 1;
149
+ }
150
+ break;
151
+ }
152
+ search = candidate + closeTag.length;
153
+ }
154
+
155
+ const closed = bodyEnd !== -1;
156
+ found.push({
157
+ start,
158
+ bodyStart,
159
+ bodyEnd: closed ? bodyEnd : haystack.length,
160
+ end: closed ? end : haystack.length,
161
+ closed
162
+ });
163
+ cursor = closed ? end : haystack.length;
164
+ }
165
+
166
+ return found;
167
+ }
168
+
169
+ /**
170
+ * Collect the class names and IDs a document references.
171
+ *
172
+ * Style sheet contents are excluded, so that a style sheet never counts as
173
+ * evidence for its own selectors. An `iframe srcdoc` value is not scanned
174
+ * either: It holds a document of its own, minified against its own symbol set.
175
+ * Over-collecting is safe here (a symbol wrongly considered used is merely
176
+ * kept), under-collecting is not.
177
+ *
178
+ * @param {string} html - Raw document markup
179
+ * @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
180
+ * @param {((text: string) => string)} [decode] - Resolves character references in attribute values
181
+ * @returns {Set<string>} Symbols to keep
182
+ */
183
+ function collectUsedSymbols(html, includeScripts, decode) {
184
+ const used = new Set();
185
+ const haystack = foldCase(html);
186
+
187
+ // Style sheets are skipped rather than cut out, so both scans share one folded
188
+ // copy and offsets keep pointing into `html`. Only the body is skipped—the start
189
+ // tag carries ordinary attributes, and `<style id="theme">` could be what
190
+ // `#theme` refers to. An unclosed element is not skipped: Reading its contents
191
+ // as markup can only add symbols, whereas ignoring the rest of the document
192
+ // would lose them.
193
+ const skipped = findRawTextElements(haystack, 'style')
194
+ .filter(element => element.closed)
195
+ .map(element => ({ start: element.bodyStart, end: element.bodyEnd }));
196
+
197
+ const addIdentifiers = (/** @type {string} */ text) => {
198
+ identifierPattern.lastIndex = 0;
199
+ let identifier;
200
+ while ((identifier = identifierPattern.exec(text))) {
201
+ used.add(identifier[0]);
202
+ }
203
+ };
204
+
205
+ const addTokens = (/** @type {string} */ text) => {
206
+ for (const token of text.split(/\s+/)) {
207
+ if (token) {
208
+ used.add(token);
209
+ }
210
+ }
211
+ };
212
+
213
+ // Both loops below walk forward, so one cursor over the skipped ranges suffices
214
+ let skipCursor = 0;
215
+ const isSkipped = (/** @type {number} */ index) => {
216
+ while (skipCursor < skipped.length && (skipped[skipCursor]?.end ?? 0) <= index) {
217
+ skipCursor++;
218
+ }
219
+ const element = skipped[skipCursor];
220
+ return element !== undefined && index >= element.start;
221
+ };
222
+
223
+ attributePattern.lastIndex = 0;
224
+ let match;
225
+ while ((match = attributePattern.exec(html))) {
226
+ const raw = match[2] ?? match[3] ?? match[4] ?? '';
227
+ if (!raw || isSkipped(match.index)) {
228
+ continue;
229
+ }
230
+ const name = (match[1] ?? '').toLowerCase();
231
+ // `class="us&#101;d"` names the class `used`, so compare against the decoded value
232
+ const value = (decode && raw.indexOf('&') !== -1) ? decode(raw) : raw;
233
+ if (name === 'class' || name === 'id' || idReferenceAttributes.has(name)) {
234
+ addTokens(value);
235
+ } else if (name.startsWith('data-')) {
236
+ // Class names are commonly parked in `data-*` attributes for scripts to apply later
237
+ addIdentifiers(value);
238
+ addTokens(value);
239
+ } else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
240
+ // Only a leading `#`: `href="/page#sec"` names a section of another document
241
+ addTokens(value.slice(1));
242
+ }
243
+ if (value.indexOf('(') !== -1) {
244
+ fragmentURLPattern.lastIndex = 0;
245
+ let reference;
246
+ while ((reference = fragmentURLPattern.exec(value))) {
247
+ addTokens(reference[1] ?? '');
248
+ }
249
+ }
250
+ }
251
+
252
+ if (includeScripts) {
253
+ // Script contents are raw text, so character references stay literal;
254
+ // an unclosed `script` runs to the end of the document, as it does in a browser
255
+ skipCursor = 0;
256
+ for (const element of findRawTextElements(haystack, 'script')) {
257
+ if (isSkipped(element.start)) {
258
+ continue;
259
+ }
260
+ const body = html.slice(element.bodyStart, element.bodyEnd);
261
+ addIdentifiers(body);
262
+ stringLiteralPattern.lastIndex = 0;
263
+ let literal;
264
+ while ((literal = stringLiteralPattern.exec(body))) {
265
+ addTokens(literal[1] ?? literal[2] ?? literal[3] ?? '');
266
+ }
267
+ }
268
+ }
269
+
270
+ return used;
271
+ }
272
+
273
+ /**
274
+ * Determine which class/ID symbols a style sheet defines but the document never references.
275
+ * @param {string} css - Style sheet contents
276
+ * @param {Set<string>} used - Symbols the document references
277
+ * @param {Array<string | RegExp>} safelist - Symbols to keep regardless
278
+ * @returns {string[]} Symbols safe to remove
279
+ */
280
+ function findUnusedSymbols(css, used, safelist) {
281
+ const reserved = new Set();
282
+ reservedAtRulePattern.lastIndex = 0;
283
+ let match;
284
+ while ((match = reservedAtRulePattern.exec(css))) {
285
+ if (match[1]) {
286
+ reserved.add(match[1]);
287
+ }
288
+ }
289
+
290
+ const unused = [];
291
+ const seen = new Set();
292
+ cssIdentifierPattern.lastIndex = 0;
293
+ while ((match = cssIdentifierPattern.exec(css))) {
294
+ const symbol = unescapeIdentifier(match[1] ?? '');
295
+ if (seen.has(symbol)) {
296
+ continue;
297
+ }
298
+ seen.add(symbol);
299
+ if (used.has(symbol) || reserved.has(symbol)) {
300
+ continue;
301
+ }
302
+ if (safelist.some(entry => {
303
+ if (!(entry instanceof RegExp)) {
304
+ return entry === symbol;
305
+ }
306
+ // `test()` advances `lastIndex` on global and sticky patterns, which would
307
+ // make a safelist entry match only every other symbol
308
+ entry.lastIndex = 0;
309
+ return entry.test(symbol);
310
+ })) {
311
+ continue;
312
+ }
313
+ unused.push(symbol);
314
+ }
315
+
316
+ return unused;
317
+ }
318
+
319
+ const unusedCSSKeys = new Set(['safelist', 'scripts']);
320
+
321
+ /**
322
+ * Normalize the `removeUnusedCSS` option into a settled configuration.
323
+ *
324
+ * A misspelled key or a safelist that is not an array would otherwise protect
325
+ * nothing, and that only surfaces as a missing rule much later—so every value
326
+ * that gets dropped is reported.
327
+ *
328
+ * @param {boolean | {safelist?: Array<string | RegExp>, scripts?: boolean} | undefined} option
329
+ * @param {(message: string) => unknown} [warn] - Receives one message per ignored value
330
+ * @returns {{safelist: Array<string | RegExp>, scripts: boolean} | null} Null when disabled
331
+ */
332
+ function normalizeUnusedCSSOptions(option, warn) {
333
+ if (!option) {
334
+ return null;
335
+ }
336
+ const report = warn ?? (() => {});
337
+ const config = /** @type {Record<string, any>} */ (typeof option === 'object' ? option : {});
338
+
339
+ for (const key of Object.keys(config)) {
340
+ if (!unusedCSSKeys.has(key)) {
341
+ report(`Ignoring unknown \`removeUnusedCSS\` key \`${key}\`—expected \`safelist\` or \`scripts\``);
342
+ }
343
+ }
344
+
345
+ /** @type {Array<string | RegExp>} */
346
+ let safelist = [];
347
+ if (config.safelist !== undefined) {
348
+ if (!Array.isArray(config.safelist)) {
349
+ report('Ignoring `removeUnusedCSS.safelist`—it takes an array of strings and regular expressions');
350
+ } else {
351
+ safelist = config.safelist.filter((/** @type {unknown} */ entry) => {
352
+ if (typeof entry === 'string' || entry instanceof RegExp) {
353
+ return true;
354
+ }
355
+ report(`Ignoring \`removeUnusedCSS.safelist\` entry of type ${typeof entry}—entries must be strings or regular expressions`);
356
+ return false;
357
+ });
358
+ }
359
+ }
360
+
361
+ if (config.scripts !== undefined && typeof config.scripts !== 'boolean') {
362
+ report('Ignoring `removeUnusedCSS.scripts`—it takes a boolean');
363
+ }
364
+
365
+ return {
366
+ safelist,
367
+ // Keeping identifiers seen in inline scripts costs a little of the reduction
368
+ // but avoids the most common breakage, so it is the default
369
+ scripts: typeof config.scripts === 'boolean' ? config.scripts : true
370
+ };
371
+ }
372
+
373
+ export {
374
+ collectUsedSymbols,
375
+ findUnusedSymbols,
376
+ normalizeUnusedCSSOptions
377
+ };