html-minifier-next 7.5.2 → 7.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,6 +3,7 @@ import { LRU, MAX_CACHE_ENTRY_SIZE, stableStringify, hashContent, identity, lowe
3
3
  import { RE_TRAILING_SEMICOLON } from './constants.js';
4
4
  import { canCollapseWhitespace, canTrimWhitespace } from './whitespace.js';
5
5
  import { wrapCSS, unwrapCSS } from './content.js';
6
+ import { findUnusedSymbols, normalizeUnusedCSSOptions } from './unused-css.js';
6
7
  import { getPreset, getPresetNames } from '../presets.js';
7
8
  import { optionDefinitions, optionDefaults } from './option-definitions.js';
8
9
 
@@ -10,6 +11,21 @@ import { optionDefinitions, optionDefaults } from './option-definitions.js';
10
11
 
11
12
  // Type definitions
12
13
 
14
+ /**
15
+ * Per-document state handed to `minifyCSS`. Its closure hangs off the memoized
16
+ * options object that every `minify()` call with those options shares, so state
17
+ * belonging to one document has to be passed in rather than captured.
18
+ *
19
+ * @typedef {{usedSymbols?: Set<string>, warned: Set<string>}} CSSContext
20
+ */
21
+
22
+ /**
23
+ * Minified style sheet plus the warnings its transform produced, cached together
24
+ * so that a cache hit can report what the transform reported
25
+ *
26
+ * @typedef {{css: string, warnings: string[]}} CSSResult
27
+ */
28
+
13
29
  /**
14
30
  * Options object produced by `processOptions` and consumed by `minifyHTML` and
15
31
  * the `lib/` helpers; normalization guarantees that the function-valued options
@@ -17,16 +33,18 @@ import { optionDefinitions, optionDefaults } from './option-definitions.js';
17
33
  * minification adds writable internal state on top of the public options
18
34
  * (set on prototype-chain forks during SVG/MathML namespace transitions)
19
35
  *
20
- * @typedef {Omit<MinifierOptions, 'preset' | 'canCollapseWhitespace' | 'canTrimWhitespace' | 'ignoreCustomComments' | 'log' | 'minifyCSS' | 'minifyJS' | 'minifyURLs' | 'minifySVG'> & {
36
+ * @typedef {Omit<MinifierOptions, 'preset' | 'canCollapseWhitespace' | 'canTrimWhitespace' | 'ignoreCustomComments' | 'log' | 'minifyCSS' | 'minifyJS' | 'minifyURLs' | 'minifySVG' | 'removeUnusedCSS'> & {
21
37
  * name: (name: string) => string,
22
38
  * log: (message: any) => unknown,
23
39
  * ignoreCustomComments: RegExp[],
24
40
  * canCollapseWhitespace: (tag: string, attrs: HTMLAttribute[], defaultFn: (tag: string) => boolean) => boolean,
25
41
  * canTrimWhitespace: (tag: string, attrs: HTMLAttribute[], defaultFn: (tag: string) => boolean) => boolean,
26
- * minifyCSS: (text: string, type?: string) => string | Promise<string>,
42
+ * minifyCSS: (text: string, type?: string, context?: CSSContext) => string | Promise<string>,
27
43
  * minifyJS: (text: string, inline?: boolean, isModule?: boolean) => string | Promise<string>,
28
44
  * minifyURLs: (text: string) => string | Promise<string>,
29
45
  * minifySVG: ((svgContent: string) => string | Promise<string>) | null,
46
+ * removeUnusedCSS: {safelist: Array<string | RegExp>, scripts: boolean} | null,
47
+ * cssContext?: CSSContext,
30
48
  * nameParent?: (name: string) => string,
31
49
  * nameHTML?: (name: string) => string,
32
50
  * insideSVG?: boolean,
@@ -82,6 +100,9 @@ const presetNamesWarned = new Set();
82
100
  // The custom-fragment ReDoS warning is security-relevant, so it reaches the
83
101
  // console even without a `log` hook—once per process, like the warnings above
84
102
  let customFragmentQuantifierWarned = false;
103
+ const unusedCSSWarned = new Set();
104
+ // Object-valued options handed a string, warned about once per distinct value
105
+ const stringValuesWarned = new Set();
85
106
 
86
107
  // Main options processor
87
108
 
@@ -101,7 +122,8 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
101
122
  minifyCSS: identity,
102
123
  minifyJS: identity,
103
124
  minifyURLs: identity,
104
- minifySVG: null
125
+ minifySVG: null,
126
+ removeUnusedCSS: null
105
127
  };
106
128
 
107
129
  const parseRegExpArray = (/** @type {unknown} */ arr) => {
@@ -136,7 +158,7 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
136
158
  Object.keys(inputOptions).forEach(function (key) {
137
159
  if (!Object.hasOwn(optionDefinitions, key) && !optionKeysExtra.has(key) && !optionKeysWarned.has(key)) {
138
160
  optionKeysWarned.add(key);
139
- warn(`HTML Minifier Next: Ignoring unknown or deprecated option “${key}” (see README for available options)`);
161
+ warn(`HTML Minifier Next: Ignoring unknown or deprecated option \`${key}\` (see README for available options)`);
140
162
  }
141
163
  });
142
164
 
@@ -166,10 +188,30 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
166
188
  return;
167
189
  }
168
190
 
191
+ // A string carries no configuration for these options. The CLI parses config
192
+ // values as JSON first, so only a value that is not JSON reaches this from
193
+ // there. (`minifyURLs` is deliberately excluded—there, a string names the site.)
194
+ const definition = /** @type {Record<string, {type?: string}>} */ (optionDefinitions)[key];
195
+ if (typeof option === 'string' && definition?.type === 'jsonObject') {
196
+ const message = `HTML Minifier Next: Ignoring \`${key}\`—it takes a boolean or an object, not a string (“${option}”)`;
197
+ if (!stringValuesWarned.has(message)) {
198
+ stringValuesWarned.add(message);
199
+ warn(message);
200
+ }
201
+ return;
202
+ }
203
+
169
204
  if (key === 'caseSensitive') {
170
205
  if (option) {
171
206
  options.name = identity;
172
207
  }
208
+ } else if (key === 'removeUnusedCSS') {
209
+ optionsDynamic.removeUnusedCSS = normalizeUnusedCSSOptions(option, message => {
210
+ if (!unusedCSSWarned.has(message)) {
211
+ unusedCSSWarned.add(message);
212
+ warn(`HTML Minifier Next: ${message}`);
213
+ }
214
+ });
173
215
  } else if (key === 'log') {
174
216
  if (typeof option === 'function') {
175
217
  options.log = option;
@@ -184,12 +226,34 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
184
226
  const cssLoader = getLightningCSS;
185
227
  const cssCache = cssMinifyCache;
186
228
 
187
- options.minifyCSS = async function (/** @type {string} */ text, /** @type {string | undefined} */ type) {
229
+ options.minifyCSS = async function (/** @type {string} */ text, /** @type {string | undefined} */ type, /** @type {CSSContext | undefined} */ context) {
188
230
  // Fast path: Nothing to minify
189
231
  if (!text || !text.trim()) {
190
232
  return text;
191
233
  }
192
234
 
235
+ // Warnings are stored with the minified result and replayed on every hit, so a
236
+ // second document with the same defect hears about it, too; `context.warned`
237
+ // then keeps one document from repeating itself. Reporting from the cache
238
+ // rather than only from the transform keeps the output independent of cache
239
+ // size and eviction. They are built and cached even when `log` is the default
240
+ // no-op, so a later document that does pass a `log` hook still gets them.
241
+ const report = (/** @type {string[]} */ messages) => {
242
+ if (!messages.length || options.log === identity) {
243
+ return;
244
+ }
245
+ const warned = context?.warned;
246
+ for (const message of messages) {
247
+ if (warned) {
248
+ if (warned.has(message)) {
249
+ continue;
250
+ }
251
+ warned.add(message);
252
+ }
253
+ options.log(message);
254
+ }
255
+ };
256
+
193
257
  // Optimization: Only process URLs if minification is enabled (not identity function)
194
258
  // This avoids expensive `replaceAsync` when URL minification is disabled
195
259
  if (options.minifyURLs !== identity) {
@@ -213,9 +277,22 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
213
277
  );
214
278
  }
215
279
 
216
- // Cache key: Content + type + options signature; large inputs are hashed to avoid huge Map keys
280
+ // Unused-symbol removal applies to style sheets only
281
+ const unusedCSSConfig = type === undefined ? options.removeUnusedCSS : undefined;
282
+ const unusedSymbols = (unusedCSSConfig && context?.usedSymbols)
283
+ ? findUnusedSymbols(text, context.usedSymbols, unusedCSSConfig.safelist)
284
+ : undefined;
285
+
286
+ // Cache key: Content + type + options signature; large inputs are hashed to avoid huge Map keys.
287
+ // The symbol list belongs in the signature: The cache outlives a single `minify()` call, so
288
+ // identical style sheets in differently marked-up documents must not share an entry.
217
289
  const inputCSS = wrapCSS(text, type);
218
- const cssSig = stableStringify({ type, opts: lightningCssOptions, cont: !!options.continueOnMinifyError });
290
+ const cssSig = stableStringify({
291
+ type,
292
+ opts: lightningCssOptions,
293
+ cont: !!options.continueOnMinifyError,
294
+ unused: unusedSymbols && unusedSymbols.length ? unusedSymbols.slice().sort() : undefined
295
+ });
219
296
  const isCacheable = inputCSS.length <= MAX_CACHE_ENTRY_SIZE;
220
297
  const cssKey = isCacheable
221
298
  ? (inputCSS.length > 2048
@@ -225,10 +302,12 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
225
302
 
226
303
  try {
227
304
  if (cssKey !== undefined) {
228
- const cached = /** @type {string | Promise<string> | undefined} */ (cssCache.get(cssKey));
305
+ const cached = /** @type {CSSResult | Promise<CSSResult> | undefined} */ (cssCache.get(cssKey));
229
306
  if (cached !== undefined) {
230
307
  // Support both resolved values and in-flight promises
231
- return await cached;
308
+ const settled = await cached;
309
+ report(settled.warnings);
310
+ return settled.css;
232
311
  }
233
312
  }
234
313
 
@@ -242,9 +321,26 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
242
321
  code: Buffer.from(inputCSS),
243
322
  minify: true,
244
323
  errorRecovery: !!options.continueOnMinifyError,
245
- ...lightningCssOptions
324
+ ...lightningCssOptions,
325
+ // Union, so that a manually supplied `unusedSymbols` list survives
326
+ ...(unusedSymbols && unusedSymbols.length
327
+ ? { unusedSymbols: lightningCssOptions.unusedSymbols ? [...new Set([...lightningCssOptions.unusedSymbols, ...unusedSymbols])] : unusedSymbols }
328
+ : {})
246
329
  });
247
330
 
331
+ // With `errorRecovery` enabled, Lightning CSS reports what it takes issue
332
+ // with instead of throwing—dropping the rule in some cases (`@property`
333
+ // with a bad `syntax`) and passing it through in others (an unknown
334
+ // at-rule), which is why the wording stops at “reported”
335
+ /** @type {string[]} */
336
+ const warnings = [];
337
+ if (result.warnings) {
338
+ for (const warning of result.warnings) {
339
+ const at = warning.loc ? ` (line ${warning.loc.line}, column ${warning.loc.column})` : '';
340
+ warnings.push(`Warning: Lightning CSS reported invalid CSS${at}: ${warning.message}`);
341
+ }
342
+ }
343
+
248
344
  const outputCSS = unwrapCSS(result.code.toString(), type);
249
345
 
250
346
  // If Lightning CSS removed significant content that looks like template syntax or UIDs, return original
@@ -259,13 +355,15 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
259
355
 
260
356
  // Preserve if output is empty and input had template syntax or UIDs
261
357
  // This catches cases where Lightning CSS removed content that should be preserved
262
- return (text.trim() && !outputCSS.trim() && (looksLikeTemplate || hasUID)) ? text : outputCSS;
358
+ const css = (text.trim() && !outputCSS.trim() && (looksLikeTemplate || hasUID)) ? text : outputCSS;
359
+ return { css, warnings };
263
360
  })();
264
361
 
265
362
  if (cssKey !== undefined) cssCache.set(cssKey, inFlight);
266
363
  const resolved = await inFlight;
267
364
  if (cssKey !== undefined) cssCache.set(cssKey, resolved);
268
- return resolved;
365
+ report(resolved.warnings);
366
+ return resolved.css;
269
367
  } catch (err) {
270
368
  if (cssKey !== undefined) cssCache.delete(cssKey);
271
369
  if (!options.continueOnMinifyError) {
@@ -515,6 +613,23 @@ const processOptions = (inputOptions, { getLightningCSS, getTerser, getSwc, getS
515
613
  optionsDynamic[key] = option;
516
614
  }
517
615
  });
616
+
617
+ // Unused-CSS removal rides along with Lightning CSS, so it silently does nothing
618
+ // when `minifyCSS` is off or replaced by a function—say so rather than let it pass
619
+ if (options.removeUnusedCSS) {
620
+ const cssOption = /** @type {Record<string, any>} */ (effectiveInput).minifyCSS;
621
+ const reason = typeof cssOption === 'function'
622
+ ? 'it does not apply when `minifyCSS` is a function'
623
+ : (options.minifyCSS === identity ? 'it requires `minifyCSS` (`--minify-css`)' : '');
624
+ if (reason) {
625
+ if (!unusedCSSWarned.has(reason)) {
626
+ unusedCSSWarned.add(reason);
627
+ warn(`HTML Minifier Next: Ignoring \`removeUnusedCSS\`—${reason}`);
628
+ }
629
+ options.removeUnusedCSS = null;
630
+ }
631
+ }
632
+
518
633
  return options;
519
634
  };
520
635
 
@@ -0,0 +1,394 @@
1
+ // Unused-CSS removal
2
+
3
+ import { findTagEnd } from './utils.js';
4
+
5
+ // Attributes whose values name elements by ID or hold space-separated ID lists
6
+ const idReferenceAttributes = new Set([
7
+ 'aria-activedescendant',
8
+ 'aria-controls',
9
+ 'aria-describedby',
10
+ 'aria-details',
11
+ 'aria-errormessage',
12
+ 'aria-flowto',
13
+ 'aria-labelledby',
14
+ 'aria-owns',
15
+ 'commandfor',
16
+ 'contextmenu',
17
+ 'for',
18
+ 'form',
19
+ 'headers',
20
+ 'itemref',
21
+ 'list',
22
+ 'popovertarget'
23
+ ]);
24
+
25
+ // Attributes whose value may be a same-document fragment URL (`#main`). `:target`
26
+ // rules and SVG sprite references (`<use href="#icon">`) rest on these, so the ID
27
+ // they name counts as used.
28
+ const fragmentReferenceAttributes = new Set([
29
+ 'action',
30
+ 'cite',
31
+ 'data',
32
+ 'formaction',
33
+ 'href',
34
+ 'poster',
35
+ 'src',
36
+ 'usemap',
37
+ 'xlink:href'
38
+ ]);
39
+
40
+ // `srcdoc` can nest, and each level costs another scan of its own markup
41
+ const SRCDOC_MAX_DEPTH = 3;
42
+
43
+ const attributePattern = /(?:^|[\s/])([-\w:.]+)\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+))/g;
44
+ const identifierPattern = /--[\w-]*|-?[A-Za-z_][\w-]*/g;
45
+ // SVG paints and filters reach elements by ID through `url(#gradient)`, in
46
+ // presentation attributes as well as in `style`
47
+ const fragmentURLPattern = /url\(\s*['"]?#([^)'"\s]+)/gi;
48
+ // Class names routinely carry characters that end a CSS identifier (`md:flex`,
49
+ // `w-1/2`, `p-[3px]`), so scripts and `data-*` values contribute whole tokens
50
+ // besides identifiers; quoted strings are where scripts keep such names
51
+ const stringLiteralPattern = /'((?:[^'\\\n]|\\.)*)'|"((?:[^"\\\n]|\\.)*)"|`((?:[^`\\]|\\.)*)`/g;
52
+
53
+ // CSS identifiers may contain escapes (`.md\:flex`, `.w-1\/2`), which have to be
54
+ // resolved before comparing them against the plain tokens found in the markup
55
+ const cssIdentifierPattern = /(?<![\w\\-])[.#]((?:[-_a-zA-Z]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])(?:[-\w]|\\[0-9a-fA-F]{1,6}[ \t\n]?|\\[^\n0-9a-fA-F])*)/g;
56
+ // `unusedSymbols` also drops `@keyframes` and `@counter-style` rules by name, so any
57
+ // name used there is off limits even when no element carries it as a class or ID
58
+ const reservedAtRulePattern = /@(?:-\w+-)?(?:keyframes|counter-style)\s+(-?[_a-zA-Z][\w-]*)/gi;
59
+ const escapePattern = /\\(?:([0-9a-fA-F]{1,6})[ \t\n]?|(.))/g;
60
+
61
+ /**
62
+ * Resolve CSS escape sequences in an identifier.
63
+ * @param {string} identifier
64
+ * @returns {string}
65
+ */
66
+ function unescapeIdentifier(identifier) {
67
+ if (identifier.indexOf('\\') === -1) {
68
+ return identifier;
69
+ }
70
+ return identifier.replace(escapePattern, (_match, hex, literal) => {
71
+ if (!hex) {
72
+ return literal;
73
+ }
74
+ // Per CSS Syntax spec, a null, surrogate, or out-of-range escape becomes U+FFFD
75
+ const code = parseInt(hex, 16);
76
+ return (code === 0 || code > 0x10FFFF || (code >= 0xD800 && code <= 0xDFFF))
77
+ ? '\uFFFD'
78
+ : String.fromCodePoint(code);
79
+ });
80
+ }
81
+
82
+ /**
83
+ * Lowercase for matching without disturbing offsets.
84
+ *
85
+ * `toLowerCase()` can change a string's length—U+0130 becomes two code units—which
86
+ * would misalign every offset taken from the result. Folding just ASCII preserves
87
+ * length, and tag names are ASCII anyway.
88
+ *
89
+ * @param {string} text
90
+ * @returns {string} Same length as `text`
91
+ */
92
+ function foldCase(text) {
93
+ const lowercased = text.toLowerCase();
94
+ return lowercased.length === text.length
95
+ ? lowercased
96
+ : text.replace(/[A-Z]+/g, uppercase => uppercase.toLowerCase());
97
+ }
98
+
99
+ /**
100
+ * Locate the bodies of a raw-text element (`style`, `script`).
101
+ *
102
+ * Scanning beats one regular expression here: A pattern permissive enough for the
103
+ * end tags browsers accept (`</script\t\n bar>`, `</script/>`) backtracks
104
+ * quadratically over a document full of near-matches, and bounding it would make
105
+ * long start tags go unrecognized.
106
+ *
107
+ * @param {string} haystack - Case-folded markup, as returned by `foldCase`
108
+ * @param {string} tagName - Lowercase element name
109
+ * @returns {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>}
110
+ */
111
+ function findRawTextElements(haystack, tagName) {
112
+ /** @type {Array<{start: number, bodyStart: number, bodyEnd: number, end: number, closed: boolean}>} */
113
+ const found = [];
114
+ const openTag = '<' + tagName;
115
+ const closeTag = '</' + tagName;
116
+ // A tag name ends at whitespace, a slash, or the closing bracket—so `<styles>`
117
+ // and `</scriptfoo>` name different elements and must not match
118
+ const endsName = (/** @type {string} */ character) =>
119
+ character === '' || character === '/' || character === '>' || /\s/.test(character);
120
+
121
+ let cursor = 0;
122
+ for (;;) {
123
+ const start = haystack.indexOf(openTag, cursor);
124
+ if (start === -1) {
125
+ break;
126
+ }
127
+ if (!endsName(haystack.charAt(start + openTag.length))) {
128
+ cursor = start + openTag.length;
129
+ continue;
130
+ }
131
+ // A quoted attribute value may hold a `>`, so the tag ends where the parser
132
+ // says it does, not at the next bracket
133
+ const startTagEnd = findTagEnd(haystack, start + openTag.length);
134
+ if (startTagEnd === -1) {
135
+ break;
136
+ }
137
+
138
+ const bodyStart = startTagEnd + 1;
139
+ let search = bodyStart;
140
+ let bodyEnd = -1;
141
+ let end = -1;
142
+ for (;;) {
143
+ const candidate = haystack.indexOf(closeTag, search);
144
+ if (candidate === -1) {
145
+ break;
146
+ }
147
+ if (endsName(haystack.charAt(candidate + closeTag.length))) {
148
+ const closeEnd = findTagEnd(haystack, candidate + closeTag.length);
149
+ if (closeEnd !== -1) {
150
+ bodyEnd = candidate;
151
+ end = closeEnd + 1;
152
+ }
153
+ break;
154
+ }
155
+ search = candidate + closeTag.length;
156
+ }
157
+
158
+ const closed = bodyEnd !== -1;
159
+ found.push({
160
+ start,
161
+ bodyStart,
162
+ bodyEnd: closed ? bodyEnd : haystack.length,
163
+ end: closed ? end : haystack.length,
164
+ closed
165
+ });
166
+ cursor = closed ? end : haystack.length;
167
+ }
168
+
169
+ return found;
170
+ }
171
+
172
+ /**
173
+ * Collect the class names and IDs a document references.
174
+ *
175
+ * Style sheet contents are excluded, so that a style sheet never counts as
176
+ * evidence for its own selectors. Over-collecting is safe here (a symbol wrongly
177
+ * considered used is merely kept), under-collecting is not.
178
+ *
179
+ * @param {string} html - Raw document markup
180
+ * @param {boolean} includeScripts - Also treat identifiers inside inline `script` elements as used
181
+ * @param {((text: string) => string)} [decode] - Resolves character references in attribute values
182
+ * @param {number} [depth] - Nesting level, counted through `srcdoc`
183
+ * @returns {Set<string>} Symbols to keep
184
+ */
185
+ function collectUsedSymbols(html, includeScripts, decode, depth = 0) {
186
+ const used = new Set();
187
+ const haystack = foldCase(html);
188
+
189
+ // Style sheets are skipped rather than cut out, so both scans share one folded
190
+ // copy and offsets keep pointing into `html`. Only the body is skipped—the start
191
+ // tag carries ordinary attributes, and `<style id="theme">` could be what
192
+ // `#theme` refers to. An unclosed element is not skipped: Reading its contents
193
+ // as markup can only add symbols, whereas ignoring the rest of the document
194
+ // would lose them.
195
+ const skipped = findRawTextElements(haystack, 'style')
196
+ .filter(element => element.closed)
197
+ .map(element => ({ start: element.bodyStart, end: element.bodyEnd }));
198
+
199
+ const addIdentifiers = (/** @type {string} */ text) => {
200
+ identifierPattern.lastIndex = 0;
201
+ let identifier;
202
+ while ((identifier = identifierPattern.exec(text))) {
203
+ used.add(identifier[0]);
204
+ }
205
+ };
206
+
207
+ const addTokens = (/** @type {string} */ text) => {
208
+ for (const token of text.split(/\s+/)) {
209
+ if (token) {
210
+ used.add(token);
211
+ }
212
+ }
213
+ };
214
+
215
+ // Both loops below walk forward, so one cursor over the skipped ranges suffices
216
+ let skipCursor = 0;
217
+ const isSkipped = (/** @type {number} */ index) => {
218
+ while (skipCursor < skipped.length && (skipped[skipCursor]?.end ?? 0) <= index) {
219
+ skipCursor++;
220
+ }
221
+ const element = skipped[skipCursor];
222
+ return element !== undefined && index >= element.start;
223
+ };
224
+
225
+ /** @type {string[]} */
226
+ const nested = [];
227
+
228
+ attributePattern.lastIndex = 0;
229
+ let match;
230
+ while ((match = attributePattern.exec(html))) {
231
+ const raw = match[2] ?? match[3] ?? match[4] ?? '';
232
+ if (!raw || isSkipped(match.index)) {
233
+ continue;
234
+ }
235
+ const name = (match[1] ?? '').toLowerCase();
236
+ // `class="us&#101;d"` names the class `used`, so compare against the decoded value
237
+ const value = (decode && raw.indexOf('&') !== -1) ? decode(raw) : raw;
238
+ if (name === 'class' || name === 'id' || idReferenceAttributes.has(name)) {
239
+ addTokens(value);
240
+ } else if (name.startsWith('data-')) {
241
+ // Class names are commonly parked in `data-*` attributes for scripts to apply later
242
+ addIdentifiers(value);
243
+ addTokens(value);
244
+ } else if (fragmentReferenceAttributes.has(name) && value.charAt(0) === '#') {
245
+ // Only a leading `#`: `href="/page#sec"` names a section of another document
246
+ addTokens(value.slice(1));
247
+ } else if (name === 'srcdoc' && depth < SRCDOC_MAX_DEPTH) {
248
+ nested.push(value);
249
+ }
250
+ if (value.indexOf('(') !== -1) {
251
+ fragmentURLPattern.lastIndex = 0;
252
+ let reference;
253
+ while ((reference = fragmentURLPattern.exec(value))) {
254
+ addTokens(reference[1] ?? '');
255
+ }
256
+ }
257
+ }
258
+
259
+ // `iframe srcdoc` holds a document of its own, whose style sheets are minified
260
+ // against this very set—so what it references has to be in it. The scan runs here
261
+ // rather than at the attribute: The patterns it shares with this one are
262
+ // module-level, so no nested scan may start while one of their loops is open.
263
+ for (const markup of nested) {
264
+ for (const symbol of collectUsedSymbols(markup, includeScripts, decode, depth + 1)) {
265
+ used.add(symbol);
266
+ }
267
+ }
268
+
269
+ if (includeScripts) {
270
+ // Script contents are raw text, so character references stay literal;
271
+ // an unclosed `script` runs to the end of the document, as it does in a browser
272
+ skipCursor = 0;
273
+ for (const element of findRawTextElements(haystack, 'script')) {
274
+ if (isSkipped(element.start)) {
275
+ continue;
276
+ }
277
+ const body = html.slice(element.bodyStart, element.bodyEnd);
278
+ addIdentifiers(body);
279
+ stringLiteralPattern.lastIndex = 0;
280
+ let literal;
281
+ while ((literal = stringLiteralPattern.exec(body))) {
282
+ addTokens(literal[1] ?? literal[2] ?? literal[3] ?? '');
283
+ }
284
+ }
285
+ }
286
+
287
+ return used;
288
+ }
289
+
290
+ /**
291
+ * Determine which class/ID symbols a style sheet defines but the document never references.
292
+ * @param {string} css - Style sheet contents
293
+ * @param {Set<string>} used - Symbols the document references
294
+ * @param {Array<string | RegExp>} safelist - Symbols to keep regardless
295
+ * @returns {string[]} Symbols safe to remove
296
+ */
297
+ function findUnusedSymbols(css, used, safelist) {
298
+ const reserved = new Set();
299
+ reservedAtRulePattern.lastIndex = 0;
300
+ let match;
301
+ while ((match = reservedAtRulePattern.exec(css))) {
302
+ if (match[1]) {
303
+ reserved.add(match[1]);
304
+ }
305
+ }
306
+
307
+ const unused = [];
308
+ const seen = new Set();
309
+ cssIdentifierPattern.lastIndex = 0;
310
+ while ((match = cssIdentifierPattern.exec(css))) {
311
+ const symbol = unescapeIdentifier(match[1] ?? '');
312
+ if (seen.has(symbol)) {
313
+ continue;
314
+ }
315
+ seen.add(symbol);
316
+ if (used.has(symbol) || reserved.has(symbol)) {
317
+ continue;
318
+ }
319
+ if (safelist.some(entry => {
320
+ if (!(entry instanceof RegExp)) {
321
+ return entry === symbol;
322
+ }
323
+ // `test()` advances `lastIndex` on global and sticky patterns, which would
324
+ // make a safelist entry match only every other symbol
325
+ entry.lastIndex = 0;
326
+ return entry.test(symbol);
327
+ })) {
328
+ continue;
329
+ }
330
+ unused.push(symbol);
331
+ }
332
+
333
+ return unused;
334
+ }
335
+
336
+ const unusedCSSKeys = new Set(['safelist', 'scripts']);
337
+
338
+ /**
339
+ * Normalize the `removeUnusedCSS` option into a settled configuration.
340
+ *
341
+ * A misspelled key or a safelist that is not an array would otherwise protect
342
+ * nothing, and that only surfaces as a missing rule much later—so every value
343
+ * that gets dropped is reported.
344
+ *
345
+ * @param {boolean | {safelist?: Array<string | RegExp>, scripts?: boolean} | undefined} option
346
+ * @param {(message: string) => unknown} [warn] - Receives one message per ignored value
347
+ * @returns {{safelist: Array<string | RegExp>, scripts: boolean} | null} Null when disabled
348
+ */
349
+ function normalizeUnusedCSSOptions(option, warn) {
350
+ if (!option) {
351
+ return null;
352
+ }
353
+ const report = warn ?? (() => {});
354
+ const config = /** @type {Record<string, any>} */ (typeof option === 'object' ? option : {});
355
+
356
+ for (const key of Object.keys(config)) {
357
+ if (!unusedCSSKeys.has(key)) {
358
+ report(`Ignoring unknown \`removeUnusedCSS\` key \`${key}\`—expected \`safelist\` or \`scripts\``);
359
+ }
360
+ }
361
+
362
+ /** @type {Array<string | RegExp>} */
363
+ let safelist = [];
364
+ if (config.safelist !== undefined) {
365
+ if (!Array.isArray(config.safelist)) {
366
+ report('Ignoring `removeUnusedCSS.safelist`—it takes an array of strings and regular expressions');
367
+ } else {
368
+ safelist = config.safelist.filter((/** @type {unknown} */ entry) => {
369
+ if (typeof entry === 'string' || entry instanceof RegExp) {
370
+ return true;
371
+ }
372
+ report(`Ignoring \`removeUnusedCSS.safelist\` entry of type ${typeof entry}—entries must be strings or regular expressions`);
373
+ return false;
374
+ });
375
+ }
376
+ }
377
+
378
+ if (config.scripts !== undefined && typeof config.scripts !== 'boolean') {
379
+ report('Ignoring `removeUnusedCSS.scripts`—it takes a boolean');
380
+ }
381
+
382
+ return {
383
+ safelist,
384
+ // Keeping identifiers seen in inline scripts costs a little of the reduction
385
+ // but avoids the most common breakage, so it is the default
386
+ scripts: typeof config.scripts === 'boolean' ? config.scripts : true
387
+ };
388
+ }
389
+
390
+ export {
391
+ collectUsedSymbols,
392
+ findUnusedSymbols,
393
+ normalizeUnusedCSSOptions
394
+ };
package/src/lib/urls.js CHANGED
@@ -12,7 +12,7 @@ const REJECTED_SCHEMES = new Set(['data:', 'javascript:', 'mailto:']);
12
12
  const DIRECTORY_INDEXES = ['index.html', 'index.htm'];
13
13
 
14
14
  /**
15
- * Get the directory portion of a pathname (up to and including the last `/`)
15
+ * Get the directory portion of a pathname (up to and including the last `/`).
16
16
  * @param {string} pathname
17
17
  * @returns {string}
18
18
  */
@@ -22,7 +22,7 @@ function getDirectory(pathname) {
22
22
  }
23
23
 
24
24
  /**
25
- * Compute a path-relative URL from a base directory to a target path
25
+ * Compute a path-relative URL from a base directory to a target path.
26
26
  * @param {string} baseDir - Base directory path (must end with `/`)
27
27
  * @param {string} targetPath - Target pathname
28
28
  * @returns {string}
@@ -63,7 +63,7 @@ function relativize(baseDir, targetPath) {
63
63
  }
64
64
 
65
65
  /**
66
- * Remove directory index from the end of a pathname
66
+ * Remove directory index from the end of a pathname.
67
67
  * @param {string} pathname
68
68
  * @returns {string}
69
69
  */
@@ -77,7 +77,7 @@ function removeDirectoryIndex(pathname) {
77
77
  }
78
78
 
79
79
  /**
80
- * Create a URL minifier function for the given site context
80
+ * Create a URL minifier function for the given site context.
81
81
  * @param {string} site - The site base URL (used to compute relative URLs)
82
82
  * @returns {(url: string) => string} Minifier function that returns the shortest URL
83
83
  */