extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,432 @@
1
+ /**
2
+ * @module research/extractor/html-to-content/extract-content/extract-content-readability
3
+ * @description Research library module.
4
+ */
5
+ import { parseHTML } from "linkedom";
6
+
7
+ interface Candidate {
8
+ score: number;
9
+ elem: any; // linkedom Element
10
+ }
11
+
12
+ /**
13
+ * ### HTML-to-Main-Content Extractor #1
14
+ * The function extracts main content with regex patterns, cleaning HTML, scoring nodes
15
+ * based on content indicators like paragraphs and id/class names, selecting
16
+ * the top candidate, extracting it, and cleaning up content around it.
17
+ *
18
+ *
19
+ * 1. Define regular expressions:
20
+ * - Various regex patterns are defined to identify content and non-content areas.
21
+ *
22
+ * 2. Define helper functions:
23
+ * - normalizeSpaces: Normalizes whitespace in a string.
24
+ * - stripTags: Removes all HTML tags from a string.
25
+ * - getTextLength: Calculates the length of text after stripping tags.
26
+ * - calculateLinkDensity: Calculates the ratio of link text to total text.
27
+ *
28
+ * 3. Clean HTML:
29
+ * - Remove unlikely candidates (e.g., ads, sidebars) from the HTML.
30
+ *
31
+ * 4. Define scoring function:
32
+ * - scoreNode: Assigns a score to an HTML node based on content and attributes.
33
+ * - Increases score for positive indicators (e.g., article, body, content tags).
34
+ * - Decreases score for negative indicators (e.g., hidden, footer, sidebar tags).
35
+ * - Adds to score based on paragraph tags and text length.
36
+ *
37
+ * 5. Find and score candidate nodes:
38
+ * - Identify potential content nodes in the cleaned HTML.
39
+ * - Score each node using the scoreNode function.
40
+ *
41
+ * 6. Select top candidate:
42
+ * - Sort candidates by score and select the highest-scoring node.
43
+ *
44
+ * 7. Extract content:
45
+ * - Use regex to extract content around the top candidate node.
46
+ *
47
+ * 8. Clean up extracted content:
48
+ * - Remove script and style tags and their contents.
49
+ * - Process anchor tags based on content density.
50
+ * - Keep only specific HTML tags (a, p, img, h1-h6, ul, ol, li).
51
+ * - Remove excess whitespace from the final content.
52
+ *
53
+ * [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
54
+ *
55
+ * @example
56
+ * var url = "https://www.nytimes.com/2024/08/28/business/telegram-ceo-pavel-durov-charged.html"
57
+ * const html = await (await fetch(url)).text();
58
+ * var articleContent = extractMainContentFromHTML(html);
59
+ * @param {Object} [options]
60
+ * @param {number} options.minContentLength default=140 - Minimum length of content to be considered valid
61
+ * @param {number} options.minScore default=20 - Minimum score for content to be considered valid
62
+ * @param {number} options.minTextLength default=25 - Minimum length of text to be considered valid
63
+ * @param {number} options.retryLength default=250 - Length to retry content extraction if initial attempt fails
64
+ * @returns {string} Extracted HTML string of main content
65
+ * @author [vtempest (2025)](https://github.com/vtempest)
66
+ * Based on [Mozilla Readability (2015)](https://github.com/mozilla/readability)
67
+ * @category Extract
68
+ */
69
+ export function extractMainContentFromHTML(
70
+ html: string,
71
+ options: {
72
+ minContentLength?: number;
73
+ minScore?: number;
74
+ minTextLength?: number;
75
+ retryLength?: number;
76
+ } = {},
77
+ ): string {
78
+ const {
79
+ minContentLength = 140,
80
+ minScore = 20,
81
+ minTextLength = 25,
82
+ retryLength = 250,
83
+ } = options;
84
+
85
+ // Define regular expressions for content identification
86
+ const positiveRe =
87
+ /article|body|content|entry|hentry|main|page|pagination|post|text|blog|story/i;
88
+ const negativeRe =
89
+ /button|combx|comment|com-|contact|figure|foot|footer|footnote|form|input|masthead|media|meta|outbrain|promo|related|scroll|shoutbox|sidebar|sponsor|shopping|tags|tool|widget/i;
90
+ const videoRe = /https?:\/\/(?:www\.)?(?:youtube|vimeo)\.com/i;
91
+
92
+ // Parse the HTML string into a document
93
+ const doc = parseHTML(html)?.document;
94
+ if (!doc) return "";
95
+
96
+ // Remove script and style tags
97
+ doc.querySelectorAll("script, style").forEach((elem) => elem.remove());
98
+
99
+ // Clean HTML by removing unlikely candidates
100
+ for (const elem of doc.querySelectorAll("*")) {
101
+ const attrs =
102
+ (elem.getAttribute("class") || "") +
103
+ " " +
104
+ (elem.getAttribute("id") || "");
105
+ if (attrs.length < 2) continue;
106
+ //test unlikely candidates
107
+ if (
108
+ !["body", "html"].includes(elem.tagName.toLowerCase()) &&
109
+ /combx|comment|community|disqus|extra|foot|header|menu|remark|rss|shoutbox|sidebar|sponsor|ad-break|agegate|pagination|pager|popup|tweet|twitter/i.test(
110
+ attrs,
111
+ ) &&
112
+ !/and|article|body|column|main|shadow/i.test(attrs)
113
+ ) {
114
+ elem.remove();
115
+ }
116
+ }
117
+
118
+ // Convert divs to paragraphs if they don't contain block elements
119
+ const divs = doc.getElementsByTagName("div");
120
+ for (const elem of divs) {
121
+ if (
122
+ !/<(?:a|blockquote|dl|div|img|ol|p|pre|table|ul)/i.test(
123
+ elem?.innerHTML?.replace(/\s+/, " "),
124
+ )
125
+ ) {
126
+ const newElem = doc.createElement("p");
127
+ for (const attr of elem.attributes) {
128
+ newElem.setAttribute(attr.name, attr.value);
129
+ }
130
+ while (elem.firstChild) {
131
+ newElem.appendChild(elem.firstChild);
132
+ }
133
+ elem.parentNode?.replaceChild(newElem, elem);
134
+ }
135
+ }
136
+
137
+ // Score nodes
138
+ const candidates: Record<string, Candidate> = {};
139
+ const elems = Array.from(doc.querySelectorAll("p, pre, td"));
140
+
141
+ for (const elem of elems) {
142
+ const parentNode = elem.parentNode;
143
+ const grandParentNode = parentNode ? parentNode.parentNode : null;
144
+
145
+ const innerText = (elem.textContent || "").trim();
146
+ const innerTextLen = innerText.length;
147
+
148
+ if (innerTextLen < minTextLength) continue;
149
+
150
+ const pKey = String(parentNode);
151
+ const gpKey = String(grandParentNode);
152
+
153
+ // Score parent and grandparent nodes
154
+ if (!candidates[pKey]) {
155
+ candidates[pKey] = scoreNode(parentNode as any, positiveRe, negativeRe);
156
+ }
157
+ if (grandParentNode && !candidates[gpKey]) {
158
+ candidates[gpKey] = scoreNode(
159
+ grandParentNode as any,
160
+ positiveRe,
161
+ negativeRe,
162
+ );
163
+ }
164
+
165
+ // Calculate score based on text content
166
+ let score = 1;
167
+ score += innerText.split(",").length;
168
+ score += Math.min(innerTextLen / 100, 3);
169
+
170
+ candidates[pKey].score += score;
171
+ if (grandParentNode) candidates[gpKey].score += score / 2;
172
+ }
173
+
174
+ // Adjust scores based on link density
175
+ for (const candidate of Object.values(candidates)) {
176
+ if (candidate && candidate.elem) {
177
+ candidate.score *= 1 - getLinkDensity(candidate.elem);
178
+ }
179
+ }
180
+
181
+ // Find the best candidate
182
+ const sortedCandidates = Object.values(candidates).sort(
183
+ (a, b) => b.score - a.score,
184
+ );
185
+ const bestCandidate = sortedCandidates[0];
186
+
187
+ let article: any;
188
+ let cleanedArticle: any;
189
+
190
+ if (bestCandidate) {
191
+ // Extract content from the best candidate and its siblings
192
+ const siblingScoreThreshold = Math.max(10, bestCandidate.score * 0.2);
193
+ article = doc.createElement("div");
194
+ const parent = bestCandidate.elem.parentNode;
195
+ const siblings = parent
196
+ ? Array.from(parent.children)
197
+ : [bestCandidate.elem];
198
+
199
+ for (let sibling of siblings) {
200
+ let append = false;
201
+ if (
202
+ sibling === bestCandidate.elem ||
203
+ (candidates[String(sibling)] &&
204
+ candidates[String(sibling)].score >= siblingScoreThreshold)
205
+ ) {
206
+ append = true;
207
+ } else if (sibling.tagName === "P") {
208
+ const linkDensity = getLinkDensity(sibling as any);
209
+ const nodeContent = sibling.textContent || "";
210
+ const nodeLength = nodeContent.length;
211
+
212
+ if (
213
+ (nodeLength > 80 && linkDensity < 0.25) ||
214
+ (nodeLength <= 80 && linkDensity === 0 && /\.( |$)/.test(nodeContent))
215
+ ) {
216
+ append = true;
217
+ }
218
+ }
219
+
220
+ if (append) article.innerHTML += sibling.innerHTML;
221
+ }
222
+
223
+ cleanedArticle = sanitize(
224
+ article,
225
+ candidates,
226
+ videoRe,
227
+ positiveRe,
228
+ negativeRe,
229
+ minTextLength,
230
+ );
231
+ // var articleLength = cleanedArticle ? cleanedArticle.textContent.length : 0;
232
+ } else {
233
+ // If no best candidate, use the body or entire document
234
+ article = doc.querySelector("body") || doc;
235
+ cleanedArticle = sanitize(
236
+ article,
237
+ candidates,
238
+ videoRe,
239
+ positiveRe,
240
+ negativeRe,
241
+ minTextLength,
242
+ );
243
+ }
244
+
245
+ return cleanedArticle ? cleanedArticle.innerHTML : "";
246
+ }
247
+
248
+ /**
249
+ * Calculates the link density of an element.
250
+ * @param {Element} elem - The element to calculate link density for
251
+ * @returns {number} The link density (ratio of link text length to total text length)
252
+ */
253
+ export function getLinkDensity(elem: any): number {
254
+ if (!elem || !elem.textContent) {
255
+ return 0;
256
+ }
257
+ const links = elem.querySelectorAll("a");
258
+ const textLength = elem.textContent.trim().length;
259
+ const linkLength = (Array.from(links) as any[]).reduce(
260
+ (total: number, link: any) => total + link.textContent.trim().length,
261
+ 0,
262
+ );
263
+ return textLength > 0 ? linkLength / textLength : 0;
264
+ }
265
+
266
+ /**
267
+ * Calculates the weight of an element based on its class and id attributes.
268
+ * @param {Element} elem - The element to calculate weight for
269
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
270
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
271
+ * @returns {number} The calculated weight
272
+ */
273
+ export function classWeight(
274
+ elem: any,
275
+ positiveRe: RegExp,
276
+ negativeRe: RegExp,
277
+ ): number {
278
+ let weight = 0;
279
+ if (!elem || !elem.getAttribute) return weight;
280
+ if (elem.getAttribute("class")) {
281
+ if (negativeRe.test(elem.getAttribute("class"))) weight -= 25;
282
+ if (positiveRe.test(elem.getAttribute("class"))) weight += 25;
283
+ }
284
+ if (elem.getAttribute("id")) {
285
+ if (negativeRe.test(elem.getAttribute("id"))) weight -= 25;
286
+ if (positiveRe.test(elem.getAttribute("id"))) weight += 25;
287
+ }
288
+ return weight;
289
+ }
290
+
291
+ /**
292
+ * Scores a node based on its tag name and attributes.
293
+ * @param {Element} elem - The element to score
294
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
295
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
296
+ * @returns {Object} An object containing the score and the element
297
+ */
298
+ export function scoreNode(
299
+ elem: any,
300
+ positiveRe: RegExp,
301
+ negativeRe: RegExp,
302
+ ): Candidate {
303
+ if (!elem || !elem.tagName) return { score: 0, elem };
304
+ const DIV_SCORES = new Set(["div", "article"]);
305
+ const BLOCK_SCORES = new Set(["pre", "td", "blockquote"]);
306
+ const BAD_ELEM_SCORES = new Set([
307
+ "address",
308
+ "ol",
309
+ "ul",
310
+ "dl",
311
+ "dd",
312
+ "dt",
313
+ "li",
314
+ "form",
315
+ "aside",
316
+ ]);
317
+ const STRUCTURE_SCORES = new Set([
318
+ "h1",
319
+ "h2",
320
+ "h3",
321
+ "h4",
322
+ "h5",
323
+ "h6",
324
+ "th",
325
+ "header",
326
+ "footer",
327
+ "nav",
328
+ ]);
329
+
330
+ let score = classWeight(elem, positiveRe, negativeRe);
331
+ const name = elem.tagName.toLowerCase();
332
+ if (DIV_SCORES.has(name)) score += 5;
333
+ else if (BLOCK_SCORES.has(name)) score += 3;
334
+ else if (BAD_ELEM_SCORES.has(name)) score -= 3;
335
+ else if (STRUCTURE_SCORES.has(name)) score -= 5;
336
+ return { score, elem };
337
+ }
338
+
339
+ /**
340
+ * Sanitizes the content by removing unwanted elements and cleaning remaining elements.
341
+ * @param {Element} node - The node to sanitize
342
+ * @param {Object} candidates - Object containing scored candidates
343
+ * @param {RegExp} videoRe - Regular expression for video URLs
344
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
345
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
346
+ * @param {number} minTextLength - Minimum text length to consider
347
+ * @returns {Element} The sanitized node
348
+ */
349
+ export function sanitize(
350
+ node: any,
351
+ candidates: Record<string, Candidate>,
352
+ videoRe: RegExp,
353
+ positiveRe: RegExp,
354
+ negativeRe: RegExp,
355
+ minTextLength: number,
356
+ ): any {
357
+ const DIV_TO_P_ELEMS = new Set([
358
+ "a",
359
+ "blockquote",
360
+ "dl",
361
+ "div",
362
+ "img",
363
+ "ol",
364
+ "p",
365
+ "pre",
366
+ "table",
367
+ "ul",
368
+ ]);
369
+
370
+ // Remove unwanted elements
371
+ for (let elem of node.querySelectorAll(
372
+ "h1, h2, h3, h4, h5, h6, form, textarea, iframe",
373
+ )) {
374
+ if (elem.tagName === "IFRAME" && videoRe.test(elem.src)) {
375
+ elem.textContent = "VIDEO";
376
+ } else {
377
+ elem.remove();
378
+ }
379
+ }
380
+
381
+ // Clean remaining elements
382
+ const allowed = new Set();
383
+ for (let elem of (
384
+ Array.from(
385
+ node.querySelectorAll("table, ul, div, aside, header, footer, section"),
386
+ ) as any[]
387
+ ).reverse()) {
388
+ if (allowed.has(elem)) continue;
389
+
390
+ const weight = classWeight(elem, positiveRe, negativeRe);
391
+ const score = candidates[String(elem)] ? candidates[String(elem)].score : 0;
392
+
393
+ if (weight + score < 0) {
394
+ elem.remove();
395
+ } else if (elem.textContent.split(",").length < 10) {
396
+ // Count various elements within the current element
397
+ const counts = {
398
+ p: elem.querySelectorAll("p").length,
399
+ img: elem.querySelectorAll("img").length,
400
+ li: Math.max(0, elem.querySelectorAll("li").length - 100),
401
+ input:
402
+ elem.querySelectorAll("input").length -
403
+ elem.querySelectorAll("input[type=hidden]").length,
404
+ a: elem.querySelectorAll("a").length,
405
+ embed: elem.querySelectorAll("embed").length,
406
+ };
407
+ let textContent = elem?.textContent || "";
408
+ textContent = textContent.trim();
409
+ const contentLength = (textContent || "").replace(/\s+/g, " ").length;
410
+
411
+ const linkDensity = getLinkDensity(elem);
412
+
413
+ // Remove element if it meets certain criteria
414
+ if (
415
+ counts.img > 1 + counts.p * 1.3 ||
416
+ (counts.li > counts.p &&
417
+ elem.tagName !== "UL" &&
418
+ elem.tagName !== "OL") ||
419
+ counts.input > counts.p / 3 ||
420
+ (contentLength < minTextLength && counts.img === 0) ||
421
+ (weight < 25 && linkDensity > 0.2) ||
422
+ (weight >= 25 && linkDensity > 0.5) ||
423
+ (counts.embed === 1 && contentLength < 75) ||
424
+ counts.embed > 1
425
+ ) {
426
+ elem.remove();
427
+ }
428
+ }
429
+ }
430
+
431
+ return node;
432
+ }