extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,432 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module research/extractor/html-to-content/extract-content/extract-content-readability
|
|
3
|
+
* @description Research library module.
|
|
4
|
+
*/
|
|
5
|
+
import { parseHTML } from "linkedom";
|
|
6
|
+
|
|
7
|
+
interface Candidate {
|
|
8
|
+
score: number;
|
|
9
|
+
elem: any; // linkedom Element
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* ### HTML-to-Main-Content Extractor #1
|
|
14
|
+
* The function extracts main content with regex patterns, cleaning HTML, scoring nodes
|
|
15
|
+
* based on content indicators like paragraphs and id/class names, selecting
|
|
16
|
+
* the top candidate, extracting it, and cleaning up content around it.
|
|
17
|
+
*
|
|
18
|
+
*
|
|
19
|
+
* 1. Define regular expressions:
|
|
20
|
+
* - Various regex patterns are defined to identify content and non-content areas.
|
|
21
|
+
*
|
|
22
|
+
* 2. Define helper functions:
|
|
23
|
+
* - normalizeSpaces: Normalizes whitespace in a string.
|
|
24
|
+
* - stripTags: Removes all HTML tags from a string.
|
|
25
|
+
* - getTextLength: Calculates the length of text after stripping tags.
|
|
26
|
+
* - calculateLinkDensity: Calculates the ratio of link text to total text.
|
|
27
|
+
*
|
|
28
|
+
* 3. Clean HTML:
|
|
29
|
+
* - Remove unlikely candidates (e.g., ads, sidebars) from the HTML.
|
|
30
|
+
*
|
|
31
|
+
* 4. Define scoring function:
|
|
32
|
+
* - scoreNode: Assigns a score to an HTML node based on content and attributes.
|
|
33
|
+
* - Increases score for positive indicators (e.g., article, body, content tags).
|
|
34
|
+
* - Decreases score for negative indicators (e.g., hidden, footer, sidebar tags).
|
|
35
|
+
* - Adds to score based on paragraph tags and text length.
|
|
36
|
+
*
|
|
37
|
+
* 5. Find and score candidate nodes:
|
|
38
|
+
* - Identify potential content nodes in the cleaned HTML.
|
|
39
|
+
* - Score each node using the scoreNode function.
|
|
40
|
+
*
|
|
41
|
+
* 6. Select top candidate:
|
|
42
|
+
* - Sort candidates by score and select the highest-scoring node.
|
|
43
|
+
*
|
|
44
|
+
* 7. Extract content:
|
|
45
|
+
* - Use regex to extract content around the top candidate node.
|
|
46
|
+
*
|
|
47
|
+
* 8. Clean up extracted content:
|
|
48
|
+
* - Remove script and style tags and their contents.
|
|
49
|
+
* - Process anchor tags based on content density.
|
|
50
|
+
* - Keep only specific HTML tags (a, p, img, h1-h6, ul, ol, li).
|
|
51
|
+
* - Remove excess whitespace from the final content.
|
|
52
|
+
*
|
|
53
|
+
* [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
|
|
54
|
+
*
|
|
55
|
+
* @example
|
|
56
|
+
* var url = "https://www.nytimes.com/2024/08/28/business/telegram-ceo-pavel-durov-charged.html"
|
|
57
|
+
* const html = await (await fetch(url)).text();
|
|
58
|
+
* var articleContent = extractMainContentFromHTML(html);
|
|
59
|
+
* @param {Object} [options]
|
|
60
|
+
* @param {number} options.minContentLength default=140 - Minimum length of content to be considered valid
|
|
61
|
+
* @param {number} options.minScore default=20 - Minimum score for content to be considered valid
|
|
62
|
+
* @param {number} options.minTextLength default=25 - Minimum length of text to be considered valid
|
|
63
|
+
* @param {number} options.retryLength default=250 - Length to retry content extraction if initial attempt fails
|
|
64
|
+
* @returns {string} Extracted HTML string of main content
|
|
65
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
66
|
+
* Based on [Mozilla Readability (2015)](https://github.com/mozilla/readability)
|
|
67
|
+
* @category Extract
|
|
68
|
+
*/
|
|
69
|
+
export function extractMainContentFromHTML(
|
|
70
|
+
html: string,
|
|
71
|
+
options: {
|
|
72
|
+
minContentLength?: number;
|
|
73
|
+
minScore?: number;
|
|
74
|
+
minTextLength?: number;
|
|
75
|
+
retryLength?: number;
|
|
76
|
+
} = {},
|
|
77
|
+
): string {
|
|
78
|
+
const {
|
|
79
|
+
minContentLength = 140,
|
|
80
|
+
minScore = 20,
|
|
81
|
+
minTextLength = 25,
|
|
82
|
+
retryLength = 250,
|
|
83
|
+
} = options;
|
|
84
|
+
|
|
85
|
+
// Define regular expressions for content identification
|
|
86
|
+
const positiveRe =
|
|
87
|
+
/article|body|content|entry|hentry|main|page|pagination|post|text|blog|story/i;
|
|
88
|
+
const negativeRe =
|
|
89
|
+
/button|combx|comment|com-|contact|figure|foot|footer|footnote|form|input|masthead|media|meta|outbrain|promo|related|scroll|shoutbox|sidebar|sponsor|shopping|tags|tool|widget/i;
|
|
90
|
+
const videoRe = /https?:\/\/(?:www\.)?(?:youtube|vimeo)\.com/i;
|
|
91
|
+
|
|
92
|
+
// Parse the HTML string into a document
|
|
93
|
+
const doc = parseHTML(html)?.document;
|
|
94
|
+
if (!doc) return "";
|
|
95
|
+
|
|
96
|
+
// Remove script and style tags
|
|
97
|
+
doc.querySelectorAll("script, style").forEach((elem) => elem.remove());
|
|
98
|
+
|
|
99
|
+
// Clean HTML by removing unlikely candidates
|
|
100
|
+
for (const elem of doc.querySelectorAll("*")) {
|
|
101
|
+
const attrs =
|
|
102
|
+
(elem.getAttribute("class") || "") +
|
|
103
|
+
" " +
|
|
104
|
+
(elem.getAttribute("id") || "");
|
|
105
|
+
if (attrs.length < 2) continue;
|
|
106
|
+
//test unlikely candidates
|
|
107
|
+
if (
|
|
108
|
+
!["body", "html"].includes(elem.tagName.toLowerCase()) &&
|
|
109
|
+
/combx|comment|community|disqus|extra|foot|header|menu|remark|rss|shoutbox|sidebar|sponsor|ad-break|agegate|pagination|pager|popup|tweet|twitter/i.test(
|
|
110
|
+
attrs,
|
|
111
|
+
) &&
|
|
112
|
+
!/and|article|body|column|main|shadow/i.test(attrs)
|
|
113
|
+
) {
|
|
114
|
+
elem.remove();
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
// Convert divs to paragraphs if they don't contain block elements
|
|
119
|
+
const divs = doc.getElementsByTagName("div");
|
|
120
|
+
for (const elem of divs) {
|
|
121
|
+
if (
|
|
122
|
+
!/<(?:a|blockquote|dl|div|img|ol|p|pre|table|ul)/i.test(
|
|
123
|
+
elem?.innerHTML?.replace(/\s+/, " "),
|
|
124
|
+
)
|
|
125
|
+
) {
|
|
126
|
+
const newElem = doc.createElement("p");
|
|
127
|
+
for (const attr of elem.attributes) {
|
|
128
|
+
newElem.setAttribute(attr.name, attr.value);
|
|
129
|
+
}
|
|
130
|
+
while (elem.firstChild) {
|
|
131
|
+
newElem.appendChild(elem.firstChild);
|
|
132
|
+
}
|
|
133
|
+
elem.parentNode?.replaceChild(newElem, elem);
|
|
134
|
+
}
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
// Score nodes
|
|
138
|
+
const candidates: Record<string, Candidate> = {};
|
|
139
|
+
const elems = Array.from(doc.querySelectorAll("p, pre, td"));
|
|
140
|
+
|
|
141
|
+
for (const elem of elems) {
|
|
142
|
+
const parentNode = elem.parentNode;
|
|
143
|
+
const grandParentNode = parentNode ? parentNode.parentNode : null;
|
|
144
|
+
|
|
145
|
+
const innerText = (elem.textContent || "").trim();
|
|
146
|
+
const innerTextLen = innerText.length;
|
|
147
|
+
|
|
148
|
+
if (innerTextLen < minTextLength) continue;
|
|
149
|
+
|
|
150
|
+
const pKey = String(parentNode);
|
|
151
|
+
const gpKey = String(grandParentNode);
|
|
152
|
+
|
|
153
|
+
// Score parent and grandparent nodes
|
|
154
|
+
if (!candidates[pKey]) {
|
|
155
|
+
candidates[pKey] = scoreNode(parentNode as any, positiveRe, negativeRe);
|
|
156
|
+
}
|
|
157
|
+
if (grandParentNode && !candidates[gpKey]) {
|
|
158
|
+
candidates[gpKey] = scoreNode(
|
|
159
|
+
grandParentNode as any,
|
|
160
|
+
positiveRe,
|
|
161
|
+
negativeRe,
|
|
162
|
+
);
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
// Calculate score based on text content
|
|
166
|
+
let score = 1;
|
|
167
|
+
score += innerText.split(",").length;
|
|
168
|
+
score += Math.min(innerTextLen / 100, 3);
|
|
169
|
+
|
|
170
|
+
candidates[pKey].score += score;
|
|
171
|
+
if (grandParentNode) candidates[gpKey].score += score / 2;
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
// Adjust scores based on link density
|
|
175
|
+
for (const candidate of Object.values(candidates)) {
|
|
176
|
+
if (candidate && candidate.elem) {
|
|
177
|
+
candidate.score *= 1 - getLinkDensity(candidate.elem);
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Find the best candidate
|
|
182
|
+
const sortedCandidates = Object.values(candidates).sort(
|
|
183
|
+
(a, b) => b.score - a.score,
|
|
184
|
+
);
|
|
185
|
+
const bestCandidate = sortedCandidates[0];
|
|
186
|
+
|
|
187
|
+
let article: any;
|
|
188
|
+
let cleanedArticle: any;
|
|
189
|
+
|
|
190
|
+
if (bestCandidate) {
|
|
191
|
+
// Extract content from the best candidate and its siblings
|
|
192
|
+
const siblingScoreThreshold = Math.max(10, bestCandidate.score * 0.2);
|
|
193
|
+
article = doc.createElement("div");
|
|
194
|
+
const parent = bestCandidate.elem.parentNode;
|
|
195
|
+
const siblings = parent
|
|
196
|
+
? Array.from(parent.children)
|
|
197
|
+
: [bestCandidate.elem];
|
|
198
|
+
|
|
199
|
+
for (let sibling of siblings) {
|
|
200
|
+
let append = false;
|
|
201
|
+
if (
|
|
202
|
+
sibling === bestCandidate.elem ||
|
|
203
|
+
(candidates[String(sibling)] &&
|
|
204
|
+
candidates[String(sibling)].score >= siblingScoreThreshold)
|
|
205
|
+
) {
|
|
206
|
+
append = true;
|
|
207
|
+
} else if (sibling.tagName === "P") {
|
|
208
|
+
const linkDensity = getLinkDensity(sibling as any);
|
|
209
|
+
const nodeContent = sibling.textContent || "";
|
|
210
|
+
const nodeLength = nodeContent.length;
|
|
211
|
+
|
|
212
|
+
if (
|
|
213
|
+
(nodeLength > 80 && linkDensity < 0.25) ||
|
|
214
|
+
(nodeLength <= 80 && linkDensity === 0 && /\.( |$)/.test(nodeContent))
|
|
215
|
+
) {
|
|
216
|
+
append = true;
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
if (append) article.innerHTML += sibling.innerHTML;
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
cleanedArticle = sanitize(
|
|
224
|
+
article,
|
|
225
|
+
candidates,
|
|
226
|
+
videoRe,
|
|
227
|
+
positiveRe,
|
|
228
|
+
negativeRe,
|
|
229
|
+
minTextLength,
|
|
230
|
+
);
|
|
231
|
+
// var articleLength = cleanedArticle ? cleanedArticle.textContent.length : 0;
|
|
232
|
+
} else {
|
|
233
|
+
// If no best candidate, use the body or entire document
|
|
234
|
+
article = doc.querySelector("body") || doc;
|
|
235
|
+
cleanedArticle = sanitize(
|
|
236
|
+
article,
|
|
237
|
+
candidates,
|
|
238
|
+
videoRe,
|
|
239
|
+
positiveRe,
|
|
240
|
+
negativeRe,
|
|
241
|
+
minTextLength,
|
|
242
|
+
);
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
return cleanedArticle ? cleanedArticle.innerHTML : "";
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* Calculates the link density of an element.
|
|
250
|
+
* @param {Element} elem - The element to calculate link density for
|
|
251
|
+
* @returns {number} The link density (ratio of link text length to total text length)
|
|
252
|
+
*/
|
|
253
|
+
export function getLinkDensity(elem: any): number {
|
|
254
|
+
if (!elem || !elem.textContent) {
|
|
255
|
+
return 0;
|
|
256
|
+
}
|
|
257
|
+
const links = elem.querySelectorAll("a");
|
|
258
|
+
const textLength = elem.textContent.trim().length;
|
|
259
|
+
const linkLength = (Array.from(links) as any[]).reduce(
|
|
260
|
+
(total: number, link: any) => total + link.textContent.trim().length,
|
|
261
|
+
0,
|
|
262
|
+
);
|
|
263
|
+
return textLength > 0 ? linkLength / textLength : 0;
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
/**
|
|
267
|
+
* Calculates the weight of an element based on its class and id attributes.
|
|
268
|
+
* @param {Element} elem - The element to calculate weight for
|
|
269
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
270
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
271
|
+
* @returns {number} The calculated weight
|
|
272
|
+
*/
|
|
273
|
+
export function classWeight(
|
|
274
|
+
elem: any,
|
|
275
|
+
positiveRe: RegExp,
|
|
276
|
+
negativeRe: RegExp,
|
|
277
|
+
): number {
|
|
278
|
+
let weight = 0;
|
|
279
|
+
if (!elem || !elem.getAttribute) return weight;
|
|
280
|
+
if (elem.getAttribute("class")) {
|
|
281
|
+
if (negativeRe.test(elem.getAttribute("class"))) weight -= 25;
|
|
282
|
+
if (positiveRe.test(elem.getAttribute("class"))) weight += 25;
|
|
283
|
+
}
|
|
284
|
+
if (elem.getAttribute("id")) {
|
|
285
|
+
if (negativeRe.test(elem.getAttribute("id"))) weight -= 25;
|
|
286
|
+
if (positiveRe.test(elem.getAttribute("id"))) weight += 25;
|
|
287
|
+
}
|
|
288
|
+
return weight;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Scores a node based on its tag name and attributes.
|
|
293
|
+
* @param {Element} elem - The element to score
|
|
294
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
295
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
296
|
+
* @returns {Object} An object containing the score and the element
|
|
297
|
+
*/
|
|
298
|
+
export function scoreNode(
|
|
299
|
+
elem: any,
|
|
300
|
+
positiveRe: RegExp,
|
|
301
|
+
negativeRe: RegExp,
|
|
302
|
+
): Candidate {
|
|
303
|
+
if (!elem || !elem.tagName) return { score: 0, elem };
|
|
304
|
+
const DIV_SCORES = new Set(["div", "article"]);
|
|
305
|
+
const BLOCK_SCORES = new Set(["pre", "td", "blockquote"]);
|
|
306
|
+
const BAD_ELEM_SCORES = new Set([
|
|
307
|
+
"address",
|
|
308
|
+
"ol",
|
|
309
|
+
"ul",
|
|
310
|
+
"dl",
|
|
311
|
+
"dd",
|
|
312
|
+
"dt",
|
|
313
|
+
"li",
|
|
314
|
+
"form",
|
|
315
|
+
"aside",
|
|
316
|
+
]);
|
|
317
|
+
const STRUCTURE_SCORES = new Set([
|
|
318
|
+
"h1",
|
|
319
|
+
"h2",
|
|
320
|
+
"h3",
|
|
321
|
+
"h4",
|
|
322
|
+
"h5",
|
|
323
|
+
"h6",
|
|
324
|
+
"th",
|
|
325
|
+
"header",
|
|
326
|
+
"footer",
|
|
327
|
+
"nav",
|
|
328
|
+
]);
|
|
329
|
+
|
|
330
|
+
let score = classWeight(elem, positiveRe, negativeRe);
|
|
331
|
+
const name = elem.tagName.toLowerCase();
|
|
332
|
+
if (DIV_SCORES.has(name)) score += 5;
|
|
333
|
+
else if (BLOCK_SCORES.has(name)) score += 3;
|
|
334
|
+
else if (BAD_ELEM_SCORES.has(name)) score -= 3;
|
|
335
|
+
else if (STRUCTURE_SCORES.has(name)) score -= 5;
|
|
336
|
+
return { score, elem };
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
/**
|
|
340
|
+
* Sanitizes the content by removing unwanted elements and cleaning remaining elements.
|
|
341
|
+
* @param {Element} node - The node to sanitize
|
|
342
|
+
* @param {Object} candidates - Object containing scored candidates
|
|
343
|
+
* @param {RegExp} videoRe - Regular expression for video URLs
|
|
344
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
345
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
346
|
+
* @param {number} minTextLength - Minimum text length to consider
|
|
347
|
+
* @returns {Element} The sanitized node
|
|
348
|
+
*/
|
|
349
|
+
export function sanitize(
|
|
350
|
+
node: any,
|
|
351
|
+
candidates: Record<string, Candidate>,
|
|
352
|
+
videoRe: RegExp,
|
|
353
|
+
positiveRe: RegExp,
|
|
354
|
+
negativeRe: RegExp,
|
|
355
|
+
minTextLength: number,
|
|
356
|
+
): any {
|
|
357
|
+
const DIV_TO_P_ELEMS = new Set([
|
|
358
|
+
"a",
|
|
359
|
+
"blockquote",
|
|
360
|
+
"dl",
|
|
361
|
+
"div",
|
|
362
|
+
"img",
|
|
363
|
+
"ol",
|
|
364
|
+
"p",
|
|
365
|
+
"pre",
|
|
366
|
+
"table",
|
|
367
|
+
"ul",
|
|
368
|
+
]);
|
|
369
|
+
|
|
370
|
+
// Remove unwanted elements
|
|
371
|
+
for (let elem of node.querySelectorAll(
|
|
372
|
+
"h1, h2, h3, h4, h5, h6, form, textarea, iframe",
|
|
373
|
+
)) {
|
|
374
|
+
if (elem.tagName === "IFRAME" && videoRe.test(elem.src)) {
|
|
375
|
+
elem.textContent = "VIDEO";
|
|
376
|
+
} else {
|
|
377
|
+
elem.remove();
|
|
378
|
+
}
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
// Clean remaining elements
|
|
382
|
+
const allowed = new Set();
|
|
383
|
+
for (let elem of (
|
|
384
|
+
Array.from(
|
|
385
|
+
node.querySelectorAll("table, ul, div, aside, header, footer, section"),
|
|
386
|
+
) as any[]
|
|
387
|
+
).reverse()) {
|
|
388
|
+
if (allowed.has(elem)) continue;
|
|
389
|
+
|
|
390
|
+
const weight = classWeight(elem, positiveRe, negativeRe);
|
|
391
|
+
const score = candidates[String(elem)] ? candidates[String(elem)].score : 0;
|
|
392
|
+
|
|
393
|
+
if (weight + score < 0) {
|
|
394
|
+
elem.remove();
|
|
395
|
+
} else if (elem.textContent.split(",").length < 10) {
|
|
396
|
+
// Count various elements within the current element
|
|
397
|
+
const counts = {
|
|
398
|
+
p: elem.querySelectorAll("p").length,
|
|
399
|
+
img: elem.querySelectorAll("img").length,
|
|
400
|
+
li: Math.max(0, elem.querySelectorAll("li").length - 100),
|
|
401
|
+
input:
|
|
402
|
+
elem.querySelectorAll("input").length -
|
|
403
|
+
elem.querySelectorAll("input[type=hidden]").length,
|
|
404
|
+
a: elem.querySelectorAll("a").length,
|
|
405
|
+
embed: elem.querySelectorAll("embed").length,
|
|
406
|
+
};
|
|
407
|
+
let textContent = elem?.textContent || "";
|
|
408
|
+
textContent = textContent.trim();
|
|
409
|
+
const contentLength = (textContent || "").replace(/\s+/g, " ").length;
|
|
410
|
+
|
|
411
|
+
const linkDensity = getLinkDensity(elem);
|
|
412
|
+
|
|
413
|
+
// Remove element if it meets certain criteria
|
|
414
|
+
if (
|
|
415
|
+
counts.img > 1 + counts.p * 1.3 ||
|
|
416
|
+
(counts.li > counts.p &&
|
|
417
|
+
elem.tagName !== "UL" &&
|
|
418
|
+
elem.tagName !== "OL") ||
|
|
419
|
+
counts.input > counts.p / 3 ||
|
|
420
|
+
(contentLength < minTextLength && counts.img === 0) ||
|
|
421
|
+
(weight < 25 && linkDensity > 0.2) ||
|
|
422
|
+
(weight >= 25 && linkDensity > 0.5) ||
|
|
423
|
+
(counts.embed === 1 && contentLength < 75) ||
|
|
424
|
+
counts.embed > 1
|
|
425
|
+
) {
|
|
426
|
+
elem.remove();
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
|
|
431
|
+
return node;
|
|
432
|
+
}
|