extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,830 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
/**
|
|
3
|
+
* @module research/extractor/html-to-content/extract-content/extract-content-mercury
|
|
4
|
+
* @description Research library module.
|
|
5
|
+
*/
|
|
6
|
+
import { parseHTML } from "linkedom";
|
|
7
|
+
import {
|
|
8
|
+
convertNodeTo,
|
|
9
|
+
stripUnlikelyCandidates,
|
|
10
|
+
convertToParagraphs,
|
|
11
|
+
cleanAttributes,
|
|
12
|
+
cleanHOnes,
|
|
13
|
+
cleanImages,
|
|
14
|
+
removeEmpty,
|
|
15
|
+
rewriteTopLevel,
|
|
16
|
+
stripJunkTags,
|
|
17
|
+
textLength,
|
|
18
|
+
linkDensity,
|
|
19
|
+
removeUnlessContent,
|
|
20
|
+
nodeIsSufficient,
|
|
21
|
+
} from "./extract-content-mercury-utils";
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* ### HTML-to-Main-Content Extractor #2
|
|
25
|
+
*
|
|
26
|
+
* 1. The algorithm starts by loading the HTML content using linkedom, a lightweight DOM parser for Node.js.
|
|
27
|
+
* 2. It then applies a series of cleaning and scoring techniques to identify the main content of
|
|
28
|
+
* the page, starting with stripping unlikely candidates (e.g., elements with class names like "comment"
|
|
29
|
+
* or "sidebar").
|
|
30
|
+
* 3. The HTML is converted into a series of paragraph elements, which are then scored based on various
|
|
31
|
+
* factors such as text length, number of commas, and the presence of certain class names or IDs.
|
|
32
|
+
* 4. The algorithm assigns scores to parent and grandparent elements based on the scores of their
|
|
33
|
+
* children, with parents receiving the full score and grandparents receiving half.
|
|
34
|
+
* 5. After scoring, the algorithm finds the top candidate element by selecting the node with the
|
|
35
|
+
* highest score.
|
|
36
|
+
* 6. The top candidate's siblings are then examined to see if they should be included in the main
|
|
37
|
+
* content, based on their scores and other factors like link density.
|
|
38
|
+
* 7. The algorithm then cleans the selected content by removing unnecessary tags, attributes, and empty
|
|
39
|
+
* elements.
|
|
40
|
+
* 8. It also handles special cases like cleaning up header tags, images, and other potentially irrelevant
|
|
41
|
+
* content.
|
|
42
|
+
* 9. Throughout the process, the algorithm uses various regular expressions and scoring heuristics to
|
|
43
|
+
* identify positive and negative indicators of content relevance.
|
|
44
|
+
* 10. Finally, the cleaned and extracted content is returned as an HTML string, representing the main
|
|
45
|
+
* body of the article or webpage.
|
|
46
|
+
*
|
|
47
|
+
* [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
|
|
48
|
+
*
|
|
49
|
+
* @param {string} html - The HTML content to extract from.
|
|
50
|
+
* @param {Object} [opts] - The options for content extraction.
|
|
51
|
+
* @param {boolean} opts.stripUnlikelyCandidates default=true - Remove elements that match non-article-
|
|
52
|
+
* like criteria first (e.g., elements with a classname of "comment").
|
|
53
|
+
* @param {boolean} opts.weightNodes default=true - Modify an element's score based on certain classNames or
|
|
54
|
+
* IDs (e.g., subtract if a node has a className of 'comment', add if a node has an ID of 'entry-content').
|
|
55
|
+
* @param {boolean} opts.cleanConditionally default=true - Clean the node to remove superfluous content
|
|
56
|
+
* like forms, ads, etc. Initially, pass in the most restrictive options which will return the highest
|
|
57
|
+
* quality content. On each failure, retry with slightly more lax options.
|
|
58
|
+
* @returns {string} The extracted content as an HTML string, or null if extraction fails.
|
|
59
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
60
|
+
* Based on [Postlight Mercury Parser (2017-)](https://github.com/postlight/parser/tree/main/src)
|
|
61
|
+
* @example var url = "https://en.wikipedia.org/wiki/David_Hilbert"
|
|
62
|
+
* var html = await (await fetch(url)).text();
|
|
63
|
+
* var content = extractMainContentFromHTML(html);
|
|
64
|
+
* console.log(content); // HTML content of main article body
|
|
65
|
+
* @category Extract
|
|
66
|
+
*/
|
|
67
|
+
export function extractMainContentFromHTML2(html, opts) {
|
|
68
|
+
opts = {
|
|
69
|
+
stripUnlikelyCandidates: true,
|
|
70
|
+
weightNodes: true,
|
|
71
|
+
cleanConditionally: true,
|
|
72
|
+
...opts,
|
|
73
|
+
};
|
|
74
|
+
|
|
75
|
+
if (!html) return;
|
|
76
|
+
const document = parseHTML(html)?.document;
|
|
77
|
+
|
|
78
|
+
if (!document) return;
|
|
79
|
+
|
|
80
|
+
var title = document.querySelector("title")?.textContent.trim();
|
|
81
|
+
|
|
82
|
+
// Cascade through our extraction-specific opts in an ordered fashion,
|
|
83
|
+
// turning them off as we try to extract content.
|
|
84
|
+
let node = getContentNode(document, title, opts);
|
|
85
|
+
|
|
86
|
+
if (nodeIsSufficient(node)) {
|
|
87
|
+
return cleanAndReturnNode(node, document);
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// We didn't succeed on first pass, one by one disable our
|
|
91
|
+
// extraction opts and try again.
|
|
92
|
+
// eslint-disable-next-line no-restricted-syntax
|
|
93
|
+
for (const key of Reflect.ownKeys(opts).filter((k) => opts[k] === true)) {
|
|
94
|
+
opts[key] = false;
|
|
95
|
+
const { document: newDocument } = parseHTML(html);
|
|
96
|
+
|
|
97
|
+
node = getContentNode(newDocument, title, opts);
|
|
98
|
+
|
|
99
|
+
if (nodeIsSufficient(node)) {
|
|
100
|
+
break;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
return cleanAndReturnNode(node, document);
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
// A list of tags that should be ignored when trying to find the top candidate
|
|
108
|
+
// for a document.
|
|
109
|
+
const NON_TOP_CANDIDATE_TAGS = [
|
|
110
|
+
"br",
|
|
111
|
+
"b",
|
|
112
|
+
"i",
|
|
113
|
+
"label",
|
|
114
|
+
"hr",
|
|
115
|
+
"area",
|
|
116
|
+
"base",
|
|
117
|
+
"basefont",
|
|
118
|
+
"input",
|
|
119
|
+
"img",
|
|
120
|
+
"link",
|
|
121
|
+
"meta",
|
|
122
|
+
];
|
|
123
|
+
|
|
124
|
+
const NON_TOP_CANDIDATE_TAGS_RE = new RegExp(
|
|
125
|
+
`^(${NON_TOP_CANDIDATE_TAGS.join("|")})$`,
|
|
126
|
+
"i"
|
|
127
|
+
);
|
|
128
|
+
|
|
129
|
+
const PHOTO_HINTS = ["figure", "photo", "image", "caption"];
|
|
130
|
+
const PHOTO_HINTS_RE = new RegExp(PHOTO_HINTS.join("|"), "i");
|
|
131
|
+
|
|
132
|
+
// A list of strings that denote a positive scoring for this content as being
|
|
133
|
+
// an article container. Checked against className and id.
|
|
134
|
+
//
|
|
135
|
+
// TODO: Perhaps have these scale based on their odds of being quality?
|
|
136
|
+
const POSITIVE_SCORE_HINTS = [
|
|
137
|
+
"article",
|
|
138
|
+
"articlecontent",
|
|
139
|
+
"instapaper_body",
|
|
140
|
+
"blog",
|
|
141
|
+
"body",
|
|
142
|
+
"content",
|
|
143
|
+
"entry-content-asset",
|
|
144
|
+
"entry",
|
|
145
|
+
"hentry",
|
|
146
|
+
"main",
|
|
147
|
+
"Normal",
|
|
148
|
+
"page",
|
|
149
|
+
"pagination",
|
|
150
|
+
"permalink",
|
|
151
|
+
"post",
|
|
152
|
+
"story",
|
|
153
|
+
"text",
|
|
154
|
+
"[-_]copy", // usatoday
|
|
155
|
+
"\\Bcopy",
|
|
156
|
+
];
|
|
157
|
+
|
|
158
|
+
// The above list, joined into a matching regular expression
|
|
159
|
+
const POSITIVE_SCORE_RE = new RegExp(POSITIVE_SCORE_HINTS.join("|"), "i");
|
|
160
|
+
|
|
161
|
+
// Readability publisher-specific guidelines
|
|
162
|
+
const READABILITY_ASSET = new RegExp("entry-content-asset", "i");
|
|
163
|
+
|
|
164
|
+
const PARAGRAPH_SCORE_TAGS = new RegExp("^(p|li|span|pre)$", "i");
|
|
165
|
+
const CHILD_CONTENT_TAGS = new RegExp("^(td|blockquote|ol|ul|dl)$", "i");
|
|
166
|
+
const BAD_TAGS = new RegExp("^(address|form)$", "i");
|
|
167
|
+
|
|
168
|
+
// A list of strings that denote a negative scoring for this content as being
|
|
169
|
+
// an article container. Checked against className and id.
|
|
170
|
+
//
|
|
171
|
+
// TODO: Perhaps have these scale based on their odds of being quality?
|
|
172
|
+
const NEGATIVE_SCORE_HINTS = [
|
|
173
|
+
"adbox",
|
|
174
|
+
"advert",
|
|
175
|
+
"author",
|
|
176
|
+
"bio",
|
|
177
|
+
"bookmark",
|
|
178
|
+
"bottom",
|
|
179
|
+
"byline",
|
|
180
|
+
"clear",
|
|
181
|
+
"com-",
|
|
182
|
+
"combx",
|
|
183
|
+
"comment",
|
|
184
|
+
"comment\\B",
|
|
185
|
+
"contact",
|
|
186
|
+
"copy",
|
|
187
|
+
"credit",
|
|
188
|
+
"crumb",
|
|
189
|
+
"date",
|
|
190
|
+
"deck",
|
|
191
|
+
"excerpt",
|
|
192
|
+
"featured", // tnr.com has a featured_content which throws us off
|
|
193
|
+
"foot",
|
|
194
|
+
"footer",
|
|
195
|
+
"footnote",
|
|
196
|
+
"graf",
|
|
197
|
+
"head",
|
|
198
|
+
"info",
|
|
199
|
+
"infotext", // newscientist.com copyright
|
|
200
|
+
"instapaper_ignore",
|
|
201
|
+
"jump",
|
|
202
|
+
"linebreak",
|
|
203
|
+
"link",
|
|
204
|
+
"masthead",
|
|
205
|
+
"media",
|
|
206
|
+
"meta",
|
|
207
|
+
"modal",
|
|
208
|
+
"outbrain", // slate.com junk
|
|
209
|
+
"promo",
|
|
210
|
+
"pr_", // autoblog - press release
|
|
211
|
+
"related",
|
|
212
|
+
"respond",
|
|
213
|
+
"roundcontent", // lifehacker restricted content warning
|
|
214
|
+
"scroll",
|
|
215
|
+
"secondary",
|
|
216
|
+
"share",
|
|
217
|
+
"shopping",
|
|
218
|
+
"shoutbox",
|
|
219
|
+
"side",
|
|
220
|
+
"sidebar",
|
|
221
|
+
"sponsor",
|
|
222
|
+
"stamp",
|
|
223
|
+
"sub",
|
|
224
|
+
"summary",
|
|
225
|
+
"tags",
|
|
226
|
+
"tools",
|
|
227
|
+
"widget",
|
|
228
|
+
];
|
|
229
|
+
// The above list, joined into a matching regular expression
|
|
230
|
+
const NEGATIVE_SCORE_RE = new RegExp(NEGATIVE_SCORE_HINTS.join("|"), "i");
|
|
231
|
+
|
|
232
|
+
// A list of selectors that specify, very clearly, either hNews or other
|
|
233
|
+
// very content-specific style content, like Blogger templates.
|
|
234
|
+
// More examples here: http://microformats.org/wiki/blog-post-formats
|
|
235
|
+
const HNEWS_CONTENT_SELECTORS = [
|
|
236
|
+
[".hentry", ".entry-content"],
|
|
237
|
+
["entry", ".entry-content"],
|
|
238
|
+
[".entry", ".entry_content"],
|
|
239
|
+
[".post", ".postbody"],
|
|
240
|
+
[".post", ".post_body"],
|
|
241
|
+
[".post", ".post-body"],
|
|
242
|
+
];
|
|
243
|
+
|
|
244
|
+
/**
|
|
245
|
+
* Normalizes spaces in a given text string.
|
|
246
|
+
* @param {string} text - The text to normalize.
|
|
247
|
+
* @returns {string} The normalized text.
|
|
248
|
+
*/
|
|
249
|
+
function normalizeSpaces(text) {
|
|
250
|
+
return text.replace(/\s{2,}(?![^<>]*<\/(pre|code|textarea)>)/g, " ").trim();
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
/**
|
|
254
|
+
* Cleans and returns the HTML of a given node.
|
|
255
|
+
* @param {Node} node - The node to clean and return.
|
|
256
|
+
* @param {Document} document - The document object.
|
|
257
|
+
* @returns {string|null} The cleaned HTML string or null if no node is provided.
|
|
258
|
+
*/
|
|
259
|
+
function cleanAndReturnNode(node, document) {
|
|
260
|
+
if (!node) {
|
|
261
|
+
return null;
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
return normalizeSpaces(node.outerHTML);
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
/**
|
|
268
|
+
* Gets the content node from the document.
|
|
269
|
+
* @param {Document} document - The document object.
|
|
270
|
+
* @param {string} title - The title of the document.
|
|
271
|
+
* @param {Object} opts - The options for content extraction.
|
|
272
|
+
* @returns {Node} The content node.
|
|
273
|
+
*/
|
|
274
|
+
function getContentNode(document, title, opts) {
|
|
275
|
+
return cleanContent(extractBestNode(document, opts), {
|
|
276
|
+
document,
|
|
277
|
+
cleanConditionally: opts.cleanConditionally,
|
|
278
|
+
title,
|
|
279
|
+
});
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
/**
|
|
283
|
+
* Gets the score of a node.
|
|
284
|
+
* @param {Node} node - The node to get the score from.
|
|
285
|
+
* @returns {number|null} The score of the node or null if no score is set.
|
|
286
|
+
*/
|
|
287
|
+
function getScore(node) {
|
|
288
|
+
return parseFloat(node.getAttribute("score")) || null;
|
|
289
|
+
}
|
|
290
|
+
|
|
291
|
+
/**
|
|
292
|
+
* Scores the number of commas in a text.
|
|
293
|
+
* @param {string} text - The text to score.
|
|
294
|
+
* @returns {number} The number of commas in the text.
|
|
295
|
+
*/
|
|
296
|
+
function scoreCommas(text) {
|
|
297
|
+
return (text.match(/,/g) || []).length;
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
/**
|
|
301
|
+
* Converts span elements to div elements.
|
|
302
|
+
* @param {Node} node - The node to convert.
|
|
303
|
+
* @param {Document} document - The document object.
|
|
304
|
+
*/
|
|
305
|
+
function convertSpans(node, document) {
|
|
306
|
+
if (node?.tagName?.toLowerCase() === "span") {
|
|
307
|
+
// convert spans to divs
|
|
308
|
+
convertNodeTo(node, document, "div");
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
|
|
312
|
+
/**
|
|
313
|
+
* Adds a score to a node and its parent elements.
|
|
314
|
+
* @param {Node} node - The node to add the score to.
|
|
315
|
+
* @param {Document} document - The document object.
|
|
316
|
+
* @param {number} score - The score to add.
|
|
317
|
+
*/
|
|
318
|
+
function addScoreTo(node, document, score) {
|
|
319
|
+
if (node) {
|
|
320
|
+
convertSpans(node, document);
|
|
321
|
+
addScore(node, document, score);
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/**
|
|
326
|
+
* Scores paragraph elements in the document.
|
|
327
|
+
* @param {Document} document - The document object.
|
|
328
|
+
* @param {boolean} weightNodes - Whether to weight nodes or not.
|
|
329
|
+
* @returns {Document} The document with scored paragraphs.
|
|
330
|
+
*/
|
|
331
|
+
function scorePs(document, weightNodes) {
|
|
332
|
+
document.querySelectorAll("p, pre").forEach((node) => {
|
|
333
|
+
if (!node.hasAttribute("score")) {
|
|
334
|
+
// The raw score for this paragraph, before we add any parent/child
|
|
335
|
+
// scores.
|
|
336
|
+
node = setScore(
|
|
337
|
+
node,
|
|
338
|
+
document,
|
|
339
|
+
getOrInitScore(node, document, weightNodes)
|
|
340
|
+
);
|
|
341
|
+
|
|
342
|
+
const parent = node.parentNode;
|
|
343
|
+
const rawScore = scoreNode(node);
|
|
344
|
+
|
|
345
|
+
addScoreTo(parent, document, rawScore, weightNodes);
|
|
346
|
+
if (parent) {
|
|
347
|
+
// Add half of the individual content score to the
|
|
348
|
+
// grandparent
|
|
349
|
+
addScoreTo(parent.parentNode, document, rawScore / 2, weightNodes);
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
});
|
|
353
|
+
|
|
354
|
+
return document;
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
/**
|
|
358
|
+
* Scores the content of the document.
|
|
359
|
+
* @param {Document} document - The document object.
|
|
360
|
+
* @param {boolean} weightNodes - Whether to weight nodes or not.
|
|
361
|
+
* @returns {Document} The document with scored content.
|
|
362
|
+
*/
|
|
363
|
+
function scoreContent(document, weightNodes = true) {
|
|
364
|
+
// First, look for special hNews based selectors and give them a big
|
|
365
|
+
// boost, if they exist
|
|
366
|
+
HNEWS_CONTENT_SELECTORS.forEach(([parentSelector, childSelector]) => {
|
|
367
|
+
document
|
|
368
|
+
.querySelectorAll(`${parentSelector} ${childSelector}`)
|
|
369
|
+
.forEach((node) => {
|
|
370
|
+
addScore(node.closest(parentSelector), document, 80);
|
|
371
|
+
});
|
|
372
|
+
});
|
|
373
|
+
|
|
374
|
+
// Doubling this again
|
|
375
|
+
// Previous solution caused a bug
|
|
376
|
+
// in which parents weren't retaining
|
|
377
|
+
// scores. This is not ideal, and
|
|
378
|
+
// should be fixed.
|
|
379
|
+
scorePs(document, weightNodes);
|
|
380
|
+
scorePs(document, weightNodes);
|
|
381
|
+
|
|
382
|
+
return document;
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
/**
|
|
386
|
+
* Scores the length of text.
|
|
387
|
+
* @param {number} textLength - The length of the text.
|
|
388
|
+
* @param {string} tagName - The tag name of the element.
|
|
389
|
+
* @returns {number} The score based on text length.
|
|
390
|
+
*/
|
|
391
|
+
function scoreLength(textLength, tagName = "p") {
|
|
392
|
+
const chunks = textLength / 50;
|
|
393
|
+
|
|
394
|
+
if (chunks > 0) {
|
|
395
|
+
let lengthBonus;
|
|
396
|
+
|
|
397
|
+
// No idea why p or pre are being tamped down here
|
|
398
|
+
// but just following the source for now
|
|
399
|
+
// Not even sure why tagName is included here,
|
|
400
|
+
// since this is only being called from the context
|
|
401
|
+
// of scoreParagraph
|
|
402
|
+
if (new RegExp("^(p|pre)$", "i").test(tagName)) {
|
|
403
|
+
lengthBonus = chunks - 2;
|
|
404
|
+
} else {
|
|
405
|
+
lengthBonus = chunks - 1.25;
|
|
406
|
+
}
|
|
407
|
+
|
|
408
|
+
return Math.min(Math.max(lengthBonus, 0), 3);
|
|
409
|
+
}
|
|
410
|
+
|
|
411
|
+
return 0;
|
|
412
|
+
}
|
|
413
|
+
|
|
414
|
+
/**
|
|
415
|
+
* Sets the score attribute of a node.
|
|
416
|
+
* @param {Node} node - The node to set the score on.
|
|
417
|
+
* @param {Document} document - The document object.
|
|
418
|
+
* @param {number} score - The score to set.
|
|
419
|
+
* @returns {Node} The node with the set score.
|
|
420
|
+
* @private
|
|
421
|
+
*/
|
|
422
|
+
export function setScore(node, document, score) {
|
|
423
|
+
node.setAttribute("score", score);
|
|
424
|
+
return node;
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
/**
|
|
428
|
+
* Scores a paragraph node.
|
|
429
|
+
* @param {Node} node - The paragraph node to score.
|
|
430
|
+
* @private
|
|
431
|
+
* @returns {number} The score of the paragraph.
|
|
432
|
+
*/
|
|
433
|
+
export function scoreParagraph(node) {
|
|
434
|
+
let score = 1;
|
|
435
|
+
const text = node.textContent.trim();
|
|
436
|
+
const textLength = text.length;
|
|
437
|
+
|
|
438
|
+
// If this paragraph is less than 25 characters, don't count it.
|
|
439
|
+
if (textLength < 25) {
|
|
440
|
+
return 0;
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
// Add points for any commas within this paragraph
|
|
444
|
+
score += scoreCommas(text);
|
|
445
|
+
|
|
446
|
+
// For every 50 characters in this paragraph, add another point. Up
|
|
447
|
+
// to 3 points.
|
|
448
|
+
score += scoreLength(textLength);
|
|
449
|
+
|
|
450
|
+
// Articles can end with short paragraphs when people are being clever
|
|
451
|
+
// but they can also end with short paragraphs setting up lists of junk
|
|
452
|
+
// that we strip. This negative tweaks junk setup paragraphs just below
|
|
453
|
+
// the cutoff threshold.
|
|
454
|
+
if (text.slice(-1) === ":") {
|
|
455
|
+
score -= 1;
|
|
456
|
+
}
|
|
457
|
+
|
|
458
|
+
return score;
|
|
459
|
+
}
|
|
460
|
+
|
|
461
|
+
// Score an individual node. Has some smarts for paragraphs, otherwise
|
|
462
|
+
// just scores based on tag.
|
|
463
|
+
function scoreNode(node) {
|
|
464
|
+
const tagName = node.tagName?.toLowerCase();
|
|
465
|
+
// if (!tagName) return 0;
|
|
466
|
+
|
|
467
|
+
// TODO: Consider ordering by most likely.
|
|
468
|
+
// E.g., if divs are a more common tag on a page,
|
|
469
|
+
// Could save doing that regex test on every node \u2013 AP
|
|
470
|
+
if (PARAGRAPH_SCORE_TAGS.test(tagName)) {
|
|
471
|
+
return scoreParagraph(node);
|
|
472
|
+
}
|
|
473
|
+
if (tagName === "div") {
|
|
474
|
+
return 5;
|
|
475
|
+
}
|
|
476
|
+
if (CHILD_CONTENT_TAGS.test(tagName)) {
|
|
477
|
+
return 3;
|
|
478
|
+
}
|
|
479
|
+
if (BAD_TAGS.test(tagName)) {
|
|
480
|
+
return -3;
|
|
481
|
+
}
|
|
482
|
+
if (tagName === "th") {
|
|
483
|
+
return -5;
|
|
484
|
+
}
|
|
485
|
+
|
|
486
|
+
return 0;
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
function addScore(node, document, amount) {
|
|
490
|
+
try {
|
|
491
|
+
const score = getOrInitScore(node, document) + amount;
|
|
492
|
+
setScore(node, document, score);
|
|
493
|
+
} catch (e) {
|
|
494
|
+
// Ignoring; error occurs in scoreNode
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
return node;
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
// Adds 1/4 of a child's score to its parent
|
|
501
|
+
function addToParent(node, document, score) {
|
|
502
|
+
const parent = node.parentNode;
|
|
503
|
+
if (parent) {
|
|
504
|
+
addScore(parent, document, score * 0.25);
|
|
505
|
+
}
|
|
506
|
+
|
|
507
|
+
return node;
|
|
508
|
+
}
|
|
509
|
+
|
|
510
|
+
// Using a variety of scoring techniques, extract the content most
|
|
511
|
+
// likely to be article text.
|
|
512
|
+
//
|
|
513
|
+
// If strip_unlikely_candidates is True, remove any elements that
|
|
514
|
+
// match certain criteria first. (Like, does this element have a
|
|
515
|
+
// classname of "comment")
|
|
516
|
+
//
|
|
517
|
+
// If weight_nodes is True, use classNames and IDs to determine the
|
|
518
|
+
// worthiness of nodes.
|
|
519
|
+
//
|
|
520
|
+
// Returns a DOM node
|
|
521
|
+
function extractBestNode(document, opts) {
|
|
522
|
+
if (opts.stripUnlikelyCandidates) {
|
|
523
|
+
document = stripUnlikelyCandidates(document);
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
document = convertToParagraphs(document);
|
|
527
|
+
document = scoreContent(document, opts.weightNodes);
|
|
528
|
+
const topCandidate = findTopCandidate(document);
|
|
529
|
+
|
|
530
|
+
return topCandidate;
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
// Clean our article content, returning a new, cleaned node.
|
|
534
|
+
function cleanContent(
|
|
535
|
+
article,
|
|
536
|
+
{ document, cleanConditionally = true, title = "", defaultCleaner = true }
|
|
537
|
+
) {
|
|
538
|
+
// Rewrite the tag name to div if it's a top level node like body or
|
|
539
|
+
// html to avoid later complications with multiple body tags.
|
|
540
|
+
rewriteTopLevel(article, document);
|
|
541
|
+
|
|
542
|
+
// Drop small images and spacer images
|
|
543
|
+
// Only do this is defaultCleaner is set to true;
|
|
544
|
+
// this can sometimes be too aggressive.
|
|
545
|
+
if (defaultCleaner) cleanImages(article, document);
|
|
546
|
+
|
|
547
|
+
// Drop certain tags like <title>, etc
|
|
548
|
+
// This is -mostly- for cleanliness, not security.
|
|
549
|
+
stripJunkTags(article, document);
|
|
550
|
+
|
|
551
|
+
// H1 tags are typically the article title, which should be extracted
|
|
552
|
+
// by the title extractor instead. If there's less than 3 of them (<3),
|
|
553
|
+
// strip them. Otherwise, turn 'em into H2s.
|
|
554
|
+
cleanHOnes(article, document);
|
|
555
|
+
|
|
556
|
+
// Clean headers
|
|
557
|
+
cleanHeaders(article, document, title);
|
|
558
|
+
|
|
559
|
+
// We used to clean UL's and OL's here, but it was leading to
|
|
560
|
+
// too many in-article lists being removed. Consider a better
|
|
561
|
+
// way to detect menus particularly and remove them.
|
|
562
|
+
// Also optionally running, since it can be overly aggressive.
|
|
563
|
+
if (defaultCleaner) cleanTags(article, document, cleanConditionally);
|
|
564
|
+
|
|
565
|
+
// Remove empty paragraph nodes
|
|
566
|
+
removeEmpty(article, document);
|
|
567
|
+
|
|
568
|
+
// Remove unnecessary attributes
|
|
569
|
+
cleanAttributes(article, document);
|
|
570
|
+
|
|
571
|
+
return article;
|
|
572
|
+
}
|
|
573
|
+
|
|
574
|
+
// After we've calculated scores, loop through all of the possible
|
|
575
|
+
// candidate nodes we found and find the one with the highest score.
|
|
576
|
+
function findTopCandidate(document) {
|
|
577
|
+
let candidate;
|
|
578
|
+
let topScore = 0;
|
|
579
|
+
|
|
580
|
+
document.querySelectorAll("[score]").forEach((node) => {
|
|
581
|
+
// Ignore tags like BR, HR, etc
|
|
582
|
+
if (NON_TOP_CANDIDATE_TAGS_RE.test(node.tagName)) {
|
|
583
|
+
return;
|
|
584
|
+
}
|
|
585
|
+
|
|
586
|
+
const score = getScore(node);
|
|
587
|
+
|
|
588
|
+
if (score > topScore) {
|
|
589
|
+
topScore = score;
|
|
590
|
+
candidate = node;
|
|
591
|
+
}
|
|
592
|
+
});
|
|
593
|
+
|
|
594
|
+
// If we don't have a candidate, return the body
|
|
595
|
+
// or whatever the first element is
|
|
596
|
+
if (!candidate) {
|
|
597
|
+
return document.body || document.querySelector("*");
|
|
598
|
+
}
|
|
599
|
+
|
|
600
|
+
candidate = mergeSiblings(candidate, topScore, document);
|
|
601
|
+
|
|
602
|
+
return candidate;
|
|
603
|
+
}
|
|
604
|
+
|
|
605
|
+
// gets and returns the score if it exists
|
|
606
|
+
// if not, initializes a score based on
|
|
607
|
+
// the node's tag type
|
|
608
|
+
function getOrInitScore(node, document, weightNodes = true) {
|
|
609
|
+
let score = getScore(node);
|
|
610
|
+
|
|
611
|
+
if (score) {
|
|
612
|
+
return score;
|
|
613
|
+
}
|
|
614
|
+
|
|
615
|
+
score = scoreNode(node);
|
|
616
|
+
|
|
617
|
+
if (weightNodes) {
|
|
618
|
+
score += getWeight(node);
|
|
619
|
+
}
|
|
620
|
+
|
|
621
|
+
addToParent(node, document, score);
|
|
622
|
+
|
|
623
|
+
return score;
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
// Get the score of a node based on its className and id.
|
|
627
|
+
function getWeight(node) {
|
|
628
|
+
const classes = node.getAttribute("class");
|
|
629
|
+
const id = node.getAttribute("id");
|
|
630
|
+
let score = 0;
|
|
631
|
+
|
|
632
|
+
if (id) {
|
|
633
|
+
// if id exists, try to score on both positive and negative
|
|
634
|
+
if (POSITIVE_SCORE_RE.test(id)) {
|
|
635
|
+
score += 25;
|
|
636
|
+
}
|
|
637
|
+
if (NEGATIVE_SCORE_RE.test(id)) {
|
|
638
|
+
score -= 25;
|
|
639
|
+
}
|
|
640
|
+
}
|
|
641
|
+
|
|
642
|
+
if (classes) {
|
|
643
|
+
if (score === 0) {
|
|
644
|
+
// if classes exist and id did not contribute to score
|
|
645
|
+
// try to score on both positive and negative
|
|
646
|
+
if (POSITIVE_SCORE_RE.test(classes)) {
|
|
647
|
+
score += 25;
|
|
648
|
+
}
|
|
649
|
+
if (NEGATIVE_SCORE_RE.test(classes)) {
|
|
650
|
+
score -= 25;
|
|
651
|
+
}
|
|
652
|
+
}
|
|
653
|
+
|
|
654
|
+
// even if score has been set by id, add score for
|
|
655
|
+
// possible photo matches
|
|
656
|
+
// "try to keep photos if we can"
|
|
657
|
+
if (PHOTO_HINTS_RE.test(classes)) {
|
|
658
|
+
score += 10;
|
|
659
|
+
}
|
|
660
|
+
|
|
661
|
+
// add 25 if class matches entry-content-asset,
|
|
662
|
+
// a class apparently instructed for use in the
|
|
663
|
+
// Readability publisher guidelines
|
|
664
|
+
// https://www.readability.com/developers/guidelines
|
|
665
|
+
if (READABILITY_ASSET.test(classes)) {
|
|
666
|
+
score += 25;
|
|
667
|
+
}
|
|
668
|
+
}
|
|
669
|
+
|
|
670
|
+
return score;
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
/**
|
|
674
|
+
* Checks if a given text appears to have an ending sentence within it.
|
|
675
|
+
* @param {string} text - The text to check for sentence endings.
|
|
676
|
+
* @returns {boolean} True if the text appears to have an ending sentence, false otherwise.
|
|
677
|
+
*/
|
|
678
|
+
function hasSentenceEnd(text) {
|
|
679
|
+
return new RegExp(".( |$)").test(text);
|
|
680
|
+
}
|
|
681
|
+
|
|
682
|
+
/**
|
|
683
|
+
* Merges siblings of the top candidate that are decently scored.
|
|
684
|
+
* This function looks through the siblings of the top candidate to see if any of them
|
|
685
|
+
* are decently scored. If they are, they may be split parts of the content
|
|
686
|
+
* (like two divs, a preamble and a body).
|
|
687
|
+
*
|
|
688
|
+
* @param {Element} candidate - The top candidate element.
|
|
689
|
+
* @param {number} topScore - The score of the top candidate.
|
|
690
|
+
* @param {Document} document - The document object.
|
|
691
|
+
* @returns {Element} The candidate element, potentially with merged siblings.
|
|
692
|
+
*/
|
|
693
|
+
function mergeSiblings(candidate, topScore, document) {
|
|
694
|
+
if (!candidate.parentNode) {
|
|
695
|
+
return candidate;
|
|
696
|
+
}
|
|
697
|
+
|
|
698
|
+
const siblingScoreThreshold = Math.max(10, topScore * 0.25);
|
|
699
|
+
const wrappingDiv = document.createElement("div");
|
|
700
|
+
|
|
701
|
+
Array.from(candidate.parentNode.children).forEach((sibling) => {
|
|
702
|
+
// Ignore tags like BR, HR, etc
|
|
703
|
+
if (NON_TOP_CANDIDATE_TAGS_RE.test(sibling.tagName)) {
|
|
704
|
+
return null;
|
|
705
|
+
}
|
|
706
|
+
|
|
707
|
+
const siblingScore = getScore(sibling);
|
|
708
|
+
if (siblingScore) {
|
|
709
|
+
if (sibling === candidate) {
|
|
710
|
+
wrappingDiv.appendChild(sibling);
|
|
711
|
+
} else {
|
|
712
|
+
let contentBonus = 0;
|
|
713
|
+
const density = linkDensity(sibling);
|
|
714
|
+
|
|
715
|
+
// If sibling has a very low link density,
|
|
716
|
+
// give it a small bonus
|
|
717
|
+
if (density < 0.05) {
|
|
718
|
+
contentBonus += 20;
|
|
719
|
+
}
|
|
720
|
+
|
|
721
|
+
// If sibling has a high link density,
|
|
722
|
+
// give it a penalty
|
|
723
|
+
if (density >= 0.5) {
|
|
724
|
+
contentBonus -= 20;
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
// If sibling node has the same class as
|
|
728
|
+
// candidate, give it a bonus
|
|
729
|
+
if (sibling.getAttribute("class") === candidate.getAttribute("class")) {
|
|
730
|
+
contentBonus += topScore * 0.2;
|
|
731
|
+
}
|
|
732
|
+
|
|
733
|
+
const newScore = siblingScore + contentBonus;
|
|
734
|
+
|
|
735
|
+
if (newScore >= siblingScoreThreshold) {
|
|
736
|
+
return wrappingDiv.appendChild(sibling);
|
|
737
|
+
}
|
|
738
|
+
if (sibling.tagName === "P") {
|
|
739
|
+
const siblingContent = sibling.textContent;
|
|
740
|
+
const siblingContentLength = textLength(siblingContent);
|
|
741
|
+
|
|
742
|
+
if (siblingContentLength > 80 && density < 0.25) {
|
|
743
|
+
return wrappingDiv.appendChild(sibling);
|
|
744
|
+
}
|
|
745
|
+
if (
|
|
746
|
+
siblingContentLength <= 80 &&
|
|
747
|
+
density === 0 &&
|
|
748
|
+
hasSentenceEnd(siblingContent)
|
|
749
|
+
) {
|
|
750
|
+
return wrappingDiv.appendChild(sibling);
|
|
751
|
+
}
|
|
752
|
+
}
|
|
753
|
+
}
|
|
754
|
+
}
|
|
755
|
+
|
|
756
|
+
return null;
|
|
757
|
+
});
|
|
758
|
+
|
|
759
|
+
if (
|
|
760
|
+
wrappingDiv.children.length === 1 &&
|
|
761
|
+
wrappingDiv?.firstElementChild === candidate
|
|
762
|
+
) {
|
|
763
|
+
return candidate;
|
|
764
|
+
}
|
|
765
|
+
|
|
766
|
+
return wrappingDiv;
|
|
767
|
+
}
|
|
768
|
+
|
|
769
|
+
function cleanTags(article, document) {
|
|
770
|
+
const CLEAN_CONDITIONALLY_TAGS = [
|
|
771
|
+
"ul",
|
|
772
|
+
"ol",
|
|
773
|
+
"table",
|
|
774
|
+
"div",
|
|
775
|
+
"button",
|
|
776
|
+
"form",
|
|
777
|
+
];
|
|
778
|
+
CLEAN_CONDITIONALLY_TAGS.forEach((tag) => {
|
|
779
|
+
article.querySelectorAll(tag).forEach((node) => {
|
|
780
|
+
const KEEP_CLASS = "parser-keep";
|
|
781
|
+
|
|
782
|
+
if (
|
|
783
|
+
node.classList.contains(KEEP_CLASS) ||
|
|
784
|
+
node.querySelector(`.${KEEP_CLASS}`)
|
|
785
|
+
)
|
|
786
|
+
return;
|
|
787
|
+
|
|
788
|
+
let weight = getScore(node);
|
|
789
|
+
if (!weight) {
|
|
790
|
+
weight = getOrInitScore(node, document);
|
|
791
|
+
setScore(node, document, weight);
|
|
792
|
+
}
|
|
793
|
+
|
|
794
|
+
if (weight < 0) {
|
|
795
|
+
node.parentNode.removeChild(node);
|
|
796
|
+
} else {
|
|
797
|
+
removeUnlessContent(node, document, weight);
|
|
798
|
+
}
|
|
799
|
+
});
|
|
800
|
+
});
|
|
801
|
+
|
|
802
|
+
return document;
|
|
803
|
+
}
|
|
804
|
+
|
|
805
|
+
function cleanHeaders(article, document, title = "") {
|
|
806
|
+
const HEADER_TAGS = ["h2", "h3", "h4", "h5", "h6"];
|
|
807
|
+
HEADER_TAGS.forEach((tag) => {
|
|
808
|
+
article.querySelectorAll(tag).forEach((header) => {
|
|
809
|
+
if (
|
|
810
|
+
header.previousElementSibling &&
|
|
811
|
+
header.previousElementSibling.tagName !== "P"
|
|
812
|
+
) {
|
|
813
|
+
header.parentNode.removeChild(header);
|
|
814
|
+
return;
|
|
815
|
+
}
|
|
816
|
+
|
|
817
|
+
if (normalizeSpaces(header.textContent) === title) {
|
|
818
|
+
header.parentNode.removeChild(header);
|
|
819
|
+
return;
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
if (getWeight(header) < 0) {
|
|
823
|
+
header.parentNode.removeChild(header);
|
|
824
|
+
return;
|
|
825
|
+
}
|
|
826
|
+
});
|
|
827
|
+
});
|
|
828
|
+
|
|
829
|
+
return document;
|
|
830
|
+
}
|