extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,830 @@
1
+ // @ts-nocheck
2
+ /**
3
+ * @module research/extractor/html-to-content/extract-content/extract-content-mercury
4
+ * @description Research library module.
5
+ */
6
+ import { parseHTML } from "linkedom";
7
+ import {
8
+ convertNodeTo,
9
+ stripUnlikelyCandidates,
10
+ convertToParagraphs,
11
+ cleanAttributes,
12
+ cleanHOnes,
13
+ cleanImages,
14
+ removeEmpty,
15
+ rewriteTopLevel,
16
+ stripJunkTags,
17
+ textLength,
18
+ linkDensity,
19
+ removeUnlessContent,
20
+ nodeIsSufficient,
21
+ } from "./extract-content-mercury-utils";
22
+
23
+ /**
24
+ * ### HTML-to-Main-Content Extractor #2
25
+ *
26
+ * 1. The algorithm starts by loading the HTML content using linkedom, a lightweight DOM parser for Node.js.
27
+ * 2. It then applies a series of cleaning and scoring techniques to identify the main content of
28
+ * the page, starting with stripping unlikely candidates (e.g., elements with class names like "comment"
29
+ * or "sidebar").
30
+ * 3. The HTML is converted into a series of paragraph elements, which are then scored based on various
31
+ * factors such as text length, number of commas, and the presence of certain class names or IDs.
32
+ * 4. The algorithm assigns scores to parent and grandparent elements based on the scores of their
33
+ * children, with parents receiving the full score and grandparents receiving half.
34
+ * 5. After scoring, the algorithm finds the top candidate element by selecting the node with the
35
+ * highest score.
36
+ * 6. The top candidate's siblings are then examined to see if they should be included in the main
37
+ * content, based on their scores and other factors like link density.
38
+ * 7. The algorithm then cleans the selected content by removing unnecessary tags, attributes, and empty
39
+ * elements.
40
+ * 8. It also handles special cases like cleaning up header tags, images, and other potentially irrelevant
41
+ * content.
42
+ * 9. Throughout the process, the algorithm uses various regular expressions and scoring heuristics to
43
+ * identify positive and negative indicators of content relevance.
44
+ * 10. Finally, the cleaned and extracted content is returned as an HTML string, representing the main
45
+ * body of the article or webpage.
46
+ *
47
+ * [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
48
+ *
49
+ * @param {string} html - The HTML content to extract from.
50
+ * @param {Object} [opts] - The options for content extraction.
51
+ * @param {boolean} opts.stripUnlikelyCandidates default=true - Remove elements that match non-article-
52
+ * like criteria first (e.g., elements with a classname of "comment").
53
+ * @param {boolean} opts.weightNodes default=true - Modify an element's score based on certain classNames or
54
+ * IDs (e.g., subtract if a node has a className of 'comment', add if a node has an ID of 'entry-content').
55
+ * @param {boolean} opts.cleanConditionally default=true - Clean the node to remove superfluous content
56
+ * like forms, ads, etc. Initially, pass in the most restrictive options which will return the highest
57
+ * quality content. On each failure, retry with slightly more lax options.
58
+ * @returns {string} The extracted content as an HTML string, or null if extraction fails.
59
+ * @author [vtempest (2025)](https://github.com/vtempest)
60
+ * Based on [Postlight Mercury Parser (2017-)](https://github.com/postlight/parser/tree/main/src)
61
+ * @example var url = "https://en.wikipedia.org/wiki/David_Hilbert"
62
+ * var html = await (await fetch(url)).text();
63
+ * var content = extractMainContentFromHTML(html);
64
+ * console.log(content); // HTML content of main article body
65
+ * @category Extract
66
+ */
67
+ export function extractMainContentFromHTML2(html, opts) {
68
+ opts = {
69
+ stripUnlikelyCandidates: true,
70
+ weightNodes: true,
71
+ cleanConditionally: true,
72
+ ...opts,
73
+ };
74
+
75
+ if (!html) return;
76
+ const document = parseHTML(html)?.document;
77
+
78
+ if (!document) return;
79
+
80
+ var title = document.querySelector("title")?.textContent.trim();
81
+
82
+ // Cascade through our extraction-specific opts in an ordered fashion,
83
+ // turning them off as we try to extract content.
84
+ let node = getContentNode(document, title, opts);
85
+
86
+ if (nodeIsSufficient(node)) {
87
+ return cleanAndReturnNode(node, document);
88
+ }
89
+
90
+ // We didn't succeed on first pass, one by one disable our
91
+ // extraction opts and try again.
92
+ // eslint-disable-next-line no-restricted-syntax
93
+ for (const key of Reflect.ownKeys(opts).filter((k) => opts[k] === true)) {
94
+ opts[key] = false;
95
+ const { document: newDocument } = parseHTML(html);
96
+
97
+ node = getContentNode(newDocument, title, opts);
98
+
99
+ if (nodeIsSufficient(node)) {
100
+ break;
101
+ }
102
+ }
103
+
104
+ return cleanAndReturnNode(node, document);
105
+ }
106
+
107
+ // A list of tags that should be ignored when trying to find the top candidate
108
+ // for a document.
109
+ const NON_TOP_CANDIDATE_TAGS = [
110
+ "br",
111
+ "b",
112
+ "i",
113
+ "label",
114
+ "hr",
115
+ "area",
116
+ "base",
117
+ "basefont",
118
+ "input",
119
+ "img",
120
+ "link",
121
+ "meta",
122
+ ];
123
+
124
+ const NON_TOP_CANDIDATE_TAGS_RE = new RegExp(
125
+ `^(${NON_TOP_CANDIDATE_TAGS.join("|")})$`,
126
+ "i"
127
+ );
128
+
129
+ const PHOTO_HINTS = ["figure", "photo", "image", "caption"];
130
+ const PHOTO_HINTS_RE = new RegExp(PHOTO_HINTS.join("|"), "i");
131
+
132
+ // A list of strings that denote a positive scoring for this content as being
133
+ // an article container. Checked against className and id.
134
+ //
135
+ // TODO: Perhaps have these scale based on their odds of being quality?
136
+ const POSITIVE_SCORE_HINTS = [
137
+ "article",
138
+ "articlecontent",
139
+ "instapaper_body",
140
+ "blog",
141
+ "body",
142
+ "content",
143
+ "entry-content-asset",
144
+ "entry",
145
+ "hentry",
146
+ "main",
147
+ "Normal",
148
+ "page",
149
+ "pagination",
150
+ "permalink",
151
+ "post",
152
+ "story",
153
+ "text",
154
+ "[-_]copy", // usatoday
155
+ "\\Bcopy",
156
+ ];
157
+
158
+ // The above list, joined into a matching regular expression
159
+ const POSITIVE_SCORE_RE = new RegExp(POSITIVE_SCORE_HINTS.join("|"), "i");
160
+
161
+ // Readability publisher-specific guidelines
162
+ const READABILITY_ASSET = new RegExp("entry-content-asset", "i");
163
+
164
+ const PARAGRAPH_SCORE_TAGS = new RegExp("^(p|li|span|pre)$", "i");
165
+ const CHILD_CONTENT_TAGS = new RegExp("^(td|blockquote|ol|ul|dl)$", "i");
166
+ const BAD_TAGS = new RegExp("^(address|form)$", "i");
167
+
168
+ // A list of strings that denote a negative scoring for this content as being
169
+ // an article container. Checked against className and id.
170
+ //
171
+ // TODO: Perhaps have these scale based on their odds of being quality?
172
+ const NEGATIVE_SCORE_HINTS = [
173
+ "adbox",
174
+ "advert",
175
+ "author",
176
+ "bio",
177
+ "bookmark",
178
+ "bottom",
179
+ "byline",
180
+ "clear",
181
+ "com-",
182
+ "combx",
183
+ "comment",
184
+ "comment\\B",
185
+ "contact",
186
+ "copy",
187
+ "credit",
188
+ "crumb",
189
+ "date",
190
+ "deck",
191
+ "excerpt",
192
+ "featured", // tnr.com has a featured_content which throws us off
193
+ "foot",
194
+ "footer",
195
+ "footnote",
196
+ "graf",
197
+ "head",
198
+ "info",
199
+ "infotext", // newscientist.com copyright
200
+ "instapaper_ignore",
201
+ "jump",
202
+ "linebreak",
203
+ "link",
204
+ "masthead",
205
+ "media",
206
+ "meta",
207
+ "modal",
208
+ "outbrain", // slate.com junk
209
+ "promo",
210
+ "pr_", // autoblog - press release
211
+ "related",
212
+ "respond",
213
+ "roundcontent", // lifehacker restricted content warning
214
+ "scroll",
215
+ "secondary",
216
+ "share",
217
+ "shopping",
218
+ "shoutbox",
219
+ "side",
220
+ "sidebar",
221
+ "sponsor",
222
+ "stamp",
223
+ "sub",
224
+ "summary",
225
+ "tags",
226
+ "tools",
227
+ "widget",
228
+ ];
229
+ // The above list, joined into a matching regular expression
230
+ const NEGATIVE_SCORE_RE = new RegExp(NEGATIVE_SCORE_HINTS.join("|"), "i");
231
+
232
+ // A list of selectors that specify, very clearly, either hNews or other
233
+ // very content-specific style content, like Blogger templates.
234
+ // More examples here: http://microformats.org/wiki/blog-post-formats
235
+ const HNEWS_CONTENT_SELECTORS = [
236
+ [".hentry", ".entry-content"],
237
+ ["entry", ".entry-content"],
238
+ [".entry", ".entry_content"],
239
+ [".post", ".postbody"],
240
+ [".post", ".post_body"],
241
+ [".post", ".post-body"],
242
+ ];
243
+
244
+ /**
245
+ * Normalizes spaces in a given text string.
246
+ * @param {string} text - The text to normalize.
247
+ * @returns {string} The normalized text.
248
+ */
249
+ function normalizeSpaces(text) {
250
+ return text.replace(/\s{2,}(?![^<>]*<\/(pre|code|textarea)>)/g, " ").trim();
251
+ }
252
+
253
+ /**
254
+ * Cleans and returns the HTML of a given node.
255
+ * @param {Node} node - The node to clean and return.
256
+ * @param {Document} document - The document object.
257
+ * @returns {string|null} The cleaned HTML string or null if no node is provided.
258
+ */
259
+ function cleanAndReturnNode(node, document) {
260
+ if (!node) {
261
+ return null;
262
+ }
263
+
264
+ return normalizeSpaces(node.outerHTML);
265
+ }
266
+
267
+ /**
268
+ * Gets the content node from the document.
269
+ * @param {Document} document - The document object.
270
+ * @param {string} title - The title of the document.
271
+ * @param {Object} opts - The options for content extraction.
272
+ * @returns {Node} The content node.
273
+ */
274
+ function getContentNode(document, title, opts) {
275
+ return cleanContent(extractBestNode(document, opts), {
276
+ document,
277
+ cleanConditionally: opts.cleanConditionally,
278
+ title,
279
+ });
280
+ }
281
+
282
+ /**
283
+ * Gets the score of a node.
284
+ * @param {Node} node - The node to get the score from.
285
+ * @returns {number|null} The score of the node or null if no score is set.
286
+ */
287
+ function getScore(node) {
288
+ return parseFloat(node.getAttribute("score")) || null;
289
+ }
290
+
291
+ /**
292
+ * Scores the number of commas in a text.
293
+ * @param {string} text - The text to score.
294
+ * @returns {number} The number of commas in the text.
295
+ */
296
+ function scoreCommas(text) {
297
+ return (text.match(/,/g) || []).length;
298
+ }
299
+
300
+ /**
301
+ * Converts span elements to div elements.
302
+ * @param {Node} node - The node to convert.
303
+ * @param {Document} document - The document object.
304
+ */
305
+ function convertSpans(node, document) {
306
+ if (node?.tagName?.toLowerCase() === "span") {
307
+ // convert spans to divs
308
+ convertNodeTo(node, document, "div");
309
+ }
310
+ }
311
+
312
+ /**
313
+ * Adds a score to a node and its parent elements.
314
+ * @param {Node} node - The node to add the score to.
315
+ * @param {Document} document - The document object.
316
+ * @param {number} score - The score to add.
317
+ */
318
+ function addScoreTo(node, document, score) {
319
+ if (node) {
320
+ convertSpans(node, document);
321
+ addScore(node, document, score);
322
+ }
323
+ }
324
+
325
+ /**
326
+ * Scores paragraph elements in the document.
327
+ * @param {Document} document - The document object.
328
+ * @param {boolean} weightNodes - Whether to weight nodes or not.
329
+ * @returns {Document} The document with scored paragraphs.
330
+ */
331
+ function scorePs(document, weightNodes) {
332
+ document.querySelectorAll("p, pre").forEach((node) => {
333
+ if (!node.hasAttribute("score")) {
334
+ // The raw score for this paragraph, before we add any parent/child
335
+ // scores.
336
+ node = setScore(
337
+ node,
338
+ document,
339
+ getOrInitScore(node, document, weightNodes)
340
+ );
341
+
342
+ const parent = node.parentNode;
343
+ const rawScore = scoreNode(node);
344
+
345
+ addScoreTo(parent, document, rawScore, weightNodes);
346
+ if (parent) {
347
+ // Add half of the individual content score to the
348
+ // grandparent
349
+ addScoreTo(parent.parentNode, document, rawScore / 2, weightNodes);
350
+ }
351
+ }
352
+ });
353
+
354
+ return document;
355
+ }
356
+
357
+ /**
358
+ * Scores the content of the document.
359
+ * @param {Document} document - The document object.
360
+ * @param {boolean} weightNodes - Whether to weight nodes or not.
361
+ * @returns {Document} The document with scored content.
362
+ */
363
+ function scoreContent(document, weightNodes = true) {
364
+ // First, look for special hNews based selectors and give them a big
365
+ // boost, if they exist
366
+ HNEWS_CONTENT_SELECTORS.forEach(([parentSelector, childSelector]) => {
367
+ document
368
+ .querySelectorAll(`${parentSelector} ${childSelector}`)
369
+ .forEach((node) => {
370
+ addScore(node.closest(parentSelector), document, 80);
371
+ });
372
+ });
373
+
374
+ // Doubling this again
375
+ // Previous solution caused a bug
376
+ // in which parents weren't retaining
377
+ // scores. This is not ideal, and
378
+ // should be fixed.
379
+ scorePs(document, weightNodes);
380
+ scorePs(document, weightNodes);
381
+
382
+ return document;
383
+ }
384
+
385
+ /**
386
+ * Scores the length of text.
387
+ * @param {number} textLength - The length of the text.
388
+ * @param {string} tagName - The tag name of the element.
389
+ * @returns {number} The score based on text length.
390
+ */
391
+ function scoreLength(textLength, tagName = "p") {
392
+ const chunks = textLength / 50;
393
+
394
+ if (chunks > 0) {
395
+ let lengthBonus;
396
+
397
+ // No idea why p or pre are being tamped down here
398
+ // but just following the source for now
399
+ // Not even sure why tagName is included here,
400
+ // since this is only being called from the context
401
+ // of scoreParagraph
402
+ if (new RegExp("^(p|pre)$", "i").test(tagName)) {
403
+ lengthBonus = chunks - 2;
404
+ } else {
405
+ lengthBonus = chunks - 1.25;
406
+ }
407
+
408
+ return Math.min(Math.max(lengthBonus, 0), 3);
409
+ }
410
+
411
+ return 0;
412
+ }
413
+
414
+ /**
415
+ * Sets the score attribute of a node.
416
+ * @param {Node} node - The node to set the score on.
417
+ * @param {Document} document - The document object.
418
+ * @param {number} score - The score to set.
419
+ * @returns {Node} The node with the set score.
420
+ * @private
421
+ */
422
+ export function setScore(node, document, score) {
423
+ node.setAttribute("score", score);
424
+ return node;
425
+ }
426
+
427
+ /**
428
+ * Scores a paragraph node.
429
+ * @param {Node} node - The paragraph node to score.
430
+ * @private
431
+ * @returns {number} The score of the paragraph.
432
+ */
433
+ export function scoreParagraph(node) {
434
+ let score = 1;
435
+ const text = node.textContent.trim();
436
+ const textLength = text.length;
437
+
438
+ // If this paragraph is less than 25 characters, don't count it.
439
+ if (textLength < 25) {
440
+ return 0;
441
+ }
442
+
443
+ // Add points for any commas within this paragraph
444
+ score += scoreCommas(text);
445
+
446
+ // For every 50 characters in this paragraph, add another point. Up
447
+ // to 3 points.
448
+ score += scoreLength(textLength);
449
+
450
+ // Articles can end with short paragraphs when people are being clever
451
+ // but they can also end with short paragraphs setting up lists of junk
452
+ // that we strip. This negative tweaks junk setup paragraphs just below
453
+ // the cutoff threshold.
454
+ if (text.slice(-1) === ":") {
455
+ score -= 1;
456
+ }
457
+
458
+ return score;
459
+ }
460
+
461
+ // Score an individual node. Has some smarts for paragraphs, otherwise
462
+ // just scores based on tag.
463
+ function scoreNode(node) {
464
+ const tagName = node.tagName?.toLowerCase();
465
+ // if (!tagName) return 0;
466
+
467
+ // TODO: Consider ordering by most likely.
468
+ // E.g., if divs are a more common tag on a page,
469
+ // Could save doing that regex test on every node \u2013 AP
470
+ if (PARAGRAPH_SCORE_TAGS.test(tagName)) {
471
+ return scoreParagraph(node);
472
+ }
473
+ if (tagName === "div") {
474
+ return 5;
475
+ }
476
+ if (CHILD_CONTENT_TAGS.test(tagName)) {
477
+ return 3;
478
+ }
479
+ if (BAD_TAGS.test(tagName)) {
480
+ return -3;
481
+ }
482
+ if (tagName === "th") {
483
+ return -5;
484
+ }
485
+
486
+ return 0;
487
+ }
488
+
489
+ function addScore(node, document, amount) {
490
+ try {
491
+ const score = getOrInitScore(node, document) + amount;
492
+ setScore(node, document, score);
493
+ } catch (e) {
494
+ // Ignoring; error occurs in scoreNode
495
+ }
496
+
497
+ return node;
498
+ }
499
+
500
+ // Adds 1/4 of a child's score to its parent
501
+ function addToParent(node, document, score) {
502
+ const parent = node.parentNode;
503
+ if (parent) {
504
+ addScore(parent, document, score * 0.25);
505
+ }
506
+
507
+ return node;
508
+ }
509
+
510
+ // Using a variety of scoring techniques, extract the content most
511
+ // likely to be article text.
512
+ //
513
+ // If strip_unlikely_candidates is True, remove any elements that
514
+ // match certain criteria first. (Like, does this element have a
515
+ // classname of "comment")
516
+ //
517
+ // If weight_nodes is True, use classNames and IDs to determine the
518
+ // worthiness of nodes.
519
+ //
520
+ // Returns a DOM node
521
+ function extractBestNode(document, opts) {
522
+ if (opts.stripUnlikelyCandidates) {
523
+ document = stripUnlikelyCandidates(document);
524
+ }
525
+
526
+ document = convertToParagraphs(document);
527
+ document = scoreContent(document, opts.weightNodes);
528
+ const topCandidate = findTopCandidate(document);
529
+
530
+ return topCandidate;
531
+ }
532
+
533
+ // Clean our article content, returning a new, cleaned node.
534
+ function cleanContent(
535
+ article,
536
+ { document, cleanConditionally = true, title = "", defaultCleaner = true }
537
+ ) {
538
+ // Rewrite the tag name to div if it's a top level node like body or
539
+ // html to avoid later complications with multiple body tags.
540
+ rewriteTopLevel(article, document);
541
+
542
+ // Drop small images and spacer images
543
+ // Only do this is defaultCleaner is set to true;
544
+ // this can sometimes be too aggressive.
545
+ if (defaultCleaner) cleanImages(article, document);
546
+
547
+ // Drop certain tags like <title>, etc
548
+ // This is -mostly- for cleanliness, not security.
549
+ stripJunkTags(article, document);
550
+
551
+ // H1 tags are typically the article title, which should be extracted
552
+ // by the title extractor instead. If there's less than 3 of them (<3),
553
+ // strip them. Otherwise, turn 'em into H2s.
554
+ cleanHOnes(article, document);
555
+
556
+ // Clean headers
557
+ cleanHeaders(article, document, title);
558
+
559
+ // We used to clean UL's and OL's here, but it was leading to
560
+ // too many in-article lists being removed. Consider a better
561
+ // way to detect menus particularly and remove them.
562
+ // Also optionally running, since it can be overly aggressive.
563
+ if (defaultCleaner) cleanTags(article, document, cleanConditionally);
564
+
565
+ // Remove empty paragraph nodes
566
+ removeEmpty(article, document);
567
+
568
+ // Remove unnecessary attributes
569
+ cleanAttributes(article, document);
570
+
571
+ return article;
572
+ }
573
+
574
+ // After we've calculated scores, loop through all of the possible
575
+ // candidate nodes we found and find the one with the highest score.
576
+ function findTopCandidate(document) {
577
+ let candidate;
578
+ let topScore = 0;
579
+
580
+ document.querySelectorAll("[score]").forEach((node) => {
581
+ // Ignore tags like BR, HR, etc
582
+ if (NON_TOP_CANDIDATE_TAGS_RE.test(node.tagName)) {
583
+ return;
584
+ }
585
+
586
+ const score = getScore(node);
587
+
588
+ if (score > topScore) {
589
+ topScore = score;
590
+ candidate = node;
591
+ }
592
+ });
593
+
594
+ // If we don't have a candidate, return the body
595
+ // or whatever the first element is
596
+ if (!candidate) {
597
+ return document.body || document.querySelector("*");
598
+ }
599
+
600
+ candidate = mergeSiblings(candidate, topScore, document);
601
+
602
+ return candidate;
603
+ }
604
+
605
+ // gets and returns the score if it exists
606
+ // if not, initializes a score based on
607
+ // the node's tag type
608
+ function getOrInitScore(node, document, weightNodes = true) {
609
+ let score = getScore(node);
610
+
611
+ if (score) {
612
+ return score;
613
+ }
614
+
615
+ score = scoreNode(node);
616
+
617
+ if (weightNodes) {
618
+ score += getWeight(node);
619
+ }
620
+
621
+ addToParent(node, document, score);
622
+
623
+ return score;
624
+ }
625
+
626
+ // Get the score of a node based on its className and id.
627
+ function getWeight(node) {
628
+ const classes = node.getAttribute("class");
629
+ const id = node.getAttribute("id");
630
+ let score = 0;
631
+
632
+ if (id) {
633
+ // if id exists, try to score on both positive and negative
634
+ if (POSITIVE_SCORE_RE.test(id)) {
635
+ score += 25;
636
+ }
637
+ if (NEGATIVE_SCORE_RE.test(id)) {
638
+ score -= 25;
639
+ }
640
+ }
641
+
642
+ if (classes) {
643
+ if (score === 0) {
644
+ // if classes exist and id did not contribute to score
645
+ // try to score on both positive and negative
646
+ if (POSITIVE_SCORE_RE.test(classes)) {
647
+ score += 25;
648
+ }
649
+ if (NEGATIVE_SCORE_RE.test(classes)) {
650
+ score -= 25;
651
+ }
652
+ }
653
+
654
+ // even if score has been set by id, add score for
655
+ // possible photo matches
656
+ // "try to keep photos if we can"
657
+ if (PHOTO_HINTS_RE.test(classes)) {
658
+ score += 10;
659
+ }
660
+
661
+ // add 25 if class matches entry-content-asset,
662
+ // a class apparently instructed for use in the
663
+ // Readability publisher guidelines
664
+ // https://www.readability.com/developers/guidelines
665
+ if (READABILITY_ASSET.test(classes)) {
666
+ score += 25;
667
+ }
668
+ }
669
+
670
+ return score;
671
+ }
672
+
673
+ /**
674
+ * Checks if a given text appears to have an ending sentence within it.
675
+ * @param {string} text - The text to check for sentence endings.
676
+ * @returns {boolean} True if the text appears to have an ending sentence, false otherwise.
677
+ */
678
+ function hasSentenceEnd(text) {
679
+ return new RegExp(".( |$)").test(text);
680
+ }
681
+
682
+ /**
683
+ * Merges siblings of the top candidate that are decently scored.
684
+ * This function looks through the siblings of the top candidate to see if any of them
685
+ * are decently scored. If they are, they may be split parts of the content
686
+ * (like two divs, a preamble and a body).
687
+ *
688
+ * @param {Element} candidate - The top candidate element.
689
+ * @param {number} topScore - The score of the top candidate.
690
+ * @param {Document} document - The document object.
691
+ * @returns {Element} The candidate element, potentially with merged siblings.
692
+ */
693
+ function mergeSiblings(candidate, topScore, document) {
694
+ if (!candidate.parentNode) {
695
+ return candidate;
696
+ }
697
+
698
+ const siblingScoreThreshold = Math.max(10, topScore * 0.25);
699
+ const wrappingDiv = document.createElement("div");
700
+
701
+ Array.from(candidate.parentNode.children).forEach((sibling) => {
702
+ // Ignore tags like BR, HR, etc
703
+ if (NON_TOP_CANDIDATE_TAGS_RE.test(sibling.tagName)) {
704
+ return null;
705
+ }
706
+
707
+ const siblingScore = getScore(sibling);
708
+ if (siblingScore) {
709
+ if (sibling === candidate) {
710
+ wrappingDiv.appendChild(sibling);
711
+ } else {
712
+ let contentBonus = 0;
713
+ const density = linkDensity(sibling);
714
+
715
+ // If sibling has a very low link density,
716
+ // give it a small bonus
717
+ if (density < 0.05) {
718
+ contentBonus += 20;
719
+ }
720
+
721
+ // If sibling has a high link density,
722
+ // give it a penalty
723
+ if (density >= 0.5) {
724
+ contentBonus -= 20;
725
+ }
726
+
727
+ // If sibling node has the same class as
728
+ // candidate, give it a bonus
729
+ if (sibling.getAttribute("class") === candidate.getAttribute("class")) {
730
+ contentBonus += topScore * 0.2;
731
+ }
732
+
733
+ const newScore = siblingScore + contentBonus;
734
+
735
+ if (newScore >= siblingScoreThreshold) {
736
+ return wrappingDiv.appendChild(sibling);
737
+ }
738
+ if (sibling.tagName === "P") {
739
+ const siblingContent = sibling.textContent;
740
+ const siblingContentLength = textLength(siblingContent);
741
+
742
+ if (siblingContentLength > 80 && density < 0.25) {
743
+ return wrappingDiv.appendChild(sibling);
744
+ }
745
+ if (
746
+ siblingContentLength <= 80 &&
747
+ density === 0 &&
748
+ hasSentenceEnd(siblingContent)
749
+ ) {
750
+ return wrappingDiv.appendChild(sibling);
751
+ }
752
+ }
753
+ }
754
+ }
755
+
756
+ return null;
757
+ });
758
+
759
+ if (
760
+ wrappingDiv.children.length === 1 &&
761
+ wrappingDiv?.firstElementChild === candidate
762
+ ) {
763
+ return candidate;
764
+ }
765
+
766
+ return wrappingDiv;
767
+ }
768
+
769
+ function cleanTags(article, document) {
770
+ const CLEAN_CONDITIONALLY_TAGS = [
771
+ "ul",
772
+ "ol",
773
+ "table",
774
+ "div",
775
+ "button",
776
+ "form",
777
+ ];
778
+ CLEAN_CONDITIONALLY_TAGS.forEach((tag) => {
779
+ article.querySelectorAll(tag).forEach((node) => {
780
+ const KEEP_CLASS = "parser-keep";
781
+
782
+ if (
783
+ node.classList.contains(KEEP_CLASS) ||
784
+ node.querySelector(`.${KEEP_CLASS}`)
785
+ )
786
+ return;
787
+
788
+ let weight = getScore(node);
789
+ if (!weight) {
790
+ weight = getOrInitScore(node, document);
791
+ setScore(node, document, weight);
792
+ }
793
+
794
+ if (weight < 0) {
795
+ node.parentNode.removeChild(node);
796
+ } else {
797
+ removeUnlessContent(node, document, weight);
798
+ }
799
+ });
800
+ });
801
+
802
+ return document;
803
+ }
804
+
805
+ function cleanHeaders(article, document, title = "") {
806
+ const HEADER_TAGS = ["h2", "h3", "h4", "h5", "h6"];
807
+ HEADER_TAGS.forEach((tag) => {
808
+ article.querySelectorAll(tag).forEach((header) => {
809
+ if (
810
+ header.previousElementSibling &&
811
+ header.previousElementSibling.tagName !== "P"
812
+ ) {
813
+ header.parentNode.removeChild(header);
814
+ return;
815
+ }
816
+
817
+ if (normalizeSpaces(header.textContent) === title) {
818
+ header.parentNode.removeChild(header);
819
+ return;
820
+ }
821
+
822
+ if (getWeight(header) < 0) {
823
+ header.parentNode.removeChild(header);
824
+ return;
825
+ }
826
+ });
827
+ });
828
+
829
+ return document;
830
+ }