extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,282 @@
1
+ // @ts-nocheck
2
+ /**
3
+ * @module research/extractor/html-to-content/html-to-basic-html
4
+ * @description Research library module.
5
+ */
6
+ import {
7
+ convertURLSafeHTMLToHTML,
8
+ convertURLToAbsoluteURL,
9
+ convertMarkdownToHTML,
10
+ } from "./html-utils";
11
+
12
+ /**
13
+ * Strip HTML to ~30 basic markup HTML tags, lists, tables, images.
14
+ * Convert anchors and relative urls to absolute urls. Basic HTML supports the same
15
+ * elements as Markdown, which is used in writing plain text. Markdown is converted
16
+ * to HTML anyways to display it, and it is better to edit basic HTML in a rich text editor.
17
+ *
18
+ * [Mozilla DOM Reference](https://developer.mozilla.org/en-US/docs/Web/API/Document_Object_Model) <br />
19
+ * [Source Code of Browser HTML DOM](https://chromium.googlesource.com/chromium/src/+/HEAD/third_party/blink/renderer/core/dom/) <br />
20
+ * [RegExp JS V8 Code](https://github.com/v8/v8/blob/94cde7c7f3fffc62f621e43f65be3d517b8a9f3d/src/regexp/regexp-compiler.cc#L3827)
21
+ * @param {string} html Any page's HTML to process
22
+ * @param {Object} [options]
23
+ * @param {boolean} options.images default=true - Whether to include images
24
+ * @param {boolean} options.links default=true - Whether to include links
25
+ * @param {boolean} options.videos default=true - Whether to include videos or not
26
+ * @param {boolean} options.formatting default=true - Whether to include formatting
27
+ * @param {string} options.url base URL for converting relative URLs to absolute
28
+ * @param {string} options.allowTags default="br,p,u,b,i ,em,strong,h1,h2,h3,h4, h5,h6,blockquote,
29
+ * code,ul,ol,li,dd,dl, table,th,tr,td,sub,sup" - Comma-separated list of allowed HTML tags.
30
+ * @param {string} options.allowedAttributes default="text,tag,href, src,type,width, height,id,data"
31
+ * List of allowed HTML attributes
32
+ * @returns {string} basic text formatting html
33
+ * @author [vtempest (2025)](https://github.com/vtempest)
34
+ * @category HTML Utilities
35
+ */
36
+ export function convertHTMLToBasicHTML(html, options = {}) {
37
+ var {
38
+ images = true,
39
+ links = true,
40
+ videos = true,
41
+ formatting = true,
42
+ url = "",
43
+ openLinksNewWindow = false,
44
+ allowTags = "br,p,u,b,i,em,strong,h1,h2,h3,h4,h5,h6,blockquote,code,\
45
+ ul,ol,li,dd,dl,table,th,tr,td,thead,tbody,sub,sup,math,iframe",
46
+ allowedAttributes = "href,src,type,width,height,id,data,target",
47
+ } = options;
48
+
49
+ // return convertMarkdownToHTML(convertMarkdownToHTML(html, false), true)
50
+
51
+ allowTags = allowTags.split(",");
52
+ if (links) allowTags.push("a");
53
+ if (images) allowTags.push("img");
54
+ if (videos)
55
+ allowTags = allowTags.concat("video,source,embed,object".split(","));
56
+
57
+ if (!formatting) allowTags = ["text"];
58
+ allowTags.push("text");
59
+
60
+ allowedAttributes = allowedAttributes
61
+ .split(",")
62
+ .concat("text,tagName".split(","));
63
+
64
+ // Convert html string to array like [{tag:"p",attr:""},{text:""}]
65
+ var basicHtml = convertHTMLToTokens(html);
66
+ if (!basicHtml) return;
67
+
68
+ basicHtml = basicHtml
69
+ .filter(
70
+ (token) =>
71
+ token.text ||
72
+ (token.tagName[0] == "/"
73
+ ? allowTags.includes(token.tagName?.substring(1)?.toLowerCase())
74
+ : allowTags.includes(token.tagName?.toLowerCase()))
75
+ )
76
+ .map((el) => {
77
+ for (var key of Object.keys(el))
78
+ if (!allowedAttributes.includes(key)) delete el[key];
79
+
80
+ var urlValue = el.href || el.src;
81
+
82
+ //non-anchor links should be opened in new window
83
+ if (urlValue && openLinksNewWindow)
84
+ if (!urlValue.startsWith("#")) el.target = "_blank";
85
+
86
+ // remove broken images
87
+ if (el.tagName?.toLowerCase() == "img") {
88
+ if (!el.src || el.src.startsWith("data:")) return false;
89
+ }
90
+
91
+ //convert relative urls to absolute urls
92
+ if (el.src) {
93
+ el.src = new URL(urlValue, url).href;
94
+ }
95
+ if (el.href) el.href = new URL(urlValue, url).href;
96
+
97
+ // convertURLToAbsoluteURL(url, urlValue);
98
+
99
+ return el;
100
+ })
101
+ .filter(Boolean)
102
+ .reduce((acc, el) => {
103
+ acc += el.text
104
+ ? `${el.text}`
105
+ : `<${el.tagName}${Object.keys(el).length > 1 ? " " : ""}${Object.keys(
106
+ el
107
+ )
108
+ .filter((key) => key != "tagName" && key != "text")
109
+ .map((key) => `${key}="${el[key]}"`)
110
+ .join(" ")}>`;
111
+ return acc;
112
+ }, "")
113
+ .replace(/<p><\/p>/g, " ")
114
+ .replace(/[\r\n\t]+/g, " ") //remove linebreaks
115
+ .replace(/ \s+/g, " ");
116
+
117
+ basicHtml = convertURLSafeHTMLToHTML(basicHtml).replace(/&nbsp;/g, " ");
118
+
119
+ // // CNN news edge case of data=attr <> inside of attr
120
+ // const reHTMLInsideDataAttr =
121
+ // /(["'])(?:(?!(?:\1|<)).)*?(?:<(?:(?!["'<>]).)*?>)?(?:(?!(?:\1|<)).)*?\1/gis;
122
+ // if (reHTMLInsideDataAttr.test(html))
123
+ // html = html.replaceAll(reHTMLInsideDataAttr, "");
124
+
125
+ return basicHtml;
126
+ }
127
+
128
+ /**
129
+ * Convert html string to array of JSON Objects tokens to translate,
130
+ * convert, or filter all elements.
131
+ * Flat array is faster than DOMParser which uses nested trees.
132
+ * @param {string} html
133
+ * @returns {array} Example [{"tag": "img","src": ""}, ...]
134
+
135
+ * @private
136
+ */
137
+ export function convertHTMLToTokens(html) {
138
+ if (!html) return;
139
+ var dom = [];
140
+
141
+ //remove script style to prevent it from counting as text
142
+ html = html
143
+ .replace(/(<(noscript|script|style)\b[^>]*>).*?(<\/\2>)/gis, "$1$3")
144
+ .replace(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi, "")
145
+ .replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, "")
146
+ .replace(/<!--[\s\S]*?-->/g, "");
147
+
148
+ const reHTMLInsideDataAttr =
149
+ /(["'])(?:(?!(?:\1|<)).)*?(?:<(?:(?!["'<>]).)*?>)?(?:(?!(?:\1|<)).)*?\1/gis;
150
+
151
+ var chunks = html.split("<");
152
+
153
+ for (var chunk of chunks) {
154
+ if (!chunk.includes(">")) continue;
155
+
156
+ var [element, text] = chunk.split(">");
157
+
158
+ if (element.includes("<")) {
159
+ if (reHTMLInsideDataAttr.test(html)) {
160
+ html = html.replaceAll(reHTMLInsideDataAttr, "");
161
+ return convertHTMLToTokens(html);
162
+ }
163
+ }
164
+
165
+ //if closing tag, add it but dont stop and also in next step
166
+ // add text after </a> as text node
167
+ if (element[0] == "/") dom.push({ tagName: element });
168
+
169
+ if (element[0] == "!") continue; //skip comments
170
+
171
+ var domElement = {};
172
+ //if has attributes
173
+ var attributesIndex = element.indexOf(" ");
174
+
175
+ if (attributesIndex == -1) {
176
+ domElement.tagName = element;
177
+ } else {
178
+ //has attributes
179
+
180
+ var tag = element.substring(0, attributesIndex);
181
+ domElement.tagName = tag;
182
+ // there can be spaces and <> inside of attr strings
183
+ //TODO cnn news edge case of data=attr <> inside of attr
184
+ //insert attr into domElement
185
+ element
186
+ .substring(attributesIndex)
187
+ .match(/ \w+=("(?:[^"\\]|\\.\s)*")/g)
188
+ ?.forEach((attr) => {
189
+ attr = attr.trim();
190
+
191
+ var key = attr.split("=")[0];
192
+ var value = attr.slice(key.length + 2, -1);
193
+ if (key == "srcset") {
194
+ key = "src";
195
+ value = value.split(",")[0].trim().split(" ")[0];
196
+ }
197
+
198
+ if (key && value) domElement[key] = value?.replace(/"/g, "");
199
+ });
200
+ }
201
+
202
+ // style and script, add their content to "content" and dont treat as text
203
+ if (["style", "script", "noscript"].includes(domElement.tagName)) {
204
+ domElement.content = text;
205
+ continue;
206
+ }
207
+
208
+ if (element[0] != "/") dom.push(domElement);
209
+
210
+ //if text node push as {text:""}
211
+ if (text) dom.push({ tagName: "text", text: text });
212
+ }
213
+
214
+ // dom = addDOMFunctions(dom);
215
+
216
+ return dom;
217
+ }
218
+
219
+ export function addDOMFunctions(domObject) {
220
+ //assign to all objects for easy chain calling
221
+ domObject = domObject || Object.prototype;
222
+
223
+ domObject = Object.assign(domObject, {
224
+ querySelectorAll: function (querySelector) {
225
+ if (querySelector.includes(","))
226
+ //multiple selectors
227
+ var selectors = querySelector.split(",").map((sel) => sel.trim());
228
+
229
+ var type = selector[0];
230
+ selector = selector.substring(1);
231
+
232
+ if (type == ".")
233
+ //class
234
+ return this.filter((el) => el.class == selector);
235
+ if (type == "#")
236
+ //id
237
+ return this.filter((el) => el.id == selector);
238
+ if (type == "[")
239
+ //attribute
240
+ return this.filter((el) => el[selector] !== undefined);
241
+ //tag
242
+ else return this.filter(({ tagName }) => tagName == selector);
243
+ },
244
+ querySelector: function (selector) {
245
+ return this.querySelectorAll(selector)[0];
246
+ },
247
+ getTextContent: function () {
248
+ return this.reduce(
249
+ (acc, { text }) => (acc += text ? text + "\n" : ""),
250
+ ""
251
+ );
252
+ },
253
+ getAttribute: function (attr) {
254
+ return this.map((el) => el[attr]).filter(Boolean);
255
+ },
256
+ getElementsByTagName: function (tag) {
257
+ return this.filter(({ tagName: t }) => t == tag).map(addDOMFunctions);
258
+ },
259
+ getElementsByClassName: function (className) {
260
+ return this.filter((el) => el.class == className);
261
+ },
262
+ getElementById: function (id) {
263
+ return this.filter((el) => el.id == id);
264
+ },
265
+ getInnerHTML: function () {
266
+ return this.reduce((acc, el) => {
267
+ acc += el.text
268
+ ? `${el.text}`
269
+ : `<${el.tagName} ${Object.keys(el)
270
+ .filter((key) => key != "tagName" && key != "text")
271
+ .map((key) => `${key}="${el[key]}"`)
272
+ .join(" ")}>`;
273
+ return acc;
274
+ }, "");
275
+ },
276
+ });
277
+
278
+ domObject.innerHTML = domObject.getInnerHTML();
279
+ domObject.textContent = domObject.getTextContent();
280
+
281
+ return domObject;
282
+ }
@@ -0,0 +1,97 @@
1
+ // @ts-nocheck
2
+ /**
3
+ * @fileoverview Utility to extract core text content from HTML documents.
4
+ * Cleans boilerplate (nav, footer, ads) to produce clean Markdown or text.
5
+ */
6
+ import { parseHTML } from "linkedom";
7
+ import { extractCite } from "../html-to-cite/extract-cite";
8
+ import { convertHTMLToBasicHTML } from "./html-to-basic-html";
9
+ import { extractHumanName } from "../html-to-cite/human-names-recognize";
10
+ import { extractMainContentFromHTML } from "./extract-content/extract-content-readability";
11
+ import { extractMainContentFromHTML2 } from "./extract-content/extract-content-mercury";
12
+
13
+ /**
14
+ * Extracts the main content and citation information from a document or HTML string
15
+ * @param {string|object} documentOrHTML - The document or HTML string to extract content from
16
+ * @param {Object} options - Optional configuration options
17
+ * @param {boolean} options.images default=true - Whether to include images in the extracted content
18
+ * @param {boolean} options.links default=true - Whether to include links in the extracted content
19
+ * @param {boolean} options.formatting default=true - Whether to preserve formatting in the extracted content
20
+ * @param {string} options.url The URL of the original document, if available, for absolutify-ing URLs
21
+ * @param {boolean} options.useExtractor2 default=false -
22
+ * false uses Mozilla Readability, true uses Postlight Mercury.
23
+ * then use the alternate if the first returns less than 200 characters
24
+ * @returns {Object} The extracted content and citation information
25
+ * @property {string} title - The title of the document
26
+ * @property {string} author_cite - The full citation for the author
27
+ * @property {string} author_short - A shortened version of the author's name
28
+ * @property {string} author - The author's name
29
+ * @property {string} date - The publication date
30
+ * @property {string} source - The source of the document
31
+ * @property {string} html - The extracted HTML content
32
+ * @author [vtempest (2025)](https://github.com/vtempest)
33
+ */
34
+ export function extractContentAndCite(documentOrHTML, options = {}) {
35
+ const {
36
+ images = true,
37
+ links = true,
38
+ formatting = true,
39
+ url = "",
40
+ useExtractor2 = 1,
41
+ minExtractedLength = 400,
42
+ } = options;
43
+
44
+ var html =
45
+ typeof documentOrHTML === "string"
46
+ ? documentOrHTML
47
+ : documentOrHTML?.documentElement?.innerHTML;
48
+
49
+ if (!html) return { error: "No HTML found" };
50
+
51
+ try {
52
+ var content1 = extractMainContentFromHTML(html, options);
53
+ } catch (e) {
54
+ console.log(e);
55
+ }
56
+ try {
57
+ var content2 = extractMainContentFromHTML2(html, options);
58
+ } catch (e) {
59
+ console.log(e);
60
+ }
61
+
62
+ //compare content lengths
63
+ var content = content1?.length > content2?.length ? content1 : content2;
64
+
65
+ // check if html is too short, if so use basic html
66
+ if (content?.replace(/<[^>]*>/g, "").length < minExtractedLength)
67
+ content = html;
68
+
69
+ //cite
70
+ var { author, author_cite, author_short, date, title, source } = extractCite(
71
+ html,
72
+ options
73
+ );
74
+
75
+ html = convertHTMLToBasicHTML(content, options);
76
+
77
+ return {
78
+ title,
79
+ author_cite,
80
+ author_short,
81
+ author,
82
+ date,
83
+ source,
84
+ html,
85
+ };
86
+ }
87
+ /**
88
+ * @typedef {Object} ExtractedContent
89
+ * @property {string} title - The title of the content
90
+ * @property {string} author_cite - The full citation for the author
91
+ * @property {string} author_short - A shortened version of the author's name
92
+ * @property {string} author - The author's name
93
+ * @property {string} date - The publication date
94
+ * @property {string} source - The source of the content
95
+ * @property {string} html - The extracted main content in HTML format
96
+ * @private
97
+ */