extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
/**
|
|
3
|
+
* @module research/extractor/html-to-content/html-to-basic-html
|
|
4
|
+
* @description Research library module.
|
|
5
|
+
*/
|
|
6
|
+
import {
|
|
7
|
+
convertURLSafeHTMLToHTML,
|
|
8
|
+
convertURLToAbsoluteURL,
|
|
9
|
+
convertMarkdownToHTML,
|
|
10
|
+
} from "./html-utils";
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Strip HTML to ~30 basic markup HTML tags, lists, tables, images.
|
|
14
|
+
* Convert anchors and relative urls to absolute urls. Basic HTML supports the same
|
|
15
|
+
* elements as Markdown, which is used in writing plain text. Markdown is converted
|
|
16
|
+
* to HTML anyways to display it, and it is better to edit basic HTML in a rich text editor.
|
|
17
|
+
*
|
|
18
|
+
* [Mozilla DOM Reference](https://developer.mozilla.org/en-US/docs/Web/API/Document_Object_Model) <br />
|
|
19
|
+
* [Source Code of Browser HTML DOM](https://chromium.googlesource.com/chromium/src/+/HEAD/third_party/blink/renderer/core/dom/) <br />
|
|
20
|
+
* [RegExp JS V8 Code](https://github.com/v8/v8/blob/94cde7c7f3fffc62f621e43f65be3d517b8a9f3d/src/regexp/regexp-compiler.cc#L3827)
|
|
21
|
+
* @param {string} html Any page's HTML to process
|
|
22
|
+
* @param {Object} [options]
|
|
23
|
+
* @param {boolean} options.images default=true - Whether to include images
|
|
24
|
+
* @param {boolean} options.links default=true - Whether to include links
|
|
25
|
+
* @param {boolean} options.videos default=true - Whether to include videos or not
|
|
26
|
+
* @param {boolean} options.formatting default=true - Whether to include formatting
|
|
27
|
+
* @param {string} options.url base URL for converting relative URLs to absolute
|
|
28
|
+
* @param {string} options.allowTags default="br,p,u,b,i ,em,strong,h1,h2,h3,h4, h5,h6,blockquote,
|
|
29
|
+
* code,ul,ol,li,dd,dl, table,th,tr,td,sub,sup" - Comma-separated list of allowed HTML tags.
|
|
30
|
+
* @param {string} options.allowedAttributes default="text,tag,href, src,type,width, height,id,data"
|
|
31
|
+
* List of allowed HTML attributes
|
|
32
|
+
* @returns {string} basic text formatting html
|
|
33
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
34
|
+
* @category HTML Utilities
|
|
35
|
+
*/
|
|
36
|
+
export function convertHTMLToBasicHTML(html, options = {}) {
|
|
37
|
+
var {
|
|
38
|
+
images = true,
|
|
39
|
+
links = true,
|
|
40
|
+
videos = true,
|
|
41
|
+
formatting = true,
|
|
42
|
+
url = "",
|
|
43
|
+
openLinksNewWindow = false,
|
|
44
|
+
allowTags = "br,p,u,b,i,em,strong,h1,h2,h3,h4,h5,h6,blockquote,code,\
|
|
45
|
+
ul,ol,li,dd,dl,table,th,tr,td,thead,tbody,sub,sup,math,iframe",
|
|
46
|
+
allowedAttributes = "href,src,type,width,height,id,data,target",
|
|
47
|
+
} = options;
|
|
48
|
+
|
|
49
|
+
// return convertMarkdownToHTML(convertMarkdownToHTML(html, false), true)
|
|
50
|
+
|
|
51
|
+
allowTags = allowTags.split(",");
|
|
52
|
+
if (links) allowTags.push("a");
|
|
53
|
+
if (images) allowTags.push("img");
|
|
54
|
+
if (videos)
|
|
55
|
+
allowTags = allowTags.concat("video,source,embed,object".split(","));
|
|
56
|
+
|
|
57
|
+
if (!formatting) allowTags = ["text"];
|
|
58
|
+
allowTags.push("text");
|
|
59
|
+
|
|
60
|
+
allowedAttributes = allowedAttributes
|
|
61
|
+
.split(",")
|
|
62
|
+
.concat("text,tagName".split(","));
|
|
63
|
+
|
|
64
|
+
// Convert html string to array like [{tag:"p",attr:""},{text:""}]
|
|
65
|
+
var basicHtml = convertHTMLToTokens(html);
|
|
66
|
+
if (!basicHtml) return;
|
|
67
|
+
|
|
68
|
+
basicHtml = basicHtml
|
|
69
|
+
.filter(
|
|
70
|
+
(token) =>
|
|
71
|
+
token.text ||
|
|
72
|
+
(token.tagName[0] == "/"
|
|
73
|
+
? allowTags.includes(token.tagName?.substring(1)?.toLowerCase())
|
|
74
|
+
: allowTags.includes(token.tagName?.toLowerCase()))
|
|
75
|
+
)
|
|
76
|
+
.map((el) => {
|
|
77
|
+
for (var key of Object.keys(el))
|
|
78
|
+
if (!allowedAttributes.includes(key)) delete el[key];
|
|
79
|
+
|
|
80
|
+
var urlValue = el.href || el.src;
|
|
81
|
+
|
|
82
|
+
//non-anchor links should be opened in new window
|
|
83
|
+
if (urlValue && openLinksNewWindow)
|
|
84
|
+
if (!urlValue.startsWith("#")) el.target = "_blank";
|
|
85
|
+
|
|
86
|
+
// remove broken images
|
|
87
|
+
if (el.tagName?.toLowerCase() == "img") {
|
|
88
|
+
if (!el.src || el.src.startsWith("data:")) return false;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
//convert relative urls to absolute urls
|
|
92
|
+
if (el.src) {
|
|
93
|
+
el.src = new URL(urlValue, url).href;
|
|
94
|
+
}
|
|
95
|
+
if (el.href) el.href = new URL(urlValue, url).href;
|
|
96
|
+
|
|
97
|
+
// convertURLToAbsoluteURL(url, urlValue);
|
|
98
|
+
|
|
99
|
+
return el;
|
|
100
|
+
})
|
|
101
|
+
.filter(Boolean)
|
|
102
|
+
.reduce((acc, el) => {
|
|
103
|
+
acc += el.text
|
|
104
|
+
? `${el.text}`
|
|
105
|
+
: `<${el.tagName}${Object.keys(el).length > 1 ? " " : ""}${Object.keys(
|
|
106
|
+
el
|
|
107
|
+
)
|
|
108
|
+
.filter((key) => key != "tagName" && key != "text")
|
|
109
|
+
.map((key) => `${key}="${el[key]}"`)
|
|
110
|
+
.join(" ")}>`;
|
|
111
|
+
return acc;
|
|
112
|
+
}, "")
|
|
113
|
+
.replace(/<p><\/p>/g, " ")
|
|
114
|
+
.replace(/[\r\n\t]+/g, " ") //remove linebreaks
|
|
115
|
+
.replace(/ \s+/g, " ");
|
|
116
|
+
|
|
117
|
+
basicHtml = convertURLSafeHTMLToHTML(basicHtml).replace(/ /g, " ");
|
|
118
|
+
|
|
119
|
+
// // CNN news edge case of data=attr <> inside of attr
|
|
120
|
+
// const reHTMLInsideDataAttr =
|
|
121
|
+
// /(["'])(?:(?!(?:\1|<)).)*?(?:<(?:(?!["'<>]).)*?>)?(?:(?!(?:\1|<)).)*?\1/gis;
|
|
122
|
+
// if (reHTMLInsideDataAttr.test(html))
|
|
123
|
+
// html = html.replaceAll(reHTMLInsideDataAttr, "");
|
|
124
|
+
|
|
125
|
+
return basicHtml;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* Convert html string to array of JSON Objects tokens to translate,
|
|
130
|
+
* convert, or filter all elements.
|
|
131
|
+
* Flat array is faster than DOMParser which uses nested trees.
|
|
132
|
+
* @param {string} html
|
|
133
|
+
* @returns {array} Example [{"tag": "img","src": ""}, ...]
|
|
134
|
+
|
|
135
|
+
* @private
|
|
136
|
+
*/
|
|
137
|
+
export function convertHTMLToTokens(html) {
|
|
138
|
+
if (!html) return;
|
|
139
|
+
var dom = [];
|
|
140
|
+
|
|
141
|
+
//remove script style to prevent it from counting as text
|
|
142
|
+
html = html
|
|
143
|
+
.replace(/(<(noscript|script|style)\b[^>]*>).*?(<\/\2>)/gis, "$1$3")
|
|
144
|
+
.replace(/<script\b[^<]*(?:(?!<\/script>)<[^<]*)*<\/script>/gi, "")
|
|
145
|
+
.replace(/<style\b[^<]*(?:(?!<\/style>)<[^<]*)*<\/style>/gi, "")
|
|
146
|
+
.replace(/<!--[\s\S]*?-->/g, "");
|
|
147
|
+
|
|
148
|
+
const reHTMLInsideDataAttr =
|
|
149
|
+
/(["'])(?:(?!(?:\1|<)).)*?(?:<(?:(?!["'<>]).)*?>)?(?:(?!(?:\1|<)).)*?\1/gis;
|
|
150
|
+
|
|
151
|
+
var chunks = html.split("<");
|
|
152
|
+
|
|
153
|
+
for (var chunk of chunks) {
|
|
154
|
+
if (!chunk.includes(">")) continue;
|
|
155
|
+
|
|
156
|
+
var [element, text] = chunk.split(">");
|
|
157
|
+
|
|
158
|
+
if (element.includes("<")) {
|
|
159
|
+
if (reHTMLInsideDataAttr.test(html)) {
|
|
160
|
+
html = html.replaceAll(reHTMLInsideDataAttr, "");
|
|
161
|
+
return convertHTMLToTokens(html);
|
|
162
|
+
}
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
//if closing tag, add it but dont stop and also in next step
|
|
166
|
+
// add text after </a> as text node
|
|
167
|
+
if (element[0] == "/") dom.push({ tagName: element });
|
|
168
|
+
|
|
169
|
+
if (element[0] == "!") continue; //skip comments
|
|
170
|
+
|
|
171
|
+
var domElement = {};
|
|
172
|
+
//if has attributes
|
|
173
|
+
var attributesIndex = element.indexOf(" ");
|
|
174
|
+
|
|
175
|
+
if (attributesIndex == -1) {
|
|
176
|
+
domElement.tagName = element;
|
|
177
|
+
} else {
|
|
178
|
+
//has attributes
|
|
179
|
+
|
|
180
|
+
var tag = element.substring(0, attributesIndex);
|
|
181
|
+
domElement.tagName = tag;
|
|
182
|
+
// there can be spaces and <> inside of attr strings
|
|
183
|
+
//TODO cnn news edge case of data=attr <> inside of attr
|
|
184
|
+
//insert attr into domElement
|
|
185
|
+
element
|
|
186
|
+
.substring(attributesIndex)
|
|
187
|
+
.match(/ \w+=("(?:[^"\\]|\\.\s)*")/g)
|
|
188
|
+
?.forEach((attr) => {
|
|
189
|
+
attr = attr.trim();
|
|
190
|
+
|
|
191
|
+
var key = attr.split("=")[0];
|
|
192
|
+
var value = attr.slice(key.length + 2, -1);
|
|
193
|
+
if (key == "srcset") {
|
|
194
|
+
key = "src";
|
|
195
|
+
value = value.split(",")[0].trim().split(" ")[0];
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
if (key && value) domElement[key] = value?.replace(/"/g, "");
|
|
199
|
+
});
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
// style and script, add their content to "content" and dont treat as text
|
|
203
|
+
if (["style", "script", "noscript"].includes(domElement.tagName)) {
|
|
204
|
+
domElement.content = text;
|
|
205
|
+
continue;
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
if (element[0] != "/") dom.push(domElement);
|
|
209
|
+
|
|
210
|
+
//if text node push as {text:""}
|
|
211
|
+
if (text) dom.push({ tagName: "text", text: text });
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
// dom = addDOMFunctions(dom);
|
|
215
|
+
|
|
216
|
+
return dom;
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
export function addDOMFunctions(domObject) {
|
|
220
|
+
//assign to all objects for easy chain calling
|
|
221
|
+
domObject = domObject || Object.prototype;
|
|
222
|
+
|
|
223
|
+
domObject = Object.assign(domObject, {
|
|
224
|
+
querySelectorAll: function (querySelector) {
|
|
225
|
+
if (querySelector.includes(","))
|
|
226
|
+
//multiple selectors
|
|
227
|
+
var selectors = querySelector.split(",").map((sel) => sel.trim());
|
|
228
|
+
|
|
229
|
+
var type = selector[0];
|
|
230
|
+
selector = selector.substring(1);
|
|
231
|
+
|
|
232
|
+
if (type == ".")
|
|
233
|
+
//class
|
|
234
|
+
return this.filter((el) => el.class == selector);
|
|
235
|
+
if (type == "#")
|
|
236
|
+
//id
|
|
237
|
+
return this.filter((el) => el.id == selector);
|
|
238
|
+
if (type == "[")
|
|
239
|
+
//attribute
|
|
240
|
+
return this.filter((el) => el[selector] !== undefined);
|
|
241
|
+
//tag
|
|
242
|
+
else return this.filter(({ tagName }) => tagName == selector);
|
|
243
|
+
},
|
|
244
|
+
querySelector: function (selector) {
|
|
245
|
+
return this.querySelectorAll(selector)[0];
|
|
246
|
+
},
|
|
247
|
+
getTextContent: function () {
|
|
248
|
+
return this.reduce(
|
|
249
|
+
(acc, { text }) => (acc += text ? text + "\n" : ""),
|
|
250
|
+
""
|
|
251
|
+
);
|
|
252
|
+
},
|
|
253
|
+
getAttribute: function (attr) {
|
|
254
|
+
return this.map((el) => el[attr]).filter(Boolean);
|
|
255
|
+
},
|
|
256
|
+
getElementsByTagName: function (tag) {
|
|
257
|
+
return this.filter(({ tagName: t }) => t == tag).map(addDOMFunctions);
|
|
258
|
+
},
|
|
259
|
+
getElementsByClassName: function (className) {
|
|
260
|
+
return this.filter((el) => el.class == className);
|
|
261
|
+
},
|
|
262
|
+
getElementById: function (id) {
|
|
263
|
+
return this.filter((el) => el.id == id);
|
|
264
|
+
},
|
|
265
|
+
getInnerHTML: function () {
|
|
266
|
+
return this.reduce((acc, el) => {
|
|
267
|
+
acc += el.text
|
|
268
|
+
? `${el.text}`
|
|
269
|
+
: `<${el.tagName} ${Object.keys(el)
|
|
270
|
+
.filter((key) => key != "tagName" && key != "text")
|
|
271
|
+
.map((key) => `${key}="${el[key]}"`)
|
|
272
|
+
.join(" ")}>`;
|
|
273
|
+
return acc;
|
|
274
|
+
}, "");
|
|
275
|
+
},
|
|
276
|
+
});
|
|
277
|
+
|
|
278
|
+
domObject.innerHTML = domObject.getInnerHTML();
|
|
279
|
+
domObject.textContent = domObject.getTextContent();
|
|
280
|
+
|
|
281
|
+
return domObject;
|
|
282
|
+
}
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
// @ts-nocheck
|
|
2
|
+
/**
|
|
3
|
+
* @fileoverview Utility to extract core text content from HTML documents.
|
|
4
|
+
* Cleans boilerplate (nav, footer, ads) to produce clean Markdown or text.
|
|
5
|
+
*/
|
|
6
|
+
import { parseHTML } from "linkedom";
|
|
7
|
+
import { extractCite } from "../html-to-cite/extract-cite";
|
|
8
|
+
import { convertHTMLToBasicHTML } from "./html-to-basic-html";
|
|
9
|
+
import { extractHumanName } from "../html-to-cite/human-names-recognize";
|
|
10
|
+
import { extractMainContentFromHTML } from "./extract-content/extract-content-readability";
|
|
11
|
+
import { extractMainContentFromHTML2 } from "./extract-content/extract-content-mercury";
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Extracts the main content and citation information from a document or HTML string
|
|
15
|
+
* @param {string|object} documentOrHTML - The document or HTML string to extract content from
|
|
16
|
+
* @param {Object} options - Optional configuration options
|
|
17
|
+
* @param {boolean} options.images default=true - Whether to include images in the extracted content
|
|
18
|
+
* @param {boolean} options.links default=true - Whether to include links in the extracted content
|
|
19
|
+
* @param {boolean} options.formatting default=true - Whether to preserve formatting in the extracted content
|
|
20
|
+
* @param {string} options.url The URL of the original document, if available, for absolutify-ing URLs
|
|
21
|
+
* @param {boolean} options.useExtractor2 default=false -
|
|
22
|
+
* false uses Mozilla Readability, true uses Postlight Mercury.
|
|
23
|
+
* then use the alternate if the first returns less than 200 characters
|
|
24
|
+
* @returns {Object} The extracted content and citation information
|
|
25
|
+
* @property {string} title - The title of the document
|
|
26
|
+
* @property {string} author_cite - The full citation for the author
|
|
27
|
+
* @property {string} author_short - A shortened version of the author's name
|
|
28
|
+
* @property {string} author - The author's name
|
|
29
|
+
* @property {string} date - The publication date
|
|
30
|
+
* @property {string} source - The source of the document
|
|
31
|
+
* @property {string} html - The extracted HTML content
|
|
32
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
33
|
+
*/
|
|
34
|
+
export function extractContentAndCite(documentOrHTML, options = {}) {
|
|
35
|
+
const {
|
|
36
|
+
images = true,
|
|
37
|
+
links = true,
|
|
38
|
+
formatting = true,
|
|
39
|
+
url = "",
|
|
40
|
+
useExtractor2 = 1,
|
|
41
|
+
minExtractedLength = 400,
|
|
42
|
+
} = options;
|
|
43
|
+
|
|
44
|
+
var html =
|
|
45
|
+
typeof documentOrHTML === "string"
|
|
46
|
+
? documentOrHTML
|
|
47
|
+
: documentOrHTML?.documentElement?.innerHTML;
|
|
48
|
+
|
|
49
|
+
if (!html) return { error: "No HTML found" };
|
|
50
|
+
|
|
51
|
+
try {
|
|
52
|
+
var content1 = extractMainContentFromHTML(html, options);
|
|
53
|
+
} catch (e) {
|
|
54
|
+
console.log(e);
|
|
55
|
+
}
|
|
56
|
+
try {
|
|
57
|
+
var content2 = extractMainContentFromHTML2(html, options);
|
|
58
|
+
} catch (e) {
|
|
59
|
+
console.log(e);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
//compare content lengths
|
|
63
|
+
var content = content1?.length > content2?.length ? content1 : content2;
|
|
64
|
+
|
|
65
|
+
// check if html is too short, if so use basic html
|
|
66
|
+
if (content?.replace(/<[^>]*>/g, "").length < minExtractedLength)
|
|
67
|
+
content = html;
|
|
68
|
+
|
|
69
|
+
//cite
|
|
70
|
+
var { author, author_cite, author_short, date, title, source } = extractCite(
|
|
71
|
+
html,
|
|
72
|
+
options
|
|
73
|
+
);
|
|
74
|
+
|
|
75
|
+
html = convertHTMLToBasicHTML(content, options);
|
|
76
|
+
|
|
77
|
+
return {
|
|
78
|
+
title,
|
|
79
|
+
author_cite,
|
|
80
|
+
author_short,
|
|
81
|
+
author,
|
|
82
|
+
date,
|
|
83
|
+
source,
|
|
84
|
+
html,
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
/**
|
|
88
|
+
* @typedef {Object} ExtractedContent
|
|
89
|
+
* @property {string} title - The title of the content
|
|
90
|
+
* @property {string} author_cite - The full citation for the author
|
|
91
|
+
* @property {string} author_short - A shortened version of the author's name
|
|
92
|
+
* @property {string} author - The author's name
|
|
93
|
+
* @property {string} date - The publication date
|
|
94
|
+
* @property {string} source - The source of the content
|
|
95
|
+
* @property {string} html - The extracted main content in HTML format
|
|
96
|
+
* @private
|
|
97
|
+
*/
|