extract-webpage 1.2.34 → 1.2.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extract-webpage.cjs.js.map +1 -1
- package/dist/extract-webpage.es.js.map +1 -1
- package/package.json +4 -4
- package/src/fs-mock.js +22 -22
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -1049
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3452 -3452
- package/src/search/index.ts +45 -45
- package/src/search/meta-search-agent-reexport.ts +38 -38
- package/src/search/url-to-html.ts +278 -278
- package/src/seektopic/seektopic-keyphrases.ts +279 -279
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -435
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -332
- package/src/url-to-content/url-to-content.ts +367 -367
- package/src/url-to-content/url-to-html.ts +436 -436
- package/src/utils/grab.ts +51 -51
|
@@ -1,367 +1,367 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
|
|
3
|
-
* Supports YouTube transcripts, PDFs, DOCX, and web articles.
|
|
4
|
-
*/
|
|
5
|
-
import { extractContentAndCite } from "../html-to-content/html-to-content";
|
|
6
|
-
import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
|
|
7
|
-
import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
|
|
8
|
-
import { scrapeURL } from "./url-to-html";
|
|
9
|
-
import grab from "../utils/grab";
|
|
10
|
-
|
|
11
|
-
/**
|
|
12
|
-
* Dynamic PDF converter to avoid bundling pdfjs at build time
|
|
13
|
-
*/
|
|
14
|
-
async function convertPDFToHTML(url: string, options: any) {
|
|
15
|
-
const { convertPDFToHTML: pdfConverter } = await import("extract-pdf");
|
|
16
|
-
return await pdfConverter(url, options);
|
|
17
|
-
}
|
|
18
|
-
|
|
19
|
-
async function isUrlPDF(url: string) {
|
|
20
|
-
try {
|
|
21
|
-
const buffer = await grab(url, { responseType: "arraybuffer", timeout: 5 });
|
|
22
|
-
if (!buffer || buffer.byteLength < 5) return false;
|
|
23
|
-
const chunk = new Uint8Array(buffer);
|
|
24
|
-
return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
|
|
25
|
-
} catch {
|
|
26
|
-
return false;
|
|
27
|
-
}
|
|
28
|
-
}
|
|
29
|
-
|
|
30
|
-
export interface ExtractContentOptions {
|
|
31
|
-
images?: boolean;
|
|
32
|
-
links?: boolean;
|
|
33
|
-
formatting?: boolean;
|
|
34
|
-
absoluteURLs?: boolean;
|
|
35
|
-
timeout?: number;
|
|
36
|
-
proxy?: string | null;
|
|
37
|
-
citeFormatMonthFull?: boolean;
|
|
38
|
-
citeFormatAuthorFull?: boolean;
|
|
39
|
-
url?: string;
|
|
40
|
-
useThirdPartyBackup?: boolean;
|
|
41
|
-
/** Preferred transcript languages when extracting YouTube videos. */
|
|
42
|
-
languages?: string[];
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
export interface ExtractedArticle {
|
|
46
|
-
cite?: string;
|
|
47
|
-
html?: string;
|
|
48
|
-
url?: string;
|
|
49
|
-
author?: string;
|
|
50
|
-
author_cite?: string;
|
|
51
|
-
author_short?: string;
|
|
52
|
-
author_type?: number | string;
|
|
53
|
-
date?: string;
|
|
54
|
-
title?: string;
|
|
55
|
-
source?: string;
|
|
56
|
-
word_count?: number;
|
|
57
|
-
format?: string;
|
|
58
|
-
error?: string | number;
|
|
59
|
-
}
|
|
60
|
-
|
|
61
|
-
type UrlLikeDocument = {
|
|
62
|
-
location?: { href?: string };
|
|
63
|
-
querySelectorAll?: (
|
|
64
|
-
selector: string,
|
|
65
|
-
) => { length: number } | ArrayLike<unknown>;
|
|
66
|
-
};
|
|
67
|
-
|
|
68
|
-
/**
|
|
69
|
-
* @typedef {Object} Article
|
|
70
|
-
* @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
|
|
71
|
-
* @property {string} html - The Basic HTML content of the article
|
|
72
|
-
* @property {string} url - The URL of the article
|
|
73
|
-
* @property {string} author - The full name of the author of the article
|
|
74
|
-
* @property {string} author_cite - Author name in Last, First Initial format
|
|
75
|
-
* @property {string} author_short - Author name in Last format
|
|
76
|
-
* @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
77
|
-
* @property {string} date - The publication date of the article
|
|
78
|
-
* @property {string} title - The title of the article
|
|
79
|
-
* @property {string} source - The source or publisher of the article
|
|
80
|
-
* @property {number} word_count - The word count of the full text (without HTML tags)
|
|
81
|
-
* @category Extract
|
|
82
|
-
*/
|
|
83
|
-
|
|
84
|
-
/**
|
|
85
|
-
* ### 🚜 Tractor the Text Extractor
|
|
86
|
-
* <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
|
|
87
|
-
*
|
|
88
|
-
* 1. Main Content Detection: Extract the main content from a URL by combining
|
|
89
|
-
* Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
|
|
90
|
-
* custom adapters for major sites for article, author, date HTML classes.
|
|
91
|
-
* 2. Basic HTML Standardization: Transform complex HTML into a simplified
|
|
92
|
-
* reading-mode format of basic HTML, making it ideal for research note archival
|
|
93
|
-
* and focused reading, with headings, images and links.
|
|
94
|
-
* 3. YouTube Transcript Processing: When a YouTube video URL is detected,
|
|
95
|
-
* retrieve the complete video transcript including both manual captions and
|
|
96
|
-
* auto-generated subtitles, maintaining proper timestamp synchronization and
|
|
97
|
-
* speaker identification where available.
|
|
98
|
-
* 4. PDF to HTML: Process PDF documents by extracting
|
|
99
|
-
* formatted text while intelligently handling line breaks, page headers,
|
|
100
|
-
* footnotes. The system analyzes text height statistics to automatically
|
|
101
|
-
* infer heading levels, creating a properly structured document hierarchy
|
|
102
|
-
* based on standard deviation from mean text size.
|
|
103
|
-
* 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
|
|
104
|
-
* (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
|
|
105
|
-
* them to HTML while preserving formatting, styles, and document structure.
|
|
106
|
-
* 6. Citation Information Extraction: Identify and extract citation metadata
|
|
107
|
-
* including author names, publication dates, sources, and titles using HTML
|
|
108
|
-
* meta tags and common class name patterns. The system validates author names
|
|
109
|
-
* against a comprehensive database of 90,000 first and last names,
|
|
110
|
-
* distinguishing between personal and organizational authors to properly
|
|
111
|
-
* format citations.
|
|
112
|
-
* 7. Author Name Formatting: Process author names by checking against
|
|
113
|
-
* known name databases, handling affixes and titles correctly, and determining
|
|
114
|
-
* whether to reverse the name order based on whether it's a personal or
|
|
115
|
-
* organizational author, ensuring proper citation formatting.
|
|
116
|
-
* @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
|
|
117
|
-
* @param {Object} [options]
|
|
118
|
-
* @param {boolean} options.images default=true - include images
|
|
119
|
-
* @param {boolean} options.links default=true - include links
|
|
120
|
-
* @param {boolean} options.formatting default=true - preserve formatting
|
|
121
|
-
* @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
|
|
122
|
-
* @param {number} options.timeout default=5 - http request timeout
|
|
123
|
-
* @returns {{
|
|
124
|
-
* title: string,
|
|
125
|
-
* author_cite: string,
|
|
126
|
-
* cite: string,
|
|
127
|
-
* author: string,
|
|
128
|
-
* date: string,
|
|
129
|
-
* source: string,
|
|
130
|
-
* html: string,
|
|
131
|
-
* word_count: number
|
|
132
|
-
* }}
|
|
133
|
-
* cite - Cite in APA Format with Author name in Last, First Initial format
|
|
134
|
-
* url - The URL of the article
|
|
135
|
-
* html - The HTML content of the article
|
|
136
|
-
* author - The author of the article
|
|
137
|
-
* author_cite - Author name in Last, First Middle format
|
|
138
|
-
* author_short - Author name in Last format
|
|
139
|
-
* author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
140
|
-
* date - The publication date of the article
|
|
141
|
-
* title - The title of the article
|
|
142
|
-
* source - The source or origin of the article
|
|
143
|
-
* word_count - The word count of the full text (without HTML tags)
|
|
144
|
-
* @category Extract
|
|
145
|
-
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
146
|
-
* @example
|
|
147
|
-
* // Extract from URL
|
|
148
|
-
* const result1 = await extractContent('https://example.com/article');
|
|
149
|
-
*
|
|
150
|
-
* // Extract from DOCX binary buffer
|
|
151
|
-
* const docxBuffer = new Uint8Array([...]); // DOCX file bytes
|
|
152
|
-
* const result2 = await extractContent(docxBuffer);
|
|
153
|
-
*
|
|
154
|
-
* // Extract from DOM object
|
|
155
|
-
* const result3 = await extractContent(document);
|
|
156
|
-
*/
|
|
157
|
-
export async function extractContent(
|
|
158
|
-
urlOrDoc:
|
|
159
|
-
| string
|
|
160
|
-
| Document
|
|
161
|
-
| UrlLikeDocument
|
|
162
|
-
| ArrayBuffer
|
|
163
|
-
| Buffer
|
|
164
|
-
| Uint8Array,
|
|
165
|
-
options: ExtractContentOptions = {},
|
|
166
|
-
): Promise<ExtractedArticle> {
|
|
167
|
-
var {
|
|
168
|
-
images = true,
|
|
169
|
-
links = true,
|
|
170
|
-
formatting = true,
|
|
171
|
-
absoluteURLs = true,
|
|
172
|
-
timeout = 5,
|
|
173
|
-
proxy = null,
|
|
174
|
-
citeFormatMonthFull = false,
|
|
175
|
-
citeFormatAuthorFull = true,
|
|
176
|
-
} = options;
|
|
177
|
-
let response: ExtractedArticle = {};
|
|
178
|
-
|
|
179
|
-
let url, isPdf, isDocxBuffer;
|
|
180
|
-
|
|
181
|
-
// Check if input is a binary buffer (DOCX)
|
|
182
|
-
if (
|
|
183
|
-
urlOrDoc instanceof ArrayBuffer ||
|
|
184
|
-
urlOrDoc instanceof Uint8Array ||
|
|
185
|
-
(typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
|
|
186
|
-
) {
|
|
187
|
-
isDocxBuffer = isBufferDOCX(urlOrDoc);
|
|
188
|
-
|
|
189
|
-
if (isDocxBuffer) {
|
|
190
|
-
// Handle DOCX binary buffer
|
|
191
|
-
response.html = await convertDOCXToHTML(urlOrDoc, options);
|
|
192
|
-
url = "buffer://docx"; // Placeholder URL for buffer input
|
|
193
|
-
} else {
|
|
194
|
-
return { error: "Binary buffer is not a valid DOCX file" };
|
|
195
|
-
}
|
|
196
|
-
} else if (
|
|
197
|
-
typeof urlOrDoc === "string" &&
|
|
198
|
-
/<\/[^>]+>/.test(urlOrDoc.trim())
|
|
199
|
-
) {
|
|
200
|
-
console.log("[extractContent] input is raw HTML string");
|
|
201
|
-
// If urlOrDoc is an HTML string, treat as HTML content
|
|
202
|
-
options.url = options.url || "";
|
|
203
|
-
|
|
204
|
-
response = extractContentAndCite(urlOrDoc, options);
|
|
205
|
-
console.log("[extractContent] extractContentAndCite (raw html) result", {
|
|
206
|
-
hasHtml: !!response?.html,
|
|
207
|
-
htmlLength: response?.html?.length || 0,
|
|
208
|
-
title: response?.title,
|
|
209
|
-
error: response?.error,
|
|
210
|
-
});
|
|
211
|
-
|
|
212
|
-
return response;
|
|
213
|
-
// if URL
|
|
214
|
-
} else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
|
|
215
|
-
url = urlOrDoc;
|
|
216
|
-
console.log("[extractContent] input is URL", { url });
|
|
217
|
-
|
|
218
|
-
// check if google doc, then extract html or pdf file
|
|
219
|
-
let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
|
|
220
|
-
if (googleDocId) {
|
|
221
|
-
url =
|
|
222
|
-
googleDocId[1] === "file"
|
|
223
|
-
? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
|
|
224
|
-
: `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
|
|
225
|
-
console.log("[extractContent] rewrote google doc url", { url });
|
|
226
|
-
}
|
|
227
|
-
|
|
228
|
-
isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
|
|
229
|
-
let youtubeID = getURLYoutubeVideo(url);
|
|
230
|
-
console.log("[extractContent] branch detection", {
|
|
231
|
-
isPdf,
|
|
232
|
-
youtubeID,
|
|
233
|
-
isDocx: url.endsWith(".docx"),
|
|
234
|
-
});
|
|
235
|
-
|
|
236
|
-
if (isPdf) {
|
|
237
|
-
// pdf checker - use dynamic import to prevent build-time evaluation
|
|
238
|
-
response = await convertPDFToHTML(url, options as any);
|
|
239
|
-
console.log("[extractContent] pdf branch result", {
|
|
240
|
-
hasHtml: !!response?.html,
|
|
241
|
-
error: response?.error,
|
|
242
|
-
});
|
|
243
|
-
} else if (url.endsWith(".docx")) {
|
|
244
|
-
response.html = await convertDOCXToHTML(url);
|
|
245
|
-
console.log("[extractContent] docx branch result", {
|
|
246
|
-
hasHtml: !!response?.html,
|
|
247
|
-
});
|
|
248
|
-
|
|
249
|
-
// check youtube
|
|
250
|
-
} else if (youtubeID) {
|
|
251
|
-
response = await convertYoutubeToText(url, options);
|
|
252
|
-
console.log("[extractContent] youtube branch result", {
|
|
253
|
-
hasHtml: !!response?.html,
|
|
254
|
-
error: response?.error,
|
|
255
|
-
});
|
|
256
|
-
} else {
|
|
257
|
-
console.log("[extractContent] scraping URL", { url, proxy });
|
|
258
|
-
|
|
259
|
-
try {
|
|
260
|
-
const html = await scrapeURL(url, {
|
|
261
|
-
proxy,
|
|
262
|
-
});
|
|
263
|
-
console.log("[extractContent] scrapeURL returned", {
|
|
264
|
-
url,
|
|
265
|
-
hasHtml: !!html,
|
|
266
|
-
htmlLength: typeof html === "string" ? html.length : 0,
|
|
267
|
-
sample: typeof html === "string" ? html.slice(0, 200) : null,
|
|
268
|
-
});
|
|
269
|
-
|
|
270
|
-
// Check if scrapeURL returned an error object instead of HTML string
|
|
271
|
-
if (typeof html !== "string" || !html) {
|
|
272
|
-
console.error("[extractContent] scrapeURL failed or returned non-string", {
|
|
273
|
-
url,
|
|
274
|
-
typeofHtml: typeof html,
|
|
275
|
-
isEmpty: !html,
|
|
276
|
-
});
|
|
277
|
-
return {
|
|
278
|
-
error: "Failed to fetch HTML content",
|
|
279
|
-
};
|
|
280
|
-
}
|
|
281
|
-
|
|
282
|
-
options.url = url;
|
|
283
|
-
response = extractContentAndCite(html, options);
|
|
284
|
-
console.log("[extractContent] extractContentAndCite result", {
|
|
285
|
-
url,
|
|
286
|
-
hasHtml: !!response?.html,
|
|
287
|
-
htmlLength: response?.html?.length || 0,
|
|
288
|
-
title: response?.title,
|
|
289
|
-
error: response?.error,
|
|
290
|
-
});
|
|
291
|
-
} catch (scrapeError) {
|
|
292
|
-
const err = scrapeError as Error;
|
|
293
|
-
console.error("[extractContent] scrapeURL threw error", {
|
|
294
|
-
url,
|
|
295
|
-
message: err?.message,
|
|
296
|
-
});
|
|
297
|
-
return {
|
|
298
|
-
error: `Failed to scrape URL: ${err?.message || String(scrapeError)}`,
|
|
299
|
-
};
|
|
300
|
-
}
|
|
301
|
-
}
|
|
302
|
-
} else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
|
|
303
|
-
//if passing in dom object document from front end
|
|
304
|
-
|
|
305
|
-
url = urlOrDoc.location.href;
|
|
306
|
-
|
|
307
|
-
//pdf checker for embedded docs
|
|
308
|
-
if (urlOrDoc?.querySelectorAll)
|
|
309
|
-
isPdf = urlOrDoc?.querySelectorAll(
|
|
310
|
-
'embed[type="application/pdf"]',
|
|
311
|
-
)?.length;
|
|
312
|
-
var youtubeID = getURLYoutubeVideo(url);
|
|
313
|
-
|
|
314
|
-
if (isPdf) {
|
|
315
|
-
response = await convertPDFToHTML(url, {});
|
|
316
|
-
} else if (youtubeID) {
|
|
317
|
-
// from front end
|
|
318
|
-
|
|
319
|
-
//if on same domain page in chrome-extension
|
|
320
|
-
options.useThirdPartyBackup = false;
|
|
321
|
-
response = await convertYoutubeToText(url, options);
|
|
322
|
-
} //pass doc to extract
|
|
323
|
-
else response = extractContentAndCite(urlOrDoc as Document, options);
|
|
324
|
-
} else {
|
|
325
|
-
// Handle other object types or invalid input
|
|
326
|
-
return {
|
|
327
|
-
error:
|
|
328
|
-
"Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
|
|
329
|
-
};
|
|
330
|
-
}
|
|
331
|
-
|
|
332
|
-
//if no text
|
|
333
|
-
if (response.error || !response.html) return { error: response.error };
|
|
334
|
-
|
|
335
|
-
//word count of full text original, no html
|
|
336
|
-
response.word_count = response.html
|
|
337
|
-
?.replace(/<[^>]*>/g, " ")
|
|
338
|
-
.split(" ").length;
|
|
339
|
-
|
|
340
|
-
//make APA cite
|
|
341
|
-
|
|
342
|
-
var { author, author_cite, author_short, date, title, source } = response;
|
|
343
|
-
|
|
344
|
-
var apa_cite_date =
|
|
345
|
-
new Date(date).getFullYear() > 1971
|
|
346
|
-
? " (" +
|
|
347
|
-
new Date(date).getFullYear() +
|
|
348
|
-
", " +
|
|
349
|
-
new Date(date).toLocaleDateString("en-US", {
|
|
350
|
-
month: citeFormatMonthFull ? "long" : "short",
|
|
351
|
-
day: "numeric",
|
|
352
|
-
}) +
|
|
353
|
-
")"
|
|
354
|
-
: ""; //"(N.D.)";
|
|
355
|
-
|
|
356
|
-
var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
|
|
357
|
-
title || ""
|
|
358
|
-
}</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
|
|
359
|
-
|
|
360
|
-
//shorten long urls by removing ?params=get used as state tracking
|
|
361
|
-
if (url && url.includes("?") && url.length > 150)
|
|
362
|
-
response.url = url.split("?")[0];
|
|
363
|
-
|
|
364
|
-
//put url on top
|
|
365
|
-
response = Object.assign({ url, cite }, response);
|
|
366
|
-
return response;
|
|
367
|
-
}
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
|
|
3
|
+
* Supports YouTube transcripts, PDFs, DOCX, and web articles.
|
|
4
|
+
*/
|
|
5
|
+
import { extractContentAndCite } from "../html-to-content/html-to-content";
|
|
6
|
+
import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
|
|
7
|
+
import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
|
|
8
|
+
import { scrapeURL } from "./url-to-html";
|
|
9
|
+
import grab from "../utils/grab";
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* Dynamic PDF converter to avoid bundling pdfjs at build time
|
|
13
|
+
*/
|
|
14
|
+
async function convertPDFToHTML(url: string, options: any) {
|
|
15
|
+
const { convertPDFToHTML: pdfConverter } = await import("extract-pdf");
|
|
16
|
+
return await pdfConverter(url, options);
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
async function isUrlPDF(url: string) {
|
|
20
|
+
try {
|
|
21
|
+
const buffer = await grab(url, { responseType: "arraybuffer", timeout: 5 });
|
|
22
|
+
if (!buffer || buffer.byteLength < 5) return false;
|
|
23
|
+
const chunk = new Uint8Array(buffer);
|
|
24
|
+
return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
|
|
25
|
+
} catch {
|
|
26
|
+
return false;
|
|
27
|
+
}
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface ExtractContentOptions {
|
|
31
|
+
images?: boolean;
|
|
32
|
+
links?: boolean;
|
|
33
|
+
formatting?: boolean;
|
|
34
|
+
absoluteURLs?: boolean;
|
|
35
|
+
timeout?: number;
|
|
36
|
+
proxy?: string | null;
|
|
37
|
+
citeFormatMonthFull?: boolean;
|
|
38
|
+
citeFormatAuthorFull?: boolean;
|
|
39
|
+
url?: string;
|
|
40
|
+
useThirdPartyBackup?: boolean;
|
|
41
|
+
/** Preferred transcript languages when extracting YouTube videos. */
|
|
42
|
+
languages?: string[];
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export interface ExtractedArticle {
|
|
46
|
+
cite?: string;
|
|
47
|
+
html?: string;
|
|
48
|
+
url?: string;
|
|
49
|
+
author?: string;
|
|
50
|
+
author_cite?: string;
|
|
51
|
+
author_short?: string;
|
|
52
|
+
author_type?: number | string;
|
|
53
|
+
date?: string;
|
|
54
|
+
title?: string;
|
|
55
|
+
source?: string;
|
|
56
|
+
word_count?: number;
|
|
57
|
+
format?: string;
|
|
58
|
+
error?: string | number;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
type UrlLikeDocument = {
|
|
62
|
+
location?: { href?: string };
|
|
63
|
+
querySelectorAll?: (
|
|
64
|
+
selector: string,
|
|
65
|
+
) => { length: number } | ArrayLike<unknown>;
|
|
66
|
+
};
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* @typedef {Object} Article
|
|
70
|
+
* @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
|
|
71
|
+
* @property {string} html - The Basic HTML content of the article
|
|
72
|
+
* @property {string} url - The URL of the article
|
|
73
|
+
* @property {string} author - The full name of the author of the article
|
|
74
|
+
* @property {string} author_cite - Author name in Last, First Initial format
|
|
75
|
+
* @property {string} author_short - Author name in Last format
|
|
76
|
+
* @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
77
|
+
* @property {string} date - The publication date of the article
|
|
78
|
+
* @property {string} title - The title of the article
|
|
79
|
+
* @property {string} source - The source or publisher of the article
|
|
80
|
+
* @property {number} word_count - The word count of the full text (without HTML tags)
|
|
81
|
+
* @category Extract
|
|
82
|
+
*/
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* ### 🚜 Tractor the Text Extractor
|
|
86
|
+
* <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
|
|
87
|
+
*
|
|
88
|
+
* 1. Main Content Detection: Extract the main content from a URL by combining
|
|
89
|
+
* Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
|
|
90
|
+
* custom adapters for major sites for article, author, date HTML classes.
|
|
91
|
+
* 2. Basic HTML Standardization: Transform complex HTML into a simplified
|
|
92
|
+
* reading-mode format of basic HTML, making it ideal for research note archival
|
|
93
|
+
* and focused reading, with headings, images and links.
|
|
94
|
+
* 3. YouTube Transcript Processing: When a YouTube video URL is detected,
|
|
95
|
+
* retrieve the complete video transcript including both manual captions and
|
|
96
|
+
* auto-generated subtitles, maintaining proper timestamp synchronization and
|
|
97
|
+
* speaker identification where available.
|
|
98
|
+
* 4. PDF to HTML: Process PDF documents by extracting
|
|
99
|
+
* formatted text while intelligently handling line breaks, page headers,
|
|
100
|
+
* footnotes. The system analyzes text height statistics to automatically
|
|
101
|
+
* infer heading levels, creating a properly structured document hierarchy
|
|
102
|
+
* based on standard deviation from mean text size.
|
|
103
|
+
* 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
|
|
104
|
+
* (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
|
|
105
|
+
* them to HTML while preserving formatting, styles, and document structure.
|
|
106
|
+
* 6. Citation Information Extraction: Identify and extract citation metadata
|
|
107
|
+
* including author names, publication dates, sources, and titles using HTML
|
|
108
|
+
* meta tags and common class name patterns. The system validates author names
|
|
109
|
+
* against a comprehensive database of 90,000 first and last names,
|
|
110
|
+
* distinguishing between personal and organizational authors to properly
|
|
111
|
+
* format citations.
|
|
112
|
+
* 7. Author Name Formatting: Process author names by checking against
|
|
113
|
+
* known name databases, handling affixes and titles correctly, and determining
|
|
114
|
+
* whether to reverse the name order based on whether it's a personal or
|
|
115
|
+
* organizational author, ensuring proper citation formatting.
|
|
116
|
+
* @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
|
|
117
|
+
* @param {Object} [options]
|
|
118
|
+
* @param {boolean} options.images default=true - include images
|
|
119
|
+
* @param {boolean} options.links default=true - include links
|
|
120
|
+
* @param {boolean} options.formatting default=true - preserve formatting
|
|
121
|
+
* @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
|
|
122
|
+
* @param {number} options.timeout default=5 - http request timeout
|
|
123
|
+
* @returns {{
|
|
124
|
+
* title: string,
|
|
125
|
+
* author_cite: string,
|
|
126
|
+
* cite: string,
|
|
127
|
+
* author: string,
|
|
128
|
+
* date: string,
|
|
129
|
+
* source: string,
|
|
130
|
+
* html: string,
|
|
131
|
+
* word_count: number
|
|
132
|
+
* }}
|
|
133
|
+
* cite - Cite in APA Format with Author name in Last, First Initial format
|
|
134
|
+
* url - The URL of the article
|
|
135
|
+
* html - The HTML content of the article
|
|
136
|
+
* author - The author of the article
|
|
137
|
+
* author_cite - Author name in Last, First Middle format
|
|
138
|
+
* author_short - Author name in Last format
|
|
139
|
+
* author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
140
|
+
* date - The publication date of the article
|
|
141
|
+
* title - The title of the article
|
|
142
|
+
* source - The source or origin of the article
|
|
143
|
+
* word_count - The word count of the full text (without HTML tags)
|
|
144
|
+
* @category Extract
|
|
145
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
146
|
+
* @example
|
|
147
|
+
* // Extract from URL
|
|
148
|
+
* const result1 = await extractContent('https://example.com/article');
|
|
149
|
+
*
|
|
150
|
+
* // Extract from DOCX binary buffer
|
|
151
|
+
* const docxBuffer = new Uint8Array([...]); // DOCX file bytes
|
|
152
|
+
* const result2 = await extractContent(docxBuffer);
|
|
153
|
+
*
|
|
154
|
+
* // Extract from DOM object
|
|
155
|
+
* const result3 = await extractContent(document);
|
|
156
|
+
*/
|
|
157
|
+
export async function extractContent(
|
|
158
|
+
urlOrDoc:
|
|
159
|
+
| string
|
|
160
|
+
| Document
|
|
161
|
+
| UrlLikeDocument
|
|
162
|
+
| ArrayBuffer
|
|
163
|
+
| Buffer
|
|
164
|
+
| Uint8Array,
|
|
165
|
+
options: ExtractContentOptions = {},
|
|
166
|
+
): Promise<ExtractedArticle> {
|
|
167
|
+
var {
|
|
168
|
+
images = true,
|
|
169
|
+
links = true,
|
|
170
|
+
formatting = true,
|
|
171
|
+
absoluteURLs = true,
|
|
172
|
+
timeout = 5,
|
|
173
|
+
proxy = null,
|
|
174
|
+
citeFormatMonthFull = false,
|
|
175
|
+
citeFormatAuthorFull = true,
|
|
176
|
+
} = options;
|
|
177
|
+
let response: ExtractedArticle = {};
|
|
178
|
+
|
|
179
|
+
let url, isPdf, isDocxBuffer;
|
|
180
|
+
|
|
181
|
+
// Check if input is a binary buffer (DOCX)
|
|
182
|
+
if (
|
|
183
|
+
urlOrDoc instanceof ArrayBuffer ||
|
|
184
|
+
urlOrDoc instanceof Uint8Array ||
|
|
185
|
+
(typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
|
|
186
|
+
) {
|
|
187
|
+
isDocxBuffer = isBufferDOCX(urlOrDoc);
|
|
188
|
+
|
|
189
|
+
if (isDocxBuffer) {
|
|
190
|
+
// Handle DOCX binary buffer
|
|
191
|
+
response.html = await convertDOCXToHTML(urlOrDoc, options);
|
|
192
|
+
url = "buffer://docx"; // Placeholder URL for buffer input
|
|
193
|
+
} else {
|
|
194
|
+
return { error: "Binary buffer is not a valid DOCX file" };
|
|
195
|
+
}
|
|
196
|
+
} else if (
|
|
197
|
+
typeof urlOrDoc === "string" &&
|
|
198
|
+
/<\/[^>]+>/.test(urlOrDoc.trim())
|
|
199
|
+
) {
|
|
200
|
+
console.log("[extractContent] input is raw HTML string");
|
|
201
|
+
// If urlOrDoc is an HTML string, treat as HTML content
|
|
202
|
+
options.url = options.url || "";
|
|
203
|
+
|
|
204
|
+
response = extractContentAndCite(urlOrDoc, options);
|
|
205
|
+
console.log("[extractContent] extractContentAndCite (raw html) result", {
|
|
206
|
+
hasHtml: !!response?.html,
|
|
207
|
+
htmlLength: response?.html?.length || 0,
|
|
208
|
+
title: response?.title,
|
|
209
|
+
error: response?.error,
|
|
210
|
+
});
|
|
211
|
+
|
|
212
|
+
return response;
|
|
213
|
+
// if URL
|
|
214
|
+
} else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
|
|
215
|
+
url = urlOrDoc;
|
|
216
|
+
console.log("[extractContent] input is URL", { url });
|
|
217
|
+
|
|
218
|
+
// check if google doc, then extract html or pdf file
|
|
219
|
+
let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
|
|
220
|
+
if (googleDocId) {
|
|
221
|
+
url =
|
|
222
|
+
googleDocId[1] === "file"
|
|
223
|
+
? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
|
|
224
|
+
: `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
|
|
225
|
+
console.log("[extractContent] rewrote google doc url", { url });
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
|
|
229
|
+
let youtubeID = getURLYoutubeVideo(url);
|
|
230
|
+
console.log("[extractContent] branch detection", {
|
|
231
|
+
isPdf,
|
|
232
|
+
youtubeID,
|
|
233
|
+
isDocx: url.endsWith(".docx"),
|
|
234
|
+
});
|
|
235
|
+
|
|
236
|
+
if (isPdf) {
|
|
237
|
+
// pdf checker - use dynamic import to prevent build-time evaluation
|
|
238
|
+
response = await convertPDFToHTML(url, options as any);
|
|
239
|
+
console.log("[extractContent] pdf branch result", {
|
|
240
|
+
hasHtml: !!response?.html,
|
|
241
|
+
error: response?.error,
|
|
242
|
+
});
|
|
243
|
+
} else if (url.endsWith(".docx")) {
|
|
244
|
+
response.html = await convertDOCXToHTML(url);
|
|
245
|
+
console.log("[extractContent] docx branch result", {
|
|
246
|
+
hasHtml: !!response?.html,
|
|
247
|
+
});
|
|
248
|
+
|
|
249
|
+
// check youtube
|
|
250
|
+
} else if (youtubeID) {
|
|
251
|
+
response = await convertYoutubeToText(url, options);
|
|
252
|
+
console.log("[extractContent] youtube branch result", {
|
|
253
|
+
hasHtml: !!response?.html,
|
|
254
|
+
error: response?.error,
|
|
255
|
+
});
|
|
256
|
+
} else {
|
|
257
|
+
console.log("[extractContent] scraping URL", { url, proxy });
|
|
258
|
+
|
|
259
|
+
try {
|
|
260
|
+
const html = await scrapeURL(url, {
|
|
261
|
+
proxy,
|
|
262
|
+
});
|
|
263
|
+
console.log("[extractContent] scrapeURL returned", {
|
|
264
|
+
url,
|
|
265
|
+
hasHtml: !!html,
|
|
266
|
+
htmlLength: typeof html === "string" ? html.length : 0,
|
|
267
|
+
sample: typeof html === "string" ? html.slice(0, 200) : null,
|
|
268
|
+
});
|
|
269
|
+
|
|
270
|
+
// Check if scrapeURL returned an error object instead of HTML string
|
|
271
|
+
if (typeof html !== "string" || !html) {
|
|
272
|
+
console.error("[extractContent] scrapeURL failed or returned non-string", {
|
|
273
|
+
url,
|
|
274
|
+
typeofHtml: typeof html,
|
|
275
|
+
isEmpty: !html,
|
|
276
|
+
});
|
|
277
|
+
return {
|
|
278
|
+
error: "Failed to fetch HTML content",
|
|
279
|
+
};
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
options.url = url;
|
|
283
|
+
response = extractContentAndCite(html, options);
|
|
284
|
+
console.log("[extractContent] extractContentAndCite result", {
|
|
285
|
+
url,
|
|
286
|
+
hasHtml: !!response?.html,
|
|
287
|
+
htmlLength: response?.html?.length || 0,
|
|
288
|
+
title: response?.title,
|
|
289
|
+
error: response?.error,
|
|
290
|
+
});
|
|
291
|
+
} catch (scrapeError) {
|
|
292
|
+
const err = scrapeError as Error;
|
|
293
|
+
console.error("[extractContent] scrapeURL threw error", {
|
|
294
|
+
url,
|
|
295
|
+
message: err?.message,
|
|
296
|
+
});
|
|
297
|
+
return {
|
|
298
|
+
error: `Failed to scrape URL: ${err?.message || String(scrapeError)}`,
|
|
299
|
+
};
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
} else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
|
|
303
|
+
//if passing in dom object document from front end
|
|
304
|
+
|
|
305
|
+
url = urlOrDoc.location.href;
|
|
306
|
+
|
|
307
|
+
//pdf checker for embedded docs
|
|
308
|
+
if (urlOrDoc?.querySelectorAll)
|
|
309
|
+
isPdf = urlOrDoc?.querySelectorAll(
|
|
310
|
+
'embed[type="application/pdf"]',
|
|
311
|
+
)?.length;
|
|
312
|
+
var youtubeID = getURLYoutubeVideo(url);
|
|
313
|
+
|
|
314
|
+
if (isPdf) {
|
|
315
|
+
response = await convertPDFToHTML(url, {});
|
|
316
|
+
} else if (youtubeID) {
|
|
317
|
+
// from front end
|
|
318
|
+
|
|
319
|
+
//if on same domain page in chrome-extension
|
|
320
|
+
options.useThirdPartyBackup = false;
|
|
321
|
+
response = await convertYoutubeToText(url, options);
|
|
322
|
+
} //pass doc to extract
|
|
323
|
+
else response = extractContentAndCite(urlOrDoc as Document, options);
|
|
324
|
+
} else {
|
|
325
|
+
// Handle other object types or invalid input
|
|
326
|
+
return {
|
|
327
|
+
error:
|
|
328
|
+
"Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
|
|
329
|
+
};
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
//if no text
|
|
333
|
+
if (response.error || !response.html) return { error: response.error };
|
|
334
|
+
|
|
335
|
+
//word count of full text original, no html
|
|
336
|
+
response.word_count = response.html
|
|
337
|
+
?.replace(/<[^>]*>/g, " ")
|
|
338
|
+
.split(" ").length;
|
|
339
|
+
|
|
340
|
+
//make APA cite
|
|
341
|
+
|
|
342
|
+
var { author, author_cite, author_short, date, title, source } = response;
|
|
343
|
+
|
|
344
|
+
var apa_cite_date =
|
|
345
|
+
new Date(date).getFullYear() > 1971
|
|
346
|
+
? " (" +
|
|
347
|
+
new Date(date).getFullYear() +
|
|
348
|
+
", " +
|
|
349
|
+
new Date(date).toLocaleDateString("en-US", {
|
|
350
|
+
month: citeFormatMonthFull ? "long" : "short",
|
|
351
|
+
day: "numeric",
|
|
352
|
+
}) +
|
|
353
|
+
")"
|
|
354
|
+
: ""; //"(N.D.)";
|
|
355
|
+
|
|
356
|
+
var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
|
|
357
|
+
title || ""
|
|
358
|
+
}</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
|
|
359
|
+
|
|
360
|
+
//shorten long urls by removing ?params=get used as state tracking
|
|
361
|
+
if (url && url.includes("?") && url.length > 150)
|
|
362
|
+
response.url = url.split("?")[0];
|
|
363
|
+
|
|
364
|
+
//put url on top
|
|
365
|
+
response = Object.assign({ url, cite }, response);
|
|
366
|
+
return response;
|
|
367
|
+
}
|