extract-webpage 1.2.34 → 1.2.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,367 +1,367 @@
1
- /**
2
- * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
- * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
- */
5
- import { extractContentAndCite } from "../html-to-content/html-to-content";
6
- import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
7
- import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
8
- import { scrapeURL } from "./url-to-html";
9
- import grab from "../utils/grab";
10
-
11
- /**
12
- * Dynamic PDF converter to avoid bundling pdfjs at build time
13
- */
14
- async function convertPDFToHTML(url: string, options: any) {
15
- const { convertPDFToHTML: pdfConverter } = await import("extract-pdf");
16
- return await pdfConverter(url, options);
17
- }
18
-
19
- async function isUrlPDF(url: string) {
20
- try {
21
- const buffer = await grab(url, { responseType: "arraybuffer", timeout: 5 });
22
- if (!buffer || buffer.byteLength < 5) return false;
23
- const chunk = new Uint8Array(buffer);
24
- return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
25
- } catch {
26
- return false;
27
- }
28
- }
29
-
30
- export interface ExtractContentOptions {
31
- images?: boolean;
32
- links?: boolean;
33
- formatting?: boolean;
34
- absoluteURLs?: boolean;
35
- timeout?: number;
36
- proxy?: string | null;
37
- citeFormatMonthFull?: boolean;
38
- citeFormatAuthorFull?: boolean;
39
- url?: string;
40
- useThirdPartyBackup?: boolean;
41
- /** Preferred transcript languages when extracting YouTube videos. */
42
- languages?: string[];
43
- }
44
-
45
- export interface ExtractedArticle {
46
- cite?: string;
47
- html?: string;
48
- url?: string;
49
- author?: string;
50
- author_cite?: string;
51
- author_short?: string;
52
- author_type?: number | string;
53
- date?: string;
54
- title?: string;
55
- source?: string;
56
- word_count?: number;
57
- format?: string;
58
- error?: string | number;
59
- }
60
-
61
- type UrlLikeDocument = {
62
- location?: { href?: string };
63
- querySelectorAll?: (
64
- selector: string,
65
- ) => { length: number } | ArrayLike<unknown>;
66
- };
67
-
68
- /**
69
- * @typedef {Object} Article
70
- * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
71
- * @property {string} html - The Basic HTML content of the article
72
- * @property {string} url - The URL of the article
73
- * @property {string} author - The full name of the author of the article
74
- * @property {string} author_cite - Author name in Last, First Initial format
75
- * @property {string} author_short - Author name in Last format
76
- * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
77
- * @property {string} date - The publication date of the article
78
- * @property {string} title - The title of the article
79
- * @property {string} source - The source or publisher of the article
80
- * @property {number} word_count - The word count of the full text (without HTML tags)
81
- * @category Extract
82
- */
83
-
84
- /**
85
- * ### 🚜 Tractor the Text Extractor
86
- * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
87
- *
88
- * 1. Main Content Detection: Extract the main content from a URL by combining
89
- * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
90
- * custom adapters for major sites for article, author, date HTML classes.
91
- * 2. Basic HTML Standardization: Transform complex HTML into a simplified
92
- * reading-mode format of basic HTML, making it ideal for research note archival
93
- * and focused reading, with headings, images and links.
94
- * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
95
- * retrieve the complete video transcript including both manual captions and
96
- * auto-generated subtitles, maintaining proper timestamp synchronization and
97
- * speaker identification where available.
98
- * 4. PDF to HTML: Process PDF documents by extracting
99
- * formatted text while intelligently handling line breaks, page headers,
100
- * footnotes. The system analyzes text height statistics to automatically
101
- * infer heading levels, creating a properly structured document hierarchy
102
- * based on standard deviation from mean text size.
103
- * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
104
- * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
105
- * them to HTML while preserving formatting, styles, and document structure.
106
- * 6. Citation Information Extraction: Identify and extract citation metadata
107
- * including author names, publication dates, sources, and titles using HTML
108
- * meta tags and common class name patterns. The system validates author names
109
- * against a comprehensive database of 90,000 first and last names,
110
- * distinguishing between personal and organizational authors to properly
111
- * format citations.
112
- * 7. Author Name Formatting: Process author names by checking against
113
- * known name databases, handling affixes and titles correctly, and determining
114
- * whether to reverse the name order based on whether it's a personal or
115
- * organizational author, ensuring proper citation formatting.
116
- * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
117
- * @param {Object} [options]
118
- * @param {boolean} options.images default=true - include images
119
- * @param {boolean} options.links default=true - include links
120
- * @param {boolean} options.formatting default=true - preserve formatting
121
- * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
122
- * @param {number} options.timeout default=5 - http request timeout
123
- * @returns {{
124
- * title: string,
125
- * author_cite: string,
126
- * cite: string,
127
- * author: string,
128
- * date: string,
129
- * source: string,
130
- * html: string,
131
- * word_count: number
132
- * }}
133
- * cite - Cite in APA Format with Author name in Last, First Initial format
134
- * url - The URL of the article
135
- * html - The HTML content of the article
136
- * author - The author of the article
137
- * author_cite - Author name in Last, First Middle format
138
- * author_short - Author name in Last format
139
- * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
140
- * date - The publication date of the article
141
- * title - The title of the article
142
- * source - The source or origin of the article
143
- * word_count - The word count of the full text (without HTML tags)
144
- * @category Extract
145
- * @author [vtempest (2025)](https://github.com/vtempest)
146
- * @example
147
- * // Extract from URL
148
- * const result1 = await extractContent('https://example.com/article');
149
- *
150
- * // Extract from DOCX binary buffer
151
- * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
152
- * const result2 = await extractContent(docxBuffer);
153
- *
154
- * // Extract from DOM object
155
- * const result3 = await extractContent(document);
156
- */
157
- export async function extractContent(
158
- urlOrDoc:
159
- | string
160
- | Document
161
- | UrlLikeDocument
162
- | ArrayBuffer
163
- | Buffer
164
- | Uint8Array,
165
- options: ExtractContentOptions = {},
166
- ): Promise<ExtractedArticle> {
167
- var {
168
- images = true,
169
- links = true,
170
- formatting = true,
171
- absoluteURLs = true,
172
- timeout = 5,
173
- proxy = null,
174
- citeFormatMonthFull = false,
175
- citeFormatAuthorFull = true,
176
- } = options;
177
- let response: ExtractedArticle = {};
178
-
179
- let url, isPdf, isDocxBuffer;
180
-
181
- // Check if input is a binary buffer (DOCX)
182
- if (
183
- urlOrDoc instanceof ArrayBuffer ||
184
- urlOrDoc instanceof Uint8Array ||
185
- (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
186
- ) {
187
- isDocxBuffer = isBufferDOCX(urlOrDoc);
188
-
189
- if (isDocxBuffer) {
190
- // Handle DOCX binary buffer
191
- response.html = await convertDOCXToHTML(urlOrDoc, options);
192
- url = "buffer://docx"; // Placeholder URL for buffer input
193
- } else {
194
- return { error: "Binary buffer is not a valid DOCX file" };
195
- }
196
- } else if (
197
- typeof urlOrDoc === "string" &&
198
- /<\/[^>]+>/.test(urlOrDoc.trim())
199
- ) {
200
- console.log("[extractContent] input is raw HTML string");
201
- // If urlOrDoc is an HTML string, treat as HTML content
202
- options.url = options.url || "";
203
-
204
- response = extractContentAndCite(urlOrDoc, options);
205
- console.log("[extractContent] extractContentAndCite (raw html) result", {
206
- hasHtml: !!response?.html,
207
- htmlLength: response?.html?.length || 0,
208
- title: response?.title,
209
- error: response?.error,
210
- });
211
-
212
- return response;
213
- // if URL
214
- } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
215
- url = urlOrDoc;
216
- console.log("[extractContent] input is URL", { url });
217
-
218
- // check if google doc, then extract html or pdf file
219
- let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
220
- if (googleDocId) {
221
- url =
222
- googleDocId[1] === "file"
223
- ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
224
- : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
225
- console.log("[extractContent] rewrote google doc url", { url });
226
- }
227
-
228
- isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
229
- let youtubeID = getURLYoutubeVideo(url);
230
- console.log("[extractContent] branch detection", {
231
- isPdf,
232
- youtubeID,
233
- isDocx: url.endsWith(".docx"),
234
- });
235
-
236
- if (isPdf) {
237
- // pdf checker - use dynamic import to prevent build-time evaluation
238
- response = await convertPDFToHTML(url, options as any);
239
- console.log("[extractContent] pdf branch result", {
240
- hasHtml: !!response?.html,
241
- error: response?.error,
242
- });
243
- } else if (url.endsWith(".docx")) {
244
- response.html = await convertDOCXToHTML(url);
245
- console.log("[extractContent] docx branch result", {
246
- hasHtml: !!response?.html,
247
- });
248
-
249
- // check youtube
250
- } else if (youtubeID) {
251
- response = await convertYoutubeToText(url, options);
252
- console.log("[extractContent] youtube branch result", {
253
- hasHtml: !!response?.html,
254
- error: response?.error,
255
- });
256
- } else {
257
- console.log("[extractContent] scraping URL", { url, proxy });
258
-
259
- try {
260
- const html = await scrapeURL(url, {
261
- proxy,
262
- });
263
- console.log("[extractContent] scrapeURL returned", {
264
- url,
265
- hasHtml: !!html,
266
- htmlLength: typeof html === "string" ? html.length : 0,
267
- sample: typeof html === "string" ? html.slice(0, 200) : null,
268
- });
269
-
270
- // Check if scrapeURL returned an error object instead of HTML string
271
- if (typeof html !== "string" || !html) {
272
- console.error("[extractContent] scrapeURL failed or returned non-string", {
273
- url,
274
- typeofHtml: typeof html,
275
- isEmpty: !html,
276
- });
277
- return {
278
- error: "Failed to fetch HTML content",
279
- };
280
- }
281
-
282
- options.url = url;
283
- response = extractContentAndCite(html, options);
284
- console.log("[extractContent] extractContentAndCite result", {
285
- url,
286
- hasHtml: !!response?.html,
287
- htmlLength: response?.html?.length || 0,
288
- title: response?.title,
289
- error: response?.error,
290
- });
291
- } catch (scrapeError) {
292
- const err = scrapeError as Error;
293
- console.error("[extractContent] scrapeURL threw error", {
294
- url,
295
- message: err?.message,
296
- });
297
- return {
298
- error: `Failed to scrape URL: ${err?.message || String(scrapeError)}`,
299
- };
300
- }
301
- }
302
- } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
303
- //if passing in dom object document from front end
304
-
305
- url = urlOrDoc.location.href;
306
-
307
- //pdf checker for embedded docs
308
- if (urlOrDoc?.querySelectorAll)
309
- isPdf = urlOrDoc?.querySelectorAll(
310
- 'embed[type="application/pdf"]',
311
- )?.length;
312
- var youtubeID = getURLYoutubeVideo(url);
313
-
314
- if (isPdf) {
315
- response = await convertPDFToHTML(url, {});
316
- } else if (youtubeID) {
317
- // from front end
318
-
319
- //if on same domain page in chrome-extension
320
- options.useThirdPartyBackup = false;
321
- response = await convertYoutubeToText(url, options);
322
- } //pass doc to extract
323
- else response = extractContentAndCite(urlOrDoc as Document, options);
324
- } else {
325
- // Handle other object types or invalid input
326
- return {
327
- error:
328
- "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
329
- };
330
- }
331
-
332
- //if no text
333
- if (response.error || !response.html) return { error: response.error };
334
-
335
- //word count of full text original, no html
336
- response.word_count = response.html
337
- ?.replace(/<[^>]*>/g, " ")
338
- .split(" ").length;
339
-
340
- //make APA cite
341
-
342
- var { author, author_cite, author_short, date, title, source } = response;
343
-
344
- var apa_cite_date =
345
- new Date(date).getFullYear() > 1971
346
- ? " (" +
347
- new Date(date).getFullYear() +
348
- ", " +
349
- new Date(date).toLocaleDateString("en-US", {
350
- month: citeFormatMonthFull ? "long" : "short",
351
- day: "numeric",
352
- }) +
353
- ")"
354
- : ""; //"(N.D.)";
355
-
356
- var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
357
- title || ""
358
- }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
359
-
360
- //shorten long urls by removing ?params=get used as state tracking
361
- if (url && url.includes("?") && url.length > 150)
362
- response.url = url.split("?")[0];
363
-
364
- //put url on top
365
- response = Object.assign({ url, cite }, response);
366
- return response;
367
- }
1
+ /**
2
+ * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
+ * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
+ */
5
+ import { extractContentAndCite } from "../html-to-content/html-to-content";
6
+ import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
7
+ import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
8
+ import { scrapeURL } from "./url-to-html";
9
+ import grab from "../utils/grab";
10
+
11
+ /**
12
+ * Dynamic PDF converter to avoid bundling pdfjs at build time
13
+ */
14
+ async function convertPDFToHTML(url: string, options: any) {
15
+ const { convertPDFToHTML: pdfConverter } = await import("extract-pdf");
16
+ return await pdfConverter(url, options);
17
+ }
18
+
19
+ async function isUrlPDF(url: string) {
20
+ try {
21
+ const buffer = await grab(url, { responseType: "arraybuffer", timeout: 5 });
22
+ if (!buffer || buffer.byteLength < 5) return false;
23
+ const chunk = new Uint8Array(buffer);
24
+ return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
25
+ } catch {
26
+ return false;
27
+ }
28
+ }
29
+
30
+ export interface ExtractContentOptions {
31
+ images?: boolean;
32
+ links?: boolean;
33
+ formatting?: boolean;
34
+ absoluteURLs?: boolean;
35
+ timeout?: number;
36
+ proxy?: string | null;
37
+ citeFormatMonthFull?: boolean;
38
+ citeFormatAuthorFull?: boolean;
39
+ url?: string;
40
+ useThirdPartyBackup?: boolean;
41
+ /** Preferred transcript languages when extracting YouTube videos. */
42
+ languages?: string[];
43
+ }
44
+
45
+ export interface ExtractedArticle {
46
+ cite?: string;
47
+ html?: string;
48
+ url?: string;
49
+ author?: string;
50
+ author_cite?: string;
51
+ author_short?: string;
52
+ author_type?: number | string;
53
+ date?: string;
54
+ title?: string;
55
+ source?: string;
56
+ word_count?: number;
57
+ format?: string;
58
+ error?: string | number;
59
+ }
60
+
61
+ type UrlLikeDocument = {
62
+ location?: { href?: string };
63
+ querySelectorAll?: (
64
+ selector: string,
65
+ ) => { length: number } | ArrayLike<unknown>;
66
+ };
67
+
68
+ /**
69
+ * @typedef {Object} Article
70
+ * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
71
+ * @property {string} html - The Basic HTML content of the article
72
+ * @property {string} url - The URL of the article
73
+ * @property {string} author - The full name of the author of the article
74
+ * @property {string} author_cite - Author name in Last, First Initial format
75
+ * @property {string} author_short - Author name in Last format
76
+ * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
77
+ * @property {string} date - The publication date of the article
78
+ * @property {string} title - The title of the article
79
+ * @property {string} source - The source or publisher of the article
80
+ * @property {number} word_count - The word count of the full text (without HTML tags)
81
+ * @category Extract
82
+ */
83
+
84
+ /**
85
+ * ### 🚜 Tractor the Text Extractor
86
+ * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
87
+ *
88
+ * 1. Main Content Detection: Extract the main content from a URL by combining
89
+ * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
90
+ * custom adapters for major sites for article, author, date HTML classes.
91
+ * 2. Basic HTML Standardization: Transform complex HTML into a simplified
92
+ * reading-mode format of basic HTML, making it ideal for research note archival
93
+ * and focused reading, with headings, images and links.
94
+ * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
95
+ * retrieve the complete video transcript including both manual captions and
96
+ * auto-generated subtitles, maintaining proper timestamp synchronization and
97
+ * speaker identification where available.
98
+ * 4. PDF to HTML: Process PDF documents by extracting
99
+ * formatted text while intelligently handling line breaks, page headers,
100
+ * footnotes. The system analyzes text height statistics to automatically
101
+ * infer heading levels, creating a properly structured document hierarchy
102
+ * based on standard deviation from mean text size.
103
+ * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
104
+ * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
105
+ * them to HTML while preserving formatting, styles, and document structure.
106
+ * 6. Citation Information Extraction: Identify and extract citation metadata
107
+ * including author names, publication dates, sources, and titles using HTML
108
+ * meta tags and common class name patterns. The system validates author names
109
+ * against a comprehensive database of 90,000 first and last names,
110
+ * distinguishing between personal and organizational authors to properly
111
+ * format citations.
112
+ * 7. Author Name Formatting: Process author names by checking against
113
+ * known name databases, handling affixes and titles correctly, and determining
114
+ * whether to reverse the name order based on whether it's a personal or
115
+ * organizational author, ensuring proper citation formatting.
116
+ * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
117
+ * @param {Object} [options]
118
+ * @param {boolean} options.images default=true - include images
119
+ * @param {boolean} options.links default=true - include links
120
+ * @param {boolean} options.formatting default=true - preserve formatting
121
+ * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
122
+ * @param {number} options.timeout default=5 - http request timeout
123
+ * @returns {{
124
+ * title: string,
125
+ * author_cite: string,
126
+ * cite: string,
127
+ * author: string,
128
+ * date: string,
129
+ * source: string,
130
+ * html: string,
131
+ * word_count: number
132
+ * }}
133
+ * cite - Cite in APA Format with Author name in Last, First Initial format
134
+ * url - The URL of the article
135
+ * html - The HTML content of the article
136
+ * author - The author of the article
137
+ * author_cite - Author name in Last, First Middle format
138
+ * author_short - Author name in Last format
139
+ * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
140
+ * date - The publication date of the article
141
+ * title - The title of the article
142
+ * source - The source or origin of the article
143
+ * word_count - The word count of the full text (without HTML tags)
144
+ * @category Extract
145
+ * @author [vtempest (2025)](https://github.com/vtempest)
146
+ * @example
147
+ * // Extract from URL
148
+ * const result1 = await extractContent('https://example.com/article');
149
+ *
150
+ * // Extract from DOCX binary buffer
151
+ * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
152
+ * const result2 = await extractContent(docxBuffer);
153
+ *
154
+ * // Extract from DOM object
155
+ * const result3 = await extractContent(document);
156
+ */
157
+ export async function extractContent(
158
+ urlOrDoc:
159
+ | string
160
+ | Document
161
+ | UrlLikeDocument
162
+ | ArrayBuffer
163
+ | Buffer
164
+ | Uint8Array,
165
+ options: ExtractContentOptions = {},
166
+ ): Promise<ExtractedArticle> {
167
+ var {
168
+ images = true,
169
+ links = true,
170
+ formatting = true,
171
+ absoluteURLs = true,
172
+ timeout = 5,
173
+ proxy = null,
174
+ citeFormatMonthFull = false,
175
+ citeFormatAuthorFull = true,
176
+ } = options;
177
+ let response: ExtractedArticle = {};
178
+
179
+ let url, isPdf, isDocxBuffer;
180
+
181
+ // Check if input is a binary buffer (DOCX)
182
+ if (
183
+ urlOrDoc instanceof ArrayBuffer ||
184
+ urlOrDoc instanceof Uint8Array ||
185
+ (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
186
+ ) {
187
+ isDocxBuffer = isBufferDOCX(urlOrDoc);
188
+
189
+ if (isDocxBuffer) {
190
+ // Handle DOCX binary buffer
191
+ response.html = await convertDOCXToHTML(urlOrDoc, options);
192
+ url = "buffer://docx"; // Placeholder URL for buffer input
193
+ } else {
194
+ return { error: "Binary buffer is not a valid DOCX file" };
195
+ }
196
+ } else if (
197
+ typeof urlOrDoc === "string" &&
198
+ /<\/[^>]+>/.test(urlOrDoc.trim())
199
+ ) {
200
+ console.log("[extractContent] input is raw HTML string");
201
+ // If urlOrDoc is an HTML string, treat as HTML content
202
+ options.url = options.url || "";
203
+
204
+ response = extractContentAndCite(urlOrDoc, options);
205
+ console.log("[extractContent] extractContentAndCite (raw html) result", {
206
+ hasHtml: !!response?.html,
207
+ htmlLength: response?.html?.length || 0,
208
+ title: response?.title,
209
+ error: response?.error,
210
+ });
211
+
212
+ return response;
213
+ // if URL
214
+ } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
215
+ url = urlOrDoc;
216
+ console.log("[extractContent] input is URL", { url });
217
+
218
+ // check if google doc, then extract html or pdf file
219
+ let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
220
+ if (googleDocId) {
221
+ url =
222
+ googleDocId[1] === "file"
223
+ ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
224
+ : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
225
+ console.log("[extractContent] rewrote google doc url", { url });
226
+ }
227
+
228
+ isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
229
+ let youtubeID = getURLYoutubeVideo(url);
230
+ console.log("[extractContent] branch detection", {
231
+ isPdf,
232
+ youtubeID,
233
+ isDocx: url.endsWith(".docx"),
234
+ });
235
+
236
+ if (isPdf) {
237
+ // pdf checker - use dynamic import to prevent build-time evaluation
238
+ response = await convertPDFToHTML(url, options as any);
239
+ console.log("[extractContent] pdf branch result", {
240
+ hasHtml: !!response?.html,
241
+ error: response?.error,
242
+ });
243
+ } else if (url.endsWith(".docx")) {
244
+ response.html = await convertDOCXToHTML(url);
245
+ console.log("[extractContent] docx branch result", {
246
+ hasHtml: !!response?.html,
247
+ });
248
+
249
+ // check youtube
250
+ } else if (youtubeID) {
251
+ response = await convertYoutubeToText(url, options);
252
+ console.log("[extractContent] youtube branch result", {
253
+ hasHtml: !!response?.html,
254
+ error: response?.error,
255
+ });
256
+ } else {
257
+ console.log("[extractContent] scraping URL", { url, proxy });
258
+
259
+ try {
260
+ const html = await scrapeURL(url, {
261
+ proxy,
262
+ });
263
+ console.log("[extractContent] scrapeURL returned", {
264
+ url,
265
+ hasHtml: !!html,
266
+ htmlLength: typeof html === "string" ? html.length : 0,
267
+ sample: typeof html === "string" ? html.slice(0, 200) : null,
268
+ });
269
+
270
+ // Check if scrapeURL returned an error object instead of HTML string
271
+ if (typeof html !== "string" || !html) {
272
+ console.error("[extractContent] scrapeURL failed or returned non-string", {
273
+ url,
274
+ typeofHtml: typeof html,
275
+ isEmpty: !html,
276
+ });
277
+ return {
278
+ error: "Failed to fetch HTML content",
279
+ };
280
+ }
281
+
282
+ options.url = url;
283
+ response = extractContentAndCite(html, options);
284
+ console.log("[extractContent] extractContentAndCite result", {
285
+ url,
286
+ hasHtml: !!response?.html,
287
+ htmlLength: response?.html?.length || 0,
288
+ title: response?.title,
289
+ error: response?.error,
290
+ });
291
+ } catch (scrapeError) {
292
+ const err = scrapeError as Error;
293
+ console.error("[extractContent] scrapeURL threw error", {
294
+ url,
295
+ message: err?.message,
296
+ });
297
+ return {
298
+ error: `Failed to scrape URL: ${err?.message || String(scrapeError)}`,
299
+ };
300
+ }
301
+ }
302
+ } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
303
+ //if passing in dom object document from front end
304
+
305
+ url = urlOrDoc.location.href;
306
+
307
+ //pdf checker for embedded docs
308
+ if (urlOrDoc?.querySelectorAll)
309
+ isPdf = urlOrDoc?.querySelectorAll(
310
+ 'embed[type="application/pdf"]',
311
+ )?.length;
312
+ var youtubeID = getURLYoutubeVideo(url);
313
+
314
+ if (isPdf) {
315
+ response = await convertPDFToHTML(url, {});
316
+ } else if (youtubeID) {
317
+ // from front end
318
+
319
+ //if on same domain page in chrome-extension
320
+ options.useThirdPartyBackup = false;
321
+ response = await convertYoutubeToText(url, options);
322
+ } //pass doc to extract
323
+ else response = extractContentAndCite(urlOrDoc as Document, options);
324
+ } else {
325
+ // Handle other object types or invalid input
326
+ return {
327
+ error:
328
+ "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
329
+ };
330
+ }
331
+
332
+ //if no text
333
+ if (response.error || !response.html) return { error: response.error };
334
+
335
+ //word count of full text original, no html
336
+ response.word_count = response.html
337
+ ?.replace(/<[^>]*>/g, " ")
338
+ .split(" ").length;
339
+
340
+ //make APA cite
341
+
342
+ var { author, author_cite, author_short, date, title, source } = response;
343
+
344
+ var apa_cite_date =
345
+ new Date(date).getFullYear() > 1971
346
+ ? " (" +
347
+ new Date(date).getFullYear() +
348
+ ", " +
349
+ new Date(date).toLocaleDateString("en-US", {
350
+ month: citeFormatMonthFull ? "long" : "short",
351
+ day: "numeric",
352
+ }) +
353
+ ")"
354
+ : ""; //"(N.D.)";
355
+
356
+ var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
357
+ title || ""
358
+ }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
359
+
360
+ //shorten long urls by removing ?params=get used as state tracking
361
+ if (url && url.includes("?") && url.length > 150)
362
+ response.url = url.split("?")[0];
363
+
364
+ //put url on top
365
+ response = Object.assign({ url, cite }, response);
366
+ return response;
367
+ }