extract-webpage 1.2.34 → 1.2.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,332 +1,332 @@
1
- /**
2
- * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
- * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
- */
5
- import grab from "grab-url";
6
- import { extractContentAndCite } from "../html-to-content/html-to-content";
7
- import { getURLYoutubeVideo, convertYoutubeToText } from "extract-youtube";
8
- import { convertPDFToHTML } from "extract-pdf";
9
- async function isUrlPDF(url: string) {
10
- try {
11
- const buffer = await grab(url, { responseType: "arraybuffer", timeout: 10 });
12
- if (!buffer || buffer.byteLength < 5) return false;
13
- const chunk = new Uint8Array(buffer);
14
- return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
15
- } catch {
16
- return false;
17
- }
18
- }
19
- import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
20
- import { scrapeURL } from "./url-to-html";
21
-
22
- export interface ExtractContentOptions {
23
- images?: boolean;
24
- links?: boolean;
25
- formatting?: boolean;
26
- absoluteURLs?: boolean;
27
- timeout?: number;
28
- proxy?: string | null;
29
- citeFormatMonthFull?: boolean;
30
- citeFormatAuthorFull?: boolean;
31
- url?: string;
32
- useThirdPartyBackup?: boolean;
33
- }
34
-
35
- export interface ExtractedArticle {
36
- cite?: string;
37
- html?: string;
38
- url?: string;
39
- author?: string;
40
- author_cite?: string;
41
- author_short?: string;
42
- author_type?: number | string;
43
- date?: string;
44
- title?: string;
45
- source?: string;
46
- word_count?: number;
47
- format?: string;
48
- error?: string | number;
49
- }
50
-
51
- type UrlLikeDocument = {
52
- location?: { href?: string };
53
- querySelectorAll?: (
54
- selector: string,
55
- ) => { length: number } | ArrayLike<unknown>;
56
- };
57
-
58
- /**
59
- * @typedef {Object} Article
60
- * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
61
- * @property {string} html - The Basic HTML content of the article
62
- * @property {string} url - The URL of the article
63
- * @property {string} author - The full name of the author of the article
64
- * @property {string} author_cite - Author name in Last, First Initial format
65
- * @property {string} author_short - Author name in Last format
66
- * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
67
- * @property {string} date - The publication date of the article
68
- * @property {string} title - The title of the article
69
- * @property {string} source - The source or publisher of the article
70
- * @property {number} word_count - The word count of the full text (without HTML tags)
71
- * @category Extract
72
- */
73
-
74
- /**
75
- * ### 🚜 Tractor the Text Extractor
76
- * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
77
- *
78
- * 1. Main Content Detection: Extract the main content from a URL by combining
79
- * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
80
- * custom adapters for major sites for article, author, date HTML classes.
81
- * 2. Basic HTML Standardization: Transform complex HTML into a simplified
82
- * reading-mode format of basic HTML, making it ideal for research note archival
83
- * and focused reading, with headings, images and links.
84
- * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
85
- * retrieve the complete video transcript including both manual captions and
86
- * auto-generated subtitles, maintaining proper timestamp synchronization and
87
- * speaker identification where available.
88
- * 4. PDF to HTML: Process PDF documents by extracting
89
- * formatted text while intelligently handling line breaks, page headers,
90
- * footnotes. The system analyzes text height statistics to automatically
91
- * infer heading levels, creating a properly structured document hierarchy
92
- * based on standard deviation from mean text size.
93
- * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
94
- * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
95
- * them to HTML while preserving formatting, styles, and document structure.
96
- * 6. Citation Information Extraction: Identify and extract citation metadata
97
- * including author names, publication dates, sources, and titles using HTML
98
- * meta tags and common class name patterns. The system validates author names
99
- * against a comprehensive database of 90,000 first and last names,
100
- * distinguishing between personal and organizational authors to properly
101
- * format citations.
102
- * 7. Author Name Formatting: Process author names by checking against
103
- * known name databases, handling affixes and titles correctly, and determining
104
- * whether to reverse the name order based on whether it's a personal or
105
- * organizational author, ensuring proper citation formatting.
106
- * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
107
- * @param {Object} [options]
108
- * @param {boolean} options.images default=true - include images
109
- * @param {boolean} options.links default=true - include links
110
- * @param {boolean} options.formatting default=true - preserve formatting
111
- * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
112
- * @param {number} options.timeout default=5 - http request timeout
113
- * @returns {{
114
- * title: string,
115
- * author_cite: string,
116
- * cite: string,
117
- * author: string,
118
- * date: string,
119
- * source: string,
120
- * html: string,
121
- * word_count: number
122
- * }}
123
- * cite - Cite in APA Format with Author name in Last, First Initial format
124
- * url - The URL of the article
125
- * html - The HTML content of the article
126
- * author - The author of the article
127
- * author_cite - Author name in Last, First Middle format
128
- * author_short - Author name in Last format
129
- * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
130
- * date - The publication date of the article
131
- * title - The title of the article
132
- * source - The source or origin of the article
133
- * word_count - The word count of the full text (without HTML tags)
134
- * @category Extract
135
- * @author [vtempest (2025)](https://github.com/vtempest)
136
- * @example
137
- * // Extract from URL
138
- * const result1 = await extractContent('https://example.com/article');
139
- *
140
- * // Extract from DOCX binary buffer
141
- * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
142
- * const result2 = await extractContent(docxBuffer);
143
- *
144
- * // Extract from DOM object
145
- * const result3 = await extractContent(document);
146
- */
147
- export async function extractContent(
148
- urlOrDoc:
149
- | string
150
- | Document
151
- | UrlLikeDocument
152
- | ArrayBuffer
153
- | Buffer
154
- | Uint8Array,
155
- options: ExtractContentOptions = {},
156
- ): Promise<ExtractedArticle> {
157
- var {
158
- images = true,
159
- links = true,
160
- formatting = true,
161
- absoluteURLs = true,
162
- timeout = 10,
163
- proxy = null,
164
- citeFormatMonthFull = false,
165
- citeFormatAuthorFull = true,
166
- } = options;
167
- let response: ExtractedArticle = {};
168
-
169
- let url, isPdf, isDocxBuffer;
170
-
171
- // Check if input is a binary buffer (DOCX)
172
- if (
173
- urlOrDoc instanceof ArrayBuffer ||
174
- urlOrDoc instanceof Uint8Array ||
175
- (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
176
- ) {
177
- isDocxBuffer = isBufferDOCX(urlOrDoc);
178
-
179
- if (isDocxBuffer) {
180
- // Handle DOCX binary buffer
181
- response.html = await convertDOCXToHTML(urlOrDoc, options);
182
- url = "buffer://docx"; // Placeholder URL for buffer input
183
- } else {
184
- return { error: "Binary buffer is not a valid DOCX file" };
185
- }
186
- } else if (
187
- typeof urlOrDoc === "string" &&
188
- /<\/[^>]+>/.test(urlOrDoc.trim())
189
- ) {
190
- console.log("[extractContent] input is raw HTML string");
191
- // If urlOrDoc is an HTML string, treat as HTML content
192
- options.url = options.url || "";
193
-
194
- response = extractContentAndCite(urlOrDoc, options);
195
- console.log("[extractContent] extractContentAndCite (raw html) result", {
196
- hasHtml: !!response?.html,
197
- htmlLength: response?.html?.length || 0,
198
- title: response?.title,
199
- error: response?.error,
200
- });
201
-
202
- return response;
203
- // if URL
204
- } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
205
- url = urlOrDoc;
206
- console.log("[extractContent] input is URL", { url });
207
-
208
- // check if google doc, then extract html or pdf file
209
- let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
210
- if (googleDocId) {
211
- url =
212
- googleDocId[1] === "file"
213
- ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
214
- : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
215
- console.log("[extractContent] rewrote google doc url", { url });
216
- }
217
-
218
- isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
219
- let youtubeID = getURLYoutubeVideo(url);
220
- console.log("[extractContent] branch detection", {
221
- isPdf,
222
- youtubeID,
223
- isDocx: url.endsWith(".docx"),
224
- });
225
-
226
- if (isPdf) {
227
- // pdf checker - use dynamic import to prevent build-time evaluation
228
- response = await convertPDFToHTML(url, options as any);
229
- console.log("[extractContent] pdf branch result", {
230
- hasHtml: !!response?.html,
231
- error: response?.error,
232
- });
233
- } else if (url.endsWith(".docx")) {
234
- response.html = await convertDOCXToHTML(url);
235
- console.log("[extractContent] docx branch result", {
236
- hasHtml: !!response?.html,
237
- });
238
-
239
- // check youtube
240
- } else if (youtubeID) {
241
- response = await convertYoutubeToText(url, options);
242
- console.log("[extractContent] youtube branch result", {
243
- hasHtml: !!response?.html,
244
- error: response?.error,
245
- });
246
- } else {
247
- console.log("[extractContent] scraping URL", { url, proxy });
248
- const html = await scrapeURL(url, {
249
- proxy,
250
- });
251
- console.log("[extractContent] scrapeURL returned", {
252
- url,
253
- hasHtml: !!html,
254
- htmlLength: typeof html === "string" ? html.length : 0,
255
- sample: typeof html === "string" ? html.slice(0, 200) : null,
256
- });
257
- options.url = url;
258
- response = extractContentAndCite(html, options);
259
- console.log("[extractContent] extractContentAndCite result", {
260
- url,
261
- hasHtml: !!response?.html,
262
- htmlLength: response?.html?.length || 0,
263
- title: response?.title,
264
- error: response?.error,
265
- });
266
- }
267
- } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
268
- //if passing in dom object document from front end
269
-
270
- url = urlOrDoc.location.href;
271
-
272
- //pdf checker for embedded docs
273
- if (urlOrDoc?.querySelectorAll)
274
- isPdf = urlOrDoc?.querySelectorAll(
275
- 'embed[type="application/pdf"]',
276
- )?.length;
277
- var youtubeID = getURLYoutubeVideo(url);
278
-
279
- if (isPdf) {
280
- response = await convertPDFToHTML(url, {});
281
- } else if (youtubeID) {
282
- // from front end
283
-
284
- //if on same domain page in chrome-extension
285
- options.useThirdPartyBackup = false;
286
- response = await convertYoutubeToText(url, options);
287
- } //pass doc to extract
288
- else response = extractContentAndCite(urlOrDoc as Document, options);
289
- } else {
290
- // Handle other object types or invalid input
291
- return {
292
- error:
293
- "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
294
- };
295
- }
296
-
297
- //if no text
298
- if (response.error || !response.html) return { error: response.error };
299
-
300
- //word count of full text original, no html
301
- response.word_count = response.html
302
- ?.replace(/<[^>]*>/g, " ")
303
- .split(" ").length;
304
-
305
- //make APA cite
306
-
307
- var { author, author_cite, author_short, date, title, source } = response;
308
-
309
- var apa_cite_date =
310
- new Date(date).getFullYear() > 1971
311
- ? " (" +
312
- new Date(date).getFullYear() +
313
- ", " +
314
- new Date(date).toLocaleDateString("en-US", {
315
- month: citeFormatMonthFull ? "long" : "short",
316
- day: "numeric",
317
- }) +
318
- ")"
319
- : ""; //"(N.D.)";
320
-
321
- var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
322
- title || ""
323
- }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
324
-
325
- //shorten long urls by removing ?params=get used as state tracking
326
- if (url && url.includes("?") && url.length > 150)
327
- response.url = url.split("?")[0];
328
-
329
- //put url on top
330
- response = Object.assign({ url, cite }, response);
331
- return response;
332
- }
1
+ /**
2
+ * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
+ * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
+ */
5
+ import grab from "grab-url";
6
+ import { extractContentAndCite } from "../html-to-content/html-to-content";
7
+ import { getURLYoutubeVideo, convertYoutubeToText } from "extract-youtube";
8
+ import { convertPDFToHTML } from "extract-pdf";
9
+ async function isUrlPDF(url: string) {
10
+ try {
11
+ const buffer = await grab(url, { responseType: "arraybuffer", timeout: 10 });
12
+ if (!buffer || buffer.byteLength < 5) return false;
13
+ const chunk = new Uint8Array(buffer);
14
+ return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
15
+ } catch {
16
+ return false;
17
+ }
18
+ }
19
+ import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
20
+ import { scrapeURL } from "./url-to-html";
21
+
22
+ export interface ExtractContentOptions {
23
+ images?: boolean;
24
+ links?: boolean;
25
+ formatting?: boolean;
26
+ absoluteURLs?: boolean;
27
+ timeout?: number;
28
+ proxy?: string | null;
29
+ citeFormatMonthFull?: boolean;
30
+ citeFormatAuthorFull?: boolean;
31
+ url?: string;
32
+ useThirdPartyBackup?: boolean;
33
+ }
34
+
35
+ export interface ExtractedArticle {
36
+ cite?: string;
37
+ html?: string;
38
+ url?: string;
39
+ author?: string;
40
+ author_cite?: string;
41
+ author_short?: string;
42
+ author_type?: number | string;
43
+ date?: string;
44
+ title?: string;
45
+ source?: string;
46
+ word_count?: number;
47
+ format?: string;
48
+ error?: string | number;
49
+ }
50
+
51
+ type UrlLikeDocument = {
52
+ location?: { href?: string };
53
+ querySelectorAll?: (
54
+ selector: string,
55
+ ) => { length: number } | ArrayLike<unknown>;
56
+ };
57
+
58
+ /**
59
+ * @typedef {Object} Article
60
+ * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
61
+ * @property {string} html - The Basic HTML content of the article
62
+ * @property {string} url - The URL of the article
63
+ * @property {string} author - The full name of the author of the article
64
+ * @property {string} author_cite - Author name in Last, First Initial format
65
+ * @property {string} author_short - Author name in Last format
66
+ * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
67
+ * @property {string} date - The publication date of the article
68
+ * @property {string} title - The title of the article
69
+ * @property {string} source - The source or publisher of the article
70
+ * @property {number} word_count - The word count of the full text (without HTML tags)
71
+ * @category Extract
72
+ */
73
+
74
+ /**
75
+ * ### 🚜 Tractor the Text Extractor
76
+ * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
77
+ *
78
+ * 1. Main Content Detection: Extract the main content from a URL by combining
79
+ * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
80
+ * custom adapters for major sites for article, author, date HTML classes.
81
+ * 2. Basic HTML Standardization: Transform complex HTML into a simplified
82
+ * reading-mode format of basic HTML, making it ideal for research note archival
83
+ * and focused reading, with headings, images and links.
84
+ * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
85
+ * retrieve the complete video transcript including both manual captions and
86
+ * auto-generated subtitles, maintaining proper timestamp synchronization and
87
+ * speaker identification where available.
88
+ * 4. PDF to HTML: Process PDF documents by extracting
89
+ * formatted text while intelligently handling line breaks, page headers,
90
+ * footnotes. The system analyzes text height statistics to automatically
91
+ * infer heading levels, creating a properly structured document hierarchy
92
+ * based on standard deviation from mean text size.
93
+ * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
94
+ * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
95
+ * them to HTML while preserving formatting, styles, and document structure.
96
+ * 6. Citation Information Extraction: Identify and extract citation metadata
97
+ * including author names, publication dates, sources, and titles using HTML
98
+ * meta tags and common class name patterns. The system validates author names
99
+ * against a comprehensive database of 90,000 first and last names,
100
+ * distinguishing between personal and organizational authors to properly
101
+ * format citations.
102
+ * 7. Author Name Formatting: Process author names by checking against
103
+ * known name databases, handling affixes and titles correctly, and determining
104
+ * whether to reverse the name order based on whether it's a personal or
105
+ * organizational author, ensuring proper citation formatting.
106
+ * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
107
+ * @param {Object} [options]
108
+ * @param {boolean} options.images default=true - include images
109
+ * @param {boolean} options.links default=true - include links
110
+ * @param {boolean} options.formatting default=true - preserve formatting
111
+ * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
112
+ * @param {number} options.timeout default=5 - http request timeout
113
+ * @returns {{
114
+ * title: string,
115
+ * author_cite: string,
116
+ * cite: string,
117
+ * author: string,
118
+ * date: string,
119
+ * source: string,
120
+ * html: string,
121
+ * word_count: number
122
+ * }}
123
+ * cite - Cite in APA Format with Author name in Last, First Initial format
124
+ * url - The URL of the article
125
+ * html - The HTML content of the article
126
+ * author - The author of the article
127
+ * author_cite - Author name in Last, First Middle format
128
+ * author_short - Author name in Last format
129
+ * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
130
+ * date - The publication date of the article
131
+ * title - The title of the article
132
+ * source - The source or origin of the article
133
+ * word_count - The word count of the full text (without HTML tags)
134
+ * @category Extract
135
+ * @author [vtempest (2025)](https://github.com/vtempest)
136
+ * @example
137
+ * // Extract from URL
138
+ * const result1 = await extractContent('https://example.com/article');
139
+ *
140
+ * // Extract from DOCX binary buffer
141
+ * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
142
+ * const result2 = await extractContent(docxBuffer);
143
+ *
144
+ * // Extract from DOM object
145
+ * const result3 = await extractContent(document);
146
+ */
147
+ export async function extractContent(
148
+ urlOrDoc:
149
+ | string
150
+ | Document
151
+ | UrlLikeDocument
152
+ | ArrayBuffer
153
+ | Buffer
154
+ | Uint8Array,
155
+ options: ExtractContentOptions = {},
156
+ ): Promise<ExtractedArticle> {
157
+ var {
158
+ images = true,
159
+ links = true,
160
+ formatting = true,
161
+ absoluteURLs = true,
162
+ timeout = 10,
163
+ proxy = null,
164
+ citeFormatMonthFull = false,
165
+ citeFormatAuthorFull = true,
166
+ } = options;
167
+ let response: ExtractedArticle = {};
168
+
169
+ let url, isPdf, isDocxBuffer;
170
+
171
+ // Check if input is a binary buffer (DOCX)
172
+ if (
173
+ urlOrDoc instanceof ArrayBuffer ||
174
+ urlOrDoc instanceof Uint8Array ||
175
+ (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
176
+ ) {
177
+ isDocxBuffer = isBufferDOCX(urlOrDoc);
178
+
179
+ if (isDocxBuffer) {
180
+ // Handle DOCX binary buffer
181
+ response.html = await convertDOCXToHTML(urlOrDoc, options);
182
+ url = "buffer://docx"; // Placeholder URL for buffer input
183
+ } else {
184
+ return { error: "Binary buffer is not a valid DOCX file" };
185
+ }
186
+ } else if (
187
+ typeof urlOrDoc === "string" &&
188
+ /<\/[^>]+>/.test(urlOrDoc.trim())
189
+ ) {
190
+ console.log("[extractContent] input is raw HTML string");
191
+ // If urlOrDoc is an HTML string, treat as HTML content
192
+ options.url = options.url || "";
193
+
194
+ response = extractContentAndCite(urlOrDoc, options);
195
+ console.log("[extractContent] extractContentAndCite (raw html) result", {
196
+ hasHtml: !!response?.html,
197
+ htmlLength: response?.html?.length || 0,
198
+ title: response?.title,
199
+ error: response?.error,
200
+ });
201
+
202
+ return response;
203
+ // if URL
204
+ } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
205
+ url = urlOrDoc;
206
+ console.log("[extractContent] input is URL", { url });
207
+
208
+ // check if google doc, then extract html or pdf file
209
+ let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
210
+ if (googleDocId) {
211
+ url =
212
+ googleDocId[1] === "file"
213
+ ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
214
+ : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
215
+ console.log("[extractContent] rewrote google doc url", { url });
216
+ }
217
+
218
+ isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
219
+ let youtubeID = getURLYoutubeVideo(url);
220
+ console.log("[extractContent] branch detection", {
221
+ isPdf,
222
+ youtubeID,
223
+ isDocx: url.endsWith(".docx"),
224
+ });
225
+
226
+ if (isPdf) {
227
+ // pdf checker - use dynamic import to prevent build-time evaluation
228
+ response = await convertPDFToHTML(url, options as any);
229
+ console.log("[extractContent] pdf branch result", {
230
+ hasHtml: !!response?.html,
231
+ error: response?.error,
232
+ });
233
+ } else if (url.endsWith(".docx")) {
234
+ response.html = await convertDOCXToHTML(url);
235
+ console.log("[extractContent] docx branch result", {
236
+ hasHtml: !!response?.html,
237
+ });
238
+
239
+ // check youtube
240
+ } else if (youtubeID) {
241
+ response = await convertYoutubeToText(url, options);
242
+ console.log("[extractContent] youtube branch result", {
243
+ hasHtml: !!response?.html,
244
+ error: response?.error,
245
+ });
246
+ } else {
247
+ console.log("[extractContent] scraping URL", { url, proxy });
248
+ const html = await scrapeURL(url, {
249
+ proxy,
250
+ });
251
+ console.log("[extractContent] scrapeURL returned", {
252
+ url,
253
+ hasHtml: !!html,
254
+ htmlLength: typeof html === "string" ? html.length : 0,
255
+ sample: typeof html === "string" ? html.slice(0, 200) : null,
256
+ });
257
+ options.url = url;
258
+ response = extractContentAndCite(html, options);
259
+ console.log("[extractContent] extractContentAndCite result", {
260
+ url,
261
+ hasHtml: !!response?.html,
262
+ htmlLength: response?.html?.length || 0,
263
+ title: response?.title,
264
+ error: response?.error,
265
+ });
266
+ }
267
+ } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
268
+ //if passing in dom object document from front end
269
+
270
+ url = urlOrDoc.location.href;
271
+
272
+ //pdf checker for embedded docs
273
+ if (urlOrDoc?.querySelectorAll)
274
+ isPdf = urlOrDoc?.querySelectorAll(
275
+ 'embed[type="application/pdf"]',
276
+ )?.length;
277
+ var youtubeID = getURLYoutubeVideo(url);
278
+
279
+ if (isPdf) {
280
+ response = await convertPDFToHTML(url, {});
281
+ } else if (youtubeID) {
282
+ // from front end
283
+
284
+ //if on same domain page in chrome-extension
285
+ options.useThirdPartyBackup = false;
286
+ response = await convertYoutubeToText(url, options);
287
+ } //pass doc to extract
288
+ else response = extractContentAndCite(urlOrDoc as Document, options);
289
+ } else {
290
+ // Handle other object types or invalid input
291
+ return {
292
+ error:
293
+ "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
294
+ };
295
+ }
296
+
297
+ //if no text
298
+ if (response.error || !response.html) return { error: response.error };
299
+
300
+ //word count of full text original, no html
301
+ response.word_count = response.html
302
+ ?.replace(/<[^>]*>/g, " ")
303
+ .split(" ").length;
304
+
305
+ //make APA cite
306
+
307
+ var { author, author_cite, author_short, date, title, source } = response;
308
+
309
+ var apa_cite_date =
310
+ new Date(date).getFullYear() > 1971
311
+ ? " (" +
312
+ new Date(date).getFullYear() +
313
+ ", " +
314
+ new Date(date).toLocaleDateString("en-US", {
315
+ month: citeFormatMonthFull ? "long" : "short",
316
+ day: "numeric",
317
+ }) +
318
+ ")"
319
+ : ""; //"(N.D.)";
320
+
321
+ var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
322
+ title || ""
323
+ }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
324
+
325
+ //shorten long urls by removing ?params=get used as state tracking
326
+ if (url && url.includes("?") && url.length > 150)
327
+ response.url = url.split("?")[0];
328
+
329
+ //put url on top
330
+ response = Object.assign({ url, cite }, response);
331
+ return response;
332
+ }