extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,332 @@
1
+ /**
2
+ * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
+ * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
+ */
5
+ import grab from "grab-url";
6
+ import { extractContentAndCite } from "../html-to-content/html-to-content";
7
+ import { getURLYoutubeVideo, convertYoutubeToText } from "extract-youtube";
8
+ import { convertPDFToHTML } from "extract-pdf";
9
+ async function isUrlPDF(url: string) {
10
+ try {
11
+ const buffer = await grab(url, { responseType: "arraybuffer", timeout: 10 });
12
+ if (!buffer || buffer.byteLength < 5) return false;
13
+ const chunk = new Uint8Array(buffer);
14
+ return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
15
+ } catch {
16
+ return false;
17
+ }
18
+ }
19
+ import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
20
+ import { scrapeURL } from "./url-to-html";
21
+
22
+ export interface ExtractContentOptions {
23
+ images?: boolean;
24
+ links?: boolean;
25
+ formatting?: boolean;
26
+ absoluteURLs?: boolean;
27
+ timeout?: number;
28
+ proxy?: string | null;
29
+ citeFormatMonthFull?: boolean;
30
+ citeFormatAuthorFull?: boolean;
31
+ url?: string;
32
+ useThirdPartyBackup?: boolean;
33
+ }
34
+
35
+ export interface ExtractedArticle {
36
+ cite?: string;
37
+ html?: string;
38
+ url?: string;
39
+ author?: string;
40
+ author_cite?: string;
41
+ author_short?: string;
42
+ author_type?: number | string;
43
+ date?: string;
44
+ title?: string;
45
+ source?: string;
46
+ word_count?: number;
47
+ format?: string;
48
+ error?: string | number;
49
+ }
50
+
51
+ type UrlLikeDocument = {
52
+ location?: { href?: string };
53
+ querySelectorAll?: (
54
+ selector: string,
55
+ ) => { length: number } | ArrayLike<unknown>;
56
+ };
57
+
58
+ /**
59
+ * @typedef {Object} Article
60
+ * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
61
+ * @property {string} html - The Basic HTML content of the article
62
+ * @property {string} url - The URL of the article
63
+ * @property {string} author - The full name of the author of the article
64
+ * @property {string} author_cite - Author name in Last, First Initial format
65
+ * @property {string} author_short - Author name in Last format
66
+ * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
67
+ * @property {string} date - The publication date of the article
68
+ * @property {string} title - The title of the article
69
+ * @property {string} source - The source or publisher of the article
70
+ * @property {number} word_count - The word count of the full text (without HTML tags)
71
+ * @category Extract
72
+ */
73
+
74
+ /**
75
+ * ### 🚜 Tractor the Text Extractor
76
+ * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
77
+ *
78
+ * 1. Main Content Detection: Extract the main content from a URL by combining
79
+ * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
80
+ * custom adapters for major sites for article, author, date HTML classes.
81
+ * 2. Basic HTML Standardization: Transform complex HTML into a simplified
82
+ * reading-mode format of basic HTML, making it ideal for research note archival
83
+ * and focused reading, with headings, images and links.
84
+ * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
85
+ * retrieve the complete video transcript including both manual captions and
86
+ * auto-generated subtitles, maintaining proper timestamp synchronization and
87
+ * speaker identification where available.
88
+ * 4. PDF to HTML: Process PDF documents by extracting
89
+ * formatted text while intelligently handling line breaks, page headers,
90
+ * footnotes. The system analyzes text height statistics to automatically
91
+ * infer heading levels, creating a properly structured document hierarchy
92
+ * based on standard deviation from mean text size.
93
+ * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
94
+ * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
95
+ * them to HTML while preserving formatting, styles, and document structure.
96
+ * 6. Citation Information Extraction: Identify and extract citation metadata
97
+ * including author names, publication dates, sources, and titles using HTML
98
+ * meta tags and common class name patterns. The system validates author names
99
+ * against a comprehensive database of 90,000 first and last names,
100
+ * distinguishing between personal and organizational authors to properly
101
+ * format citations.
102
+ * 7. Author Name Formatting: Process author names by checking against
103
+ * known name databases, handling affixes and titles correctly, and determining
104
+ * whether to reverse the name order based on whether it's a personal or
105
+ * organizational author, ensuring proper citation formatting.
106
+ * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
107
+ * @param {Object} [options]
108
+ * @param {boolean} options.images default=true - include images
109
+ * @param {boolean} options.links default=true - include links
110
+ * @param {boolean} options.formatting default=true - preserve formatting
111
+ * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
112
+ * @param {number} options.timeout default=5 - http request timeout
113
+ * @returns {{
114
+ * title: string,
115
+ * author_cite: string,
116
+ * cite: string,
117
+ * author: string,
118
+ * date: string,
119
+ * source: string,
120
+ * html: string,
121
+ * word_count: number
122
+ * }}
123
+ * cite - Cite in APA Format with Author name in Last, First Initial format
124
+ * url - The URL of the article
125
+ * html - The HTML content of the article
126
+ * author - The author of the article
127
+ * author_cite - Author name in Last, First Middle format
128
+ * author_short - Author name in Last format
129
+ * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
130
+ * date - The publication date of the article
131
+ * title - The title of the article
132
+ * source - The source or origin of the article
133
+ * word_count - The word count of the full text (without HTML tags)
134
+ * @category Extract
135
+ * @author [vtempest (2025)](https://github.com/vtempest)
136
+ * @example
137
+ * // Extract from URL
138
+ * const result1 = await extractContent('https://example.com/article');
139
+ *
140
+ * // Extract from DOCX binary buffer
141
+ * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
142
+ * const result2 = await extractContent(docxBuffer);
143
+ *
144
+ * // Extract from DOM object
145
+ * const result3 = await extractContent(document);
146
+ */
147
+ export async function extractContent(
148
+ urlOrDoc:
149
+ | string
150
+ | Document
151
+ | UrlLikeDocument
152
+ | ArrayBuffer
153
+ | Buffer
154
+ | Uint8Array,
155
+ options: ExtractContentOptions = {},
156
+ ): Promise<ExtractedArticle> {
157
+ var {
158
+ images = true,
159
+ links = true,
160
+ formatting = true,
161
+ absoluteURLs = true,
162
+ timeout = 10,
163
+ proxy = null,
164
+ citeFormatMonthFull = false,
165
+ citeFormatAuthorFull = true,
166
+ } = options;
167
+ let response: ExtractedArticle = {};
168
+
169
+ let url, isPdf, isDocxBuffer;
170
+
171
+ // Check if input is a binary buffer (DOCX)
172
+ if (
173
+ urlOrDoc instanceof ArrayBuffer ||
174
+ urlOrDoc instanceof Uint8Array ||
175
+ (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
176
+ ) {
177
+ isDocxBuffer = isBufferDOCX(urlOrDoc);
178
+
179
+ if (isDocxBuffer) {
180
+ // Handle DOCX binary buffer
181
+ response.html = await convertDOCXToHTML(urlOrDoc, options);
182
+ url = "buffer://docx"; // Placeholder URL for buffer input
183
+ } else {
184
+ return { error: "Binary buffer is not a valid DOCX file" };
185
+ }
186
+ } else if (
187
+ typeof urlOrDoc === "string" &&
188
+ /<\/[^>]+>/.test(urlOrDoc.trim())
189
+ ) {
190
+ console.log("[extractContent] input is raw HTML string");
191
+ // If urlOrDoc is an HTML string, treat as HTML content
192
+ options.url = options.url || "";
193
+
194
+ response = extractContentAndCite(urlOrDoc, options);
195
+ console.log("[extractContent] extractContentAndCite (raw html) result", {
196
+ hasHtml: !!response?.html,
197
+ htmlLength: response?.html?.length || 0,
198
+ title: response?.title,
199
+ error: response?.error,
200
+ });
201
+
202
+ return response;
203
+ // if URL
204
+ } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
205
+ url = urlOrDoc;
206
+ console.log("[extractContent] input is URL", { url });
207
+
208
+ // check if google doc, then extract html or pdf file
209
+ let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
210
+ if (googleDocId) {
211
+ url =
212
+ googleDocId[1] === "file"
213
+ ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
214
+ : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
215
+ console.log("[extractContent] rewrote google doc url", { url });
216
+ }
217
+
218
+ isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
219
+ let youtubeID = getURLYoutubeVideo(url);
220
+ console.log("[extractContent] branch detection", {
221
+ isPdf,
222
+ youtubeID,
223
+ isDocx: url.endsWith(".docx"),
224
+ });
225
+
226
+ if (isPdf) {
227
+ // pdf checker - use dynamic import to prevent build-time evaluation
228
+ response = await convertPDFToHTML(url, options as any);
229
+ console.log("[extractContent] pdf branch result", {
230
+ hasHtml: !!response?.html,
231
+ error: response?.error,
232
+ });
233
+ } else if (url.endsWith(".docx")) {
234
+ response.html = await convertDOCXToHTML(url);
235
+ console.log("[extractContent] docx branch result", {
236
+ hasHtml: !!response?.html,
237
+ });
238
+
239
+ // check youtube
240
+ } else if (youtubeID) {
241
+ response = await convertYoutubeToText(url, options);
242
+ console.log("[extractContent] youtube branch result", {
243
+ hasHtml: !!response?.html,
244
+ error: response?.error,
245
+ });
246
+ } else {
247
+ console.log("[extractContent] scraping URL", { url, proxy });
248
+ const html = await scrapeURL(url, {
249
+ proxy,
250
+ });
251
+ console.log("[extractContent] scrapeURL returned", {
252
+ url,
253
+ hasHtml: !!html,
254
+ htmlLength: typeof html === "string" ? html.length : 0,
255
+ sample: typeof html === "string" ? html.slice(0, 200) : null,
256
+ });
257
+ options.url = url;
258
+ response = extractContentAndCite(html, options);
259
+ console.log("[extractContent] extractContentAndCite result", {
260
+ url,
261
+ hasHtml: !!response?.html,
262
+ htmlLength: response?.html?.length || 0,
263
+ title: response?.title,
264
+ error: response?.error,
265
+ });
266
+ }
267
+ } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
268
+ //if passing in dom object document from front end
269
+
270
+ url = urlOrDoc.location.href;
271
+
272
+ //pdf checker for embedded docs
273
+ if (urlOrDoc?.querySelectorAll)
274
+ isPdf = urlOrDoc?.querySelectorAll(
275
+ 'embed[type="application/pdf"]',
276
+ )?.length;
277
+ var youtubeID = getURLYoutubeVideo(url);
278
+
279
+ if (isPdf) {
280
+ response = await convertPDFToHTML(url, {});
281
+ } else if (youtubeID) {
282
+ // from front end
283
+
284
+ //if on same domain page in chrome-extension
285
+ options.useThirdPartyBackup = false;
286
+ response = await convertYoutubeToText(url, options);
287
+ } //pass doc to extract
288
+ else response = extractContentAndCite(urlOrDoc as Document, options);
289
+ } else {
290
+ // Handle other object types or invalid input
291
+ return {
292
+ error:
293
+ "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
294
+ };
295
+ }
296
+
297
+ //if no text
298
+ if (response.error || !response.html) return { error: response.error };
299
+
300
+ //word count of full text original, no html
301
+ response.word_count = response.html
302
+ ?.replace(/<[^>]*>/g, " ")
303
+ .split(" ").length;
304
+
305
+ //make APA cite
306
+
307
+ var { author, author_cite, author_short, date, title, source } = response;
308
+
309
+ var apa_cite_date =
310
+ new Date(date).getFullYear() > 1971
311
+ ? " (" +
312
+ new Date(date).getFullYear() +
313
+ ", " +
314
+ new Date(date).toLocaleDateString("en-US", {
315
+ month: citeFormatMonthFull ? "long" : "short",
316
+ day: "numeric",
317
+ }) +
318
+ ")"
319
+ : ""; //"(N.D.)";
320
+
321
+ var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
322
+ title || ""
323
+ }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
324
+
325
+ //shorten long urls by removing ?params=get used as state tracking
326
+ if (url && url.includes("?") && url.length > 150)
327
+ response.url = url.split("?")[0];
328
+
329
+ //put url on top
330
+ response = Object.assign({ url, cite }, response);
331
+ return response;
332
+ }
@@ -0,0 +1,368 @@
1
+ /**
2
+ * @fileoverview Unit tests for URL content extraction
3
+ */
4
+ import { extractContent } from "../url-to-content";
5
+ import { scrapeURL } from "../url-to-html";
6
+ import { extractContentAndCite } from "../../html-to-content/html-to-content";
7
+ import { convertYoutubeToText } from "../youtube-helpers";
8
+ import { convertPDFToHTML } from "extract-pdf";
9
+
10
+ // Mock dependencies
11
+ jest.mock("../url-to-html", () => ({
12
+ scrapeURL: jest.fn(),
13
+ }));
14
+
15
+ jest.mock("../../html-to-content/html-to-content", () => ({
16
+ extractContentAndCite: jest.fn(),
17
+ }));
18
+
19
+ jest.mock("../youtube-helpers", () => ({
20
+ getURLYoutubeVideo: jest.fn(),
21
+ convertYoutubeToText: jest.fn(),
22
+ }));
23
+
24
+ jest.mock("extract-pdf", () => ({
25
+ convertPDFToHTML: jest.fn(),
26
+ }));
27
+
28
+ jest.mock("grab-url", () => jest.fn());
29
+
30
+ const mockScrapeURL = scrapeURL as jest.MockedFunction<typeof scrapeURL>;
31
+ const mockExtractContentAndCite = extractContentAndCite as jest.MockedFunction<
32
+ typeof extractContentAndCite
33
+ >;
34
+ const mockConvertYoutubeToText =
35
+ convertYoutubeToText as jest.MockedFunction<typeof convertYoutubeToText>;
36
+ const mockConvertPDFToHTML = convertPDFToHTML as jest.MockedFunction<
37
+ typeof convertPDFToHTML
38
+ >;
39
+
40
+ describe("extractContent", () => {
41
+ beforeEach(() => {
42
+ jest.clearAllMocks();
43
+ });
44
+
45
+ describe("URL extraction", () => {
46
+ it("should extract content from a regular URL", async () => {
47
+ const mockHtml = "<html><body><article>Test content</article></body></html>";
48
+ const mockExtracted = {
49
+ title: "Test Article",
50
+ html: "<p>Test content</p>",
51
+ author: "John Doe",
52
+ date: "2024-01-01",
53
+ source: "Example",
54
+ word_count: 2,
55
+ };
56
+
57
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
58
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
59
+
60
+ const result = await extractContent("https://example.com/article");
61
+
62
+ expect(mockScrapeURL).toHaveBeenCalledWith("https://example.com/article", {
63
+ proxy: null,
64
+ });
65
+ expect(mockExtractContentAndCite).toHaveBeenCalledWith(mockHtml, {
66
+ url: "https://example.com/article",
67
+ images: true,
68
+ links: true,
69
+ formatting: true,
70
+ absoluteURLs: true,
71
+ timeout: 10,
72
+ proxy: null,
73
+ citeFormatMonthFull: false,
74
+ citeFormatAuthorFull: true,
75
+ });
76
+ expect(result.title).toBe("Test Article");
77
+ expect(result.html).toBe("<p>Test content</p>");
78
+ });
79
+
80
+ it("should handle scrapeURL returning error object", async () => {
81
+ mockScrapeURL.mockResolvedValueOnce({
82
+ error: "HTTP error: 403 Forbidden",
83
+ } as any);
84
+
85
+ const result = await extractContent("https://blocked-site.com/article");
86
+
87
+ expect(result.error).toBe("HTTP error: 403 Forbidden");
88
+ expect(mockExtractContentAndCite).not.toHaveBeenCalled();
89
+ });
90
+
91
+ it("should handle scrapeURL returning null", async () => {
92
+ mockScrapeURL.mockResolvedValueOnce(null as any);
93
+
94
+ const result = await extractContent("https://null-response.com/article");
95
+
96
+ expect(result.error).toBe("Failed to fetch HTML content");
97
+ expect(mockExtractContentAndCite).not.toHaveBeenCalled();
98
+ });
99
+
100
+ it("should handle scrapeURL returning undefined", async () => {
101
+ mockScrapeURL.mockResolvedValueOnce(undefined as any);
102
+
103
+ const result = await extractContent(
104
+ "https://undefined-response.com/article"
105
+ );
106
+
107
+ expect(result.error).toBe("Failed to fetch HTML content");
108
+ expect(mockExtractContentAndCite).not.toHaveBeenCalled();
109
+ });
110
+
111
+ it("should handle scrapeURL returning empty string", async () => {
112
+ mockScrapeURL.mockResolvedValueOnce("");
113
+
114
+ const result = await extractContent("https://empty-response.com/article");
115
+
116
+ expect(result.error).toBe("Failed to fetch HTML content");
117
+ expect(mockExtractContentAndCite).not.toHaveBeenCalled();
118
+ });
119
+
120
+ it("should handle scrapeURL throwing an error", async () => {
121
+ mockScrapeURL.mockRejectedValueOnce(new Error("Network timeout"));
122
+
123
+ await expect(
124
+ extractContent("https://timeout.com/article")
125
+ ).rejects.toThrow("Network timeout");
126
+ });
127
+
128
+ it("should handle extraction returning no HTML", async () => {
129
+ const mockHtml = "<html><body>Content</body></html>";
130
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
131
+ mockExtractContentAndCite.mockReturnValueOnce({
132
+ title: "Article",
133
+ html: "", // Empty HTML
134
+ } as any);
135
+
136
+ const result = await extractContent("https://example.com/empty");
137
+
138
+ expect(result.error).toBeUndefined(); // This is actually a success case with empty content
139
+ });
140
+
141
+ it("should extract from raw HTML string", async () => {
142
+ const mockHtml = "<html><body><p>Test content</p></body></html>";
143
+ const mockExtracted = {
144
+ title: "Test",
145
+ html: "<p>Test content</p>",
146
+ word_count: 2,
147
+ };
148
+
149
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
150
+
151
+ const result = await extractContent(mockHtml, { url: "https://example.com" });
152
+
153
+ expect(mockScrapeURL).not.toHaveBeenCalled();
154
+ expect(mockExtractContentAndCite).toHaveBeenCalledWith(mockHtml, expect.any(Object));
155
+ expect(result.title).toBe("Test");
156
+ });
157
+
158
+ it("should handle extraction with custom options", async () => {
159
+ const mockHtml = "<html><body>Content</body></html>";
160
+ const mockExtracted = {
161
+ title: "Test",
162
+ html: "<p>Content</p>",
163
+ word_count: 1,
164
+ };
165
+
166
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
167
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
168
+
169
+ await extractContent("https://example.com/article", {
170
+ images: false,
171
+ links: false,
172
+ formatting: false,
173
+ timeout: 15,
174
+ proxy: "https://proxy.example.com",
175
+ });
176
+
177
+ expect(mockScrapeURL).toHaveBeenCalledWith("https://example.com/article", {
178
+ proxy: "https://proxy.example.com",
179
+ });
180
+ expect(mockExtractContentAndCite).toHaveBeenCalledWith(
181
+ mockHtml,
182
+ expect.objectContaining({
183
+ images: false,
184
+ links: false,
185
+ formatting: false,
186
+ timeout: 15,
187
+ proxy: "https://proxy.example.com",
188
+ })
189
+ );
190
+ });
191
+ });
192
+
193
+ describe("YouTube extraction", () => {
194
+ it("should extract YouTube video transcript", async () => {
195
+ const youtubeHelpers = require("../youtube-helpers");
196
+ youtubeHelpers.getURLYoutubeVideo.mockReturnValueOnce("dQw4w9WgXcQ");
197
+
198
+ const mockTranscript = {
199
+ title: "Test Video",
200
+ html: "<p>Transcript content</p>",
201
+ source: "YouTube",
202
+ word_count: 2,
203
+ };
204
+
205
+ mockConvertYoutubeToText.mockResolvedValueOnce(mockTranscript);
206
+
207
+ const result = await extractContent(
208
+ "https://www.youtube.com/watch?v=dQw4w9WgXcQ"
209
+ );
210
+
211
+ expect(mockConvertYoutubeToText).toHaveBeenCalledWith(
212
+ "https://www.youtube.com/watch?v=dQw4w9WgXcQ",
213
+ expect.any(Object)
214
+ );
215
+ expect(result.title).toBe("Test Video");
216
+ expect(result.source).toBe("YouTube");
217
+ expect(mockScrapeURL).not.toHaveBeenCalled();
218
+ });
219
+ });
220
+
221
+ describe("PDF extraction", () => {
222
+ it("should extract content from PDF URLs", async () => {
223
+ const mockPdfContent = {
224
+ title: "PDF Document",
225
+ html: "<p>PDF content</p>",
226
+ word_count: 2,
227
+ };
228
+
229
+ mockConvertPDFToHTML.mockResolvedValueOnce(mockPdfContent as any);
230
+
231
+ const result = await extractContent("https://example.com/document.pdf");
232
+
233
+ expect(mockConvertPDFToHTML).toHaveBeenCalledWith(
234
+ "https://example.com/document.pdf",
235
+ expect.any(Object)
236
+ );
237
+ expect(result.title).toBe("PDF Document");
238
+ expect(mockScrapeURL).not.toHaveBeenCalled();
239
+ });
240
+ });
241
+
242
+ describe("Google Docs extraction", () => {
243
+ it("should rewrite Google Doc URLs to export format", async () => {
244
+ const mockHtml = "<html><body>Google Doc content</body></html>";
245
+ const mockExtracted = {
246
+ title: "Google Doc",
247
+ html: "<p>Content</p>",
248
+ word_count: 1,
249
+ };
250
+
251
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
252
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
253
+
254
+ await extractContent(
255
+ "https://docs.google.com/document/d/ABC123/edit"
256
+ );
257
+
258
+ expect(mockScrapeURL).toHaveBeenCalledWith(
259
+ "https://docs.google.com/document/d/ABC123/export?format=html",
260
+ expect.any(Object)
261
+ );
262
+ });
263
+
264
+ it("should rewrite Google Drive file URLs", async () => {
265
+ const mockPdfContent = {
266
+ title: "Drive PDF",
267
+ html: "<p>PDF from Drive</p>",
268
+ word_count: 3,
269
+ };
270
+
271
+ mockConvertPDFToHTML.mockResolvedValueOnce(mockPdfContent as any);
272
+
273
+ await extractContent(
274
+ "https://drive.google.com/file/d/ABC123/view"
275
+ );
276
+
277
+ expect(mockConvertPDFToHTML).toHaveBeenCalledWith(
278
+ "https://drive.google.com/uc?export=download&id=ABC123",
279
+ expect.any(Object)
280
+ );
281
+ });
282
+ });
283
+
284
+ describe("Error handling", () => {
285
+ it("should return error if extraction returns error", async () => {
286
+ const mockHtml = "<html><body>Content</body></html>";
287
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
288
+ mockExtractContentAndCite.mockReturnValueOnce({
289
+ error: "Failed to parse content",
290
+ } as any);
291
+
292
+ const result = await extractContent("https://example.com/article");
293
+
294
+ expect(result.error).toBe("Failed to parse content");
295
+ });
296
+
297
+ it("should return error for invalid input type", async () => {
298
+ const result = await extractContent({ invalid: "object" } as any);
299
+
300
+ expect(result.error).toContain("Invalid input type");
301
+ });
302
+ });
303
+
304
+ describe("Word count and citation", () => {
305
+ it("should calculate word count from HTML", async () => {
306
+ const mockHtml = "<html><body>Content</body></html>";
307
+ const mockExtracted = {
308
+ title: "Test Article",
309
+ html: "<p>This is a test article with some words</p>",
310
+ author_cite: "Doe, J.",
311
+ date: "2024-01-15",
312
+ source: "Example",
313
+ };
314
+
315
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
316
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
317
+
318
+ const result = await extractContent("https://example.com/article");
319
+
320
+ expect(result.word_count).toBeGreaterThan(0);
321
+ });
322
+
323
+ it("should generate APA citation", async () => {
324
+ const mockHtml = "<html><body>Content</body></html>";
325
+ const mockExtracted = {
326
+ title: "Test Article",
327
+ html: "<p>Content</p>",
328
+ author_cite: "Smith, J.",
329
+ date: "2024-01-15",
330
+ source: "Example News",
331
+ word_count: 1,
332
+ };
333
+
334
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
335
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
336
+
337
+ const result = await extractContent("https://example.com/article");
338
+
339
+ expect(result.cite).toContain("Smith, J.");
340
+ expect(result.cite).toContain("(2024");
341
+ expect(result.cite).toContain("Test Article");
342
+ expect(result.cite).toContain("Example News");
343
+ });
344
+
345
+ it("should shorten long URLs in citation", async () => {
346
+ const longUrl =
347
+ "https://example.com/article?" +
348
+ "param1=value1&param2=value2&param3=value3&" +
349
+ "tracking=12345&sessionid=abcdefghijklmnopqrstuvwxyz&" +
350
+ "utm_source=test&utm_medium=test&utm_campaign=test";
351
+
352
+ const mockHtml = "<html><body>Content</body></html>";
353
+ const mockExtracted = {
354
+ title: "Article",
355
+ html: "<p>Content</p>",
356
+ word_count: 1,
357
+ };
358
+
359
+ mockScrapeURL.mockResolvedValueOnce(mockHtml);
360
+ mockExtractContentAndCite.mockReturnValueOnce(mockExtracted);
361
+
362
+ const result = await extractContent(longUrl);
363
+
364
+ expect(result.url).toBe("https://example.com/article");
365
+ expect(result.url).not.toContain("?");
366
+ });
367
+ });
368
+ });