extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,318 @@
1
+ /**
2
+ * @module research/extractor/url-to-content/is-url-porn
3
+ * @description Research library module.
4
+ */
5
+ export interface IsURLPornOptions {
6
+ url?: string;
7
+ title?: string;
8
+ threshold?: number;
9
+ }
10
+
11
+ /**
12
+ * Determines if content is likely adult/porn based on configurable threshold
13
+ *
14
+ * @param {Object} options - Configuration object
15
+ * @param {string} [options.url] - URL to analyze (optional)
16
+ * @param {string} [options.title] - Page title to analyze (optional)
17
+ * @param {number} [options.threshold] - Probability threshold (0.5 default)
18
+ * @returns {boolean} True if likelihood exceeds threshold, false otherwise
19
+ * @author [vtempest (2025)](https://github.com/vtempest)
20
+ * @example
21
+ * isURLPorn({
22
+ * title: "Hot deals on sexy cars",
23
+ * threshold: 0.8
24
+ * });
25
+ * console.log(isPorn2); // false (low confidence)
26
+ */
27
+ export function isURLPorn(options: IsURLPornOptions = {}): boolean {
28
+ if (!options || typeof options !== "object") {
29
+ throw new Error("Options parameter must be an object");
30
+ }
31
+
32
+ const { url = "", title = "", threshold = 0.5 } = options;
33
+
34
+ if (threshold < 0 || threshold > 1) {
35
+ throw new Error("Threshold must be between 0 and 1");
36
+ }
37
+
38
+ if (!url && !title) {
39
+ return false;
40
+ }
41
+
42
+ const likelihood = calculateLikelihood(url, title);
43
+ return likelihood >= threshold;
44
+ }
45
+
46
+ /**
47
+ * Configuration object containing patterns and keywords for adult content detection
48
+ */
49
+ const ADULT_CONTENT_CONFIG = {
50
+ keywords: [
51
+ "adult",
52
+ "adultvideo",
53
+ "adultmovie",
54
+ "adultfilm",
55
+ "anal",
56
+ "bdsm",
57
+ "bestiality",
58
+ "bondage",
59
+ "boob",
60
+ "boobs",
61
+ "boobies",
62
+ "breast",
63
+ "bukkake",
64
+ "cameltoe",
65
+ "creampie",
66
+ "cock",
67
+ "cuckold",
68
+ "cunt",
69
+ "deepthroat",
70
+ "erotic",
71
+ "escort",
72
+ "facesitting",
73
+ "facial",
74
+ "felching",
75
+ "fetish",
76
+ "fisting",
77
+ "gloryhole",
78
+ "gonzo",
79
+ "hentai",
80
+ "incest",
81
+ "lesbian",
82
+ "lolicon",
83
+ "naked",
84
+ "naughty",
85
+ "nude",
86
+ "orgasm",
87
+ "orgy",
88
+ "pegging",
89
+ "penis",
90
+ "playboy",
91
+ "porn",
92
+ "pornography",
93
+ "pussy",
94
+ "rimjob",
95
+ "scat",
96
+ "semen",
97
+ "sperm",
98
+ "sexvideo",
99
+ "transsexual",
100
+ "transgender",
101
+ "threesome",
102
+ "twink",
103
+ "upskirt",
104
+ "vagina",
105
+ "virgin",
106
+ "whore",
107
+ "xxx",
108
+ "yaoi",
109
+ "yiff",
110
+ "youporn",
111
+ ],
112
+ urlPatterns: [
113
+ "porn",
114
+ "xxx",
115
+ "adult",
116
+ "nude",
117
+ "naked",
118
+ "escort",
119
+ "fetish",
120
+ "tube",
121
+ "redtube",
122
+ "pornhub",
123
+ "xnxx",
124
+ "xhamster",
125
+ "youporn",
126
+ "xvideos",
127
+ "spankbang",
128
+ "eporner",
129
+ "tnaflix",
130
+ ],
131
+ titlePhrases: [
132
+ "free porn",
133
+ "xxx videos",
134
+ "adult videos",
135
+ "sex videos",
136
+ "nude pics",
137
+ "naked girls",
138
+ "hot babes",
139
+ "sexy women",
140
+ "porn tube",
141
+ "adult tube",
142
+ "sex tube",
143
+ "xxx tube",
144
+ ],
145
+ strongIndicators: [
146
+ "pornhub",
147
+ "xnxx",
148
+ "xhamster",
149
+ "youporn",
150
+ "xvideos",
151
+ "redtube",
152
+ "porn",
153
+ "xxx",
154
+ "adult",
155
+ "nude",
156
+ "naked",
157
+ "sex",
158
+ ],
159
+ };
160
+
161
+ /**
162
+ * Calculates the likelihood (0-1) that content contains adult material
163
+ *
164
+ * @param {string} url - URL to analyze
165
+ * @param {string} title - Title to analyze
166
+ * @returns {number} Likelihood score between 0 and 1
167
+ */
168
+ function calculateLikelihood(url: string, title: string): number {
169
+ let totalScore = 0;
170
+ let maxPossibleScore = 0;
171
+ const flags = "gi";
172
+
173
+ // Analyze URL
174
+ if (url && typeof url === "string") {
175
+ const urlScore = analyzeUrl(url, flags);
176
+ totalScore += urlScore.score;
177
+ maxPossibleScore += urlScore.maxScore;
178
+ }
179
+
180
+ // Analyze title
181
+ if (title && typeof title === "string") {
182
+ const titleScore = analyzeTitle(title, flags);
183
+ totalScore += titleScore.score;
184
+ maxPossibleScore += titleScore.maxScore;
185
+ }
186
+
187
+ // Normalize score to 0-1 range
188
+ if (maxPossibleScore === 0) {
189
+ return 0;
190
+ }
191
+
192
+ const normalizedScore = Math.min(totalScore / maxPossibleScore, 1);
193
+
194
+ // Apply sigmoid function to create more realistic probability distribution
195
+ return applySigmoid(normalizedScore);
196
+ }
197
+
198
+ /**
199
+ * Analyzes a URL for adult content indicators
200
+ *
201
+ * @param {string} url - URL to analyze
202
+ * @param {string} flags - Regex flags
203
+ * @returns {Object} Score object with score and maxScore properties
204
+ */
205
+ function analyzeUrl(url: string, flags: string): { score: number; maxScore: number } {
206
+ let score = 0;
207
+ const maxScore = 100;
208
+
209
+ try {
210
+ // Parse URL components
211
+ const urlObj = new URL(url);
212
+ const hostname = urlObj.hostname.toLowerCase();
213
+ const pathname = urlObj.pathname.toLowerCase();
214
+ const searchParams = urlObj.search.toLowerCase();
215
+ const fullUrl = url.toLowerCase();
216
+
217
+ // Check for strong indicators in hostname (highest weight)
218
+ ADULT_CONTENT_CONFIG.strongIndicators.forEach((indicator) => {
219
+ if (hostname.includes(indicator)) {
220
+ score += 30; // High confidence
221
+ }
222
+ });
223
+
224
+ // Check URL patterns in all components
225
+ ADULT_CONTENT_CONFIG.urlPatterns.forEach((pattern) => {
226
+ const regex = new RegExp(pattern, flags);
227
+ if (regex.test(hostname)) score += 20;
228
+ if (regex.test(pathname)) score += 15;
229
+ if (regex.test(searchParams)) score += 10;
230
+ });
231
+
232
+ // Check general keywords in full URL
233
+ ADULT_CONTENT_CONFIG.keywords.forEach((keyword) => {
234
+ const regex = new RegExp(`\\b${escapeRegex(keyword)}\\b`, flags);
235
+ if (regex.test(fullUrl)) {
236
+ score += 5;
237
+ }
238
+ });
239
+ } catch (error) {
240
+ // If URL parsing fails, treat as plain text
241
+ ADULT_CONTENT_CONFIG.urlPatterns.forEach((pattern) => {
242
+ const regex = new RegExp(pattern, flags);
243
+ if (regex.test(url)) {
244
+ score += 15;
245
+ }
246
+ });
247
+
248
+ ADULT_CONTENT_CONFIG.keywords.forEach((keyword) => {
249
+ const regex = new RegExp(`\\b${escapeRegex(keyword)}\\b`, flags);
250
+ if (regex.test(url)) {
251
+ score += 5;
252
+ }
253
+ });
254
+ }
255
+
256
+ return { score: Math.min(score, maxScore), maxScore };
257
+ }
258
+
259
+ /**
260
+ * Analyzes a title for adult content indicators
261
+ *
262
+ * @param {string} title - Title to analyze
263
+ * @param {string} flags - Regex flags
264
+ * @returns {Object} Score object with score and maxScore properties
265
+ */
266
+ function analyzeTitle(title: string, flags: string): { score: number; maxScore: number } {
267
+ let score = 0;
268
+ const maxScore = 100;
269
+
270
+ // Check title-specific phrases (highest weight for titles)
271
+ ADULT_CONTENT_CONFIG.titlePhrases.forEach((phrase) => {
272
+ const regex = new RegExp(escapeRegex(phrase), flags);
273
+ if (regex.test(title)) {
274
+ score += 25;
275
+ }
276
+ });
277
+
278
+ // Check strong indicators
279
+ ADULT_CONTENT_CONFIG.strongIndicators.forEach((indicator) => {
280
+ const regex = new RegExp(`\\b${escapeRegex(indicator)}\\b`, flags);
281
+ if (regex.test(title)) {
282
+ score += 20;
283
+ }
284
+ });
285
+
286
+ // Check general keywords
287
+ ADULT_CONTENT_CONFIG.keywords.forEach((keyword) => {
288
+ const regex = new RegExp(`\\b${escapeRegex(keyword)}\\b`, flags);
289
+ if (regex.test(title)) {
290
+ score += 8;
291
+ }
292
+ });
293
+
294
+ return { score: Math.min(score, maxScore), maxScore };
295
+ }
296
+
297
+ /**
298
+ * Escapes special regex characters in a string
299
+ *
300
+ * @param {string} string - String to escape
301
+ * @returns {string} Escaped string
302
+ */
303
+ function escapeRegex(string: string): string {
304
+ return string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
305
+ }
306
+
307
+ /**
308
+ * Applies sigmoid function to normalize score to more realistic probability
309
+ *
310
+ * @param {number} x - Input value
311
+ * @returns {number} Sigmoid output between 0 and 1
312
+ */
313
+ function applySigmoid(x: number): number {
314
+ // Adjust sigmoid steepness for more realistic distribution
315
+ const steepness = 6;
316
+ const midpoint = 0.5;
317
+ return 1 / (1 + Math.exp(-steepness * (x - midpoint)));
318
+ }
@@ -0,0 +1,367 @@
1
+ /**
2
+ * @fileoverview High-level orchestrator for extracting content from any URL or binary buffer.
3
+ * Supports YouTube transcripts, PDFs, DOCX, and web articles.
4
+ */
5
+ import { extractContentAndCite } from "../html-to-content/html-to-content";
6
+ import { getURLYoutubeVideo, convertYoutubeToText } from "./youtube-helpers";
7
+ import { convertDOCXToHTML, isBufferDOCX } from "./docx-to-content";
8
+ import { scrapeURL } from "./url-to-html";
9
+ import grab from "../utils/grab";
10
+
11
+ /**
12
+ * Dynamic PDF converter to avoid bundling pdfjs at build time
13
+ */
14
+ async function convertPDFToHTML(url: string, options: any) {
15
+ const { convertPDFToHTML: pdfConverter } = await import("extract-pdf");
16
+ return await pdfConverter(url, options);
17
+ }
18
+
19
+ async function isUrlPDF(url: string) {
20
+ try {
21
+ const buffer = await grab(url, { responseType: "arraybuffer", timeout: 5 });
22
+ if (!buffer || buffer.byteLength < 5) return false;
23
+ const chunk = new Uint8Array(buffer);
24
+ return chunk[0] === 0x25 && chunk[1] === 0x50 && chunk[2] === 0x44 && chunk[3] === 0x46 && chunk[4] === 0x2d;
25
+ } catch {
26
+ return false;
27
+ }
28
+ }
29
+
30
+ export interface ExtractContentOptions {
31
+ images?: boolean;
32
+ links?: boolean;
33
+ formatting?: boolean;
34
+ absoluteURLs?: boolean;
35
+ timeout?: number;
36
+ proxy?: string | null;
37
+ citeFormatMonthFull?: boolean;
38
+ citeFormatAuthorFull?: boolean;
39
+ url?: string;
40
+ useThirdPartyBackup?: boolean;
41
+ /** Preferred transcript languages when extracting YouTube videos. */
42
+ languages?: string[];
43
+ }
44
+
45
+ export interface ExtractedArticle {
46
+ cite?: string;
47
+ html?: string;
48
+ url?: string;
49
+ author?: string;
50
+ author_cite?: string;
51
+ author_short?: string;
52
+ author_type?: number | string;
53
+ date?: string;
54
+ title?: string;
55
+ source?: string;
56
+ word_count?: number;
57
+ format?: string;
58
+ error?: string | number;
59
+ }
60
+
61
+ type UrlLikeDocument = {
62
+ location?: { href?: string };
63
+ querySelectorAll?: (
64
+ selector: string,
65
+ ) => { length: number } | ArrayLike<unknown>;
66
+ };
67
+
68
+ /**
69
+ * @typedef {Object} Article
70
+ * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
71
+ * @property {string} html - The Basic HTML content of the article
72
+ * @property {string} url - The URL of the article
73
+ * @property {string} author - The full name of the author of the article
74
+ * @property {string} author_cite - Author name in Last, First Initial format
75
+ * @property {string} author_short - Author name in Last format
76
+ * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
77
+ * @property {string} date - The publication date of the article
78
+ * @property {string} title - The title of the article
79
+ * @property {string} source - The source or publisher of the article
80
+ * @property {number} word_count - The word count of the full text (without HTML tags)
81
+ * @category Extract
82
+ */
83
+
84
+ /**
85
+ * ### 🚜 Tractor the Text Extractor
86
+ * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
87
+ *
88
+ * 1. Main Content Detection: Extract the main content from a URL by combining
89
+ * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
90
+ * custom adapters for major sites for article, author, date HTML classes.
91
+ * 2. Basic HTML Standardization: Transform complex HTML into a simplified
92
+ * reading-mode format of basic HTML, making it ideal for research note archival
93
+ * and focused reading, with headings, images and links.
94
+ * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
95
+ * retrieve the complete video transcript including both manual captions and
96
+ * auto-generated subtitles, maintaining proper timestamp synchronization and
97
+ * speaker identification where available.
98
+ * 4. PDF to HTML: Process PDF documents by extracting
99
+ * formatted text while intelligently handling line breaks, page headers,
100
+ * footnotes. The system analyzes text height statistics to automatically
101
+ * infer heading levels, creating a properly structured document hierarchy
102
+ * based on standard deviation from mean text size.
103
+ * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
104
+ * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
105
+ * them to HTML while preserving formatting, styles, and document structure.
106
+ * 6. Citation Information Extraction: Identify and extract citation metadata
107
+ * including author names, publication dates, sources, and titles using HTML
108
+ * meta tags and common class name patterns. The system validates author names
109
+ * against a comprehensive database of 90,000 first and last names,
110
+ * distinguishing between personal and organizational authors to properly
111
+ * format citations.
112
+ * 7. Author Name Formatting: Process author names by checking against
113
+ * known name databases, handling affixes and titles correctly, and determining
114
+ * whether to reverse the name order based on whether it's a personal or
115
+ * organizational author, ensuring proper citation formatting.
116
+ * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
117
+ * @param {Object} [options]
118
+ * @param {boolean} options.images default=true - include images
119
+ * @param {boolean} options.links default=true - include links
120
+ * @param {boolean} options.formatting default=true - preserve formatting
121
+ * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
122
+ * @param {number} options.timeout default=5 - http request timeout
123
+ * @returns {{
124
+ * title: string,
125
+ * author_cite: string,
126
+ * cite: string,
127
+ * author: string,
128
+ * date: string,
129
+ * source: string,
130
+ * html: string,
131
+ * word_count: number
132
+ * }}
133
+ * cite - Cite in APA Format with Author name in Last, First Initial format
134
+ * url - The URL of the article
135
+ * html - The HTML content of the article
136
+ * author - The author of the article
137
+ * author_cite - Author name in Last, First Middle format
138
+ * author_short - Author name in Last format
139
+ * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
140
+ * date - The publication date of the article
141
+ * title - The title of the article
142
+ * source - The source or origin of the article
143
+ * word_count - The word count of the full text (without HTML tags)
144
+ * @category Extract
145
+ * @author [vtempest (2025)](https://github.com/vtempest)
146
+ * @example
147
+ * // Extract from URL
148
+ * const result1 = await extractContent('https://example.com/article');
149
+ *
150
+ * // Extract from DOCX binary buffer
151
+ * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
152
+ * const result2 = await extractContent(docxBuffer);
153
+ *
154
+ * // Extract from DOM object
155
+ * const result3 = await extractContent(document);
156
+ */
157
+ export async function extractContent(
158
+ urlOrDoc:
159
+ | string
160
+ | Document
161
+ | UrlLikeDocument
162
+ | ArrayBuffer
163
+ | Buffer
164
+ | Uint8Array,
165
+ options: ExtractContentOptions = {},
166
+ ): Promise<ExtractedArticle> {
167
+ var {
168
+ images = true,
169
+ links = true,
170
+ formatting = true,
171
+ absoluteURLs = true,
172
+ timeout = 5,
173
+ proxy = null,
174
+ citeFormatMonthFull = false,
175
+ citeFormatAuthorFull = true,
176
+ } = options;
177
+ let response: ExtractedArticle = {};
178
+
179
+ let url, isPdf, isDocxBuffer;
180
+
181
+ // Check if input is a binary buffer (DOCX)
182
+ if (
183
+ urlOrDoc instanceof ArrayBuffer ||
184
+ urlOrDoc instanceof Uint8Array ||
185
+ (typeof Buffer !== "undefined" && Buffer.isBuffer(urlOrDoc))
186
+ ) {
187
+ isDocxBuffer = isBufferDOCX(urlOrDoc);
188
+
189
+ if (isDocxBuffer) {
190
+ // Handle DOCX binary buffer
191
+ response.html = await convertDOCXToHTML(urlOrDoc, options);
192
+ url = "buffer://docx"; // Placeholder URL for buffer input
193
+ } else {
194
+ return { error: "Binary buffer is not a valid DOCX file" };
195
+ }
196
+ } else if (
197
+ typeof urlOrDoc === "string" &&
198
+ /<\/[^>]+>/.test(urlOrDoc.trim())
199
+ ) {
200
+ console.log("[extractContent] input is raw HTML string");
201
+ // If urlOrDoc is an HTML string, treat as HTML content
202
+ options.url = options.url || "";
203
+
204
+ response = extractContentAndCite(urlOrDoc, options);
205
+ console.log("[extractContent] extractContentAndCite (raw html) result", {
206
+ hasHtml: !!response?.html,
207
+ htmlLength: response?.html?.length || 0,
208
+ title: response?.title,
209
+ error: response?.error,
210
+ });
211
+
212
+ return response;
213
+ // if URL
214
+ } else if (typeof urlOrDoc === "string" && urlOrDoc.startsWith("http")) {
215
+ url = urlOrDoc;
216
+ console.log("[extractContent] input is URL", { url });
217
+
218
+ // check if google doc, then extract html or pdf file
219
+ let googleDocId = url.match(/google\.com\/(file|document)\/d\/([\w-]+)/);
220
+ if (googleDocId) {
221
+ url =
222
+ googleDocId[1] === "file"
223
+ ? `https://drive.google.com/uc?export=download&id=${googleDocId[2]}`
224
+ : `https://docs.google.com/document/d/${googleDocId[2]}/export?format=html`;
225
+ console.log("[extractContent] rewrote google doc url", { url });
226
+ }
227
+
228
+ isPdf = url.endsWith(".pdf") || (await isUrlPDF(url));
229
+ let youtubeID = getURLYoutubeVideo(url);
230
+ console.log("[extractContent] branch detection", {
231
+ isPdf,
232
+ youtubeID,
233
+ isDocx: url.endsWith(".docx"),
234
+ });
235
+
236
+ if (isPdf) {
237
+ // pdf checker - use dynamic import to prevent build-time evaluation
238
+ response = await convertPDFToHTML(url, options as any);
239
+ console.log("[extractContent] pdf branch result", {
240
+ hasHtml: !!response?.html,
241
+ error: response?.error,
242
+ });
243
+ } else if (url.endsWith(".docx")) {
244
+ response.html = await convertDOCXToHTML(url);
245
+ console.log("[extractContent] docx branch result", {
246
+ hasHtml: !!response?.html,
247
+ });
248
+
249
+ // check youtube
250
+ } else if (youtubeID) {
251
+ response = await convertYoutubeToText(url, options);
252
+ console.log("[extractContent] youtube branch result", {
253
+ hasHtml: !!response?.html,
254
+ error: response?.error,
255
+ });
256
+ } else {
257
+ console.log("[extractContent] scraping URL", { url, proxy });
258
+
259
+ try {
260
+ const html = await scrapeURL(url, {
261
+ proxy,
262
+ });
263
+ console.log("[extractContent] scrapeURL returned", {
264
+ url,
265
+ hasHtml: !!html,
266
+ htmlLength: typeof html === "string" ? html.length : 0,
267
+ sample: typeof html === "string" ? html.slice(0, 200) : null,
268
+ });
269
+
270
+ // Check if scrapeURL returned an error object instead of HTML string
271
+ if (typeof html !== "string" || !html) {
272
+ console.error("[extractContent] scrapeURL failed or returned non-string", {
273
+ url,
274
+ typeofHtml: typeof html,
275
+ isEmpty: !html,
276
+ });
277
+ return {
278
+ error: "Failed to fetch HTML content",
279
+ };
280
+ }
281
+
282
+ options.url = url;
283
+ response = extractContentAndCite(html, options);
284
+ console.log("[extractContent] extractContentAndCite result", {
285
+ url,
286
+ hasHtml: !!response?.html,
287
+ htmlLength: response?.html?.length || 0,
288
+ title: response?.title,
289
+ error: response?.error,
290
+ });
291
+ } catch (scrapeError) {
292
+ const err = scrapeError as Error;
293
+ console.error("[extractContent] scrapeURL threw error", {
294
+ url,
295
+ message: err?.message,
296
+ });
297
+ return {
298
+ error: `Failed to scrape URL: ${err?.message || String(scrapeError)}`,
299
+ };
300
+ }
301
+ }
302
+ } else if (typeof urlOrDoc == "object" && urlOrDoc.location) {
303
+ //if passing in dom object document from front end
304
+
305
+ url = urlOrDoc.location.href;
306
+
307
+ //pdf checker for embedded docs
308
+ if (urlOrDoc?.querySelectorAll)
309
+ isPdf = urlOrDoc?.querySelectorAll(
310
+ 'embed[type="application/pdf"]',
311
+ )?.length;
312
+ var youtubeID = getURLYoutubeVideo(url);
313
+
314
+ if (isPdf) {
315
+ response = await convertPDFToHTML(url, {});
316
+ } else if (youtubeID) {
317
+ // from front end
318
+
319
+ //if on same domain page in chrome-extension
320
+ options.useThirdPartyBackup = false;
321
+ response = await convertYoutubeToText(url, options);
322
+ } //pass doc to extract
323
+ else response = extractContentAndCite(urlOrDoc as Document, options);
324
+ } else {
325
+ // Handle other object types or invalid input
326
+ return {
327
+ error:
328
+ "Invalid input type. Expected URL string, DOM object, or DOCX binary buffer.",
329
+ };
330
+ }
331
+
332
+ //if no text
333
+ if (response.error || !response.html) return { error: response.error };
334
+
335
+ //word count of full text original, no html
336
+ response.word_count = response.html
337
+ ?.replace(/<[^>]*>/g, " ")
338
+ .split(" ").length;
339
+
340
+ //make APA cite
341
+
342
+ var { author, author_cite, author_short, date, title, source } = response;
343
+
344
+ var apa_cite_date =
345
+ new Date(date).getFullYear() > 1971
346
+ ? " (" +
347
+ new Date(date).getFullYear() +
348
+ ", " +
349
+ new Date(date).toLocaleDateString("en-US", {
350
+ month: citeFormatMonthFull ? "long" : "short",
351
+ day: "numeric",
352
+ }) +
353
+ ")"
354
+ : ""; //"(N.D.)";
355
+
356
+ var cite = `${author_cite || source || " "}${apa_cite_date}. <b>${
357
+ title || ""
358
+ }</b>. <i>${source || ""}</i>. <a href="${url}" target="_blank">${url}</a>`;
359
+
360
+ //shorten long urls by removing ?params=get used as state tracking
361
+ if (url && url.includes("?") && url.length > 150)
362
+ response.url = url.split("?")[0];
363
+
364
+ //put url on top
365
+ response = Object.assign({ url, cite }, response);
366
+ return response;
367
+ }