extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,48 @@
1
+ /**
2
+ * @fileoverview Utility for splitting text into semantic chunks for RAG and NLP tasks.
3
+ * Uses complex regex patterns to identify structural elements like lists, tables, and code.
4
+ *
5
+ * ### Split Text by Semantic Characters
6
+ * <img width="350px" src="https://i.imgur.com/RpXf5as.png" />
7
+ *
8
+ *
9
+ * Splits document text into semantic chunks based on various textual and structural
10
+ * elements like HTML, markdown, and paragraphs.
11
+ *
12
+ * This function performs a comprehensive tokenization of the input text, considering a wide range
13
+ * of semantic elements and structural patterns commonly found in documents.It uses regular
14
+ * expressions to identify and separate the following elements:
15
+ *
16
+ * 1. Headings(Setext - style, Markdown, and HTML - style)
17
+ * 2. Citations(e.g., [1])
18
+ * 3. List items(bulleted, numbered, lettered, or task lists, including nested up to three levels)
19
+ * 4. Block quotes(including nested quotes and citations, up to three levels)
20
+ * 5. Code blocks(fenced, indented, or HTML pre / code tags)
21
+ * 6. Tables(Markdown, grid tables, and HTML tables)
22
+ * 7. Horizontal rules(Markdown and HTML hr tag)
23
+ * 8. Standalone lines or phrases(including single - line blocks and HTML elements)
24
+ * 9. Sentences or phrases ending with punctuation(including ellipsis and Unicode punctuation)
25
+ * 10. Quoted text, parenthetical phrases, or bracketed content
26
+ * 11. Paragraphs
27
+ * 12. HTML - like tags and their content(including self - closing tags and attributes)
28
+ * 13. LaTeX - style math expressions(inline and block)
29
+ * 14. Any remaining content(fallback)
30
+ *
31
+ * The function applies various length constraints to each type of element to ensure reasonable
32
+ * chunk sizes.It also handles nested structures and special cases like code blocks and math
33
+ * expressions.
34
+ *
35
+ * [Sentence RAG Benchmarks](https://superlinked.com/vectorhub/articles/evaluation-rag-retrieval-chunking-methods)
36
+ *
37
+ * @author[Jina AI(2024)](https://gist.github.com/hanxiao/3f60354cf6dc5ac698bc9154163b4e6a)
38
+ * @param { string } text - The input text to be split into semantic chunks.
39
+ * @param { Object }[options = {}] - Optional configuration options(currently unused).
40
+ * @returns { Array.<string> } An array of text chunks, each representing a semantic unit of the document.
41
+ * @category Topics
42
+ * @example
43
+ * const text = "# Heading\n\nThis is a paragraph.\n\n- List item 1\n- List item 2\n\n";
44
+ * const chunks = splitTextSemanticChars(text);
45
+ * console.log(chunks);
46
+ * // Output: ['# Heading', 'This is a paragraph.', '- List item 1', '- List item 2']
47
+ */
48
+ export declare function splitTextSemanticChars(text: any, options?: {}): unknown[];
@@ -0,0 +1,35 @@
1
+ /**
2
+ * @fileoverview
3
+ * Splits text into sentences, handling 220+ common abbreviations,
4
+ * and inferring acronyms, numbers, URLs, times, names, etc.
5
+ *
6
+ * @param inputText - The text to be split into sentences.
7
+ * @param options - Configuration options for sentence splitting.
8
+ * @returns An array of sentences.
9
+ * @author [vtempest (2025)](https://github.com/vtempest)
10
+ * @license MIT
11
+ * @example
12
+ * ```ts
13
+ * const text = "Dr. Smith went to the U.S. He met Mr. Jones.";
14
+ * const sentences = splitTextToSentences(text);
15
+ * ["Dr. Smith went to the U.S.", "He met Mr. Jones."]
16
+ * ```
17
+ */
18
+ export declare function splitTextToSentences(inputText: string, options?: SplitSentencesOptions): string[];
19
+ export type SplitSentencesOptions = {
20
+ /**
21
+ * Split on HTML tags like P, DIV, UL, OL.
22
+ * @default true
23
+ */
24
+ splitOnHtmlTags?: boolean;
25
+ /**
26
+ * Minimum size for a sentence.
27
+ * @default 20
28
+ */
29
+ minSize?: number;
30
+ /**
31
+ * Maximum size for a sentence.
32
+ * @default 500
33
+ */
34
+ maxSize?: number;
35
+ };
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Topic token tuple shape used by downstream ranking logic:
3
+ * [term, termCategory, uniqueness, metadata]
4
+ */
5
+ export type TopicToken = [string, number, number, string];
6
+ /**
7
+ * Trie structure for phrase completion lookup keyed by first two letters, then full token.
8
+ */
9
+ type PhraseEntry = [string | null, number, number];
10
+ export type PhrasesModel = Record<string, Record<string, PhraseEntry[]>>;
11
+ export interface ConvertTextToTokensOptions {
12
+ phrasesModel: PhrasesModel;
13
+ typosModel?: Record<string, string>;
14
+ checkTypos?: 0 | 1;
15
+ ignoreStopWords?: 0 | 1;
16
+ checkRootWords?: 0 | 1;
17
+ }
18
+ /**
19
+ * @typedef {Object} Token
20
+ * @property {number} termCategory - The category of the term
21
+ * @property {number} uniqueness - The uniqueness score of the term
22
+ * @property {string} term - The actual term or phrase
23
+ */
24
+ /**
25
+ * ### Convert Text Query to Topic Phrase Tokens
26
+ * <img width="350px" src="https://i.imgur.com/NDrmSRQ.png" />
27
+ *
28
+ * Returns a list of phrases that are found in Wiki Titles/ dictionary phrases World Model
29
+ * that match the input phrase, or just the single word if found. Search results will be
30
+ * more accurate if we infer likely phrases and search for those words occuring together and
31
+ * not just split into words and find frequency. Examples are "white house" or "state of the art"
32
+ * which should be searched as a phrase but would return different context if split into words.
33
+ * As Led Zeppelin famously put it: \u266b "'Cause you know sometimes words have two meanings."
34
+ *
35
+ * @param {string} phrase
36
+ * @param {Object} [options]
37
+ * @param {Object} options.phrasesModel - remote model
38
+ * @param {Object} options.typosModel - remote model
39
+ * @param {number} options.checkTypos - check for typos
40
+ * @param {number} options.ignoreStopWords - ignore 300+ overused words
41
+ * @param {number} options.checkRootWords - check for word's root stem
42
+ * @returns {Array<{termCategory: number, uniqueness: number, term: string}>}
43
+ * @example
44
+ * const result = convertTextToTokens("The president of the united states is in the white house", { phrasesModel, typosModel });
45
+ * console.log(result);
46
+ *
47
+ * @author [vtempest (2025)](https://github.com/vtempest)
48
+ * @category Topics
49
+ */
50
+ export declare function convertTextToTokens(phrase: string, options?: Partial<ConvertTextToTokensOptions>): TopicToken[];
51
+ export {};
@@ -0,0 +1,12 @@
1
+ /**
2
+ * @fileoverview Utility for identifying common stop words (e.g., "the", "and", "is").
3
+ * Based on the SpaCy English stop word list.
4
+ */
5
+ /**
6
+ * Checks word is in [320 commonly ignored "stop words
7
+ * "](https://raw.githubusercontent.com/igorbrigadir/stopwords/master/en/spacy.txt)
8
+ * in queries, using efficient JS Set method
9
+ * @param {string} word
10
+ * @returns {Boolean}
11
+ */
12
+ export declare function isWordCommonIgnored(word: any): boolean;
@@ -0,0 +1,16 @@
1
+ /**
2
+ * @fileoverview Implementation of the Porter Stemmer algorithm for word normalization.
3
+ * Used to reduce words to their root form (e.g., "running" to "run").
4
+ */
5
+ /**
6
+ * Stems a word using the [Porter
7
+ * Stemmer](https://snowballstem.org/algorithms/porter/stemmer.html)
8
+ * for removing inflectional endings like "ing", "ist", "ize".
9
+ *
10
+ * @author [Porter, M. (1980)](https://tartarus.org/martin/PorterStemmer/)
11
+ * @param word - The word to be stemmed
12
+ * @returns The stemmed word
13
+ * @example const rootWord = stemWordToRoot("running"); // returns "run"
14
+ * @category Topics
15
+ */
16
+ export declare function stemWordToRoot(word: string): string;
@@ -0,0 +1,22 @@
1
+ /**
2
+ * Converts a DOCX document to HTML
3
+ *
4
+ * @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input - DOCX input to convert
5
+ * @param {DocxOptions} [options] - Conversion options
6
+ * @returns {Promise<string>} The converted HTML
7
+ * @throws {Error} If conversion fails
8
+ * @category Extract
9
+ * @example
10
+ * const html = await convertDOCXToHTML('https://example.com/doc.docx');
11
+ * const html = await convertDOCXToHTML(fileInput.files[0]);
12
+ */
13
+ export declare function convertDOCXToHTML(input: any, options?: {}): Promise<string>;
14
+ /**
15
+ * Detects if a binary buffer is a DOCX file by checking the file signature
16
+ * DOCX files are ZIP archives with specific internal structure
17
+ *
18
+ * @param {ArrayBuffer|Buffer|Uint8Array} buffer - Binary buffer to check
19
+ * @returns {boolean} True if buffer appears to be a DOCX file
20
+ * @category Extract
21
+ */
22
+ export declare function isBufferDOCX(buffer: any): boolean;
@@ -0,0 +1,26 @@
1
+ /**
2
+ * @module research/extractor/url-to-content/is-url-porn
3
+ * @description Research library module.
4
+ */
5
+ export interface IsURLPornOptions {
6
+ url?: string;
7
+ title?: string;
8
+ threshold?: number;
9
+ }
10
+ /**
11
+ * Determines if content is likely adult/porn based on configurable threshold
12
+ *
13
+ * @param {Object} options - Configuration object
14
+ * @param {string} [options.url] - URL to analyze (optional)
15
+ * @param {string} [options.title] - Page title to analyze (optional)
16
+ * @param {number} [options.threshold] - Probability threshold (0.5 default)
17
+ * @returns {boolean} True if likelihood exceeds threshold, false otherwise
18
+ * @author [vtempest (2025)](https://github.com/vtempest)
19
+ * @example
20
+ * isURLPorn({
21
+ * title: "Hot deals on sexy cars",
22
+ * threshold: 0.8
23
+ * });
24
+ * console.log(isPorn2); // false (low confidence)
25
+ */
26
+ export declare function isURLPorn(options?: IsURLPornOptions): boolean;
@@ -0,0 +1,127 @@
1
+ export interface ExtractContentOptions {
2
+ images?: boolean;
3
+ links?: boolean;
4
+ formatting?: boolean;
5
+ absoluteURLs?: boolean;
6
+ timeout?: number;
7
+ proxy?: string | null;
8
+ citeFormatMonthFull?: boolean;
9
+ citeFormatAuthorFull?: boolean;
10
+ url?: string;
11
+ useThirdPartyBackup?: boolean;
12
+ /** Preferred transcript languages when extracting YouTube videos. */
13
+ languages?: string[];
14
+ }
15
+ export interface ExtractedArticle {
16
+ cite?: string;
17
+ html?: string;
18
+ url?: string;
19
+ author?: string;
20
+ author_cite?: string;
21
+ author_short?: string;
22
+ author_type?: number | string;
23
+ date?: string;
24
+ title?: string;
25
+ source?: string;
26
+ word_count?: number;
27
+ format?: string;
28
+ error?: string | number;
29
+ }
30
+ type UrlLikeDocument = {
31
+ location?: {
32
+ href?: string;
33
+ };
34
+ querySelectorAll?: (selector: string) => {
35
+ length: number;
36
+ } | ArrayLike<unknown>;
37
+ };
38
+ /**
39
+ * @typedef {Object} Article
40
+ * @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
41
+ * @property {string} html - The Basic HTML content of the article
42
+ * @property {string} url - The URL of the article
43
+ * @property {string} author - The full name of the author of the article
44
+ * @property {string} author_cite - Author name in Last, First Initial format
45
+ * @property {string} author_short - Author name in Last format
46
+ * @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
47
+ * @property {string} date - The publication date of the article
48
+ * @property {string} title - The title of the article
49
+ * @property {string} source - The source or publisher of the article
50
+ * @property {number} word_count - The word count of the full text (without HTML tags)
51
+ * @category Extract
52
+ */
53
+ /**
54
+ * ### 🚜 Tractor the Text Extractor
55
+ * <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
56
+ *
57
+ * 1. Main Content Detection: Extract the main content from a URL by combining
58
+ * Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
59
+ * custom adapters for major sites for article, author, date HTML classes.
60
+ * 2. Basic HTML Standardization: Transform complex HTML into a simplified
61
+ * reading-mode format of basic HTML, making it ideal for research note archival
62
+ * and focused reading, with headings, images and links.
63
+ * 3. YouTube Transcript Processing: When a YouTube video URL is detected,
64
+ * retrieve the complete video transcript including both manual captions and
65
+ * auto-generated subtitles, maintaining proper timestamp synchronization and
66
+ * speaker identification where available.
67
+ * 4. PDF to HTML: Process PDF documents by extracting
68
+ * formatted text while intelligently handling line breaks, page headers,
69
+ * footnotes. The system analyzes text height statistics to automatically
70
+ * infer heading levels, creating a properly structured document hierarchy
71
+ * based on standard deviation from mean text size.
72
+ * 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
73
+ * (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
74
+ * them to HTML while preserving formatting, styles, and document structure.
75
+ * 6. Citation Information Extraction: Identify and extract citation metadata
76
+ * including author names, publication dates, sources, and titles using HTML
77
+ * meta tags and common class name patterns. The system validates author names
78
+ * against a comprehensive database of 90,000 first and last names,
79
+ * distinguishing between personal and organizational authors to properly
80
+ * format citations.
81
+ * 7. Author Name Formatting: Process author names by checking against
82
+ * known name databases, handling affixes and titles correctly, and determining
83
+ * whether to reverse the name order based on whether it's a personal or
84
+ * organizational author, ensuring proper citation formatting.
85
+ * @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
86
+ * @param {Object} [options]
87
+ * @param {boolean} options.images default=true - include images
88
+ * @param {boolean} options.links default=true - include links
89
+ * @param {boolean} options.formatting default=true - preserve formatting
90
+ * @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
91
+ * @param {number} options.timeout default=5 - http request timeout
92
+ * @returns {{
93
+ * title: string,
94
+ * author_cite: string,
95
+ * cite: string,
96
+ * author: string,
97
+ * date: string,
98
+ * source: string,
99
+ * html: string,
100
+ * word_count: number
101
+ * }}
102
+ * cite - Cite in APA Format with Author name in Last, First Initial format
103
+ * url - The URL of the article
104
+ * html - The HTML content of the article
105
+ * author - The author of the article
106
+ * author_cite - Author name in Last, First Middle format
107
+ * author_short - Author name in Last format
108
+ * author_type - Author type ["single", "two-author", "more-than-two", "organization"]
109
+ * date - The publication date of the article
110
+ * title - The title of the article
111
+ * source - The source or origin of the article
112
+ * word_count - The word count of the full text (without HTML tags)
113
+ * @category Extract
114
+ * @author [vtempest (2025)](https://github.com/vtempest)
115
+ * @example
116
+ * // Extract from URL
117
+ * const result1 = await extractContent('https://example.com/article');
118
+ *
119
+ * // Extract from DOCX binary buffer
120
+ * const docxBuffer = new Uint8Array([...]); // DOCX file bytes
121
+ * const result2 = await extractContent(docxBuffer);
122
+ *
123
+ * // Extract from DOM object
124
+ * const result3 = await extractContent(document);
125
+ */
126
+ export declare function extractContent(urlOrDoc: string | Document | UrlLikeDocument | ArrayBuffer | Buffer | Uint8Array, options?: ExtractContentOptions): Promise<ExtractedArticle>;
127
+ export {};
@@ -0,0 +1,60 @@
1
+ /**
2
+ * ### Tardigrade the Web Crawler
3
+ * <img src="https://i.imgur.com/iuzpcvD.png" width="350px" />
4
+ *
5
+ * 1. **Use Fetch API, check for bot detection.** Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
6
+ * Scraping internet pages is a [free speech right
7
+ * ](https://blog.apify.com/is-web-scraping-legal/).
8
+ * 2. Features: timeout, redirects, default UA, referer as google, and bot
9
+ * detection checking. <br />
10
+ * 3. If fetch method does not get needed HTML, use Docker proxy as backup.
11
+ *
12
+ * 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
13
+ * container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
14
+ * secondary in-page API requests after the initial page request, including user login and cookie storage.
15
+ * 5. Bypass Cloudflare bot check: A webpage proxy that request through Chromium (puppeteer) - can be used
16
+ * to bypass Cloudflare anti bot using cookie id javascript method.
17
+ * 6. Send your request to the server with the port 3000 and add your URL to the "url"
18
+ * query string like this: `http://localhost:3000/?url=https://example.org`
19
+ *
20
+ * 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
21
+ * and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
22
+ * [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
23
+ * [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
24
+ * [Proxy-Cheap](https://app.proxy-cheap.com/order)
25
+ * [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
26
+ *
27
+ * @param {string} url - any domain's URL
28
+ * @param {Object} [options]
29
+ * @param {number} options.timeout default=5 - abort request if not retrived, in seconds
30
+ * @param {number} options.maxRedirects default=3 - max redirects to follow
31
+ * @param {number} options.checkBotDetection default=true - check for bot detection messages
32
+ * @param {number} options.changeReferer default=true - set referer as google
33
+ * @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
34
+ * @param {string} options.proxy default=false - use proxy url
35
+ * @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
36
+ * @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
37
+ * @category Extract
38
+ * @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
39
+ * @author [vtempest (2025)](https://github.com/vtempest)
40
+ */
41
+ export declare function scrapeURL(url: any, options?: {}): Promise<any>;
42
+ /**
43
+ * As backup, scrape with JINA to get html
44
+ * @param {string} url
45
+ * @returns {Promise<string>}
46
+ */
47
+ export declare function scrapeJINA(url: any): Promise<any>;
48
+ /**
49
+ * Fetches and parses the robots.txt file for a given URL.
50
+ * @param {string} url - The base URL to fetch the robots.txt from.
51
+ * @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
52
+ */
53
+ export declare function fetchScrapingRules(url: any): Promise<{
54
+ directives: {};
55
+ crawlDelay: {};
56
+ sitemaps: any[];
57
+ preferredHost: any;
58
+ } | {
59
+ error: string;
60
+ }>;
@@ -0,0 +1,23 @@
1
+ /**
2
+ * @fileoverview Adapter helpers bridging the `extract-youtube` transcript API to
3
+ * the `getURLYoutubeVideo` / `convertYoutubeToText` helpers expected by the URL
4
+ * content extractor. Keeps the extractor decoupled from the transcript library's
5
+ * concrete API surface.
6
+ */
7
+ /**
8
+ * Extracts the 11-character YouTube video id from a URL, if present.
9
+ *
10
+ * @param {string} url - A URL that may point to a YouTube video.
11
+ * @returns {string | null} The video id, or null when the URL is not a YouTube link.
12
+ */
13
+ export declare function getURLYoutubeVideo(url: string): string | null;
14
+ /**
15
+ * Fetches a YouTube video transcript and returns it as a simple HTML document.
16
+ *
17
+ * @param {string} url - The YouTube video URL.
18
+ * @param {{ languages?: string[] }} [options] - Optional transcript languages.
19
+ * @returns {Promise<Record<string, any>>} An extraction response with `html`, or `{ error }`.
20
+ */
21
+ export declare function convertYoutubeToText(url: string, options?: {
22
+ languages?: string[];
23
+ }): Promise<Record<string, any>>;
@@ -0,0 +1,70 @@
1
+ /**
2
+ * Fetch youtube.com video's webpage HTML for embedded transcript.
3
+ * If blocked, use scraper of alternative sites providing transcripts.
4
+ * @param {string} videoUrl
5
+ * @param {Object} [options]
6
+ * @param {boolean} options.addTimestamps default=true -
7
+ * true to return timestamps, default true
8
+ * @param {boolean} options.timeout default=5 - http request timeout
9
+ * @return {{content: string, timestamps: string, word_count: number}}
10
+ * where content is the full text of the transcript,
11
+ * timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
12
+ * and word_count is the number of words in the transcript.
13
+ * @category Extract
14
+ * @author [vtempest (2025)](https://github.com/vtempest)
15
+ */
16
+ export declare function convertYoutubeToText(videoUrl: any, options?: {}): Promise<{
17
+ html: any;
18
+ word_count: any;
19
+ source: string;
20
+ date: any;
21
+ title: any;
22
+ author_cite: any;
23
+ length: number;
24
+ }>;
25
+ /**
26
+ * Test if URL is to youtube video and return video id if true
27
+ * @param {string} url - youtube video URL
28
+ * @returns {string|boolean} video ID or false
29
+ * @private
30
+ */
31
+ export declare function getURLYoutubeVideo(url: any): any;
32
+ /**
33
+ * Fetch-based scraper of youtubetotranscript.com
34
+ * @returns {Object} content, timestamps - where content is the full text of
35
+ * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
36
+ */
37
+ export declare function fetchViaYoutubeToTranscriptCom(videoId: any, options?: {}): Promise<{
38
+ error: number;
39
+ content?: undefined;
40
+ title?: undefined;
41
+ author_cite?: undefined;
42
+ timestamps?: undefined;
43
+ } | {
44
+ content: string;
45
+ title: any;
46
+ author_cite: any;
47
+ timestamps: any[];
48
+ error?: undefined;
49
+ }>;
50
+ /** ========== NOT WORKING ========== */
51
+ /**
52
+ * Get YouTube transcript of most YouTube videos,
53
+ * except if disabled by uploader
54
+ * fetch-based scraper of youtubetranscript.com
55
+ *
56
+ * @param {string} videoUrl
57
+ * @returns {Object} where content is the full text of
58
+ * the transcript, and timestamps is an array of [characterIndex, timeSeconds]
59
+ * @private
60
+ */
61
+ export declare function fetchViaYoutubeTranscript(videoId: any, options?: {}): Promise<{
62
+ error: number;
63
+ content?: undefined;
64
+ timestamps?: undefined;
65
+ } | {
66
+ content: string;
67
+ timestamps: any[];
68
+ error?: undefined;
69
+ }>;
70
+ export declare function extractYouTubeInfo(videoId: any, options?: {}): Promise<{}>;
@@ -0,0 +1,4 @@
1
+ import { Document } from 'chat-agent-toolkit';
2
+ export declare const getDocumentsFromLinks: ({ links }: {
3
+ links: string[];
4
+ }) => Promise<Document<Record<string, any>>[]>;
@@ -0,0 +1,18 @@
1
+ /**
2
+ * Fetch wrapper for grabbing binary content
3
+ * Replacement for grab-url package using standard fetch API
4
+ */
5
+ export interface GrabOptions {
6
+ responseType?: "text" | "arraybuffer";
7
+ /** Timeout in seconds */
8
+ timeout?: number;
9
+ method?: string;
10
+ headers?: Record<string, string>;
11
+ body?: string;
12
+ }
13
+ export default function grab(url: string, options?: GrabOptions & {
14
+ responseType?: "text";
15
+ }): Promise<string>;
16
+ export default function grab(url: string, options: GrabOptions & {
17
+ responseType: "arraybuffer";
18
+ }): Promise<ArrayBuffer>;
package/package.json ADDED
@@ -0,0 +1,109 @@
1
+ {
2
+ "name": "extract-webpage",
3
+ "version": "1.2.5",
4
+ "module": "./dist/extract-webpage.es.js",
5
+ "description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
6
+ "author": "vtempest <grokthiscontact@gmail.com>",
7
+ "license": "rights.institute/PROSPER",
8
+ "repository": {
9
+ "type": "git",
10
+ "url": "https://github.com/OpenSourceAGI/qwksearch-research-agent",
11
+ "directory": "packages/extract-webpage"
12
+ },
13
+ "main": "./dist/extract-webpage.cjs.js",
14
+ "types": "./dist/index.d.ts",
15
+ "exports": {
16
+ ".": {
17
+ "types": "./dist/index.d.ts",
18
+ "import": "./dist/extract-webpage.es.js",
19
+ "require": "./dist/extract-webpage.cjs.js"
20
+ },
21
+ "./search": {
22
+ "types": "./src/search/index.ts",
23
+ "react-server": "./src/search/index.ts",
24
+ "import": "./src/search/index.ts",
25
+ "require": "./src/search/index.ts"
26
+ },
27
+ "./*": {
28
+ "types": "./src/*.ts",
29
+ "react-server": "./src/*",
30
+ "import": "./src/*",
31
+ "require": "./src/*"
32
+ }
33
+ },
34
+ "files": [
35
+ "dist",
36
+ "src"
37
+ ],
38
+ "typesVersions": {
39
+ "*": {
40
+ "*": [
41
+ "src/*"
42
+ ]
43
+ }
44
+ },
45
+ "scripts": {
46
+ "build": "vite build",
47
+ "test": "vitest",
48
+ "ship": "npm run build && npx standard-version --release-as patch; rm CHANGELOG.md; npm publish",
49
+ "test-ui": "vitest --ui --watch",
50
+ "make": "rm -rf dist/*; NODE_OPTIONS=--max-old-space-size=15192 BUN_JSC_forceRAMSize=15192 vite build "
51
+ },
52
+ "peerDependencies": {
53
+ "next": ">=15.0.0"
54
+ },
55
+ "peerDependenciesMeta": {
56
+ "next": {
57
+ "optional": true
58
+ }
59
+ },
60
+ "devDependencies": {
61
+ "@tsconfig/svelte": "^5.0.8",
62
+ "@types/node": "^22.0.0",
63
+ "@vitest/ui": "^4.0.18",
64
+ "axios": "^1.13.6",
65
+ "clsx": "^2.1.1",
66
+ "next": "^16.2.10",
67
+ "react": "^19.2.4",
68
+ "react-dom": "^19.2.4",
69
+ "terser": "^5.46.0",
70
+ "typedoc-plugin-markdown": "^4.10.0",
71
+ "typescript": "^5.9.3",
72
+ "vinext": "1.0.0-beta.0",
73
+ "vite": "^8.1.3",
74
+ "vite-plugin-dts": "^5.0.3",
75
+ "vite-plugin-node-polyfills": "^0.28.0",
76
+ "vitest": "^4.0.18"
77
+ },
78
+ "dependencies": {
79
+ "@huggingface/transformers": "^3.8.1",
80
+ "ai": "^5.0.0",
81
+ "chat-agent-toolkit": "^1.2.3",
82
+ "chrono-node": "^2.9.0",
83
+ "drizzle-orm": "^0.45.1",
84
+ "extract-pdf": "^0.1.1",
85
+ "extract-youtube": "^1.0.3",
86
+ "highlight.js": "^11.11.1",
87
+ "html-entities": "^2.6.0",
88
+ "js-yaml": "^4.1.1",
89
+ "jsdom": "^28.1.0",
90
+ "jszip": "^3.10.1",
91
+ "linkedom": "^0.18.12",
92
+ "marked": "^17.0.4",
93
+ "node-fetch": "^3.3.2",
94
+ "qwksearch-api-client": "^0.0.12",
95
+ "tldts": "^7.0.25",
96
+ "youtube-po-token-generator": "^0.6.0",
97
+ "zod": "^4.3.6"
98
+ },
99
+ "keywords": [
100
+ "nlp",
101
+ "autocomplete",
102
+ "knowledge-graph",
103
+ "keywords",
104
+ "search-algorithm",
105
+ "hacktoberfest",
106
+ "mind-map",
107
+ "ai-search"
108
+ ]
109
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Runtime env-var accessor.
3
+ * Uses `process.env` for Next.js. In Cloudflare Workers with vinext,
4
+ * this would need to be adapted to use cloudflare:workers.
5
+ */
6
+ export function getEnv(key: string): string | undefined {
7
+ return process.env[key];
8
+ }