extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for splitting text into semantic chunks for RAG and NLP tasks.
|
|
3
|
+
* Uses complex regex patterns to identify structural elements like lists, tables, and code.
|
|
4
|
+
*
|
|
5
|
+
* ### Split Text by Semantic Characters
|
|
6
|
+
* <img width="350px" src="https://i.imgur.com/RpXf5as.png" />
|
|
7
|
+
*
|
|
8
|
+
*
|
|
9
|
+
* Splits document text into semantic chunks based on various textual and structural
|
|
10
|
+
* elements like HTML, markdown, and paragraphs.
|
|
11
|
+
*
|
|
12
|
+
* This function performs a comprehensive tokenization of the input text, considering a wide range
|
|
13
|
+
* of semantic elements and structural patterns commonly found in documents.It uses regular
|
|
14
|
+
* expressions to identify and separate the following elements:
|
|
15
|
+
*
|
|
16
|
+
* 1. Headings(Setext - style, Markdown, and HTML - style)
|
|
17
|
+
* 2. Citations(e.g., [1])
|
|
18
|
+
* 3. List items(bulleted, numbered, lettered, or task lists, including nested up to three levels)
|
|
19
|
+
* 4. Block quotes(including nested quotes and citations, up to three levels)
|
|
20
|
+
* 5. Code blocks(fenced, indented, or HTML pre / code tags)
|
|
21
|
+
* 6. Tables(Markdown, grid tables, and HTML tables)
|
|
22
|
+
* 7. Horizontal rules(Markdown and HTML hr tag)
|
|
23
|
+
* 8. Standalone lines or phrases(including single - line blocks and HTML elements)
|
|
24
|
+
* 9. Sentences or phrases ending with punctuation(including ellipsis and Unicode punctuation)
|
|
25
|
+
* 10. Quoted text, parenthetical phrases, or bracketed content
|
|
26
|
+
* 11. Paragraphs
|
|
27
|
+
* 12. HTML - like tags and their content(including self - closing tags and attributes)
|
|
28
|
+
* 13. LaTeX - style math expressions(inline and block)
|
|
29
|
+
* 14. Any remaining content(fallback)
|
|
30
|
+
*
|
|
31
|
+
* The function applies various length constraints to each type of element to ensure reasonable
|
|
32
|
+
* chunk sizes.It also handles nested structures and special cases like code blocks and math
|
|
33
|
+
* expressions.
|
|
34
|
+
*
|
|
35
|
+
* [Sentence RAG Benchmarks](https://superlinked.com/vectorhub/articles/evaluation-rag-retrieval-chunking-methods)
|
|
36
|
+
*
|
|
37
|
+
* @author[Jina AI(2024)](https://gist.github.com/hanxiao/3f60354cf6dc5ac698bc9154163b4e6a)
|
|
38
|
+
* @param { string } text - The input text to be split into semantic chunks.
|
|
39
|
+
* @param { Object }[options = {}] - Optional configuration options(currently unused).
|
|
40
|
+
* @returns { Array.<string> } An array of text chunks, each representing a semantic unit of the document.
|
|
41
|
+
* @category Topics
|
|
42
|
+
* @example
|
|
43
|
+
* const text = "# Heading\n\nThis is a paragraph.\n\n- List item 1\n- List item 2\n\n";
|
|
44
|
+
* const chunks = splitTextSemanticChars(text);
|
|
45
|
+
* console.log(chunks);
|
|
46
|
+
* // Output: ['# Heading', 'This is a paragraph.', '- List item 1', '- List item 2']
|
|
47
|
+
*/
|
|
48
|
+
export declare function splitTextSemanticChars(text: any, options?: {}): unknown[];
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview
|
|
3
|
+
* Splits text into sentences, handling 220+ common abbreviations,
|
|
4
|
+
* and inferring acronyms, numbers, URLs, times, names, etc.
|
|
5
|
+
*
|
|
6
|
+
* @param inputText - The text to be split into sentences.
|
|
7
|
+
* @param options - Configuration options for sentence splitting.
|
|
8
|
+
* @returns An array of sentences.
|
|
9
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
10
|
+
* @license MIT
|
|
11
|
+
* @example
|
|
12
|
+
* ```ts
|
|
13
|
+
* const text = "Dr. Smith went to the U.S. He met Mr. Jones.";
|
|
14
|
+
* const sentences = splitTextToSentences(text);
|
|
15
|
+
* ["Dr. Smith went to the U.S.", "He met Mr. Jones."]
|
|
16
|
+
* ```
|
|
17
|
+
*/
|
|
18
|
+
export declare function splitTextToSentences(inputText: string, options?: SplitSentencesOptions): string[];
|
|
19
|
+
export type SplitSentencesOptions = {
|
|
20
|
+
/**
|
|
21
|
+
* Split on HTML tags like P, DIV, UL, OL.
|
|
22
|
+
* @default true
|
|
23
|
+
*/
|
|
24
|
+
splitOnHtmlTags?: boolean;
|
|
25
|
+
/**
|
|
26
|
+
* Minimum size for a sentence.
|
|
27
|
+
* @default 20
|
|
28
|
+
*/
|
|
29
|
+
minSize?: number;
|
|
30
|
+
/**
|
|
31
|
+
* Maximum size for a sentence.
|
|
32
|
+
* @default 500
|
|
33
|
+
*/
|
|
34
|
+
maxSize?: number;
|
|
35
|
+
};
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Topic token tuple shape used by downstream ranking logic:
|
|
3
|
+
* [term, termCategory, uniqueness, metadata]
|
|
4
|
+
*/
|
|
5
|
+
export type TopicToken = [string, number, number, string];
|
|
6
|
+
/**
|
|
7
|
+
* Trie structure for phrase completion lookup keyed by first two letters, then full token.
|
|
8
|
+
*/
|
|
9
|
+
type PhraseEntry = [string | null, number, number];
|
|
10
|
+
export type PhrasesModel = Record<string, Record<string, PhraseEntry[]>>;
|
|
11
|
+
export interface ConvertTextToTokensOptions {
|
|
12
|
+
phrasesModel: PhrasesModel;
|
|
13
|
+
typosModel?: Record<string, string>;
|
|
14
|
+
checkTypos?: 0 | 1;
|
|
15
|
+
ignoreStopWords?: 0 | 1;
|
|
16
|
+
checkRootWords?: 0 | 1;
|
|
17
|
+
}
|
|
18
|
+
/**
|
|
19
|
+
* @typedef {Object} Token
|
|
20
|
+
* @property {number} termCategory - The category of the term
|
|
21
|
+
* @property {number} uniqueness - The uniqueness score of the term
|
|
22
|
+
* @property {string} term - The actual term or phrase
|
|
23
|
+
*/
|
|
24
|
+
/**
|
|
25
|
+
* ### Convert Text Query to Topic Phrase Tokens
|
|
26
|
+
* <img width="350px" src="https://i.imgur.com/NDrmSRQ.png" />
|
|
27
|
+
*
|
|
28
|
+
* Returns a list of phrases that are found in Wiki Titles/ dictionary phrases World Model
|
|
29
|
+
* that match the input phrase, or just the single word if found. Search results will be
|
|
30
|
+
* more accurate if we infer likely phrases and search for those words occuring together and
|
|
31
|
+
* not just split into words and find frequency. Examples are "white house" or "state of the art"
|
|
32
|
+
* which should be searched as a phrase but would return different context if split into words.
|
|
33
|
+
* As Led Zeppelin famously put it: \u266b "'Cause you know sometimes words have two meanings."
|
|
34
|
+
*
|
|
35
|
+
* @param {string} phrase
|
|
36
|
+
* @param {Object} [options]
|
|
37
|
+
* @param {Object} options.phrasesModel - remote model
|
|
38
|
+
* @param {Object} options.typosModel - remote model
|
|
39
|
+
* @param {number} options.checkTypos - check for typos
|
|
40
|
+
* @param {number} options.ignoreStopWords - ignore 300+ overused words
|
|
41
|
+
* @param {number} options.checkRootWords - check for word's root stem
|
|
42
|
+
* @returns {Array<{termCategory: number, uniqueness: number, term: string}>}
|
|
43
|
+
* @example
|
|
44
|
+
* const result = convertTextToTokens("The president of the united states is in the white house", { phrasesModel, typosModel });
|
|
45
|
+
* console.log(result);
|
|
46
|
+
*
|
|
47
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
48
|
+
* @category Topics
|
|
49
|
+
*/
|
|
50
|
+
export declare function convertTextToTokens(phrase: string, options?: Partial<ConvertTextToTokensOptions>): TopicToken[];
|
|
51
|
+
export {};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for identifying common stop words (e.g., "the", "and", "is").
|
|
3
|
+
* Based on the SpaCy English stop word list.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Checks word is in [320 commonly ignored "stop words
|
|
7
|
+
* "](https://raw.githubusercontent.com/igorbrigadir/stopwords/master/en/spacy.txt)
|
|
8
|
+
* in queries, using efficient JS Set method
|
|
9
|
+
* @param {string} word
|
|
10
|
+
* @returns {Boolean}
|
|
11
|
+
*/
|
|
12
|
+
export declare function isWordCommonIgnored(word: any): boolean;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Implementation of the Porter Stemmer algorithm for word normalization.
|
|
3
|
+
* Used to reduce words to their root form (e.g., "running" to "run").
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Stems a word using the [Porter
|
|
7
|
+
* Stemmer](https://snowballstem.org/algorithms/porter/stemmer.html)
|
|
8
|
+
* for removing inflectional endings like "ing", "ist", "ize".
|
|
9
|
+
*
|
|
10
|
+
* @author [Porter, M. (1980)](https://tartarus.org/martin/PorterStemmer/)
|
|
11
|
+
* @param word - The word to be stemmed
|
|
12
|
+
* @returns The stemmed word
|
|
13
|
+
* @example const rootWord = stemWordToRoot("running"); // returns "run"
|
|
14
|
+
* @category Topics
|
|
15
|
+
*/
|
|
16
|
+
export declare function stemWordToRoot(word: string): string;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Converts a DOCX document to HTML
|
|
3
|
+
*
|
|
4
|
+
* @param {string|File|Blob|ArrayBuffer|Buffer|Uint8Array} input - DOCX input to convert
|
|
5
|
+
* @param {DocxOptions} [options] - Conversion options
|
|
6
|
+
* @returns {Promise<string>} The converted HTML
|
|
7
|
+
* @throws {Error} If conversion fails
|
|
8
|
+
* @category Extract
|
|
9
|
+
* @example
|
|
10
|
+
* const html = await convertDOCXToHTML('https://example.com/doc.docx');
|
|
11
|
+
* const html = await convertDOCXToHTML(fileInput.files[0]);
|
|
12
|
+
*/
|
|
13
|
+
export declare function convertDOCXToHTML(input: any, options?: {}): Promise<string>;
|
|
14
|
+
/**
|
|
15
|
+
* Detects if a binary buffer is a DOCX file by checking the file signature
|
|
16
|
+
* DOCX files are ZIP archives with specific internal structure
|
|
17
|
+
*
|
|
18
|
+
* @param {ArrayBuffer|Buffer|Uint8Array} buffer - Binary buffer to check
|
|
19
|
+
* @returns {boolean} True if buffer appears to be a DOCX file
|
|
20
|
+
* @category Extract
|
|
21
|
+
*/
|
|
22
|
+
export declare function isBufferDOCX(buffer: any): boolean;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module research/extractor/url-to-content/is-url-porn
|
|
3
|
+
* @description Research library module.
|
|
4
|
+
*/
|
|
5
|
+
export interface IsURLPornOptions {
|
|
6
|
+
url?: string;
|
|
7
|
+
title?: string;
|
|
8
|
+
threshold?: number;
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* Determines if content is likely adult/porn based on configurable threshold
|
|
12
|
+
*
|
|
13
|
+
* @param {Object} options - Configuration object
|
|
14
|
+
* @param {string} [options.url] - URL to analyze (optional)
|
|
15
|
+
* @param {string} [options.title] - Page title to analyze (optional)
|
|
16
|
+
* @param {number} [options.threshold] - Probability threshold (0.5 default)
|
|
17
|
+
* @returns {boolean} True if likelihood exceeds threshold, false otherwise
|
|
18
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
19
|
+
* @example
|
|
20
|
+
* isURLPorn({
|
|
21
|
+
* title: "Hot deals on sexy cars",
|
|
22
|
+
* threshold: 0.8
|
|
23
|
+
* });
|
|
24
|
+
* console.log(isPorn2); // false (low confidence)
|
|
25
|
+
*/
|
|
26
|
+
export declare function isURLPorn(options?: IsURLPornOptions): boolean;
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
export interface ExtractContentOptions {
|
|
2
|
+
images?: boolean;
|
|
3
|
+
links?: boolean;
|
|
4
|
+
formatting?: boolean;
|
|
5
|
+
absoluteURLs?: boolean;
|
|
6
|
+
timeout?: number;
|
|
7
|
+
proxy?: string | null;
|
|
8
|
+
citeFormatMonthFull?: boolean;
|
|
9
|
+
citeFormatAuthorFull?: boolean;
|
|
10
|
+
url?: string;
|
|
11
|
+
useThirdPartyBackup?: boolean;
|
|
12
|
+
/** Preferred transcript languages when extracting YouTube videos. */
|
|
13
|
+
languages?: string[];
|
|
14
|
+
}
|
|
15
|
+
export interface ExtractedArticle {
|
|
16
|
+
cite?: string;
|
|
17
|
+
html?: string;
|
|
18
|
+
url?: string;
|
|
19
|
+
author?: string;
|
|
20
|
+
author_cite?: string;
|
|
21
|
+
author_short?: string;
|
|
22
|
+
author_type?: number | string;
|
|
23
|
+
date?: string;
|
|
24
|
+
title?: string;
|
|
25
|
+
source?: string;
|
|
26
|
+
word_count?: number;
|
|
27
|
+
format?: string;
|
|
28
|
+
error?: string | number;
|
|
29
|
+
}
|
|
30
|
+
type UrlLikeDocument = {
|
|
31
|
+
location?: {
|
|
32
|
+
href?: string;
|
|
33
|
+
};
|
|
34
|
+
querySelectorAll?: (selector: string) => {
|
|
35
|
+
length: number;
|
|
36
|
+
} | ArrayLike<unknown>;
|
|
37
|
+
};
|
|
38
|
+
/**
|
|
39
|
+
* @typedef {Object} Article
|
|
40
|
+
* @property {string} cite - Cite in APA Format with Author name in Last, First Initial format
|
|
41
|
+
* @property {string} html - The Basic HTML content of the article
|
|
42
|
+
* @property {string} url - The URL of the article
|
|
43
|
+
* @property {string} author - The full name of the author of the article
|
|
44
|
+
* @property {string} author_cite - Author name in Last, First Initial format
|
|
45
|
+
* @property {string} author_short - Author name in Last format
|
|
46
|
+
* @property {number} author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
47
|
+
* @property {string} date - The publication date of the article
|
|
48
|
+
* @property {string} title - The title of the article
|
|
49
|
+
* @property {string} source - The source or publisher of the article
|
|
50
|
+
* @property {number} word_count - The word count of the full text (without HTML tags)
|
|
51
|
+
* @category Extract
|
|
52
|
+
*/
|
|
53
|
+
/**
|
|
54
|
+
* ### 🚜 Tractor the Text Extractor
|
|
55
|
+
* <img width="350px" src="https://i.imgur.com/o8NTXxY.png" />
|
|
56
|
+
*
|
|
57
|
+
* 1. Main Content Detection: Extract the main content from a URL by combining
|
|
58
|
+
* Mozilla Readability and Postlight Mercury algorithms, utilizing over 100
|
|
59
|
+
* custom adapters for major sites for article, author, date HTML classes.
|
|
60
|
+
* 2. Basic HTML Standardization: Transform complex HTML into a simplified
|
|
61
|
+
* reading-mode format of basic HTML, making it ideal for research note archival
|
|
62
|
+
* and focused reading, with headings, images and links.
|
|
63
|
+
* 3. YouTube Transcript Processing: When a YouTube video URL is detected,
|
|
64
|
+
* retrieve the complete video transcript including both manual captions and
|
|
65
|
+
* auto-generated subtitles, maintaining proper timestamp synchronization and
|
|
66
|
+
* speaker identification where available.
|
|
67
|
+
* 4. PDF to HTML: Process PDF documents by extracting
|
|
68
|
+
* formatted text while intelligently handling line breaks, page headers,
|
|
69
|
+
* footnotes. The system analyzes text height statistics to automatically
|
|
70
|
+
* infer heading levels, creating a properly structured document hierarchy
|
|
71
|
+
* based on standard deviation from mean text size.
|
|
72
|
+
* 5. DOCX Binary Buffer Processing: Accept DOCX files as binary buffers
|
|
73
|
+
* (ArrayBuffer, Buffer, or Uint8Array) and automatically detect and convert
|
|
74
|
+
* them to HTML while preserving formatting, styles, and document structure.
|
|
75
|
+
* 6. Citation Information Extraction: Identify and extract citation metadata
|
|
76
|
+
* including author names, publication dates, sources, and titles using HTML
|
|
77
|
+
* meta tags and common class name patterns. The system validates author names
|
|
78
|
+
* against a comprehensive database of 90,000 first and last names,
|
|
79
|
+
* distinguishing between personal and organizational authors to properly
|
|
80
|
+
* format citations.
|
|
81
|
+
* 7. Author Name Formatting: Process author names by checking against
|
|
82
|
+
* known name databases, handling affixes and titles correctly, and determining
|
|
83
|
+
* whether to reverse the name order based on whether it's a personal or
|
|
84
|
+
* organizational author, ensuring proper citation formatting.
|
|
85
|
+
* @param {document|string|ArrayBuffer|Buffer|Uint8Array} urlOrDoc - url, dom object with article content, or binary buffer (DOCX)
|
|
86
|
+
* @param {Object} [options]
|
|
87
|
+
* @param {boolean} options.images default=true - include images
|
|
88
|
+
* @param {boolean} options.links default=true - include links
|
|
89
|
+
* @param {boolean} options.formatting default=true - preserve formatting
|
|
90
|
+
* @param {boolean} options.absoluteURLs default=true - convert URLs to absolute
|
|
91
|
+
* @param {number} options.timeout default=5 - http request timeout
|
|
92
|
+
* @returns {{
|
|
93
|
+
* title: string,
|
|
94
|
+
* author_cite: string,
|
|
95
|
+
* cite: string,
|
|
96
|
+
* author: string,
|
|
97
|
+
* date: string,
|
|
98
|
+
* source: string,
|
|
99
|
+
* html: string,
|
|
100
|
+
* word_count: number
|
|
101
|
+
* }}
|
|
102
|
+
* cite - Cite in APA Format with Author name in Last, First Initial format
|
|
103
|
+
* url - The URL of the article
|
|
104
|
+
* html - The HTML content of the article
|
|
105
|
+
* author - The author of the article
|
|
106
|
+
* author_cite - Author name in Last, First Middle format
|
|
107
|
+
* author_short - Author name in Last format
|
|
108
|
+
* author_type - Author type ["single", "two-author", "more-than-two", "organization"]
|
|
109
|
+
* date - The publication date of the article
|
|
110
|
+
* title - The title of the article
|
|
111
|
+
* source - The source or origin of the article
|
|
112
|
+
* word_count - The word count of the full text (without HTML tags)
|
|
113
|
+
* @category Extract
|
|
114
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
115
|
+
* @example
|
|
116
|
+
* // Extract from URL
|
|
117
|
+
* const result1 = await extractContent('https://example.com/article');
|
|
118
|
+
*
|
|
119
|
+
* // Extract from DOCX binary buffer
|
|
120
|
+
* const docxBuffer = new Uint8Array([...]); // DOCX file bytes
|
|
121
|
+
* const result2 = await extractContent(docxBuffer);
|
|
122
|
+
*
|
|
123
|
+
* // Extract from DOM object
|
|
124
|
+
* const result3 = await extractContent(document);
|
|
125
|
+
*/
|
|
126
|
+
export declare function extractContent(urlOrDoc: string | Document | UrlLikeDocument | ArrayBuffer | Buffer | Uint8Array, options?: ExtractContentOptions): Promise<ExtractedArticle>;
|
|
127
|
+
export {};
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ### Tardigrade the Web Crawler
|
|
3
|
+
* <img src="https://i.imgur.com/iuzpcvD.png" width="350px" />
|
|
4
|
+
*
|
|
5
|
+
* 1. **Use Fetch API, check for bot detection.** Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
|
|
6
|
+
* Scraping internet pages is a [free speech right
|
|
7
|
+
* ](https://blog.apify.com/is-web-scraping-legal/).
|
|
8
|
+
* 2. Features: timeout, redirects, default UA, referer as google, and bot
|
|
9
|
+
* detection checking. <br />
|
|
10
|
+
* 3. If fetch method does not get needed HTML, use Docker proxy as backup.
|
|
11
|
+
*
|
|
12
|
+
* 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
|
|
13
|
+
* container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
|
|
14
|
+
* secondary in-page API requests after the initial page request, including user login and cookie storage.
|
|
15
|
+
* 5. Bypass Cloudflare bot check: A webpage proxy that request through Chromium (puppeteer) - can be used
|
|
16
|
+
* to bypass Cloudflare anti bot using cookie id javascript method.
|
|
17
|
+
* 6. Send your request to the server with the port 3000 and add your URL to the "url"
|
|
18
|
+
* query string like this: `http://localhost:3000/?url=https://example.org`
|
|
19
|
+
*
|
|
20
|
+
* 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
|
|
21
|
+
* and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
|
|
22
|
+
* [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
|
|
23
|
+
* [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
|
|
24
|
+
* [Proxy-Cheap](https://app.proxy-cheap.com/order)
|
|
25
|
+
* [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
|
|
26
|
+
*
|
|
27
|
+
* @param {string} url - any domain's URL
|
|
28
|
+
* @param {Object} [options]
|
|
29
|
+
* @param {number} options.timeout default=5 - abort request if not retrived, in seconds
|
|
30
|
+
* @param {number} options.maxRedirects default=3 - max redirects to follow
|
|
31
|
+
* @param {number} options.checkBotDetection default=true - check for bot detection messages
|
|
32
|
+
* @param {number} options.changeReferer default=true - set referer as google
|
|
33
|
+
* @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
|
|
34
|
+
* @param {string} options.proxy default=false - use proxy url
|
|
35
|
+
* @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
|
|
36
|
+
* @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
|
|
37
|
+
* @category Extract
|
|
38
|
+
* @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
|
|
39
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
40
|
+
*/
|
|
41
|
+
export declare function scrapeURL(url: any, options?: {}): Promise<any>;
|
|
42
|
+
/**
|
|
43
|
+
* As backup, scrape with JINA to get html
|
|
44
|
+
* @param {string} url
|
|
45
|
+
* @returns {Promise<string>}
|
|
46
|
+
*/
|
|
47
|
+
export declare function scrapeJINA(url: any): Promise<any>;
|
|
48
|
+
/**
|
|
49
|
+
* Fetches and parses the robots.txt file for a given URL.
|
|
50
|
+
* @param {string} url - The base URL to fetch the robots.txt from.
|
|
51
|
+
* @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
|
|
52
|
+
*/
|
|
53
|
+
export declare function fetchScrapingRules(url: any): Promise<{
|
|
54
|
+
directives: {};
|
|
55
|
+
crawlDelay: {};
|
|
56
|
+
sitemaps: any[];
|
|
57
|
+
preferredHost: any;
|
|
58
|
+
} | {
|
|
59
|
+
error: string;
|
|
60
|
+
}>;
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Adapter helpers bridging the `extract-youtube` transcript API to
|
|
3
|
+
* the `getURLYoutubeVideo` / `convertYoutubeToText` helpers expected by the URL
|
|
4
|
+
* content extractor. Keeps the extractor decoupled from the transcript library's
|
|
5
|
+
* concrete API surface.
|
|
6
|
+
*/
|
|
7
|
+
/**
|
|
8
|
+
* Extracts the 11-character YouTube video id from a URL, if present.
|
|
9
|
+
*
|
|
10
|
+
* @param {string} url - A URL that may point to a YouTube video.
|
|
11
|
+
* @returns {string | null} The video id, or null when the URL is not a YouTube link.
|
|
12
|
+
*/
|
|
13
|
+
export declare function getURLYoutubeVideo(url: string): string | null;
|
|
14
|
+
/**
|
|
15
|
+
* Fetches a YouTube video transcript and returns it as a simple HTML document.
|
|
16
|
+
*
|
|
17
|
+
* @param {string} url - The YouTube video URL.
|
|
18
|
+
* @param {{ languages?: string[] }} [options] - Optional transcript languages.
|
|
19
|
+
* @returns {Promise<Record<string, any>>} An extraction response with `html`, or `{ error }`.
|
|
20
|
+
*/
|
|
21
|
+
export declare function convertYoutubeToText(url: string, options?: {
|
|
22
|
+
languages?: string[];
|
|
23
|
+
}): Promise<Record<string, any>>;
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fetch youtube.com video's webpage HTML for embedded transcript.
|
|
3
|
+
* If blocked, use scraper of alternative sites providing transcripts.
|
|
4
|
+
* @param {string} videoUrl
|
|
5
|
+
* @param {Object} [options]
|
|
6
|
+
* @param {boolean} options.addTimestamps default=true -
|
|
7
|
+
* true to return timestamps, default true
|
|
8
|
+
* @param {boolean} options.timeout default=5 - http request timeout
|
|
9
|
+
* @return {{content: string, timestamps: string, word_count: number}}
|
|
10
|
+
* where content is the full text of the transcript,
|
|
11
|
+
* timestamps is a string of comma-separated [characterIndex, timeSeconds] pairs,
|
|
12
|
+
* and word_count is the number of words in the transcript.
|
|
13
|
+
* @category Extract
|
|
14
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
15
|
+
*/
|
|
16
|
+
export declare function convertYoutubeToText(videoUrl: any, options?: {}): Promise<{
|
|
17
|
+
html: any;
|
|
18
|
+
word_count: any;
|
|
19
|
+
source: string;
|
|
20
|
+
date: any;
|
|
21
|
+
title: any;
|
|
22
|
+
author_cite: any;
|
|
23
|
+
length: number;
|
|
24
|
+
}>;
|
|
25
|
+
/**
|
|
26
|
+
* Test if URL is to youtube video and return video id if true
|
|
27
|
+
* @param {string} url - youtube video URL
|
|
28
|
+
* @returns {string|boolean} video ID or false
|
|
29
|
+
* @private
|
|
30
|
+
*/
|
|
31
|
+
export declare function getURLYoutubeVideo(url: any): any;
|
|
32
|
+
/**
|
|
33
|
+
* Fetch-based scraper of youtubetotranscript.com
|
|
34
|
+
* @returns {Object} content, timestamps - where content is the full text of
|
|
35
|
+
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
36
|
+
*/
|
|
37
|
+
export declare function fetchViaYoutubeToTranscriptCom(videoId: any, options?: {}): Promise<{
|
|
38
|
+
error: number;
|
|
39
|
+
content?: undefined;
|
|
40
|
+
title?: undefined;
|
|
41
|
+
author_cite?: undefined;
|
|
42
|
+
timestamps?: undefined;
|
|
43
|
+
} | {
|
|
44
|
+
content: string;
|
|
45
|
+
title: any;
|
|
46
|
+
author_cite: any;
|
|
47
|
+
timestamps: any[];
|
|
48
|
+
error?: undefined;
|
|
49
|
+
}>;
|
|
50
|
+
/** ========== NOT WORKING ========== */
|
|
51
|
+
/**
|
|
52
|
+
* Get YouTube transcript of most YouTube videos,
|
|
53
|
+
* except if disabled by uploader
|
|
54
|
+
* fetch-based scraper of youtubetranscript.com
|
|
55
|
+
*
|
|
56
|
+
* @param {string} videoUrl
|
|
57
|
+
* @returns {Object} where content is the full text of
|
|
58
|
+
* the transcript, and timestamps is an array of [characterIndex, timeSeconds]
|
|
59
|
+
* @private
|
|
60
|
+
*/
|
|
61
|
+
export declare function fetchViaYoutubeTranscript(videoId: any, options?: {}): Promise<{
|
|
62
|
+
error: number;
|
|
63
|
+
content?: undefined;
|
|
64
|
+
timestamps?: undefined;
|
|
65
|
+
} | {
|
|
66
|
+
content: string;
|
|
67
|
+
timestamps: any[];
|
|
68
|
+
error?: undefined;
|
|
69
|
+
}>;
|
|
70
|
+
export declare function extractYouTubeInfo(videoId: any, options?: {}): Promise<{}>;
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Fetch wrapper for grabbing binary content
|
|
3
|
+
* Replacement for grab-url package using standard fetch API
|
|
4
|
+
*/
|
|
5
|
+
export interface GrabOptions {
|
|
6
|
+
responseType?: "text" | "arraybuffer";
|
|
7
|
+
/** Timeout in seconds */
|
|
8
|
+
timeout?: number;
|
|
9
|
+
method?: string;
|
|
10
|
+
headers?: Record<string, string>;
|
|
11
|
+
body?: string;
|
|
12
|
+
}
|
|
13
|
+
export default function grab(url: string, options?: GrabOptions & {
|
|
14
|
+
responseType?: "text";
|
|
15
|
+
}): Promise<string>;
|
|
16
|
+
export default function grab(url: string, options: GrabOptions & {
|
|
17
|
+
responseType: "arraybuffer";
|
|
18
|
+
}): Promise<ArrayBuffer>;
|
package/package.json
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "extract-webpage",
|
|
3
|
+
"version": "1.2.5",
|
|
4
|
+
"module": "./dist/extract-webpage.es.js",
|
|
5
|
+
"description": "Search, extract, cite, and outline the web for a topic with AI Research Agent.",
|
|
6
|
+
"author": "vtempest <grokthiscontact@gmail.com>",
|
|
7
|
+
"license": "rights.institute/PROSPER",
|
|
8
|
+
"repository": {
|
|
9
|
+
"type": "git",
|
|
10
|
+
"url": "https://github.com/OpenSourceAGI/qwksearch-research-agent",
|
|
11
|
+
"directory": "packages/extract-webpage"
|
|
12
|
+
},
|
|
13
|
+
"main": "./dist/extract-webpage.cjs.js",
|
|
14
|
+
"types": "./dist/index.d.ts",
|
|
15
|
+
"exports": {
|
|
16
|
+
".": {
|
|
17
|
+
"types": "./dist/index.d.ts",
|
|
18
|
+
"import": "./dist/extract-webpage.es.js",
|
|
19
|
+
"require": "./dist/extract-webpage.cjs.js"
|
|
20
|
+
},
|
|
21
|
+
"./search": {
|
|
22
|
+
"types": "./src/search/index.ts",
|
|
23
|
+
"react-server": "./src/search/index.ts",
|
|
24
|
+
"import": "./src/search/index.ts",
|
|
25
|
+
"require": "./src/search/index.ts"
|
|
26
|
+
},
|
|
27
|
+
"./*": {
|
|
28
|
+
"types": "./src/*.ts",
|
|
29
|
+
"react-server": "./src/*",
|
|
30
|
+
"import": "./src/*",
|
|
31
|
+
"require": "./src/*"
|
|
32
|
+
}
|
|
33
|
+
},
|
|
34
|
+
"files": [
|
|
35
|
+
"dist",
|
|
36
|
+
"src"
|
|
37
|
+
],
|
|
38
|
+
"typesVersions": {
|
|
39
|
+
"*": {
|
|
40
|
+
"*": [
|
|
41
|
+
"src/*"
|
|
42
|
+
]
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
"scripts": {
|
|
46
|
+
"build": "vite build",
|
|
47
|
+
"test": "vitest",
|
|
48
|
+
"ship": "npm run build && npx standard-version --release-as patch; rm CHANGELOG.md; npm publish",
|
|
49
|
+
"test-ui": "vitest --ui --watch",
|
|
50
|
+
"make": "rm -rf dist/*; NODE_OPTIONS=--max-old-space-size=15192 BUN_JSC_forceRAMSize=15192 vite build "
|
|
51
|
+
},
|
|
52
|
+
"peerDependencies": {
|
|
53
|
+
"next": ">=15.0.0"
|
|
54
|
+
},
|
|
55
|
+
"peerDependenciesMeta": {
|
|
56
|
+
"next": {
|
|
57
|
+
"optional": true
|
|
58
|
+
}
|
|
59
|
+
},
|
|
60
|
+
"devDependencies": {
|
|
61
|
+
"@tsconfig/svelte": "^5.0.8",
|
|
62
|
+
"@types/node": "^22.0.0",
|
|
63
|
+
"@vitest/ui": "^4.0.18",
|
|
64
|
+
"axios": "^1.13.6",
|
|
65
|
+
"clsx": "^2.1.1",
|
|
66
|
+
"next": "^16.2.10",
|
|
67
|
+
"react": "^19.2.4",
|
|
68
|
+
"react-dom": "^19.2.4",
|
|
69
|
+
"terser": "^5.46.0",
|
|
70
|
+
"typedoc-plugin-markdown": "^4.10.0",
|
|
71
|
+
"typescript": "^5.9.3",
|
|
72
|
+
"vinext": "1.0.0-beta.0",
|
|
73
|
+
"vite": "^8.1.3",
|
|
74
|
+
"vite-plugin-dts": "^5.0.3",
|
|
75
|
+
"vite-plugin-node-polyfills": "^0.28.0",
|
|
76
|
+
"vitest": "^4.0.18"
|
|
77
|
+
},
|
|
78
|
+
"dependencies": {
|
|
79
|
+
"@huggingface/transformers": "^3.8.1",
|
|
80
|
+
"ai": "^5.0.0",
|
|
81
|
+
"chat-agent-toolkit": "^1.2.3",
|
|
82
|
+
"chrono-node": "^2.9.0",
|
|
83
|
+
"drizzle-orm": "^0.45.1",
|
|
84
|
+
"extract-pdf": "^0.1.1",
|
|
85
|
+
"extract-youtube": "^1.0.3",
|
|
86
|
+
"highlight.js": "^11.11.1",
|
|
87
|
+
"html-entities": "^2.6.0",
|
|
88
|
+
"js-yaml": "^4.1.1",
|
|
89
|
+
"jsdom": "^28.1.0",
|
|
90
|
+
"jszip": "^3.10.1",
|
|
91
|
+
"linkedom": "^0.18.12",
|
|
92
|
+
"marked": "^17.0.4",
|
|
93
|
+
"node-fetch": "^3.3.2",
|
|
94
|
+
"qwksearch-api-client": "^0.0.12",
|
|
95
|
+
"tldts": "^7.0.25",
|
|
96
|
+
"youtube-po-token-generator": "^0.6.0",
|
|
97
|
+
"zod": "^4.3.6"
|
|
98
|
+
},
|
|
99
|
+
"keywords": [
|
|
100
|
+
"nlp",
|
|
101
|
+
"autocomplete",
|
|
102
|
+
"knowledge-graph",
|
|
103
|
+
"keywords",
|
|
104
|
+
"search-algorithm",
|
|
105
|
+
"hacktoberfest",
|
|
106
|
+
"mind-map",
|
|
107
|
+
"ai-search"
|
|
108
|
+
]
|
|
109
|
+
}
|