extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extracts the author from the document and validates it as a human name
|
|
3
|
+
*
|
|
4
|
+
* @param {Document} document
|
|
5
|
+
* @returns {object|null} author_cite, author_short, author_type - or null if no valid author found
|
|
6
|
+
*/
|
|
7
|
+
export declare function extractAuthor(document: any): {
|
|
8
|
+
author_cite: any;
|
|
9
|
+
author_short: any;
|
|
10
|
+
author_type: number;
|
|
11
|
+
};
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
export interface ExtractCiteResult {
|
|
2
|
+
author?: string;
|
|
3
|
+
author_cite?: string;
|
|
4
|
+
date?: string;
|
|
5
|
+
title?: string;
|
|
6
|
+
source?: string;
|
|
7
|
+
}
|
|
8
|
+
export interface ExtractCiteOptions {
|
|
9
|
+
url?: string;
|
|
10
|
+
}
|
|
11
|
+
/**
|
|
12
|
+
* ### \u1f4da\u1f48e Extract Expert Excerpt
|
|
13
|
+
* <img width="350px" src="https://i.imgur.com/4GOOM9s.jpeg" />
|
|
14
|
+
*
|
|
15
|
+
* Extract author, date, source, and title from HTML using meta tags
|
|
16
|
+
* and common class names. Validates human name from author string to check
|
|
17
|
+
* against common list of 90k first names, last names,and organizations to infer
|
|
18
|
+
* if it should be reversed starting by author last name (accounting for affixes/titles),
|
|
19
|
+
* since organizations are not reversed.
|
|
20
|
+
* [Article Extraction Benchmark](https://github.com/scrapinghub/article-extraction-benchmark?tab=readme-ov-file#results)
|
|
21
|
+
* @param {Document | string} document dom object or html string with article content
|
|
22
|
+
* @param {ExtractCiteOptions} [options={}]
|
|
23
|
+
* @returns {ExtractCiteResult | null} An object containing extracted citation information.
|
|
24
|
+
* @category Extract
|
|
25
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
26
|
+
*/
|
|
27
|
+
export declare function extractCite(document: Document | string, options?: ExtractCiteOptions): {
|
|
28
|
+
author: string;
|
|
29
|
+
author_cite: any;
|
|
30
|
+
date: string;
|
|
31
|
+
title: string;
|
|
32
|
+
source: string;
|
|
33
|
+
};
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
declare const FAST_PREPEND = "";
|
|
2
|
+
declare const MIN_SEGMENT_LEN = 6;
|
|
3
|
+
declare const MAX_SEGMENT_LEN = 52;
|
|
4
|
+
declare const DATE_EXPRESSIONS: string;
|
|
5
|
+
declare const SLOW_PREPEND = "";
|
|
6
|
+
declare const FREE_TEXT_EXPRESSIONS = ".//*[self::div or self::h2 or self::h3 or self::h4 or self::li or self::p or self::span or self::time or self::ul]/text()";
|
|
7
|
+
declare const THREE_COMP_REGEX_A: RegExp;
|
|
8
|
+
declare const THREE_COMP_REGEX_B: RegExp;
|
|
9
|
+
declare const TWO_COMP_REGEX: RegExp;
|
|
10
|
+
declare const YEAR_PATTERN: RegExp;
|
|
11
|
+
declare const COPYRIGHT_PATTERN: RegExp;
|
|
12
|
+
declare const THREE_PATTERN: RegExp;
|
|
13
|
+
declare const THREE_CATCH: RegExp;
|
|
14
|
+
declare const THREE_LOOSE_PATTERN: RegExp;
|
|
15
|
+
declare const THREE_LOOSE_CATCH: RegExp;
|
|
16
|
+
declare const SELECT_YMD_PATTERN: RegExp;
|
|
17
|
+
declare const SELECT_YMD_YEAR: RegExp;
|
|
18
|
+
declare const YMD_YEAR: RegExp;
|
|
19
|
+
declare const DATESTRINGS_PATTERN: RegExp;
|
|
20
|
+
declare const DATESTRINGS_CATCH: RegExp;
|
|
21
|
+
declare const SLASHES_PATTERN: RegExp;
|
|
22
|
+
declare const SLASHES_YEAR: RegExp;
|
|
23
|
+
declare const YYYYMM_PATTERN: RegExp;
|
|
24
|
+
declare const YYYYMM_CATCH: RegExp;
|
|
25
|
+
declare const MMYYYY_PATTERN: RegExp;
|
|
26
|
+
declare const MMYYYY_YEAR: RegExp;
|
|
27
|
+
declare const SIMPLE_PATTERN: RegExp;
|
|
28
|
+
declare const YMD_PATTERN: RegExp;
|
|
29
|
+
declare const TIMESTAMP_PATTERN: RegExp;
|
|
30
|
+
declare function discard_unwanted(tree: any): any[];
|
|
31
|
+
declare function extract_url_date(testurl: any, options: any): string;
|
|
32
|
+
declare function regex_parse(string: any): Date;
|
|
33
|
+
declare function custom_parse(string: any, outputformat: any, min_date: any, max_date: any): any;
|
|
34
|
+
declare function external_date_parser(string: any, outputformat: any): string;
|
|
35
|
+
declare function try_date_expr(string: any, outputformat: any, extensive_search: any, min_date: any, max_date: any): any;
|
|
36
|
+
declare function img_search(tree: any, options: any): string;
|
|
37
|
+
declare function pattern_search(text: any, date_pattern: any, options: any): any;
|
|
38
|
+
declare function json_search(tree: any, options: any): any;
|
|
39
|
+
declare function idiosyncrasies_search(htmlstring: any, options: any): any;
|
|
40
|
+
export { discard_unwanted, extract_url_date, regex_parse, custom_parse, external_date_parser, try_date_expr, img_search, pattern_search, json_search, idiosyncrasies_search, DATE_EXPRESSIONS, FAST_PREPEND, SLOW_PREPEND, FREE_TEXT_EXPRESSIONS, MAX_SEGMENT_LEN, MIN_SEGMENT_LEN, YEAR_PATTERN, YMD_PATTERN, COPYRIGHT_PATTERN, TIMESTAMP_PATTERN, THREE_PATTERN, THREE_CATCH, THREE_LOOSE_PATTERN, THREE_LOOSE_CATCH, SELECT_YMD_PATTERN, SELECT_YMD_YEAR, YMD_YEAR, DATESTRINGS_PATTERN, DATESTRINGS_CATCH, SLASHES_PATTERN, SLASHES_YEAR, YYYYMM_PATTERN, YYYYMM_CATCH, MMYYYY_PATTERN, MMYYYY_YEAR, SIMPLE_PATTERN, THREE_COMP_REGEX_A, THREE_COMP_REGEX_B, TWO_COMP_REGEX, };
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Validation and filtering logic for extracted date candidates.
|
|
3
|
+
* Ensures dates fall within plausible ranges and meet format requirements.
|
|
4
|
+
*/
|
|
5
|
+
declare function is_valid_date(date_input: any, outputformat: any, earliest: any, latest: any): boolean;
|
|
6
|
+
declare function is_valid_format(outputformat: any): boolean;
|
|
7
|
+
declare function plausible_year_filter(htmlstring: any, pattern: any, yearpat: any, earliest: any, latest: any, incomplete?: boolean): Map<any, any>;
|
|
8
|
+
declare function compare_values(reference: any, attempt: any, options: any): any;
|
|
9
|
+
declare function filter_ymd_candidate(bestmatch: any, pattern: any, original_date: any, copyear: any, outputformat: any, min_date: any, max_date: any): any;
|
|
10
|
+
declare function convert_date(datestring: any, inputformat: any, outputformat: any): any;
|
|
11
|
+
declare function check_extracted_reference(reference: any, options: any): string;
|
|
12
|
+
declare function check_date_input(date_object: any, default_date: any): any;
|
|
13
|
+
declare function get_min_date(min_date: any): any;
|
|
14
|
+
declare function get_max_date(max_date: any): any;
|
|
15
|
+
export { is_valid_date, is_valid_format, plausible_year_filter, compare_values, filter_ymd_candidate, convert_date, check_extracted_reference, check_date_input, get_min_date, get_max_date };
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extract date from document using various methods
|
|
3
|
+
*
|
|
4
|
+
* @param {Document} document - DOM object with article content
|
|
5
|
+
* @param {string} url - URL of the page
|
|
6
|
+
* @returns {string|null} Extracted date or null if not found
|
|
7
|
+
*/
|
|
8
|
+
export declare function extractDateQuick(document: any, url: any): any;
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
declare const DATE_ATTRIBUTES: Set<string>;
|
|
2
|
+
declare const NAME_MODIFIED: Set<string>;
|
|
3
|
+
declare const PROPERTY_MODIFIED: Set<string>;
|
|
4
|
+
declare const ITEMPROP_ATTRS_ORIGINAL: Set<string>;
|
|
5
|
+
declare const ITEMPROP_ATTRS_MODIFIED: Set<string>;
|
|
6
|
+
declare const ITEMPROP_ATTRS: Set<string>;
|
|
7
|
+
declare const CLASS_ATTRS: Set<string>;
|
|
8
|
+
declare const NON_DIGITS_REGEX: RegExp;
|
|
9
|
+
export { DATE_ATTRIBUTES, NAME_MODIFIED, PROPERTY_MODIFIED, ITEMPROP_ATTRS_ORIGINAL, ITEMPROP_ATTRS_MODIFIED, ITEMPROP_ATTRS, CLASS_ATTRS, NON_DIGITS_REGEX, };
|
|
10
|
+
/**
|
|
11
|
+
* Extract date from document using various methods
|
|
12
|
+
*
|
|
13
|
+
* @param {Document} htmlobject - DOM object with article content
|
|
14
|
+
* @param {boolean} [extensive_search=true] - perform extensive search if true
|
|
15
|
+
* @param {boolean} [original_date=false] - return original date if true
|
|
16
|
+
* @param {string} [outputformat="%Y-%m-%d"] - output format
|
|
17
|
+
* @param {string} [url=null] - URL of the page
|
|
18
|
+
* @param {boolean} [verbose=false] - log debug messages if true
|
|
19
|
+
* @param {Date} [min_date=null] - minimum date to consider
|
|
20
|
+
* @param {Date} [max_date=null] - maximum date to consider
|
|
21
|
+
* @param {boolean} [deferred_url_extractor=false] - if true, do not extract date from URL
|
|
22
|
+
* @returns {string|null} Extracted date or null if not found
|
|
23
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
24
|
+
* Based on [Barbaresi (2020)](https://github.com/adbar/htmldate/)
|
|
25
|
+
*/
|
|
26
|
+
export declare function extractDate(htmlobject: any, extensive_search?: boolean, original_date?: boolean, outputformat?: string, url?: any, verbose?: boolean, min_date?: any, max_date?: any, deferred_url_extractor?: boolean): any;
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for identifying, extracting, and normalizing document titles from HTML.
|
|
3
|
+
* Handles metadata, selectors, and breadcrumb cleaning.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Extract and clean title from document
|
|
7
|
+
*
|
|
8
|
+
* @param {Document} document - DOM object with article content
|
|
9
|
+
* @returns {string} Extracted and cleaned title
|
|
10
|
+
*/
|
|
11
|
+
export declare function extractTitle(document: any): string;
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
export interface ExtractHumanNameOptions {
|
|
2
|
+
formatCiteShortenAuthor?: boolean;
|
|
3
|
+
maxAuthorsBeforeEtAl?: number;
|
|
4
|
+
}
|
|
5
|
+
/**
|
|
6
|
+
* Validates and formats author names properly handling multiple authors and multi-word names
|
|
7
|
+
*
|
|
8
|
+
* @param {string} author - The author name string(s) to be processed
|
|
9
|
+
* @param {ExtractHumanNameOptions} [options={}] - Configuration options
|
|
10
|
+
* @returns {object} Formatted author information for citation
|
|
11
|
+
*/
|
|
12
|
+
export declare function extractHumanName(author: string, options?: ExtractHumanNameOptions): {
|
|
13
|
+
author_cite: any;
|
|
14
|
+
author_short: any;
|
|
15
|
+
author_type: number;
|
|
16
|
+
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
export interface CiteMetadata {
|
|
2
|
+
author?: string;
|
|
3
|
+
date?: string;
|
|
4
|
+
title?: string;
|
|
5
|
+
source?: string;
|
|
6
|
+
}
|
|
7
|
+
/**
|
|
8
|
+
* Extract cite info from common property names in webpage's metadata
|
|
9
|
+
* @param {Document} doc dom object of document
|
|
10
|
+
* @returns {CiteMetadata} author, date, title, source
|
|
11
|
+
*/
|
|
12
|
+
export declare function extractCiteFromMetadata(doc: Document): CiteMetadata;
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for extracting and normalizing domain names from URLs.
|
|
3
|
+
* Handles subdomains and TLD cleaning for source attribution.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Extract TLD and hostname from domain in Regex. There's [two or more part
|
|
7
|
+
* TLDs](https://en.wikipedia.org/wiki/List_of_Internet_top-level_domains)
|
|
8
|
+
* so it is hard to tell if host.secondTLD.tld or host.tld is correct way
|
|
9
|
+
* to get root domain (e.g. abc.go.jp, abc.co.uk)
|
|
10
|
+
* @param {string} domain
|
|
11
|
+
* @returns {string} rootDomain
|
|
12
|
+
*/
|
|
13
|
+
export declare function convertURLToDomain(domain: any): any;
|
|
14
|
+
/**
|
|
15
|
+
* Checks if a string is a valid URL.
|
|
16
|
+
* @param {string} string
|
|
17
|
+
* @returns {boolean} true if the string is a valid URL
|
|
18
|
+
* @private
|
|
19
|
+
*/
|
|
20
|
+
export declare function isURLValid(string: any): boolean;
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module research/extractor/html-to-content/extract-content/extract-content-mercury-utils
|
|
3
|
+
* @description Research library module.
|
|
4
|
+
*/
|
|
5
|
+
declare function normalizeSpaces(text: any): any;
|
|
6
|
+
declare function paragraphize(node: any, document: any, br?: boolean): any;
|
|
7
|
+
declare function getAttrs(node: any): unknown;
|
|
8
|
+
declare function convertNodeTo(node: any, document: any, tag?: string): any;
|
|
9
|
+
declare function brsToPs(document: any): any;
|
|
10
|
+
declare function convertToParagraphs(document: any): any;
|
|
11
|
+
declare function cleanImages(article: any, document: any): any;
|
|
12
|
+
declare function stripJunkTags(article: any, document: any, tags?: any[]): any;
|
|
13
|
+
declare function cleanHOnes(article: any, document: any): any;
|
|
14
|
+
declare function cleanAttributes(article: any, document: any): any;
|
|
15
|
+
declare function removeEmpty(article: any): any;
|
|
16
|
+
declare function removeUnlessContent(node: any, weight: any): void;
|
|
17
|
+
declare function rewriteTopLevel(article: any, document: any): any;
|
|
18
|
+
declare function textLength(text: any): any;
|
|
19
|
+
declare function linkDensity(node: any): number;
|
|
20
|
+
declare function stripTags(text: any, document: any): any;
|
|
21
|
+
declare function stripUnlikelyCandidates(document: any): any;
|
|
22
|
+
declare function withinComment(node: any): boolean;
|
|
23
|
+
declare function nodeIsSufficient(node: any): boolean;
|
|
24
|
+
declare function isWordpress(document: any): boolean;
|
|
25
|
+
declare function setAttr(node: any, attr: any, val: any): any;
|
|
26
|
+
declare function setAttrs(node: any, attrs: any): any;
|
|
27
|
+
export { normalizeSpaces, paragraphize, getAttrs, convertNodeTo, brsToPs, convertToParagraphs, cleanImages, stripJunkTags, cleanHOnes, cleanAttributes, removeEmpty, rewriteTopLevel, textLength, linkDensity, stripTags, stripUnlikelyCandidates, withinComment, nodeIsSufficient, isWordpress, setAttr, removeUnlessContent, setAttrs, };
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ### HTML-to-Main-Content Extractor #2
|
|
3
|
+
*
|
|
4
|
+
* 1. The algorithm starts by loading the HTML content using linkedom, a lightweight DOM parser for Node.js.
|
|
5
|
+
* 2. It then applies a series of cleaning and scoring techniques to identify the main content of
|
|
6
|
+
* the page, starting with stripping unlikely candidates (e.g., elements with class names like "comment"
|
|
7
|
+
* or "sidebar").
|
|
8
|
+
* 3. The HTML is converted into a series of paragraph elements, which are then scored based on various
|
|
9
|
+
* factors such as text length, number of commas, and the presence of certain class names or IDs.
|
|
10
|
+
* 4. The algorithm assigns scores to parent and grandparent elements based on the scores of their
|
|
11
|
+
* children, with parents receiving the full score and grandparents receiving half.
|
|
12
|
+
* 5. After scoring, the algorithm finds the top candidate element by selecting the node with the
|
|
13
|
+
* highest score.
|
|
14
|
+
* 6. The top candidate's siblings are then examined to see if they should be included in the main
|
|
15
|
+
* content, based on their scores and other factors like link density.
|
|
16
|
+
* 7. The algorithm then cleans the selected content by removing unnecessary tags, attributes, and empty
|
|
17
|
+
* elements.
|
|
18
|
+
* 8. It also handles special cases like cleaning up header tags, images, and other potentially irrelevant
|
|
19
|
+
* content.
|
|
20
|
+
* 9. Throughout the process, the algorithm uses various regular expressions and scoring heuristics to
|
|
21
|
+
* identify positive and negative indicators of content relevance.
|
|
22
|
+
* 10. Finally, the cleaned and extracted content is returned as an HTML string, representing the main
|
|
23
|
+
* body of the article or webpage.
|
|
24
|
+
*
|
|
25
|
+
* [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
|
|
26
|
+
*
|
|
27
|
+
* @param {string} html - The HTML content to extract from.
|
|
28
|
+
* @param {Object} [opts] - The options for content extraction.
|
|
29
|
+
* @param {boolean} opts.stripUnlikelyCandidates default=true - Remove elements that match non-article-
|
|
30
|
+
* like criteria first (e.g., elements with a classname of "comment").
|
|
31
|
+
* @param {boolean} opts.weightNodes default=true - Modify an element's score based on certain classNames or
|
|
32
|
+
* IDs (e.g., subtract if a node has a className of 'comment', add if a node has an ID of 'entry-content').
|
|
33
|
+
* @param {boolean} opts.cleanConditionally default=true - Clean the node to remove superfluous content
|
|
34
|
+
* like forms, ads, etc. Initially, pass in the most restrictive options which will return the highest
|
|
35
|
+
* quality content. On each failure, retry with slightly more lax options.
|
|
36
|
+
* @returns {string} The extracted content as an HTML string, or null if extraction fails.
|
|
37
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
38
|
+
* Based on [Postlight Mercury Parser (2017-)](https://github.com/postlight/parser/tree/main/src)
|
|
39
|
+
* @example var url = "https://en.wikipedia.org/wiki/David_Hilbert"
|
|
40
|
+
* var html = await (await fetch(url)).text();
|
|
41
|
+
* var content = extractMainContentFromHTML(html);
|
|
42
|
+
* console.log(content); // HTML content of main article body
|
|
43
|
+
* @category Extract
|
|
44
|
+
*/
|
|
45
|
+
export declare function extractMainContentFromHTML2(html: any, opts: any): any;
|
|
46
|
+
/**
|
|
47
|
+
* Sets the score attribute of a node.
|
|
48
|
+
* @param {Node} node - The node to set the score on.
|
|
49
|
+
* @param {Document} document - The document object.
|
|
50
|
+
* @param {number} score - The score to set.
|
|
51
|
+
* @returns {Node} The node with the set score.
|
|
52
|
+
* @private
|
|
53
|
+
*/
|
|
54
|
+
export declare function setScore(node: any, document: any, score: any): any;
|
|
55
|
+
/**
|
|
56
|
+
* Scores a paragraph node.
|
|
57
|
+
* @param {Node} node - The paragraph node to score.
|
|
58
|
+
* @private
|
|
59
|
+
* @returns {number} The score of the paragraph.
|
|
60
|
+
*/
|
|
61
|
+
export declare function scoreParagraph(node: any): number;
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
interface Candidate {
|
|
2
|
+
score: number;
|
|
3
|
+
elem: any;
|
|
4
|
+
}
|
|
5
|
+
/**
|
|
6
|
+
* ### HTML-to-Main-Content Extractor #1
|
|
7
|
+
* The function extracts main content with regex patterns, cleaning HTML, scoring nodes
|
|
8
|
+
* based on content indicators like paragraphs and id/class names, selecting
|
|
9
|
+
* the top candidate, extracting it, and cleaning up content around it.
|
|
10
|
+
*
|
|
11
|
+
*
|
|
12
|
+
* 1. Define regular expressions:
|
|
13
|
+
* - Various regex patterns are defined to identify content and non-content areas.
|
|
14
|
+
*
|
|
15
|
+
* 2. Define helper functions:
|
|
16
|
+
* - normalizeSpaces: Normalizes whitespace in a string.
|
|
17
|
+
* - stripTags: Removes all HTML tags from a string.
|
|
18
|
+
* - getTextLength: Calculates the length of text after stripping tags.
|
|
19
|
+
* - calculateLinkDensity: Calculates the ratio of link text to total text.
|
|
20
|
+
*
|
|
21
|
+
* 3. Clean HTML:
|
|
22
|
+
* - Remove unlikely candidates (e.g., ads, sidebars) from the HTML.
|
|
23
|
+
*
|
|
24
|
+
* 4. Define scoring function:
|
|
25
|
+
* - scoreNode: Assigns a score to an HTML node based on content and attributes.
|
|
26
|
+
* - Increases score for positive indicators (e.g., article, body, content tags).
|
|
27
|
+
* - Decreases score for negative indicators (e.g., hidden, footer, sidebar tags).
|
|
28
|
+
* - Adds to score based on paragraph tags and text length.
|
|
29
|
+
*
|
|
30
|
+
* 5. Find and score candidate nodes:
|
|
31
|
+
* - Identify potential content nodes in the cleaned HTML.
|
|
32
|
+
* - Score each node using the scoreNode function.
|
|
33
|
+
*
|
|
34
|
+
* 6. Select top candidate:
|
|
35
|
+
* - Sort candidates by score and select the highest-scoring node.
|
|
36
|
+
*
|
|
37
|
+
* 7. Extract content:
|
|
38
|
+
* - Use regex to extract content around the top candidate node.
|
|
39
|
+
*
|
|
40
|
+
* 8. Clean up extracted content:
|
|
41
|
+
* - Remove script and style tags and their contents.
|
|
42
|
+
* - Process anchor tags based on content density.
|
|
43
|
+
* - Keep only specific HTML tags (a, p, img, h1-h6, ul, ol, li).
|
|
44
|
+
* - Remove excess whitespace from the final content.
|
|
45
|
+
*
|
|
46
|
+
* [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
|
|
47
|
+
*
|
|
48
|
+
* @example
|
|
49
|
+
* var url = "https://www.nytimes.com/2024/08/28/business/telegram-ceo-pavel-durov-charged.html"
|
|
50
|
+
* const html = await (await fetch(url)).text();
|
|
51
|
+
* var articleContent = extractMainContentFromHTML(html);
|
|
52
|
+
* @param {Object} [options]
|
|
53
|
+
* @param {number} options.minContentLength default=140 - Minimum length of content to be considered valid
|
|
54
|
+
* @param {number} options.minScore default=20 - Minimum score for content to be considered valid
|
|
55
|
+
* @param {number} options.minTextLength default=25 - Minimum length of text to be considered valid
|
|
56
|
+
* @param {number} options.retryLength default=250 - Length to retry content extraction if initial attempt fails
|
|
57
|
+
* @returns {string} Extracted HTML string of main content
|
|
58
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
59
|
+
* Based on [Mozilla Readability (2015)](https://github.com/mozilla/readability)
|
|
60
|
+
* @category Extract
|
|
61
|
+
*/
|
|
62
|
+
export declare function extractMainContentFromHTML(html: string, options?: {
|
|
63
|
+
minContentLength?: number;
|
|
64
|
+
minScore?: number;
|
|
65
|
+
minTextLength?: number;
|
|
66
|
+
retryLength?: number;
|
|
67
|
+
}): string;
|
|
68
|
+
/**
|
|
69
|
+
* Calculates the link density of an element.
|
|
70
|
+
* @param {Element} elem - The element to calculate link density for
|
|
71
|
+
* @returns {number} The link density (ratio of link text length to total text length)
|
|
72
|
+
*/
|
|
73
|
+
export declare function getLinkDensity(elem: any): number;
|
|
74
|
+
/**
|
|
75
|
+
* Calculates the weight of an element based on its class and id attributes.
|
|
76
|
+
* @param {Element} elem - The element to calculate weight for
|
|
77
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
78
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
79
|
+
* @returns {number} The calculated weight
|
|
80
|
+
*/
|
|
81
|
+
export declare function classWeight(elem: any, positiveRe: RegExp, negativeRe: RegExp): number;
|
|
82
|
+
/**
|
|
83
|
+
* Scores a node based on its tag name and attributes.
|
|
84
|
+
* @param {Element} elem - The element to score
|
|
85
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
86
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
87
|
+
* @returns {Object} An object containing the score and the element
|
|
88
|
+
*/
|
|
89
|
+
export declare function scoreNode(elem: any, positiveRe: RegExp, negativeRe: RegExp): Candidate;
|
|
90
|
+
/**
|
|
91
|
+
* Sanitizes the content by removing unwanted elements and cleaning remaining elements.
|
|
92
|
+
* @param {Element} node - The node to sanitize
|
|
93
|
+
* @param {Object} candidates - Object containing scored candidates
|
|
94
|
+
* @param {RegExp} videoRe - Regular expression for video URLs
|
|
95
|
+
* @param {RegExp} positiveRe - Regular expression for positive indicators
|
|
96
|
+
* @param {RegExp} negativeRe - Regular expression for negative indicators
|
|
97
|
+
* @param {number} minTextLength - Minimum text length to consider
|
|
98
|
+
* @returns {Element} The sanitized node
|
|
99
|
+
*/
|
|
100
|
+
export declare function sanitize(node: any, candidates: Record<string, Candidate>, videoRe: RegExp, positiveRe: RegExp, negativeRe: RegExp, minTextLength: number): any;
|
|
101
|
+
export {};
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Strip HTML to ~30 basic markup HTML tags, lists, tables, images.
|
|
3
|
+
* Convert anchors and relative urls to absolute urls. Basic HTML supports the same
|
|
4
|
+
* elements as Markdown, which is used in writing plain text. Markdown is converted
|
|
5
|
+
* to HTML anyways to display it, and it is better to edit basic HTML in a rich text editor.
|
|
6
|
+
*
|
|
7
|
+
* [Mozilla DOM Reference](https://developer.mozilla.org/en-US/docs/Web/API/Document_Object_Model) <br />
|
|
8
|
+
* [Source Code of Browser HTML DOM](https://chromium.googlesource.com/chromium/src/+/HEAD/third_party/blink/renderer/core/dom/) <br />
|
|
9
|
+
* [RegExp JS V8 Code](https://github.com/v8/v8/blob/94cde7c7f3fffc62f621e43f65be3d517b8a9f3d/src/regexp/regexp-compiler.cc#L3827)
|
|
10
|
+
* @param {string} html Any page's HTML to process
|
|
11
|
+
* @param {Object} [options]
|
|
12
|
+
* @param {boolean} options.images default=true - Whether to include images
|
|
13
|
+
* @param {boolean} options.links default=true - Whether to include links
|
|
14
|
+
* @param {boolean} options.videos default=true - Whether to include videos or not
|
|
15
|
+
* @param {boolean} options.formatting default=true - Whether to include formatting
|
|
16
|
+
* @param {string} options.url base URL for converting relative URLs to absolute
|
|
17
|
+
* @param {string} options.allowTags default="br,p,u,b,i ,em,strong,h1,h2,h3,h4, h5,h6,blockquote,
|
|
18
|
+
* code,ul,ol,li,dd,dl, table,th,tr,td,sub,sup" - Comma-separated list of allowed HTML tags.
|
|
19
|
+
* @param {string} options.allowedAttributes default="text,tag,href, src,type,width, height,id,data"
|
|
20
|
+
* List of allowed HTML attributes
|
|
21
|
+
* @returns {string} basic text formatting html
|
|
22
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
23
|
+
* @category HTML Utilities
|
|
24
|
+
*/
|
|
25
|
+
export declare function convertHTMLToBasicHTML(html: any, options?: {}): any[];
|
|
26
|
+
/**
|
|
27
|
+
* Convert html string to array of JSON Objects tokens to translate,
|
|
28
|
+
* convert, or filter all elements.
|
|
29
|
+
* Flat array is faster than DOMParser which uses nested trees.
|
|
30
|
+
* @param {string} html
|
|
31
|
+
* @returns {array} Example [{"tag": "img","src": ""}, ...]
|
|
32
|
+
|
|
33
|
+
* @private
|
|
34
|
+
*/
|
|
35
|
+
export declare function convertHTMLToTokens(html: any): any[];
|
|
36
|
+
export declare function addDOMFunctions(domObject: any): any;
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extracts the main content and citation information from a document or HTML string
|
|
3
|
+
* @param {string|object} documentOrHTML - The document or HTML string to extract content from
|
|
4
|
+
* @param {Object} options - Optional configuration options
|
|
5
|
+
* @param {boolean} options.images default=true - Whether to include images in the extracted content
|
|
6
|
+
* @param {boolean} options.links default=true - Whether to include links in the extracted content
|
|
7
|
+
* @param {boolean} options.formatting default=true - Whether to preserve formatting in the extracted content
|
|
8
|
+
* @param {string} options.url The URL of the original document, if available, for absolutify-ing URLs
|
|
9
|
+
* @param {boolean} options.useExtractor2 default=false -
|
|
10
|
+
* false uses Mozilla Readability, true uses Postlight Mercury.
|
|
11
|
+
* then use the alternate if the first returns less than 200 characters
|
|
12
|
+
* @returns {Object} The extracted content and citation information
|
|
13
|
+
* @property {string} title - The title of the document
|
|
14
|
+
* @property {string} author_cite - The full citation for the author
|
|
15
|
+
* @property {string} author_short - A shortened version of the author's name
|
|
16
|
+
* @property {string} author - The author's name
|
|
17
|
+
* @property {string} date - The publication date
|
|
18
|
+
* @property {string} source - The source of the document
|
|
19
|
+
* @property {string} html - The extracted HTML content
|
|
20
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
21
|
+
*/
|
|
22
|
+
export declare function extractContentAndCite(documentOrHTML: any, options?: {}): {
|
|
23
|
+
error: string;
|
|
24
|
+
title?: undefined;
|
|
25
|
+
author_cite?: undefined;
|
|
26
|
+
author_short?: undefined;
|
|
27
|
+
author?: undefined;
|
|
28
|
+
date?: undefined;
|
|
29
|
+
source?: undefined;
|
|
30
|
+
html?: undefined;
|
|
31
|
+
} | {
|
|
32
|
+
title: string;
|
|
33
|
+
author_cite: any;
|
|
34
|
+
author_short: any;
|
|
35
|
+
author: string;
|
|
36
|
+
date: string;
|
|
37
|
+
source: string;
|
|
38
|
+
html: any;
|
|
39
|
+
error?: undefined;
|
|
40
|
+
};
|
|
41
|
+
/**
|
|
42
|
+
* @typedef {Object} ExtractedContent
|
|
43
|
+
* @property {string} title - The title of the content
|
|
44
|
+
* @property {string} author_cite - The full citation for the author
|
|
45
|
+
* @property {string} author_short - A shortened version of the author's name
|
|
46
|
+
* @property {string} author - The author's name
|
|
47
|
+
* @property {string} date - The publication date
|
|
48
|
+
* @property {string} source - The source of the content
|
|
49
|
+
* @property {string} html - The extracted main content in HTML format
|
|
50
|
+
* @private
|
|
51
|
+
*/
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module research/extractor/html-to-content/html-utils
|
|
3
|
+
* @description Research library module.
|
|
4
|
+
*/
|
|
5
|
+
/**
|
|
6
|
+
* Converts URL-safe escaped HTML codes like &"'`’ & to standard HTML or in reverse.
|
|
7
|
+
* @param {string} str - The string to process.
|
|
8
|
+
* @param {boolean} toStandardHTML default=true - If true, converts url-safe codes
|
|
9
|
+
* to standard HTML. If false, converts standard HTML to url-safe codes.
|
|
10
|
+
* @return {string} The processed string.
|
|
11
|
+
* @category HTML Utilities
|
|
12
|
+
* @example
|
|
13
|
+
* var normalHTML = convertURLSafeHTMLToHTML('<p>This & that © 2023 '+
|
|
14
|
+
* '"Quotes"'Apostrophes' €100 ☺</p>', true)
|
|
15
|
+
* console.log(normalHTML) // "<p>This & that \u00a9 2023 "Quotes" 'Apostrophes' \u20ac100 \u263a</p>"
|
|
16
|
+
*/
|
|
17
|
+
export declare function convertURLSafeHTMLToHTML(str: any, toStandardHTML?: boolean): any;
|
|
18
|
+
/**
|
|
19
|
+
* Convert relative URL to absolute URL using base URL.
|
|
20
|
+
* @param {string} base base url of the domain
|
|
21
|
+
* @param {string} relative partial urls like ../images/image.jpg #hash
|
|
22
|
+
* @returns {string} absolute URL
|
|
23
|
+
* @example
|
|
24
|
+
* var absoluteURL = convertURLToAbsoluteURL('https://example.com', 'images/image.jpg')
|
|
25
|
+
* console.log(absoluteURL) // Returns: "https://example.com/images/image.jpg"
|
|
26
|
+
* var absoluteURL = convertURLToAbsoluteURL('https://example.com', '//images/image.jpg')
|
|
27
|
+
* console.log(absoluteURL) // Returns: "https:images/image.jpg"
|
|
28
|
+
* @category HTML Utilities
|
|
29
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
30
|
+
*/
|
|
31
|
+
export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
|
|
32
|
+
/**
|
|
33
|
+
* Converts Markdown text to HTML. It handles the following Markdown elements:
|
|
34
|
+
* - Headers (h1 to h6)
|
|
35
|
+
* - Bold text
|
|
36
|
+
* - Italic text
|
|
37
|
+
* - Unordered lists
|
|
38
|
+
* - Ordered lists
|
|
39
|
+
* - Paragraphs
|
|
40
|
+
* - Images
|
|
41
|
+
* - Links
|
|
42
|
+
* - Code blocks
|
|
43
|
+
* @param {string} content - The Markdown or HTML content to be converted.
|
|
44
|
+
* @param {boolean} toHtml - default=true - If true, converts Markdown to HTML.
|
|
45
|
+
* If false, converts HTML to Markdown.
|
|
46
|
+
* @returns {string} The resulting HTML string.
|
|
47
|
+
* @category HTML Utilities
|
|
48
|
+
* @example
|
|
49
|
+
* const markdown = "# Header\n\nThis is **bold** and *italic* text.\n\n* List item 1\n* List item 2";
|
|
50
|
+
* const html = convertMarkdownToHTML(markdown);
|
|
51
|
+
* console.log(html);
|
|
52
|
+
* // Output:
|
|
53
|
+
* // <h1>Header</h1>
|
|
54
|
+
* // <p>This is <strong>bold</strong> and <em>italic</em> text.</p>
|
|
55
|
+
* // <ul><li>List item 1</li><li>List item 2</li></ul>
|
|
56
|
+
*/
|
|
57
|
+
export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
|
|
58
|
+
export declare function convertHTMLToMarkdown(html: any): any;
|
|
59
|
+
/**
|
|
60
|
+
* Copy HTML to clipboard. When pasting into rich text field,
|
|
61
|
+
* pastes rich text. When pasting into plain text field, pastes:
|
|
62
|
+
* plain text, html, or markdown.
|
|
63
|
+
*
|
|
64
|
+
* @param {string} html - The HTML content to be copied.
|
|
65
|
+
* @param {object} options - The options object.
|
|
66
|
+
* @param {number} options.pastePlainFormat -
|
|
67
|
+
* default=0
|
|
68
|
+
* 0 - plain text
|
|
69
|
+
* 1 - markdown
|
|
70
|
+
* 2 - html
|
|
71
|
+
* @returns {Promise<void>} - A promise that resolves when
|
|
72
|
+
* the HTML is copied to the clipboard.
|
|
73
|
+
* @category HTML Utilities
|
|
74
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
75
|
+
*/
|
|
76
|
+
export declare function copyHTMLToClipboard(html: any, options?: {}): Promise<void>;
|
package/dist/index.d.ts
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Research Agent Library entry point.
|
|
3
|
+
* Exports various specialized agents, tools, and utilities for AI-driven research.
|
|
4
|
+
*
|
|
5
|
+
* @author vtempest <grokthiscontact@gmail.com>
|
|
6
|
+
* @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
|
|
7
|
+
* to get a dual-use commercial license to remove the GPL requirements.
|
|
8
|
+
*/
|
|
9
|
+
export * from './search/search-web';
|
|
10
|
+
export * from './search';
|
|
11
|
+
export * from './tokenize/word-to-root-stem';
|
|
12
|
+
export * from './tokenize/suggest-complete-word';
|
|
13
|
+
export * from './tokenize/text-to-topic-tokens';
|
|
14
|
+
export * from './tokenize/text-to-sentences';
|
|
15
|
+
export * from './tokenize/text-to-chunks';
|
|
16
|
+
export * from './url-to-content/url-to-content';
|
|
17
|
+
export * from './url-to-content/url-to-html';
|
|
18
|
+
export * from './html-to-cite/url-to-domain';
|
|
19
|
+
export * from './url-to-content/youtube-to-text';
|
|
20
|
+
export * from './url-to-content/docx-to-content';
|
|
21
|
+
export * from './html-to-content/html-to-content';
|
|
22
|
+
export * from './html-to-content/extract-content/extract-content-readability';
|
|
23
|
+
export * from './html-to-content/extract-content/extract-content-mercury';
|
|
24
|
+
export * from './html-to-content/html-to-basic-html';
|
|
25
|
+
export * from './html-to-cite/extract-cite';
|
|
26
|
+
export * from './html-to-content/html-utils';
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @module extract-webpage/search
|
|
3
|
+
* @description Re-exports MetaSearchAgent from agent-toolkit with search functions
|
|
4
|
+
*/
|
|
5
|
+
export declare const searchHandlers: {
|
|
6
|
+
webSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
7
|
+
academicSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
8
|
+
writingAssistant: import('chat-agent-toolkit').MetaSearchAgent;
|
|
9
|
+
wolframAlphaSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
10
|
+
youtubeSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
11
|
+
redditSearch: import('chat-agent-toolkit').MetaSearchAgent;
|
|
12
|
+
};
|
|
13
|
+
export { MetaSearchAgent, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
|
|
14
|
+
export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Re-exports MetaSearchAgent from chat-agent-toolkit
|
|
3
|
+
* @deprecated Import from 'chat-agent-toolkit' instead
|
|
4
|
+
*/
|
|
5
|
+
export { MetaSearchAgent as default, searchHandlers, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, } from 'chat-agent-toolkit';
|
|
6
|
+
export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
|
|
7
|
+
export { LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
|
|
8
|
+
export { webSearchResponsePrompt, webSearchRetrieverPrompt, webSearchRetrieverFewShots, writingAssistantPrompt, } from 'chat-agent-toolkit';
|