extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,11 @@
1
+ /**
2
+ * Extracts the author from the document and validates it as a human name
3
+ *
4
+ * @param {Document} document
5
+ * @returns {object|null} author_cite, author_short, author_type - or null if no valid author found
6
+ */
7
+ export declare function extractAuthor(document: any): {
8
+ author_cite: any;
9
+ author_short: any;
10
+ author_type: number;
11
+ };
@@ -0,0 +1,33 @@
1
+ export interface ExtractCiteResult {
2
+ author?: string;
3
+ author_cite?: string;
4
+ date?: string;
5
+ title?: string;
6
+ source?: string;
7
+ }
8
+ export interface ExtractCiteOptions {
9
+ url?: string;
10
+ }
11
+ /**
12
+ * ### \u1f4da\u1f48e Extract Expert Excerpt
13
+ * <img width="350px" src="https://i.imgur.com/4GOOM9s.jpeg" />
14
+ *
15
+ * Extract author, date, source, and title from HTML using meta tags
16
+ * and common class names. Validates human name from author string to check
17
+ * against common list of 90k first names, last names,and organizations to infer
18
+ * if it should be reversed starting by author last name (accounting for affixes/titles),
19
+ * since organizations are not reversed.
20
+ * [Article Extraction Benchmark](https://github.com/scrapinghub/article-extraction-benchmark?tab=readme-ov-file#results)
21
+ * @param {Document | string} document dom object or html string with article content
22
+ * @param {ExtractCiteOptions} [options={}]
23
+ * @returns {ExtractCiteResult | null} An object containing extracted citation information.
24
+ * @category Extract
25
+ * @author [vtempest (2025)](https://github.com/vtempest)
26
+ */
27
+ export declare function extractCite(document: Document | string, options?: ExtractCiteOptions): {
28
+ author: string;
29
+ author_cite: any;
30
+ date: string;
31
+ title: string;
32
+ source: string;
33
+ };
@@ -0,0 +1,40 @@
1
+ declare const FAST_PREPEND = "";
2
+ declare const MIN_SEGMENT_LEN = 6;
3
+ declare const MAX_SEGMENT_LEN = 52;
4
+ declare const DATE_EXPRESSIONS: string;
5
+ declare const SLOW_PREPEND = "";
6
+ declare const FREE_TEXT_EXPRESSIONS = ".//*[self::div or self::h2 or self::h3 or self::h4 or self::li or self::p or self::span or self::time or self::ul]/text()";
7
+ declare const THREE_COMP_REGEX_A: RegExp;
8
+ declare const THREE_COMP_REGEX_B: RegExp;
9
+ declare const TWO_COMP_REGEX: RegExp;
10
+ declare const YEAR_PATTERN: RegExp;
11
+ declare const COPYRIGHT_PATTERN: RegExp;
12
+ declare const THREE_PATTERN: RegExp;
13
+ declare const THREE_CATCH: RegExp;
14
+ declare const THREE_LOOSE_PATTERN: RegExp;
15
+ declare const THREE_LOOSE_CATCH: RegExp;
16
+ declare const SELECT_YMD_PATTERN: RegExp;
17
+ declare const SELECT_YMD_YEAR: RegExp;
18
+ declare const YMD_YEAR: RegExp;
19
+ declare const DATESTRINGS_PATTERN: RegExp;
20
+ declare const DATESTRINGS_CATCH: RegExp;
21
+ declare const SLASHES_PATTERN: RegExp;
22
+ declare const SLASHES_YEAR: RegExp;
23
+ declare const YYYYMM_PATTERN: RegExp;
24
+ declare const YYYYMM_CATCH: RegExp;
25
+ declare const MMYYYY_PATTERN: RegExp;
26
+ declare const MMYYYY_YEAR: RegExp;
27
+ declare const SIMPLE_PATTERN: RegExp;
28
+ declare const YMD_PATTERN: RegExp;
29
+ declare const TIMESTAMP_PATTERN: RegExp;
30
+ declare function discard_unwanted(tree: any): any[];
31
+ declare function extract_url_date(testurl: any, options: any): string;
32
+ declare function regex_parse(string: any): Date;
33
+ declare function custom_parse(string: any, outputformat: any, min_date: any, max_date: any): any;
34
+ declare function external_date_parser(string: any, outputformat: any): string;
35
+ declare function try_date_expr(string: any, outputformat: any, extensive_search: any, min_date: any, max_date: any): any;
36
+ declare function img_search(tree: any, options: any): string;
37
+ declare function pattern_search(text: any, date_pattern: any, options: any): any;
38
+ declare function json_search(tree: any, options: any): any;
39
+ declare function idiosyncrasies_search(htmlstring: any, options: any): any;
40
+ export { discard_unwanted, extract_url_date, regex_parse, custom_parse, external_date_parser, try_date_expr, img_search, pattern_search, json_search, idiosyncrasies_search, DATE_EXPRESSIONS, FAST_PREPEND, SLOW_PREPEND, FREE_TEXT_EXPRESSIONS, MAX_SEGMENT_LEN, MIN_SEGMENT_LEN, YEAR_PATTERN, YMD_PATTERN, COPYRIGHT_PATTERN, TIMESTAMP_PATTERN, THREE_PATTERN, THREE_CATCH, THREE_LOOSE_PATTERN, THREE_LOOSE_CATCH, SELECT_YMD_PATTERN, SELECT_YMD_YEAR, YMD_YEAR, DATESTRINGS_PATTERN, DATESTRINGS_CATCH, SLASHES_PATTERN, SLASHES_YEAR, YYYYMM_PATTERN, YYYYMM_CATCH, MMYYYY_PATTERN, MMYYYY_YEAR, SIMPLE_PATTERN, THREE_COMP_REGEX_A, THREE_COMP_REGEX_B, TWO_COMP_REGEX, };
@@ -0,0 +1,15 @@
1
+ /**
2
+ * @fileoverview Validation and filtering logic for extracted date candidates.
3
+ * Ensures dates fall within plausible ranges and meet format requirements.
4
+ */
5
+ declare function is_valid_date(date_input: any, outputformat: any, earliest: any, latest: any): boolean;
6
+ declare function is_valid_format(outputformat: any): boolean;
7
+ declare function plausible_year_filter(htmlstring: any, pattern: any, yearpat: any, earliest: any, latest: any, incomplete?: boolean): Map<any, any>;
8
+ declare function compare_values(reference: any, attempt: any, options: any): any;
9
+ declare function filter_ymd_candidate(bestmatch: any, pattern: any, original_date: any, copyear: any, outputformat: any, min_date: any, max_date: any): any;
10
+ declare function convert_date(datestring: any, inputformat: any, outputformat: any): any;
11
+ declare function check_extracted_reference(reference: any, options: any): string;
12
+ declare function check_date_input(date_object: any, default_date: any): any;
13
+ declare function get_min_date(min_date: any): any;
14
+ declare function get_max_date(max_date: any): any;
15
+ export { is_valid_date, is_valid_format, plausible_year_filter, compare_values, filter_ymd_candidate, convert_date, check_extracted_reference, check_date_input, get_min_date, get_max_date };
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Extract date from document using various methods
3
+ *
4
+ * @param {Document} document - DOM object with article content
5
+ * @param {string} url - URL of the page
6
+ * @returns {string|null} Extracted date or null if not found
7
+ */
8
+ export declare function extractDateQuick(document: any, url: any): any;
@@ -0,0 +1,26 @@
1
+ declare const DATE_ATTRIBUTES: Set<string>;
2
+ declare const NAME_MODIFIED: Set<string>;
3
+ declare const PROPERTY_MODIFIED: Set<string>;
4
+ declare const ITEMPROP_ATTRS_ORIGINAL: Set<string>;
5
+ declare const ITEMPROP_ATTRS_MODIFIED: Set<string>;
6
+ declare const ITEMPROP_ATTRS: Set<string>;
7
+ declare const CLASS_ATTRS: Set<string>;
8
+ declare const NON_DIGITS_REGEX: RegExp;
9
+ export { DATE_ATTRIBUTES, NAME_MODIFIED, PROPERTY_MODIFIED, ITEMPROP_ATTRS_ORIGINAL, ITEMPROP_ATTRS_MODIFIED, ITEMPROP_ATTRS, CLASS_ATTRS, NON_DIGITS_REGEX, };
10
+ /**
11
+ * Extract date from document using various methods
12
+ *
13
+ * @param {Document} htmlobject - DOM object with article content
14
+ * @param {boolean} [extensive_search=true] - perform extensive search if true
15
+ * @param {boolean} [original_date=false] - return original date if true
16
+ * @param {string} [outputformat="%Y-%m-%d"] - output format
17
+ * @param {string} [url=null] - URL of the page
18
+ * @param {boolean} [verbose=false] - log debug messages if true
19
+ * @param {Date} [min_date=null] - minimum date to consider
20
+ * @param {Date} [max_date=null] - maximum date to consider
21
+ * @param {boolean} [deferred_url_extractor=false] - if true, do not extract date from URL
22
+ * @returns {string|null} Extracted date or null if not found
23
+ * @author [vtempest (2025)](https://github.com/vtempest)
24
+ * Based on [Barbaresi (2020)](https://github.com/adbar/htmldate/)
25
+ */
26
+ export declare function extractDate(htmlobject: any, extensive_search?: boolean, original_date?: boolean, outputformat?: string, url?: any, verbose?: boolean, min_date?: any, max_date?: any, deferred_url_extractor?: boolean): any;
@@ -0,0 +1,7 @@
1
+ /**
2
+ * Extract source from document using common class names
3
+ *
4
+ * @param {document} document document or dom object with article content
5
+ * @returns {object} source
6
+ */
7
+ export declare function extractSource(document: any): any;
@@ -0,0 +1,11 @@
1
+ /**
2
+ * @fileoverview Utility for identifying, extracting, and normalizing document titles from HTML.
3
+ * Handles metadata, selectors, and breadcrumb cleaning.
4
+ */
5
+ /**
6
+ * Extract and clean title from document
7
+ *
8
+ * @param {Document} document - DOM object with article content
9
+ * @returns {string} Extracted and cleaned title
10
+ */
11
+ export declare function extractTitle(document: any): string;
@@ -0,0 +1,16 @@
1
+ export interface ExtractHumanNameOptions {
2
+ formatCiteShortenAuthor?: boolean;
3
+ maxAuthorsBeforeEtAl?: number;
4
+ }
5
+ /**
6
+ * Validates and formats author names properly handling multiple authors and multi-word names
7
+ *
8
+ * @param {string} author - The author name string(s) to be processed
9
+ * @param {ExtractHumanNameOptions} [options={}] - Configuration options
10
+ * @returns {object} Formatted author information for citation
11
+ */
12
+ export declare function extractHumanName(author: string, options?: ExtractHumanNameOptions): {
13
+ author_cite: any;
14
+ author_short: any;
15
+ author_type: number;
16
+ };
@@ -0,0 +1,12 @@
1
+ export interface CiteMetadata {
2
+ author?: string;
3
+ date?: string;
4
+ title?: string;
5
+ source?: string;
6
+ }
7
+ /**
8
+ * Extract cite info from common property names in webpage's metadata
9
+ * @param {Document} doc dom object of document
10
+ * @returns {CiteMetadata} author, date, title, source
11
+ */
12
+ export declare function extractCiteFromMetadata(doc: Document): CiteMetadata;
@@ -0,0 +1,20 @@
1
+ /**
2
+ * @fileoverview Utility for extracting and normalizing domain names from URLs.
3
+ * Handles subdomains and TLD cleaning for source attribution.
4
+ */
5
+ /**
6
+ * Extract TLD and hostname from domain in Regex. There's [two or more part
7
+ * TLDs](https://en.wikipedia.org/wiki/List_of_Internet_top-level_domains)
8
+ * so it is hard to tell if host.secondTLD.tld or host.tld is correct way
9
+ * to get root domain (e.g. abc.go.jp, abc.co.uk)
10
+ * @param {string} domain
11
+ * @returns {string} rootDomain
12
+ */
13
+ export declare function convertURLToDomain(domain: any): any;
14
+ /**
15
+ * Checks if a string is a valid URL.
16
+ * @param {string} string
17
+ * @returns {boolean} true if the string is a valid URL
18
+ * @private
19
+ */
20
+ export declare function isURLValid(string: any): boolean;
@@ -0,0 +1,27 @@
1
+ /**
2
+ * @module research/extractor/html-to-content/extract-content/extract-content-mercury-utils
3
+ * @description Research library module.
4
+ */
5
+ declare function normalizeSpaces(text: any): any;
6
+ declare function paragraphize(node: any, document: any, br?: boolean): any;
7
+ declare function getAttrs(node: any): unknown;
8
+ declare function convertNodeTo(node: any, document: any, tag?: string): any;
9
+ declare function brsToPs(document: any): any;
10
+ declare function convertToParagraphs(document: any): any;
11
+ declare function cleanImages(article: any, document: any): any;
12
+ declare function stripJunkTags(article: any, document: any, tags?: any[]): any;
13
+ declare function cleanHOnes(article: any, document: any): any;
14
+ declare function cleanAttributes(article: any, document: any): any;
15
+ declare function removeEmpty(article: any): any;
16
+ declare function removeUnlessContent(node: any, weight: any): void;
17
+ declare function rewriteTopLevel(article: any, document: any): any;
18
+ declare function textLength(text: any): any;
19
+ declare function linkDensity(node: any): number;
20
+ declare function stripTags(text: any, document: any): any;
21
+ declare function stripUnlikelyCandidates(document: any): any;
22
+ declare function withinComment(node: any): boolean;
23
+ declare function nodeIsSufficient(node: any): boolean;
24
+ declare function isWordpress(document: any): boolean;
25
+ declare function setAttr(node: any, attr: any, val: any): any;
26
+ declare function setAttrs(node: any, attrs: any): any;
27
+ export { normalizeSpaces, paragraphize, getAttrs, convertNodeTo, brsToPs, convertToParagraphs, cleanImages, stripJunkTags, cleanHOnes, cleanAttributes, removeEmpty, rewriteTopLevel, textLength, linkDensity, stripTags, stripUnlikelyCandidates, withinComment, nodeIsSufficient, isWordpress, setAttr, removeUnlessContent, setAttrs, };
@@ -0,0 +1,61 @@
1
+ /**
2
+ * ### HTML-to-Main-Content Extractor #2
3
+ *
4
+ * 1. The algorithm starts by loading the HTML content using linkedom, a lightweight DOM parser for Node.js.
5
+ * 2. It then applies a series of cleaning and scoring techniques to identify the main content of
6
+ * the page, starting with stripping unlikely candidates (e.g., elements with class names like "comment"
7
+ * or "sidebar").
8
+ * 3. The HTML is converted into a series of paragraph elements, which are then scored based on various
9
+ * factors such as text length, number of commas, and the presence of certain class names or IDs.
10
+ * 4. The algorithm assigns scores to parent and grandparent elements based on the scores of their
11
+ * children, with parents receiving the full score and grandparents receiving half.
12
+ * 5. After scoring, the algorithm finds the top candidate element by selecting the node with the
13
+ * highest score.
14
+ * 6. The top candidate's siblings are then examined to see if they should be included in the main
15
+ * content, based on their scores and other factors like link density.
16
+ * 7. The algorithm then cleans the selected content by removing unnecessary tags, attributes, and empty
17
+ * elements.
18
+ * 8. It also handles special cases like cleaning up header tags, images, and other potentially irrelevant
19
+ * content.
20
+ * 9. Throughout the process, the algorithm uses various regular expressions and scoring heuristics to
21
+ * identify positive and negative indicators of content relevance.
22
+ * 10. Finally, the cleaned and extracted content is returned as an HTML string, representing the main
23
+ * body of the article or webpage.
24
+ *
25
+ * [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
26
+ *
27
+ * @param {string} html - The HTML content to extract from.
28
+ * @param {Object} [opts] - The options for content extraction.
29
+ * @param {boolean} opts.stripUnlikelyCandidates default=true - Remove elements that match non-article-
30
+ * like criteria first (e.g., elements with a classname of "comment").
31
+ * @param {boolean} opts.weightNodes default=true - Modify an element's score based on certain classNames or
32
+ * IDs (e.g., subtract if a node has a className of 'comment', add if a node has an ID of 'entry-content').
33
+ * @param {boolean} opts.cleanConditionally default=true - Clean the node to remove superfluous content
34
+ * like forms, ads, etc. Initially, pass in the most restrictive options which will return the highest
35
+ * quality content. On each failure, retry with slightly more lax options.
36
+ * @returns {string} The extracted content as an HTML string, or null if extraction fails.
37
+ * @author [vtempest (2025)](https://github.com/vtempest)
38
+ * Based on [Postlight Mercury Parser (2017-)](https://github.com/postlight/parser/tree/main/src)
39
+ * @example var url = "https://en.wikipedia.org/wiki/David_Hilbert"
40
+ * var html = await (await fetch(url)).text();
41
+ * var content = extractMainContentFromHTML(html);
42
+ * console.log(content); // HTML content of main article body
43
+ * @category Extract
44
+ */
45
+ export declare function extractMainContentFromHTML2(html: any, opts: any): any;
46
+ /**
47
+ * Sets the score attribute of a node.
48
+ * @param {Node} node - The node to set the score on.
49
+ * @param {Document} document - The document object.
50
+ * @param {number} score - The score to set.
51
+ * @returns {Node} The node with the set score.
52
+ * @private
53
+ */
54
+ export declare function setScore(node: any, document: any, score: any): any;
55
+ /**
56
+ * Scores a paragraph node.
57
+ * @param {Node} node - The paragraph node to score.
58
+ * @private
59
+ * @returns {number} The score of the paragraph.
60
+ */
61
+ export declare function scoreParagraph(node: any): number;
@@ -0,0 +1,101 @@
1
+ interface Candidate {
2
+ score: number;
3
+ elem: any;
4
+ }
5
+ /**
6
+ * ### HTML-to-Main-Content Extractor #1
7
+ * The function extracts main content with regex patterns, cleaning HTML, scoring nodes
8
+ * based on content indicators like paragraphs and id/class names, selecting
9
+ * the top candidate, extracting it, and cleaning up content around it.
10
+ *
11
+ *
12
+ * 1. Define regular expressions:
13
+ * - Various regex patterns are defined to identify content and non-content areas.
14
+ *
15
+ * 2. Define helper functions:
16
+ * - normalizeSpaces: Normalizes whitespace in a string.
17
+ * - stripTags: Removes all HTML tags from a string.
18
+ * - getTextLength: Calculates the length of text after stripping tags.
19
+ * - calculateLinkDensity: Calculates the ratio of link text to total text.
20
+ *
21
+ * 3. Clean HTML:
22
+ * - Remove unlikely candidates (e.g., ads, sidebars) from the HTML.
23
+ *
24
+ * 4. Define scoring function:
25
+ * - scoreNode: Assigns a score to an HTML node based on content and attributes.
26
+ * - Increases score for positive indicators (e.g., article, body, content tags).
27
+ * - Decreases score for negative indicators (e.g., hidden, footer, sidebar tags).
28
+ * - Adds to score based on paragraph tags and text length.
29
+ *
30
+ * 5. Find and score candidate nodes:
31
+ * - Identify potential content nodes in the cleaned HTML.
32
+ * - Score each node using the scoreNode function.
33
+ *
34
+ * 6. Select top candidate:
35
+ * - Sort candidates by score and select the highest-scoring node.
36
+ *
37
+ * 7. Extract content:
38
+ * - Use regex to extract content around the top candidate node.
39
+ *
40
+ * 8. Clean up extracted content:
41
+ * - Remove script and style tags and their contents.
42
+ * - Process anchor tags based on content density.
43
+ * - Keep only specific HTML tags (a, p, img, h1-h6, ul, ol, li).
44
+ * - Remove excess whitespace from the final content.
45
+ *
46
+ * [Article Extraction Benchmark](https://trafilatura.readthedocs.io/en/latest/evaluation.html)
47
+ *
48
+ * @example
49
+ * var url = "https://www.nytimes.com/2024/08/28/business/telegram-ceo-pavel-durov-charged.html"
50
+ * const html = await (await fetch(url)).text();
51
+ * var articleContent = extractMainContentFromHTML(html);
52
+ * @param {Object} [options]
53
+ * @param {number} options.minContentLength default=140 - Minimum length of content to be considered valid
54
+ * @param {number} options.minScore default=20 - Minimum score for content to be considered valid
55
+ * @param {number} options.minTextLength default=25 - Minimum length of text to be considered valid
56
+ * @param {number} options.retryLength default=250 - Length to retry content extraction if initial attempt fails
57
+ * @returns {string} Extracted HTML string of main content
58
+ * @author [vtempest (2025)](https://github.com/vtempest)
59
+ * Based on [Mozilla Readability (2015)](https://github.com/mozilla/readability)
60
+ * @category Extract
61
+ */
62
+ export declare function extractMainContentFromHTML(html: string, options?: {
63
+ minContentLength?: number;
64
+ minScore?: number;
65
+ minTextLength?: number;
66
+ retryLength?: number;
67
+ }): string;
68
+ /**
69
+ * Calculates the link density of an element.
70
+ * @param {Element} elem - The element to calculate link density for
71
+ * @returns {number} The link density (ratio of link text length to total text length)
72
+ */
73
+ export declare function getLinkDensity(elem: any): number;
74
+ /**
75
+ * Calculates the weight of an element based on its class and id attributes.
76
+ * @param {Element} elem - The element to calculate weight for
77
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
78
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
79
+ * @returns {number} The calculated weight
80
+ */
81
+ export declare function classWeight(elem: any, positiveRe: RegExp, negativeRe: RegExp): number;
82
+ /**
83
+ * Scores a node based on its tag name and attributes.
84
+ * @param {Element} elem - The element to score
85
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
86
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
87
+ * @returns {Object} An object containing the score and the element
88
+ */
89
+ export declare function scoreNode(elem: any, positiveRe: RegExp, negativeRe: RegExp): Candidate;
90
+ /**
91
+ * Sanitizes the content by removing unwanted elements and cleaning remaining elements.
92
+ * @param {Element} node - The node to sanitize
93
+ * @param {Object} candidates - Object containing scored candidates
94
+ * @param {RegExp} videoRe - Regular expression for video URLs
95
+ * @param {RegExp} positiveRe - Regular expression for positive indicators
96
+ * @param {RegExp} negativeRe - Regular expression for negative indicators
97
+ * @param {number} minTextLength - Minimum text length to consider
98
+ * @returns {Element} The sanitized node
99
+ */
100
+ export declare function sanitize(node: any, candidates: Record<string, Candidate>, videoRe: RegExp, positiveRe: RegExp, negativeRe: RegExp, minTextLength: number): any;
101
+ export {};
@@ -0,0 +1,36 @@
1
+ /**
2
+ * Strip HTML to ~30 basic markup HTML tags, lists, tables, images.
3
+ * Convert anchors and relative urls to absolute urls. Basic HTML supports the same
4
+ * elements as Markdown, which is used in writing plain text. Markdown is converted
5
+ * to HTML anyways to display it, and it is better to edit basic HTML in a rich text editor.
6
+ *
7
+ * [Mozilla DOM Reference](https://developer.mozilla.org/en-US/docs/Web/API/Document_Object_Model) <br />
8
+ * [Source Code of Browser HTML DOM](https://chromium.googlesource.com/chromium/src/+/HEAD/third_party/blink/renderer/core/dom/) <br />
9
+ * [RegExp JS V8 Code](https://github.com/v8/v8/blob/94cde7c7f3fffc62f621e43f65be3d517b8a9f3d/src/regexp/regexp-compiler.cc#L3827)
10
+ * @param {string} html Any page's HTML to process
11
+ * @param {Object} [options]
12
+ * @param {boolean} options.images default=true - Whether to include images
13
+ * @param {boolean} options.links default=true - Whether to include links
14
+ * @param {boolean} options.videos default=true - Whether to include videos or not
15
+ * @param {boolean} options.formatting default=true - Whether to include formatting
16
+ * @param {string} options.url base URL for converting relative URLs to absolute
17
+ * @param {string} options.allowTags default="br,p,u,b,i ,em,strong,h1,h2,h3,h4, h5,h6,blockquote,
18
+ * code,ul,ol,li,dd,dl, table,th,tr,td,sub,sup" - Comma-separated list of allowed HTML tags.
19
+ * @param {string} options.allowedAttributes default="text,tag,href, src,type,width, height,id,data"
20
+ * List of allowed HTML attributes
21
+ * @returns {string} basic text formatting html
22
+ * @author [vtempest (2025)](https://github.com/vtempest)
23
+ * @category HTML Utilities
24
+ */
25
+ export declare function convertHTMLToBasicHTML(html: any, options?: {}): any[];
26
+ /**
27
+ * Convert html string to array of JSON Objects tokens to translate,
28
+ * convert, or filter all elements.
29
+ * Flat array is faster than DOMParser which uses nested trees.
30
+ * @param {string} html
31
+ * @returns {array} Example [{"tag": "img","src": ""}, ...]
32
+
33
+ * @private
34
+ */
35
+ export declare function convertHTMLToTokens(html: any): any[];
36
+ export declare function addDOMFunctions(domObject: any): any;
@@ -0,0 +1,51 @@
1
+ /**
2
+ * Extracts the main content and citation information from a document or HTML string
3
+ * @param {string|object} documentOrHTML - The document or HTML string to extract content from
4
+ * @param {Object} options - Optional configuration options
5
+ * @param {boolean} options.images default=true - Whether to include images in the extracted content
6
+ * @param {boolean} options.links default=true - Whether to include links in the extracted content
7
+ * @param {boolean} options.formatting default=true - Whether to preserve formatting in the extracted content
8
+ * @param {string} options.url The URL of the original document, if available, for absolutify-ing URLs
9
+ * @param {boolean} options.useExtractor2 default=false -
10
+ * false uses Mozilla Readability, true uses Postlight Mercury.
11
+ * then use the alternate if the first returns less than 200 characters
12
+ * @returns {Object} The extracted content and citation information
13
+ * @property {string} title - The title of the document
14
+ * @property {string} author_cite - The full citation for the author
15
+ * @property {string} author_short - A shortened version of the author's name
16
+ * @property {string} author - The author's name
17
+ * @property {string} date - The publication date
18
+ * @property {string} source - The source of the document
19
+ * @property {string} html - The extracted HTML content
20
+ * @author [vtempest (2025)](https://github.com/vtempest)
21
+ */
22
+ export declare function extractContentAndCite(documentOrHTML: any, options?: {}): {
23
+ error: string;
24
+ title?: undefined;
25
+ author_cite?: undefined;
26
+ author_short?: undefined;
27
+ author?: undefined;
28
+ date?: undefined;
29
+ source?: undefined;
30
+ html?: undefined;
31
+ } | {
32
+ title: string;
33
+ author_cite: any;
34
+ author_short: any;
35
+ author: string;
36
+ date: string;
37
+ source: string;
38
+ html: any;
39
+ error?: undefined;
40
+ };
41
+ /**
42
+ * @typedef {Object} ExtractedContent
43
+ * @property {string} title - The title of the content
44
+ * @property {string} author_cite - The full citation for the author
45
+ * @property {string} author_short - A shortened version of the author's name
46
+ * @property {string} author - The author's name
47
+ * @property {string} date - The publication date
48
+ * @property {string} source - The source of the content
49
+ * @property {string} html - The extracted main content in HTML format
50
+ * @private
51
+ */
@@ -0,0 +1,76 @@
1
+ /**
2
+ * @module research/extractor/html-to-content/html-utils
3
+ * @description Research library module.
4
+ */
5
+ /**
6
+ * Converts URL-safe escaped HTML codes like &"'`&rsquo; & to standard HTML or in reverse.
7
+ * @param {string} str - The string to process.
8
+ * @param {boolean} toStandardHTML default=true - If true, converts url-safe codes
9
+ * to standard HTML. If false, converts standard HTML to url-safe codes.
10
+ * @return {string} The processed string.
11
+ * @category HTML Utilities
12
+ * @example
13
+ * var normalHTML = convertURLSafeHTMLToHTML('&lt;p&gt;This &amp; that &copy; 2023 '+
14
+ * '&quot;Quotes&quot;&#39;Apostrophes&#39; &euro;100 &#x263A;&lt;/p&gt;', true)
15
+ * console.log(normalHTML) // "<p>This & that \u00a9 2023 "Quotes" 'Apostrophes' \u20ac100 \u263a</p>"
16
+ */
17
+ export declare function convertURLSafeHTMLToHTML(str: any, toStandardHTML?: boolean): any;
18
+ /**
19
+ * Convert relative URL to absolute URL using base URL.
20
+ * @param {string} base base url of the domain
21
+ * @param {string} relative partial urls like ../images/image.jpg #hash
22
+ * @returns {string} absolute URL
23
+ * @example
24
+ * var absoluteURL = convertURLToAbsoluteURL('https://example.com', 'images/image.jpg')
25
+ * console.log(absoluteURL) // Returns: "https://example.com/images/image.jpg"
26
+ * var absoluteURL = convertURLToAbsoluteURL('https://example.com', '//images/image.jpg')
27
+ * console.log(absoluteURL) // Returns: "https:images/image.jpg"
28
+ * @category HTML Utilities
29
+ * @author [vtempest (2025)](https://github.com/vtempest)
30
+ */
31
+ export declare function convertURLToAbsoluteURL(base: any, relative: any): any;
32
+ /**
33
+ * Converts Markdown text to HTML. It handles the following Markdown elements:
34
+ * - Headers (h1 to h6)
35
+ * - Bold text
36
+ * - Italic text
37
+ * - Unordered lists
38
+ * - Ordered lists
39
+ * - Paragraphs
40
+ * - Images
41
+ * - Links
42
+ * - Code blocks
43
+ * @param {string} content - The Markdown or HTML content to be converted.
44
+ * @param {boolean} toHtml - default=true - If true, converts Markdown to HTML.
45
+ * If false, converts HTML to Markdown.
46
+ * @returns {string} The resulting HTML string.
47
+ * @category HTML Utilities
48
+ * @example
49
+ * const markdown = "# Header\n\nThis is **bold** and *italic* text.\n\n* List item 1\n* List item 2";
50
+ * const html = convertMarkdownToHTML(markdown);
51
+ * console.log(html);
52
+ * // Output:
53
+ * // <h1>Header</h1>
54
+ * // <p>This is <strong>bold</strong> and <em>italic</em> text.</p>
55
+ * // <ul><li>List item 1</li><li>List item 2</li></ul>
56
+ */
57
+ export declare function convertMarkdownToHTML(content: any, toHtml?: boolean): any;
58
+ export declare function convertHTMLToMarkdown(html: any): any;
59
+ /**
60
+ * Copy HTML to clipboard. When pasting into rich text field,
61
+ * pastes rich text. When pasting into plain text field, pastes:
62
+ * plain text, html, or markdown.
63
+ *
64
+ * @param {string} html - The HTML content to be copied.
65
+ * @param {object} options - The options object.
66
+ * @param {number} options.pastePlainFormat -
67
+ * default=0
68
+ * 0 - plain text
69
+ * 1 - markdown
70
+ * 2 - html
71
+ * @returns {Promise<void>} - A promise that resolves when
72
+ * the HTML is copied to the clipboard.
73
+ * @category HTML Utilities
74
+ * @author [vtempest (2025)](https://github.com/vtempest)
75
+ */
76
+ export declare function copyHTMLToClipboard(html: any, options?: {}): Promise<void>;
@@ -0,0 +1,26 @@
1
+ /**
2
+ * @fileoverview Research Agent Library entry point.
3
+ * Exports various specialized agents, tools, and utilities for AI-driven research.
4
+ *
5
+ * @author vtempest <grokthiscontact@gmail.com>
6
+ * @license AGPL-3.0 Organizations should email grokthiscontact@gmail.com
7
+ * to get a dual-use commercial license to remove the GPL requirements.
8
+ */
9
+ export * from './search/search-web';
10
+ export * from './search';
11
+ export * from './tokenize/word-to-root-stem';
12
+ export * from './tokenize/suggest-complete-word';
13
+ export * from './tokenize/text-to-topic-tokens';
14
+ export * from './tokenize/text-to-sentences';
15
+ export * from './tokenize/text-to-chunks';
16
+ export * from './url-to-content/url-to-content';
17
+ export * from './url-to-content/url-to-html';
18
+ export * from './html-to-cite/url-to-domain';
19
+ export * from './url-to-content/youtube-to-text';
20
+ export * from './url-to-content/docx-to-content';
21
+ export * from './html-to-content/html-to-content';
22
+ export * from './html-to-content/extract-content/extract-content-readability';
23
+ export * from './html-to-content/extract-content/extract-content-mercury';
24
+ export * from './html-to-content/html-to-basic-html';
25
+ export * from './html-to-cite/extract-cite';
26
+ export * from './html-to-content/html-utils';
@@ -0,0 +1,14 @@
1
+ /**
2
+ * @module extract-webpage/search
3
+ * @description Re-exports MetaSearchAgent from agent-toolkit with search functions
4
+ */
5
+ export declare const searchHandlers: {
6
+ webSearch: import('chat-agent-toolkit').MetaSearchAgent;
7
+ academicSearch: import('chat-agent-toolkit').MetaSearchAgent;
8
+ writingAssistant: import('chat-agent-toolkit').MetaSearchAgent;
9
+ wolframAlphaSearch: import('chat-agent-toolkit').MetaSearchAgent;
10
+ youtubeSearch: import('chat-agent-toolkit').MetaSearchAgent;
11
+ redditSearch: import('chat-agent-toolkit').MetaSearchAgent;
12
+ };
13
+ export { MetaSearchAgent, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
14
+ export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
@@ -0,0 +1,8 @@
1
+ /**
2
+ * @fileoverview Re-exports MetaSearchAgent from chat-agent-toolkit
3
+ * @deprecated Import from 'chat-agent-toolkit' instead
4
+ */
5
+ export { MetaSearchAgent as default, searchHandlers, generateSuggestions, groupAndSummarizeDocs, buildFallbackDocs, rerankDocs, processDocs, normalizeSourcesOutput, splitTextIntoChunks, } from 'chat-agent-toolkit';
6
+ export type { MetaSearchAgentType, Config, ChatTurnMessage, SearchingEvent, FewShotExample, Document, } from 'chat-agent-toolkit';
7
+ export { LineOutputParser, LineListOutputParser, formatChatHistoryAsString, } from 'chat-agent-toolkit';
8
+ export { webSearchResponsePrompt, webSearchRetrieverPrompt, webSearchRetrieverFewShots, writingAssistantPrompt, } from 'chat-agent-toolkit';