extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,47 @@
1
+ /**
2
+ * Search Web via SearXNG metasearch of all major search engines.
3
+ */
4
+ export declare function searchWeb(query: string, options?: SearchOptions): Promise<SearxngSearchResult[] | SearchResponse>;
5
+ interface SearxngSearchOptions {
6
+ categories?: string[];
7
+ engines?: string[];
8
+ language?: string;
9
+ pageno?: number;
10
+ }
11
+ export declare const searchSearxng: (query: string, opts?: SearxngSearchOptions) => Promise<{
12
+ results: SearxngSearchResult[];
13
+ suggestions: string[];
14
+ }>;
15
+ interface SearchOptions {
16
+ category?: string | number;
17
+ recency?: string;
18
+ privateSearxng?: string | boolean | null;
19
+ maxRetries?: number;
20
+ page?: number;
21
+ safesearch?: boolean;
22
+ lang?: string;
23
+ proxy?: string | null;
24
+ useProxy?: boolean;
25
+ }
26
+ export interface SearxngSearchResult {
27
+ title: string;
28
+ url: string;
29
+ snippet?: string;
30
+ domain?: string;
31
+ favicon?: string;
32
+ score?: number;
33
+ source?: string;
34
+ date?: string;
35
+ img_src?: string;
36
+ thumbnail_src?: string;
37
+ thumbnail?: string;
38
+ content?: string;
39
+ author?: string;
40
+ iframe_src?: string;
41
+ }
42
+ export interface SearchResponse {
43
+ results: SearxngSearchResult[];
44
+ suggestions: string[];
45
+ infoboxes?: any[];
46
+ }
47
+ export {};
@@ -0,0 +1,33 @@
1
+ /**
2
+ * Search Web via SearXNG metasearch of all major search engines.
3
+ * Options are 10 search categories, recency, and how many
4
+ * times to retry other domains if first time fails.
5
+ * SearXNG is a free internet metasearch engine which aggregates results from
6
+ * more than [180+ search sources](https://docs.searxng.org/user/configured_engines.html).
7
+ *
8
+ * [Searxng Overview](https://medium.com/@elmo92/search-in-peace-with-searxng-an-alternative-search-engine-that-keeps-your-searches-private-accd8cddd6fc)
9
+ * [Searxng Installation Guide](https://github.com/searxng/searxng-docker/tree/master)
10
+ *
11
+ * ![google_dead](https://i.imgur.com/6rRpaY1.png)
12
+ * @param {string} query - The search query string.
13
+ * @param {Object} [options]
14
+ * @param {string} options.category default=general - ["general", "news", "videos", "images",
15
+ * "science","it", "files", "social+media", "map", "music"]
16
+ * @param {string} options.recency default=all - ["all", "day", "week", "month", "year"]
17
+ * @param {string|boolean} options.privateSearxng default=null - Use your custom domain SearXNG
18
+ * @param {number} options.maxRetries default=3 - Maximum number of retry attempts if the initial search fails.
19
+ * @param {number} options.page default=1 - The page number to retrieve.
20
+ * @param {boolean} options.safesearch default=false - Whether to block adult content.
21
+ * @param {string} options.lang default="en-US" - The language to use for the search.
22
+ * @param {string} options.proxy default=false - Use corsproxy.io to access in frontend JS
23
+ * @returns {Promise<Array<{title: string, url: string, snippet: string, domain: string, favicon: string, path: string, engines: string[]}>>} An array of search result objects.
24
+ * @example const advancedResults = await searchWeb('Node.js', {
25
+ * category: 2,
26
+ * recency: 1,
27
+ * maxRetries: 5
28
+ * });
29
+ * @category Search
30
+ * @author [vtempest (2025)](https://github.com/vtempest)
31
+ * [Heiser, M., Tauber, A., Flament, A., et al. (2014-)](https://github.com/searxng/searxng/graphs/contributors)
32
+ */
33
+ export declare function searchWeb(query: any, options?: any): Promise<any>;
@@ -0,0 +1,20 @@
1
+ interface TavilySearchOptions {
2
+ searchDepth?: "basic" | "advanced";
3
+ maxResults?: number;
4
+ includeDomains?: string[];
5
+ excludeDomains?: string[];
6
+ }
7
+ interface TavilySearchResult {
8
+ title: string;
9
+ url: string;
10
+ content: string;
11
+ score: number;
12
+ raw_content?: string;
13
+ }
14
+ export declare const searchTavily: (query: string, opts?: TavilySearchOptions) => Promise<{
15
+ results: TavilySearchResult[];
16
+ suggestions: string[];
17
+ }>;
18
+ export declare const getTavilyApiKey: () => any;
19
+ export declare const isTavilyConfigured: () => boolean;
20
+ export {};
@@ -0,0 +1,62 @@
1
+ /**
2
+ * ### Tardigrade the Web Crawler
3
+ *
4
+ * 1. **Use Fetch API, check for bot detection.
5
+ * Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
6
+ * Scraping internet pages is a [free speech right
7
+ * ](https://blog.apify.com/is-web-scraping-legal/).
8
+ * 2. Features: timeout, redirects, default UA, referer as google, and bot
9
+ * detection checking. <br />
10
+ * 3. If fetch method does not get needed HTML, use Docker proxy as backup.
11
+ *
12
+ * 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
13
+ * container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
14
+ * secondary in-page API requests after the initial page request, including user login and cookie storage.
15
+ * 5. Bypass Cloudflare bot check: A webpage proxy that request
16
+ * through Chromium (puppeteer) - can be used to bypass Cloudflare
17
+ * anti bot using cookie id javascript method.
18
+ * 6. Send your request to the server with the port 3000 and add your URL to the "url"
19
+ * query string like this: `http://localhost:3000/?url=https://example.org`
20
+ *
21
+ * 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
22
+ * and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
23
+ * [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
24
+ * [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
25
+ * [Proxy-Cheap](https://app.proxy-cheap.com/order)
26
+ * [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
27
+ *
28
+ * @param {string} url - any domain's URL
29
+ * @param {Object} [options]
30
+ * @param {number} options.timeout default=5 - abort request if not retrived, in seconds
31
+ * @param {number} options.maxRedirects default=3 - max redirects to follow
32
+ * @param {number} options.checkBotDetection default=true - check for bot detection messages
33
+ * @param {number} options.changeReferer default=true - set referer as google
34
+ * @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
35
+ * @param {string} options.proxy default=false - use proxy url
36
+ * @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
37
+ * @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
38
+ * @category Extract
39
+ * @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
40
+ * @author [vtempest (2025)](https://github.com/vtempest)
41
+ * @license MIT
42
+ */
43
+ export declare function scrapeURL(url: any, options?: any): Promise<any>;
44
+ /**
45
+ * As backup, scrape with JINA to get html
46
+ * @param {string} url
47
+ * @returns {Promise<string>}
48
+ */
49
+ export declare function scrapeJINA(url: any, timeout?: number): Promise<string>;
50
+ /**
51
+ * Fetches and parses the robots.txt file for a given URL.
52
+ * @param {string} url - The base URL to fetch the robots.txt from.
53
+ * @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
54
+ */
55
+ export declare function fetchScrapingRules(url: any): Promise<{
56
+ directives: {};
57
+ crawlDelay: {};
58
+ sitemaps: any[];
59
+ preferredHost: any;
60
+ } | {
61
+ error: string;
62
+ }>;
@@ -0,0 +1,28 @@
1
+ import { KeyphraseEntry } from './types';
2
+ /**
3
+ * Folds smaller keyphrases that are near-subsets of larger ones, merging their
4
+ * weights and sentence lists upward.
5
+ *
6
+ * Two phrases are considered overlapping when they share all-but-one word
7
+ * (i.e., the number of words in the smaller phrase that do **not** appear in
8
+ * the larger phrase is fewer than 2). When a merge occurs:
9
+ * - The larger phrase absorbs a fraction of the smaller one's weight
10
+ * (`smallWeight / largerWordCount`).
11
+ * - Sentence indices are union-merged.
12
+ * - If the smaller phrase actually outweighs the larger at that point, the
13
+ * larger entry adopts the smaller phrase's text (best representative wins).
14
+ *
15
+ * Phrases are processed largest-first so that supersets are always evaluated
16
+ * before their constituent sub-phrases.
17
+ *
18
+ * @param keyphrases - Raw scored keyphrases, any order.
19
+ * @returns Deduplicated, folded array \u2014 larger representative phrases only.
20
+ *
21
+ * @example
22
+ * const folded = foldSubphrases([
23
+ * { keyphrase: "machine learning", words: 2, weight: 10, sentences: [0, 1] },
24
+ * { keyphrase: "machine", words: 1, weight: 4, sentences: [0, 2] },
25
+ * ]);
26
+ * // "machine" is absorbed into "machine learning"
27
+ */
28
+ export declare function foldSubphrases(keyphrases: KeyphraseEntry[]): KeyphraseEntry[];
@@ -0,0 +1,27 @@
1
+ import { NgramMap, TopicToken } from './types';
2
+ /**
3
+ * Extracts noun-anchored edge-grams from a token array and accumulates them into `nGrams`.
4
+ *
5
+ * An edge-gram is a contiguous slice of `nGramSize` tokens where:
6
+ * - The **first** and **last** tokens are topic entities (noun or wiki title).
7
+ * - Every token is either a topic entity or a common stop word (e.g. "of", "the").
8
+ * - Every token meets the `minWordLength` character threshold.
9
+ *
10
+ * This allows natural multi-word keyphrases like "state of the art" or
11
+ * "machine learning" while ignoring pure function-word sequences.
12
+ *
13
+ * @param nGramSize - Number of tokens in the slice to evaluate.
14
+ * @param terms - Full token array for the current sentence.
15
+ * @param index - Start position for this slice within `terms`.
16
+ * @param nGrams - Accumulator map mutated in-place: `nGrams[size][phrase] = [sentenceIdx, ...]`.
17
+ * @param minWordLength - Minimum character length for any word to be included.
18
+ * @param sentenceIndex - Index of the originating sentence, appended to the phrase's entry.
19
+ * @returns The same `nGrams` reference (mutated).
20
+ *
21
+ * @example
22
+ * const terms: TopicToken[] = [["machine", 1, 4, ""], ["learning", 1, 5, ""]];
23
+ * const nGrams: NgramMap = {};
24
+ * extractNounEdgeGrams(2, terms, 0, nGrams, 3, 0);
25
+ * // nGrams[2]["machine learning"] === [0]
26
+ */
27
+ export declare function extractNounEdgeGrams(nGramSize: number, terms: TopicToken[], index: number, nGrams: NgramMap, minWordLength: number, sentenceIndex: number): NgramMap;
@@ -0,0 +1,28 @@
1
+ import { SentenceEntry, RankOptions } from './types';
2
+ /**
3
+ * ### TextRank: Rank Sentences by Centrality to Shared Keyphrases
4
+ *
5
+ * Builds a weighted undirected graph where each node is a sentence and edges
6
+ * connect sentences that share keyphrases. Sentence importance is estimated by
7
+ * a random-walk simulation (analogous to PageRank): nodes visited more often
8
+ * during the walk are considered more central to the document's key concepts.
9
+ *
10
+ * **Key optimisations vs. na\u00efve implementation:**
11
+ * - Weighted sampling via a single O(degree) scan instead of an O(weight) array.
12
+ * - `Set`-based keyphrase intersection (O(1) lookup) instead of `Array.includes`.
13
+ * - `Map<text, index>` for O(1) weight increment instead of O(n) linear scan.
14
+ * - Flat `Map<string, Map<string, number>>` graph \u2014 no object-method overhead.
15
+ * - Periodic forced reset prevents the walk from getting stuck in dense clusters.
16
+ *
17
+ * **References:**
18
+ * 1. Zhao & Xie (2021) \u2014 "An Improved TextRank Multi-feature Fusion Algorithm"
19
+ * https://iopscience.iop.org/article/10.1088/1742-6596/2078/1/012021/pdf
20
+ * 2. Pan et al. (2019) \u2014 "An improved TextRank keywords extraction algorithm"
21
+ * https://dl.acm.org/doi/10.1145/3321408.3326659
22
+ *
23
+ * @param sentencesWithKeyphrases - Sentences with pre-attached keyphrase lists.
24
+ * @param options - Walk parameters.
25
+ * @returns The same array with `weight` set on each sentence, or `undefined`
26
+ * if no edges exist (no shared keyphrases between any pair).
27
+ */
28
+ export declare function rankSentencesCentralToKeyphrase(sentencesWithKeyphrases: SentenceEntry[], options?: RankOptions): SentenceEntry[] | undefined;
@@ -0,0 +1,53 @@
1
+ import { KeyphraseEntry, SEEKTOPICOptions, SEEKTOPICResult } from './types';
2
+ /**
3
+ * ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
4
+ *
5
+ * Pulls the most important phrases and sentences out of any document.
6
+ * Given raw text, it returns a ranked list of key concepts (e.g. "neural
7
+ * network", "climate change") and, optionally, the sentences that best
8
+ * summarise the document around those concepts.
9
+ *
10
+ * <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
11
+ *
12
+ * **How it works \u2014 8-step pipeline:**
13
+ *
14
+ * 1. **Clean** \u2014 strip HTML tags and entities.
15
+ * 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
16
+ * 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
17
+ * to extract the most descriptive key topic phrases labeling the document.
18
+ * 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
19
+ * for all sentences vs topic phrases, finding the top relevant sentences per topic.
20
+ * 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
21
+ * n-grams (1-N words).
22
+ * 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
23
+ * 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
24
+ * 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
25
+ *
26
+ * <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
27
+ * width="550px" controls />
28
+ *
29
+ * @param docText - Plain text or HTML document to analyse.
30
+ * @param options - Optional tuning parameters (word limits, thresholds, query bias).
31
+ * @returns
32
+ * - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
33
+ * - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
34
+ * `keyphrases`, and the full `sentences` array.
35
+ *
36
+ * @example
37
+ * // Fast: just extract keyphrases
38
+ * const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
39
+ * // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
40
+ *
41
+ * @example
42
+ * // Full: keyphrases + summary sentences, biased toward a search query
43
+ * const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
44
+ * phrasesModel,
45
+ * optionSkipRanking: false,
46
+ * heavyWeightQuery: "transformer attention",
47
+ * limitTopSentences: 5,
48
+ * }) as SEEKTOPICResult;
49
+ *
50
+ * @author [ai-research-agent (2024)](https://airesearch.js.org)
51
+ * @category Topics
52
+ */
53
+ export declare function extractSEEKTOPIC(docText: string, options?: SEEKTOPICOptions): Promise<KeyphraseEntry[] | SEEKTOPICResult>;
@@ -0,0 +1,86 @@
1
+ import { TopicToken, PhrasesModel } from '../tokenize/text-to-topic-tokens';
2
+ export type { TopicToken, PhrasesModel };
3
+ /** Map from n-gram size \u2192 { phrase text \u2192 sentence indices } */
4
+ export type NgramMap = Record<number, Record<string, number[]>>;
5
+ /**
6
+ * A ranked keyphrase with its scoring metadata and the sentence indices it appears in.
7
+ */
8
+ export interface KeyphraseEntry {
9
+ /** The keyphrase text */
10
+ keyphrase: string;
11
+ /** Indices of sentences containing this keyphrase (or comma-separated string form) */
12
+ sentences: number[] | string;
13
+ /** Sentences representing this keyphrase directly with their relevance similarity */
14
+ topSentences?: Array<{
15
+ text: string;
16
+ similarity: number;
17
+ }>;
18
+ /** Number of words in the keyphrase */
19
+ words: number;
20
+ /** Composite ranking weight */
21
+ weight: number;
22
+ /** True if the phrase is a Wikipedia-linked entity */
23
+ wiki?: boolean;
24
+ }
25
+ /**
26
+ * A sentence node used during graph construction and TextRank scoring.
27
+ */
28
+ export interface SentenceEntry {
29
+ text: string;
30
+ index: number;
31
+ keyphrases: Array<{
32
+ keyphrase: string;
33
+ weight: number;
34
+ }>;
35
+ weight: number;
36
+ }
37
+ /**
38
+ * A sentence in the final SEEKTOPIC output \u2014 keyphrases are resolved to strings.
39
+ */
40
+ export interface SentenceResult {
41
+ text: string;
42
+ index: number;
43
+ keyphrases: string[];
44
+ weight: number;
45
+ }
46
+ /** Options for {@link extractSEEKTOPIC} */
47
+ export interface SEEKTOPICOptions {
48
+ /** Trie phrases model for wiki-phrase tokenization */
49
+ phrasesModel?: PhrasesModel;
50
+ /** Maximum words per keyphrase (default 2) */
51
+ maxWords?: number;
52
+ /** Minimum words per keyphrase (default 1) */
53
+ minWords?: number;
54
+ /** Minimum character length of any word in a keyphrase (default 3) */
55
+ minWordLength?: number;
56
+ /** Fraction of all keyphrases to use as the TextRank graph limit (default 0.5) */
57
+ topKeyphrasesPercent?: number;
58
+ /** Max sentences to return in full-ranking mode (default 5) */
59
+ limitTopSentences?: number;
60
+ /** Max keyphrases to return (default 10) */
61
+ limitTopKeyphrases?: number;
62
+ /** Minimum character length of the full keyphrase string (default 5) */
63
+ minKeyPhraseLength?: number;
64
+ /** Query string to bias keyphrase weights toward (default "") */
65
+ heavyWeightQuery?: string;
66
+ /** Strip HTML tags before processing (default true) */
67
+ removeHTML?: boolean;
68
+ /** Return only keyphrases without running TextRank (default true) */
69
+ optionSkipRanking?: boolean;
70
+ getEnv?: (key: string) => string | undefined;
71
+ }
72
+ /** Full SEEKTOPIC output when `optionSkipRanking` is false */
73
+ export interface SEEKTOPICResult {
74
+ topSentences: SentenceResult[];
75
+ keyphrases: Array<Omit<KeyphraseEntry, "sentences"> & {
76
+ sentences: string;
77
+ }>;
78
+ sentences: string[];
79
+ }
80
+ /** Options for {@link rankSentencesCentralToKeyphrase} */
81
+ export interface RankOptions {
82
+ /** Number of random-walk steps (default 1000) */
83
+ iterations?: number;
84
+ /** Steps between forced random resets to avoid cluster traps (default 100) */
85
+ resetInterval?: number;
86
+ }
@@ -0,0 +1,89 @@
1
+ /**
2
+ * Text embeddings convert words or phrases into numerical vectors in a high-dimensional
3
+ * space, where each dimension represents a semantic feature extracted by a model like
4
+ * MiniLM-L6-v2. In this concept space, words with similar meanings have vectors that
5
+ * are close together, allowing for quantitative comparisons of semantic similarity.
6
+ * These vector representations enable powerful applications in natural language processing,
7
+ * including semantic search, text classification, and clustering, by leveraging the
8
+ * geometric properties of the embedding space to capture and analyze the relationships
9
+ * between words and concepts.
10
+ * [Text Embeddings, Classification, and Semantic Search
11
+ * (Youtube)](https://www.youtube.com/watch?v=sNa_uiqSlJo&t=129s)
12
+ *
13
+ * <img src="https://i.imgur.com/wtJqEqX.png" width="350" />
14
+ * @param {string} text - The text to embed.
15
+ * @param {Object} [options]
16
+ * @param {AutoTokenizer} options.pipeline
17
+ * - The pipeline to use for embedding.
18
+ * @param {number} options.precision default=4 - The number of decimal places to round to.
19
+ * @returns {Promise<{embeddingsDict: Object.<string, number[]>, embedding: number[]}>}
20
+ * @category Similarity
21
+ */
22
+ export declare function convertTextToEmbedding(text: string | string[], options?: any): Promise<any>;
23
+ /**
24
+ * Calculate the semantic similarity between one text and a list of
25
+ * other sentences by comparing their embeddings.
26
+ * https://huggingface.co/docs/api-inference/detailed_parameters#sentence-similarity-task
27
+ *
28
+ * <img src="https://i.imgur.com/ex2UWnu.png" width="350px" />
29
+ * @param {string} source_sentence The string that you wish to
30
+ * compare the other strings with. This can be a phrase, sentence,
31
+ * or longer passage, depending on the model being used.
32
+ * @param {Array<string>} sentences A list of strings which will be compared
33
+ * against the source_sentence.
34
+ * @param {Object} [options]
35
+ * @param {string} options.model default="sentence-transformers/all-MiniLM-L6-v2"
36
+ * @param {string} options.HF_API_KEY Required https://huggingface.co/settings/tokens
37
+ * @returns array of 0-1 similarity scores for each sentence
38
+ * @category Similarity
39
+ */
40
+ export declare function weighRelevanceConceptVectorAPI(source_sentence: any, sentences: any, options?: {}): Promise<string | {
41
+ error: string;
42
+ }>;
43
+ /**
44
+ * Initialize HuggingFace Transformers pipeline for embedding text.
45
+ *
46
+ * <img src="https://i.imgur.com/3R5Tsrf.png" width="350px" />
47
+ * @param {Object} [options]
48
+ * @param {string} options.pipelineName default "feature-extraction",
49
+ * @param {string} options.modelName default="Xenova/all-MiniLM-L6-v2" -
50
+ * The name of the model to use
51
+ * @returns {Promise<import("@huggingface/transformers").AutoTokenizer>} The pipeline.
52
+ * @category Similarity
53
+ */
54
+ export declare function getEmbeddingModel(options?: any): Promise<any>;
55
+ /**
56
+ * [Cosine similarity](https://en.wikipedia.org/wiki/Cosine_similarity) gets similarity of two
57
+ * vectors by whether they have the same direction (similar) or are poles apart. Cosine similarity
58
+ * is often used with text representations to compare how similar two documents or sentences
59
+ * are to each other. The output of cosine similarity ranges from -1 to 1, where -1 means the
60
+ * two vectors are completely dissimilar, and 1 indicates maximum similarity.
61
+ * @param {Array<number>} vectorA
62
+ * @param {Array<number>} vectorB
63
+ * @returns {number} -1 to 1 similarity score
64
+ */
65
+ export declare function calculateCosineSimilarity(vectorA: any, vectorB: any): number;
66
+ /**
67
+ * Rerank documents's chunks based on relevance to query,
68
+ * based on cosine similarity of their concept vectors generated
69
+ * by a 20MB MiniLM transformer model downloaded locally.
70
+ *
71
+ * [A Complete Overview of Word Embeddings](https://www.youtube.com/watch?v=5MaWmXwxFNQ&t=323s)
72
+ * @param {Array<string>} documents
73
+ * @param {string} query
74
+ * @param {Object} [options]
75
+ * @returns {Promise<Array<{content: string, similarity: number}>>}
76
+ * @category Similarity
77
+ */
78
+ export declare function weighRelevanceConceptVector(documents: any, query: any, options?: {}): Promise<any>;
79
+ /**
80
+ * Rerank documents's chunks based on relevance to multiple queries,
81
+ * optimizing by embedding documents only once.
82
+ *
83
+ * @param {Array<string>} documents
84
+ * @param {Array<string>} queries
85
+ * @param {Object} [options]
86
+ * @returns {Promise<Object<string, Array<{content: string, similarity: number}>>>}
87
+ * @category Similarity
88
+ */
89
+ export declare function weighRelevanceConceptVectorMultiple(documents: any, queries: any, options?: {}): Promise<{}>;
@@ -0,0 +1,22 @@
1
+ import { KeyphraseEntry, PhrasesModel } from './types';
2
+ /**
3
+ * Applies two weighting passes to a keyphrase list:
4
+ *
5
+ * 1. **Wiki-entity bonus** \u2014 if any token in the phrase is a Wikipedia-linked
6
+ * entity (POS tag 5), the keyphrase weight is doubled and `wiki` is set.
7
+ * 2. **IDF domain-specificity** \u2014 weight is multiplied by the average POS
8
+ * uniqueness score across the phrase's tokens. Rare, domain-specific terms
9
+ * (high uniqueness) receive a larger multiplier than common nouns.
10
+ * 3. **Heavy-query bias** (optional) \u2014 if `heavyWeightQuery` is set, keyphrases
11
+ * that closely match the query words receive a large additive bonus, allowing
12
+ * dynamic re-ranking when a user clicks a term or arrives via a search query.
13
+ *
14
+ * @param keyphrases - Keyphrases to weight; mutated in-place for efficiency.
15
+ * @param phrasesModel - Trie model passed to the tokenizer (may be undefined).
16
+ * @param heavyWeightQuery - Space-separated query string to bias ranking toward.
17
+ * @returns The same array with updated `weight` and `wiki` fields.
18
+ *
19
+ * @example
20
+ * const weighted = weightKeyphrasesBySpecificity(keyphrases, model, "self attention");
21
+ */
22
+ export declare function weightKeyphrasesBySpecificity(keyphrases: KeyphraseEntry[], phrasesModel: PhrasesModel | undefined, heavyWeightQuery: string): KeyphraseEntry[];
File without changes
@@ -0,0 +1,64 @@
1
+ /**
2
+ * Provides search query autocomplete/suggestions from various search engines.
3
+ */
4
+ /**
5
+ * Autocomplete function type
6
+ */
7
+ type AutocompleteFunction = (query: string, locale?: string) => Promise<string[]>;
8
+ /**
9
+ * Baidu autocomplete
10
+ */
11
+ export declare function baidu(query: string, _locale?: string): Promise<string[]>;
12
+ /**
13
+ * Brave autocomplete
14
+ */
15
+ export declare function brave(query: string, _locale?: string): Promise<string[]>;
16
+ /**
17
+ * DuckDuckGo autocomplete
18
+ */
19
+ export declare function duckduckgo(query: string, locale?: string): Promise<string[]>;
20
+ /**
21
+ * Google autocomplete
22
+ */
23
+ export declare function google(query: string, locale?: string): Promise<string[]>;
24
+ /**
25
+ * Qwant autocomplete
26
+ */
27
+ export declare function qwant(query: string, locale?: string): Promise<string[]>;
28
+ /**
29
+ * Startpage autocomplete
30
+ */
31
+ export declare function startpage(query: string, locale?: string): Promise<string[]>;
32
+ /**
33
+ * Wikipedia autocomplete
34
+ */
35
+ export declare function wikipedia(query: string, locale?: string): Promise<string[]>;
36
+ /**
37
+ * Yandex autocomplete
38
+ */
39
+ export declare function yandex(query: string, _locale?: string): Promise<string[]>;
40
+ /**
41
+ * Available autocomplete backends
42
+ */
43
+ export declare const backends: {
44
+ [key: string]: AutocompleteFunction;
45
+ };
46
+ /**
47
+ * Get autocomplete suggestions from a specific backend
48
+ *
49
+ * @param backendName - Name of the autocomplete backend
50
+ * @param query - Search query
51
+ * @param locale - Locale/language code (e.g., 'en-US', 'de-DE')
52
+ * @returns Array of suggestion strings
53
+ */
54
+ export declare function searchAutocomplete(backendName: string, query: string, locale?: string): Promise<string[]>;
55
+ /**
56
+ * Get autocomplete suggestions from multiple backends and merge them
57
+ *
58
+ * @param backendNames - Array of backend names to query
59
+ * @param query - Search query
60
+ * @param locale - Locale/language code
61
+ * @returns Merged and deduplicated array of suggestions
62
+ */
63
+ export declare function searchAutocompleteMulti(backendNames: string[], query: string, locale?: string): Promise<string[]>;
64
+ export {};
@@ -0,0 +1,48 @@
1
+ /**
2
+ * @fileoverview Utility for suggesting word and phrase completions based on a trie model.
3
+ * Used for search autocomplete and real-time query suggestions.
4
+ */
5
+ export interface SuggestCompletionsOptions {
6
+ phrasesModel: Record<string, Record<string, Array<[string | null, number, number]>>>;
7
+ limitMaxResults?: number;
8
+ numberOfLastWordsToCheck?: number;
9
+ optionShowFullQuery?: boolean;
10
+ }
11
+ export interface SuggestionResult {
12
+ name?: string;
13
+ word?: string;
14
+ phrase?: string;
15
+ }
16
+ /**
17
+ * ### Autocomplete Topic Phrase Completions
18
+ * <img width="350px" src="https://i.imgur.com/0k5mO76.png" />
19
+ *
20
+ * Completes the query with the most likely next words for phrases.
21
+ * If typing 2+ letters of a word, returns all possible words matching those few letters.
22
+ *
23
+ *
24
+ * @param {string} query - The input query which can be pertial words or phrases.
25
+ * @param {Object} [options]
26
+ * @param {Object} options.phrasesModel - A custom phrases model to use for autocomplete suggestions.
27
+ * @param {number} options.limitMaxResults default=10 - The maximum number of autocomplete suggestions to return.
28
+ * @param {number} options.numberOfLastWordsToCheck default=5 - The number of last words in the query to check for phrase completions.
29
+ * @returns {Promise<Array<Object>>} An array of autocomplete suggestions, each containing either a 'phrase' or 'word' property.
30
+ * @example
31
+ * // Basic usage
32
+ * const suggestions = await suggestNextWordCompletions("self att");
33
+ * // Possible output: [{ phrase: "self attention" }, { phrase: "self attract" }, { phrase: "self attack" }]
34
+ *
35
+ * @example
36
+ * // Using options
37
+ * const customModel = await import("./custom-phrases-model.json");
38
+ * const suggestions = await suggestNextWordCompletions("artificial int", {
39
+ * phrasesModel: customModel,
40
+ * limitMaxResults: 5,
41
+ * numberOfLastWordsToCheck: 3
42
+ * });
43
+ * // Possible output: [{ phrase: "artificial intelligence" }, { phrase: "artificial interpretation" }]
44
+ *
45
+ * @author [vtempest (2025)](https://github.com/vtempest)
46
+ * @category Topics
47
+ */
48
+ export declare function suggestNextWordCompletions(query: string, options?: Partial<SuggestCompletionsOptions>): Promise<SuggestionResult[] | undefined>;