extract-webpage 1.2.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +212 -0
- package/dist/config/env.d.ts +6 -0
- package/dist/config/index.d.ts +23 -0
- package/dist/config/serverRegistry.d.ts +7 -0
- package/dist/config/types.d.ts +4 -0
- package/dist/extract-webpage.cjs.js +2 -0
- package/dist/extract-webpage.cjs.js.map +1 -0
- package/dist/extract-webpage.es.js +5 -0
- package/dist/extract-webpage.es.js.map +1 -0
- package/dist/html-to-cite/extract-author.d.ts +11 -0
- package/dist/html-to-cite/extract-cite.d.ts +33 -0
- package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
- package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
- package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
- package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
- package/dist/html-to-cite/extract-source.d.ts +7 -0
- package/dist/html-to-cite/extract-title.d.ts +11 -0
- package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
- package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
- package/dist/html-to-cite/url-to-domain.d.ts +20 -0
- package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
- package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
- package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
- package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
- package/dist/html-to-content/html-to-content.d.ts +51 -0
- package/dist/html-to-content/html-utils.d.ts +76 -0
- package/dist/index.d.ts +26 -0
- package/dist/search/index.d.ts +14 -0
- package/dist/search/meta-search-agent-reexport.d.ts +8 -0
- package/dist/search/public-searxng.d.ts +47 -0
- package/dist/search/search-web.d.ts +33 -0
- package/dist/search/tavily.d.ts +20 -0
- package/dist/search/url-to-html.d.ts +62 -0
- package/dist/seektopic/fold-keyphrases.d.ts +28 -0
- package/dist/seektopic/ngrams.d.ts +27 -0
- package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
- package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
- package/dist/seektopic/types.d.ts +86 -0
- package/dist/seektopic/vector-search.d.ts +89 -0
- package/dist/seektopic/weight-keyphrases.d.ts +22 -0
- package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
- package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
- package/dist/tokenize/suggest-complete-word.d.ts +48 -0
- package/dist/tokenize/text-to-chunks.d.ts +48 -0
- package/dist/tokenize/text-to-sentences.d.ts +35 -0
- package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
- package/dist/tokenize/word-is-ignored.d.ts +12 -0
- package/dist/tokenize/word-to-root-stem.d.ts +16 -0
- package/dist/url-to-content/docx-to-content.d.ts +22 -0
- package/dist/url-to-content/is-url-adult.d.ts +26 -0
- package/dist/url-to-content/url-to-content.d.ts +127 -0
- package/dist/url-to-content/url-to-html.d.ts +60 -0
- package/dist/url-to-content/youtube-helpers.d.ts +23 -0
- package/dist/url-to-content/youtube-to-text.d.ts +70 -0
- package/dist/utils/documents.d.ts +4 -0
- package/dist/utils/grab.d.ts +18 -0
- package/package.json +109 -0
- package/src/config/env.ts +8 -0
- package/src/config/index.ts +233 -0
- package/src/config/serverRegistry.ts +24 -0
- package/src/config/types.ts +17 -0
- package/src/fs-mock.js +22 -0
- package/src/global.d.ts +8 -0
- package/src/html-to-cite/extract-author.ts +125 -0
- package/src/html-to-cite/extract-cite.ts +97 -0
- package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
- package/src/html-to-cite/extract-date/date-validators.ts +191 -0
- package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
- package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
- package/src/html-to-cite/extract-source.ts +30 -0
- package/src/html-to-cite/extract-title.ts +78 -0
- package/src/html-to-cite/human-names-92k.json +1 -0
- package/src/html-to-cite/human-names-recognize.ts +396 -0
- package/src/html-to-cite/metadata-to-cite.ts +73 -0
- package/src/html-to-cite/url-to-domain.ts +50 -0
- package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
- package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
- package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
- package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
- package/src/html-to-content/html-to-basic-html.ts +282 -0
- package/src/html-to-content/html-to-content.ts +97 -0
- package/src/html-to-content/html-utils.ts +398 -0
- package/src/index.ts +29 -0
- package/src/search/__tests__/public-searxng.test.ts +529 -0
- package/src/search/index.ts +43 -0
- package/src/search/meta-search-agent-reexport.ts +38 -0
- package/src/search/public-searxng.ts +470 -0
- package/src/search/search-web.ts +668 -0
- package/src/search/tavily.ts +106 -0
- package/src/search/url-to-html.ts +278 -0
- package/src/seektopic/fold-keyphrases.ts +87 -0
- package/src/seektopic/ngrams.ts +64 -0
- package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
- package/src/seektopic/seektopic-keyphrases.ts +279 -0
- package/src/seektopic/types.ts +92 -0
- package/src/seektopic/vector-search.ts +232 -0
- package/src/seektopic/weight-keyphrases.ts +59 -0
- package/src/suggest-next-words/autocomplete-ai.ts +38 -0
- package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
- package/src/tokenize/suggest-complete-word.ts +137 -0
- package/src/tokenize/text-to-chunks.ts +150 -0
- package/src/tokenize/text-to-sentences.ts +614 -0
- package/src/tokenize/text-to-topic-tokens.ts +175 -0
- package/src/tokenize/word-is-ignored.ts +53 -0
- package/src/tokenize/word-to-root-stem.ts +151 -0
- package/src/types.d.ts +130 -0
- package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
- package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
- package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
- package/src/url-to-content/docx-to-content.ts +702 -0
- package/src/url-to-content/is-url-adult.ts +318 -0
- package/src/url-to-content/url-to-content.ts +367 -0
- package/src/url-to-content/url-to-html.ts +436 -0
- package/src/url-to-content/youtube-helpers.ts +64 -0
- package/src/url-to-content/youtube-to-text.ts +468 -0
- package/src/utils/documents.ts +71 -0
- package/src/utils/grab.ts +51 -0
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Search Web via SearXNG metasearch of all major search engines.
|
|
3
|
+
*/
|
|
4
|
+
export declare function searchWeb(query: string, options?: SearchOptions): Promise<SearxngSearchResult[] | SearchResponse>;
|
|
5
|
+
interface SearxngSearchOptions {
|
|
6
|
+
categories?: string[];
|
|
7
|
+
engines?: string[];
|
|
8
|
+
language?: string;
|
|
9
|
+
pageno?: number;
|
|
10
|
+
}
|
|
11
|
+
export declare const searchSearxng: (query: string, opts?: SearxngSearchOptions) => Promise<{
|
|
12
|
+
results: SearxngSearchResult[];
|
|
13
|
+
suggestions: string[];
|
|
14
|
+
}>;
|
|
15
|
+
interface SearchOptions {
|
|
16
|
+
category?: string | number;
|
|
17
|
+
recency?: string;
|
|
18
|
+
privateSearxng?: string | boolean | null;
|
|
19
|
+
maxRetries?: number;
|
|
20
|
+
page?: number;
|
|
21
|
+
safesearch?: boolean;
|
|
22
|
+
lang?: string;
|
|
23
|
+
proxy?: string | null;
|
|
24
|
+
useProxy?: boolean;
|
|
25
|
+
}
|
|
26
|
+
export interface SearxngSearchResult {
|
|
27
|
+
title: string;
|
|
28
|
+
url: string;
|
|
29
|
+
snippet?: string;
|
|
30
|
+
domain?: string;
|
|
31
|
+
favicon?: string;
|
|
32
|
+
score?: number;
|
|
33
|
+
source?: string;
|
|
34
|
+
date?: string;
|
|
35
|
+
img_src?: string;
|
|
36
|
+
thumbnail_src?: string;
|
|
37
|
+
thumbnail?: string;
|
|
38
|
+
content?: string;
|
|
39
|
+
author?: string;
|
|
40
|
+
iframe_src?: string;
|
|
41
|
+
}
|
|
42
|
+
export interface SearchResponse {
|
|
43
|
+
results: SearxngSearchResult[];
|
|
44
|
+
suggestions: string[];
|
|
45
|
+
infoboxes?: any[];
|
|
46
|
+
}
|
|
47
|
+
export {};
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Search Web via SearXNG metasearch of all major search engines.
|
|
3
|
+
* Options are 10 search categories, recency, and how many
|
|
4
|
+
* times to retry other domains if first time fails.
|
|
5
|
+
* SearXNG is a free internet metasearch engine which aggregates results from
|
|
6
|
+
* more than [180+ search sources](https://docs.searxng.org/user/configured_engines.html).
|
|
7
|
+
*
|
|
8
|
+
* [Searxng Overview](https://medium.com/@elmo92/search-in-peace-with-searxng-an-alternative-search-engine-that-keeps-your-searches-private-accd8cddd6fc)
|
|
9
|
+
* [Searxng Installation Guide](https://github.com/searxng/searxng-docker/tree/master)
|
|
10
|
+
*
|
|
11
|
+
* 
|
|
12
|
+
* @param {string} query - The search query string.
|
|
13
|
+
* @param {Object} [options]
|
|
14
|
+
* @param {string} options.category default=general - ["general", "news", "videos", "images",
|
|
15
|
+
* "science","it", "files", "social+media", "map", "music"]
|
|
16
|
+
* @param {string} options.recency default=all - ["all", "day", "week", "month", "year"]
|
|
17
|
+
* @param {string|boolean} options.privateSearxng default=null - Use your custom domain SearXNG
|
|
18
|
+
* @param {number} options.maxRetries default=3 - Maximum number of retry attempts if the initial search fails.
|
|
19
|
+
* @param {number} options.page default=1 - The page number to retrieve.
|
|
20
|
+
* @param {boolean} options.safesearch default=false - Whether to block adult content.
|
|
21
|
+
* @param {string} options.lang default="en-US" - The language to use for the search.
|
|
22
|
+
* @param {string} options.proxy default=false - Use corsproxy.io to access in frontend JS
|
|
23
|
+
* @returns {Promise<Array<{title: string, url: string, snippet: string, domain: string, favicon: string, path: string, engines: string[]}>>} An array of search result objects.
|
|
24
|
+
* @example const advancedResults = await searchWeb('Node.js', {
|
|
25
|
+
* category: 2,
|
|
26
|
+
* recency: 1,
|
|
27
|
+
* maxRetries: 5
|
|
28
|
+
* });
|
|
29
|
+
* @category Search
|
|
30
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
31
|
+
* [Heiser, M., Tauber, A., Flament, A., et al. (2014-)](https://github.com/searxng/searxng/graphs/contributors)
|
|
32
|
+
*/
|
|
33
|
+
export declare function searchWeb(query: any, options?: any): Promise<any>;
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
interface TavilySearchOptions {
|
|
2
|
+
searchDepth?: "basic" | "advanced";
|
|
3
|
+
maxResults?: number;
|
|
4
|
+
includeDomains?: string[];
|
|
5
|
+
excludeDomains?: string[];
|
|
6
|
+
}
|
|
7
|
+
interface TavilySearchResult {
|
|
8
|
+
title: string;
|
|
9
|
+
url: string;
|
|
10
|
+
content: string;
|
|
11
|
+
score: number;
|
|
12
|
+
raw_content?: string;
|
|
13
|
+
}
|
|
14
|
+
export declare const searchTavily: (query: string, opts?: TavilySearchOptions) => Promise<{
|
|
15
|
+
results: TavilySearchResult[];
|
|
16
|
+
suggestions: string[];
|
|
17
|
+
}>;
|
|
18
|
+
export declare const getTavilyApiKey: () => any;
|
|
19
|
+
export declare const isTavilyConfigured: () => boolean;
|
|
20
|
+
export {};
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* ### Tardigrade the Web Crawler
|
|
3
|
+
*
|
|
4
|
+
* 1. **Use Fetch API, check for bot detection.
|
|
5
|
+
* Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
|
|
6
|
+
* Scraping internet pages is a [free speech right
|
|
7
|
+
* ](https://blog.apify.com/is-web-scraping-legal/).
|
|
8
|
+
* 2. Features: timeout, redirects, default UA, referer as google, and bot
|
|
9
|
+
* detection checking. <br />
|
|
10
|
+
* 3. If fetch method does not get needed HTML, use Docker proxy as backup.
|
|
11
|
+
*
|
|
12
|
+
* 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
|
|
13
|
+
* container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
|
|
14
|
+
* secondary in-page API requests after the initial page request, including user login and cookie storage.
|
|
15
|
+
* 5. Bypass Cloudflare bot check: A webpage proxy that request
|
|
16
|
+
* through Chromium (puppeteer) - can be used to bypass Cloudflare
|
|
17
|
+
* anti bot using cookie id javascript method.
|
|
18
|
+
* 6. Send your request to the server with the port 3000 and add your URL to the "url"
|
|
19
|
+
* query string like this: `http://localhost:3000/?url=https://example.org`
|
|
20
|
+
*
|
|
21
|
+
* 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
|
|
22
|
+
* and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
|
|
23
|
+
* [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
|
|
24
|
+
* [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
|
|
25
|
+
* [Proxy-Cheap](https://app.proxy-cheap.com/order)
|
|
26
|
+
* [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
|
|
27
|
+
*
|
|
28
|
+
* @param {string} url - any domain's URL
|
|
29
|
+
* @param {Object} [options]
|
|
30
|
+
* @param {number} options.timeout default=5 - abort request if not retrived, in seconds
|
|
31
|
+
* @param {number} options.maxRedirects default=3 - max redirects to follow
|
|
32
|
+
* @param {number} options.checkBotDetection default=true - check for bot detection messages
|
|
33
|
+
* @param {number} options.changeReferer default=true - set referer as google
|
|
34
|
+
* @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
|
|
35
|
+
* @param {string} options.proxy default=false - use proxy url
|
|
36
|
+
* @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
|
|
37
|
+
* @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
|
|
38
|
+
* @category Extract
|
|
39
|
+
* @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
|
|
40
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
41
|
+
* @license MIT
|
|
42
|
+
*/
|
|
43
|
+
export declare function scrapeURL(url: any, options?: any): Promise<any>;
|
|
44
|
+
/**
|
|
45
|
+
* As backup, scrape with JINA to get html
|
|
46
|
+
* @param {string} url
|
|
47
|
+
* @returns {Promise<string>}
|
|
48
|
+
*/
|
|
49
|
+
export declare function scrapeJINA(url: any, timeout?: number): Promise<string>;
|
|
50
|
+
/**
|
|
51
|
+
* Fetches and parses the robots.txt file for a given URL.
|
|
52
|
+
* @param {string} url - The base URL to fetch the robots.txt from.
|
|
53
|
+
* @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
|
|
54
|
+
*/
|
|
55
|
+
export declare function fetchScrapingRules(url: any): Promise<{
|
|
56
|
+
directives: {};
|
|
57
|
+
crawlDelay: {};
|
|
58
|
+
sitemaps: any[];
|
|
59
|
+
preferredHost: any;
|
|
60
|
+
} | {
|
|
61
|
+
error: string;
|
|
62
|
+
}>;
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import { KeyphraseEntry } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* Folds smaller keyphrases that are near-subsets of larger ones, merging their
|
|
4
|
+
* weights and sentence lists upward.
|
|
5
|
+
*
|
|
6
|
+
* Two phrases are considered overlapping when they share all-but-one word
|
|
7
|
+
* (i.e., the number of words in the smaller phrase that do **not** appear in
|
|
8
|
+
* the larger phrase is fewer than 2). When a merge occurs:
|
|
9
|
+
* - The larger phrase absorbs a fraction of the smaller one's weight
|
|
10
|
+
* (`smallWeight / largerWordCount`).
|
|
11
|
+
* - Sentence indices are union-merged.
|
|
12
|
+
* - If the smaller phrase actually outweighs the larger at that point, the
|
|
13
|
+
* larger entry adopts the smaller phrase's text (best representative wins).
|
|
14
|
+
*
|
|
15
|
+
* Phrases are processed largest-first so that supersets are always evaluated
|
|
16
|
+
* before their constituent sub-phrases.
|
|
17
|
+
*
|
|
18
|
+
* @param keyphrases - Raw scored keyphrases, any order.
|
|
19
|
+
* @returns Deduplicated, folded array \u2014 larger representative phrases only.
|
|
20
|
+
*
|
|
21
|
+
* @example
|
|
22
|
+
* const folded = foldSubphrases([
|
|
23
|
+
* { keyphrase: "machine learning", words: 2, weight: 10, sentences: [0, 1] },
|
|
24
|
+
* { keyphrase: "machine", words: 1, weight: 4, sentences: [0, 2] },
|
|
25
|
+
* ]);
|
|
26
|
+
* // "machine" is absorbed into "machine learning"
|
|
27
|
+
*/
|
|
28
|
+
export declare function foldSubphrases(keyphrases: KeyphraseEntry[]): KeyphraseEntry[];
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
import { NgramMap, TopicToken } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* Extracts noun-anchored edge-grams from a token array and accumulates them into `nGrams`.
|
|
4
|
+
*
|
|
5
|
+
* An edge-gram is a contiguous slice of `nGramSize` tokens where:
|
|
6
|
+
* - The **first** and **last** tokens are topic entities (noun or wiki title).
|
|
7
|
+
* - Every token is either a topic entity or a common stop word (e.g. "of", "the").
|
|
8
|
+
* - Every token meets the `minWordLength` character threshold.
|
|
9
|
+
*
|
|
10
|
+
* This allows natural multi-word keyphrases like "state of the art" or
|
|
11
|
+
* "machine learning" while ignoring pure function-word sequences.
|
|
12
|
+
*
|
|
13
|
+
* @param nGramSize - Number of tokens in the slice to evaluate.
|
|
14
|
+
* @param terms - Full token array for the current sentence.
|
|
15
|
+
* @param index - Start position for this slice within `terms`.
|
|
16
|
+
* @param nGrams - Accumulator map mutated in-place: `nGrams[size][phrase] = [sentenceIdx, ...]`.
|
|
17
|
+
* @param minWordLength - Minimum character length for any word to be included.
|
|
18
|
+
* @param sentenceIndex - Index of the originating sentence, appended to the phrase's entry.
|
|
19
|
+
* @returns The same `nGrams` reference (mutated).
|
|
20
|
+
*
|
|
21
|
+
* @example
|
|
22
|
+
* const terms: TopicToken[] = [["machine", 1, 4, ""], ["learning", 1, 5, ""]];
|
|
23
|
+
* const nGrams: NgramMap = {};
|
|
24
|
+
* extractNounEdgeGrams(2, terms, 0, nGrams, 3, 0);
|
|
25
|
+
* // nGrams[2]["machine learning"] === [0]
|
|
26
|
+
*/
|
|
27
|
+
export declare function extractNounEdgeGrams(nGramSize: number, terms: TopicToken[], index: number, nGrams: NgramMap, minWordLength: number, sentenceIndex: number): NgramMap;
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
import { SentenceEntry, RankOptions } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* ### TextRank: Rank Sentences by Centrality to Shared Keyphrases
|
|
4
|
+
*
|
|
5
|
+
* Builds a weighted undirected graph where each node is a sentence and edges
|
|
6
|
+
* connect sentences that share keyphrases. Sentence importance is estimated by
|
|
7
|
+
* a random-walk simulation (analogous to PageRank): nodes visited more often
|
|
8
|
+
* during the walk are considered more central to the document's key concepts.
|
|
9
|
+
*
|
|
10
|
+
* **Key optimisations vs. na\u00efve implementation:**
|
|
11
|
+
* - Weighted sampling via a single O(degree) scan instead of an O(weight) array.
|
|
12
|
+
* - `Set`-based keyphrase intersection (O(1) lookup) instead of `Array.includes`.
|
|
13
|
+
* - `Map<text, index>` for O(1) weight increment instead of O(n) linear scan.
|
|
14
|
+
* - Flat `Map<string, Map<string, number>>` graph \u2014 no object-method overhead.
|
|
15
|
+
* - Periodic forced reset prevents the walk from getting stuck in dense clusters.
|
|
16
|
+
*
|
|
17
|
+
* **References:**
|
|
18
|
+
* 1. Zhao & Xie (2021) \u2014 "An Improved TextRank Multi-feature Fusion Algorithm"
|
|
19
|
+
* https://iopscience.iop.org/article/10.1088/1742-6596/2078/1/012021/pdf
|
|
20
|
+
* 2. Pan et al. (2019) \u2014 "An improved TextRank keywords extraction algorithm"
|
|
21
|
+
* https://dl.acm.org/doi/10.1145/3321408.3326659
|
|
22
|
+
*
|
|
23
|
+
* @param sentencesWithKeyphrases - Sentences with pre-attached keyphrase lists.
|
|
24
|
+
* @param options - Walk parameters.
|
|
25
|
+
* @returns The same array with `weight` set on each sentence, or `undefined`
|
|
26
|
+
* if no edges exist (no shared keyphrases between any pair).
|
|
27
|
+
*/
|
|
28
|
+
export declare function rankSentencesCentralToKeyphrase(sentencesWithKeyphrases: SentenceEntry[], options?: RankOptions): SentenceEntry[] | undefined;
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { KeyphraseEntry, SEEKTOPICOptions, SEEKTOPICResult } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* ### SEEKTOPIC \u2014 Keyphrase & Sentence Extraction
|
|
4
|
+
*
|
|
5
|
+
* Pulls the most important phrases and sentences out of any document.
|
|
6
|
+
* Given raw text, it returns a ranked list of key concepts (e.g. "neural
|
|
7
|
+
* network", "climate change") and, optionally, the sentences that best
|
|
8
|
+
* summarise the document around those concepts.
|
|
9
|
+
*
|
|
10
|
+
* <img src="https://i.imgur.com/gZ4kI1V.png" width="360px" />
|
|
11
|
+
*
|
|
12
|
+
* **How it works \u2014 8-step pipeline:**
|
|
13
|
+
*
|
|
14
|
+
* 1. **Clean** \u2014 strip HTML tags and entities.
|
|
15
|
+
* 2. **Split** \u2014 break text into sentences, respecting abbreviations and URLs.
|
|
16
|
+
* 3. **Topic Extraction** *(LLM Path)* \u2014 sends the first 5000 words to an LLM
|
|
17
|
+
* to extract the most descriptive key topic phrases labeling the document.
|
|
18
|
+
* 4. **Vector Search** *(LLM Path)* \u2014 calculates cosine similarity embeddings
|
|
19
|
+
* for all sentences vs topic phrases, finding the top relevant sentences per topic.
|
|
20
|
+
* 5. **Tokenise & Extract** *(Fallback)* \u2014 if LLM fails, identify noun-anchored
|
|
21
|
+
* n-grams (1-N words).
|
|
22
|
+
* 6. **Score & Fold** *(Fallback)* \u2014 weight phrases and merge subsumed phrases.
|
|
23
|
+
* 7. **Re-weight** *(Fallback)* \u2014 boost Wikipedia entities and rare domain terms.
|
|
24
|
+
* 8. **TextRank** *(Fallback)* \u2014 build sentence similarity graph and run random walk.
|
|
25
|
+
*
|
|
26
|
+
* <video src="https://github.com/user-attachments/assets/73348d63-7671-4e20-8df9-29a13d5b0768"
|
|
27
|
+
* width="550px" controls />
|
|
28
|
+
*
|
|
29
|
+
* @param docText - Plain text or HTML document to analyse.
|
|
30
|
+
* @param options - Optional tuning parameters (word limits, thresholds, query bias).
|
|
31
|
+
* @returns
|
|
32
|
+
* - `optionSkipRanking: true` (default) \u2192 `KeyphraseEntry[]` sorted by weight.
|
|
33
|
+
* - `optionSkipRanking: false` \u2192 `SEEKTOPICResult` with `topSentences`,
|
|
34
|
+
* `keyphrases`, and the full `sentences` array.
|
|
35
|
+
*
|
|
36
|
+
* @example
|
|
37
|
+
* // Fast: just extract keyphrases
|
|
38
|
+
* const keyphrases = extractSEEKTOPIC(articleText, { phrasesModel });
|
|
39
|
+
* // \u2192 [{ keyphrase: "machine learning", weight: 84 }, ...]
|
|
40
|
+
*
|
|
41
|
+
* @example
|
|
42
|
+
* // Full: keyphrases + summary sentences, biased toward a search query
|
|
43
|
+
* const { topSentences, keyphrases } = extractSEEKTOPIC(articleText, {
|
|
44
|
+
* phrasesModel,
|
|
45
|
+
* optionSkipRanking: false,
|
|
46
|
+
* heavyWeightQuery: "transformer attention",
|
|
47
|
+
* limitTopSentences: 5,
|
|
48
|
+
* }) as SEEKTOPICResult;
|
|
49
|
+
*
|
|
50
|
+
* @author [ai-research-agent (2024)](https://airesearch.js.org)
|
|
51
|
+
* @category Topics
|
|
52
|
+
*/
|
|
53
|
+
export declare function extractSEEKTOPIC(docText: string, options?: SEEKTOPICOptions): Promise<KeyphraseEntry[] | SEEKTOPICResult>;
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
import { TopicToken, PhrasesModel } from '../tokenize/text-to-topic-tokens';
|
|
2
|
+
export type { TopicToken, PhrasesModel };
|
|
3
|
+
/** Map from n-gram size \u2192 { phrase text \u2192 sentence indices } */
|
|
4
|
+
export type NgramMap = Record<number, Record<string, number[]>>;
|
|
5
|
+
/**
|
|
6
|
+
* A ranked keyphrase with its scoring metadata and the sentence indices it appears in.
|
|
7
|
+
*/
|
|
8
|
+
export interface KeyphraseEntry {
|
|
9
|
+
/** The keyphrase text */
|
|
10
|
+
keyphrase: string;
|
|
11
|
+
/** Indices of sentences containing this keyphrase (or comma-separated string form) */
|
|
12
|
+
sentences: number[] | string;
|
|
13
|
+
/** Sentences representing this keyphrase directly with their relevance similarity */
|
|
14
|
+
topSentences?: Array<{
|
|
15
|
+
text: string;
|
|
16
|
+
similarity: number;
|
|
17
|
+
}>;
|
|
18
|
+
/** Number of words in the keyphrase */
|
|
19
|
+
words: number;
|
|
20
|
+
/** Composite ranking weight */
|
|
21
|
+
weight: number;
|
|
22
|
+
/** True if the phrase is a Wikipedia-linked entity */
|
|
23
|
+
wiki?: boolean;
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* A sentence node used during graph construction and TextRank scoring.
|
|
27
|
+
*/
|
|
28
|
+
export interface SentenceEntry {
|
|
29
|
+
text: string;
|
|
30
|
+
index: number;
|
|
31
|
+
keyphrases: Array<{
|
|
32
|
+
keyphrase: string;
|
|
33
|
+
weight: number;
|
|
34
|
+
}>;
|
|
35
|
+
weight: number;
|
|
36
|
+
}
|
|
37
|
+
/**
|
|
38
|
+
* A sentence in the final SEEKTOPIC output \u2014 keyphrases are resolved to strings.
|
|
39
|
+
*/
|
|
40
|
+
export interface SentenceResult {
|
|
41
|
+
text: string;
|
|
42
|
+
index: number;
|
|
43
|
+
keyphrases: string[];
|
|
44
|
+
weight: number;
|
|
45
|
+
}
|
|
46
|
+
/** Options for {@link extractSEEKTOPIC} */
|
|
47
|
+
export interface SEEKTOPICOptions {
|
|
48
|
+
/** Trie phrases model for wiki-phrase tokenization */
|
|
49
|
+
phrasesModel?: PhrasesModel;
|
|
50
|
+
/** Maximum words per keyphrase (default 2) */
|
|
51
|
+
maxWords?: number;
|
|
52
|
+
/** Minimum words per keyphrase (default 1) */
|
|
53
|
+
minWords?: number;
|
|
54
|
+
/** Minimum character length of any word in a keyphrase (default 3) */
|
|
55
|
+
minWordLength?: number;
|
|
56
|
+
/** Fraction of all keyphrases to use as the TextRank graph limit (default 0.5) */
|
|
57
|
+
topKeyphrasesPercent?: number;
|
|
58
|
+
/** Max sentences to return in full-ranking mode (default 5) */
|
|
59
|
+
limitTopSentences?: number;
|
|
60
|
+
/** Max keyphrases to return (default 10) */
|
|
61
|
+
limitTopKeyphrases?: number;
|
|
62
|
+
/** Minimum character length of the full keyphrase string (default 5) */
|
|
63
|
+
minKeyPhraseLength?: number;
|
|
64
|
+
/** Query string to bias keyphrase weights toward (default "") */
|
|
65
|
+
heavyWeightQuery?: string;
|
|
66
|
+
/** Strip HTML tags before processing (default true) */
|
|
67
|
+
removeHTML?: boolean;
|
|
68
|
+
/** Return only keyphrases without running TextRank (default true) */
|
|
69
|
+
optionSkipRanking?: boolean;
|
|
70
|
+
getEnv?: (key: string) => string | undefined;
|
|
71
|
+
}
|
|
72
|
+
/** Full SEEKTOPIC output when `optionSkipRanking` is false */
|
|
73
|
+
export interface SEEKTOPICResult {
|
|
74
|
+
topSentences: SentenceResult[];
|
|
75
|
+
keyphrases: Array<Omit<KeyphraseEntry, "sentences"> & {
|
|
76
|
+
sentences: string;
|
|
77
|
+
}>;
|
|
78
|
+
sentences: string[];
|
|
79
|
+
}
|
|
80
|
+
/** Options for {@link rankSentencesCentralToKeyphrase} */
|
|
81
|
+
export interface RankOptions {
|
|
82
|
+
/** Number of random-walk steps (default 1000) */
|
|
83
|
+
iterations?: number;
|
|
84
|
+
/** Steps between forced random resets to avoid cluster traps (default 100) */
|
|
85
|
+
resetInterval?: number;
|
|
86
|
+
}
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Text embeddings convert words or phrases into numerical vectors in a high-dimensional
|
|
3
|
+
* space, where each dimension represents a semantic feature extracted by a model like
|
|
4
|
+
* MiniLM-L6-v2. In this concept space, words with similar meanings have vectors that
|
|
5
|
+
* are close together, allowing for quantitative comparisons of semantic similarity.
|
|
6
|
+
* These vector representations enable powerful applications in natural language processing,
|
|
7
|
+
* including semantic search, text classification, and clustering, by leveraging the
|
|
8
|
+
* geometric properties of the embedding space to capture and analyze the relationships
|
|
9
|
+
* between words and concepts.
|
|
10
|
+
* [Text Embeddings, Classification, and Semantic Search
|
|
11
|
+
* (Youtube)](https://www.youtube.com/watch?v=sNa_uiqSlJo&t=129s)
|
|
12
|
+
*
|
|
13
|
+
* <img src="https://i.imgur.com/wtJqEqX.png" width="350" />
|
|
14
|
+
* @param {string} text - The text to embed.
|
|
15
|
+
* @param {Object} [options]
|
|
16
|
+
* @param {AutoTokenizer} options.pipeline
|
|
17
|
+
* - The pipeline to use for embedding.
|
|
18
|
+
* @param {number} options.precision default=4 - The number of decimal places to round to.
|
|
19
|
+
* @returns {Promise<{embeddingsDict: Object.<string, number[]>, embedding: number[]}>}
|
|
20
|
+
* @category Similarity
|
|
21
|
+
*/
|
|
22
|
+
export declare function convertTextToEmbedding(text: string | string[], options?: any): Promise<any>;
|
|
23
|
+
/**
|
|
24
|
+
* Calculate the semantic similarity between one text and a list of
|
|
25
|
+
* other sentences by comparing their embeddings.
|
|
26
|
+
* https://huggingface.co/docs/api-inference/detailed_parameters#sentence-similarity-task
|
|
27
|
+
*
|
|
28
|
+
* <img src="https://i.imgur.com/ex2UWnu.png" width="350px" />
|
|
29
|
+
* @param {string} source_sentence The string that you wish to
|
|
30
|
+
* compare the other strings with. This can be a phrase, sentence,
|
|
31
|
+
* or longer passage, depending on the model being used.
|
|
32
|
+
* @param {Array<string>} sentences A list of strings which will be compared
|
|
33
|
+
* against the source_sentence.
|
|
34
|
+
* @param {Object} [options]
|
|
35
|
+
* @param {string} options.model default="sentence-transformers/all-MiniLM-L6-v2"
|
|
36
|
+
* @param {string} options.HF_API_KEY Required https://huggingface.co/settings/tokens
|
|
37
|
+
* @returns array of 0-1 similarity scores for each sentence
|
|
38
|
+
* @category Similarity
|
|
39
|
+
*/
|
|
40
|
+
export declare function weighRelevanceConceptVectorAPI(source_sentence: any, sentences: any, options?: {}): Promise<string | {
|
|
41
|
+
error: string;
|
|
42
|
+
}>;
|
|
43
|
+
/**
|
|
44
|
+
* Initialize HuggingFace Transformers pipeline for embedding text.
|
|
45
|
+
*
|
|
46
|
+
* <img src="https://i.imgur.com/3R5Tsrf.png" width="350px" />
|
|
47
|
+
* @param {Object} [options]
|
|
48
|
+
* @param {string} options.pipelineName default "feature-extraction",
|
|
49
|
+
* @param {string} options.modelName default="Xenova/all-MiniLM-L6-v2" -
|
|
50
|
+
* The name of the model to use
|
|
51
|
+
* @returns {Promise<import("@huggingface/transformers").AutoTokenizer>} The pipeline.
|
|
52
|
+
* @category Similarity
|
|
53
|
+
*/
|
|
54
|
+
export declare function getEmbeddingModel(options?: any): Promise<any>;
|
|
55
|
+
/**
|
|
56
|
+
* [Cosine similarity](https://en.wikipedia.org/wiki/Cosine_similarity) gets similarity of two
|
|
57
|
+
* vectors by whether they have the same direction (similar) or are poles apart. Cosine similarity
|
|
58
|
+
* is often used with text representations to compare how similar two documents or sentences
|
|
59
|
+
* are to each other. The output of cosine similarity ranges from -1 to 1, where -1 means the
|
|
60
|
+
* two vectors are completely dissimilar, and 1 indicates maximum similarity.
|
|
61
|
+
* @param {Array<number>} vectorA
|
|
62
|
+
* @param {Array<number>} vectorB
|
|
63
|
+
* @returns {number} -1 to 1 similarity score
|
|
64
|
+
*/
|
|
65
|
+
export declare function calculateCosineSimilarity(vectorA: any, vectorB: any): number;
|
|
66
|
+
/**
|
|
67
|
+
* Rerank documents's chunks based on relevance to query,
|
|
68
|
+
* based on cosine similarity of their concept vectors generated
|
|
69
|
+
* by a 20MB MiniLM transformer model downloaded locally.
|
|
70
|
+
*
|
|
71
|
+
* [A Complete Overview of Word Embeddings](https://www.youtube.com/watch?v=5MaWmXwxFNQ&t=323s)
|
|
72
|
+
* @param {Array<string>} documents
|
|
73
|
+
* @param {string} query
|
|
74
|
+
* @param {Object} [options]
|
|
75
|
+
* @returns {Promise<Array<{content: string, similarity: number}>>}
|
|
76
|
+
* @category Similarity
|
|
77
|
+
*/
|
|
78
|
+
export declare function weighRelevanceConceptVector(documents: any, query: any, options?: {}): Promise<any>;
|
|
79
|
+
/**
|
|
80
|
+
* Rerank documents's chunks based on relevance to multiple queries,
|
|
81
|
+
* optimizing by embedding documents only once.
|
|
82
|
+
*
|
|
83
|
+
* @param {Array<string>} documents
|
|
84
|
+
* @param {Array<string>} queries
|
|
85
|
+
* @param {Object} [options]
|
|
86
|
+
* @returns {Promise<Object<string, Array<{content: string, similarity: number}>>>}
|
|
87
|
+
* @category Similarity
|
|
88
|
+
*/
|
|
89
|
+
export declare function weighRelevanceConceptVectorMultiple(documents: any, queries: any, options?: {}): Promise<{}>;
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { KeyphraseEntry, PhrasesModel } from './types';
|
|
2
|
+
/**
|
|
3
|
+
* Applies two weighting passes to a keyphrase list:
|
|
4
|
+
*
|
|
5
|
+
* 1. **Wiki-entity bonus** \u2014 if any token in the phrase is a Wikipedia-linked
|
|
6
|
+
* entity (POS tag 5), the keyphrase weight is doubled and `wiki` is set.
|
|
7
|
+
* 2. **IDF domain-specificity** \u2014 weight is multiplied by the average POS
|
|
8
|
+
* uniqueness score across the phrase's tokens. Rare, domain-specific terms
|
|
9
|
+
* (high uniqueness) receive a larger multiplier than common nouns.
|
|
10
|
+
* 3. **Heavy-query bias** (optional) \u2014 if `heavyWeightQuery` is set, keyphrases
|
|
11
|
+
* that closely match the query words receive a large additive bonus, allowing
|
|
12
|
+
* dynamic re-ranking when a user clicks a term or arrives via a search query.
|
|
13
|
+
*
|
|
14
|
+
* @param keyphrases - Keyphrases to weight; mutated in-place for efficiency.
|
|
15
|
+
* @param phrasesModel - Trie model passed to the tokenizer (may be undefined).
|
|
16
|
+
* @param heavyWeightQuery - Space-separated query string to bias ranking toward.
|
|
17
|
+
* @returns The same array with updated `weight` and `wiki` fields.
|
|
18
|
+
*
|
|
19
|
+
* @example
|
|
20
|
+
* const weighted = weightKeyphrasesBySpecificity(keyphrases, model, "self attention");
|
|
21
|
+
*/
|
|
22
|
+
export declare function weightKeyphrasesBySpecificity(keyphrases: KeyphraseEntry[], phrasesModel: PhrasesModel | undefined, heavyWeightQuery: string): KeyphraseEntry[];
|
|
File without changes
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Provides search query autocomplete/suggestions from various search engines.
|
|
3
|
+
*/
|
|
4
|
+
/**
|
|
5
|
+
* Autocomplete function type
|
|
6
|
+
*/
|
|
7
|
+
type AutocompleteFunction = (query: string, locale?: string) => Promise<string[]>;
|
|
8
|
+
/**
|
|
9
|
+
* Baidu autocomplete
|
|
10
|
+
*/
|
|
11
|
+
export declare function baidu(query: string, _locale?: string): Promise<string[]>;
|
|
12
|
+
/**
|
|
13
|
+
* Brave autocomplete
|
|
14
|
+
*/
|
|
15
|
+
export declare function brave(query: string, _locale?: string): Promise<string[]>;
|
|
16
|
+
/**
|
|
17
|
+
* DuckDuckGo autocomplete
|
|
18
|
+
*/
|
|
19
|
+
export declare function duckduckgo(query: string, locale?: string): Promise<string[]>;
|
|
20
|
+
/**
|
|
21
|
+
* Google autocomplete
|
|
22
|
+
*/
|
|
23
|
+
export declare function google(query: string, locale?: string): Promise<string[]>;
|
|
24
|
+
/**
|
|
25
|
+
* Qwant autocomplete
|
|
26
|
+
*/
|
|
27
|
+
export declare function qwant(query: string, locale?: string): Promise<string[]>;
|
|
28
|
+
/**
|
|
29
|
+
* Startpage autocomplete
|
|
30
|
+
*/
|
|
31
|
+
export declare function startpage(query: string, locale?: string): Promise<string[]>;
|
|
32
|
+
/**
|
|
33
|
+
* Wikipedia autocomplete
|
|
34
|
+
*/
|
|
35
|
+
export declare function wikipedia(query: string, locale?: string): Promise<string[]>;
|
|
36
|
+
/**
|
|
37
|
+
* Yandex autocomplete
|
|
38
|
+
*/
|
|
39
|
+
export declare function yandex(query: string, _locale?: string): Promise<string[]>;
|
|
40
|
+
/**
|
|
41
|
+
* Available autocomplete backends
|
|
42
|
+
*/
|
|
43
|
+
export declare const backends: {
|
|
44
|
+
[key: string]: AutocompleteFunction;
|
|
45
|
+
};
|
|
46
|
+
/**
|
|
47
|
+
* Get autocomplete suggestions from a specific backend
|
|
48
|
+
*
|
|
49
|
+
* @param backendName - Name of the autocomplete backend
|
|
50
|
+
* @param query - Search query
|
|
51
|
+
* @param locale - Locale/language code (e.g., 'en-US', 'de-DE')
|
|
52
|
+
* @returns Array of suggestion strings
|
|
53
|
+
*/
|
|
54
|
+
export declare function searchAutocomplete(backendName: string, query: string, locale?: string): Promise<string[]>;
|
|
55
|
+
/**
|
|
56
|
+
* Get autocomplete suggestions from multiple backends and merge them
|
|
57
|
+
*
|
|
58
|
+
* @param backendNames - Array of backend names to query
|
|
59
|
+
* @param query - Search query
|
|
60
|
+
* @param locale - Locale/language code
|
|
61
|
+
* @returns Merged and deduplicated array of suggestions
|
|
62
|
+
*/
|
|
63
|
+
export declare function searchAutocompleteMulti(backendNames: string[], query: string, locale?: string): Promise<string[]>;
|
|
64
|
+
export {};
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* @fileoverview Utility for suggesting word and phrase completions based on a trie model.
|
|
3
|
+
* Used for search autocomplete and real-time query suggestions.
|
|
4
|
+
*/
|
|
5
|
+
export interface SuggestCompletionsOptions {
|
|
6
|
+
phrasesModel: Record<string, Record<string, Array<[string | null, number, number]>>>;
|
|
7
|
+
limitMaxResults?: number;
|
|
8
|
+
numberOfLastWordsToCheck?: number;
|
|
9
|
+
optionShowFullQuery?: boolean;
|
|
10
|
+
}
|
|
11
|
+
export interface SuggestionResult {
|
|
12
|
+
name?: string;
|
|
13
|
+
word?: string;
|
|
14
|
+
phrase?: string;
|
|
15
|
+
}
|
|
16
|
+
/**
|
|
17
|
+
* ### Autocomplete Topic Phrase Completions
|
|
18
|
+
* <img width="350px" src="https://i.imgur.com/0k5mO76.png" />
|
|
19
|
+
*
|
|
20
|
+
* Completes the query with the most likely next words for phrases.
|
|
21
|
+
* If typing 2+ letters of a word, returns all possible words matching those few letters.
|
|
22
|
+
*
|
|
23
|
+
*
|
|
24
|
+
* @param {string} query - The input query which can be pertial words or phrases.
|
|
25
|
+
* @param {Object} [options]
|
|
26
|
+
* @param {Object} options.phrasesModel - A custom phrases model to use for autocomplete suggestions.
|
|
27
|
+
* @param {number} options.limitMaxResults default=10 - The maximum number of autocomplete suggestions to return.
|
|
28
|
+
* @param {number} options.numberOfLastWordsToCheck default=5 - The number of last words in the query to check for phrase completions.
|
|
29
|
+
* @returns {Promise<Array<Object>>} An array of autocomplete suggestions, each containing either a 'phrase' or 'word' property.
|
|
30
|
+
* @example
|
|
31
|
+
* // Basic usage
|
|
32
|
+
* const suggestions = await suggestNextWordCompletions("self att");
|
|
33
|
+
* // Possible output: [{ phrase: "self attention" }, { phrase: "self attract" }, { phrase: "self attack" }]
|
|
34
|
+
*
|
|
35
|
+
* @example
|
|
36
|
+
* // Using options
|
|
37
|
+
* const customModel = await import("./custom-phrases-model.json");
|
|
38
|
+
* const suggestions = await suggestNextWordCompletions("artificial int", {
|
|
39
|
+
* phrasesModel: customModel,
|
|
40
|
+
* limitMaxResults: 5,
|
|
41
|
+
* numberOfLastWordsToCheck: 3
|
|
42
|
+
* });
|
|
43
|
+
* // Possible output: [{ phrase: "artificial intelligence" }, { phrase: "artificial interpretation" }]
|
|
44
|
+
*
|
|
45
|
+
* @author [vtempest (2025)](https://github.com/vtempest)
|
|
46
|
+
* @category Topics
|
|
47
|
+
*/
|
|
48
|
+
export declare function suggestNextWordCompletions(query: string, options?: Partial<SuggestCompletionsOptions>): Promise<SuggestionResult[] | undefined>;
|