extract-webpage 1.2.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (117) hide show
  1. package/README.md +212 -0
  2. package/dist/config/env.d.ts +6 -0
  3. package/dist/config/index.d.ts +23 -0
  4. package/dist/config/serverRegistry.d.ts +7 -0
  5. package/dist/config/types.d.ts +4 -0
  6. package/dist/extract-webpage.cjs.js +2 -0
  7. package/dist/extract-webpage.cjs.js.map +1 -0
  8. package/dist/extract-webpage.es.js +5 -0
  9. package/dist/extract-webpage.es.js.map +1 -0
  10. package/dist/html-to-cite/extract-author.d.ts +11 -0
  11. package/dist/html-to-cite/extract-cite.d.ts +33 -0
  12. package/dist/html-to-cite/extract-date/date-extractors.d.ts +40 -0
  13. package/dist/html-to-cite/extract-date/date-validators.d.ts +15 -0
  14. package/dist/html-to-cite/extract-date/extract-date-quick.d.ts +8 -0
  15. package/dist/html-to-cite/extract-date/extract-date.d.ts +26 -0
  16. package/dist/html-to-cite/extract-source.d.ts +7 -0
  17. package/dist/html-to-cite/extract-title.d.ts +11 -0
  18. package/dist/html-to-cite/human-names-recognize.d.ts +16 -0
  19. package/dist/html-to-cite/metadata-to-cite.d.ts +12 -0
  20. package/dist/html-to-cite/url-to-domain.d.ts +20 -0
  21. package/dist/html-to-content/extract-content/extract-content-mercury-utils.d.ts +27 -0
  22. package/dist/html-to-content/extract-content/extract-content-mercury.d.ts +61 -0
  23. package/dist/html-to-content/extract-content/extract-content-readability.d.ts +101 -0
  24. package/dist/html-to-content/html-to-basic-html.d.ts +36 -0
  25. package/dist/html-to-content/html-to-content.d.ts +51 -0
  26. package/dist/html-to-content/html-utils.d.ts +76 -0
  27. package/dist/index.d.ts +26 -0
  28. package/dist/search/index.d.ts +14 -0
  29. package/dist/search/meta-search-agent-reexport.d.ts +8 -0
  30. package/dist/search/public-searxng.d.ts +47 -0
  31. package/dist/search/search-web.d.ts +33 -0
  32. package/dist/search/tavily.d.ts +20 -0
  33. package/dist/search/url-to-html.d.ts +62 -0
  34. package/dist/seektopic/fold-keyphrases.d.ts +28 -0
  35. package/dist/seektopic/ngrams.d.ts +27 -0
  36. package/dist/seektopic/rank-sentences-keyphrases.d.ts +28 -0
  37. package/dist/seektopic/seektopic-keyphrases.d.ts +53 -0
  38. package/dist/seektopic/types.d.ts +86 -0
  39. package/dist/seektopic/vector-search.d.ts +89 -0
  40. package/dist/seektopic/weight-keyphrases.d.ts +22 -0
  41. package/dist/suggest-next-words/autocomplete-ai.d.ts +0 -0
  42. package/dist/suggest-next-words/autocomplete-search-engines.d.ts +64 -0
  43. package/dist/tokenize/suggest-complete-word.d.ts +48 -0
  44. package/dist/tokenize/text-to-chunks.d.ts +48 -0
  45. package/dist/tokenize/text-to-sentences.d.ts +35 -0
  46. package/dist/tokenize/text-to-topic-tokens.d.ts +51 -0
  47. package/dist/tokenize/word-is-ignored.d.ts +12 -0
  48. package/dist/tokenize/word-to-root-stem.d.ts +16 -0
  49. package/dist/url-to-content/docx-to-content.d.ts +22 -0
  50. package/dist/url-to-content/is-url-adult.d.ts +26 -0
  51. package/dist/url-to-content/url-to-content.d.ts +127 -0
  52. package/dist/url-to-content/url-to-html.d.ts +60 -0
  53. package/dist/url-to-content/youtube-helpers.d.ts +23 -0
  54. package/dist/url-to-content/youtube-to-text.d.ts +70 -0
  55. package/dist/utils/documents.d.ts +4 -0
  56. package/dist/utils/grab.d.ts +18 -0
  57. package/package.json +109 -0
  58. package/src/config/env.ts +8 -0
  59. package/src/config/index.ts +233 -0
  60. package/src/config/serverRegistry.ts +24 -0
  61. package/src/config/types.ts +17 -0
  62. package/src/fs-mock.js +22 -0
  63. package/src/global.d.ts +8 -0
  64. package/src/html-to-cite/extract-author.ts +125 -0
  65. package/src/html-to-cite/extract-cite.ts +97 -0
  66. package/src/html-to-cite/extract-date/date-extractors.ts +484 -0
  67. package/src/html-to-cite/extract-date/date-validators.ts +191 -0
  68. package/src/html-to-cite/extract-date/extract-date-quick.ts +184 -0
  69. package/src/html-to-cite/extract-date/extract-date.ts +1049 -0
  70. package/src/html-to-cite/extract-source.ts +30 -0
  71. package/src/html-to-cite/extract-title.ts +78 -0
  72. package/src/html-to-cite/human-names-92k.json +1 -0
  73. package/src/html-to-cite/human-names-recognize.ts +396 -0
  74. package/src/html-to-cite/metadata-to-cite.ts +73 -0
  75. package/src/html-to-cite/url-to-domain.ts +50 -0
  76. package/src/html-to-content/extract-content/extract-content-mercury-utils.ts +696 -0
  77. package/src/html-to-content/extract-content/extract-content-mercury.ts +830 -0
  78. package/src/html-to-content/extract-content/extract-content-readability.ts +432 -0
  79. package/src/html-to-content/extract-content/extract-selectors-per-domain.json +3453 -0
  80. package/src/html-to-content/html-to-basic-html.ts +282 -0
  81. package/src/html-to-content/html-to-content.ts +97 -0
  82. package/src/html-to-content/html-utils.ts +398 -0
  83. package/src/index.ts +29 -0
  84. package/src/search/__tests__/public-searxng.test.ts +529 -0
  85. package/src/search/index.ts +43 -0
  86. package/src/search/meta-search-agent-reexport.ts +38 -0
  87. package/src/search/public-searxng.ts +470 -0
  88. package/src/search/search-web.ts +668 -0
  89. package/src/search/tavily.ts +106 -0
  90. package/src/search/url-to-html.ts +278 -0
  91. package/src/seektopic/fold-keyphrases.ts +87 -0
  92. package/src/seektopic/ngrams.ts +64 -0
  93. package/src/seektopic/rank-sentences-keyphrases.ts +132 -0
  94. package/src/seektopic/seektopic-keyphrases.ts +279 -0
  95. package/src/seektopic/types.ts +92 -0
  96. package/src/seektopic/vector-search.ts +232 -0
  97. package/src/seektopic/weight-keyphrases.ts +59 -0
  98. package/src/suggest-next-words/autocomplete-ai.ts +38 -0
  99. package/src/suggest-next-words/autocomplete-search-engines.ts +435 -0
  100. package/src/tokenize/suggest-complete-word.ts +137 -0
  101. package/src/tokenize/text-to-chunks.ts +150 -0
  102. package/src/tokenize/text-to-sentences.ts +614 -0
  103. package/src/tokenize/text-to-topic-tokens.ts +175 -0
  104. package/src/tokenize/word-is-ignored.ts +53 -0
  105. package/src/tokenize/word-to-root-stem.ts +151 -0
  106. package/src/types.d.ts +130 -0
  107. package/src/url-to-content/.fuse_hidden003bd28a0000000d +332 -0
  108. package/src/url-to-content/__tests__/url-to-content.test.ts +368 -0
  109. package/src/url-to-content/__tests__/url-to-html.test.ts +301 -0
  110. package/src/url-to-content/docx-to-content.ts +702 -0
  111. package/src/url-to-content/is-url-adult.ts +318 -0
  112. package/src/url-to-content/url-to-content.ts +367 -0
  113. package/src/url-to-content/url-to-html.ts +436 -0
  114. package/src/url-to-content/youtube-helpers.ts +64 -0
  115. package/src/url-to-content/youtube-to-text.ts +468 -0
  116. package/src/utils/documents.ts +71 -0
  117. package/src/utils/grab.ts +51 -0
@@ -0,0 +1,106 @@
1
+ /**
2
+ * @module research/search/tavily
3
+ * @description Research library module.
4
+ */
5
+ import axios from "axios";
6
+ import configManager from "../config";
7
+
8
+ interface TavilySearchOptions {
9
+ searchDepth?: "basic" | "advanced";
10
+ maxResults?: number;
11
+ includeDomains?: string[];
12
+ excludeDomains?: string[];
13
+ }
14
+
15
+ interface TavilySearchResult {
16
+ title: string;
17
+ url: string;
18
+ content: string;
19
+ score: number;
20
+ raw_content?: string;
21
+ }
22
+
23
+ interface TavilyResponse {
24
+ results: TavilySearchResult[];
25
+ query: string;
26
+ }
27
+
28
+ export const searchTavily = async (
29
+ query: string,
30
+ opts?: TavilySearchOptions,
31
+ ): Promise<{ results: TavilySearchResult[]; suggestions: string[] }> => {
32
+ const tavilyApiKey =
33
+ configManager.getConfig("search.tavilyApiKey", "") ||
34
+ (typeof process !== "undefined" ? process.env.TAVILY_API_KEY : "") ||
35
+ "";
36
+
37
+ if (!tavilyApiKey) {
38
+ throw new Error(
39
+ "Tavily API key not configured. Please add your API key in Settings > Search.",
40
+ );
41
+ }
42
+
43
+ // Sanitize query - Tavily has a maximum query length limit (~400 characters)
44
+ // If the query is too long, truncate it to avoid 400 Bad Request errors
45
+ let sanitizedQuery = query.trim();
46
+ if (sanitizedQuery.length > 400) {
47
+ console.warn(`[Tavily] Query too long (${sanitizedQuery.length} chars), truncating to 400 chars`);
48
+ sanitizedQuery = sanitizedQuery.slice(0, 400);
49
+ }
50
+
51
+ try {
52
+ const response = await axios.post<TavilyResponse>(
53
+ "https://api.tavily.com/search",
54
+ {
55
+ api_key: tavilyApiKey,
56
+ query: sanitizedQuery,
57
+ search_depth: opts?.searchDepth || "basic",
58
+ max_results: opts?.maxResults || 10,
59
+ include_domains: opts?.includeDomains || [],
60
+ exclude_domains: opts?.excludeDomains || [],
61
+ include_answer: false,
62
+ include_raw_content: false,
63
+ },
64
+ {
65
+ headers: {
66
+ "Content-Type": "application/json",
67
+ },
68
+ },
69
+ );
70
+
71
+ const results = response.data.results || [];
72
+
73
+ return {
74
+ results: results.map((r) => ({
75
+ title: r.title,
76
+ url: r.url,
77
+ content: r.content,
78
+ score: r.score,
79
+ raw_content: r.raw_content,
80
+ })),
81
+ suggestions: [],
82
+ };
83
+ } catch (error: any) {
84
+ console.error("Tavily search error:", error);
85
+
86
+ // Provide more context for debugging
87
+ if (error.response?.status === 400) {
88
+ console.error("Tavily 400 error - Query length:", sanitizedQuery.length);
89
+ console.error("Tavily 400 error - Query preview:", sanitizedQuery.slice(0, 200));
90
+ }
91
+
92
+ throw new Error(
93
+ `Tavily search failed: ${error.response?.data?.error || error.message}`,
94
+ );
95
+ }
96
+ };
97
+
98
+ export const getTavilyApiKey = () =>
99
+ configManager.getConfig("search.tavilyApiKey", "") ||
100
+ (typeof process !== "undefined" ? process.env.TAVILY_API_KEY : "") ||
101
+ "";
102
+
103
+ export const isTavilyConfigured = () => {
104
+ const apiKey = getTavilyApiKey();
105
+ return apiKey && apiKey.length > 0;
106
+ };
@@ -0,0 +1,278 @@
1
+ import grab from "../utils/grab";
2
+
3
+ /**
4
+ * ### Tardigrade the Web Crawler
5
+ *
6
+ * 1. **Use Fetch API, check for bot detection.
7
+ * Scrape any domain's URL to get its HTML, JSON, or arraybuffer.<br />
8
+ * Scraping internet pages is a [free speech right
9
+ * ](https://blog.apify.com/is-web-scraping-legal/).
10
+ * 2. Features: timeout, redirects, default UA, referer as google, and bot
11
+ * detection checking. <br />
12
+ * 3. If fetch method does not get needed HTML, use Docker proxy as backup.
13
+ *
14
+ * 4. [Setup Docker](https://github.com/vtempest/ai-research-agent/tree/master/src/crawler)
15
+ * container with NodeJS server API renders with puppeteer DOM to get all HTML loaded by
16
+ * secondary in-page API requests after the initial page request, including user login and cookie storage.
17
+ * 5. Bypass Cloudflare bot check: A webpage proxy that request
18
+ * through Chromium (puppeteer) - can be used to bypass Cloudflare
19
+ * anti bot using cookie id javascript method.
20
+ * 6. Send your request to the server with the port 3000 and add your URL to the "url"
21
+ * query string like this: `http://localhost:3000/?url=https://example.org`
22
+ *
23
+ * 7. Optional: Setup residential IP proxy to access sites that IP-block datacenters
24
+ * and manage rotation with [Scrapoxy](https://scrapoxy.io). Recommended:
25
+ * [Hypeproxy](https://hypeproxy.io/products/static-residential-proxies)
26
+ * [NinjasProxy](https://ninjasproxy.com/residential-proxies/)
27
+ * [Proxy-Cheap](https://app.proxy-cheap.com/order)
28
+ * [LiveProxies](https://liveproxies.io/rotating-residential-proxies-pricing)
29
+ *
30
+ * @param {string} url - any domain's URL
31
+ * @param {Object} [options]
32
+ * @param {number} options.timeout default=5 - abort request if not retrived, in seconds
33
+ * @param {number} options.maxRedirects default=3 - max redirects to follow
34
+ * @param {number} options.checkBotDetection default=true - check for bot detection messages
35
+ * @param {number} options.changeReferer default=true - set referer as google
36
+ * @param {number} options.userAgentIndex default=0 - index of [google bot, default chrome]
37
+ * @param {string} options.proxy default=false - use proxy url
38
+ * @param {boolean} options.checkRobotsAllowed default=false - check robots.txt rules
39
+ * @returns {Promise<string>} - HTML, JSON, arraybuffer, or error object
40
+ * @category Extract
41
+ * @example await scrapeURL("https://hckrnews.com", {timeout: 5, userAgentIndex: 1})
42
+ * @author [vtempest (2025)](https://github.com/vtempest)
43
+ * @license MIT
44
+ */
45
+ export async function scrapeURL(url, options = {} as any) {
46
+ // try {
47
+ let {
48
+ timeout = 5,
49
+ checkBotDetection = true,
50
+ maxRedirects = 3,
51
+ changeReferer = 0,
52
+ userAgentIndex = 0,
53
+ proxy = null,
54
+ useProxyAsBackup = true,
55
+ checkRobotsAllowed = false,
56
+ } = options;
57
+
58
+ if (checkRobotsAllowed) {
59
+ const rules = await fetchScrapingRules(url);
60
+ if (!isAllowedToScrape(rules, url)) {
61
+ return { error: "Robots.txt forbids to scrape there" };
62
+ }
63
+ }
64
+
65
+ if (proxy) url = proxy + url;
66
+
67
+ var userAgentStrings = [
68
+ "Chrome/41.0.2272.96 Mobile Safari/537.36 (compatible ; Googlebot/2.1 ; +http://www.google.com/bot.html)",
69
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_4) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.83 Safari/537.36,gzip(gfe)",
70
+ ];
71
+
72
+ var headers = {
73
+ ...options,
74
+ "User-Agent": userAgentStrings[userAgentIndex],
75
+ signal: AbortSignal.timeout(timeout * 1000),
76
+ accept:
77
+ "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
78
+ "accept-language": "en-US,en;q=0.9",
79
+ };
80
+
81
+ if (changeReferer) headers["Referer"] = "https://www.google.com/";
82
+
83
+ let response;
84
+ try {
85
+ response = await grab(url, {
86
+ ...headers,
87
+ responseType: "raw",
88
+ });
89
+ } catch (e) {
90
+ return { error: "Error in fetch", msg: e.message };
91
+ }
92
+
93
+ if (response.redirected) {
94
+ if (maxRedirects <= 0) return { error: "Max redirects exceeded" };
95
+ maxRedirects--;
96
+ options = { ...options, maxRedirects };
97
+
98
+ return scrapeURL(response.url, options);
99
+ }
100
+
101
+ //return based on content type
102
+ const contentType = response.headers.get("Content-Type");
103
+
104
+ // if (contentType.includes("application/json")) {
105
+ // return await response.json();
106
+ // } else
107
+ // if (contentType.includes("text")) {
108
+ var html = await response.text();
109
+
110
+ if (checkBotDetection && checkHTMLForBotDetection(html)) {
111
+ html = await scrapeJINA(url, timeout);
112
+
113
+ //if all methods fail -- return jina
114
+ if (checkBotDetection && checkHTMLForBotDetection(html))
115
+ return { error: "Bot detected" }; //, html: response.html };
116
+ }
117
+
118
+ return html;
119
+ }
120
+
121
+ /**
122
+ * As backup, scrape with JINA to get html
123
+ * @param {string} url
124
+ * @returns {Promise<string>}
125
+ */
126
+ export async function scrapeJINA(url, timeout = 15) {
127
+ let articleExtract = "";
128
+ try {
129
+ articleExtract = await grab("https://r.jina.ai/" + url, {
130
+ timeout: timeout * 1000,
131
+ responseType: "text",
132
+ });
133
+ } catch {
134
+ return "";
135
+ }
136
+
137
+ //convert Title: to <title>
138
+ var title = articleExtract.match(/Title: (.*)/)?.[1];
139
+
140
+ if (articleExtract.includes("===============\n"))
141
+ articleExtract = articleExtract
142
+ .split("===============\n")
143
+ .slice(1)
144
+ .join(" ");
145
+
146
+ var match = articleExtract.match(/Markdown Content:([\s\S]*)/);
147
+ articleExtract = match ? match[1] : articleExtract;
148
+
149
+ // articleExtract = convertMarkdownToHTML(articleExtract);
150
+
151
+ if (title) articleExtract = "<title>" + title + "</title>" + articleExtract;
152
+
153
+ return articleExtract;
154
+ }
155
+
156
+ /**
157
+ * Check html for bot block messages
158
+ * @param {string} html
159
+ * @returns {Boolean} true if bot detection message found
160
+ */
161
+ function checkHTMLForBotDetection(html) {
162
+ var commonBlocks = [
163
+ "Error 403 - Unavailable",
164
+ "The security system for this website has been triggered",
165
+ "You do not have permission to view this page.",
166
+ "Our systems have detected unusual traffic from your computer network.",
167
+ "Your request has been blocked due to a network policy.",
168
+ "Cloudflare Ray ID found ",
169
+ "Please verify you are a human",
170
+ "Our systems have detected unusual traffic activity from your network. Please complete this reCAPTCHA",
171
+ "Sorry, we just need to make sure you're not a robot",
172
+ "Access to this page has been denied",
173
+ "<p>Please enable JS and disable any ad blocker",
174
+ "Please make sure your browser supports JavaScript",
175
+ "Please complete the security check to access",
176
+ "https://errors.edgesuite.net",
177
+ "Please enable JS and disable any ad blocker",
178
+ "The resource you are looking for might have been removed, had its name changed, or is temporarily unavailable.",
179
+ "We\u2019re currently checking your connection. This shouldn\u2019t take long.",
180
+ "Generated by cloudfront (CloudFront)",
181
+ "You don't have permission to access",
182
+ "The request could not be satisfied.",
183
+ "Enable JavaScript and cookies to continue",
184
+ "Something went wrong. Wait a moment and try again.",
185
+ "You\u2019re using a web browser that isn\u2019t supported",
186
+ "403 Forbidden",
187
+ "504 Gateway Timeout",
188
+ "You\u2019re Temporarily Blocked",
189
+ "Our systems have detected unusual activity",
190
+ "Agree & Join LinkedIn",
191
+ "Verifying you are human. This may take a few seconds",
192
+ "500 Internal Server Error",
193
+ "By clicking Continue to join or sign in, you agree to LinkedIn",
194
+ "Enable JS in your browser",
195
+ "Verifying you are human",
196
+ "Your request has been blocked",
197
+ "You've been blocked by network security",
198
+ "You've hit the rate limit.",
199
+ ];
200
+
201
+ return (
202
+ html &&
203
+ typeof html?.indexOf !== "undefined" &&
204
+ commonBlocks.filter((m) => html?.indexOf(m) > -1).length > 0
205
+ );
206
+ }
207
+
208
+ /**
209
+ * Fetches and parses the robots.txt file for a given URL.
210
+ * @param {string} url - The base URL to fetch the robots.txt from.
211
+ * @returns {Promise<Object>} A JSON object representing the parsed robots.txt.
212
+ */
213
+ export async function fetchScrapingRules(url) {
214
+ const robotsUrl = `https://${url.split("//")[1].split("/")[0]}/robots.txt`;
215
+ let content;
216
+ try {
217
+ content = await grab(robotsUrl, { responseType: "text" });
218
+ } catch (e) {
219
+ return { error: "No robots.txt found" };
220
+ }
221
+
222
+ const rules = {
223
+ directives: {},
224
+ crawlDelay: {},
225
+ sitemaps: [],
226
+ preferredHost: null,
227
+ };
228
+ let currentUserAgents = [];
229
+
230
+ const lines = content.split("\n");
231
+ for (const line of lines) {
232
+ const [directive, value] = line.split(":").map((s) => s.trim());
233
+ switch (directive.toLowerCase()) {
234
+ case "user-agent":
235
+ currentUserAgents = [value.toLowerCase()];
236
+ break;
237
+ case "disallow":
238
+ case "allow":
239
+ for (const ua of currentUserAgents) {
240
+ rules.directives[ua] = rules.directives[ua] || [];
241
+ rules.directives[ua].push({
242
+ path: value,
243
+ allow: directive.toLowerCase() === "allow",
244
+ });
245
+ }
246
+ break;
247
+ case "crawl-delay":
248
+ for (const ua of currentUserAgents) {
249
+ rules.crawlDelay[ua] = parseFloat(value);
250
+ }
251
+ break;
252
+ case "sitemap":
253
+ rules.sitemaps.push(value);
254
+ break;
255
+ case "host":
256
+ rules.preferredHost = value.toLowerCase();
257
+ break;
258
+ }
259
+ }
260
+ return rules;
261
+ }
262
+
263
+ /**
264
+ * Checks if a given path is allowed for a specific user agent.
265
+ * //TODO cache rules per domain
266
+ * @param {Object} rules - The parsed rules from robots.txt.
267
+ * @param {string} path - The path to check.
268
+ * @param {string} [userAgent='*'] - The user agent to check for.
269
+ * @returns {boolean} True if the path is allowed, false otherwise.
270
+ */
271
+ function isAllowedToScrape(rules, path, userAgent = "*") {
272
+ const relevantRules =
273
+ rules.directives[userAgent.toLowerCase()] || rules.directives["*"] || [];
274
+ for (const rule of relevantRules)
275
+ if (path.startsWith(rule.path)) return rule.allow;
276
+
277
+ return true; // If no rules match, it's allowed by default
278
+ }
@@ -0,0 +1,87 @@
1
+ /**
2
+ * @fileoverview Utility for grouping and merging similar keyphrases.
3
+ * Handles normalization and score aggregation for related terms.
4
+ */
5
+ import type { KeyphraseEntry } from "./types";
6
+
7
+ /**
8
+ * Folds smaller keyphrases that are near-subsets of larger ones, merging their
9
+ * weights and sentence lists upward.
10
+ *
11
+ * Two phrases are considered overlapping when they share all-but-one word
12
+ * (i.e., the number of words in the smaller phrase that do **not** appear in
13
+ * the larger phrase is fewer than 2). When a merge occurs:
14
+ * - The larger phrase absorbs a fraction of the smaller one's weight
15
+ * (`smallWeight / largerWordCount`).
16
+ * - Sentence indices are union-merged.
17
+ * - If the smaller phrase actually outweighs the larger at that point, the
18
+ * larger entry adopts the smaller phrase's text (best representative wins).
19
+ *
20
+ * Phrases are processed largest-first so that supersets are always evaluated
21
+ * before their constituent sub-phrases.
22
+ *
23
+ * @param keyphrases - Raw scored keyphrases, any order.
24
+ * @returns Deduplicated, folded array \u2014 larger representative phrases only.
25
+ *
26
+ * @example
27
+ * const folded = foldSubphrases([
28
+ * { keyphrase: "machine learning", words: 2, weight: 10, sentences: [0, 1] },
29
+ * { keyphrase: "machine", words: 1, weight: 4, sentences: [0, 2] },
30
+ * ]);
31
+ * // "machine" is absorbed into "machine learning"
32
+ */
33
+ export function foldSubphrases(keyphrases: KeyphraseEntry[]): KeyphraseEntry[] {
34
+ // Largest n-grams first so supersets are already in `folded` when we check
35
+ const sorted = keyphrases.slice().sort((a, b) => b.words - a.words);
36
+ const folded: KeyphraseEntry[] = [];
37
+
38
+ outer: for (const curr of sorted) {
39
+ const currWords = curr.keyphrase.split(" ");
40
+
41
+ for (const existing of folded) {
42
+ const existingWords = existing.keyphrase.split(" ");
43
+
44
+ // Count how many words in curr are absent from existing
45
+ let diff = 0;
46
+ for (const w of currWords) {
47
+ if (!existingWords.includes(w) && ++diff >= 2) break;
48
+ }
49
+
50
+ if (diff < 2) {
51
+ // Absorb curr's weight (proportionally discounted by target word count)
52
+ existing.weight += curr.weight / existingWords.length;
53
+
54
+ // Union of sentence indices
55
+ const existingSentences = Array.isArray(existing.sentences)
56
+ ? existing.sentences
57
+ : typeof existing.sentences === "string"
58
+ ? existing.sentences.split(",").map(Number).filter((n) => !isNaN(n))
59
+ : [];
60
+ const currSentences = Array.isArray(curr.sentences)
61
+ ? curr.sentences
62
+ : typeof curr.sentences === "string"
63
+ ? curr.sentences.split(",").map(Number).filter((n) => !isNaN(n))
64
+ : [];
65
+
66
+ const sentSet = new Set<number>(existingSentences);
67
+ for (const s of currSentences) sentSet.add(s);
68
+ existing.sentences = Array.from(sentSet);
69
+
70
+ // Promote curr as the representative if it now outweighs existing
71
+ if (existing.weight < curr.weight) {
72
+ existing.keyphrase = curr.keyphrase;
73
+ existing.words = curr.words;
74
+ }
75
+
76
+ continue outer; // merged \u2014 skip push
77
+ }
78
+ }
79
+
80
+ // No compatible existing phrase found \u2014 add as new entry
81
+ if (curr.sentences.length >= 1) {
82
+ folded.push({ ...curr });
83
+ }
84
+ }
85
+
86
+ return folded;
87
+ }
@@ -0,0 +1,64 @@
1
+ /**
2
+ * @fileoverview Utility for extracting noun-anchored n-grams from tokenized text.
3
+ * Implements logic to find high-quality candidate phrases.
4
+ */
5
+ import type { NgramMap, TopicToken } from "./types";
6
+ import { isWordCommonIgnored } from "../tokenize/word-is-ignored";
7
+
8
+ /** POS tag values that qualify as topic-worthy terms (noun=1, wiki-entity=5) */
9
+ const TOPIC_TAGS = new Set([1, 5]);
10
+
11
+ /**
12
+ * Extracts noun-anchored edge-grams from a token array and accumulates them into `nGrams`.
13
+ *
14
+ * An edge-gram is a contiguous slice of `nGramSize` tokens where:
15
+ * - The **first** and **last** tokens are topic entities (noun or wiki title).
16
+ * - Every token is either a topic entity or a common stop word (e.g. "of", "the").
17
+ * - Every token meets the `minWordLength` character threshold.
18
+ *
19
+ * This allows natural multi-word keyphrases like "state of the art" or
20
+ * "machine learning" while ignoring pure function-word sequences.
21
+ *
22
+ * @param nGramSize - Number of tokens in the slice to evaluate.
23
+ * @param terms - Full token array for the current sentence.
24
+ * @param index - Start position for this slice within `terms`.
25
+ * @param nGrams - Accumulator map mutated in-place: `nGrams[size][phrase] = [sentenceIdx, ...]`.
26
+ * @param minWordLength - Minimum character length for any word to be included.
27
+ * @param sentenceIndex - Index of the originating sentence, appended to the phrase's entry.
28
+ * @returns The same `nGrams` reference (mutated).
29
+ *
30
+ * @example
31
+ * const terms: TopicToken[] = [["machine", 1, 4, ""], ["learning", 1, 5, ""]];
32
+ * const nGrams: NgramMap = {};
33
+ * extractNounEdgeGrams(2, terms, 0, nGrams, 3, 0);
34
+ * // nGrams[2]["machine learning"] === [0]
35
+ */
36
+ export function extractNounEdgeGrams(
37
+ nGramSize: number,
38
+ terms: TopicToken[],
39
+ index: number,
40
+ nGrams: NgramMap,
41
+ minWordLength: number,
42
+ sentenceIndex: number,
43
+ ): NgramMap {
44
+ // Bail early if the slice would extend past the end of the array
45
+ if (index + nGramSize - 1 >= terms.length) return nGrams;
46
+
47
+ const slice = terms.slice(index, index + nGramSize);
48
+
49
+ // Edge tokens must be topic entities
50
+ if (!TOPIC_TAGS.has(slice[0][1]) || !TOPIC_TAGS.has(slice[nGramSize - 1][1])) return nGrams;
51
+
52
+ // Every token must meet length + category requirements
53
+ for (let i = 0; i < slice.length; i++) {
54
+ const [word, tag] = slice[i];
55
+ if (word.length < minWordLength) return nGrams;
56
+ if (!TOPIC_TAGS.has(tag) && !isWordCommonIgnored(word)) return nGrams;
57
+ }
58
+
59
+ const bucket = nGrams[nGramSize] ?? (nGrams[nGramSize] = {});
60
+ const phrase = slice.map(t => t[0]).join(" ");
61
+ (bucket[phrase] ?? (bucket[phrase] = [])).push(sentenceIndex);
62
+
63
+ return nGrams;
64
+ }
@@ -0,0 +1,132 @@
1
+ /**
2
+ * @fileoverview Implementation of TextRank algorithm for ranking sentences based on keyphrase centrality.
3
+ * Uses random walk simulations to identify the most relevant sentences in a document.
4
+ */
5
+ import type { SentenceEntry, RankOptions } from "./types";
6
+
7
+ /**
8
+ * Selects the next vertex in a random walk using weighted sampling.
9
+ *
10
+ * Instead of building a distribution array (O(totalWeight) space per step),
11
+ * this scans the adjacency map once with a single random draw \u2014 O(degree) time
12
+ * and O(1) additional space.
13
+ *
14
+ * @param adjacent - Map of neighbour key \u2192 edge weight.
15
+ * @returns The selected neighbour key.
16
+ */
17
+ function weightedRandom(adjacent: Map<string, number>): string {
18
+ let total = 0;
19
+ for (const w of adjacent.values()) total += w;
20
+
21
+ let r = Math.random() * total;
22
+ for (const [key, w] of adjacent) {
23
+ r -= w;
24
+ if (r <= 0) return key;
25
+ }
26
+
27
+ // Floating-point rounding safety: return the last entry
28
+ return [...adjacent.keys()][adjacent.size - 1];
29
+ }
30
+
31
+ /**
32
+ * ### TextRank: Rank Sentences by Centrality to Shared Keyphrases
33
+ *
34
+ * Builds a weighted undirected graph where each node is a sentence and edges
35
+ * connect sentences that share keyphrases. Sentence importance is estimated by
36
+ * a random-walk simulation (analogous to PageRank): nodes visited more often
37
+ * during the walk are considered more central to the document's key concepts.
38
+ *
39
+ * **Key optimisations vs. na\u00efve implementation:**
40
+ * - Weighted sampling via a single O(degree) scan instead of an O(weight) array.
41
+ * - `Set`-based keyphrase intersection (O(1) lookup) instead of `Array.includes`.
42
+ * - `Map<text, index>` for O(1) weight increment instead of O(n) linear scan.
43
+ * - Flat `Map<string, Map<string, number>>` graph \u2014 no object-method overhead.
44
+ * - Periodic forced reset prevents the walk from getting stuck in dense clusters.
45
+ *
46
+ * **References:**
47
+ * 1. Zhao & Xie (2021) \u2014 "An Improved TextRank Multi-feature Fusion Algorithm"
48
+ * https://iopscience.iop.org/article/10.1088/1742-6596/2078/1/012021/pdf
49
+ * 2. Pan et al. (2019) \u2014 "An improved TextRank keywords extraction algorithm"
50
+ * https://dl.acm.org/doi/10.1145/3321408.3326659
51
+ *
52
+ * @param sentencesWithKeyphrases - Sentences with pre-attached keyphrase lists.
53
+ * @param options - Walk parameters.
54
+ * @returns The same array with `weight` set on each sentence, or `undefined`
55
+ * if no edges exist (no shared keyphrases between any pair).
56
+ */
57
+ export function rankSentencesCentralToKeyphrase(
58
+ sentencesWithKeyphrases: SentenceEntry[],
59
+ options: RankOptions = {},
60
+ ): SentenceEntry[] | undefined {
61
+ const { iterations = 1000, resetInterval = 100 } = options;
62
+
63
+ // \u2500\u2500 Build graph \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
64
+ // graph: sentence text \u2192 { neighbour text \u2192 edge weight }
65
+ const graph = new Map<string, Map<string, number>>();
66
+
67
+ // text \u2192 array index for O(1) weight increments during the walk
68
+ const textToIndex = new Map<string, number>();
69
+
70
+ const sentences: SentenceEntry[] = sentencesWithKeyphrases.map((s, i) => {
71
+ textToIndex.set(s.text, i);
72
+ return { ...s, weight: 0 };
73
+ });
74
+
75
+ for (let i = 0; i < sentences.length; i++) {
76
+ const kp1 = sentences[i].keyphrases;
77
+
78
+ for (let j = i + 1; j < sentences.length; j++) {
79
+ const kp2 = sentences[j].keyphrases;
80
+
81
+ // Compare shorter list against a Set built from the longer list
82
+ let longerList = kp1;
83
+ let shorterList = kp2;
84
+ if (kp2.length > kp1.length) { longerList = kp2; shorterList = kp1; }
85
+
86
+ const shorterSet = new Set(shorterList.map(k => k.keyphrase));
87
+
88
+ let edgeWeight = 0;
89
+ for (const k of longerList) {
90
+ if (shorterSet.has(k.keyphrase)) edgeWeight += k.weight / 100;
91
+ }
92
+
93
+ if (edgeWeight > 0) {
94
+ const ti = sentences[i].text;
95
+ const tj = sentences[j].text;
96
+
97
+ if (!graph.has(ti)) graph.set(ti, new Map());
98
+ if (!graph.has(tj)) graph.set(tj, new Map());
99
+ graph.get(ti)!.set(tj, edgeWeight);
100
+ graph.get(tj)!.set(ti, edgeWeight);
101
+ }
102
+ }
103
+ }
104
+
105
+ const allVertices = [...graph.keys()];
106
+ if (allVertices.length === 0) return undefined;
107
+
108
+ // \u2500\u2500 Random walk \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500
109
+ let currentKey = allVertices[Math.floor(Math.random() * allVertices.length)];
110
+
111
+ for (let i = 0; i < iterations; i++) {
112
+ // Periodic reset \u2014 escape dense clusters and improve global coverage
113
+ if (i % resetInterval === 0) {
114
+ currentKey = allVertices[Math.floor(Math.random() * allVertices.length)];
115
+ }
116
+
117
+ const adjacent = graph.get(currentKey);
118
+ if (!adjacent || adjacent.size === 0) {
119
+ currentKey = allVertices[Math.floor(Math.random() * allVertices.length)];
120
+ continue;
121
+ }
122
+
123
+ const nextKey = weightedRandom(adjacent);
124
+
125
+ const idx = textToIndex.get(nextKey);
126
+ if (idx !== undefined) sentences[idx].weight++;
127
+
128
+ currentKey = nextKey;
129
+ }
130
+
131
+ return sentences;
132
+ }